Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
49 changes: 0 additions & 49 deletions sentry_sdk/ai/monitoring.py

This file was deleted.

29 changes: 21 additions & 8 deletions sentry_sdk/integrations/anthropic.py
Original file line number Diff line number Diff line change
Expand Up @@ -5,7 +5,6 @@
from typing import TYPE_CHECKING, cast

import sentry_sdk
from sentry_sdk.ai.monitoring import record_token_usage
from sentry_sdk.ai.utils import (
GEN_AI_ALLOWED_MESSAGE_ROLES,
normalize_message_roles,
Expand Down Expand Up @@ -661,13 +660,27 @@ def _set_output_data(
span, SPANDATA.GEN_AI_RESPONSE_TEXT, output_messages["response"]
)

record_token_usage(
span,
input_tokens=input_tokens,
output_tokens=output_tokens,
input_tokens_cached=cache_read_input_tokens,
input_tokens_cache_write=cache_write_input_tokens,
)
if input_tokens is not None:
span.set_attribute(SPANDATA.GEN_AI_USAGE_INPUT_TOKENS, input_tokens)

if cache_read_input_tokens is not None:
span.set_attribute(
SPANDATA.GEN_AI_USAGE_CACHE_READ_INPUT_TOKENS, cache_read_input_tokens
)

if cache_write_input_tokens is not None:
span.set_attribute(
SPANDATA.GEN_AI_USAGE_CACHE_CREATION_INPUT_TOKENS,
cache_write_input_tokens,
)

if output_tokens is not None:
span.set_attribute(SPANDATA.GEN_AI_USAGE_OUTPUT_TOKENS, output_tokens)

if input_tokens is not None and output_tokens is not None:
span.set_attribute(
SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS, input_tokens + output_tokens
)


def _sentry_patched_create_sync(f: "Any", *args: "Any", **kwargs: "Any") -> "Any":
Expand Down
67 changes: 51 additions & 16 deletions sentry_sdk/integrations/cohere.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,7 +3,6 @@
from typing import TYPE_CHECKING

from sentry_sdk import consts
from sentry_sdk.ai.monitoring import record_token_usage
from sentry_sdk.ai.utils import set_data_normalized
from sentry_sdk.consts import SPANDATA
from sentry_sdk.traces import Span
Expand Down Expand Up @@ -119,17 +118,47 @@ def collect_chat_response_fields(

if hasattr(res, "meta"):
if hasattr(res.meta, "billed_units"):
record_token_usage(
span,
input_tokens=res.meta.billed_units.input_tokens,
output_tokens=res.meta.billed_units.output_tokens,
)
if res.meta.billed_units.input_tokens is not None:
span.set_attribute(
SPANDATA.GEN_AI_USAGE_INPUT_TOKENS,
res.meta.billed_units.input_tokens,
)

if res.meta.billed_units.output_tokens is not None:
span.set_attribute(
SPANDATA.GEN_AI_USAGE_OUTPUT_TOKENS,
res.meta.billed_units.output_tokens,
)

if (
res.meta.billed_units.input_tokens is not None
and res.meta.billed_units.output_tokens is not None
):
span.set_attribute(
SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS,
res.meta.billed_units.input_tokens
+ res.meta.billed_units.output_tokens,
)
elif hasattr(res.meta, "tokens"):
record_token_usage(
span,
input_tokens=res.meta.tokens.input_tokens,
output_tokens=res.meta.tokens.output_tokens,
)
if res.meta.tokens.input_tokens is not None:
span.set_attribute(
SPANDATA.GEN_AI_USAGE_INPUT_TOKENS, res.meta.tokens.input_tokens
)

if res.meta.tokens.output_tokens is not None:
span.set_attribute(
SPANDATA.GEN_AI_USAGE_OUTPUT_TOKENS,
res.meta.tokens.output_tokens,
)

if (
res.meta.tokens.input_tokens is not None
and res.meta.tokens.output_tokens is not None
):
span.set_attribute(
SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS,
res.meta.tokens.input_tokens + res.meta.tokens.output_tokens,
)

if hasattr(res.meta, "warnings"):
set_data_normalized(span, SPANDATA.AI_WARNINGS, res.meta.warnings)
Expand Down Expand Up @@ -272,11 +301,17 @@ def new_embed(*args: "Any", **kwargs: "Any") -> "Any":
and hasattr(res.meta, "billed_units")
and hasattr(res.meta.billed_units, "input_tokens")
):
record_token_usage(
span,
input_tokens=res.meta.billed_units.input_tokens,
total_tokens=res.meta.billed_units.input_tokens,
)
if res.meta.billed_units.input_tokens is not None:
span.set_attribute(
SPANDATA.GEN_AI_USAGE_INPUT_TOKENS,
res.meta.billed_units.input_tokens,
)

span.set_attribute(
SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS,
res.meta.billed_units.input_tokens,
)

return res

return new_embed
73 changes: 54 additions & 19 deletions sentry_sdk/integrations/huggingface_hub.py
Original file line number Diff line number Diff line change
Expand Up @@ -4,7 +4,6 @@
from typing import TYPE_CHECKING, cast

import sentry_sdk
from sentry_sdk.ai.monitoring import record_token_usage
from sentry_sdk.ai.utils import set_data_normalized
from sentry_sdk.consts import OP, SPANDATA
from sentry_sdk.integrations import DidNotEnable, Integration, _check_minimum_version
Expand Down Expand Up @@ -209,17 +208,34 @@ def new_huggingface_task(*args: "Any", **kwargs: "Any") -> "Any":
text_response,
)

if usage is not None:
record_token_usage(
span,
input_tokens=usage.prompt_tokens,
output_tokens=usage.completion_tokens,
total_tokens=usage.total_tokens,
if usage is not None and usage.prompt_tokens is not None:
span.set_attribute(
SPANDATA.GEN_AI_USAGE_INPUT_TOKENS, usage.prompt_tokens
)

if usage is not None and usage.completion_tokens is not None:
span.set_attribute(
SPANDATA.GEN_AI_USAGE_OUTPUT_TOKENS, usage.completion_tokens
)

if usage is not None and usage.total_tokens is not None:
span.set_attribute(
SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS, usage.total_tokens
)

elif (
usage is not None
and usage.prompt_tokens is not None
and usage.completion_tokens is not None
):
span.set_attribute(
SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS,
usage.prompt_tokens + usage.completion_tokens,
)
elif tokens_used > 0:
record_token_usage(
span,
total_tokens=tokens_used,
span.set_attribute(
SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS,
tokens_used,
)

# If the response is not a generator (meaning a streaming response)
Expand Down Expand Up @@ -276,9 +292,8 @@ def new_details_iterator() -> "Iterable[Any]":
)

if tokens_used > 0:
record_token_usage(
span,
total_tokens=tokens_used,
span.set_attribute(
SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS, tokens_used
)

span.__exit__(None, None, None)
Expand Down Expand Up @@ -368,12 +383,32 @@ def new_iterator() -> "Iterable[ChatCompletionStreamOutput]":
text_response,
)

if usage is not None:
record_token_usage(
span,
input_tokens=usage.prompt_tokens,
output_tokens=usage.completion_tokens,
total_tokens=usage.total_tokens,
if usage is None:
span.__exit__(None, None, None)
return

if usage.prompt_tokens is not None:
span.set_attribute(
SPANDATA.GEN_AI_USAGE_INPUT_TOKENS, usage.prompt_tokens
)

if usage.completion_tokens is not None:
span.set_attribute(
SPANDATA.GEN_AI_USAGE_OUTPUT_TOKENS,
usage.completion_tokens,
)

if usage.total_tokens is not None:
span.set_attribute(
SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS, usage.total_tokens
)
elif (
usage.prompt_tokens is not None
and usage.completion_tokens is not None
):
span.set_attribute(
SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS,
usage.prompt_tokens + usage.completion_tokens,
)

span.__exit__(None, None, None)
Expand Down
26 changes: 19 additions & 7 deletions sentry_sdk/integrations/litellm.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,7 +3,6 @@

import sentry_sdk
from sentry_sdk import consts
from sentry_sdk.ai.monitoring import record_token_usage
from sentry_sdk.ai.utils import (
set_data_normalized,
transform_openai_content_part,
Expand Down Expand Up @@ -213,12 +212,25 @@ def _success_callback(
# Record token usage
if hasattr(completion_response, "usage"):
usage = completion_response.usage
record_token_usage(
span,
input_tokens=getattr(usage, "prompt_tokens", None),
output_tokens=getattr(usage, "completion_tokens", None),
total_tokens=getattr(usage, "total_tokens", None),
)

input_tokens = getattr(usage, "prompt_tokens", None)
if input_tokens is not None:
span.set_attribute(SPANDATA.GEN_AI_USAGE_INPUT_TOKENS, input_tokens)

output_tokens = getattr(usage, "completion_tokens", None)
if output_tokens is not None:
span.set_attribute(SPANDATA.GEN_AI_USAGE_OUTPUT_TOKENS, output_tokens)

total_tokens = getattr(usage, "total_tokens", None)
if (
total_tokens is None
and input_tokens is not None
and output_tokens is not None
):
total_tokens = input_tokens + output_tokens

if total_tokens is not None:
span.set_attribute(SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS, total_tokens)

finally:
is_streaming = kwargs.get("stream")
Expand Down
Loading
Loading