Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
7 changes: 7 additions & 0 deletions sentry_sdk/ai/monitoring.py
Original file line number Diff line number Diff line change
Expand Up @@ -192,6 +192,13 @@ def record_token_usage(
output_tokens_reasoning: "Optional[int]" = None,
total_tokens: "Optional[int]" = None,
) -> None:
warnings.warn(
"record_token_usage() is deprecated and will be removed in version 3.0 of sentry-sdk. "
"Use the manual span API instead, e.g. span.set_attribute() (in streaming mode) or span.set_data().",
DeprecationWarning,
stacklevel=2,
)

# TODO: move pipeline name elsewhere
ai_pipeline_name = get_ai_pipeline_name()
if ai_pipeline_name:
Expand Down
25 changes: 17 additions & 8 deletions sentry_sdk/integrations/anthropic.py
Original file line number Diff line number Diff line change
Expand Up @@ -5,7 +5,6 @@
from typing import TYPE_CHECKING, cast

import sentry_sdk
from sentry_sdk.ai.monitoring import record_token_usage
from sentry_sdk.ai.utils import (
GEN_AI_ALLOWED_MESSAGE_ROLES,
get_start_span_function,
Expand Down Expand Up @@ -714,13 +713,23 @@
span, SPANDATA.GEN_AI_RESPONSE_TEXT, output_messages["response"]
)

record_token_usage(
span,
input_tokens=input_tokens,
output_tokens=output_tokens,
input_tokens_cached=cache_read_input_tokens,
input_tokens_cache_write=cache_write_input_tokens,
)
if input_tokens is not None:
set_on_span(SPANDATA.GEN_AI_USAGE_INPUT_TOKENS, input_tokens)

if cache_read_input_tokens is not None:
set_on_span(SPANDATA.GEN_AI_USAGE_INPUT_TOKENS_CACHED, cache_read_input_tokens)

if cache_write_input_tokens is not None:
set_on_span(
SPANDATA.GEN_AI_USAGE_INPUT_TOKENS_CACHE_WRITE,
cache_write_input_tokens,
)

if output_tokens is not None:
set_on_span(SPANDATA.GEN_AI_USAGE_OUTPUT_TOKENS, output_tokens)

if input_tokens is not None and output_tokens is not None:
set_on_span(SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS, input_tokens + output_tokens)

Check warning on line 732 in sentry_sdk/integrations/anthropic.py

View check run for this annotation

@sentry/warden / warden: code-review

Inlined token recording drops the AI pipeline name

The modified OpenAI token-usage helper writes token attributes directly but no longer preserves `record_token_usage()`'s `GEN_AI_PIPELINE_NAME` side effect. When an OpenAI call runs inside `@ai_track`, its span will therefore lose the pipeline name. Please retain that context-variable write in the inlined path. The reported Anthropic location is not among the changed files.
Comment thread
alexander-alderman-webb marked this conversation as resolved.


def _sentry_patched_create_sync(f: "Any", *args: "Any", **kwargs: "Any") -> "Any":
Expand Down
74 changes: 58 additions & 16 deletions sentry_sdk/integrations/cohere.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,7 +3,6 @@
from typing import TYPE_CHECKING

from sentry_sdk import consts
from sentry_sdk.ai.monitoring import record_token_usage
from sentry_sdk.ai.utils import get_start_span_function, set_data_normalized
from sentry_sdk.consts import SPANDATA
from sentry_sdk.traces import StreamedSpan
Expand Down Expand Up @@ -134,18 +133,52 @@ def collect_chat_response_fields(
set_data_normalized(span, "ai." + attr, getattr(res, attr))

if hasattr(res, "meta"):
set_on_span = (
span.set_attribute if isinstance(span, StreamedSpan) else span.set_data
)

if hasattr(res.meta, "billed_units"):
record_token_usage(
span,
input_tokens=res.meta.billed_units.input_tokens,
output_tokens=res.meta.billed_units.output_tokens,
)
if res.meta.billed_units.input_tokens is not None:
set_on_span(
SPANDATA.GEN_AI_USAGE_INPUT_TOKENS,
res.meta.billed_units.input_tokens,
)

if res.meta.billed_units.output_tokens is not None:
set_on_span(
SPANDATA.GEN_AI_USAGE_OUTPUT_TOKENS,
res.meta.billed_units.output_tokens,
)

if (
res.meta.billed_units.input_tokens is not None
and res.meta.billed_units.output_tokens is not None
):
set_on_span(
SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS,
res.meta.billed_units.input_tokens
+ res.meta.billed_units.output_tokens,
)
elif hasattr(res.meta, "tokens"):
record_token_usage(
span,
input_tokens=res.meta.tokens.input_tokens,
output_tokens=res.meta.tokens.output_tokens,
)
if res.meta.tokens.input_tokens is not None:
set_on_span(
SPANDATA.GEN_AI_USAGE_INPUT_TOKENS, res.meta.tokens.input_tokens
)

if res.meta.tokens.output_tokens is not None:
set_on_span(
SPANDATA.GEN_AI_USAGE_OUTPUT_TOKENS,
res.meta.tokens.output_tokens,
)

if (
res.meta.tokens.input_tokens is not None
and res.meta.tokens.output_tokens is not None
):
set_on_span(
SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS,
res.meta.tokens.input_tokens + res.meta.tokens.output_tokens,
)
Comment thread
cursor[bot] marked this conversation as resolved.

if hasattr(res.meta, "warnings"):
set_data_normalized(span, SPANDATA.AI_WARNINGS, res.meta.warnings)
Expand Down Expand Up @@ -268,13 +301,17 @@ def new_embed(*args: "Any", **kwargs: "Any") -> "Any":
"sentry.origin": CohereIntegration.origin,
},
)

set_on_span = span_ctx.set_attribute
else:
span_ctx = get_start_span_function()(
op=consts.OP.COHERE_EMBEDDINGS_CREATE,
name="Cohere Embedding Creation",
origin=CohereIntegration.origin,
)

set_on_span = span_ctx.set_data

with span_ctx as span:
if "texts" in kwargs and _should_record(integration, "inputs"):
if isinstance(kwargs["texts"], str):
Expand Down Expand Up @@ -302,11 +339,16 @@ def new_embed(*args: "Any", **kwargs: "Any") -> "Any":
and hasattr(res.meta, "billed_units")
and hasattr(res.meta.billed_units, "input_tokens")
):
record_token_usage(
span,
input_tokens=res.meta.billed_units.input_tokens,
total_tokens=res.meta.billed_units.input_tokens,
)
if res.meta.billed_units.input_tokens is not None:
set_on_span(
SPANDATA.GEN_AI_USAGE_INPUT_TOKENS,
res.meta.billed_units.input_tokens,
)

set_on_span(
SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS,
res.meta.billed_units.input_tokens,
)
return res

return new_embed
79 changes: 55 additions & 24 deletions sentry_sdk/integrations/huggingface_hub.py
Original file line number Diff line number Diff line change
Expand Up @@ -4,7 +4,6 @@
from typing import TYPE_CHECKING, cast

import sentry_sdk
from sentry_sdk.ai.monitoring import record_token_usage
from sentry_sdk.ai.utils import (
_set_span_data_attribute,
get_start_span_function,
Expand All @@ -13,7 +12,6 @@
from sentry_sdk.consts import OP, SPANDATA
from sentry_sdk.integrations import DidNotEnable, Integration
from sentry_sdk.scope import should_send_default_pii
from sentry_sdk.traces import StreamedSpan
from sentry_sdk.tracing_utils import has_span_streaming_enabled
from sentry_sdk.utils import (
capture_internal_exceptions,
Expand All @@ -23,13 +21,12 @@
)

if TYPE_CHECKING:
from typing import Any, Callable, Iterable, Union
from typing import Any, Callable, Iterable

from huggingface_hub import (
ChatCompletionStreamOutput,
)

from sentry_sdk.tracing import Span

try:
import huggingface_hub.inference._client
Expand Down Expand Up @@ -97,7 +94,6 @@
model = hf_client.model or kwargs.get("model") or ""
operation_name = op.split(".")[-1]

span: "Union[Span, StreamedSpan]"
if has_span_streaming_enabled(client.options):
span = sentry_sdk.traces.start_span(
name=f"{operation_name} {model}",
Expand All @@ -106,12 +102,16 @@
"sentry.origin": HuggingfaceHubIntegration.origin,
},
)

set_on_span = span.set_attribute
else:
span = get_start_span_function()(
op=op,
name=f"{operation_name} {model}",
origin=HuggingfaceHubIntegration.origin,
)

set_on_span = span.set_data

Check warning on line 114 in sentry_sdk/integrations/huggingface_hub.py

View check run for this annotation

@sentry/warden / warden: find-bugs

Inlining token recording drops gen_ai.pipeline.name from @ai_track

OpenAI’s inlined token-usage helper writes token attributes directly but omits `record_token_usage()`’s `GEN_AI_PIPELINE_NAME` side effect. When an OpenAI call runs inside `@ai_track`, its integration span no longer receives the pipeline name. Preserve that context-variable write in the inlined path.
span.__enter__()

_set_span_data_attribute(span, SPANDATA.GEN_AI_OPERATION_NAME, operation_name)
Expand Down Expand Up @@ -258,16 +258,30 @@
)

if usage is not None:
record_token_usage(
span,
input_tokens=usage.prompt_tokens,
output_tokens=usage.completion_tokens,
total_tokens=usage.total_tokens,
)
if usage is not None and usage.prompt_tokens is not None:
set_on_span(SPANDATA.GEN_AI_USAGE_INPUT_TOKENS, usage.prompt_tokens)

if usage is not None and usage.completion_tokens is not None:
set_on_span(
SPANDATA.GEN_AI_USAGE_OUTPUT_TOKENS, usage.completion_tokens
)

if usage is not None and usage.total_tokens is not None:
set_on_span(SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS, usage.total_tokens)

elif (
usage is not None
and usage.prompt_tokens is not None
and usage.completion_tokens is not None
):
set_on_span(
SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS,
usage.prompt_tokens + usage.completion_tokens,
)
elif tokens_used > 0:
record_token_usage(
span,
total_tokens=tokens_used,
set_on_span(
SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS,

Check warning on line 283 in sentry_sdk/integrations/huggingface_hub.py

View check run for this annotation

@sentry/warden / warden: code-review

[3JF-TLT] Inlined token recording drops the AI pipeline name (additional location)

The modified OpenAI token-usage helper writes token attributes directly but no longer preserves `record_token_usage()`'s `GEN_AI_PIPELINE_NAME` side effect. When an OpenAI call runs inside `@ai_track`, its span will therefore lose the pipeline name. Please retain that context-variable write in the inlined path. The reported Anthropic location is not among the changed files.
tokens_used,
)

# If the response is not a generator (meaning a streaming response)
Expand Down Expand Up @@ -331,10 +345,7 @@
)

if tokens_used > 0:
record_token_usage(
span,
total_tokens=tokens_used,
)
set_on_span(SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS, tokens_used)

span.__exit__(None, None, None)

Expand Down Expand Up @@ -443,12 +454,32 @@
text_response,
)

if usage is not None:
record_token_usage(
span,
input_tokens=usage.prompt_tokens,
output_tokens=usage.completion_tokens,
total_tokens=usage.total_tokens,
if usage is None:
span.__exit__(None, None, None)
return

if usage.prompt_tokens is not None:
set_on_span(
SPANDATA.GEN_AI_USAGE_INPUT_TOKENS, usage.prompt_tokens
)

if usage.completion_tokens is not None:
set_on_span(
SPANDATA.GEN_AI_USAGE_OUTPUT_TOKENS,
usage.completion_tokens,
)

if usage.total_tokens is not None:
set_on_span(
SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS, usage.total_tokens
)
elif (
usage.prompt_tokens is not None
and usage.completion_tokens is not None
):
set_on_span(
SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS,
usage.prompt_tokens + usage.completion_tokens,
)

span.__exit__(None, None, None)
Expand Down
29 changes: 23 additions & 6 deletions sentry_sdk/integrations/litellm.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,7 +3,6 @@

import sentry_sdk
from sentry_sdk import consts
from sentry_sdk.ai.monitoring import record_token_usage
from sentry_sdk.ai.utils import (
get_start_span_function,
set_data_normalized,
Expand All @@ -14,6 +13,7 @@
from sentry_sdk.consts import SPANDATA
from sentry_sdk.integrations import DidNotEnable, Integration
from sentry_sdk.scope import should_send_default_pii
from sentry_sdk.traces import StreamedSpan
from sentry_sdk.tracing_utils import (
has_span_streaming_enabled,
should_truncate_gen_ai_input,
Expand Down Expand Up @@ -257,13 +257,30 @@ def _success_callback(
# Record token usage
if hasattr(completion_response, "usage"):
usage = completion_response.usage
record_token_usage(
span,
input_tokens=getattr(usage, "prompt_tokens", None),
output_tokens=getattr(usage, "completion_tokens", None),
total_tokens=getattr(usage, "total_tokens", None),

set_on_span = (
span.set_attribute if isinstance(span, StreamedSpan) else span.set_data
)

input_tokens = getattr(usage, "prompt_tokens", None)
if input_tokens is not None:
set_on_span(SPANDATA.GEN_AI_USAGE_INPUT_TOKENS, input_tokens)

output_tokens = getattr(usage, "completion_tokens", None)
if output_tokens is not None:
set_on_span(SPANDATA.GEN_AI_USAGE_OUTPUT_TOKENS, output_tokens)

total_tokens = getattr(usage, "total_tokens", None)
if (
total_tokens is None
and input_tokens is not None
and output_tokens is not None
):
total_tokens = input_tokens + output_tokens

if total_tokens is not None:
set_on_span(SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS, total_tokens)

finally:
is_streaming = kwargs.get("stream")
# Callback is fired multiple times when streaming a response.
Expand Down
Loading
Loading