From 6f265277cf98b68351181e8b35d62aec2d40d28f Mon Sep 17 00:00:00 2001 From: Alexander Alderman Webb Date: Wed, 7 Oct 2026 15:54:48 +0200 Subject: [PATCH 1/4] chore: Deprecate ai.monitoring.record_token_usage() --- sentry_sdk/ai/monitoring.py | 7 +++ sentry_sdk/integrations/anthropic.py | 25 +++++++--- sentry_sdk/integrations/cohere.py | 73 ++++++++++++++++++++++------ sentry_sdk/integrations/litellm.py | 29 ++++++++--- sentry_sdk/integrations/openai.py | 63 +++++++++++++++++------- 5 files changed, 150 insertions(+), 47 deletions(-) diff --git a/sentry_sdk/ai/monitoring.py b/sentry_sdk/ai/monitoring.py index 05aabd96a9..f58c49ac26 100644 --- a/sentry_sdk/ai/monitoring.py +++ b/sentry_sdk/ai/monitoring.py @@ -192,6 +192,13 @@ def record_token_usage( output_tokens_reasoning: "Optional[int]" = None, total_tokens: "Optional[int]" = None, ) -> None: + warnings.warn( + "record_token_usage() is deprecated and will be removed in version 3.0 of sentry-sdk. " + "Use the manual span API instead, e.g. Span.set_data().", + DeprecationWarning, + stacklevel=2, + ) + # TODO: move pipeline name elsewhere ai_pipeline_name = get_ai_pipeline_name() if ai_pipeline_name: diff --git a/sentry_sdk/integrations/anthropic.py b/sentry_sdk/integrations/anthropic.py index 23bdbbb06e..4b04bc1cfc 100644 --- a/sentry_sdk/integrations/anthropic.py +++ b/sentry_sdk/integrations/anthropic.py @@ -5,7 +5,6 @@ from typing import TYPE_CHECKING, cast import sentry_sdk -from sentry_sdk.ai.monitoring import record_token_usage from sentry_sdk.ai.utils import ( GEN_AI_ALLOWED_MESSAGE_ROLES, get_start_span_function, @@ -714,13 +713,23 @@ def _set_output_data( span, SPANDATA.GEN_AI_RESPONSE_TEXT, output_messages["response"] ) - record_token_usage( - span, - input_tokens=input_tokens, - output_tokens=output_tokens, - input_tokens_cached=cache_read_input_tokens, - input_tokens_cache_write=cache_write_input_tokens, - ) + if input_tokens is not None: + set_on_span(SPANDATA.GEN_AI_USAGE_INPUT_TOKENS, input_tokens) + + if cache_read_input_tokens is not None: + set_on_span(SPANDATA.GEN_AI_USAGE_INPUT_TOKENS_CACHED, cache_read_input_tokens) + + if cache_write_input_tokens is not None: + set_on_span( + SPANDATA.GEN_AI_USAGE_INPUT_TOKENS_CACHE_WRITE, + cache_write_input_tokens, + ) + + if output_tokens is not None: + set_on_span(SPANDATA.GEN_AI_USAGE_OUTPUT_TOKENS, output_tokens) + + if input_tokens is not None and output_tokens is not None: + set_on_span(SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS, input_tokens + output_tokens) def _sentry_patched_create_sync(f: "Any", *args: "Any", **kwargs: "Any") -> "Any": diff --git a/sentry_sdk/integrations/cohere.py b/sentry_sdk/integrations/cohere.py index 9cac912a2b..6fcc1f4d2a 100644 --- a/sentry_sdk/integrations/cohere.py +++ b/sentry_sdk/integrations/cohere.py @@ -3,7 +3,6 @@ from typing import TYPE_CHECKING from sentry_sdk import consts -from sentry_sdk.ai.monitoring import record_token_usage from sentry_sdk.ai.utils import get_start_span_function, set_data_normalized from sentry_sdk.consts import SPANDATA from sentry_sdk.traces import StreamedSpan @@ -134,19 +133,52 @@ def collect_chat_response_fields( set_data_normalized(span, "ai." + attr, getattr(res, attr)) if hasattr(res, "meta"): - if hasattr(res.meta, "billed_units"): - record_token_usage( - span, - input_tokens=res.meta.billed_units.input_tokens, - output_tokens=res.meta.billed_units.output_tokens, + set_on_span = ( + span.set_attribute if isinstance(span, StreamedSpan) else span.set_data + ) + + if res.meta.billed_units.input_tokens is not None: + set_on_span( + SPANDATA.GEN_AI_USAGE_INPUT_TOKENS, + res.meta.billed_units.input_tokens, ) - elif hasattr(res.meta, "tokens"): - record_token_usage( - span, - input_tokens=res.meta.tokens.input_tokens, - output_tokens=res.meta.tokens.output_tokens, + + if res.meta.billed_units.output_tokens is not None: + set_on_span( + SPANDATA.GEN_AI_USAGE_OUTPUT_TOKENS, + res.meta.billed_units.output_tokens, ) + if ( + res.meta.billed_units.input_tokens is not None + and res.meta.billed_units.output_tokens is not None + ): + set_on_span( + SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS, + res.meta.billed_units.input_tokens + + res.meta.billed_units.output_tokens, + ) + elif hasattr(res.meta, "tokens"): + if res.meta.tokens.input_tokens is not None: + set_on_span( + SPANDATA.GEN_AI_USAGE_INPUT_TOKENS, res.meta.tokens.input_tokens + ) + + if res.meta.tokens.output_tokens is not None: + set_on_span( + SPANDATA.GEN_AI_USAGE_OUTPUT_TOKENS, + res.meta.tokens.output_tokens, + ) + + if ( + res.meta.tokens.input_tokens is not None + and res.meta.tokens.output_tokens is not None + ): + set_on_span( + SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS, + res.meta.tokens.input_tokens + res.meta.tokens.output_tokens, + ) + if hasattr(res.meta, "warnings"): set_data_normalized(span, SPANDATA.AI_WARNINGS, res.meta.warnings) @@ -268,6 +300,8 @@ def new_embed(*args: "Any", **kwargs: "Any") -> "Any": "sentry.origin": CohereIntegration.origin, }, ) + + set_on_span = span_ctx.set_attribute else: span_ctx = get_start_span_function()( op=consts.OP.COHERE_EMBEDDINGS_CREATE, @@ -275,6 +309,8 @@ def new_embed(*args: "Any", **kwargs: "Any") -> "Any": origin=CohereIntegration.origin, ) + set_on_span = span_ctx.set_data + with span_ctx as span: if "texts" in kwargs and _should_record(integration, "inputs"): if isinstance(kwargs["texts"], str): @@ -302,11 +338,16 @@ def new_embed(*args: "Any", **kwargs: "Any") -> "Any": and hasattr(res.meta, "billed_units") and hasattr(res.meta.billed_units, "input_tokens") ): - record_token_usage( - span, - input_tokens=res.meta.billed_units.input_tokens, - total_tokens=res.meta.billed_units.input_tokens, - ) + if res.meta.billed_units.input_tokens is not None: + set_on_span( + SPANDATA.GEN_AI_USAGE_INPUT_TOKENS, + res.meta.billed_units.input_tokens, + ) + + set_on_span( + SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS, + res.meta.billed_units.input_tokens, + ) return res return new_embed diff --git a/sentry_sdk/integrations/litellm.py b/sentry_sdk/integrations/litellm.py index 74adddb7ec..ce94441a4d 100644 --- a/sentry_sdk/integrations/litellm.py +++ b/sentry_sdk/integrations/litellm.py @@ -3,7 +3,6 @@ import sentry_sdk from sentry_sdk import consts -from sentry_sdk.ai.monitoring import record_token_usage from sentry_sdk.ai.utils import ( get_start_span_function, set_data_normalized, @@ -14,6 +13,7 @@ from sentry_sdk.consts import SPANDATA from sentry_sdk.integrations import DidNotEnable, Integration from sentry_sdk.scope import should_send_default_pii +from sentry_sdk.traces import StreamedSpan from sentry_sdk.tracing_utils import ( has_span_streaming_enabled, should_truncate_gen_ai_input, @@ -257,13 +257,30 @@ def _success_callback( # Record token usage if hasattr(completion_response, "usage"): usage = completion_response.usage - record_token_usage( - span, - input_tokens=getattr(usage, "prompt_tokens", None), - output_tokens=getattr(usage, "completion_tokens", None), - total_tokens=getattr(usage, "total_tokens", None), + + set_on_span = ( + span.set_attribute if isinstance(span, StreamedSpan) else span.set_data ) + input_tokens = getattr(usage, "prompt_tokens", None) + if input_tokens is not None: + set_on_span(SPANDATA.GEN_AI_USAGE_INPUT_TOKENS, input_tokens) + + output_tokens = getattr(usage, "completion_tokens", None) + if output_tokens is not None: + set_on_span(SPANDATA.GEN_AI_USAGE_OUTPUT_TOKENS, output_tokens) + + total_tokens = getattr(usage, "total_tokens", None) + if ( + total_tokens is None + and input_tokens is not None + and output_tokens is not None + ): + total_tokens = input_tokens + output_tokens + + if total_tokens is not None: + set_on_span(SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS, total_tokens) + finally: is_streaming = kwargs.get("stream") # Callback is fired multiple times when streaming a response. diff --git a/sentry_sdk/integrations/openai.py b/sentry_sdk/integrations/openai.py index 742b693245..04ae33daff 100644 --- a/sentry_sdk/integrations/openai.py +++ b/sentry_sdk/integrations/openai.py @@ -34,7 +34,6 @@ from sentry_sdk.ai._openai_responses_api import ( _transform_tool_definitions as _transform_tool_definitions_responses, ) -from sentry_sdk.ai.monitoring import record_token_usage from sentry_sdk.ai.utils import ( get_start_span_function, normalize_message_roles, @@ -239,21 +238,36 @@ def _calculate_completions_token_usage( if hasattr(choice, "message") and hasattr(choice.message, "content"): output_tokens += count_tokens(choice.message.content) + set_on_span = ( + span.set_attribute if isinstance(span, StreamedSpan) else span.set_data + ) + # Do not set token data if it is 0 input_tokens = input_tokens or None + if input_tokens is not None: + set_on_span(SPANDATA.GEN_AI_USAGE_INPUT_TOKENS, input_tokens) + input_tokens_cached = input_tokens_cached or None + if input_tokens_cached is not None: + set_on_span(SPANDATA.GEN_AI_USAGE_INPUT_TOKENS_CACHED, input_tokens_cached) + output_tokens = output_tokens or None + if output_tokens is not None: + set_on_span(SPANDATA.GEN_AI_USAGE_OUTPUT_TOKENS, output_tokens) + output_tokens_reasoning = output_tokens_reasoning or None + if output_tokens_reasoning is not None: + set_on_span( + SPANDATA.GEN_AI_USAGE_OUTPUT_TOKENS_REASONING, + output_tokens_reasoning, + ) + total_tokens = total_tokens or None + if total_tokens is None and input_tokens is not None and output_tokens is not None: + total_tokens = input_tokens + output_tokens - record_token_usage( - span, - input_tokens=input_tokens, - input_tokens_cached=input_tokens_cached, - output_tokens=output_tokens, - output_tokens_reasoning=output_tokens_reasoning, - total_tokens=total_tokens, - ) + if total_tokens is not None: + set_on_span(SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS, total_tokens) def _calculate_responses_token_usage( @@ -317,21 +331,36 @@ def _calculate_responses_token_usage( if hasattr(content_item, "text"): output_tokens += count_tokens(content_item.text) + set_on_span = ( + span.set_attribute if isinstance(span, StreamedSpan) else span.set_data + ) + # Do not set token data if it is 0 input_tokens = input_tokens or None + if input_tokens is not None: + set_on_span(SPANDATA.GEN_AI_USAGE_INPUT_TOKENS, input_tokens) + input_tokens_cached = input_tokens_cached or None + if input_tokens_cached is not None: + set_on_span(SPANDATA.GEN_AI_USAGE_INPUT_TOKENS_CACHED, input_tokens_cached) + output_tokens = output_tokens or None + if output_tokens is not None: + set_on_span(SPANDATA.GEN_AI_USAGE_OUTPUT_TOKENS, output_tokens) + output_tokens_reasoning = output_tokens_reasoning or None + if output_tokens_reasoning is not None: + set_on_span( + SPANDATA.GEN_AI_USAGE_OUTPUT_TOKENS_REASONING, + output_tokens_reasoning, + ) + total_tokens = total_tokens or None + if total_tokens is None and input_tokens is not None and output_tokens is not None: + total_tokens = input_tokens + output_tokens - record_token_usage( - span, - input_tokens=input_tokens, - input_tokens_cached=input_tokens_cached, - output_tokens=output_tokens, - output_tokens_reasoning=output_tokens_reasoning, - total_tokens=total_tokens, - ) + if total_tokens is not None: + set_on_span(SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS, total_tokens) def _set_responses_api_input_data( From 44318e61228e0ff03646b5a31ad2639bec8e751c Mon Sep 17 00:00:00 2001 From: Alex Alderman Webb Date: Wed, 7 Oct 2026 16:05:33 +0200 Subject: [PATCH 2/4] Update sentry_sdk/ai/monitoring.py Co-authored-by: Ivana Kellyer --- sentry_sdk/ai/monitoring.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/sentry_sdk/ai/monitoring.py b/sentry_sdk/ai/monitoring.py index f58c49ac26..232d2bb5f0 100644 --- a/sentry_sdk/ai/monitoring.py +++ b/sentry_sdk/ai/monitoring.py @@ -194,7 +194,7 @@ def record_token_usage( ) -> None: warnings.warn( "record_token_usage() is deprecated and will be removed in version 3.0 of sentry-sdk. " - "Use the manual span API instead, e.g. Span.set_data().", + "Use the manual span API instead, e.g. span.set_attribute() (in streaming mode) or span.set_data().", DeprecationWarning, stacklevel=2, ) From 436b9e1b2c42515c173720c7f2c6e35261288b7c Mon Sep 17 00:00:00 2001 From: Alexander Alderman Webb Date: Wed, 7 Oct 2026 16:14:06 +0200 Subject: [PATCH 3/4] restore hasattr(res.meta, billed_units) --- sentry_sdk/integrations/cohere.py | 39 ++++++++++++++++--------------- 1 file changed, 20 insertions(+), 19 deletions(-) diff --git a/sentry_sdk/integrations/cohere.py b/sentry_sdk/integrations/cohere.py index 6fcc1f4d2a..0b1d1ebef4 100644 --- a/sentry_sdk/integrations/cohere.py +++ b/sentry_sdk/integrations/cohere.py @@ -137,27 +137,28 @@ def collect_chat_response_fields( span.set_attribute if isinstance(span, StreamedSpan) else span.set_data ) - if res.meta.billed_units.input_tokens is not None: - set_on_span( - SPANDATA.GEN_AI_USAGE_INPUT_TOKENS, - res.meta.billed_units.input_tokens, - ) + if hasattr(res.meta, "billed_units"): + if res.meta.billed_units.input_tokens is not None: + set_on_span( + SPANDATA.GEN_AI_USAGE_INPUT_TOKENS, + res.meta.billed_units.input_tokens, + ) - if res.meta.billed_units.output_tokens is not None: - set_on_span( - SPANDATA.GEN_AI_USAGE_OUTPUT_TOKENS, - res.meta.billed_units.output_tokens, - ) + if res.meta.billed_units.output_tokens is not None: + set_on_span( + SPANDATA.GEN_AI_USAGE_OUTPUT_TOKENS, + res.meta.billed_units.output_tokens, + ) - if ( - res.meta.billed_units.input_tokens is not None - and res.meta.billed_units.output_tokens is not None - ): - set_on_span( - SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS, - res.meta.billed_units.input_tokens - + res.meta.billed_units.output_tokens, - ) + if ( + res.meta.billed_units.input_tokens is not None + and res.meta.billed_units.output_tokens is not None + ): + set_on_span( + SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS, + res.meta.billed_units.input_tokens + + res.meta.billed_units.output_tokens, + ) elif hasattr(res.meta, "tokens"): if res.meta.tokens.input_tokens is not None: set_on_span( From cb7a639bb32f4b41d5ba3d9ce3d1a42c2c7c6b7d Mon Sep 17 00:00:00 2001 From: Alexander Alderman Webb Date: Thu, 8 Oct 2026 10:09:06 +0200 Subject: [PATCH 4/4] add huggingface_hub --- sentry_sdk/integrations/huggingface_hub.py | 79 +++++++++++++++------- 1 file changed, 55 insertions(+), 24 deletions(-) diff --git a/sentry_sdk/integrations/huggingface_hub.py b/sentry_sdk/integrations/huggingface_hub.py index 5980c298c0..b2e6b3ef71 100644 --- a/sentry_sdk/integrations/huggingface_hub.py +++ b/sentry_sdk/integrations/huggingface_hub.py @@ -4,7 +4,6 @@ from typing import TYPE_CHECKING, cast import sentry_sdk -from sentry_sdk.ai.monitoring import record_token_usage from sentry_sdk.ai.utils import ( _set_span_data_attribute, get_start_span_function, @@ -13,7 +12,6 @@ from sentry_sdk.consts import OP, SPANDATA from sentry_sdk.integrations import DidNotEnable, Integration from sentry_sdk.scope import should_send_default_pii -from sentry_sdk.traces import StreamedSpan from sentry_sdk.tracing_utils import has_span_streaming_enabled from sentry_sdk.utils import ( capture_internal_exceptions, @@ -23,13 +21,12 @@ ) if TYPE_CHECKING: - from typing import Any, Callable, Iterable, Union + from typing import Any, Callable, Iterable from huggingface_hub import ( ChatCompletionStreamOutput, ) - from sentry_sdk.tracing import Span try: import huggingface_hub.inference._client @@ -97,7 +94,6 @@ def new_huggingface_task(*args: "Any", **kwargs: "Any") -> "Any": model = hf_client.model or kwargs.get("model") or "" operation_name = op.split(".")[-1] - span: "Union[Span, StreamedSpan]" if has_span_streaming_enabled(client.options): span = sentry_sdk.traces.start_span( name=f"{operation_name} {model}", @@ -106,12 +102,16 @@ def new_huggingface_task(*args: "Any", **kwargs: "Any") -> "Any": "sentry.origin": HuggingfaceHubIntegration.origin, }, ) + + set_on_span = span.set_attribute else: span = get_start_span_function()( op=op, name=f"{operation_name} {model}", origin=HuggingfaceHubIntegration.origin, ) + + set_on_span = span.set_data span.__enter__() _set_span_data_attribute(span, SPANDATA.GEN_AI_OPERATION_NAME, operation_name) @@ -258,16 +258,30 @@ def new_huggingface_task(*args: "Any", **kwargs: "Any") -> "Any": ) if usage is not None: - record_token_usage( - span, - input_tokens=usage.prompt_tokens, - output_tokens=usage.completion_tokens, - total_tokens=usage.total_tokens, - ) + if usage is not None and usage.prompt_tokens is not None: + set_on_span(SPANDATA.GEN_AI_USAGE_INPUT_TOKENS, usage.prompt_tokens) + + if usage is not None and usage.completion_tokens is not None: + set_on_span( + SPANDATA.GEN_AI_USAGE_OUTPUT_TOKENS, usage.completion_tokens + ) + + if usage is not None and usage.total_tokens is not None: + set_on_span(SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS, usage.total_tokens) + + elif ( + usage is not None + and usage.prompt_tokens is not None + and usage.completion_tokens is not None + ): + set_on_span( + SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS, + usage.prompt_tokens + usage.completion_tokens, + ) elif tokens_used > 0: - record_token_usage( - span, - total_tokens=tokens_used, + set_on_span( + SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS, + tokens_used, ) # If the response is not a generator (meaning a streaming response) @@ -331,10 +345,7 @@ def new_details_iterator() -> "Iterable[Any]": ) if tokens_used > 0: - record_token_usage( - span, - total_tokens=tokens_used, - ) + set_on_span(SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS, tokens_used) span.__exit__(None, None, None) @@ -443,12 +454,32 @@ def new_iterator() -> "Iterable[ChatCompletionStreamOutput]": text_response, ) - if usage is not None: - record_token_usage( - span, - input_tokens=usage.prompt_tokens, - output_tokens=usage.completion_tokens, - total_tokens=usage.total_tokens, + if usage is None: + span.__exit__(None, None, None) + return + + if usage.prompt_tokens is not None: + set_on_span( + SPANDATA.GEN_AI_USAGE_INPUT_TOKENS, usage.prompt_tokens + ) + + if usage.completion_tokens is not None: + set_on_span( + SPANDATA.GEN_AI_USAGE_OUTPUT_TOKENS, + usage.completion_tokens, + ) + + if usage.total_tokens is not None: + set_on_span( + SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS, usage.total_tokens + ) + elif ( + usage.prompt_tokens is not None + and usage.completion_tokens is not None + ): + set_on_span( + SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS, + usage.prompt_tokens + usage.completion_tokens, ) span.__exit__(None, None, None)