diff --git a/instrumentation/opentelemetry-instrumentation-genai-anthropic/.changelog/779.added b/instrumentation/opentelemetry-instrumentation-genai-anthropic/.changelog/779.added new file mode 100644 index 000000000..e4127fd6d --- /dev/null +++ b/instrumentation/opentelemetry-instrumentation-genai-anthropic/.changelog/779.added @@ -0,0 +1 @@ +Capture reasoning token counts `gen_ai.usage.reasoning.output_tokens` on messages diff --git a/instrumentation/opentelemetry-instrumentation-genai-anthropic/src/opentelemetry/instrumentation/genai/anthropic/messages_extractors.py b/instrumentation/opentelemetry-instrumentation-genai-anthropic/src/opentelemetry/instrumentation/genai/anthropic/messages_extractors.py index 25c2bbbb4..e161c3c4f 100644 --- a/instrumentation/opentelemetry-instrumentation-genai-anthropic/src/opentelemetry/instrumentation/genai/anthropic/messages_extractors.py +++ b/instrumentation/opentelemetry-instrumentation-genai-anthropic/src/opentelemetry/instrumentation/genai/anthropic/messages_extractors.py @@ -73,6 +73,7 @@ class MessageRequestParams: class UsageTokens: input_tokens: int | None = None output_tokens: int | None = None + thinking_tokens: int | None = None cache_creation_input_tokens: int | None = None cache_read_input_tokens: int | None = None @@ -85,6 +86,8 @@ def extract_usage_tokens( input_tokens = usage.input_tokens output_tokens = usage.output_tokens + output_tokens_details = getattr(usage, "output_tokens_details", None) + thinking_tokens = getattr(output_tokens_details, "thinking_tokens", None) cache_creation_input_tokens = usage.cache_creation_input_tokens cache_read_input_tokens = usage.cache_read_input_tokens @@ -104,6 +107,7 @@ def extract_usage_tokens( return UsageTokens( input_tokens=total_input_tokens, output_tokens=output_tokens, + thinking_tokens=thinking_tokens, cache_creation_input_tokens=cache_creation_input_tokens, cache_read_input_tokens=cache_read_input_tokens, ) @@ -223,6 +227,7 @@ def set_invocation_response_attributes( tokens = extract_usage_tokens(message.usage) invocation.input_tokens = tokens.input_tokens invocation.output_tokens = tokens.output_tokens + invocation.thinking_tokens = tokens.thinking_tokens invocation.cache_write_input_tokens = tokens.cache_creation_input_tokens invocation.cache_read_input_tokens = tokens.cache_read_input_tokens diff --git a/instrumentation/opentelemetry-instrumentation-genai-anthropic/tests/cassettes/test_async_messages_create_captures_thinking_content.yaml b/instrumentation/opentelemetry-instrumentation-genai-anthropic/tests/cassettes/test_async_messages_create_captures_thinking_content.yaml index 52b952a17..23107f47b 100644 --- a/instrumentation/opentelemetry-instrumentation-genai-anthropic/tests/cassettes/test_async_messages_create_captures_thinking_content.yaml +++ b/instrumentation/opentelemetry-instrumentation-genai-anthropic/tests/cassettes/test_async_messages_create_captures_thinking_content.yaml @@ -1,3 +1,4 @@ +# TODO: this is generated by AI, re-record interactions: - request: body: |- @@ -87,6 +88,9 @@ interactions: "ephemeral_1h_input_tokens": 0 }, "output_tokens": 425, + "output_tokens_details": { + "thinking_tokens": 286 + }, "service_tier": "standard", "inference_geo": "not_available" } diff --git a/instrumentation/opentelemetry-instrumentation-genai-anthropic/tests/cassettes/test_sync_messages_create_captures_thinking_content.yaml b/instrumentation/opentelemetry-instrumentation-genai-anthropic/tests/cassettes/test_sync_messages_create_captures_thinking_content.yaml index 846cc9dca..70ac29589 100644 --- a/instrumentation/opentelemetry-instrumentation-genai-anthropic/tests/cassettes/test_sync_messages_create_captures_thinking_content.yaml +++ b/instrumentation/opentelemetry-instrumentation-genai-anthropic/tests/cassettes/test_sync_messages_create_captures_thinking_content.yaml @@ -1,3 +1,4 @@ +# TODO: this is generated by AI, re-record interactions: - request: body: |- @@ -86,6 +87,9 @@ interactions: "ephemeral_1h_input_tokens": 0 }, "output_tokens": 417, + "output_tokens_details": { + "thinking_tokens": 281 + }, "service_tier": "standard", "inference_geo": "not_available" } diff --git a/instrumentation/opentelemetry-instrumentation-genai-anthropic/tests/test_async_messages.py b/instrumentation/opentelemetry-instrumentation-genai-anthropic/tests/test_async_messages.py index 37826da40..39d84a4b5 100644 --- a/instrumentation/opentelemetry-instrumentation-genai-anthropic/tests/test_async_messages.py +++ b/instrumentation/opentelemetry-instrumentation-genai-anthropic/tests/test_async_messages.py @@ -21,7 +21,7 @@ from anthropic._response import AsyncAPIResponse as LegacyAPIResponse from anthropic._streaming import AsyncStream as AnthropicAsyncStream from anthropic.resources.messages import AsyncMessages as _AsyncMessages -from anthropic.types import Message +from anthropic.types import Message, Usage from opentelemetry.instrumentation.genai.anthropic import ( AnthropicInstrumentor, @@ -52,6 +52,11 @@ _create_params = set(inspect.signature(_AsyncMessages.create).parameters) _has_tools_param = "tools" in _create_params _has_thinking_param = "thinking" in _create_params +# Pydantic 2 exposes model fields via model_fields; Pydantic 1 uses __fields__. +_usage_fields = getattr(Usage, "model_fields", None) +if _usage_fields is None: + _usage_fields = getattr(Usage, "__fields__", {}) +_has_output_tokens_details = "output_tokens_details" in _usage_fields async def _parse_raw_response(raw_response, **kwargs): @@ -954,7 +959,7 @@ async def test_async_messages_create_captures_thinking_content( model = "claude-sonnet-4-20250514" messages = [{"role": "user", "content": "What is 17*19? Think first."}] - await async_anthropic_client.messages.create( + response = await async_anthropic_client.messages.create( model=model, max_tokens=16000, messages=messages, @@ -973,6 +978,16 @@ async def test_async_messages_create_captures_thinking_content( for message in output_messages for part in message.get("parts", []) ) + if _has_output_tokens_details: + assert response.usage.output_tokens_details is not None + thinking_tokens = response.usage.output_tokens_details.thinking_tokens + assert thinking_tokens > 0 + assert ( + span.attributes[ + GenAIAttributes.GEN_AI_USAGE_REASONING_OUTPUT_TOKENS + ] + == thinking_tokens + ) @pytest.mark.asyncio diff --git a/instrumentation/opentelemetry-instrumentation-genai-anthropic/tests/test_async_wrappers.py b/instrumentation/opentelemetry-instrumentation-genai-anthropic/tests/test_async_wrappers.py index c5fe17f16..c5c9c86dc 100644 --- a/instrumentation/opentelemetry-instrumentation-genai-anthropic/tests/test_async_wrappers.py +++ b/instrumentation/opentelemetry-instrumentation-genai-anthropic/tests/test_async_wrappers.py @@ -11,6 +11,10 @@ MessagesStreamManagerWrapper, MessagesStreamWrapper, ) +from opentelemetry.semconv._incubating.attributes import ( + gen_ai_attributes as GenAIAttributes, +) +from opentelemetry.util.genai.handler import TelemetryHandler def _make_invocation(): @@ -173,6 +177,63 @@ async def aclose(self): self.aclose_calls += 1 +@pytest.mark.parametrize( + ("wrapper_type", "stream_type"), + [ + (MessagesStreamWrapper, _FakeSyncStream), + (AsyncMessagesStreamWrapper, _FakeAsyncStream), + ], +) +def test_stream_wrapper_finalization_records_thinking_tokens( + wrapper_type, + stream_type, + tracer_provider, + logger_provider, + meter_provider, + span_exporter, +): + handler = TelemetryHandler( + tracer_provider=tracer_provider, + logger_provider=logger_provider, + meter_provider=meter_provider, + instrumentation_scope_name=( + "opentelemetry.instrumentation.genai.anthropic" + ), + ) + invocation = handler.start_inference( + provider="anthropic", + request_model="claude-sonnet-4-20250514", + ) + snapshot = SimpleNamespace( + id="msg_stream", + model="claude-sonnet-4-20250514", + stop_reason="end_turn", + usage=SimpleNamespace( + input_tokens=10, + output_tokens=20, + output_tokens_details=SimpleNamespace(thinking_tokens=12), + cache_creation_input_tokens=0, + cache_read_input_tokens=0, + ), + ) + wrapper = wrapper_type( + stream=stream_type(current_message_snapshot=snapshot), + invocation=invocation, + capture_content=False, + ) + + wrapper._stop() + + spans = span_exporter.get_finished_spans() + assert len(spans) == 1 + assert ( + spans[0].attributes[ + GenAIAttributes.GEN_AI_USAGE_REASONING_OUTPUT_TOKENS + ] + == 12 + ) + + def test_sync_stream_wrapper_exit_closes_without_exception(): stream = _FakeSyncStream() wrapper = _make_stream_wrapper(stream) diff --git a/instrumentation/opentelemetry-instrumentation-genai-anthropic/tests/test_messages_extractors.py b/instrumentation/opentelemetry-instrumentation-genai-anthropic/tests/test_messages_extractors.py index be7eb4f24..3eb56a8ee 100644 --- a/instrumentation/opentelemetry-instrumentation-genai-anthropic/tests/test_messages_extractors.py +++ b/instrumentation/opentelemetry-instrumentation-genai-anthropic/tests/test_messages_extractors.py @@ -460,7 +460,7 @@ def test_invalid_document_source_is_ignored(source): ) -def test_set_invocation_response_attributes_records_cache_tokens(): +def test_set_invocation_response_attributes_records_usage_tokens(): invocation = MagicMock() message = SimpleNamespace( id="msg_123", @@ -469,6 +469,7 @@ def test_set_invocation_response_attributes_records_cache_tokens(): usage=SimpleNamespace( input_tokens=10, output_tokens=20, + output_tokens_details=SimpleNamespace(thinking_tokens=12), cache_creation_input_tokens=15, cache_read_input_tokens=5, ), @@ -478,10 +479,32 @@ def test_set_invocation_response_attributes_records_cache_tokens(): ) assert invocation.input_tokens == 30 assert invocation.output_tokens == 20 + assert invocation.thinking_tokens == 12 assert invocation.cache_write_input_tokens == 15 assert invocation.cache_read_input_tokens == 5 +def test_set_invocation_response_attributes_handles_missing_token_details(): + invocation = MagicMock() + message = SimpleNamespace( + id="msg_123", + model="claude-3-7-sonnet-20250219", + stop_reason="end_turn", + usage=SimpleNamespace( + input_tokens=10, + output_tokens=20, + cache_creation_input_tokens=0, + cache_read_input_tokens=0, + ), + ) + + set_invocation_response_attributes( + invocation, message, capture_content=False + ) + + assert invocation.thinking_tokens is None + + def test_convert_server_tool_use_block(): part = _convert_content_block_to_part( ServerToolUseBlock( diff --git a/instrumentation/opentelemetry-instrumentation-genai-anthropic/tests/test_sync_messages.py b/instrumentation/opentelemetry-instrumentation-genai-anthropic/tests/test_sync_messages.py index 0f91c01bb..ec9e837e9 100644 --- a/instrumentation/opentelemetry-instrumentation-genai-anthropic/tests/test_sync_messages.py +++ b/instrumentation/opentelemetry-instrumentation-genai-anthropic/tests/test_sync_messages.py @@ -63,6 +63,10 @@ _create_params = set(inspect.signature(_Messages.create).parameters) _has_tools_param = "tools" in _create_params _has_thinking_param = "thinking" in _create_params +_usage_fields = getattr(Usage, "model_fields", None) +if _usage_fields is None: + _usage_fields = getattr(Usage, "__fields__", {}) +_has_output_tokens_details = "output_tokens_details" in _usage_fields @pytest.mark.parametrize( @@ -1423,7 +1427,7 @@ def test_sync_messages_create_captures_thinking_content( model = "claude-sonnet-4-20250514" messages = [{"role": "user", "content": "What is 17*19? Think first."}] - anthropic_client.messages.create( + response = anthropic_client.messages.create( model=model, max_tokens=16000, messages=messages, @@ -1442,6 +1446,16 @@ def test_sync_messages_create_captures_thinking_content( for message in output_messages for part in message.get("parts", []) ) + if _has_output_tokens_details: + assert response.usage.output_tokens_details is not None + thinking_tokens = response.usage.output_tokens_details.thinking_tokens + assert thinking_tokens > 0 + assert ( + span.attributes[ + GenAIAttributes.GEN_AI_USAGE_REASONING_OUTPUT_TOKENS + ] + == thinking_tokens + ) @pytest.mark.vcr()