From a81017564c83b2536d7a03b78b2016a07124282c Mon Sep 17 00:00:00 2001 From: Alexander Alderman Webb Date: Tue, 6 Oct 2026 10:06:42 +0200 Subject: [PATCH 1/8] test(openai): Stop directly calling _calculate_completions_token_usage() --- tests/integrations/openai/test_openai.py | 764 +++++++++++++---------- 1 file changed, 430 insertions(+), 334 deletions(-) diff --git a/tests/integrations/openai/test_openai.py b/tests/integrations/openai/test_openai.py index 96866886ec..0e515d5ab6 100644 --- a/tests/integrations/openai/test_openai.py +++ b/tests/integrations/openai/test_openai.py @@ -27,6 +27,15 @@ from openai.types.chat.chat_completion_chunk import ChoiceDelta from openai.types.create_embedding_response import Usage as EmbeddingTokenUsage +try: + from openai.types.completion_usage import ( + CompletionTokensDetails, + PromptTokensDetails, + ) +except ImportError: + CompletionTokensDetails = None + PromptTokensDetails = None + try: from openai.types.chat import ( ChatCompletionCustomToolParam, @@ -65,11 +74,7 @@ from sentry_sdk import start_transaction from sentry_sdk.consts import OP, SPANDATA -from sentry_sdk.integrations.openai import ( - OpenAIIntegration, - _calculate_completions_token_usage, - _calculate_responses_token_usage, -) +from sentry_sdk.integrations.openai import OpenAIIntegration from sentry_sdk.integrations.stdlib import StdlibIntegration from sentry_sdk.utils import safe_serialize @@ -4676,366 +4681,457 @@ async def test_span_origin_embeddings_async( assert event["spans"][0]["origin"] == "auto.ai.openai" -def test_completions_token_usage_from_response(): - """Token counts are extracted from response.usage using Completions API field names.""" - span = mock.MagicMock() - - def count_tokens(msg): - return len(str(msg)) - - response = mock.MagicMock() - response.usage = mock.MagicMock() - response.usage.completion_tokens = 10 - response.usage.prompt_tokens = 20 - response.usage.total_tokens = 30 - messages = [] - streaming_message_responses = [] - - with mock.patch( - "sentry_sdk.integrations.openai.record_token_usage" - ) as mock_record_token_usage: - _calculate_completions_token_usage( - messages=messages, - response=response, - span=span, - streaming_message_responses=streaming_message_responses, - streaming_message_total_token_usage=None, - count_tokens=count_tokens, - ) - mock_record_token_usage.assert_called_once_with( - span, - input_tokens=20, - input_tokens_cached=None, - output_tokens=10, - output_tokens_reasoning=None, - total_tokens=30, - ) +@pytest.mark.skipif( + OPENAI_VERSION is None or OPENAI_VERSION < (1, 51, 0), + reason="Previous versions do not expose cached input tokens. See https://github.com/openai/openai-python/commit/7c8c11158c4e0b63fef495c32447d6e31870073f.", +) +@pytest.mark.parametrize("span_streaming", [True, False]) +@pytest.mark.parametrize("stream_gen_ai_spans", [True, False]) +def test_completions_token_usage_with_detailed_fields( + sentry_init, + capture_events, + capture_items, + nonstreaming_chat_completions_model_response, + get_model_response, + span_streaming, + stream_gen_ai_spans, +): + """Cached and reasoning token counts are extracted from prompt_tokens_details and completion_tokens_details.""" + sentry_init( + integrations=[OpenAIIntegration()], + disabled_integrations=[StdlibIntegration], + traces_sample_rate=1.0, + stream_gen_ai_spans=stream_gen_ai_spans, + trace_lifecycle="stream" if span_streaming else "static", + ) + client = OpenAI(api_key="z") + returned_stream = get_model_response( + nonstreaming_chat_completions_model_response( + response_id="chat-id", + response_model="gpt-3.5-turbo", + message_content="the model response", + created=10000000, + usage=CompletionUsage( + prompt_tokens=20, + prompt_tokens_details=PromptTokensDetails(cached_tokens=5), + completion_tokens=10, + completion_tokens_details=CompletionTokensDetails(reasoning_tokens=8), + total_tokens=30, + ), + ), + serialize_pydantic=True, + ) -def test_completions_token_usage_with_detailed_fields(): - """Cached and reasoning token counts are extracted from prompt_tokens_details and completion_tokens_details.""" - span = mock.MagicMock() - - def count_tokens(msg): - return len(str(msg)) - - response = mock.MagicMock() - response.usage = mock.MagicMock() - response.usage.prompt_tokens = 20 - response.usage.prompt_tokens_details = mock.MagicMock() - response.usage.prompt_tokens_details.cached_tokens = 5 - response.usage.completion_tokens = 10 - response.usage.completion_tokens_details = mock.MagicMock() - response.usage.completion_tokens_details.reasoning_tokens = 8 - response.usage.total_tokens = 30 - - with mock.patch( - "sentry_sdk.integrations.openai.record_token_usage" - ) as mock_record_token_usage: - _calculate_completions_token_usage( - messages=[], - response=response, - span=span, - streaming_message_responses=[], - streaming_message_total_token_usage=None, - count_tokens=count_tokens, - ) - mock_record_token_usage.assert_called_once_with( - span, - input_tokens=20, - input_tokens_cached=5, - output_tokens=10, - output_tokens_reasoning=8, - total_tokens=30, - ) + if span_streaming or stream_gen_ai_spans: + items = capture_items("span") + + with mock.patch.object( + client.chat._client._client, + "send", + return_value=returned_stream, + ), start_transaction(name="openai tx"): + client.chat.completions.create( + model="some-model", + messages=[{"role": "user", "content": "hello"}], + ) + + sentry_sdk.flush() + (span,) = (item.payload for item in items) + + assert span["attributes"][SPANDATA.GEN_AI_USAGE_INPUT_TOKENS] == 20 + assert span["attributes"][SPANDATA.GEN_AI_USAGE_INPUT_TOKENS_CACHED] == 5 + assert span["attributes"][SPANDATA.GEN_AI_USAGE_OUTPUT_TOKENS] == 10 + assert span["attributes"][SPANDATA.GEN_AI_USAGE_OUTPUT_TOKENS_REASONING] == 8 + assert span["attributes"][SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS] == 30 + else: + events = capture_events() + + with mock.patch.object( + client.chat._client._client, + "send", + return_value=returned_stream, + ), start_transaction(name="openai tx"): + client.chat.completions.create( + model="some-model", + messages=[{"role": "user", "content": "hello"}], + ) + + (event,) = events + (span,) = event["spans"] + + assert span["data"][SPANDATA.GEN_AI_USAGE_INPUT_TOKENS] == 20 + assert span["data"][SPANDATA.GEN_AI_USAGE_INPUT_TOKENS_CACHED] == 5 + assert span["data"][SPANDATA.GEN_AI_USAGE_OUTPUT_TOKENS] == 10 + assert span["data"][SPANDATA.GEN_AI_USAGE_OUTPUT_TOKENS_REASONING] == 8 + assert span["data"][SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS] == 30 -def test_completions_token_usage_manual_input_counting(): +@pytest.mark.parametrize("span_streaming", [True, False]) +@pytest.mark.parametrize("stream_gen_ai_spans", [True, False]) +def test_completions_token_usage_manual_input_counting( + sentry_init, + capture_events, + capture_items, + nonstreaming_chat_completions_model_response, + get_model_response, + span_streaming, + stream_gen_ai_spans, +): """When prompt_tokens is missing, input tokens are counted manually from messages.""" - span = mock.MagicMock() - - def count_tokens(msg): - return len(str(msg)) - - response = mock.MagicMock() - response.usage = mock.MagicMock() - response.usage.completion_tokens = 10 - response.usage.total_tokens = 10 - messages = [ - {"content": "one"}, - {"content": "two"}, - {"content": "three"}, - ] - streaming_message_responses = [] - - with mock.patch( - "sentry_sdk.integrations.openai.record_token_usage" - ) as mock_record_token_usage: - _calculate_completions_token_usage( - messages=messages, - response=response, - span=span, - streaming_message_responses=streaming_message_responses, - streaming_message_total_token_usage=None, - count_tokens=count_tokens, - ) - mock_record_token_usage.assert_called_once_with( - span, - input_tokens=11, - input_tokens_cached=None, - output_tokens=10, - output_tokens_reasoning=None, - total_tokens=10, - ) + sentry_init( + integrations=[ + OpenAIIntegration(tiktoken_encoding_name=tiktoken_encoding_if_installed()) + ], + disabled_integrations=[StdlibIntegration], + traces_sample_rate=1.0, + stream_gen_ai_spans=stream_gen_ai_spans, + trace_lifecycle="stream" if span_streaming else "static", + ) + + client = OpenAI(api_key="z") + returned_stream = get_model_response( + nonstreaming_chat_completions_model_response( + response_id="chat-id", + response_model="gpt-3.5-turbo", + message_content="the model response", + created=10000000, + usage=CompletionUsage( + prompt_tokens=0, + completion_tokens=10, + total_tokens=10, + ), + ), + serialize_pydantic=True, + ) + if span_streaming or stream_gen_ai_spans: + items = capture_items("span") -def test_completions_token_usage_manual_output_counting_streaming(): - """When completion_tokens is missing, output tokens are counted from streaming responses.""" - span = mock.MagicMock() + with mock.patch.object( + client.chat._client._client, + "send", + return_value=returned_stream, + ), start_transaction(name="openai tx"): + client.chat.completions.create( + model="some-model", + messages=[ + {"role": "user", "content": "one"}, + {"role": "user", "content": "two"}, + {"role": "user", "content": "three"}, + ], + ) - def count_tokens(msg): - return len(str(msg)) + sentry_sdk.flush() + (span,) = (item.payload for item in items) - response = mock.MagicMock() - response.usage = mock.MagicMock() - response.usage.prompt_tokens = 20 - response.usage.total_tokens = 20 - messages = [] - streaming_message_responses = [ - "one", - "two", - "three", - ] + assert span["attributes"][SPANDATA.GEN_AI_USAGE_OUTPUT_TOKENS] == 10 + assert span["attributes"][SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS] == 10 + if tiktoken_encoding_if_installed(): + assert span["attributes"][SPANDATA.GEN_AI_USAGE_INPUT_TOKENS] == 3 - with mock.patch( - "sentry_sdk.integrations.openai.record_token_usage" - ) as mock_record_token_usage: - _calculate_completions_token_usage( - messages=messages, - response=response, - span=span, - streaming_message_responses=streaming_message_responses, - streaming_message_total_token_usage=None, - count_tokens=count_tokens, - ) - mock_record_token_usage.assert_called_once_with( - span, - input_tokens=20, - input_tokens_cached=None, - output_tokens=11, - output_tokens_reasoning=None, - total_tokens=20, + else: + events = capture_events() + + with mock.patch.object( + client.chat._client._client, + "send", + return_value=returned_stream, + ), start_transaction(name="openai tx"): + client.chat.completions.create( + model="some-model", + messages=[ + {"role": "user", "content": "one"}, + {"role": "user", "content": "two"}, + {"role": "user", "content": "three"}, + ], + ) + + (event,) = events + (span,) = event["spans"] + + assert span["data"][SPANDATA.GEN_AI_USAGE_OUTPUT_TOKENS] == 10 + assert span["data"][SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS] == 10 + if tiktoken_encoding_if_installed(): + assert span["data"][SPANDATA.GEN_AI_USAGE_INPUT_TOKENS] == 3 + + +@pytest.mark.parametrize("span_streaming", [True, False]) +@pytest.mark.parametrize("stream_gen_ai_spans", [True, False]) +@pytest.mark.skipif( + OPENAI_VERSION is None or OPENAI_VERSION < (1, 26, 0), + reason="Previous versions do not report token usage when streaming. See https://github.com/openai/openai-python/commit/6cc515874f5f4b26b35f408d6afc3c14b4dfe3b0.", +) +def test_completions_token_usage_from_streaming_response( + sentry_init, + capture_events, + capture_items, + get_model_response, + server_side_event_chunks, + streaming_chat_completions_model_response, + span_streaming, + stream_gen_ai_spans, +): + """Token counts are extracted from the final chunk's usage when streaming.""" + sentry_init( + integrations=[OpenAIIntegration()], + disabled_integrations=[StdlibIntegration], + traces_sample_rate=1.0, + stream_gen_ai_spans=stream_gen_ai_spans, + trace_lifecycle="stream" if span_streaming else "static", + ) + + client = OpenAI(api_key="z") + returned_stream = get_model_response( + server_side_event_chunks( + streaming_chat_completions_model_response, + include_event_type=False, ) + ) + if span_streaming or stream_gen_ai_spans: + items = capture_items("span") -def test_completions_token_usage_manual_output_counting_choices(): + with mock.patch.object( + client.chat._client._client, + "send", + return_value=returned_stream, + ), start_transaction(name="openai tx"): + response_stream = client.chat.completions.create( + model="some-model", + messages=[{"role": "user", "content": "hello"}], + stream=True, + ) + for _ in response_stream: + pass + + sentry_sdk.flush() + (span,) = (item.payload for item in items) + + assert span["attributes"][SPANDATA.GEN_AI_USAGE_INPUT_TOKENS] == 10 + assert span["attributes"][SPANDATA.GEN_AI_USAGE_OUTPUT_TOKENS] == 20 + assert span["attributes"][SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS] == 30 + + else: + events = capture_events() + + with mock.patch.object( + client.chat._client._client, + "send", + return_value=returned_stream, + ), start_transaction(name="openai tx"): + response_stream = client.chat.completions.create( + model="some-model", + messages=[{"role": "user", "content": "hello"}], + stream=True, + ) + for _ in response_stream: + pass + + (event,) = events + (span,) = event["spans"] + + assert span["data"][SPANDATA.GEN_AI_USAGE_INPUT_TOKENS] == 10 + assert span["data"][SPANDATA.GEN_AI_USAGE_OUTPUT_TOKENS] == 20 + assert span["data"][SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS] == 30 + + +@pytest.mark.parametrize("span_streaming", [True, False]) +@pytest.mark.parametrize("stream_gen_ai_spans", [True, False]) +def test_completions_token_usage_manual_output_counting_choices( + sentry_init, + capture_events, + capture_items, + get_model_response, + span_streaming, + stream_gen_ai_spans, +): """When completion_tokens is missing, output tokens are counted from response.choices.""" - span = mock.MagicMock() - - def count_tokens(msg): - return len(str(msg)) - - response = mock.MagicMock() - response.usage = mock.MagicMock() - response.usage.prompt_tokens = 20 - response.usage.total_tokens = 20 - response.choices = [ - Choice( - index=0, - finish_reason="stop", - message=ChatCompletionMessage(role="assistant", content="one"), - ), - Choice( - index=1, - finish_reason="stop", - message=ChatCompletionMessage(role="assistant", content="two"), - ), - Choice( - index=2, - finish_reason="stop", - message=ChatCompletionMessage(role="assistant", content="three"), - ), - ] - messages = [] - streaming_message_responses = None - - with mock.patch( - "sentry_sdk.integrations.openai.record_token_usage" - ) as mock_record_token_usage: - _calculate_completions_token_usage( - messages=messages, - response=response, - span=span, - streaming_message_responses=streaming_message_responses, - streaming_message_total_token_usage=None, - count_tokens=count_tokens, - ) - mock_record_token_usage.assert_called_once_with( - span, - input_tokens=20, - input_tokens_cached=None, - output_tokens=11, - output_tokens_reasoning=None, - total_tokens=20, - ) + sentry_init( + integrations=[ + OpenAIIntegration(tiktoken_encoding_name=tiktoken_encoding_if_installed()) + ], + disabled_integrations=[StdlibIntegration], + traces_sample_rate=1.0, + stream_gen_ai_spans=stream_gen_ai_spans, + trace_lifecycle="stream" if span_streaming else "static", + ) + client = OpenAI(api_key="z") + returned_stream = get_model_response( + ChatCompletion( + id="chat-id", + choices=[ + Choice( + index=0, + finish_reason="stop", + message=ChatCompletionMessage(role="assistant", content="one"), + ), + Choice( + index=1, + finish_reason="stop", + message=ChatCompletionMessage(role="assistant", content="two"), + ), + Choice( + index=2, + finish_reason="stop", + message=ChatCompletionMessage(role="assistant", content="three"), + ), + ], + created=10000000, + model="gpt-3.5-turbo", + object="chat.completion", + usage=CompletionUsage( + prompt_tokens=20, + completion_tokens=0, + total_tokens=20, + ), + ), + serialize_pydantic=True, + ) -def test_completions_token_usage_no_usage_data(): - """When response has no usage data and no streaming responses, all tokens are None.""" - span = mock.MagicMock() + if span_streaming or stream_gen_ai_spans: + items = capture_items("span") - def count_tokens(msg): - return len(str(msg)) + with mock.patch.object( + client.chat._client._client, + "send", + return_value=returned_stream, + ), start_transaction(name="openai tx"): + client.chat.completions.create( + model="some-model", + messages=[{"role": "user", "content": "hello"}], + ) - response = mock.MagicMock() - messages = [] - streaming_message_responses = None + sentry_sdk.flush() + (span,) = (item.payload for item in items) - with mock.patch( - "sentry_sdk.integrations.openai.record_token_usage" - ) as mock_record_token_usage: - _calculate_completions_token_usage( - messages=messages, - response=response, - span=span, - streaming_message_responses=streaming_message_responses, - streaming_message_total_token_usage=None, - count_tokens=count_tokens, - ) - mock_record_token_usage.assert_called_once_with( - span, - input_tokens=None, - input_tokens_cached=None, - output_tokens=None, - output_tokens_reasoning=None, - total_tokens=None, - ) + assert span["attributes"][SPANDATA.GEN_AI_USAGE_INPUT_TOKENS] == 20 + assert span["attributes"][SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS] == 20 + if tiktoken_encoding_if_installed(): + assert span["attributes"][SPANDATA.GEN_AI_USAGE_OUTPUT_TOKENS] == 3 + else: + events = capture_events() -@pytest.mark.skipif(SKIP_RESPONSES_TESTS, reason="Responses API not available") -def test_responses_token_usage_from_response(): - """Token counts including cached and reasoning tokens are extracted from Responses API.""" - span = mock.MagicMock() - - def count_tokens(msg): - return len(str(msg)) - - response = mock.MagicMock() - response.usage = mock.MagicMock() - response.usage.input_tokens = 20 - response.usage.input_tokens_details = mock.MagicMock() - response.usage.input_tokens_details.cached_tokens = 5 - response.usage.output_tokens = 10 - response.usage.output_tokens_details = mock.MagicMock() - response.usage.output_tokens_details.reasoning_tokens = 8 - response.usage.total_tokens = 30 - input = [] - - with mock.patch( - "sentry_sdk.integrations.openai.record_token_usage" - ) as mock_record_token_usage: - _calculate_responses_token_usage(input, response, span, None, count_tokens) - mock_record_token_usage.assert_called_once_with( - span, - input_tokens=20, - input_tokens_cached=5, - output_tokens=10, - output_tokens_reasoning=8, - total_tokens=30, - ) + with mock.patch.object( + client.chat._client._client, + "send", + return_value=returned_stream, + ), start_transaction(name="openai tx"): + client.chat.completions.create( + model="some-model", + messages=[{"role": "user", "content": "hello"}], + ) + (event,) = events + (span,) = event["spans"] -@pytest.mark.skipif(SKIP_RESPONSES_TESTS, reason="Responses API not available") -def test_responses_token_usage_no_usage_data(): - """When Responses API response has no usage data, all tokens are None.""" - span = mock.MagicMock() - - def count_tokens(msg): - return len(str(msg)) - - response = mock.MagicMock() - response.usage = None - input = [] - streaming_message_responses = None - - with mock.patch( - "sentry_sdk.integrations.openai.record_token_usage" - ) as mock_record_token_usage: - _calculate_responses_token_usage( - input, response, span, streaming_message_responses, count_tokens - ) - mock_record_token_usage.assert_called_once_with( - span, - input_tokens=None, - input_tokens_cached=None, - output_tokens=None, - output_tokens_reasoning=None, - total_tokens=None, - ) + assert span["data"][SPANDATA.GEN_AI_USAGE_INPUT_TOKENS] == 20 + assert span["data"][SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS] == 20 + if tiktoken_encoding_if_installed(): + assert span["data"][SPANDATA.GEN_AI_USAGE_OUTPUT_TOKENS] == 3 +@pytest.mark.parametrize("span_streaming", [True, False]) +@pytest.mark.parametrize("stream_gen_ai_spans", [True, False]) @pytest.mark.skipif(SKIP_RESPONSES_TESTS, reason="Responses API not available") -def test_responses_token_usage_manual_output_counting_response_output(): +def test_responses_token_usage_manual_output_counting_response_output( + sentry_init, + capture_events, + capture_items, + get_model_response, + span_streaming, + stream_gen_ai_spans, +): """When output_tokens is missing, output tokens are counted from response.output.""" - span = mock.MagicMock() - - def count_tokens(msg): - return len(str(msg)) - - response = mock.MagicMock() - response.usage = mock.MagicMock() - response.usage.input_tokens = 20 - response.usage.total_tokens = 20 - response.output = [ - ResponseOutputMessage( - id="msg-1", - content=[ - ResponseOutputText( - annotations=[], - text="one", - type="output_text", - ), + sentry_init( + integrations=[ + OpenAIIntegration(tiktoken_encoding_name=tiktoken_encoding_if_installed()) + ], + disabled_integrations=[StdlibIntegration], + traces_sample_rate=1.0, + stream_gen_ai_spans=stream_gen_ai_spans, + trace_lifecycle="stream" if span_streaming else "static", + ) + + client = OpenAI(api_key="z") + returned_stream = get_model_response( + Response( + id="chat-id", + output=[ + ResponseOutputMessage( + id="message-id", + content=[ + ResponseOutputText( + annotations=[], + text=content, + type="output_text", + ) + ], + role="assistant", + status="completed", + type="message", + ) + for content in ("one", "two", "three") ], - role="assistant", - status="completed", - type="message", - ), - ResponseOutputMessage( - id="msg-2", - content=[ - ResponseOutputText( - annotations=[], - text="two", - type="output_text", + parallel_tool_calls=False, + tool_choice="none", + tools=[], + created_at=10000000, + model="response-model-id", + object="response", + usage=ResponseUsage( + input_tokens=20, + input_tokens_details=InputTokensDetails( + cached_tokens=0, + cache_write_tokens=0, ), - ResponseOutputText( - annotations=[], - text="three", - type="output_text", + output_tokens=0, + output_tokens_details=OutputTokensDetails( + reasoning_tokens=0, ), - ], - role="assistant", - status="completed", - type="message", + total_tokens=20, + ), ), - ] - input = [] - streaming_message_responses = None - - with mock.patch( - "sentry_sdk.integrations.openai.record_token_usage" - ) as mock_record_token_usage: - _calculate_responses_token_usage( - input, response, span, streaming_message_responses, count_tokens - ) - mock_record_token_usage.assert_called_once_with( - span, - input_tokens=20, - input_tokens_cached=None, - output_tokens=11, - output_tokens_reasoning=None, - total_tokens=20, - ) + serialize_pydantic=True, + ) + + if span_streaming or stream_gen_ai_spans: + items = capture_items("span") + + with mock.patch.object( + client.responses._client._client, + "send", + return_value=returned_stream, + ), start_transaction(name="openai tx"): + client.responses.create(model="gpt-4o", input="hello") + + sentry_sdk.flush() + (span,) = (item.payload for item in items) + + assert span["attributes"][SPANDATA.GEN_AI_USAGE_INPUT_TOKENS] == 20 + assert span["attributes"][SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS] == 20 + if tiktoken_encoding_if_installed(): + assert span["attributes"][SPANDATA.GEN_AI_USAGE_OUTPUT_TOKENS] == 3 + + else: + events = capture_events() + + with mock.patch.object( + client.responses._client._client, + "send", + return_value=returned_stream, + ), start_transaction(name="openai tx"): + client.responses.create(model="gpt-4o", input="hello") + + (event,) = events + (span,) = event["spans"] + + assert span["data"][SPANDATA.GEN_AI_USAGE_INPUT_TOKENS] == 20 + assert span["data"][SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS] == 20 + if tiktoken_encoding_if_installed(): + assert span["data"][SPANDATA.GEN_AI_USAGE_OUTPUT_TOKENS] == 3 @pytest.mark.parametrize("span_streaming", [True, False]) From 2c7ae6553cf9b17a94f7f4594a111e658a72c3dd Mon Sep 17 00:00:00 2001 From: Alexander Alderman Webb Date: Tue, 6 Oct 2026 14:48:10 +0200 Subject: [PATCH 2/8] Use nonstreaming_responses_model_response fixture --- tests/conftest.py | 68 +++--- .../integrations/langchain/test_langchain.py | 16 +- tests/integrations/openai/test_openai.py | 27 +-- .../openai_agents/test_openai_agents.py | 221 ++++++++++++++++-- 4 files changed, 258 insertions(+), 74 deletions(-) diff --git a/tests/conftest.py b/tests/conftest.py index 9145506be2..fdcc35ad95 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -1479,42 +1479,38 @@ def openai_embedding_model_response(): @pytest.fixture def nonstreaming_responses_model_response(): - return openai.types.responses.Response( - id="resp_123", - output=[ - openai.types.responses.ResponseOutputMessage( - id="msg_123", - type="message", - status="completed", - content=[ - openai.types.responses.ResponseOutputText( - text="Hello, how can I help you?", - type="output_text", - annotations=[], - ) - ], - role="assistant", - ) - ], - parallel_tool_calls=False, - tool_choice="none", - tools=[], - created_at=10000000, - model="gpt-4", - object="response", - usage=openai.types.responses.ResponseUsage( - input_tokens=10, - input_tokens_details=openai.types.responses.response_usage.InputTokensDetails( - cached_tokens=4, - cache_write_tokens=6, - ), - output_tokens=20, - output_tokens_details=openai.types.responses.response_usage.OutputTokensDetails( - reasoning_tokens=5, - ), - total_tokens=30, - ), - ) + def inner( + message_contents, + usage, + ): + return openai.types.responses.Response( + id="resp_123", + output=[ + openai.types.responses.ResponseOutputMessage( + id="msg_123", + type="message", + status="completed", + content=[ + openai.types.responses.ResponseOutputText( + text=message_content, + type="output_text", + annotations=[], + ) + ], + role="assistant", + ) + for message_content in message_contents + ], + parallel_tool_calls=False, + tool_choice="none", + tools=[], + created_at=10000000, + model="gpt-4", + object="response", + usage=usage, + ) + + return inner @pytest.fixture diff --git a/tests/integrations/langchain/test_langchain.py b/tests/integrations/langchain/test_langchain.py index a6fa363744..fe3886ce02 100644 --- a/tests/integrations/langchain/test_langchain.py +++ b/tests/integrations/langchain/test_langchain.py @@ -752,7 +752,21 @@ def test_langchain_create_agent( ) model_response = get_model_response( - nonstreaming_responses_model_response, + nonstreaming_responses_model_response( + message_contents=("Hello, how can I help you?",), + usage=ResponseUsage( + input_tokens=10, + input_tokens_details=InputTokensDetails( + cached_tokens=4, + cache_write_tokens=6, + ), + output_tokens=20, + output_tokens_details=OutputTokensDetails( + reasoning_tokens=5, + ), + total_tokens=30, + ), + ), serialize_pydantic=True, request_headers={ "X-Stainless-Raw-Response": "True", diff --git a/tests/integrations/openai/test_openai.py b/tests/integrations/openai/test_openai.py index 0e515d5ab6..d60c62fd59 100644 --- a/tests/integrations/openai/test_openai.py +++ b/tests/integrations/openai/test_openai.py @@ -5041,6 +5041,7 @@ def test_responses_token_usage_manual_output_counting_response_output( capture_events, capture_items, get_model_response, + nonstreaming_responses_model_response, span_streaming, stream_gen_ai_spans, ): @@ -5057,30 +5058,8 @@ def test_responses_token_usage_manual_output_counting_response_output( client = OpenAI(api_key="z") returned_stream = get_model_response( - Response( - id="chat-id", - output=[ - ResponseOutputMessage( - id="message-id", - content=[ - ResponseOutputText( - annotations=[], - text=content, - type="output_text", - ) - ], - role="assistant", - status="completed", - type="message", - ) - for content in ("one", "two", "three") - ], - parallel_tool_calls=False, - tool_choice="none", - tools=[], - created_at=10000000, - model="response-model-id", - object="response", + nonstreaming_responses_model_response( + message_contents=("one", "two", "three"), usage=ResponseUsage( input_tokens=20, input_tokens_details=InputTokensDetails( diff --git a/tests/integrations/openai_agents/test_openai_agents.py b/tests/integrations/openai_agents/test_openai_agents.py index 8bc49a2c9f..b36436592b 100644 --- a/tests/integrations/openai_agents/test_openai_agents.py +++ b/tests/integrations/openai_agents/test_openai_agents.py @@ -390,7 +390,22 @@ def some_function(a: str, b: list[int]) -> str: agent = test_agent.clone(model=model, tools=tools) response = get_model_response( - nonstreaming_responses_model_response, serialize_pydantic=True + nonstreaming_responses_model_response( + message_contents=("Hello, how can I help you?",), + usage=ResponseUsage( + input_tokens=10, + input_tokens_details=InputTokensDetails( + cached_tokens=4, + cache_write_tokens=6, + ), + output_tokens=20, + output_tokens_details=OutputTokensDetails( + reasoning_tokens=5, + ), + total_tokens=30, + ), + ), + serialize_pydantic=True, ) if span_streaming: @@ -513,7 +528,22 @@ async def test_agent_invocation_span_no_pii( agent = test_agent.clone(model=model) response = get_model_response( - nonstreaming_responses_model_response, serialize_pydantic=True + nonstreaming_responses_model_response( + message_contents=("Hello, how can I help you?",), + usage=ResponseUsage( + input_tokens=10, + input_tokens_details=InputTokensDetails( + cached_tokens=4, + cache_write_tokens=6, + ), + output_tokens=20, + output_tokens_details=OutputTokensDetails( + reasoning_tokens=5, + ), + total_tokens=30, + ), + ), + serialize_pydantic=True, ) if span_streaming: @@ -764,7 +794,22 @@ async def test_invoke_agent_span_data_collection_inputs( agent = test_agent.clone(model=model) response = get_model_response( - nonstreaming_responses_model_response, serialize_pydantic=True + nonstreaming_responses_model_response( + message_contents=("Hello, how can I help you?",), + usage=ResponseUsage( + input_tokens=10, + input_tokens_details=InputTokensDetails( + cached_tokens=4, + cache_write_tokens=6, + ), + output_tokens=20, + output_tokens_details=OutputTokensDetails( + reasoning_tokens=5, + ), + total_tokens=30, + ), + ), + serialize_pydantic=True, ) system_message = { @@ -942,7 +987,22 @@ async def test_invoke_agent_span_data_collection_outputs( agent = test_agent.clone(model=model) response = get_model_response( - nonstreaming_responses_model_response, serialize_pydantic=True + nonstreaming_responses_model_response( + message_contents=("Hello, how can I help you?",), + usage=ResponseUsage( + input_tokens=10, + input_tokens_details=InputTokensDetails( + cached_tokens=4, + cache_write_tokens=6, + ), + output_tokens=20, + output_tokens_details=OutputTokensDetails( + reasoning_tokens=5, + ), + total_tokens=30, + ), + ), + serialize_pydantic=True, ) if span_streaming: @@ -1107,7 +1167,22 @@ async def test_data_collection_inputs( agent = test_agent.clone(model=model, tools=[simple_test_tool]) response = get_model_response( - nonstreaming_responses_model_response, serialize_pydantic=True + nonstreaming_responses_model_response( + message_contents=("Hello, how can I help you?",), + usage=ResponseUsage( + input_tokens=10, + input_tokens_details=InputTokensDetails( + cached_tokens=4, + cache_write_tokens=6, + ), + output_tokens=20, + output_tokens_details=OutputTokensDetails( + reasoning_tokens=5, + ), + total_tokens=30, + ), + ), + serialize_pydantic=True, ) init_kwargs = { @@ -1554,7 +1629,22 @@ async def test_agent_invocation_span( agent = test_agent_with_instructions(instructions).clone(model=model) response = get_model_response( - nonstreaming_responses_model_response, serialize_pydantic=True + nonstreaming_responses_model_response( + message_contents=("Hello, how can I help you?",), + usage=ResponseUsage( + input_tokens=10, + input_tokens_details=InputTokensDetails( + cached_tokens=4, + cache_write_tokens=6, + ), + output_tokens=20, + output_tokens_details=OutputTokensDetails( + reasoning_tokens=5, + ), + total_tokens=30, + ), + ), + serialize_pydantic=True, ) if span_streaming: @@ -1787,7 +1877,22 @@ async def test_client_span_custom_model( agent = test_agent_custom_model.clone(model=model) response = get_model_response( - nonstreaming_responses_model_response, serialize_pydantic=True + nonstreaming_responses_model_response( + message_contents=("Hello, how can I help you?",), + usage=ResponseUsage( + input_tokens=10, + input_tokens_details=InputTokensDetails( + cached_tokens=4, + cache_write_tokens=6, + ), + output_tokens=20, + output_tokens_details=OutputTokensDetails( + reasoning_tokens=5, + ), + total_tokens=30, + ), + ), + serialize_pydantic=True, ) if span_streaming or stream_gen_ai_spans: @@ -1870,7 +1975,22 @@ def test_agent_invocation_span_sync_no_pii( agent = test_agent.clone(model=model) response = get_model_response( - nonstreaming_responses_model_response, serialize_pydantic=True + nonstreaming_responses_model_response( + message_contents=("Hello, how can I help you?",), + usage=ResponseUsage( + input_tokens=10, + input_tokens_details=InputTokensDetails( + cached_tokens=4, + cache_write_tokens=6, + ), + output_tokens=20, + output_tokens_details=OutputTokensDetails( + reasoning_tokens=5, + ), + total_tokens=30, + ), + ), + serialize_pydantic=True, ) if span_streaming: @@ -2262,7 +2382,22 @@ def test_agent_invocation_span_sync( agent = test_agent_with_instructions(instructions).clone(model=model) response = get_model_response( - nonstreaming_responses_model_response, serialize_pydantic=True + nonstreaming_responses_model_response( + message_contents=("Hello, how can I help you?",), + usage=ResponseUsage( + input_tokens=10, + input_tokens_details=InputTokensDetails( + cached_tokens=4, + cache_write_tokens=6, + ), + output_tokens=20, + output_tokens_details=OutputTokensDetails( + reasoning_tokens=5, + ), + total_tokens=30, + ), + ), + serialize_pydantic=True, ) if span_streaming: @@ -5210,7 +5345,22 @@ async def test_multiple_agents_asyncio( agent = test_agent.clone(model=model) response = get_model_response( - nonstreaming_responses_model_response, serialize_pydantic=True + nonstreaming_responses_model_response( + message_contents=("Hello, how can I help you?",), + usage=ResponseUsage( + input_tokens=10, + input_tokens_details=InputTokensDetails( + cached_tokens=4, + cache_write_tokens=6, + ), + output_tokens=20, + output_tokens_details=OutputTokensDetails( + reasoning_tokens=5, + ), + total_tokens=30, + ), + ), + serialize_pydantic=True, ) if span_streaming: @@ -6805,7 +6955,22 @@ async def test_conversation_id_on_all_spans( agent = test_agent.clone(model=model) response = get_model_response( - nonstreaming_responses_model_response, serialize_pydantic=True + nonstreaming_responses_model_response( + message_contents=("Hello, how can I help you?",), + usage=ResponseUsage( + input_tokens=10, + input_tokens_details=InputTokensDetails( + cached_tokens=4, + cache_write_tokens=6, + ), + output_tokens=20, + output_tokens_details=OutputTokensDetails( + reasoning_tokens=5, + ), + total_tokens=30, + ), + ), + serialize_pydantic=True, ) if span_streaming: @@ -7202,7 +7367,22 @@ async def test_no_conversation_id_when_not_provided( agent = test_agent.clone(model=model) response = get_model_response( - nonstreaming_responses_model_response, serialize_pydantic=True + nonstreaming_responses_model_response( + message_contents=("Hello, how can I help you?",), + usage=ResponseUsage( + input_tokens=10, + input_tokens_details=InputTokensDetails( + cached_tokens=4, + cache_write_tokens=6, + ), + output_tokens=20, + output_tokens_details=OutputTokensDetails( + reasoning_tokens=5, + ), + total_tokens=30, + ), + ), + serialize_pydantic=True, ) if span_streaming: @@ -7341,7 +7521,22 @@ async def test_runner_run_with_starting_agent_kwarg( agent = test_agent.clone(model=model) response = get_model_response( - nonstreaming_responses_model_response, serialize_pydantic=True + nonstreaming_responses_model_response( + message_contents=("Hello, how can I help you?",), + usage=ResponseUsage( + input_tokens=10, + input_tokens_details=InputTokensDetails( + cached_tokens=4, + cache_write_tokens=6, + ), + output_tokens=20, + output_tokens_details=OutputTokensDetails( + reasoning_tokens=5, + ), + total_tokens=30, + ), + ), + serialize_pydantic=True, ) with patch.object( From b4f368825c69c250e56d58667c3febf89f6d0a3c Mon Sep 17 00:00:00 2001 From: Alexander Alderman Webb Date: Tue, 6 Oct 2026 14:57:33 +0200 Subject: [PATCH 3/8] add types --- tests/conftest.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/tests/conftest.py b/tests/conftest.py index fdcc35ad95..0c1ab3005e 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -1480,8 +1480,8 @@ def openai_embedding_model_response(): @pytest.fixture def nonstreaming_responses_model_response(): def inner( - message_contents, - usage, + message_contents: "Iterator[str]", + usage: "Iterator[openai.types.responses.ResponseUsage]", ): return openai.types.responses.Response( id="resp_123", From 3139530c2dfaa4700d7959c9ca560884673e50b9 Mon Sep 17 00:00:00 2001 From: Alexander Alderman Webb Date: Tue, 6 Oct 2026 15:06:39 +0200 Subject: [PATCH 4/8] fix model input --- tests/integrations/openai/test_openai.py | 14 ++++++-------- 1 file changed, 6 insertions(+), 8 deletions(-) diff --git a/tests/integrations/openai/test_openai.py b/tests/integrations/openai/test_openai.py index d60c62fd59..f9ea04f035 100644 --- a/tests/integrations/openai/test_openai.py +++ b/tests/integrations/openai/test_openai.py @@ -4816,9 +4816,8 @@ def test_completions_token_usage_manual_input_counting( client.chat.completions.create( model="some-model", messages=[ - {"role": "user", "content": "one"}, - {"role": "user", "content": "two"}, - {"role": "user", "content": "three"}, + {"role": "system", "content": "You are a helpful assistant."}, + {"role": "user", "content": "hello"}, ], ) @@ -4828,7 +4827,7 @@ def test_completions_token_usage_manual_input_counting( assert span["attributes"][SPANDATA.GEN_AI_USAGE_OUTPUT_TOKENS] == 10 assert span["attributes"][SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS] == 10 if tiktoken_encoding_if_installed(): - assert span["attributes"][SPANDATA.GEN_AI_USAGE_INPUT_TOKENS] == 3 + assert span["attributes"][SPANDATA.GEN_AI_USAGE_INPUT_TOKENS] == 7 else: events = capture_events() @@ -4841,9 +4840,8 @@ def test_completions_token_usage_manual_input_counting( client.chat.completions.create( model="some-model", messages=[ - {"role": "user", "content": "one"}, - {"role": "user", "content": "two"}, - {"role": "user", "content": "three"}, + {"role": "system", "content": "You are a helpful assistant."}, + {"role": "user", "content": "hello"}, ], ) @@ -4853,7 +4851,7 @@ def test_completions_token_usage_manual_input_counting( assert span["data"][SPANDATA.GEN_AI_USAGE_OUTPUT_TOKENS] == 10 assert span["data"][SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS] == 10 if tiktoken_encoding_if_installed(): - assert span["data"][SPANDATA.GEN_AI_USAGE_INPUT_TOKENS] == 3 + assert span["data"][SPANDATA.GEN_AI_USAGE_INPUT_TOKENS] == 7 @pytest.mark.parametrize("span_streaming", [True, False]) From 7a440e42b976c9feba5ce6699a6e482f09745c5a Mon Sep 17 00:00:00 2001 From: Alexander Alderman Webb Date: Tue, 6 Oct 2026 15:15:44 +0200 Subject: [PATCH 5/8] parametrize tomkens in streaming fixture --- tests/conftest.py | 207 +++++++++++---------- tests/integrations/litellm/test_litellm.py | 16 +- tests/integrations/openai/test_openai.py | 30 ++- 3 files changed, 138 insertions(+), 115 deletions(-) diff --git a/tests/conftest.py b/tests/conftest.py index 0c1ab3005e..dbef15fc85 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -1317,116 +1317,117 @@ def inner(request_headers=None): @pytest.fixture def streaming_chat_completions_model_response(): - return [ - openai.types.chat.ChatCompletionChunk( - id="chatcmpl-test", - object="chat.completion.chunk", - created=10000000, - model="gpt-3.5-turbo", - choices=[ - openai.types.chat.chat_completion_chunk.Choice( - index=0, - delta=openai.types.chat.chat_completion_chunk.ChoiceDelta( - role="assistant" + def inner( + usage: "openai.types.CompletionUsage", + ): + return [ + openai.types.chat.ChatCompletionChunk( + id="chatcmpl-test", + object="chat.completion.chunk", + created=10000000, + model="gpt-3.5-turbo", + choices=[ + openai.types.chat.chat_completion_chunk.Choice( + index=0, + delta=openai.types.chat.chat_completion_chunk.ChoiceDelta( + role="assistant" + ), + finish_reason=None, ), - finish_reason=None, - ), - ], - ), - openai.types.chat.ChatCompletionChunk( - id="chatcmpl-test", - object="chat.completion.chunk", - created=10000000, - model="gpt-3.5-turbo", - choices=[ - openai.types.chat.chat_completion_chunk.Choice( - index=0, - delta=openai.types.chat.chat_completion_chunk.ChoiceDelta( - content="Tes" + ], + ), + openai.types.chat.ChatCompletionChunk( + id="chatcmpl-test", + object="chat.completion.chunk", + created=10000000, + model="gpt-3.5-turbo", + choices=[ + openai.types.chat.chat_completion_chunk.Choice( + index=0, + delta=openai.types.chat.chat_completion_chunk.ChoiceDelta( + content="Tes" + ), + finish_reason=None, ), - finish_reason=None, - ), - ], - ), - openai.types.chat.ChatCompletionChunk( - id="chatcmpl-test", - object="chat.completion.chunk", - created=10000000, - model="gpt-3.5-turbo", - choices=[ - openai.types.chat.chat_completion_chunk.Choice( - index=0, - delta=openai.types.chat.chat_completion_chunk.ChoiceDelta( - content="t r" + ], + ), + openai.types.chat.ChatCompletionChunk( + id="chatcmpl-test", + object="chat.completion.chunk", + created=10000000, + model="gpt-3.5-turbo", + choices=[ + openai.types.chat.chat_completion_chunk.Choice( + index=0, + delta=openai.types.chat.chat_completion_chunk.ChoiceDelta( + content="t r" + ), + finish_reason=None, ), - finish_reason=None, - ), - ], - ), - openai.types.chat.ChatCompletionChunk( - id="chatcmpl-test", - object="chat.completion.chunk", - created=10000000, - model="gpt-3.5-turbo", - choices=[ - openai.types.chat.chat_completion_chunk.Choice( - index=0, - delta=openai.types.chat.chat_completion_chunk.ChoiceDelta( - content="esp" + ], + ), + openai.types.chat.ChatCompletionChunk( + id="chatcmpl-test", + object="chat.completion.chunk", + created=10000000, + model="gpt-3.5-turbo", + choices=[ + openai.types.chat.chat_completion_chunk.Choice( + index=0, + delta=openai.types.chat.chat_completion_chunk.ChoiceDelta( + content="esp" + ), + finish_reason=None, ), - finish_reason=None, - ), - ], - ), - openai.types.chat.ChatCompletionChunk( - id="chatcmpl-test", - object="chat.completion.chunk", - created=10000000, - model="gpt-3.5-turbo", - choices=[ - openai.types.chat.chat_completion_chunk.Choice( - index=0, - delta=openai.types.chat.chat_completion_chunk.ChoiceDelta( - content="ons" + ], + ), + openai.types.chat.ChatCompletionChunk( + id="chatcmpl-test", + object="chat.completion.chunk", + created=10000000, + model="gpt-3.5-turbo", + choices=[ + openai.types.chat.chat_completion_chunk.Choice( + index=0, + delta=openai.types.chat.chat_completion_chunk.ChoiceDelta( + content="ons" + ), + finish_reason=None, ), - finish_reason=None, - ), - ], - ), - openai.types.chat.ChatCompletionChunk( - id="chatcmpl-test", - object="chat.completion.chunk", - created=10000000, - model="gpt-3.5-turbo", - choices=[ - openai.types.chat.chat_completion_chunk.Choice( - index=0, - delta=openai.types.chat.chat_completion_chunk.ChoiceDelta( - content="e" + ], + ), + openai.types.chat.ChatCompletionChunk( + id="chatcmpl-test", + object="chat.completion.chunk", + created=10000000, + model="gpt-3.5-turbo", + choices=[ + openai.types.chat.chat_completion_chunk.Choice( + index=0, + delta=openai.types.chat.chat_completion_chunk.ChoiceDelta( + content="e" + ), + finish_reason=None, ), - finish_reason=None, - ), - ], - ), - openai.types.chat.ChatCompletionChunk( - id="chatcmpl-test", - object="chat.completion.chunk", - created=10000000, - model="gpt-3.5-turbo", - choices=[ - openai.types.chat.chat_completion_chunk.Choice( - index=0, - delta=openai.types.chat.chat_completion_chunk.ChoiceDelta(), - finish_reason="stop", - ), - ], - usage=openai.types.CompletionUsage( - prompt_tokens=10, - completion_tokens=20, - total_tokens=30, + ], ), - ), - ] + openai.types.chat.ChatCompletionChunk( + id="chatcmpl-test", + object="chat.completion.chunk", + created=10000000, + model="gpt-3.5-turbo", + choices=[ + openai.types.chat.chat_completion_chunk.Choice( + index=0, + delta=openai.types.chat.chat_completion_chunk.ChoiceDelta(), + finish_reason="stop", + ), + ], + usage=usage, + ), + ] + + return inner @pytest.fixture diff --git a/tests/integrations/litellm/test_litellm.py b/tests/integrations/litellm/test_litellm.py index a5d5c2ae9f..076ee6d7f8 100644 --- a/tests/integrations/litellm/test_litellm.py +++ b/tests/integrations/litellm/test_litellm.py @@ -596,7 +596,13 @@ def test_streaming_chat_completion( model_response = get_model_response( server_side_event_chunks( - streaming_chat_completions_model_response, + streaming_chat_completions_model_response( + usage=CompletionUsage( + prompt_tokens=10, + completion_tokens=20, + total_tokens=30, + ), + ), include_event_type=False, ), request_headers={"X-Stainless-Raw-Response": "true"}, @@ -713,7 +719,13 @@ async def test_async_streaming_chat_completion( model_response = get_model_response( async_iterator( server_side_event_chunks( - streaming_chat_completions_model_response, + streaming_chat_completions_model_response( + usage=CompletionUsage( + prompt_tokens=10, + completion_tokens=20, + total_tokens=30, + ), + ), include_event_type=False, ), ), diff --git a/tests/integrations/openai/test_openai.py b/tests/integrations/openai/test_openai.py index f9ea04f035..241504f776 100644 --- a/tests/integrations/openai/test_openai.py +++ b/tests/integrations/openai/test_openai.py @@ -4860,7 +4860,7 @@ def test_completions_token_usage_manual_input_counting( OPENAI_VERSION is None or OPENAI_VERSION < (1, 26, 0), reason="Previous versions do not report token usage when streaming. See https://github.com/openai/openai-python/commit/6cc515874f5f4b26b35f408d6afc3c14b4dfe3b0.", ) -def test_completions_token_usage_from_streaming_response( +def test_completions_token_usage_manual_output_counting_streaming( sentry_init, capture_events, capture_items, @@ -4870,9 +4870,11 @@ def test_completions_token_usage_from_streaming_response( span_streaming, stream_gen_ai_spans, ): - """Token counts are extracted from the final chunk's usage when streaming.""" + """When completion_tokens is missing, output tokens are counted from streamed content.""" sentry_init( - integrations=[OpenAIIntegration()], + integrations=[ + OpenAIIntegration(tiktoken_encoding_name=tiktoken_encoding_if_installed()) + ], disabled_integrations=[StdlibIntegration], traces_sample_rate=1.0, stream_gen_ai_spans=stream_gen_ai_spans, @@ -4882,7 +4884,13 @@ def test_completions_token_usage_from_streaming_response( client = OpenAI(api_key="z") returned_stream = get_model_response( server_side_event_chunks( - streaming_chat_completions_model_response, + streaming_chat_completions_model_response( + usage=CompletionUsage( + prompt_tokens=20, + completion_tokens=0, + total_tokens=20, + ) + ), include_event_type=False, ) ) @@ -4906,9 +4914,10 @@ def test_completions_token_usage_from_streaming_response( sentry_sdk.flush() (span,) = (item.payload for item in items) - assert span["attributes"][SPANDATA.GEN_AI_USAGE_INPUT_TOKENS] == 10 - assert span["attributes"][SPANDATA.GEN_AI_USAGE_OUTPUT_TOKENS] == 20 - assert span["attributes"][SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS] == 30 + assert span["attributes"][SPANDATA.GEN_AI_USAGE_INPUT_TOKENS] == 20 + assert span["attributes"][SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS] == 20 + if tiktoken_encoding_if_installed(): + assert span["attributes"][SPANDATA.GEN_AI_USAGE_OUTPUT_TOKENS] == 2 else: events = capture_events() @@ -4929,9 +4938,10 @@ def test_completions_token_usage_from_streaming_response( (event,) = events (span,) = event["spans"] - assert span["data"][SPANDATA.GEN_AI_USAGE_INPUT_TOKENS] == 10 - assert span["data"][SPANDATA.GEN_AI_USAGE_OUTPUT_TOKENS] == 20 - assert span["data"][SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS] == 30 + assert span["data"][SPANDATA.GEN_AI_USAGE_INPUT_TOKENS] == 20 + assert span["data"][SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS] == 20 + if tiktoken_encoding_if_installed(): + assert span["data"][SPANDATA.GEN_AI_USAGE_OUTPUT_TOKENS] == 2 @pytest.mark.parametrize("span_streaming", [True, False]) From 2d16b1a59051c146f2f5ca1192581b3c5d2f62b6 Mon Sep 17 00:00:00 2001 From: Alexander Alderman Webb Date: Tue, 6 Oct 2026 15:20:47 +0200 Subject: [PATCH 6/8] . --- tests/conftest.py | 2 +- tests/integrations/openai/test_openai.py | 14 ++++++++------ 2 files changed, 9 insertions(+), 7 deletions(-) diff --git a/tests/conftest.py b/tests/conftest.py index dbef15fc85..ba40ae2761 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -1482,7 +1482,7 @@ def openai_embedding_model_response(): def nonstreaming_responses_model_response(): def inner( message_contents: "Iterator[str]", - usage: "Iterator[openai.types.responses.ResponseUsage]", + usage: "openai.types.responses.ResponseUsage", ): return openai.types.responses.Response( id="resp_123", diff --git a/tests/integrations/openai/test_openai.py b/tests/integrations/openai/test_openai.py index 241504f776..18d45f987f 100644 --- a/tests/integrations/openai/test_openai.py +++ b/tests/integrations/openai/test_openai.py @@ -4816,8 +4816,9 @@ def test_completions_token_usage_manual_input_counting( client.chat.completions.create( model="some-model", messages=[ - {"role": "system", "content": "You are a helpful assistant."}, - {"role": "user", "content": "hello"}, + {"content": "one"}, + {"content": "two"}, + {"content": "three"}, ], ) @@ -4827,7 +4828,7 @@ def test_completions_token_usage_manual_input_counting( assert span["attributes"][SPANDATA.GEN_AI_USAGE_OUTPUT_TOKENS] == 10 assert span["attributes"][SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS] == 10 if tiktoken_encoding_if_installed(): - assert span["attributes"][SPANDATA.GEN_AI_USAGE_INPUT_TOKENS] == 7 + assert span["attributes"][SPANDATA.GEN_AI_USAGE_INPUT_TOKENS] == 3 else: events = capture_events() @@ -4840,8 +4841,9 @@ def test_completions_token_usage_manual_input_counting( client.chat.completions.create( model="some-model", messages=[ - {"role": "system", "content": "You are a helpful assistant."}, - {"role": "user", "content": "hello"}, + {"content": "one"}, + {"content": "two"}, + {"content": "three"}, ], ) @@ -4851,7 +4853,7 @@ def test_completions_token_usage_manual_input_counting( assert span["data"][SPANDATA.GEN_AI_USAGE_OUTPUT_TOKENS] == 10 assert span["data"][SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS] == 10 if tiktoken_encoding_if_installed(): - assert span["data"][SPANDATA.GEN_AI_USAGE_INPUT_TOKENS] == 7 + assert span["data"][SPANDATA.GEN_AI_USAGE_INPUT_TOKENS] == 3 @pytest.mark.parametrize("span_streaming", [True, False]) From 584e80d75ed055fc51966f4bcb8784b192373c5b Mon Sep 17 00:00:00 2001 From: Alexander Alderman Webb Date: Tue, 6 Oct 2026 15:31:55 +0200 Subject: [PATCH 7/8] preserve existing response chunks --- tests/conftest.py | 92 +++++----------------- tests/integrations/litellm/test_litellm.py | 1 + tests/integrations/openai/test_openai.py | 7 +- 3 files changed, 23 insertions(+), 77 deletions(-) diff --git a/tests/conftest.py b/tests/conftest.py index ba40ae2761..8bd57cc3a9 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -1319,6 +1319,7 @@ def inner(request_headers=None): def streaming_chat_completions_model_response(): def inner( usage: "openai.types.CompletionUsage", + message_contents: "Iterator[str]", ): return [ openai.types.chat.ChatCompletionChunk( @@ -1336,81 +1337,24 @@ def inner( ), ], ), - openai.types.chat.ChatCompletionChunk( - id="chatcmpl-test", - object="chat.completion.chunk", - created=10000000, - model="gpt-3.5-turbo", - choices=[ - openai.types.chat.chat_completion_chunk.Choice( - index=0, - delta=openai.types.chat.chat_completion_chunk.ChoiceDelta( - content="Tes" - ), - finish_reason=None, - ), - ], - ), - openai.types.chat.ChatCompletionChunk( - id="chatcmpl-test", - object="chat.completion.chunk", - created=10000000, - model="gpt-3.5-turbo", - choices=[ - openai.types.chat.chat_completion_chunk.Choice( - index=0, - delta=openai.types.chat.chat_completion_chunk.ChoiceDelta( - content="t r" - ), - finish_reason=None, - ), - ], - ), - openai.types.chat.ChatCompletionChunk( - id="chatcmpl-test", - object="chat.completion.chunk", - created=10000000, - model="gpt-3.5-turbo", - choices=[ - openai.types.chat.chat_completion_chunk.Choice( - index=0, - delta=openai.types.chat.chat_completion_chunk.ChoiceDelta( - content="esp" - ), - finish_reason=None, - ), - ], - ), - openai.types.chat.ChatCompletionChunk( - id="chatcmpl-test", - object="chat.completion.chunk", - created=10000000, - model="gpt-3.5-turbo", - choices=[ - openai.types.chat.chat_completion_chunk.Choice( - index=0, - delta=openai.types.chat.chat_completion_chunk.ChoiceDelta( - content="ons" + *[ + openai.types.chat.ChatCompletionChunk( + id="chatcmpl-test", + object="chat.completion.chunk", + created=10000000, + model="gpt-3.5-turbo", + choices=[ + openai.types.chat.chat_completion_chunk.Choice( + index=0, + delta=openai.types.chat.chat_completion_chunk.ChoiceDelta( + content=message_content + ), + finish_reason=None, ), - finish_reason=None, - ), - ], - ), - openai.types.chat.ChatCompletionChunk( - id="chatcmpl-test", - object="chat.completion.chunk", - created=10000000, - model="gpt-3.5-turbo", - choices=[ - openai.types.chat.chat_completion_chunk.Choice( - index=0, - delta=openai.types.chat.chat_completion_chunk.ChoiceDelta( - content="e" - ), - finish_reason=None, - ), - ], - ), + ], + ) + for message_content in message_contents + ], openai.types.chat.ChatCompletionChunk( id="chatcmpl-test", object="chat.completion.chunk", diff --git a/tests/integrations/litellm/test_litellm.py b/tests/integrations/litellm/test_litellm.py index 076ee6d7f8..4947dcf903 100644 --- a/tests/integrations/litellm/test_litellm.py +++ b/tests/integrations/litellm/test_litellm.py @@ -725,6 +725,7 @@ async def test_async_streaming_chat_completion( completion_tokens=20, total_tokens=30, ), + message_contents=("Tes", "t r", "esp", "ons", "e"), ), include_event_type=False, ), diff --git a/tests/integrations/openai/test_openai.py b/tests/integrations/openai/test_openai.py index 18d45f987f..e578b03fdc 100644 --- a/tests/integrations/openai/test_openai.py +++ b/tests/integrations/openai/test_openai.py @@ -4891,7 +4891,8 @@ def test_completions_token_usage_manual_output_counting_streaming( prompt_tokens=20, completion_tokens=0, total_tokens=20, - ) + ), + message_contents=("one", " two", " three"), ), include_event_type=False, ) @@ -4919,7 +4920,7 @@ def test_completions_token_usage_manual_output_counting_streaming( assert span["attributes"][SPANDATA.GEN_AI_USAGE_INPUT_TOKENS] == 20 assert span["attributes"][SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS] == 20 if tiktoken_encoding_if_installed(): - assert span["attributes"][SPANDATA.GEN_AI_USAGE_OUTPUT_TOKENS] == 2 + assert span["attributes"][SPANDATA.GEN_AI_USAGE_OUTPUT_TOKENS] == 3 else: events = capture_events() @@ -4943,7 +4944,7 @@ def test_completions_token_usage_manual_output_counting_streaming( assert span["data"][SPANDATA.GEN_AI_USAGE_INPUT_TOKENS] == 20 assert span["data"][SPANDATA.GEN_AI_USAGE_TOTAL_TOKENS] == 20 if tiktoken_encoding_if_installed(): - assert span["data"][SPANDATA.GEN_AI_USAGE_OUTPUT_TOKENS] == 2 + assert span["data"][SPANDATA.GEN_AI_USAGE_OUTPUT_TOKENS] == 3 @pytest.mark.parametrize("span_streaming", [True, False]) From f80a3b6449fadb51b24872933a4fdf99f2b2e7a1 Mon Sep 17 00:00:00 2001 From: Alexander Alderman Webb Date: Tue, 6 Oct 2026 15:35:39 +0200 Subject: [PATCH 8/8] . --- tests/integrations/litellm/test_litellm.py | 1 + 1 file changed, 1 insertion(+) diff --git a/tests/integrations/litellm/test_litellm.py b/tests/integrations/litellm/test_litellm.py index 4947dcf903..e475eaf346 100644 --- a/tests/integrations/litellm/test_litellm.py +++ b/tests/integrations/litellm/test_litellm.py @@ -602,6 +602,7 @@ def test_streaming_chat_completion( completion_tokens=20, total_tokens=30, ), + message_contents=("Tes", "t r", "esp", "ons", "e"), ), include_event_type=False, ),