Skip to content

Commit 2eb8a2b

Browse files
breken-aidustinbyrne
authored andcommitted
fix(ai): capture usage and output of incomplete Responses streams
A Responses API stream cut short, for example by max_output_tokens, ends on response.incomplete instead of response.completed. The stream state only read usage and output from response.completed, so these generations were captured with no token counts and empty output even though the stop reason was set. Read them from every terminal event (completed, incomplete, failed), as the stop reason already does and as posthog-js does.
1 parent 824faf3 commit 2eb8a2b

3 files changed

Lines changed: 100 additions & 4 deletions

File tree

Lines changed: 5 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,5 @@
1+
---
2+
pypi/posthog: patch
3+
---
4+
5+
Capture token usage and output for OpenAI Responses streams that end incomplete, such as when `max_output_tokens` is reached.

‎posthog/ai/openai/openai_converter.py‎

Lines changed: 16 additions & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -520,6 +520,15 @@ def extract_openai_usage_from_response(response: Any) -> TokenUsage:
520520
return result
521521

522522

523+
# Stream events that carry the final Responses API response, with its usage and
524+
# output. Exactly one of them ends a stream.
525+
_RESPONSES_TERMINAL_EVENT_TYPES = (
526+
"response.completed",
527+
"response.incomplete",
528+
"response.failed",
529+
)
530+
531+
523532
def extract_openai_usage_from_chunk(
524533
chunk: Any, provider_type: str = "chat"
525534
) -> TokenUsage:
@@ -577,8 +586,10 @@ def extract_openai_usage_from_chunk(
577586
usage["raw_usage"] = serialized
578587

579588
elif provider_type == "responses":
580-
# For Responses API, usage is only in chunk.response.usage for completed events
581-
if hasattr(chunk, "type") and chunk.type == "response.completed":
589+
# For Responses API, usage is only in chunk.response.usage on the terminal
590+
# event. A run cut short (e.g. by max_output_tokens) ends on
591+
# response.incomplete instead of response.completed, and is still billed.
592+
if getattr(chunk, "type", None) in _RESPONSES_TERMINAL_EVENT_TYPES:
582593
if (
583594
hasattr(chunk, "response")
584595
and hasattr(chunk.response, "usage")
@@ -634,7 +645,8 @@ def extract_openai_content_from_chunk(
634645
Returns:
635646
For "chat": text content (str), or an audio/refusal delta block (dict),
636647
if present. For "responses": the full `response.output` list on the
637-
`response.completed` event. None otherwise.
648+
terminal (`response.completed`, `response.incomplete` or
649+
`response.failed`) event. None otherwise.
638650
"""
639651

640652
if provider_type == "chat":
@@ -663,7 +675,7 @@ def extract_openai_content_from_chunk(
663675

664676
elif provider_type == "responses":
665677
# Responses API format
666-
if hasattr(chunk, "type") and chunk.type == "response.completed":
678+
if getattr(chunk, "type", None) in _RESPONSES_TERMINAL_EVENT_TYPES:
667679
if hasattr(chunk, "response") and chunk.response:
668680
res = chunk.response
669681
if res.output:

‎posthog/test/ai/openai/test_openai.py‎

Lines changed: 79 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -1916,6 +1916,85 @@ def test_streaming_responses_api_extracts_model_from_response_object(mock_client
19161916
assert props["$ai_model"] == "gpt-4o-mini-stored"
19171917

19181918

1919+
def test_streaming_responses_api_captures_usage_and_output_when_incomplete(
1920+
mock_client,
1921+
):
1922+
"""A stream cut short by max_output_tokens ends on response.incomplete, not
1923+
response.completed. Its usage and partial output are still billed and must be
1924+
captured."""
1925+
from openai.types.responses import ResponseIncompleteEvent
1926+
from openai.types.responses.response import IncompleteDetails
1927+
1928+
incomplete_response = Response(
1929+
id="resp_incomplete",
1930+
model="gpt-4o-mini",
1931+
object="response",
1932+
created_at=1741476542,
1933+
status="incomplete",
1934+
error=None,
1935+
incomplete_details=IncompleteDetails(reason="max_output_tokens"),
1936+
instructions=None,
1937+
max_output_tokens=16,
1938+
tools=[],
1939+
tool_choice="auto",
1940+
output=[
1941+
ResponseOutputMessage(
1942+
id="msg_123",
1943+
type="message",
1944+
role="assistant",
1945+
status="incomplete",
1946+
content=[
1947+
ResponseOutputText(
1948+
type="output_text",
1949+
text="Once upon a time",
1950+
annotations=[],
1951+
)
1952+
],
1953+
)
1954+
],
1955+
parallel_tool_calls=True,
1956+
previous_response_id=None,
1957+
usage=make_response_usage(
1958+
input_tokens=20,
1959+
output_tokens=16,
1960+
total_tokens=36,
1961+
),
1962+
user=None,
1963+
metadata={},
1964+
)
1965+
chunks = [
1966+
ResponseIncompleteEvent(
1967+
type="response.incomplete",
1968+
sequence_number=1,
1969+
response=incomplete_response,
1970+
)
1971+
]
1972+
1973+
with patch("openai.resources.responses.Responses.create") as mock_create:
1974+
mock_create.return_value = iter(chunks)
1975+
1976+
client = OpenAI(api_key="test-key", posthog_client=mock_client)
1977+
response_generator = client.responses.create(
1978+
model="gpt-4o-mini",
1979+
input=[{"role": "user", "content": "Tell me a story"}],
1980+
max_output_tokens=16,
1981+
stream=True,
1982+
posthog_distinct_id="test-id",
1983+
)
1984+
list(response_generator)
1985+
1986+
props = mock_client.capture.call_args[1]["properties"]
1987+
assert props["$ai_stop_reason"] == "max_output_tokens"
1988+
assert props["$ai_input_tokens"] == 20
1989+
assert props["$ai_output_tokens"] == 16
1990+
assert props["$ai_output_choices"] == [
1991+
{
1992+
"role": "assistant",
1993+
"content": [{"type": "text", "text": "Once upon a time"}],
1994+
}
1995+
]
1996+
1997+
19191998
def test_non_streaming_extracts_model_from_response(mock_client):
19201999
"""Test that non-streaming calls extract model from response when not in kwargs."""
19212000
# Create a response with model but we won't pass model in kwargs

0 commit comments

Comments
 (0)