From 2df121c82179e00de4dcfc436c58ff710b13c181 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Tue, 11 Aug 2026 20:21:17 -0700 Subject: [PATCH] fix(passthrough): bill streamed Responses calls that end incomplete Streams that terminate with response.incomplete (e.g. max_output_tokens reached) carry real usage in the terminal event but were rebuilt as None and logged at zero spend, letting callers bypass budget enforcement. Parse response.incomplete alongside response.completed when reconstructing the streamed response. --- .../llms/openai/responses/transformation.py | 11 +-- .../openai_passthrough_logging_handler.py | 2 +- ...test_openai_passthrough_logging_handler.py | 71 +++++++++++++++++++ 3 files changed, 78 insertions(+), 6 deletions(-) diff --git a/litellm/llms/openai/responses/transformation.py b/litellm/llms/openai/responses/transformation.py index 11a82172316..568beed8664 100644 --- a/litellm/llms/openai/responses/transformation.py +++ b/litellm/llms/openai/responses/transformation.py @@ -354,12 +354,13 @@ class OpenAIResponsesAPIConfig(BaseResponsesAPIConfig): return event_pydantic_model.model_construct(**parsed_chunk) @staticmethod - def parse_completed_response_from_stream_chunks(all_chunks: list[str]) -> ResponsesAPIResponse | None: + def parse_terminal_response_from_stream_chunks(all_chunks: list[str]) -> ResponsesAPIResponse | None: for chunk_str in reversed(all_chunks): - try: - return ResponseCompletedEvent.model_validate_json(chunk_str.removeprefix("data: ")).response - except ValueError: - continue + for event_model in (ResponseCompletedEvent, ResponseIncompleteEvent): + try: + return event_model.model_validate_json(chunk_str.removeprefix("data: ")).response + except ValueError: + continue return None @staticmethod diff --git a/litellm/proxy/pass_through_endpoints/llm_provider_handlers/openai_passthrough_logging_handler.py b/litellm/proxy/pass_through_endpoints/llm_provider_handlers/openai_passthrough_logging_handler.py index 97c2c682a58..1614784a0dc 100644 --- a/litellm/proxy/pass_through_endpoints/llm_provider_handlers/openai_passthrough_logging_handler.py +++ b/litellm/proxy/pass_through_endpoints/llm_provider_handlers/openai_passthrough_logging_handler.py @@ -542,7 +542,7 @@ class OpenAIPassthroughLoggingHandler(BasePassthroughLoggingHandler): handler: Final = OpenAIPassthroughLoggingHandler() handler_instance: Final = handler complete_response: Final = ( - OpenAIResponsesAPIConfig.parse_completed_response_from_stream_chunks(all_chunks=all_chunks) + OpenAIResponsesAPIConfig.parse_terminal_response_from_stream_chunks(all_chunks=all_chunks) if is_responses else handler._build_complete_streaming_response( all_chunks=all_chunks, diff --git a/tests/test_litellm/proxy/pass_through_endpoints/llm_provider_handlers/test_openai_passthrough_logging_handler.py b/tests/test_litellm/proxy/pass_through_endpoints/llm_provider_handlers/test_openai_passthrough_logging_handler.py index 6b48d575281..b0576960355 100644 --- a/tests/test_litellm/proxy/pass_through_endpoints/llm_provider_handlers/test_openai_passthrough_logging_handler.py +++ b/tests/test_litellm/proxy/pass_through_endpoints/llm_provider_handlers/test_openai_passthrough_logging_handler.py @@ -593,6 +593,77 @@ class TestOpenAIPassthroughLoggingHandler: call_type="responses", ) + @patch(f"{OpenAIPassthroughLoggingHandler.__module__}.get_standard_logging_object_payload") + @patch("litellm.completion_cost", return_value=2.1e-06) + def test_streaming_responses_incomplete_event_is_billed(self, mock_completion_cost, mock_get_standard_logging): + response_id = "resp_INCOMPLETESENTINEL0123456789ab" + incomplete_event = { + "type": "response.incomplete", + "sequence_number": 5, + "response": { + "id": response_id, + "object": "response", + "created_at": 1786374786, + "status": "incomplete", + "model": "gpt-4o-mini-2024-07-18", + "output": [ + { + "id": "msg_abc", + "type": "message", + "status": "incomplete", + "role": "assistant", + "content": [ + { + "type": "output_text", + "text": "OK", + "annotations": [], + } + ], + } + ], + "usage": { + "input_tokens": 14, + "input_tokens_details": {"cached_tokens": 0}, + "output_tokens": 32, + "output_tokens_details": {"reasoning_tokens": 0}, + "total_tokens": 46, + }, + "error": None, + "incomplete_details": {"reason": "max_output_tokens"}, + "instructions": None, + "metadata": {}, + "parallel_tool_calls": True, + "temperature": 1.0, + "tool_choice": "auto", + "tools": [], + "top_p": 1.0, + }, + } + logging_obj = self._create_mock_logging_obj() + + result = OpenAIPassthroughLoggingHandler._handle_logging_openai_collected_chunks( + litellm_logging_obj=logging_obj, + passthrough_success_handler_obj=MagicMock(), + url_route="https://api.openai.com/v1/responses", + request_body={"model": "gpt-4o-mini", "stream": True}, + endpoint_type=MagicMock(), + start_time=self.start_time, + all_chunks=[f"data: {json.dumps(incomplete_event)}"], + end_time=self.end_time, + ) + + response = result["result"] + assert response.id == response_id + assert response.status == "incomplete" + assert response.usage.output_tokens == 32 + assert result["kwargs"]["response_cost"] == 2.1e-06 + mock_completion_cost.assert_called_once_with( + completion_response=response, + model="gpt-4o-mini", + custom_llm_provider="openai", + call_type="responses", + ) + @patch(f"{OpenAIPassthroughLoggingHandler.__module__}.get_standard_logging_object_payload") @patch("litellm.completion_cost") def test_streaming_responses_without_completed_event_returns_none(