diff --git a/litellm/llms/openai/responses/transformation.py b/litellm/llms/openai/responses/transformation.py index 11a82172316..568beed8664 100644 --- a/litellm/llms/openai/responses/transformation.py +++ b/litellm/llms/openai/responses/transformation.py @@ -354,12 +354,13 @@ class OpenAIResponsesAPIConfig(BaseResponsesAPIConfig): return event_pydantic_model.model_construct(**parsed_chunk) @staticmethod - def parse_completed_response_from_stream_chunks(all_chunks: list[str]) -> ResponsesAPIResponse | None: + def parse_terminal_response_from_stream_chunks(all_chunks: list[str]) -> ResponsesAPIResponse | None: for chunk_str in reversed(all_chunks): - try: - return ResponseCompletedEvent.model_validate_json(chunk_str.removeprefix("data: ")).response - except ValueError: - continue + for event_model in (ResponseCompletedEvent, ResponseIncompleteEvent): + try: + return event_model.model_validate_json(chunk_str.removeprefix("data: ")).response + except ValueError: + continue return None @staticmethod diff --git a/litellm/proxy/pass_through_endpoints/llm_provider_handlers/openai_passthrough_logging_handler.py b/litellm/proxy/pass_through_endpoints/llm_provider_handlers/openai_passthrough_logging_handler.py index 97c2c682a58..1614784a0dc 100644 --- a/litellm/proxy/pass_through_endpoints/llm_provider_handlers/openai_passthrough_logging_handler.py +++ b/litellm/proxy/pass_through_endpoints/llm_provider_handlers/openai_passthrough_logging_handler.py @@ -542,7 +542,7 @@ class OpenAIPassthroughLoggingHandler(BasePassthroughLoggingHandler): handler: Final = OpenAIPassthroughLoggingHandler() handler_instance: Final = handler complete_response: Final = ( - OpenAIResponsesAPIConfig.parse_completed_response_from_stream_chunks(all_chunks=all_chunks) + OpenAIResponsesAPIConfig.parse_terminal_response_from_stream_chunks(all_chunks=all_chunks) if is_responses else handler._build_complete_streaming_response( all_chunks=all_chunks, diff --git a/tests/test_litellm/proxy/pass_through_endpoints/llm_provider_handlers/test_openai_passthrough_logging_handler.py b/tests/test_litellm/proxy/pass_through_endpoints/llm_provider_handlers/test_openai_passthrough_logging_handler.py index 6b48d575281..b0576960355 100644 --- a/tests/test_litellm/proxy/pass_through_endpoints/llm_provider_handlers/test_openai_passthrough_logging_handler.py +++ b/tests/test_litellm/proxy/pass_through_endpoints/llm_provider_handlers/test_openai_passthrough_logging_handler.py @@ -593,6 +593,77 @@ class TestOpenAIPassthroughLoggingHandler: call_type="responses", ) + @patch(f"{OpenAIPassthroughLoggingHandler.__module__}.get_standard_logging_object_payload") + @patch("litellm.completion_cost", return_value=2.1e-06) + def test_streaming_responses_incomplete_event_is_billed(self, mock_completion_cost, mock_get_standard_logging): + response_id = "resp_INCOMPLETESENTINEL0123456789ab" + incomplete_event = { + "type": "response.incomplete", + "sequence_number": 5, + "response": { + "id": response_id, + "object": "response", + "created_at": 1786374786, + "status": "incomplete", + "model": "gpt-4o-mini-2024-07-18", + "output": [ + { + "id": "msg_abc", + "type": "message", + "status": "incomplete", + "role": "assistant", + "content": [ + { + "type": "output_text", + "text": "OK", + "annotations": [], + } + ], + } + ], + "usage": { + "input_tokens": 14, + "input_tokens_details": {"cached_tokens": 0}, + "output_tokens": 32, + "output_tokens_details": {"reasoning_tokens": 0}, + "total_tokens": 46, + }, + "error": None, + "incomplete_details": {"reason": "max_output_tokens"}, + "instructions": None, + "metadata": {}, + "parallel_tool_calls": True, + "temperature": 1.0, + "tool_choice": "auto", + "tools": [], + "top_p": 1.0, + }, + } + logging_obj = self._create_mock_logging_obj() + + result = OpenAIPassthroughLoggingHandler._handle_logging_openai_collected_chunks( + litellm_logging_obj=logging_obj, + passthrough_success_handler_obj=MagicMock(), + url_route="https://api.openai.com/v1/responses", + request_body={"model": "gpt-4o-mini", "stream": True}, + endpoint_type=MagicMock(), + start_time=self.start_time, + all_chunks=[f"data: {json.dumps(incomplete_event)}"], + end_time=self.end_time, + ) + + response = result["result"] + assert response.id == response_id + assert response.status == "incomplete" + assert response.usage.output_tokens == 32 + assert result["kwargs"]["response_cost"] == 2.1e-06 + mock_completion_cost.assert_called_once_with( + completion_response=response, + model="gpt-4o-mini", + custom_llm_provider="openai", + call_type="responses", + ) + @patch(f"{OpenAIPassthroughLoggingHandler.__module__}.get_standard_logging_object_payload") @patch("litellm.completion_cost") def test_streaming_responses_without_completed_event_returns_none(