From 0e89bf60fe1cce00107b5727dc9ff81f5cb9a1f2 Mon Sep 17 00:00:00 2001 From: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Tue, 11 Aug 2026 02:15:49 +0000 Subject: [PATCH 1/9] feat(dashscope): add latest Model Studio models to the cost map Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- ...odel_prices_and_context_window_backup.json | 114 ++++++++++++++++++ model_prices_and_context_window.json | 114 ++++++++++++++++++ .../test_dashscope_cost_calculator.py | 43 +++++++ 3 files changed, 271 insertions(+) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 951c114b0a9..632112a01fc 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -12888,6 +12888,103 @@ "supports_system_messages": true, "supports_tool_choice": false }, + "dashscope/deepseek-v4-flash": { + "cache_read_input_token_cost": 4e-08, + "input_cost_per_token": 2e-07, + "litellm_provider": "dashscope", + "max_input_tokens": 1000000, + "max_output_tokens": 393216, + "max_tokens": 393216, + "mode": "chat", + "output_cost_per_token": 4e-07, + "source": "https://www.alibabacloud.com/help/en/model-studio/models", + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "dashscope/deepseek-v4-flash-0731": { + "cache_read_input_token_cost": 4e-08, + "input_cost_per_token": 2e-07, + "litellm_provider": "dashscope", + "max_input_tokens": 1000000, + "max_output_tokens": 393216, + "max_tokens": 393216, + "mode": "chat", + "output_cost_per_token": 4e-07, + "source": "https://www.alibabacloud.com/help/en/model-studio/models", + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "dashscope/deepseek-v4-pro": { + "cache_read_input_token_cost": 2e-07, + "input_cost_per_token": 2.4e-06, + "litellm_provider": "dashscope", + "max_input_tokens": 1000000, + "max_output_tokens": 393216, + "max_tokens": 393216, + "mode": "chat", + "output_cost_per_token": 4.8e-06, + "source": "https://www.alibabacloud.com/help/en/model-studio/models", + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "dashscope/glm-5.1": { + "cache_read_input_token_cost": 2.6e-07, + "input_cost_per_token": 1.4e-06, + "litellm_provider": "dashscope", + "max_input_tokens": 202745, + "max_output_tokens": 131072, + "max_tokens": 131072, + "mode": "chat", + "output_cost_per_token": 4.4e-06, + "source": "https://www.alibabacloud.com/help/en/model-studio/models", + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "dashscope/glm-5.2": { + "cache_read_input_token_cost": 2.8e-07, + "input_cost_per_token": 1.4e-06, + "litellm_provider": "dashscope", + "max_input_tokens": 1048576, + "max_output_tokens": 131072, + "max_tokens": 131072, + "mode": "chat", + "output_cost_per_token": 4.4e-06, + "source": "https://www.alibabacloud.com/help/en/model-studio/models", + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "dashscope/kimi-k2.7-code": { + "cache_read_input_token_cost": 1.9e-07, + "input_cost_per_token": 9.5e-07, + "litellm_provider": "dashscope", + "max_input_tokens": 229376, + "max_output_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "output_cost_per_token": 4e-06, + "source": "https://www.alibabacloud.com/help/en/model-studio/models", + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, "dashscope/qwen-coder": { "input_cost_per_token": 3e-07, "litellm_provider": "dashscope", @@ -13681,6 +13778,23 @@ } ] }, + "dashscope/qwen3.8-max": { + "cache_read_input_token_cost": 2.5e-07, + "input_cost_per_token": 2e-06, + "litellm_provider": "dashscope", + "max_input_tokens": 991808, + "max_output_tokens": 131072, + "max_tokens": 131072, + "mode": "chat", + "output_cost_per_token": 6e-06, + "source": "https://www.alibabacloud.com/help/en/model-studio/models", + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, "dashscope/qwq-plus": { "input_cost_per_token": 8e-07, "litellm_provider": "dashscope", diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 951c114b0a9..632112a01fc 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -12888,6 +12888,103 @@ "supports_system_messages": true, "supports_tool_choice": false }, + "dashscope/deepseek-v4-flash": { + "cache_read_input_token_cost": 4e-08, + "input_cost_per_token": 2e-07, + "litellm_provider": "dashscope", + "max_input_tokens": 1000000, + "max_output_tokens": 393216, + "max_tokens": 393216, + "mode": "chat", + "output_cost_per_token": 4e-07, + "source": "https://www.alibabacloud.com/help/en/model-studio/models", + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "dashscope/deepseek-v4-flash-0731": { + "cache_read_input_token_cost": 4e-08, + "input_cost_per_token": 2e-07, + "litellm_provider": "dashscope", + "max_input_tokens": 1000000, + "max_output_tokens": 393216, + "max_tokens": 393216, + "mode": "chat", + "output_cost_per_token": 4e-07, + "source": "https://www.alibabacloud.com/help/en/model-studio/models", + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "dashscope/deepseek-v4-pro": { + "cache_read_input_token_cost": 2e-07, + "input_cost_per_token": 2.4e-06, + "litellm_provider": "dashscope", + "max_input_tokens": 1000000, + "max_output_tokens": 393216, + "max_tokens": 393216, + "mode": "chat", + "output_cost_per_token": 4.8e-06, + "source": "https://www.alibabacloud.com/help/en/model-studio/models", + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "dashscope/glm-5.1": { + "cache_read_input_token_cost": 2.6e-07, + "input_cost_per_token": 1.4e-06, + "litellm_provider": "dashscope", + "max_input_tokens": 202745, + "max_output_tokens": 131072, + "max_tokens": 131072, + "mode": "chat", + "output_cost_per_token": 4.4e-06, + "source": "https://www.alibabacloud.com/help/en/model-studio/models", + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "dashscope/glm-5.2": { + "cache_read_input_token_cost": 2.8e-07, + "input_cost_per_token": 1.4e-06, + "litellm_provider": "dashscope", + "max_input_tokens": 1048576, + "max_output_tokens": 131072, + "max_tokens": 131072, + "mode": "chat", + "output_cost_per_token": 4.4e-06, + "source": "https://www.alibabacloud.com/help/en/model-studio/models", + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "dashscope/kimi-k2.7-code": { + "cache_read_input_token_cost": 1.9e-07, + "input_cost_per_token": 9.5e-07, + "litellm_provider": "dashscope", + "max_input_tokens": 229376, + "max_output_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "output_cost_per_token": 4e-06, + "source": "https://www.alibabacloud.com/help/en/model-studio/models", + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, "dashscope/qwen-coder": { "input_cost_per_token": 3e-07, "litellm_provider": "dashscope", @@ -13681,6 +13778,23 @@ } ] }, + "dashscope/qwen3.8-max": { + "cache_read_input_token_cost": 2.5e-07, + "input_cost_per_token": 2e-06, + "litellm_provider": "dashscope", + "max_input_tokens": 991808, + "max_output_tokens": 131072, + "max_tokens": 131072, + "mode": "chat", + "output_cost_per_token": 6e-06, + "source": "https://www.alibabacloud.com/help/en/model-studio/models", + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, "dashscope/qwq-plus": { "input_cost_per_token": 8e-07, "litellm_provider": "dashscope", diff --git a/tests/test_litellm/llms/dashscope/test_dashscope_cost_calculator.py b/tests/test_litellm/llms/dashscope/test_dashscope_cost_calculator.py index 6041a8c8377..f21a5fdff5e 100644 --- a/tests/test_litellm/llms/dashscope/test_dashscope_cost_calculator.py +++ b/tests/test_litellm/llms/dashscope/test_dashscope_cost_calculator.py @@ -196,6 +196,49 @@ class TestDashscopeCostCalculator: assert math.isclose(prompt_cost, expected_prompt_cost, rel_tol=1e-10) assert math.isclose(completion_cost, expected_completion_cost, rel_tol=1e-10) + @pytest.mark.parametrize( + "model, input_cost, output_cost, cache_read_cost, max_input_tokens, max_output_tokens", + ( + ("qwen3.8-max", 2e-06, 6e-06, 2.5e-07, 991808, 131072), + ("deepseek-v4-pro", 2.4e-06, 4.8e-06, 2e-07, 1000000, 393216), + ("deepseek-v4-flash", 2e-07, 4e-07, 4e-08, 1000000, 393216), + ("deepseek-v4-flash-0731", 2e-07, 4e-07, 4e-08, 1000000, 393216), + ("glm-5.1", 1.4e-06, 4.4e-06, 2.6e-07, 202745, 131072), + ("glm-5.2", 1.4e-06, 4.4e-06, 2.8e-07, 1048576, 131072), + ("kimi-k2.7-code", 9.5e-07, 4e-06, 1.9e-07, 229376, 16384), + ), + ) + def test_dashscope_latest_model_pricing( + self, + model: str, + input_cost: float, + output_cost: float, + cache_read_cost: float, + max_input_tokens: int, + max_output_tokens: int, + ) -> None: + """ + Model Studio International (Singapore) pricing and context limits for the models + added for Qwen3.8, DeepSeek V4, GLM 5 and Kimi K2.7 + """ + usage = Usage( + prompt_tokens=10000, + completion_tokens=1000, + total_tokens=11000, + prompt_tokens_details=PromptTokensDetailsWrapper(cached_tokens=4000), + ) + + prompt_cost, completion_cost = dashscope_cost_per_token(model=model, usage=usage) + + model_info = litellm.get_model_info(f"dashscope/{model}") + assert model_info["max_input_tokens"] == max_input_tokens + assert model_info["max_output_tokens"] == max_output_tokens + assert model_info["supports_reasoning"] is True + + expected_prompt_cost = 4000 * cache_read_cost + 6000 * input_cost + assert math.isclose(prompt_cost, expected_prompt_cost, rel_tol=1e-10) + assert math.isclose(completion_cost, 1000 * output_cost, rel_tol=1e-10) + def test_dashscope_tiered_pricing_exceeding_highest_tier(self): """ Tests tiered pricing when token count exceeds the highest defined tier range. From 0cd28c5c409e2a15af06a49e431b2c8af71f96df Mon Sep 17 00:00:00 2001 From: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Tue, 11 Aug 2026 02:32:46 +0000 Subject: [PATCH 2/9] chore(dashscope): drop cost map regression test Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../test_dashscope_cost_calculator.py | 43 ------------------- 1 file changed, 43 deletions(-) diff --git a/tests/test_litellm/llms/dashscope/test_dashscope_cost_calculator.py b/tests/test_litellm/llms/dashscope/test_dashscope_cost_calculator.py index f21a5fdff5e..6041a8c8377 100644 --- a/tests/test_litellm/llms/dashscope/test_dashscope_cost_calculator.py +++ b/tests/test_litellm/llms/dashscope/test_dashscope_cost_calculator.py @@ -196,49 +196,6 @@ class TestDashscopeCostCalculator: assert math.isclose(prompt_cost, expected_prompt_cost, rel_tol=1e-10) assert math.isclose(completion_cost, expected_completion_cost, rel_tol=1e-10) - @pytest.mark.parametrize( - "model, input_cost, output_cost, cache_read_cost, max_input_tokens, max_output_tokens", - ( - ("qwen3.8-max", 2e-06, 6e-06, 2.5e-07, 991808, 131072), - ("deepseek-v4-pro", 2.4e-06, 4.8e-06, 2e-07, 1000000, 393216), - ("deepseek-v4-flash", 2e-07, 4e-07, 4e-08, 1000000, 393216), - ("deepseek-v4-flash-0731", 2e-07, 4e-07, 4e-08, 1000000, 393216), - ("glm-5.1", 1.4e-06, 4.4e-06, 2.6e-07, 202745, 131072), - ("glm-5.2", 1.4e-06, 4.4e-06, 2.8e-07, 1048576, 131072), - ("kimi-k2.7-code", 9.5e-07, 4e-06, 1.9e-07, 229376, 16384), - ), - ) - def test_dashscope_latest_model_pricing( - self, - model: str, - input_cost: float, - output_cost: float, - cache_read_cost: float, - max_input_tokens: int, - max_output_tokens: int, - ) -> None: - """ - Model Studio International (Singapore) pricing and context limits for the models - added for Qwen3.8, DeepSeek V4, GLM 5 and Kimi K2.7 - """ - usage = Usage( - prompt_tokens=10000, - completion_tokens=1000, - total_tokens=11000, - prompt_tokens_details=PromptTokensDetailsWrapper(cached_tokens=4000), - ) - - prompt_cost, completion_cost = dashscope_cost_per_token(model=model, usage=usage) - - model_info = litellm.get_model_info(f"dashscope/{model}") - assert model_info["max_input_tokens"] == max_input_tokens - assert model_info["max_output_tokens"] == max_output_tokens - assert model_info["supports_reasoning"] is True - - expected_prompt_cost = 4000 * cache_read_cost + 6000 * input_cost - assert math.isclose(prompt_cost, expected_prompt_cost, rel_tol=1e-10) - assert math.isclose(completion_cost, 1000 * output_cost, rel_tol=1e-10) - def test_dashscope_tiered_pricing_exceeding_highest_tier(self): """ Tests tiered pricing when token count exceeds the highest defined tier range. From c8655c38251695c1c892071449fe22cb809ac295 Mon Sep 17 00:00:00 2001 From: william-xue <20151622+william-xue@users.noreply.github.com> Date: Tue, 11 Aug 2026 18:29:08 +0800 Subject: [PATCH 3/9] fix(proxy): track streamed passthrough Responses cost --- .../openai_passthrough_logging_handler.py | 44 ++++++++--- ...test_openai_passthrough_logging_handler.py | 74 +++++++++++++++++++ 2 files changed, 108 insertions(+), 10 deletions(-) diff --git a/litellm/proxy/pass_through_endpoints/llm_provider_handlers/openai_passthrough_logging_handler.py b/litellm/proxy/pass_through_endpoints/llm_provider_handlers/openai_passthrough_logging_handler.py index f79b589f6b3..28aa09d5a94 100644 --- a/litellm/proxy/pass_through_endpoints/llm_provider_handlers/openai_passthrough_logging_handler.py +++ b/litellm/proxy/pass_through_endpoints/llm_provider_handlers/openai_passthrough_logging_handler.py @@ -26,7 +26,7 @@ from litellm.proxy.pass_through_endpoints.llm_provider_handlers.base_passthrough from litellm.proxy.pass_through_endpoints.success_handler import ( PassThroughEndpointLogging, ) -from litellm.types.llms.openai import ResponsesAPIResponse +from litellm.types.llms.openai import ResponseCompletedEvent, ResponsesAPIResponse from litellm.types.passthrough_endpoints.pass_through_endpoints import ( EndpointType, PassthroughStandardLoggingPayload, @@ -464,7 +464,7 @@ class OpenAIPassthroughLoggingHandler(BasePassthroughLoggingHandler): def _build_complete_streaming_response( self, - all_chunks: list, + all_chunks: list[str], litellm_logging_obj: LiteLLMLoggingObj, model: str, ) -> ModelResponse | TextCompletionResponse | None: @@ -518,6 +518,15 @@ class OpenAIPassthroughLoggingHandler(BasePassthroughLoggingHandler): verbose_proxy_logger.error("Error building complete streaming response: %s", e) return None + @staticmethod + def _build_complete_streaming_responses_response(all_chunks: list[str]) -> ResponsesAPIResponse | None: + for chunk_str in reversed(all_chunks): + try: + return ResponseCompletedEvent.model_validate_json(chunk_str.removeprefix("data: ")).response + except ValueError: + continue + return None + @staticmethod def _handle_logging_openai_collected_chunks( litellm_logging_obj: LiteLLMLoggingObj, @@ -536,13 +545,19 @@ class OpenAIPassthroughLoggingHandler(BasePassthroughLoggingHandler): # Extract model from request body model: Final = request_body.get("model", "gpt-4o") + is_responses: Final = OpenAIPassthroughLoggingHandler.is_openai_responses_route(url_route) + # Build complete response from chunks using our streaming handler handler: Final = OpenAIPassthroughLoggingHandler() handler_instance: Final = handler - complete_response: Final = handler._build_complete_streaming_response( - all_chunks=all_chunks, - litellm_logging_obj=litellm_logging_obj, - model=model, + complete_response: Final = ( + handler._build_complete_streaming_responses_response(all_chunks=all_chunks) + if is_responses + else handler._build_complete_streaming_response( + all_chunks=all_chunks, + litellm_logging_obj=litellm_logging_obj, + model=model, + ) ) if complete_response is None: @@ -554,10 +569,19 @@ class OpenAIPassthroughLoggingHandler(BasePassthroughLoggingHandler): custom_llm_provider: Final = litellm_logging_obj.model_call_details.get("custom_llm_provider", "openai") # Calculate cost using LiteLLM's cost calculator - response_cost: Final = litellm.completion_cost( - completion_response=complete_response, - model=model, - custom_llm_provider=custom_llm_provider, + response_cost: Final = ( + litellm.completion_cost( + completion_response=complete_response, + model=model, + custom_llm_provider=custom_llm_provider, + call_type="responses", + ) + if is_responses + else litellm.completion_cost( + completion_response=complete_response, + model=model, + custom_llm_provider=custom_llm_provider, + ) ) # Preserve existing litellm_params to maintain metadata tags diff --git a/tests/test_litellm/proxy/pass_through_endpoints/llm_provider_handlers/test_openai_passthrough_logging_handler.py b/tests/test_litellm/proxy/pass_through_endpoints/llm_provider_handlers/test_openai_passthrough_logging_handler.py index 401ea2ef589..9e5da39af88 100644 --- a/tests/test_litellm/proxy/pass_through_endpoints/llm_provider_handlers/test_openai_passthrough_logging_handler.py +++ b/tests/test_litellm/proxy/pass_through_endpoints/llm_provider_handlers/test_openai_passthrough_logging_handler.py @@ -519,6 +519,80 @@ class TestOpenAIPassthroughLoggingHandler: assert result is None # Placeholder implementation + @patch(f"{OpenAIPassthroughLoggingHandler.__module__}.get_standard_logging_object_payload") + @patch("litellm.completion_cost", return_value=3.3e-06) + def test_streaming_responses_cost_uses_completed_response( + self, mock_completion_cost, mock_get_standard_logging + ): + response_id = "resp_PROOFSENTINEL0123456789abcdef" + completed_event = { + "type": "response.completed", + "sequence_number": 8, + "response": { + "id": response_id, + "object": "response", + "created_at": 1786374786, + "status": "completed", + "model": "gpt-4o-mini-2024-07-18", + "output": [ + { + "id": "msg_abc", + "type": "message", + "status": "completed", + "role": "assistant", + "content": [ + { + "type": "output_text", + "text": "OK", + "annotations": [], + } + ], + } + ], + "usage": { + "input_tokens": 14, + "input_tokens_details": {"cached_tokens": 0}, + "output_tokens": 2, + "output_tokens_details": {"reasoning_tokens": 0}, + "total_tokens": 16, + }, + "error": None, + "incomplete_details": None, + "instructions": None, + "metadata": {}, + "parallel_tool_calls": True, + "temperature": 1.0, + "tool_choice": "auto", + "tools": [], + "top_p": 1.0, + }, + } + logging_obj = self._create_mock_logging_obj() + + result = OpenAIPassthroughLoggingHandler._handle_logging_openai_collected_chunks( + litellm_logging_obj=logging_obj, + passthrough_success_handler_obj=MagicMock(), + url_route="https://api.openai.com/v1/responses", + request_body={"model": "gpt-4o-mini", "stream": True}, + endpoint_type=MagicMock(), + start_time=self.start_time, + all_chunks=[f"data: {json.dumps(completed_event)}", "data: [DONE]"], + end_time=self.end_time, + ) + + response = result["result"] + assert response.id == response_id + assert response.model == "gpt-4o-mini-2024-07-18" + assert response.usage.input_tokens == 14 + assert response.usage.output_tokens == 2 + assert result["kwargs"]["response_cost"] == 3.3e-06 + mock_completion_cost.assert_called_once_with( + completion_response=response, + model="gpt-4o-mini", + custom_llm_provider="openai", + call_type="responses", + ) + @patch("litellm.completion_cost") @patch( "litellm.litellm_core_utils.litellm_logging.get_standard_logging_object_payload" From dc30e1816d89c7124db6f20bdb6d68296b5ab6d2 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Tue, 11 Aug 2026 20:11:53 -0700 Subject: [PATCH 4/9] refactor(passthrough): move Responses stream terminal-event parsing into OpenAI provider config Addresses review feedback: the ResponseCompletedEvent SSE parsing now lives in OpenAIResponsesAPIConfig next to the other Responses stream event handling, and the proxy logging handler calls it. Adds coverage for streams that end without a response.completed event. --- .../llms/openai/responses/transformation.py | 9 +++++++ .../openai_passthrough_logging_handler.py | 13 ++-------- ...test_openai_passthrough_logging_handler.py | 24 +++++++++++++++++++ 3 files changed, 35 insertions(+), 11 deletions(-) diff --git a/litellm/llms/openai/responses/transformation.py b/litellm/llms/openai/responses/transformation.py index f12a034b6ad..11a82172316 100644 --- a/litellm/llms/openai/responses/transformation.py +++ b/litellm/llms/openai/responses/transformation.py @@ -353,6 +353,15 @@ class OpenAIResponsesAPIConfig(BaseResponsesAPIConfig): ) return event_pydantic_model.model_construct(**parsed_chunk) + @staticmethod + def parse_completed_response_from_stream_chunks(all_chunks: list[str]) -> ResponsesAPIResponse | None: + for chunk_str in reversed(all_chunks): + try: + return ResponseCompletedEvent.model_validate_json(chunk_str.removeprefix("data: ")).response + except ValueError: + continue + return None + @staticmethod def get_event_model_class(event_type: str) -> Any: """ diff --git a/litellm/proxy/pass_through_endpoints/llm_provider_handlers/openai_passthrough_logging_handler.py b/litellm/proxy/pass_through_endpoints/llm_provider_handlers/openai_passthrough_logging_handler.py index 28aa09d5a94..97c2c682a58 100644 --- a/litellm/proxy/pass_through_endpoints/llm_provider_handlers/openai_passthrough_logging_handler.py +++ b/litellm/proxy/pass_through_endpoints/llm_provider_handlers/openai_passthrough_logging_handler.py @@ -26,7 +26,7 @@ from litellm.proxy.pass_through_endpoints.llm_provider_handlers.base_passthrough from litellm.proxy.pass_through_endpoints.success_handler import ( PassThroughEndpointLogging, ) -from litellm.types.llms.openai import ResponseCompletedEvent, ResponsesAPIResponse +from litellm.types.llms.openai import ResponsesAPIResponse from litellm.types.passthrough_endpoints.pass_through_endpoints import ( EndpointType, PassthroughStandardLoggingPayload, @@ -518,15 +518,6 @@ class OpenAIPassthroughLoggingHandler(BasePassthroughLoggingHandler): verbose_proxy_logger.error("Error building complete streaming response: %s", e) return None - @staticmethod - def _build_complete_streaming_responses_response(all_chunks: list[str]) -> ResponsesAPIResponse | None: - for chunk_str in reversed(all_chunks): - try: - return ResponseCompletedEvent.model_validate_json(chunk_str.removeprefix("data: ")).response - except ValueError: - continue - return None - @staticmethod def _handle_logging_openai_collected_chunks( litellm_logging_obj: LiteLLMLoggingObj, @@ -551,7 +542,7 @@ class OpenAIPassthroughLoggingHandler(BasePassthroughLoggingHandler): handler: Final = OpenAIPassthroughLoggingHandler() handler_instance: Final = handler complete_response: Final = ( - handler._build_complete_streaming_responses_response(all_chunks=all_chunks) + OpenAIResponsesAPIConfig.parse_completed_response_from_stream_chunks(all_chunks=all_chunks) if is_responses else handler._build_complete_streaming_response( all_chunks=all_chunks, diff --git a/tests/test_litellm/proxy/pass_through_endpoints/llm_provider_handlers/test_openai_passthrough_logging_handler.py b/tests/test_litellm/proxy/pass_through_endpoints/llm_provider_handlers/test_openai_passthrough_logging_handler.py index 9e5da39af88..6b48d575281 100644 --- a/tests/test_litellm/proxy/pass_through_endpoints/llm_provider_handlers/test_openai_passthrough_logging_handler.py +++ b/tests/test_litellm/proxy/pass_through_endpoints/llm_provider_handlers/test_openai_passthrough_logging_handler.py @@ -593,6 +593,30 @@ class TestOpenAIPassthroughLoggingHandler: call_type="responses", ) + @patch(f"{OpenAIPassthroughLoggingHandler.__module__}.get_standard_logging_object_payload") + @patch("litellm.completion_cost") + def test_streaming_responses_without_completed_event_returns_none( + self, mock_completion_cost, mock_get_standard_logging + ): + logging_obj = self._create_mock_logging_obj() + + result = OpenAIPassthroughLoggingHandler._handle_logging_openai_collected_chunks( + litellm_logging_obj=logging_obj, + passthrough_success_handler_obj=MagicMock(), + url_route="https://api.openai.com/v1/responses", + request_body={"model": "gpt-4o-mini", "stream": True}, + endpoint_type=MagicMock(), + start_time=self.start_time, + all_chunks=[ + 'data: {"type": "response.created", "sequence_number": 0}', + 'data: {"type": "response.output_text.delta", "sequence_number": 1, "delta": "OK"}', + ], + end_time=self.end_time, + ) + + assert result == {"result": None, "kwargs": {}} + mock_completion_cost.assert_not_called() + @patch("litellm.completion_cost") @patch( "litellm.litellm_core_utils.litellm_logging.get_standard_logging_object_payload" From 2df121c82179e00de4dcfc436c58ff710b13c181 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Tue, 11 Aug 2026 20:21:17 -0700 Subject: [PATCH 5/9] fix(passthrough): bill streamed Responses calls that end incomplete Streams that terminate with response.incomplete (e.g. max_output_tokens reached) carry real usage in the terminal event but were rebuilt as None and logged at zero spend, letting callers bypass budget enforcement. Parse response.incomplete alongside response.completed when reconstructing the streamed response. --- .../llms/openai/responses/transformation.py | 11 +-- .../openai_passthrough_logging_handler.py | 2 +- ...test_openai_passthrough_logging_handler.py | 71 +++++++++++++++++++ 3 files changed, 78 insertions(+), 6 deletions(-) diff --git a/litellm/llms/openai/responses/transformation.py b/litellm/llms/openai/responses/transformation.py index 11a82172316..568beed8664 100644 --- a/litellm/llms/openai/responses/transformation.py +++ b/litellm/llms/openai/responses/transformation.py @@ -354,12 +354,13 @@ class OpenAIResponsesAPIConfig(BaseResponsesAPIConfig): return event_pydantic_model.model_construct(**parsed_chunk) @staticmethod - def parse_completed_response_from_stream_chunks(all_chunks: list[str]) -> ResponsesAPIResponse | None: + def parse_terminal_response_from_stream_chunks(all_chunks: list[str]) -> ResponsesAPIResponse | None: for chunk_str in reversed(all_chunks): - try: - return ResponseCompletedEvent.model_validate_json(chunk_str.removeprefix("data: ")).response - except ValueError: - continue + for event_model in (ResponseCompletedEvent, ResponseIncompleteEvent): + try: + return event_model.model_validate_json(chunk_str.removeprefix("data: ")).response + except ValueError: + continue return None @staticmethod diff --git a/litellm/proxy/pass_through_endpoints/llm_provider_handlers/openai_passthrough_logging_handler.py b/litellm/proxy/pass_through_endpoints/llm_provider_handlers/openai_passthrough_logging_handler.py index 97c2c682a58..1614784a0dc 100644 --- a/litellm/proxy/pass_through_endpoints/llm_provider_handlers/openai_passthrough_logging_handler.py +++ b/litellm/proxy/pass_through_endpoints/llm_provider_handlers/openai_passthrough_logging_handler.py @@ -542,7 +542,7 @@ class OpenAIPassthroughLoggingHandler(BasePassthroughLoggingHandler): handler: Final = OpenAIPassthroughLoggingHandler() handler_instance: Final = handler complete_response: Final = ( - OpenAIResponsesAPIConfig.parse_completed_response_from_stream_chunks(all_chunks=all_chunks) + OpenAIResponsesAPIConfig.parse_terminal_response_from_stream_chunks(all_chunks=all_chunks) if is_responses else handler._build_complete_streaming_response( all_chunks=all_chunks, diff --git a/tests/test_litellm/proxy/pass_through_endpoints/llm_provider_handlers/test_openai_passthrough_logging_handler.py b/tests/test_litellm/proxy/pass_through_endpoints/llm_provider_handlers/test_openai_passthrough_logging_handler.py index 6b48d575281..b0576960355 100644 --- a/tests/test_litellm/proxy/pass_through_endpoints/llm_provider_handlers/test_openai_passthrough_logging_handler.py +++ b/tests/test_litellm/proxy/pass_through_endpoints/llm_provider_handlers/test_openai_passthrough_logging_handler.py @@ -593,6 +593,77 @@ class TestOpenAIPassthroughLoggingHandler: call_type="responses", ) + @patch(f"{OpenAIPassthroughLoggingHandler.__module__}.get_standard_logging_object_payload") + @patch("litellm.completion_cost", return_value=2.1e-06) + def test_streaming_responses_incomplete_event_is_billed(self, mock_completion_cost, mock_get_standard_logging): + response_id = "resp_INCOMPLETESENTINEL0123456789ab" + incomplete_event = { + "type": "response.incomplete", + "sequence_number": 5, + "response": { + "id": response_id, + "object": "response", + "created_at": 1786374786, + "status": "incomplete", + "model": "gpt-4o-mini-2024-07-18", + "output": [ + { + "id": "msg_abc", + "type": "message", + "status": "incomplete", + "role": "assistant", + "content": [ + { + "type": "output_text", + "text": "OK", + "annotations": [], + } + ], + } + ], + "usage": { + "input_tokens": 14, + "input_tokens_details": {"cached_tokens": 0}, + "output_tokens": 32, + "output_tokens_details": {"reasoning_tokens": 0}, + "total_tokens": 46, + }, + "error": None, + "incomplete_details": {"reason": "max_output_tokens"}, + "instructions": None, + "metadata": {}, + "parallel_tool_calls": True, + "temperature": 1.0, + "tool_choice": "auto", + "tools": [], + "top_p": 1.0, + }, + } + logging_obj = self._create_mock_logging_obj() + + result = OpenAIPassthroughLoggingHandler._handle_logging_openai_collected_chunks( + litellm_logging_obj=logging_obj, + passthrough_success_handler_obj=MagicMock(), + url_route="https://api.openai.com/v1/responses", + request_body={"model": "gpt-4o-mini", "stream": True}, + endpoint_type=MagicMock(), + start_time=self.start_time, + all_chunks=[f"data: {json.dumps(incomplete_event)}"], + end_time=self.end_time, + ) + + response = result["result"] + assert response.id == response_id + assert response.status == "incomplete" + assert response.usage.output_tokens == 32 + assert result["kwargs"]["response_cost"] == 2.1e-06 + mock_completion_cost.assert_called_once_with( + completion_response=response, + model="gpt-4o-mini", + custom_llm_provider="openai", + call_type="responses", + ) + @patch(f"{OpenAIPassthroughLoggingHandler.__module__}.get_standard_logging_object_payload") @patch("litellm.completion_cost") def test_streaming_responses_without_completed_event_returns_none( From 7a55ca811b6d57fdc356260102619c10aec95c04 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 12 Aug 2026 03:45:34 +0000 Subject: [PATCH 6/9] fix(responses): init completed_response on bridge streaming iterator (#35413) LiteLLMCompletionStreamingIterator overrides __init__ without calling super().__init__(), so completed_response was only set once the stream reached RESPONSE_COMPLETED. On a mid-stream provider error the router's _extract_partial_responses_usage read source_iterator.completed_response during fallback recovery and raised AttributeError, masking the real provider error (e.g. Anthropic 529) and bypassing configured retries and fallbacks. Initialize the attribute to None so recovery degrades to no partial usage instead of crashing. Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../streaming_iterator.py | 1 + ...st_router_aresponses_streaming_fallback.py | 29 +++++++++++++++++++ 2 files changed, 30 insertions(+) diff --git a/litellm/responses/litellm_completion_transformation/streaming_iterator.py b/litellm/responses/litellm_completion_transformation/streaming_iterator.py index ddd05075763..5a7380a55b8 100644 --- a/litellm/responses/litellm_completion_transformation/streaming_iterator.py +++ b/litellm/responses/litellm_completion_transformation/streaming_iterator.py @@ -82,6 +82,7 @@ class LiteLLMCompletionStreamingIterator(ResponsesAPIStreamingIterator): self.sent_output_item_done_event: bool = False self.sent_annotation_events: bool = False self.litellm_model_response: ModelResponse | TextCompletionResponse | None = None + self.completed_response: Any = None self.final_text: str = "" self._cached_item_id: str | None = None self._cached_response_id: str | None = None diff --git a/tests/router_unit_tests/test_router_aresponses_streaming_fallback.py b/tests/router_unit_tests/test_router_aresponses_streaming_fallback.py index 2fb7bdfceb5..17124a94a8f 100644 --- a/tests/router_unit_tests/test_router_aresponses_streaming_fallback.py +++ b/tests/router_unit_tests/test_router_aresponses_streaming_fallback.py @@ -90,6 +90,35 @@ def test_extract_partial_responses_usage_no_completed_response(): assert usage is None +def test_extract_partial_responses_usage_bridge_iterator_no_completed_response(): + """ + Regression for #35411: the bridge iterator + (LiteLLMCompletionStreamingIterator) overrides __init__ without calling + super().__init__(), so completed_response was never set until the stream + reached RESPONSE_COMPLETED. On a mid-stream provider error (before + completion) the fallback recovery path read source_iterator.completed_response + and raised AttributeError, masking the real provider error and bypassing + fallbacks. The attribute must always exist and default to None. + """ + from litellm.responses.litellm_completion_transformation.streaming_iterator import ( + LiteLLMCompletionStreamingIterator, + ) + + wrapper = MagicMock() + wrapper.logging_obj = MagicMock() + iterator = LiteLLMCompletionStreamingIterator( + model="anthropic/claude-sonnet-4-5", + litellm_custom_stream_wrapper=wrapper, + request_input="hi", + responses_api_request={}, + ) + + assert iterator.completed_response is None + # No chat chunks collected yet and no completed_response → must return + # None instead of raising AttributeError. + assert Router._extract_partial_responses_usage(iterator) is None + + # -------- _combine_responses_fallback_usage -------- From 8bfb7772e4effe3f699f5ac327320cc9a6b20f9b Mon Sep 17 00:00:00 2001 From: yucheng-berri Date: Tue, 11 Aug 2026 20:52:43 -0700 Subject: [PATCH 7/9] fix(batches): attribute Anthropic passthrough batch cost to the creating key, team and tags (#36468) The Anthropic batch create never persisted the creating key's hashed token or its request tags on the managed object, so when CheckBatchCost billed the batch hours later there was nothing to attribute it to. Key spend, key budgets and tag spend never moved for batch usage. Persist both from the create, the way the Vertex passthrough already does, and register the batch only from the collection route. An id-scoped route cannot rebuild the unified object id, because it embeds the model and the model comes from the create's request body, so it could only claim a row it did not create or fail the model_object_id unique constraint. The shared metadata helpers, the route predicate and the registration-result logging now live in batch_attribution instead of being copied per provider. The Anthropic write previously logged success unconditionally, before the fire-and-forget task had run. Resolves LIT-5288 --- litellm/constants.py | 2 + .../proxy/hooks/proxy_track_cost_callback.py | 9 +- .../anthropic_passthrough_logging_handler.py | 46 +++-- .../batch_attribution.py | 78 +++++++++ .../vertex_passthrough_logging_handler.py | 67 ++------ .../proxy/test_managed_files_hook.py | 4 +- ...t_anthropic_passthrough_logging_handler.py | 117 +++++++++++++ .../test_batch_attribution.py | 159 ++++++++++++++++++ 8 files changed, 403 insertions(+), 79 deletions(-) create mode 100644 litellm/proxy/pass_through_endpoints/llm_provider_handlers/batch_attribution.py create mode 100644 tests/test_litellm/proxy/pass_through_endpoints/llm_provider_handlers/test_batch_attribution.py diff --git a/litellm/constants.py b/litellm/constants.py index c9d9ff155ff..8c3541067a5 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -472,6 +472,8 @@ EMAIL_BUDGET_ALERT_MAX_SPEND_ALERT_PERCENTAGE: Final = float( ### ANTHROPIC CONSTANTS ### ANTHROPIC_TOKEN_COUNTING_BETA_VERSION = os.getenv("ANTHROPIC_TOKEN_COUNTING_BETA_VERSION", "token-counting-2024-11-01") ANTHROPIC_SKILLS_API_BETA_VERSION: Final = "skills-2025-10-02" +ANTHROPIC_BATCHES_ROUTE: Final = "/v1/messages/batches" +VERTEX_BATCH_PREDICTION_JOBS_ROUTE: Final = "batchPredictionJobs" ANTHROPIC_WEB_SEARCH_TOOL_MAX_USES: Final = { "low": 1, "medium": 5, diff --git a/litellm/proxy/hooks/proxy_track_cost_callback.py b/litellm/proxy/hooks/proxy_track_cost_callback.py index 0e22b5324c1..4551680e1b4 100644 --- a/litellm/proxy/hooks/proxy_track_cost_callback.py +++ b/litellm/proxy/hooks/proxy_track_cost_callback.py @@ -39,11 +39,10 @@ _UNATTRIBUTED_TRACKABLE_CALL_TYPES: Final[frozenset[str]] = frozenset( CallTypes.pass_through.value, CallTypes.llm_passthrough_route.value, CallTypes.allm_passthrough_route.value, - # CheckBatchCost's synthetic logging_obj for a completed managed batch only ever - # carries user_api_key_user_id (from LiteLLM_ManagedObjectTable.created_by) and - # user_api_key_team_id (from .team_id) -- both are None for batches created with - # the master key or a team-less key, since the table never stores the raw key - # hash. The batch already incurred real provider cost, so track it regardless. + # CheckBatchCost's synthetic logging_obj for a completed managed batch carries + # whatever LiteLLM_ManagedObjectTable stored at create time, and all of it is + # None for a batch created before those columns were persisted, or by the master + # key. The batch already incurred real provider cost, so track it regardless. CallTypes.aretrieve_batch.value, } ) diff --git a/litellm/proxy/pass_through_endpoints/llm_provider_handlers/anthropic_passthrough_logging_handler.py b/litellm/proxy/pass_through_endpoints/llm_provider_handlers/anthropic_passthrough_logging_handler.py index 9fb967e570f..d0ca61e1fd1 100644 --- a/litellm/proxy/pass_through_endpoints/llm_provider_handlers/anthropic_passthrough_logging_handler.py +++ b/litellm/proxy/pass_through_endpoints/llm_provider_handlers/anthropic_passthrough_logging_handler.py @@ -1,3 +1,4 @@ +import asyncio import json from collections.abc import Sequence from datetime import datetime @@ -7,6 +8,7 @@ import httpx import litellm from litellm._logging import verbose_proxy_logger +from litellm.constants import ANTHROPIC_BATCHES_ROUTE from litellm.litellm_core_utils.core_helpers import map_finish_reason from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj from litellm.litellm_core_utils.litellm_logging import use_custom_pricing_for_model @@ -20,6 +22,12 @@ from litellm.llms.anthropic.chat.handler import ( from litellm.llms.anthropic.chat.transformation import AnthropicConfig from litellm.proxy._types import PassThroughEndpointLoggingTypedDict from litellm.proxy.auth.auth_utils import get_end_user_id_from_request_body +from litellm.proxy.pass_through_endpoints.llm_provider_handlers.batch_attribution import ( + is_collection_route, + log_batch_registration_result, + optional_str, + request_tags_from_metadata, +) from litellm.types.passthrough_endpoints.pass_through_endpoints import ( PassthroughStandardLoggingPayload, ) @@ -833,13 +841,14 @@ class AnthropicPassthroughLoggingHandler: # Store the managed object for cost tracking # This will be picked up by check_batch_cost polling mechanism - AnthropicPassthroughLoggingHandler._store_batch_managed_object( - unified_object_id=unified_object_id, - batch_object=litellm_batch_response, - model_object_id=batch_id, - logging_obj=logging_obj, - **kwargs, - ) + if is_collection_route(url_route, ANTHROPIC_BATCHES_ROUTE): + AnthropicPassthroughLoggingHandler._store_batch_managed_object( + unified_object_id=unified_object_id, + batch_object=litellm_batch_response, + model_object_id=batch_id, + logging_obj=logging_obj, + **kwargs, + ) # Create a batch job response for logging litellm_model_response = ModelResponse() @@ -964,8 +973,12 @@ class AnthropicPassthroughLoggingHandler: **kwargs, ) -> None: """ - Store batch managed object for cost tracking. + Register a newly created batch for cost tracking. This will be picked up by the check_batch_cost polling mechanism. + + Only the create reaches here, so the row records the creating key and its tags. + An id-scoped route cannot rebuild the unified object id anyway: the model comes + from the create's request body, which a retrieve does not have. """ try: # Get the managed files hook from the logging object @@ -981,7 +994,7 @@ class AnthropicPassthroughLoggingHandler: user_api_key_dict: Final = UserAPIKeyAuth( user_id=_request_metadata.get("user_api_key_user_id", "default-user"), - api_key="", + api_key=optional_str(_request_metadata.get("user_api_key")), team_id=_request_metadata.get("user_api_key_team_id"), team_alias=None, user_role=LitellmUserRoles.CUSTOMER, # Use proper enum value @@ -1003,9 +1016,7 @@ class AnthropicPassthroughLoggingHandler: ) # Store the unified object for batch cost tracking - import asyncio - - asyncio.create_task( + task: Final = asyncio.create_task( managed_files_hook.store_unified_object_id( unified_object_id=unified_object_id, file_object=batch_object, @@ -1013,13 +1024,14 @@ class AnthropicPassthroughLoggingHandler: model_object_id=model_object_id, file_purpose="batch", user_api_key_dict=user_api_key_dict, + request_tags=request_tags_from_metadata(_request_metadata), + persist_attribution=True, ) ) - - verbose_proxy_logger.info( - "Stored Anthropic batch managed object with unified_object_id=%s, batch_id=%s", - unified_object_id, - model_object_id, + task.add_done_callback( + lambda finished: log_batch_registration_result( + finished, "Anthropic", unified_object_id, model_object_id, is_batch_create=True + ) ) else: verbose_proxy_logger.warning( diff --git a/litellm/proxy/pass_through_endpoints/llm_provider_handlers/batch_attribution.py b/litellm/proxy/pass_through_endpoints/llm_provider_handlers/batch_attribution.py new file mode 100644 index 00000000000..e94145b3efe --- /dev/null +++ b/litellm/proxy/pass_through_endpoints/llm_provider_handlers/batch_attribution.py @@ -0,0 +1,78 @@ +"""Spend attribution for batches created through a passthrough endpoint. + +The creating key and its tags are read off the passthrough request's metadata and +persisted on the managed object row, because the batch cost lands hours later in a +background poll that has no request to read them from. +""" + +import asyncio +from collections.abc import Mapping, Sequence +from typing import Final + +from litellm._logging import verbose_proxy_logger + + +def optional_str(value: object) -> str | None: + return value if isinstance(value, str) else None + + +def _optional_str_tuple(value: object) -> tuple[str, ...] | None: + if not isinstance(value, list): + return None + items: Final[Sequence[object]] = value + return tuple(tag for tag in items if isinstance(tag, str)) + + +def is_collection_route(url_route: str, collection_suffix: str) -> bool: + """Whether the route addresses the batch collection itself rather than one batch. + A POST to the collection is the create; every id-scoped route is a retrieve, + results or cancel. + """ + return url_route.split("?")[0].rstrip("/").endswith(collection_suffix) + + +def request_tags_from_metadata(request_metadata: Mapping[str, object]) -> tuple[str, ...] | None: + """Tags for the batch-cost spend row: the request's own tags when it sent any, + otherwise the key's tags, which auth exposes as user_api_key_auth_metadata (a + tagged key does not put its tags in the top-level metadata "tags" on the + passthrough path) + """ + tags: Final = _optional_str_tuple(request_metadata.get("tags")) + if tags: + return tags + key_auth_metadata: Final = request_metadata.get("user_api_key_auth_metadata") + if isinstance(key_auth_metadata, dict): + return _optional_str_tuple(key_auth_metadata.get("tags")) + return None + + +def log_batch_registration_result( + finished: asyncio.Task[None], + provider: str, + unified_object_id: str, + model_object_id: str, + is_batch_create: bool, +) -> None: + """Report the outcome of the fire-and-forget managed object write. A create that + fails is not retried by a later poll, so its cost is never tracked at all. + """ + error: Final = finished.exception() if not finished.cancelled() else None + if finished.cancelled() or error is not None: + consequence: Final = ( + "its cost will not be tracked" if is_batch_create else "its status and output file may be stale" + ) + verbose_proxy_logger.error( + "Failed to store %s batch managed object with unified_object_id=%s, batch_id=%s; %s: %s", + provider, + unified_object_id, + model_object_id, + consequence, + error, + ) + return + verbose_proxy_logger.info( + "Stored %s batch managed object with unified_object_id=%s, batch_id=%s", + provider, + unified_object_id, + model_object_id, + ) diff --git a/litellm/proxy/pass_through_endpoints/llm_provider_handlers/vertex_passthrough_logging_handler.py b/litellm/proxy/pass_through_endpoints/llm_provider_handlers/vertex_passthrough_logging_handler.py index 7dee0e4a364..621b3ff9c83 100644 --- a/litellm/proxy/pass_through_endpoints/llm_provider_handlers/vertex_passthrough_logging_handler.py +++ b/litellm/proxy/pass_through_endpoints/llm_provider_handlers/vertex_passthrough_logging_handler.py @@ -1,6 +1,5 @@ import asyncio import re -from collections.abc import Mapping from datetime import datetime from typing import TYPE_CHECKING, Any, Final, cast from urllib.parse import urlparse @@ -9,6 +8,7 @@ import httpx import litellm from litellm._logging import verbose_proxy_logger +from litellm.constants import VERTEX_BATCH_PREDICTION_JOBS_ROUTE from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj from litellm.llms.vertex_ai.gemini.vertex_and_google_ai_studio_gemini import ( ModelResponseIterator as VertexModelResponseIterator, @@ -18,6 +18,12 @@ from litellm.llms.vertex_ai.vector_stores.search_api.transformation import ( ) from litellm.llms.vertex_ai.videos.transformation import VertexAIVideoConfig from litellm.proxy._types import PassThroughEndpointLoggingTypedDict +from litellm.proxy.pass_through_endpoints.llm_provider_handlers.batch_attribution import ( + is_collection_route, + log_batch_registration_result, + optional_str, + request_tags_from_metadata, +) from litellm.types.utils import ( Choices, EmbeddingResponse, @@ -41,32 +47,6 @@ else: EndpointType = Any -def _optional_str(value: object) -> str | None: - return value if isinstance(value, str) else None - - -def _optional_str_tuple(value: object) -> tuple[str, ...] | None: - if not isinstance(value, list): - return None - items: Final = cast(list[object], value) # cast-ok: isinstance-narrowed; element type unknown - return tuple(tag for tag in items if isinstance(tag, str)) - - -def _request_tags(request_metadata: Mapping[str, object]) -> tuple[str, ...] | None: - """Tags for the batch-cost spend row: the request's own tags when it sent any, - otherwise the key's tags, which auth exposes as user_api_key_auth_metadata (a - tagged key does not put its tags in the top-level metadata "tags" on the - passthrough path) - """ - tags: Final = _optional_str_tuple(request_metadata.get("tags")) - if tags: - return tags - key_auth_metadata: Final = request_metadata.get("user_api_key_auth_metadata") - if isinstance(key_auth_metadata, dict): - return _optional_str_tuple(key_auth_metadata.get("tags")) - return None - - class VertexPassthroughLoggingHandler: @staticmethod def vertex_passthrough_handler( @@ -685,7 +665,7 @@ class VertexPassthroughLoggingHandler: # Store the managed object for cost tracking # This will be picked up by check_batch_cost polling mechanism - is_batch_create: Final = url_route.split("?")[0].rstrip("/").endswith("batchPredictionJobs") + is_batch_create: Final = is_collection_route(url_route, VERTEX_BATCH_PREDICTION_JOBS_ROUTE) VertexPassthroughLoggingHandler._store_batch_managed_object( unified_object_id=unified_object_id, batch_object=litellm_batch_response, @@ -809,29 +789,6 @@ class VertexPassthroughLoggingHandler: "kwargs": kwargs, } - @staticmethod - def _log_batch_registration_result( - finished: asyncio.Task, unified_object_id: str, model_object_id: str, is_batch_create: bool - ) -> None: - error: Final = finished.exception() if not finished.cancelled() else None - if finished.cancelled() or error is not None: - consequence: Final = ( - "its cost will not be tracked" if is_batch_create else "its status and output file may be stale" - ) - verbose_proxy_logger.error( - "Failed to store batch managed object with unified_object_id=%s, batch_id=%s; %s: %s", - unified_object_id, - model_object_id, - consequence, - error, - ) - return - verbose_proxy_logger.info( - "Stored batch managed object with unified_object_id=%s, batch_id=%s", - unified_object_id, - model_object_id, - ) - @staticmethod def _store_batch_managed_object( unified_object_id: str, @@ -863,7 +820,7 @@ class VertexPassthroughLoggingHandler: user_api_key_dict: Final = UserAPIKeyAuth( user_id=_request_metadata.get("user_api_key_user_id", "default-user"), - api_key=_optional_str(_request_metadata.get("user_api_key")), + api_key=optional_str(_request_metadata.get("user_api_key")), team_id=_request_metadata.get("user_api_key_team_id"), team_alias=None, user_role=LitellmUserRoles.CUSTOMER, # Use proper enum value @@ -893,14 +850,14 @@ class VertexPassthroughLoggingHandler: model_object_id=model_object_id, file_purpose="batch", user_api_key_dict=user_api_key_dict, - request_tags=_request_tags(_request_metadata), + request_tags=request_tags_from_metadata(_request_metadata), persist_attribution=is_batch_create, create_if_missing=is_batch_create, ) ) task.add_done_callback( - lambda finished: VertexPassthroughLoggingHandler._log_batch_registration_result( - finished, unified_object_id, model_object_id, is_batch_create + lambda finished: log_batch_registration_result( + finished, "Vertex AI", unified_object_id, model_object_id, is_batch_create ) ) else: diff --git a/tests/test_litellm/enterprise/proxy/test_managed_files_hook.py b/tests/test_litellm/enterprise/proxy/test_managed_files_hook.py index 34cd0cabc2c..d260f79a09a 100644 --- a/tests/test_litellm/enterprise/proxy/test_managed_files_hook.py +++ b/tests/test_litellm/enterprise/proxy/test_managed_files_hook.py @@ -604,8 +604,8 @@ async def test_create_still_upserts_and_claims_attribution(): @pytest.mark.asyncio async def test_default_callers_still_create_their_rows(): - """create_if_missing defaults to True, so the fine-tune, Responses and Anthropic - callers, none of which pass it, keep upserting exactly as before.""" + """create_if_missing defaults to True, so the fine-tune, Responses and managed + /v1/batches callers, none of which passes it, keep upserting exactly as before.""" managed_files, mock_prisma = _make_object_store_instance() await managed_files.store_unified_object_id( diff --git a/tests/test_litellm/proxy/pass_through_endpoints/llm_provider_handlers/test_anthropic_passthrough_logging_handler.py b/tests/test_litellm/proxy/pass_through_endpoints/llm_provider_handlers/test_anthropic_passthrough_logging_handler.py index 947a7a64beb..1c98e3ce535 100644 --- a/tests/test_litellm/proxy/pass_through_endpoints/llm_provider_handlers/test_anthropic_passthrough_logging_handler.py +++ b/tests/test_litellm/proxy/pass_through_endpoints/llm_provider_handlers/test_anthropic_passthrough_logging_handler.py @@ -1,3 +1,4 @@ +import asyncio import json import os import sys @@ -17,6 +18,13 @@ from litellm.proxy.pass_through_endpoints.llm_provider_handlers.anthropic_passth ) +async def _drain_tasks(): + """Await the fire-and-forget managed object write and let its done callback run.""" + pending = [t for t in asyncio.all_tasks() if t is not asyncio.current_task()] + await asyncio.gather(*pending, return_exceptions=True) + await asyncio.sleep(0) + + class TestAnthropicLoggingHandlerModelFallback: """Test the model fallback logic in the anthropic passthrough logging handler.""" @@ -925,6 +933,114 @@ class TestAnthropicBatchPassthroughCostTracking: assert call_kwargs["user_api_key_dict"].user_id == expected_user_id assert call_kwargs["user_api_key_dict"].team_id == expected_team_id + async def _store_with_metadata(self, mock_logging_obj, metadata): + mock_managed_files_hook = MagicMock() + mock_managed_files_hook.store_unified_object_id = AsyncMock() + with ( + patch("litellm.proxy.proxy_server.proxy_logging_obj") as mock_pl, + patch( + "litellm.proxy.pass_through_endpoints.llm_provider_handlers.batch_attribution.verbose_proxy_logger" + ), + ): + mock_pl.get_proxy_hook.return_value = mock_managed_files_hook + AnthropicPassthroughLoggingHandler._store_batch_managed_object( + unified_object_id="uoi", + batch_object={"id": "b1", "object": "batch", "status": "validating"}, + model_object_id="b1", + logging_obj=mock_logging_obj, + litellm_params={"metadata": metadata}, + ) + await _drain_tasks() + mock_managed_files_hook.store_unified_object_id.assert_awaited_once() + return mock_managed_files_hook.store_unified_object_id.call_args[1] + + @pytest.mark.asyncio + async def test_create_persists_key_hash_and_tags(self, mock_logging_obj): + """Regression (LIT-5288): the batch create must persist the creating key's hashed + token and its tags so CheckBatchCost can attribute the batch-cost spend row to the + key, team and tags. Before this fix the stored api_key was always "" and no tags + were stored, so key/team/tag spend and budgets never moved for batch usage.""" + call_kwargs = await self._store_with_metadata( + mock_logging_obj, + { + "user_api_key": "hashed-key-a", + "user_api_key_user_id": "alice", + "user_api_key_team_id": "team-alpha", + "user_api_key_auth_metadata": {"tags": ["env:prod", 7, "team:ml"]}, + }, + ) + + assert call_kwargs["user_api_key_dict"].api_key == "hashed-key-a" + assert call_kwargs["request_tags"] == ("env:prod", "team:ml") + assert call_kwargs["persist_attribution"] is True + + @pytest.mark.asyncio + async def test_failed_create_write_is_reported_not_swallowed(self, mock_logging_obj): + """The managed object write is fire-and-forget, and only the create writes the row, + so a failed create is never back-filled by a later retrieve and that batch's cost + is never tracked. The failure has to reach the log instead of being reported as a + success.""" + mock_managed_files_hook = MagicMock() + mock_managed_files_hook.store_unified_object_id = AsyncMock( + side_effect=RuntimeError("db down") + ) + with ( + patch("litellm.proxy.proxy_server.proxy_logging_obj") as mock_pl, + patch( + "litellm.proxy.pass_through_endpoints.llm_provider_handlers.batch_attribution.verbose_proxy_logger" + ) as mock_logger, + ): + mock_pl.get_proxy_hook.return_value = mock_managed_files_hook + AnthropicPassthroughLoggingHandler._store_batch_managed_object( + unified_object_id="uoi", + batch_object={"id": "b1", "object": "batch", "status": "validating"}, + model_object_id="b1", + logging_obj=mock_logging_obj, + litellm_params={"metadata": {"user_api_key": "hashed-key-a"}}, + ) + await _drain_tasks() + + mock_logger.info.assert_not_called() + mock_logger.error.assert_called_once() + assert "its cost will not be tracked" in mock_logger.error.call_args[0] + assert "Anthropic" in mock_logger.error.call_args[0] + + @pytest.mark.parametrize( + "url_route, registers", + [ + ("https://api.anthropic.com/v1/messages/batches", True), + ("https://api.anthropic.com/v1/messages/batches/", True), + ("https://api.anthropic.com/v1/messages/batches?limit=20", True), + ("https://api.anthropic.com/v1/messages/batches/msgbatch_123", False), + ("https://api.anthropic.com/v1/messages/batches/msgbatch_123/results", False), + ("https://api.anthropic.com/v1/messages/batches/msgbatch_123/cancel", False), + ], + ) + def test_batch_is_registered_from_the_create_route_only( + self, mock_logging_obj, mock_httpx_response, mock_request_body, url_route, registers + ): + """Only a POST to the collection route registers the batch. Every id-scoped route + is a retrieve, results or cancel, and none of them can rebuild the unified object + id anyway: it embeds the model, which comes from the create's request body. Before + this gate an id-scoped route reached the store with a mismatched id, where it could + only either claim a row it did not create or fail the model_object_id unique + constraint.""" + with patch.object( + AnthropicPassthroughLoggingHandler, "_store_batch_managed_object" + ) as mock_store: + AnthropicPassthroughLoggingHandler.batch_creation_handler( + httpx_response=mock_httpx_response, + logging_obj=mock_logging_obj, + url_route=url_route, + result="success", + start_time=datetime.now(), + end_time=datetime.now(), + cache_hit=False, + request_body=mock_request_body, + ) + + assert mock_store.call_count == (1 if registers else 0) + def test_batch_creation_handler_failure_status_code( self, mock_logging_obj, mock_request_body ): @@ -978,6 +1094,7 @@ class TestAnthropicBatchPassthroughCostTracking: batch_object=batch_object, model_object_id="msgbatch_123", logging_obj=mock_logging_obj, + is_batch_create=True, user_id="test-user", ) diff --git a/tests/test_litellm/proxy/pass_through_endpoints/llm_provider_handlers/test_batch_attribution.py b/tests/test_litellm/proxy/pass_through_endpoints/llm_provider_handlers/test_batch_attribution.py new file mode 100644 index 00000000000..5109cbb5991 --- /dev/null +++ b/tests/test_litellm/proxy/pass_through_endpoints/llm_provider_handlers/test_batch_attribution.py @@ -0,0 +1,159 @@ +import asyncio +from unittest.mock import patch + +import pytest + +from litellm.proxy.pass_through_endpoints.llm_provider_handlers.batch_attribution import ( + is_collection_route, + log_batch_registration_result, + optional_str, + request_tags_from_metadata, +) + + +@pytest.mark.parametrize( + "value, expected", + [("a", "a"), ("", ""), (None, None), (7, None), (["a"], None)], +) +def test_optional_str(value, expected): + assert optional_str(value) == expected + + +class TestRequestTagsFromMetadata: + """Tags for the batch-cost spend row. These feed LiteLLM_ManagedObjectTable.request_tags, + which is the only record of the creating request's tags by the time CheckBatchCost bills + the batch hours later.""" + + @pytest.mark.parametrize( + "metadata, expected", + [ + # a request that sent its own tags (x-litellm-tags header or body metadata) + ({"tags": ["req:a", "req:b"]}, ("req:a", "req:b")), + # request tags win over the key's own tags + ( + {"tags": ["req:a"], "user_api_key_auth_metadata": {"tags": ["key:b"]}}, + ("req:a",), + ), + # no request tags: fall back to the tags the key itself carries, because a + # tagged key does not put its tags in the top-level metadata on this path + ({"user_api_key_auth_metadata": {"tags": ["key:b"]}}, ("key:b",)), + # an empty request tag list is not a selection, so the key's tags still apply + ( + {"tags": [], "user_api_key_auth_metadata": {"tags": ["key:b"]}}, + ("key:b",), + ), + # neither: no tags on the spend row + ({}, None), + # order is preserved, so the spend row is reproducible + ({"tags": ["z", "a", "m"]}, ("z", "a", "m")), + ], + ) + def test_precedence(self, metadata, expected): + assert request_tags_from_metadata(metadata) == expected + + @pytest.mark.parametrize( + "raw, expected", + [ + # non-string entries are dropped rather than crashing the create + (["env:prod", 7, None, "team:ml"], ("env:prod", "team:ml")), + # nothing usable survives, so this is treated as no request tags at all + ([7, None], None), + # a non-list is not a tag list + ("env:prod", None), + ({"env": "prod"}, None), + (None, None), + ], + ) + def test_malformed_tags_are_dropped(self, raw, expected): + assert request_tags_from_metadata({"tags": raw}) == expected + + def test_malformed_key_auth_metadata_is_ignored(self): + assert request_tags_from_metadata({"user_api_key_auth_metadata": "nope"}) is None + + +@pytest.mark.parametrize( + "url_route, suffix, expected", + [ + ("https://api.anthropic.com/v1/messages/batches", "/v1/messages/batches", True), + ("https://api.anthropic.com/v1/messages/batches/", "/v1/messages/batches", True), + ("https://api.anthropic.com/v1/messages/batches?limit=20", "/v1/messages/batches", True), + ("https://api.anthropic.com/v1/messages/batches/msgbatch_1", "/v1/messages/batches", False), + # a proxied base with a path prefix still resolves, because this is a suffix match + ("https://gateway.internal/anthropic/v1/messages/batches", "/v1/messages/batches", True), + ("https://aiplatform.googleapis.com/v1/projects/p/locations/l/batchPredictionJobs", "batchPredictionJobs", True), + ("https://aiplatform.googleapis.com/v1/projects/p/locations/l/batchPredictionJobs/9", "batchPredictionJobs", False), + ], +) +def test_is_collection_route(url_route, suffix, expected): + assert is_collection_route(url_route, suffix) is expected + + +class TestLogBatchRegistrationResult: + """The managed object write is fire and forget, so its outcome only ever reaches an + operator through this log line.""" + + @staticmethod + async def _finished_task(coro): + task = asyncio.ensure_future(coro) + await asyncio.gather(task, return_exceptions=True) + return task + + @pytest.mark.asyncio + async def test_success_names_the_provider(self): + async def ok(): + return None + + task = await self._finished_task(ok()) + with patch( + "litellm.proxy.pass_through_endpoints.llm_provider_handlers.batch_attribution.verbose_proxy_logger" + ) as logger: + log_batch_registration_result(task, "Anthropic", "uoi", "b1", is_batch_create=True) + + logger.error.assert_not_called() + logger.info.assert_called_once() + assert "Anthropic" in logger.info.call_args[0] + + @pytest.mark.asyncio + async def test_a_failed_create_says_the_cost_is_lost(self): + async def boom(): + raise RuntimeError("db down") + + task = await self._finished_task(boom()) + with patch( + "litellm.proxy.pass_through_endpoints.llm_provider_handlers.batch_attribution.verbose_proxy_logger" + ) as logger: + log_batch_registration_result(task, "Vertex AI", "uoi", "b1", is_batch_create=True) + + logger.info.assert_not_called() + assert "its cost will not be tracked" in logger.error.call_args[0] + + @pytest.mark.asyncio + async def test_a_failed_refresh_says_the_row_is_stale(self): + async def boom(): + raise RuntimeError("db down") + + task = await self._finished_task(boom()) + with patch( + "litellm.proxy.pass_through_endpoints.llm_provider_handlers.batch_attribution.verbose_proxy_logger" + ) as logger: + log_batch_registration_result(task, "Vertex AI", "uoi", "b1", is_batch_create=False) + + logger.info.assert_not_called() + assert "its status and output file may be stale" in logger.error.call_args[0] + + @pytest.mark.asyncio + async def test_a_cancelled_write_is_reported_not_reraised(self): + async def slow(): + await asyncio.sleep(60) + + task = asyncio.ensure_future(slow()) + await asyncio.sleep(0) + task.cancel() + await asyncio.gather(task, return_exceptions=True) + with patch( + "litellm.proxy.pass_through_endpoints.llm_provider_handlers.batch_attribution.verbose_proxy_logger" + ) as logger: + log_batch_registration_result(task, "Anthropic", "uoi", "b1", is_batch_create=True) + + logger.info.assert_not_called() + logger.error.assert_called_once() From 5e14649c5420312b10187071b52b6afaf262da47 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Tue, 11 Aug 2026 21:01:18 -0700 Subject: [PATCH 8/9] fix(passthrough): bill streamed Responses calls that end failed A stream can terminate with a response.failed event that still reports consumed tokens; those were rebuilt as None and logged at zero spend. Parse response.failed alongside completed and incomplete, matching the buffered path, which prices any terminal response that reports usage. --- .../llms/openai/responses/transformation.py | 2 +- ...test_openai_passthrough_logging_handler.py | 57 +++++++++++++++++++ 2 files changed, 58 insertions(+), 1 deletion(-) diff --git a/litellm/llms/openai/responses/transformation.py b/litellm/llms/openai/responses/transformation.py index 568beed8664..b2a69564908 100644 --- a/litellm/llms/openai/responses/transformation.py +++ b/litellm/llms/openai/responses/transformation.py @@ -356,7 +356,7 @@ class OpenAIResponsesAPIConfig(BaseResponsesAPIConfig): @staticmethod def parse_terminal_response_from_stream_chunks(all_chunks: list[str]) -> ResponsesAPIResponse | None: for chunk_str in reversed(all_chunks): - for event_model in (ResponseCompletedEvent, ResponseIncompleteEvent): + for event_model in (ResponseCompletedEvent, ResponseIncompleteEvent, ResponseFailedEvent): try: return event_model.model_validate_json(chunk_str.removeprefix("data: ")).response except ValueError: diff --git a/tests/test_litellm/proxy/pass_through_endpoints/llm_provider_handlers/test_openai_passthrough_logging_handler.py b/tests/test_litellm/proxy/pass_through_endpoints/llm_provider_handlers/test_openai_passthrough_logging_handler.py index b0576960355..1735308bda0 100644 --- a/tests/test_litellm/proxy/pass_through_endpoints/llm_provider_handlers/test_openai_passthrough_logging_handler.py +++ b/tests/test_litellm/proxy/pass_through_endpoints/llm_provider_handlers/test_openai_passthrough_logging_handler.py @@ -664,6 +664,63 @@ class TestOpenAIPassthroughLoggingHandler: call_type="responses", ) + @patch(f"{OpenAIPassthroughLoggingHandler.__module__}.get_standard_logging_object_payload") + @patch("litellm.completion_cost", return_value=1.4e-06) + def test_streaming_responses_failed_event_is_billed(self, mock_completion_cost, mock_get_standard_logging): + response_id = "resp_FAILEDSENTINEL0123456789abcd" + failed_event = { + "type": "response.failed", + "sequence_number": 4, + "response": { + "id": response_id, + "object": "response", + "created_at": 1786374786, + "status": "failed", + "model": "gpt-4o-mini-2024-07-18", + "output": [], + "usage": { + "input_tokens": 14, + "input_tokens_details": {"cached_tokens": 0}, + "output_tokens": 7, + "output_tokens_details": {"reasoning_tokens": 0}, + "total_tokens": 21, + }, + "error": {"code": "server_error", "message": "The model failed to generate a response"}, + "incomplete_details": None, + "instructions": None, + "metadata": {}, + "parallel_tool_calls": True, + "temperature": 1.0, + "tool_choice": "auto", + "tools": [], + "top_p": 1.0, + }, + } + logging_obj = self._create_mock_logging_obj() + + result = OpenAIPassthroughLoggingHandler._handle_logging_openai_collected_chunks( + litellm_logging_obj=logging_obj, + passthrough_success_handler_obj=MagicMock(), + url_route="https://api.openai.com/v1/responses", + request_body={"model": "gpt-4o-mini", "stream": True}, + endpoint_type=MagicMock(), + start_time=self.start_time, + all_chunks=[f"data: {json.dumps(failed_event)}"], + end_time=self.end_time, + ) + + response = result["result"] + assert response.id == response_id + assert response.status == "failed" + assert response.usage.total_tokens == 21 + assert result["kwargs"]["response_cost"] == 1.4e-06 + mock_completion_cost.assert_called_once_with( + completion_response=response, + model="gpt-4o-mini", + custom_llm_provider="openai", + call_type="responses", + ) + @patch(f"{OpenAIPassthroughLoggingHandler.__module__}.get_standard_logging_object_payload") @patch("litellm.completion_cost") def test_streaming_responses_without_completed_event_returns_none( From 08a73740ec3dba37d3109af6b1a9bdadb680781e Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Tue, 11 Aug 2026 21:28:55 -0700 Subject: [PATCH 9/9] fix(passthrough): keep prompt/completion token split for streamed OpenAI rows --- .../openai_passthrough_logging_handler.py | 11 +++- ...test_openai_passthrough_logging_handler.py | 50 +++++++++++++++++++ 2 files changed, 59 insertions(+), 2 deletions(-) diff --git a/litellm/proxy/pass_through_endpoints/llm_provider_handlers/openai_passthrough_logging_handler.py b/litellm/proxy/pass_through_endpoints/llm_provider_handlers/openai_passthrough_logging_handler.py index 1614784a0dc..b40de351847 100644 --- a/litellm/proxy/pass_through_endpoints/llm_provider_handlers/openai_passthrough_logging_handler.py +++ b/litellm/proxy/pass_through_endpoints/llm_provider_handlers/openai_passthrough_logging_handler.py @@ -583,6 +583,8 @@ class OpenAIPassthroughLoggingHandler(BasePassthroughLoggingHandler): "response_cost": response_cost, "model": model, "custom_llm_provider": custom_llm_provider, + "call_type": litellm_logging_obj.call_type, + "messages": litellm_logging_obj.model_call_details.get("messages"), "litellm_params": existing_litellm_params.copy(), } @@ -599,8 +601,11 @@ class OpenAIPassthroughLoggingHandler(BasePassthroughLoggingHandler): user ) - # Create standard logging object - get_standard_logging_object_payload( + # Attach the payload to kwargs so the success handler adopts it; + # its later rebuild runs on a copy whose Responses usage was + # coerced to chat shape and serializes as total_tokens only, + # zeroing the prompt/completion split in spend logs. + standard_logging_object: Final = get_standard_logging_object_payload( kwargs=kwargs, init_response_obj=complete_response, start_time=start_time, @@ -608,6 +613,8 @@ class OpenAIPassthroughLoggingHandler(BasePassthroughLoggingHandler): logging_obj=litellm_logging_obj, status="success", ) + if standard_logging_object is not None: + kwargs["standard_logging_object"] = standard_logging_object # Update logging object with cost information litellm_logging_obj.model_call_details["model"] = model diff --git a/tests/test_litellm/proxy/pass_through_endpoints/llm_provider_handlers/test_openai_passthrough_logging_handler.py b/tests/test_litellm/proxy/pass_through_endpoints/llm_provider_handlers/test_openai_passthrough_logging_handler.py index 1735308bda0..05051ab3745 100644 --- a/tests/test_litellm/proxy/pass_through_endpoints/llm_provider_handlers/test_openai_passthrough_logging_handler.py +++ b/tests/test_litellm/proxy/pass_through_endpoints/llm_provider_handlers/test_openai_passthrough_logging_handler.py @@ -586,6 +586,7 @@ class TestOpenAIPassthroughLoggingHandler: assert response.usage.input_tokens == 14 assert response.usage.output_tokens == 2 assert result["kwargs"]["response_cost"] == 3.3e-06 + assert result["kwargs"]["standard_logging_object"] is mock_get_standard_logging.return_value mock_completion_cost.assert_called_once_with( completion_response=response, model="gpt-4o-mini", @@ -657,6 +658,7 @@ class TestOpenAIPassthroughLoggingHandler: assert response.status == "incomplete" assert response.usage.output_tokens == 32 assert result["kwargs"]["response_cost"] == 2.1e-06 + assert result["kwargs"]["standard_logging_object"] is mock_get_standard_logging.return_value mock_completion_cost.assert_called_once_with( completion_response=response, model="gpt-4o-mini", @@ -714,6 +716,7 @@ class TestOpenAIPassthroughLoggingHandler: assert response.status == "failed" assert response.usage.total_tokens == 21 assert result["kwargs"]["response_cost"] == 1.4e-06 + assert result["kwargs"]["standard_logging_object"] is mock_get_standard_logging.return_value mock_completion_cost.assert_called_once_with( completion_response=response, model="gpt-4o-mini", @@ -721,6 +724,53 @@ class TestOpenAIPassthroughLoggingHandler: call_type="responses", ) + @patch(f"{OpenAIPassthroughLoggingHandler.__module__}.get_standard_logging_object_payload", return_value=None) + @patch("litellm.completion_cost", return_value=3.3e-06) + def test_streaming_responses_none_payload_is_not_attached(self, mock_completion_cost, mock_get_standard_logging): + completed_event = { + "type": "response.completed", + "sequence_number": 8, + "response": { + "id": "resp_NONEPAYLOADSENTINEL0123456789", + "object": "response", + "created_at": 1786374786, + "status": "completed", + "model": "gpt-4o-mini-2024-07-18", + "output": [], + "usage": { + "input_tokens": 14, + "input_tokens_details": {"cached_tokens": 0}, + "output_tokens": 2, + "output_tokens_details": {"reasoning_tokens": 0}, + "total_tokens": 16, + }, + "error": None, + "incomplete_details": None, + "instructions": None, + "metadata": {}, + "parallel_tool_calls": True, + "temperature": 1.0, + "tool_choice": "auto", + "tools": [], + "top_p": 1.0, + }, + } + logging_obj = self._create_mock_logging_obj() + + result = OpenAIPassthroughLoggingHandler._handle_logging_openai_collected_chunks( + litellm_logging_obj=logging_obj, + passthrough_success_handler_obj=MagicMock(), + url_route="https://api.openai.com/v1/responses", + request_body={"model": "gpt-4o-mini", "stream": True}, + endpoint_type=MagicMock(), + start_time=self.start_time, + all_chunks=[f"data: {json.dumps(completed_event)}", "data: [DONE]"], + end_time=self.end_time, + ) + + assert "standard_logging_object" not in result["kwargs"] + assert result["kwargs"]["response_cost"] == 3.3e-06 + @patch(f"{OpenAIPassthroughLoggingHandler.__module__}.get_standard_logging_object_payload") @patch("litellm.completion_cost") def test_streaming_responses_without_completed_event_returns_none(