diff --git a/litellm/litellm_core_utils/streaming_handler.py b/litellm/litellm_core_utils/streaming_handler.py index 60dbf7c644a..c2b0ca5e9db 100644 --- a/litellm/litellm_core_utils/streaming_handler.py +++ b/litellm/litellm_core_utils/streaming_handler.py @@ -1465,6 +1465,7 @@ class CustomStreamWrapper: if self.stream_options is not None and self.stream_options["include_usage"] is True: model_response.choices = [] return model_response + self._record_usage_only_chunk(model_response=model_response) return ## CHECK FOR TOOL USE @@ -1691,6 +1692,17 @@ class CustomStreamWrapper: model_response.choices[0].finish_reason = "tool_calls" return model_response + def _record_usage_only_chunk(self, model_response: "ModelResponseStream") -> None: + """ + Keep provider usage-only chunks (e.g. OpenRouter's post-finish chunk, which carries a + provider-reported cost) available to cost tracking. They are never returned to the + caller; ``stream_options.include_usage`` only controls what the caller sees. + """ + if getattr(model_response, "usage", None) is None: + return + model_response.choices = [] + self.chunks.append(model_response) + @staticmethod def _propagate_usage_cost_to_hidden_params( response: "ModelResponse", diff --git a/tests/test_litellm/litellm_core_utils/test_streaming_handler.py b/tests/test_litellm/litellm_core_utils/test_streaming_handler.py index 514714136fd..22daaf64dfd 100644 --- a/tests/test_litellm/litellm_core_utils/test_streaming_handler.py +++ b/tests/test_litellm/litellm_core_utils/test_streaming_handler.py @@ -1507,6 +1507,74 @@ async def test_openrouter_streaming_cost_after_finish_reason(logging_obj: Loggin assert usage_chunks[-1].usage.cost == 0.00025 +@pytest.mark.asyncio +async def test_openrouter_streaming_usage_only_chunk_without_stream_options( + logging_obj: Logging, +): + """ + Regression: OpenRouter's post-finish chunk has `choices: []`. When the caller did not + pass stream_options.include_usage it was dropped before cost tracking, so the + provider-reported cost never reached the assembled response. + """ + from litellm.cost_calculator import get_response_cost_from_hidden_params + from litellm.utils import ModelResponseListIterator + + chunk1 = ModelResponseStream( + id="chatcmpl-or", + created=1742056047, + model="openrouter/claude", + choices=[ + StreamingChoices( + finish_reason=None, index=0, delta=Delta(content="Hi", role="assistant") + ) + ], + usage=None, + ) + chunk2 = ModelResponseStream( + id="chatcmpl-or", + created=1742056048, + model="openrouter/claude", + choices=[ + StreamingChoices(finish_reason="stop", index=0, delta=Delta(content="")) + ], + usage=None, + ) + usage_only_chunk = ModelResponseStream( + id="chatcmpl-or", + created=1742056049, + model="openrouter/claude", + choices=[], + usage=Usage( + completion_tokens=5, prompt_tokens=10, total_tokens=15, cost=0.00025 + ), + ) + + response = CustomStreamWrapper( + completion_stream=ModelResponseListIterator( + model_responses=[chunk1, chunk2, usage_only_chunk] + ), + model="openrouter/claude", + custom_llm_provider="openrouter", + logging_obj=logging_obj, + ) + + collected_chunks = [chunk async for chunk in response] + + assert all(getattr(chunk, "usage", None) is None for chunk in collected_chunks) + + complete_response = litellm.stream_chunk_builder( + chunks=response.chunks, + messages=[{"role": "user", "content": "Hey"}], + ) + assert complete_response is not None + assert complete_response.usage.cost == 0.00025 + + CustomStreamWrapper._propagate_usage_cost_to_hidden_params(complete_response) + assert ( + get_response_cost_from_hidden_params(complete_response._hidden_params) == 0.00025 + ) + + def test_openrouter_streaming_cost_propagates_to_hidden_params(): """ Verify that provider-reported cost from usage.cost flows into