From 375659ef042378a9b9733db0b61f57de3a824342 Mon Sep 17 00:00:00 2001 From: Mateo Wang <277851410+mateo-berri@users.noreply.github.com> Date: Mon, 29 Jun 2026 20:29:49 -0700 Subject: [PATCH] fix(proxy): emit x-litellm-response-cost header on /messages and /generateContent (LIT-4076) (#31675) * fix(proxy): emit x-litellm-response-cost header on /messages and /generateContent (LIT-4076) The Anthropic /v1/messages and Google native :generateContent routes return TypedDict results (AnthropicMessagesResponse, GenerateContentResponseBody) that are plain dicts at runtime and cannot hold a _hidden_params attribute. The cost is computed by update_response_metadata, but ResponseMetadata.apply() only persists _hidden_params back when the result object has that attribute, so for those two routes the computed response_cost was dropped. The non-streaming header build in base_process_llm_request then saw an empty response_cost and get_custom_headers filtered the x-litellm-response-cost header out, even though the other x-litellm-* headers still appeared. The non-streaming success path now recovers the cost from the logging object when the response cannot carry _hidden_params, preferring the value already stored in model_call_details and recomputing from the same calculator only when it has not been stored yet. Object responses (ModelResponse, ResponsesAPIResponse) keep their existing behavior, so chat/completions, /responses, and the Anthropic error path that intentionally emits a zero cost are unaffected. Streaming stays out of scope because the header is emitted at stream start, before the cost is known. * fix(proxy): also recover response cost header for /generateContent responses with _hidden_params (LIT-4076) * fix(proxy): compute generateContent response cost synchronously so cost header is emitted (LIT-4076) * fix(lint): suppress BLE001 on generate_content cost normalization guard The defensive blind except keeps cost normalization from ever breaking the response path; mark it noqa so it does not breach the strict-rule budget. --- litellm/litellm_core_utils/litellm_logging.py | 37 +++ litellm/proxy/common_request_processing.py | 29 +- .../test_litellm_logging.py | 65 ++++ .../proxy/test_common_request_processing.py | 286 ++++++++++++++++++ 4 files changed, 416 insertions(+), 1 deletion(-) diff --git a/litellm/litellm_core_utils/litellm_logging.py b/litellm/litellm_core_utils/litellm_logging.py index 2457d117b81..245084b7c4a 100644 --- a/litellm/litellm_core_utils/litellm_logging.py +++ b/litellm/litellm_core_utils/litellm_logging.py @@ -1384,6 +1384,10 @@ class Logging(LiteLLMLoggingBaseClass): if cache_hit is True: return 0.0 + transformed_result = self._generate_content_result_as_model_response(result) + if transformed_result is not None: + result = transformed_result + if isinstance(result, BaseModel) and hasattr(result, "_hidden_params"): hidden_params = getattr(result, "_hidden_params", {}) if ( @@ -1463,6 +1467,39 @@ class Logging(LiteLLMLoggingBaseClass): return None + def _generate_content_result_as_model_response(self, result: object) -> Optional[ModelResponse]: + """ + Native Google :generateContent bodies report token usage under + ``usageMetadata``, which the cost calculator does not read, so a raw body + always costs 0. The async success path already transforms it into a + ``ModelResponse`` before costing; do the same transformation here so the + synchronously-built ``x-litellm-response-cost`` header carries the real + cost. Returns ``None`` (leaving the original result untouched) for other + call types, for already-transformed ``ModelResponse`` results, and on any + transformation failure. + """ + if self.call_type not in ( + CallTypes.generate_content.value, + CallTypes.agenerate_content.value, + ): + return None + if isinstance(result, ModelResponse) or not isinstance(result, (BaseModel, dict)): + return None + try: + import httpx + + completion_response = result.model_dump(by_alias=True) if isinstance(result, BaseModel) else dict(result) + return litellm.VertexGeminiConfig()._transform_google_generate_content_to_openai_model_response( + completion_response=completion_response, + model_response=ModelResponse(), + model=self.model or "", + logging_obj=self, + raw_response=httpx.Response(status_code=200, headers={}), + ) + except Exception as e: # noqa: BLE001 - cost normalization must never break the response path + verbose_logger.debug(f"generate_content response cost normalization failed: {e}") + return None + async def _response_cost_calculator_async( self, result: Union[ diff --git a/litellm/proxy/common_request_processing.py b/litellm/proxy/common_request_processing.py index 4d931a47e9d..97f7d51970c 100644 --- a/litellm/proxy/common_request_processing.py +++ b/litellm/proxy/common_request_processing.py @@ -1184,6 +1184,26 @@ class ProxyBaseLLMRequestProcessing: model_id = model_info.get("id", "") or "" return model_id + @staticmethod + def _response_cost_from_logging_obj( + *, + response: Any, + logging_obj: LiteLLMLoggingObj, + ) -> float | str: + """ + Recover the response cost when the response never recorded one in its + ``_hidden_params``: Anthropic /v1/messages returns a TypedDict that cannot + hold the attribute at all, and Google :generateContent carries + ``_hidden_params`` but no synchronously-populated ``response_cost``. In both + cases the cost is read back from the logging object instead, recomputing from + the same calculator only when it has not been stored yet. + """ + stored_cost = logging_obj.model_call_details.get("response_cost") + if isinstance(stored_cost, (int, float)): + return float(stored_cost) + recomputed_cost = logging_obj._response_cost_calculator(result=response) + return recomputed_cost if isinstance(recomputed_cost, (int, float)) else "" + def _debug_log_request_payload(self) -> None: """Log request payload at DEBUG level, truncating if too large.""" if not verbose_proxy_logger.isEnabledFor(logging.DEBUG): @@ -1687,6 +1707,13 @@ class ProxyBaseLLMRequestProcessing: hidden_params = getattr(response, "_hidden_params", {}) or {} # get any updated response headers additional_headers = hidden_params.get("additional_headers", {}) or {} + recover_response_cost = not response_cost and hidden_params.get("response_cost") is None + response_cost_for_headers = ( + self._response_cost_from_logging_obj(response=response, logging_obj=logging_obj) or "" + if recover_response_cost + else response_cost + ) + fastapi_response.headers.update( ProxyBaseLLMRequestProcessing.get_custom_headers( user_api_key_dict=user_api_key_dict, @@ -1695,7 +1722,7 @@ class ProxyBaseLLMRequestProcessing: cache_key=cache_key, api_base=api_base, version=version, - response_cost=response_cost, + response_cost=response_cost_for_headers, model_region=getattr(user_api_key_dict, "allowed_model_region", ""), fastest_response_batch_completion=fastest_response_batch_completion, request_data=self.data, diff --git a/tests/test_litellm/litellm_core_utils/test_litellm_logging.py b/tests/test_litellm/litellm_core_utils/test_litellm_logging.py index 5a8ac0313a7..dcbc5002ee4 100644 --- a/tests/test_litellm/litellm_core_utils/test_litellm_logging.py +++ b/tests/test_litellm/litellm_core_utils/test_litellm_logging.py @@ -1424,6 +1424,71 @@ def test_response_cost_calculator_with_response_cost_in_hidden_params(logging_ob assert response_cost > 100 +def test_response_cost_calculator_native_generate_content_body_uses_usage_metadata(): + """ + Regression for LIT-4076: a native Google :generateContent body reports tokens + under ``usageMetadata`` rather than ``usage``, so the cost calculator read 0 + tokens and returned 0.0 synchronously. The calculator now transforms the native + body (as the async logging path does) so the cost is the real non-zero amount. + """ + from litellm.types.llms.vertex_ai import GenerateContentResponseBody + from litellm.types.utils import ModelResponse, Usage + + logging_obj = LitellmLogging( + model="gemini-2.5-flash", + messages=[{"role": "user", "content": "Hey"}], + stream=False, + call_type="agenerate_content", + start_time=time.time(), + litellm_call_id="lit4076", + function_id="lit4076", + ) + logging_obj.model_call_details["custom_llm_provider"] = "gemini" + logging_obj.optional_params = {} + + native_body = GenerateContentResponseBody( + candidates=[{"content": {"parts": [{"text": "hi"}], "role": "model"}, "finishReason": "STOP"}], + usageMetadata={ + "promptTokenCount": 1000, + "candidatesTokenCount": 500, + "totalTokenCount": 1500, + }, + ) + + expected_cost = litellm.completion_cost( + completion_response=ModelResponse( + model="gemini-2.5-flash", + usage=Usage(prompt_tokens=1000, completion_tokens=500, total_tokens=1500), + ), + model="gemini-2.5-flash", + custom_llm_provider="gemini", + ) + assert expected_cost > 0 + + cost = logging_obj._response_cost_calculator(result=native_body) + assert cost == pytest.approx(expected_cost) + + +def test_response_cost_calculator_does_not_transform_non_generate_content_dict(): + """The native-body transform must only run for generate_content call types, so a + plain dict on a chat completion call is left untouched (no spurious Gemini cost).""" + logging_obj = LitellmLogging( + model="gpt-4o", + messages=[{"role": "user", "content": "Hey"}], + stream=False, + call_type="acompletion", + start_time=time.time(), + litellm_call_id="lit4076-2", + function_id="lit4076-2", + ) + logging_obj.optional_params = {} + + cost = logging_obj._response_cost_calculator( + result={"usageMetadata": {"promptTokenCount": 1000, "candidatesTokenCount": 500}} + ) + assert not cost + + def test_sentry_event_scrubber_initialization(monkeypatch): # Step 1: Create a fake sentry_sdk.scrubber module mock_event_scrubber_instance = MagicMock() diff --git a/tests/test_litellm/proxy/test_common_request_processing.py b/tests/test_litellm/proxy/test_common_request_processing.py index c77ee2b784d..1d0dafed171 100644 --- a/tests/test_litellm/proxy/test_common_request_processing.py +++ b/tests/test_litellm/proxy/test_common_request_processing.py @@ -4066,3 +4066,289 @@ class TestAllmPassthroughStreamingProviderGate: streamed = [chunk async for chunk in result.body_iterator] assert streamed == chunks mock_handler.assert_not_awaited() + + +class TestResponseCostHeaderForTypedDictResponses: + """ + Regression for LIT-4076. x-litellm-response-cost went missing on Anthropic + /v1/messages and Google :generateContent even though it appeared on + /chat/completions and /responses. /v1/messages returns a TypedDict that cannot + hold _hidden_params at all, and :generateContent carries _hidden_params but no + synchronously-populated response_cost. In both cases the raw response_cost is + empty at header-build time. The non-streaming header build now recovers the cost + from the logging object whenever the response itself never recorded one, while + leaving object responses (ModelResponse etc.) untouched. + """ + + def _build_logging_obj(self, *, model_call_details, response_cost_calculator): + logging_obj = MagicMock() + logging_obj.litellm_call_id = "call-lit4076" + logging_obj.cost_breakdown = None + logging_obj.model_call_details = model_call_details + logging_obj._response_cost_calculator = response_cost_calculator + logging_obj._enqueue_deferred_logging = None + logging_obj._on_deferred_stream_complete = None + return logging_obj + + async def _drive_non_streaming(self, *, monkeypatch, response, logging_obj, route_type): + import litellm.proxy.common_request_processing as crp + from litellm.proxy._types import UserAPIKeyAuth as RealUserAPIKeyAuth + + async def fake_route_request(**kwargs): + async def _llm_call(): + return response + + return _llm_call() + + monkeypatch.setattr(crp, "route_request", fake_route_request) + + async def fake_post_call_success_hook(data, user_api_key_dict, response): + return response + + proxy_logging_obj = MagicMock(spec=ProxyLogging) + proxy_logging_obj.during_call_hook = AsyncMock(return_value=None) + proxy_logging_obj.update_request_status = AsyncMock(return_value=None) + proxy_logging_obj.post_call_response_headers_hook = AsyncMock(return_value={}) + proxy_logging_obj.post_call_success_hook = fake_post_call_success_hook + + fastapi_response = Response() + processing_obj = ProxyBaseLLMRequestProcessing(data={"litellm_logging_obj": logging_obj}) + + with patch.object( + ProxyBaseLLMRequestProcessing, + "_has_post_call_guardrails", + return_value=False, + ): + await processing_obj.base_process_llm_request( + request=MagicMock(spec=Request, headers={}), + fastapi_response=fastapi_response, + user_api_key_dict=RealUserAPIKeyAuth(api_key="sk-test"), + route_type=route_type, + proxy_logging_obj=proxy_logging_obj, + general_settings={}, + proxy_config=MagicMock(spec=ProxyConfig), + select_data_generator=None, + llm_router=None, + skip_pre_call_logic=True, + ) + return fastapi_response + + @pytest.mark.asyncio + async def test_messages_typeddict_emits_cost_header_from_stored_cost(self, monkeypatch): + from litellm.types.utils import AnthropicMessagesResponse + + response = AnthropicMessagesResponse( + id="msg_1", + type="message", + role="assistant", + content=[{"type": "text", "text": "hi"}], + model="claude-haiku-4-5", + usage={"input_tokens": 10, "output_tokens": 5}, + ) + recompute = MagicMock(return_value=999.0) + logging_obj = self._build_logging_obj( + model_call_details={"response_cost": 0.00123}, + response_cost_calculator=recompute, + ) + + fastapi_response = await self._drive_non_streaming( + monkeypatch=monkeypatch, + response=response, + logging_obj=logging_obj, + route_type="anthropic_messages", + ) + + assert fastapi_response.headers["x-litellm-response-cost"] == "0.00123" + recompute.assert_not_called() + + @pytest.mark.asyncio + async def test_generate_content_typeddict_emits_cost_header_via_recompute(self, monkeypatch): + from litellm.types.llms.vertex_ai import GenerateContentResponseBody + + response = GenerateContentResponseBody( + candidates=[{"content": {"parts": [{"text": "hi"}], "role": "model"}}], + usageMetadata={ + "promptTokenCount": 10, + "candidatesTokenCount": 5, + "totalTokenCount": 15, + }, + ) + recompute = MagicMock(return_value=0.00456) + logging_obj = self._build_logging_obj( + model_call_details={}, + response_cost_calculator=recompute, + ) + + fastapi_response = await self._drive_non_streaming( + monkeypatch=monkeypatch, + response=response, + logging_obj=logging_obj, + route_type="agenerate_content", + ) + + assert fastapi_response.headers["x-litellm-response-cost"] == "0.00456" + recompute.assert_called_once() + assert recompute.call_args.kwargs["result"] is response + + @pytest.mark.asyncio + async def test_generate_content_emits_real_nonzero_cost_header_from_usage_metadata(self, monkeypatch): + """ + End-to-end regression for LIT-4076 using the real cost calculator (not a + mock). A native :generateContent body reports tokens under usageMetadata, + which the cost calculator did not read, so the synchronously-recovered + cost was 0.0 and the header was dropped even though the async logging path + billed a real non-zero amount. The header must now carry the true cost. + """ + from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj + from litellm.types.llms.vertex_ai import GenerateContentResponseBody + from litellm.types.utils import ModelResponse, Usage + + response = GenerateContentResponseBody( + candidates=[{"content": {"parts": [{"text": "hi"}], "role": "model"}, "finishReason": "STOP"}], + usageMetadata={ + "promptTokenCount": 1000, + "candidatesTokenCount": 500, + "totalTokenCount": 1500, + }, + ) + + real_logging = LiteLLMLoggingObj( + model="gemini-2.5-flash", + messages=[{"role": "user", "content": "hi"}], + stream=False, + call_type="agenerate_content", + start_time=None, + litellm_call_id="call-lit4076-real", + function_id="fn", + ) + real_logging.model_call_details["custom_llm_provider"] = "gemini" + real_logging.optional_params = {} + + logging_obj = self._build_logging_obj( + model_call_details={}, + response_cost_calculator=real_logging._response_cost_calculator, + ) + + fastapi_response = await self._drive_non_streaming( + monkeypatch=monkeypatch, + response=response, + logging_obj=logging_obj, + route_type="agenerate_content", + ) + + expected_cost = litellm.completion_cost( + completion_response=ModelResponse( + model="gemini-2.5-flash", + usage=Usage(prompt_tokens=1000, completion_tokens=500, total_tokens=1500), + ), + model="gemini-2.5-flash", + custom_llm_provider="gemini", + ) + assert expected_cost > 0 + assert float(fastapi_response.headers["x-litellm-response-cost"]) == pytest.approx(expected_cost) + + @pytest.mark.asyncio + async def test_generate_content_with_hidden_params_emits_cost_header(self, monkeypatch): + """ + Models the real :generateContent response: it DOES carry a _hidden_params + attribute (which is why x-litellm-model-group / x-litellm-model-api-base + appear), but no response_cost is populated synchronously at header-build + time. The cost is only available on the logging object. The previous + ``not hasattr(response, "_hidden_params")`` guard skipped recovery here, so + x-litellm-response-cost went missing even though the cost was computed. + """ + from types import SimpleNamespace + + response = SimpleNamespace( + _hidden_params={ + "additional_headers": {"x-litellm-model-group": "gemini-2.5-flash"}, + } + ) + recompute = MagicMock(return_value=999.0) + logging_obj = self._build_logging_obj( + model_call_details={"response_cost": 0.0004521}, + response_cost_calculator=recompute, + ) + + fastapi_response = await self._drive_non_streaming( + monkeypatch=monkeypatch, + response=response, + logging_obj=logging_obj, + route_type="agenerate_content", + ) + + assert fastapi_response.headers["x-litellm-response-cost"] == "0.0004521" + assert fastapi_response.headers["x-litellm-model-group"] == "gemini-2.5-flash" + recompute.assert_not_called() + + @pytest.mark.asyncio + async def test_generate_content_with_hidden_params_zero_cost_drops_header(self, monkeypatch): + """ + A recovered cost of 0 must normalize to a dropped header, exactly like + /chat/completions, so :generateContent does not start emitting + x-litellm-response-cost: 0.0 where nothing was emitted before. + """ + from types import SimpleNamespace + + response = SimpleNamespace( + _hidden_params={ + "additional_headers": {"x-litellm-model-group": "gemini-2.5-flash"}, + } + ) + recompute = MagicMock(return_value=999.0) + logging_obj = self._build_logging_obj( + model_call_details={"response_cost": 0.0}, + response_cost_calculator=recompute, + ) + + fastapi_response = await self._drive_non_streaming( + monkeypatch=monkeypatch, + response=response, + logging_obj=logging_obj, + route_type="agenerate_content", + ) + + assert "x-litellm-response-cost" not in fastapi_response.headers + recompute.assert_not_called() + + @pytest.mark.asyncio + async def test_object_response_with_hidden_params_is_unaffected(self, monkeypatch): + from types import SimpleNamespace + + response = SimpleNamespace(_hidden_params={"response_cost": 0.009}) + recompute = MagicMock(side_effect=AssertionError("must not recompute for object responses")) + logging_obj = self._build_logging_obj( + model_call_details={"response_cost": 123.0}, + response_cost_calculator=recompute, + ) + + fastapi_response = await self._drive_non_streaming( + monkeypatch=monkeypatch, + response=response, + logging_obj=logging_obj, + route_type="acompletion", + ) + + assert fastapi_response.headers["x-litellm-response-cost"] == "0.009" + recompute.assert_not_called() + + @pytest.mark.asyncio + async def test_object_response_zero_cost_drops_header_like_chat_completions(self, monkeypatch): + from types import SimpleNamespace + + response = SimpleNamespace(_hidden_params={"response_cost": 0.0}) + recompute = MagicMock(side_effect=AssertionError("must not recompute for object responses")) + logging_obj = self._build_logging_obj( + model_call_details={"response_cost": 0.00789}, + response_cost_calculator=recompute, + ) + + fastapi_response = await self._drive_non_streaming( + monkeypatch=monkeypatch, + response=response, + logging_obj=logging_obj, + route_type="acompletion", + ) + + assert "x-litellm-response-cost" not in fastapi_response.headers + recompute.assert_not_called()