fix(proxy): emit x-litellm-response-cost header on /messages and /generateContent (LIT-4076) (#31675)

* fix(proxy): emit x-litellm-response-cost header on /messages and /generateContent (LIT-4076)

The Anthropic /v1/messages and Google native :generateContent routes return
TypedDict results (AnthropicMessagesResponse, GenerateContentResponseBody) that
are plain dicts at runtime and cannot hold a _hidden_params attribute. The cost
is computed by update_response_metadata, but ResponseMetadata.apply() only
persists _hidden_params back when the result object has that attribute, so for
those two routes the computed response_cost was dropped. The non-streaming
header build in base_process_llm_request then saw an empty response_cost and
get_custom_headers filtered the x-litellm-response-cost header out, even though
the other x-litellm-* headers still appeared.

The non-streaming success path now recovers the cost from the logging object
when the response cannot carry _hidden_params, preferring the value already
stored in model_call_details and recomputing from the same calculator only when
it has not been stored yet. Object responses (ModelResponse, ResponsesAPIResponse)
keep their existing behavior, so chat/completions, /responses, and the Anthropic
error path that intentionally emits a zero cost are unaffected. Streaming stays
out of scope because the header is emitted at stream start, before the cost is
known.

* fix(proxy): also recover response cost header for /generateContent responses with _hidden_params (LIT-4076)

* fix(proxy): compute generateContent response cost synchronously so cost header is emitted (LIT-4076)

* fix(lint): suppress BLE001 on generate_content cost normalization guard

The defensive blind except keeps cost normalization from ever breaking the
response path; mark it noqa so it does not breach the strict-rule budget.
This commit is contained in:
Mateo Wang 2026-06-29 20:29:49 -07:00 • committed by GitHub
parent d295b76655
commit 375659ef04
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
4 changed files with 416 additions and 1 deletions

View file

@ -1384,6 +1384,10 @@ class Logging(LiteLLMLoggingBaseClass):
if cache_hit is True:
return 0.0
transformed_result = self._generate_content_result_as_model_response(result)
if transformed_result is not None:
result = transformed_result
if isinstance(result, BaseModel) and hasattr(result, "_hidden_params"):
hidden_params = getattr(result, "_hidden_params", {})
if (
@ -1463,6 +1467,39 @@ class Logging(LiteLLMLoggingBaseClass):
return None
def _generate_content_result_as_model_response(self, result: object) -> Optional[ModelResponse]:
"""
Native Google :generateContent bodies report token usage under
``usageMetadata``, which the cost calculator does not read, so a raw body
always costs 0. The async success path already transforms it into a
``ModelResponse`` before costing; do the same transformation here so the
synchronously-built ``x-litellm-response-cost`` header carries the real
cost. Returns ``None`` (leaving the original result untouched) for other
call types, for already-transformed ``ModelResponse`` results, and on any
transformation failure.
"""
if self.call_type not in (
CallTypes.generate_content.value,
CallTypes.agenerate_content.value,
):
return None
if isinstance(result, ModelResponse) or not isinstance(result, (BaseModel, dict)):
return None
try:
import httpx
completion_response = result.model_dump(by_alias=True) if isinstance(result, BaseModel) else dict(result)
return litellm.VertexGeminiConfig()._transform_google_generate_content_to_openai_model_response(
completion_response=completion_response,
model_response=ModelResponse(),
model=self.model or "",
logging_obj=self,
raw_response=httpx.Response(status_code=200, headers={}),
)
except Exception as e: # noqa: BLE001 - cost normalization must never break the response path
verbose_logger.debug(f"generate_content response cost normalization failed: {e}")
return None
async def _response_cost_calculator_async(
self,
result: Union[

View file

@ -1184,6 +1184,26 @@ class ProxyBaseLLMRequestProcessing:
model_id = model_info.get("id", "") or ""
return model_id
@staticmethod
def _response_cost_from_logging_obj(
*,
response: Any,
logging_obj: LiteLLMLoggingObj,
) -> float | str:
"""
Recover the response cost when the response never recorded one in its
``_hidden_params``: Anthropic /v1/messages returns a TypedDict that cannot
hold the attribute at all, and Google :generateContent carries
``_hidden_params`` but no synchronously-populated ``response_cost``. In both
cases the cost is read back from the logging object instead, recomputing from
the same calculator only when it has not been stored yet.
"""
stored_cost = logging_obj.model_call_details.get("response_cost")
if isinstance(stored_cost, (int, float)):
return float(stored_cost)
recomputed_cost = logging_obj._response_cost_calculator(result=response)
return recomputed_cost if isinstance(recomputed_cost, (int, float)) else ""
def _debug_log_request_payload(self) -> None:
"""Log request payload at DEBUG level, truncating if too large."""
if not verbose_proxy_logger.isEnabledFor(logging.DEBUG):
@ -1687,6 +1707,13 @@ class ProxyBaseLLMRequestProcessing:
hidden_params = getattr(response, "_hidden_params", {}) or {} # get any updated response headers
additional_headers = hidden_params.get("additional_headers", {}) or {}
recover_response_cost = not response_cost and hidden_params.get("response_cost") is None
response_cost_for_headers = (
self._response_cost_from_logging_obj(response=response, logging_obj=logging_obj) or ""
if recover_response_cost
else response_cost
)
fastapi_response.headers.update(
ProxyBaseLLMRequestProcessing.get_custom_headers(
user_api_key_dict=user_api_key_dict,
@ -1695,7 +1722,7 @@ class ProxyBaseLLMRequestProcessing:
cache_key=cache_key,
api_base=api_base,
version=version,
response_cost=response_cost,
response_cost=response_cost_for_headers,
model_region=getattr(user_api_key_dict, "allowed_model_region", ""),
fastest_response_batch_completion=fastest_response_batch_completion,
request_data=self.data,

View file

@ -1424,6 +1424,71 @@ def test_response_cost_calculator_with_response_cost_in_hidden_params(logging_ob
assert response_cost > 100
def test_response_cost_calculator_native_generate_content_body_uses_usage_metadata():
"""
Regression for LIT-4076: a native Google :generateContent body reports tokens
under ``usageMetadata`` rather than ``usage``, so the cost calculator read 0
tokens and returned 0.0 synchronously. The calculator now transforms the native
body (as the async logging path does) so the cost is the real non-zero amount.
"""
from litellm.types.llms.vertex_ai import GenerateContentResponseBody
from litellm.types.utils import ModelResponse, Usage
logging_obj = LitellmLogging(
model="gemini-2.5-flash",
messages=[{"role": "user", "content": "Hey"}],
stream=False,
call_type="agenerate_content",
start_time=time.time(),
litellm_call_id="lit4076",
function_id="lit4076",
)
logging_obj.model_call_details["custom_llm_provider"] = "gemini"
logging_obj.optional_params = {}
native_body = GenerateContentResponseBody(
candidates=[{"content": {"parts": [{"text": "hi"}], "role": "model"}, "finishReason": "STOP"}],
usageMetadata={
"promptTokenCount": 1000,
"candidatesTokenCount": 500,
"totalTokenCount": 1500,
},
)
expected_cost = litellm.completion_cost(
completion_response=ModelResponse(
model="gemini-2.5-flash",
usage=Usage(prompt_tokens=1000, completion_tokens=500, total_tokens=1500),
),
model="gemini-2.5-flash",
custom_llm_provider="gemini",
)
assert expected_cost > 0
cost = logging_obj._response_cost_calculator(result=native_body)
assert cost == pytest.approx(expected_cost)
def test_response_cost_calculator_does_not_transform_non_generate_content_dict():
"""The native-body transform must only run for generate_content call types, so a
plain dict on a chat completion call is left untouched (no spurious Gemini cost)."""
logging_obj = LitellmLogging(
model="gpt-4o",
messages=[{"role": "user", "content": "Hey"}],
stream=False,
call_type="acompletion",
start_time=time.time(),
litellm_call_id="lit4076-2",
function_id="lit4076-2",
)
logging_obj.optional_params = {}
cost = logging_obj._response_cost_calculator(
result={"usageMetadata": {"promptTokenCount": 1000, "candidatesTokenCount": 500}}
)
assert not cost
def test_sentry_event_scrubber_initialization(monkeypatch):
# Step 1: Create a fake sentry_sdk.scrubber module
mock_event_scrubber_instance = MagicMock()

View file

@ -4066,3 +4066,289 @@ class TestAllmPassthroughStreamingProviderGate:
streamed = [chunk async for chunk in result.body_iterator]
assert streamed == chunks
mock_handler.assert_not_awaited()
class TestResponseCostHeaderForTypedDictResponses:
"""
Regression for LIT-4076. x-litellm-response-cost went missing on Anthropic
/v1/messages and Google :generateContent even though it appeared on
/chat/completions and /responses. /v1/messages returns a TypedDict that cannot
hold _hidden_params at all, and :generateContent carries _hidden_params but no
synchronously-populated response_cost. In both cases the raw response_cost is
empty at header-build time. The non-streaming header build now recovers the cost
from the logging object whenever the response itself never recorded one, while
leaving object responses (ModelResponse etc.) untouched.
"""
def _build_logging_obj(self, *, model_call_details, response_cost_calculator):
logging_obj = MagicMock()
logging_obj.litellm_call_id = "call-lit4076"
logging_obj.cost_breakdown = None
logging_obj.model_call_details = model_call_details
logging_obj._response_cost_calculator = response_cost_calculator
logging_obj._enqueue_deferred_logging = None
logging_obj._on_deferred_stream_complete = None
return logging_obj
async def _drive_non_streaming(self, *, monkeypatch, response, logging_obj, route_type):
import litellm.proxy.common_request_processing as crp
from litellm.proxy._types import UserAPIKeyAuth as RealUserAPIKeyAuth
async def fake_route_request(**kwargs):
async def _llm_call():
return response
return _llm_call()
monkeypatch.setattr(crp, "route_request", fake_route_request)
async def fake_post_call_success_hook(data, user_api_key_dict, response):
return response
proxy_logging_obj = MagicMock(spec=ProxyLogging)
proxy_logging_obj.during_call_hook = AsyncMock(return_value=None)
proxy_logging_obj.update_request_status = AsyncMock(return_value=None)
proxy_logging_obj.post_call_response_headers_hook = AsyncMock(return_value={})
proxy_logging_obj.post_call_success_hook = fake_post_call_success_hook
fastapi_response = Response()
processing_obj = ProxyBaseLLMRequestProcessing(data={"litellm_logging_obj": logging_obj})
with patch.object(
ProxyBaseLLMRequestProcessing,
"_has_post_call_guardrails",
return_value=False,
):
await processing_obj.base_process_llm_request(
request=MagicMock(spec=Request, headers={}),
fastapi_response=fastapi_response,
user_api_key_dict=RealUserAPIKeyAuth(api_key="sk-test"),
route_type=route_type,
proxy_logging_obj=proxy_logging_obj,
general_settings={},
proxy_config=MagicMock(spec=ProxyConfig),
select_data_generator=None,
llm_router=None,
skip_pre_call_logic=True,
)
return fastapi_response
@pytest.mark.asyncio
async def test_messages_typeddict_emits_cost_header_from_stored_cost(self, monkeypatch):
from litellm.types.utils import AnthropicMessagesResponse
response = AnthropicMessagesResponse(
id="msg_1",
type="message",
role="assistant",
content=[{"type": "text", "text": "hi"}],
model="claude-haiku-4-5",
usage={"input_tokens": 10, "output_tokens": 5},
)
recompute = MagicMock(return_value=999.0)
logging_obj = self._build_logging_obj(
model_call_details={"response_cost": 0.00123},
response_cost_calculator=recompute,
)
fastapi_response = await self._drive_non_streaming(
monkeypatch=monkeypatch,
response=response,
logging_obj=logging_obj,
route_type="anthropic_messages",
)
assert fastapi_response.headers["x-litellm-response-cost"] == "0.00123"
recompute.assert_not_called()
@pytest.mark.asyncio
async def test_generate_content_typeddict_emits_cost_header_via_recompute(self, monkeypatch):
from litellm.types.llms.vertex_ai import GenerateContentResponseBody
response = GenerateContentResponseBody(
candidates=[{"content": {"parts": [{"text": "hi"}], "role": "model"}}],
usageMetadata={
"promptTokenCount": 10,
"candidatesTokenCount": 5,
"totalTokenCount": 15,
},
)
recompute = MagicMock(return_value=0.00456)
logging_obj = self._build_logging_obj(
model_call_details={},
response_cost_calculator=recompute,
)
fastapi_response = await self._drive_non_streaming(
monkeypatch=monkeypatch,
response=response,
logging_obj=logging_obj,
route_type="agenerate_content",
)
assert fastapi_response.headers["x-litellm-response-cost"] == "0.00456"
recompute.assert_called_once()
assert recompute.call_args.kwargs["result"] is response
@pytest.mark.asyncio
async def test_generate_content_emits_real_nonzero_cost_header_from_usage_metadata(self, monkeypatch):
"""
End-to-end regression for LIT-4076 using the real cost calculator (not a
mock). A native :generateContent body reports tokens under usageMetadata,
which the cost calculator did not read, so the synchronously-recovered
cost was 0.0 and the header was dropped even though the async logging path
billed a real non-zero amount. The header must now carry the true cost.
"""
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
from litellm.types.llms.vertex_ai import GenerateContentResponseBody
from litellm.types.utils import ModelResponse, Usage
response = GenerateContentResponseBody(
candidates=[{"content": {"parts": [{"text": "hi"}], "role": "model"}, "finishReason": "STOP"}],
usageMetadata={
"promptTokenCount": 1000,
"candidatesTokenCount": 500,
"totalTokenCount": 1500,
},
)
real_logging = LiteLLMLoggingObj(
model="gemini-2.5-flash",
messages=[{"role": "user", "content": "hi"}],
stream=False,
call_type="agenerate_content",
start_time=None,
litellm_call_id="call-lit4076-real",
function_id="fn",
)
real_logging.model_call_details["custom_llm_provider"] = "gemini"
real_logging.optional_params = {}
logging_obj = self._build_logging_obj(
model_call_details={},
response_cost_calculator=real_logging._response_cost_calculator,
)
fastapi_response = await self._drive_non_streaming(
monkeypatch=monkeypatch,
response=response,
logging_obj=logging_obj,
route_type="agenerate_content",
)
expected_cost = litellm.completion_cost(
completion_response=ModelResponse(
model="gemini-2.5-flash",
usage=Usage(prompt_tokens=1000, completion_tokens=500, total_tokens=1500),
),
model="gemini-2.5-flash",
custom_llm_provider="gemini",
)
assert expected_cost > 0
assert float(fastapi_response.headers["x-litellm-response-cost"]) == pytest.approx(expected_cost)
@pytest.mark.asyncio
async def test_generate_content_with_hidden_params_emits_cost_header(self, monkeypatch):
"""
Models the real :generateContent response: it DOES carry a _hidden_params
attribute (which is why x-litellm-model-group / x-litellm-model-api-base
appear), but no response_cost is populated synchronously at header-build
time. The cost is only available on the logging object. The previous
``not hasattr(response, "_hidden_params")`` guard skipped recovery here, so
x-litellm-response-cost went missing even though the cost was computed.
"""
from types import SimpleNamespace
response = SimpleNamespace(
_hidden_params={
"additional_headers": {"x-litellm-model-group": "gemini-2.5-flash"},
}
)
recompute = MagicMock(return_value=999.0)
logging_obj = self._build_logging_obj(
model_call_details={"response_cost": 0.0004521},
response_cost_calculator=recompute,
)
fastapi_response = await self._drive_non_streaming(
monkeypatch=monkeypatch,
response=response,
logging_obj=logging_obj,
route_type="agenerate_content",
)
assert fastapi_response.headers["x-litellm-response-cost"] == "0.0004521"
assert fastapi_response.headers["x-litellm-model-group"] == "gemini-2.5-flash"
recompute.assert_not_called()
@pytest.mark.asyncio
async def test_generate_content_with_hidden_params_zero_cost_drops_header(self, monkeypatch):
"""
A recovered cost of 0 must normalize to a dropped header, exactly like
/chat/completions, so :generateContent does not start emitting
x-litellm-response-cost: 0.0 where nothing was emitted before.
"""
from types import SimpleNamespace
response = SimpleNamespace(
_hidden_params={
"additional_headers": {"x-litellm-model-group": "gemini-2.5-flash"},
}
)
recompute = MagicMock(return_value=999.0)
logging_obj = self._build_logging_obj(
model_call_details={"response_cost": 0.0},
response_cost_calculator=recompute,
)
fastapi_response = await self._drive_non_streaming(
monkeypatch=monkeypatch,
response=response,
logging_obj=logging_obj,
route_type="agenerate_content",
)
assert "x-litellm-response-cost" not in fastapi_response.headers
recompute.assert_not_called()
@pytest.mark.asyncio
async def test_object_response_with_hidden_params_is_unaffected(self, monkeypatch):
from types import SimpleNamespace
response = SimpleNamespace(_hidden_params={"response_cost": 0.009})
recompute = MagicMock(side_effect=AssertionError("must not recompute for object responses"))
logging_obj = self._build_logging_obj(
model_call_details={"response_cost": 123.0},
response_cost_calculator=recompute,
)
fastapi_response = await self._drive_non_streaming(
monkeypatch=monkeypatch,
response=response,
logging_obj=logging_obj,
route_type="acompletion",
)
assert fastapi_response.headers["x-litellm-response-cost"] == "0.009"
recompute.assert_not_called()
@pytest.mark.asyncio
async def test_object_response_zero_cost_drops_header_like_chat_completions(self, monkeypatch):
from types import SimpleNamespace
response = SimpleNamespace(_hidden_params={"response_cost": 0.0})
recompute = MagicMock(side_effect=AssertionError("must not recompute for object responses"))
logging_obj = self._build_logging_obj(
model_call_details={"response_cost": 0.00789},
response_cost_calculator=recompute,
)
fastapi_response = await self._drive_non_streaming(
monkeypatch=monkeypatch,
response=response,
logging_obj=logging_obj,
route_type="acompletion",
)
assert "x-litellm-response-cost" not in fastapi_response.headers
recompute.assert_not_called()