mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
fix(proxy): emit x-litellm-response-cost header on /messages and /generateContent (LIT-4076) (#31675)
* fix(proxy): emit x-litellm-response-cost header on /messages and /generateContent (LIT-4076) The Anthropic /v1/messages and Google native :generateContent routes return TypedDict results (AnthropicMessagesResponse, GenerateContentResponseBody) that are plain dicts at runtime and cannot hold a _hidden_params attribute. The cost is computed by update_response_metadata, but ResponseMetadata.apply() only persists _hidden_params back when the result object has that attribute, so for those two routes the computed response_cost was dropped. The non-streaming header build in base_process_llm_request then saw an empty response_cost and get_custom_headers filtered the x-litellm-response-cost header out, even though the other x-litellm-* headers still appeared. The non-streaming success path now recovers the cost from the logging object when the response cannot carry _hidden_params, preferring the value already stored in model_call_details and recomputing from the same calculator only when it has not been stored yet. Object responses (ModelResponse, ResponsesAPIResponse) keep their existing behavior, so chat/completions, /responses, and the Anthropic error path that intentionally emits a zero cost are unaffected. Streaming stays out of scope because the header is emitted at stream start, before the cost is known. * fix(proxy): also recover response cost header for /generateContent responses with _hidden_params (LIT-4076) * fix(proxy): compute generateContent response cost synchronously so cost header is emitted (LIT-4076) * fix(lint): suppress BLE001 on generate_content cost normalization guard The defensive blind except keeps cost normalization from ever breaking the response path; mark it noqa so it does not breach the strict-rule budget.
This commit is contained in:
parent
d295b76655
commit
375659ef04
4 changed files with 416 additions and 1 deletions
|
|
@ -1384,6 +1384,10 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
if cache_hit is True:
|
||||
return 0.0
|
||||
|
||||
transformed_result = self._generate_content_result_as_model_response(result)
|
||||
if transformed_result is not None:
|
||||
result = transformed_result
|
||||
|
||||
if isinstance(result, BaseModel) and hasattr(result, "_hidden_params"):
|
||||
hidden_params = getattr(result, "_hidden_params", {})
|
||||
if (
|
||||
|
|
@ -1463,6 +1467,39 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
|
||||
return None
|
||||
|
||||
def _generate_content_result_as_model_response(self, result: object) -> Optional[ModelResponse]:
|
||||
"""
|
||||
Native Google :generateContent bodies report token usage under
|
||||
``usageMetadata``, which the cost calculator does not read, so a raw body
|
||||
always costs 0. The async success path already transforms it into a
|
||||
``ModelResponse`` before costing; do the same transformation here so the
|
||||
synchronously-built ``x-litellm-response-cost`` header carries the real
|
||||
cost. Returns ``None`` (leaving the original result untouched) for other
|
||||
call types, for already-transformed ``ModelResponse`` results, and on any
|
||||
transformation failure.
|
||||
"""
|
||||
if self.call_type not in (
|
||||
CallTypes.generate_content.value,
|
||||
CallTypes.agenerate_content.value,
|
||||
):
|
||||
return None
|
||||
if isinstance(result, ModelResponse) or not isinstance(result, (BaseModel, dict)):
|
||||
return None
|
||||
try:
|
||||
import httpx
|
||||
|
||||
completion_response = result.model_dump(by_alias=True) if isinstance(result, BaseModel) else dict(result)
|
||||
return litellm.VertexGeminiConfig()._transform_google_generate_content_to_openai_model_response(
|
||||
completion_response=completion_response,
|
||||
model_response=ModelResponse(),
|
||||
model=self.model or "",
|
||||
logging_obj=self,
|
||||
raw_response=httpx.Response(status_code=200, headers={}),
|
||||
)
|
||||
except Exception as e: # noqa: BLE001 - cost normalization must never break the response path
|
||||
verbose_logger.debug(f"generate_content response cost normalization failed: {e}")
|
||||
return None
|
||||
|
||||
async def _response_cost_calculator_async(
|
||||
self,
|
||||
result: Union[
|
||||
|
|
|
|||
|
|
@ -1184,6 +1184,26 @@ class ProxyBaseLLMRequestProcessing:
|
|||
model_id = model_info.get("id", "") or ""
|
||||
return model_id
|
||||
|
||||
@staticmethod
|
||||
def _response_cost_from_logging_obj(
|
||||
*,
|
||||
response: Any,
|
||||
logging_obj: LiteLLMLoggingObj,
|
||||
) -> float | str:
|
||||
"""
|
||||
Recover the response cost when the response never recorded one in its
|
||||
``_hidden_params``: Anthropic /v1/messages returns a TypedDict that cannot
|
||||
hold the attribute at all, and Google :generateContent carries
|
||||
``_hidden_params`` but no synchronously-populated ``response_cost``. In both
|
||||
cases the cost is read back from the logging object instead, recomputing from
|
||||
the same calculator only when it has not been stored yet.
|
||||
"""
|
||||
stored_cost = logging_obj.model_call_details.get("response_cost")
|
||||
if isinstance(stored_cost, (int, float)):
|
||||
return float(stored_cost)
|
||||
recomputed_cost = logging_obj._response_cost_calculator(result=response)
|
||||
return recomputed_cost if isinstance(recomputed_cost, (int, float)) else ""
|
||||
|
||||
def _debug_log_request_payload(self) -> None:
|
||||
"""Log request payload at DEBUG level, truncating if too large."""
|
||||
if not verbose_proxy_logger.isEnabledFor(logging.DEBUG):
|
||||
|
|
@ -1687,6 +1707,13 @@ class ProxyBaseLLMRequestProcessing:
|
|||
hidden_params = getattr(response, "_hidden_params", {}) or {} # get any updated response headers
|
||||
additional_headers = hidden_params.get("additional_headers", {}) or {}
|
||||
|
||||
recover_response_cost = not response_cost and hidden_params.get("response_cost") is None
|
||||
response_cost_for_headers = (
|
||||
self._response_cost_from_logging_obj(response=response, logging_obj=logging_obj) or ""
|
||||
if recover_response_cost
|
||||
else response_cost
|
||||
)
|
||||
|
||||
fastapi_response.headers.update(
|
||||
ProxyBaseLLMRequestProcessing.get_custom_headers(
|
||||
user_api_key_dict=user_api_key_dict,
|
||||
|
|
@ -1695,7 +1722,7 @@ class ProxyBaseLLMRequestProcessing:
|
|||
cache_key=cache_key,
|
||||
api_base=api_base,
|
||||
version=version,
|
||||
response_cost=response_cost,
|
||||
response_cost=response_cost_for_headers,
|
||||
model_region=getattr(user_api_key_dict, "allowed_model_region", ""),
|
||||
fastest_response_batch_completion=fastest_response_batch_completion,
|
||||
request_data=self.data,
|
||||
|
|
|
|||
|
|
@ -1424,6 +1424,71 @@ def test_response_cost_calculator_with_response_cost_in_hidden_params(logging_ob
|
|||
assert response_cost > 100
|
||||
|
||||
|
||||
def test_response_cost_calculator_native_generate_content_body_uses_usage_metadata():
|
||||
"""
|
||||
Regression for LIT-4076: a native Google :generateContent body reports tokens
|
||||
under ``usageMetadata`` rather than ``usage``, so the cost calculator read 0
|
||||
tokens and returned 0.0 synchronously. The calculator now transforms the native
|
||||
body (as the async logging path does) so the cost is the real non-zero amount.
|
||||
"""
|
||||
from litellm.types.llms.vertex_ai import GenerateContentResponseBody
|
||||
from litellm.types.utils import ModelResponse, Usage
|
||||
|
||||
logging_obj = LitellmLogging(
|
||||
model="gemini-2.5-flash",
|
||||
messages=[{"role": "user", "content": "Hey"}],
|
||||
stream=False,
|
||||
call_type="agenerate_content",
|
||||
start_time=time.time(),
|
||||
litellm_call_id="lit4076",
|
||||
function_id="lit4076",
|
||||
)
|
||||
logging_obj.model_call_details["custom_llm_provider"] = "gemini"
|
||||
logging_obj.optional_params = {}
|
||||
|
||||
native_body = GenerateContentResponseBody(
|
||||
candidates=[{"content": {"parts": [{"text": "hi"}], "role": "model"}, "finishReason": "STOP"}],
|
||||
usageMetadata={
|
||||
"promptTokenCount": 1000,
|
||||
"candidatesTokenCount": 500,
|
||||
"totalTokenCount": 1500,
|
||||
},
|
||||
)
|
||||
|
||||
expected_cost = litellm.completion_cost(
|
||||
completion_response=ModelResponse(
|
||||
model="gemini-2.5-flash",
|
||||
usage=Usage(prompt_tokens=1000, completion_tokens=500, total_tokens=1500),
|
||||
),
|
||||
model="gemini-2.5-flash",
|
||||
custom_llm_provider="gemini",
|
||||
)
|
||||
assert expected_cost > 0
|
||||
|
||||
cost = logging_obj._response_cost_calculator(result=native_body)
|
||||
assert cost == pytest.approx(expected_cost)
|
||||
|
||||
|
||||
def test_response_cost_calculator_does_not_transform_non_generate_content_dict():
|
||||
"""The native-body transform must only run for generate_content call types, so a
|
||||
plain dict on a chat completion call is left untouched (no spurious Gemini cost)."""
|
||||
logging_obj = LitellmLogging(
|
||||
model="gpt-4o",
|
||||
messages=[{"role": "user", "content": "Hey"}],
|
||||
stream=False,
|
||||
call_type="acompletion",
|
||||
start_time=time.time(),
|
||||
litellm_call_id="lit4076-2",
|
||||
function_id="lit4076-2",
|
||||
)
|
||||
logging_obj.optional_params = {}
|
||||
|
||||
cost = logging_obj._response_cost_calculator(
|
||||
result={"usageMetadata": {"promptTokenCount": 1000, "candidatesTokenCount": 500}}
|
||||
)
|
||||
assert not cost
|
||||
|
||||
|
||||
def test_sentry_event_scrubber_initialization(monkeypatch):
|
||||
# Step 1: Create a fake sentry_sdk.scrubber module
|
||||
mock_event_scrubber_instance = MagicMock()
|
||||
|
|
|
|||
|
|
@ -4066,3 +4066,289 @@ class TestAllmPassthroughStreamingProviderGate:
|
|||
streamed = [chunk async for chunk in result.body_iterator]
|
||||
assert streamed == chunks
|
||||
mock_handler.assert_not_awaited()
|
||||
|
||||
|
||||
class TestResponseCostHeaderForTypedDictResponses:
|
||||
"""
|
||||
Regression for LIT-4076. x-litellm-response-cost went missing on Anthropic
|
||||
/v1/messages and Google :generateContent even though it appeared on
|
||||
/chat/completions and /responses. /v1/messages returns a TypedDict that cannot
|
||||
hold _hidden_params at all, and :generateContent carries _hidden_params but no
|
||||
synchronously-populated response_cost. In both cases the raw response_cost is
|
||||
empty at header-build time. The non-streaming header build now recovers the cost
|
||||
from the logging object whenever the response itself never recorded one, while
|
||||
leaving object responses (ModelResponse etc.) untouched.
|
||||
"""
|
||||
|
||||
def _build_logging_obj(self, *, model_call_details, response_cost_calculator):
|
||||
logging_obj = MagicMock()
|
||||
logging_obj.litellm_call_id = "call-lit4076"
|
||||
logging_obj.cost_breakdown = None
|
||||
logging_obj.model_call_details = model_call_details
|
||||
logging_obj._response_cost_calculator = response_cost_calculator
|
||||
logging_obj._enqueue_deferred_logging = None
|
||||
logging_obj._on_deferred_stream_complete = None
|
||||
return logging_obj
|
||||
|
||||
async def _drive_non_streaming(self, *, monkeypatch, response, logging_obj, route_type):
|
||||
import litellm.proxy.common_request_processing as crp
|
||||
from litellm.proxy._types import UserAPIKeyAuth as RealUserAPIKeyAuth
|
||||
|
||||
async def fake_route_request(**kwargs):
|
||||
async def _llm_call():
|
||||
return response
|
||||
|
||||
return _llm_call()
|
||||
|
||||
monkeypatch.setattr(crp, "route_request", fake_route_request)
|
||||
|
||||
async def fake_post_call_success_hook(data, user_api_key_dict, response):
|
||||
return response
|
||||
|
||||
proxy_logging_obj = MagicMock(spec=ProxyLogging)
|
||||
proxy_logging_obj.during_call_hook = AsyncMock(return_value=None)
|
||||
proxy_logging_obj.update_request_status = AsyncMock(return_value=None)
|
||||
proxy_logging_obj.post_call_response_headers_hook = AsyncMock(return_value={})
|
||||
proxy_logging_obj.post_call_success_hook = fake_post_call_success_hook
|
||||
|
||||
fastapi_response = Response()
|
||||
processing_obj = ProxyBaseLLMRequestProcessing(data={"litellm_logging_obj": logging_obj})
|
||||
|
||||
with patch.object(
|
||||
ProxyBaseLLMRequestProcessing,
|
||||
"_has_post_call_guardrails",
|
||||
return_value=False,
|
||||
):
|
||||
await processing_obj.base_process_llm_request(
|
||||
request=MagicMock(spec=Request, headers={}),
|
||||
fastapi_response=fastapi_response,
|
||||
user_api_key_dict=RealUserAPIKeyAuth(api_key="sk-test"),
|
||||
route_type=route_type,
|
||||
proxy_logging_obj=proxy_logging_obj,
|
||||
general_settings={},
|
||||
proxy_config=MagicMock(spec=ProxyConfig),
|
||||
select_data_generator=None,
|
||||
llm_router=None,
|
||||
skip_pre_call_logic=True,
|
||||
)
|
||||
return fastapi_response
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_messages_typeddict_emits_cost_header_from_stored_cost(self, monkeypatch):
|
||||
from litellm.types.utils import AnthropicMessagesResponse
|
||||
|
||||
response = AnthropicMessagesResponse(
|
||||
id="msg_1",
|
||||
type="message",
|
||||
role="assistant",
|
||||
content=[{"type": "text", "text": "hi"}],
|
||||
model="claude-haiku-4-5",
|
||||
usage={"input_tokens": 10, "output_tokens": 5},
|
||||
)
|
||||
recompute = MagicMock(return_value=999.0)
|
||||
logging_obj = self._build_logging_obj(
|
||||
model_call_details={"response_cost": 0.00123},
|
||||
response_cost_calculator=recompute,
|
||||
)
|
||||
|
||||
fastapi_response = await self._drive_non_streaming(
|
||||
monkeypatch=monkeypatch,
|
||||
response=response,
|
||||
logging_obj=logging_obj,
|
||||
route_type="anthropic_messages",
|
||||
)
|
||||
|
||||
assert fastapi_response.headers["x-litellm-response-cost"] == "0.00123"
|
||||
recompute.assert_not_called()
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_generate_content_typeddict_emits_cost_header_via_recompute(self, monkeypatch):
|
||||
from litellm.types.llms.vertex_ai import GenerateContentResponseBody
|
||||
|
||||
response = GenerateContentResponseBody(
|
||||
candidates=[{"content": {"parts": [{"text": "hi"}], "role": "model"}}],
|
||||
usageMetadata={
|
||||
"promptTokenCount": 10,
|
||||
"candidatesTokenCount": 5,
|
||||
"totalTokenCount": 15,
|
||||
},
|
||||
)
|
||||
recompute = MagicMock(return_value=0.00456)
|
||||
logging_obj = self._build_logging_obj(
|
||||
model_call_details={},
|
||||
response_cost_calculator=recompute,
|
||||
)
|
||||
|
||||
fastapi_response = await self._drive_non_streaming(
|
||||
monkeypatch=monkeypatch,
|
||||
response=response,
|
||||
logging_obj=logging_obj,
|
||||
route_type="agenerate_content",
|
||||
)
|
||||
|
||||
assert fastapi_response.headers["x-litellm-response-cost"] == "0.00456"
|
||||
recompute.assert_called_once()
|
||||
assert recompute.call_args.kwargs["result"] is response
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_generate_content_emits_real_nonzero_cost_header_from_usage_metadata(self, monkeypatch):
|
||||
"""
|
||||
End-to-end regression for LIT-4076 using the real cost calculator (not a
|
||||
mock). A native :generateContent body reports tokens under usageMetadata,
|
||||
which the cost calculator did not read, so the synchronously-recovered
|
||||
cost was 0.0 and the header was dropped even though the async logging path
|
||||
billed a real non-zero amount. The header must now carry the true cost.
|
||||
"""
|
||||
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
|
||||
from litellm.types.llms.vertex_ai import GenerateContentResponseBody
|
||||
from litellm.types.utils import ModelResponse, Usage
|
||||
|
||||
response = GenerateContentResponseBody(
|
||||
candidates=[{"content": {"parts": [{"text": "hi"}], "role": "model"}, "finishReason": "STOP"}],
|
||||
usageMetadata={
|
||||
"promptTokenCount": 1000,
|
||||
"candidatesTokenCount": 500,
|
||||
"totalTokenCount": 1500,
|
||||
},
|
||||
)
|
||||
|
||||
real_logging = LiteLLMLoggingObj(
|
||||
model="gemini-2.5-flash",
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
stream=False,
|
||||
call_type="agenerate_content",
|
||||
start_time=None,
|
||||
litellm_call_id="call-lit4076-real",
|
||||
function_id="fn",
|
||||
)
|
||||
real_logging.model_call_details["custom_llm_provider"] = "gemini"
|
||||
real_logging.optional_params = {}
|
||||
|
||||
logging_obj = self._build_logging_obj(
|
||||
model_call_details={},
|
||||
response_cost_calculator=real_logging._response_cost_calculator,
|
||||
)
|
||||
|
||||
fastapi_response = await self._drive_non_streaming(
|
||||
monkeypatch=monkeypatch,
|
||||
response=response,
|
||||
logging_obj=logging_obj,
|
||||
route_type="agenerate_content",
|
||||
)
|
||||
|
||||
expected_cost = litellm.completion_cost(
|
||||
completion_response=ModelResponse(
|
||||
model="gemini-2.5-flash",
|
||||
usage=Usage(prompt_tokens=1000, completion_tokens=500, total_tokens=1500),
|
||||
),
|
||||
model="gemini-2.5-flash",
|
||||
custom_llm_provider="gemini",
|
||||
)
|
||||
assert expected_cost > 0
|
||||
assert float(fastapi_response.headers["x-litellm-response-cost"]) == pytest.approx(expected_cost)
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_generate_content_with_hidden_params_emits_cost_header(self, monkeypatch):
|
||||
"""
|
||||
Models the real :generateContent response: it DOES carry a _hidden_params
|
||||
attribute (which is why x-litellm-model-group / x-litellm-model-api-base
|
||||
appear), but no response_cost is populated synchronously at header-build
|
||||
time. The cost is only available on the logging object. The previous
|
||||
``not hasattr(response, "_hidden_params")`` guard skipped recovery here, so
|
||||
x-litellm-response-cost went missing even though the cost was computed.
|
||||
"""
|
||||
from types import SimpleNamespace
|
||||
|
||||
response = SimpleNamespace(
|
||||
_hidden_params={
|
||||
"additional_headers": {"x-litellm-model-group": "gemini-2.5-flash"},
|
||||
}
|
||||
)
|
||||
recompute = MagicMock(return_value=999.0)
|
||||
logging_obj = self._build_logging_obj(
|
||||
model_call_details={"response_cost": 0.0004521},
|
||||
response_cost_calculator=recompute,
|
||||
)
|
||||
|
||||
fastapi_response = await self._drive_non_streaming(
|
||||
monkeypatch=monkeypatch,
|
||||
response=response,
|
||||
logging_obj=logging_obj,
|
||||
route_type="agenerate_content",
|
||||
)
|
||||
|
||||
assert fastapi_response.headers["x-litellm-response-cost"] == "0.0004521"
|
||||
assert fastapi_response.headers["x-litellm-model-group"] == "gemini-2.5-flash"
|
||||
recompute.assert_not_called()
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_generate_content_with_hidden_params_zero_cost_drops_header(self, monkeypatch):
|
||||
"""
|
||||
A recovered cost of 0 must normalize to a dropped header, exactly like
|
||||
/chat/completions, so :generateContent does not start emitting
|
||||
x-litellm-response-cost: 0.0 where nothing was emitted before.
|
||||
"""
|
||||
from types import SimpleNamespace
|
||||
|
||||
response = SimpleNamespace(
|
||||
_hidden_params={
|
||||
"additional_headers": {"x-litellm-model-group": "gemini-2.5-flash"},
|
||||
}
|
||||
)
|
||||
recompute = MagicMock(return_value=999.0)
|
||||
logging_obj = self._build_logging_obj(
|
||||
model_call_details={"response_cost": 0.0},
|
||||
response_cost_calculator=recompute,
|
||||
)
|
||||
|
||||
fastapi_response = await self._drive_non_streaming(
|
||||
monkeypatch=monkeypatch,
|
||||
response=response,
|
||||
logging_obj=logging_obj,
|
||||
route_type="agenerate_content",
|
||||
)
|
||||
|
||||
assert "x-litellm-response-cost" not in fastapi_response.headers
|
||||
recompute.assert_not_called()
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_object_response_with_hidden_params_is_unaffected(self, monkeypatch):
|
||||
from types import SimpleNamespace
|
||||
|
||||
response = SimpleNamespace(_hidden_params={"response_cost": 0.009})
|
||||
recompute = MagicMock(side_effect=AssertionError("must not recompute for object responses"))
|
||||
logging_obj = self._build_logging_obj(
|
||||
model_call_details={"response_cost": 123.0},
|
||||
response_cost_calculator=recompute,
|
||||
)
|
||||
|
||||
fastapi_response = await self._drive_non_streaming(
|
||||
monkeypatch=monkeypatch,
|
||||
response=response,
|
||||
logging_obj=logging_obj,
|
||||
route_type="acompletion",
|
||||
)
|
||||
|
||||
assert fastapi_response.headers["x-litellm-response-cost"] == "0.009"
|
||||
recompute.assert_not_called()
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_object_response_zero_cost_drops_header_like_chat_completions(self, monkeypatch):
|
||||
from types import SimpleNamespace
|
||||
|
||||
response = SimpleNamespace(_hidden_params={"response_cost": 0.0})
|
||||
recompute = MagicMock(side_effect=AssertionError("must not recompute for object responses"))
|
||||
logging_obj = self._build_logging_obj(
|
||||
model_call_details={"response_cost": 0.00789},
|
||||
response_cost_calculator=recompute,
|
||||
)
|
||||
|
||||
fastapi_response = await self._drive_non_streaming(
|
||||
monkeypatch=monkeypatch,
|
||||
response=response,
|
||||
logging_obj=logging_obj,
|
||||
route_type="acompletion",
|
||||
)
|
||||
|
||||
assert "x-litellm-response-cost" not in fastapi_response.headers
|
||||
recompute.assert_not_called()
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue