From a7be4d7797a0745895df8679f84a5f8c948d6453 Mon Sep 17 00:00:00 2001 From: Ben Langfeld Date: Wed, 16 Sep 2026 22:19:03 -0400 Subject: [PATCH] fix(proxy): drop an upstream cost when the gateway has no price An opted-in request against an unpriced deployment kept whatever cost the upstream had reported, so usage.cost could hand back a figure the gateway never charged, which is the one thing the field is meant to rule out Also drops the banned type: ignore comments, which do nothing since pyrightconfig disables them, and trims the docstrings back to the one rule that is not obvious from the code --- litellm/proxy/common_request_processing.py | 40 +++---------------- .../proxy/test_common_request_processing.py | 16 ++++++-- 2 files changed, 19 insertions(+), 37 deletions(-) diff --git a/litellm/proxy/common_request_processing.py b/litellm/proxy/common_request_processing.py index fe873f2c2c1..869f21cff02 100644 --- a/litellm/proxy/common_request_processing.py +++ b/litellm/proxy/common_request_processing.py @@ -2821,8 +2821,6 @@ class ProxyBaseLLMRequestProcessing: else llm_cost_for_headers ) - # Same value the x-litellm-response-cost header carries below, so the body and - # the header cannot report different costs for one request. self._maybe_set_usage_cost(request, response, response_cost_for_headers) # Always return the client-requested model name (not provider-prefixed internal identifiers) @@ -4098,28 +4096,11 @@ class ProxyBaseLLMRequestProcessing: @staticmethod def _maybe_set_usage_cost(request: Request, response: object, cost: float | str | None) -> None: - """ - Record the gateway's cost on ``usage.cost``, if this request asked for it. - - Kept as one entry point so the decision and the write are exercised together - rather than only through a full request. - """ if ProxyBaseLLMRequestProcessing._should_include_cost_in_usage(request): ProxyBaseLLMRequestProcessing._set_usage_cost(response, cost) @staticmethod def _should_include_cost_in_usage(request: Request) -> bool: - """ - Whether this request asked for the gateway's cost on ``usage.cost``. - - The decision is per-request. A caller that did not ask gets the body it would - have received before this existed, so turning the feature on for a proxy cannot - change the response shape for consumers who never opted in. - - The ``x-litellm-include-cost-in-usage`` header wins over - ``litellm.include_cost_in_usage`` in both directions, so a deployment that turns - it on globally can still be opted out of for one request. - """ header_value: Final = request.headers.get(X_LITELLM_INCLUDE_COST_IN_USAGE) if header_value is not None and header_value.strip() != "": return header_value.strip().lower() in ("true", "1", "yes") @@ -4128,25 +4109,16 @@ class ProxyBaseLLMRequestProcessing: @staticmethod def _set_usage_cost(response: object, cost: float | str | None) -> None: """ - Record what the gateway charged on the response's usage object. - - ``usage.cost`` carries one meaning: the figure this gateway billed, which is the - same value ``x-litellm-response-cost`` reports. Some upstreams (OpenRouter) send - a cost of their own under that name - a different number, computed by a - different party, against pricing we did not apply - so it is replaced rather - than left in place. A caller reading ``usage.cost`` should never have to know - which deployment served the request to know what the number means. - - A non-numeric cost means the deployment carries no configured price. The field - is then left unset, so an unpriced request stays distinguishable from a free - one - the same reason ``x-litellm-response-cost`` is omitted rather than sent as - ``0.0``. A real ``0.0``, as unbilled non-inference calls produce, is recorded. + usage.cost means what this gateway charged, so an upstream provider's own figure is + dropped rather than left behind: an unpriced deployment reports no cost at all, and a + caller reading the field should not get a number from whoever happened to serve it. """ - if isinstance(cost, bool) or not isinstance(cost, (int, float)): - return usage: Final = getattr(response, "usage", None) if not isinstance(usage, Usage): return + if isinstance(cost, bool) or not isinstance(cost, (int, float)): + del usage.cost + return usage.cost = float(cost) @staticmethod diff --git a/tests/test_litellm/proxy/test_common_request_processing.py b/tests/test_litellm/proxy/test_common_request_processing.py index 55948c92ff8..71f9ac23514 100644 --- a/tests/test_litellm/proxy/test_common_request_processing.py +++ b/tests/test_litellm/proxy/test_common_request_processing.py @@ -9318,7 +9318,7 @@ def _request_with_headers(**headers: str) -> Request: def _response_with_usage(**usage_kwargs: object): from litellm.types.utils import ModelResponse, Usage - return ModelResponse(usage=Usage(**usage_kwargs)) # type: ignore[arg-type] + return ModelResponse(usage=Usage(**usage_kwargs)) class TestIncludeCostInUsage: @@ -9351,7 +9351,7 @@ class TestIncludeCostInUsage: def test_cost_is_recorded_on_usage(self): response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16) ProxyBaseLLMRequestProcessing._set_usage_cost(response, 5.85e-06) - assert response.usage.cost == 5.85e-06 # type: ignore[attr-defined] + assert response.model_dump()["usage"]["cost"] == 5.85e-06 def test_a_provider_supplied_cost_is_replaced(self): """ @@ -9361,7 +9361,17 @@ class TestIncludeCostInUsage: """ response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16, cost=8.775e-06) ProxyBaseLLMRequestProcessing._set_usage_cost(response, 5.85e-06) - assert response.usage.cost == 5.85e-06 # type: ignore[attr-defined] + assert response.model_dump()["usage"]["cost"] == 5.85e-06 + + @pytest.mark.parametrize("unpriced", ["", None, "None"]) + def test_an_unpriced_deployment_drops_a_provider_supplied_cost(self, unpriced): + """ + An upstream's own figure must not survive as the answer when the gateway has no + price of its own, or an opted-in caller reads a number the gateway never charged. + """ + response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16, cost=8.775e-06) + ProxyBaseLLMRequestProcessing._set_usage_cost(response, unpriced) + assert "cost" not in response.model_dump()["usage"] @pytest.mark.parametrize("unpriced", ["", None, "None"]) def test_an_unpriced_deployment_leaves_the_field_absent(self, unpriced):