fix(proxy): drop an upstream cost when the gateway has no price

An opted-in request against an unpriced deployment kept whatever cost the upstream had reported, so usage.cost could hand back a figure the gateway never charged, which is the one thing the field is meant to rule out

Also drops the banned type: ignore comments, which do nothing since pyrightconfig disables them, and trims the docstrings back to the one rule that is not obvious from the code
This commit is contained in:
Ben Langfeld 2026-09-16 22:19:03 -04:00
parent aa617a9f09
commit a7be4d7797
No known key found for this signature in database
2 changed files with 19 additions and 37 deletions

View file

@ -2821,8 +2821,6 @@ class ProxyBaseLLMRequestProcessing:
else llm_cost_for_headers
)
# Same value the x-litellm-response-cost header carries below, so the body and
# the header cannot report different costs for one request.
self._maybe_set_usage_cost(request, response, response_cost_for_headers)
# Always return the client-requested model name (not provider-prefixed internal identifiers)
@ -4098,28 +4096,11 @@ class ProxyBaseLLMRequestProcessing:
@staticmethod
def _maybe_set_usage_cost(request: Request, response: object, cost: float | str | None) -> None:
"""
Record the gateway's cost on ``usage.cost``, if this request asked for it.
Kept as one entry point so the decision and the write are exercised together
rather than only through a full request.
"""
if ProxyBaseLLMRequestProcessing._should_include_cost_in_usage(request):
ProxyBaseLLMRequestProcessing._set_usage_cost(response, cost)
@staticmethod
def _should_include_cost_in_usage(request: Request) -> bool:
"""
Whether this request asked for the gateway's cost on ``usage.cost``.
The decision is per-request. A caller that did not ask gets the body it would
have received before this existed, so turning the feature on for a proxy cannot
change the response shape for consumers who never opted in.
The ``x-litellm-include-cost-in-usage`` header wins over
``litellm.include_cost_in_usage`` in both directions, so a deployment that turns
it on globally can still be opted out of for one request.
"""
header_value: Final = request.headers.get(X_LITELLM_INCLUDE_COST_IN_USAGE)
if header_value is not None and header_value.strip() != "":
return header_value.strip().lower() in ("true", "1", "yes")
@ -4128,25 +4109,16 @@ class ProxyBaseLLMRequestProcessing:
@staticmethod
def _set_usage_cost(response: object, cost: float | str | None) -> None:
"""
Record what the gateway charged on the response's usage object.
``usage.cost`` carries one meaning: the figure this gateway billed, which is the
same value ``x-litellm-response-cost`` reports. Some upstreams (OpenRouter) send
a cost of their own under that name - a different number, computed by a
different party, against pricing we did not apply - so it is replaced rather
than left in place. A caller reading ``usage.cost`` should never have to know
which deployment served the request to know what the number means.
A non-numeric cost means the deployment carries no configured price. The field
is then left unset, so an unpriced request stays distinguishable from a free
one - the same reason ``x-litellm-response-cost`` is omitted rather than sent as
``0.0``. A real ``0.0``, as unbilled non-inference calls produce, is recorded.
usage.cost means what this gateway charged, so an upstream provider's own figure is
dropped rather than left behind: an unpriced deployment reports no cost at all, and a
caller reading the field should not get a number from whoever happened to serve it.
"""
if isinstance(cost, bool) or not isinstance(cost, (int, float)):
return
usage: Final = getattr(response, "usage", None)
if not isinstance(usage, Usage):
return
if isinstance(cost, bool) or not isinstance(cost, (int, float)):
del usage.cost
return
usage.cost = float(cost)
@staticmethod

View file

@ -9318,7 +9318,7 @@ def _request_with_headers(**headers: str) -> Request:
def _response_with_usage(**usage_kwargs: object):
from litellm.types.utils import ModelResponse, Usage
return ModelResponse(usage=Usage(**usage_kwargs)) # type: ignore[arg-type]
return ModelResponse(usage=Usage(**usage_kwargs))
class TestIncludeCostInUsage:
@ -9351,7 +9351,7 @@ class TestIncludeCostInUsage:
def test_cost_is_recorded_on_usage(self):
response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16)
ProxyBaseLLMRequestProcessing._set_usage_cost(response, 5.85e-06)
assert response.usage.cost == 5.85e-06 # type: ignore[attr-defined]
assert response.model_dump()["usage"]["cost"] == 5.85e-06
def test_a_provider_supplied_cost_is_replaced(self):
"""
@ -9361,7 +9361,17 @@ class TestIncludeCostInUsage:
"""
response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16, cost=8.775e-06)
ProxyBaseLLMRequestProcessing._set_usage_cost(response, 5.85e-06)
assert response.usage.cost == 5.85e-06 # type: ignore[attr-defined]
assert response.model_dump()["usage"]["cost"] == 5.85e-06
@pytest.mark.parametrize("unpriced", ["", None, "None"])
def test_an_unpriced_deployment_drops_a_provider_supplied_cost(self, unpriced):
"""
An upstream's own figure must not survive as the answer when the gateway has no
price of its own, or an opted-in caller reads a number the gateway never charged.
"""
response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16, cost=8.775e-06)
ProxyBaseLLMRequestProcessing._set_usage_cost(response, unpriced)
assert "cost" not in response.model_dump()["usage"]
@pytest.mark.parametrize("unpriced", ["", None, "None"])
def test_an_unpriced_deployment_leaves_the_field_absent(self, unpriced):