From 5e6fbc749d609a04b85d4173bf96e004fcae5b0f Mon Sep 17 00:00:00 2001 From: Ben Langfeld Date: Wed, 16 Sep 2026 21:15:24 -0400 Subject: [PATCH 1/3] Return the gateway's cost on usage.cost for non-streaming responses Cost is reachable today only as the x-litellm-response-cost header, which clients built on SDKs that abstract the transport cannot read. Streaming responses already carry the figure on usage.cost behind include_cost_in_streaming_usage; this extends the same field to non-streaming responses, opt-in per request via the x-litellm-include-cost-in-usage header or litellm.include_cost_in_usage. The injected value is the one already computed for the response header, so body and header cannot report different costs for the same request, and no cost is calculated that was not being calculated already. usage.cost is a declared field on Usage whose constructor deletes it when unset, so a caller that did not opt in receives an unchanged response and an unpriced deployment leaves the field absent rather than recording a free call. A real 0.0, as unbilled non-inference calls produce, is still recorded. A cost an upstream provider supplied under the same name is replaced rather than preserved: it is a different number computed by a different party, and a caller should not have to know which deployment served the request to know what the field means. Co-Authored-By: Claude Opus 5 --- litellm/__init__.py | 1 + litellm/constants.py | 1 + litellm/proxy/common_request_processing.py | 48 +++++++++++ .../proxy/test_common_request_processing.py | 81 +++++++++++++++++++ 4 files changed, 131 insertions(+) diff --git a/litellm/__init__.py b/litellm/__init__.py index a5c638e4a9a..c5580f35c86 100644 --- a/litellm/__init__.py +++ b/litellm/__init__.py @@ -371,6 +371,7 @@ banned_keywords_list: Optional[Union[str, List]] = None llm_guard_mode: Literal["all", "key-specific", "request-specific"] = "all" guardrail_name_config_map: Dict[str, GuardrailItem] = {} include_cost_in_streaming_usage: bool = False +include_cost_in_usage: bool = False reasoning_auto_summary: bool = False ### PROMPTS #### from litellm.types.prompts.init_prompts import PromptSpec diff --git a/litellm/constants.py b/litellm/constants.py index 338fe0f6b85..da259974018 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -1493,6 +1493,7 @@ MAXIMUM_TRACEBACK_LINES_TO_LOG: Final = int(os.getenv("MAXIMUM_TRACEBACK_LINES_T # Headers to control callbacks X_LITELLM_DISABLE_CALLBACKS: Final = "x-litellm-disable-callbacks" +X_LITELLM_INCLUDE_COST_IN_USAGE: Final = "x-litellm-include-cost-in-usage" LITELLM_METADATA_FIELD: Final = "litellm_metadata" OLD_LITELLM_METADATA_FIELD: Final = "metadata" RETURN_RAW_MODEL_NAME_METADATA_KEY: Final = "_complexity_router_return_raw_model_name" diff --git a/litellm/proxy/common_request_processing.py b/litellm/proxy/common_request_processing.py index 2f39e6c71bc..23dc64f0c1a 100644 --- a/litellm/proxy/common_request_processing.py +++ b/litellm/proxy/common_request_processing.py @@ -43,6 +43,7 @@ from litellm.constants import ( STREAM_SSE_DATA_PREFIX, STREAM_SSE_KEEPALIVE_PING_BYTES, UNSAFE_PROXY_RESPONSE_HEADERS, + X_LITELLM_INCLUDE_COST_IN_USAGE, ) from litellm.integrations.custom_guardrail import CustomGuardrail from litellm.litellm_core_utils.core_helpers import ( @@ -2820,6 +2821,11 @@ class ProxyBaseLLMRequestProcessing: else llm_cost_for_headers ) + # Same value the x-litellm-response-cost header carries below, so the body and + # the header cannot report different costs for one request. + if self._should_include_cost_in_usage(request): + self._set_usage_cost(response, response_cost_for_headers) + # Always return the client-requested model name (not provider-prefixed internal identifiers) # for OpenAI-compatible responses. if requested_model_from_client: @@ -4091,6 +4097,48 @@ class ProxyBaseLLMRequestProcessing: return cost_from_logging_obj return ProxyBaseLLMRequestProcessing._completion_cost_or_none(model_response, model_name, service_tier) + @staticmethod + def _should_include_cost_in_usage(request: Request) -> bool: + """ + Whether this request asked for the gateway's cost on ``usage.cost``. + + The decision is per-request. A caller that did not ask gets the body it would + have received before this existed, so turning the feature on for a proxy cannot + change the response shape for consumers who never opted in. + + The ``x-litellm-include-cost-in-usage`` header wins over + ``litellm.include_cost_in_usage`` in both directions, so a deployment that turns + it on globally can still be opted out of for one request. + """ + header_value: Final = request.headers.get(X_LITELLM_INCLUDE_COST_IN_USAGE) + if header_value is not None and header_value.strip() != "": + return header_value.strip().lower() in ("true", "1", "yes") + return bool(getattr(litellm, "include_cost_in_usage", False)) + + @staticmethod + def _set_usage_cost(response: object, cost: float | str | None) -> None: + """ + Record what the gateway charged on the response's usage object. + + ``usage.cost`` carries one meaning: the figure this gateway billed, which is the + same value ``x-litellm-response-cost`` reports. Some upstreams (OpenRouter) send + a cost of their own under that name - a different number, computed by a + different party, against pricing we did not apply - so it is replaced rather + than left in place. A caller reading ``usage.cost`` should never have to know + which deployment served the request to know what the number means. + + A non-numeric cost means the deployment carries no configured price. The field + is then left unset, so an unpriced request stays distinguishable from a free + one - the same reason ``x-litellm-response-cost`` is omitted rather than sent as + ``0.0``. A real ``0.0``, as unbilled non-inference calls produce, is recorded. + """ + if isinstance(cost, bool) or not isinstance(cost, (int, float)): + return + usage: Final = getattr(response, "usage", None) + if not isinstance(usage, Usage): + return + usage.cost = float(cost) + @staticmethod def _inject_cost_into_usage_dict( obj: dict, model_name: str, litellm_logging_obj: LiteLLMLoggingObj | None = None diff --git a/tests/test_litellm/proxy/test_common_request_processing.py b/tests/test_litellm/proxy/test_common_request_processing.py index 4ac687625c2..2c298413586 100644 --- a/tests/test_litellm/proxy/test_common_request_processing.py +++ b/tests/test_litellm/proxy/test_common_request_processing.py @@ -9307,3 +9307,84 @@ class TestErrorLogCarriesCallId: record: Final = caplog.records[-1] assert record.litellm_call_id == call_id assert call_id in record.getMessage() + + +def _request_with_headers(**headers: str) -> Request: + """A Request carrying the given headers, built the way the proxy receives one.""" + encoded: Final = [(k.replace("_", "-").encode(), v.encode()) for k, v in headers.items()] + return Request({"type": "http", "method": "POST", "headers": encoded}) + + +def _response_with_usage(**usage_kwargs: object): + from litellm.types.utils import ModelResponse, Usage + + return ModelResponse(usage=Usage(**usage_kwargs)) # type: ignore[arg-type] + + +class TestIncludeCostInUsage: + """ + usage.cost carries the gateway's own figure, opt-in per request. + + The invariants that matter are that a caller who did not ask sees no change at + all, and that a caller who did ask can read the field without first working out + which deployment served the request. + """ + + def test_off_by_default(self): + assert ProxyBaseLLMRequestProcessing._should_include_cost_in_usage(_request_with_headers()) is False + + @pytest.mark.parametrize("value", ["true", "TRUE", "1", "yes", " true "]) + def test_header_opts_in(self, value): + request: Final = _request_with_headers(x_litellm_include_cost_in_usage=value) + assert ProxyBaseLLMRequestProcessing._should_include_cost_in_usage(request) is True + + @pytest.mark.parametrize("value", ["false", "0", "no", "anything-else"]) + def test_header_opts_out(self, value, monkeypatch): + monkeypatch.setattr(litellm, "include_cost_in_usage", True, raising=False) + request: Final = _request_with_headers(x_litellm_include_cost_in_usage=value) + assert ProxyBaseLLMRequestProcessing._should_include_cost_in_usage(request) is False + + def test_setting_applies_when_no_header_is_sent(self, monkeypatch): + monkeypatch.setattr(litellm, "include_cost_in_usage", True, raising=False) + assert ProxyBaseLLMRequestProcessing._should_include_cost_in_usage(_request_with_headers()) is True + + def test_cost_is_recorded_on_usage(self): + response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16) + ProxyBaseLLMRequestProcessing._set_usage_cost(response, 5.85e-06) + assert response.usage.cost == 5.85e-06 # type: ignore[attr-defined] + + def test_a_provider_supplied_cost_is_replaced(self): + """ + OpenRouter reports its own cost under this name. It is a different number, + computed by a different party, so the gateway's figure has to win - otherwise + the meaning of the field would depend on which deployment served the request. + """ + response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16, cost=8.775e-06) + ProxyBaseLLMRequestProcessing._set_usage_cost(response, 5.85e-06) + assert response.usage.cost == 5.85e-06 # type: ignore[attr-defined] + + @pytest.mark.parametrize("unpriced", ["", None, "None"]) + def test_an_unpriced_deployment_leaves_the_field_absent(self, unpriced): + """ + Absence is not zero. A deployment with no configured price must not serialize + as a free call, which is the failure mode that looks like a real result. + """ + response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16) + ProxyBaseLLMRequestProcessing._set_usage_cost(response, unpriced) + assert "cost" not in response.model_dump()["usage"] + + def test_a_real_zero_is_recorded(self): + """Unbilled non-inference calls cost 0.0, and that zero is an answer.""" + response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16) + ProxyBaseLLMRequestProcessing._set_usage_cost(response, 0.0) + assert response.model_dump()["usage"]["cost"] == 0.0 + + def test_untouched_usage_does_not_serialize_a_cost(self): + """The opted-out path has to be byte-identical, not merely null-valued.""" + response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16) + assert "cost" not in response.model_dump()["usage"] + + def test_a_response_without_usage_is_left_alone(self): + sentinel: Final = SimpleNamespace(usage=None) + ProxyBaseLLMRequestProcessing._set_usage_cost(sentinel, 5.85e-06) + assert sentinel.usage is None From aa617a9f09ade41ce23fac5862a873a720a1c656 Mon Sep 17 00:00:00 2001 From: Ben Langfeld Date: Wed, 16 Sep 2026 21:39:50 -0400 Subject: [PATCH 2/3] Make the usage.cost call site one testable entry point The call site was a two-line conditional inside _process_llm_request, so the injecting branch was only reachable through a full request and went uncovered. Collapsing it into _maybe_set_usage_cost leaves one unconditional line at the call site and puts the opt-in decision next to the write, where both branches can be exercised directly. Co-Authored-By: Claude Opus 5 --- litellm/proxy/common_request_processing.py | 14 ++++++++++++-- .../proxy/test_common_request_processing.py | 12 ++++++++++++ 2 files changed, 24 insertions(+), 2 deletions(-) diff --git a/litellm/proxy/common_request_processing.py b/litellm/proxy/common_request_processing.py index 23dc64f0c1a..fe873f2c2c1 100644 --- a/litellm/proxy/common_request_processing.py +++ b/litellm/proxy/common_request_processing.py @@ -2823,8 +2823,7 @@ class ProxyBaseLLMRequestProcessing: # Same value the x-litellm-response-cost header carries below, so the body and # the header cannot report different costs for one request. - if self._should_include_cost_in_usage(request): - self._set_usage_cost(response, response_cost_for_headers) + self._maybe_set_usage_cost(request, response, response_cost_for_headers) # Always return the client-requested model name (not provider-prefixed internal identifiers) # for OpenAI-compatible responses. @@ -4097,6 +4096,17 @@ class ProxyBaseLLMRequestProcessing: return cost_from_logging_obj return ProxyBaseLLMRequestProcessing._completion_cost_or_none(model_response, model_name, service_tier) + @staticmethod + def _maybe_set_usage_cost(request: Request, response: object, cost: float | str | None) -> None: + """ + Record the gateway's cost on ``usage.cost``, if this request asked for it. + + Kept as one entry point so the decision and the write are exercised together + rather than only through a full request. + """ + if ProxyBaseLLMRequestProcessing._should_include_cost_in_usage(request): + ProxyBaseLLMRequestProcessing._set_usage_cost(response, cost) + @staticmethod def _should_include_cost_in_usage(request: Request) -> bool: """ diff --git a/tests/test_litellm/proxy/test_common_request_processing.py b/tests/test_litellm/proxy/test_common_request_processing.py index 2c298413586..55948c92ff8 100644 --- a/tests/test_litellm/proxy/test_common_request_processing.py +++ b/tests/test_litellm/proxy/test_common_request_processing.py @@ -9384,6 +9384,18 @@ class TestIncludeCostInUsage: response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16) assert "cost" not in response.model_dump()["usage"] + def test_opted_in_request_gets_the_cost_recorded(self): + """The decision and the write, exercised together as the request path runs them.""" + response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16) + request: Final = _request_with_headers(x_litellm_include_cost_in_usage="true") + ProxyBaseLLMRequestProcessing._maybe_set_usage_cost(request, response, 5.85e-06) + assert response.model_dump()["usage"]["cost"] == 5.85e-06 + + def test_opted_out_request_is_left_untouched(self): + response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16) + ProxyBaseLLMRequestProcessing._maybe_set_usage_cost(_request_with_headers(), response, 5.85e-06) + assert "cost" not in response.model_dump()["usage"] + def test_a_response_without_usage_is_left_alone(self): sentinel: Final = SimpleNamespace(usage=None) ProxyBaseLLMRequestProcessing._set_usage_cost(sentinel, 5.85e-06) From a7be4d7797a0745895df8679f84a5f8c948d6453 Mon Sep 17 00:00:00 2001 From: Ben Langfeld Date: Wed, 16 Sep 2026 22:19:03 -0400 Subject: [PATCH 3/3] fix(proxy): drop an upstream cost when the gateway has no price An opted-in request against an unpriced deployment kept whatever cost the upstream had reported, so usage.cost could hand back a figure the gateway never charged, which is the one thing the field is meant to rule out Also drops the banned type: ignore comments, which do nothing since pyrightconfig disables them, and trims the docstrings back to the one rule that is not obvious from the code --- litellm/proxy/common_request_processing.py | 40 +++---------------- .../proxy/test_common_request_processing.py | 16 ++++++-- 2 files changed, 19 insertions(+), 37 deletions(-) diff --git a/litellm/proxy/common_request_processing.py b/litellm/proxy/common_request_processing.py index fe873f2c2c1..869f21cff02 100644 --- a/litellm/proxy/common_request_processing.py +++ b/litellm/proxy/common_request_processing.py @@ -2821,8 +2821,6 @@ class ProxyBaseLLMRequestProcessing: else llm_cost_for_headers ) - # Same value the x-litellm-response-cost header carries below, so the body and - # the header cannot report different costs for one request. self._maybe_set_usage_cost(request, response, response_cost_for_headers) # Always return the client-requested model name (not provider-prefixed internal identifiers) @@ -4098,28 +4096,11 @@ class ProxyBaseLLMRequestProcessing: @staticmethod def _maybe_set_usage_cost(request: Request, response: object, cost: float | str | None) -> None: - """ - Record the gateway's cost on ``usage.cost``, if this request asked for it. - - Kept as one entry point so the decision and the write are exercised together - rather than only through a full request. - """ if ProxyBaseLLMRequestProcessing._should_include_cost_in_usage(request): ProxyBaseLLMRequestProcessing._set_usage_cost(response, cost) @staticmethod def _should_include_cost_in_usage(request: Request) -> bool: - """ - Whether this request asked for the gateway's cost on ``usage.cost``. - - The decision is per-request. A caller that did not ask gets the body it would - have received before this existed, so turning the feature on for a proxy cannot - change the response shape for consumers who never opted in. - - The ``x-litellm-include-cost-in-usage`` header wins over - ``litellm.include_cost_in_usage`` in both directions, so a deployment that turns - it on globally can still be opted out of for one request. - """ header_value: Final = request.headers.get(X_LITELLM_INCLUDE_COST_IN_USAGE) if header_value is not None and header_value.strip() != "": return header_value.strip().lower() in ("true", "1", "yes") @@ -4128,25 +4109,16 @@ class ProxyBaseLLMRequestProcessing: @staticmethod def _set_usage_cost(response: object, cost: float | str | None) -> None: """ - Record what the gateway charged on the response's usage object. - - ``usage.cost`` carries one meaning: the figure this gateway billed, which is the - same value ``x-litellm-response-cost`` reports. Some upstreams (OpenRouter) send - a cost of their own under that name - a different number, computed by a - different party, against pricing we did not apply - so it is replaced rather - than left in place. A caller reading ``usage.cost`` should never have to know - which deployment served the request to know what the number means. - - A non-numeric cost means the deployment carries no configured price. The field - is then left unset, so an unpriced request stays distinguishable from a free - one - the same reason ``x-litellm-response-cost`` is omitted rather than sent as - ``0.0``. A real ``0.0``, as unbilled non-inference calls produce, is recorded. + usage.cost means what this gateway charged, so an upstream provider's own figure is + dropped rather than left behind: an unpriced deployment reports no cost at all, and a + caller reading the field should not get a number from whoever happened to serve it. """ - if isinstance(cost, bool) or not isinstance(cost, (int, float)): - return usage: Final = getattr(response, "usage", None) if not isinstance(usage, Usage): return + if isinstance(cost, bool) or not isinstance(cost, (int, float)): + del usage.cost + return usage.cost = float(cost) @staticmethod diff --git a/tests/test_litellm/proxy/test_common_request_processing.py b/tests/test_litellm/proxy/test_common_request_processing.py index 55948c92ff8..71f9ac23514 100644 --- a/tests/test_litellm/proxy/test_common_request_processing.py +++ b/tests/test_litellm/proxy/test_common_request_processing.py @@ -9318,7 +9318,7 @@ def _request_with_headers(**headers: str) -> Request: def _response_with_usage(**usage_kwargs: object): from litellm.types.utils import ModelResponse, Usage - return ModelResponse(usage=Usage(**usage_kwargs)) # type: ignore[arg-type] + return ModelResponse(usage=Usage(**usage_kwargs)) class TestIncludeCostInUsage: @@ -9351,7 +9351,7 @@ class TestIncludeCostInUsage: def test_cost_is_recorded_on_usage(self): response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16) ProxyBaseLLMRequestProcessing._set_usage_cost(response, 5.85e-06) - assert response.usage.cost == 5.85e-06 # type: ignore[attr-defined] + assert response.model_dump()["usage"]["cost"] == 5.85e-06 def test_a_provider_supplied_cost_is_replaced(self): """ @@ -9361,7 +9361,17 @@ class TestIncludeCostInUsage: """ response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16, cost=8.775e-06) ProxyBaseLLMRequestProcessing._set_usage_cost(response, 5.85e-06) - assert response.usage.cost == 5.85e-06 # type: ignore[attr-defined] + assert response.model_dump()["usage"]["cost"] == 5.85e-06 + + @pytest.mark.parametrize("unpriced", ["", None, "None"]) + def test_an_unpriced_deployment_drops_a_provider_supplied_cost(self, unpriced): + """ + An upstream's own figure must not survive as the answer when the gateway has no + price of its own, or an opted-in caller reads a number the gateway never charged. + """ + response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16, cost=8.775e-06) + ProxyBaseLLMRequestProcessing._set_usage_cost(response, unpriced) + assert "cost" not in response.model_dump()["usage"] @pytest.mark.parametrize("unpriced", ["", None, "None"]) def test_an_unpriced_deployment_leaves_the_field_absent(self, unpriced):