From 5e6fbc749d609a04b85d4173bf96e004fcae5b0f Mon Sep 17 00:00:00 2001 From: Ben Langfeld Date: Wed, 16 Sep 2026 21:15:24 -0400 Subject: [PATCH] Return the gateway's cost on usage.cost for non-streaming responses Cost is reachable today only as the x-litellm-response-cost header, which clients built on SDKs that abstract the transport cannot read. Streaming responses already carry the figure on usage.cost behind include_cost_in_streaming_usage; this extends the same field to non-streaming responses, opt-in per request via the x-litellm-include-cost-in-usage header or litellm.include_cost_in_usage. The injected value is the one already computed for the response header, so body and header cannot report different costs for the same request, and no cost is calculated that was not being calculated already. usage.cost is a declared field on Usage whose constructor deletes it when unset, so a caller that did not opt in receives an unchanged response and an unpriced deployment leaves the field absent rather than recording a free call. A real 0.0, as unbilled non-inference calls produce, is still recorded. A cost an upstream provider supplied under the same name is replaced rather than preserved: it is a different number computed by a different party, and a caller should not have to know which deployment served the request to know what the field means. Co-Authored-By: Claude Opus 5 --- litellm/__init__.py | 1 + litellm/constants.py | 1 + litellm/proxy/common_request_processing.py | 48 +++++++++++ .../proxy/test_common_request_processing.py | 81 +++++++++++++++++++ 4 files changed, 131 insertions(+) diff --git a/litellm/__init__.py b/litellm/__init__.py index a5c638e4a9a..c5580f35c86 100644 --- a/litellm/__init__.py +++ b/litellm/__init__.py @@ -371,6 +371,7 @@ banned_keywords_list: Optional[Union[str, List]] = None llm_guard_mode: Literal["all", "key-specific", "request-specific"] = "all" guardrail_name_config_map: Dict[str, GuardrailItem] = {} include_cost_in_streaming_usage: bool = False +include_cost_in_usage: bool = False reasoning_auto_summary: bool = False ### PROMPTS #### from litellm.types.prompts.init_prompts import PromptSpec diff --git a/litellm/constants.py b/litellm/constants.py index 338fe0f6b85..da259974018 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -1493,6 +1493,7 @@ MAXIMUM_TRACEBACK_LINES_TO_LOG: Final = int(os.getenv("MAXIMUM_TRACEBACK_LINES_T # Headers to control callbacks X_LITELLM_DISABLE_CALLBACKS: Final = "x-litellm-disable-callbacks" +X_LITELLM_INCLUDE_COST_IN_USAGE: Final = "x-litellm-include-cost-in-usage" LITELLM_METADATA_FIELD: Final = "litellm_metadata" OLD_LITELLM_METADATA_FIELD: Final = "metadata" RETURN_RAW_MODEL_NAME_METADATA_KEY: Final = "_complexity_router_return_raw_model_name" diff --git a/litellm/proxy/common_request_processing.py b/litellm/proxy/common_request_processing.py index 2f39e6c71bc..23dc64f0c1a 100644 --- a/litellm/proxy/common_request_processing.py +++ b/litellm/proxy/common_request_processing.py @@ -43,6 +43,7 @@ from litellm.constants import ( STREAM_SSE_DATA_PREFIX, STREAM_SSE_KEEPALIVE_PING_BYTES, UNSAFE_PROXY_RESPONSE_HEADERS, + X_LITELLM_INCLUDE_COST_IN_USAGE, ) from litellm.integrations.custom_guardrail import CustomGuardrail from litellm.litellm_core_utils.core_helpers import ( @@ -2820,6 +2821,11 @@ class ProxyBaseLLMRequestProcessing: else llm_cost_for_headers ) + # Same value the x-litellm-response-cost header carries below, so the body and + # the header cannot report different costs for one request. + if self._should_include_cost_in_usage(request): + self._set_usage_cost(response, response_cost_for_headers) + # Always return the client-requested model name (not provider-prefixed internal identifiers) # for OpenAI-compatible responses. if requested_model_from_client: @@ -4091,6 +4097,48 @@ class ProxyBaseLLMRequestProcessing: return cost_from_logging_obj return ProxyBaseLLMRequestProcessing._completion_cost_or_none(model_response, model_name, service_tier) + @staticmethod + def _should_include_cost_in_usage(request: Request) -> bool: + """ + Whether this request asked for the gateway's cost on ``usage.cost``. + + The decision is per-request. A caller that did not ask gets the body it would + have received before this existed, so turning the feature on for a proxy cannot + change the response shape for consumers who never opted in. + + The ``x-litellm-include-cost-in-usage`` header wins over + ``litellm.include_cost_in_usage`` in both directions, so a deployment that turns + it on globally can still be opted out of for one request. + """ + header_value: Final = request.headers.get(X_LITELLM_INCLUDE_COST_IN_USAGE) + if header_value is not None and header_value.strip() != "": + return header_value.strip().lower() in ("true", "1", "yes") + return bool(getattr(litellm, "include_cost_in_usage", False)) + + @staticmethod + def _set_usage_cost(response: object, cost: float | str | None) -> None: + """ + Record what the gateway charged on the response's usage object. + + ``usage.cost`` carries one meaning: the figure this gateway billed, which is the + same value ``x-litellm-response-cost`` reports. Some upstreams (OpenRouter) send + a cost of their own under that name - a different number, computed by a + different party, against pricing we did not apply - so it is replaced rather + than left in place. A caller reading ``usage.cost`` should never have to know + which deployment served the request to know what the number means. + + A non-numeric cost means the deployment carries no configured price. The field + is then left unset, so an unpriced request stays distinguishable from a free + one - the same reason ``x-litellm-response-cost`` is omitted rather than sent as + ``0.0``. A real ``0.0``, as unbilled non-inference calls produce, is recorded. + """ + if isinstance(cost, bool) or not isinstance(cost, (int, float)): + return + usage: Final = getattr(response, "usage", None) + if not isinstance(usage, Usage): + return + usage.cost = float(cost) + @staticmethod def _inject_cost_into_usage_dict( obj: dict, model_name: str, litellm_logging_obj: LiteLLMLoggingObj | None = None diff --git a/tests/test_litellm/proxy/test_common_request_processing.py b/tests/test_litellm/proxy/test_common_request_processing.py index 4ac687625c2..2c298413586 100644 --- a/tests/test_litellm/proxy/test_common_request_processing.py +++ b/tests/test_litellm/proxy/test_common_request_processing.py @@ -9307,3 +9307,84 @@ class TestErrorLogCarriesCallId: record: Final = caplog.records[-1] assert record.litellm_call_id == call_id assert call_id in record.getMessage() + + +def _request_with_headers(**headers: str) -> Request: + """A Request carrying the given headers, built the way the proxy receives one.""" + encoded: Final = [(k.replace("_", "-").encode(), v.encode()) for k, v in headers.items()] + return Request({"type": "http", "method": "POST", "headers": encoded}) + + +def _response_with_usage(**usage_kwargs: object): + from litellm.types.utils import ModelResponse, Usage + + return ModelResponse(usage=Usage(**usage_kwargs)) # type: ignore[arg-type] + + +class TestIncludeCostInUsage: + """ + usage.cost carries the gateway's own figure, opt-in per request. + + The invariants that matter are that a caller who did not ask sees no change at + all, and that a caller who did ask can read the field without first working out + which deployment served the request. + """ + + def test_off_by_default(self): + assert ProxyBaseLLMRequestProcessing._should_include_cost_in_usage(_request_with_headers()) is False + + @pytest.mark.parametrize("value", ["true", "TRUE", "1", "yes", " true "]) + def test_header_opts_in(self, value): + request: Final = _request_with_headers(x_litellm_include_cost_in_usage=value) + assert ProxyBaseLLMRequestProcessing._should_include_cost_in_usage(request) is True + + @pytest.mark.parametrize("value", ["false", "0", "no", "anything-else"]) + def test_header_opts_out(self, value, monkeypatch): + monkeypatch.setattr(litellm, "include_cost_in_usage", True, raising=False) + request: Final = _request_with_headers(x_litellm_include_cost_in_usage=value) + assert ProxyBaseLLMRequestProcessing._should_include_cost_in_usage(request) is False + + def test_setting_applies_when_no_header_is_sent(self, monkeypatch): + monkeypatch.setattr(litellm, "include_cost_in_usage", True, raising=False) + assert ProxyBaseLLMRequestProcessing._should_include_cost_in_usage(_request_with_headers()) is True + + def test_cost_is_recorded_on_usage(self): + response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16) + ProxyBaseLLMRequestProcessing._set_usage_cost(response, 5.85e-06) + assert response.usage.cost == 5.85e-06 # type: ignore[attr-defined] + + def test_a_provider_supplied_cost_is_replaced(self): + """ + OpenRouter reports its own cost under this name. It is a different number, + computed by a different party, so the gateway's figure has to win - otherwise + the meaning of the field would depend on which deployment served the request. + """ + response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16, cost=8.775e-06) + ProxyBaseLLMRequestProcessing._set_usage_cost(response, 5.85e-06) + assert response.usage.cost == 5.85e-06 # type: ignore[attr-defined] + + @pytest.mark.parametrize("unpriced", ["", None, "None"]) + def test_an_unpriced_deployment_leaves_the_field_absent(self, unpriced): + """ + Absence is not zero. A deployment with no configured price must not serialize + as a free call, which is the failure mode that looks like a real result. + """ + response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16) + ProxyBaseLLMRequestProcessing._set_usage_cost(response, unpriced) + assert "cost" not in response.model_dump()["usage"] + + def test_a_real_zero_is_recorded(self): + """Unbilled non-inference calls cost 0.0, and that zero is an answer.""" + response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16) + ProxyBaseLLMRequestProcessing._set_usage_cost(response, 0.0) + assert response.model_dump()["usage"]["cost"] == 0.0 + + def test_untouched_usage_does_not_serialize_a_cost(self): + """The opted-out path has to be byte-identical, not merely null-valued.""" + response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16) + assert "cost" not in response.model_dump()["usage"] + + def test_a_response_without_usage_is_left_alone(self): + sentinel: Final = SimpleNamespace(usage=None) + ProxyBaseLLMRequestProcessing._set_usage_cost(sentinel, 5.85e-06) + assert sentinel.usage is None