Return the gateway's cost on usage.cost for non-streaming responses

Cost is reachable today only as the x-litellm-response-cost header, which clients built on SDKs that abstract the transport cannot read. Streaming responses already carry the figure on usage.cost behind include_cost_in_streaming_usage; this extends the same field to non-streaming responses, opt-in per request via the x-litellm-include-cost-in-usage header or litellm.include_cost_in_usage.

The injected value is the one already computed for the response header, so body and header cannot report different costs for the same request, and no cost is calculated that was not being calculated already.

usage.cost is a declared field on Usage whose constructor deletes it when unset, so a caller that did not opt in receives an unchanged response and an unpriced deployment leaves the field absent rather than recording a free call. A real 0.0, as unbilled non-inference calls produce, is still recorded.

A cost an upstream provider supplied under the same name is replaced rather than preserved: it is a different number computed by a different party, and a caller should not have to know which deployment served the request to know what the field means.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
Ben Langfeld 2026-09-16 21:15:24 -04:00
parent 38676aa599
commit 5e6fbc749d
No known key found for this signature in database
4 changed files with 131 additions and 0 deletions

View file

@ -371,6 +371,7 @@ banned_keywords_list: Optional[Union[str, List]] = None
llm_guard_mode: Literal["all", "key-specific", "request-specific"] = "all"
guardrail_name_config_map: Dict[str, GuardrailItem] = {}
include_cost_in_streaming_usage: bool = False
include_cost_in_usage: bool = False
reasoning_auto_summary: bool = False
### PROMPTS ####
from litellm.types.prompts.init_prompts import PromptSpec

View file

@ -1493,6 +1493,7 @@ MAXIMUM_TRACEBACK_LINES_TO_LOG: Final = int(os.getenv("MAXIMUM_TRACEBACK_LINES_T
# Headers to control callbacks
X_LITELLM_DISABLE_CALLBACKS: Final = "x-litellm-disable-callbacks"
X_LITELLM_INCLUDE_COST_IN_USAGE: Final = "x-litellm-include-cost-in-usage"
LITELLM_METADATA_FIELD: Final = "litellm_metadata"
OLD_LITELLM_METADATA_FIELD: Final = "metadata"
RETURN_RAW_MODEL_NAME_METADATA_KEY: Final = "_complexity_router_return_raw_model_name"

View file

@ -43,6 +43,7 @@ from litellm.constants import (
STREAM_SSE_DATA_PREFIX,
STREAM_SSE_KEEPALIVE_PING_BYTES,
UNSAFE_PROXY_RESPONSE_HEADERS,
X_LITELLM_INCLUDE_COST_IN_USAGE,
)
from litellm.integrations.custom_guardrail import CustomGuardrail
from litellm.litellm_core_utils.core_helpers import (
@ -2820,6 +2821,11 @@ class ProxyBaseLLMRequestProcessing:
else llm_cost_for_headers
)
# Same value the x-litellm-response-cost header carries below, so the body and
# the header cannot report different costs for one request.
if self._should_include_cost_in_usage(request):
self._set_usage_cost(response, response_cost_for_headers)
# Always return the client-requested model name (not provider-prefixed internal identifiers)
# for OpenAI-compatible responses.
if requested_model_from_client:
@ -4091,6 +4097,48 @@ class ProxyBaseLLMRequestProcessing:
return cost_from_logging_obj
return ProxyBaseLLMRequestProcessing._completion_cost_or_none(model_response, model_name, service_tier)
@staticmethod
def _should_include_cost_in_usage(request: Request) -> bool:
"""
Whether this request asked for the gateway's cost on ``usage.cost``.
The decision is per-request. A caller that did not ask gets the body it would
have received before this existed, so turning the feature on for a proxy cannot
change the response shape for consumers who never opted in.
The ``x-litellm-include-cost-in-usage`` header wins over
``litellm.include_cost_in_usage`` in both directions, so a deployment that turns
it on globally can still be opted out of for one request.
"""
header_value: Final = request.headers.get(X_LITELLM_INCLUDE_COST_IN_USAGE)
if header_value is not None and header_value.strip() != "":
return header_value.strip().lower() in ("true", "1", "yes")
return bool(getattr(litellm, "include_cost_in_usage", False))
@staticmethod
def _set_usage_cost(response: object, cost: float | str | None) -> None:
"""
Record what the gateway charged on the response's usage object.
``usage.cost`` carries one meaning: the figure this gateway billed, which is the
same value ``x-litellm-response-cost`` reports. Some upstreams (OpenRouter) send
a cost of their own under that name - a different number, computed by a
different party, against pricing we did not apply - so it is replaced rather
than left in place. A caller reading ``usage.cost`` should never have to know
which deployment served the request to know what the number means.
A non-numeric cost means the deployment carries no configured price. The field
is then left unset, so an unpriced request stays distinguishable from a free
one - the same reason ``x-litellm-response-cost`` is omitted rather than sent as
``0.0``. A real ``0.0``, as unbilled non-inference calls produce, is recorded.
"""
if isinstance(cost, bool) or not isinstance(cost, (int, float)):
return
usage: Final = getattr(response, "usage", None)
if not isinstance(usage, Usage):
return
usage.cost = float(cost)
@staticmethod
def _inject_cost_into_usage_dict(
obj: dict, model_name: str, litellm_logging_obj: LiteLLMLoggingObj | None = None

View file

@ -9307,3 +9307,84 @@ class TestErrorLogCarriesCallId:
record: Final = caplog.records[-1]
assert record.litellm_call_id == call_id
assert call_id in record.getMessage()
def _request_with_headers(**headers: str) -> Request:
"""A Request carrying the given headers, built the way the proxy receives one."""
encoded: Final = [(k.replace("_", "-").encode(), v.encode()) for k, v in headers.items()]
return Request({"type": "http", "method": "POST", "headers": encoded})
def _response_with_usage(**usage_kwargs: object):
from litellm.types.utils import ModelResponse, Usage
return ModelResponse(usage=Usage(**usage_kwargs)) # type: ignore[arg-type]
class TestIncludeCostInUsage:
"""
usage.cost carries the gateway's own figure, opt-in per request.
The invariants that matter are that a caller who did not ask sees no change at
all, and that a caller who did ask can read the field without first working out
which deployment served the request.
"""
def test_off_by_default(self):
assert ProxyBaseLLMRequestProcessing._should_include_cost_in_usage(_request_with_headers()) is False
@pytest.mark.parametrize("value", ["true", "TRUE", "1", "yes", " true "])
def test_header_opts_in(self, value):
request: Final = _request_with_headers(x_litellm_include_cost_in_usage=value)
assert ProxyBaseLLMRequestProcessing._should_include_cost_in_usage(request) is True
@pytest.mark.parametrize("value", ["false", "0", "no", "anything-else"])
def test_header_opts_out(self, value, monkeypatch):
monkeypatch.setattr(litellm, "include_cost_in_usage", True, raising=False)
request: Final = _request_with_headers(x_litellm_include_cost_in_usage=value)
assert ProxyBaseLLMRequestProcessing._should_include_cost_in_usage(request) is False
def test_setting_applies_when_no_header_is_sent(self, monkeypatch):
monkeypatch.setattr(litellm, "include_cost_in_usage", True, raising=False)
assert ProxyBaseLLMRequestProcessing._should_include_cost_in_usage(_request_with_headers()) is True
def test_cost_is_recorded_on_usage(self):
response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16)
ProxyBaseLLMRequestProcessing._set_usage_cost(response, 5.85e-06)
assert response.usage.cost == 5.85e-06 # type: ignore[attr-defined]
def test_a_provider_supplied_cost_is_replaced(self):
"""
OpenRouter reports its own cost under this name. It is a different number,
computed by a different party, so the gateway's figure has to win - otherwise
the meaning of the field would depend on which deployment served the request.
"""
response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16, cost=8.775e-06)
ProxyBaseLLMRequestProcessing._set_usage_cost(response, 5.85e-06)
assert response.usage.cost == 5.85e-06 # type: ignore[attr-defined]
@pytest.mark.parametrize("unpriced", ["", None, "None"])
def test_an_unpriced_deployment_leaves_the_field_absent(self, unpriced):
"""
Absence is not zero. A deployment with no configured price must not serialize
as a free call, which is the failure mode that looks like a real result.
"""
response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16)
ProxyBaseLLMRequestProcessing._set_usage_cost(response, unpriced)
assert "cost" not in response.model_dump()["usage"]
def test_a_real_zero_is_recorded(self):
"""Unbilled non-inference calls cost 0.0, and that zero is an answer."""
response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16)
ProxyBaseLLMRequestProcessing._set_usage_cost(response, 0.0)
assert response.model_dump()["usage"]["cost"] == 0.0
def test_untouched_usage_does_not_serialize_a_cost(self):
"""The opted-out path has to be byte-identical, not merely null-valued."""
response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16)
assert "cost" not in response.model_dump()["usage"]
def test_a_response_without_usage_is_left_alone(self):
sentinel: Final = SimpleNamespace(usage=None)
ProxyBaseLLMRequestProcessing._set_usage_cost(sentinel, 5.85e-06)
assert sentinel.usage is None