diff --git a/litellm/__init__.py b/litellm/__init__.py index a5c638e4a9a..c5580f35c86 100644 --- a/litellm/__init__.py +++ b/litellm/__init__.py @@ -371,6 +371,7 @@ banned_keywords_list: Optional[Union[str, List]] = None llm_guard_mode: Literal["all", "key-specific", "request-specific"] = "all" guardrail_name_config_map: Dict[str, GuardrailItem] = {} include_cost_in_streaming_usage: bool = False +include_cost_in_usage: bool = False reasoning_auto_summary: bool = False ### PROMPTS #### from litellm.types.prompts.init_prompts import PromptSpec diff --git a/litellm/constants.py b/litellm/constants.py index 338fe0f6b85..da259974018 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -1493,6 +1493,7 @@ MAXIMUM_TRACEBACK_LINES_TO_LOG: Final = int(os.getenv("MAXIMUM_TRACEBACK_LINES_T # Headers to control callbacks X_LITELLM_DISABLE_CALLBACKS: Final = "x-litellm-disable-callbacks" +X_LITELLM_INCLUDE_COST_IN_USAGE: Final = "x-litellm-include-cost-in-usage" LITELLM_METADATA_FIELD: Final = "litellm_metadata" OLD_LITELLM_METADATA_FIELD: Final = "metadata" RETURN_RAW_MODEL_NAME_METADATA_KEY: Final = "_complexity_router_return_raw_model_name" diff --git a/litellm/proxy/common_request_processing.py b/litellm/proxy/common_request_processing.py index 2f39e6c71bc..23dc64f0c1a 100644 --- a/litellm/proxy/common_request_processing.py +++ b/litellm/proxy/common_request_processing.py @@ -43,6 +43,7 @@ from litellm.constants import ( STREAM_SSE_DATA_PREFIX, STREAM_SSE_KEEPALIVE_PING_BYTES, UNSAFE_PROXY_RESPONSE_HEADERS, + X_LITELLM_INCLUDE_COST_IN_USAGE, ) from litellm.integrations.custom_guardrail import CustomGuardrail from litellm.litellm_core_utils.core_helpers import ( @@ -2820,6 +2821,11 @@ class ProxyBaseLLMRequestProcessing: else llm_cost_for_headers ) + # Same value the x-litellm-response-cost header carries below, so the body and + # the header cannot report different costs for one request. + if self._should_include_cost_in_usage(request): + self._set_usage_cost(response, response_cost_for_headers) + # Always return the client-requested model name (not provider-prefixed internal identifiers) # for OpenAI-compatible responses. if requested_model_from_client: @@ -4091,6 +4097,48 @@ class ProxyBaseLLMRequestProcessing: return cost_from_logging_obj return ProxyBaseLLMRequestProcessing._completion_cost_or_none(model_response, model_name, service_tier) + @staticmethod + def _should_include_cost_in_usage(request: Request) -> bool: + """ + Whether this request asked for the gateway's cost on ``usage.cost``. + + The decision is per-request. A caller that did not ask gets the body it would + have received before this existed, so turning the feature on for a proxy cannot + change the response shape for consumers who never opted in. + + The ``x-litellm-include-cost-in-usage`` header wins over + ``litellm.include_cost_in_usage`` in both directions, so a deployment that turns + it on globally can still be opted out of for one request. + """ + header_value: Final = request.headers.get(X_LITELLM_INCLUDE_COST_IN_USAGE) + if header_value is not None and header_value.strip() != "": + return header_value.strip().lower() in ("true", "1", "yes") + return bool(getattr(litellm, "include_cost_in_usage", False)) + + @staticmethod + def _set_usage_cost(response: object, cost: float | str | None) -> None: + """ + Record what the gateway charged on the response's usage object. + + ``usage.cost`` carries one meaning: the figure this gateway billed, which is the + same value ``x-litellm-response-cost`` reports. Some upstreams (OpenRouter) send + a cost of their own under that name - a different number, computed by a + different party, against pricing we did not apply - so it is replaced rather + than left in place. A caller reading ``usage.cost`` should never have to know + which deployment served the request to know what the number means. + + A non-numeric cost means the deployment carries no configured price. The field + is then left unset, so an unpriced request stays distinguishable from a free + one - the same reason ``x-litellm-response-cost`` is omitted rather than sent as + ``0.0``. A real ``0.0``, as unbilled non-inference calls produce, is recorded. + """ + if isinstance(cost, bool) or not isinstance(cost, (int, float)): + return + usage: Final = getattr(response, "usage", None) + if not isinstance(usage, Usage): + return + usage.cost = float(cost) + @staticmethod def _inject_cost_into_usage_dict( obj: dict, model_name: str, litellm_logging_obj: LiteLLMLoggingObj | None = None diff --git a/tests/test_litellm/proxy/test_common_request_processing.py b/tests/test_litellm/proxy/test_common_request_processing.py index 4ac687625c2..2c298413586 100644 --- a/tests/test_litellm/proxy/test_common_request_processing.py +++ b/tests/test_litellm/proxy/test_common_request_processing.py @@ -9307,3 +9307,84 @@ class TestErrorLogCarriesCallId: record: Final = caplog.records[-1] assert record.litellm_call_id == call_id assert call_id in record.getMessage() + + +def _request_with_headers(**headers: str) -> Request: + """A Request carrying the given headers, built the way the proxy receives one.""" + encoded: Final = [(k.replace("_", "-").encode(), v.encode()) for k, v in headers.items()] + return Request({"type": "http", "method": "POST", "headers": encoded}) + + +def _response_with_usage(**usage_kwargs: object): + from litellm.types.utils import ModelResponse, Usage + + return ModelResponse(usage=Usage(**usage_kwargs)) # type: ignore[arg-type] + + +class TestIncludeCostInUsage: + """ + usage.cost carries the gateway's own figure, opt-in per request. + + The invariants that matter are that a caller who did not ask sees no change at + all, and that a caller who did ask can read the field without first working out + which deployment served the request. + """ + + def test_off_by_default(self): + assert ProxyBaseLLMRequestProcessing._should_include_cost_in_usage(_request_with_headers()) is False + + @pytest.mark.parametrize("value", ["true", "TRUE", "1", "yes", " true "]) + def test_header_opts_in(self, value): + request: Final = _request_with_headers(x_litellm_include_cost_in_usage=value) + assert ProxyBaseLLMRequestProcessing._should_include_cost_in_usage(request) is True + + @pytest.mark.parametrize("value", ["false", "0", "no", "anything-else"]) + def test_header_opts_out(self, value, monkeypatch): + monkeypatch.setattr(litellm, "include_cost_in_usage", True, raising=False) + request: Final = _request_with_headers(x_litellm_include_cost_in_usage=value) + assert ProxyBaseLLMRequestProcessing._should_include_cost_in_usage(request) is False + + def test_setting_applies_when_no_header_is_sent(self, monkeypatch): + monkeypatch.setattr(litellm, "include_cost_in_usage", True, raising=False) + assert ProxyBaseLLMRequestProcessing._should_include_cost_in_usage(_request_with_headers()) is True + + def test_cost_is_recorded_on_usage(self): + response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16) + ProxyBaseLLMRequestProcessing._set_usage_cost(response, 5.85e-06) + assert response.usage.cost == 5.85e-06 # type: ignore[attr-defined] + + def test_a_provider_supplied_cost_is_replaced(self): + """ + OpenRouter reports its own cost under this name. It is a different number, + computed by a different party, so the gateway's figure has to win - otherwise + the meaning of the field would depend on which deployment served the request. + """ + response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16, cost=8.775e-06) + ProxyBaseLLMRequestProcessing._set_usage_cost(response, 5.85e-06) + assert response.usage.cost == 5.85e-06 # type: ignore[attr-defined] + + @pytest.mark.parametrize("unpriced", ["", None, "None"]) + def test_an_unpriced_deployment_leaves_the_field_absent(self, unpriced): + """ + Absence is not zero. A deployment with no configured price must not serialize + as a free call, which is the failure mode that looks like a real result. + """ + response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16) + ProxyBaseLLMRequestProcessing._set_usage_cost(response, unpriced) + assert "cost" not in response.model_dump()["usage"] + + def test_a_real_zero_is_recorded(self): + """Unbilled non-inference calls cost 0.0, and that zero is an answer.""" + response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16) + ProxyBaseLLMRequestProcessing._set_usage_cost(response, 0.0) + assert response.model_dump()["usage"]["cost"] == 0.0 + + def test_untouched_usage_does_not_serialize_a_cost(self): + """The opted-out path has to be byte-identical, not merely null-valued.""" + response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16) + assert "cost" not in response.model_dump()["usage"] + + def test_a_response_without_usage_is_left_alone(self): + sentinel: Final = SimpleNamespace(usage=None) + ProxyBaseLLMRequestProcessing._set_usage_cost(sentinel, 5.85e-06) + assert sentinel.usage is None