diff --git a/litellm/__init__.py b/litellm/__init__.py index 5d10737e876..7bf950b05c9 100644 --- a/litellm/__init__.py +++ b/litellm/__init__.py @@ -369,6 +369,7 @@ banned_keywords_list: Optional[Union[str, List]] = None llm_guard_mode: Literal["all", "key-specific", "request-specific"] = "all" guardrail_name_config_map: Dict[str, GuardrailItem] = {} include_cost_in_streaming_usage: bool = False +include_cost_in_usage: bool = False reasoning_auto_summary: bool = False ### PROMPTS #### from litellm.types.prompts.init_prompts import PromptSpec diff --git a/litellm/constants.py b/litellm/constants.py index a292b654778..d45797b8267 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -1535,6 +1535,7 @@ PASSTHROUGH_UPSTREAM_ERROR_BODY_MAX_LOG_CHARS: Final = 4096 # Headers to control callbacks X_LITELLM_DISABLE_CALLBACKS: Final = "x-litellm-disable-callbacks" +X_LITELLM_INCLUDE_COST_IN_USAGE: Final = "x-litellm-include-cost-in-usage" LITELLM_METADATA_FIELD: Final = "litellm_metadata" OLD_LITELLM_METADATA_FIELD: Final = "metadata" RETURN_RAW_MODEL_NAME_METADATA_KEY: Final = "_complexity_router_return_raw_model_name" diff --git a/litellm/proxy/common_request_processing.py b/litellm/proxy/common_request_processing.py index 15610da9aec..b2d8174912e 100644 --- a/litellm/proxy/common_request_processing.py +++ b/litellm/proxy/common_request_processing.py @@ -45,6 +45,7 @@ from litellm.constants import ( STREAM_SSE_DATA_PREFIX, STREAM_SSE_KEEPALIVE_PING_BYTES, UNSAFE_PROXY_RESPONSE_HEADERS, + X_LITELLM_INCLUDE_COST_IN_USAGE, ) from litellm.integrations.custom_guardrail import CustomGuardrail from litellm.litellm_core_utils.bug_report import ( @@ -2912,6 +2913,8 @@ class ProxyBaseLLMRequestProcessing: else llm_cost_for_headers ) + self._maybe_set_usage_cost(request, response, response_cost_for_headers) + # Always return the client-requested model name (not provider-prefixed internal identifiers) # for OpenAI-compatible responses. if requested_model_from_client: @@ -4231,6 +4234,33 @@ class ProxyBaseLLMRequestProcessing: return cost_from_logging_obj return ProxyBaseLLMRequestProcessing._completion_cost_or_none(model_response, model_name, service_tier) + @staticmethod + def _maybe_set_usage_cost(request: Request, response: object, cost: float | str | None) -> None: + if ProxyBaseLLMRequestProcessing._should_include_cost_in_usage(request): + ProxyBaseLLMRequestProcessing._set_usage_cost(response, cost) + + @staticmethod + def _should_include_cost_in_usage(request: Request) -> bool: + header_value: Final = request.headers.get(X_LITELLM_INCLUDE_COST_IN_USAGE) + if header_value is not None and header_value.strip() != "": + return header_value.strip().lower() in ("true", "1", "yes") + return bool(getattr(litellm, "include_cost_in_usage", False)) + + @staticmethod + def _set_usage_cost(response: object, cost: float | str | None) -> None: + """ + usage.cost means what this gateway charged, so an upstream provider's own figure is + dropped rather than left behind: an unpriced deployment reports no cost at all, and a + caller reading the field should not get a number from whoever happened to serve it. + """ + usage: Final = getattr(response, "usage", None) + if not isinstance(usage, Usage): + return + if isinstance(cost, bool) or not isinstance(cost, (int, float)): + del usage.cost + return + usage.cost = float(cost) + @staticmethod def _inject_cost_into_usage_dict( obj: dict, model_name: str, litellm_logging_obj: LiteLLMLoggingObj | None = None diff --git a/tests/test_litellm/proxy/test_common_request_processing.py b/tests/test_litellm/proxy/test_common_request_processing.py index c17f41a8b8f..707e5ede3f7 100644 --- a/tests/test_litellm/proxy/test_common_request_processing.py +++ b/tests/test_litellm/proxy/test_common_request_processing.py @@ -10226,3 +10226,106 @@ class TestStreamingContainerOwnershipRecordedBeforeDone: assert tuple(chunk for chunk, _ in observed) == self.CHUNKS assert tuple(count for _, count in observed) == (0, 0, 0, 0) recorder.assert_awaited_once() + + +def _request_with_headers(**headers: str) -> Request: + """A Request carrying the given headers, built the way the proxy receives one.""" + encoded: Final = [(k.replace("_", "-").encode(), v.encode()) for k, v in headers.items()] + return Request({"type": "http", "method": "POST", "headers": encoded}) + + +def _response_with_usage(**usage_kwargs: object): + from litellm.types.utils import ModelResponse, Usage + + return ModelResponse(usage=Usage(**usage_kwargs)) + + +class TestIncludeCostInUsage: + """ + usage.cost carries the gateway's own figure, opt-in per request. + + The invariants that matter are that a caller who did not ask sees no change at + all, and that a caller who did ask can read the field without first working out + which deployment served the request. + """ + + def test_off_by_default(self): + assert ProxyBaseLLMRequestProcessing._should_include_cost_in_usage(_request_with_headers()) is False + + @pytest.mark.parametrize("value", ["true", "TRUE", "1", "yes", " true "]) + def test_header_opts_in(self, value): + request: Final = _request_with_headers(x_litellm_include_cost_in_usage=value) + assert ProxyBaseLLMRequestProcessing._should_include_cost_in_usage(request) is True + + @pytest.mark.parametrize("value", ["false", "0", "no", "anything-else"]) + def test_header_opts_out(self, value, monkeypatch): + monkeypatch.setattr(litellm, "include_cost_in_usage", True, raising=False) + request: Final = _request_with_headers(x_litellm_include_cost_in_usage=value) + assert ProxyBaseLLMRequestProcessing._should_include_cost_in_usage(request) is False + + def test_setting_applies_when_no_header_is_sent(self, monkeypatch): + monkeypatch.setattr(litellm, "include_cost_in_usage", True, raising=False) + assert ProxyBaseLLMRequestProcessing._should_include_cost_in_usage(_request_with_headers()) is True + + def test_cost_is_recorded_on_usage(self): + response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16) + ProxyBaseLLMRequestProcessing._set_usage_cost(response, 5.85e-06) + assert response.model_dump()["usage"]["cost"] == 5.85e-06 + + def test_a_provider_supplied_cost_is_replaced(self): + """ + OpenRouter reports its own cost under this name. It is a different number, + computed by a different party, so the gateway's figure has to win - otherwise + the meaning of the field would depend on which deployment served the request. + """ + response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16, cost=8.775e-06) + ProxyBaseLLMRequestProcessing._set_usage_cost(response, 5.85e-06) + assert response.model_dump()["usage"]["cost"] == 5.85e-06 + + @pytest.mark.parametrize("unpriced", ["", None, "None"]) + def test_an_unpriced_deployment_drops_a_provider_supplied_cost(self, unpriced): + """ + An upstream's own figure must not survive as the answer when the gateway has no + price of its own, or an opted-in caller reads a number the gateway never charged. + """ + response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16, cost=8.775e-06) + ProxyBaseLLMRequestProcessing._set_usage_cost(response, unpriced) + assert "cost" not in response.model_dump()["usage"] + + @pytest.mark.parametrize("unpriced", ["", None, "None"]) + def test_an_unpriced_deployment_leaves_the_field_absent(self, unpriced): + """ + Absence is not zero. A deployment with no configured price must not serialize + as a free call, which is the failure mode that looks like a real result. + """ + response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16) + ProxyBaseLLMRequestProcessing._set_usage_cost(response, unpriced) + assert "cost" not in response.model_dump()["usage"] + + def test_a_real_zero_is_recorded(self): + """Unbilled non-inference calls cost 0.0, and that zero is an answer.""" + response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16) + ProxyBaseLLMRequestProcessing._set_usage_cost(response, 0.0) + assert response.model_dump()["usage"]["cost"] == 0.0 + + def test_untouched_usage_does_not_serialize_a_cost(self): + """The opted-out path has to be byte-identical, not merely null-valued.""" + response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16) + assert "cost" not in response.model_dump()["usage"] + + def test_opted_in_request_gets_the_cost_recorded(self): + """The decision and the write, exercised together as the request path runs them.""" + response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16) + request: Final = _request_with_headers(x_litellm_include_cost_in_usage="true") + ProxyBaseLLMRequestProcessing._maybe_set_usage_cost(request, response, 5.85e-06) + assert response.model_dump()["usage"]["cost"] == 5.85e-06 + + def test_opted_out_request_is_left_untouched(self): + response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16) + ProxyBaseLLMRequestProcessing._maybe_set_usage_cost(_request_with_headers(), response, 5.85e-06) + assert "cost" not in response.model_dump()["usage"] + + def test_a_response_without_usage_is_left_alone(self): + sentinel: Final = SimpleNamespace(usage=None) + ProxyBaseLLMRequestProcessing._set_usage_cost(sentinel, 5.85e-06) + assert sentinel.usage is None