This commit is contained in:
Ben Langfeld 2026-09-27 12:57:18 -03:00 • committed by GitHub
commit 9a0867715c
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
4 changed files with 135 additions and 0 deletions

View file

@ -369,6 +369,7 @@ banned_keywords_list: Optional[Union[str, List]] = None
llm_guard_mode: Literal["all", "key-specific", "request-specific"] = "all"
guardrail_name_config_map: Dict[str, GuardrailItem] = {}
include_cost_in_streaming_usage: bool = False
include_cost_in_usage: bool = False
reasoning_auto_summary: bool = False
### PROMPTS ####
from litellm.types.prompts.init_prompts import PromptSpec

View file

@ -1535,6 +1535,7 @@ PASSTHROUGH_UPSTREAM_ERROR_BODY_MAX_LOG_CHARS: Final = 4096
# Headers to control callbacks
X_LITELLM_DISABLE_CALLBACKS: Final = "x-litellm-disable-callbacks"
X_LITELLM_INCLUDE_COST_IN_USAGE: Final = "x-litellm-include-cost-in-usage"
LITELLM_METADATA_FIELD: Final = "litellm_metadata"
OLD_LITELLM_METADATA_FIELD: Final = "metadata"
RETURN_RAW_MODEL_NAME_METADATA_KEY: Final = "_complexity_router_return_raw_model_name"

View file

@ -45,6 +45,7 @@ from litellm.constants import (
STREAM_SSE_DATA_PREFIX,
STREAM_SSE_KEEPALIVE_PING_BYTES,
UNSAFE_PROXY_RESPONSE_HEADERS,
X_LITELLM_INCLUDE_COST_IN_USAGE,
)
from litellm.integrations.custom_guardrail import CustomGuardrail
from litellm.litellm_core_utils.bug_report import (
@ -2912,6 +2913,8 @@ class ProxyBaseLLMRequestProcessing:
else llm_cost_for_headers
)
self._maybe_set_usage_cost(request, response, response_cost_for_headers)
# Always return the client-requested model name (not provider-prefixed internal identifiers)
# for OpenAI-compatible responses.
if requested_model_from_client:
@ -4231,6 +4234,33 @@ class ProxyBaseLLMRequestProcessing:
return cost_from_logging_obj
return ProxyBaseLLMRequestProcessing._completion_cost_or_none(model_response, model_name, service_tier)
@staticmethod
def _maybe_set_usage_cost(request: Request, response: object, cost: float | str | None) -> None:
if ProxyBaseLLMRequestProcessing._should_include_cost_in_usage(request):
ProxyBaseLLMRequestProcessing._set_usage_cost(response, cost)
@staticmethod
def _should_include_cost_in_usage(request: Request) -> bool:
header_value: Final = request.headers.get(X_LITELLM_INCLUDE_COST_IN_USAGE)
if header_value is not None and header_value.strip() != "":
return header_value.strip().lower() in ("true", "1", "yes")
return bool(getattr(litellm, "include_cost_in_usage", False))
@staticmethod
def _set_usage_cost(response: object, cost: float | str | None) -> None:
"""
usage.cost means what this gateway charged, so an upstream provider's own figure is
dropped rather than left behind: an unpriced deployment reports no cost at all, and a
caller reading the field should not get a number from whoever happened to serve it.
"""
usage: Final = getattr(response, "usage", None)
if not isinstance(usage, Usage):
return
if isinstance(cost, bool) or not isinstance(cost, (int, float)):
del usage.cost
return
usage.cost = float(cost)
@staticmethod
def _inject_cost_into_usage_dict(
obj: dict, model_name: str, litellm_logging_obj: LiteLLMLoggingObj | None = None

View file

@ -10226,3 +10226,106 @@ class TestStreamingContainerOwnershipRecordedBeforeDone:
assert tuple(chunk for chunk, _ in observed) == self.CHUNKS
assert tuple(count for _, count in observed) == (0, 0, 0, 0)
recorder.assert_awaited_once()
def _request_with_headers(**headers: str) -> Request:
"""A Request carrying the given headers, built the way the proxy receives one."""
encoded: Final = [(k.replace("_", "-").encode(), v.encode()) for k, v in headers.items()]
return Request({"type": "http", "method": "POST", "headers": encoded})
def _response_with_usage(**usage_kwargs: object):
from litellm.types.utils import ModelResponse, Usage
return ModelResponse(usage=Usage(**usage_kwargs))
class TestIncludeCostInUsage:
"""
usage.cost carries the gateway's own figure, opt-in per request.
The invariants that matter are that a caller who did not ask sees no change at
all, and that a caller who did ask can read the field without first working out
which deployment served the request.
"""
def test_off_by_default(self):
assert ProxyBaseLLMRequestProcessing._should_include_cost_in_usage(_request_with_headers()) is False
@pytest.mark.parametrize("value", ["true", "TRUE", "1", "yes", " true "])
def test_header_opts_in(self, value):
request: Final = _request_with_headers(x_litellm_include_cost_in_usage=value)
assert ProxyBaseLLMRequestProcessing._should_include_cost_in_usage(request) is True
@pytest.mark.parametrize("value", ["false", "0", "no", "anything-else"])
def test_header_opts_out(self, value, monkeypatch):
monkeypatch.setattr(litellm, "include_cost_in_usage", True, raising=False)
request: Final = _request_with_headers(x_litellm_include_cost_in_usage=value)
assert ProxyBaseLLMRequestProcessing._should_include_cost_in_usage(request) is False
def test_setting_applies_when_no_header_is_sent(self, monkeypatch):
monkeypatch.setattr(litellm, "include_cost_in_usage", True, raising=False)
assert ProxyBaseLLMRequestProcessing._should_include_cost_in_usage(_request_with_headers()) is True
def test_cost_is_recorded_on_usage(self):
response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16)
ProxyBaseLLMRequestProcessing._set_usage_cost(response, 5.85e-06)
assert response.model_dump()["usage"]["cost"] == 5.85e-06
def test_a_provider_supplied_cost_is_replaced(self):
"""
OpenRouter reports its own cost under this name. It is a different number,
computed by a different party, so the gateway's figure has to win - otherwise
the meaning of the field would depend on which deployment served the request.
"""
response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16, cost=8.775e-06)
ProxyBaseLLMRequestProcessing._set_usage_cost(response, 5.85e-06)
assert response.model_dump()["usage"]["cost"] == 5.85e-06
@pytest.mark.parametrize("unpriced", ["", None, "None"])
def test_an_unpriced_deployment_drops_a_provider_supplied_cost(self, unpriced):
"""
An upstream's own figure must not survive as the answer when the gateway has no
price of its own, or an opted-in caller reads a number the gateway never charged.
"""
response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16, cost=8.775e-06)
ProxyBaseLLMRequestProcessing._set_usage_cost(response, unpriced)
assert "cost" not in response.model_dump()["usage"]
@pytest.mark.parametrize("unpriced", ["", None, "None"])
def test_an_unpriced_deployment_leaves_the_field_absent(self, unpriced):
"""
Absence is not zero. A deployment with no configured price must not serialize
as a free call, which is the failure mode that looks like a real result.
"""
response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16)
ProxyBaseLLMRequestProcessing._set_usage_cost(response, unpriced)
assert "cost" not in response.model_dump()["usage"]
def test_a_real_zero_is_recorded(self):
"""Unbilled non-inference calls cost 0.0, and that zero is an answer."""
response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16)
ProxyBaseLLMRequestProcessing._set_usage_cost(response, 0.0)
assert response.model_dump()["usage"]["cost"] == 0.0
def test_untouched_usage_does_not_serialize_a_cost(self):
"""The opted-out path has to be byte-identical, not merely null-valued."""
response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16)
assert "cost" not in response.model_dump()["usage"]
def test_opted_in_request_gets_the_cost_recorded(self):
"""The decision and the write, exercised together as the request path runs them."""
response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16)
request: Final = _request_with_headers(x_litellm_include_cost_in_usage="true")
ProxyBaseLLMRequestProcessing._maybe_set_usage_cost(request, response, 5.85e-06)
assert response.model_dump()["usage"]["cost"] == 5.85e-06
def test_opted_out_request_is_left_untouched(self):
response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16)
ProxyBaseLLMRequestProcessing._maybe_set_usage_cost(_request_with_headers(), response, 5.85e-06)
assert "cost" not in response.model_dump()["usage"]
def test_a_response_without_usage_is_left_alone(self):
sentinel: Final = SimpleNamespace(usage=None)
ProxyBaseLLMRequestProcessing._set_usage_cost(sentinel, 5.85e-06)
assert sentinel.usage is None