mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-03 02:22:24 +00:00
Merge 6e670e4a82 into ff462f7a77
This commit is contained in:
commit
9a0867715c
4 changed files with 135 additions and 0 deletions
|
|
@ -369,6 +369,7 @@ banned_keywords_list: Optional[Union[str, List]] = None
|
|||
llm_guard_mode: Literal["all", "key-specific", "request-specific"] = "all"
|
||||
guardrail_name_config_map: Dict[str, GuardrailItem] = {}
|
||||
include_cost_in_streaming_usage: bool = False
|
||||
include_cost_in_usage: bool = False
|
||||
reasoning_auto_summary: bool = False
|
||||
### PROMPTS ####
|
||||
from litellm.types.prompts.init_prompts import PromptSpec
|
||||
|
|
|
|||
|
|
@ -1535,6 +1535,7 @@ PASSTHROUGH_UPSTREAM_ERROR_BODY_MAX_LOG_CHARS: Final = 4096
|
|||
|
||||
# Headers to control callbacks
|
||||
X_LITELLM_DISABLE_CALLBACKS: Final = "x-litellm-disable-callbacks"
|
||||
X_LITELLM_INCLUDE_COST_IN_USAGE: Final = "x-litellm-include-cost-in-usage"
|
||||
LITELLM_METADATA_FIELD: Final = "litellm_metadata"
|
||||
OLD_LITELLM_METADATA_FIELD: Final = "metadata"
|
||||
RETURN_RAW_MODEL_NAME_METADATA_KEY: Final = "_complexity_router_return_raw_model_name"
|
||||
|
|
|
|||
|
|
@ -45,6 +45,7 @@ from litellm.constants import (
|
|||
STREAM_SSE_DATA_PREFIX,
|
||||
STREAM_SSE_KEEPALIVE_PING_BYTES,
|
||||
UNSAFE_PROXY_RESPONSE_HEADERS,
|
||||
X_LITELLM_INCLUDE_COST_IN_USAGE,
|
||||
)
|
||||
from litellm.integrations.custom_guardrail import CustomGuardrail
|
||||
from litellm.litellm_core_utils.bug_report import (
|
||||
|
|
@ -2912,6 +2913,8 @@ class ProxyBaseLLMRequestProcessing:
|
|||
else llm_cost_for_headers
|
||||
)
|
||||
|
||||
self._maybe_set_usage_cost(request, response, response_cost_for_headers)
|
||||
|
||||
# Always return the client-requested model name (not provider-prefixed internal identifiers)
|
||||
# for OpenAI-compatible responses.
|
||||
if requested_model_from_client:
|
||||
|
|
@ -4231,6 +4234,33 @@ class ProxyBaseLLMRequestProcessing:
|
|||
return cost_from_logging_obj
|
||||
return ProxyBaseLLMRequestProcessing._completion_cost_or_none(model_response, model_name, service_tier)
|
||||
|
||||
@staticmethod
|
||||
def _maybe_set_usage_cost(request: Request, response: object, cost: float | str | None) -> None:
|
||||
if ProxyBaseLLMRequestProcessing._should_include_cost_in_usage(request):
|
||||
ProxyBaseLLMRequestProcessing._set_usage_cost(response, cost)
|
||||
|
||||
@staticmethod
|
||||
def _should_include_cost_in_usage(request: Request) -> bool:
|
||||
header_value: Final = request.headers.get(X_LITELLM_INCLUDE_COST_IN_USAGE)
|
||||
if header_value is not None and header_value.strip() != "":
|
||||
return header_value.strip().lower() in ("true", "1", "yes")
|
||||
return bool(getattr(litellm, "include_cost_in_usage", False))
|
||||
|
||||
@staticmethod
|
||||
def _set_usage_cost(response: object, cost: float | str | None) -> None:
|
||||
"""
|
||||
usage.cost means what this gateway charged, so an upstream provider's own figure is
|
||||
dropped rather than left behind: an unpriced deployment reports no cost at all, and a
|
||||
caller reading the field should not get a number from whoever happened to serve it.
|
||||
"""
|
||||
usage: Final = getattr(response, "usage", None)
|
||||
if not isinstance(usage, Usage):
|
||||
return
|
||||
if isinstance(cost, bool) or not isinstance(cost, (int, float)):
|
||||
del usage.cost
|
||||
return
|
||||
usage.cost = float(cost)
|
||||
|
||||
@staticmethod
|
||||
def _inject_cost_into_usage_dict(
|
||||
obj: dict, model_name: str, litellm_logging_obj: LiteLLMLoggingObj | None = None
|
||||
|
|
|
|||
|
|
@ -10226,3 +10226,106 @@ class TestStreamingContainerOwnershipRecordedBeforeDone:
|
|||
assert tuple(chunk for chunk, _ in observed) == self.CHUNKS
|
||||
assert tuple(count for _, count in observed) == (0, 0, 0, 0)
|
||||
recorder.assert_awaited_once()
|
||||
|
||||
|
||||
def _request_with_headers(**headers: str) -> Request:
|
||||
"""A Request carrying the given headers, built the way the proxy receives one."""
|
||||
encoded: Final = [(k.replace("_", "-").encode(), v.encode()) for k, v in headers.items()]
|
||||
return Request({"type": "http", "method": "POST", "headers": encoded})
|
||||
|
||||
|
||||
def _response_with_usage(**usage_kwargs: object):
|
||||
from litellm.types.utils import ModelResponse, Usage
|
||||
|
||||
return ModelResponse(usage=Usage(**usage_kwargs))
|
||||
|
||||
|
||||
class TestIncludeCostInUsage:
|
||||
"""
|
||||
usage.cost carries the gateway's own figure, opt-in per request.
|
||||
|
||||
The invariants that matter are that a caller who did not ask sees no change at
|
||||
all, and that a caller who did ask can read the field without first working out
|
||||
which deployment served the request.
|
||||
"""
|
||||
|
||||
def test_off_by_default(self):
|
||||
assert ProxyBaseLLMRequestProcessing._should_include_cost_in_usage(_request_with_headers()) is False
|
||||
|
||||
@pytest.mark.parametrize("value", ["true", "TRUE", "1", "yes", " true "])
|
||||
def test_header_opts_in(self, value):
|
||||
request: Final = _request_with_headers(x_litellm_include_cost_in_usage=value)
|
||||
assert ProxyBaseLLMRequestProcessing._should_include_cost_in_usage(request) is True
|
||||
|
||||
@pytest.mark.parametrize("value", ["false", "0", "no", "anything-else"])
|
||||
def test_header_opts_out(self, value, monkeypatch):
|
||||
monkeypatch.setattr(litellm, "include_cost_in_usage", True, raising=False)
|
||||
request: Final = _request_with_headers(x_litellm_include_cost_in_usage=value)
|
||||
assert ProxyBaseLLMRequestProcessing._should_include_cost_in_usage(request) is False
|
||||
|
||||
def test_setting_applies_when_no_header_is_sent(self, monkeypatch):
|
||||
monkeypatch.setattr(litellm, "include_cost_in_usage", True, raising=False)
|
||||
assert ProxyBaseLLMRequestProcessing._should_include_cost_in_usage(_request_with_headers()) is True
|
||||
|
||||
def test_cost_is_recorded_on_usage(self):
|
||||
response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16)
|
||||
ProxyBaseLLMRequestProcessing._set_usage_cost(response, 5.85e-06)
|
||||
assert response.model_dump()["usage"]["cost"] == 5.85e-06
|
||||
|
||||
def test_a_provider_supplied_cost_is_replaced(self):
|
||||
"""
|
||||
OpenRouter reports its own cost under this name. It is a different number,
|
||||
computed by a different party, so the gateway's figure has to win - otherwise
|
||||
the meaning of the field would depend on which deployment served the request.
|
||||
"""
|
||||
response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16, cost=8.775e-06)
|
||||
ProxyBaseLLMRequestProcessing._set_usage_cost(response, 5.85e-06)
|
||||
assert response.model_dump()["usage"]["cost"] == 5.85e-06
|
||||
|
||||
@pytest.mark.parametrize("unpriced", ["", None, "None"])
|
||||
def test_an_unpriced_deployment_drops_a_provider_supplied_cost(self, unpriced):
|
||||
"""
|
||||
An upstream's own figure must not survive as the answer when the gateway has no
|
||||
price of its own, or an opted-in caller reads a number the gateway never charged.
|
||||
"""
|
||||
response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16, cost=8.775e-06)
|
||||
ProxyBaseLLMRequestProcessing._set_usage_cost(response, unpriced)
|
||||
assert "cost" not in response.model_dump()["usage"]
|
||||
|
||||
@pytest.mark.parametrize("unpriced", ["", None, "None"])
|
||||
def test_an_unpriced_deployment_leaves_the_field_absent(self, unpriced):
|
||||
"""
|
||||
Absence is not zero. A deployment with no configured price must not serialize
|
||||
as a free call, which is the failure mode that looks like a real result.
|
||||
"""
|
||||
response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16)
|
||||
ProxyBaseLLMRequestProcessing._set_usage_cost(response, unpriced)
|
||||
assert "cost" not in response.model_dump()["usage"]
|
||||
|
||||
def test_a_real_zero_is_recorded(self):
|
||||
"""Unbilled non-inference calls cost 0.0, and that zero is an answer."""
|
||||
response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16)
|
||||
ProxyBaseLLMRequestProcessing._set_usage_cost(response, 0.0)
|
||||
assert response.model_dump()["usage"]["cost"] == 0.0
|
||||
|
||||
def test_untouched_usage_does_not_serialize_a_cost(self):
|
||||
"""The opted-out path has to be byte-identical, not merely null-valued."""
|
||||
response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16)
|
||||
assert "cost" not in response.model_dump()["usage"]
|
||||
|
||||
def test_opted_in_request_gets_the_cost_recorded(self):
|
||||
"""The decision and the write, exercised together as the request path runs them."""
|
||||
response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16)
|
||||
request: Final = _request_with_headers(x_litellm_include_cost_in_usage="true")
|
||||
ProxyBaseLLMRequestProcessing._maybe_set_usage_cost(request, response, 5.85e-06)
|
||||
assert response.model_dump()["usage"]["cost"] == 5.85e-06
|
||||
|
||||
def test_opted_out_request_is_left_untouched(self):
|
||||
response: Final = _response_with_usage(prompt_tokens=11, completion_tokens=5, total_tokens=16)
|
||||
ProxyBaseLLMRequestProcessing._maybe_set_usage_cost(_request_with_headers(), response, 5.85e-06)
|
||||
assert "cost" not in response.model_dump()["usage"]
|
||||
|
||||
def test_a_response_without_usage_is_left_alone(self):
|
||||
sentinel: Final = SimpleNamespace(usage=None)
|
||||
ProxyBaseLLMRequestProcessing._set_usage_cost(sentinel, 5.85e-06)
|
||||
assert sentinel.usage is None
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue