mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-03 02:22:24 +00:00
fix(cost): cover per-second pricing dimension in deployment custom rates
_model_has_known_pricing blocks registration for models whose cost-map entry has either per-token or per-second pricing, but the rate extraction only handled per-token rates, so a deployment's custom input_cost_per_second was silently ignored and the model's public rate billed instead. Extend the extraction to return per-second rates too (folding an optional output_cost_per_second into litellm's single custom_cost_per_second value, matching what a registered per-second entry would bill) and thread custom_cost_per_second through response_cost_calculator into completion_cost, which already accepts it Also isolate the unknown-model registration test with monkeypatch so it no longer leaks an open_ai_chat_completion_models entry into other tests
This commit is contained in:
parent
75a6c0ee31
commit
0b89c8f079
5 changed files with 82 additions and 36 deletions
|
|
@ -1857,6 +1857,7 @@ def response_cost_calculator(
|
|||
base_model: Optional[str] = None,
|
||||
custom_pricing: Optional[bool] = None,
|
||||
custom_cost_per_token: Optional[CostPerToken] = None,
|
||||
custom_cost_per_second: Optional[float] = None,
|
||||
prompt: str = "",
|
||||
standard_built_in_tools_params: Optional[StandardBuiltInToolsParams] = None,
|
||||
litellm_model_name: Optional[str] = None,
|
||||
|
|
@ -1895,6 +1896,7 @@ def response_cost_calculator(
|
|||
optional_params=optional_params,
|
||||
custom_pricing=custom_pricing,
|
||||
custom_cost_per_token=custom_cost_per_token,
|
||||
custom_cost_per_second=custom_cost_per_second,
|
||||
base_model=base_model,
|
||||
prompt=prompt,
|
||||
standard_built_in_tools_params=standard_built_in_tools_params,
|
||||
|
|
|
|||
|
|
@ -1525,8 +1525,12 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
# litellm.model_cost (model-cost-map reloads replace the map wholesale,
|
||||
# after which cost calc would fall back to the model's public price).
|
||||
custom_cost_per_token: Optional[CostPerToken] = None
|
||||
custom_cost_per_second: Optional[float] = None
|
||||
if custom_pricing is True:
|
||||
custom_cost_per_token = get_custom_cost_per_token_from_litellm_params(
|
||||
(
|
||||
custom_cost_per_token,
|
||||
custom_cost_per_second,
|
||||
) = get_custom_cost_from_litellm_params(
|
||||
litellm_params=(
|
||||
self.litellm_params if hasattr(self, "litellm_params") else None
|
||||
)
|
||||
|
|
@ -1555,6 +1559,7 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
"optional_params": self.optional_params,
|
||||
"custom_pricing": custom_pricing,
|
||||
"custom_cost_per_token": custom_cost_per_token,
|
||||
"custom_cost_per_second": custom_cost_per_second,
|
||||
"prompt": prompt,
|
||||
"standard_built_in_tools_params": self.standard_built_in_tools_params,
|
||||
"router_model_id": router_model_id,
|
||||
|
|
@ -4788,17 +4793,21 @@ def use_custom_pricing_for_model(litellm_params: Optional[dict]) -> bool:
|
|||
return False
|
||||
|
||||
|
||||
def get_custom_cost_per_token_from_litellm_params(
|
||||
def get_custom_cost_from_litellm_params(
|
||||
litellm_params: Optional[dict],
|
||||
) -> Optional[CostPerToken]:
|
||||
) -> Tuple[Optional[CostPerToken], Optional[float]]:
|
||||
"""
|
||||
Extract explicit per-token rates from litellm_params or its
|
||||
Extract explicit (per-token, per-second) rates from litellm_params or its
|
||||
metadata/litellm_metadata model_info, mirroring the lookup order of
|
||||
`use_custom_pricing_for_model`. A literal 0 is a valid rate (e.g. zero-cost
|
||||
BYOK deployments). Returns None unless both base rates are set.
|
||||
BYOK deployments). Per-token pricing requires both base rates. Per-second
|
||||
pricing requires input_cost_per_second (the same rule pricing registration
|
||||
uses in main.py); an output_cost_per_second is folded into litellm's
|
||||
single custom_cost_per_second rate so the billed total matches a
|
||||
registered per-second cost-map entry.
|
||||
"""
|
||||
if litellm_params is None:
|
||||
return None
|
||||
return None, None
|
||||
|
||||
sources = [litellm_params]
|
||||
for metadata_key in ("metadata", "litellm_metadata"):
|
||||
|
|
@ -4810,21 +4819,25 @@ def get_custom_cost_per_token_from_litellm_params(
|
|||
for source in sources:
|
||||
input_cost = source.get("input_cost_per_token")
|
||||
output_cost = source.get("output_cost_per_token")
|
||||
if input_cost is None or output_cost is None:
|
||||
continue
|
||||
custom_cost: CostPerToken = {
|
||||
"input_cost_per_token": input_cost,
|
||||
"output_cost_per_token": output_cost,
|
||||
}
|
||||
for cache_key in (
|
||||
"cache_read_input_token_cost",
|
||||
"cache_creation_input_token_cost",
|
||||
):
|
||||
if source.get(cache_key) is not None:
|
||||
custom_cost[cache_key] = source[cache_key] # type: ignore[literal-required]
|
||||
return custom_cost
|
||||
if input_cost is not None and output_cost is not None:
|
||||
custom_cost: CostPerToken = {
|
||||
"input_cost_per_token": input_cost,
|
||||
"output_cost_per_token": output_cost,
|
||||
}
|
||||
for cache_key in (
|
||||
"cache_read_input_token_cost",
|
||||
"cache_creation_input_token_cost",
|
||||
):
|
||||
if source.get(cache_key) is not None:
|
||||
custom_cost[cache_key] = source[cache_key] # type: ignore[literal-required]
|
||||
return custom_cost, None
|
||||
input_cost_per_second = source.get("input_cost_per_second")
|
||||
if input_cost_per_second is not None:
|
||||
return None, input_cost_per_second + (
|
||||
source.get("output_cost_per_second") or 0.0
|
||||
)
|
||||
|
||||
return None
|
||||
return None, None
|
||||
|
||||
|
||||
def is_valid_sha256_hash(value: str) -> bool:
|
||||
|
|
|
|||
|
|
@ -1092,8 +1092,10 @@ def _model_has_known_pricing(model: str, custom_llm_provider: str) -> bool:
|
|||
shared entry and re-prices every other request for that model in the
|
||||
process (e.g. a zero-cost BYOK deployment zeroing the real model's
|
||||
billing). Known models get their per-request rates via
|
||||
custom_cost_per_token at cost-calculation time instead; registration is
|
||||
only needed so unknown models can be priced at all.
|
||||
custom_cost_per_token / custom_cost_per_second at cost-calculation time
|
||||
instead (see get_custom_cost_from_litellm_params, which covers the same
|
||||
pricing dimensions checked here); registration is only needed so unknown
|
||||
models can be priced at all.
|
||||
"""
|
||||
for key in (model, f"{custom_llm_provider}/{model}"):
|
||||
entry = litellm.model_cost.get(key)
|
||||
|
|
|
|||
|
|
@ -2639,3 +2639,27 @@ def test_response_cost_calculator_custom_pricing_survives_model_cost_reload(
|
|||
result=_response_with_usage("claude-opus-4-6")
|
||||
)
|
||||
assert cost_after_reload == pytest.approx(expected)
|
||||
|
||||
|
||||
def test_response_cost_calculator_uses_deployment_per_second_rates():
|
||||
"""A deployment priced per second must be billed at its own rates, not the
|
||||
concrete model's public per-token price (same custom-pricing path as the
|
||||
per-token tests above, for the other pricing dimension)."""
|
||||
logging_obj = _logging_obj_with_custom_pricing(
|
||||
litellm_params={
|
||||
"metadata": {
|
||||
"model_info": {
|
||||
"id": "per-second-deployment-id-not-in-model-cost",
|
||||
"input_cost_per_second": 0.001,
|
||||
"output_cost_per_second": 0.002,
|
||||
}
|
||||
}
|
||||
},
|
||||
model="anthropic/claude-opus-4-6",
|
||||
)
|
||||
response = _response_with_usage("claude-opus-4-6")
|
||||
response._response_ms = 5_000
|
||||
|
||||
cost = logging_obj._response_cost_calculator(result=response)
|
||||
|
||||
assert cost == pytest.approx((0.001 + 0.002) * 5)
|
||||
|
|
|
|||
|
|
@ -1988,19 +1988,24 @@ def test_completion_custom_pricing_does_not_overwrite_canonical_model_cost(
|
|||
assert response._hidden_params["response_cost"] == 0.0
|
||||
|
||||
|
||||
def test_completion_custom_pricing_still_registers_unknown_model():
|
||||
def test_completion_custom_pricing_still_registers_unknown_model(monkeypatch):
|
||||
model = "openai/unknown-custom-priced-model-xyz"
|
||||
monkeypatch.setattr(litellm, "model_cost", dict(litellm.model_cost))
|
||||
monkeypatch.setattr(
|
||||
litellm,
|
||||
"open_ai_chat_completion_models",
|
||||
set(litellm.open_ai_chat_completion_models),
|
||||
)
|
||||
litellm.model_cost.pop(model, None)
|
||||
try:
|
||||
litellm.completion(
|
||||
model=model,
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
mock_response="ok",
|
||||
api_key="sk-test",
|
||||
input_cost_per_token=1e-07,
|
||||
output_cost_per_token=2e-07,
|
||||
)
|
||||
assert litellm.model_cost[model]["input_cost_per_token"] == 1e-07
|
||||
assert litellm.model_cost[model]["output_cost_per_token"] == 2e-07
|
||||
finally:
|
||||
litellm.model_cost.pop(model, None)
|
||||
|
||||
litellm.completion(
|
||||
model=model,
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
mock_response="ok",
|
||||
api_key="sk-test",
|
||||
input_cost_per_token=1e-07,
|
||||
output_cost_per_token=2e-07,
|
||||
)
|
||||
|
||||
assert litellm.model_cost[model]["input_cost_per_token"] == 1e-07
|
||||
assert litellm.model_cost[model]["output_cost_per_token"] == 2e-07
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue