fix(cost): cover per-second pricing dimension in deployment custom rates

_model_has_known_pricing blocks registration for models whose cost-map
entry has either per-token or per-second pricing, but the rate
extraction only handled per-token rates, so a deployment's custom
input_cost_per_second was silently ignored and the model's public rate
billed instead. Extend the extraction to return per-second rates too
(folding an optional output_cost_per_second into litellm's single
custom_cost_per_second value, matching what a registered per-second
entry would bill) and thread custom_cost_per_second through
response_cost_calculator into completion_cost, which already accepts it

Also isolate the unknown-model registration test with monkeypatch so it
no longer leaks an open_ai_chat_completion_models entry into other tests
This commit is contained in:
Filippo Mattia Menghi 2026-06-10 10:24:06 +02:00
parent 75a6c0ee31
commit 0b89c8f079
5 changed files with 82 additions and 36 deletions

View file

@ -1857,6 +1857,7 @@ def response_cost_calculator(
base_model: Optional[str] = None,
custom_pricing: Optional[bool] = None,
custom_cost_per_token: Optional[CostPerToken] = None,
custom_cost_per_second: Optional[float] = None,
prompt: str = "",
standard_built_in_tools_params: Optional[StandardBuiltInToolsParams] = None,
litellm_model_name: Optional[str] = None,
@ -1895,6 +1896,7 @@ def response_cost_calculator(
optional_params=optional_params,
custom_pricing=custom_pricing,
custom_cost_per_token=custom_cost_per_token,
custom_cost_per_second=custom_cost_per_second,
base_model=base_model,
prompt=prompt,
standard_built_in_tools_params=standard_built_in_tools_params,

View file

@ -1525,8 +1525,12 @@ class Logging(LiteLLMLoggingBaseClass):
# litellm.model_cost (model-cost-map reloads replace the map wholesale,
# after which cost calc would fall back to the model's public price).
custom_cost_per_token: Optional[CostPerToken] = None
custom_cost_per_second: Optional[float] = None
if custom_pricing is True:
custom_cost_per_token = get_custom_cost_per_token_from_litellm_params(
(
custom_cost_per_token,
custom_cost_per_second,
) = get_custom_cost_from_litellm_params(
litellm_params=(
self.litellm_params if hasattr(self, "litellm_params") else None
)
@ -1555,6 +1559,7 @@ class Logging(LiteLLMLoggingBaseClass):
"optional_params": self.optional_params,
"custom_pricing": custom_pricing,
"custom_cost_per_token": custom_cost_per_token,
"custom_cost_per_second": custom_cost_per_second,
"prompt": prompt,
"standard_built_in_tools_params": self.standard_built_in_tools_params,
"router_model_id": router_model_id,
@ -4788,17 +4793,21 @@ def use_custom_pricing_for_model(litellm_params: Optional[dict]) -> bool:
return False
def get_custom_cost_per_token_from_litellm_params(
def get_custom_cost_from_litellm_params(
litellm_params: Optional[dict],
) -> Optional[CostPerToken]:
) -> Tuple[Optional[CostPerToken], Optional[float]]:
"""
Extract explicit per-token rates from litellm_params or its
Extract explicit (per-token, per-second) rates from litellm_params or its
metadata/litellm_metadata model_info, mirroring the lookup order of
`use_custom_pricing_for_model`. A literal 0 is a valid rate (e.g. zero-cost
BYOK deployments). Returns None unless both base rates are set.
BYOK deployments). Per-token pricing requires both base rates. Per-second
pricing requires input_cost_per_second (the same rule pricing registration
uses in main.py); an output_cost_per_second is folded into litellm's
single custom_cost_per_second rate so the billed total matches a
registered per-second cost-map entry.
"""
if litellm_params is None:
return None
return None, None
sources = [litellm_params]
for metadata_key in ("metadata", "litellm_metadata"):
@ -4810,21 +4819,25 @@ def get_custom_cost_per_token_from_litellm_params(
for source in sources:
input_cost = source.get("input_cost_per_token")
output_cost = source.get("output_cost_per_token")
if input_cost is None or output_cost is None:
continue
custom_cost: CostPerToken = {
"input_cost_per_token": input_cost,
"output_cost_per_token": output_cost,
}
for cache_key in (
"cache_read_input_token_cost",
"cache_creation_input_token_cost",
):
if source.get(cache_key) is not None:
custom_cost[cache_key] = source[cache_key] # type: ignore[literal-required]
return custom_cost
if input_cost is not None and output_cost is not None:
custom_cost: CostPerToken = {
"input_cost_per_token": input_cost,
"output_cost_per_token": output_cost,
}
for cache_key in (
"cache_read_input_token_cost",
"cache_creation_input_token_cost",
):
if source.get(cache_key) is not None:
custom_cost[cache_key] = source[cache_key] # type: ignore[literal-required]
return custom_cost, None
input_cost_per_second = source.get("input_cost_per_second")
if input_cost_per_second is not None:
return None, input_cost_per_second + (
source.get("output_cost_per_second") or 0.0
)
return None
return None, None
def is_valid_sha256_hash(value: str) -> bool:

View file

@ -1092,8 +1092,10 @@ def _model_has_known_pricing(model: str, custom_llm_provider: str) -> bool:
shared entry and re-prices every other request for that model in the
process (e.g. a zero-cost BYOK deployment zeroing the real model's
billing). Known models get their per-request rates via
custom_cost_per_token at cost-calculation time instead; registration is
only needed so unknown models can be priced at all.
custom_cost_per_token / custom_cost_per_second at cost-calculation time
instead (see get_custom_cost_from_litellm_params, which covers the same
pricing dimensions checked here); registration is only needed so unknown
models can be priced at all.
"""
for key in (model, f"{custom_llm_provider}/{model}"):
entry = litellm.model_cost.get(key)

View file

@ -2639,3 +2639,27 @@ def test_response_cost_calculator_custom_pricing_survives_model_cost_reload(
result=_response_with_usage("claude-opus-4-6")
)
assert cost_after_reload == pytest.approx(expected)
def test_response_cost_calculator_uses_deployment_per_second_rates():
"""A deployment priced per second must be billed at its own rates, not the
concrete model's public per-token price (same custom-pricing path as the
per-token tests above, for the other pricing dimension)."""
logging_obj = _logging_obj_with_custom_pricing(
litellm_params={
"metadata": {
"model_info": {
"id": "per-second-deployment-id-not-in-model-cost",
"input_cost_per_second": 0.001,
"output_cost_per_second": 0.002,
}
}
},
model="anthropic/claude-opus-4-6",
)
response = _response_with_usage("claude-opus-4-6")
response._response_ms = 5_000
cost = logging_obj._response_cost_calculator(result=response)
assert cost == pytest.approx((0.001 + 0.002) * 5)

View file

@ -1988,19 +1988,24 @@ def test_completion_custom_pricing_does_not_overwrite_canonical_model_cost(
assert response._hidden_params["response_cost"] == 0.0
def test_completion_custom_pricing_still_registers_unknown_model():
def test_completion_custom_pricing_still_registers_unknown_model(monkeypatch):
model = "openai/unknown-custom-priced-model-xyz"
monkeypatch.setattr(litellm, "model_cost", dict(litellm.model_cost))
monkeypatch.setattr(
litellm,
"open_ai_chat_completion_models",
set(litellm.open_ai_chat_completion_models),
)
litellm.model_cost.pop(model, None)
try:
litellm.completion(
model=model,
messages=[{"role": "user", "content": "hi"}],
mock_response="ok",
api_key="sk-test",
input_cost_per_token=1e-07,
output_cost_per_token=2e-07,
)
assert litellm.model_cost[model]["input_cost_per_token"] == 1e-07
assert litellm.model_cost[model]["output_cost_per_token"] == 2e-07
finally:
litellm.model_cost.pop(model, None)
litellm.completion(
model=model,
messages=[{"role": "user", "content": "hi"}],
mock_response="ok",
api_key="sk-test",
input_cost_per_token=1e-07,
output_cost_per_token=2e-07,
)
assert litellm.model_cost[model]["input_cost_per_token"] == 1e-07
assert litellm.model_cost[model]["output_cost_per_token"] == 2e-07