This commit is contained in:
Eduardo Pessin 2026-10-03 16:26:55 -04:00 • committed by GitHub
commit bd1fbc74e2
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
2 changed files with 105 additions and 1 deletions

View file

@ -812,6 +812,33 @@ def _cost_map_entry_prices_anything(entry: Mapping[str, object]) -> bool:
)
def _has_rate(model: str, custom_llm_provider: str | None) -> bool:
# A registered entry is a real answer even when it prices at zero; an unregistered
# model is not, because `get_model_info` invents zero rates for it rather than
# raising. Hence truthiness here, not `is not None`.
if model in litellm.model_cost:
return True
try:
info: Final = litellm.get_model_info(model=model, custom_llm_provider=custom_llm_provider)
except Exception: # noqa: BLE001 # get_model_info raises bare Exception for some unmapped models
return False
return any(info.get(key) for key in ("input_cost_per_token", "output_cost_per_token", "input_cost_per_second"))
def _priced_provider_response_model(
provider_response_model: str | None,
completion_response_model: str | None,
custom_llm_provider: str | None,
) -> str | None:
if provider_response_model is None:
return None
if _has_rate(provider_response_model, custom_llm_provider):
return provider_response_model
if completion_response_model is not None and _has_rate(completion_response_model, custom_llm_provider):
return completion_response_model
return provider_response_model
def _select_model_name_for_cost_calc(
model: str | None,
completion_response: object | None,
@ -858,7 +885,15 @@ def _select_model_name_for_cost_calc(
return_model = model
elif base_model is not None or provider_response_model is not None:
return_model = base_model if base_model is not None else provider_response_model
return_model = (
base_model
if base_model is not None
else _priced_provider_response_model(
provider_response_model,
completion_response_model if isinstance(completion_response_model, str) else None,
custom_llm_provider,
)
)
elif completion_response_model is None and hidden_params is not None:
if hidden_params.get("model", None) is not None and len(hidden_params["model"]) > 0:

View file

@ -5639,3 +5639,72 @@ def test_completion_cost_bills_base_when_gemini_serves_on_demand(
)
assert cost == pytest.approx(100 * 0.001 + 50 * 0.002)
def test_an_unpriced_provider_slug_does_not_bill_the_turn_at_zero():
"""Regression test for #42161.
`_select_model_name_for_cost_calc` prefers `provider_response_model` over
`response.model`, and `get_model_info` answers a model it has never heard of with
zero rates instead of raising, so a turn reporting an unregistered slug was billed
0.0 with nothing in the logs to explain it. Anthropic hits this by naming the dated
build in `message_start` while the map carries the family.
The invariant: what the provider reports must not change the price of a turn the
price map cannot quote it by.
"""
from litellm.cost_calculator import completion_cost
from litellm.types.utils import Choices, Message, ModelResponse, Usage
unpriced = "claude-test-42161"
assert unpriced not in litellm.model_cost
def _response(reported: str | None) -> ModelResponse:
response = ModelResponse(
model="anthropic/claude-opus-5",
choices=[Choices(message=Message(content="x"))],
)
response.usage = Usage(prompt_tokens=38, completion_tokens=24, total_tokens=62)
response._hidden_params = {} if reported is None else {"provider_response_model": reported}
return response
baseline = completion_cost(completion_response=_response(None), custom_llm_provider="anthropic")
assert baseline > 0
reported_unpriced = completion_cost(completion_response=_response(unpriced), custom_llm_provider="anthropic")
assert reported_unpriced == baseline
def test_a_priced_provider_slug_still_decides_the_rate():
"""The fallback must not cost the reported name its precedence.
When the provider names something the map does price, that is the most specific
truth about what served the turn and it still decides the rate.
"""
from litellm.cost_calculator import completion_cost
from litellm.types.utils import Choices, Message, ModelResponse, Usage
priced = "test-model-priced-42161"
litellm.register_model(
model_cost={
priced: {
"input_cost_per_token": 1e-05,
"output_cost_per_token": 2e-05,
"litellm_provider": "anthropic",
"mode": "chat",
}
}
)
def _response(reported: str | None, model: str) -> ModelResponse:
response = ModelResponse(model=model, choices=[Choices(message=Message(content="x"))])
response.usage = Usage(prompt_tokens=38, completion_tokens=24, total_tokens=62)
response._hidden_params = {} if reported is None else {"provider_response_model": reported}
return response
reported = completion_cost(
completion_response=_response(priced, "anthropic/claude-opus-5"),
custom_llm_provider="anthropic",
)
expected = 38 * 1e-05 + 24 * 2e-05
assert reported == pytest.approx(expected)