fix(cost): a provider slug the price map lacks no longer bills the turn at zero

Anthropic names the dated build in `message_start`, so a streamed turn carries
`provider_response_model="claude-opus-5-20250930"` while the price map holds
`claude-opus-5`. `_select_model_name_for_cost_calc` prefers the reported name
over `response.model`, and `get_model_info` answers a model it has never heard
of with zero rates rather than raising -- so the turn is billed 0.0 with nothing
in the logs to explain it.

Same usage, only that field varying, on main:

    provider_response_model absent                  ->  0.00079
    provider_response_model=claude-opus-5           ->  0.00079
    provider_response_model=claude-opus-5-20250930  ->  0.0

Missing is harmless; the damage is a name that is present and unpriced.

The reported name still wins whenever it resolves to a rate -- it is the most
specific truth about what served the turn. It is only passed over when charging
by it would produce a silent zero, and then the response's own model is used:
the name the deployment resolved to, and the one an unstreamed turn is already
priced by.

This is not specific to Anthropic or to streaming. Any provider reporting a more
specific slug than the map carries -- dated, regional, build-suffixed -- is
affected, and only streamed turns carry the field, which is why it reads as a
streaming bug from the outside.

Reported in #42161.
This commit is contained in:
Eduardo Pessin 2026-09-22 13:08:42 +01:00 • committed by eduardopessin
parent c19ce71bcd
commit c5ed8284bd
2 changed files with 122 additions and 1 deletions

View file

@ -800,6 +800,52 @@ def _cost_map_entry_prices_anything(entry: Mapping[str, object]) -> bool:
)
def _has_rate(model: str, custom_llm_provider: str | None) -> bool:
"""Whether the price map can quote this name.
``get_model_info`` answers a model it does not know with zero rates rather than
raising, so asking it is not enough to tell "free" from "unknown".
"""
if model in litellm.model_cost:
return True
try:
info: Final = litellm.get_model_info(model=model, custom_llm_provider=custom_llm_provider)
except Exception:
return False
return any(
info.get(key)
for key in ("input_cost_per_token", "output_cost_per_token", "input_cost_per_second")
)
def _priced_provider_response_model(
provider_response_model: str | None,
completion_response_model: str | None,
custom_llm_provider: str | None,
) -> str | None:
"""The provider's own model name, unless charging by it would bill the turn at zero.
Providers report a build rather than a family: Anthropic's ``message_start`` names
``claude-opus-5-20250930`` while the price map carries ``claude-opus-5``. Preferring
the reported name is right when it is priced — it is the most specific truth about
what served the turn — but when the map has never heard of it the turn is billed
``0.0`` with nothing in the logs to say why, because ``get_model_info`` returns zero
rates for an unknown model instead of raising.
Falling back to the response's own model keeps the previous, working behaviour for
that case: it is the name the deployment resolved to, and it is what an unstreamed
turn — which carries no ``provider_response_model`` at all — is already priced by.
"""
if provider_response_model is None:
return None
if _has_rate(provider_response_model, custom_llm_provider):
return provider_response_model
if completion_response_model is not None and _has_rate(completion_response_model, custom_llm_provider):
return completion_response_model
# Neither is priced: keep the provider's name, so the zero that follows is reported
# against what actually served the turn.
return provider_response_model
def _select_model_name_for_cost_calc(
model: str | None,
completion_response: object | None,
@ -846,7 +892,13 @@ def _select_model_name_for_cost_calc(
return_model = model
elif base_model is not None or provider_response_model is not None:
return_model = base_model if base_model is not None else provider_response_model
return_model = (
base_model
if base_model is not None
else _priced_provider_response_model(
provider_response_model, completion_response_model, custom_llm_provider
)
)
elif completion_response_model is None and hidden_params is not None:
if hidden_params.get("model", None) is not None and len(hidden_params["model"]) > 0:

View file

@ -5319,3 +5319,72 @@ def test_completion_cost_is_zero_when_explicit_rates_are_zero(monkeypatch: pytes
)
assert cost == 0.0
def test_dated_provider_slug_does_not_bill_the_turn_at_zero():
"""A provider reporting a build the price map does not carry must not zero the turn.
Anthropic names the dated build in `message_start`, so a streamed turn carries
`provider_response_model="claude-opus-5-20250930"` while the map holds
`claude-opus-5`. `_select_model_name_for_cost_calc` prefers the reported name, and
`get_model_info` answers an unknown model with zero rates rather than raising, so the
turn was billed 0.0 with nothing in the logs to explain it.
Unstreamed turns carry no `provider_response_model` and were unaffected, which is why
this looked like a streaming bug. Regression test for #42161.
"""
from litellm.cost_calculator import completion_cost
from litellm.types.utils import Choices, Message, ModelResponse, Usage
def _response(reported: str | None) -> ModelResponse:
response = ModelResponse(
model="anthropic/claude-opus-5",
choices=[Choices(message=Message(content="x"))],
)
response.usage = Usage(prompt_tokens=38, completion_tokens=24, total_tokens=62)
response._hidden_params = (
{} if reported is None else {"provider_response_model": reported}
)
return response
assert "claude-opus-5" in litellm.model_cost
assert "claude-opus-5-20250930" not in litellm.model_cost
baseline = completion_cost(
completion_response=_response(None), custom_llm_provider="anthropic"
)
assert baseline > 0
dated = completion_cost(
completion_response=_response("claude-opus-5-20250930"),
custom_llm_provider="anthropic",
)
assert dated == baseline
def test_a_priced_provider_slug_still_wins_over_the_response_model():
"""The fallback must not cost the reported name its precedence.
When the provider names something the map *does* price, that is the most specific
truth about what served the turn and it still decides the rate.
"""
from litellm.cost_calculator import completion_cost
from litellm.types.utils import Choices, Message, ModelResponse, Usage
def _response(reported: str | None, model: str) -> ModelResponse:
response = ModelResponse(
model=model, choices=[Choices(message=Message(content="x"))]
)
response.usage = Usage(prompt_tokens=38, completion_tokens=24, total_tokens=62)
response._hidden_params = (
{} if reported is None else {"provider_response_model": reported}
)
return response
reported_haiku = completion_cost(
completion_response=_response("claude-haiku-4-5", "anthropic/claude-opus-5"),
custom_llm_provider="anthropic",
)
haiku_directly = completion_cost(
completion_response=_response(None, "anthropic/claude-haiku-4-5"),
custom_llm_provider="anthropic",
)
assert reported_haiku == haiku_directly