mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
fix(cost): a provider slug the price map lacks no longer bills the turn at zero
Anthropic names the dated build in `message_start`, so a streamed turn carries
`provider_response_model="claude-opus-5-20250930"` while the price map holds
`claude-opus-5`. `_select_model_name_for_cost_calc` prefers the reported name
over `response.model`, and `get_model_info` answers a model it has never heard
of with zero rates rather than raising -- so the turn is billed 0.0 with nothing
in the logs to explain it.
Same usage, only that field varying, on main:
provider_response_model absent -> 0.00079
provider_response_model=claude-opus-5 -> 0.00079
provider_response_model=claude-opus-5-20250930 -> 0.0
Missing is harmless; the damage is a name that is present and unpriced.
The reported name still wins whenever it resolves to a rate -- it is the most
specific truth about what served the turn. It is only passed over when charging
by it would produce a silent zero, and then the response's own model is used:
the name the deployment resolved to, and the one an unstreamed turn is already
priced by.
This is not specific to Anthropic or to streaming. Any provider reporting a more
specific slug than the map carries -- dated, regional, build-suffixed -- is
affected, and only streamed turns carry the field, which is why it reads as a
streaming bug from the outside.
Reported in #42161.
This commit is contained in:
parent
c19ce71bcd
commit
c5ed8284bd
2 changed files with 122 additions and 1 deletions
|
|
@ -800,6 +800,52 @@ def _cost_map_entry_prices_anything(entry: Mapping[str, object]) -> bool:
|
|||
)
|
||||
|
||||
|
||||
def _has_rate(model: str, custom_llm_provider: str | None) -> bool:
|
||||
"""Whether the price map can quote this name.
|
||||
|
||||
``get_model_info`` answers a model it does not know with zero rates rather than
|
||||
raising, so asking it is not enough to tell "free" from "unknown".
|
||||
"""
|
||||
if model in litellm.model_cost:
|
||||
return True
|
||||
try:
|
||||
info: Final = litellm.get_model_info(model=model, custom_llm_provider=custom_llm_provider)
|
||||
except Exception:
|
||||
return False
|
||||
return any(
|
||||
info.get(key)
|
||||
for key in ("input_cost_per_token", "output_cost_per_token", "input_cost_per_second")
|
||||
)
|
||||
|
||||
|
||||
def _priced_provider_response_model(
|
||||
provider_response_model: str | None,
|
||||
completion_response_model: str | None,
|
||||
custom_llm_provider: str | None,
|
||||
) -> str | None:
|
||||
"""The provider's own model name, unless charging by it would bill the turn at zero.
|
||||
|
||||
Providers report a build rather than a family: Anthropic's ``message_start`` names
|
||||
``claude-opus-5-20250930`` while the price map carries ``claude-opus-5``. Preferring
|
||||
the reported name is right when it is priced — it is the most specific truth about
|
||||
what served the turn — but when the map has never heard of it the turn is billed
|
||||
``0.0`` with nothing in the logs to say why, because ``get_model_info`` returns zero
|
||||
rates for an unknown model instead of raising.
|
||||
|
||||
Falling back to the response's own model keeps the previous, working behaviour for
|
||||
that case: it is the name the deployment resolved to, and it is what an unstreamed
|
||||
turn — which carries no ``provider_response_model`` at all — is already priced by.
|
||||
"""
|
||||
if provider_response_model is None:
|
||||
return None
|
||||
if _has_rate(provider_response_model, custom_llm_provider):
|
||||
return provider_response_model
|
||||
if completion_response_model is not None and _has_rate(completion_response_model, custom_llm_provider):
|
||||
return completion_response_model
|
||||
# Neither is priced: keep the provider's name, so the zero that follows is reported
|
||||
# against what actually served the turn.
|
||||
return provider_response_model
|
||||
|
||||
def _select_model_name_for_cost_calc(
|
||||
model: str | None,
|
||||
completion_response: object | None,
|
||||
|
|
@ -846,7 +892,13 @@ def _select_model_name_for_cost_calc(
|
|||
return_model = model
|
||||
|
||||
elif base_model is not None or provider_response_model is not None:
|
||||
return_model = base_model if base_model is not None else provider_response_model
|
||||
return_model = (
|
||||
base_model
|
||||
if base_model is not None
|
||||
else _priced_provider_response_model(
|
||||
provider_response_model, completion_response_model, custom_llm_provider
|
||||
)
|
||||
)
|
||||
|
||||
elif completion_response_model is None and hidden_params is not None:
|
||||
if hidden_params.get("model", None) is not None and len(hidden_params["model"]) > 0:
|
||||
|
|
|
|||
|
|
@ -5319,3 +5319,72 @@ def test_completion_cost_is_zero_when_explicit_rates_are_zero(monkeypatch: pytes
|
|||
)
|
||||
|
||||
assert cost == 0.0
|
||||
def test_dated_provider_slug_does_not_bill_the_turn_at_zero():
|
||||
"""A provider reporting a build the price map does not carry must not zero the turn.
|
||||
|
||||
Anthropic names the dated build in `message_start`, so a streamed turn carries
|
||||
`provider_response_model="claude-opus-5-20250930"` while the map holds
|
||||
`claude-opus-5`. `_select_model_name_for_cost_calc` prefers the reported name, and
|
||||
`get_model_info` answers an unknown model with zero rates rather than raising, so the
|
||||
turn was billed 0.0 with nothing in the logs to explain it.
|
||||
|
||||
Unstreamed turns carry no `provider_response_model` and were unaffected, which is why
|
||||
this looked like a streaming bug. Regression test for #42161.
|
||||
"""
|
||||
from litellm.cost_calculator import completion_cost
|
||||
from litellm.types.utils import Choices, Message, ModelResponse, Usage
|
||||
|
||||
def _response(reported: str | None) -> ModelResponse:
|
||||
response = ModelResponse(
|
||||
model="anthropic/claude-opus-5",
|
||||
choices=[Choices(message=Message(content="x"))],
|
||||
)
|
||||
response.usage = Usage(prompt_tokens=38, completion_tokens=24, total_tokens=62)
|
||||
response._hidden_params = (
|
||||
{} if reported is None else {"provider_response_model": reported}
|
||||
)
|
||||
return response
|
||||
|
||||
assert "claude-opus-5" in litellm.model_cost
|
||||
assert "claude-opus-5-20250930" not in litellm.model_cost
|
||||
|
||||
baseline = completion_cost(
|
||||
completion_response=_response(None), custom_llm_provider="anthropic"
|
||||
)
|
||||
assert baseline > 0
|
||||
|
||||
dated = completion_cost(
|
||||
completion_response=_response("claude-opus-5-20250930"),
|
||||
custom_llm_provider="anthropic",
|
||||
)
|
||||
assert dated == baseline
|
||||
|
||||
|
||||
def test_a_priced_provider_slug_still_wins_over_the_response_model():
|
||||
"""The fallback must not cost the reported name its precedence.
|
||||
|
||||
When the provider names something the map *does* price, that is the most specific
|
||||
truth about what served the turn and it still decides the rate.
|
||||
"""
|
||||
from litellm.cost_calculator import completion_cost
|
||||
from litellm.types.utils import Choices, Message, ModelResponse, Usage
|
||||
|
||||
def _response(reported: str | None, model: str) -> ModelResponse:
|
||||
response = ModelResponse(
|
||||
model=model, choices=[Choices(message=Message(content="x"))]
|
||||
)
|
||||
response.usage = Usage(prompt_tokens=38, completion_tokens=24, total_tokens=62)
|
||||
response._hidden_params = (
|
||||
{} if reported is None else {"provider_response_model": reported}
|
||||
)
|
||||
return response
|
||||
|
||||
reported_haiku = completion_cost(
|
||||
completion_response=_response("claude-haiku-4-5", "anthropic/claude-opus-5"),
|
||||
custom_llm_provider="anthropic",
|
||||
)
|
||||
haiku_directly = completion_cost(
|
||||
completion_response=_response(None, "anthropic/claude-haiku-4-5"),
|
||||
custom_llm_provider="anthropic",
|
||||
)
|
||||
assert reported_haiku == haiku_directly
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue