diff --git a/litellm/cost_calculator.py b/litellm/cost_calculator.py index a279b9f0903..c97748d4631 100644 --- a/litellm/cost_calculator.py +++ b/litellm/cost_calculator.py @@ -800,6 +800,52 @@ def _cost_map_entry_prices_anything(entry: Mapping[str, object]) -> bool: ) +def _has_rate(model: str, custom_llm_provider: str | None) -> bool: + """Whether the price map can quote this name. + + ``get_model_info`` answers a model it does not know with zero rates rather than + raising, so asking it is not enough to tell "free" from "unknown". + """ + if model in litellm.model_cost: + return True + try: + info: Final = litellm.get_model_info(model=model, custom_llm_provider=custom_llm_provider) + except Exception: + return False + return any( + info.get(key) + for key in ("input_cost_per_token", "output_cost_per_token", "input_cost_per_second") + ) + + +def _priced_provider_response_model( + provider_response_model: str | None, + completion_response_model: str | None, + custom_llm_provider: str | None, +) -> str | None: + """The provider's own model name, unless charging by it would bill the turn at zero. + + Providers report a build rather than a family: Anthropic's ``message_start`` names + ``claude-opus-5-20250930`` while the price map carries ``claude-opus-5``. Preferring + the reported name is right when it is priced — it is the most specific truth about + what served the turn — but when the map has never heard of it the turn is billed + ``0.0`` with nothing in the logs to say why, because ``get_model_info`` returns zero + rates for an unknown model instead of raising. + + Falling back to the response's own model keeps the previous, working behaviour for + that case: it is the name the deployment resolved to, and it is what an unstreamed + turn — which carries no ``provider_response_model`` at all — is already priced by. + """ + if provider_response_model is None: + return None + if _has_rate(provider_response_model, custom_llm_provider): + return provider_response_model + if completion_response_model is not None and _has_rate(completion_response_model, custom_llm_provider): + return completion_response_model + # Neither is priced: keep the provider's name, so the zero that follows is reported + # against what actually served the turn. + return provider_response_model + def _select_model_name_for_cost_calc( model: str | None, completion_response: object | None, @@ -846,7 +892,13 @@ def _select_model_name_for_cost_calc( return_model = model elif base_model is not None or provider_response_model is not None: - return_model = base_model if base_model is not None else provider_response_model + return_model = ( + base_model + if base_model is not None + else _priced_provider_response_model( + provider_response_model, completion_response_model, custom_llm_provider + ) + ) elif completion_response_model is None and hidden_params is not None: if hidden_params.get("model", None) is not None and len(hidden_params["model"]) > 0: diff --git a/tests/test_litellm/test_cost_calculator.py b/tests/test_litellm/test_cost_calculator.py index 99dea6366f9..860f35f4c94 100644 --- a/tests/test_litellm/test_cost_calculator.py +++ b/tests/test_litellm/test_cost_calculator.py @@ -5319,3 +5319,72 @@ def test_completion_cost_is_zero_when_explicit_rates_are_zero(monkeypatch: pytes ) assert cost == 0.0 +def test_dated_provider_slug_does_not_bill_the_turn_at_zero(): + """A provider reporting a build the price map does not carry must not zero the turn. + + Anthropic names the dated build in `message_start`, so a streamed turn carries + `provider_response_model="claude-opus-5-20250930"` while the map holds + `claude-opus-5`. `_select_model_name_for_cost_calc` prefers the reported name, and + `get_model_info` answers an unknown model with zero rates rather than raising, so the + turn was billed 0.0 with nothing in the logs to explain it. + + Unstreamed turns carry no `provider_response_model` and were unaffected, which is why + this looked like a streaming bug. Regression test for #42161. + """ + from litellm.cost_calculator import completion_cost + from litellm.types.utils import Choices, Message, ModelResponse, Usage + + def _response(reported: str | None) -> ModelResponse: + response = ModelResponse( + model="anthropic/claude-opus-5", + choices=[Choices(message=Message(content="x"))], + ) + response.usage = Usage(prompt_tokens=38, completion_tokens=24, total_tokens=62) + response._hidden_params = ( + {} if reported is None else {"provider_response_model": reported} + ) + return response + + assert "claude-opus-5" in litellm.model_cost + assert "claude-opus-5-20250930" not in litellm.model_cost + + baseline = completion_cost( + completion_response=_response(None), custom_llm_provider="anthropic" + ) + assert baseline > 0 + + dated = completion_cost( + completion_response=_response("claude-opus-5-20250930"), + custom_llm_provider="anthropic", + ) + assert dated == baseline + + +def test_a_priced_provider_slug_still_wins_over_the_response_model(): + """The fallback must not cost the reported name its precedence. + + When the provider names something the map *does* price, that is the most specific + truth about what served the turn and it still decides the rate. + """ + from litellm.cost_calculator import completion_cost + from litellm.types.utils import Choices, Message, ModelResponse, Usage + + def _response(reported: str | None, model: str) -> ModelResponse: + response = ModelResponse( + model=model, choices=[Choices(message=Message(content="x"))] + ) + response.usage = Usage(prompt_tokens=38, completion_tokens=24, total_tokens=62) + response._hidden_params = ( + {} if reported is None else {"provider_response_model": reported} + ) + return response + + reported_haiku = completion_cost( + completion_response=_response("claude-haiku-4-5", "anthropic/claude-opus-5"), + custom_llm_provider="anthropic", + ) + haiku_directly = completion_cost( + completion_response=_response(None, "anthropic/claude-haiku-4-5"), + custom_llm_provider="anthropic", + ) + assert reported_haiku == haiku_directly