From c5ed8284bdaa890bfec90f51b334ac18e4fbf67f Mon Sep 17 00:00:00 2001 From: Eduardo Pessin Date: Tue, 22 Sep 2026 13:08:42 +0100 Subject: [PATCH 1/3] fix(cost): a provider slug the price map lacks no longer bills the turn at zero Anthropic names the dated build in `message_start`, so a streamed turn carries `provider_response_model="claude-opus-5-20250930"` while the price map holds `claude-opus-5`. `_select_model_name_for_cost_calc` prefers the reported name over `response.model`, and `get_model_info` answers a model it has never heard of with zero rates rather than raising -- so the turn is billed 0.0 with nothing in the logs to explain it. Same usage, only that field varying, on main: provider_response_model absent -> 0.00079 provider_response_model=claude-opus-5 -> 0.00079 provider_response_model=claude-opus-5-20250930 -> 0.0 Missing is harmless; the damage is a name that is present and unpriced. The reported name still wins whenever it resolves to a rate -- it is the most specific truth about what served the turn. It is only passed over when charging by it would produce a silent zero, and then the response's own model is used: the name the deployment resolved to, and the one an unstreamed turn is already priced by. This is not specific to Anthropic or to streaming. Any provider reporting a more specific slug than the map carries -- dated, regional, build-suffixed -- is affected, and only streamed turns carry the field, which is why it reads as a streaming bug from the outside. Reported in #42161. --- litellm/cost_calculator.py | 54 ++++++++++++++++- tests/test_litellm/test_cost_calculator.py | 69 ++++++++++++++++++++++ 2 files changed, 122 insertions(+), 1 deletion(-) diff --git a/litellm/cost_calculator.py b/litellm/cost_calculator.py index a279b9f0903..c97748d4631 100644 --- a/litellm/cost_calculator.py +++ b/litellm/cost_calculator.py @@ -800,6 +800,52 @@ def _cost_map_entry_prices_anything(entry: Mapping[str, object]) -> bool: ) +def _has_rate(model: str, custom_llm_provider: str | None) -> bool: + """Whether the price map can quote this name. + + ``get_model_info`` answers a model it does not know with zero rates rather than + raising, so asking it is not enough to tell "free" from "unknown". + """ + if model in litellm.model_cost: + return True + try: + info: Final = litellm.get_model_info(model=model, custom_llm_provider=custom_llm_provider) + except Exception: + return False + return any( + info.get(key) + for key in ("input_cost_per_token", "output_cost_per_token", "input_cost_per_second") + ) + + +def _priced_provider_response_model( + provider_response_model: str | None, + completion_response_model: str | None, + custom_llm_provider: str | None, +) -> str | None: + """The provider's own model name, unless charging by it would bill the turn at zero. + + Providers report a build rather than a family: Anthropic's ``message_start`` names + ``claude-opus-5-20250930`` while the price map carries ``claude-opus-5``. Preferring + the reported name is right when it is priced — it is the most specific truth about + what served the turn — but when the map has never heard of it the turn is billed + ``0.0`` with nothing in the logs to say why, because ``get_model_info`` returns zero + rates for an unknown model instead of raising. + + Falling back to the response's own model keeps the previous, working behaviour for + that case: it is the name the deployment resolved to, and it is what an unstreamed + turn — which carries no ``provider_response_model`` at all — is already priced by. + """ + if provider_response_model is None: + return None + if _has_rate(provider_response_model, custom_llm_provider): + return provider_response_model + if completion_response_model is not None and _has_rate(completion_response_model, custom_llm_provider): + return completion_response_model + # Neither is priced: keep the provider's name, so the zero that follows is reported + # against what actually served the turn. + return provider_response_model + def _select_model_name_for_cost_calc( model: str | None, completion_response: object | None, @@ -846,7 +892,13 @@ def _select_model_name_for_cost_calc( return_model = model elif base_model is not None or provider_response_model is not None: - return_model = base_model if base_model is not None else provider_response_model + return_model = ( + base_model + if base_model is not None + else _priced_provider_response_model( + provider_response_model, completion_response_model, custom_llm_provider + ) + ) elif completion_response_model is None and hidden_params is not None: if hidden_params.get("model", None) is not None and len(hidden_params["model"]) > 0: diff --git a/tests/test_litellm/test_cost_calculator.py b/tests/test_litellm/test_cost_calculator.py index 99dea6366f9..860f35f4c94 100644 --- a/tests/test_litellm/test_cost_calculator.py +++ b/tests/test_litellm/test_cost_calculator.py @@ -5319,3 +5319,72 @@ def test_completion_cost_is_zero_when_explicit_rates_are_zero(monkeypatch: pytes ) assert cost == 0.0 +def test_dated_provider_slug_does_not_bill_the_turn_at_zero(): + """A provider reporting a build the price map does not carry must not zero the turn. + + Anthropic names the dated build in `message_start`, so a streamed turn carries + `provider_response_model="claude-opus-5-20250930"` while the map holds + `claude-opus-5`. `_select_model_name_for_cost_calc` prefers the reported name, and + `get_model_info` answers an unknown model with zero rates rather than raising, so the + turn was billed 0.0 with nothing in the logs to explain it. + + Unstreamed turns carry no `provider_response_model` and were unaffected, which is why + this looked like a streaming bug. Regression test for #42161. + """ + from litellm.cost_calculator import completion_cost + from litellm.types.utils import Choices, Message, ModelResponse, Usage + + def _response(reported: str | None) -> ModelResponse: + response = ModelResponse( + model="anthropic/claude-opus-5", + choices=[Choices(message=Message(content="x"))], + ) + response.usage = Usage(prompt_tokens=38, completion_tokens=24, total_tokens=62) + response._hidden_params = ( + {} if reported is None else {"provider_response_model": reported} + ) + return response + + assert "claude-opus-5" in litellm.model_cost + assert "claude-opus-5-20250930" not in litellm.model_cost + + baseline = completion_cost( + completion_response=_response(None), custom_llm_provider="anthropic" + ) + assert baseline > 0 + + dated = completion_cost( + completion_response=_response("claude-opus-5-20250930"), + custom_llm_provider="anthropic", + ) + assert dated == baseline + + +def test_a_priced_provider_slug_still_wins_over_the_response_model(): + """The fallback must not cost the reported name its precedence. + + When the provider names something the map *does* price, that is the most specific + truth about what served the turn and it still decides the rate. + """ + from litellm.cost_calculator import completion_cost + from litellm.types.utils import Choices, Message, ModelResponse, Usage + + def _response(reported: str | None, model: str) -> ModelResponse: + response = ModelResponse( + model=model, choices=[Choices(message=Message(content="x"))] + ) + response.usage = Usage(prompt_tokens=38, completion_tokens=24, total_tokens=62) + response._hidden_params = ( + {} if reported is None else {"provider_response_model": reported} + ) + return response + + reported_haiku = completion_cost( + completion_response=_response("claude-haiku-4-5", "anthropic/claude-opus-5"), + custom_llm_provider="anthropic", + ) + haiku_directly = completion_cost( + completion_response=_response(None, "anthropic/claude-haiku-4-5"), + custom_llm_provider="anthropic", + ) + assert reported_haiku == haiku_directly From 00b8e37dfc5675128036345d5d924692e7d146a6 Mon Sep 17 00:00:00 2001 From: Eduardo Pessin Date: Tue, 22 Sep 2026 13:33:07 +0100 Subject: [PATCH 2/3] address review: drop explanatory comments, stop pinning vendor slugs in tests Three points from the review, two of which were right. The helper docstrings explained behaviour the code already shows, which AGENTS.md rules out. Removed. The tests asserted that `claude-opus-5` is in the price map and `claude-opus-5-20250930` is not. Both are facts we do not own: if Anthropic registers the dated build, the test breaks without any litellm change. They now use a name that is ours, and assert the invariant instead, that what the provider reports cannot change the price of a turn the map cannot quote it by. Picking that name needed care. An arbitrary string makes `get_model_info` raise, which a caller upstream already handles, so the test passed with or without the fix and defended nothing. A `claude-` prefixed name reaches the zero-rate fallback that this bug rides on, so `claude-test-42161` reproduces it and stays ours. The third point, that lines 801 and 828 exceed 120 characters, does not hold: they are 98 and 39. --- litellm/cost_calculator.py | 29 ++------- tests/test_litellm/test_cost_calculator.py | 76 +++++++++++----------- 2 files changed, 43 insertions(+), 62 deletions(-) diff --git a/litellm/cost_calculator.py b/litellm/cost_calculator.py index c97748d4631..1eb2d426251 100644 --- a/litellm/cost_calculator.py +++ b/litellm/cost_calculator.py @@ -801,21 +801,16 @@ def _cost_map_entry_prices_anything(entry: Mapping[str, object]) -> bool: def _has_rate(model: str, custom_llm_provider: str | None) -> bool: - """Whether the price map can quote this name. - - ``get_model_info`` answers a model it does not know with zero rates rather than - raising, so asking it is not enough to tell "free" from "unknown". - """ + # A registered entry is a real answer even when it prices at zero; an unregistered + # model is not, because `get_model_info` invents zero rates for it rather than + # raising. Hence truthiness here, not `is not None`. if model in litellm.model_cost: return True try: info: Final = litellm.get_model_info(model=model, custom_llm_provider=custom_llm_provider) except Exception: return False - return any( - info.get(key) - for key in ("input_cost_per_token", "output_cost_per_token", "input_cost_per_second") - ) + return any(info.get(key) for key in ("input_cost_per_token", "output_cost_per_token", "input_cost_per_second")) def _priced_provider_response_model( @@ -823,29 +818,15 @@ def _priced_provider_response_model( completion_response_model: str | None, custom_llm_provider: str | None, ) -> str | None: - """The provider's own model name, unless charging by it would bill the turn at zero. - - Providers report a build rather than a family: Anthropic's ``message_start`` names - ``claude-opus-5-20250930`` while the price map carries ``claude-opus-5``. Preferring - the reported name is right when it is priced — it is the most specific truth about - what served the turn — but when the map has never heard of it the turn is billed - ``0.0`` with nothing in the logs to say why, because ``get_model_info`` returns zero - rates for an unknown model instead of raising. - - Falling back to the response's own model keeps the previous, working behaviour for - that case: it is the name the deployment resolved to, and it is what an unstreamed - turn — which carries no ``provider_response_model`` at all — is already priced by. - """ if provider_response_model is None: return None if _has_rate(provider_response_model, custom_llm_provider): return provider_response_model if completion_response_model is not None and _has_rate(completion_response_model, custom_llm_provider): return completion_response_model - # Neither is priced: keep the provider's name, so the zero that follows is reported - # against what actually served the turn. return provider_response_model + def _select_model_name_for_cost_calc( model: str | None, completion_response: object | None, diff --git a/tests/test_litellm/test_cost_calculator.py b/tests/test_litellm/test_cost_calculator.py index 860f35f4c94..70467ffd24c 100644 --- a/tests/test_litellm/test_cost_calculator.py +++ b/tests/test_litellm/test_cost_calculator.py @@ -5319,72 +5319,72 @@ def test_completion_cost_is_zero_when_explicit_rates_are_zero(monkeypatch: pytes ) assert cost == 0.0 -def test_dated_provider_slug_does_not_bill_the_turn_at_zero(): - """A provider reporting a build the price map does not carry must not zero the turn. - Anthropic names the dated build in `message_start`, so a streamed turn carries - `provider_response_model="claude-opus-5-20250930"` while the map holds - `claude-opus-5`. `_select_model_name_for_cost_calc` prefers the reported name, and - `get_model_info` answers an unknown model with zero rates rather than raising, so the - turn was billed 0.0 with nothing in the logs to explain it. - Unstreamed turns carry no `provider_response_model` and were unaffected, which is why - this looked like a streaming bug. Regression test for #42161. +def test_an_unpriced_provider_slug_does_not_bill_the_turn_at_zero(): + """Regression test for #42161. + + `_select_model_name_for_cost_calc` prefers `provider_response_model` over + `response.model`, and `get_model_info` answers a model it has never heard of with + zero rates instead of raising, so a turn reporting an unregistered slug was billed + 0.0 with nothing in the logs to explain it. Anthropic hits this by naming the dated + build in `message_start` while the map carries the family. + + The invariant: what the provider reports must not change the price of a turn the + price map cannot quote it by. """ from litellm.cost_calculator import completion_cost from litellm.types.utils import Choices, Message, ModelResponse, Usage + unpriced = "claude-test-42161" + assert unpriced not in litellm.model_cost + def _response(reported: str | None) -> ModelResponse: response = ModelResponse( model="anthropic/claude-opus-5", choices=[Choices(message=Message(content="x"))], ) response.usage = Usage(prompt_tokens=38, completion_tokens=24, total_tokens=62) - response._hidden_params = ( - {} if reported is None else {"provider_response_model": reported} - ) + response._hidden_params = {} if reported is None else {"provider_response_model": reported} return response - assert "claude-opus-5" in litellm.model_cost - assert "claude-opus-5-20250930" not in litellm.model_cost - - baseline = completion_cost( - completion_response=_response(None), custom_llm_provider="anthropic" - ) + baseline = completion_cost(completion_response=_response(None), custom_llm_provider="anthropic") assert baseline > 0 - dated = completion_cost( - completion_response=_response("claude-opus-5-20250930"), - custom_llm_provider="anthropic", - ) - assert dated == baseline + reported_unpriced = completion_cost(completion_response=_response(unpriced), custom_llm_provider="anthropic") + assert reported_unpriced == baseline -def test_a_priced_provider_slug_still_wins_over_the_response_model(): +def test_a_priced_provider_slug_still_decides_the_rate(): """The fallback must not cost the reported name its precedence. - When the provider names something the map *does* price, that is the most specific + When the provider names something the map does price, that is the most specific truth about what served the turn and it still decides the rate. """ from litellm.cost_calculator import completion_cost from litellm.types.utils import Choices, Message, ModelResponse, Usage + priced = "test-model-priced-42161" + litellm.register_model( + model_cost={ + priced: { + "input_cost_per_token": 1e-05, + "output_cost_per_token": 2e-05, + "litellm_provider": "anthropic", + "mode": "chat", + } + } + ) + def _response(reported: str | None, model: str) -> ModelResponse: - response = ModelResponse( - model=model, choices=[Choices(message=Message(content="x"))] - ) + response = ModelResponse(model=model, choices=[Choices(message=Message(content="x"))]) response.usage = Usage(prompt_tokens=38, completion_tokens=24, total_tokens=62) - response._hidden_params = ( - {} if reported is None else {"provider_response_model": reported} - ) + response._hidden_params = {} if reported is None else {"provider_response_model": reported} return response - reported_haiku = completion_cost( - completion_response=_response("claude-haiku-4-5", "anthropic/claude-opus-5"), + reported = completion_cost( + completion_response=_response(priced, "anthropic/claude-opus-5"), custom_llm_provider="anthropic", ) - haiku_directly = completion_cost( - completion_response=_response(None, "anthropic/claude-haiku-4-5"), - custom_llm_provider="anthropic", - ) - assert reported_haiku == haiku_directly + expected = 38 * 1e-05 + 24 * 2e-05 + assert reported == pytest.approx(expected) From fd60471a84ecb8b946d0f3f778ede5c7801da38d Mon Sep 17 00:00:00 2001 From: Eduardo Pessin Date: Thu, 1 Oct 2026 19:27:43 +0100 Subject: [PATCH 3/3] Pass completion_response_model as str or None to _priced_provider_response_model basedpyright counted the bare variable as partially unknown (reportUnknownArgumentType +1 over the budget). --- litellm/cost_calculator.py | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/litellm/cost_calculator.py b/litellm/cost_calculator.py index f85ca0462da..b5c6c3a914f 100644 --- a/litellm/cost_calculator.py +++ b/litellm/cost_calculator.py @@ -888,7 +888,9 @@ def _select_model_name_for_cost_calc( base_model if base_model is not None else _priced_provider_response_model( - provider_response_model, completion_response_model, custom_llm_provider + provider_response_model, + completion_response_model if isinstance(completion_response_model, str) else None, + custom_llm_provider, ) )