diff --git a/litellm/cost_calculator.py b/litellm/cost_calculator.py index e358636c105..88aacdd30a6 100644 --- a/litellm/cost_calculator.py +++ b/litellm/cost_calculator.py @@ -812,6 +812,33 @@ def _cost_map_entry_prices_anything(entry: Mapping[str, object]) -> bool: ) +def _has_rate(model: str, custom_llm_provider: str | None) -> bool: + # A registered entry is a real answer even when it prices at zero; an unregistered + # model is not, because `get_model_info` invents zero rates for it rather than + # raising. Hence truthiness here, not `is not None`. + if model in litellm.model_cost: + return True + try: + info: Final = litellm.get_model_info(model=model, custom_llm_provider=custom_llm_provider) + except Exception: # noqa: BLE001 # get_model_info raises bare Exception for some unmapped models + return False + return any(info.get(key) for key in ("input_cost_per_token", "output_cost_per_token", "input_cost_per_second")) + + +def _priced_provider_response_model( + provider_response_model: str | None, + completion_response_model: str | None, + custom_llm_provider: str | None, +) -> str | None: + if provider_response_model is None: + return None + if _has_rate(provider_response_model, custom_llm_provider): + return provider_response_model + if completion_response_model is not None and _has_rate(completion_response_model, custom_llm_provider): + return completion_response_model + return provider_response_model + + def _select_model_name_for_cost_calc( model: str | None, completion_response: object | None, @@ -858,7 +885,15 @@ def _select_model_name_for_cost_calc( return_model = model elif base_model is not None or provider_response_model is not None: - return_model = base_model if base_model is not None else provider_response_model + return_model = ( + base_model + if base_model is not None + else _priced_provider_response_model( + provider_response_model, + completion_response_model if isinstance(completion_response_model, str) else None, + custom_llm_provider, + ) + ) elif completion_response_model is None and hidden_params is not None: if hidden_params.get("model", None) is not None and len(hidden_params["model"]) > 0: diff --git a/tests/unit/test_cost_calculator.py b/tests/unit/test_cost_calculator.py index 36e188e82d6..51527851104 100644 --- a/tests/unit/test_cost_calculator.py +++ b/tests/unit/test_cost_calculator.py @@ -5639,3 +5639,72 @@ def test_completion_cost_bills_base_when_gemini_serves_on_demand( ) assert cost == pytest.approx(100 * 0.001 + 50 * 0.002) + + +def test_an_unpriced_provider_slug_does_not_bill_the_turn_at_zero(): + """Regression test for #42161. + + `_select_model_name_for_cost_calc` prefers `provider_response_model` over + `response.model`, and `get_model_info` answers a model it has never heard of with + zero rates instead of raising, so a turn reporting an unregistered slug was billed + 0.0 with nothing in the logs to explain it. Anthropic hits this by naming the dated + build in `message_start` while the map carries the family. + + The invariant: what the provider reports must not change the price of a turn the + price map cannot quote it by. + """ + from litellm.cost_calculator import completion_cost + from litellm.types.utils import Choices, Message, ModelResponse, Usage + + unpriced = "claude-test-42161" + assert unpriced not in litellm.model_cost + + def _response(reported: str | None) -> ModelResponse: + response = ModelResponse( + model="anthropic/claude-opus-5", + choices=[Choices(message=Message(content="x"))], + ) + response.usage = Usage(prompt_tokens=38, completion_tokens=24, total_tokens=62) + response._hidden_params = {} if reported is None else {"provider_response_model": reported} + return response + + baseline = completion_cost(completion_response=_response(None), custom_llm_provider="anthropic") + assert baseline > 0 + + reported_unpriced = completion_cost(completion_response=_response(unpriced), custom_llm_provider="anthropic") + assert reported_unpriced == baseline + + +def test_a_priced_provider_slug_still_decides_the_rate(): + """The fallback must not cost the reported name its precedence. + + When the provider names something the map does price, that is the most specific + truth about what served the turn and it still decides the rate. + """ + from litellm.cost_calculator import completion_cost + from litellm.types.utils import Choices, Message, ModelResponse, Usage + + priced = "test-model-priced-42161" + litellm.register_model( + model_cost={ + priced: { + "input_cost_per_token": 1e-05, + "output_cost_per_token": 2e-05, + "litellm_provider": "anthropic", + "mode": "chat", + } + } + ) + + def _response(reported: str | None, model: str) -> ModelResponse: + response = ModelResponse(model=model, choices=[Choices(message=Message(content="x"))]) + response.usage = Usage(prompt_tokens=38, completion_tokens=24, total_tokens=62) + response._hidden_params = {} if reported is None else {"provider_response_model": reported} + return response + + reported = completion_cost( + completion_response=_response(priced, "anthropic/claude-opus-5"), + custom_llm_provider="anthropic", + ) + expected = 38 * 1e-05 + 24 * 2e-05 + assert reported == pytest.approx(expected)