From 1955d41beb581708e13eb3641f7cfe2fb9f4afae Mon Sep 17 00:00:00 2001 From: Darsh Joshi Date: Sat, 3 Oct 2026 11:52:24 -0400 Subject: [PATCH] fix(cost): count character rates only when picking the deployment entry Character rates were added to the fields every pricing check accepts, so the realtime fallback also read a character-only deployment as priced. Realtime bills per token, so it took that deployment's zero token cost and never reached the session model's rates Keep character rates out of the shared field set and pass them only where the deployment's own entry is chosen, and cover a mapped speech model whose deployment overrides the public character rate --- litellm/cost_calculator.py | 23 ++++---- tests/unit/test_cost_calculator.py | 84 ++++++++++++++++++++++++++++++ 2 files changed, 95 insertions(+), 12 deletions(-) diff --git a/litellm/cost_calculator.py b/litellm/cost_calculator.py index 809adb9c678..b31ec951e30 100644 --- a/litellm/cost_calculator.py +++ b/litellm/cost_calculator.py @@ -800,21 +800,20 @@ def _get_hidden_str_for_cost_calc(hidden_params: object, key: str) -> str | None _NON_TOKEN_RATE_FIELDS: Final = frozenset( - { - "cost_per_second", - "input_cost_per_second", - "output_cost_per_second", - "input_cost_per_query", - "input_cost_per_character", - "output_cost_per_character", - "tiered_pricing", - } + {"cost_per_second", "input_cost_per_second", "output_cost_per_second", "input_cost_per_query", "tiered_pricing"} ) +# Count only when picking a deployment's own entry: speech bills per character, realtime bills per token +_PER_CHARACTER_RATE_FIELDS: Final = frozenset({"input_cost_per_character", "output_cost_per_character"}) -def _cost_map_entry_prices_anything(entry: Mapping[str, object]) -> bool: +def _cost_map_entry_prices_anything( + entry: Mapping[str, object], extra_rate_fields: frozenset[str] = frozenset() +) -> bool: return any( - value is not None and (field in _NON_TOKEN_RATE_FIELDS or ("cost_per" in field and "token" in field)) + value is not None + and ( + field in _NON_TOKEN_RATE_FIELDS or field in extra_rate_fields or ("cost_per" in field and "token" in field) + ) for field, value in entry.items() ) @@ -857,7 +856,7 @@ def _select_model_name_for_cost_calc( if custom_pricing is True: if router_model_id is not None and router_model_id in litellm.model_cost: entry: Final = litellm.model_cost[router_model_id] - if _cost_map_entry_prices_anything(entry): + if _cost_map_entry_prices_anything(entry, extra_rate_fields=_PER_CHARACTER_RATE_FIELDS): return_model = router_model_id else: return_model = model diff --git a/tests/unit/test_cost_calculator.py b/tests/unit/test_cost_calculator.py index 6ee3e5548e7..f31fc100488 100644 --- a/tests/unit/test_cost_calculator.py +++ b/tests/unit/test_cost_calculator.py @@ -1099,6 +1099,90 @@ def test_per_character_priced_speech_deployment_bills_its_own_rate(_local_model_ assert cost == pytest.approx(len(prompt) * rate) +def test_per_character_rate_on_mapped_speech_deployment_beats_the_public_rate(_local_model_cost_map: None) -> None: + from litellm import Router + + public_rate: Final = litellm.model_cost["tts-1"]["input_cost_per_character"] + deployment_rate: Final = public_rate / 3 + prompt: Final = "abcdefghijklm" + router: Final = Router( + model_list=[ + { + "model_name": "discounted-tts", + "litellm_params": { + "model": "openai/tts-1", + "api_key": "sk-fake", + "input_cost_per_character": deployment_rate, + }, + }, + ] + ) + router_model_id: Final = router.model_list[0]["model_info"]["id"] + + cost: Final = completion_cost( + completion_response=None, + model="openai/tts-1", + custom_llm_provider="openai", + call_type="speech", + prompt=prompt, + custom_pricing=True, + router_model_id=router_model_id, + ) + + assert cost == pytest.approx(len(prompt) * deployment_rate) + assert cost != pytest.approx(len(prompt) * public_rate) + + +def test_per_character_rate_on_realtime_deployment_keeps_session_token_pricing( + _local_model_cost_map: None, monkeypatch: pytest.MonkeyPatch +) -> None: + """Realtime is billed per token, so a character rate alone must not stop the + lookup at the deployment and bill the session nothing.""" + from litellm.types.utils import CompletionTokensDetailsWrapper + + model: Final = "gpt-realtime" + deployment_key: Final = "deployment-id-for-a-character-priced-realtime-group" + monkeypatch.setitem( + litellm.model_cost, + deployment_key, + {"litellm_provider": "openai", "mode": "realtime", "input_cost_per_character": 1e-6}, + ) + logging_object: Final = LiteLLMRealtimeStreamLoggingObject( + usage=Usage( + prompt_tokens=120, + completion_tokens=60, + total_tokens=180, + prompt_tokens_details=PromptTokensDetailsWrapper(text_tokens=20, audio_tokens=100), + completion_tokens_details=CompletionTokensDetailsWrapper(text_tokens=10, audio_tokens=50), + ), + results=[ + {"type": "session.created", "session": {"model": model}}, + { + "type": "response.done", + "response": {"usage": {"input_tokens": 120, "output_tokens": 60, "total_tokens": 180}}, + }, + ], + ) + + public_cost: Final = completion_cost( + completion_response=logging_object, + model=model, + call_type=CallTypes.arealtime.value, + custom_llm_provider="openai", + ) + deployment_cost: Final = completion_cost( + completion_response=logging_object, + model=model, + call_type=CallTypes.arealtime.value, + custom_llm_provider="openai", + custom_pricing=True, + router_model_id=deployment_key, + ) + + assert public_cost > 0 + assert deployment_cost == pytest.approx(public_cost, rel=1e-9) + + def test_azure_realtime_cost_calculator(_local_model_cost_map): cost = handle_realtime_stream_cost_calculation(