From 49cd17ee102160e154f81e8949fee9223c628918 Mon Sep 17 00:00:00 2001 From: Darsh Joshi Date: Fri, 2 Oct 2026 19:01:49 -0400 Subject: [PATCH 1/2] fix(cost): bill per-character rates set on a deployment A deployment priced only through input_cost_per_character or output_cost_per_character was treated as unpriced, so cost lookup fell back to the shared model key. The router strips custom pricing from that key, so audio_speech calls on such a deployment logged spend 0 and sent no x-litellm-response-cost header Count both character rates as pricing when picking the deployment's own cost entry Fixes #44200 --- litellm/cost_calculator.py | 10 +++++++++- tests/unit/test_cost_calculator.py | 32 ++++++++++++++++++++++++++++++ 2 files changed, 41 insertions(+), 1 deletion(-) diff --git a/litellm/cost_calculator.py b/litellm/cost_calculator.py index 41a7ef1ab64..809adb9c678 100644 --- a/litellm/cost_calculator.py +++ b/litellm/cost_calculator.py @@ -800,7 +800,15 @@ def _get_hidden_str_for_cost_calc(hidden_params: object, key: str) -> str | None _NON_TOKEN_RATE_FIELDS: Final = frozenset( - {"cost_per_second", "input_cost_per_second", "output_cost_per_second", "input_cost_per_query", "tiered_pricing"} + { + "cost_per_second", + "input_cost_per_second", + "output_cost_per_second", + "input_cost_per_query", + "input_cost_per_character", + "output_cost_per_character", + "tiered_pricing", + } ) diff --git a/tests/unit/test_cost_calculator.py b/tests/unit/test_cost_calculator.py index 36e188e82d6..6ee3e5548e7 100644 --- a/tests/unit/test_cost_calculator.py +++ b/tests/unit/test_cost_calculator.py @@ -1067,6 +1067,38 @@ def test_per_query_priced_rerank_deployment_completion_cost_is_nonzero(): assert cost == pytest.approx(3 * 0.001) +def test_per_character_priced_speech_deployment_bills_its_own_rate(_local_model_cost_map: None) -> None: + from litellm import Router + + rate: Final = 1e-8 + prompt: Final = "abcdefghijklm" + router: Final = Router( + model_list=[ + { + "model_name": "custom-tts", + "litellm_params": { + "model": "openai/custom-tts-unmapped", + "api_key": "sk-fake", + "input_cost_per_character": rate, + }, + }, + ] + ) + router_model_id: Final = router.model_list[0]["model_info"]["id"] + + cost: Final = completion_cost( + completion_response=None, + model="openai/custom-tts-unmapped", + custom_llm_provider="openai", + call_type="speech", + prompt=prompt, + custom_pricing=True, + router_model_id=router_model_id, + ) + + assert cost == pytest.approx(len(prompt) * rate) + + def test_azure_realtime_cost_calculator(_local_model_cost_map): cost = handle_realtime_stream_cost_calculation( From 1955d41beb581708e13eb3641f7cfe2fb9f4afae Mon Sep 17 00:00:00 2001 From: Darsh Joshi Date: Sat, 3 Oct 2026 11:52:24 -0400 Subject: [PATCH 2/2] fix(cost): count character rates only when picking the deployment entry Character rates were added to the fields every pricing check accepts, so the realtime fallback also read a character-only deployment as priced. Realtime bills per token, so it took that deployment's zero token cost and never reached the session model's rates Keep character rates out of the shared field set and pass them only where the deployment's own entry is chosen, and cover a mapped speech model whose deployment overrides the public character rate --- litellm/cost_calculator.py | 23 ++++---- tests/unit/test_cost_calculator.py | 84 ++++++++++++++++++++++++++++++ 2 files changed, 95 insertions(+), 12 deletions(-) diff --git a/litellm/cost_calculator.py b/litellm/cost_calculator.py index 809adb9c678..b31ec951e30 100644 --- a/litellm/cost_calculator.py +++ b/litellm/cost_calculator.py @@ -800,21 +800,20 @@ def _get_hidden_str_for_cost_calc(hidden_params: object, key: str) -> str | None _NON_TOKEN_RATE_FIELDS: Final = frozenset( - { - "cost_per_second", - "input_cost_per_second", - "output_cost_per_second", - "input_cost_per_query", - "input_cost_per_character", - "output_cost_per_character", - "tiered_pricing", - } + {"cost_per_second", "input_cost_per_second", "output_cost_per_second", "input_cost_per_query", "tiered_pricing"} ) +# Count only when picking a deployment's own entry: speech bills per character, realtime bills per token +_PER_CHARACTER_RATE_FIELDS: Final = frozenset({"input_cost_per_character", "output_cost_per_character"}) -def _cost_map_entry_prices_anything(entry: Mapping[str, object]) -> bool: +def _cost_map_entry_prices_anything( + entry: Mapping[str, object], extra_rate_fields: frozenset[str] = frozenset() +) -> bool: return any( - value is not None and (field in _NON_TOKEN_RATE_FIELDS or ("cost_per" in field and "token" in field)) + value is not None + and ( + field in _NON_TOKEN_RATE_FIELDS or field in extra_rate_fields or ("cost_per" in field and "token" in field) + ) for field, value in entry.items() ) @@ -857,7 +856,7 @@ def _select_model_name_for_cost_calc( if custom_pricing is True: if router_model_id is not None and router_model_id in litellm.model_cost: entry: Final = litellm.model_cost[router_model_id] - if _cost_map_entry_prices_anything(entry): + if _cost_map_entry_prices_anything(entry, extra_rate_fields=_PER_CHARACTER_RATE_FIELDS): return_model = router_model_id else: return_model = model diff --git a/tests/unit/test_cost_calculator.py b/tests/unit/test_cost_calculator.py index 6ee3e5548e7..f31fc100488 100644 --- a/tests/unit/test_cost_calculator.py +++ b/tests/unit/test_cost_calculator.py @@ -1099,6 +1099,90 @@ def test_per_character_priced_speech_deployment_bills_its_own_rate(_local_model_ assert cost == pytest.approx(len(prompt) * rate) +def test_per_character_rate_on_mapped_speech_deployment_beats_the_public_rate(_local_model_cost_map: None) -> None: + from litellm import Router + + public_rate: Final = litellm.model_cost["tts-1"]["input_cost_per_character"] + deployment_rate: Final = public_rate / 3 + prompt: Final = "abcdefghijklm" + router: Final = Router( + model_list=[ + { + "model_name": "discounted-tts", + "litellm_params": { + "model": "openai/tts-1", + "api_key": "sk-fake", + "input_cost_per_character": deployment_rate, + }, + }, + ] + ) + router_model_id: Final = router.model_list[0]["model_info"]["id"] + + cost: Final = completion_cost( + completion_response=None, + model="openai/tts-1", + custom_llm_provider="openai", + call_type="speech", + prompt=prompt, + custom_pricing=True, + router_model_id=router_model_id, + ) + + assert cost == pytest.approx(len(prompt) * deployment_rate) + assert cost != pytest.approx(len(prompt) * public_rate) + + +def test_per_character_rate_on_realtime_deployment_keeps_session_token_pricing( + _local_model_cost_map: None, monkeypatch: pytest.MonkeyPatch +) -> None: + """Realtime is billed per token, so a character rate alone must not stop the + lookup at the deployment and bill the session nothing.""" + from litellm.types.utils import CompletionTokensDetailsWrapper + + model: Final = "gpt-realtime" + deployment_key: Final = "deployment-id-for-a-character-priced-realtime-group" + monkeypatch.setitem( + litellm.model_cost, + deployment_key, + {"litellm_provider": "openai", "mode": "realtime", "input_cost_per_character": 1e-6}, + ) + logging_object: Final = LiteLLMRealtimeStreamLoggingObject( + usage=Usage( + prompt_tokens=120, + completion_tokens=60, + total_tokens=180, + prompt_tokens_details=PromptTokensDetailsWrapper(text_tokens=20, audio_tokens=100), + completion_tokens_details=CompletionTokensDetailsWrapper(text_tokens=10, audio_tokens=50), + ), + results=[ + {"type": "session.created", "session": {"model": model}}, + { + "type": "response.done", + "response": {"usage": {"input_tokens": 120, "output_tokens": 60, "total_tokens": 180}}, + }, + ], + ) + + public_cost: Final = completion_cost( + completion_response=logging_object, + model=model, + call_type=CallTypes.arealtime.value, + custom_llm_provider="openai", + ) + deployment_cost: Final = completion_cost( + completion_response=logging_object, + model=model, + call_type=CallTypes.arealtime.value, + custom_llm_provider="openai", + custom_pricing=True, + router_model_id=deployment_key, + ) + + assert public_cost > 0 + assert deployment_cost == pytest.approx(public_cost, rel=1e-9) + + def test_azure_realtime_cost_calculator(_local_model_cost_map): cost = handle_realtime_stream_cost_calculation(