diff --git a/litellm/cost_calculator.py b/litellm/cost_calculator.py index e358636c105..1bf2da87448 100644 --- a/litellm/cost_calculator.py +++ b/litellm/cost_calculator.py @@ -803,11 +803,18 @@ def _get_hidden_str_for_cost_calc(hidden_params: object, key: str) -> str | None _NON_TOKEN_RATE_FIELDS: Final = frozenset( {"cost_per_second", "input_cost_per_second", "output_cost_per_second", "input_cost_per_query", "tiered_pricing"} ) +# Count only when picking a deployment's own entry: speech bills per character, realtime bills per token +_PER_CHARACTER_RATE_FIELDS: Final = frozenset({"input_cost_per_character", "output_cost_per_character"}) -def _cost_map_entry_prices_anything(entry: Mapping[str, object]) -> bool: +def _cost_map_entry_prices_anything( + entry: Mapping[str, object], extra_rate_fields: frozenset[str] = frozenset() +) -> bool: return any( - value is not None and (field in _NON_TOKEN_RATE_FIELDS or ("cost_per" in field and "token" in field)) + value is not None + and ( + field in _NON_TOKEN_RATE_FIELDS or field in extra_rate_fields or ("cost_per" in field and "token" in field) + ) for field, value in entry.items() ) @@ -850,7 +857,7 @@ def _select_model_name_for_cost_calc( if custom_pricing is True: if router_model_id is not None and router_model_id in litellm.model_cost: entry: Final = litellm.model_cost[router_model_id] - if _cost_map_entry_prices_anything(entry): + if _cost_map_entry_prices_anything(entry, extra_rate_fields=_PER_CHARACTER_RATE_FIELDS): return_model = router_model_id else: return_model = model diff --git a/tests/unit/test_cost_calculator.py b/tests/unit/test_cost_calculator.py index 9a7c22f6f82..5e314cb46fd 100644 --- a/tests/unit/test_cost_calculator.py +++ b/tests/unit/test_cost_calculator.py @@ -1068,6 +1068,122 @@ def test_per_query_priced_rerank_deployment_completion_cost_is_nonzero(): assert cost == pytest.approx(3 * 0.001) +def test_per_character_priced_speech_deployment_bills_its_own_rate(_local_model_cost_map: None) -> None: + from litellm import Router + + rate: Final = 1e-8 + prompt: Final = "abcdefghijklm" + router: Final = Router( + model_list=[ + { + "model_name": "custom-tts", + "litellm_params": { + "model": "openai/custom-tts-unmapped", + "api_key": "sk-fake", + "input_cost_per_character": rate, + }, + }, + ] + ) + router_model_id: Final = router.model_list[0]["model_info"]["id"] + + cost: Final = completion_cost( + completion_response=None, + model="openai/custom-tts-unmapped", + custom_llm_provider="openai", + call_type="speech", + prompt=prompt, + custom_pricing=True, + router_model_id=router_model_id, + ) + + assert cost == pytest.approx(len(prompt) * rate) + + +def test_per_character_rate_on_mapped_speech_deployment_beats_the_public_rate(_local_model_cost_map: None) -> None: + from litellm import Router + + public_rate: Final = litellm.model_cost["tts-1"]["input_cost_per_character"] + deployment_rate: Final = public_rate / 3 + prompt: Final = "abcdefghijklm" + router: Final = Router( + model_list=[ + { + "model_name": "discounted-tts", + "litellm_params": { + "model": "openai/tts-1", + "api_key": "sk-fake", + "input_cost_per_character": deployment_rate, + }, + }, + ] + ) + router_model_id: Final = router.model_list[0]["model_info"]["id"] + + cost: Final = completion_cost( + completion_response=None, + model="openai/tts-1", + custom_llm_provider="openai", + call_type="speech", + prompt=prompt, + custom_pricing=True, + router_model_id=router_model_id, + ) + + assert cost == pytest.approx(len(prompt) * deployment_rate) + assert cost != pytest.approx(len(prompt) * public_rate) + + +def test_per_character_rate_on_realtime_deployment_keeps_session_token_pricing( + _local_model_cost_map: None, monkeypatch: pytest.MonkeyPatch +) -> None: + """Realtime is billed per token, so a character rate alone must not stop the + lookup at the deployment and bill the session nothing.""" + from litellm.types.utils import CompletionTokensDetailsWrapper + + model: Final = "gpt-realtime" + deployment_key: Final = "deployment-id-for-a-character-priced-realtime-group" + monkeypatch.setitem( + litellm.model_cost, + deployment_key, + {"litellm_provider": "openai", "mode": "realtime", "input_cost_per_character": 1e-6}, + ) + logging_object: Final = LiteLLMRealtimeStreamLoggingObject( + usage=Usage( + prompt_tokens=120, + completion_tokens=60, + total_tokens=180, + prompt_tokens_details=PromptTokensDetailsWrapper(text_tokens=20, audio_tokens=100), + completion_tokens_details=CompletionTokensDetailsWrapper(text_tokens=10, audio_tokens=50), + ), + results=[ + {"type": "session.created", "session": {"model": model}}, + { + "type": "response.done", + "response": {"usage": {"input_tokens": 120, "output_tokens": 60, "total_tokens": 180}}, + }, + ], + ) + + public_cost: Final = completion_cost( + completion_response=logging_object, + model=model, + call_type=CallTypes.arealtime.value, + custom_llm_provider="openai", + ) + deployment_cost: Final = completion_cost( + completion_response=logging_object, + model=model, + call_type=CallTypes.arealtime.value, + custom_llm_provider="openai", + custom_pricing=True, + router_model_id=deployment_key, + ) + + assert public_cost > 0 + assert deployment_cost == pytest.approx(public_cost, rel=1e-9) + + def test_azure_realtime_cost_calculator(_local_model_cost_map): cost = handle_realtime_stream_cost_calculation(