From 1f5005ec54b01bb82de1850e45c155902802d55b Mon Sep 17 00:00:00 2001 From: JingHao-Leon Date: Sat, 3 Oct 2026 01:28:20 +0800 Subject: [PATCH] fix(cost): recognize character rates in deployment cost-map selection MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A deployment whose only rates are input_cost_per_character / output_cost_per_character (e.g. a custom OpenAI-compatible TTS backend) failed _cost_map_entry_prices_anything(): the character fields are neither in _NON_TOKEN_RATE_FIELDS nor token-named, so the "prices anything" check returned False. With custom_pricing=True the router_model_id was then never selected and audio_speech (aspeech) resolved its price through the bare model name — which is not in the cost map — producing spend = 0 and no x-litellm-response-cost header (issue #44200). Add both character fields to _NON_TOKEN_RATE_FIELDS so character-only deployments are selected like per-second and per-query deployments. --- litellm/cost_calculator.py | 10 +++++- tests/unit/test_cost_calculator.py | 58 +++++++++++++++++++++--------- 2 files changed, 50 insertions(+), 18 deletions(-) diff --git a/litellm/cost_calculator.py b/litellm/cost_calculator.py index 238b7cc3fdd..8700f2639ef 100644 --- a/litellm/cost_calculator.py +++ b/litellm/cost_calculator.py @@ -800,7 +800,15 @@ def _get_hidden_str_for_cost_calc(hidden_params: object, key: str) -> str | None _NON_TOKEN_RATE_FIELDS: Final = frozenset( - {"cost_per_second", "input_cost_per_second", "output_cost_per_second", "input_cost_per_query", "tiered_pricing"} + { + "cost_per_second", + "input_cost_per_second", + "output_cost_per_second", + "input_cost_per_query", + "input_cost_per_character", + "output_cost_per_character", + "tiered_pricing", + } ) diff --git a/tests/unit/test_cost_calculator.py b/tests/unit/test_cost_calculator.py index 36e188e82d6..934eb837f97 100644 --- a/tests/unit/test_cost_calculator.py +++ b/tests/unit/test_cost_calculator.py @@ -186,7 +186,9 @@ def test_response_cost_calculator_keeps_optional_params_out_of_hidden_params(): assert optional_params["aws_session_token"] == "session-secret" -def test_embedding_success_logging_and_spend_log_carry_no_forwarded_credentials(monkeypatch: pytest.MonkeyPatch) -> None: +def test_embedding_success_logging_and_spend_log_carry_no_forwarded_credentials( + monkeypatch: pytest.MonkeyPatch, +) -> None: from litellm.proxy import proxy_server from litellm.proxy.spend_tracking.spend_tracking_utils import _get_proxy_server_request_for_spend_logs_payload @@ -235,10 +237,6 @@ def test_embedding_success_logging_and_spend_log_carry_no_forwarded_credentials( assert logging_obj.optional_params["extra_headers"] == {"x-goog-api-key": "goog-secret"} - - - - def test_realtime_stream_combines_text_and_audio_token_details(): """Realtime response.done usage with input_token_details / output_token_details.""" from litellm.cost_calculator import RealtimeAPITokenUsageProcessor @@ -1353,8 +1351,6 @@ def test_bedrock_cost_calculator_comparison_with_without_cache(): print(f"Cost with cache: {cost_with_cache}") - - def test_gemini_25_explicit_caching_cost_direct_usage(): """ Test that Gemini 2.5 models correctly calculate costs with explicit caching. @@ -1923,8 +1919,6 @@ def test_cost_margin_with_discount(monkeypatch): print(f" - Expected: ${expected_cost:.6f}") - - def test_completion_cost_extracts_service_tier_from_response(_local_model_cost_map): """Test that completion_cost extracts service_tier from completion_response object.""" from litellm import completion_cost @@ -2674,8 +2668,6 @@ def test_gemini_without_cache_tokens_details(): print("✅ Gemini without cacheTokensDetails works correctly") - - def test_additional_costs_only_for_azure_ai(_local_model_cost_map): """ Test that _get_additional_costs is only called for azure_ai provider. @@ -3220,9 +3212,7 @@ def test_cost_per_token_resolves_per_second_rate_precedence( model: Final = "test-chat-per-second-rate-precedence" entry: Final = {**pricing_fields, "litellm_provider": "together_ai", "mode": "chat"} - litellm.register_model( - model_cost={model: entry} - ) + litellm.register_model(model_cost={model: entry}) assert cost_per_token( model=model, @@ -3647,6 +3637,42 @@ def test_combine_usage_objects_sums_mirrored_cache_write_fields_once(): assert combined_pair.prompt_tokens_details.cache_creation_tokens == 100 +def test_select_model_name_selects_character_priced_deployment(_local_model_cost_map): + """ + A deployment whose only rate is input_cost_per_character must be selected + by router_model_id: the aspeech cost path resolves its price through the + deployment entry, and a character-only entry failing the "prices anything" + check silently produced spend = 0 (issue #44200). + """ + from litellm.cost_calculator import _select_model_name_for_cost_calc + + router_model_id = "openai/qwen-audio-3.1-tts-flash-uuid" + litellm.model_cost[router_model_id] = { + "input_cost_per_character": 1e-8, + "output_cost_per_character": 0.0, + "litellm_provider": "openai", + } + + selected = _select_model_name_for_cost_calc( + model="qwen-audio-3.1-tts-flash", + completion_response=None, + custom_pricing=True, + custom_llm_provider="openai", + router_model_id=router_model_id, + ) + + assert selected == router_model_id + + +def test_cost_map_entry_prices_anything_recognizes_character_rates(): + from litellm.cost_calculator import _cost_map_entry_prices_anything + + assert _cost_map_entry_prices_anything({"input_cost_per_character": 1e-8}) is True + assert _cost_map_entry_prices_anything({"output_cost_per_character": 0.0}) is True + assert _cost_map_entry_prices_anything({"input_cost_per_token": 1e-6}) is True + assert _cost_map_entry_prices_anything({"mode": "audio_speech"}) is False + + def test_select_model_name_strips_unregistered_alias_prefix(_local_model_cost_map): """A router-facing model_name alias containing "/" whose leading segment is NOT a registered provider must not be double-prefixed into a non-existent cost key. @@ -4785,9 +4811,7 @@ def test_xai_batch_tier_discounts_the_long_context_rate_like_the_flat_batch_rate assert info[f"{prefix}_above_200k_tokens_batches"] < info[f"{prefix}_above_200k_tokens"] -@pytest.mark.parametrize( - ("prompt_tokens", "tier"), [(200_000, "_above_200k_tokens_batches"), (199_999, "_batches")] -) +@pytest.mark.parametrize(("prompt_tokens", "tier"), [(200_000, "_above_200k_tokens_batches"), (199_999, "_batches")]) def test_xai_batch_cost_calculator_bills_the_200k_batch_tier_inclusively( _local_model_cost_map: None, prompt_tokens: int, tier: str ) -> None: