From ca5bcfc8a8cf5cf5d406557a506fe62458835060 Mon Sep 17 00:00:00 2001 From: shivam Date: Thu, 16 Jul 2026 00:48:16 +0000 Subject: [PATCH] fix(cost): price Gemini audio/video prompt tokens at tier-aware text rate Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../litellm_core_utils/llm_cost_calc/utils.py | 12 ++- .../llm_cost_calc/test_llm_cost_calc_utils.py | 81 +++++++++++++++++++ 2 files changed, 90 insertions(+), 3 deletions(-) diff --git a/litellm/litellm_core_utils/llm_cost_calc/utils.py b/litellm/litellm_core_utils/llm_cost_calc/utils.py index 33bf546c239..2a5322416de 100644 --- a/litellm/litellm_core_utils/llm_cost_calc/utils.py +++ b/litellm/litellm_core_utils/llm_cost_calc/utils.py @@ -581,7 +581,10 @@ def _calculate_input_cost( ### AUDIO COST if prompt_tokens_details["audio_tokens"]: audio_cost_key = _get_service_tier_cost_key("input_cost_per_audio_token", service_tier) - prompt_cost += calculate_cost_component(model_info, audio_cost_key, prompt_tokens_details["audio_tokens"]) + if model_info.get(audio_cost_key) is None: + prompt_cost += float(prompt_tokens_details["audio_tokens"]) * prompt_base_cost + else: + prompt_cost += calculate_cost_component(model_info, audio_cost_key, prompt_tokens_details["audio_tokens"]) ### IMAGE TOKEN COST if prompt_tokens_details["image_tokens"]: @@ -596,8 +599,11 @@ def _calculate_input_cost( if prompt_tokens_details["video_tokens"]: video_token_cost_key = "input_cost_per_video_token" if model_info.get(video_token_cost_key) is None: - video_token_cost_key = "input_cost_per_token" - prompt_cost += calculate_cost_component(model_info, video_token_cost_key, prompt_tokens_details["video_tokens"]) + prompt_cost += float(prompt_tokens_details["video_tokens"]) * prompt_base_cost + else: + prompt_cost += calculate_cost_component( + model_info, video_token_cost_key, prompt_tokens_details["video_tokens"] + ) ### CACHE WRITING COST - Now uses tiered pricing if ( diff --git a/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py b/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py index b156faf3ea6..b84cda71ea2 100644 --- a/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py +++ b/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py @@ -337,6 +337,87 @@ def test_video_input_tokens_gemini_omni_flash_preview(): ) +def test_audio_input_tokens_gemini_priced_at_text_rate(): + """Regression for LIT-4474: audio prompt tokens must fall back to the text input rate. + + gemini-3-pro-preview declares supports_audio_input but has no input_cost_per_audio_token, + so audio tokens were previously billed at $0. + """ + model = "gemini-3-pro-preview" + os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True" + litellm.model_cost = litellm.get_model_cost_map(url="") + + model_cost_map = litellm.model_cost[f"gemini/{model}"] + assert model_cost_map.get("input_cost_per_audio_token") is None + + text_tokens = 1000 + audio_tokens = 5000 + prompt_tokens = text_tokens + audio_tokens + usage = Usage( + completion_tokens=0, + prompt_tokens=prompt_tokens, + total_tokens=prompt_tokens, + prompt_tokens_details=PromptTokensDetailsWrapper( + text_tokens=text_tokens, + audio_tokens=audio_tokens, + ), + ) + + prompt_cost, _ = generic_cost_per_token( + model=model, + usage=usage, + custom_llm_provider="gemini", + ) + + assert round(prompt_cost, 10) == round( + model_cost_map["input_cost_per_token"] * prompt_tokens, + 10, + ) + assert round(prompt_cost, 10) != round( + model_cost_map["input_cost_per_token"] * text_tokens, + 10, + ) + + +def test_audio_video_input_tokens_gemini_use_above_200k_tier(): + """Regression for LIT-4474: audio/video fallbacks must honor the >200k long-context tier. + + Falling back to the raw un-tiered input_cost_per_token undercounts when the request crosses + the 200k boundary; audio and video should be priced at input_cost_per_token_above_200k_tokens + like text. + """ + model = "gemini-3-pro-preview" + os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True" + litellm.model_cost = litellm.get_model_cost_map(url="") + + model_cost_map = litellm.model_cost[f"gemini/{model}"] + hi_rate = model_cost_map["input_cost_per_token_above_200k_tokens"] + assert hi_rate != model_cost_map["input_cost_per_token"] + + text_tokens = 100000 + audio_tokens = 60000 + video_tokens = 80000 + prompt_tokens = text_tokens + audio_tokens + video_tokens + usage = Usage( + completion_tokens=0, + prompt_tokens=prompt_tokens, + total_tokens=prompt_tokens, + prompt_tokens_details=PromptTokensDetailsWrapper( + text_tokens=text_tokens, + audio_tokens=audio_tokens, + video_tokens=video_tokens, + ), + ) + + prompt_cost, _ = generic_cost_per_token( + model=model, + usage=usage, + custom_llm_provider="gemini", + ) + + assert round(prompt_cost, 10) == round(hi_rate * prompt_tokens, 10) + + def test_video_tokens_fallback_to_base_cost(): """Video output tokens fall back to the base output rate when output_cost_per_video_token is not set.""" from unittest.mock import patch