From 8ab082e4acf5fcab0eaf87bb0deb65b665df6a1f Mon Sep 17 00:00:00 2001 From: Akhilesh Arora Date: Thu, 21 May 2026 18:00:30 +0200 Subject: [PATCH] fix(cost): apply tiered input rate to image tokens above threshold gemini-2.5-pro and other models charge 2x input_cost_per_token above 200k input tokens. Text tokens correctly pick up the tiered rate via prompt_base_cost (resolved in _get_token_base_cost). Image tokens did not. When input_cost_per_image_token is missing, _calculate_input_cost fell back to looking up input_cost_per_token from model_info via calculate_cost_component, which has no notion of _above__tokens and always returns the base rate. A 250k all-image-token gemini-2.5-pro request was billed 250_000 * 1.25e-6 = $0.3125 instead of 250_000 * 2.5e-6 = $0.625. Multiply by prompt_base_cost (the already-resolved tier rate) when the image-specific key is absent. Mirrors the treatment text tokens get on line 556. --- .../litellm_core_utils/llm_cost_calc/utils.py | 22 ++++--- .../llm_cost_calc/test_llm_cost_calc_utils.py | 61 +++++++++++++++++++ 2 files changed, 75 insertions(+), 8 deletions(-) diff --git a/litellm/litellm_core_utils/llm_cost_calc/utils.py b/litellm/litellm_core_utils/llm_cost_calc/utils.py index 59d0465e6d4..bdd90202ea6 100644 --- a/litellm/litellm_core_utils/llm_cost_calc/utils.py +++ b/litellm/litellm_core_utils/llm_cost_calc/utils.py @@ -569,14 +569,20 @@ def _calculate_input_cost( ### IMAGE TOKEN COST if prompt_tokens_details["image_tokens"]: - # For image token costs: - # First check if input_cost_per_image_token is available. If not, default to generic input_cost_per_token. - image_token_cost_key = "input_cost_per_image_token" - if model_info.get(image_token_cost_key) is None: - image_token_cost_key = "input_cost_per_token" - prompt_cost += calculate_cost_component( - model_info, image_token_cost_key, prompt_tokens_details["image_tokens"] - ) + # If input_cost_per_image_token is defined, use it directly. + # Otherwise charge at prompt_base_cost (the text rate already resolved + # for tiered keys like input_cost_per_token_above_200k_tokens), not the + # un-tiered input_cost_per_token from model_info. + if model_info.get("input_cost_per_image_token") is not None: + prompt_cost += calculate_cost_component( + model_info, + "input_cost_per_image_token", + prompt_tokens_details["image_tokens"], + ) + else: + prompt_cost += ( + float(prompt_tokens_details["image_tokens"]) * prompt_base_cost + ) ### CACHE WRITING COST - Now uses tiered pricing if ( diff --git a/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py b/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py index a7a2b7720d7..c5438176025 100644 --- a/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py +++ b/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py @@ -298,6 +298,67 @@ def test_generic_cost_per_token_above_200k_tokens(): ) +def test_input_image_tokens_use_custom_pricing_when_set(): + """When input_cost_per_image_token IS defined, it must be used in preference + to prompt_base_cost. Covers the `if` branch of the image fallback fix.""" + from unittest.mock import patch + + mock_model_info = { + "input_cost_per_token": 1e-6, + "output_cost_per_token": 2e-6, + "input_cost_per_image_token": 3e-6, + } + usage = Usage( + prompt_tokens=1000, + completion_tokens=0, + total_tokens=1000, + prompt_tokens_details=PromptTokensDetailsWrapper( + audio_tokens=None, cached_tokens=None, text_tokens=0, image_tokens=1000 + ), + ) + with patch( + "litellm.litellm_core_utils.llm_cost_calc.utils.get_model_info", + return_value=mock_model_info, + ): + prompt_cost, _ = generic_cost_per_token( + model="test-model", usage=usage, custom_llm_provider="gemini" + ) + assert round(prompt_cost, 12) == round(1000 * 3e-6, 12) + + +def test_image_tokens_above_200k_uses_tiered_rate(): + """ + gemini-2.5-pro charges 2x input_cost_per_token above 200k. The model_cost + entry has no input_cost_per_image_token, so image tokens above the + threshold must also use the above-200k rate, not the base rate. + """ + model = "gemini-2.5-pro" + custom_llm_provider = "vertex_ai" + os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True" + litellm.model_cost = litellm.get_model_cost_map(url="") + + model_cost_map = litellm.model_cost[model] + assert "input_cost_per_image_token" not in model_cost_map + above_200k_rate = model_cost_map["input_cost_per_token_above_200k_tokens"] + + prompt_tokens = 250_000 + usage = Usage( + prompt_tokens=prompt_tokens, + completion_tokens=0, + total_tokens=prompt_tokens, + prompt_tokens_details=PromptTokensDetailsWrapper( + audio_tokens=None, + cached_tokens=None, + text_tokens=0, + image_tokens=prompt_tokens, + ), + ) + prompt_cost, _ = generic_cost_per_token( + model=model, usage=usage, custom_llm_provider=custom_llm_provider + ) + assert round(prompt_cost, 10) == round(prompt_tokens * above_200k_rate, 10) + + def test_generic_cost_per_token_gpt54_above_272k_tokens(): """GPT-5.4/5.4-pro: prompts >272K input tokens priced at 2x input, 1.5x output.""" model = "gpt-5.4"