mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-04 02:31:27 +00:00
fix(cost): apply tiered input rate to image tokens above threshold
gemini-2.5-pro and other models charge 2x input_cost_per_token above 200k input tokens. Text tokens correctly pick up the tiered rate via prompt_base_cost (resolved in _get_token_base_cost). Image tokens did not. When input_cost_per_image_token is missing, _calculate_input_cost fell back to looking up input_cost_per_token from model_info via calculate_cost_component, which has no notion of _above_<X>_tokens and always returns the base rate. A 250k all-image-token gemini-2.5-pro request was billed 250_000 * 1.25e-6 = $0.3125 instead of 250_000 * 2.5e-6 = $0.625. Multiply by prompt_base_cost (the already-resolved tier rate) when the image-specific key is absent. Mirrors the treatment text tokens get on line 556.
This commit is contained in:
parent
697a90ea77
commit
8ab082e4ac
2 changed files with 75 additions and 8 deletions
|
|
@ -569,14 +569,20 @@ def _calculate_input_cost(
|
|||
|
||||
### IMAGE TOKEN COST
|
||||
if prompt_tokens_details["image_tokens"]:
|
||||
# For image token costs:
|
||||
# First check if input_cost_per_image_token is available. If not, default to generic input_cost_per_token.
|
||||
image_token_cost_key = "input_cost_per_image_token"
|
||||
if model_info.get(image_token_cost_key) is None:
|
||||
image_token_cost_key = "input_cost_per_token"
|
||||
prompt_cost += calculate_cost_component(
|
||||
model_info, image_token_cost_key, prompt_tokens_details["image_tokens"]
|
||||
)
|
||||
# If input_cost_per_image_token is defined, use it directly.
|
||||
# Otherwise charge at prompt_base_cost (the text rate already resolved
|
||||
# for tiered keys like input_cost_per_token_above_200k_tokens), not the
|
||||
# un-tiered input_cost_per_token from model_info.
|
||||
if model_info.get("input_cost_per_image_token") is not None:
|
||||
prompt_cost += calculate_cost_component(
|
||||
model_info,
|
||||
"input_cost_per_image_token",
|
||||
prompt_tokens_details["image_tokens"],
|
||||
)
|
||||
else:
|
||||
prompt_cost += (
|
||||
float(prompt_tokens_details["image_tokens"]) * prompt_base_cost
|
||||
)
|
||||
|
||||
### CACHE WRITING COST - Now uses tiered pricing
|
||||
if (
|
||||
|
|
|
|||
|
|
@ -298,6 +298,67 @@ def test_generic_cost_per_token_above_200k_tokens():
|
|||
)
|
||||
|
||||
|
||||
def test_input_image_tokens_use_custom_pricing_when_set():
|
||||
"""When input_cost_per_image_token IS defined, it must be used in preference
|
||||
to prompt_base_cost. Covers the `if` branch of the image fallback fix."""
|
||||
from unittest.mock import patch
|
||||
|
||||
mock_model_info = {
|
||||
"input_cost_per_token": 1e-6,
|
||||
"output_cost_per_token": 2e-6,
|
||||
"input_cost_per_image_token": 3e-6,
|
||||
}
|
||||
usage = Usage(
|
||||
prompt_tokens=1000,
|
||||
completion_tokens=0,
|
||||
total_tokens=1000,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(
|
||||
audio_tokens=None, cached_tokens=None, text_tokens=0, image_tokens=1000
|
||||
),
|
||||
)
|
||||
with patch(
|
||||
"litellm.litellm_core_utils.llm_cost_calc.utils.get_model_info",
|
||||
return_value=mock_model_info,
|
||||
):
|
||||
prompt_cost, _ = generic_cost_per_token(
|
||||
model="test-model", usage=usage, custom_llm_provider="gemini"
|
||||
)
|
||||
assert round(prompt_cost, 12) == round(1000 * 3e-6, 12)
|
||||
|
||||
|
||||
def test_image_tokens_above_200k_uses_tiered_rate():
|
||||
"""
|
||||
gemini-2.5-pro charges 2x input_cost_per_token above 200k. The model_cost
|
||||
entry has no input_cost_per_image_token, so image tokens above the
|
||||
threshold must also use the above-200k rate, not the base rate.
|
||||
"""
|
||||
model = "gemini-2.5-pro"
|
||||
custom_llm_provider = "vertex_ai"
|
||||
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
|
||||
litellm.model_cost = litellm.get_model_cost_map(url="")
|
||||
|
||||
model_cost_map = litellm.model_cost[model]
|
||||
assert "input_cost_per_image_token" not in model_cost_map
|
||||
above_200k_rate = model_cost_map["input_cost_per_token_above_200k_tokens"]
|
||||
|
||||
prompt_tokens = 250_000
|
||||
usage = Usage(
|
||||
prompt_tokens=prompt_tokens,
|
||||
completion_tokens=0,
|
||||
total_tokens=prompt_tokens,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(
|
||||
audio_tokens=None,
|
||||
cached_tokens=None,
|
||||
text_tokens=0,
|
||||
image_tokens=prompt_tokens,
|
||||
),
|
||||
)
|
||||
prompt_cost, _ = generic_cost_per_token(
|
||||
model=model, usage=usage, custom_llm_provider=custom_llm_provider
|
||||
)
|
||||
assert round(prompt_cost, 10) == round(prompt_tokens * above_200k_rate, 10)
|
||||
|
||||
|
||||
def test_generic_cost_per_token_gpt54_above_272k_tokens():
|
||||
"""GPT-5.4/5.4-pro: prompts >272K input tokens priced at 2x input, 1.5x output."""
|
||||
model = "gpt-5.4"
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue