mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-06 02:48:13 +00:00
fix(cost): bill audio input tokens at base rate when model has no audio price
Audio prompt tokens are reported separately from text tokens and subtracted from the text bucket before pricing. When a model ships without input_cost_per_audio_token (37 shipped models including gemini-2.5-pro, gemini-3-pro-preview, and xai grok-4-1-fast), those audio tokens were billed at $0 instead of falling back to input_cost_per_token. Image and video input tokens already fall back to the base rate, and output audio falls back to the base completion rate; input audio was the only modality that silently zeroed.
This commit is contained in:
parent
d2d158f271
commit
3e9bd6db8a
2 changed files with 66 additions and 1 deletions
|
|
@ -662,7 +662,11 @@ def _calculate_input_cost(
|
|||
|
||||
### AUDIO COST
|
||||
if prompt_tokens_details["audio_tokens"]:
|
||||
audio_cost_key: Final = _get_service_tier_cost_key("input_cost_per_audio_token", service_tier)
|
||||
tier_audio_cost_key: Final = _get_service_tier_cost_key("input_cost_per_audio_token", service_tier)
|
||||
has_audio_price: Final = (
|
||||
model_info.get(tier_audio_cost_key) is not None or model_info.get("input_cost_per_audio_token") is not None
|
||||
)
|
||||
audio_cost_key: Final = tier_audio_cost_key if has_audio_price else "input_cost_per_token"
|
||||
prompt_cost += calculate_cost_component(model_info, audio_cost_key, prompt_tokens_details["audio_tokens"])
|
||||
|
||||
### IMAGE TOKEN COST
|
||||
|
|
|
|||
|
|
@ -1449,6 +1449,67 @@ def test_string_cost_values_edge_cases():
|
|||
assert round(completion_cost, 12) == round(expected_completion_cost, 12)
|
||||
|
||||
|
||||
def test_audio_input_tokens_fall_back_to_base_rate_when_no_audio_price():
|
||||
"""Audio input tokens must bill at input_cost_per_token when the model has no audio rate.
|
||||
|
||||
Regression: models like gemini-2.5-pro report audio prompt tokens separately from text
|
||||
but ship without input_cost_per_audio_token, which previously billed those tokens at $0.
|
||||
"""
|
||||
from unittest.mock import patch
|
||||
|
||||
mock_model_info = {
|
||||
"input_cost_per_token": 1.25e-6,
|
||||
"output_cost_per_token": 1e-5,
|
||||
}
|
||||
|
||||
usage = Usage(
|
||||
prompt_tokens=9700,
|
||||
completion_tokens=0,
|
||||
total_tokens=9700,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(
|
||||
audio_tokens=9600, cached_tokens=0, text_tokens=100, image_tokens=None
|
||||
),
|
||||
)
|
||||
|
||||
with patch(
|
||||
"litellm.litellm_core_utils.llm_cost_calc.utils.get_model_info",
|
||||
return_value=mock_model_info,
|
||||
):
|
||||
prompt_cost, _ = generic_cost_per_token(model="gemini-2.5-pro", usage=usage, custom_llm_provider="vertex_ai")
|
||||
|
||||
expected_prompt_cost = (100 + 9600) * 1.25e-6
|
||||
assert round(prompt_cost, 12) == round(expected_prompt_cost, 12)
|
||||
|
||||
|
||||
def test_audio_input_tokens_use_audio_rate_when_present():
|
||||
"""When input_cost_per_audio_token exists it takes precedence over the base rate."""
|
||||
from unittest.mock import patch
|
||||
|
||||
mock_model_info = {
|
||||
"input_cost_per_token": 1.25e-6,
|
||||
"input_cost_per_audio_token": 5e-6,
|
||||
"output_cost_per_token": 1e-5,
|
||||
}
|
||||
|
||||
usage = Usage(
|
||||
prompt_tokens=9700,
|
||||
completion_tokens=0,
|
||||
total_tokens=9700,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(
|
||||
audio_tokens=9600, cached_tokens=0, text_tokens=100, image_tokens=None
|
||||
),
|
||||
)
|
||||
|
||||
with patch(
|
||||
"litellm.litellm_core_utils.llm_cost_calc.utils.get_model_info",
|
||||
return_value=mock_model_info,
|
||||
):
|
||||
prompt_cost, _ = generic_cost_per_token(model="gemini-2.5-pro", usage=usage, custom_llm_provider="vertex_ai")
|
||||
|
||||
expected_prompt_cost = 100 * 1.25e-6 + 9600 * 5e-6
|
||||
assert round(prompt_cost, 12) == round(expected_prompt_cost, 12)
|
||||
|
||||
|
||||
def test_string_cost_values_with_threshold():
|
||||
"""Test that string cost values work correctly with threshold pricing."""
|
||||
from unittest.mock import patch
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue