mirror of
https://github.com/BerriAI/litellm.git
synced 2026-08-28 05:25:59 +00:00
fix(cost-map): correct gemini-live native-audio text input rate
This commit is contained in:
parent
2548e960f1
commit
d565860f60
3 changed files with 21 additions and 11 deletions
|
|
@ -20355,7 +20355,7 @@
|
|||
"gemini-live-2.5-flash-preview-native-audio-09-2025": {
|
||||
"cache_read_input_token_cost": 7.5e-08,
|
||||
"input_cost_per_audio_token": 3e-06,
|
||||
"input_cost_per_token": 3e-07,
|
||||
"input_cost_per_token": 5e-07,
|
||||
"litellm_provider": "vertex_ai-language-models",
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 65535,
|
||||
|
|
@ -20399,7 +20399,7 @@
|
|||
"gemini/gemini-live-2.5-flash-preview-native-audio-09-2025": {
|
||||
"cache_read_input_token_cost": 7.5e-08,
|
||||
"input_cost_per_audio_token": 3e-06,
|
||||
"input_cost_per_token": 3e-07,
|
||||
"input_cost_per_token": 5e-07,
|
||||
"litellm_provider": "gemini",
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 65535,
|
||||
|
|
|
|||
|
|
@ -20355,7 +20355,7 @@
|
|||
"gemini-live-2.5-flash-preview-native-audio-09-2025": {
|
||||
"cache_read_input_token_cost": 7.5e-08,
|
||||
"input_cost_per_audio_token": 3e-06,
|
||||
"input_cost_per_token": 3e-07,
|
||||
"input_cost_per_token": 5e-07,
|
||||
"litellm_provider": "vertex_ai-language-models",
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 65535,
|
||||
|
|
@ -20399,7 +20399,7 @@
|
|||
"gemini/gemini-live-2.5-flash-preview-native-audio-09-2025": {
|
||||
"cache_read_input_token_cost": 7.5e-08,
|
||||
"input_cost_per_audio_token": 3e-06,
|
||||
"input_cost_per_token": 3e-07,
|
||||
"input_cost_per_token": 5e-07,
|
||||
"litellm_provider": "gemini",
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 65535,
|
||||
|
|
|
|||
|
|
@ -20,6 +20,11 @@ NATIVE_AUDIO_KEYS: Final = tuple(
|
|||
for suffix in ("latest", "preview-09-2025", "preview-12-2025")
|
||||
)
|
||||
|
||||
LIVE_NATIVE_AUDIO_KEYS: Final = (
|
||||
"gemini-live-2.5-flash-preview-native-audio-09-2025",
|
||||
"gemini/gemini-live-2.5-flash-preview-native-audio-09-2025",
|
||||
)
|
||||
|
||||
FLASH_TTS_INPUT: Final = 5e-07
|
||||
FLASH_TTS_AUDIO_OUTPUT: Final = 1e-05
|
||||
PRO_TTS_INPUT: Final = 1e-06
|
||||
|
|
@ -45,10 +50,15 @@ PUBLISHED_RATES: Final = {
|
|||
"output_cost_per_token": NATIVE_AUDIO_TEXT_OUTPUT,
|
||||
"output_cost_per_audio_token": NATIVE_AUDIO_AUDIO_OUTPUT,
|
||||
}
|
||||
for key in NATIVE_AUDIO_KEYS
|
||||
for key in (*NATIVE_AUDIO_KEYS, *LIVE_NATIVE_AUDIO_KEYS)
|
||||
},
|
||||
}
|
||||
ALL_KEYS: Final = tuple(PUBLISHED_RATES)
|
||||
NATIVE_AUDIO_BILLING_CASES: Final = (
|
||||
*((key, "gemini") for key in NATIVE_AUDIO_KEYS),
|
||||
("gemini-live-2.5-flash-preview-native-audio-09-2025", "vertex_ai"),
|
||||
("gemini/gemini-live-2.5-flash-preview-native-audio-09-2025", "gemini"),
|
||||
)
|
||||
LONG_CONTEXT_TIER_FIELDS: Final = (
|
||||
"input_cost_per_token_above_200k_tokens",
|
||||
"output_cost_per_token_above_200k_tokens",
|
||||
|
|
@ -118,8 +128,8 @@ def test_tts_audio_output_is_billed_at_the_audio_rate(
|
|||
assert completion_cost == pytest.approx(49 * audio_output_rate)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", NATIVE_AUDIO_KEYS)
|
||||
def test_native_audio_output_is_billed_at_the_audio_rate(model: str, local_model_cost_map):
|
||||
@pytest.mark.parametrize("model, provider", NATIVE_AUDIO_BILLING_CASES)
|
||||
def test_native_audio_output_is_billed_at_the_audio_rate(model: str, provider: str, local_model_cost_map):
|
||||
usage: Final = Usage(
|
||||
prompt_tokens=377,
|
||||
completion_tokens=84,
|
||||
|
|
@ -127,18 +137,18 @@ def test_native_audio_output_is_billed_at_the_audio_rate(model: str, local_model
|
|||
prompt_tokens_details=PromptTokensDetailsWrapper(text_tokens=377),
|
||||
completion_tokens_details=CompletionTokensDetailsWrapper(audio_tokens=48, reasoning_tokens=36, text_tokens=0),
|
||||
)
|
||||
prompt_cost, completion_cost = generic_cost_per_token(model=model, usage=usage, custom_llm_provider="gemini")
|
||||
prompt_cost, completion_cost = generic_cost_per_token(model=model, usage=usage, custom_llm_provider=provider)
|
||||
assert prompt_cost == pytest.approx(377 * NATIVE_AUDIO_TEXT_INPUT)
|
||||
assert completion_cost == pytest.approx(48 * NATIVE_AUDIO_AUDIO_OUTPUT + 36 * NATIVE_AUDIO_TEXT_OUTPUT)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", NATIVE_AUDIO_KEYS)
|
||||
def test_native_audio_input_is_billed_at_the_audio_rate(model: str, local_model_cost_map):
|
||||
@pytest.mark.parametrize("model, provider", NATIVE_AUDIO_BILLING_CASES)
|
||||
def test_native_audio_input_is_billed_at_the_audio_rate(model: str, provider: str, local_model_cost_map):
|
||||
usage: Final = Usage(
|
||||
prompt_tokens=1000,
|
||||
completion_tokens=0,
|
||||
total_tokens=1000,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(text_tokens=100, audio_tokens=900),
|
||||
)
|
||||
prompt_cost, _ = generic_cost_per_token(model=model, usage=usage, custom_llm_provider="gemini")
|
||||
prompt_cost, _ = generic_cost_per_token(model=model, usage=usage, custom_llm_provider=provider)
|
||||
assert prompt_cost == pytest.approx(100 * NATIVE_AUDIO_TEXT_INPUT + 900 * NATIVE_AUDIO_AUDIO_INPUT)
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue