fix(cost): correct gemini-live-2.5-flash-native-audio limits and capabilities

Google's model card for model ID gemini-live-2.5-flash-native-audio gives a
128K context window and 64K maximum output tokens, and marks structured
output, context caching and URL context as not supported. Its modality list
is text in and out, image in, audio in and out, and video in, with no
document input of any kind.

The entry advertised a 1M context window, an off-by-one 65535 output cap, and
three capability flags the vendor marks unsupported. Context caching is the
fourth and is handled in the cached-fields change alongside its two preview
siblings.

Both the bare id and vertex_ai/gemini-live-2.5-flash-native-audio resolve to
this single entry, so the test drives the corrected values through both.
This commit is contained in:
Marty Sullivan 2026-09-07 04:33:28 -04:00
parent 4db0efe3c7
commit 9058a84192
3 changed files with 37 additions and 12 deletions

View file

@ -23830,9 +23830,9 @@
"input_cost_per_audio_token": 3e-06,
"input_cost_per_token": 5e-07,
"litellm_provider": "vertex_ai-language-models",
"max_input_tokens": 1048576,
"max_output_tokens": 65535,
"max_tokens": 65535,
"max_input_tokens": 131072,
"max_output_tokens": 65536,
"max_tokens": 65536,
"mode": "realtime",
"output_cost_per_audio_token": 1.2e-05,
"output_cost_per_token": 2e-06,
@ -23855,12 +23855,12 @@
"supports_audio_output": true,
"supports_function_calling": true,
"supports_parallel_function_calling": true,
"supports_pdf_input": true,
"supports_pdf_input": false,
"supports_prompt_caching": false,
"supports_response_schema": true,
"supports_response_schema": false,
"supports_system_messages": true,
"supports_tool_choice": true,
"supports_url_context": true,
"supports_url_context": false,
"supports_vision": true,
"supports_web_search": true,
"search_context_cost_per_query": {

View file

@ -23830,9 +23830,9 @@
"input_cost_per_audio_token": 3e-06,
"input_cost_per_token": 5e-07,
"litellm_provider": "vertex_ai-language-models",
"max_input_tokens": 1048576,
"max_output_tokens": 65535,
"max_tokens": 65535,
"max_input_tokens": 131072,
"max_output_tokens": 65536,
"max_tokens": 65536,
"mode": "realtime",
"output_cost_per_audio_token": 1.2e-05,
"output_cost_per_token": 2e-06,
@ -23855,12 +23855,12 @@
"supports_audio_output": true,
"supports_function_calling": true,
"supports_parallel_function_calling": true,
"supports_pdf_input": true,
"supports_pdf_input": false,
"supports_prompt_caching": false,
"supports_response_schema": true,
"supports_response_schema": false,
"supports_system_messages": true,
"supports_tool_choice": true,
"supports_url_context": true,
"supports_url_context": false,
"supports_vision": true,
"supports_web_search": true,
"search_context_cost_per_query": {

View file

@ -4453,6 +4453,31 @@ def test_gemini_live_native_audio_ga_realtime_cost(_local_model_cost_map: None)
assert cost == pytest.approx(expected_cost, rel=1e-9)
@pytest.mark.parametrize(
"model",
["gemini-live-2.5-flash-native-audio", "vertex_ai/gemini-live-2.5-flash-native-audio"],
)
def test_gemini_live_native_audio_limits_and_capabilities_match_vendor_model_card(
_local_model_cost_map: None, model: str
) -> None:
"""Google's card for model ID gemini-live-2.5-flash-native-audio gives a 128K context window and
64K maximum output tokens, and marks structured output and URL context as not supported. Its
modality list is text, image, audio and video, with no document input, so pdf input cannot be
advertised either. The entry claimed a 1M window, an off-by-one 65535 output cap, and all three
capabilities.
Both ids resolve to the one bare entry, which is why no vertex_ai/-prefixed twin is needed.
"""
info = litellm.get_model_info(model)
assert info["max_input_tokens"] == 131072
assert info["max_output_tokens"] == 65536
assert info["max_tokens"] == 65536
assert info["supports_response_schema"] is False
assert info["supports_url_context"] is False
assert info["supports_pdf_input"] is False
@pytest.mark.parametrize(
"priceless_entry",
[