From 9058a84192e1e4fab2f7da27b7a2ccb24a6b989d Mon Sep 17 00:00:00 2001 From: Marty Sullivan Date: Mon, 7 Sep 2026 04:33:28 -0400 Subject: [PATCH] fix(cost): correct gemini-live-2.5-flash-native-audio limits and capabilities Google's model card for model ID gemini-live-2.5-flash-native-audio gives a 128K context window and 64K maximum output tokens, and marks structured output, context caching and URL context as not supported. Its modality list is text in and out, image in, audio in and out, and video in, with no document input of any kind. The entry advertised a 1M context window, an off-by-one 65535 output cap, and three capability flags the vendor marks unsupported. Context caching is the fourth and is handled in the cached-fields change alongside its two preview siblings. Both the bare id and vertex_ai/gemini-live-2.5-flash-native-audio resolve to this single entry, so the test drives the corrected values through both. --- ...odel_prices_and_context_window_backup.json | 12 ++++----- model_prices_and_context_window.json | 12 ++++----- tests/test_litellm/test_cost_calculator.py | 25 +++++++++++++++++++ 3 files changed, 37 insertions(+), 12 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 4c2428b5e15..1faadebc4e6 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -23830,9 +23830,9 @@ "input_cost_per_audio_token": 3e-06, "input_cost_per_token": 5e-07, "litellm_provider": "vertex_ai-language-models", - "max_input_tokens": 1048576, - "max_output_tokens": 65535, - "max_tokens": 65535, + "max_input_tokens": 131072, + "max_output_tokens": 65536, + "max_tokens": 65536, "mode": "realtime", "output_cost_per_audio_token": 1.2e-05, "output_cost_per_token": 2e-06, @@ -23855,12 +23855,12 @@ "supports_audio_output": true, "supports_function_calling": true, "supports_parallel_function_calling": true, - "supports_pdf_input": true, + "supports_pdf_input": false, "supports_prompt_caching": false, - "supports_response_schema": true, + "supports_response_schema": false, "supports_system_messages": true, "supports_tool_choice": true, - "supports_url_context": true, + "supports_url_context": false, "supports_vision": true, "supports_web_search": true, "search_context_cost_per_query": { diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 4c2428b5e15..1faadebc4e6 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -23830,9 +23830,9 @@ "input_cost_per_audio_token": 3e-06, "input_cost_per_token": 5e-07, "litellm_provider": "vertex_ai-language-models", - "max_input_tokens": 1048576, - "max_output_tokens": 65535, - "max_tokens": 65535, + "max_input_tokens": 131072, + "max_output_tokens": 65536, + "max_tokens": 65536, "mode": "realtime", "output_cost_per_audio_token": 1.2e-05, "output_cost_per_token": 2e-06, @@ -23855,12 +23855,12 @@ "supports_audio_output": true, "supports_function_calling": true, "supports_parallel_function_calling": true, - "supports_pdf_input": true, + "supports_pdf_input": false, "supports_prompt_caching": false, - "supports_response_schema": true, + "supports_response_schema": false, "supports_system_messages": true, "supports_tool_choice": true, - "supports_url_context": true, + "supports_url_context": false, "supports_vision": true, "supports_web_search": true, "search_context_cost_per_query": { diff --git a/tests/test_litellm/test_cost_calculator.py b/tests/test_litellm/test_cost_calculator.py index d25a08f6da8..4c440524431 100644 --- a/tests/test_litellm/test_cost_calculator.py +++ b/tests/test_litellm/test_cost_calculator.py @@ -4453,6 +4453,31 @@ def test_gemini_live_native_audio_ga_realtime_cost(_local_model_cost_map: None) assert cost == pytest.approx(expected_cost, rel=1e-9) +@pytest.mark.parametrize( + "model", + ["gemini-live-2.5-flash-native-audio", "vertex_ai/gemini-live-2.5-flash-native-audio"], +) +def test_gemini_live_native_audio_limits_and_capabilities_match_vendor_model_card( + _local_model_cost_map: None, model: str +) -> None: + """Google's card for model ID gemini-live-2.5-flash-native-audio gives a 128K context window and + 64K maximum output tokens, and marks structured output and URL context as not supported. Its + modality list is text, image, audio and video, with no document input, so pdf input cannot be + advertised either. The entry claimed a 1M window, an off-by-one 65535 output cap, and all three + capabilities. + + Both ids resolve to the one bare entry, which is why no vertex_ai/-prefixed twin is needed. + """ + info = litellm.get_model_info(model) + + assert info["max_input_tokens"] == 131072 + assert info["max_output_tokens"] == 65536 + assert info["max_tokens"] == 65536 + assert info["supports_response_schema"] is False + assert info["supports_url_context"] is False + assert info["supports_pdf_input"] is False + + @pytest.mark.parametrize( "priceless_entry", [