From 4db0efe3c759110db20995ad788dbf3bf1aed63d Mon Sep 17 00:00:00 2001 From: Marty Sullivan Date: Mon, 7 Sep 2026 04:30:13 -0400 Subject: [PATCH] fix(pricing): correct cached-token fields on realtime cost-map entries azure/gpt-realtime-2 was the only member of the gpt-realtime-2 family priced on one side of its cached-audio meter. Azure publishes that meter as "gpt-realtime-2 Audio cd inp Gl 1M Tokens" at 0.4 per 1M and charges the same rate for the write that populates the cache and the read that hits it, so cache_creation_input_audio_token_cost lands at 4e-07, matching azure/gpt-realtime-2.1, azure/gpt-realtime-2.1-mini and the openai gpt-realtime-2 entry. No cost path reads that field yet, so this corrects what get_model_info reports rather than what anything bills. The gemini Live entries go the other way. Google's Vertex context-caching page publishes separate supported-model lists for implicit and explicit caching, and no Live or native-audio model is in either one. Its pricing page prints N/A in both cached-input columns for every Gemini 2.5 Flash Live API row, where plain 2.5 Flash and 2.5 Flash-Lite both carry real cached prices, and the Vertex model card for the family marks context caching not supported outright. Vertex never reports cachedContentTokenCount on a Live session either, including for a byte-identical 7,021-token prefix replayed across sessions minutes apart, which is well past the 2,048-token minimum the same page sets for the Gemini 2 family. So the 7.5e-08 on the two preview siblings priced something the provider does not sell, and supports_prompt_caching on all three claimed a capability the model does not have. The rate comes out. The flag is set to false rather than removed, because get_model_info maps an absent key to None, and None is how this map spells "nobody checked" across the 2,788 entries that omit it, where false records the vendor's documented no. Both readers of the flag gate on `is True`, so nothing bills or behaves differently either way. Only the cached fields change on the two 09-2025 preview entries. Their source field points at the Gemini API pricing page rather than the Vertex one, so they describe a different surface with its own published limits, and their context windows are left alone rather than assumed to match the Vertex model card that drives the GA entry. Tests cover all three halves: the family invariant that a cached audio read implies an equal cached audio write, a cached count on a Live entry leaving the bill at the fresh-input total instead of adding the old 7.5e-08, and supports_prompt_caching answering false for all three entries while still answering true for 2.5 Flash, so the false cannot be a swallowed lookup error. --- ...odel_prices_and_context_window_backup.json | 9 +-- model_prices_and_context_window.json | 9 +-- tests/test_litellm/test_cost_calculator.py | 77 ++++++++++++++++++- 3 files changed, 84 insertions(+), 11 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 6c6e65927f9..4c2428b5e15 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -5422,6 +5422,7 @@ "supports_tool_choice": true }, "azure/gpt-realtime-2": { + "cache_creation_input_audio_token_cost": 4e-07, "cache_read_input_audio_token_cost": 4e-07, "cache_read_input_token_cost": 4e-07, "deprecation_date": "2026-08-31", @@ -23855,7 +23856,7 @@ "supports_function_calling": true, "supports_parallel_function_calling": true, "supports_pdf_input": true, - "supports_prompt_caching": true, + "supports_prompt_caching": false, "supports_response_schema": true, "supports_system_messages": true, "supports_tool_choice": true, @@ -23871,7 +23872,6 @@ "input_cost_per_image_token": 3e-06 }, "gemini-live-2.5-flash-preview-native-audio-09-2025": { - "cache_read_input_token_cost": 7.5e-08, "input_cost_per_audio_token": 3e-06, "input_cost_per_token": 5e-07, "litellm_provider": "vertex_ai-language-models", @@ -23900,7 +23900,7 @@ "supports_function_calling": true, "supports_parallel_function_calling": true, "supports_pdf_input": true, - "supports_prompt_caching": true, + "supports_prompt_caching": false, "supports_response_schema": true, "supports_system_messages": true, "supports_tool_choice": true, @@ -23915,7 +23915,6 @@ "gemini_native_audio": true }, "gemini/gemini-live-2.5-flash-preview-native-audio-09-2025": { - "cache_read_input_token_cost": 7.5e-08, "input_cost_per_audio_token": 3e-06, "input_cost_per_token": 5e-07, "litellm_provider": "gemini", @@ -23945,7 +23944,7 @@ "supports_function_calling": true, "supports_parallel_function_calling": true, "supports_pdf_input": true, - "supports_prompt_caching": true, + "supports_prompt_caching": false, "supports_response_schema": true, "supports_system_messages": true, "supports_tool_choice": true, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 6c6e65927f9..4c2428b5e15 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -5422,6 +5422,7 @@ "supports_tool_choice": true }, "azure/gpt-realtime-2": { + "cache_creation_input_audio_token_cost": 4e-07, "cache_read_input_audio_token_cost": 4e-07, "cache_read_input_token_cost": 4e-07, "deprecation_date": "2026-08-31", @@ -23855,7 +23856,7 @@ "supports_function_calling": true, "supports_parallel_function_calling": true, "supports_pdf_input": true, - "supports_prompt_caching": true, + "supports_prompt_caching": false, "supports_response_schema": true, "supports_system_messages": true, "supports_tool_choice": true, @@ -23871,7 +23872,6 @@ "input_cost_per_image_token": 3e-06 }, "gemini-live-2.5-flash-preview-native-audio-09-2025": { - "cache_read_input_token_cost": 7.5e-08, "input_cost_per_audio_token": 3e-06, "input_cost_per_token": 5e-07, "litellm_provider": "vertex_ai-language-models", @@ -23900,7 +23900,7 @@ "supports_function_calling": true, "supports_parallel_function_calling": true, "supports_pdf_input": true, - "supports_prompt_caching": true, + "supports_prompt_caching": false, "supports_response_schema": true, "supports_system_messages": true, "supports_tool_choice": true, @@ -23915,7 +23915,6 @@ "gemini_native_audio": true }, "gemini/gemini-live-2.5-flash-preview-native-audio-09-2025": { - "cache_read_input_token_cost": 7.5e-08, "input_cost_per_audio_token": 3e-06, "input_cost_per_token": 5e-07, "litellm_provider": "gemini", @@ -23945,7 +23944,7 @@ "supports_function_calling": true, "supports_parallel_function_calling": true, "supports_pdf_input": true, - "supports_prompt_caching": true, + "supports_prompt_caching": false, "supports_response_schema": true, "supports_system_messages": true, "supports_tool_choice": true, diff --git a/tests/test_litellm/test_cost_calculator.py b/tests/test_litellm/test_cost_calculator.py index 1945e5ffac5..d25a08f6da8 100644 --- a/tests/test_litellm/test_cost_calculator.py +++ b/tests/test_litellm/test_cost_calculator.py @@ -25,7 +25,7 @@ from litellm.types.utils import ( PromptTokensDetailsWrapper, Usage, ) -from litellm.utils import TranscriptionResponse +from litellm.utils import TranscriptionResponse, supports_prompt_caching @pytest.fixture @@ -288,6 +288,81 @@ def test_github_copilot_mai_code_1_flash_pricing(_local_model_cost_map, model): assert completion_usd == pytest.approx(500 * 4.5e-06) +GPT_REALTIME_2_FAMILY: Final = ( + "azure/gpt-realtime-2", + "azure/gpt-realtime-2.1", + "azure/gpt-realtime-2.1-mini", + "gpt-realtime-2", + "gpt-realtime-2.1", + "gpt-realtime-2.1-mini", +) + + +def test_gpt_realtime_2_family_prices_audio_cache_writes_and_reads_alike(_local_model_cost_map: None) -> None: + """Azure publishes one cached-audio meter per gpt-realtime-2 deployment, charged at the same rate for + the write that populates the cache and the read that hits it. azure/gpt-realtime-2 carried only the + read side, so it was the one family member reporting no cache-creation audio price for a deployment + whose meter publishes one.""" + audio_cache_rates: Final = { + model: ( + litellm.model_cost[model].get("cache_read_input_audio_token_cost"), + litellm.model_cost[model].get("cache_creation_input_audio_token_cost"), + ) + for model in GPT_REALTIME_2_FAMILY + } + + assert all(read is not None and write == read for read, write in audio_cache_rates.values()), audio_cache_rates + assert audio_cache_rates["azure/gpt-realtime-2"] == (4e-07, 4e-07) + + +GEMINI_LIVE_NATIVE_AUDIO_CASES: Final = ( + ("gemini-live-2.5-flash-native-audio", "vertex_ai"), + ("gemini-live-2.5-flash-preview-native-audio-09-2025", "vertex_ai"), + ("gemini/gemini-live-2.5-flash-preview-native-audio-09-2025", "gemini"), +) + + +@pytest.mark.parametrize(("model", "provider"), GEMINI_LIVE_NATIVE_AUDIO_CASES) +def test_gemini_live_native_audio_carries_no_cached_input_rate( + _local_model_cost_map: None, model: str, provider: str +) -> None: + """Google publishes no cached-input price for the Live API. Its pricing table prints N/A in both + cached columns for every Gemini 2.5 Flash Live API row, and no Live or native-audio model appears + under either implicit or explicit context caching. Two of these entries priced a cached read at + 7.5e-08 regardless, which billed 0.008 here. With no invented rate the cached tokens drop out of + the bill, which is inert in practice because Vertex reports no cachedContentTokenCount on a Live + session.""" + assert litellm.get_model_info(model, custom_llm_provider=provider)["cache_read_input_token_cost"] is None + + prompt_usd, _ = cost_per_token( + model=model, + prompt_tokens=101_000, + completion_tokens=0, + custom_llm_provider=provider, + usage_object=Usage( + prompt_tokens=101_000, + completion_tokens=0, + prompt_tokens_details=PromptTokensDetailsWrapper(cached_tokens=100_000), + ), + ) + + assert prompt_usd == pytest.approx(1_000 * 5e-07) + + +@pytest.mark.parametrize(("model", "provider"), GEMINI_LIVE_NATIVE_AUDIO_CASES) +def test_gemini_live_native_audio_declares_prompt_caching_unsupported( + _local_model_cost_map: None, model: str, provider: str +) -> None: + """The capability claim is what made the absent cached rate read as a pricing gap rather than a + vendor limitation. The flag has to say False rather than go missing: get_model_info maps an absent + key to None, which is how this map spells "nobody checked", where False records the vendor's + documented no. The 2.5 Flash control is load-bearing because supports_prompt_caching turns any + lookup error into False, so without it a broken lookup would read as a pass.""" + assert litellm.get_model_info(model, custom_llm_provider=provider)["supports_prompt_caching"] is False + assert supports_prompt_caching(model=model, custom_llm_provider=provider) is False + assert supports_prompt_caching(model="gemini-2.5-flash", custom_llm_provider="vertex_ai") is True + + def test_cost_calculator_with_usage(_local_model_cost_map, monkeypatch): usage = Usage(