fix(pricing): correct cached-token fields on realtime cost-map entries

azure/gpt-realtime-2 was the only member of the gpt-realtime-2 family priced
on one side of its cached-audio meter. Azure publishes that meter as
"gpt-realtime-2 Audio cd inp Gl 1M Tokens" at 0.4 per 1M and charges the
same rate for the write that populates the cache and the read that hits it,
so cache_creation_input_audio_token_cost lands at 4e-07, matching
azure/gpt-realtime-2.1, azure/gpt-realtime-2.1-mini and the openai
gpt-realtime-2 entry. No cost path reads that field yet, so this corrects
what get_model_info reports rather than what anything bills.

The gemini Live entries go the other way. Google's Vertex context-caching
page publishes separate supported-model lists for implicit and explicit
caching, and no Live or native-audio model is in either one. Its pricing
page prints N/A in both cached-input columns for every Gemini 2.5 Flash
Live API row, where plain 2.5 Flash and 2.5 Flash-Lite both carry real
cached prices, and the Vertex model card for the family marks context
caching not supported outright. Vertex never reports cachedContentTokenCount
on a Live session either, including for a byte-identical 7,021-token prefix
replayed across sessions minutes apart, which is well past the 2,048-token
minimum the same page sets for the Gemini 2 family.

So the 7.5e-08 on the two preview siblings priced something the provider does
not sell, and supports_prompt_caching on all three claimed a capability the
model does not have. The rate comes out. The flag is set to false rather than
removed, because get_model_info maps an absent key to None, and None is how
this map spells "nobody checked" across the 2,788 entries that omit it, where
false records the vendor's documented no. Both readers of the flag gate on
`is True`, so nothing bills or behaves differently either way.

Only the cached fields change on the two 09-2025 preview entries. Their
source field points at the Gemini API pricing page rather than the Vertex
one, so they describe a different surface with its own published limits, and
their context windows are left alone rather than assumed to match the Vertex
model card that drives the GA entry.

Tests cover all three halves: the family invariant that a cached audio read
implies an equal cached audio write, a cached count on a Live entry leaving
the bill at the fresh-input total instead of adding the old 7.5e-08, and
supports_prompt_caching answering false for all three entries while still
answering true for 2.5 Flash, so the false cannot be a swallowed lookup
error.
This commit is contained in:
Marty Sullivan 2026-09-07 04:30:13 -04:00
parent 168a0055a2
commit 4db0efe3c7
3 changed files with 84 additions and 11 deletions

View file

@ -5422,6 +5422,7 @@
"supports_tool_choice": true
},
"azure/gpt-realtime-2": {
"cache_creation_input_audio_token_cost": 4e-07,
"cache_read_input_audio_token_cost": 4e-07,
"cache_read_input_token_cost": 4e-07,
"deprecation_date": "2026-08-31",
@ -23855,7 +23856,7 @@
"supports_function_calling": true,
"supports_parallel_function_calling": true,
"supports_pdf_input": true,
"supports_prompt_caching": true,
"supports_prompt_caching": false,
"supports_response_schema": true,
"supports_system_messages": true,
"supports_tool_choice": true,
@ -23871,7 +23872,6 @@
"input_cost_per_image_token": 3e-06
},
"gemini-live-2.5-flash-preview-native-audio-09-2025": {
"cache_read_input_token_cost": 7.5e-08,
"input_cost_per_audio_token": 3e-06,
"input_cost_per_token": 5e-07,
"litellm_provider": "vertex_ai-language-models",
@ -23900,7 +23900,7 @@
"supports_function_calling": true,
"supports_parallel_function_calling": true,
"supports_pdf_input": true,
"supports_prompt_caching": true,
"supports_prompt_caching": false,
"supports_response_schema": true,
"supports_system_messages": true,
"supports_tool_choice": true,
@ -23915,7 +23915,6 @@
"gemini_native_audio": true
},
"gemini/gemini-live-2.5-flash-preview-native-audio-09-2025": {
"cache_read_input_token_cost": 7.5e-08,
"input_cost_per_audio_token": 3e-06,
"input_cost_per_token": 5e-07,
"litellm_provider": "gemini",
@ -23945,7 +23944,7 @@
"supports_function_calling": true,
"supports_parallel_function_calling": true,
"supports_pdf_input": true,
"supports_prompt_caching": true,
"supports_prompt_caching": false,
"supports_response_schema": true,
"supports_system_messages": true,
"supports_tool_choice": true,

View file

@ -5422,6 +5422,7 @@
"supports_tool_choice": true
},
"azure/gpt-realtime-2": {
"cache_creation_input_audio_token_cost": 4e-07,
"cache_read_input_audio_token_cost": 4e-07,
"cache_read_input_token_cost": 4e-07,
"deprecation_date": "2026-08-31",
@ -23855,7 +23856,7 @@
"supports_function_calling": true,
"supports_parallel_function_calling": true,
"supports_pdf_input": true,
"supports_prompt_caching": true,
"supports_prompt_caching": false,
"supports_response_schema": true,
"supports_system_messages": true,
"supports_tool_choice": true,
@ -23871,7 +23872,6 @@
"input_cost_per_image_token": 3e-06
},
"gemini-live-2.5-flash-preview-native-audio-09-2025": {
"cache_read_input_token_cost": 7.5e-08,
"input_cost_per_audio_token": 3e-06,
"input_cost_per_token": 5e-07,
"litellm_provider": "vertex_ai-language-models",
@ -23900,7 +23900,7 @@
"supports_function_calling": true,
"supports_parallel_function_calling": true,
"supports_pdf_input": true,
"supports_prompt_caching": true,
"supports_prompt_caching": false,
"supports_response_schema": true,
"supports_system_messages": true,
"supports_tool_choice": true,
@ -23915,7 +23915,6 @@
"gemini_native_audio": true
},
"gemini/gemini-live-2.5-flash-preview-native-audio-09-2025": {
"cache_read_input_token_cost": 7.5e-08,
"input_cost_per_audio_token": 3e-06,
"input_cost_per_token": 5e-07,
"litellm_provider": "gemini",
@ -23945,7 +23944,7 @@
"supports_function_calling": true,
"supports_parallel_function_calling": true,
"supports_pdf_input": true,
"supports_prompt_caching": true,
"supports_prompt_caching": false,
"supports_response_schema": true,
"supports_system_messages": true,
"supports_tool_choice": true,

View file

@ -25,7 +25,7 @@ from litellm.types.utils import (
PromptTokensDetailsWrapper,
Usage,
)
from litellm.utils import TranscriptionResponse
from litellm.utils import TranscriptionResponse, supports_prompt_caching
@pytest.fixture
@ -288,6 +288,81 @@ def test_github_copilot_mai_code_1_flash_pricing(_local_model_cost_map, model):
assert completion_usd == pytest.approx(500 * 4.5e-06)
GPT_REALTIME_2_FAMILY: Final = (
"azure/gpt-realtime-2",
"azure/gpt-realtime-2.1",
"azure/gpt-realtime-2.1-mini",
"gpt-realtime-2",
"gpt-realtime-2.1",
"gpt-realtime-2.1-mini",
)
def test_gpt_realtime_2_family_prices_audio_cache_writes_and_reads_alike(_local_model_cost_map: None) -> None:
"""Azure publishes one cached-audio meter per gpt-realtime-2 deployment, charged at the same rate for
the write that populates the cache and the read that hits it. azure/gpt-realtime-2 carried only the
read side, so it was the one family member reporting no cache-creation audio price for a deployment
whose meter publishes one."""
audio_cache_rates: Final = {
model: (
litellm.model_cost[model].get("cache_read_input_audio_token_cost"),
litellm.model_cost[model].get("cache_creation_input_audio_token_cost"),
)
for model in GPT_REALTIME_2_FAMILY
}
assert all(read is not None and write == read for read, write in audio_cache_rates.values()), audio_cache_rates
assert audio_cache_rates["azure/gpt-realtime-2"] == (4e-07, 4e-07)
GEMINI_LIVE_NATIVE_AUDIO_CASES: Final = (
("gemini-live-2.5-flash-native-audio", "vertex_ai"),
("gemini-live-2.5-flash-preview-native-audio-09-2025", "vertex_ai"),
("gemini/gemini-live-2.5-flash-preview-native-audio-09-2025", "gemini"),
)
@pytest.mark.parametrize(("model", "provider"), GEMINI_LIVE_NATIVE_AUDIO_CASES)
def test_gemini_live_native_audio_carries_no_cached_input_rate(
_local_model_cost_map: None, model: str, provider: str
) -> None:
"""Google publishes no cached-input price for the Live API. Its pricing table prints N/A in both
cached columns for every Gemini 2.5 Flash Live API row, and no Live or native-audio model appears
under either implicit or explicit context caching. Two of these entries priced a cached read at
7.5e-08 regardless, which billed 0.008 here. With no invented rate the cached tokens drop out of
the bill, which is inert in practice because Vertex reports no cachedContentTokenCount on a Live
session."""
assert litellm.get_model_info(model, custom_llm_provider=provider)["cache_read_input_token_cost"] is None
prompt_usd, _ = cost_per_token(
model=model,
prompt_tokens=101_000,
completion_tokens=0,
custom_llm_provider=provider,
usage_object=Usage(
prompt_tokens=101_000,
completion_tokens=0,
prompt_tokens_details=PromptTokensDetailsWrapper(cached_tokens=100_000),
),
)
assert prompt_usd == pytest.approx(1_000 * 5e-07)
@pytest.mark.parametrize(("model", "provider"), GEMINI_LIVE_NATIVE_AUDIO_CASES)
def test_gemini_live_native_audio_declares_prompt_caching_unsupported(
_local_model_cost_map: None, model: str, provider: str
) -> None:
"""The capability claim is what made the absent cached rate read as a pricing gap rather than a
vendor limitation. The flag has to say False rather than go missing: get_model_info maps an absent
key to None, which is how this map spells "nobody checked", where False records the vendor's
documented no. The 2.5 Flash control is load-bearing because supports_prompt_caching turns any
lookup error into False, so without it a broken lookup would read as a pass."""
assert litellm.get_model_info(model, custom_llm_provider=provider)["supports_prompt_caching"] is False
assert supports_prompt_caching(model=model, custom_llm_provider=provider) is False
assert supports_prompt_caching(model="gemini-2.5-flash", custom_llm_provider="vertex_ai") is True
def test_cost_calculator_with_usage(_local_model_cost_map, monkeypatch):
usage = Usage(