fix(realtime): reconcile model metadata and test contracts

This commit is contained in:
Emerson Gomes 2026-09-22 18:27:12 -05:00
parent ea84863217
commit 257c7f2e0a
No known key found for this signature in database
GPG key ID: D3DF28AB5D1B5E17
11 changed files with 1012 additions and 97 deletions

View file

@ -6533,9 +6533,9 @@ class BaseLLMHTTPHandler:
try:
configured_client: Final = openai_client.with_options(
timeout=timeout,
set_default_headers={ # mutable-ok: OpenAI SDK accepts a mutable custom-header mapping
key: str(value) for key, value in (extra_headers or {}).items()
},
set_default_headers=MappingProxyType(
{key: str(value) for key, value in (extra_headers or MappingProxyType({})).items()}
),
)
raw_response: Final = await configured_client.post(
"/realtime/translations/client_secrets",
@ -6669,9 +6669,9 @@ class BaseLLMHTTPHandler:
timeout=timeout,
set_default_headers={ # mutable-ok: OpenAI SDK accepts a mutable custom-header mapping
"Content-Type": "application/sdp",
**{ # mutable-ok: caller headers are normalized into the SDK header mapping
key: str(value) for key, value in (extra_headers or {}).items()
},
**MappingProxyType(
{key: str(value) for key, value in (extra_headers or MappingProxyType({})).items()}
),
},
)
translation_response: Final = await configured_client.post(
@ -6684,9 +6684,9 @@ class BaseLLMHTTPHandler:
RealtimeSessionCreateRequestParam,
session_data,
)
sdk_extra_headers: Final = { # mutable-ok: OpenAI SDK accepts a mutable custom-header mapping
key: str(value) for key, value in (extra_headers or {}).items()
}
sdk_extra_headers: Final = MappingProxyType(
{key: str(value) for key, value in (extra_headers or MappingProxyType({})).items()}
)
raw_response: Final = await openai_client.realtime.calls.with_raw_response.create(
sdp=sdp_text,
session=realtime_session_data,

View file

@ -7,7 +7,7 @@ This requires websockets, and is currently only supported on LiteLLM Proxy.
import ssl
from collections.abc import Mapping
from contextlib import AbstractAsyncContextManager
from types import TracebackType
from types import MappingProxyType, TracebackType
from typing import Any, Final, cast
from openai import AsyncOpenAI, omit
@ -180,13 +180,11 @@ class OpenAIRealtime(OpenAIChatCompletion):
additional_headers=headers,
max_size=REALTIME_WEBSOCKET_MAX_MESSAGE_SIZE_BYTES,
ssl=ssl_config,
**({"open_timeout": timeout} if timeout is not None else {}),
**(MappingProxyType({"open_timeout": timeout}) if timeout is not None else MappingProxyType({})),
)
openai_client: Final = client
model_query: Final = query_params.get("model")
extra_query: Final = { # mutable-ok: OpenAI SDK accepts a mutable query-parameter mapping
key: value for key, value in query_params.items() if key != "model"
}
extra_query: Final = MappingProxyType({key: value for key, value in query_params.items() if key != "model"})
sdk_model: Final = omit if query_params.get("intent") == "transcription" else model_query or model
sdk_connection_manager: Final = openai_client.realtime.connect(
model=sdk_model,
@ -194,8 +192,8 @@ class OpenAIRealtime(OpenAIChatCompletion):
extra_headers=headers,
websocket_connection_options={ # mutable-ok: OpenAI SDK forwards a mutable options mapping
"max_size": REALTIME_WEBSOCKET_MAX_MESSAGE_SIZE_BYTES,
**({"ssl": ssl_config} if url.startswith("wss://") else {}),
**({"open_timeout": timeout} if timeout is not None else {}),
**(MappingProxyType({"ssl": ssl_config}) if url.startswith("wss://") else MappingProxyType({})),
**(MappingProxyType({"open_timeout": timeout}) if timeout is not None else MappingProxyType({})),
},
max_retries=0,
)

View file

@ -1,4 +1,229 @@
{
"azure/gpt-live-transcribe": {
"input_cost_per_second": 0.0002833333333333333,
"litellm_provider": "azure",
"mode": "audio_transcription",
"source": "https://techcommunity.microsoft.com/blog/azure-ai-foundry-blog/introducing-gpt-transcribe-and-gpt-live-transcribe-in-microsoft-foundry/4541740",
"supported_endpoints": [
"/v1/realtime",
"/v1/realtime/transcription_sessions"
],
"supported_modalities": [
"audio",
"text"
],
"supported_output_modalities": [
"text"
],
"supports_audio_input": true,
"supports_native_streaming": true
},
"azure/gpt-realtime-2.1-2026-07-07": {
"cache_creation_input_audio_token_cost": 4e-07,
"cache_read_input_audio_token_cost": 4e-07,
"cache_read_input_token_cost": 4e-07,
"input_cost_per_audio_token": 3.2e-05,
"input_cost_per_image_token": 5e-06,
"input_cost_per_token": 4e-06,
"litellm_provider": "azure",
"max_input_tokens": 32000,
"max_output_tokens": 4096,
"max_tokens": 4096,
"mode": "realtime",
"output_cost_per_audio_token": 6.4e-05,
"output_cost_per_token": 2.4e-05,
"source": "https://learn.microsoft.com/en-us/azure/foundry/foundry-models/concepts/models-sold-directly-by-azure",
"supported_endpoints": [
"/v1/realtime"
],
"supported_modalities": [
"text",
"image",
"audio"
],
"supported_output_modalities": [
"text",
"audio"
],
"supports_audio_input": true,
"supports_audio_output": true,
"supports_function_calling": true,
"supports_parallel_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"cache_read_input_image_token_cost": 5e-07,
"supports_prompt_caching": true,
"supports_reasoning": true
},
"azure/gpt-realtime-2.1-mini-2026-07-07": {
"cache_creation_input_audio_token_cost": 3e-07,
"cache_read_input_audio_token_cost": 3e-07,
"cache_read_input_token_cost": 6e-08,
"input_cost_per_audio_token": 1e-05,
"input_cost_per_image_token": 8e-07,
"input_cost_per_token": 6e-07,
"litellm_provider": "azure",
"max_input_tokens": 32000,
"max_output_tokens": 4096,
"max_tokens": 4096,
"mode": "realtime",
"output_cost_per_audio_token": 2e-05,
"output_cost_per_token": 2.4e-06,
"source": "https://learn.microsoft.com/en-us/azure/foundry/foundry-models/concepts/models-sold-directly-by-azure",
"supported_endpoints": [
"/v1/realtime"
],
"supported_modalities": [
"text",
"image",
"audio"
],
"supported_output_modalities": [
"text",
"audio"
],
"supports_audio_input": true,
"supports_audio_output": true,
"supports_function_calling": true,
"supports_parallel_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"cache_read_input_image_token_cost": 8e-08,
"supports_prompt_caching": true,
"supports_reasoning": true
},
"azure/gpt-realtime-translate": {
"litellm_provider": "azure",
"max_input_tokens": 32000,
"max_output_tokens": 4096,
"max_tokens": 4096,
"mode": "realtime",
"output_cost_per_second": 0.0005666666666666667,
"source": "https://techcommunity.microsoft.com/blog/azure-ai-foundry-blog/a-new-chapter-for-realtime-ai-reasoning-translation-and-real-time-transcription/4517124",
"supported_endpoints": [
"/v1/realtime/translations",
"/v1/realtime/translations/client_secrets",
"/v1/realtime/translations/calls"
],
"supported_modalities": [
"audio"
],
"supported_output_modalities": [
"audio",
"text"
],
"supports_audio_input": true,
"supports_audio_output": true,
"supports_native_streaming": true
},
"azure/gpt-realtime-translate-2026-05-06": {
"litellm_provider": "azure",
"max_input_tokens": 32000,
"max_output_tokens": 4096,
"max_tokens": 4096,
"mode": "realtime",
"output_cost_per_second": 0.0005666666666666667,
"source": "https://techcommunity.microsoft.com/blog/azure-ai-foundry-blog/a-new-chapter-for-realtime-ai-reasoning-translation-and-real-time-transcription/4517124",
"supported_endpoints": [
"/v1/realtime/translations",
"/v1/realtime/translations/client_secrets",
"/v1/realtime/translations/calls"
],
"supported_modalities": [
"audio"
],
"supported_output_modalities": [
"audio",
"text"
],
"supports_audio_input": true,
"supports_audio_output": true,
"supports_native_streaming": true
},
"azure/gpt-realtime-translate-2026-05-07": {
"litellm_provider": "azure",
"max_input_tokens": 32000,
"max_output_tokens": 4096,
"max_tokens": 4096,
"mode": "realtime",
"output_cost_per_second": 0.0005666666666666667,
"source": "https://ai.azure.com/catalog/models/gpt-realtime-translate",
"supported_endpoints": [
"/v1/realtime/translations",
"/v1/realtime/translations/client_secrets",
"/v1/realtime/translations/calls"
],
"supported_modalities": [
"audio"
],
"supported_output_modalities": [
"audio",
"text"
],
"supports_audio_input": true,
"supports_audio_output": true,
"supports_native_streaming": true
},
"azure/gpt-realtime-whisper-2026-05-06": {
"input_cost_per_second": 0.0002833333333333333,
"litellm_provider": "azure",
"mode": "audio_transcription",
"source": "https://learn.microsoft.com/en-us/azure/foundry/openai/concepts/gpt-realtime-whisper",
"supported_endpoints": [
"/v1/realtime",
"/v1/realtime/transcription_sessions"
],
"supported_modalities": [
"audio"
],
"supported_output_modalities": [
"text"
],
"supports_audio_input": true,
"max_input_tokens": 32000,
"max_output_tokens": 4096,
"max_tokens": 4096
},
"azure/gpt-realtime-whisper-2026-05-07": {
"input_cost_per_second": 0.0002833333333333333,
"litellm_provider": "azure",
"mode": "audio_transcription",
"source": "https://learn.microsoft.com/en-us/azure/foundry/openai/concepts/gpt-realtime-whisper",
"supported_endpoints": [
"/v1/realtime",
"/v1/realtime/transcription_sessions"
],
"supported_modalities": [
"audio"
],
"supported_output_modalities": [
"text"
],
"supports_audio_input": true,
"max_input_tokens": 32000,
"max_output_tokens": 4096,
"max_tokens": 4096
},
"azure/gpt-transcribe": {
"input_cost_per_second": 7.5e-05,
"litellm_provider": "azure",
"mode": "audio_transcription",
"source": "https://techcommunity.microsoft.com/blog/azure-ai-foundry-blog/introducing-gpt-transcribe-and-gpt-live-transcribe-in-microsoft-foundry/4541740",
"supported_endpoints": [
"/v1/audio/transcriptions",
"/v1/realtime",
"/v1/realtime/transcription_sessions"
],
"supported_modalities": [
"audio",
"text"
],
"supported_output_modalities": [
"text"
],
"supports_audio_input": true,
"supports_native_streaming": true
},
"sample_spec": {
"code_interpreter_cost_per_session": 0.0,
"computer_use_input_cost_per_1k_tokens": 0.0,
@ -5821,6 +6046,7 @@
"azure/gpt-realtime-2.1": {
"cache_creation_input_audio_token_cost": 4e-07,
"cache_read_input_audio_token_cost": 4e-07,
"cache_read_input_image_token_cost": 5e-07,
"cache_read_input_token_cost": 4e-07,
"deprecation_date": "2027-06-25",
"input_cost_per_audio_token": 3.2e-05,
@ -5850,12 +6076,15 @@
"supports_audio_output": true,
"supports_function_calling": true,
"supports_parallel_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_system_messages": true,
"supports_tool_choice": true
},
"azure/gpt-realtime-2.1-mini": {
"cache_creation_input_audio_token_cost": 3e-07,
"cache_read_input_audio_token_cost": 3e-07,
"cache_read_input_image_token_cost": 8e-08,
"cache_read_input_token_cost": 6e-08,
"deprecation_date": "2027-06-25",
"input_cost_per_audio_token": 1e-05,
@ -5885,6 +6114,8 @@
"supports_audio_output": true,
"supports_function_calling": true,
"supports_parallel_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_system_messages": true,
"supports_tool_choice": true
},
@ -25462,7 +25693,6 @@
"cache_read_input_token_cost": 2e-07,
"cache_read_input_token_cost_above_200k_tokens": 4e-07,
"cache_read_input_token_cost_above_200k_tokens_priority": 7.2e-07,
"cache_read_input_token_cost_batches": 1e-07,
"cache_read_input_token_cost_flex": 1e-07,
"cache_read_input_token_cost_priority": 3.6e-07,
"deprecation_date": "2027-05-28",
@ -25555,7 +25785,6 @@
},
"gemini-3.1-flash-image": {
"cache_read_input_token_cost": 5e-08,
"cache_read_input_token_cost_batches": 2.5e-08,
"cache_read_input_token_cost_flex": 2.5e-08,
"deprecation_date": "2027-05-28",
"input_cost_per_image": 0.00056,
@ -25639,7 +25868,6 @@
},
"gemini-3.1-flash-lite-image": {
"cache_read_input_token_cost": 2.5e-08,
"cache_read_input_token_cost_batches": 1.25e-08,
"cache_read_input_token_cost_flex": 1.25e-08,
"input_cost_per_image": 0.00028,
"input_cost_per_token": 2.5e-07,
@ -25732,7 +25960,6 @@
"cache_read_input_audio_token_cost": 5e-08,
"deprecation_date": "2027-05-07",
"cache_read_input_token_cost": 2.5e-08,
"cache_read_input_token_cost_batches": 1.25e-08,
"cache_read_input_token_cost_flex": 1.25e-08,
"cache_read_input_token_cost_priority": 4.5e-08,
"input_cost_per_audio_token": 5e-07,
@ -25792,7 +26019,6 @@
"gemini-3.5-flash-lite": {
"deprecation_date": "2027-07-21",
"cache_read_input_token_cost": 3e-08,
"cache_read_input_token_cost_batches": 1.5e-08,
"cache_read_input_token_cost_flex": 1.5e-08,
"cache_read_input_token_cost_priority": 5.4e-08,
"input_cost_per_token": 3e-07,
@ -25849,7 +26075,6 @@
},
"deep-research-pro-preview-12-2025": {
"cache_read_input_token_cost": 2e-07,
"cache_read_input_token_cost_batches": 1e-07,
"input_cost_per_image": 0.0011,
"input_cost_per_token": 2e-06,
"input_cost_per_token_batches": 1e-06,
@ -26455,7 +26680,6 @@
"prompt_cache_min_tokens": 4096,
"deprecation_date": "2027-05-19",
"cache_read_input_token_cost": 1.5e-07,
"cache_read_input_token_cost_batches": 7.5e-08,
"input_cost_per_token": 1.5e-06,
"input_cost_per_audio_token": 1.5e-06,
"litellm_provider": "vertex_ai",
@ -26515,7 +26739,6 @@
"vertex_ai/gemini-3.6-flash": {
"prompt_cache_min_tokens": 4096,
"cache_read_input_token_cost": 7.5e-08,
"cache_read_input_token_cost_batches": 3.75e-08,
"cache_read_input_token_cost_flex": 3.75e-08,
"input_cost_per_token": 7.5e-07,
"input_cost_per_token_batches": 3.75e-07,
@ -26573,7 +26796,6 @@
"vertex_ai/gemini-3.7-flash": {
"prompt_cache_min_tokens": 4096,
"cache_read_input_token_cost": 7.5e-08,
"cache_read_input_token_cost_batches": 3.75e-08,
"cache_read_input_token_cost_flex": 3.75e-08,
"input_cost_per_token": 7.5e-07,
"input_cost_per_token_batches": 3.75e-07,
@ -26632,7 +26854,6 @@
"vertex_ai/gemini-3.8-flash": {
"prompt_cache_min_tokens": 4096,
"cache_read_input_token_cost": 7.5e-08,
"cache_read_input_token_cost_batches": 3.75e-08,
"cache_read_input_token_cost_flex": 3.75e-08,
"input_cost_per_token": 7.5e-07,
"input_cost_per_token_batches": 3.75e-07,
@ -28380,7 +28601,6 @@
"prompt_cache_min_tokens": 4096,
"deprecation_date": "2027-05-19",
"cache_read_input_token_cost": 1.5e-07,
"cache_read_input_token_cost_batches": 7.5e-08,
"input_cost_per_audio_token": 1.5e-06,
"input_cost_per_token": 1.5e-06,
"litellm_provider": "vertex_ai-language-models",
@ -28440,7 +28660,6 @@
"gemini-3.6-flash": {
"prompt_cache_min_tokens": 4096,
"cache_read_input_token_cost": 7.5e-08,
"cache_read_input_token_cost_batches": 3.75e-08,
"cache_read_input_token_cost_flex": 3.75e-08,
"input_cost_per_token": 7.5e-07,
"input_cost_per_token_batches": 3.75e-07,
@ -28498,7 +28717,6 @@
"gemini-3.7-flash": {
"prompt_cache_min_tokens": 4096,
"cache_read_input_token_cost": 7.5e-08,
"cache_read_input_token_cost_batches": 3.75e-08,
"cache_read_input_token_cost_flex": 3.75e-08,
"input_cost_per_token": 7.5e-07,
"input_cost_per_token_batches": 3.75e-07,
@ -28557,7 +28775,6 @@
"gemini-3.8-flash": {
"prompt_cache_min_tokens": 4096,
"cache_read_input_token_cost": 7.5e-08,
"cache_read_input_token_cost_batches": 3.75e-08,
"cache_read_input_token_cost_flex": 3.75e-08,
"input_cost_per_token": 7.5e-07,
"input_cost_per_token_batches": 3.75e-07,
@ -33245,7 +33462,6 @@
},
"gpt-5-2025-08-07": {
"cache_read_input_token_cost": 1.25e-07,
"cache_read_input_token_cost_batches": 6.25e-08,
"cache_read_input_token_cost_flex": 6.25e-08,
"cache_read_input_token_cost_priority": 2.5e-07,
"deprecation_date": "2026-12-11",
@ -33426,7 +33642,6 @@
},
"gpt-5-mini-2025-08-07": {
"cache_read_input_token_cost": 2.5e-08,
"cache_read_input_token_cost_batches": 1.25e-08,
"cache_read_input_token_cost_flex": 1.25e-08,
"cache_read_input_token_cost_priority": 4.5e-08,
"deprecation_date": "2026-12-11",
@ -33527,7 +33742,6 @@
},
"gpt-5-nano-2025-08-07": {
"cache_read_input_token_cost": 5e-09,
"cache_read_input_token_cost_batches": 2.5e-09,
"cache_read_input_token_cost_flex": 2.5e-09,
"deprecation_date": "2026-12-11",
"input_cost_per_token": 5e-08,
@ -33715,6 +33929,7 @@
"gpt-realtime-2.1": {
"cache_creation_input_audio_token_cost": 4e-07,
"cache_read_input_audio_token_cost": 4e-07,
"cache_read_input_image_token_cost": 5e-07,
"cache_read_input_token_cost": 4e-07,
"input_cost_per_audio_token": 3.2e-05,
"input_cost_per_image_token": 5e-06,
@ -33745,12 +33960,15 @@
"supports_audio_output": true,
"supports_function_calling": true,
"supports_parallel_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_system_messages": true,
"supports_tool_choice": true
},
"gpt-realtime-2.1-mini": {
"cache_creation_input_audio_token_cost": 3e-07,
"cache_read_input_audio_token_cost": 3e-07,
"cache_read_input_image_token_cost": 8e-08,
"cache_read_input_token_cost": 6e-08,
"input_cost_per_audio_token": 1e-05,
"input_cost_per_image_token": 8e-07,
@ -33781,6 +33999,8 @@
"supports_audio_output": true,
"supports_function_calling": true,
"supports_parallel_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_system_messages": true,
"supports_tool_choice": true
},
@ -47859,7 +48079,6 @@
"cache_read_input_token_cost": 2e-07,
"cache_read_input_token_cost_above_200k_tokens": 4e-07,
"cache_read_input_token_cost_above_200k_tokens_priority": 7.2e-07,
"cache_read_input_token_cost_batches": 1e-07,
"cache_read_input_token_cost_flex": 1e-07,
"cache_read_input_token_cost_priority": 3.6e-07,
"deprecation_date": "2027-05-28",
@ -47904,7 +48123,6 @@
},
"vertex_ai/gemini-3.1-flash-image": {
"cache_read_input_token_cost": 5e-08,
"cache_read_input_token_cost_batches": 2.5e-08,
"cache_read_input_token_cost_flex": 2.5e-08,
"deprecation_date": "2027-05-28",
"input_cost_per_image": 0.00056,
@ -47940,7 +48158,6 @@
},
"vertex_ai/gemini-3.1-flash-lite-image": {
"cache_read_input_token_cost": 2.5e-08,
"cache_read_input_token_cost_batches": 1.25e-08,
"cache_read_input_token_cost_flex": 1.25e-08,
"input_cost_per_image": 0.00028,
"input_cost_per_token": 2.5e-07,
@ -48033,7 +48250,6 @@
"cache_read_input_audio_token_cost": 5e-08,
"deprecation_date": "2027-05-07",
"cache_read_input_token_cost": 2.5e-08,
"cache_read_input_token_cost_batches": 1.25e-08,
"cache_read_input_token_cost_flex": 1.25e-08,
"cache_read_input_token_cost_priority": 4.5e-08,
"input_cost_per_audio_token": 5e-07,
@ -48094,7 +48310,6 @@
"vertex_ai/gemini-3.5-flash-lite": {
"deprecation_date": "2027-07-21",
"cache_read_input_token_cost": 3e-08,
"cache_read_input_token_cost_batches": 1.5e-08,
"cache_read_input_token_cost_flex": 1.5e-08,
"cache_read_input_token_cost_priority": 5.4e-08,
"input_cost_per_token": 3e-07,
@ -48152,7 +48367,6 @@
},
"vertex_ai/deep-research-pro-preview-12-2025": {
"cache_read_input_token_cost": 2e-07,
"cache_read_input_token_cost_batches": 1e-07,
"input_cost_per_image": 0.0011,
"input_cost_per_token": 2e-06,
"input_cost_per_token_batches": 1e-06,
@ -57348,7 +57562,8 @@
"supported_output_modalities": [
"text"
],
"supports_audio_input": true
"supports_audio_input": true,
"supports_native_streaming": true
},
"gpt-live-transcribe": {
"input_cost_per_second": 0.000283333333333,
@ -57366,7 +57581,8 @@
"supported_output_modalities": [
"text"
],
"supports_audio_input": true
"supports_audio_input": true,
"supports_native_streaming": true
},
"gpt-live-1": {
"input_cost_per_second": 0.000833333333333,
@ -57392,7 +57608,13 @@
"max_output_tokens": 2000,
"max_tokens": 2000,
"mode": "realtime",
"output_cost_per_second": 0.0005666666666666667,
"source": "https://developers.openai.com/api/docs/pricing",
"supported_endpoints": [
"/v1/realtime/translations",
"/v1/realtime/translations/client_secrets",
"/v1/realtime/translations/calls"
],
"supported_modalities": [
"audio"
],
@ -57401,7 +57623,8 @@
"audio"
],
"supports_audio_input": true,
"supports_audio_output": true
"supports_audio_output": true,
"supports_native_streaming": true
},
"claude-mythos-5": {
"supports_anthropic_compaction": true,

View file

@ -1,4 +1,229 @@
{
"azure/gpt-live-transcribe": {
"input_cost_per_second": 0.0002833333333333333,
"litellm_provider": "azure",
"mode": "audio_transcription",
"source": "https://techcommunity.microsoft.com/blog/azure-ai-foundry-blog/introducing-gpt-transcribe-and-gpt-live-transcribe-in-microsoft-foundry/4541740",
"supported_endpoints": [
"/v1/realtime",
"/v1/realtime/transcription_sessions"
],
"supported_modalities": [
"audio",
"text"
],
"supported_output_modalities": [
"text"
],
"supports_audio_input": true,
"supports_native_streaming": true
},
"azure/gpt-realtime-2.1-2026-07-07": {
"cache_creation_input_audio_token_cost": 4e-07,
"cache_read_input_audio_token_cost": 4e-07,
"cache_read_input_token_cost": 4e-07,
"input_cost_per_audio_token": 3.2e-05,
"input_cost_per_image_token": 5e-06,
"input_cost_per_token": 4e-06,
"litellm_provider": "azure",
"max_input_tokens": 32000,
"max_output_tokens": 4096,
"max_tokens": 4096,
"mode": "realtime",
"output_cost_per_audio_token": 6.4e-05,
"output_cost_per_token": 2.4e-05,
"source": "https://learn.microsoft.com/en-us/azure/foundry/foundry-models/concepts/models-sold-directly-by-azure",
"supported_endpoints": [
"/v1/realtime"
],
"supported_modalities": [
"text",
"image",
"audio"
],
"supported_output_modalities": [
"text",
"audio"
],
"supports_audio_input": true,
"supports_audio_output": true,
"supports_function_calling": true,
"supports_parallel_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"cache_read_input_image_token_cost": 5e-07,
"supports_prompt_caching": true,
"supports_reasoning": true
},
"azure/gpt-realtime-2.1-mini-2026-07-07": {
"cache_creation_input_audio_token_cost": 3e-07,
"cache_read_input_audio_token_cost": 3e-07,
"cache_read_input_token_cost": 6e-08,
"input_cost_per_audio_token": 1e-05,
"input_cost_per_image_token": 8e-07,
"input_cost_per_token": 6e-07,
"litellm_provider": "azure",
"max_input_tokens": 32000,
"max_output_tokens": 4096,
"max_tokens": 4096,
"mode": "realtime",
"output_cost_per_audio_token": 2e-05,
"output_cost_per_token": 2.4e-06,
"source": "https://learn.microsoft.com/en-us/azure/foundry/foundry-models/concepts/models-sold-directly-by-azure",
"supported_endpoints": [
"/v1/realtime"
],
"supported_modalities": [
"text",
"image",
"audio"
],
"supported_output_modalities": [
"text",
"audio"
],
"supports_audio_input": true,
"supports_audio_output": true,
"supports_function_calling": true,
"supports_parallel_function_calling": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"cache_read_input_image_token_cost": 8e-08,
"supports_prompt_caching": true,
"supports_reasoning": true
},
"azure/gpt-realtime-translate": {
"litellm_provider": "azure",
"max_input_tokens": 32000,
"max_output_tokens": 4096,
"max_tokens": 4096,
"mode": "realtime",
"output_cost_per_second": 0.0005666666666666667,
"source": "https://techcommunity.microsoft.com/blog/azure-ai-foundry-blog/a-new-chapter-for-realtime-ai-reasoning-translation-and-real-time-transcription/4517124",
"supported_endpoints": [
"/v1/realtime/translations",
"/v1/realtime/translations/client_secrets",
"/v1/realtime/translations/calls"
],
"supported_modalities": [
"audio"
],
"supported_output_modalities": [
"audio",
"text"
],
"supports_audio_input": true,
"supports_audio_output": true,
"supports_native_streaming": true
},
"azure/gpt-realtime-translate-2026-05-06": {
"litellm_provider": "azure",
"max_input_tokens": 32000,
"max_output_tokens": 4096,
"max_tokens": 4096,
"mode": "realtime",
"output_cost_per_second": 0.0005666666666666667,
"source": "https://techcommunity.microsoft.com/blog/azure-ai-foundry-blog/a-new-chapter-for-realtime-ai-reasoning-translation-and-real-time-transcription/4517124",
"supported_endpoints": [
"/v1/realtime/translations",
"/v1/realtime/translations/client_secrets",
"/v1/realtime/translations/calls"
],
"supported_modalities": [
"audio"
],
"supported_output_modalities": [
"audio",
"text"
],
"supports_audio_input": true,
"supports_audio_output": true,
"supports_native_streaming": true
},
"azure/gpt-realtime-translate-2026-05-07": {
"litellm_provider": "azure",
"max_input_tokens": 32000,
"max_output_tokens": 4096,
"max_tokens": 4096,
"mode": "realtime",
"output_cost_per_second": 0.0005666666666666667,
"source": "https://ai.azure.com/catalog/models/gpt-realtime-translate",
"supported_endpoints": [
"/v1/realtime/translations",
"/v1/realtime/translations/client_secrets",
"/v1/realtime/translations/calls"
],
"supported_modalities": [
"audio"
],
"supported_output_modalities": [
"audio",
"text"
],
"supports_audio_input": true,
"supports_audio_output": true,
"supports_native_streaming": true
},
"azure/gpt-realtime-whisper-2026-05-06": {
"input_cost_per_second": 0.0002833333333333333,
"litellm_provider": "azure",
"mode": "audio_transcription",
"source": "https://learn.microsoft.com/en-us/azure/foundry/openai/concepts/gpt-realtime-whisper",
"supported_endpoints": [
"/v1/realtime",
"/v1/realtime/transcription_sessions"
],
"supported_modalities": [
"audio"
],
"supported_output_modalities": [
"text"
],
"supports_audio_input": true,
"max_input_tokens": 32000,
"max_output_tokens": 4096,
"max_tokens": 4096
},
"azure/gpt-realtime-whisper-2026-05-07": {
"input_cost_per_second": 0.0002833333333333333,
"litellm_provider": "azure",
"mode": "audio_transcription",
"source": "https://learn.microsoft.com/en-us/azure/foundry/openai/concepts/gpt-realtime-whisper",
"supported_endpoints": [
"/v1/realtime",
"/v1/realtime/transcription_sessions"
],
"supported_modalities": [
"audio"
],
"supported_output_modalities": [
"text"
],
"supports_audio_input": true,
"max_input_tokens": 32000,
"max_output_tokens": 4096,
"max_tokens": 4096
},
"azure/gpt-transcribe": {
"input_cost_per_second": 7.5e-05,
"litellm_provider": "azure",
"mode": "audio_transcription",
"source": "https://techcommunity.microsoft.com/blog/azure-ai-foundry-blog/introducing-gpt-transcribe-and-gpt-live-transcribe-in-microsoft-foundry/4541740",
"supported_endpoints": [
"/v1/audio/transcriptions",
"/v1/realtime",
"/v1/realtime/transcription_sessions"
],
"supported_modalities": [
"audio",
"text"
],
"supported_output_modalities": [
"text"
],
"supports_audio_input": true,
"supports_native_streaming": true
},
"sample_spec": {
"code_interpreter_cost_per_session": 0.0,
"computer_use_input_cost_per_1k_tokens": 0.0,
@ -5821,6 +6046,7 @@
"azure/gpt-realtime-2.1": {
"cache_creation_input_audio_token_cost": 4e-07,
"cache_read_input_audio_token_cost": 4e-07,
"cache_read_input_image_token_cost": 5e-07,
"cache_read_input_token_cost": 4e-07,
"deprecation_date": "2027-06-25",
"input_cost_per_audio_token": 3.2e-05,
@ -5850,12 +6076,15 @@
"supports_audio_output": true,
"supports_function_calling": true,
"supports_parallel_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_system_messages": true,
"supports_tool_choice": true
},
"azure/gpt-realtime-2.1-mini": {
"cache_creation_input_audio_token_cost": 3e-07,
"cache_read_input_audio_token_cost": 3e-07,
"cache_read_input_image_token_cost": 8e-08,
"cache_read_input_token_cost": 6e-08,
"deprecation_date": "2027-06-25",
"input_cost_per_audio_token": 1e-05,
@ -5885,6 +6114,8 @@
"supports_audio_output": true,
"supports_function_calling": true,
"supports_parallel_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_system_messages": true,
"supports_tool_choice": true
},
@ -25462,7 +25693,6 @@
"cache_read_input_token_cost": 2e-07,
"cache_read_input_token_cost_above_200k_tokens": 4e-07,
"cache_read_input_token_cost_above_200k_tokens_priority": 7.2e-07,
"cache_read_input_token_cost_batches": 1e-07,
"cache_read_input_token_cost_flex": 1e-07,
"cache_read_input_token_cost_priority": 3.6e-07,
"deprecation_date": "2027-05-28",
@ -25555,7 +25785,6 @@
},
"gemini-3.1-flash-image": {
"cache_read_input_token_cost": 5e-08,
"cache_read_input_token_cost_batches": 2.5e-08,
"cache_read_input_token_cost_flex": 2.5e-08,
"deprecation_date": "2027-05-28",
"input_cost_per_image": 0.00056,
@ -25639,7 +25868,6 @@
},
"gemini-3.1-flash-lite-image": {
"cache_read_input_token_cost": 2.5e-08,
"cache_read_input_token_cost_batches": 1.25e-08,
"cache_read_input_token_cost_flex": 1.25e-08,
"input_cost_per_image": 0.00028,
"input_cost_per_token": 2.5e-07,
@ -25732,7 +25960,6 @@
"cache_read_input_audio_token_cost": 5e-08,
"deprecation_date": "2027-05-07",
"cache_read_input_token_cost": 2.5e-08,
"cache_read_input_token_cost_batches": 1.25e-08,
"cache_read_input_token_cost_flex": 1.25e-08,
"cache_read_input_token_cost_priority": 4.5e-08,
"input_cost_per_audio_token": 5e-07,
@ -25792,7 +26019,6 @@
"gemini-3.5-flash-lite": {
"deprecation_date": "2027-07-21",
"cache_read_input_token_cost": 3e-08,
"cache_read_input_token_cost_batches": 1.5e-08,
"cache_read_input_token_cost_flex": 1.5e-08,
"cache_read_input_token_cost_priority": 5.4e-08,
"input_cost_per_token": 3e-07,
@ -25849,7 +26075,6 @@
},
"deep-research-pro-preview-12-2025": {
"cache_read_input_token_cost": 2e-07,
"cache_read_input_token_cost_batches": 1e-07,
"input_cost_per_image": 0.0011,
"input_cost_per_token": 2e-06,
"input_cost_per_token_batches": 1e-06,
@ -26455,7 +26680,6 @@
"prompt_cache_min_tokens": 4096,
"deprecation_date": "2027-05-19",
"cache_read_input_token_cost": 1.5e-07,
"cache_read_input_token_cost_batches": 7.5e-08,
"input_cost_per_token": 1.5e-06,
"input_cost_per_audio_token": 1.5e-06,
"litellm_provider": "vertex_ai",
@ -26515,7 +26739,6 @@
"vertex_ai/gemini-3.6-flash": {
"prompt_cache_min_tokens": 4096,
"cache_read_input_token_cost": 7.5e-08,
"cache_read_input_token_cost_batches": 3.75e-08,
"cache_read_input_token_cost_flex": 3.75e-08,
"input_cost_per_token": 7.5e-07,
"input_cost_per_token_batches": 3.75e-07,
@ -26573,7 +26796,6 @@
"vertex_ai/gemini-3.7-flash": {
"prompt_cache_min_tokens": 4096,
"cache_read_input_token_cost": 7.5e-08,
"cache_read_input_token_cost_batches": 3.75e-08,
"cache_read_input_token_cost_flex": 3.75e-08,
"input_cost_per_token": 7.5e-07,
"input_cost_per_token_batches": 3.75e-07,
@ -26632,7 +26854,6 @@
"vertex_ai/gemini-3.8-flash": {
"prompt_cache_min_tokens": 4096,
"cache_read_input_token_cost": 7.5e-08,
"cache_read_input_token_cost_batches": 3.75e-08,
"cache_read_input_token_cost_flex": 3.75e-08,
"input_cost_per_token": 7.5e-07,
"input_cost_per_token_batches": 3.75e-07,
@ -28380,7 +28601,6 @@
"prompt_cache_min_tokens": 4096,
"deprecation_date": "2027-05-19",
"cache_read_input_token_cost": 1.5e-07,
"cache_read_input_token_cost_batches": 7.5e-08,
"input_cost_per_audio_token": 1.5e-06,
"input_cost_per_token": 1.5e-06,
"litellm_provider": "vertex_ai-language-models",
@ -28440,7 +28660,6 @@
"gemini-3.6-flash": {
"prompt_cache_min_tokens": 4096,
"cache_read_input_token_cost": 7.5e-08,
"cache_read_input_token_cost_batches": 3.75e-08,
"cache_read_input_token_cost_flex": 3.75e-08,
"input_cost_per_token": 7.5e-07,
"input_cost_per_token_batches": 3.75e-07,
@ -28498,7 +28717,6 @@
"gemini-3.7-flash": {
"prompt_cache_min_tokens": 4096,
"cache_read_input_token_cost": 7.5e-08,
"cache_read_input_token_cost_batches": 3.75e-08,
"cache_read_input_token_cost_flex": 3.75e-08,
"input_cost_per_token": 7.5e-07,
"input_cost_per_token_batches": 3.75e-07,
@ -28557,7 +28775,6 @@
"gemini-3.8-flash": {
"prompt_cache_min_tokens": 4096,
"cache_read_input_token_cost": 7.5e-08,
"cache_read_input_token_cost_batches": 3.75e-08,
"cache_read_input_token_cost_flex": 3.75e-08,
"input_cost_per_token": 7.5e-07,
"input_cost_per_token_batches": 3.75e-07,
@ -33245,7 +33462,6 @@
},
"gpt-5-2025-08-07": {
"cache_read_input_token_cost": 1.25e-07,
"cache_read_input_token_cost_batches": 6.25e-08,
"cache_read_input_token_cost_flex": 6.25e-08,
"cache_read_input_token_cost_priority": 2.5e-07,
"deprecation_date": "2026-12-11",
@ -33426,7 +33642,6 @@
},
"gpt-5-mini-2025-08-07": {
"cache_read_input_token_cost": 2.5e-08,
"cache_read_input_token_cost_batches": 1.25e-08,
"cache_read_input_token_cost_flex": 1.25e-08,
"cache_read_input_token_cost_priority": 4.5e-08,
"deprecation_date": "2026-12-11",
@ -33527,7 +33742,6 @@
},
"gpt-5-nano-2025-08-07": {
"cache_read_input_token_cost": 5e-09,
"cache_read_input_token_cost_batches": 2.5e-09,
"cache_read_input_token_cost_flex": 2.5e-09,
"deprecation_date": "2026-12-11",
"input_cost_per_token": 5e-08,
@ -33715,6 +33929,7 @@
"gpt-realtime-2.1": {
"cache_creation_input_audio_token_cost": 4e-07,
"cache_read_input_audio_token_cost": 4e-07,
"cache_read_input_image_token_cost": 5e-07,
"cache_read_input_token_cost": 4e-07,
"input_cost_per_audio_token": 3.2e-05,
"input_cost_per_image_token": 5e-06,
@ -33745,12 +33960,15 @@
"supports_audio_output": true,
"supports_function_calling": true,
"supports_parallel_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_system_messages": true,
"supports_tool_choice": true
},
"gpt-realtime-2.1-mini": {
"cache_creation_input_audio_token_cost": 3e-07,
"cache_read_input_audio_token_cost": 3e-07,
"cache_read_input_image_token_cost": 8e-08,
"cache_read_input_token_cost": 6e-08,
"input_cost_per_audio_token": 1e-05,
"input_cost_per_image_token": 8e-07,
@ -33781,6 +33999,8 @@
"supports_audio_output": true,
"supports_function_calling": true,
"supports_parallel_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_system_messages": true,
"supports_tool_choice": true
},
@ -47859,7 +48079,6 @@
"cache_read_input_token_cost": 2e-07,
"cache_read_input_token_cost_above_200k_tokens": 4e-07,
"cache_read_input_token_cost_above_200k_tokens_priority": 7.2e-07,
"cache_read_input_token_cost_batches": 1e-07,
"cache_read_input_token_cost_flex": 1e-07,
"cache_read_input_token_cost_priority": 3.6e-07,
"deprecation_date": "2027-05-28",
@ -47904,7 +48123,6 @@
},
"vertex_ai/gemini-3.1-flash-image": {
"cache_read_input_token_cost": 5e-08,
"cache_read_input_token_cost_batches": 2.5e-08,
"cache_read_input_token_cost_flex": 2.5e-08,
"deprecation_date": "2027-05-28",
"input_cost_per_image": 0.00056,
@ -47940,7 +48158,6 @@
},
"vertex_ai/gemini-3.1-flash-lite-image": {
"cache_read_input_token_cost": 2.5e-08,
"cache_read_input_token_cost_batches": 1.25e-08,
"cache_read_input_token_cost_flex": 1.25e-08,
"input_cost_per_image": 0.00028,
"input_cost_per_token": 2.5e-07,
@ -48033,7 +48250,6 @@
"cache_read_input_audio_token_cost": 5e-08,
"deprecation_date": "2027-05-07",
"cache_read_input_token_cost": 2.5e-08,
"cache_read_input_token_cost_batches": 1.25e-08,
"cache_read_input_token_cost_flex": 1.25e-08,
"cache_read_input_token_cost_priority": 4.5e-08,
"input_cost_per_audio_token": 5e-07,
@ -48094,7 +48310,6 @@
"vertex_ai/gemini-3.5-flash-lite": {
"deprecation_date": "2027-07-21",
"cache_read_input_token_cost": 3e-08,
"cache_read_input_token_cost_batches": 1.5e-08,
"cache_read_input_token_cost_flex": 1.5e-08,
"cache_read_input_token_cost_priority": 5.4e-08,
"input_cost_per_token": 3e-07,
@ -48152,7 +48367,6 @@
},
"vertex_ai/deep-research-pro-preview-12-2025": {
"cache_read_input_token_cost": 2e-07,
"cache_read_input_token_cost_batches": 1e-07,
"input_cost_per_image": 0.0011,
"input_cost_per_token": 2e-06,
"input_cost_per_token_batches": 1e-06,
@ -57348,7 +57562,8 @@
"supported_output_modalities": [
"text"
],
"supports_audio_input": true
"supports_audio_input": true,
"supports_native_streaming": true
},
"gpt-live-transcribe": {
"input_cost_per_second": 0.000283333333333,
@ -57366,7 +57581,8 @@
"supported_output_modalities": [
"text"
],
"supports_audio_input": true
"supports_audio_input": true,
"supports_native_streaming": true
},
"gpt-live-1": {
"input_cost_per_second": 0.000833333333333,
@ -57392,7 +57608,13 @@
"max_output_tokens": 2000,
"max_tokens": 2000,
"mode": "realtime",
"output_cost_per_second": 0.0005666666666666667,
"source": "https://developers.openai.com/api/docs/pricing",
"supported_endpoints": [
"/v1/realtime/translations",
"/v1/realtime/translations/client_secrets",
"/v1/realtime/translations/calls"
],
"supported_modalities": [
"audio"
],
@ -57401,7 +57623,8 @@
"audio"
],
"supports_audio_input": true,
"supports_audio_output": true
"supports_audio_output": true,
"supports_native_streaming": true
},
"claude-mythos-5": {
"supports_anthropic_compaction": true,

View file

@ -29,6 +29,7 @@ from litellm.llms.gemini.image_generation.cost_calculator import (
from litellm.llms.vertex_ai.image_generation.cost_calculator import (
cost_calculator as vertex_image_generation_cost_calculator,
)
from litellm.types.llms.base import CachedTokensDetails
from litellm.types.utils import (
CacheCreationTokenDetails,
CompletionTokensDetailsWrapper,
@ -42,6 +43,39 @@ from litellm.types.utils import (
)
def test_realtime_cached_modality_breakdown_matches_prompt_cost(_local_model_cost_map):
model: Final = "gpt-realtime-2.1-mini"
rates: Final = litellm.model_cost[model]
usage: Final = Usage(
prompt_tokens=1000,
completion_tokens=0,
total_tokens=1000,
prompt_tokens_details=PromptTokensDetailsWrapper(
text_tokens=400,
audio_tokens=400,
image_tokens=200,
cached_tokens=300,
cached_tokens_details=CachedTokensDetails(text_tokens=100, audio_tokens=150, image_tokens=50),
),
)
prompt_cost, _ = generic_cost_per_token(model=model, usage=usage, custom_llm_provider="openai")
breakdown: Final = get_token_type_cost_breakdown(model=model, custom_llm_provider="openai", usage=usage)
cached_cost: Final = (
100 * rates["cache_read_input_token_cost"]
+ 150 * rates["cache_read_input_audio_token_cost"]
+ 50 * rates["cache_read_input_image_token_cost"]
)
uncached_cost: Final = (
300 * rates["input_cost_per_token"]
+ 250 * rates["input_cost_per_audio_token"]
+ 150 * rates["input_cost_per_image_token"]
)
assert breakdown.cache_read_cost == pytest.approx(cached_cost)
assert prompt_cost == pytest.approx(uncached_cost + cached_cost)
@pytest.fixture
def _local_model_cost_map(monkeypatch: pytest.MonkeyPatch) -> None:
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
@ -299,8 +333,6 @@ def test_reasoning_tokens_gemini(_local_model_cost_map):
)
def test_image_tokens_with_custom_pricing():
"""Test that image_tokens in completion are properly costed with output_cost_per_image_token."""
from unittest.mock import patch
@ -1950,6 +1982,10 @@ def test_cache_writing_cost_with_zero_creation_tokens_and_ephemeral_details():
prompt_tokens_details: PromptTokensDetailsResult = {
"cache_hit_tokens": 0,
"cache_hit_audio_tokens": 0,
"cached_text_tokens": 0,
"cached_audio_tokens": 0,
"cached_image_tokens": 0,
"has_cached_tokens_details": False,
"cache_creation_tokens": 0,
"cache_creation_token_details": CacheCreationTokenDetails(
ephemeral_5m_input_tokens=100,
@ -2185,10 +2221,6 @@ def test_vertex_image_generation_cost_falls_back_to_flat_image_pricing(_local_mo
assert round(cost, 10) == round(expected_cost, 10)
def test_query_count_is_free_without_a_per_query_price(_local_model_cost_map):
usage = Usage(
prompt_tokens=0,
@ -2367,8 +2399,6 @@ def test_vertex_global_or_absent_location_no_uplift(vertex_location, _local_mode
assert base == located
def test_vertex_uplift_invalid_multiplier_defaults_to_one():
"""A malformed multiplier in the cost map degrades to base pricing, never raises."""
from litellm.litellm_core_utils.llm_cost_calc.utils import (
@ -3585,8 +3615,6 @@ def test_route_image_generation_cost_openai_honors_deployment_input_cost_per_ima
assert cost == pytest.approx(0.07)
@pytest.mark.parametrize(
("custom_llm_provider", "model"),
[

View file

@ -1,3 +1,4 @@
import json
from unittest.mock import AsyncMock, MagicMock, patch
import pytest

View file

@ -1121,7 +1121,7 @@ async def test_translation_client_secret_rejects_disallowed_nested_transcription
),
)
with pytest.raises(Exception, match="Tried to access gpt-live-transcribe"):
with pytest.raises(Exception, match=r"gpt-live-transcribe.*not available for this API key"):
await _prepare_client_secret_session(
req=req,
user_api_key_dict=UserAPIKeyAuth(models=["gpt-realtime-translate"]),

View file

@ -160,10 +160,6 @@ def test_cost_calculator_with_response_cost_in_additional_headers():
assert result == 1000
def test_realtime_stream_combines_text_and_audio_token_details():
"""Realtime response.done usage with input_token_details / output_token_details."""
from litellm.cost_calculator import RealtimeAPITokenUsageProcessor
@ -587,7 +583,7 @@ def test_completion_cost_image_generation_reads_deployment_model_info_price_from
assert cost == pytest.approx(0.08)
def test_completion_cost_image_generation_registered_deployment_price_keeps_map_token_rates(
def test_completion_cost_image_generation_registered_deployment_applies_custom_image_rate(
_local_model_cost_map: None, monkeypatch: pytest.MonkeyPatch
) -> None:
deployment_id: Final = "gemini-image-deployment-priced-per-image"
@ -1041,8 +1037,6 @@ def test_bedrock_cost_calculator_comparison_with_without_cache():
print(f"Cost with cache: {cost_with_cache}")
def test_gemini_25_explicit_caching_cost_direct_usage():
"""
Test that Gemini 2.5 models correctly calculate costs with explicit caching.
@ -1611,8 +1605,6 @@ def test_cost_margin_with_discount(monkeypatch):
print(f" - Expected: ${expected_cost:.6f}")
def test_completion_cost_extracts_service_tier_from_response(_local_model_cost_map):
"""Test that completion_cost extracts service_tier from completion_response object."""
from litellm import completion_cost
@ -2363,8 +2355,6 @@ def test_gemini_without_cache_tokens_details():
print("✅ Gemini without cacheTokensDetails works correctly")
def test_additional_costs_only_for_azure_ai(_local_model_cost_map):
"""
Test that _get_additional_costs is only called for azure_ai provider.
@ -4633,3 +4623,65 @@ def test_gemini_live_native_audio_limits_and_capabilities_match_vendor_model_car
assert info["supports_response_schema"] is False
assert info["supports_url_context"] is False
assert info["supports_pdf_input"] is False
@pytest.mark.parametrize("provider", ("openai", "azure"))
@pytest.mark.parametrize("model", ("gpt-realtime-2.1", "gpt-realtime-2.1-mini"))
def test_realtime_cached_multimodal_token_cost(_local_model_cost_map, provider: str, model: str):
model_name: Final = f"azure/{model}" if provider == "azure" else model
rates: Final = litellm.model_cost[model_name]
events: Final[OpenAIRealtimeStreamList] = [
{"type": "session.created", "session": {"model": model}},
{
"type": "response.done",
"response": {
"usage": {
"input_tokens": 1000,
"output_tokens": 300,
"total_tokens": 1300,
"input_token_details": {
"text_tokens": 400,
"audio_tokens": 400,
"image_tokens": 200,
"cached_tokens": 300,
"cached_tokens_details": {"text_tokens": 100, "audio_tokens": 150, "image_tokens": 50},
},
"output_token_details": {"text_tokens": 100, "audio_tokens": 100, "reasoning_tokens": 100},
}
},
},
]
combined: Final = RealtimeAPITokenUsageProcessor.collect_and_combine_usage_from_realtime_stream_results(events)
actual: Final = handle_realtime_stream_cost_calculation(
results=events,
combined_usage_object=combined,
custom_llm_provider=provider,
litellm_model_name=model_name,
)
expected: Final = (
300 * rates["input_cost_per_token"]
+ 250 * rates["input_cost_per_audio_token"]
+ 150 * rates["input_cost_per_image_token"]
+ 100 * rates["cache_read_input_token_cost"]
+ 150 * rates["cache_read_input_audio_token_cost"]
+ 50 * rates["cache_read_input_image_token_cost"]
+ 200 * rates["output_cost_per_token"]
+ 100 * rates["output_cost_per_audio_token"]
)
assert actual == pytest.approx(expected)
def test_realtime_translation_duration_cost(_local_model_cost_map):
from litellm.cost_calculator import handle_realtime_translation_cost_calculation
model: Final = "gpt-realtime-translate"
events: Final[OpenAIRealtimeStreamList] = [
{"type": "session.closed", "usage": {"type": "duration", "output_seconds": 2.0}}
]
actual: Final = handle_realtime_translation_cost_calculation(
results=events,
custom_llm_provider="openai",
litellm_model_name=model,
)
assert actual == pytest.approx(2 * litellm.model_cost[model]["output_cost_per_second"])

View file

@ -478,3 +478,36 @@ def test_unregistered_provider_guard_flags_only_labels_nobody_registered():
"unknown_root-new_family_models",
"vertex_ai-new_family_models",
]
@pytest.mark.parametrize("model", ("gpt-realtime-2.1", "gpt-realtime-2.1-mini"))
def test_realtime_family_cache_image_rate_tracks_azure(prices: dict, model: str):
openai: Final = prices[model]
azure: Final = prices[f"azure/{model}"]
assert openai["cache_read_input_image_token_cost"] > 0
assert azure["cache_read_input_image_token_cost"] == openai["cache_read_input_image_token_cost"]
assert azure["input_cost_per_image_token"] >= azure["cache_read_input_image_token_cost"]
@pytest.mark.parametrize(
"model,mode",
(
("gpt-realtime-translate", "realtime"),
("gpt-live-transcribe", "audio_transcription"),
("gpt-transcribe", "audio_transcription"),
),
)
def test_azure_realtime_specialized_models_follow_openai_modes(prices: dict, model: str, mode: str):
openai: Final = prices[model]
azure: Final = prices[f"azure/{model}"]
assert openai["mode"] == azure["mode"] == mode
assert azure["supports_audio_input"] is True
assert azure["supported_endpoints"]
def test_model_prices_backup_is_synchronized(prices: dict):
backup: Final = json.loads(BACKUP_PRICES_PATH.read_text())
assert backup == prices

View file

@ -259,6 +259,21 @@ async def test_construct_url_v1_protocol():
assert url.count("/realtime") == 1
def test_construct_url_translation_protocol():
from litellm.llms.azure.realtime.handler import AzureOpenAIRealtime
url = AzureOpenAIRealtime()._construct_url(
api_base="https://my-endpoint.openai.azure.com",
model="translate-deployment",
api_version=None,
realtime_protocol="GA",
query_params={"model": "translate-deployment"},
realtime_mode="translation",
)
assert url == "wss://my-endpoint.openai.azure.com/openai/v1/realtime/translations?model=translate-deployment"
@pytest.mark.asyncio
@pytest.mark.parametrize("protocol", ["ga", "Ga", "gA", "V1", "v1", "GA"])
async def test_construct_url_case_insensitive_protocol(protocol):

View file

@ -10227,7 +10227,7 @@ export interface paths {
* WebSocket: realtime_websocket_endpoint
* @description WebSocket connection endpoint
*/
get: operations["websocket_realtime_websocket_endpoint_get_3"];
get: operations["websocket_realtime_websocket_endpoint_get_6"];
put?: never;
post?: never;
delete?: never;
@ -10294,6 +10294,60 @@ export interface paths {
patch?: never;
trace?: never;
};
"/openai/v1/realtime/translations": {
parameters: {
query?: never;
header?: never;
path?: never;
cookie?: never;
};
/**
* WebSocket: realtime_websocket_endpoint
* @description WebSocket connection endpoint
*/
get: operations["websocket_realtime_websocket_endpoint_get_3"];
put?: never;
post?: never;
delete?: never;
options?: never;
head?: never;
patch?: never;
trace?: never;
};
"/openai/v1/realtime/translations/calls": {
parameters: {
query?: never;
header?: never;
path?: never;
cookie?: never;
};
get?: never;
put?: never;
/** Proxy Realtime Calls */
post: operations["proxy_realtime_calls_openai_v1_realtime_translations_calls_post"];
delete?: never;
options?: never;
head?: never;
patch?: never;
trace?: never;
};
"/openai/v1/realtime/translations/client_secrets": {
parameters: {
query?: never;
header?: never;
path?: never;
cookie?: never;
};
get?: never;
put?: never;
/** Create Realtime Client Secret */
post: operations["create_realtime_client_secret_openai_v1_realtime_translations_client_secrets_post"];
delete?: never;
options?: never;
head?: never;
patch?: never;
trace?: never;
};
"/openai/v1/responses": {
parameters: {
query?: never;
@ -13109,7 +13163,7 @@ export interface paths {
* WebSocket: realtime_websocket_endpoint
* @description WebSocket connection endpoint
*/
get: operations["websocket_realtime_websocket_endpoint_get"];
get: operations["websocket_realtime_websocket_endpoint_get_4"];
put?: never;
post?: never;
delete?: never;
@ -13176,6 +13230,60 @@ export interface paths {
patch?: never;
trace?: never;
};
"/realtime/translations": {
parameters: {
query?: never;
header?: never;
path?: never;
cookie?: never;
};
/**
* WebSocket: realtime_websocket_endpoint
* @description WebSocket connection endpoint
*/
get: operations["websocket_realtime_websocket_endpoint_get"];
put?: never;
post?: never;
delete?: never;
options?: never;
head?: never;
patch?: never;
trace?: never;
};
"/realtime/translations/calls": {
parameters: {
query?: never;
header?: never;
path?: never;
cookie?: never;
};
get?: never;
put?: never;
/** Proxy Realtime Calls */
post: operations["proxy_realtime_calls_realtime_translations_calls_post"];
delete?: never;
options?: never;
head?: never;
patch?: never;
trace?: never;
};
"/realtime/translations/client_secrets": {
parameters: {
query?: never;
header?: never;
path?: never;
cookie?: never;
};
get?: never;
put?: never;
/** Create Realtime Client Secret */
post: operations["create_realtime_client_secret_realtime_translations_client_secrets_post"];
delete?: never;
options?: never;
head?: never;
patch?: never;
trace?: never;
};
"/register": {
parameters: {
query?: never;
@ -20205,7 +20313,7 @@ export interface paths {
* WebSocket: realtime_websocket_endpoint
* @description WebSocket connection endpoint
*/
get: operations["websocket_realtime_websocket_endpoint_get_2"];
get: operations["websocket_realtime_websocket_endpoint_get_5"];
put?: never;
post?: never;
delete?: never;
@ -20272,6 +20380,60 @@ export interface paths {
patch?: never;
trace?: never;
};
"/v1/realtime/translations": {
parameters: {
query?: never;
header?: never;
path?: never;
cookie?: never;
};
/**
* WebSocket: realtime_websocket_endpoint
* @description WebSocket connection endpoint
*/
get: operations["websocket_realtime_websocket_endpoint_get_2"];
put?: never;
post?: never;
delete?: never;
options?: never;
head?: never;
patch?: never;
trace?: never;
};
"/v1/realtime/translations/calls": {
parameters: {
query?: never;
header?: never;
path?: never;
cookie?: never;
};
get?: never;
put?: never;
/** Proxy Realtime Calls */
post: operations["proxy_realtime_calls_v1_realtime_translations_calls_post"];
delete?: never;
options?: never;
head?: never;
patch?: never;
trace?: never;
};
"/v1/realtime/translations/client_secrets": {
parameters: {
query?: never;
header?: never;
path?: never;
cookie?: never;
};
get?: never;
put?: never;
/** Create Realtime Client Secret */
post: operations["create_realtime_client_secret_v1_realtime_translations_client_secrets_post"];
delete?: never;
options?: never;
head?: never;
patch?: never;
trace?: never;
};
"/v1/rerank": {
parameters: {
query?: never;
@ -26043,7 +26205,7 @@ export interface components {
* CallTypes
* @enum {string}
*/
CallTypes: "embedding" | "aembedding" | "completion" | "acompletion" | "atext_completion" | "text_completion" | "image_generation" | "aimage_generation" | "image_edit" | "aimage_edit" | "moderation" | "amoderation" | "atranscription" | "transcription" | "aspeech" | "speech" | "rerank" | "arerank" | "search" | "asearch" | "_arealtime" | "_aresponses_websocket" | "create_batch" | "acreate_batch" | "aretrieve_batch" | "retrieve_batch" | "acancel_batch" | "cancel_batch" | "pass_through_endpoint" | "anthropic_messages" | "aanthropic_messages" | "get_assistants" | "aget_assistants" | "create_assistants" | "acreate_assistants" | "delete_assistant" | "adelete_assistant" | "acreate_thread" | "create_thread" | "aget_thread" | "get_thread" | "a_add_message" | "add_message" | "aget_messages" | "get_messages" | "arun_thread" | "run_thread" | "arun_thread_stream" | "run_thread_stream" | "afile_retrieve" | "file_retrieve" | "afile_delete" | "file_delete" | "afile_list" | "file_list" | "acreate_file" | "create_file" | "afile_content" | "file_content" | "create_fine_tuning_job" | "acreate_fine_tuning_job" | "create_video" | "acreate_video" | "video_generation" | "avideo_generation" | "avideo_retrieve" | "video_retrieve" | "avideo_content" | "video_content" | "video_remix" | "avideo_remix" | "video_list" | "avideo_list" | "video_retrieve_job" | "avideo_retrieve_job" | "video_delete" | "avideo_delete" | "video_create_character" | "avideo_create_character" | "video_get_character" | "avideo_get_character" | "video_edit" | "avideo_edit" | "video_extension" | "avideo_extension" | "vector_store_file_create" | "avector_store_file_create" | "vector_store_file_list" | "avector_store_file_list" | "vector_store_file_retrieve" | "avector_store_file_retrieve" | "vector_store_file_content" | "avector_store_file_content" | "vector_store_file_update" | "avector_store_file_update" | "vector_store_file_delete" | "avector_store_file_delete" | "vector_store_create" | "avector_store_create" | "vector_store_search" | "avector_store_search" | "ingest" | "aingest" | "query" | "aquery" | "create_interaction" | "acreate_interaction" | "create_container" | "acreate_container" | "list_containers" | "alist_containers" | "retrieve_container" | "aretrieve_container" | "delete_container" | "adelete_container" | "list_container_files" | "alist_container_files" | "upload_container_file" | "aupload_container_file" | "create_sandbox" | "acreate_sandbox" | "delete_sandbox" | "adelete_sandbox" | "run_code" | "arun_code" | "code_interpreter_tool" | "acode_interpreter_tool" | "acancel_fine_tuning_job" | "cancel_fine_tuning_job" | "alist_fine_tuning_jobs" | "list_fine_tuning_jobs" | "aretrieve_fine_tuning_job" | "retrieve_fine_tuning_job" | "responses" | "aresponses" | "alist_input_items" | "llm_passthrough_route" | "allm_passthrough_route" | "generate_content" | "agenerate_content" | "generate_content_stream" | "agenerate_content_stream" | "ocr" | "aocr" | "call_mcp_tool" | "list_mcp_tools" | "asend_message" | "send_message" | "acreate_skill";
CallTypes: "embedding" | "aembedding" | "completion" | "acompletion" | "atext_completion" | "text_completion" | "image_generation" | "aimage_generation" | "image_edit" | "aimage_edit" | "moderation" | "amoderation" | "atranscription" | "transcription" | "aspeech" | "speech" | "rerank" | "arerank" | "search" | "asearch" | "_arealtime" | "_aresponses_websocket" | "acreate_realtime_client_secret" | "arealtime_calls" | "acreate_realtime_transcription_session" | "acreate_realtime_translation_client_secret" | "arealtime_translation_calls" | "create_batch" | "acreate_batch" | "aretrieve_batch" | "retrieve_batch" | "acancel_batch" | "cancel_batch" | "pass_through_endpoint" | "anthropic_messages" | "aanthropic_messages" | "get_assistants" | "aget_assistants" | "create_assistants" | "acreate_assistants" | "delete_assistant" | "adelete_assistant" | "acreate_thread" | "create_thread" | "aget_thread" | "get_thread" | "a_add_message" | "add_message" | "aget_messages" | "get_messages" | "arun_thread" | "run_thread" | "arun_thread_stream" | "run_thread_stream" | "afile_retrieve" | "file_retrieve" | "afile_delete" | "file_delete" | "afile_list" | "file_list" | "acreate_file" | "create_file" | "afile_content" | "file_content" | "create_fine_tuning_job" | "acreate_fine_tuning_job" | "create_video" | "acreate_video" | "video_generation" | "avideo_generation" | "avideo_retrieve" | "video_retrieve" | "avideo_content" | "video_content" | "video_remix" | "avideo_remix" | "video_list" | "avideo_list" | "video_retrieve_job" | "avideo_retrieve_job" | "video_delete" | "avideo_delete" | "video_create_character" | "avideo_create_character" | "video_get_character" | "avideo_get_character" | "video_edit" | "avideo_edit" | "video_extension" | "avideo_extension" | "vector_store_file_create" | "avector_store_file_create" | "vector_store_file_list" | "avector_store_file_list" | "vector_store_file_retrieve" | "avector_store_file_retrieve" | "vector_store_file_content" | "avector_store_file_content" | "vector_store_file_update" | "avector_store_file_update" | "vector_store_file_delete" | "avector_store_file_delete" | "vector_store_create" | "avector_store_create" | "vector_store_search" | "avector_store_search" | "ingest" | "aingest" | "query" | "aquery" | "create_interaction" | "acreate_interaction" | "create_container" | "acreate_container" | "list_containers" | "alist_containers" | "retrieve_container" | "aretrieve_container" | "delete_container" | "adelete_container" | "list_container_files" | "alist_container_files" | "upload_container_file" | "aupload_container_file" | "create_sandbox" | "acreate_sandbox" | "delete_sandbox" | "adelete_sandbox" | "run_code" | "arun_code" | "code_interpreter_tool" | "acode_interpreter_tool" | "acancel_fine_tuning_job" | "cancel_fine_tuning_job" | "alist_fine_tuning_jobs" | "list_fine_tuning_jobs" | "aretrieve_fine_tuning_job" | "retrieve_fine_tuning_job" | "responses" | "aresponses" | "alist_input_items" | "llm_passthrough_route" | "allm_passthrough_route" | "generate_content" | "agenerate_content" | "generate_content_stream" | "agenerate_content_stream" | "ocr" | "aocr" | "call_mcp_tool" | "list_mcp_tools" | "asend_message" | "send_message" | "acreate_skill";
/** CallbackDelete */
CallbackDelete: {
/** Callback Name */
@ -27123,6 +27285,12 @@ export interface components {
* @description opt-in to RFC 8628 verification_uri_complete for the CLI SSO device flow, pre-filling the user_code in the browser. Off by default; intended for same-host clients where the device that starts the flow and the browser run on the same machine
*/
allow_cli_sso_verification_uri_complete?: boolean | null;
/**
* Allow Non Billable Realtime Protocols
* @description Allow Realtime WebRTC setup endpoints whose inference usage bypasses LiteLLM spend tracking and budget enforcement
* @default false
*/
allow_non_billable_realtime_protocols: boolean;
/**
* Allow Unmanaged Response Ids
* @description If True, lets keys address Responses API ids that this proxy did not issue (raw provider ids, or ids issued before response-id encryption was configured). Such an id carries no owner, so no ownership check can run on it; ids this proxy did issue keep full ownership enforcement. Off by default, in which case an unrecognized response id is rejected with 403
@ -56324,7 +56492,7 @@ export interface operations {
};
};
};
websocket_realtime_websocket_endpoint_get_3: {
websocket_realtime_websocket_endpoint_get_6: {
parameters: {
query?: never;
header?: never;
@ -56402,6 +56570,64 @@ export interface operations {
};
};
};
websocket_realtime_websocket_endpoint_get_3: {
parameters: {
query?: never;
header?: never;
path?: never;
cookie?: never;
};
requestBody?: never;
responses: {
/** @description WebSocket Protocol Switched */
101: {
headers: {
[name: string]: unknown;
};
content?: never;
};
};
};
proxy_realtime_calls_openai_v1_realtime_translations_calls_post: {
parameters: {
query?: never;
header?: never;
path?: never;
cookie?: never;
};
requestBody?: never;
responses: {
/** @description Successful Response */
200: {
headers: {
[name: string]: unknown;
};
content: {
"application/json": unknown;
};
};
};
};
create_realtime_client_secret_openai_v1_realtime_translations_client_secrets_post: {
parameters: {
query?: never;
header?: never;
path?: never;
cookie?: never;
};
requestBody?: never;
responses: {
/** @description Successful Response */
200: {
headers: {
[name: string]: unknown;
};
content: {
"application/json": components["schemas"]["RealtimeClientSecretResponse"];
};
};
};
};
responses_api_openai_v1_responses_post: {
parameters: {
query?: never;
@ -59409,7 +59635,7 @@ export interface operations {
};
};
};
websocket_realtime_websocket_endpoint_get: {
websocket_realtime_websocket_endpoint_get_4: {
parameters: {
query?: never;
header?: never;
@ -59487,6 +59713,64 @@ export interface operations {
};
};
};
websocket_realtime_websocket_endpoint_get: {
parameters: {
query?: never;
header?: never;
path?: never;
cookie?: never;
};
requestBody?: never;
responses: {
/** @description WebSocket Protocol Switched */
101: {
headers: {
[name: string]: unknown;
};
content?: never;
};
};
};
proxy_realtime_calls_realtime_translations_calls_post: {
parameters: {
query?: never;
header?: never;
path?: never;
cookie?: never;
};
requestBody?: never;
responses: {
/** @description Successful Response */
200: {
headers: {
[name: string]: unknown;
};
content: {
"application/json": unknown;
};
};
};
};
create_realtime_client_secret_realtime_translations_client_secrets_post: {
parameters: {
query?: never;
header?: never;
path?: never;
cookie?: never;
};
requestBody?: never;
responses: {
/** @description Successful Response */
200: {
headers: {
[name: string]: unknown;
};
content: {
"application/json": components["schemas"]["RealtimeClientSecretResponse"];
};
};
};
};
register_client_register_post: {
parameters: {
query?: {
@ -68594,7 +68878,7 @@ export interface operations {
};
};
};
websocket_realtime_websocket_endpoint_get_2: {
websocket_realtime_websocket_endpoint_get_5: {
parameters: {
query?: never;
header?: never;
@ -68672,6 +68956,64 @@ export interface operations {
};
};
};
websocket_realtime_websocket_endpoint_get_2: {
parameters: {
query?: never;
header?: never;
path?: never;
cookie?: never;
};
requestBody?: never;
responses: {
/** @description WebSocket Protocol Switched */
101: {
headers: {
[name: string]: unknown;
};
content?: never;
};
};
};
proxy_realtime_calls_v1_realtime_translations_calls_post: {
parameters: {
query?: never;
header?: never;
path?: never;
cookie?: never;
};
requestBody?: never;
responses: {
/** @description Successful Response */
200: {
headers: {
[name: string]: unknown;
};
content: {
"application/json": unknown;
};
};
};
};
create_realtime_client_secret_v1_realtime_translations_client_secrets_post: {
parameters: {
query?: never;
header?: never;
path?: never;
cookie?: never;
};
requestBody?: never;
responses: {
/** @description Successful Response */
200: {
headers: {
[name: string]: unknown;
};
content: {
"application/json": components["schemas"]["RealtimeClientSecretResponse"];
};
};
};
};
rerank_v1_rerank_post: {
parameters: {
query?: never;