mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-04 02:31:27 +00:00
fix(realtime): reconcile model metadata and test contracts
This commit is contained in:
parent
ea84863217
commit
257c7f2e0a
11 changed files with 1012 additions and 97 deletions
|
|
@ -6533,9 +6533,9 @@ class BaseLLMHTTPHandler:
|
|||
try:
|
||||
configured_client: Final = openai_client.with_options(
|
||||
timeout=timeout,
|
||||
set_default_headers={ # mutable-ok: OpenAI SDK accepts a mutable custom-header mapping
|
||||
key: str(value) for key, value in (extra_headers or {}).items()
|
||||
},
|
||||
set_default_headers=MappingProxyType(
|
||||
{key: str(value) for key, value in (extra_headers or MappingProxyType({})).items()}
|
||||
),
|
||||
)
|
||||
raw_response: Final = await configured_client.post(
|
||||
"/realtime/translations/client_secrets",
|
||||
|
|
@ -6669,9 +6669,9 @@ class BaseLLMHTTPHandler:
|
|||
timeout=timeout,
|
||||
set_default_headers={ # mutable-ok: OpenAI SDK accepts a mutable custom-header mapping
|
||||
"Content-Type": "application/sdp",
|
||||
**{ # mutable-ok: caller headers are normalized into the SDK header mapping
|
||||
key: str(value) for key, value in (extra_headers or {}).items()
|
||||
},
|
||||
**MappingProxyType(
|
||||
{key: str(value) for key, value in (extra_headers or MappingProxyType({})).items()}
|
||||
),
|
||||
},
|
||||
)
|
||||
translation_response: Final = await configured_client.post(
|
||||
|
|
@ -6684,9 +6684,9 @@ class BaseLLMHTTPHandler:
|
|||
RealtimeSessionCreateRequestParam,
|
||||
session_data,
|
||||
)
|
||||
sdk_extra_headers: Final = { # mutable-ok: OpenAI SDK accepts a mutable custom-header mapping
|
||||
key: str(value) for key, value in (extra_headers or {}).items()
|
||||
}
|
||||
sdk_extra_headers: Final = MappingProxyType(
|
||||
{key: str(value) for key, value in (extra_headers or MappingProxyType({})).items()}
|
||||
)
|
||||
raw_response: Final = await openai_client.realtime.calls.with_raw_response.create(
|
||||
sdp=sdp_text,
|
||||
session=realtime_session_data,
|
||||
|
|
|
|||
|
|
@ -7,7 +7,7 @@ This requires websockets, and is currently only supported on LiteLLM Proxy.
|
|||
import ssl
|
||||
from collections.abc import Mapping
|
||||
from contextlib import AbstractAsyncContextManager
|
||||
from types import TracebackType
|
||||
from types import MappingProxyType, TracebackType
|
||||
from typing import Any, Final, cast
|
||||
|
||||
from openai import AsyncOpenAI, omit
|
||||
|
|
@ -180,13 +180,11 @@ class OpenAIRealtime(OpenAIChatCompletion):
|
|||
additional_headers=headers,
|
||||
max_size=REALTIME_WEBSOCKET_MAX_MESSAGE_SIZE_BYTES,
|
||||
ssl=ssl_config,
|
||||
**({"open_timeout": timeout} if timeout is not None else {}),
|
||||
**(MappingProxyType({"open_timeout": timeout}) if timeout is not None else MappingProxyType({})),
|
||||
)
|
||||
openai_client: Final = client
|
||||
model_query: Final = query_params.get("model")
|
||||
extra_query: Final = { # mutable-ok: OpenAI SDK accepts a mutable query-parameter mapping
|
||||
key: value for key, value in query_params.items() if key != "model"
|
||||
}
|
||||
extra_query: Final = MappingProxyType({key: value for key, value in query_params.items() if key != "model"})
|
||||
sdk_model: Final = omit if query_params.get("intent") == "transcription" else model_query or model
|
||||
sdk_connection_manager: Final = openai_client.realtime.connect(
|
||||
model=sdk_model,
|
||||
|
|
@ -194,8 +192,8 @@ class OpenAIRealtime(OpenAIChatCompletion):
|
|||
extra_headers=headers,
|
||||
websocket_connection_options={ # mutable-ok: OpenAI SDK forwards a mutable options mapping
|
||||
"max_size": REALTIME_WEBSOCKET_MAX_MESSAGE_SIZE_BYTES,
|
||||
**({"ssl": ssl_config} if url.startswith("wss://") else {}),
|
||||
**({"open_timeout": timeout} if timeout is not None else {}),
|
||||
**(MappingProxyType({"ssl": ssl_config}) if url.startswith("wss://") else MappingProxyType({})),
|
||||
**(MappingProxyType({"open_timeout": timeout}) if timeout is not None else MappingProxyType({})),
|
||||
},
|
||||
max_retries=0,
|
||||
)
|
||||
|
|
|
|||
|
|
@ -1,4 +1,229 @@
|
|||
{
|
||||
"azure/gpt-live-transcribe": {
|
||||
"input_cost_per_second": 0.0002833333333333333,
|
||||
"litellm_provider": "azure",
|
||||
"mode": "audio_transcription",
|
||||
"source": "https://techcommunity.microsoft.com/blog/azure-ai-foundry-blog/introducing-gpt-transcribe-and-gpt-live-transcribe-in-microsoft-foundry/4541740",
|
||||
"supported_endpoints": [
|
||||
"/v1/realtime",
|
||||
"/v1/realtime/transcription_sessions"
|
||||
],
|
||||
"supported_modalities": [
|
||||
"audio",
|
||||
"text"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"text"
|
||||
],
|
||||
"supports_audio_input": true,
|
||||
"supports_native_streaming": true
|
||||
},
|
||||
"azure/gpt-realtime-2.1-2026-07-07": {
|
||||
"cache_creation_input_audio_token_cost": 4e-07,
|
||||
"cache_read_input_audio_token_cost": 4e-07,
|
||||
"cache_read_input_token_cost": 4e-07,
|
||||
"input_cost_per_audio_token": 3.2e-05,
|
||||
"input_cost_per_image_token": 5e-06,
|
||||
"input_cost_per_token": 4e-06,
|
||||
"litellm_provider": "azure",
|
||||
"max_input_tokens": 32000,
|
||||
"max_output_tokens": 4096,
|
||||
"max_tokens": 4096,
|
||||
"mode": "realtime",
|
||||
"output_cost_per_audio_token": 6.4e-05,
|
||||
"output_cost_per_token": 2.4e-05,
|
||||
"source": "https://learn.microsoft.com/en-us/azure/foundry/foundry-models/concepts/models-sold-directly-by-azure",
|
||||
"supported_endpoints": [
|
||||
"/v1/realtime"
|
||||
],
|
||||
"supported_modalities": [
|
||||
"text",
|
||||
"image",
|
||||
"audio"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"text",
|
||||
"audio"
|
||||
],
|
||||
"supports_audio_input": true,
|
||||
"supports_audio_output": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"cache_read_input_image_token_cost": 5e-07,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true
|
||||
},
|
||||
"azure/gpt-realtime-2.1-mini-2026-07-07": {
|
||||
"cache_creation_input_audio_token_cost": 3e-07,
|
||||
"cache_read_input_audio_token_cost": 3e-07,
|
||||
"cache_read_input_token_cost": 6e-08,
|
||||
"input_cost_per_audio_token": 1e-05,
|
||||
"input_cost_per_image_token": 8e-07,
|
||||
"input_cost_per_token": 6e-07,
|
||||
"litellm_provider": "azure",
|
||||
"max_input_tokens": 32000,
|
||||
"max_output_tokens": 4096,
|
||||
"max_tokens": 4096,
|
||||
"mode": "realtime",
|
||||
"output_cost_per_audio_token": 2e-05,
|
||||
"output_cost_per_token": 2.4e-06,
|
||||
"source": "https://learn.microsoft.com/en-us/azure/foundry/foundry-models/concepts/models-sold-directly-by-azure",
|
||||
"supported_endpoints": [
|
||||
"/v1/realtime"
|
||||
],
|
||||
"supported_modalities": [
|
||||
"text",
|
||||
"image",
|
||||
"audio"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"text",
|
||||
"audio"
|
||||
],
|
||||
"supports_audio_input": true,
|
||||
"supports_audio_output": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"cache_read_input_image_token_cost": 8e-08,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true
|
||||
},
|
||||
"azure/gpt-realtime-translate": {
|
||||
"litellm_provider": "azure",
|
||||
"max_input_tokens": 32000,
|
||||
"max_output_tokens": 4096,
|
||||
"max_tokens": 4096,
|
||||
"mode": "realtime",
|
||||
"output_cost_per_second": 0.0005666666666666667,
|
||||
"source": "https://techcommunity.microsoft.com/blog/azure-ai-foundry-blog/a-new-chapter-for-realtime-ai-reasoning-translation-and-real-time-transcription/4517124",
|
||||
"supported_endpoints": [
|
||||
"/v1/realtime/translations",
|
||||
"/v1/realtime/translations/client_secrets",
|
||||
"/v1/realtime/translations/calls"
|
||||
],
|
||||
"supported_modalities": [
|
||||
"audio"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"audio",
|
||||
"text"
|
||||
],
|
||||
"supports_audio_input": true,
|
||||
"supports_audio_output": true,
|
||||
"supports_native_streaming": true
|
||||
},
|
||||
"azure/gpt-realtime-translate-2026-05-06": {
|
||||
"litellm_provider": "azure",
|
||||
"max_input_tokens": 32000,
|
||||
"max_output_tokens": 4096,
|
||||
"max_tokens": 4096,
|
||||
"mode": "realtime",
|
||||
"output_cost_per_second": 0.0005666666666666667,
|
||||
"source": "https://techcommunity.microsoft.com/blog/azure-ai-foundry-blog/a-new-chapter-for-realtime-ai-reasoning-translation-and-real-time-transcription/4517124",
|
||||
"supported_endpoints": [
|
||||
"/v1/realtime/translations",
|
||||
"/v1/realtime/translations/client_secrets",
|
||||
"/v1/realtime/translations/calls"
|
||||
],
|
||||
"supported_modalities": [
|
||||
"audio"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"audio",
|
||||
"text"
|
||||
],
|
||||
"supports_audio_input": true,
|
||||
"supports_audio_output": true,
|
||||
"supports_native_streaming": true
|
||||
},
|
||||
"azure/gpt-realtime-translate-2026-05-07": {
|
||||
"litellm_provider": "azure",
|
||||
"max_input_tokens": 32000,
|
||||
"max_output_tokens": 4096,
|
||||
"max_tokens": 4096,
|
||||
"mode": "realtime",
|
||||
"output_cost_per_second": 0.0005666666666666667,
|
||||
"source": "https://ai.azure.com/catalog/models/gpt-realtime-translate",
|
||||
"supported_endpoints": [
|
||||
"/v1/realtime/translations",
|
||||
"/v1/realtime/translations/client_secrets",
|
||||
"/v1/realtime/translations/calls"
|
||||
],
|
||||
"supported_modalities": [
|
||||
"audio"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"audio",
|
||||
"text"
|
||||
],
|
||||
"supports_audio_input": true,
|
||||
"supports_audio_output": true,
|
||||
"supports_native_streaming": true
|
||||
},
|
||||
"azure/gpt-realtime-whisper-2026-05-06": {
|
||||
"input_cost_per_second": 0.0002833333333333333,
|
||||
"litellm_provider": "azure",
|
||||
"mode": "audio_transcription",
|
||||
"source": "https://learn.microsoft.com/en-us/azure/foundry/openai/concepts/gpt-realtime-whisper",
|
||||
"supported_endpoints": [
|
||||
"/v1/realtime",
|
||||
"/v1/realtime/transcription_sessions"
|
||||
],
|
||||
"supported_modalities": [
|
||||
"audio"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"text"
|
||||
],
|
||||
"supports_audio_input": true,
|
||||
"max_input_tokens": 32000,
|
||||
"max_output_tokens": 4096,
|
||||
"max_tokens": 4096
|
||||
},
|
||||
"azure/gpt-realtime-whisper-2026-05-07": {
|
||||
"input_cost_per_second": 0.0002833333333333333,
|
||||
"litellm_provider": "azure",
|
||||
"mode": "audio_transcription",
|
||||
"source": "https://learn.microsoft.com/en-us/azure/foundry/openai/concepts/gpt-realtime-whisper",
|
||||
"supported_endpoints": [
|
||||
"/v1/realtime",
|
||||
"/v1/realtime/transcription_sessions"
|
||||
],
|
||||
"supported_modalities": [
|
||||
"audio"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"text"
|
||||
],
|
||||
"supports_audio_input": true,
|
||||
"max_input_tokens": 32000,
|
||||
"max_output_tokens": 4096,
|
||||
"max_tokens": 4096
|
||||
},
|
||||
"azure/gpt-transcribe": {
|
||||
"input_cost_per_second": 7.5e-05,
|
||||
"litellm_provider": "azure",
|
||||
"mode": "audio_transcription",
|
||||
"source": "https://techcommunity.microsoft.com/blog/azure-ai-foundry-blog/introducing-gpt-transcribe-and-gpt-live-transcribe-in-microsoft-foundry/4541740",
|
||||
"supported_endpoints": [
|
||||
"/v1/audio/transcriptions",
|
||||
"/v1/realtime",
|
||||
"/v1/realtime/transcription_sessions"
|
||||
],
|
||||
"supported_modalities": [
|
||||
"audio",
|
||||
"text"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"text"
|
||||
],
|
||||
"supports_audio_input": true,
|
||||
"supports_native_streaming": true
|
||||
},
|
||||
"sample_spec": {
|
||||
"code_interpreter_cost_per_session": 0.0,
|
||||
"computer_use_input_cost_per_1k_tokens": 0.0,
|
||||
|
|
@ -5821,6 +6046,7 @@
|
|||
"azure/gpt-realtime-2.1": {
|
||||
"cache_creation_input_audio_token_cost": 4e-07,
|
||||
"cache_read_input_audio_token_cost": 4e-07,
|
||||
"cache_read_input_image_token_cost": 5e-07,
|
||||
"cache_read_input_token_cost": 4e-07,
|
||||
"deprecation_date": "2027-06-25",
|
||||
"input_cost_per_audio_token": 3.2e-05,
|
||||
|
|
@ -5850,12 +6076,15 @@
|
|||
"supports_audio_output": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"azure/gpt-realtime-2.1-mini": {
|
||||
"cache_creation_input_audio_token_cost": 3e-07,
|
||||
"cache_read_input_audio_token_cost": 3e-07,
|
||||
"cache_read_input_image_token_cost": 8e-08,
|
||||
"cache_read_input_token_cost": 6e-08,
|
||||
"deprecation_date": "2027-06-25",
|
||||
"input_cost_per_audio_token": 1e-05,
|
||||
|
|
@ -5885,6 +6114,8 @@
|
|||
"supports_audio_output": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
|
|
@ -25462,7 +25693,6 @@
|
|||
"cache_read_input_token_cost": 2e-07,
|
||||
"cache_read_input_token_cost_above_200k_tokens": 4e-07,
|
||||
"cache_read_input_token_cost_above_200k_tokens_priority": 7.2e-07,
|
||||
"cache_read_input_token_cost_batches": 1e-07,
|
||||
"cache_read_input_token_cost_flex": 1e-07,
|
||||
"cache_read_input_token_cost_priority": 3.6e-07,
|
||||
"deprecation_date": "2027-05-28",
|
||||
|
|
@ -25555,7 +25785,6 @@
|
|||
},
|
||||
"gemini-3.1-flash-image": {
|
||||
"cache_read_input_token_cost": 5e-08,
|
||||
"cache_read_input_token_cost_batches": 2.5e-08,
|
||||
"cache_read_input_token_cost_flex": 2.5e-08,
|
||||
"deprecation_date": "2027-05-28",
|
||||
"input_cost_per_image": 0.00056,
|
||||
|
|
@ -25639,7 +25868,6 @@
|
|||
},
|
||||
"gemini-3.1-flash-lite-image": {
|
||||
"cache_read_input_token_cost": 2.5e-08,
|
||||
"cache_read_input_token_cost_batches": 1.25e-08,
|
||||
"cache_read_input_token_cost_flex": 1.25e-08,
|
||||
"input_cost_per_image": 0.00028,
|
||||
"input_cost_per_token": 2.5e-07,
|
||||
|
|
@ -25732,7 +25960,6 @@
|
|||
"cache_read_input_audio_token_cost": 5e-08,
|
||||
"deprecation_date": "2027-05-07",
|
||||
"cache_read_input_token_cost": 2.5e-08,
|
||||
"cache_read_input_token_cost_batches": 1.25e-08,
|
||||
"cache_read_input_token_cost_flex": 1.25e-08,
|
||||
"cache_read_input_token_cost_priority": 4.5e-08,
|
||||
"input_cost_per_audio_token": 5e-07,
|
||||
|
|
@ -25792,7 +26019,6 @@
|
|||
"gemini-3.5-flash-lite": {
|
||||
"deprecation_date": "2027-07-21",
|
||||
"cache_read_input_token_cost": 3e-08,
|
||||
"cache_read_input_token_cost_batches": 1.5e-08,
|
||||
"cache_read_input_token_cost_flex": 1.5e-08,
|
||||
"cache_read_input_token_cost_priority": 5.4e-08,
|
||||
"input_cost_per_token": 3e-07,
|
||||
|
|
@ -25849,7 +26075,6 @@
|
|||
},
|
||||
"deep-research-pro-preview-12-2025": {
|
||||
"cache_read_input_token_cost": 2e-07,
|
||||
"cache_read_input_token_cost_batches": 1e-07,
|
||||
"input_cost_per_image": 0.0011,
|
||||
"input_cost_per_token": 2e-06,
|
||||
"input_cost_per_token_batches": 1e-06,
|
||||
|
|
@ -26455,7 +26680,6 @@
|
|||
"prompt_cache_min_tokens": 4096,
|
||||
"deprecation_date": "2027-05-19",
|
||||
"cache_read_input_token_cost": 1.5e-07,
|
||||
"cache_read_input_token_cost_batches": 7.5e-08,
|
||||
"input_cost_per_token": 1.5e-06,
|
||||
"input_cost_per_audio_token": 1.5e-06,
|
||||
"litellm_provider": "vertex_ai",
|
||||
|
|
@ -26515,7 +26739,6 @@
|
|||
"vertex_ai/gemini-3.6-flash": {
|
||||
"prompt_cache_min_tokens": 4096,
|
||||
"cache_read_input_token_cost": 7.5e-08,
|
||||
"cache_read_input_token_cost_batches": 3.75e-08,
|
||||
"cache_read_input_token_cost_flex": 3.75e-08,
|
||||
"input_cost_per_token": 7.5e-07,
|
||||
"input_cost_per_token_batches": 3.75e-07,
|
||||
|
|
@ -26573,7 +26796,6 @@
|
|||
"vertex_ai/gemini-3.7-flash": {
|
||||
"prompt_cache_min_tokens": 4096,
|
||||
"cache_read_input_token_cost": 7.5e-08,
|
||||
"cache_read_input_token_cost_batches": 3.75e-08,
|
||||
"cache_read_input_token_cost_flex": 3.75e-08,
|
||||
"input_cost_per_token": 7.5e-07,
|
||||
"input_cost_per_token_batches": 3.75e-07,
|
||||
|
|
@ -26632,7 +26854,6 @@
|
|||
"vertex_ai/gemini-3.8-flash": {
|
||||
"prompt_cache_min_tokens": 4096,
|
||||
"cache_read_input_token_cost": 7.5e-08,
|
||||
"cache_read_input_token_cost_batches": 3.75e-08,
|
||||
"cache_read_input_token_cost_flex": 3.75e-08,
|
||||
"input_cost_per_token": 7.5e-07,
|
||||
"input_cost_per_token_batches": 3.75e-07,
|
||||
|
|
@ -28380,7 +28601,6 @@
|
|||
"prompt_cache_min_tokens": 4096,
|
||||
"deprecation_date": "2027-05-19",
|
||||
"cache_read_input_token_cost": 1.5e-07,
|
||||
"cache_read_input_token_cost_batches": 7.5e-08,
|
||||
"input_cost_per_audio_token": 1.5e-06,
|
||||
"input_cost_per_token": 1.5e-06,
|
||||
"litellm_provider": "vertex_ai-language-models",
|
||||
|
|
@ -28440,7 +28660,6 @@
|
|||
"gemini-3.6-flash": {
|
||||
"prompt_cache_min_tokens": 4096,
|
||||
"cache_read_input_token_cost": 7.5e-08,
|
||||
"cache_read_input_token_cost_batches": 3.75e-08,
|
||||
"cache_read_input_token_cost_flex": 3.75e-08,
|
||||
"input_cost_per_token": 7.5e-07,
|
||||
"input_cost_per_token_batches": 3.75e-07,
|
||||
|
|
@ -28498,7 +28717,6 @@
|
|||
"gemini-3.7-flash": {
|
||||
"prompt_cache_min_tokens": 4096,
|
||||
"cache_read_input_token_cost": 7.5e-08,
|
||||
"cache_read_input_token_cost_batches": 3.75e-08,
|
||||
"cache_read_input_token_cost_flex": 3.75e-08,
|
||||
"input_cost_per_token": 7.5e-07,
|
||||
"input_cost_per_token_batches": 3.75e-07,
|
||||
|
|
@ -28557,7 +28775,6 @@
|
|||
"gemini-3.8-flash": {
|
||||
"prompt_cache_min_tokens": 4096,
|
||||
"cache_read_input_token_cost": 7.5e-08,
|
||||
"cache_read_input_token_cost_batches": 3.75e-08,
|
||||
"cache_read_input_token_cost_flex": 3.75e-08,
|
||||
"input_cost_per_token": 7.5e-07,
|
||||
"input_cost_per_token_batches": 3.75e-07,
|
||||
|
|
@ -33245,7 +33462,6 @@
|
|||
},
|
||||
"gpt-5-2025-08-07": {
|
||||
"cache_read_input_token_cost": 1.25e-07,
|
||||
"cache_read_input_token_cost_batches": 6.25e-08,
|
||||
"cache_read_input_token_cost_flex": 6.25e-08,
|
||||
"cache_read_input_token_cost_priority": 2.5e-07,
|
||||
"deprecation_date": "2026-12-11",
|
||||
|
|
@ -33426,7 +33642,6 @@
|
|||
},
|
||||
"gpt-5-mini-2025-08-07": {
|
||||
"cache_read_input_token_cost": 2.5e-08,
|
||||
"cache_read_input_token_cost_batches": 1.25e-08,
|
||||
"cache_read_input_token_cost_flex": 1.25e-08,
|
||||
"cache_read_input_token_cost_priority": 4.5e-08,
|
||||
"deprecation_date": "2026-12-11",
|
||||
|
|
@ -33527,7 +33742,6 @@
|
|||
},
|
||||
"gpt-5-nano-2025-08-07": {
|
||||
"cache_read_input_token_cost": 5e-09,
|
||||
"cache_read_input_token_cost_batches": 2.5e-09,
|
||||
"cache_read_input_token_cost_flex": 2.5e-09,
|
||||
"deprecation_date": "2026-12-11",
|
||||
"input_cost_per_token": 5e-08,
|
||||
|
|
@ -33715,6 +33929,7 @@
|
|||
"gpt-realtime-2.1": {
|
||||
"cache_creation_input_audio_token_cost": 4e-07,
|
||||
"cache_read_input_audio_token_cost": 4e-07,
|
||||
"cache_read_input_image_token_cost": 5e-07,
|
||||
"cache_read_input_token_cost": 4e-07,
|
||||
"input_cost_per_audio_token": 3.2e-05,
|
||||
"input_cost_per_image_token": 5e-06,
|
||||
|
|
@ -33745,12 +33960,15 @@
|
|||
"supports_audio_output": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"gpt-realtime-2.1-mini": {
|
||||
"cache_creation_input_audio_token_cost": 3e-07,
|
||||
"cache_read_input_audio_token_cost": 3e-07,
|
||||
"cache_read_input_image_token_cost": 8e-08,
|
||||
"cache_read_input_token_cost": 6e-08,
|
||||
"input_cost_per_audio_token": 1e-05,
|
||||
"input_cost_per_image_token": 8e-07,
|
||||
|
|
@ -33781,6 +33999,8 @@
|
|||
"supports_audio_output": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
|
|
@ -47859,7 +48079,6 @@
|
|||
"cache_read_input_token_cost": 2e-07,
|
||||
"cache_read_input_token_cost_above_200k_tokens": 4e-07,
|
||||
"cache_read_input_token_cost_above_200k_tokens_priority": 7.2e-07,
|
||||
"cache_read_input_token_cost_batches": 1e-07,
|
||||
"cache_read_input_token_cost_flex": 1e-07,
|
||||
"cache_read_input_token_cost_priority": 3.6e-07,
|
||||
"deprecation_date": "2027-05-28",
|
||||
|
|
@ -47904,7 +48123,6 @@
|
|||
},
|
||||
"vertex_ai/gemini-3.1-flash-image": {
|
||||
"cache_read_input_token_cost": 5e-08,
|
||||
"cache_read_input_token_cost_batches": 2.5e-08,
|
||||
"cache_read_input_token_cost_flex": 2.5e-08,
|
||||
"deprecation_date": "2027-05-28",
|
||||
"input_cost_per_image": 0.00056,
|
||||
|
|
@ -47940,7 +48158,6 @@
|
|||
},
|
||||
"vertex_ai/gemini-3.1-flash-lite-image": {
|
||||
"cache_read_input_token_cost": 2.5e-08,
|
||||
"cache_read_input_token_cost_batches": 1.25e-08,
|
||||
"cache_read_input_token_cost_flex": 1.25e-08,
|
||||
"input_cost_per_image": 0.00028,
|
||||
"input_cost_per_token": 2.5e-07,
|
||||
|
|
@ -48033,7 +48250,6 @@
|
|||
"cache_read_input_audio_token_cost": 5e-08,
|
||||
"deprecation_date": "2027-05-07",
|
||||
"cache_read_input_token_cost": 2.5e-08,
|
||||
"cache_read_input_token_cost_batches": 1.25e-08,
|
||||
"cache_read_input_token_cost_flex": 1.25e-08,
|
||||
"cache_read_input_token_cost_priority": 4.5e-08,
|
||||
"input_cost_per_audio_token": 5e-07,
|
||||
|
|
@ -48094,7 +48310,6 @@
|
|||
"vertex_ai/gemini-3.5-flash-lite": {
|
||||
"deprecation_date": "2027-07-21",
|
||||
"cache_read_input_token_cost": 3e-08,
|
||||
"cache_read_input_token_cost_batches": 1.5e-08,
|
||||
"cache_read_input_token_cost_flex": 1.5e-08,
|
||||
"cache_read_input_token_cost_priority": 5.4e-08,
|
||||
"input_cost_per_token": 3e-07,
|
||||
|
|
@ -48152,7 +48367,6 @@
|
|||
},
|
||||
"vertex_ai/deep-research-pro-preview-12-2025": {
|
||||
"cache_read_input_token_cost": 2e-07,
|
||||
"cache_read_input_token_cost_batches": 1e-07,
|
||||
"input_cost_per_image": 0.0011,
|
||||
"input_cost_per_token": 2e-06,
|
||||
"input_cost_per_token_batches": 1e-06,
|
||||
|
|
@ -57348,7 +57562,8 @@
|
|||
"supported_output_modalities": [
|
||||
"text"
|
||||
],
|
||||
"supports_audio_input": true
|
||||
"supports_audio_input": true,
|
||||
"supports_native_streaming": true
|
||||
},
|
||||
"gpt-live-transcribe": {
|
||||
"input_cost_per_second": 0.000283333333333,
|
||||
|
|
@ -57366,7 +57581,8 @@
|
|||
"supported_output_modalities": [
|
||||
"text"
|
||||
],
|
||||
"supports_audio_input": true
|
||||
"supports_audio_input": true,
|
||||
"supports_native_streaming": true
|
||||
},
|
||||
"gpt-live-1": {
|
||||
"input_cost_per_second": 0.000833333333333,
|
||||
|
|
@ -57392,7 +57608,13 @@
|
|||
"max_output_tokens": 2000,
|
||||
"max_tokens": 2000,
|
||||
"mode": "realtime",
|
||||
"output_cost_per_second": 0.0005666666666666667,
|
||||
"source": "https://developers.openai.com/api/docs/pricing",
|
||||
"supported_endpoints": [
|
||||
"/v1/realtime/translations",
|
||||
"/v1/realtime/translations/client_secrets",
|
||||
"/v1/realtime/translations/calls"
|
||||
],
|
||||
"supported_modalities": [
|
||||
"audio"
|
||||
],
|
||||
|
|
@ -57401,7 +57623,8 @@
|
|||
"audio"
|
||||
],
|
||||
"supports_audio_input": true,
|
||||
"supports_audio_output": true
|
||||
"supports_audio_output": true,
|
||||
"supports_native_streaming": true
|
||||
},
|
||||
"claude-mythos-5": {
|
||||
"supports_anthropic_compaction": true,
|
||||
|
|
|
|||
|
|
@ -1,4 +1,229 @@
|
|||
{
|
||||
"azure/gpt-live-transcribe": {
|
||||
"input_cost_per_second": 0.0002833333333333333,
|
||||
"litellm_provider": "azure",
|
||||
"mode": "audio_transcription",
|
||||
"source": "https://techcommunity.microsoft.com/blog/azure-ai-foundry-blog/introducing-gpt-transcribe-and-gpt-live-transcribe-in-microsoft-foundry/4541740",
|
||||
"supported_endpoints": [
|
||||
"/v1/realtime",
|
||||
"/v1/realtime/transcription_sessions"
|
||||
],
|
||||
"supported_modalities": [
|
||||
"audio",
|
||||
"text"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"text"
|
||||
],
|
||||
"supports_audio_input": true,
|
||||
"supports_native_streaming": true
|
||||
},
|
||||
"azure/gpt-realtime-2.1-2026-07-07": {
|
||||
"cache_creation_input_audio_token_cost": 4e-07,
|
||||
"cache_read_input_audio_token_cost": 4e-07,
|
||||
"cache_read_input_token_cost": 4e-07,
|
||||
"input_cost_per_audio_token": 3.2e-05,
|
||||
"input_cost_per_image_token": 5e-06,
|
||||
"input_cost_per_token": 4e-06,
|
||||
"litellm_provider": "azure",
|
||||
"max_input_tokens": 32000,
|
||||
"max_output_tokens": 4096,
|
||||
"max_tokens": 4096,
|
||||
"mode": "realtime",
|
||||
"output_cost_per_audio_token": 6.4e-05,
|
||||
"output_cost_per_token": 2.4e-05,
|
||||
"source": "https://learn.microsoft.com/en-us/azure/foundry/foundry-models/concepts/models-sold-directly-by-azure",
|
||||
"supported_endpoints": [
|
||||
"/v1/realtime"
|
||||
],
|
||||
"supported_modalities": [
|
||||
"text",
|
||||
"image",
|
||||
"audio"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"text",
|
||||
"audio"
|
||||
],
|
||||
"supports_audio_input": true,
|
||||
"supports_audio_output": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"cache_read_input_image_token_cost": 5e-07,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true
|
||||
},
|
||||
"azure/gpt-realtime-2.1-mini-2026-07-07": {
|
||||
"cache_creation_input_audio_token_cost": 3e-07,
|
||||
"cache_read_input_audio_token_cost": 3e-07,
|
||||
"cache_read_input_token_cost": 6e-08,
|
||||
"input_cost_per_audio_token": 1e-05,
|
||||
"input_cost_per_image_token": 8e-07,
|
||||
"input_cost_per_token": 6e-07,
|
||||
"litellm_provider": "azure",
|
||||
"max_input_tokens": 32000,
|
||||
"max_output_tokens": 4096,
|
||||
"max_tokens": 4096,
|
||||
"mode": "realtime",
|
||||
"output_cost_per_audio_token": 2e-05,
|
||||
"output_cost_per_token": 2.4e-06,
|
||||
"source": "https://learn.microsoft.com/en-us/azure/foundry/foundry-models/concepts/models-sold-directly-by-azure",
|
||||
"supported_endpoints": [
|
||||
"/v1/realtime"
|
||||
],
|
||||
"supported_modalities": [
|
||||
"text",
|
||||
"image",
|
||||
"audio"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"text",
|
||||
"audio"
|
||||
],
|
||||
"supports_audio_input": true,
|
||||
"supports_audio_output": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"cache_read_input_image_token_cost": 8e-08,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true
|
||||
},
|
||||
"azure/gpt-realtime-translate": {
|
||||
"litellm_provider": "azure",
|
||||
"max_input_tokens": 32000,
|
||||
"max_output_tokens": 4096,
|
||||
"max_tokens": 4096,
|
||||
"mode": "realtime",
|
||||
"output_cost_per_second": 0.0005666666666666667,
|
||||
"source": "https://techcommunity.microsoft.com/blog/azure-ai-foundry-blog/a-new-chapter-for-realtime-ai-reasoning-translation-and-real-time-transcription/4517124",
|
||||
"supported_endpoints": [
|
||||
"/v1/realtime/translations",
|
||||
"/v1/realtime/translations/client_secrets",
|
||||
"/v1/realtime/translations/calls"
|
||||
],
|
||||
"supported_modalities": [
|
||||
"audio"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"audio",
|
||||
"text"
|
||||
],
|
||||
"supports_audio_input": true,
|
||||
"supports_audio_output": true,
|
||||
"supports_native_streaming": true
|
||||
},
|
||||
"azure/gpt-realtime-translate-2026-05-06": {
|
||||
"litellm_provider": "azure",
|
||||
"max_input_tokens": 32000,
|
||||
"max_output_tokens": 4096,
|
||||
"max_tokens": 4096,
|
||||
"mode": "realtime",
|
||||
"output_cost_per_second": 0.0005666666666666667,
|
||||
"source": "https://techcommunity.microsoft.com/blog/azure-ai-foundry-blog/a-new-chapter-for-realtime-ai-reasoning-translation-and-real-time-transcription/4517124",
|
||||
"supported_endpoints": [
|
||||
"/v1/realtime/translations",
|
||||
"/v1/realtime/translations/client_secrets",
|
||||
"/v1/realtime/translations/calls"
|
||||
],
|
||||
"supported_modalities": [
|
||||
"audio"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"audio",
|
||||
"text"
|
||||
],
|
||||
"supports_audio_input": true,
|
||||
"supports_audio_output": true,
|
||||
"supports_native_streaming": true
|
||||
},
|
||||
"azure/gpt-realtime-translate-2026-05-07": {
|
||||
"litellm_provider": "azure",
|
||||
"max_input_tokens": 32000,
|
||||
"max_output_tokens": 4096,
|
||||
"max_tokens": 4096,
|
||||
"mode": "realtime",
|
||||
"output_cost_per_second": 0.0005666666666666667,
|
||||
"source": "https://ai.azure.com/catalog/models/gpt-realtime-translate",
|
||||
"supported_endpoints": [
|
||||
"/v1/realtime/translations",
|
||||
"/v1/realtime/translations/client_secrets",
|
||||
"/v1/realtime/translations/calls"
|
||||
],
|
||||
"supported_modalities": [
|
||||
"audio"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"audio",
|
||||
"text"
|
||||
],
|
||||
"supports_audio_input": true,
|
||||
"supports_audio_output": true,
|
||||
"supports_native_streaming": true
|
||||
},
|
||||
"azure/gpt-realtime-whisper-2026-05-06": {
|
||||
"input_cost_per_second": 0.0002833333333333333,
|
||||
"litellm_provider": "azure",
|
||||
"mode": "audio_transcription",
|
||||
"source": "https://learn.microsoft.com/en-us/azure/foundry/openai/concepts/gpt-realtime-whisper",
|
||||
"supported_endpoints": [
|
||||
"/v1/realtime",
|
||||
"/v1/realtime/transcription_sessions"
|
||||
],
|
||||
"supported_modalities": [
|
||||
"audio"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"text"
|
||||
],
|
||||
"supports_audio_input": true,
|
||||
"max_input_tokens": 32000,
|
||||
"max_output_tokens": 4096,
|
||||
"max_tokens": 4096
|
||||
},
|
||||
"azure/gpt-realtime-whisper-2026-05-07": {
|
||||
"input_cost_per_second": 0.0002833333333333333,
|
||||
"litellm_provider": "azure",
|
||||
"mode": "audio_transcription",
|
||||
"source": "https://learn.microsoft.com/en-us/azure/foundry/openai/concepts/gpt-realtime-whisper",
|
||||
"supported_endpoints": [
|
||||
"/v1/realtime",
|
||||
"/v1/realtime/transcription_sessions"
|
||||
],
|
||||
"supported_modalities": [
|
||||
"audio"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"text"
|
||||
],
|
||||
"supports_audio_input": true,
|
||||
"max_input_tokens": 32000,
|
||||
"max_output_tokens": 4096,
|
||||
"max_tokens": 4096
|
||||
},
|
||||
"azure/gpt-transcribe": {
|
||||
"input_cost_per_second": 7.5e-05,
|
||||
"litellm_provider": "azure",
|
||||
"mode": "audio_transcription",
|
||||
"source": "https://techcommunity.microsoft.com/blog/azure-ai-foundry-blog/introducing-gpt-transcribe-and-gpt-live-transcribe-in-microsoft-foundry/4541740",
|
||||
"supported_endpoints": [
|
||||
"/v1/audio/transcriptions",
|
||||
"/v1/realtime",
|
||||
"/v1/realtime/transcription_sessions"
|
||||
],
|
||||
"supported_modalities": [
|
||||
"audio",
|
||||
"text"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"text"
|
||||
],
|
||||
"supports_audio_input": true,
|
||||
"supports_native_streaming": true
|
||||
},
|
||||
"sample_spec": {
|
||||
"code_interpreter_cost_per_session": 0.0,
|
||||
"computer_use_input_cost_per_1k_tokens": 0.0,
|
||||
|
|
@ -5821,6 +6046,7 @@
|
|||
"azure/gpt-realtime-2.1": {
|
||||
"cache_creation_input_audio_token_cost": 4e-07,
|
||||
"cache_read_input_audio_token_cost": 4e-07,
|
||||
"cache_read_input_image_token_cost": 5e-07,
|
||||
"cache_read_input_token_cost": 4e-07,
|
||||
"deprecation_date": "2027-06-25",
|
||||
"input_cost_per_audio_token": 3.2e-05,
|
||||
|
|
@ -5850,12 +6076,15 @@
|
|||
"supports_audio_output": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"azure/gpt-realtime-2.1-mini": {
|
||||
"cache_creation_input_audio_token_cost": 3e-07,
|
||||
"cache_read_input_audio_token_cost": 3e-07,
|
||||
"cache_read_input_image_token_cost": 8e-08,
|
||||
"cache_read_input_token_cost": 6e-08,
|
||||
"deprecation_date": "2027-06-25",
|
||||
"input_cost_per_audio_token": 1e-05,
|
||||
|
|
@ -5885,6 +6114,8 @@
|
|||
"supports_audio_output": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
|
|
@ -25462,7 +25693,6 @@
|
|||
"cache_read_input_token_cost": 2e-07,
|
||||
"cache_read_input_token_cost_above_200k_tokens": 4e-07,
|
||||
"cache_read_input_token_cost_above_200k_tokens_priority": 7.2e-07,
|
||||
"cache_read_input_token_cost_batches": 1e-07,
|
||||
"cache_read_input_token_cost_flex": 1e-07,
|
||||
"cache_read_input_token_cost_priority": 3.6e-07,
|
||||
"deprecation_date": "2027-05-28",
|
||||
|
|
@ -25555,7 +25785,6 @@
|
|||
},
|
||||
"gemini-3.1-flash-image": {
|
||||
"cache_read_input_token_cost": 5e-08,
|
||||
"cache_read_input_token_cost_batches": 2.5e-08,
|
||||
"cache_read_input_token_cost_flex": 2.5e-08,
|
||||
"deprecation_date": "2027-05-28",
|
||||
"input_cost_per_image": 0.00056,
|
||||
|
|
@ -25639,7 +25868,6 @@
|
|||
},
|
||||
"gemini-3.1-flash-lite-image": {
|
||||
"cache_read_input_token_cost": 2.5e-08,
|
||||
"cache_read_input_token_cost_batches": 1.25e-08,
|
||||
"cache_read_input_token_cost_flex": 1.25e-08,
|
||||
"input_cost_per_image": 0.00028,
|
||||
"input_cost_per_token": 2.5e-07,
|
||||
|
|
@ -25732,7 +25960,6 @@
|
|||
"cache_read_input_audio_token_cost": 5e-08,
|
||||
"deprecation_date": "2027-05-07",
|
||||
"cache_read_input_token_cost": 2.5e-08,
|
||||
"cache_read_input_token_cost_batches": 1.25e-08,
|
||||
"cache_read_input_token_cost_flex": 1.25e-08,
|
||||
"cache_read_input_token_cost_priority": 4.5e-08,
|
||||
"input_cost_per_audio_token": 5e-07,
|
||||
|
|
@ -25792,7 +26019,6 @@
|
|||
"gemini-3.5-flash-lite": {
|
||||
"deprecation_date": "2027-07-21",
|
||||
"cache_read_input_token_cost": 3e-08,
|
||||
"cache_read_input_token_cost_batches": 1.5e-08,
|
||||
"cache_read_input_token_cost_flex": 1.5e-08,
|
||||
"cache_read_input_token_cost_priority": 5.4e-08,
|
||||
"input_cost_per_token": 3e-07,
|
||||
|
|
@ -25849,7 +26075,6 @@
|
|||
},
|
||||
"deep-research-pro-preview-12-2025": {
|
||||
"cache_read_input_token_cost": 2e-07,
|
||||
"cache_read_input_token_cost_batches": 1e-07,
|
||||
"input_cost_per_image": 0.0011,
|
||||
"input_cost_per_token": 2e-06,
|
||||
"input_cost_per_token_batches": 1e-06,
|
||||
|
|
@ -26455,7 +26680,6 @@
|
|||
"prompt_cache_min_tokens": 4096,
|
||||
"deprecation_date": "2027-05-19",
|
||||
"cache_read_input_token_cost": 1.5e-07,
|
||||
"cache_read_input_token_cost_batches": 7.5e-08,
|
||||
"input_cost_per_token": 1.5e-06,
|
||||
"input_cost_per_audio_token": 1.5e-06,
|
||||
"litellm_provider": "vertex_ai",
|
||||
|
|
@ -26515,7 +26739,6 @@
|
|||
"vertex_ai/gemini-3.6-flash": {
|
||||
"prompt_cache_min_tokens": 4096,
|
||||
"cache_read_input_token_cost": 7.5e-08,
|
||||
"cache_read_input_token_cost_batches": 3.75e-08,
|
||||
"cache_read_input_token_cost_flex": 3.75e-08,
|
||||
"input_cost_per_token": 7.5e-07,
|
||||
"input_cost_per_token_batches": 3.75e-07,
|
||||
|
|
@ -26573,7 +26796,6 @@
|
|||
"vertex_ai/gemini-3.7-flash": {
|
||||
"prompt_cache_min_tokens": 4096,
|
||||
"cache_read_input_token_cost": 7.5e-08,
|
||||
"cache_read_input_token_cost_batches": 3.75e-08,
|
||||
"cache_read_input_token_cost_flex": 3.75e-08,
|
||||
"input_cost_per_token": 7.5e-07,
|
||||
"input_cost_per_token_batches": 3.75e-07,
|
||||
|
|
@ -26632,7 +26854,6 @@
|
|||
"vertex_ai/gemini-3.8-flash": {
|
||||
"prompt_cache_min_tokens": 4096,
|
||||
"cache_read_input_token_cost": 7.5e-08,
|
||||
"cache_read_input_token_cost_batches": 3.75e-08,
|
||||
"cache_read_input_token_cost_flex": 3.75e-08,
|
||||
"input_cost_per_token": 7.5e-07,
|
||||
"input_cost_per_token_batches": 3.75e-07,
|
||||
|
|
@ -28380,7 +28601,6 @@
|
|||
"prompt_cache_min_tokens": 4096,
|
||||
"deprecation_date": "2027-05-19",
|
||||
"cache_read_input_token_cost": 1.5e-07,
|
||||
"cache_read_input_token_cost_batches": 7.5e-08,
|
||||
"input_cost_per_audio_token": 1.5e-06,
|
||||
"input_cost_per_token": 1.5e-06,
|
||||
"litellm_provider": "vertex_ai-language-models",
|
||||
|
|
@ -28440,7 +28660,6 @@
|
|||
"gemini-3.6-flash": {
|
||||
"prompt_cache_min_tokens": 4096,
|
||||
"cache_read_input_token_cost": 7.5e-08,
|
||||
"cache_read_input_token_cost_batches": 3.75e-08,
|
||||
"cache_read_input_token_cost_flex": 3.75e-08,
|
||||
"input_cost_per_token": 7.5e-07,
|
||||
"input_cost_per_token_batches": 3.75e-07,
|
||||
|
|
@ -28498,7 +28717,6 @@
|
|||
"gemini-3.7-flash": {
|
||||
"prompt_cache_min_tokens": 4096,
|
||||
"cache_read_input_token_cost": 7.5e-08,
|
||||
"cache_read_input_token_cost_batches": 3.75e-08,
|
||||
"cache_read_input_token_cost_flex": 3.75e-08,
|
||||
"input_cost_per_token": 7.5e-07,
|
||||
"input_cost_per_token_batches": 3.75e-07,
|
||||
|
|
@ -28557,7 +28775,6 @@
|
|||
"gemini-3.8-flash": {
|
||||
"prompt_cache_min_tokens": 4096,
|
||||
"cache_read_input_token_cost": 7.5e-08,
|
||||
"cache_read_input_token_cost_batches": 3.75e-08,
|
||||
"cache_read_input_token_cost_flex": 3.75e-08,
|
||||
"input_cost_per_token": 7.5e-07,
|
||||
"input_cost_per_token_batches": 3.75e-07,
|
||||
|
|
@ -33245,7 +33462,6 @@
|
|||
},
|
||||
"gpt-5-2025-08-07": {
|
||||
"cache_read_input_token_cost": 1.25e-07,
|
||||
"cache_read_input_token_cost_batches": 6.25e-08,
|
||||
"cache_read_input_token_cost_flex": 6.25e-08,
|
||||
"cache_read_input_token_cost_priority": 2.5e-07,
|
||||
"deprecation_date": "2026-12-11",
|
||||
|
|
@ -33426,7 +33642,6 @@
|
|||
},
|
||||
"gpt-5-mini-2025-08-07": {
|
||||
"cache_read_input_token_cost": 2.5e-08,
|
||||
"cache_read_input_token_cost_batches": 1.25e-08,
|
||||
"cache_read_input_token_cost_flex": 1.25e-08,
|
||||
"cache_read_input_token_cost_priority": 4.5e-08,
|
||||
"deprecation_date": "2026-12-11",
|
||||
|
|
@ -33527,7 +33742,6 @@
|
|||
},
|
||||
"gpt-5-nano-2025-08-07": {
|
||||
"cache_read_input_token_cost": 5e-09,
|
||||
"cache_read_input_token_cost_batches": 2.5e-09,
|
||||
"cache_read_input_token_cost_flex": 2.5e-09,
|
||||
"deprecation_date": "2026-12-11",
|
||||
"input_cost_per_token": 5e-08,
|
||||
|
|
@ -33715,6 +33929,7 @@
|
|||
"gpt-realtime-2.1": {
|
||||
"cache_creation_input_audio_token_cost": 4e-07,
|
||||
"cache_read_input_audio_token_cost": 4e-07,
|
||||
"cache_read_input_image_token_cost": 5e-07,
|
||||
"cache_read_input_token_cost": 4e-07,
|
||||
"input_cost_per_audio_token": 3.2e-05,
|
||||
"input_cost_per_image_token": 5e-06,
|
||||
|
|
@ -33745,12 +33960,15 @@
|
|||
"supports_audio_output": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"gpt-realtime-2.1-mini": {
|
||||
"cache_creation_input_audio_token_cost": 3e-07,
|
||||
"cache_read_input_audio_token_cost": 3e-07,
|
||||
"cache_read_input_image_token_cost": 8e-08,
|
||||
"cache_read_input_token_cost": 6e-08,
|
||||
"input_cost_per_audio_token": 1e-05,
|
||||
"input_cost_per_image_token": 8e-07,
|
||||
|
|
@ -33781,6 +33999,8 @@
|
|||
"supports_audio_output": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
|
|
@ -47859,7 +48079,6 @@
|
|||
"cache_read_input_token_cost": 2e-07,
|
||||
"cache_read_input_token_cost_above_200k_tokens": 4e-07,
|
||||
"cache_read_input_token_cost_above_200k_tokens_priority": 7.2e-07,
|
||||
"cache_read_input_token_cost_batches": 1e-07,
|
||||
"cache_read_input_token_cost_flex": 1e-07,
|
||||
"cache_read_input_token_cost_priority": 3.6e-07,
|
||||
"deprecation_date": "2027-05-28",
|
||||
|
|
@ -47904,7 +48123,6 @@
|
|||
},
|
||||
"vertex_ai/gemini-3.1-flash-image": {
|
||||
"cache_read_input_token_cost": 5e-08,
|
||||
"cache_read_input_token_cost_batches": 2.5e-08,
|
||||
"cache_read_input_token_cost_flex": 2.5e-08,
|
||||
"deprecation_date": "2027-05-28",
|
||||
"input_cost_per_image": 0.00056,
|
||||
|
|
@ -47940,7 +48158,6 @@
|
|||
},
|
||||
"vertex_ai/gemini-3.1-flash-lite-image": {
|
||||
"cache_read_input_token_cost": 2.5e-08,
|
||||
"cache_read_input_token_cost_batches": 1.25e-08,
|
||||
"cache_read_input_token_cost_flex": 1.25e-08,
|
||||
"input_cost_per_image": 0.00028,
|
||||
"input_cost_per_token": 2.5e-07,
|
||||
|
|
@ -48033,7 +48250,6 @@
|
|||
"cache_read_input_audio_token_cost": 5e-08,
|
||||
"deprecation_date": "2027-05-07",
|
||||
"cache_read_input_token_cost": 2.5e-08,
|
||||
"cache_read_input_token_cost_batches": 1.25e-08,
|
||||
"cache_read_input_token_cost_flex": 1.25e-08,
|
||||
"cache_read_input_token_cost_priority": 4.5e-08,
|
||||
"input_cost_per_audio_token": 5e-07,
|
||||
|
|
@ -48094,7 +48310,6 @@
|
|||
"vertex_ai/gemini-3.5-flash-lite": {
|
||||
"deprecation_date": "2027-07-21",
|
||||
"cache_read_input_token_cost": 3e-08,
|
||||
"cache_read_input_token_cost_batches": 1.5e-08,
|
||||
"cache_read_input_token_cost_flex": 1.5e-08,
|
||||
"cache_read_input_token_cost_priority": 5.4e-08,
|
||||
"input_cost_per_token": 3e-07,
|
||||
|
|
@ -48152,7 +48367,6 @@
|
|||
},
|
||||
"vertex_ai/deep-research-pro-preview-12-2025": {
|
||||
"cache_read_input_token_cost": 2e-07,
|
||||
"cache_read_input_token_cost_batches": 1e-07,
|
||||
"input_cost_per_image": 0.0011,
|
||||
"input_cost_per_token": 2e-06,
|
||||
"input_cost_per_token_batches": 1e-06,
|
||||
|
|
@ -57348,7 +57562,8 @@
|
|||
"supported_output_modalities": [
|
||||
"text"
|
||||
],
|
||||
"supports_audio_input": true
|
||||
"supports_audio_input": true,
|
||||
"supports_native_streaming": true
|
||||
},
|
||||
"gpt-live-transcribe": {
|
||||
"input_cost_per_second": 0.000283333333333,
|
||||
|
|
@ -57366,7 +57581,8 @@
|
|||
"supported_output_modalities": [
|
||||
"text"
|
||||
],
|
||||
"supports_audio_input": true
|
||||
"supports_audio_input": true,
|
||||
"supports_native_streaming": true
|
||||
},
|
||||
"gpt-live-1": {
|
||||
"input_cost_per_second": 0.000833333333333,
|
||||
|
|
@ -57392,7 +57608,13 @@
|
|||
"max_output_tokens": 2000,
|
||||
"max_tokens": 2000,
|
||||
"mode": "realtime",
|
||||
"output_cost_per_second": 0.0005666666666666667,
|
||||
"source": "https://developers.openai.com/api/docs/pricing",
|
||||
"supported_endpoints": [
|
||||
"/v1/realtime/translations",
|
||||
"/v1/realtime/translations/client_secrets",
|
||||
"/v1/realtime/translations/calls"
|
||||
],
|
||||
"supported_modalities": [
|
||||
"audio"
|
||||
],
|
||||
|
|
@ -57401,7 +57623,8 @@
|
|||
"audio"
|
||||
],
|
||||
"supports_audio_input": true,
|
||||
"supports_audio_output": true
|
||||
"supports_audio_output": true,
|
||||
"supports_native_streaming": true
|
||||
},
|
||||
"claude-mythos-5": {
|
||||
"supports_anthropic_compaction": true,
|
||||
|
|
|
|||
|
|
@ -29,6 +29,7 @@ from litellm.llms.gemini.image_generation.cost_calculator import (
|
|||
from litellm.llms.vertex_ai.image_generation.cost_calculator import (
|
||||
cost_calculator as vertex_image_generation_cost_calculator,
|
||||
)
|
||||
from litellm.types.llms.base import CachedTokensDetails
|
||||
from litellm.types.utils import (
|
||||
CacheCreationTokenDetails,
|
||||
CompletionTokensDetailsWrapper,
|
||||
|
|
@ -42,6 +43,39 @@ from litellm.types.utils import (
|
|||
)
|
||||
|
||||
|
||||
def test_realtime_cached_modality_breakdown_matches_prompt_cost(_local_model_cost_map):
|
||||
model: Final = "gpt-realtime-2.1-mini"
|
||||
rates: Final = litellm.model_cost[model]
|
||||
usage: Final = Usage(
|
||||
prompt_tokens=1000,
|
||||
completion_tokens=0,
|
||||
total_tokens=1000,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(
|
||||
text_tokens=400,
|
||||
audio_tokens=400,
|
||||
image_tokens=200,
|
||||
cached_tokens=300,
|
||||
cached_tokens_details=CachedTokensDetails(text_tokens=100, audio_tokens=150, image_tokens=50),
|
||||
),
|
||||
)
|
||||
|
||||
prompt_cost, _ = generic_cost_per_token(model=model, usage=usage, custom_llm_provider="openai")
|
||||
breakdown: Final = get_token_type_cost_breakdown(model=model, custom_llm_provider="openai", usage=usage)
|
||||
cached_cost: Final = (
|
||||
100 * rates["cache_read_input_token_cost"]
|
||||
+ 150 * rates["cache_read_input_audio_token_cost"]
|
||||
+ 50 * rates["cache_read_input_image_token_cost"]
|
||||
)
|
||||
uncached_cost: Final = (
|
||||
300 * rates["input_cost_per_token"]
|
||||
+ 250 * rates["input_cost_per_audio_token"]
|
||||
+ 150 * rates["input_cost_per_image_token"]
|
||||
)
|
||||
|
||||
assert breakdown.cache_read_cost == pytest.approx(cached_cost)
|
||||
assert prompt_cost == pytest.approx(uncached_cost + cached_cost)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def _local_model_cost_map(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
|
||||
|
|
@ -299,8 +333,6 @@ def test_reasoning_tokens_gemini(_local_model_cost_map):
|
|||
)
|
||||
|
||||
|
||||
|
||||
|
||||
def test_image_tokens_with_custom_pricing():
|
||||
"""Test that image_tokens in completion are properly costed with output_cost_per_image_token."""
|
||||
from unittest.mock import patch
|
||||
|
|
@ -1950,6 +1982,10 @@ def test_cache_writing_cost_with_zero_creation_tokens_and_ephemeral_details():
|
|||
prompt_tokens_details: PromptTokensDetailsResult = {
|
||||
"cache_hit_tokens": 0,
|
||||
"cache_hit_audio_tokens": 0,
|
||||
"cached_text_tokens": 0,
|
||||
"cached_audio_tokens": 0,
|
||||
"cached_image_tokens": 0,
|
||||
"has_cached_tokens_details": False,
|
||||
"cache_creation_tokens": 0,
|
||||
"cache_creation_token_details": CacheCreationTokenDetails(
|
||||
ephemeral_5m_input_tokens=100,
|
||||
|
|
@ -2185,10 +2221,6 @@ def test_vertex_image_generation_cost_falls_back_to_flat_image_pricing(_local_mo
|
|||
assert round(cost, 10) == round(expected_cost, 10)
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
def test_query_count_is_free_without_a_per_query_price(_local_model_cost_map):
|
||||
usage = Usage(
|
||||
prompt_tokens=0,
|
||||
|
|
@ -2367,8 +2399,6 @@ def test_vertex_global_or_absent_location_no_uplift(vertex_location, _local_mode
|
|||
assert base == located
|
||||
|
||||
|
||||
|
||||
|
||||
def test_vertex_uplift_invalid_multiplier_defaults_to_one():
|
||||
"""A malformed multiplier in the cost map degrades to base pricing, never raises."""
|
||||
from litellm.litellm_core_utils.llm_cost_calc.utils import (
|
||||
|
|
@ -3585,8 +3615,6 @@ def test_route_image_generation_cost_openai_honors_deployment_input_cost_per_ima
|
|||
assert cost == pytest.approx(0.07)
|
||||
|
||||
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("custom_llm_provider", "model"),
|
||||
[
|
||||
|
|
|
|||
|
|
@ -1,3 +1,4 @@
|
|||
import json
|
||||
from unittest.mock import AsyncMock, MagicMock, patch
|
||||
|
||||
import pytest
|
||||
|
|
|
|||
|
|
@ -1121,7 +1121,7 @@ async def test_translation_client_secret_rejects_disallowed_nested_transcription
|
|||
),
|
||||
)
|
||||
|
||||
with pytest.raises(Exception, match="Tried to access gpt-live-transcribe"):
|
||||
with pytest.raises(Exception, match=r"gpt-live-transcribe.*not available for this API key"):
|
||||
await _prepare_client_secret_session(
|
||||
req=req,
|
||||
user_api_key_dict=UserAPIKeyAuth(models=["gpt-realtime-translate"]),
|
||||
|
|
|
|||
|
|
@ -160,10 +160,6 @@ def test_cost_calculator_with_response_cost_in_additional_headers():
|
|||
assert result == 1000
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
def test_realtime_stream_combines_text_and_audio_token_details():
|
||||
"""Realtime response.done usage with input_token_details / output_token_details."""
|
||||
from litellm.cost_calculator import RealtimeAPITokenUsageProcessor
|
||||
|
|
@ -587,7 +583,7 @@ def test_completion_cost_image_generation_reads_deployment_model_info_price_from
|
|||
assert cost == pytest.approx(0.08)
|
||||
|
||||
|
||||
def test_completion_cost_image_generation_registered_deployment_price_keeps_map_token_rates(
|
||||
def test_completion_cost_image_generation_registered_deployment_applies_custom_image_rate(
|
||||
_local_model_cost_map: None, monkeypatch: pytest.MonkeyPatch
|
||||
) -> None:
|
||||
deployment_id: Final = "gemini-image-deployment-priced-per-image"
|
||||
|
|
@ -1041,8 +1037,6 @@ def test_bedrock_cost_calculator_comparison_with_without_cache():
|
|||
print(f"Cost with cache: {cost_with_cache}")
|
||||
|
||||
|
||||
|
||||
|
||||
def test_gemini_25_explicit_caching_cost_direct_usage():
|
||||
"""
|
||||
Test that Gemini 2.5 models correctly calculate costs with explicit caching.
|
||||
|
|
@ -1611,8 +1605,6 @@ def test_cost_margin_with_discount(monkeypatch):
|
|||
print(f" - Expected: ${expected_cost:.6f}")
|
||||
|
||||
|
||||
|
||||
|
||||
def test_completion_cost_extracts_service_tier_from_response(_local_model_cost_map):
|
||||
"""Test that completion_cost extracts service_tier from completion_response object."""
|
||||
from litellm import completion_cost
|
||||
|
|
@ -2363,8 +2355,6 @@ def test_gemini_without_cache_tokens_details():
|
|||
print("✅ Gemini without cacheTokensDetails works correctly")
|
||||
|
||||
|
||||
|
||||
|
||||
def test_additional_costs_only_for_azure_ai(_local_model_cost_map):
|
||||
"""
|
||||
Test that _get_additional_costs is only called for azure_ai provider.
|
||||
|
|
@ -4633,3 +4623,65 @@ def test_gemini_live_native_audio_limits_and_capabilities_match_vendor_model_car
|
|||
assert info["supports_response_schema"] is False
|
||||
assert info["supports_url_context"] is False
|
||||
assert info["supports_pdf_input"] is False
|
||||
|
||||
@pytest.mark.parametrize("provider", ("openai", "azure"))
|
||||
@pytest.mark.parametrize("model", ("gpt-realtime-2.1", "gpt-realtime-2.1-mini"))
|
||||
def test_realtime_cached_multimodal_token_cost(_local_model_cost_map, provider: str, model: str):
|
||||
model_name: Final = f"azure/{model}" if provider == "azure" else model
|
||||
rates: Final = litellm.model_cost[model_name]
|
||||
events: Final[OpenAIRealtimeStreamList] = [
|
||||
{"type": "session.created", "session": {"model": model}},
|
||||
{
|
||||
"type": "response.done",
|
||||
"response": {
|
||||
"usage": {
|
||||
"input_tokens": 1000,
|
||||
"output_tokens": 300,
|
||||
"total_tokens": 1300,
|
||||
"input_token_details": {
|
||||
"text_tokens": 400,
|
||||
"audio_tokens": 400,
|
||||
"image_tokens": 200,
|
||||
"cached_tokens": 300,
|
||||
"cached_tokens_details": {"text_tokens": 100, "audio_tokens": 150, "image_tokens": 50},
|
||||
},
|
||||
"output_token_details": {"text_tokens": 100, "audio_tokens": 100, "reasoning_tokens": 100},
|
||||
}
|
||||
},
|
||||
},
|
||||
]
|
||||
combined: Final = RealtimeAPITokenUsageProcessor.collect_and_combine_usage_from_realtime_stream_results(events)
|
||||
actual: Final = handle_realtime_stream_cost_calculation(
|
||||
results=events,
|
||||
combined_usage_object=combined,
|
||||
custom_llm_provider=provider,
|
||||
litellm_model_name=model_name,
|
||||
)
|
||||
expected: Final = (
|
||||
300 * rates["input_cost_per_token"]
|
||||
+ 250 * rates["input_cost_per_audio_token"]
|
||||
+ 150 * rates["input_cost_per_image_token"]
|
||||
+ 100 * rates["cache_read_input_token_cost"]
|
||||
+ 150 * rates["cache_read_input_audio_token_cost"]
|
||||
+ 50 * rates["cache_read_input_image_token_cost"]
|
||||
+ 200 * rates["output_cost_per_token"]
|
||||
+ 100 * rates["output_cost_per_audio_token"]
|
||||
)
|
||||
|
||||
assert actual == pytest.approx(expected)
|
||||
|
||||
|
||||
def test_realtime_translation_duration_cost(_local_model_cost_map):
|
||||
from litellm.cost_calculator import handle_realtime_translation_cost_calculation
|
||||
|
||||
model: Final = "gpt-realtime-translate"
|
||||
events: Final[OpenAIRealtimeStreamList] = [
|
||||
{"type": "session.closed", "usage": {"type": "duration", "output_seconds": 2.0}}
|
||||
]
|
||||
actual: Final = handle_realtime_translation_cost_calculation(
|
||||
results=events,
|
||||
custom_llm_provider="openai",
|
||||
litellm_model_name=model,
|
||||
)
|
||||
|
||||
assert actual == pytest.approx(2 * litellm.model_cost[model]["output_cost_per_second"])
|
||||
|
|
|
|||
|
|
@ -478,3 +478,36 @@ def test_unregistered_provider_guard_flags_only_labels_nobody_registered():
|
|||
"unknown_root-new_family_models",
|
||||
"vertex_ai-new_family_models",
|
||||
]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", ("gpt-realtime-2.1", "gpt-realtime-2.1-mini"))
|
||||
def test_realtime_family_cache_image_rate_tracks_azure(prices: dict, model: str):
|
||||
openai: Final = prices[model]
|
||||
azure: Final = prices[f"azure/{model}"]
|
||||
|
||||
assert openai["cache_read_input_image_token_cost"] > 0
|
||||
assert azure["cache_read_input_image_token_cost"] == openai["cache_read_input_image_token_cost"]
|
||||
assert azure["input_cost_per_image_token"] >= azure["cache_read_input_image_token_cost"]
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model,mode",
|
||||
(
|
||||
("gpt-realtime-translate", "realtime"),
|
||||
("gpt-live-transcribe", "audio_transcription"),
|
||||
("gpt-transcribe", "audio_transcription"),
|
||||
),
|
||||
)
|
||||
def test_azure_realtime_specialized_models_follow_openai_modes(prices: dict, model: str, mode: str):
|
||||
openai: Final = prices[model]
|
||||
azure: Final = prices[f"azure/{model}"]
|
||||
|
||||
assert openai["mode"] == azure["mode"] == mode
|
||||
assert azure["supports_audio_input"] is True
|
||||
assert azure["supported_endpoints"]
|
||||
|
||||
|
||||
def test_model_prices_backup_is_synchronized(prices: dict):
|
||||
backup: Final = json.loads(BACKUP_PRICES_PATH.read_text())
|
||||
|
||||
assert backup == prices
|
||||
|
|
|
|||
|
|
@ -259,6 +259,21 @@ async def test_construct_url_v1_protocol():
|
|||
assert url.count("/realtime") == 1
|
||||
|
||||
|
||||
def test_construct_url_translation_protocol():
|
||||
from litellm.llms.azure.realtime.handler import AzureOpenAIRealtime
|
||||
|
||||
url = AzureOpenAIRealtime()._construct_url(
|
||||
api_base="https://my-endpoint.openai.azure.com",
|
||||
model="translate-deployment",
|
||||
api_version=None,
|
||||
realtime_protocol="GA",
|
||||
query_params={"model": "translate-deployment"},
|
||||
realtime_mode="translation",
|
||||
)
|
||||
|
||||
assert url == "wss://my-endpoint.openai.azure.com/openai/v1/realtime/translations?model=translate-deployment"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.parametrize("protocol", ["ga", "Ga", "gA", "V1", "v1", "GA"])
|
||||
async def test_construct_url_case_insensitive_protocol(protocol):
|
||||
|
|
|
|||
356
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
356
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
|
|
@ -10227,7 +10227,7 @@ export interface paths {
|
|||
* WebSocket: realtime_websocket_endpoint
|
||||
* @description WebSocket connection endpoint
|
||||
*/
|
||||
get: operations["websocket_realtime_websocket_endpoint_get_3"];
|
||||
get: operations["websocket_realtime_websocket_endpoint_get_6"];
|
||||
put?: never;
|
||||
post?: never;
|
||||
delete?: never;
|
||||
|
|
@ -10294,6 +10294,60 @@ export interface paths {
|
|||
patch?: never;
|
||||
trace?: never;
|
||||
};
|
||||
"/openai/v1/realtime/translations": {
|
||||
parameters: {
|
||||
query?: never;
|
||||
header?: never;
|
||||
path?: never;
|
||||
cookie?: never;
|
||||
};
|
||||
/**
|
||||
* WebSocket: realtime_websocket_endpoint
|
||||
* @description WebSocket connection endpoint
|
||||
*/
|
||||
get: operations["websocket_realtime_websocket_endpoint_get_3"];
|
||||
put?: never;
|
||||
post?: never;
|
||||
delete?: never;
|
||||
options?: never;
|
||||
head?: never;
|
||||
patch?: never;
|
||||
trace?: never;
|
||||
};
|
||||
"/openai/v1/realtime/translations/calls": {
|
||||
parameters: {
|
||||
query?: never;
|
||||
header?: never;
|
||||
path?: never;
|
||||
cookie?: never;
|
||||
};
|
||||
get?: never;
|
||||
put?: never;
|
||||
/** Proxy Realtime Calls */
|
||||
post: operations["proxy_realtime_calls_openai_v1_realtime_translations_calls_post"];
|
||||
delete?: never;
|
||||
options?: never;
|
||||
head?: never;
|
||||
patch?: never;
|
||||
trace?: never;
|
||||
};
|
||||
"/openai/v1/realtime/translations/client_secrets": {
|
||||
parameters: {
|
||||
query?: never;
|
||||
header?: never;
|
||||
path?: never;
|
||||
cookie?: never;
|
||||
};
|
||||
get?: never;
|
||||
put?: never;
|
||||
/** Create Realtime Client Secret */
|
||||
post: operations["create_realtime_client_secret_openai_v1_realtime_translations_client_secrets_post"];
|
||||
delete?: never;
|
||||
options?: never;
|
||||
head?: never;
|
||||
patch?: never;
|
||||
trace?: never;
|
||||
};
|
||||
"/openai/v1/responses": {
|
||||
parameters: {
|
||||
query?: never;
|
||||
|
|
@ -13109,7 +13163,7 @@ export interface paths {
|
|||
* WebSocket: realtime_websocket_endpoint
|
||||
* @description WebSocket connection endpoint
|
||||
*/
|
||||
get: operations["websocket_realtime_websocket_endpoint_get"];
|
||||
get: operations["websocket_realtime_websocket_endpoint_get_4"];
|
||||
put?: never;
|
||||
post?: never;
|
||||
delete?: never;
|
||||
|
|
@ -13176,6 +13230,60 @@ export interface paths {
|
|||
patch?: never;
|
||||
trace?: never;
|
||||
};
|
||||
"/realtime/translations": {
|
||||
parameters: {
|
||||
query?: never;
|
||||
header?: never;
|
||||
path?: never;
|
||||
cookie?: never;
|
||||
};
|
||||
/**
|
||||
* WebSocket: realtime_websocket_endpoint
|
||||
* @description WebSocket connection endpoint
|
||||
*/
|
||||
get: operations["websocket_realtime_websocket_endpoint_get"];
|
||||
put?: never;
|
||||
post?: never;
|
||||
delete?: never;
|
||||
options?: never;
|
||||
head?: never;
|
||||
patch?: never;
|
||||
trace?: never;
|
||||
};
|
||||
"/realtime/translations/calls": {
|
||||
parameters: {
|
||||
query?: never;
|
||||
header?: never;
|
||||
path?: never;
|
||||
cookie?: never;
|
||||
};
|
||||
get?: never;
|
||||
put?: never;
|
||||
/** Proxy Realtime Calls */
|
||||
post: operations["proxy_realtime_calls_realtime_translations_calls_post"];
|
||||
delete?: never;
|
||||
options?: never;
|
||||
head?: never;
|
||||
patch?: never;
|
||||
trace?: never;
|
||||
};
|
||||
"/realtime/translations/client_secrets": {
|
||||
parameters: {
|
||||
query?: never;
|
||||
header?: never;
|
||||
path?: never;
|
||||
cookie?: never;
|
||||
};
|
||||
get?: never;
|
||||
put?: never;
|
||||
/** Create Realtime Client Secret */
|
||||
post: operations["create_realtime_client_secret_realtime_translations_client_secrets_post"];
|
||||
delete?: never;
|
||||
options?: never;
|
||||
head?: never;
|
||||
patch?: never;
|
||||
trace?: never;
|
||||
};
|
||||
"/register": {
|
||||
parameters: {
|
||||
query?: never;
|
||||
|
|
@ -20205,7 +20313,7 @@ export interface paths {
|
|||
* WebSocket: realtime_websocket_endpoint
|
||||
* @description WebSocket connection endpoint
|
||||
*/
|
||||
get: operations["websocket_realtime_websocket_endpoint_get_2"];
|
||||
get: operations["websocket_realtime_websocket_endpoint_get_5"];
|
||||
put?: never;
|
||||
post?: never;
|
||||
delete?: never;
|
||||
|
|
@ -20272,6 +20380,60 @@ export interface paths {
|
|||
patch?: never;
|
||||
trace?: never;
|
||||
};
|
||||
"/v1/realtime/translations": {
|
||||
parameters: {
|
||||
query?: never;
|
||||
header?: never;
|
||||
path?: never;
|
||||
cookie?: never;
|
||||
};
|
||||
/**
|
||||
* WebSocket: realtime_websocket_endpoint
|
||||
* @description WebSocket connection endpoint
|
||||
*/
|
||||
get: operations["websocket_realtime_websocket_endpoint_get_2"];
|
||||
put?: never;
|
||||
post?: never;
|
||||
delete?: never;
|
||||
options?: never;
|
||||
head?: never;
|
||||
patch?: never;
|
||||
trace?: never;
|
||||
};
|
||||
"/v1/realtime/translations/calls": {
|
||||
parameters: {
|
||||
query?: never;
|
||||
header?: never;
|
||||
path?: never;
|
||||
cookie?: never;
|
||||
};
|
||||
get?: never;
|
||||
put?: never;
|
||||
/** Proxy Realtime Calls */
|
||||
post: operations["proxy_realtime_calls_v1_realtime_translations_calls_post"];
|
||||
delete?: never;
|
||||
options?: never;
|
||||
head?: never;
|
||||
patch?: never;
|
||||
trace?: never;
|
||||
};
|
||||
"/v1/realtime/translations/client_secrets": {
|
||||
parameters: {
|
||||
query?: never;
|
||||
header?: never;
|
||||
path?: never;
|
||||
cookie?: never;
|
||||
};
|
||||
get?: never;
|
||||
put?: never;
|
||||
/** Create Realtime Client Secret */
|
||||
post: operations["create_realtime_client_secret_v1_realtime_translations_client_secrets_post"];
|
||||
delete?: never;
|
||||
options?: never;
|
||||
head?: never;
|
||||
patch?: never;
|
||||
trace?: never;
|
||||
};
|
||||
"/v1/rerank": {
|
||||
parameters: {
|
||||
query?: never;
|
||||
|
|
@ -26043,7 +26205,7 @@ export interface components {
|
|||
* CallTypes
|
||||
* @enum {string}
|
||||
*/
|
||||
CallTypes: "embedding" | "aembedding" | "completion" | "acompletion" | "atext_completion" | "text_completion" | "image_generation" | "aimage_generation" | "image_edit" | "aimage_edit" | "moderation" | "amoderation" | "atranscription" | "transcription" | "aspeech" | "speech" | "rerank" | "arerank" | "search" | "asearch" | "_arealtime" | "_aresponses_websocket" | "create_batch" | "acreate_batch" | "aretrieve_batch" | "retrieve_batch" | "acancel_batch" | "cancel_batch" | "pass_through_endpoint" | "anthropic_messages" | "aanthropic_messages" | "get_assistants" | "aget_assistants" | "create_assistants" | "acreate_assistants" | "delete_assistant" | "adelete_assistant" | "acreate_thread" | "create_thread" | "aget_thread" | "get_thread" | "a_add_message" | "add_message" | "aget_messages" | "get_messages" | "arun_thread" | "run_thread" | "arun_thread_stream" | "run_thread_stream" | "afile_retrieve" | "file_retrieve" | "afile_delete" | "file_delete" | "afile_list" | "file_list" | "acreate_file" | "create_file" | "afile_content" | "file_content" | "create_fine_tuning_job" | "acreate_fine_tuning_job" | "create_video" | "acreate_video" | "video_generation" | "avideo_generation" | "avideo_retrieve" | "video_retrieve" | "avideo_content" | "video_content" | "video_remix" | "avideo_remix" | "video_list" | "avideo_list" | "video_retrieve_job" | "avideo_retrieve_job" | "video_delete" | "avideo_delete" | "video_create_character" | "avideo_create_character" | "video_get_character" | "avideo_get_character" | "video_edit" | "avideo_edit" | "video_extension" | "avideo_extension" | "vector_store_file_create" | "avector_store_file_create" | "vector_store_file_list" | "avector_store_file_list" | "vector_store_file_retrieve" | "avector_store_file_retrieve" | "vector_store_file_content" | "avector_store_file_content" | "vector_store_file_update" | "avector_store_file_update" | "vector_store_file_delete" | "avector_store_file_delete" | "vector_store_create" | "avector_store_create" | "vector_store_search" | "avector_store_search" | "ingest" | "aingest" | "query" | "aquery" | "create_interaction" | "acreate_interaction" | "create_container" | "acreate_container" | "list_containers" | "alist_containers" | "retrieve_container" | "aretrieve_container" | "delete_container" | "adelete_container" | "list_container_files" | "alist_container_files" | "upload_container_file" | "aupload_container_file" | "create_sandbox" | "acreate_sandbox" | "delete_sandbox" | "adelete_sandbox" | "run_code" | "arun_code" | "code_interpreter_tool" | "acode_interpreter_tool" | "acancel_fine_tuning_job" | "cancel_fine_tuning_job" | "alist_fine_tuning_jobs" | "list_fine_tuning_jobs" | "aretrieve_fine_tuning_job" | "retrieve_fine_tuning_job" | "responses" | "aresponses" | "alist_input_items" | "llm_passthrough_route" | "allm_passthrough_route" | "generate_content" | "agenerate_content" | "generate_content_stream" | "agenerate_content_stream" | "ocr" | "aocr" | "call_mcp_tool" | "list_mcp_tools" | "asend_message" | "send_message" | "acreate_skill";
|
||||
CallTypes: "embedding" | "aembedding" | "completion" | "acompletion" | "atext_completion" | "text_completion" | "image_generation" | "aimage_generation" | "image_edit" | "aimage_edit" | "moderation" | "amoderation" | "atranscription" | "transcription" | "aspeech" | "speech" | "rerank" | "arerank" | "search" | "asearch" | "_arealtime" | "_aresponses_websocket" | "acreate_realtime_client_secret" | "arealtime_calls" | "acreate_realtime_transcription_session" | "acreate_realtime_translation_client_secret" | "arealtime_translation_calls" | "create_batch" | "acreate_batch" | "aretrieve_batch" | "retrieve_batch" | "acancel_batch" | "cancel_batch" | "pass_through_endpoint" | "anthropic_messages" | "aanthropic_messages" | "get_assistants" | "aget_assistants" | "create_assistants" | "acreate_assistants" | "delete_assistant" | "adelete_assistant" | "acreate_thread" | "create_thread" | "aget_thread" | "get_thread" | "a_add_message" | "add_message" | "aget_messages" | "get_messages" | "arun_thread" | "run_thread" | "arun_thread_stream" | "run_thread_stream" | "afile_retrieve" | "file_retrieve" | "afile_delete" | "file_delete" | "afile_list" | "file_list" | "acreate_file" | "create_file" | "afile_content" | "file_content" | "create_fine_tuning_job" | "acreate_fine_tuning_job" | "create_video" | "acreate_video" | "video_generation" | "avideo_generation" | "avideo_retrieve" | "video_retrieve" | "avideo_content" | "video_content" | "video_remix" | "avideo_remix" | "video_list" | "avideo_list" | "video_retrieve_job" | "avideo_retrieve_job" | "video_delete" | "avideo_delete" | "video_create_character" | "avideo_create_character" | "video_get_character" | "avideo_get_character" | "video_edit" | "avideo_edit" | "video_extension" | "avideo_extension" | "vector_store_file_create" | "avector_store_file_create" | "vector_store_file_list" | "avector_store_file_list" | "vector_store_file_retrieve" | "avector_store_file_retrieve" | "vector_store_file_content" | "avector_store_file_content" | "vector_store_file_update" | "avector_store_file_update" | "vector_store_file_delete" | "avector_store_file_delete" | "vector_store_create" | "avector_store_create" | "vector_store_search" | "avector_store_search" | "ingest" | "aingest" | "query" | "aquery" | "create_interaction" | "acreate_interaction" | "create_container" | "acreate_container" | "list_containers" | "alist_containers" | "retrieve_container" | "aretrieve_container" | "delete_container" | "adelete_container" | "list_container_files" | "alist_container_files" | "upload_container_file" | "aupload_container_file" | "create_sandbox" | "acreate_sandbox" | "delete_sandbox" | "adelete_sandbox" | "run_code" | "arun_code" | "code_interpreter_tool" | "acode_interpreter_tool" | "acancel_fine_tuning_job" | "cancel_fine_tuning_job" | "alist_fine_tuning_jobs" | "list_fine_tuning_jobs" | "aretrieve_fine_tuning_job" | "retrieve_fine_tuning_job" | "responses" | "aresponses" | "alist_input_items" | "llm_passthrough_route" | "allm_passthrough_route" | "generate_content" | "agenerate_content" | "generate_content_stream" | "agenerate_content_stream" | "ocr" | "aocr" | "call_mcp_tool" | "list_mcp_tools" | "asend_message" | "send_message" | "acreate_skill";
|
||||
/** CallbackDelete */
|
||||
CallbackDelete: {
|
||||
/** Callback Name */
|
||||
|
|
@ -27123,6 +27285,12 @@ export interface components {
|
|||
* @description opt-in to RFC 8628 verification_uri_complete for the CLI SSO device flow, pre-filling the user_code in the browser. Off by default; intended for same-host clients where the device that starts the flow and the browser run on the same machine
|
||||
*/
|
||||
allow_cli_sso_verification_uri_complete?: boolean | null;
|
||||
/**
|
||||
* Allow Non Billable Realtime Protocols
|
||||
* @description Allow Realtime WebRTC setup endpoints whose inference usage bypasses LiteLLM spend tracking and budget enforcement
|
||||
* @default false
|
||||
*/
|
||||
allow_non_billable_realtime_protocols: boolean;
|
||||
/**
|
||||
* Allow Unmanaged Response Ids
|
||||
* @description If True, lets keys address Responses API ids that this proxy did not issue (raw provider ids, or ids issued before response-id encryption was configured). Such an id carries no owner, so no ownership check can run on it; ids this proxy did issue keep full ownership enforcement. Off by default, in which case an unrecognized response id is rejected with 403
|
||||
|
|
@ -56324,7 +56492,7 @@ export interface operations {
|
|||
};
|
||||
};
|
||||
};
|
||||
websocket_realtime_websocket_endpoint_get_3: {
|
||||
websocket_realtime_websocket_endpoint_get_6: {
|
||||
parameters: {
|
||||
query?: never;
|
||||
header?: never;
|
||||
|
|
@ -56402,6 +56570,64 @@ export interface operations {
|
|||
};
|
||||
};
|
||||
};
|
||||
websocket_realtime_websocket_endpoint_get_3: {
|
||||
parameters: {
|
||||
query?: never;
|
||||
header?: never;
|
||||
path?: never;
|
||||
cookie?: never;
|
||||
};
|
||||
requestBody?: never;
|
||||
responses: {
|
||||
/** @description WebSocket Protocol Switched */
|
||||
101: {
|
||||
headers: {
|
||||
[name: string]: unknown;
|
||||
};
|
||||
content?: never;
|
||||
};
|
||||
};
|
||||
};
|
||||
proxy_realtime_calls_openai_v1_realtime_translations_calls_post: {
|
||||
parameters: {
|
||||
query?: never;
|
||||
header?: never;
|
||||
path?: never;
|
||||
cookie?: never;
|
||||
};
|
||||
requestBody?: never;
|
||||
responses: {
|
||||
/** @description Successful Response */
|
||||
200: {
|
||||
headers: {
|
||||
[name: string]: unknown;
|
||||
};
|
||||
content: {
|
||||
"application/json": unknown;
|
||||
};
|
||||
};
|
||||
};
|
||||
};
|
||||
create_realtime_client_secret_openai_v1_realtime_translations_client_secrets_post: {
|
||||
parameters: {
|
||||
query?: never;
|
||||
header?: never;
|
||||
path?: never;
|
||||
cookie?: never;
|
||||
};
|
||||
requestBody?: never;
|
||||
responses: {
|
||||
/** @description Successful Response */
|
||||
200: {
|
||||
headers: {
|
||||
[name: string]: unknown;
|
||||
};
|
||||
content: {
|
||||
"application/json": components["schemas"]["RealtimeClientSecretResponse"];
|
||||
};
|
||||
};
|
||||
};
|
||||
};
|
||||
responses_api_openai_v1_responses_post: {
|
||||
parameters: {
|
||||
query?: never;
|
||||
|
|
@ -59409,7 +59635,7 @@ export interface operations {
|
|||
};
|
||||
};
|
||||
};
|
||||
websocket_realtime_websocket_endpoint_get: {
|
||||
websocket_realtime_websocket_endpoint_get_4: {
|
||||
parameters: {
|
||||
query?: never;
|
||||
header?: never;
|
||||
|
|
@ -59487,6 +59713,64 @@ export interface operations {
|
|||
};
|
||||
};
|
||||
};
|
||||
websocket_realtime_websocket_endpoint_get: {
|
||||
parameters: {
|
||||
query?: never;
|
||||
header?: never;
|
||||
path?: never;
|
||||
cookie?: never;
|
||||
};
|
||||
requestBody?: never;
|
||||
responses: {
|
||||
/** @description WebSocket Protocol Switched */
|
||||
101: {
|
||||
headers: {
|
||||
[name: string]: unknown;
|
||||
};
|
||||
content?: never;
|
||||
};
|
||||
};
|
||||
};
|
||||
proxy_realtime_calls_realtime_translations_calls_post: {
|
||||
parameters: {
|
||||
query?: never;
|
||||
header?: never;
|
||||
path?: never;
|
||||
cookie?: never;
|
||||
};
|
||||
requestBody?: never;
|
||||
responses: {
|
||||
/** @description Successful Response */
|
||||
200: {
|
||||
headers: {
|
||||
[name: string]: unknown;
|
||||
};
|
||||
content: {
|
||||
"application/json": unknown;
|
||||
};
|
||||
};
|
||||
};
|
||||
};
|
||||
create_realtime_client_secret_realtime_translations_client_secrets_post: {
|
||||
parameters: {
|
||||
query?: never;
|
||||
header?: never;
|
||||
path?: never;
|
||||
cookie?: never;
|
||||
};
|
||||
requestBody?: never;
|
||||
responses: {
|
||||
/** @description Successful Response */
|
||||
200: {
|
||||
headers: {
|
||||
[name: string]: unknown;
|
||||
};
|
||||
content: {
|
||||
"application/json": components["schemas"]["RealtimeClientSecretResponse"];
|
||||
};
|
||||
};
|
||||
};
|
||||
};
|
||||
register_client_register_post: {
|
||||
parameters: {
|
||||
query?: {
|
||||
|
|
@ -68594,7 +68878,7 @@ export interface operations {
|
|||
};
|
||||
};
|
||||
};
|
||||
websocket_realtime_websocket_endpoint_get_2: {
|
||||
websocket_realtime_websocket_endpoint_get_5: {
|
||||
parameters: {
|
||||
query?: never;
|
||||
header?: never;
|
||||
|
|
@ -68672,6 +68956,64 @@ export interface operations {
|
|||
};
|
||||
};
|
||||
};
|
||||
websocket_realtime_websocket_endpoint_get_2: {
|
||||
parameters: {
|
||||
query?: never;
|
||||
header?: never;
|
||||
path?: never;
|
||||
cookie?: never;
|
||||
};
|
||||
requestBody?: never;
|
||||
responses: {
|
||||
/** @description WebSocket Protocol Switched */
|
||||
101: {
|
||||
headers: {
|
||||
[name: string]: unknown;
|
||||
};
|
||||
content?: never;
|
||||
};
|
||||
};
|
||||
};
|
||||
proxy_realtime_calls_v1_realtime_translations_calls_post: {
|
||||
parameters: {
|
||||
query?: never;
|
||||
header?: never;
|
||||
path?: never;
|
||||
cookie?: never;
|
||||
};
|
||||
requestBody?: never;
|
||||
responses: {
|
||||
/** @description Successful Response */
|
||||
200: {
|
||||
headers: {
|
||||
[name: string]: unknown;
|
||||
};
|
||||
content: {
|
||||
"application/json": unknown;
|
||||
};
|
||||
};
|
||||
};
|
||||
};
|
||||
create_realtime_client_secret_v1_realtime_translations_client_secrets_post: {
|
||||
parameters: {
|
||||
query?: never;
|
||||
header?: never;
|
||||
path?: never;
|
||||
cookie?: never;
|
||||
};
|
||||
requestBody?: never;
|
||||
responses: {
|
||||
/** @description Successful Response */
|
||||
200: {
|
||||
headers: {
|
||||
[name: string]: unknown;
|
||||
};
|
||||
content: {
|
||||
"application/json": components["schemas"]["RealtimeClientSecretResponse"];
|
||||
};
|
||||
};
|
||||
};
|
||||
};
|
||||
rerank_v1_rerank_post: {
|
||||
parameters: {
|
||||
query?: never;
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue