From 3686f6a005cd1ca1915aa2542256ad8470eb35bf Mon Sep 17 00:00:00 2001 From: jibanez-staticduo Date: Wed, 9 Sep 2026 11:17:09 +0200 Subject: [PATCH] fix(chatgpt): select live transport from model metadata --- litellm/llms/chatgpt/realtime.py | 11 +++++++- ...odel_prices_and_context_window_backup.json | 10 ++++++++ model_prices_and_context_window.json | 10 ++++++++ .../llms/chatgpt/test_realtime.py | 25 ++++++++++++++++++- 4 files changed, 54 insertions(+), 2 deletions(-) diff --git a/litellm/llms/chatgpt/realtime.py b/litellm/llms/chatgpt/realtime.py index 66a703eb616..93a79b7d159 100644 --- a/litellm/llms/chatgpt/realtime.py +++ b/litellm/llms/chatgpt/realtime.py @@ -9,6 +9,7 @@ from litellm.llms.openai.realtime.handler import OpenAIRealtime from litellm.llms.openai.realtime.http_transformation import OpenAIRealtimeHTTPConfig from litellm.types.realtime import RealtimeQueryParams from litellm.types.router import GenericLiteLLMParams +from litellm.utils import get_model_info from .common_utils import CHATGPT_API_BASE from .responses.transformation import ChatGPTResponsesAPIConfig @@ -34,6 +35,14 @@ def realtime_headers( } +def realtime_endpoint(model: str) -> str: + try: + model_info: Final = get_model_info(model, custom_llm_provider="chatgpt") + except Exception: + return "realtime" + return "live" if "/v1/live" in (model_info.get("supported_endpoints") or ()) else "realtime" + + class ChatGPTRealtime(OpenAIRealtime): def __init__(self, params: GenericLiteLLMParams, headers: Mapping[str, str]) -> None: super().__init__() @@ -50,7 +59,7 @@ class ChatGPTRealtime(OpenAIRealtime): def _construct_url(self, api_base: str, query_params: RealtimeQueryParams) -> str: base: Final = URL(api_base) - endpoint: Final = "live" if query_params.get("model") == "gpt-live-1-codex" else "realtime" + endpoint: Final = realtime_endpoint(query_params.get("model", "")) if self._call_id: return str( base.copy_with( diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 54ebdc85be9..ae4432e167d 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -27832,6 +27832,16 @@ "max_tokens": 8191, "mode": "embedding" }, + "chatgpt/gpt-live-1-codex": { + "litellm_provider": "chatgpt", + "mode": "realtime", + "supported_endpoints": [ + "/v1/realtime/calls", + "/v1/live" + ], + "supports_audio_input": true, + "supports_audio_output": true + }, "chatgpt/gpt-5.4": { "litellm_provider": "chatgpt", "max_input_tokens": 1050000, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 54ebdc85be9..ae4432e167d 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -27832,6 +27832,16 @@ "max_tokens": 8191, "mode": "embedding" }, + "chatgpt/gpt-live-1-codex": { + "litellm_provider": "chatgpt", + "mode": "realtime", + "supported_endpoints": [ + "/v1/realtime/calls", + "/v1/live" + ], + "supports_audio_input": true, + "supports_audio_output": true + }, "chatgpt/gpt-5.4": { "litellm_provider": "chatgpt", "max_input_tokens": 1050000, diff --git a/tests/test_litellm/llms/chatgpt/test_realtime.py b/tests/test_litellm/llms/chatgpt/test_realtime.py index 7ce8d5f298a..239a5a12280 100644 --- a/tests/test_litellm/llms/chatgpt/test_realtime.py +++ b/tests/test_litellm/llms/chatgpt/test_realtime.py @@ -40,7 +40,7 @@ async def test_chatgpt_call_keeps_oauth_and_frameless_session(chatgpt_tokens): @pytest.mark.parametrize("model,endpoint", [("gpt-realtime-1.5", "realtime"), ("gpt-live-1-codex", "live")]) -def test_realtime_uses_platform_endpoint_with_oauth_headers(model, endpoint, chatgpt_tokens): +def test_realtime_uses_platform_endpoint_with_oauth_headers(model, endpoint, chatgpt_tokens, local_model_cost_map): handler = ChatGPTRealtime( GenericLiteLLMParams(), { @@ -55,3 +55,26 @@ def test_realtime_uses_platform_endpoint_with_oauth_headers(model, endpoint, cha assert headers["Authorization"] == "Bearer test-token-default" assert "authorization" not in headers assert headers["openai-alpha"] == "quicksilver=v2" + + +@pytest.mark.parametrize("endpoint", ["live", "realtime"]) +@pytest.mark.parametrize("call_id", [None, "rtc_metadata"]) +def test_realtime_routes_new_models_using_registered_metadata(endpoint, call_id, chatgpt_tokens, local_model_cost_map): + model = "metadata-voice-model" + litellm.register_model({f"chatgpt/{model}": { + "litellm_provider": "chatgpt", "mode": "realtime", "supported_endpoints": [f"/v1/{endpoint}"] + }}) + handler = ChatGPTRealtime(GenericLiteLLMParams(chatgpt_realtime_call_id=call_id), {}) + expected = ( + f"wss://api.openai.com/v1/{endpoint}?model={model}" if call_id is None + else f"wss://api.openai.com/v1/live/{call_id}" if endpoint == "live" + else f"wss://api.openai.com/v1/realtime?call_id={call_id}" + ) + assert handler._construct_url("https://api.openai.com/v1", {"model": model}) == expected + + +def test_realtime_unknown_model_keeps_standard_endpoint(chatgpt_tokens, local_model_cost_map): + handler = ChatGPTRealtime(GenericLiteLLMParams(), {}) + assert handler._construct_url("https://api.openai.com/v1", {"model": "unknown-voice-model"}) == ( + "wss://api.openai.com/v1/realtime?model=unknown-voice-model" + )