feat(chatgpt): support Codex image and realtime routes

This commit is contained in:
jibanez-staticduo 2026-09-09 07:20:37 +02:00
parent ee7c7e14f3
commit bce8aa1dcc
No known key found for this signature in database
21 changed files with 1010 additions and 14 deletions

View file

@ -0,0 +1,52 @@
# ChatGPT OAuth routes used by Codex
The ChatGPT provider supports image generation and editing, structured Responses output, Realtime WebSockets, and direct WebRTC offer exchange using the existing ChatGPT OAuth credentials configured with `CHATGPT_TOKEN_DIR` and `CHATGPT_AUTH_FILE`
## Models and transports
| Model | Proxy route | Upstream route |
| --- | --- | --- |
| `gpt-image-2` | `/v1/images/generations` | ChatGPT `/backend-api/codex/images/generations` |
| `gpt-image-2` | `/v1/images/edits` | ChatGPT `/backend-api/codex/images/edits` |
| `codex-auto-review` | `/v1/responses` | ChatGPT `/backend-api/codex/responses` |
| `gpt-realtime-1.5` | `/v1/realtime` WebSocket | OpenAI `/v1/realtime` with ChatGPT OAuth |
| `gpt-live-1-codex` | `/v1/realtime/calls`, then `/v1/live/{call_id}` | ChatGPT call signaling, OpenAI live sideband |
| `gpt-4o-mini-transcribe` | Nested in Realtime `session.audio.input.transcription.model` | Same Realtime session |
Codex uses `gpt-image-2` for both image routes. There is no separate `gpt-image-2-edit` model in that contract. Memory extraction and consolidation use ordinary Responses models and need no additional transport
Register deployments with `litellm_params.model: chatgpt/<model>`. These auxiliary models do not need to appear in the conversation model selector. This contribution uses the existing single-account ChatGPT authentication contract
## Images
Both synchronous and asynchronous SDK calls support image generation and editing. Edits accept the usual `image` file input or Codex's JSON `images` array, containing one to five `{ "image_url": "data:image/png;base64,..." }` references. PNG, JPEG, WEBP, and HTTPS references are accepted. Masks and mixing `image` with `images` are rejected
```python
await litellm.aimage_edit(
model="chatgpt/gpt-image-2",
prompt="Change the blue circle to red",
images=[{"image_url": image_data_url}],
)
```
## Guardian
The Responses transformation preserves `text.format`, including strict JSON schemas. Custom Codex providers use `/responses` for `codex-auto-review`. The separate native `/guardian` route can be disabled by the upstream service; registering an alias does not enable that route
## Realtime and GPT-Live
Standard Realtime WebSockets use the OpenAI host with the selected ChatGPT access token and account header. Client protocol headers are forwarded without forwarding the client's proxy authorization header
Codex WebRTC offers can use JSON or multipart `sdp` and `session` fields with an ordinary LiteLLM key. Signaling preserves the session shape and the `intent` and `architecture` query parameters. Existing raw SDP requests authenticated with an encrypted ephemeral key retain their existing route
For GPT-Live, send `OpenAI-Alpha: quicksilver=v2`, `intent=quicksilver&architecture=avas`, and a Codex voice such as `sol`. Do not add the GA `session.type: realtime` field to a Frameless Bidi session. A generic `Voice session access denied` error can mean an unsupported voice; it is not sufficient evidence of missing account entitlement
The returned `Location` contains an encrypted call identifier. The sideband connection must use the same LiteLLM bearer key and retain access to the requested alias. The identifier expires after one hour and remains valid across proxy workers sharing the same salt key. OAuth tokens are never returned to clients
The direct `/v1/live?model=gpt-live-1-codex` WebSocket is forwarded, but upstream access can differ from WebRTC. A successful WebRTC call does not establish permission for direct live sessions. Upstream errors remain visible; the proxy does not replace the model, voice, or transport silently
## Verification
The integration was exercised with real image generation and JSON edits, strict Guardian JSON output, Realtime text and audio output, audio transcription, and a GPT-Live WebRTC session with a successful sideband context acknowledgment. Expired, malformed, tampered, and wrong-owner call identifiers have regression coverage
References: [Codex source](https://github.com/openai/codex), [OpenClaw voice authentication](https://docs.openclaw.ai/providers/openai/voice-and-speech), and [Pi Codex signaling implementation](https://github.com/monotykamary/pi-better-openai/blob/main/src/live/transport.ts). These upstream capabilities may change independently of LiteLLM

View file

@ -102,7 +102,9 @@ async def aimage_generation(*args, **kwargs) -> ImageResponse:
ctx: Final = contextvars.copy_context()
func_with_context: Final = partial(ctx.run, func)
_, custom_llm_provider, _, _ = get_llm_provider(model=model, api_base=kwargs.get("api_base", None))
_, custom_llm_provider, _, _ = get_llm_provider(
model=model, api_base=kwargs.get("api_base", None), litellm_params=GenericLiteLLMParams(**kwargs)
)
# Await normally
init_response: Final = await loop.run_in_executor(None, func_with_context)
@ -227,6 +229,7 @@ def image_generation(
model=model,
custom_llm_provider=custom_llm_provider,
api_base=api_base,
litellm_params=GenericLiteLLMParams(**kwargs),
)
else:
model = "dall-e-2"
@ -377,6 +380,7 @@ def image_generation(
# Providers using llm_http_handler
#########################################################
elif custom_llm_provider in (
litellm.LlmProviders.CHATGPT,
litellm.LlmProviders.RECRAFT,
litellm.LlmProviders.AIML,
litellm.LlmProviders.GEMINI,
@ -785,6 +789,7 @@ def image_edit(
model, custom_llm_provider, _, _ = get_llm_provider(
model=model or DEFAULT_IMAGE_ENDPOINT_MODEL,
custom_llm_provider=custom_llm_provider,
litellm_params=litellm_params,
)
# Check for custom provider
@ -965,9 +970,9 @@ def image_edit(
@client
async def aimage_edit(
image: FileTypes | list[FileTypes],
model: str,
prompt: str,
image: FileTypes | list[FileTypes] | None = None,
model: str = "",
prompt: str = "",
mask: str | None = None,
n: int | None = None,
quality: str | ImageGenerationRequestQuality | None = None,
@ -1002,14 +1007,12 @@ async def aimage_edit(
# get custom llm provider so we can use this for mapping exceptions
if custom_llm_provider is None:
_, custom_llm_provider, _, _ = litellm.get_llm_provider(
model=model, api_base=local_vars.get("base_url", None)
model=model, api_base=local_vars.get("base_url", None), litellm_params=GenericLiteLLMParams(**kwargs)
)
images: Final = image if isinstance(image, list) else [image]
func: Final = partial(
image_edit,
image=images,
image=image,
prompt=prompt,
mask=mask,
model=model,

View file

@ -0,0 +1,137 @@
import base64
from collections.abc import Mapping, Sequence
from types import MappingProxyType
from typing import Final
from httpx._types import FileTypes as HTTPFileTypes
from httpx._types import RequestFiles
from pydantic import BaseModel, ConfigDict, Field, TypeAdapter
from litellm.llms.openai.image_edit.transformation import OpenAIImageEditConfig
from litellm.llms.openai.image_generation.gpt_transformation import GPTImageGenerationConfig
from litellm.types.llms.openai import AllMessageValues, FileTypes
from litellm.types.router import GenericLiteLLMParams
from .common_utils import CHATGPT_API_BASE
from .responses.transformation import ChatGPTResponsesAPIConfig
class ReferenceImage(BaseModel):
model_config = ConfigDict(extra="forbid")
image_url: str = Field(pattern=r"^(data:image/(png|jpeg|webp);base64,|https://)")
def encode_reference(file: HTTPFileTypes) -> dict[str, str]: # mutable-ok: image handler requires dictionaries
content: Final = file[1] if isinstance(file, tuple) else file
content_type: Final = file[2] if isinstance(file, tuple) and len(file) >= 3 else "image/png"
if content_type not in ("image/png", "image/jpeg", "image/webp"):
raise ValueError("Reference images must be PNG, JPEG, or WEBP")
raw: Final = (
content.encode() if isinstance(content, str) else content if isinstance(content, bytes) else content.read()
)
return { # mutable-ok: JSON request serialization
"image_url": f"data:{content_type};base64," + base64.b64encode(raw).decode("ascii")
}
def image_headers(
headers: Mapping[str, object], model: str, params: Mapping[str, object]
) -> dict[str, object]: # mutable-ok: image handler requires dictionaries
auth_headers: Final = ChatGPTResponsesAPIConfig().validate_environment(
headers={}, # mutable-ok: Responses adapter header contract
model=model,
litellm_params=GenericLiteLLMParams.model_validate(params),
)
return {**headers, **auth_headers, "accept": "application/json"} # mutable-ok: JSON request serialization
class ChatGPTImageGenerationConfig(GPTImageGenerationConfig):
def validate_environment(
self,
headers: Mapping[str, object],
model: str,
messages: Sequence[AllMessageValues],
optional_params: Mapping[str, object],
litellm_params: Mapping[str, object],
api_key: str | None = None,
api_base: str | None = None,
) -> dict[str, object]: # mutable-ok: image handler requires dictionaries
return image_headers(headers, model, litellm_params)
def get_complete_url(
self,
api_base: str | None,
api_key: str | None,
model: str,
optional_params: Mapping[str, object],
litellm_params: Mapping[str, object],
stream: bool | None = None,
) -> str:
return f"{(api_base or CHATGPT_API_BASE).rstrip('/')}/images/generations"
def transform_image_generation_request(
self,
model: str,
prompt: str,
optional_params: Mapping[str, object],
litellm_params: Mapping[str, object],
headers: Mapping[str, object],
) -> dict[str, object]: # mutable-ok: image handler requires dictionaries
return {"model": model, "prompt": prompt, **optional_params} # mutable-ok: JSON request serialization
class ChatGPTImageEditConfig(OpenAIImageEditConfig):
def validate_environment(
self,
headers: Mapping[str, object],
model: str,
api_key: str | None = None,
litellm_params: Mapping[str, object] | None = None,
api_base: str | None = None,
) -> dict[str, object]: # mutable-ok: image handler requires dictionaries
return image_headers(headers, model, litellm_params or MappingProxyType({}))
def get_complete_url(self, model: str, api_base: str | None, litellm_params: Mapping[str, object]) -> str:
return f"{(api_base or CHATGPT_API_BASE).rstrip('/')}/images/edits"
def use_multipart_form_data(self) -> bool:
return False
def transform_image_edit_request(
self,
model: str,
prompt: str | None,
image: FileTypes | None,
image_edit_optional_request_params: Mapping[str, object],
litellm_params: GenericLiteLLMParams,
headers: Mapping[str, object],
) -> tuple[dict[str, object], RequestFiles]: # mutable-ok: image handler requires dictionaries
if image_edit_optional_request_params.get("mask") is not None:
raise ValueError("ChatGPT image editing does not support masks")
references: Final = getattr(litellm_params, "images", None)
if references is not None:
if image:
raise ValueError("Specify only one of image or images")
validated: Final = TypeAdapter(tuple[ReferenceImage, ...]).validate_python(references)
if not 1 <= len(validated) <= 5:
raise ValueError("images must contain between 1 and 5 reference images")
return { # mutable-ok: JSON request serialization
"model": model,
"prompt": prompt,
**image_edit_optional_request_params,
"images": tuple(item.model_dump() for item in validated),
}, ()
data, files = super().transform_image_edit_request(
model,
prompt,
image,
dict(image_edit_optional_request_params), # mutable-ok: parent edit adapter requires dictionaries
litellm_params,
dict(headers), # mutable-ok: parent edit adapter requires dictionaries
)
parts: Final = files.items() if isinstance(files, Mapping) else files
encoded: Final = tuple(encode_reference(file) for field, file in parts if field == "image[]")
if not 1 <= len(encoded) <= 5:
raise ValueError("images must contain between 1 and 5 reference images")
return {**data, "images": encoded}, () # mutable-ok: JSON request serialization

View file

@ -0,0 +1,124 @@
from collections.abc import Mapping
from types import MappingProxyType
from typing import Final
from httpx import URL
from pydantic import TypeAdapter
from litellm.llms.openai.realtime.handler import OpenAIRealtime
from litellm.llms.openai.realtime.http_transformation import OpenAIRealtimeHTTPConfig
from litellm.types.realtime import RealtimeQueryParams
from litellm.types.router import GenericLiteLLMParams
from .common_utils import CHATGPT_API_BASE
from .responses.transformation import ChatGPTResponsesAPIConfig
def realtime_headers(
params: GenericLiteLLMParams, headers: Mapping[str, str]
) -> dict[str, str]: # mutable-ok: HTTP handler header contract
forwarded: Final = MappingProxyType(
{
key.lower(): value
for key, value in headers.items()
if key.lower() in ("openai-alpha", "openai-beta", "x-session-id", "x-oai-attestation")
}
)
return { # mutable-ok: HTTP handler updates headers
**ChatGPTResponsesAPIConfig().validate_environment(
headers={}, # mutable-ok: Responses adapter header contract
model="",
litellm_params=params,
),
**forwarded,
}
class ChatGPTRealtime(OpenAIRealtime):
def __init__(self, params: GenericLiteLLMParams, headers: Mapping[str, str]) -> None:
super().__init__()
self._profile_headers = realtime_headers(params, headers)
self._call_id = TypeAdapter(str | None).validate_python(getattr(params, "chatgpt_realtime_call_id", None))
def _get_additional_headers(
self, api_key: str, *, openai_beta_realtime: bool = False
) -> dict[str, str]: # mutable-ok: HTTP handler header contract
return { # mutable-ok: HTTP handler updates headers
**(MappingProxyType({"OpenAI-Beta": "realtime=v1"}) if openai_beta_realtime else MappingProxyType({})),
**self._profile_headers,
}
def _construct_url(self, api_base: str, query_params: RealtimeQueryParams) -> str:
base: Final = URL(api_base)
endpoint: Final = "live" if query_params.get("model") == "gpt-live-1-codex" else "realtime"
if self._call_id:
return str(
base.copy_with(
scheme="wss" if base.scheme in ("https", "wss") else "ws",
path=f"{base.path.rstrip('/')}/{endpoint}/{self._call_id}"
if endpoint == "live"
else f"{base.path.rstrip('/')}/realtime",
params=() if endpoint == "live" else (("call_id", self._call_id),),
)
)
return str(
base.copy_with(
scheme="wss" if base.scheme in ("https", "wss") else "ws",
path=f"{base.path.rstrip('/')}/{endpoint}",
params=query_params,
)
)
class ChatGPTRealtimeHTTPConfig(OpenAIRealtimeHTTPConfig):
realtime_calls_json: Final = True
def __init__(self, params: GenericLiteLLMParams) -> None:
self._params = params
def get_api_base(
self,
api_base: str | None,
**kwargs: object, # kwargs-ok: provider interface accepts optional credentials
) -> str:
return api_base or CHATGPT_API_BASE
def get_api_key(
self,
api_key: str | None,
**kwargs: object, # kwargs-ok: provider interface accepts optional credentials
) -> str:
return "chatgpt-oauth"
def get_realtime_calls_url(self, api_base: str | None, model: str, api_version: str | None = None) -> str:
query: Final = TypeAdapter(Mapping[str, str]).validate_python(
getattr(self._params, "extra_query", None) or MappingProxyType({})
)
return str(URL(f"{self.get_api_base(api_base).rstrip('/')}/realtime/calls", params=query))
def get_realtime_calls_headers(
self, ephemeral_key: str
) -> dict[str, str]: # mutable-ok: HTTP handler header contract
return realtime_headers(self._params, MappingProxyType({}))
def validate_environment(
self,
headers: Mapping[str, str],
model: str,
api_key: str | None = None,
) -> dict[str, str]: # mutable-ok: HTTP handler header contract
return { # mutable-ok: HTTP handler updates headers
**realtime_headers(self._params, headers),
"Content-Type": "application/json",
}
def get_complete_url(self, api_base: str | None, model: str, api_version: str | None = None) -> str:
return "https://api.openai.com/v1/realtime/client_secrets"
def get_transcription_session_url(
self,
api_base: str | None,
model: str,
api_version: str | None = None,
) -> str:
return "https://api.openai.com/v1/realtime/transcription_sessions"

View file

@ -102,6 +102,7 @@ class ChatGPTResponsesAPIConfig(OpenAIResponsesAPIConfig):
"reasoning",
"previous_response_id",
"truncation",
"text",
}
return {k: v for k, v in request.items() if k in allowed_keys}

View file

@ -6411,7 +6411,7 @@ class BaseLLMHTTPHandler:
# Build multipart form data: sdp + session JSON
session_data: Final = session_config or {}
if "type" not in session_data:
if "type" not in session_data and not getattr(provider_config, "realtime_calls_json", False):
session_data["type"] = "realtime"
if "model" not in session_data and model:
session_data["model"] = model
@ -6434,6 +6434,13 @@ class BaseLLMHTTPHandler:
)
try:
if getattr(provider_config, "realtime_calls_json", False):
return await async_httpx_client.post(
url=url,
headers=headers,
json={"sdp": sdp_text, "session": session_data}, # mutable-ok: JSON signaling payload
timeout=timeout,
)
return await async_httpx_client.post(
url=url,
headers=headers,

View file

@ -399,6 +399,9 @@ class LiteLLMRoutes(enum.Enum):
"/realtime?{model}",
"/v1/realtime?{model}",
"/openai/v1/realtime?{model}",
"/live",
"/v1/live",
"/v1/live/{call_id}",
# realtime (GA WebRTC HTTP routes)
"/realtime/client_secrets",
"/v1/realtime/client_secrets",

View file

@ -11592,6 +11592,19 @@ async def _reject_realtime_session(
await _release_realtime_budget_reservation(user_api_key_dict)
@app.websocket("/v1/live/{call_id}")
async def codex_live_sideband_endpoint(
websocket: WebSocket,
call_id: str,
user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth_websocket),
) -> None:
from litellm.proxy.realtime_endpoints.codex import codex_realtime_sideband
await codex_realtime_sideband(websocket, call_id, user_api_key_dict)
@app.websocket("/v1/live")
@app.websocket("/live")
@app.websocket("/openai/v1/realtime")
@app.websocket("/v1/realtime")
@app.websocket("/realtime")
@ -11599,12 +11612,18 @@ async def realtime_websocket_endpoint(
websocket: WebSocket,
model: str | None = fastapi.Query(None, description="The model to use for the websocket connection."),
intent: str | None = fastapi.Query(None, description="The intent of the websocket connection."),
call_id: str | None = None,
guardrails: str | None = fastapi.Query(
None,
description="Comma-separated list of guardrail names to apply to this request.",
),
user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth_websocket),
):
if call_id is not None and call_id.startswith("rtc_litellm_"):
from litellm.proxy.realtime_endpoints.codex import codex_realtime_sideband
await codex_realtime_sideband(websocket, call_id, user_api_key_dict)
return
requested_protocols: Final = [
p.strip() for p in (websocket.headers.get("sec-websocket-protocol") or "").split(",") if p.strip()
]
@ -11635,7 +11654,10 @@ async def realtime_websocket_endpoint(
await websocket.accept(**accept_kwargs)
# Only use explicit parameters, not all query params
query_params: Final = cast(RealtimeQueryParams, dict(_realtime_query_params_template(model, intent)))
query_params: Final = cast(
RealtimeQueryParams,
dict(_realtime_query_params_template(model, intent) + ((("call_id", call_id),) if call_id is not None else ())),
)
data: dict[str, object] = {
"model": route_model,

View file

@ -0,0 +1,188 @@
import base64
import hashlib
import json
import time
from types import MappingProxyType
from typing import Final
from urllib.parse import urlsplit
import httpx
from fastapi import HTTPException, Request, Response, WebSocket
from pydantic import BaseModel, Field
from litellm.llms.base_llm.chat.transformation import BaseLLMException
from litellm.proxy._types import ProxyException, UserAPIKeyAuth
from litellm.proxy.auth.auth_checks import can_key_call_resolved_model
from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
from litellm.proxy.common_utils.encrypt_decrypt_utils import decrypt_value_helper, encrypt_value_helper
from litellm.proxy.spend_tracking.budget_reservation import release_or_invalidate_budget_reservation
from litellm.types.realtime import RealtimeQueryParams, RealtimeSessionConfig
class CodexRealtimeOffer(BaseModel):
sdp: str = Field(min_length=1)
session: RealtimeSessionConfig
class CodexRealtimeCall(BaseModel):
call_id: str = Field(pattern=r"^rtc_[A-Za-z0-9_-]+$")
model: str
alias: str
owner: str
expires_at: float
class ChatGPTCallRouting(BaseModel):
model: str
def encode_call(call: CodexRealtimeCall) -> str:
encrypted: Final = encrypt_value_helper(call.model_dump_json())
return "rtc_litellm_" + base64.urlsafe_b64encode(encrypted.encode()).decode().rstrip("=")
def decode_call(token: str, authorization: str) -> CodexRealtimeCall:
try:
if not token.startswith("rtc_litellm_"):
raise ValueError("Invalid call prefix")
encoded: Final = token.removeprefix("rtc_litellm_")
encrypted: Final = base64.b64decode(encoded + "=" * (-len(encoded) % 4), altchars=b"-_", validate=True)
plaintext: Final = decrypt_value_helper(encrypted.decode(), key="codex_realtime_call")
call: Final = CodexRealtimeCall.model_validate_json(plaintext or "")
except (ValueError, TypeError, UnicodeError) as exc:
raise HTTPException(403, "Invalid realtime call") from exc
if call.expires_at < time.time() or call.owner != hashlib.sha256(authorization.encode()).hexdigest():
raise HTTPException(403, "Invalid or expired realtime call")
return call
async def read_codex_offer(request: Request) -> CodexRealtimeOffer:
if request.headers.get("content-type", "").startswith("multipart/form-data"):
form: Final = await request.form()
return CodexRealtimeOffer.model_validate(
MappingProxyType({"sdp": form.get("sdp"), "session": json.loads(str(form.get("session", "{}")))})
)
return CodexRealtimeOffer.model_validate(await request.json())
async def create_codex_realtime_call(request: Request) -> Response:
from litellm.proxy import proxy_server as server
from litellm.proxy.common_request_processing import ProxyBaseLLMRequestProcessing
try:
offer: Final = await read_codex_offer(request)
except ValueError as exc:
raise HTTPException(400, "Invalid realtime offer: expected sdp and session") from exc
model: Final = offer.session.model
if not model:
raise HTTPException(400, "session.model is required")
auth: Final = await user_api_key_auth(
request=request,
api_key=request.headers.get("authorization", ""),
azure_api_key_header="",
anthropic_api_key_header=None,
google_ai_studio_api_key_header=None,
azure_apim_header=None,
custom_litellm_key_header=None,
)
try:
await can_key_call_resolved_model(
model=model,
llm_model_list=server.llm_model_list,
valid_token=auth,
llm_router=server.llm_router,
)
data: Final = { # mutable-ok: proxy processor enriches request data
"model": model,
"sdp_body": offer.sdp.encode(),
"session": offer.session.model_dump(exclude_none=True),
"openai_ephemeral_key": "",
"extra_query": { # mutable-ok: router request parameters
key: value for key, value in request.query_params.items() if key in ("intent", "architecture")
},
"extra_headers": { # mutable-ok: router request headers
key: value
for key, value in request.headers.items()
if key in ("openai-alpha", "openai-beta", "x-session-id", "x-oai-attestation")
},
}
processor: Final = ProxyBaseLLMRequestProcessing(data=data)
processed, _ = await processor.common_processing_pre_call_logic(
request=request,
general_settings=server.general_settings,
user_api_key_dict=auth,
version=server.version,
proxy_logging_obj=server.proxy_logging_obj,
proxy_config=server.proxy_config,
user_model=server.user_model,
user_temperature=server.user_temperature,
user_request_timeout=server.user_request_timeout,
user_max_tokens=server.user_max_tokens,
user_api_base=server.user_api_base,
model=model,
route_type="arealtime_calls",
)
result: Final = await server.route_request(
data=processed,
route_type="arealtime_calls",
llm_router=server.llm_router,
user_model=server.user_model,
)
try:
response: Final = await result
except BaseLLMException as exc:
raise HTTPException(exc.status_code, str(exc)) from exc
if not isinstance(response, httpx.Response):
raise HTTPException(502, "Invalid realtime signaling response")
routing_data: Final = response.extensions.get("chatgpt_realtime")
if response.is_error:
return Response(response.content, status_code=response.status_code, media_type="application/json")
if not routing_data:
raise HTTPException(400, "Direct call signaling requires a ChatGPT deployment")
routing: Final = ChatGPTCallRouting.model_validate(routing_data)
call_id: Final = urlsplit(response.headers.get("location", "")).path.rstrip("/").rsplit("/", 1)[-1]
call: Final = CodexRealtimeCall(
call_id=call_id,
model=routing.model,
alias=model,
owner=hashlib.sha256(request.headers.get("authorization", "").encode()).hexdigest(),
expires_at=time.time() + 3600,
)
token: Final = encode_call(call)
return Response(
response.content,
status_code=response.status_code,
media_type="application/sdp",
headers=MappingProxyType({"Location": f"/v1/realtime/calls/{token}"}),
)
finally:
await release_or_invalidate_budget_reservation(budget_reservation=auth.budget_reservation)
async def codex_realtime_sideband(websocket: WebSocket, token: str, auth: UserAPIKeyAuth) -> None:
import litellm
from litellm.proxy import proxy_server as server
try:
try:
call: Final = decode_call(token, websocket.headers.get("authorization", ""))
await can_key_call_resolved_model(
model=call.alias,
llm_model_list=server.llm_model_list,
valid_token=auth,
llm_router=server.llm_router,
)
except (HTTPException, ProxyException):
await websocket.close(code=1008, reason="Invalid realtime call")
return
await websocket.accept()
query: Final[RealtimeQueryParams] = {"model": call.model}
await litellm._arealtime( # pyright: ignore[reportPrivateUsage] # internal proxy entrypoint for an already authorized call
model=f"chatgpt/{call.model}",
websocket=websocket,
chatgpt_realtime_call_id=call.call_id,
query_params=query,
user_api_key_dict=auth,
)
finally:
await release_or_invalidate_budget_reservation(budget_reservation=auth.budget_reservation)

View file

@ -375,6 +375,11 @@ async def proxy_realtime_calls(
request: Request,
fastapi_response: Response,
) -> Response:
if request.headers.get("content-type", "").split(";", 1)[0] in ("application/json", "multipart/form-data"):
from litellm.proxy.realtime_endpoints.codex import create_codex_realtime_call
return await create_codex_realtime_call(request)
from litellm.proxy.proxy_server import (
add_litellm_data_to_request,
general_settings,

View file

@ -89,7 +89,11 @@ def _get_realtime_http_provider_config(
)
provider_config: BaseRealtimeHTTPConfig | None = None
if custom_llm_provider in LlmProviders._member_map_.values():
if custom_llm_provider == "chatgpt":
from litellm.llms.chatgpt.realtime import ChatGPTRealtimeHTTPConfig
provider_config = ChatGPTRealtimeHTTPConfig(litellm_params)
elif custom_llm_provider in LlmProviders._member_map_.values():
provider_config = ProviderConfigManager.get_provider_realtime_http_config(
model="",
provider=LlmProviders(custom_llm_provider),
@ -137,6 +141,7 @@ async def acreate_realtime_client_secret(
model=model_name,
api_base=litellm_params.api_base,
api_key=litellm_params.api_key,
litellm_params=litellm_params,
)
(
provider_config,
@ -205,6 +210,7 @@ async def acreate_realtime_transcription_session(
model=model_name,
api_base=litellm_params.api_base,
api_key=litellm_params.api_key,
litellm_params=litellm_params,
)
(
provider_config,
@ -266,6 +272,7 @@ async def arealtime_calls(
model=model_name,
api_base=litellm_params.api_base,
api_key=litellm_params.api_key,
litellm_params=litellm_params,
)
provider_config, resolved_api_base, _ = _get_realtime_http_provider_config(
custom_llm_provider=custom_llm_provider,
@ -282,7 +289,7 @@ async def arealtime_calls(
litellm_params={"api_base": resolved_api_base},
custom_llm_provider=custom_llm_provider,
)
return await base_llm_http_handler.async_realtime_calls_handler(
response: Final = await base_llm_http_handler.async_realtime_calls_handler(
api_base=resolved_api_base,
openai_ephemeral_key=openai_ephemeral_key,
sdp_body=sdp_body,
@ -295,6 +302,13 @@ async def arealtime_calls(
client=kwargs.get("client"),
api_version=litellm_params.api_version,
)
if custom_llm_provider == "chatgpt":
response.extensions["chatgpt_realtime"] = MappingProxyType(
{
"model": model_name,
}
)
return response
async def vertex_access_token_resolver(
@ -366,6 +380,7 @@ async def _arealtime(
model=model,
api_base=api_base,
api_key=api_key,
litellm_params=litellm_params,
)
# If the client supplied `model` in the URL, ensure it uses the normalized
@ -439,6 +454,20 @@ async def _arealtime(
user_api_key_dict=kwargs.get("user_api_key_dict"),
litellm_metadata=_build_litellm_metadata(kwargs),
)
elif _custom_llm_provider == "chatgpt":
from litellm.llms.chatgpt.realtime import ChatGPTRealtime
await ChatGPTRealtime(litellm_params, websocket.headers).async_realtime(
model=model,
websocket=websocket,
logging_obj=litellm_logging_obj,
api_base=api_base or "https://api.openai.com/v1",
api_key="chatgpt-oauth",
timeout=timeout,
query_params=query_params,
user_api_key_dict=kwargs.get("user_api_key_dict"),
litellm_metadata=_build_litellm_metadata(kwargs),
)
elif _custom_llm_provider == "openai":
api_base = dynamic_api_base or litellm_params.api_base or litellm.api_base or "https://api.openai.com/"
# set API KEY

View file

@ -49,6 +49,7 @@ class RealtimeModalityResponseTransformOutput(TypedDict):
class RealtimeQueryParams(TypedDict, total=False):
model: str
intent: str | None
call_id: ReadOnly[str]
# Add more fields as needed

View file

@ -9040,6 +9040,10 @@ class ProviderConfigManager:
model: str,
provider: LlmProviders,
) -> BaseImageGenerationConfig | None:
if LlmProviders.CHATGPT == provider:
from litellm.llms.chatgpt.images import ChatGPTImageGenerationConfig
return ChatGPTImageGenerationConfig()
if LlmProviders.OPENAI == provider:
from litellm.llms.openai.image_generation import (
get_openai_image_generation_config,
@ -9237,6 +9241,10 @@ class ProviderConfigManager:
model: str,
provider: LlmProviders,
) -> BaseImageEditConfig | None:
if LlmProviders.CHATGPT == provider:
from litellm.llms.chatgpt.images import ChatGPTImageEditConfig
return ChatGPTImageEditConfig()
if LlmProviders.OPENAI == provider:
from litellm.llms.openai.image_edit import get_openai_image_edit_config

View file

@ -0,0 +1,22 @@
import json
import time
import pytest
@pytest.fixture
def chatgpt_tokens(tmp_path, monkeypatch):
monkeypatch.setenv("CHATGPT_TOKEN_DIR", str(tmp_path))
monkeypatch.setenv("CHATGPT_AUTH_FILE", "auth.json")
for profile in ("default", "account2", "account3"):
name = "auth.json" if profile == "default" else profile + ".json"
(tmp_path / name).write_text(
json.dumps(
{
"access_token": "test-token-" + profile,
"account_id": "test-account-" + profile,
"expires_at": time.time() + 3600,
}
)
)
return str(tmp_path)

View file

@ -19,6 +19,29 @@ from litellm.llms.chatgpt.responses.transformation import ChatGPTResponsesAPICon
class TestChatGPTResponsesAPITransformation:
def test_guardian_preserves_strict_output_schema(self):
text = {
"format": {
"type": "json_schema",
"name": "review",
"strict": True,
"schema": {
"type": "object",
"properties": {"allowed": {"type": "boolean"}},
"required": ["allowed"],
"additionalProperties": False,
},
}
}
request = ChatGPTResponsesAPIConfig().transform_responses_api_request(
model="codex-auto-review",
input="Review the command pwd",
response_api_optional_request_params={"text": text},
litellm_params=GenericLiteLLMParams(),
headers={},
)
assert request["text"] == text
@pytest.mark.parametrize(
"model_name",
[

View file

@ -0,0 +1,106 @@
import base64
import httpx
import pytest
import litellm
from litellm.llms.chatgpt.images import ChatGPTImageEditConfig, ChatGPTImageGenerationConfig
from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler
from litellm.types.router import GenericLiteLLMParams
def test_generation_routes_with_chatgpt_oauth(chatgpt_tokens):
requests = []
def respond(request):
requests.append(request)
return httpx.Response(200, json={"created": 1, "data": [{"b64_json": "aGVsbG8="}]})
client = HTTPHandler()
client.client = httpx.Client(transport=httpx.MockTransport(respond))
result = litellm.image_generation(
model="chatgpt/gpt-image-2",
prompt="blue circle",
client=client,
quality="auto",
size="auto",
background="auto",
)
assert result.data[0].b64_json == "aGVsbG8="
assert str(requests[0].url) == "https://chatgpt.com/backend-api/codex/images/generations"
assert requests[0].headers["authorization"] == "Bearer test-token-" + "default"
assert requests[0].headers["chatgpt-account-id"] == "test-account-" + "default"
assert b'"model":"gpt-image-2"' in requests[0].content
def test_codex_json_edit_survives_sdk_dispatch(chatgpt_tokens):
requests = []
def respond(request):
requests.append(request)
return httpx.Response(200, json={"created": 1, "data": [{"b64_json": "aGVsbG8="}]})
client = HTTPHandler()
client.client = httpx.Client(transport=httpx.MockTransport(respond))
references = [{"image_url": "data:image/png;base64,aGVsbG8="}]
result = litellm.image_edit(
model="chatgpt/gpt-image-2",
prompt="red circle",
images=references,
client=client,
quality="auto",
size="auto",
)
assert result.data[0].b64_json == "aGVsbG8="
assert str(requests[0].url) == "https://chatgpt.com/backend-api/codex/images/edits"
import json
assert json.loads(requests[0].content)["images"] == references
@pytest.mark.parametrize(
"references", [[], [{"image_url": "file:///etc/passwd"}], [{}], [{"image_url": "https://example.com/a.png"}] * 6]
)
def test_edit_rejects_invalid_references(references):
with pytest.raises(ValueError, match=r"images must contain|validation error"):
ChatGPTImageEditConfig().transform_image_edit_request(
"gpt-image-2", "edit", None, {}, GenericLiteLLMParams(images=references), {}
)
def test_edit_converts_multipart_image_bytes():
data, files = ChatGPTImageEditConfig().transform_image_edit_request(
"gpt-image-2", "edit", b"example", {}, GenericLiteLLMParams(), {}
)
assert not files
assert base64.b64decode(data["images"][0]["image_url"].split(",", 1)[1]) == b"example"
def test_image_auth_does_not_accept_inbound_override(chatgpt_tokens):
headers = ChatGPTImageGenerationConfig().validate_environment(
{"Authorization": "Bearer wrong"}, "gpt-image-2", [], {}, {"chatgpt_token_dir": chatgpt_tokens}
)
assert headers["Authorization"] == "Bearer test-token-default"
@pytest.mark.asyncio
async def test_async_codex_edit_without_multipart_image(chatgpt_tokens):
requests = []
def respond(request):
requests.append(request)
return httpx.Response(200, json={"created": 1, "data": [{"b64_json": "aGVsbG8="}]})
client = AsyncHTTPHandler()
client.client = httpx.AsyncClient(transport=httpx.MockTransport(respond))
response = await litellm.aimage_edit(
model="chatgpt/gpt-image-2",
prompt="red circle",
client=client,
images=[{"image_url": "data:image/png;base64,aGVsbG8="}],
chatgpt_auth_profile="account3",
)
assert response.data[0].b64_json == "aGVsbG8="
assert str(requests[0].url).endswith("/codex/images/edits")
assert requests[0].headers["content-type"] == "application/json"
await client.client.aclose()

View file

@ -0,0 +1,57 @@
import json
import httpx
import pytest
import litellm
from litellm.llms.chatgpt.realtime import ChatGPTRealtime
from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler
from litellm.types.router import GenericLiteLLMParams
@pytest.mark.asyncio
async def test_chatgpt_call_keeps_oauth_and_frameless_session(chatgpt_tokens):
requests = []
def respond(request):
requests.append(request)
return httpx.Response(201, text="v=0\r\n", headers={"location": "/v1/realtime/calls/rtc_test"})
client = AsyncHTTPHandler()
client.client = httpx.AsyncClient(transport=httpx.MockTransport(respond))
response = await litellm.arealtime_calls(
model="chatgpt/gpt-live-1-codex",
openai_ephemeral_key="",
sdp_body=b"v=0\r\n",
session={"model": "chatgpt/gpt-live-1-codex", "audio": {"output": {"voice": "sol"}}},
extra_query={"intent": "quicksilver", "architecture": "avas"},
extra_headers={"openai-alpha": "quicksilver=v2"},
client=client,
)
assert response.status_code == 201
assert requests[0].url.path == "/backend-api/codex/realtime/calls"
assert requests[0].url.params["architecture"] == "avas"
assert requests[0].headers["authorization"] == "Bearer test-token-" + "default"
assert json.loads(requests[0].content) == {
"sdp": "v=0\r\n",
"session": {"model": "gpt-live-1-codex", "audio": {"output": {"voice": "sol"}}},
}
await client.client.aclose()
@pytest.mark.parametrize("model,endpoint", [("gpt-realtime-1.5", "realtime"), ("gpt-live-1-codex", "live")])
def test_realtime_uses_platform_endpoint_with_oauth_headers(model, endpoint, chatgpt_tokens):
handler = ChatGPTRealtime(
GenericLiteLLMParams(),
{
"authorization": "Bearer proxy-key",
"openai-alpha": "quicksilver=v2",
},
)
assert handler._construct_url("https://api.openai.com/v1", {"model": model}) == (
f"wss://api.openai.com/v1/{endpoint}?model={model}"
)
headers = handler._get_additional_headers("unused")
assert headers["Authorization"] == "Bearer test-token-default"
assert "authorization" not in headers
assert headers["openai-alpha"] == "quicksilver=v2"

View file

@ -17,6 +17,24 @@ from litellm.proxy.auth.auth_checks_organization import _user_is_org_admin
from litellm.proxy.auth.route_checks import RouteChecks
@pytest.mark.parametrize("route", ["/live", "/v1/live", "/v1/live/rtc_litellm_test"])
def test_codex_live_routes_allow_inference_keys(route: str):
from litellm.proxy.auth.auth_checks import _allowed_routes_check
assert RouteChecks.is_llm_api_route(route)
assert _allowed_routes_check(user_route=route, allowed_routes=["openai_routes"])
token = UserAPIKeyAuth(allowed_routes=["llm_api_routes"])
RouteChecks.is_virtual_key_allowed_to_call_route(route=route, valid_token=token)
RouteChecks.non_proxy_admin_allowed_routes_check(
user_obj=None,
_user_role=LitellmUserRoles.INTERNAL_USER.value,
route=route,
request=Request({"type": "http", "path": route, "query_string": b"", "headers": []}),
valid_token=token,
request_data={},
)
def test_non_admin_config_update_route_rejected():
"""Test that non-admin users are rejected when trying to call /config/update"""

View file

@ -0,0 +1,76 @@
import hashlib
import time
import pytest
from fastapi import HTTPException, WebSocket
from litellm.proxy._types import ProxyException, UserAPIKeyAuth
from litellm.proxy.realtime_endpoints import codex
from litellm.proxy.realtime_endpoints.codex import CodexRealtimeCall, decode_call, encode_call
def test_sideband_token_binds_owner_and_model(monkeypatch):
monkeypatch.setenv("LITELLM_SALT_KEY", "test-only-salt-for-codex-realtime")
call = CodexRealtimeCall(
call_id="rtc_test",
model="gpt-live-1-codex",
alias="gpt-live-1-codex",
owner=hashlib.sha256(b"Bearer test-owner").hexdigest(),
expires_at=time.time() + 300,
)
token = encode_call(call)
assert "/" not in token
assert decode_call(token, "Bearer test-owner") == call
with pytest.raises(HTTPException) as error:
decode_call(token, "Bearer different-owner")
assert error.value.status_code == 403
with pytest.raises(HTTPException):
decode_call(token[:30] + "tampered" + token[30:], "Bearer test-owner")
def test_sideband_rejects_expired_token(monkeypatch):
monkeypatch.setenv("LITELLM_SALT_KEY", "test-only-salt-for-codex-realtime")
call = CodexRealtimeCall(
call_id="rtc_test",
model="gpt-realtime-1.5",
alias="gpt-realtime-1.5",
owner=hashlib.sha256(b"Bearer test-owner").hexdigest(),
expires_at=time.time() - 1,
)
with pytest.raises(HTTPException):
decode_call(encode_call(call), "Bearer test-owner")
@pytest.mark.parametrize("token", ["", "rtc_other", "rtc_litellm_%%%%", "rtc_litellm_a"])
def test_sideband_rejects_malformed_tokens(token):
with pytest.raises(HTTPException):
decode_call(token, "Bearer test-owner")
@pytest.mark.asyncio
async def test_sideband_rejects_revoked_model_access(monkeypatch):
monkeypatch.setenv("LITELLM_SALT_KEY", "test-only-salt-for-codex-realtime")
call = CodexRealtimeCall(
call_id="rtc_test",
model="gpt-live-1-codex",
alias="voice",
owner=hashlib.sha256(b"Bearer test-owner").hexdigest(),
expires_at=time.time() + 300,
)
sent = []
async def receive():
return {"type": "websocket.connect"}
async def send(message):
sent.append(message)
async def deny_model(**kwargs):
raise ProxyException("Model access revoked", "auth_error", "model", 403)
monkeypatch.setattr(codex, "can_key_call_resolved_model", deny_model)
websocket = WebSocket(
{"type": "websocket", "headers": [(b"authorization", b"Bearer test-owner")]}, receive, send
)
await codex.codex_realtime_sideband(websocket, encode_call(call), UserAPIKeyAuth())
assert sent == [{"type": "websocket.close", "code": 1008, "reason": "Invalid realtime call"}]

View file

@ -61,7 +61,7 @@ def _run_client_secret(session, model, monkeypatch):
captured.update(kwargs)
return object()
def mock_get_llm_provider(model, api_base, api_key):
def mock_get_llm_provider(model, api_base, api_key, litellm_params=None):
return model, "openai", None, api_base
monkeypatch.setattr(realtime_main, "get_llm_provider", mock_get_llm_provider)
@ -161,7 +161,7 @@ async def test_arealtime_vertex_branch_resolves_credentials_under_a_bound(monkey
async def hanging_token_refresh(**kwargs):
await asyncio.sleep(30)
def mock_get_llm_provider(model, api_base, api_key):
def mock_get_llm_provider(model, api_base, api_key, litellm_params=None):
return model, "vertex_ai", None, api_base
monkeypatch.setattr(realtime_main, "get_llm_provider", mock_get_llm_provider)

View file

@ -8276,6 +8276,26 @@ export interface paths {
patch?: never;
trace?: never;
};
"/live": {
parameters: {
query?: never;
header?: never;
path?: never;
cookie?: never;
};
/**
* WebSocket: realtime_websocket_endpoint
* @description WebSocket connection endpoint
*/
get: operations["websocket_realtime_websocket_endpoint_get_4"];
put?: never;
post?: never;
delete?: never;
options?: never;
head?: never;
patch?: never;
trace?: never;
};
"/login": {
parameters: {
query?: never;
@ -18396,6 +18416,46 @@ export interface paths {
patch?: never;
trace?: never;
};
"/v1/live": {
parameters: {
query?: never;
header?: never;
path?: never;
cookie?: never;
};
/**
* WebSocket: realtime_websocket_endpoint
* @description WebSocket connection endpoint
*/
get: operations["websocket_realtime_websocket_endpoint_get_5"];
put?: never;
post?: never;
delete?: never;
options?: never;
head?: never;
patch?: never;
trace?: never;
};
"/v1/live/": {
parameters: {
query?: never;
header?: never;
path?: never;
cookie?: never;
};
/**
* WebSocket: codex_live_sideband_endpoint
* @description WebSocket connection endpoint
*/
get: operations["websocket_codex_live_sideband_endpoint"];
put?: never;
post?: never;
delete?: never;
options?: never;
head?: never;
patch?: never;
trace?: never;
};
"/v1/mcp/access_groups": {
parameters: {
query?: never;
@ -50644,6 +50704,24 @@ export interface operations {
};
};
};
websocket_realtime_websocket_endpoint_get_4: {
parameters: {
query?: never;
header?: never;
path?: never;
cookie?: never;
};
requestBody?: never;
responses: {
/** @description WebSocket Protocol Switched */
101: {
headers: {
[name: string]: unknown;
};
content?: never;
};
};
};
login_login_post: {
parameters: {
query?: never;
@ -63000,6 +63078,42 @@ export interface operations {
};
};
};
websocket_realtime_websocket_endpoint_get_5: {
parameters: {
query?: never;
header?: never;
path?: never;
cookie?: never;
};
requestBody?: never;
responses: {
/** @description WebSocket Protocol Switched */
101: {
headers: {
[name: string]: unknown;
};
content?: never;
};
};
};
websocket_codex_live_sideband_endpoint: {
parameters: {
query?: never;
header?: never;
path?: never;
cookie?: never;
};
requestBody?: never;
responses: {
/** @description WebSocket Protocol Switched */
101: {
headers: {
[name: string]: unknown;
};
content?: never;
};
};
};
get_mcp_access_groups_v1_mcp_access_groups_get: {
parameters: {
query?: never;