From 1d4a72a38563c284595a775908aaa60ced94c37c Mon Sep 17 00:00:00 2001 From: Emerson Gomes Date: Fri, 5 Jun 2026 08:29:55 -0500 Subject: [PATCH] Address PR review: session_type validation, model auth fix, cost perf, billing fallback, detail/docs cleanup --- litellm/cost_calculator.py | 13 ++-- litellm/proxy/realtime_endpoints/endpoints.py | 4 +- litellm/realtime_api/README.md | 62 ++----------------- litellm/realtime_api/main.py | 6 ++ 4 files changed, 21 insertions(+), 64 deletions(-) diff --git a/litellm/cost_calculator.py b/litellm/cost_calculator.py index aea1ba3f220..69309c6b4c3 100644 --- a/litellm/cost_calculator.py +++ b/litellm/cost_calculator.py @@ -2534,11 +2534,12 @@ def handle_realtime_stream_cost_calculation( break # exit if we find a valid model total_cost = input_cost_per_token + output_cost_per_token - total_cost += handle_realtime_transcription_cost_calculation( - results=results, - custom_llm_provider=custom_llm_provider, - litellm_model_name=litellm_model_name, - ) + if any(r.get("type") == _TRANSCRIPTION_COMPLETED_EVENT_TYPE for r in results): + total_cost += handle_realtime_transcription_cost_calculation( + results=results, + custom_llm_provider=custom_llm_provider, + litellm_model_name=litellm_model_name, + ) return total_cost @@ -2602,7 +2603,7 @@ def _get_transcription_model_name_from_results( transcription = ( (session.get("audio", {}) or {}).get("input", {}) or {} ).get("transcription", {}) or session.get("input_audio_transcription", {}) - model = (transcription or {}).get("model") + model = (transcription or {}).get("model") or session.get("model") if model: return model return None diff --git a/litellm/proxy/realtime_endpoints/endpoints.py b/litellm/proxy/realtime_endpoints/endpoints.py index a2afd868ec4..94b1384b712 100644 --- a/litellm/proxy/realtime_endpoints/endpoints.py +++ b/litellm/proxy/realtime_endpoints/endpoints.py @@ -291,6 +291,8 @@ async def proxy_realtime_calls( user_id = decoded_payload.get("user_id") or None team_id = decoded_payload.get("team_id") or None session_type = decoded_payload.get("session_type") or "realtime" + if session_type not in ("realtime", "transcription"): + session_type = "realtime" else: # Backward compatibility: older tokens contained only encrypted upstream key. openai_ephemeral_key = decrypted_token_value @@ -464,7 +466,7 @@ async def create_realtime_transcription_session( ) if isinstance(e, HTTPException): raise ProxyException( - message=getattr(e, "message", str(e)), + message=getattr(e, "detail", getattr(e, "message", str(e))), type=getattr(e, "type", "None"), param=getattr(e, "param", "None"), code=getattr(e, "status_code", http_status.HTTP_400_BAD_REQUEST), diff --git a/litellm/realtime_api/README.md b/litellm/realtime_api/README.md index cf3d2973cfc..d810de2f24f 100644 --- a/litellm/realtime_api/README.md +++ b/litellm/realtime_api/README.md @@ -1,61 +1,9 @@ Abstraction / Routing logic for OpenAI's `/v1/realtime` endpoints. -## Realtime transcription (`gpt-realtime-whisper`) +Supported endpoints: +- WebSocket: `/v1/realtime` (with `intent=transcription` for transcription-only sessions) +- HTTP: `/v1/realtime/client_secrets`, `/v1/realtime/transcription_sessions` -`gpt-realtime-whisper` is the low-latency streaming speech-to-text model. It is a -Realtime transcription session, not the file-based `/audio/transcriptions` path. Use -the standard `gpt-4o-transcribe` / `whisper-1` models for request/response or file -transcription; use `gpt-realtime-whisper` for live streaming transcript deltas. +Supported providers: OpenAI, Azure OpenAI, Bedrock, Vertex AI, xAI. -Both OpenAI and Azure OpenAI (Microsoft Foundry) are supported. Cost is tracked by input -audio duration (OpenAI: $0.017/minute), derived from the -`conversation.item.input_audio_transcription.completed` usage events. - -### WebSocket - -Connect to the proxy realtime WebSocket with `intent=transcription`, then send a -`session.update` configuring a transcription session: - -``` -wss:///v1/realtime?model=gpt-realtime-whisper&intent=transcription -``` - -```json -{ - "type": "session.update", - "session": { - "type": "transcription", - "audio": { - "input": { - "format": { "type": "audio/pcm", "rate": 24000 }, - "transcription": { "model": "gpt-realtime-whisper", "language": "en" } - } - } - } -} -``` - -Append audio with `input_audio_buffer.append`, then `input_audio_buffer.commit` (when not -using server VAD). Listen for `conversation.item.input_audio_transcription.delta` and -`.completed` events. The proxy does not auto-trigger `response.create` for transcription -sessions. - -### Ephemeral transcription session (WebRTC) - -`POST /v1/realtime/transcription_sessions` mints an ephemeral session for browser/WebRTC -clients. The returned `client_secret.value` is encrypted by the proxy and exchanged via -`POST /v1/realtime/calls`. - -```bash -curl https:///v1/realtime/transcription_sessions \ - -H "Authorization: Bearer $LITELLM_KEY" \ - -H "Content-Type: application/json" \ - -d '{ - "input_audio_format": "pcm16", - "input_audio_transcription": { "model": "gpt-realtime-whisper", "language": "en" } - }' -``` - -For Azure, route to an `azure/gpt-realtime-whisper` deployment; the proxy targets -`/openai/realtime/transcription_sessions?api-version=...` and forwards -`intent=transcription` on the WebSocket. \ No newline at end of file +For user-facing documentation and usage examples, see the litellm-docs repo. \ No newline at end of file diff --git a/litellm/realtime_api/main.py b/litellm/realtime_api/main.py index 60f6fabd561..1883d1ea19f 100644 --- a/litellm/realtime_api/main.py +++ b/litellm/realtime_api/main.py @@ -212,6 +212,12 @@ async def acreate_realtime_transcription_session( custom_llm_provider=custom_llm_provider, ) request_data = req.model_dump(exclude_none=True, exclude={"model"}) + # Ensure the upstream body's input_audio_transcription.model matches the + # authorized routing model. This prevents a caller from supplying an allowed + # top-level model for auth while sneaking a different model into the nested + # transcription config that gets forwarded to the provider. + if isinstance(request_data.get("input_audio_transcription"), dict): + request_data["input_audio_transcription"]["model"] = model_name return await base_llm_http_handler.async_realtime_transcription_session_handler( api_base=resolved_api_base, api_key=resolved_api_key,