Address PR review: session_type validation, model auth fix, cost perf, billing fallback, detail/docs cleanup

This commit is contained in:
Emerson Gomes 2026-06-05 08:29:55 -05:00
parent 312f51a10b
commit 1d4a72a385
No known key found for this signature in database
GPG key ID: D3DF28AB5D1B5E17
4 changed files with 21 additions and 64 deletions

View file

@ -2534,11 +2534,12 @@ def handle_realtime_stream_cost_calculation(
break # exit if we find a valid model
total_cost = input_cost_per_token + output_cost_per_token
total_cost += handle_realtime_transcription_cost_calculation(
results=results,
custom_llm_provider=custom_llm_provider,
litellm_model_name=litellm_model_name,
)
if any(r.get("type") == _TRANSCRIPTION_COMPLETED_EVENT_TYPE for r in results):
total_cost += handle_realtime_transcription_cost_calculation(
results=results,
custom_llm_provider=custom_llm_provider,
litellm_model_name=litellm_model_name,
)
return total_cost
@ -2602,7 +2603,7 @@ def _get_transcription_model_name_from_results(
transcription = (
(session.get("audio", {}) or {}).get("input", {}) or {}
).get("transcription", {}) or session.get("input_audio_transcription", {})
model = (transcription or {}).get("model")
model = (transcription or {}).get("model") or session.get("model")
if model:
return model
return None

View file

@ -291,6 +291,8 @@ async def proxy_realtime_calls(
user_id = decoded_payload.get("user_id") or None
team_id = decoded_payload.get("team_id") or None
session_type = decoded_payload.get("session_type") or "realtime"
if session_type not in ("realtime", "transcription"):
session_type = "realtime"
else:
# Backward compatibility: older tokens contained only encrypted upstream key.
openai_ephemeral_key = decrypted_token_value
@ -464,7 +466,7 @@ async def create_realtime_transcription_session(
)
if isinstance(e, HTTPException):
raise ProxyException(
message=getattr(e, "message", str(e)),
message=getattr(e, "detail", getattr(e, "message", str(e))),
type=getattr(e, "type", "None"),
param=getattr(e, "param", "None"),
code=getattr(e, "status_code", http_status.HTTP_400_BAD_REQUEST),

View file

@ -1,61 +1,9 @@
Abstraction / Routing logic for OpenAI's `/v1/realtime` endpoints.
## Realtime transcription (`gpt-realtime-whisper`)
Supported endpoints:
- WebSocket: `/v1/realtime` (with `intent=transcription` for transcription-only sessions)
- HTTP: `/v1/realtime/client_secrets`, `/v1/realtime/transcription_sessions`
`gpt-realtime-whisper` is the low-latency streaming speech-to-text model. It is a
Realtime transcription session, not the file-based `/audio/transcriptions` path. Use
the standard `gpt-4o-transcribe` / `whisper-1` models for request/response or file
transcription; use `gpt-realtime-whisper` for live streaming transcript deltas.
Supported providers: OpenAI, Azure OpenAI, Bedrock, Vertex AI, xAI.
Both OpenAI and Azure OpenAI (Microsoft Foundry) are supported. Cost is tracked by input
audio duration (OpenAI: $0.017/minute), derived from the
`conversation.item.input_audio_transcription.completed` usage events.
### WebSocket
Connect to the proxy realtime WebSocket with `intent=transcription`, then send a
`session.update` configuring a transcription session:
```
wss://<proxy>/v1/realtime?model=gpt-realtime-whisper&intent=transcription
```
```json
{
"type": "session.update",
"session": {
"type": "transcription",
"audio": {
"input": {
"format": { "type": "audio/pcm", "rate": 24000 },
"transcription": { "model": "gpt-realtime-whisper", "language": "en" }
}
}
}
}
```
Append audio with `input_audio_buffer.append`, then `input_audio_buffer.commit` (when not
using server VAD). Listen for `conversation.item.input_audio_transcription.delta` and
`.completed` events. The proxy does not auto-trigger `response.create` for transcription
sessions.
### Ephemeral transcription session (WebRTC)
`POST /v1/realtime/transcription_sessions` mints an ephemeral session for browser/WebRTC
clients. The returned `client_secret.value` is encrypted by the proxy and exchanged via
`POST /v1/realtime/calls`.
```bash
curl https://<proxy>/v1/realtime/transcription_sessions \
-H "Authorization: Bearer $LITELLM_KEY" \
-H "Content-Type: application/json" \
-d '{
"input_audio_format": "pcm16",
"input_audio_transcription": { "model": "gpt-realtime-whisper", "language": "en" }
}'
```
For Azure, route to an `azure/gpt-realtime-whisper` deployment; the proxy targets
`/openai/realtime/transcription_sessions?api-version=...` and forwards
`intent=transcription` on the WebSocket.
For user-facing documentation and usage examples, see the litellm-docs repo.

View file

@ -212,6 +212,12 @@ async def acreate_realtime_transcription_session(
custom_llm_provider=custom_llm_provider,
)
request_data = req.model_dump(exclude_none=True, exclude={"model"})
# Ensure the upstream body's input_audio_transcription.model matches the
# authorized routing model. This prevents a caller from supplying an allowed
# top-level model for auth while sneaking a different model into the nested
# transcription config that gets forwarded to the provider.
if isinstance(request_data.get("input_audio_transcription"), dict):
request_data["input_audio_transcription"]["model"] = model_name
return await base_llm_http_handler.async_realtime_transcription_session_handler(
api_base=resolved_api_base,
api_key=resolved_api_key,