fix(proxy): allow non-admin virtual keys to call GA Realtime WebRTC HTTP routes (#30089)

* fix(proxy): allow non-admin virtual keys to call GA Realtime WebRTC HTTP routes

Add the realtime WebRTC HTTP sub-routes (/realtime/client_secrets,
/realtime/calls and their /v1 + /openai/v1 variants) to
LiteLLMRoutes.openai_routes so is_llm_api_route() classifies them as
LLM API routes. Without this, non-admin virtual keys received
401 'Only proxy admin can be used to generate, delete, update info
for new keys/users/teams' when calling these endpoints.

Fixes #29923

* fix(proxy): validate session.model for realtime routes in model-access check

The GA Realtime WebRTC HTTP routes resolve the effective model from the
nested session.model (falling back to the top-level model), but the auth
layer's get_model_from_request() only extracted the top-level model. A
model-restricted virtual key could therefore place a disallowed model in
session.model, leave the top-level model unset, and skip can_key_call_model()
entirely - obtaining an ephemeral token for a model it is not allowed to use.

Extract session.model for the realtime client_secrets/calls routes so the
model-access check runs against the model the request will actually use.
Legitimate callers are unaffected; their permitted model still validates.

Relates to https://github.com/BerriAI/litellm/issues/29923

* fix(proxy): classify realtime transcription_sessions routes as LLM API routes

Add the GA Realtime WebRTC transcription_sessions HTTP routes to
openai_routes so is_llm_api_route() returns True for them, matching the
client_secrets and calls routes already fixed. These endpoints are
registered with user_api_key_auth in realtime_endpoints/endpoints.py, so
without this a non-admin virtual key calling
POST /v1/realtime/transcription_sessions would hit the admin-only 401
branch. Extends the regression test parametrization accordingly.

---------

Co-authored-by: habonlaci <4699494+habonlaci@users.noreply.github.com>
This commit is contained in:
Habon Laszlo 2026-06-17 12:56:24 +02:00 • committed by GitHub
parent cf2db415b8
commit 682bb6caad
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
4 changed files with 103 additions and 0 deletions

View file

@ -361,6 +361,16 @@ class LiteLLMRoutes(enum.Enum):
"/realtime?{model}",
"/v1/realtime?{model}",
"/openai/v1/realtime?{model}",
# realtime (GA WebRTC HTTP routes)
"/realtime/client_secrets",
"/v1/realtime/client_secrets",
"/openai/v1/realtime/client_secrets",
"/realtime/calls",
"/v1/realtime/calls",
"/openai/v1/realtime/calls",
"/realtime/transcription_sessions",
"/v1/realtime/transcription_sessions",
"/openai/v1/realtime/transcription_sessions",
# responses API
"/responses",
"/v1/responses",

View file

@ -1227,6 +1227,14 @@ _MODEL_ROUTING_BODY_TARGET_MODEL_ROUTE_MARKERS = (
"/vector_stores",
)
_MODEL_ROUTING_COMPLETION_MODEL_ROUTE_MARKERS = ("/evals",)
# Realtime WebRTC routes carry the effective model inside the nested
# ``session.model`` field (see realtime_endpoints.endpoints), so the model the
# request will actually use is not present at the top level. Extract it here so
# can_key_call_model() validates the real target model.
_MODEL_ROUTING_SESSION_MODEL_ROUTE_MARKERS = (
"/realtime/client_secrets",
"/realtime/calls",
)
_MODEL_ROUTING_ID_FIELDS = (
"file_id",
"input_file_id",
@ -1409,6 +1417,12 @@ def _extract_model_candidates_from_request(
_append_model_candidates(candidates, body_model)
if uses_body_target_model_sources or not body_model:
_append_model_candidates(candidates, request_data.get("target_model_names"))
if _route_matches_any_marker(
route=route, markers=_MODEL_ROUTING_SESSION_MODEL_ROUTE_MARKERS
):
session = request_data.get("session")
if isinstance(session, dict):
_append_model_candidates(candidates, session.get("model"))
if uses_completion_model_sources and isinstance(
request_data.get("completion"), dict
):

View file

@ -530,6 +530,59 @@ def test_get_model_from_request_handles_managed_id_decoder_failures():
)
@pytest.mark.parametrize(
"route",
[
"/realtime/client_secrets",
"/v1/realtime/client_secrets",
"/openai/v1/realtime/client_secrets",
"/realtime/calls",
"/v1/realtime/calls",
"/openai/v1/realtime/calls",
],
)
def test_get_model_from_request_extracts_realtime_session_model(route):
"""The effective realtime model lives in ``session.model`` (not the
top-level ``model``). It must be surfaced so can_key_call_model() can
validate the model a restricted key is actually requesting.
Regression test for the model-access bypass on the GA Realtime WebRTC
HTTP routes (https://github.com/BerriAI/litellm/issues/29923).
"""
assert (
get_model_from_request(
request_data={"session": {"type": "realtime", "model": "gpt-realtime"}},
route=route,
)
== "gpt-realtime"
)
def test_get_model_from_request_realtime_includes_top_level_and_session_model():
"""When both top-level and session model are present, both are returned so
neither path can smuggle a disallowed model past the model-access check."""
models = get_model_from_request(
request_data={
"model": "gpt-4o-realtime-preview",
"session": {"type": "realtime", "model": "gpt-realtime"},
},
route="/v1/realtime/client_secrets",
)
assert models == ["gpt-4o-realtime-preview", "gpt-realtime"]
def test_get_model_from_request_ignores_session_model_on_non_realtime_routes():
"""A nested ``session.model`` must not leak into model resolution for
unrelated routes."""
assert (
get_model_from_request(
request_data={"session": {"type": "realtime", "model": "gpt-realtime"}},
route="/v1/chat/completions",
)
is None
)
def test_abbreviate_api_key():
assert abbreviate_api_key("sk-test-1234") == "sk-...1234"

View file

@ -460,6 +460,32 @@ def test_mcp_inference_routes_classified_as_llm_api(route):
assert RouteChecks.is_management_route(route=route) is False
@pytest.mark.parametrize(
"route",
[
"/realtime/client_secrets",
"/v1/realtime/client_secrets",
"/openai/v1/realtime/client_secrets",
"/realtime/calls",
"/v1/realtime/calls",
"/openai/v1/realtime/calls",
"/realtime/transcription_sessions",
"/v1/realtime/transcription_sessions",
"/openai/v1/realtime/transcription_sessions",
],
)
def test_realtime_webrtc_http_routes_classified_as_llm_api(route):
"""GA Realtime WebRTC HTTP routes must be classified as LLM API routes so
non-admin virtual keys can call them instead of hitting the admin-only
401 branch in non_proxy_admin_allowed_routes_check.
Regression test for https://github.com/BerriAI/litellm/issues/29923
"""
assert RouteChecks.is_llm_api_route(route=route) is True
assert RouteChecks.is_management_route(route=route) is False
def test_virtual_key_allowed_routes_with_litellm_routes_member_name_denied():
"""Test that virtual key is denied when route is not in the allowed LiteLLMRoutes group"""