mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-10 03:28:53 +00:00
fix(proxy): allow non-admin virtual keys to call GA Realtime WebRTC HTTP routes (#30089)
* fix(proxy): allow non-admin virtual keys to call GA Realtime WebRTC HTTP routes Add the realtime WebRTC HTTP sub-routes (/realtime/client_secrets, /realtime/calls and their /v1 + /openai/v1 variants) to LiteLLMRoutes.openai_routes so is_llm_api_route() classifies them as LLM API routes. Without this, non-admin virtual keys received 401 'Only proxy admin can be used to generate, delete, update info for new keys/users/teams' when calling these endpoints. Fixes #29923 * fix(proxy): validate session.model for realtime routes in model-access check The GA Realtime WebRTC HTTP routes resolve the effective model from the nested session.model (falling back to the top-level model), but the auth layer's get_model_from_request() only extracted the top-level model. A model-restricted virtual key could therefore place a disallowed model in session.model, leave the top-level model unset, and skip can_key_call_model() entirely - obtaining an ephemeral token for a model it is not allowed to use. Extract session.model for the realtime client_secrets/calls routes so the model-access check runs against the model the request will actually use. Legitimate callers are unaffected; their permitted model still validates. Relates to https://github.com/BerriAI/litellm/issues/29923 * fix(proxy): classify realtime transcription_sessions routes as LLM API routes Add the GA Realtime WebRTC transcription_sessions HTTP routes to openai_routes so is_llm_api_route() returns True for them, matching the client_secrets and calls routes already fixed. These endpoints are registered with user_api_key_auth in realtime_endpoints/endpoints.py, so without this a non-admin virtual key calling POST /v1/realtime/transcription_sessions would hit the admin-only 401 branch. Extends the regression test parametrization accordingly. --------- Co-authored-by: habonlaci <4699494+habonlaci@users.noreply.github.com>
This commit is contained in:
parent
cf2db415b8
commit
682bb6caad
4 changed files with 103 additions and 0 deletions
|
|
@ -361,6 +361,16 @@ class LiteLLMRoutes(enum.Enum):
|
|||
"/realtime?{model}",
|
||||
"/v1/realtime?{model}",
|
||||
"/openai/v1/realtime?{model}",
|
||||
# realtime (GA WebRTC HTTP routes)
|
||||
"/realtime/client_secrets",
|
||||
"/v1/realtime/client_secrets",
|
||||
"/openai/v1/realtime/client_secrets",
|
||||
"/realtime/calls",
|
||||
"/v1/realtime/calls",
|
||||
"/openai/v1/realtime/calls",
|
||||
"/realtime/transcription_sessions",
|
||||
"/v1/realtime/transcription_sessions",
|
||||
"/openai/v1/realtime/transcription_sessions",
|
||||
# responses API
|
||||
"/responses",
|
||||
"/v1/responses",
|
||||
|
|
|
|||
|
|
@ -1227,6 +1227,14 @@ _MODEL_ROUTING_BODY_TARGET_MODEL_ROUTE_MARKERS = (
|
|||
"/vector_stores",
|
||||
)
|
||||
_MODEL_ROUTING_COMPLETION_MODEL_ROUTE_MARKERS = ("/evals",)
|
||||
# Realtime WebRTC routes carry the effective model inside the nested
|
||||
# ``session.model`` field (see realtime_endpoints.endpoints), so the model the
|
||||
# request will actually use is not present at the top level. Extract it here so
|
||||
# can_key_call_model() validates the real target model.
|
||||
_MODEL_ROUTING_SESSION_MODEL_ROUTE_MARKERS = (
|
||||
"/realtime/client_secrets",
|
||||
"/realtime/calls",
|
||||
)
|
||||
_MODEL_ROUTING_ID_FIELDS = (
|
||||
"file_id",
|
||||
"input_file_id",
|
||||
|
|
@ -1409,6 +1417,12 @@ def _extract_model_candidates_from_request(
|
|||
_append_model_candidates(candidates, body_model)
|
||||
if uses_body_target_model_sources or not body_model:
|
||||
_append_model_candidates(candidates, request_data.get("target_model_names"))
|
||||
if _route_matches_any_marker(
|
||||
route=route, markers=_MODEL_ROUTING_SESSION_MODEL_ROUTE_MARKERS
|
||||
):
|
||||
session = request_data.get("session")
|
||||
if isinstance(session, dict):
|
||||
_append_model_candidates(candidates, session.get("model"))
|
||||
if uses_completion_model_sources and isinstance(
|
||||
request_data.get("completion"), dict
|
||||
):
|
||||
|
|
|
|||
|
|
@ -530,6 +530,59 @@ def test_get_model_from_request_handles_managed_id_decoder_failures():
|
|||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"route",
|
||||
[
|
||||
"/realtime/client_secrets",
|
||||
"/v1/realtime/client_secrets",
|
||||
"/openai/v1/realtime/client_secrets",
|
||||
"/realtime/calls",
|
||||
"/v1/realtime/calls",
|
||||
"/openai/v1/realtime/calls",
|
||||
],
|
||||
)
|
||||
def test_get_model_from_request_extracts_realtime_session_model(route):
|
||||
"""The effective realtime model lives in ``session.model`` (not the
|
||||
top-level ``model``). It must be surfaced so can_key_call_model() can
|
||||
validate the model a restricted key is actually requesting.
|
||||
|
||||
Regression test for the model-access bypass on the GA Realtime WebRTC
|
||||
HTTP routes (https://github.com/BerriAI/litellm/issues/29923).
|
||||
"""
|
||||
assert (
|
||||
get_model_from_request(
|
||||
request_data={"session": {"type": "realtime", "model": "gpt-realtime"}},
|
||||
route=route,
|
||||
)
|
||||
== "gpt-realtime"
|
||||
)
|
||||
|
||||
|
||||
def test_get_model_from_request_realtime_includes_top_level_and_session_model():
|
||||
"""When both top-level and session model are present, both are returned so
|
||||
neither path can smuggle a disallowed model past the model-access check."""
|
||||
models = get_model_from_request(
|
||||
request_data={
|
||||
"model": "gpt-4o-realtime-preview",
|
||||
"session": {"type": "realtime", "model": "gpt-realtime"},
|
||||
},
|
||||
route="/v1/realtime/client_secrets",
|
||||
)
|
||||
assert models == ["gpt-4o-realtime-preview", "gpt-realtime"]
|
||||
|
||||
|
||||
def test_get_model_from_request_ignores_session_model_on_non_realtime_routes():
|
||||
"""A nested ``session.model`` must not leak into model resolution for
|
||||
unrelated routes."""
|
||||
assert (
|
||||
get_model_from_request(
|
||||
request_data={"session": {"type": "realtime", "model": "gpt-realtime"}},
|
||||
route="/v1/chat/completions",
|
||||
)
|
||||
is None
|
||||
)
|
||||
|
||||
|
||||
def test_abbreviate_api_key():
|
||||
assert abbreviate_api_key("sk-test-1234") == "sk-...1234"
|
||||
|
||||
|
|
|
|||
|
|
@ -460,6 +460,32 @@ def test_mcp_inference_routes_classified_as_llm_api(route):
|
|||
assert RouteChecks.is_management_route(route=route) is False
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"route",
|
||||
[
|
||||
"/realtime/client_secrets",
|
||||
"/v1/realtime/client_secrets",
|
||||
"/openai/v1/realtime/client_secrets",
|
||||
"/realtime/calls",
|
||||
"/v1/realtime/calls",
|
||||
"/openai/v1/realtime/calls",
|
||||
"/realtime/transcription_sessions",
|
||||
"/v1/realtime/transcription_sessions",
|
||||
"/openai/v1/realtime/transcription_sessions",
|
||||
],
|
||||
)
|
||||
def test_realtime_webrtc_http_routes_classified_as_llm_api(route):
|
||||
"""GA Realtime WebRTC HTTP routes must be classified as LLM API routes so
|
||||
non-admin virtual keys can call them instead of hitting the admin-only
|
||||
401 branch in non_proxy_admin_allowed_routes_check.
|
||||
|
||||
Regression test for https://github.com/BerriAI/litellm/issues/29923
|
||||
"""
|
||||
|
||||
assert RouteChecks.is_llm_api_route(route=route) is True
|
||||
assert RouteChecks.is_management_route(route=route) is False
|
||||
|
||||
|
||||
def test_virtual_key_allowed_routes_with_litellm_routes_member_name_denied():
|
||||
"""Test that virtual key is denied when route is not in the allowed LiteLLMRoutes group"""
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue