From 5437b86a5e9f1e7669b3ec572d0b7498ead196be Mon Sep 17 00:00:00 2001 From: shrey kharbanda Date: Mon, 28 Sep 2026 15:24:08 -0700 Subject: [PATCH] fix(proxy): resolve listed wildcard ids in /v1/models/{id} and honor the UI's "true" /v1/models/{model_id} now follows the same wildcard rule as the listing, so any id /v1/models returns can be looked up by id, with the same return_wildcard_routes override. The setting also counts as on for the string "true", matching what the Admin UI switch shows. The e2e restores whatever the database held before the test instead of deleting the setting. --- litellm/proxy/proxy_server.py | 19 ++++--- .../proxy/response_api_endpoints/endpoints.py | 6 +-- .../test_model_list_wildcard_routes_e2e.py | 50 +++++++++++++++---- .../response_api_endpoints/test_endpoints.py | 5 +- tests/unit/proxy/test_proxy_server.py | 45 ++++++++++++++++- .../src/components/networking.test.ts | 2 - .../src/components/networking.tsx | 2 - ui/litellm-dashboard/src/lib/http/schema.d.ts | 8 ++- 8 files changed, 108 insertions(+), 29 deletions(-) diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index 9dd176cac15..f63b65e00d0 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -11342,6 +11342,16 @@ async def _deployment_hidden_by_listing_callbacks(deployment: Deployment, user_a return listed_name in await _names_hidden_by_listing_callbacks(user_api_key_dict, (listed_name,)) +def _includes_wildcard_routes(return_wildcard_routes: bool | None, settings: Mapping[str, object]) -> bool: + """Whether a listing or lookup includes wildcard routes such as `openai/*`. A request's + own `return_wildcard_routes` wins; without one, `model_list_return_wildcard_routes` + counts as on exactly when the admin UI's switch shows it on (`true` or `"true"`).""" + if return_wildcard_routes is not None: + return return_wildcard_routes + configured: Final = settings.get("model_list_return_wildcard_routes") + return configured is True or configured == "true" + + @router.get("/v1/models", dependencies=[Depends(user_api_key_auth)], tags=["model management"]) @router.get( "/models", dependencies=[Depends(user_api_key_auth)], tags=["model management"] @@ -11456,11 +11466,7 @@ async def model_list( hidden_names: Final = blocked_names | unhealthy_names - include_wildcard_routes: Final = ( - settings.get("model_list_return_wildcard_routes") is True - if return_wildcard_routes is None - else return_wildcard_routes - ) + include_wildcard_routes: Final = _includes_wildcard_routes(return_wildcard_routes, settings) # If scope=expand and user has admin privileges, return all proxy models if should_expand_scope: @@ -11609,6 +11615,7 @@ async def model_info( user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth), team_id: str | None = None, healthy_only: bool | None = False, + return_wildcard_routes: bool | None = None, ): """ Retrieve information about a specific model accessible to your API key. @@ -11643,7 +11650,7 @@ async def model_info( team_id=team_id, include_model_access_groups=False, only_model_access_groups=False, - return_wildcard_routes=False, + return_wildcard_routes=_includes_wildcard_routes(return_wildcard_routes, settings), user_api_key_cache=user_api_key_cache, ) diff --git a/litellm/proxy/response_api_endpoints/endpoints.py b/litellm/proxy/response_api_endpoints/endpoints.py index 84221da1c79..692615005a8 100644 --- a/litellm/proxy/response_api_endpoints/endpoints.py +++ b/litellm/proxy/response_api_endpoints/endpoints.py @@ -478,13 +478,11 @@ async def cursor_model_list( verify models via `GET {base}/models` (the OpenAI SDK contract). Without this route those requests fall through to the Cursor Cloud Agents passthrough, which demands a Cursor API key and 401s, so key verification silently fails before any - chat request is ever sent. Delegates to the standard `/v1/models` handler. + chat request is ever sent. Delegates to the standard `/v1/models` handler with + wildcard routes left out, since Cursor offers every listed id as a callable model. """ from litellm.proxy.proxy_server import model_list - # Cursor offers every listed id as a model and a wildcard route such as - # `openai/*` is not callable, so the proxy-wide model_list_return_wildcard_routes - # default is pinned off here; Cursor sends no query params to opt out itself. return await model_list(user_api_key_dict=user_api_key_dict, return_wildcard_routes=False) diff --git a/tests/e2e/management/test_model_list_wildcard_routes_e2e.py b/tests/e2e/management/test_model_list_wildcard_routes_e2e.py index 9a9cea73bce..26f754ba421 100644 --- a/tests/e2e/management/test_model_list_wildcard_routes_e2e.py +++ b/tests/e2e/management/test_model_list_wildcard_routes_e2e.py @@ -3,8 +3,8 @@ default for whether GET /v1/models lists a wildcard route such as `openai/*` nex the models it expands to. The setting is written through /config/field/update, the route behind the admin UI's -General Settings toggle, and deleted on teardown so the shared proxy goes back to -leaving wildcard routes out. The other listings the suites run either pass +General Settings toggle, and teardown puts back whatever the database held before, so +the shared proxy ends the test the way it started. The other listings the suites run either pass return_wildcard_routes explicitly or only check that a named model is present, so the window with the setting on changes none of them. The wildcard deployment gets a unique prefix instead of `openai/*`, which would claim every `openai/...` request the other suites send. @@ -12,16 +12,16 @@ claim every `openai/...` request the other suites send. from __future__ import annotations -from typing import Final +from typing import Final, Literal import pytest -from pydantic import BaseModel +from pydantic import BaseModel, ConfigDict, JsonValue, RootModel from e2e_config import unique_marker from e2e_http import NoBody, Success, unwrap from lifecycle import ResourceManager from management_client import ManagementClient -from models import LiteLLMParamsBody, ModelsListParams, ModelsListResponse +from models import ConfigListParams, LiteLLMParamsBody, ModelsListParams, ModelsListResponse pytestmark = pytest.mark.e2e @@ -31,7 +31,7 @@ _DUMMY_API_KEY: Final = "e2e-dummy-key" class ConfigFieldUpdateBody(BaseModel): field_name: str - field_value: bool + field_value: JsonValue config_type: str = "general_settings" @@ -40,12 +40,23 @@ class ConfigFieldDeleteBody(BaseModel): config_type: str = "general_settings" -def _write_setting(client: ManagementClient, enabled: bool) -> None: +class StoredField(BaseModel): + model_config = ConfigDict(extra="ignore") + field_name: str + field_value: JsonValue = None + source: Literal["config", "db", "env", "default", "unset"] = "unset" + + +class StoredFieldList(RootModel[tuple[StoredField, ...]]): + pass + + +def _write_setting(client: ManagementClient, value: JsonValue) -> None: _ = unwrap( client.proxy.transport.post( "/config/field/update", headers=client.proxy.transport.master, - json=ConfigFieldUpdateBody(field_name=_SETTING, field_value=enabled), + json=ConfigFieldUpdateBody(field_name=_SETTING, field_value=value), response_type=NoBody, ) ) @@ -62,6 +73,25 @@ def _delete_setting(client: ManagementClient) -> None: ) +def _stored_setting(client: ManagementClient) -> StoredField | None: + fields: Final = unwrap( + client.proxy.transport.get( + "/config/list", + headers=client.proxy.transport.master, + params=ConfigListParams(config_type="general_settings"), + response_type=StoredFieldList, + ) + ).root + return next((field for field in fields if field.field_name == _SETTING and field.source == "db"), None) + + +def _restore_setting(client: ManagementClient, stored: StoredField | None) -> None: + if stored is None: + _delete_setting(client) + return + _write_setting(client, stored.field_value) + + def _await_listing(client: ManagementClient, pattern: str, query: BaseModel, *, listed: bool) -> None: """Poll /v1/models under `query` on every replica until each one lists `pattern`, or each one leaves it out, per `listed`.""" @@ -84,9 +114,11 @@ class TestModelListWildcardRoutesSetting: model_id = client.proxy.create_model(pattern, LiteLLMParamsBody(model="openai/*", api_key=_DUMMY_API_KEY)) resources.defer(lambda: client.proxy.delete_model(model_id)) + stored: Final = _stored_setting(client) + resources.defer(lambda: _restore_setting(client, stored)) + _write_setting(client, False) _await_listing(client, pattern, NoBody(), listed=False) - resources.defer(lambda: _delete_setting(client)) _write_setting(client, True) _await_listing(client, pattern, NoBody(), listed=True) _await_listing(client, pattern, ModelsListParams(return_wildcard_routes=False), listed=False) diff --git a/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py b/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py index cd470c8879f..f789ed8efc6 100644 --- a/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py +++ b/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py @@ -1351,8 +1351,9 @@ def test_cursor_models_route_delegates_to_model_list(): assert response.status_code == 200, f"{path}: {response.text}" assert response.json() == model_payload assert mock_model_list.call_count == 2 - # Cursor lists every id it gets, so the proxy-wide wildcard default must not reach it - assert all(call.kwargs["return_wildcard_routes"] is False for call in mock_model_list.call_args_list) + assert all(call.kwargs["return_wildcard_routes"] is False for call in mock_model_list.call_args_list), ( + "Cursor offers every listed id, so model_list_return_wildcard_routes must not reach it" + ) finally: app.dependency_overrides.pop(user_api_key_auth, None) diff --git a/tests/unit/proxy/test_proxy_server.py b/tests/unit/proxy/test_proxy_server.py index 352b12f0f50..40efdff8295 100644 --- a/tests/unit/proxy/test_proxy_server.py +++ b/tests/unit/proxy/test_proxy_server.py @@ -3138,16 +3138,19 @@ async def _listed_model_ids( return [model["id"] for model in response["data"]] +@pytest.mark.parametrize("setting", [True, "true"]) @pytest.mark.parametrize( "user_role, query", [(LitellmUserRoles.INTERNAL_USER, {}), (LitellmUserRoles.PROXY_ADMIN, {"scope": "expand"})], ) @pytest.mark.asyncio async def test_model_list_return_wildcard_routes_setting_matches_query_param( - openai_wildcard_router, monkeypatch, user_role, query + openai_wildcard_router, monkeypatch, user_role, query, setting ): requested = await _listed_model_ids(monkeypatch, {}, user_role, return_wildcard_routes=True, **query) - from_setting = await _listed_model_ids(monkeypatch, {"model_list_return_wildcard_routes": True}, user_role, **query) + from_setting = await _listed_model_ids( + monkeypatch, {"model_list_return_wildcard_routes": setting}, user_role, **query + ) assert "openai/*" in from_setting, from_setting assert from_setting == requested @@ -3158,6 +3161,8 @@ async def test_model_list_return_wildcard_routes_setting_matches_query_param( [ ({}, {}), ({"model_list_return_wildcard_routes": "false"}, {}), + ({"model_list_return_wildcard_routes": "True"}, {}), + ({"model_list_return_wildcard_routes": 1}, {}), ({"model_list_return_wildcard_routes": True}, {"return_wildcard_routes": False}), ], ) @@ -3171,6 +3176,42 @@ async def test_model_list_leaves_out_wildcard_routes_unless_enabled( assert "openai/*" not in listed, listed +async def _model_info_resolves(model_id: str, **query: bool) -> bool: + from fastapi import HTTPException + + try: + response = await litellm.proxy.proxy_server.model_info( + model_id=model_id, + user_api_key_dict=UserAPIKeyAuth(api_key="sk-test", user_role=LitellmUserRoles.INTERNAL_USER), + **query, + ) + except HTTPException as e: + assert e.status_code == 404, e.detail + return False + return response["id"] == model_id + + +@pytest.mark.parametrize( + "general_settings, query, wildcard_listed", + [ + ({}, {}, False), + ({"model_list_return_wildcard_routes": True}, {}, True), + ({"model_list_return_wildcard_routes": "true"}, {}, True), + ({"model_list_return_wildcard_routes": True}, {"return_wildcard_routes": False}, False), + ({}, {"return_wildcard_routes": True}, True), + ], +) +@pytest.mark.asyncio +async def test_model_info_resolves_exactly_the_ids_model_list_lists( + openai_wildcard_router, monkeypatch, general_settings, query, wildcard_listed +): + listed = await _listed_model_ids(monkeypatch, general_settings, LitellmUserRoles.INTERNAL_USER, **query) + resolvable = [model_id for model_id in ("gpt-4o", "openai/*") if await _model_info_resolves(model_id, **query)] + + assert ("openai/*" in listed) is wildcard_listed, listed + assert resolvable == [model_id for model_id in ("gpt-4o", "openai/*") if model_id in listed], (listed, resolvable) + + @pytest.mark.asyncio async def test_model_list_return_wildcard_routes_is_an_admin_ui_toggle(openai_wildcard_router, monkeypatch): stored_general_settings: Final = {"model_list_return_wildcard_routes": True} diff --git a/ui/litellm-dashboard/src/components/networking.test.ts b/ui/litellm-dashboard/src/components/networking.test.ts index 1378b7b9eeb..18488e21555 100644 --- a/ui/litellm-dashboard/src/components/networking.test.ts +++ b/ui/litellm-dashboard/src/components/networking.test.ts @@ -168,8 +168,6 @@ describe("modelAvailableCall", () => { global.fetch = currentFetch; }); - // An omitted return_wildcard_routes takes the proxy's model_list_return_wildcard_routes - // default, so the dashboard's own pickers always send it. it.each([ [undefined, "False"], [false, "False"], diff --git a/ui/litellm-dashboard/src/components/networking.tsx b/ui/litellm-dashboard/src/components/networking.tsx index affe24187f4..0c9251e5ac8 100644 --- a/ui/litellm-dashboard/src/components/networking.tsx +++ b/ui/litellm-dashboard/src/components/networking.tsx @@ -1940,8 +1940,6 @@ export const modelAvailableCall = async ( accessToken, query: { include_model_access_groups: "True", - // Sent either way: an omitted value falls back to the proxy's - // general_settings.model_list_return_wildcard_routes, not to false. return_wildcard_routes: return_wildcard_routes === true ? "True" : "False", only_model_access_groups: only_model_access_groups === true ? "True" : undefined, team_id: teamID || undefined, diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index 4f82a475fb8..79089fcef75 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -3761,7 +3761,8 @@ export interface paths { * verify models via `GET {base}/models` (the OpenAI SDK contract). Without this * route those requests fall through to the Cursor Cloud Agents passthrough, which * demands a Cursor API key and 401s, so key verification silently fails before any - * chat request is ever sent. Delegates to the standard `/v1/models` handler. + * chat request is ever sent. Delegates to the standard `/v1/models` handler with + * wildcard routes left out, since Cursor offers every listed id as a callable model. */ get: operations["cursor_model_list_cursor_models_get"]; put?: never; @@ -3787,7 +3788,8 @@ export interface paths { * verify models via `GET {base}/models` (the OpenAI SDK contract). Without this * route those requests fall through to the Cursor Cloud Agents passthrough, which * demands a Cursor API key and 401s, so key verification silently fails before any - * chat request is ever sent. Delegates to the standard `/v1/models` handler. + * chat request is ever sent. Delegates to the standard `/v1/models` handler with + * wildcard routes left out, since Cursor offers every listed id as a callable model. */ get: operations["cursor_model_list_cursor_v1_models_get"]; put?: never; @@ -59947,6 +59949,7 @@ export interface operations { query?: { team_id?: string | null; healthy_only?: boolean | null; + return_wildcard_routes?: boolean | null; }; header?: never; path: { @@ -73252,6 +73255,7 @@ export interface operations { query?: { team_id?: string | null; healthy_only?: boolean | null; + return_wildcard_routes?: boolean | null; }; header?: never; path: {