diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index 9dd176cac15..f63b65e00d0 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -11342,6 +11342,16 @@ async def _deployment_hidden_by_listing_callbacks(deployment: Deployment, user_a return listed_name in await _names_hidden_by_listing_callbacks(user_api_key_dict, (listed_name,)) +def _includes_wildcard_routes(return_wildcard_routes: bool | None, settings: Mapping[str, object]) -> bool: + """Whether a listing or lookup includes wildcard routes such as `openai/*`. A request's + own `return_wildcard_routes` wins; without one, `model_list_return_wildcard_routes` + counts as on exactly when the admin UI's switch shows it on (`true` or `"true"`).""" + if return_wildcard_routes is not None: + return return_wildcard_routes + configured: Final = settings.get("model_list_return_wildcard_routes") + return configured is True or configured == "true" + + @router.get("/v1/models", dependencies=[Depends(user_api_key_auth)], tags=["model management"]) @router.get( "/models", dependencies=[Depends(user_api_key_auth)], tags=["model management"] @@ -11456,11 +11466,7 @@ async def model_list( hidden_names: Final = blocked_names | unhealthy_names - include_wildcard_routes: Final = ( - settings.get("model_list_return_wildcard_routes") is True - if return_wildcard_routes is None - else return_wildcard_routes - ) + include_wildcard_routes: Final = _includes_wildcard_routes(return_wildcard_routes, settings) # If scope=expand and user has admin privileges, return all proxy models if should_expand_scope: @@ -11609,6 +11615,7 @@ async def model_info( user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth), team_id: str | None = None, healthy_only: bool | None = False, + return_wildcard_routes: bool | None = None, ): """ Retrieve information about a specific model accessible to your API key. @@ -11643,7 +11650,7 @@ async def model_info( team_id=team_id, include_model_access_groups=False, only_model_access_groups=False, - return_wildcard_routes=False, + return_wildcard_routes=_includes_wildcard_routes(return_wildcard_routes, settings), user_api_key_cache=user_api_key_cache, ) diff --git a/litellm/proxy/response_api_endpoints/endpoints.py b/litellm/proxy/response_api_endpoints/endpoints.py index 84221da1c79..692615005a8 100644 --- a/litellm/proxy/response_api_endpoints/endpoints.py +++ b/litellm/proxy/response_api_endpoints/endpoints.py @@ -478,13 +478,11 @@ async def cursor_model_list( verify models via `GET {base}/models` (the OpenAI SDK contract). Without this route those requests fall through to the Cursor Cloud Agents passthrough, which demands a Cursor API key and 401s, so key verification silently fails before any - chat request is ever sent. Delegates to the standard `/v1/models` handler. + chat request is ever sent. Delegates to the standard `/v1/models` handler with + wildcard routes left out, since Cursor offers every listed id as a callable model. """ from litellm.proxy.proxy_server import model_list - # Cursor offers every listed id as a model and a wildcard route such as - # `openai/*` is not callable, so the proxy-wide model_list_return_wildcard_routes - # default is pinned off here; Cursor sends no query params to opt out itself. return await model_list(user_api_key_dict=user_api_key_dict, return_wildcard_routes=False) diff --git a/tests/e2e/management/test_model_list_wildcard_routes_e2e.py b/tests/e2e/management/test_model_list_wildcard_routes_e2e.py index 9a9cea73bce..26f754ba421 100644 --- a/tests/e2e/management/test_model_list_wildcard_routes_e2e.py +++ b/tests/e2e/management/test_model_list_wildcard_routes_e2e.py @@ -3,8 +3,8 @@ default for whether GET /v1/models lists a wildcard route such as `openai/*` nex the models it expands to. The setting is written through /config/field/update, the route behind the admin UI's -General Settings toggle, and deleted on teardown so the shared proxy goes back to -leaving wildcard routes out. The other listings the suites run either pass +General Settings toggle, and teardown puts back whatever the database held before, so +the shared proxy ends the test the way it started. The other listings the suites run either pass return_wildcard_routes explicitly or only check that a named model is present, so the window with the setting on changes none of them. The wildcard deployment gets a unique prefix instead of `openai/*`, which would claim every `openai/...` request the other suites send. @@ -12,16 +12,16 @@ claim every `openai/...` request the other suites send. from __future__ import annotations -from typing import Final +from typing import Final, Literal import pytest -from pydantic import BaseModel +from pydantic import BaseModel, ConfigDict, JsonValue, RootModel from e2e_config import unique_marker from e2e_http import NoBody, Success, unwrap from lifecycle import ResourceManager from management_client import ManagementClient -from models import LiteLLMParamsBody, ModelsListParams, ModelsListResponse +from models import ConfigListParams, LiteLLMParamsBody, ModelsListParams, ModelsListResponse pytestmark = pytest.mark.e2e @@ -31,7 +31,7 @@ _DUMMY_API_KEY: Final = "e2e-dummy-key" class ConfigFieldUpdateBody(BaseModel): field_name: str - field_value: bool + field_value: JsonValue config_type: str = "general_settings" @@ -40,12 +40,23 @@ class ConfigFieldDeleteBody(BaseModel): config_type: str = "general_settings" -def _write_setting(client: ManagementClient, enabled: bool) -> None: +class StoredField(BaseModel): + model_config = ConfigDict(extra="ignore") + field_name: str + field_value: JsonValue = None + source: Literal["config", "db", "env", "default", "unset"] = "unset" + + +class StoredFieldList(RootModel[tuple[StoredField, ...]]): + pass + + +def _write_setting(client: ManagementClient, value: JsonValue) -> None: _ = unwrap( client.proxy.transport.post( "/config/field/update", headers=client.proxy.transport.master, - json=ConfigFieldUpdateBody(field_name=_SETTING, field_value=enabled), + json=ConfigFieldUpdateBody(field_name=_SETTING, field_value=value), response_type=NoBody, ) ) @@ -62,6 +73,25 @@ def _delete_setting(client: ManagementClient) -> None: ) +def _stored_setting(client: ManagementClient) -> StoredField | None: + fields: Final = unwrap( + client.proxy.transport.get( + "/config/list", + headers=client.proxy.transport.master, + params=ConfigListParams(config_type="general_settings"), + response_type=StoredFieldList, + ) + ).root + return next((field for field in fields if field.field_name == _SETTING and field.source == "db"), None) + + +def _restore_setting(client: ManagementClient, stored: StoredField | None) -> None: + if stored is None: + _delete_setting(client) + return + _write_setting(client, stored.field_value) + + def _await_listing(client: ManagementClient, pattern: str, query: BaseModel, *, listed: bool) -> None: """Poll /v1/models under `query` on every replica until each one lists `pattern`, or each one leaves it out, per `listed`.""" @@ -84,9 +114,11 @@ class TestModelListWildcardRoutesSetting: model_id = client.proxy.create_model(pattern, LiteLLMParamsBody(model="openai/*", api_key=_DUMMY_API_KEY)) resources.defer(lambda: client.proxy.delete_model(model_id)) + stored: Final = _stored_setting(client) + resources.defer(lambda: _restore_setting(client, stored)) + _write_setting(client, False) _await_listing(client, pattern, NoBody(), listed=False) - resources.defer(lambda: _delete_setting(client)) _write_setting(client, True) _await_listing(client, pattern, NoBody(), listed=True) _await_listing(client, pattern, ModelsListParams(return_wildcard_routes=False), listed=False) diff --git a/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py b/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py index cd470c8879f..f789ed8efc6 100644 --- a/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py +++ b/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py @@ -1351,8 +1351,9 @@ def test_cursor_models_route_delegates_to_model_list(): assert response.status_code == 200, f"{path}: {response.text}" assert response.json() == model_payload assert mock_model_list.call_count == 2 - # Cursor lists every id it gets, so the proxy-wide wildcard default must not reach it - assert all(call.kwargs["return_wildcard_routes"] is False for call in mock_model_list.call_args_list) + assert all(call.kwargs["return_wildcard_routes"] is False for call in mock_model_list.call_args_list), ( + "Cursor offers every listed id, so model_list_return_wildcard_routes must not reach it" + ) finally: app.dependency_overrides.pop(user_api_key_auth, None) diff --git a/tests/unit/proxy/test_proxy_server.py b/tests/unit/proxy/test_proxy_server.py index 352b12f0f50..40efdff8295 100644 --- a/tests/unit/proxy/test_proxy_server.py +++ b/tests/unit/proxy/test_proxy_server.py @@ -3138,16 +3138,19 @@ async def _listed_model_ids( return [model["id"] for model in response["data"]] +@pytest.mark.parametrize("setting", [True, "true"]) @pytest.mark.parametrize( "user_role, query", [(LitellmUserRoles.INTERNAL_USER, {}), (LitellmUserRoles.PROXY_ADMIN, {"scope": "expand"})], ) @pytest.mark.asyncio async def test_model_list_return_wildcard_routes_setting_matches_query_param( - openai_wildcard_router, monkeypatch, user_role, query + openai_wildcard_router, monkeypatch, user_role, query, setting ): requested = await _listed_model_ids(monkeypatch, {}, user_role, return_wildcard_routes=True, **query) - from_setting = await _listed_model_ids(monkeypatch, {"model_list_return_wildcard_routes": True}, user_role, **query) + from_setting = await _listed_model_ids( + monkeypatch, {"model_list_return_wildcard_routes": setting}, user_role, **query + ) assert "openai/*" in from_setting, from_setting assert from_setting == requested @@ -3158,6 +3161,8 @@ async def test_model_list_return_wildcard_routes_setting_matches_query_param( [ ({}, {}), ({"model_list_return_wildcard_routes": "false"}, {}), + ({"model_list_return_wildcard_routes": "True"}, {}), + ({"model_list_return_wildcard_routes": 1}, {}), ({"model_list_return_wildcard_routes": True}, {"return_wildcard_routes": False}), ], ) @@ -3171,6 +3176,42 @@ async def test_model_list_leaves_out_wildcard_routes_unless_enabled( assert "openai/*" not in listed, listed +async def _model_info_resolves(model_id: str, **query: bool) -> bool: + from fastapi import HTTPException + + try: + response = await litellm.proxy.proxy_server.model_info( + model_id=model_id, + user_api_key_dict=UserAPIKeyAuth(api_key="sk-test", user_role=LitellmUserRoles.INTERNAL_USER), + **query, + ) + except HTTPException as e: + assert e.status_code == 404, e.detail + return False + return response["id"] == model_id + + +@pytest.mark.parametrize( + "general_settings, query, wildcard_listed", + [ + ({}, {}, False), + ({"model_list_return_wildcard_routes": True}, {}, True), + ({"model_list_return_wildcard_routes": "true"}, {}, True), + ({"model_list_return_wildcard_routes": True}, {"return_wildcard_routes": False}, False), + ({}, {"return_wildcard_routes": True}, True), + ], +) +@pytest.mark.asyncio +async def test_model_info_resolves_exactly_the_ids_model_list_lists( + openai_wildcard_router, monkeypatch, general_settings, query, wildcard_listed +): + listed = await _listed_model_ids(monkeypatch, general_settings, LitellmUserRoles.INTERNAL_USER, **query) + resolvable = [model_id for model_id in ("gpt-4o", "openai/*") if await _model_info_resolves(model_id, **query)] + + assert ("openai/*" in listed) is wildcard_listed, listed + assert resolvable == [model_id for model_id in ("gpt-4o", "openai/*") if model_id in listed], (listed, resolvable) + + @pytest.mark.asyncio async def test_model_list_return_wildcard_routes_is_an_admin_ui_toggle(openai_wildcard_router, monkeypatch): stored_general_settings: Final = {"model_list_return_wildcard_routes": True} diff --git a/ui/litellm-dashboard/src/components/networking.test.ts b/ui/litellm-dashboard/src/components/networking.test.ts index 1378b7b9eeb..18488e21555 100644 --- a/ui/litellm-dashboard/src/components/networking.test.ts +++ b/ui/litellm-dashboard/src/components/networking.test.ts @@ -168,8 +168,6 @@ describe("modelAvailableCall", () => { global.fetch = currentFetch; }); - // An omitted return_wildcard_routes takes the proxy's model_list_return_wildcard_routes - // default, so the dashboard's own pickers always send it. it.each([ [undefined, "False"], [false, "False"], diff --git a/ui/litellm-dashboard/src/components/networking.tsx b/ui/litellm-dashboard/src/components/networking.tsx index affe24187f4..0c9251e5ac8 100644 --- a/ui/litellm-dashboard/src/components/networking.tsx +++ b/ui/litellm-dashboard/src/components/networking.tsx @@ -1940,8 +1940,6 @@ export const modelAvailableCall = async ( accessToken, query: { include_model_access_groups: "True", - // Sent either way: an omitted value falls back to the proxy's - // general_settings.model_list_return_wildcard_routes, not to false. return_wildcard_routes: return_wildcard_routes === true ? "True" : "False", only_model_access_groups: only_model_access_groups === true ? "True" : undefined, team_id: teamID || undefined, diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index 4f82a475fb8..79089fcef75 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -3761,7 +3761,8 @@ export interface paths { * verify models via `GET {base}/models` (the OpenAI SDK contract). Without this * route those requests fall through to the Cursor Cloud Agents passthrough, which * demands a Cursor API key and 401s, so key verification silently fails before any - * chat request is ever sent. Delegates to the standard `/v1/models` handler. + * chat request is ever sent. Delegates to the standard `/v1/models` handler with + * wildcard routes left out, since Cursor offers every listed id as a callable model. */ get: operations["cursor_model_list_cursor_models_get"]; put?: never; @@ -3787,7 +3788,8 @@ export interface paths { * verify models via `GET {base}/models` (the OpenAI SDK contract). Without this * route those requests fall through to the Cursor Cloud Agents passthrough, which * demands a Cursor API key and 401s, so key verification silently fails before any - * chat request is ever sent. Delegates to the standard `/v1/models` handler. + * chat request is ever sent. Delegates to the standard `/v1/models` handler with + * wildcard routes left out, since Cursor offers every listed id as a callable model. */ get: operations["cursor_model_list_cursor_v1_models_get"]; put?: never; @@ -59947,6 +59949,7 @@ export interface operations { query?: { team_id?: string | null; healthy_only?: boolean | null; + return_wildcard_routes?: boolean | null; }; header?: never; path: { @@ -73252,6 +73255,7 @@ export interface operations { query?: { team_id?: string | null; healthy_only?: boolean | null; + return_wildcard_routes?: boolean | null; }; header?: never; path: {