From b841d9a6c21558486523a96ad281220741966567 Mon Sep 17 00:00:00 2001 From: shrey kharbanda Date: Mon, 28 Sep 2026 12:54:40 -0700 Subject: [PATCH 1/6] feat(proxy): add model_list_return_wildcard_routes setting for /v1/models general_settings.model_list_return_wildcard_routes: true makes /v1/models list wildcard routes such as openai/* for every caller, the same as passing return_wildcard_routes=true. An explicit return_wildcard_routes=false on the request still leaves them out --- litellm/proxy/_types.py | 8 +++ litellm/proxy/proxy_server.py | 16 ++++- tests/unit/proxy/test_proxy_server.py | 61 +++++++++++++++++++ ui/litellm-dashboard/src/lib/http/schema.d.ts | 13 ++++ 4 files changed, 95 insertions(+), 3 deletions(-) diff --git a/litellm/proxy/_types.py b/litellm/proxy/_types.py index 3d8d701d15b..41a9341c6cf 100644 --- a/litellm/proxy/_types.py +++ b/litellm/proxy/_types.py @@ -2856,6 +2856,14 @@ class ConfigGeneralSettings(LiteLLMPydanticObjectBase): "hidden model can still be called." ), ) + model_list_return_wildcard_routes: bool | None = Field( + None, + description=( + "When true, `/models` lists wildcard routes such as `openai/*` next to the models they " + "expand to, for every caller, without needing `return_wildcard_routes=true` per request. " + "A request can still pass `return_wildcard_routes=false` to leave them out." + ), + ) alerting: list | None = Field( None, description="List of alerting integrations - e.g. `alerting: ['slack', 'webhook', 'email']`. 'slack' posts Slack-format messages to any Slack-compatible webhook (Slack, Rocket.Chat, Mattermost); 'webhook' posts structured JSON budget alerts to WEBHOOK_URL", diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index ae559f30857..9c45fd23b7a 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -11349,7 +11349,7 @@ async def _deployment_hidden_by_listing_callbacks(deployment: Deployment, user_a async def model_list( request: Request = None, # pyright: ignore[reportArgumentType] # FastAPI always injects the Request; the None default only serves direct in-process callers user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth), - return_wildcard_routes: bool | None = False, + return_wildcard_routes: bool | None = None, team_id: str | None = None, include_model_access_groups: bool | None = False, only_model_access_groups: bool | None = False, @@ -11364,6 +11364,10 @@ async def model_list( This is just for compatibility with openai projects like aider. Query Parameters: + - return_wildcard_routes: When true, also list wildcard routes (e.g. `openai/*`) + next to the models they expand to. Defaults to + `general_settings.model_list_return_wildcard_routes`, which is + false unless set; pass `false` to leave them out regardless. - include_metadata: Include additional metadata in the response with fallback information - fallback_type: Type of fallbacks to include ("general", "context_window", "content_policy") Defaults to "general" when include_metadata=true @@ -11452,6 +11456,12 @@ async def model_list( hidden_names: Final = blocked_names | unhealthy_names + include_wildcard_routes: Final = ( + settings.get("model_list_return_wildcard_routes") is True + if return_wildcard_routes is None + else return_wildcard_routes + ) + # If scope=expand and user has admin privileges, return all proxy models if should_expand_scope: # Get all proxy models as if user is a proxy admin @@ -11475,7 +11485,7 @@ async def model_list( proxy_model_list=proxy_model_list, user_model=None, infer_model_from_keys=False, - return_wildcard_routes=return_wildcard_routes or False, + return_wildcard_routes=include_wildcard_routes, llm_router=llm_router, model_access_groups=model_access_groups, include_model_access_groups=include_model_access_groups or False, @@ -11535,7 +11545,7 @@ async def model_list( team_id=team_id, include_model_access_groups=include_model_access_groups or False, only_model_access_groups=only_model_access_groups or False, - return_wildcard_routes=return_wildcard_routes or False, + return_wildcard_routes=include_wildcard_routes, user_api_key_cache=user_api_key_cache, ) diff --git a/tests/unit/proxy/test_proxy_server.py b/tests/unit/proxy/test_proxy_server.py index 8947da4d9fc..8c676e97c99 100644 --- a/tests/unit/proxy/test_proxy_server.py +++ b/tests/unit/proxy/test_proxy_server.py @@ -3108,3 +3108,64 @@ def test_get_litellm_model_info(data): ): get_litellm_model_info(model=model) get_info_mock.assert_called_once_with(data["expected"]) + + +@pytest.fixture +def openai_wildcard_router(monkeypatch: pytest.MonkeyPatch) -> litellm.Router: + router = litellm.Router( + model_list=[ + {"model_name": "gpt-4o", "litellm_params": {"model": "openai/gpt-4o", "api_key": "sk-fake"}}, + {"model_name": "openai/*", "litellm_params": {"model": "openai/*", "api_key": "sk-fake"}}, + ] + ) + monkeypatch.setattr(litellm.proxy.proxy_server, "llm_router", router) + monkeypatch.setattr(litellm.proxy.proxy_server, "llm_model_list", router.model_list) + monkeypatch.setattr(litellm.proxy.proxy_server, "prisma_client", None) + monkeypatch.setattr(litellm.proxy.proxy_server, "user_model", None) + return router + + +async def _listed_model_ids( + monkeypatch: pytest.MonkeyPatch, + general_settings: dict[str, object], + user_role: LitellmUserRoles, + **query: str | bool, +) -> list[str]: + monkeypatch.setattr(litellm.proxy.proxy_server, "general_settings", general_settings) + response = await litellm.proxy.proxy_server.model_list( + user_api_key_dict=UserAPIKeyAuth(api_key="sk-test", user_role=user_role), **query + ) + return [model["id"] for model in response["data"]] + + +@pytest.mark.parametrize( + "user_role, query", + [(LitellmUserRoles.INTERNAL_USER, {}), (LitellmUserRoles.PROXY_ADMIN, {"scope": "expand"})], +) +@pytest.mark.asyncio +async def test_model_list_return_wildcard_routes_setting_matches_query_param( + openai_wildcard_router, monkeypatch, user_role, query +): + requested = await _listed_model_ids(monkeypatch, {}, user_role, return_wildcard_routes=True, **query) + from_setting = await _listed_model_ids(monkeypatch, {"model_list_return_wildcard_routes": True}, user_role, **query) + + assert "openai/*" in from_setting, from_setting + assert from_setting == requested + + +@pytest.mark.parametrize( + "general_settings, query", + [ + ({}, {}), + ({"model_list_return_wildcard_routes": "false"}, {}), + ({"model_list_return_wildcard_routes": True}, {"return_wildcard_routes": False}), + ], +) +@pytest.mark.asyncio +async def test_model_list_leaves_out_wildcard_routes_unless_enabled( + openai_wildcard_router, monkeypatch, general_settings, query +): + listed = await _listed_model_ids(monkeypatch, general_settings, LitellmUserRoles.INTERNAL_USER, **query) + + assert "gpt-4o" in listed, listed + assert "openai/*" not in listed, listed diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index 16325970ed1..d0becba1ba1 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -9779,6 +9779,10 @@ export interface paths { * This is just for compatibility with openai projects like aider. * * Query Parameters: + * - return_wildcard_routes: When true, also list wildcard routes (e.g. `openai/*`) + * next to the models they expand to. Defaults to + * `general_settings.model_list_return_wildcard_routes`, which is + * false unless set; pass `false` to leave them out regardless. * - include_metadata: Include additional metadata in the response with fallback information * - fallback_type: Type of fallbacks to include ("general", "context_window", "content_policy") * Defaults to "general" when include_metadata=true @@ -20059,6 +20063,10 @@ export interface paths { * This is just for compatibility with openai projects like aider. * * Query Parameters: + * - return_wildcard_routes: When true, also list wildcard routes (e.g. `openai/*`) + * next to the models they expand to. Defaults to + * `general_settings.model_list_return_wildcard_routes`, which is + * false unless set; pass `false` to leave them out regardless. * - include_metadata: Include additional metadata in the response with fallback information * - fallback_type: Type of fallbacks to include ("general", "context_window", "content_policy") * Defaults to "general" when include_metadata=true @@ -28379,6 +28387,11 @@ export interface components { * @description When true, `/models`, `/v1/models/{id}` and `/model/info` hide models whose backing deployments are all unhealthy, for every caller, without needing `healthy_only=true` per request. Requires `background_health_checks: true`, and keeps deployment health state cached without turning on `enable_health_check_routing`, so routing is unaffected. With no health state nothing is hidden. Hiding is presentation-only, a hidden model can still be called. */ model_list_healthy_only?: boolean | null; + /** + * Model List Return Wildcard Routes + * @description When true, `/models` lists wildcard routes such as `openai/*` next to the models they expand to, for every caller, without needing `return_wildcard_routes=true` per request. A request can still pass `return_wildcard_routes=false` to leave them out. + */ + model_list_return_wildcard_routes?: boolean | null; /** * Otel * @description [BETA] OpenTelemetry support - this might change, use with caution. From 811f1510ec3420cbc9a959e3ac72240fd20308c4 Mon Sep 17 00:00:00 2001 From: shrey kharbanda Date: Mon, 28 Sep 2026 13:40:56 -0700 Subject: [PATCH 2/6] feat(proxy): show model_list_return_wildcard_routes in the admin UI, add e2e Adding the field to _GENERAL_SETTINGS_CONFIG_LIST_FIELD_TYPES makes /config/list return it, so the General Settings page renders it as a toggle that saves through /config/field/update. The e2e test flips it on through that route, checks every replica lists the wildcard route in /v1/models and still leaves it out for return_wildcard_routes=false, then deletes it and checks the listing goes back. --- litellm/proxy/proxy_server.py | 1 + tests/e2e/coverage_registry/mgmt.yaml | 1 + .../test_model_list_wildcard_routes_e2e.py | 95 +++++++++++++++++++ tests/unit/proxy/test_proxy_server.py | 33 +++++++ 4 files changed, 130 insertions(+) create mode 100644 tests/e2e/management/test_model_list_wildcard_routes_e2e.py diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index 9c45fd23b7a..9dd176cac15 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -18038,6 +18038,7 @@ _GENERAL_SETTINGS_CONFIG_LIST_FIELD_TYPES: Final[Mapping[str, str]] = MappingPro "apply_user_budget_to_team_keys": "Boolean", "user_api_key_cache_max_size": "Integer", "transcribe_media_buckets": "List", + "model_list_return_wildcard_routes": "Boolean", } ) diff --git a/tests/e2e/coverage_registry/mgmt.yaml b/tests/e2e/coverage_registry/mgmt.yaml index e1a840b1239..6fada6b992d 100644 --- a/tests/e2e/coverage_registry/mgmt.yaml +++ b/tests/e2e/coverage_registry/mgmt.yaml @@ -72,6 +72,7 @@ - {id: mgmt.callback.list.happy_path, module: mgmt, tier: P2, surface: api, assertions: [happy_path], source: "callback_management_endpoints.py", rationale: "Callback config (smoke)"} - {id: mgmt.cost_tracking.estimate.happy_path, module: mgmt, tier: P2, surface: api, assertions: [happy_path], source: "cost_tracking_settings.py", rationale: "Cost estimate (smoke)"} - {id: mgmt.router_settings.update.happy_path, module: mgmt, tier: P2, surface: api, assertions: [happy_path], source: "router_settings_endpoints.py", rationale: "Router config (smoke)"} +- {id: mgmt.general_settings.model_list_return_wildcard_routes.persists, module: mgmt, tier: P1, surface: api, assertions: [persists], source: "proxy_server.py:model_list", rationale: "The admin-set default lists wildcard routes in /v1/models on every replica, and return_wildcard_routes=false still leaves them out"} - {id: mgmt.jwt_key_mapping.new.happy_path, module: mgmt, tier: P2, surface: api, assertions: [happy_path], source: "jwt_key_mapping_endpoints.py", rationale: "JWT->key mapping (smoke)"} - {id: mgmt.compliance.gdpr.happy_path, module: mgmt, tier: P2, surface: api, assertions: [happy_path], source: "compliance_endpoints.py", rationale: "GDPR ops (smoke)"} - {id: mgmt.tool_management.list.happy_path, module: mgmt, tier: P2, surface: api, assertions: [happy_path], source: "tool_management_endpoints.py", rationale: "Tool inventory (smoke)"} diff --git a/tests/e2e/management/test_model_list_wildcard_routes_e2e.py b/tests/e2e/management/test_model_list_wildcard_routes_e2e.py new file mode 100644 index 00000000000..5aa81821d59 --- /dev/null +++ b/tests/e2e/management/test_model_list_wildcard_routes_e2e.py @@ -0,0 +1,95 @@ +"""Live e2e: `general_settings.model_list_return_wildcard_routes`, the proxy-wide +default for whether GET /v1/models lists a wildcard route such as `openai/*` next to +the models it expands to. + +The setting is written through /config/field/update, the route behind the admin UI's +General Settings toggle, and deleted on teardown so the shared proxy goes back to +leaving wildcard routes out. Every other listing the suites run passes +return_wildcard_routes explicitly, so the window with the setting on changes none of +them. The wildcard deployment gets a unique prefix instead of `openai/*`, which would +claim every `openai/...` request the other suites send. +""" + +from __future__ import annotations + +from typing import Final + +import pytest +from pydantic import BaseModel + +from e2e_config import unique_marker +from e2e_http import NoBody, Success, unwrap +from lifecycle import ResourceManager +from management_client import ManagementClient +from models import LiteLLMParamsBody, ModelsListParams, ModelsListResponse + +pytestmark = pytest.mark.e2e + +_SETTING: Final = "model_list_return_wildcard_routes" +_DUMMY_API_KEY: Final = "e2e-dummy-key" + + +class ConfigFieldUpdateBody(BaseModel): + field_name: str + field_value: bool + config_type: str = "general_settings" + + +class ConfigFieldDeleteBody(BaseModel): + field_name: str + config_type: str = "general_settings" + + +def _write_setting(client: ManagementClient, enabled: bool) -> None: + _ = unwrap( + client.proxy.transport.post( + "/config/field/update", + headers=client.proxy.transport.master, + json=ConfigFieldUpdateBody(field_name=_SETTING, field_value=enabled), + response_type=NoBody, + ) + ) + + +def _delete_setting(client: ManagementClient) -> None: + _ = unwrap( + client.proxy.transport.post( + "/config/field/delete", + headers=client.proxy.transport.master, + json=ConfigFieldDeleteBody(field_name=_SETTING), + response_type=NoBody, + ) + ) + + +def _await_listing(client: ManagementClient, pattern: str, query: BaseModel, *, listed: bool) -> None: + """Poll /v1/models under `query` on every replica until each one lists `pattern`, + or each one leaves it out, per `listed`.""" + _ = client.proxy.read_back_everywhere( + "/v1/models", + params=query, + response_type=ModelsListResponse, + converged=lambda result: ( + isinstance(result, Success) and any(entry.id == pattern for entry in result.data.data) is listed + ), + ) + + +class TestModelListWildcardRoutesSetting: + @pytest.mark.covers("mgmt.general_settings.model_list_return_wildcard_routes.persists") + def test_setting_lists_wildcard_route_unless_request_opts_out( + self, client: ManagementClient, resources: ResourceManager + ) -> None: + pattern = f"e2e-wildcard-{unique_marker()}/*" + model_id = client.proxy.create_model(pattern, LiteLLMParamsBody(model="openai/*", api_key=_DUMMY_API_KEY)) + resources.defer(lambda: client.proxy.delete_model(model_id)) + + _await_listing(client, pattern, NoBody(), listed=False) + + resources.defer(lambda: _delete_setting(client)) + _write_setting(client, True) + _await_listing(client, pattern, NoBody(), listed=True) + _await_listing(client, pattern, ModelsListParams(return_wildcard_routes=False), listed=False) + + _delete_setting(client) + _await_listing(client, pattern, NoBody(), listed=False) diff --git a/tests/unit/proxy/test_proxy_server.py b/tests/unit/proxy/test_proxy_server.py index 8c676e97c99..352b12f0f50 100644 --- a/tests/unit/proxy/test_proxy_server.py +++ b/tests/unit/proxy/test_proxy_server.py @@ -3169,3 +3169,36 @@ async def test_model_list_leaves_out_wildcard_routes_unless_enabled( assert "gpt-4o" in listed, listed assert "openai/*" not in listed, listed + + +@pytest.mark.asyncio +async def test_model_list_return_wildcard_routes_is_an_admin_ui_toggle(openai_wildcard_router, monkeypatch): + stored_general_settings: Final = {"model_list_return_wildcard_routes": True} + config_table: Final = MagicMock() + config_table.find_first = AsyncMock( + side_effect=lambda where: ( + MagicMock(param_value=stored_general_settings) if where["param_name"] == "general_settings" else None + ) + ) + ui_prisma_client: Final = MagicMock() + ui_prisma_client.db.litellm_config = config_table + monkeypatch.setattr(litellm.proxy.proxy_server, "general_settings", {}) + monkeypatch.setattr(litellm.proxy.proxy_server, "proxy_config", ProxyConfig()) + await litellm.proxy.proxy_server.proxy_config._update_general_settings(stored_general_settings) + + monkeypatch.setattr(litellm.proxy.proxy_server, "prisma_client", ui_prisma_client) + ui_fields: Final = { + field.field_name: field + for field in await litellm.proxy.proxy_server.get_config_list( + config_type="general_settings", + user_api_key_dict=UserAPIKeyAuth(api_key="sk-test", user_role=LitellmUserRoles.PROXY_ADMIN), + ) + } + monkeypatch.setattr(litellm.proxy.proxy_server, "prisma_client", None) + response: Final = await litellm.proxy.proxy_server.model_list( + user_api_key_dict=UserAPIKeyAuth(api_key="sk-test", user_role=LitellmUserRoles.INTERNAL_USER) + ) + + toggle: Final = ui_fields["model_list_return_wildcard_routes"] + assert (toggle.field_type, toggle.field_value, toggle.stored_in_db) == ("Boolean", True, True) + assert "openai/*" in [model["id"] for model in response["data"]] From 7eb134a14c377a9c35433a8cc7b0087081b7c7f7 Mon Sep 17 00:00:00 2001 From: shrey kharbanda Date: Mon, 28 Sep 2026 13:47:23 -0700 Subject: [PATCH 3/6] fix(ui): send return_wildcard_routes=False from modelAvailableCall modelAvailableCall left the param off when it wanted wildcard routes hidden, which used to mean false. It now means the proxy's model_list_return_wildcard_routes default, so turning that on would have put openai/* into the dashboard's own pickers (playground, guardrails, policies, users). Sending False keeps them as they were on every proxy version. --- .../src/components/networking.test.ts | 31 +++++++++++++++++++ .../src/components/networking.tsx | 4 ++- 2 files changed, 34 insertions(+), 1 deletion(-) diff --git a/ui/litellm-dashboard/src/components/networking.test.ts b/ui/litellm-dashboard/src/components/networking.test.ts index b964231804e..1378b7b9eeb 100644 --- a/ui/litellm-dashboard/src/components/networking.test.ts +++ b/ui/litellm-dashboard/src/components/networking.test.ts @@ -157,6 +157,37 @@ describe("modelInfoCall", () => { }); }); +describe("modelAvailableCall", () => { + let currentFetch: typeof global.fetch; + + beforeEach(() => { + currentFetch = global.fetch; + }); + + afterEach(() => { + global.fetch = currentFetch; + }); + + // An omitted return_wildcard_routes takes the proxy's model_list_return_wildcard_routes + // default, so the dashboard's own pickers always send it. + it.each([ + [undefined, "False"], + [false, "False"], + [true, "True"], + ])("sends return_wildcard_routes=%s as %s", async (returnWildcardRoutes, sent) => { + const mockFetch = vi + .fn() + .mockResolvedValue({ ok: true, text: vi.fn().mockResolvedValue(JSON.stringify({ data: [] })) } as any); + global.fetch = mockFetch as any; + + await Networking.modelAvailableCall("token", "user", "Admin", returnWildcardRoutes); + + const parsed = new URL(mockFetch.mock.calls[0][0] as string, "http://example.com"); + expect(parsed.pathname).toBe("/models"); + expect(parsed.searchParams.get("return_wildcard_routes")).toBe(sent); + }); +}); + describe("daily activity helpers", () => { const startTime = new Date("2025-02-12T00:00:00.000Z"); const endTime = new Date("2025-02-19T00:00:00.000Z"); diff --git a/ui/litellm-dashboard/src/components/networking.tsx b/ui/litellm-dashboard/src/components/networking.tsx index aeb41687788..affe24187f4 100644 --- a/ui/litellm-dashboard/src/components/networking.tsx +++ b/ui/litellm-dashboard/src/components/networking.tsx @@ -1940,7 +1940,9 @@ export const modelAvailableCall = async ( accessToken, query: { include_model_access_groups: "True", - return_wildcard_routes: return_wildcard_routes === true ? "True" : undefined, + // Sent either way: an omitted value falls back to the proxy's + // general_settings.model_list_return_wildcard_routes, not to false. + return_wildcard_routes: return_wildcard_routes === true ? "True" : "False", only_model_access_groups: only_model_access_groups === true ? "True" : undefined, team_id: teamID || undefined, scope: scope || undefined, From 249a9184bd491f857daa954c1dade508137abfbf Mon Sep 17 00:00:00 2001 From: shrey kharbanda Date: Mon, 28 Sep 2026 14:06:19 -0700 Subject: [PATCH 4/6] fix(proxy): keep /cursor/v1/models off the wildcard routes default Cursor offers every id /cursor/v1/models returns and sends no query params, so with model_list_return_wildcard_routes on it would list openai/* with no way to opt out. Pin return_wildcard_routes=False there, as before the setting, and say which listings the setting leaves alone in its description. --- litellm/proxy/_types.py | 7 ++++--- litellm/proxy/response_api_endpoints/endpoints.py | 5 ++++- .../e2e/management/test_model_list_wildcard_routes_e2e.py | 6 +++--- .../proxy/response_api_endpoints/test_endpoints.py | 2 ++ ui/litellm-dashboard/src/lib/http/schema.d.ts | 2 +- 5 files changed, 14 insertions(+), 8 deletions(-) diff --git a/litellm/proxy/_types.py b/litellm/proxy/_types.py index 41a9341c6cf..f5837abd587 100644 --- a/litellm/proxy/_types.py +++ b/litellm/proxy/_types.py @@ -2859,9 +2859,10 @@ class ConfigGeneralSettings(LiteLLMPydanticObjectBase): model_list_return_wildcard_routes: bool | None = Field( None, description=( - "When true, `/models` lists wildcard routes such as `openai/*` next to the models they " - "expand to, for every caller, without needing `return_wildcard_routes=true` per request. " - "A request can still pass `return_wildcard_routes=false` to leave them out." + "When true, `/v1/models` lists wildcard routes such as `openai/*` next to the models they " + "expand to, without the caller passing `return_wildcard_routes=true`. A request that passes " + "`return_wildcard_routes=false` still leaves them out, as do the dashboard's model pickers " + "and `/cursor/v1/models`." ), ) alerting: list | None = Field( diff --git a/litellm/proxy/response_api_endpoints/endpoints.py b/litellm/proxy/response_api_endpoints/endpoints.py index c5d702ad65a..84221da1c79 100644 --- a/litellm/proxy/response_api_endpoints/endpoints.py +++ b/litellm/proxy/response_api_endpoints/endpoints.py @@ -482,7 +482,10 @@ async def cursor_model_list( """ from litellm.proxy.proxy_server import model_list - return await model_list(user_api_key_dict=user_api_key_dict) + # Cursor offers every listed id as a model and a wildcard route such as + # `openai/*` is not callable, so the proxy-wide model_list_return_wildcard_routes + # default is pinned off here; Cursor sends no query params to opt out itself. + return await model_list(user_api_key_dict=user_api_key_dict, return_wildcard_routes=False) @router.post( diff --git a/tests/e2e/management/test_model_list_wildcard_routes_e2e.py b/tests/e2e/management/test_model_list_wildcard_routes_e2e.py index 5aa81821d59..9a9cea73bce 100644 --- a/tests/e2e/management/test_model_list_wildcard_routes_e2e.py +++ b/tests/e2e/management/test_model_list_wildcard_routes_e2e.py @@ -4,9 +4,9 @@ the models it expands to. The setting is written through /config/field/update, the route behind the admin UI's General Settings toggle, and deleted on teardown so the shared proxy goes back to -leaving wildcard routes out. Every other listing the suites run passes -return_wildcard_routes explicitly, so the window with the setting on changes none of -them. The wildcard deployment gets a unique prefix instead of `openai/*`, which would +leaving wildcard routes out. The other listings the suites run either pass +return_wildcard_routes explicitly or only check that a named model is present, so the +window with the setting on changes none of them. The wildcard deployment gets a unique prefix instead of `openai/*`, which would claim every `openai/...` request the other suites send. """ diff --git a/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py b/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py index 656dc33e88c..cd470c8879f 100644 --- a/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py +++ b/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py @@ -1351,6 +1351,8 @@ def test_cursor_models_route_delegates_to_model_list(): assert response.status_code == 200, f"{path}: {response.text}" assert response.json() == model_payload assert mock_model_list.call_count == 2 + # Cursor lists every id it gets, so the proxy-wide wildcard default must not reach it + assert all(call.kwargs["return_wildcard_routes"] is False for call in mock_model_list.call_args_list) finally: app.dependency_overrides.pop(user_api_key_auth, None) diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index d0becba1ba1..4f82a475fb8 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -28389,7 +28389,7 @@ export interface components { model_list_healthy_only?: boolean | null; /** * Model List Return Wildcard Routes - * @description When true, `/models` lists wildcard routes such as `openai/*` next to the models they expand to, for every caller, without needing `return_wildcard_routes=true` per request. A request can still pass `return_wildcard_routes=false` to leave them out. + * @description When true, `/v1/models` lists wildcard routes such as `openai/*` next to the models they expand to, without the caller passing `return_wildcard_routes=true`. A request that passes `return_wildcard_routes=false` still leaves them out, as do the dashboard's model pickers and `/cursor/v1/models`. */ model_list_return_wildcard_routes?: boolean | null; /** From 5437b86a5e9f1e7669b3ec572d0b7498ead196be Mon Sep 17 00:00:00 2001 From: shrey kharbanda Date: Mon, 28 Sep 2026 15:24:08 -0700 Subject: [PATCH 5/6] fix(proxy): resolve listed wildcard ids in /v1/models/{id} and honor the UI's "true" /v1/models/{model_id} now follows the same wildcard rule as the listing, so any id /v1/models returns can be looked up by id, with the same return_wildcard_routes override. The setting also counts as on for the string "true", matching what the Admin UI switch shows. The e2e restores whatever the database held before the test instead of deleting the setting. --- litellm/proxy/proxy_server.py | 19 ++++--- .../proxy/response_api_endpoints/endpoints.py | 6 +-- .../test_model_list_wildcard_routes_e2e.py | 50 +++++++++++++++---- .../response_api_endpoints/test_endpoints.py | 5 +- tests/unit/proxy/test_proxy_server.py | 45 ++++++++++++++++- .../src/components/networking.test.ts | 2 - .../src/components/networking.tsx | 2 - ui/litellm-dashboard/src/lib/http/schema.d.ts | 8 ++- 8 files changed, 108 insertions(+), 29 deletions(-) diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index 9dd176cac15..f63b65e00d0 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -11342,6 +11342,16 @@ async def _deployment_hidden_by_listing_callbacks(deployment: Deployment, user_a return listed_name in await _names_hidden_by_listing_callbacks(user_api_key_dict, (listed_name,)) +def _includes_wildcard_routes(return_wildcard_routes: bool | None, settings: Mapping[str, object]) -> bool: + """Whether a listing or lookup includes wildcard routes such as `openai/*`. A request's + own `return_wildcard_routes` wins; without one, `model_list_return_wildcard_routes` + counts as on exactly when the admin UI's switch shows it on (`true` or `"true"`).""" + if return_wildcard_routes is not None: + return return_wildcard_routes + configured: Final = settings.get("model_list_return_wildcard_routes") + return configured is True or configured == "true" + + @router.get("/v1/models", dependencies=[Depends(user_api_key_auth)], tags=["model management"]) @router.get( "/models", dependencies=[Depends(user_api_key_auth)], tags=["model management"] @@ -11456,11 +11466,7 @@ async def model_list( hidden_names: Final = blocked_names | unhealthy_names - include_wildcard_routes: Final = ( - settings.get("model_list_return_wildcard_routes") is True - if return_wildcard_routes is None - else return_wildcard_routes - ) + include_wildcard_routes: Final = _includes_wildcard_routes(return_wildcard_routes, settings) # If scope=expand and user has admin privileges, return all proxy models if should_expand_scope: @@ -11609,6 +11615,7 @@ async def model_info( user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth), team_id: str | None = None, healthy_only: bool | None = False, + return_wildcard_routes: bool | None = None, ): """ Retrieve information about a specific model accessible to your API key. @@ -11643,7 +11650,7 @@ async def model_info( team_id=team_id, include_model_access_groups=False, only_model_access_groups=False, - return_wildcard_routes=False, + return_wildcard_routes=_includes_wildcard_routes(return_wildcard_routes, settings), user_api_key_cache=user_api_key_cache, ) diff --git a/litellm/proxy/response_api_endpoints/endpoints.py b/litellm/proxy/response_api_endpoints/endpoints.py index 84221da1c79..692615005a8 100644 --- a/litellm/proxy/response_api_endpoints/endpoints.py +++ b/litellm/proxy/response_api_endpoints/endpoints.py @@ -478,13 +478,11 @@ async def cursor_model_list( verify models via `GET {base}/models` (the OpenAI SDK contract). Without this route those requests fall through to the Cursor Cloud Agents passthrough, which demands a Cursor API key and 401s, so key verification silently fails before any - chat request is ever sent. Delegates to the standard `/v1/models` handler. + chat request is ever sent. Delegates to the standard `/v1/models` handler with + wildcard routes left out, since Cursor offers every listed id as a callable model. """ from litellm.proxy.proxy_server import model_list - # Cursor offers every listed id as a model and a wildcard route such as - # `openai/*` is not callable, so the proxy-wide model_list_return_wildcard_routes - # default is pinned off here; Cursor sends no query params to opt out itself. return await model_list(user_api_key_dict=user_api_key_dict, return_wildcard_routes=False) diff --git a/tests/e2e/management/test_model_list_wildcard_routes_e2e.py b/tests/e2e/management/test_model_list_wildcard_routes_e2e.py index 9a9cea73bce..26f754ba421 100644 --- a/tests/e2e/management/test_model_list_wildcard_routes_e2e.py +++ b/tests/e2e/management/test_model_list_wildcard_routes_e2e.py @@ -3,8 +3,8 @@ default for whether GET /v1/models lists a wildcard route such as `openai/*` nex the models it expands to. The setting is written through /config/field/update, the route behind the admin UI's -General Settings toggle, and deleted on teardown so the shared proxy goes back to -leaving wildcard routes out. The other listings the suites run either pass +General Settings toggle, and teardown puts back whatever the database held before, so +the shared proxy ends the test the way it started. The other listings the suites run either pass return_wildcard_routes explicitly or only check that a named model is present, so the window with the setting on changes none of them. The wildcard deployment gets a unique prefix instead of `openai/*`, which would claim every `openai/...` request the other suites send. @@ -12,16 +12,16 @@ claim every `openai/...` request the other suites send. from __future__ import annotations -from typing import Final +from typing import Final, Literal import pytest -from pydantic import BaseModel +from pydantic import BaseModel, ConfigDict, JsonValue, RootModel from e2e_config import unique_marker from e2e_http import NoBody, Success, unwrap from lifecycle import ResourceManager from management_client import ManagementClient -from models import LiteLLMParamsBody, ModelsListParams, ModelsListResponse +from models import ConfigListParams, LiteLLMParamsBody, ModelsListParams, ModelsListResponse pytestmark = pytest.mark.e2e @@ -31,7 +31,7 @@ _DUMMY_API_KEY: Final = "e2e-dummy-key" class ConfigFieldUpdateBody(BaseModel): field_name: str - field_value: bool + field_value: JsonValue config_type: str = "general_settings" @@ -40,12 +40,23 @@ class ConfigFieldDeleteBody(BaseModel): config_type: str = "general_settings" -def _write_setting(client: ManagementClient, enabled: bool) -> None: +class StoredField(BaseModel): + model_config = ConfigDict(extra="ignore") + field_name: str + field_value: JsonValue = None + source: Literal["config", "db", "env", "default", "unset"] = "unset" + + +class StoredFieldList(RootModel[tuple[StoredField, ...]]): + pass + + +def _write_setting(client: ManagementClient, value: JsonValue) -> None: _ = unwrap( client.proxy.transport.post( "/config/field/update", headers=client.proxy.transport.master, - json=ConfigFieldUpdateBody(field_name=_SETTING, field_value=enabled), + json=ConfigFieldUpdateBody(field_name=_SETTING, field_value=value), response_type=NoBody, ) ) @@ -62,6 +73,25 @@ def _delete_setting(client: ManagementClient) -> None: ) +def _stored_setting(client: ManagementClient) -> StoredField | None: + fields: Final = unwrap( + client.proxy.transport.get( + "/config/list", + headers=client.proxy.transport.master, + params=ConfigListParams(config_type="general_settings"), + response_type=StoredFieldList, + ) + ).root + return next((field for field in fields if field.field_name == _SETTING and field.source == "db"), None) + + +def _restore_setting(client: ManagementClient, stored: StoredField | None) -> None: + if stored is None: + _delete_setting(client) + return + _write_setting(client, stored.field_value) + + def _await_listing(client: ManagementClient, pattern: str, query: BaseModel, *, listed: bool) -> None: """Poll /v1/models under `query` on every replica until each one lists `pattern`, or each one leaves it out, per `listed`.""" @@ -84,9 +114,11 @@ class TestModelListWildcardRoutesSetting: model_id = client.proxy.create_model(pattern, LiteLLMParamsBody(model="openai/*", api_key=_DUMMY_API_KEY)) resources.defer(lambda: client.proxy.delete_model(model_id)) + stored: Final = _stored_setting(client) + resources.defer(lambda: _restore_setting(client, stored)) + _write_setting(client, False) _await_listing(client, pattern, NoBody(), listed=False) - resources.defer(lambda: _delete_setting(client)) _write_setting(client, True) _await_listing(client, pattern, NoBody(), listed=True) _await_listing(client, pattern, ModelsListParams(return_wildcard_routes=False), listed=False) diff --git a/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py b/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py index cd470c8879f..f789ed8efc6 100644 --- a/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py +++ b/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py @@ -1351,8 +1351,9 @@ def test_cursor_models_route_delegates_to_model_list(): assert response.status_code == 200, f"{path}: {response.text}" assert response.json() == model_payload assert mock_model_list.call_count == 2 - # Cursor lists every id it gets, so the proxy-wide wildcard default must not reach it - assert all(call.kwargs["return_wildcard_routes"] is False for call in mock_model_list.call_args_list) + assert all(call.kwargs["return_wildcard_routes"] is False for call in mock_model_list.call_args_list), ( + "Cursor offers every listed id, so model_list_return_wildcard_routes must not reach it" + ) finally: app.dependency_overrides.pop(user_api_key_auth, None) diff --git a/tests/unit/proxy/test_proxy_server.py b/tests/unit/proxy/test_proxy_server.py index 352b12f0f50..40efdff8295 100644 --- a/tests/unit/proxy/test_proxy_server.py +++ b/tests/unit/proxy/test_proxy_server.py @@ -3138,16 +3138,19 @@ async def _listed_model_ids( return [model["id"] for model in response["data"]] +@pytest.mark.parametrize("setting", [True, "true"]) @pytest.mark.parametrize( "user_role, query", [(LitellmUserRoles.INTERNAL_USER, {}), (LitellmUserRoles.PROXY_ADMIN, {"scope": "expand"})], ) @pytest.mark.asyncio async def test_model_list_return_wildcard_routes_setting_matches_query_param( - openai_wildcard_router, monkeypatch, user_role, query + openai_wildcard_router, monkeypatch, user_role, query, setting ): requested = await _listed_model_ids(monkeypatch, {}, user_role, return_wildcard_routes=True, **query) - from_setting = await _listed_model_ids(monkeypatch, {"model_list_return_wildcard_routes": True}, user_role, **query) + from_setting = await _listed_model_ids( + monkeypatch, {"model_list_return_wildcard_routes": setting}, user_role, **query + ) assert "openai/*" in from_setting, from_setting assert from_setting == requested @@ -3158,6 +3161,8 @@ async def test_model_list_return_wildcard_routes_setting_matches_query_param( [ ({}, {}), ({"model_list_return_wildcard_routes": "false"}, {}), + ({"model_list_return_wildcard_routes": "True"}, {}), + ({"model_list_return_wildcard_routes": 1}, {}), ({"model_list_return_wildcard_routes": True}, {"return_wildcard_routes": False}), ], ) @@ -3171,6 +3176,42 @@ async def test_model_list_leaves_out_wildcard_routes_unless_enabled( assert "openai/*" not in listed, listed +async def _model_info_resolves(model_id: str, **query: bool) -> bool: + from fastapi import HTTPException + + try: + response = await litellm.proxy.proxy_server.model_info( + model_id=model_id, + user_api_key_dict=UserAPIKeyAuth(api_key="sk-test", user_role=LitellmUserRoles.INTERNAL_USER), + **query, + ) + except HTTPException as e: + assert e.status_code == 404, e.detail + return False + return response["id"] == model_id + + +@pytest.mark.parametrize( + "general_settings, query, wildcard_listed", + [ + ({}, {}, False), + ({"model_list_return_wildcard_routes": True}, {}, True), + ({"model_list_return_wildcard_routes": "true"}, {}, True), + ({"model_list_return_wildcard_routes": True}, {"return_wildcard_routes": False}, False), + ({}, {"return_wildcard_routes": True}, True), + ], +) +@pytest.mark.asyncio +async def test_model_info_resolves_exactly_the_ids_model_list_lists( + openai_wildcard_router, monkeypatch, general_settings, query, wildcard_listed +): + listed = await _listed_model_ids(monkeypatch, general_settings, LitellmUserRoles.INTERNAL_USER, **query) + resolvable = [model_id for model_id in ("gpt-4o", "openai/*") if await _model_info_resolves(model_id, **query)] + + assert ("openai/*" in listed) is wildcard_listed, listed + assert resolvable == [model_id for model_id in ("gpt-4o", "openai/*") if model_id in listed], (listed, resolvable) + + @pytest.mark.asyncio async def test_model_list_return_wildcard_routes_is_an_admin_ui_toggle(openai_wildcard_router, monkeypatch): stored_general_settings: Final = {"model_list_return_wildcard_routes": True} diff --git a/ui/litellm-dashboard/src/components/networking.test.ts b/ui/litellm-dashboard/src/components/networking.test.ts index 1378b7b9eeb..18488e21555 100644 --- a/ui/litellm-dashboard/src/components/networking.test.ts +++ b/ui/litellm-dashboard/src/components/networking.test.ts @@ -168,8 +168,6 @@ describe("modelAvailableCall", () => { global.fetch = currentFetch; }); - // An omitted return_wildcard_routes takes the proxy's model_list_return_wildcard_routes - // default, so the dashboard's own pickers always send it. it.each([ [undefined, "False"], [false, "False"], diff --git a/ui/litellm-dashboard/src/components/networking.tsx b/ui/litellm-dashboard/src/components/networking.tsx index affe24187f4..0c9251e5ac8 100644 --- a/ui/litellm-dashboard/src/components/networking.tsx +++ b/ui/litellm-dashboard/src/components/networking.tsx @@ -1940,8 +1940,6 @@ export const modelAvailableCall = async ( accessToken, query: { include_model_access_groups: "True", - // Sent either way: an omitted value falls back to the proxy's - // general_settings.model_list_return_wildcard_routes, not to false. return_wildcard_routes: return_wildcard_routes === true ? "True" : "False", only_model_access_groups: only_model_access_groups === true ? "True" : undefined, team_id: teamID || undefined, diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index 4f82a475fb8..79089fcef75 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -3761,7 +3761,8 @@ export interface paths { * verify models via `GET {base}/models` (the OpenAI SDK contract). Without this * route those requests fall through to the Cursor Cloud Agents passthrough, which * demands a Cursor API key and 401s, so key verification silently fails before any - * chat request is ever sent. Delegates to the standard `/v1/models` handler. + * chat request is ever sent. Delegates to the standard `/v1/models` handler with + * wildcard routes left out, since Cursor offers every listed id as a callable model. */ get: operations["cursor_model_list_cursor_models_get"]; put?: never; @@ -3787,7 +3788,8 @@ export interface paths { * verify models via `GET {base}/models` (the OpenAI SDK contract). Without this * route those requests fall through to the Cursor Cloud Agents passthrough, which * demands a Cursor API key and 401s, so key verification silently fails before any - * chat request is ever sent. Delegates to the standard `/v1/models` handler. + * chat request is ever sent. Delegates to the standard `/v1/models` handler with + * wildcard routes left out, since Cursor offers every listed id as a callable model. */ get: operations["cursor_model_list_cursor_v1_models_get"]; put?: never; @@ -59947,6 +59949,7 @@ export interface operations { query?: { team_id?: string | null; healthy_only?: boolean | null; + return_wildcard_routes?: boolean | null; }; header?: never; path: { @@ -73252,6 +73255,7 @@ export interface operations { query?: { team_id?: string | null; healthy_only?: boolean | null; + return_wildcard_routes?: boolean | null; }; header?: never; path: { From 47b80b6a3dc5d3e1c9d7bb23352cf3d5ba21793e Mon Sep 17 00:00:00 2001 From: shrey kharbanda Date: Mon, 28 Sep 2026 15:33:56 -0700 Subject: [PATCH 6/6] chore(proxy): regenerate the lazy OpenAPI snapshot and drop an assert in except The Cursor model listing docstring changed, so the snapshot served for unloaded features is regenerated. The model_info test helper re-raises a non-404 instead of asserting inside the except block (PT017). --- litellm/proxy/_lazy_openapi_snapshot.json | 4 ++-- tests/unit/proxy/test_proxy_server.py | 3 ++- 2 files changed, 4 insertions(+), 3 deletions(-) diff --git a/litellm/proxy/_lazy_openapi_snapshot.json b/litellm/proxy/_lazy_openapi_snapshot.json index db57aa4f046..b7365488ffb 100644 --- a/litellm/proxy/_lazy_openapi_snapshot.json +++ b/litellm/proxy/_lazy_openapi_snapshot.json @@ -25220,7 +25220,7 @@ }, "/cursor/models": { "get": { - "description": "OpenAI-compatible model listing for the Cursor BYOK base URL.\n\nClients pointed at `/cursor` as an OpenAI-compatible base URL resolve and\nverify models via `GET {base}/models` (the OpenAI SDK contract). Without this\nroute those requests fall through to the Cursor Cloud Agents passthrough, which\ndemands a Cursor API key and 401s, so key verification silently fails before any\nchat request is ever sent. Delegates to the standard `/v1/models` handler.", + "description": "OpenAI-compatible model listing for the Cursor BYOK base URL.\n\nClients pointed at `/cursor` as an OpenAI-compatible base URL resolve and\nverify models via `GET {base}/models` (the OpenAI SDK contract). Without this\nroute those requests fall through to the Cursor Cloud Agents passthrough, which\ndemands a Cursor API key and 401s, so key verification silently fails before any\nchat request is ever sent. Delegates to the standard `/v1/models` handler with\nwildcard routes left out, since Cursor offers every listed id as a callable model.", "operationId": "cursor_model_list_cursor_models_get", "responses": { "200": { @@ -25245,7 +25245,7 @@ }, "/cursor/v1/models": { "get": { - "description": "OpenAI-compatible model listing for the Cursor BYOK base URL.\n\nClients pointed at `/cursor` as an OpenAI-compatible base URL resolve and\nverify models via `GET {base}/models` (the OpenAI SDK contract). Without this\nroute those requests fall through to the Cursor Cloud Agents passthrough, which\ndemands a Cursor API key and 401s, so key verification silently fails before any\nchat request is ever sent. Delegates to the standard `/v1/models` handler.", + "description": "OpenAI-compatible model listing for the Cursor BYOK base URL.\n\nClients pointed at `/cursor` as an OpenAI-compatible base URL resolve and\nverify models via `GET {base}/models` (the OpenAI SDK contract). Without this\nroute those requests fall through to the Cursor Cloud Agents passthrough, which\ndemands a Cursor API key and 401s, so key verification silently fails before any\nchat request is ever sent. Delegates to the standard `/v1/models` handler with\nwildcard routes left out, since Cursor offers every listed id as a callable model.", "operationId": "cursor_model_list_cursor_v1_models_get", "responses": { "200": { diff --git a/tests/unit/proxy/test_proxy_server.py b/tests/unit/proxy/test_proxy_server.py index 40efdff8295..0284cb2b3d4 100644 --- a/tests/unit/proxy/test_proxy_server.py +++ b/tests/unit/proxy/test_proxy_server.py @@ -3186,7 +3186,8 @@ async def _model_info_resolves(model_id: str, **query: bool) -> bool: **query, ) except HTTPException as e: - assert e.status_code == 404, e.detail + if e.status_code != 404: + raise return False return response["id"] == model_id