fix(proxy): resolve listed wildcard ids in /v1/models/{id} and honor the UI's "true"

/v1/models/{model_id} now follows the same wildcard rule as the listing, so any
id /v1/models returns can be looked up by id, with the same
return_wildcard_routes override. The setting also counts as on for the string
"true", matching what the Admin UI switch shows. The e2e restores whatever the
database held before the test instead of deleting the setting.
This commit is contained in:
shrey kharbanda 2026-09-28 15:24:08 -07:00
parent 249a9184bd
commit 5437b86a5e
8 changed files with 108 additions and 29 deletions

View file

@ -11342,6 +11342,16 @@ async def _deployment_hidden_by_listing_callbacks(deployment: Deployment, user_a
return listed_name in await _names_hidden_by_listing_callbacks(user_api_key_dict, (listed_name,))
def _includes_wildcard_routes(return_wildcard_routes: bool | None, settings: Mapping[str, object]) -> bool:
"""Whether a listing or lookup includes wildcard routes such as `openai/*`. A request's
own `return_wildcard_routes` wins; without one, `model_list_return_wildcard_routes`
counts as on exactly when the admin UI's switch shows it on (`true` or `"true"`)."""
if return_wildcard_routes is not None:
return return_wildcard_routes
configured: Final = settings.get("model_list_return_wildcard_routes")
return configured is True or configured == "true"
@router.get("/v1/models", dependencies=[Depends(user_api_key_auth)], tags=["model management"])
@router.get(
"/models", dependencies=[Depends(user_api_key_auth)], tags=["model management"]
@ -11456,11 +11466,7 @@ async def model_list(
hidden_names: Final = blocked_names | unhealthy_names
include_wildcard_routes: Final = (
settings.get("model_list_return_wildcard_routes") is True
if return_wildcard_routes is None
else return_wildcard_routes
)
include_wildcard_routes: Final = _includes_wildcard_routes(return_wildcard_routes, settings)
# If scope=expand and user has admin privileges, return all proxy models
if should_expand_scope:
@ -11609,6 +11615,7 @@ async def model_info(
user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth),
team_id: str | None = None,
healthy_only: bool | None = False,
return_wildcard_routes: bool | None = None,
):
"""
Retrieve information about a specific model accessible to your API key.
@ -11643,7 +11650,7 @@ async def model_info(
team_id=team_id,
include_model_access_groups=False,
only_model_access_groups=False,
return_wildcard_routes=False,
return_wildcard_routes=_includes_wildcard_routes(return_wildcard_routes, settings),
user_api_key_cache=user_api_key_cache,
)

View file

@ -478,13 +478,11 @@ async def cursor_model_list(
verify models via `GET {base}/models` (the OpenAI SDK contract). Without this
route those requests fall through to the Cursor Cloud Agents passthrough, which
demands a Cursor API key and 401s, so key verification silently fails before any
chat request is ever sent. Delegates to the standard `/v1/models` handler.
chat request is ever sent. Delegates to the standard `/v1/models` handler with
wildcard routes left out, since Cursor offers every listed id as a callable model.
"""
from litellm.proxy.proxy_server import model_list
# Cursor offers every listed id as a model and a wildcard route such as
# `openai/*` is not callable, so the proxy-wide model_list_return_wildcard_routes
# default is pinned off here; Cursor sends no query params to opt out itself.
return await model_list(user_api_key_dict=user_api_key_dict, return_wildcard_routes=False)

View file

@ -3,8 +3,8 @@ default for whether GET /v1/models lists a wildcard route such as `openai/*` nex
the models it expands to.
The setting is written through /config/field/update, the route behind the admin UI's
General Settings toggle, and deleted on teardown so the shared proxy goes back to
leaving wildcard routes out. The other listings the suites run either pass
General Settings toggle, and teardown puts back whatever the database held before, so
the shared proxy ends the test the way it started. The other listings the suites run either pass
return_wildcard_routes explicitly or only check that a named model is present, so the
window with the setting on changes none of them. The wildcard deployment gets a unique prefix instead of `openai/*`, which would
claim every `openai/...` request the other suites send.
@ -12,16 +12,16 @@ claim every `openai/...` request the other suites send.
from __future__ import annotations
from typing import Final
from typing import Final, Literal
import pytest
from pydantic import BaseModel
from pydantic import BaseModel, ConfigDict, JsonValue, RootModel
from e2e_config import unique_marker
from e2e_http import NoBody, Success, unwrap
from lifecycle import ResourceManager
from management_client import ManagementClient
from models import LiteLLMParamsBody, ModelsListParams, ModelsListResponse
from models import ConfigListParams, LiteLLMParamsBody, ModelsListParams, ModelsListResponse
pytestmark = pytest.mark.e2e
@ -31,7 +31,7 @@ _DUMMY_API_KEY: Final = "e2e-dummy-key"
class ConfigFieldUpdateBody(BaseModel):
field_name: str
field_value: bool
field_value: JsonValue
config_type: str = "general_settings"
@ -40,12 +40,23 @@ class ConfigFieldDeleteBody(BaseModel):
config_type: str = "general_settings"
def _write_setting(client: ManagementClient, enabled: bool) -> None:
class StoredField(BaseModel):
model_config = ConfigDict(extra="ignore")
field_name: str
field_value: JsonValue = None
source: Literal["config", "db", "env", "default", "unset"] = "unset"
class StoredFieldList(RootModel[tuple[StoredField, ...]]):
pass
def _write_setting(client: ManagementClient, value: JsonValue) -> None:
_ = unwrap(
client.proxy.transport.post(
"/config/field/update",
headers=client.proxy.transport.master,
json=ConfigFieldUpdateBody(field_name=_SETTING, field_value=enabled),
json=ConfigFieldUpdateBody(field_name=_SETTING, field_value=value),
response_type=NoBody,
)
)
@ -62,6 +73,25 @@ def _delete_setting(client: ManagementClient) -> None:
)
def _stored_setting(client: ManagementClient) -> StoredField | None:
fields: Final = unwrap(
client.proxy.transport.get(
"/config/list",
headers=client.proxy.transport.master,
params=ConfigListParams(config_type="general_settings"),
response_type=StoredFieldList,
)
).root
return next((field for field in fields if field.field_name == _SETTING and field.source == "db"), None)
def _restore_setting(client: ManagementClient, stored: StoredField | None) -> None:
if stored is None:
_delete_setting(client)
return
_write_setting(client, stored.field_value)
def _await_listing(client: ManagementClient, pattern: str, query: BaseModel, *, listed: bool) -> None:
"""Poll /v1/models under `query` on every replica until each one lists `pattern`,
or each one leaves it out, per `listed`."""
@ -84,9 +114,11 @@ class TestModelListWildcardRoutesSetting:
model_id = client.proxy.create_model(pattern, LiteLLMParamsBody(model="openai/*", api_key=_DUMMY_API_KEY))
resources.defer(lambda: client.proxy.delete_model(model_id))
stored: Final = _stored_setting(client)
resources.defer(lambda: _restore_setting(client, stored))
_write_setting(client, False)
_await_listing(client, pattern, NoBody(), listed=False)
resources.defer(lambda: _delete_setting(client))
_write_setting(client, True)
_await_listing(client, pattern, NoBody(), listed=True)
_await_listing(client, pattern, ModelsListParams(return_wildcard_routes=False), listed=False)

View file

@ -1351,8 +1351,9 @@ def test_cursor_models_route_delegates_to_model_list():
assert response.status_code == 200, f"{path}: {response.text}"
assert response.json() == model_payload
assert mock_model_list.call_count == 2
# Cursor lists every id it gets, so the proxy-wide wildcard default must not reach it
assert all(call.kwargs["return_wildcard_routes"] is False for call in mock_model_list.call_args_list)
assert all(call.kwargs["return_wildcard_routes"] is False for call in mock_model_list.call_args_list), (
"Cursor offers every listed id, so model_list_return_wildcard_routes must not reach it"
)
finally:
app.dependency_overrides.pop(user_api_key_auth, None)

View file

@ -3138,16 +3138,19 @@ async def _listed_model_ids(
return [model["id"] for model in response["data"]]
@pytest.mark.parametrize("setting", [True, "true"])
@pytest.mark.parametrize(
"user_role, query",
[(LitellmUserRoles.INTERNAL_USER, {}), (LitellmUserRoles.PROXY_ADMIN, {"scope": "expand"})],
)
@pytest.mark.asyncio
async def test_model_list_return_wildcard_routes_setting_matches_query_param(
openai_wildcard_router, monkeypatch, user_role, query
openai_wildcard_router, monkeypatch, user_role, query, setting
):
requested = await _listed_model_ids(monkeypatch, {}, user_role, return_wildcard_routes=True, **query)
from_setting = await _listed_model_ids(monkeypatch, {"model_list_return_wildcard_routes": True}, user_role, **query)
from_setting = await _listed_model_ids(
monkeypatch, {"model_list_return_wildcard_routes": setting}, user_role, **query
)
assert "openai/*" in from_setting, from_setting
assert from_setting == requested
@ -3158,6 +3161,8 @@ async def test_model_list_return_wildcard_routes_setting_matches_query_param(
[
({}, {}),
({"model_list_return_wildcard_routes": "false"}, {}),
({"model_list_return_wildcard_routes": "True"}, {}),
({"model_list_return_wildcard_routes": 1}, {}),
({"model_list_return_wildcard_routes": True}, {"return_wildcard_routes": False}),
],
)
@ -3171,6 +3176,42 @@ async def test_model_list_leaves_out_wildcard_routes_unless_enabled(
assert "openai/*" not in listed, listed
async def _model_info_resolves(model_id: str, **query: bool) -> bool:
from fastapi import HTTPException
try:
response = await litellm.proxy.proxy_server.model_info(
model_id=model_id,
user_api_key_dict=UserAPIKeyAuth(api_key="sk-test", user_role=LitellmUserRoles.INTERNAL_USER),
**query,
)
except HTTPException as e:
assert e.status_code == 404, e.detail
return False
return response["id"] == model_id
@pytest.mark.parametrize(
"general_settings, query, wildcard_listed",
[
({}, {}, False),
({"model_list_return_wildcard_routes": True}, {}, True),
({"model_list_return_wildcard_routes": "true"}, {}, True),
({"model_list_return_wildcard_routes": True}, {"return_wildcard_routes": False}, False),
({}, {"return_wildcard_routes": True}, True),
],
)
@pytest.mark.asyncio
async def test_model_info_resolves_exactly_the_ids_model_list_lists(
openai_wildcard_router, monkeypatch, general_settings, query, wildcard_listed
):
listed = await _listed_model_ids(monkeypatch, general_settings, LitellmUserRoles.INTERNAL_USER, **query)
resolvable = [model_id for model_id in ("gpt-4o", "openai/*") if await _model_info_resolves(model_id, **query)]
assert ("openai/*" in listed) is wildcard_listed, listed
assert resolvable == [model_id for model_id in ("gpt-4o", "openai/*") if model_id in listed], (listed, resolvable)
@pytest.mark.asyncio
async def test_model_list_return_wildcard_routes_is_an_admin_ui_toggle(openai_wildcard_router, monkeypatch):
stored_general_settings: Final = {"model_list_return_wildcard_routes": True}

View file

@ -168,8 +168,6 @@ describe("modelAvailableCall", () => {
global.fetch = currentFetch;
});
// An omitted return_wildcard_routes takes the proxy's model_list_return_wildcard_routes
// default, so the dashboard's own pickers always send it.
it.each([
[undefined, "False"],
[false, "False"],

View file

@ -1940,8 +1940,6 @@ export const modelAvailableCall = async (
accessToken,
query: {
include_model_access_groups: "True",
// Sent either way: an omitted value falls back to the proxy's
// general_settings.model_list_return_wildcard_routes, not to false.
return_wildcard_routes: return_wildcard_routes === true ? "True" : "False",
only_model_access_groups: only_model_access_groups === true ? "True" : undefined,
team_id: teamID || undefined,

View file

@ -3761,7 +3761,8 @@ export interface paths {
* verify models via `GET {base}/models` (the OpenAI SDK contract). Without this
* route those requests fall through to the Cursor Cloud Agents passthrough, which
* demands a Cursor API key and 401s, so key verification silently fails before any
* chat request is ever sent. Delegates to the standard `/v1/models` handler.
* chat request is ever sent. Delegates to the standard `/v1/models` handler with
* wildcard routes left out, since Cursor offers every listed id as a callable model.
*/
get: operations["cursor_model_list_cursor_models_get"];
put?: never;
@ -3787,7 +3788,8 @@ export interface paths {
* verify models via `GET {base}/models` (the OpenAI SDK contract). Without this
* route those requests fall through to the Cursor Cloud Agents passthrough, which
* demands a Cursor API key and 401s, so key verification silently fails before any
* chat request is ever sent. Delegates to the standard `/v1/models` handler.
* chat request is ever sent. Delegates to the standard `/v1/models` handler with
* wildcard routes left out, since Cursor offers every listed id as a callable model.
*/
get: operations["cursor_model_list_cursor_v1_models_get"];
put?: never;
@ -59947,6 +59949,7 @@ export interface operations {
query?: {
team_id?: string | null;
healthy_only?: boolean | null;
return_wildcard_routes?: boolean | null;
};
header?: never;
path: {
@ -73252,6 +73255,7 @@ export interface operations {
query?: {
team_id?: string | null;
healthy_only?: boolean | null;
return_wildcard_routes?: boolean | null;
};
header?: never;
path: {