mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-06 02:48:13 +00:00
fix(proxy): resolve listed wildcard ids in /v1/models/{id} and honor the UI's "true"
/v1/models/{model_id} now follows the same wildcard rule as the listing, so any
id /v1/models returns can be looked up by id, with the same
return_wildcard_routes override. The setting also counts as on for the string
"true", matching what the Admin UI switch shows. The e2e restores whatever the
database held before the test instead of deleting the setting.
This commit is contained in:
parent
249a9184bd
commit
5437b86a5e
8 changed files with 108 additions and 29 deletions
|
|
@ -11342,6 +11342,16 @@ async def _deployment_hidden_by_listing_callbacks(deployment: Deployment, user_a
|
|||
return listed_name in await _names_hidden_by_listing_callbacks(user_api_key_dict, (listed_name,))
|
||||
|
||||
|
||||
def _includes_wildcard_routes(return_wildcard_routes: bool | None, settings: Mapping[str, object]) -> bool:
|
||||
"""Whether a listing or lookup includes wildcard routes such as `openai/*`. A request's
|
||||
own `return_wildcard_routes` wins; without one, `model_list_return_wildcard_routes`
|
||||
counts as on exactly when the admin UI's switch shows it on (`true` or `"true"`)."""
|
||||
if return_wildcard_routes is not None:
|
||||
return return_wildcard_routes
|
||||
configured: Final = settings.get("model_list_return_wildcard_routes")
|
||||
return configured is True or configured == "true"
|
||||
|
||||
|
||||
@router.get("/v1/models", dependencies=[Depends(user_api_key_auth)], tags=["model management"])
|
||||
@router.get(
|
||||
"/models", dependencies=[Depends(user_api_key_auth)], tags=["model management"]
|
||||
|
|
@ -11456,11 +11466,7 @@ async def model_list(
|
|||
|
||||
hidden_names: Final = blocked_names | unhealthy_names
|
||||
|
||||
include_wildcard_routes: Final = (
|
||||
settings.get("model_list_return_wildcard_routes") is True
|
||||
if return_wildcard_routes is None
|
||||
else return_wildcard_routes
|
||||
)
|
||||
include_wildcard_routes: Final = _includes_wildcard_routes(return_wildcard_routes, settings)
|
||||
|
||||
# If scope=expand and user has admin privileges, return all proxy models
|
||||
if should_expand_scope:
|
||||
|
|
@ -11609,6 +11615,7 @@ async def model_info(
|
|||
user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth),
|
||||
team_id: str | None = None,
|
||||
healthy_only: bool | None = False,
|
||||
return_wildcard_routes: bool | None = None,
|
||||
):
|
||||
"""
|
||||
Retrieve information about a specific model accessible to your API key.
|
||||
|
|
@ -11643,7 +11650,7 @@ async def model_info(
|
|||
team_id=team_id,
|
||||
include_model_access_groups=False,
|
||||
only_model_access_groups=False,
|
||||
return_wildcard_routes=False,
|
||||
return_wildcard_routes=_includes_wildcard_routes(return_wildcard_routes, settings),
|
||||
user_api_key_cache=user_api_key_cache,
|
||||
)
|
||||
|
||||
|
|
|
|||
|
|
@ -478,13 +478,11 @@ async def cursor_model_list(
|
|||
verify models via `GET {base}/models` (the OpenAI SDK contract). Without this
|
||||
route those requests fall through to the Cursor Cloud Agents passthrough, which
|
||||
demands a Cursor API key and 401s, so key verification silently fails before any
|
||||
chat request is ever sent. Delegates to the standard `/v1/models` handler.
|
||||
chat request is ever sent. Delegates to the standard `/v1/models` handler with
|
||||
wildcard routes left out, since Cursor offers every listed id as a callable model.
|
||||
"""
|
||||
from litellm.proxy.proxy_server import model_list
|
||||
|
||||
# Cursor offers every listed id as a model and a wildcard route such as
|
||||
# `openai/*` is not callable, so the proxy-wide model_list_return_wildcard_routes
|
||||
# default is pinned off here; Cursor sends no query params to opt out itself.
|
||||
return await model_list(user_api_key_dict=user_api_key_dict, return_wildcard_routes=False)
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -3,8 +3,8 @@ default for whether GET /v1/models lists a wildcard route such as `openai/*` nex
|
|||
the models it expands to.
|
||||
|
||||
The setting is written through /config/field/update, the route behind the admin UI's
|
||||
General Settings toggle, and deleted on teardown so the shared proxy goes back to
|
||||
leaving wildcard routes out. The other listings the suites run either pass
|
||||
General Settings toggle, and teardown puts back whatever the database held before, so
|
||||
the shared proxy ends the test the way it started. The other listings the suites run either pass
|
||||
return_wildcard_routes explicitly or only check that a named model is present, so the
|
||||
window with the setting on changes none of them. The wildcard deployment gets a unique prefix instead of `openai/*`, which would
|
||||
claim every `openai/...` request the other suites send.
|
||||
|
|
@ -12,16 +12,16 @@ claim every `openai/...` request the other suites send.
|
|||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Final
|
||||
from typing import Final, Literal
|
||||
|
||||
import pytest
|
||||
from pydantic import BaseModel
|
||||
from pydantic import BaseModel, ConfigDict, JsonValue, RootModel
|
||||
|
||||
from e2e_config import unique_marker
|
||||
from e2e_http import NoBody, Success, unwrap
|
||||
from lifecycle import ResourceManager
|
||||
from management_client import ManagementClient
|
||||
from models import LiteLLMParamsBody, ModelsListParams, ModelsListResponse
|
||||
from models import ConfigListParams, LiteLLMParamsBody, ModelsListParams, ModelsListResponse
|
||||
|
||||
pytestmark = pytest.mark.e2e
|
||||
|
||||
|
|
@ -31,7 +31,7 @@ _DUMMY_API_KEY: Final = "e2e-dummy-key"
|
|||
|
||||
class ConfigFieldUpdateBody(BaseModel):
|
||||
field_name: str
|
||||
field_value: bool
|
||||
field_value: JsonValue
|
||||
config_type: str = "general_settings"
|
||||
|
||||
|
||||
|
|
@ -40,12 +40,23 @@ class ConfigFieldDeleteBody(BaseModel):
|
|||
config_type: str = "general_settings"
|
||||
|
||||
|
||||
def _write_setting(client: ManagementClient, enabled: bool) -> None:
|
||||
class StoredField(BaseModel):
|
||||
model_config = ConfigDict(extra="ignore")
|
||||
field_name: str
|
||||
field_value: JsonValue = None
|
||||
source: Literal["config", "db", "env", "default", "unset"] = "unset"
|
||||
|
||||
|
||||
class StoredFieldList(RootModel[tuple[StoredField, ...]]):
|
||||
pass
|
||||
|
||||
|
||||
def _write_setting(client: ManagementClient, value: JsonValue) -> None:
|
||||
_ = unwrap(
|
||||
client.proxy.transport.post(
|
||||
"/config/field/update",
|
||||
headers=client.proxy.transport.master,
|
||||
json=ConfigFieldUpdateBody(field_name=_SETTING, field_value=enabled),
|
||||
json=ConfigFieldUpdateBody(field_name=_SETTING, field_value=value),
|
||||
response_type=NoBody,
|
||||
)
|
||||
)
|
||||
|
|
@ -62,6 +73,25 @@ def _delete_setting(client: ManagementClient) -> None:
|
|||
)
|
||||
|
||||
|
||||
def _stored_setting(client: ManagementClient) -> StoredField | None:
|
||||
fields: Final = unwrap(
|
||||
client.proxy.transport.get(
|
||||
"/config/list",
|
||||
headers=client.proxy.transport.master,
|
||||
params=ConfigListParams(config_type="general_settings"),
|
||||
response_type=StoredFieldList,
|
||||
)
|
||||
).root
|
||||
return next((field for field in fields if field.field_name == _SETTING and field.source == "db"), None)
|
||||
|
||||
|
||||
def _restore_setting(client: ManagementClient, stored: StoredField | None) -> None:
|
||||
if stored is None:
|
||||
_delete_setting(client)
|
||||
return
|
||||
_write_setting(client, stored.field_value)
|
||||
|
||||
|
||||
def _await_listing(client: ManagementClient, pattern: str, query: BaseModel, *, listed: bool) -> None:
|
||||
"""Poll /v1/models under `query` on every replica until each one lists `pattern`,
|
||||
or each one leaves it out, per `listed`."""
|
||||
|
|
@ -84,9 +114,11 @@ class TestModelListWildcardRoutesSetting:
|
|||
model_id = client.proxy.create_model(pattern, LiteLLMParamsBody(model="openai/*", api_key=_DUMMY_API_KEY))
|
||||
resources.defer(lambda: client.proxy.delete_model(model_id))
|
||||
|
||||
stored: Final = _stored_setting(client)
|
||||
resources.defer(lambda: _restore_setting(client, stored))
|
||||
_write_setting(client, False)
|
||||
_await_listing(client, pattern, NoBody(), listed=False)
|
||||
|
||||
resources.defer(lambda: _delete_setting(client))
|
||||
_write_setting(client, True)
|
||||
_await_listing(client, pattern, NoBody(), listed=True)
|
||||
_await_listing(client, pattern, ModelsListParams(return_wildcard_routes=False), listed=False)
|
||||
|
|
|
|||
|
|
@ -1351,8 +1351,9 @@ def test_cursor_models_route_delegates_to_model_list():
|
|||
assert response.status_code == 200, f"{path}: {response.text}"
|
||||
assert response.json() == model_payload
|
||||
assert mock_model_list.call_count == 2
|
||||
# Cursor lists every id it gets, so the proxy-wide wildcard default must not reach it
|
||||
assert all(call.kwargs["return_wildcard_routes"] is False for call in mock_model_list.call_args_list)
|
||||
assert all(call.kwargs["return_wildcard_routes"] is False for call in mock_model_list.call_args_list), (
|
||||
"Cursor offers every listed id, so model_list_return_wildcard_routes must not reach it"
|
||||
)
|
||||
finally:
|
||||
app.dependency_overrides.pop(user_api_key_auth, None)
|
||||
|
||||
|
|
|
|||
|
|
@ -3138,16 +3138,19 @@ async def _listed_model_ids(
|
|||
return [model["id"] for model in response["data"]]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("setting", [True, "true"])
|
||||
@pytest.mark.parametrize(
|
||||
"user_role, query",
|
||||
[(LitellmUserRoles.INTERNAL_USER, {}), (LitellmUserRoles.PROXY_ADMIN, {"scope": "expand"})],
|
||||
)
|
||||
@pytest.mark.asyncio
|
||||
async def test_model_list_return_wildcard_routes_setting_matches_query_param(
|
||||
openai_wildcard_router, monkeypatch, user_role, query
|
||||
openai_wildcard_router, monkeypatch, user_role, query, setting
|
||||
):
|
||||
requested = await _listed_model_ids(monkeypatch, {}, user_role, return_wildcard_routes=True, **query)
|
||||
from_setting = await _listed_model_ids(monkeypatch, {"model_list_return_wildcard_routes": True}, user_role, **query)
|
||||
from_setting = await _listed_model_ids(
|
||||
monkeypatch, {"model_list_return_wildcard_routes": setting}, user_role, **query
|
||||
)
|
||||
|
||||
assert "openai/*" in from_setting, from_setting
|
||||
assert from_setting == requested
|
||||
|
|
@ -3158,6 +3161,8 @@ async def test_model_list_return_wildcard_routes_setting_matches_query_param(
|
|||
[
|
||||
({}, {}),
|
||||
({"model_list_return_wildcard_routes": "false"}, {}),
|
||||
({"model_list_return_wildcard_routes": "True"}, {}),
|
||||
({"model_list_return_wildcard_routes": 1}, {}),
|
||||
({"model_list_return_wildcard_routes": True}, {"return_wildcard_routes": False}),
|
||||
],
|
||||
)
|
||||
|
|
@ -3171,6 +3176,42 @@ async def test_model_list_leaves_out_wildcard_routes_unless_enabled(
|
|||
assert "openai/*" not in listed, listed
|
||||
|
||||
|
||||
async def _model_info_resolves(model_id: str, **query: bool) -> bool:
|
||||
from fastapi import HTTPException
|
||||
|
||||
try:
|
||||
response = await litellm.proxy.proxy_server.model_info(
|
||||
model_id=model_id,
|
||||
user_api_key_dict=UserAPIKeyAuth(api_key="sk-test", user_role=LitellmUserRoles.INTERNAL_USER),
|
||||
**query,
|
||||
)
|
||||
except HTTPException as e:
|
||||
assert e.status_code == 404, e.detail
|
||||
return False
|
||||
return response["id"] == model_id
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"general_settings, query, wildcard_listed",
|
||||
[
|
||||
({}, {}, False),
|
||||
({"model_list_return_wildcard_routes": True}, {}, True),
|
||||
({"model_list_return_wildcard_routes": "true"}, {}, True),
|
||||
({"model_list_return_wildcard_routes": True}, {"return_wildcard_routes": False}, False),
|
||||
({}, {"return_wildcard_routes": True}, True),
|
||||
],
|
||||
)
|
||||
@pytest.mark.asyncio
|
||||
async def test_model_info_resolves_exactly_the_ids_model_list_lists(
|
||||
openai_wildcard_router, monkeypatch, general_settings, query, wildcard_listed
|
||||
):
|
||||
listed = await _listed_model_ids(monkeypatch, general_settings, LitellmUserRoles.INTERNAL_USER, **query)
|
||||
resolvable = [model_id for model_id in ("gpt-4o", "openai/*") if await _model_info_resolves(model_id, **query)]
|
||||
|
||||
assert ("openai/*" in listed) is wildcard_listed, listed
|
||||
assert resolvable == [model_id for model_id in ("gpt-4o", "openai/*") if model_id in listed], (listed, resolvable)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_model_list_return_wildcard_routes_is_an_admin_ui_toggle(openai_wildcard_router, monkeypatch):
|
||||
stored_general_settings: Final = {"model_list_return_wildcard_routes": True}
|
||||
|
|
|
|||
|
|
@ -168,8 +168,6 @@ describe("modelAvailableCall", () => {
|
|||
global.fetch = currentFetch;
|
||||
});
|
||||
|
||||
// An omitted return_wildcard_routes takes the proxy's model_list_return_wildcard_routes
|
||||
// default, so the dashboard's own pickers always send it.
|
||||
it.each([
|
||||
[undefined, "False"],
|
||||
[false, "False"],
|
||||
|
|
|
|||
|
|
@ -1940,8 +1940,6 @@ export const modelAvailableCall = async (
|
|||
accessToken,
|
||||
query: {
|
||||
include_model_access_groups: "True",
|
||||
// Sent either way: an omitted value falls back to the proxy's
|
||||
// general_settings.model_list_return_wildcard_routes, not to false.
|
||||
return_wildcard_routes: return_wildcard_routes === true ? "True" : "False",
|
||||
only_model_access_groups: only_model_access_groups === true ? "True" : undefined,
|
||||
team_id: teamID || undefined,
|
||||
|
|
|
|||
8
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
8
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
|
|
@ -3761,7 +3761,8 @@ export interface paths {
|
|||
* verify models via `GET {base}/models` (the OpenAI SDK contract). Without this
|
||||
* route those requests fall through to the Cursor Cloud Agents passthrough, which
|
||||
* demands a Cursor API key and 401s, so key verification silently fails before any
|
||||
* chat request is ever sent. Delegates to the standard `/v1/models` handler.
|
||||
* chat request is ever sent. Delegates to the standard `/v1/models` handler with
|
||||
* wildcard routes left out, since Cursor offers every listed id as a callable model.
|
||||
*/
|
||||
get: operations["cursor_model_list_cursor_models_get"];
|
||||
put?: never;
|
||||
|
|
@ -3787,7 +3788,8 @@ export interface paths {
|
|||
* verify models via `GET {base}/models` (the OpenAI SDK contract). Without this
|
||||
* route those requests fall through to the Cursor Cloud Agents passthrough, which
|
||||
* demands a Cursor API key and 401s, so key verification silently fails before any
|
||||
* chat request is ever sent. Delegates to the standard `/v1/models` handler.
|
||||
* chat request is ever sent. Delegates to the standard `/v1/models` handler with
|
||||
* wildcard routes left out, since Cursor offers every listed id as a callable model.
|
||||
*/
|
||||
get: operations["cursor_model_list_cursor_v1_models_get"];
|
||||
put?: never;
|
||||
|
|
@ -59947,6 +59949,7 @@ export interface operations {
|
|||
query?: {
|
||||
team_id?: string | null;
|
||||
healthy_only?: boolean | null;
|
||||
return_wildcard_routes?: boolean | null;
|
||||
};
|
||||
header?: never;
|
||||
path: {
|
||||
|
|
@ -73252,6 +73255,7 @@ export interface operations {
|
|||
query?: {
|
||||
team_id?: string | null;
|
||||
healthy_only?: boolean | null;
|
||||
return_wildcard_routes?: boolean | null;
|
||||
};
|
||||
header?: never;
|
||||
path: {
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue