diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index 7cd116c23d7..4e94db894fd 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -14466,11 +14466,24 @@ async def _fetch_db_models_for_search( filter for `team_public_model_name` instead and keep the DB cost bounded by `search`. """ - db_where_condition: Final[dict[str, Any]] = { - "model_name": {"contains": search_lower, "mode": "insensitive"} if model_name is None else model_name - } - if db_model_ids_in_router: - db_where_condition["model_id"] = {"not": {"in": list(db_model_ids_in_router)}} + db_where_condition: Final[dict[str, Any]] = ( + { + "AND": [ + { + "OR": [ + {"model_name": {"contains": search_lower, "mode": "insensitive"}}, + # JSON string_contains is case-sensitive on Postgres (see + # note above); router-side matching below covers the + # case-insensitive path for rows already in the router. + {"litellm_params": {"path": ["model"], "string_contains": search_lower}}, + ] + }, + *( [{"model_id": {"not": {"in": list(db_model_ids_in_router)}}}] if db_model_ids_in_router else [] ), + ] + } + if model_name is None + else {"model_name": model_name} + ) # Unsorted searches only need enough DB rows to fill the current # page after counting router-side matches. Sorted searches need @@ -14566,7 +14579,13 @@ async def _apply_search_filter_to_models( if search_lower in (m.get("model_name") or "").lower(): return True team_public_model_name: Final = (m.get("model_info") or {}).get("team_public_model_name") or "" - return search_lower in team_public_model_name.lower() + if search_lower in team_public_model_name.lower(): + return True + # Also match the underlying LiteLLM model name (e.g. + # "openrouter/deepseek/deepseek-chat"), so users can find + # deployments by typing the provider or the upstream model id. + litellm_model: Final = (m.get("litellm_params") or {}).get("model") or "" + return search_lower in litellm_model.lower() # Filter models in router by search term, dropping BYOK rows that # belong to teams the caller is not a member of so search can't leak @@ -14687,6 +14706,7 @@ def _sort_models( "updated_at", "costs", "status", + "blocked", ]: return all_models @@ -14744,6 +14764,11 @@ def _sort_models( db_model: Final = model_info.get("db_model", False) return db_model + elif sort_by == "blocked": + # Routing status: False (active) comes before True (paused) for asc, + # so `sortBy=blocked&sortOrder=asc` surfaces the active deployments. + return bool(model_info.get("blocked", False)) + return None try: @@ -14783,11 +14808,20 @@ def _matches_model_info_filters( exclude_auto_routers: bool | None, access_group: str | None, wildcard_only: bool | None, + blocked: bool | None = None, ) -> bool: if exclude_auto_routers is True and _is_auto_router_model(model): return False if isinstance(access_group, str) and not _model_in_access_group(model, access_group): return False + # Routing-status filter. Guarded on `is True` / `is False` because direct + # calls that bypass FastAPI pass the truthy Query sentinel as the default, + # which must not filter (same pattern as `exclude_auto_routers`). Entries + # without a `blocked` flag (e.g. A2A agents) are neither active nor paused, + # so they drop out of both filtered views. + if blocked is True or blocked is False: + if (model.get("model_info") or {}).get("blocked") is not blocked: + return False return wildcard_only is not True or "*" in str(model.get("model_name") or "") @@ -15086,12 +15120,19 @@ async def model_info_v2( ), sortBy: str | None = fastapi.Query( None, - description="Field to sort by. Options: model_name, created_at, updated_at, costs, status", + description="Field to sort by. Options: model_name, created_at, updated_at, costs, status, blocked", ), sortOrder: str | None = fastapi.Query( "asc", description="Sort order. Options: asc, desc", ), + blocked: bool | None = fastapi.Query( + None, + description=( + "Filter by routing status: false = active deployments, true = paused (blocked) " + "deployments. Omit to return both." + ), + ), exclude_auto_routers: bool | None = fastapi.Query( False, description=( @@ -15125,7 +15166,8 @@ async def model_info_v2( search: Case-insensitive partial match on model name or team public name. modelId: Return a single deployment by LiteLLM model id. teamId: Filter to models with direct access or team membership for this team id. - sortBy / sortOrder: Sort by model_name, created_at, updated_at, costs, or status. + sortBy / sortOrder: Sort by model_name, created_at, updated_at, costs, status, or blocked. + blocked: Filter by routing status (false = active, true = paused). access_group: Only return deployments in this model access group. wildcard_only: Only return deployments whose `model_name` contains `*`. @@ -15282,7 +15324,8 @@ async def model_info_v2( # `is True` because direct-call tests bypass FastAPI, so the Query default arrives as a # truthy sentinel object rather than False. all_models = [ - m for m in all_models if _matches_model_info_filters(m, exclude_auto_routers, access_group, wildcard_only) + m for m in all_models + if _matches_model_info_filters(m, exclude_auto_routers, access_group, wildcard_only, blocked) ] # Update total count to include agents diff --git a/tests/test_litellm/proxy/proxy_server/test_routes_model_info.py b/tests/test_litellm/proxy/proxy_server/test_routes_model_info.py index 5175d92084c..1b911e89e25 100644 --- a/tests/test_litellm/proxy/proxy_server/test_routes_model_info.py +++ b/tests/test_litellm/proxy/proxy_server/test_routes_model_info.py @@ -802,6 +802,122 @@ def test_v2_model_info_exclude_auto_routers_paginates_over_the_filtered_set(clie assert len(payload["data"]) == 1 +# --------------------------------------------------------------------------- +# GET /v2/model/info?blocked / ?sortBy=blocked +# --------------------------------------------------------------------------- + + +@pytest.fixture +def routing_status_router(monkeypatch): + """Router with one paused (blocked) deployment and two active ones.""" + model_list = [ + { + "model_name": "paused-model", + "litellm_params": {"model": "openai/paused-model"}, + "model_info": {"id": "paused-1", "db_model": True, "blocked": True}, + }, + { + "model_name": "active-model", + "litellm_params": {"model": "openai/active-model"}, + "model_info": {"id": "active-1", "db_model": True, "blocked": False}, + }, + { + "model_name": "another-active", + "litellm_params": {"model": "openai/another-active"}, + "model_info": {"id": "active-2", "db_model": True, "blocked": False}, + }, + ] + from unittest.mock import AsyncMock + + router = MagicMock() + router.model_list = model_list + monkeypatch.setattr(proxy_server, "llm_router", router) + monkeypatch.setattr(proxy_server, "llm_model_list", model_list) + monkeypatch.setattr(proxy_server, "prisma_client", MagicMock()) + monkeypatch.setattr(proxy_server, "user_model", None) + monkeypatch.setattr(proxy_server.proxy_config, "get_config", AsyncMock(return_value={})) + monkeypatch.setattr( + proxy_server, + "_apply_search_filter_to_models", + AsyncMock(side_effect=lambda all_models, **kw: (all_models, len(all_models))), + ) + monkeypatch.setattr(proxy_server, "_enrich_model_info_with_litellm_data", lambda model, **kw: model) + + import litellm.proxy.agent_endpoints.model_list_helpers as mlh + + monkeypatch.setattr(mlh, "append_agents_to_model_info", AsyncMock(side_effect=lambda models, **kw: models)) + yield router + + +def test_v2_model_info_blocked_filter_returns_only_paused(client, auth_as, routing_status_router): + """`?blocked=true` keeps just the paused deployments.""" + with auth_as(): + response = client.get("/v2/model/info", params={"blocked": "true"}) + assert response.status_code == 200 + payload = response.json() + assert _model_names(payload) == ["paused-model"] + assert payload["total_count"] == 1 + + +def test_v2_model_info_blocked_filter_returns_only_active(client, auth_as, routing_status_router): + """`?blocked=false` keeps just the active deployments.""" + with auth_as(): + response = client.get("/v2/model/info", params={"blocked": "false"}) + assert response.status_code == 200 + payload = response.json() + assert _model_names(payload) == ["active-model", "another-active"] + assert payload["total_count"] == 2 + + +def test_v2_model_info_sort_by_blocked_puts_active_first(client, auth_as, routing_status_router): + """`sortBy=blocked&sortOrder=asc` surfaces the active deployments first.""" + with auth_as(): + response = client.get("/v2/model/info", params={"sortBy": "blocked", "sortOrder": "asc"}) + assert response.status_code == 200 + assert _model_names(response.json()) == ["active-model", "another-active", "paused-model"] + + +def test_v2_model_info_search_matches_litellm_model_name(client, auth_as, monkeypatch): + """`search` also hits the underlying LiteLLM model name (e.g. a provider + prefix), not just the public model name.""" + model_list = [ + { + "model_name": "paused-model", + "litellm_params": {"model": "openai/paused-model"}, + "model_info": {"id": "paused-1", "db_model": True, "blocked": True}, + }, + { + "model_name": "active-model", + "litellm_params": {"model": "openai/active-model"}, + "model_info": {"id": "active-1", "db_model": True, "blocked": False}, + }, + ] + from unittest.mock import AsyncMock + + router = MagicMock() + router.model_list = model_list + monkeypatch.setattr(proxy_server, "llm_router", router) + monkeypatch.setattr(proxy_server, "llm_model_list", model_list) + monkeypatch.setattr(proxy_server, "prisma_client", MagicMock()) + monkeypatch.setattr(proxy_server, "user_model", None) + monkeypatch.setattr(proxy_server.proxy_config, "get_config", AsyncMock(return_value={})) + + async def fake_fetch_db_models_for_search(**kwargs): + return [], 0 + + monkeypatch.setattr(proxy_server, "_fetch_db_models_for_search", fake_fetch_db_models_for_search) + monkeypatch.setattr(proxy_server, "_enrich_model_info_with_litellm_data", lambda model, **kw: model) + + import litellm.proxy.agent_endpoints.model_list_helpers as mlh + + monkeypatch.setattr(mlh, "append_agents_to_model_info", AsyncMock(side_effect=lambda models, **kw: models)) + + with auth_as(): + response = client.get("/v2/model/info", params={"search": "openai/paused"}) + assert response.status_code == 200 + assert _model_names(response.json()) == ["paused-model"] + + @pytest.mark.asyncio async def test_model_info_v2_query_sentinel_does_not_filter(monkeypatch, mixed_auto_router_router): """Called directly (not through FastAPI) the default arrives as a truthy Query object. diff --git a/ui/litellm-dashboard/src/app/(dashboard)/hooks/models/useModels.ts b/ui/litellm-dashboard/src/app/(dashboard)/hooks/models/useModels.ts index b5cc329d4c9..bae33426e0a 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/hooks/models/useModels.ts +++ b/ui/litellm-dashboard/src/app/(dashboard)/hooks/models/useModels.ts @@ -42,6 +42,7 @@ export const useModelsInfo = ( modelName?: string, accessGroup?: string, wildcardOnly: boolean = false, + blocked?: boolean, ) => { const { accessToken, userId, userRole } = useAuthorized(); return useQuery({ @@ -62,6 +63,8 @@ export const useModelsInfo = ( ...(excludeAutoRouters && { excludeAutoRouters: "true" }), ...(accessGroup && { accessGroup }), ...(wildcardOnly && { wildcardOnly: "true" }), + // `blocked !== undefined` (not truthiness): false is a meaningful filter value. + ...(blocked !== undefined && { blocked }), }, }), queryFn: async () => @@ -80,6 +83,7 @@ export const useModelsInfo = ( modelName, accessGroup, wildcardOnly, + blocked, ), enabled: Boolean(accessToken && userId && userRole), }); diff --git a/ui/litellm-dashboard/src/app/(dashboard)/models-and-endpoints/components/AllModelsTab.tsx b/ui/litellm-dashboard/src/app/(dashboard)/models-and-endpoints/components/AllModelsTab.tsx index 2217bca0fa0..a7ead51bdc5 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/models-and-endpoints/components/AllModelsTab.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/models-and-endpoints/components/AllModelsTab.tsx @@ -30,6 +30,7 @@ import { isModelTableSortColumnId, MODEL_NAME_COLUMN_ID, MODEL_TABLE_SORT_COLUMN_IDS, + ROUTING_STATUS_COLUMN_ID, toServerSortField, } from "./ModelsTableColumns"; @@ -54,6 +55,7 @@ const TABLE_STATE = { view_mode: parseAsStringLiteral(MODEL_VIEW_MODES).withDefault("current_team"), filter_team: parseAsString.withDefault(PERSONAL_TEAM_VALUE), access_group: parseAsString.withDefault(""), + status: parseAsString.withDefault(""), sort_by: parseAsStringLiteral(MODEL_TABLE_SORT_COLUMN_IDS), sort_order: parseAsStringLiteral(["asc", "desc"] as const).withDefault("asc"), page: boundedInteger(1, MAX_PAGE, 1), @@ -88,6 +90,10 @@ const AllModelsTab = ({ const modelViewMode = tableState.view_mode; const selectedTeamValue = tableState.filter_team; const selectedModelAccessGroupFilter = tableState.access_group || null; + const routingStatusFilter = + tableState.status === "active" || tableState.status === "paused" ? tableState.status : null; + const blockedForQuery: boolean | undefined = + routingStatusFilter === null ? undefined : routingStatusFilter === "active"; const pagination = useMemo( () => ({ pageIndex: tableState.page - 1, pageSize: tableState.page_size }), [tableState.page, tableState.page_size], @@ -142,6 +148,7 @@ const AllModelsTab = ({ modelNameForQuery, accessGroupForQuery, wildcardOnlyForQuery, + blockedForQuery, ); const isLoading = isLoadingModelsInfo || isLoadingModelCostMap; @@ -169,8 +176,9 @@ const AllModelsTab = ({ ? { id: MODEL_NAME_COLUMN_ID, value: selectedModelGroup } : null, selectedModelAccessGroupFilter ? { id: ACCESS_GROUPS_COLUMN_ID, value: selectedModelAccessGroupFilter } : null, + routingStatusFilter ? { id: ROUTING_STATUS_COLUMN_ID, value: routingStatusFilter } : null, ].filter((entry) => entry !== null), - [selectedModelGroup, selectedModelAccessGroupFilter], + [selectedModelGroup, selectedModelAccessGroupFilter, routingStatusFilter], ); const handleSearchChange = useCallback( @@ -184,8 +192,13 @@ const AllModelsTab = ({ const next = functionalUpdate(updater, columnFilters); const modelGroup = next.find((entry) => entry.id === MODEL_NAME_COLUMN_ID)?.value; const accessGroup = next.find((entry) => entry.id === ACCESS_GROUPS_COLUMN_ID)?.value; + const status = next.find((entry) => entry.id === ROUTING_STATUS_COLUMN_ID)?.value; setSelectedModelGroup(typeof modelGroup === "string" ? modelGroup : ALL_MODEL_GROUPS_VALUE); - void setTableState({ access_group: typeof accessGroup === "string" ? accessGroup : null, page: null }); + void setTableState({ + access_group: typeof accessGroup === "string" ? accessGroup : null, + status: typeof status === "string" ? status : null, + page: null, + }); }; const handleSortingChange: OnChangeFn = (updater) => { diff --git a/ui/litellm-dashboard/src/app/(dashboard)/models-and-endpoints/components/AllModelsTable.test.tsx b/ui/litellm-dashboard/src/app/(dashboard)/models-and-endpoints/components/AllModelsTable.test.tsx index cc0169d745c..f3cecc4e37a 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/models-and-endpoints/components/AllModelsTable.test.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/models-and-endpoints/components/AllModelsTable.test.tsx @@ -76,13 +76,14 @@ const row = (modelId: string): HTMLElement => { }; describe("AllModelsTable", () => { - it("renders the nine design columns and hides Source behind the Columns menu", async () => { + it("renders the ten design columns and hides Source behind the Columns menu", async () => { const user = userEvent.setup(); render(); for (const header of [ "Model ID", "Model Information", + "Status", "Credentials", "Created By", "Updated At", @@ -95,17 +96,30 @@ describe("AllModelsTable", () => { } expect(screen.queryByRole("columnheader", { name: /^source$/i })).not.toBeInTheDocument(); - expect(screen.queryByRole("columnheader", { name: /^status$/i })).not.toBeInTheDocument(); expect(screen.queryByText("DB Model")).not.toBeInTheDocument(); await user.click(screen.getByRole("button", { name: /columns/i })); - expect(screen.queryByRole("menuitemcheckbox", { name: /status/i })).not.toBeInTheDocument(); await user.click(await screen.findByRole("menuitemcheckbox", { name: /^source$/i })); expect(await screen.findByRole("columnheader", { name: /^source$/i })).toBeInTheDocument(); expect(await screen.findByText("DB Model")).toBeInTheDocument(); }); + it("reports the routing status picked in the Filters drawer", async () => { + const user = userEvent.setup(); + const onColumnFiltersChange = vi.fn(); + render(); + + await user.click(screen.getByTestId("datatable-filters-trigger")); + await user.click(screen.getByRole("combobox", { name: /filter by status/i })); + await user.click(await screen.findByRole("option", { name: "Paused" })); + await user.click(screen.getByTestId("filter-drawer-apply")); + + const updater = onColumnFiltersChange.mock.calls.at(-1)?.[0]; + const next = typeof updater === "function" ? updater([]) : updater; + expect(next).toEqual([{ id: "model_info_blocked", value: "paused" }]); + }); + it("opens the model detail from the model ID cell", async () => { const user = userEvent.setup(); const onModelIdClick = vi.fn(); diff --git a/ui/litellm-dashboard/src/app/(dashboard)/models-and-endpoints/components/AllModelsTable.tsx b/ui/litellm-dashboard/src/app/(dashboard)/models-and-endpoints/components/AllModelsTable.tsx index 2a52bdfb46e..e9176bfdf81 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/models-and-endpoints/components/AllModelsTable.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/models-and-endpoints/components/AllModelsTable.tsx @@ -21,6 +21,7 @@ import { ACCESS_GROUPS_COLUMN_ID, getModelsTableColumns, MODEL_NAME_COLUMN_ID, + ROUTING_STATUS_COLUMN_ID, STATUS_COLUMN_ID, } from "./ModelsTableColumns"; @@ -36,8 +37,15 @@ const ALL_PROXY_MODELS_LABEL = "All Proxy Models"; const FILTER_LABELS: Record = { [MODEL_NAME_COLUMN_ID]: "Public Model Name", [ACCESS_GROUPS_COLUMN_ID]: "Model Access Group", + [ROUTING_STATUS_COLUMN_ID]: "Status", }; +const ROUTING_STATUS_FILTER_VALUES = { + all: "All Statuses", + active: "Active", + paused: "Paused", +} as const; + const VIEW_MODE_LABELS: Record = { current_team: "Current Team Models", all: ALL_PROXY_MODELS_LABEL, @@ -167,6 +175,9 @@ export function AllModelsTable({ if (columnId === MODEL_NAME_COLUMN_ID && raw === WILDCARD_MODEL_GROUP_VALUE) { return "Wildcard Models (*)"; } + if (columnId === ROUTING_STATUS_COLUMN_ID) { + return ROUTING_STATUS_FILTER_VALUES[raw as keyof typeof ROUTING_STATUS_FILTER_VALUES] ?? raw; + } return raw; }; @@ -288,6 +299,29 @@ export function AllModelsTable({ emptyText="No models found" /> + + + = { [STATUS_COLUMN_ID]: "status", [CREATED_BY_COLUMN_ID]: "created_at", [UPDATED_AT_COLUMN_ID]: "updated_at", + [ROUTING_STATUS_COLUMN_ID]: "blocked", }; export const toServerSortField = (columnId: string): string => COLUMN_ID_TO_SERVER_SORT_FIELD[columnId] ?? columnId; @@ -259,6 +262,14 @@ function AccessGroupsCell({ accessGroups }: { accessGroups: string[] | null }) { ); } +function RoutingStatusCell({ blocked }: { blocked: boolean | null | undefined }) { + return blocked === true ? ( + + ) : ( + + ); +} + interface ModelRowActionsProps { model: ModelData; userRole: string; @@ -444,6 +455,16 @@ export const getModelsTableColumns = ({ minSize: 90, cell: ({ row }) => , }, + { + id: ROUTING_STATUS_COLUMN_ID, + accessorFn: (row) => row.model_info?.blocked === true, + meta: { title: "Status", skeleton: "badge" }, + header: ({ column }) => , + enableSorting: true, + size: 110, + minSize: 90, + cell: ({ row }) => , + }, { id: TEAM_ID_COLUMN_ID, accessorFn: (row) => row.model_info.team_id ?? "", diff --git a/ui/litellm-dashboard/src/components/networking.tsx b/ui/litellm-dashboard/src/components/networking.tsx index 42dc5350b49..556b4c4f009 100644 --- a/ui/litellm-dashboard/src/components/networking.tsx +++ b/ui/litellm-dashboard/src/components/networking.tsx @@ -1669,6 +1669,7 @@ export const modelInfoCall = async ( modelName?: string, accessGroup?: string, wildcardOnly?: boolean, + blocked?: boolean, ) => { /** * Get all models on proxy @@ -1706,6 +1707,9 @@ export const modelInfoCall = async ( if (wildcardOnly) { params.append("wildcard_only", "true"); } + if (blocked !== undefined) { + params.append("blocked", blocked ? "true" : "false"); + } if (params.toString()) { url += `?${params.toString()}`; }