From d94b227906a67d6f283c890f4080259e0706160d Mon Sep 17 00:00:00 2001
From: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
Date: Mon, 14 Sep 2026 12:27:24 +0000
Subject: [PATCH 001/160] fix(mcp): stable ordering for MCP servers list in
Admin UI
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
---
.../mcp_management_endpoints.py | 9 ++-
.../test_mcp_management_endpoints.py | 70 +++++++++++++++++++
.../_components/mcp_servers.test.tsx | 29 +++++++-
.../mcp-servers/_components/mcp_servers.tsx | 49 ++++++-------
4 files changed, 127 insertions(+), 30 deletions(-)
diff --git a/litellm/proxy/management_endpoints/mcp_management_endpoints.py b/litellm/proxy/management_endpoints/mcp_management_endpoints.py
index 4c97bbaf5de..1dac2fd3a77 100644
--- a/litellm/proxy/management_endpoints/mcp_management_endpoints.py
+++ b/litellm/proxy/management_endpoints/mcp_management_endpoints.py
@@ -1024,6 +1024,9 @@ if MCP_AVAILABLE:
return {"servers": registry_servers}
## FastAPI Routes
+ def _mcp_server_display_order(server: LiteLLM_MCPServerTable) -> tuple[str, str]:
+ return ((server.server_name or server.alias or server.server_id).lower(), server.server_id)
+
def _get_user_mcp_management_mode() -> UserMCPManagementMode:
from litellm.proxy.proxy_server import (
general_settings as proxy_general_settings,
@@ -1174,10 +1177,12 @@ if MCP_AVAILABLE:
detail="You do not have permission to view MCP servers for this team.",
)
- redacted_mcp_servers = await _get_team_scoped_mcp_server_list(sanitized_team_id)
+ redacted_mcp_servers = sorted(
+ await _get_team_scoped_mcp_server_list(sanitized_team_id), key=_mcp_server_display_order
+ )
else:
servers: Final = await _resolve_accessible_mcp_servers(user_api_key_dict)
- redacted_mcp_servers = _redact_mcp_credentials_list(servers)
+ redacted_mcp_servers = sorted(_redact_mcp_credentials_list(servers), key=_mcp_server_display_order)
if connected_app_view is True and is_ui_session_credential(user_api_key_dict):
reachable_ids: Final = await _connected_app_reachable_server_ids(user_api_key_dict)
diff --git a/tests/test_litellm/proxy/management_endpoints/test_mcp_management_endpoints.py b/tests/test_litellm/proxy/management_endpoints/test_mcp_management_endpoints.py
index 5e00e7d75be..96bb0e35f0f 100644
--- a/tests/test_litellm/proxy/management_endpoints/test_mcp_management_endpoints.py
+++ b/tests/test_litellm/proxy/management_endpoints/test_mcp_management_endpoints.py
@@ -1535,6 +1535,76 @@ class TestTeamScopedMCPServerAccess:
result = await fetch_all_mcp_servers(user_api_key_dict=mock_user_auth, team_id="any-team-id")
assert len(result) == 1
+ @pytest.mark.asyncio
+ async def test_team_scoped_list_is_sorted_by_display_name(self):
+ """Set-derived resolution order must not leak to the client."""
+ mock_user_auth = generate_mock_user_api_key_auth(
+ user_role=LitellmUserRoles.PROXY_ADMIN,
+ user_id="admin_user",
+ )
+ unsorted = [
+ generate_mock_mcp_server_db_record(server_id="s-zeta", alias="zeta"),
+ generate_mock_mcp_server_db_record(server_id="s-alpha", alias="Alpha"),
+ generate_mock_mcp_server_db_record(server_id="s-mid", alias="mid"),
+ ]
+
+ with (
+ patch(
+ "litellm.proxy.management_endpoints.mcp_management_endpoints._user_has_admin_view",
+ return_value=True,
+ ),
+ patch(
+ "litellm.proxy.management_endpoints.mcp_management_endpoints._get_team_scoped_mcp_server_list",
+ AsyncMock(return_value=unsorted),
+ ),
+ ):
+ from litellm.proxy.management_endpoints.mcp_management_endpoints import (
+ fetch_all_mcp_servers,
+ )
+
+ result = await fetch_all_mcp_servers(user_api_key_dict=mock_user_auth, team_id="any-team-id")
+ assert [s.server_id for s in result] == ["s-alpha", "s-mid", "s-zeta"]
+
+
+class TestFetchAllMCPServersOrdering:
+ @pytest.mark.asyncio
+ async def test_list_is_sorted_by_display_name_regardless_of_resolution_order(self):
+ """The registry resolves ids through a set, so the response must impose its own order."""
+ mock_user_auth = generate_mock_user_api_key_auth(
+ user_role=LitellmUserRoles.PROXY_ADMIN,
+ user_id="admin_user",
+ )
+ first_order = [
+ generate_mock_mcp_server_db_record(server_id="s-zeta", alias="zeta"),
+ generate_mock_mcp_server_db_record(server_id="s-alpha", alias="Alpha"),
+ generate_mock_mcp_server_db_record(server_id="s-mid", alias="mid"),
+ ]
+ second_order = list(reversed(first_order))
+
+ for resolved in (first_order, second_order):
+ mock_manager = MagicMock()
+ mock_manager.get_all_allowed_mcp_servers = AsyncMock(return_value=resolved)
+ with (
+ patch(
+ "litellm.proxy.management_endpoints.mcp_management_endpoints.global_mcp_server_manager",
+ mock_manager,
+ ),
+ patch(
+ "litellm.proxy.management_endpoints.mcp_management_endpoints._user_has_admin_view",
+ return_value=True,
+ ),
+ patch(
+ "litellm.proxy.management_endpoints.mcp_management_endpoints.build_effective_auth_contexts",
+ AsyncMock(return_value=[mock_user_auth]),
+ ),
+ ):
+ from litellm.proxy.management_endpoints.mcp_management_endpoints import (
+ fetch_all_mcp_servers,
+ )
+
+ result = await fetch_all_mcp_servers(user_api_key_dict=mock_user_auth)
+ assert [s.server_id for s in result] == ["s-alpha", "s-mid", "s-zeta"]
+
@pytest.mark.asyncio
async def test_restricted_virtual_key_cannot_use_team_id_filter(self):
"""Restricted virtual keys must not bypass access limits via team_id."""
diff --git a/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/mcp_servers.test.tsx b/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/mcp_servers.test.tsx
index 8abb8855e3d..390c42f1ea3 100644
--- a/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/mcp_servers.test.tsx
+++ b/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/mcp_servers.test.tsx
@@ -3,7 +3,8 @@ import { render, waitFor, screen, act, within } from "@testing-library/react";
import userEvent from "@testing-library/user-event";
import { describe, it, expect, vi, beforeEach } from "vitest";
import { QueryClient, QueryClientProvider } from "@tanstack/react-query";
-import MCPServers from "./mcp_servers";
+import MCPServers, { compareServers } from "./mcp_servers";
+import type { MCPServer } from "@/components/mcp_tools/types";
import * as networking from "@/components/networking";
// Mock the networking module
@@ -29,6 +30,32 @@ const createQueryClient = () =>
},
});
+describe("compareServers", () => {
+ const server = (server_id: string, name: string, created_at = ""): MCPServer =>
+ ({ server_id, server_name: name, created_at, updated_at: created_at }) as MCPServer;
+
+ const shuffled = [server("c", "github"), server("a", "slack"), server("b", "Jira")];
+
+ it("orders servers without timestamps by name so config.yaml servers render in a stable order", () => {
+ const byCreated = [...shuffled].sort((a, b) => compareServers(a, b, "created_desc")).map((s) => s.server_id);
+ const byUpdated = [...shuffled].sort((a, b) => compareServers(a, b, "updated_desc")).map((s) => s.server_id);
+ const byHealth = [...shuffled].sort((a, b) => compareServers(a, b, "health")).map((s) => s.server_id);
+
+ expect(byCreated).toEqual(["c", "b", "a"]);
+ expect(byUpdated).toEqual(["c", "b", "a"]);
+ expect(byHealth).toEqual(["c", "b", "a"]);
+ });
+
+ it("keeps newest-first when timestamps differ", () => {
+ const newest = server("new", "zzz", "2026-02-01T00:00:00Z");
+ const oldest = server("old", "aaa", "2026-01-01T00:00:00Z");
+ expect([oldest, newest].sort((a, b) => compareServers(a, b, "created_desc")).map((s) => s.server_id)).toEqual([
+ "new",
+ "old",
+ ]);
+ });
+});
+
describe("MCPServers", () => {
const defaultProps = {
accessToken: "123",
diff --git a/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/mcp_servers.tsx b/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/mcp_servers.tsx
index e6148d5d997..902e8811996 100644
--- a/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/mcp_servers.tsx
+++ b/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/mcp_servers.tsx
@@ -45,7 +45,7 @@ import { TOOLS_OAUTH_UI_STATE_KEY } from "@/hooks/mcpOAuthUtils";
import UserEnvVarsModal from "./UserEnvVarsModal";
import { listMCPUserEnvVarStatus } from "@/components/networking";
-type SortKey = "created_desc" | "updated_desc" | "name_asc" | "health";
+export type SortKey = "created_desc" | "updated_desc" | "name_asc" | "health";
const SORT_OPTIONS: { value: SortKey; label: string }[] = [
{ value: "created_desc", label: "Recently created" },
@@ -60,32 +60,33 @@ const HEALTH_RANK: Record = {
healthy: 2,
};
-const compareServers = (a: MCPServer, b: MCPServer, sort: SortKey): number => {
+const compareByName = (a: MCPServer, b: MCPServer): number => {
+ const nameA = (a.server_name || a.alias || a.server_id).toLowerCase();
+ const nameB = (b.server_name || b.alias || b.server_id).toLowerCase();
+ return nameA.localeCompare(nameB) || a.server_id.localeCompare(b.server_id);
+};
+
+const compareByTimestampDesc = (a: string | null | undefined, b: string | null | undefined): number => {
+ const ta = a ? new Date(a).getTime() : 0;
+ const tb = b ? new Date(b).getTime() : 0;
+ return tb - ta;
+};
+
+export const compareServers = (a: MCPServer, b: MCPServer, sort: SortKey): number => {
switch (sort) {
- case "name_asc": {
- const nameA = (a.server_name || a.alias || a.server_id).toLowerCase();
- const nameB = (b.server_name || b.alias || b.server_id).toLowerCase();
- return nameA.localeCompare(nameB);
- }
- case "updated_desc": {
- const ta = a.updated_at ? new Date(a.updated_at).getTime() : 0;
- const tb = b.updated_at ? new Date(b.updated_at).getTime() : 0;
- return tb - ta;
- }
+ case "name_asc":
+ return compareByName(a, b);
+ case "updated_desc":
+ return compareByTimestampDesc(a.updated_at, b.updated_at) || compareByName(a, b);
case "health": {
const ra = HEALTH_RANK[a.status ?? "unknown"] ?? 1;
const rb = HEALTH_RANK[b.status ?? "unknown"] ?? 1;
if (ra !== rb) return ra - rb;
- const ta = a.created_at ? new Date(a.created_at).getTime() : 0;
- const tb = b.created_at ? new Date(b.created_at).getTime() : 0;
- return tb - ta;
+ return compareByTimestampDesc(a.created_at, b.created_at) || compareByName(a, b);
}
case "created_desc":
- default: {
- const ta = a.created_at ? new Date(a.created_at).getTime() : 0;
- const tb = b.created_at ? new Date(b.created_at).getTime() : 0;
- return tb - ta;
- }
+ default:
+ return compareByTimestampDesc(a.created_at, b.created_at) || compareByName(a, b);
}
};
@@ -297,13 +298,7 @@ const MCPServers: React.FC = ({ accessToken, userRole, userID })
server.mcp_access_groups?.some((g: any) => (typeof g === "string" ? g === group : g && g.name === group)),
);
}
- const sorted = [...filtered].sort((a, b) => {
- if (!a.created_at && !b.created_at) return 0;
- if (!a.created_at) return 1;
- if (!b.created_at) return -1;
- return new Date(b.created_at).getTime() - new Date(a.created_at).getTime();
- });
- setFilteredServers(sorted);
+ setFilteredServers(filtered);
},
[serversWithHealth],
);
From aec592b734a91c3d068f89f0977561b733bc926f Mon Sep 17 00:00:00 2001
From: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
Date: Mon, 14 Sep 2026 12:43:50 +0000
Subject: [PATCH 002/160] test(mcp): drop internal patches from ordering tests
and regenerate API types
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
---
.../test_mcp_management_endpoints.py | 47 ++++++-------------
ui/litellm-dashboard/src/lib/http/schema.d.ts | 2 -
2 files changed, 14 insertions(+), 35 deletions(-)
diff --git a/tests/test_litellm/proxy/management_endpoints/test_mcp_management_endpoints.py b/tests/test_litellm/proxy/management_endpoints/test_mcp_management_endpoints.py
index 96bb0e35f0f..6381bb7b9fc 100644
--- a/tests/test_litellm/proxy/management_endpoints/test_mcp_management_endpoints.py
+++ b/tests/test_litellm/proxy/management_endpoints/test_mcp_management_endpoints.py
@@ -1535,38 +1535,19 @@ class TestTeamScopedMCPServerAccess:
result = await fetch_all_mcp_servers(user_api_key_dict=mock_user_auth, team_id="any-team-id")
assert len(result) == 1
- @pytest.mark.asyncio
- async def test_team_scoped_list_is_sorted_by_display_name(self):
- """Set-derived resolution order must not leak to the client."""
- mock_user_auth = generate_mock_user_api_key_auth(
- user_role=LitellmUserRoles.PROXY_ADMIN,
- user_id="admin_user",
- )
- unsorted = [
- generate_mock_mcp_server_db_record(server_id="s-zeta", alias="zeta"),
- generate_mock_mcp_server_db_record(server_id="s-alpha", alias="Alpha"),
- generate_mock_mcp_server_db_record(server_id="s-mid", alias="mid"),
- ]
-
- with (
- patch(
- "litellm.proxy.management_endpoints.mcp_management_endpoints._user_has_admin_view",
- return_value=True,
- ),
- patch(
- "litellm.proxy.management_endpoints.mcp_management_endpoints._get_team_scoped_mcp_server_list",
- AsyncMock(return_value=unsorted),
- ),
- ):
- from litellm.proxy.management_endpoints.mcp_management_endpoints import (
- fetch_all_mcp_servers,
- )
-
- result = await fetch_all_mcp_servers(user_api_key_dict=mock_user_auth, team_id="any-team-id")
- assert [s.server_id for s in result] == ["s-alpha", "s-mid", "s-zeta"]
-
class TestFetchAllMCPServersOrdering:
+ def test_display_order_is_case_insensitive_name_then_id(self):
+ servers = [
+ generate_mock_mcp_server_db_record(server_id="s-2", alias="github"),
+ generate_mock_mcp_server_db_record(server_id="s-1", alias="github"),
+ generate_mock_mcp_server_db_record(server_id="s-0", alias="Slack"),
+ generate_mock_mcp_server_db_record(server_id="s-3", alias="confluence"),
+ ]
+
+ ordered = sorted(servers, key=mgmt_endpoints._mcp_server_display_order)
+ assert [s.server_id for s in ordered] == ["s-3", "s-1", "s-2", "s-0"]
+
@pytest.mark.asyncio
async def test_list_is_sorted_by_display_name_regardless_of_resolution_order(self):
"""The registry resolves ids through a set, so the response must impose its own order."""
@@ -1585,15 +1566,15 @@ class TestFetchAllMCPServersOrdering:
mock_manager = MagicMock()
mock_manager.get_all_allowed_mcp_servers = AsyncMock(return_value=resolved)
with (
- patch(
+ patch( # test-quality-ok: the route reads a module-global manager with no injection seam
"litellm.proxy.management_endpoints.mcp_management_endpoints.global_mcp_server_manager",
mock_manager,
),
- patch(
+ patch( # test-quality-ok: admin view is derived from module-global proxy settings
"litellm.proxy.management_endpoints.mcp_management_endpoints._user_has_admin_view",
return_value=True,
),
- patch(
+ patch( # test-quality-ok: auth contexts need a live prisma client
"litellm.proxy.management_endpoints.mcp_management_endpoints.build_effective_auth_contexts",
AsyncMock(return_value=[mock_user_auth]),
),
diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts
index 7eadaa6c991..839aa52fa84 100644
--- a/ui/litellm-dashboard/src/lib/http/schema.d.ts
+++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts
@@ -16781,7 +16781,6 @@ export interface paths {
* - permissions: Optional[dict] - [Not Implemented Yet] User-specific permissions, eg. turning off pii masking.
* - metadata: Optional[dict] - Metadata for user, store information for user. Example metadata = {"team": "core-infra", "app": "app2", "email": "ishaan@berri.ai" }
* - max_parallel_requests: Optional[int] - Rate limit a user based on the number of parallel requests. Raises 429 error, if user's parallel requests > x.
- * - soft_budget: Optional[float] - Get alerts when user crosses given budget, doesn't block requests.
* - model_max_budget: Optional[dict] - Model-specific max budget for user. [Docs](https://docs.litellm.ai/docs/proxy/users#add-model-specific-budgets-to-keys)
* - budget_fallbacks: Optional[Dict[str, List[str]]] - Per-model fallback chain tried in order when that model's own `model_max_budget` is exceeded, e.g. {"gpt-4o": ["gpt-4o-mini"]}.
* - model_rpm_limit: Optional[float] - Model-specific rpm limit for user. [Docs](https://docs.litellm.ai/docs/proxy/users#add-model-specific-limits-to-keys)
@@ -16887,7 +16886,6 @@ export interface paths {
* - permissions: Optional[dict] - [Not Implemented Yet] User-specific permissions, eg. turning off pii masking.
* - metadata: Optional[dict] - Metadata for user, store information for user. Example metadata = {"team": "core-infra", "app": "app2", "email": "ishaan@berri.ai" }
* - max_parallel_requests: Optional[int] - Rate limit a user based on the number of parallel requests. Raises 429 error, if user's parallel requests > x.
- * - soft_budget: Optional[float] - Get alerts when user crosses given budget, doesn't block requests.
* - model_max_budget: Optional[dict] - Model-specific max budget for user. [Docs](https://docs.litellm.ai/docs/proxy/users#add-model-specific-budgets-to-keys)
* - budget_fallbacks: Optional[Dict[str, List[str]]] - Per-model fallback chain tried in order when that model's own `model_max_budget` is exceeded, e.g. {"gpt-4o": ["gpt-4o-mini"]}.
* - model_rpm_limit: Optional[float] - Model-specific rpm limit for user. [Docs](https://docs.litellm.ai/docs/proxy/users#add-model-specific-limits-to-keys)
From 067514bc76794fc3f9a66b9083f1cf52e6728427 Mon Sep 17 00:00:00 2001
From: Tejas Chopra
Date: Wed, 16 Sep 2026 21:33:26 -0700
Subject: [PATCH 003/160] fix(responses): patch custom_tool_call_output in
place on guardrail write-back
---
.../guardrail_translation/handler.py | 2 +-
...test_openai_responses_guardrail_handler.py | 50 +++++++++++++++++++
2 files changed, 51 insertions(+), 1 deletion(-)
diff --git a/litellm/llms/openai/responses/guardrail_translation/handler.py b/litellm/llms/openai/responses/guardrail_translation/handler.py
index 1ab4811b8a0..9739ca95089 100644
--- a/litellm/llms/openai/responses/guardrail_translation/handler.py
+++ b/litellm/llms/openai/responses/guardrail_translation/handler.py
@@ -209,7 +209,7 @@ _TOOL_CALL_PAYLOAD_EVENT_TYPES: Final = _TOOL_CALL_PAYLOAD_DELTA_EVENT_TYPES | f
)
_OUTPUT_ITEM_EVENT_TYPES: Final = frozenset({"response.output_item.added", "response.output_item.done"})
_PATCHABLE_ITEM_FIELDS: Final[Mapping[str, str]] = MappingProxyType(
- {"function_call_output": "output", "message": "content"}
+ {"function_call_output": "output", "custom_tool_call_output": "output", "message": "content"}
)
_EMPTY_RESPONSES_REQUEST: Final[ResponsesAPIOptionalRequestParams] = {}
diff --git a/tests/test_litellm/llms/openai/responses/test_openai_responses_guardrail_handler.py b/tests/test_litellm/llms/openai/responses/test_openai_responses_guardrail_handler.py
index c714b5d378a..ea58a48a3d5 100644
--- a/tests/test_litellm/llms/openai/responses/test_openai_responses_guardrail_handler.py
+++ b/tests/test_litellm/llms/openai/responses/test_openai_responses_guardrail_handler.py
@@ -2223,6 +2223,56 @@ class TestStructuredMessagesWriteBack:
}
assert result["input"][3] == {"role": "user", "content": "What is the codename?"}
+ @pytest.mark.asyncio
+ async def test_codex_custom_tool_items_survive_tool_output_compression(self):
+ handler = OpenAIResponsesHandler()
+ additional_tools_item = {
+ "type": "additional_tools",
+ "tools": [{"type": "custom", "name": "exec", "description": "Run a JavaScript snippet"}],
+ }
+ reasoning_item = {
+ "id": "rs_456",
+ "type": "reasoning",
+ "summary": [],
+ "encrypted_content": "gAAAAA-signed-reasoning",
+ }
+ custom_tool_call_item = {
+ "id": "ctc_456",
+ "type": "custom_tool_call",
+ "call_id": "call_exec",
+ "name": "exec",
+ "input": 'const r = await tools.exec_command({"cmd": "cat memo.txt"});\ntext(r.output);',
+ "status": "completed",
+ }
+ data = {
+ "model": "gpt-5.6",
+ "input": [
+ additional_tools_item,
+ {"role": "user", "content": "What is the codename?"},
+ reasoning_item,
+ custom_tool_call_item,
+ {
+ "type": "custom_tool_call_output",
+ "call_id": "call_exec",
+ "output": [
+ {"type": "input_text", "text": "Script completed\nOutput:\n"},
+ {"type": "input_text", "text": "memo " * 400},
+ ],
+ },
+ ],
+ }
+
+ result = await handler.process_input_messages(data, ToolOutputRewriteGuardrail())
+
+ assert result["input"][0] is additional_tools_item
+ assert result["input"][1] == {"role": "user", "content": "What is the codename?"}
+ assert result["input"][2] is reasoning_item
+ assert result["input"][3] is custom_tool_call_item
+ assert result["input"][4]["type"] == "custom_tool_call_output"
+ assert result["input"][4]["call_id"] == "call_exec"
+ assert COMPRESSED_MARKER in str(result["input"][4]["output"])
+ assert len(result["input"]) == 5
+
@pytest.mark.asyncio
async def test_web_search_call_item_preserved_verbatim(self):
handler = OpenAIResponsesHandler()
From b84f8b6a772d859a6ad762e8429549b86b877ca0 Mon Sep 17 00:00:00 2001
From: yassin
Date: Thu, 17 Sep 2026 19:24:14 +0000
Subject: [PATCH 004/160] feat(agents): attach access groups to agents and
enforce them for models, MCP servers and agent calls
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
---
.../migration.sql | 2 +
.../litellm_proxy_extras/schema.prisma | 1 +
.../mcp_server/auth/user_api_key_auth_mcp.py | 37 ++-
litellm/proxy/_lazy_openapi_snapshot.json | 42 +++
litellm/proxy/_types.py | 9 +-
.../proxy/agent_endpoints/agent_registry.py | 15 +-
.../auth/agent_access_groups.py | 80 ++++++
.../auth/agent_permission_handler.py | 33 ++-
litellm/proxy/auth/auth_checks.py | 37 ++-
.../access_group_endpoints.py | 72 +++++
litellm/proxy/schema.prisma | 1 +
litellm/types/agents.py | 3 +
schema.prisma | 1 +
.../auth/test_user_api_key_auth_mcp.py | 59 ++++
.../auth/test_agent_access_groups.py | 130 +++++++++
.../auth/test_agent_permission_handler.py | 83 ++++++
.../agent_endpoints/test_agent_registry.py | 133 +++++++++
.../proxy/auth/test_auth_checks.py | 91 ++++++
.../test_access_group_endpoints.py | 271 ++++++++----------
.../agents/_components/AgentFormKit.tsx | 2 +
.../add_agent_form.integration.test.tsx | 3 +
.../_components/add_agent_form.test.tsx | 30 ++
.../agents/_components/add_agent_form.tsx | 21 ++
.../agents/_components/agent_config.ts | 5 +
.../agent_info.integration.test.tsx | 7 +
.../agents/_components/agent_info.test.tsx | 70 +++++
.../agents/_components/agent_info.tsx | 53 +++-
.../src/components/agents/types.ts | 1 +
.../src/components/networking.tsx | 1 +
ui/litellm-dashboard/src/lib/http/schema.d.ts | 6 +
30 files changed, 1138 insertions(+), 161 deletions(-)
create mode 100644 litellm-proxy-extras/litellm_proxy_extras/migrations/20260917000000_add_agent_access_group_ids/migration.sql
create mode 100644 litellm/proxy/agent_endpoints/auth/agent_access_groups.py
create mode 100644 tests/test_litellm/proxy/agent_endpoints/auth/test_agent_access_groups.py
diff --git a/litellm-proxy-extras/litellm_proxy_extras/migrations/20260917000000_add_agent_access_group_ids/migration.sql b/litellm-proxy-extras/litellm_proxy_extras/migrations/20260917000000_add_agent_access_group_ids/migration.sql
new file mode 100644
index 00000000000..d594b0056df
--- /dev/null
+++ b/litellm-proxy-extras/litellm_proxy_extras/migrations/20260917000000_add_agent_access_group_ids/migration.sql
@@ -0,0 +1,2 @@
+-- AlterTable
+ALTER TABLE "LiteLLM_AgentsTable" ADD COLUMN IF NOT EXISTS "access_group_ids" TEXT[] DEFAULT ARRAY[]::TEXT[];
diff --git a/litellm-proxy-extras/litellm_proxy_extras/schema.prisma b/litellm-proxy-extras/litellm_proxy_extras/schema.prisma
index 139fb031671..e0b52dd77de 100644
--- a/litellm-proxy-extras/litellm_proxy_extras/schema.prisma
+++ b/litellm-proxy-extras/litellm_proxy_extras/schema.prisma
@@ -71,6 +71,7 @@ model LiteLLM_AgentsTable {
static_headers Json? @default("{}")
extra_headers String[] @default([])
agent_access_groups String[] @default([])
+ access_group_ids String[] @default([])
object_permission_id String?
object_permission LiteLLM_ObjectPermissionTable? @relation(fields: [object_permission_id], references: [object_permission_id])
spend Float @default(0.0)
diff --git a/litellm/proxy/_experimental/mcp_server/auth/user_api_key_auth_mcp.py b/litellm/proxy/_experimental/mcp_server/auth/user_api_key_auth_mcp.py
index b0d57cb6228..bbb3d30864f 100644
--- a/litellm/proxy/_experimental/mcp_server/auth/user_api_key_auth_mcp.py
+++ b/litellm/proxy/_experimental/mcp_server/auth/user_api_key_auth_mcp.py
@@ -1549,10 +1549,18 @@ class MCPRequestHandler:
allowed_mcp_servers_for_agent: Final = await MCPRequestHandler._get_allowed_mcp_servers_for_agent(
user_api_key_auth
)
- if len(allowed_mcp_servers_for_agent) > 0:
+ agent_access_group_servers: Final = await MCPRequestHandler._get_agent_access_group_server_ceiling(
+ user_api_key_auth
+ )
+ if len(allowed_mcp_servers_for_agent) > 0 or agent_access_group_servers is not None:
has_lower_level_mcp_restrictions = True
- # Intersect: agent can only use servers allowed by BOTH key/team AND agent config
- allowed_mcp_servers = [s for s in allowed_mcp_servers if s in allowed_mcp_servers_for_agent]
+ # Intersect: agent can only use servers allowed by key/team AND agent config AND agent access groups
+ allowed_mcp_servers = [
+ s
+ for s in allowed_mcp_servers
+ if (len(allowed_mcp_servers_for_agent) == 0 or s in allowed_mcp_servers_for_agent)
+ and (agent_access_group_servers is None or s in agent_access_group_servers)
+ ]
verbose_logger.debug(
"Applied agent intersection filter. Final allowed servers: %s", allowed_mcp_servers
)
@@ -3137,6 +3145,29 @@ class MCPRequestHandler:
verbose_logger.warning("Failed to get allowed MCP servers for agent: %s", e)
return []
+ @staticmethod
+ async def _get_agent_access_group_server_ceiling(
+ user_api_key_auth: UserAPIKeyAuth,
+ ) -> frozenset[str] | None:
+ """
+ Server IDs the agent's attached unified access groups (``LiteLLM_AgentsTable.access_group_ids``)
+ allow, or None when the agent has none attached. Unlike the object_permission path above, an
+ attached group set that names no servers is an empty ceiling and denies every server.
+ """
+ from litellm.proxy._experimental.mcp_server.mcp_server_manager import (
+ global_mcp_server_manager,
+ )
+ from litellm.proxy.agent_endpoints.auth.agent_access_groups import (
+ resolve_agent_access_group_ceiling,
+ )
+
+ if not user_api_key_auth.agent_id:
+ return None
+ ceiling: Final = await resolve_agent_access_group_ceiling(user_api_key_auth.agent_id)
+ if ceiling is None:
+ return None
+ return frozenset(global_mcp_server_manager.expand_permission_list(sorted(ceiling.mcp_server_ids)))
+
@staticmethod
async def _get_agent_tool_permissions_for_server(
server_id: str,
diff --git a/litellm/proxy/_lazy_openapi_snapshot.json b/litellm/proxy/_lazy_openapi_snapshot.json
index 74f38b3ca6d..cf547dc89d5 100644
--- a/litellm/proxy/_lazy_openapi_snapshot.json
+++ b/litellm/proxy/_lazy_openapi_snapshot.json
@@ -2357,6 +2357,20 @@
},
"AgentConfig": {
"properties": {
+ "access_group_ids": {
+ "anyOf": [
+ {
+ "items": {
+ "type": "string"
+ },
+ "type": "array"
+ },
+ {
+ "type": "null"
+ }
+ ],
+ "title": "Access Group Ids"
+ },
"agent_card_params": {
"$ref": "#/components/schemas/AgentCard"
},
@@ -2683,6 +2697,20 @@
},
"AgentResponse": {
"properties": {
+ "access_group_ids": {
+ "anyOf": [
+ {
+ "items": {
+ "type": "string"
+ },
+ "type": "array"
+ },
+ {
+ "type": "null"
+ }
+ ],
+ "title": "Access Group Ids"
+ },
"agent_card_params": {
"additionalProperties": true,
"title": "Agent Card Params",
@@ -3471,6 +3499,20 @@
},
"PatchAgentRequest": {
"properties": {
+ "access_group_ids": {
+ "anyOf": [
+ {
+ "items": {
+ "type": "string"
+ },
+ "type": "array"
+ },
+ {
+ "type": "null"
+ }
+ ],
+ "title": "Access Group Ids"
+ },
"agent_card_params": {
"$ref": "#/components/schemas/AgentCard"
},
diff --git a/litellm/proxy/_types.py b/litellm/proxy/_types.py
index 680c63393e8..3a7125acd1f 100644
--- a/litellm/proxy/_types.py
+++ b/litellm/proxy/_types.py
@@ -4127,6 +4127,11 @@ class ProxyErrorTypes(str, enum.Enum):
Project does not have access to the model
"""
+ agent_model_access_denied = "agent_model_access_denied"
+ """
+ The agent behind the key does not have access to the model
+ """
+
model_cost_map_missing = "model_cost_map_missing"
expired_key = "expired_key"
@@ -4201,7 +4206,7 @@ class ProxyErrorTypes(str, enum.Enum):
@classmethod
def get_model_access_error_type_for_object(
- cls, object_type: Literal["key", "user", "team", "org", "project"]
+ cls, object_type: Literal["key", "user", "team", "org", "project", "agent"]
) -> "ProxyErrorTypes":
"""
Get the model access error type for object_type
@@ -4216,6 +4221,8 @@ class ProxyErrorTypes(str, enum.Enum):
return cls.org_model_access_denied
elif object_type == "project":
return cls.project_model_access_denied
+ elif object_type == "agent":
+ return cls.agent_model_access_denied
@classmethod
def get_vector_store_access_error_type_for_object(
diff --git a/litellm/proxy/agent_endpoints/agent_registry.py b/litellm/proxy/agent_endpoints/agent_registry.py
index c7b6bca72cf..c8948d8d70e 100644
--- a/litellm/proxy/agent_endpoints/agent_registry.py
+++ b/litellm/proxy/agent_endpoints/agent_registry.py
@@ -7,6 +7,7 @@ from types import MappingProxyType
from typing import TYPE_CHECKING, Final, NamedTuple, Protocol, TypedDict
from pydantic import TypeAdapter, ValidationError
+from typing_extensions import ReadOnly
import litellm
from litellm.constants import REDACTED_BY_LITELM_STRING
@@ -37,6 +38,7 @@ class AgentRecordDump(TypedDict):
agent_card_params: dict[str, object]
static_headers: dict[str, str] | None
extra_headers: list[str] | None
+ access_group_ids: ReadOnly[Sequence[str] | None]
object_permission: dict[str, object] | None
spend: float
tpm_limit: int | None
@@ -284,6 +286,12 @@ def _resolved_agent_param_value(
return _MISSING_AGENT_PARAM
+def _patched_access_group_ids(agent: PatchAgentRequest) -> Mapping[str, object]:
+ if "access_group_ids" not in agent:
+ return MappingProxyType({})
+ return MappingProxyType({"access_group_ids": tuple(dict.fromkeys(agent.get("access_group_ids") or ()))})
+
+
def _restore_redacted_litellm_params(
incoming: Mapping[str, object],
existing: Mapping[str, object],
@@ -516,6 +524,7 @@ class AgentRegistry:
static_headers_val: Final[str | None] = safe_dumps(dict(static_headers_obj)) if static_headers_obj else None
extra_headers_val: Final = agent.get("extra_headers")
+ access_group_ids_val: Final = agent.get("access_group_ids")
create_data: Final[dict[str, object]] = {
"agent_name": agent_name,
@@ -532,6 +541,8 @@ class AgentRegistry:
create_data["static_headers"] = static_headers_val
if extra_headers_val is not None:
create_data["extra_headers"] = extra_headers_val
+ if access_group_ids_val is not None:
+ create_data["access_group_ids"] = tuple(dict.fromkeys(access_group_ids_val))
if object_permission_id is not None:
create_data["object_permission_id"] = object_permission_id
@@ -601,7 +612,7 @@ class AgentRegistry:
existing_agent: Final[Mapping[str, object]] = dict(existing_record)
augment_agent: Final = {**existing_agent, **agent}
- update_data: Final[dict[str, object]] = {}
+ update_data: Final[dict[str, object]] = {**_patched_access_group_ids(agent)}
if augment_agent.get("agent_name"):
update_data["agent_name"] = augment_agent.get("agent_name")
if "litellm_params" in agent:
@@ -703,6 +714,7 @@ class AgentRegistry:
safe_dumps(dict(static_headers_obj_u)) if static_headers_obj_u is not None else safe_dumps({})
)
extra_headers_val_u: Final = agent.get("extra_headers") or []
+ access_group_ids_val_u: Final = tuple(dict.fromkeys(agent.get("access_group_ids") or ()))
update_data: Final[dict[str, object]] = {
"agent_name": agent_name,
@@ -710,6 +722,7 @@ class AgentRegistry:
"agent_card_params": agent_card_params,
"static_headers": static_headers_val_u,
"extra_headers": extra_headers_val_u,
+ "access_group_ids": access_group_ids_val_u,
"updated_by": updated_by,
"updated_at": datetime.now(timezone.utc),
}
diff --git a/litellm/proxy/agent_endpoints/auth/agent_access_groups.py b/litellm/proxy/agent_endpoints/auth/agent_access_groups.py
new file mode 100644
index 00000000000..4a579de679e
--- /dev/null
+++ b/litellm/proxy/agent_endpoints/auth/agent_access_groups.py
@@ -0,0 +1,80 @@
+"""
+Ceiling that an agent's attached access groups place on requests made with that agent's key.
+
+Keys and teams use access groups as grants. An agent uses them the way it already uses its
+``object_permission``: the union of the attached groups caps what the agent's key can reach,
+on top of whatever the key and team allow. A group that cannot be loaded contributes nothing,
+so a missing or unreadable group can only narrow the agent, never widen it.
+"""
+
+import asyncio
+from collections.abc import Awaitable, Callable
+from dataclasses import dataclass
+from typing import Final, TypeAlias
+
+from fastapi import HTTPException
+
+from litellm._logging import verbose_proxy_logger
+from litellm.proxy._types import LiteLLM_AccessGroupTable
+from litellm.types.agents import AgentResponse
+
+AgentLoader: TypeAlias = Callable[[str], Awaitable[AgentResponse | None]]
+AccessGroupLoader: TypeAlias = Callable[[str], Awaitable[LiteLLM_AccessGroupTable | None]]
+
+
+@dataclass(frozen=True, slots=True)
+class AgentAccessGroupCeiling:
+ """Everything the agent's attached access groups allow. An empty set denies that resource kind."""
+
+ access_group_ids: tuple[str, ...]
+ models: frozenset[str]
+ mcp_server_ids: frozenset[str]
+ agent_ids: frozenset[str]
+
+
+async def _load_agent(agent_id: str) -> AgentResponse | None:
+ from litellm.proxy.common_utils.registry_read_through import get_agent_with_read_through
+
+ return await get_agent_with_read_through(agent_id)
+
+
+async def _load_access_group(access_group_id: str) -> LiteLLM_AccessGroupTable | None:
+ from litellm.proxy.auth.auth_checks import get_access_object
+ from litellm.proxy.proxy_server import prisma_client, proxy_logging_obj, user_api_key_cache
+
+ if prisma_client is None:
+ verbose_proxy_logger.warning("Agent access group %s cannot be loaded without a DB", access_group_id)
+ return None
+ try:
+ return await get_access_object(
+ access_group_id=access_group_id,
+ prisma_client=prisma_client,
+ user_api_key_cache=user_api_key_cache,
+ proxy_logging_obj=proxy_logging_obj,
+ )
+ except HTTPException as e:
+ verbose_proxy_logger.warning(
+ "Agent access group %s could not be loaded, treating it as empty: %s", access_group_id, e.detail
+ )
+ return None
+
+
+async def resolve_agent_access_group_ceiling(
+ agent_id: str,
+ load_agent: AgentLoader = _load_agent,
+ load_access_group: AccessGroupLoader = _load_access_group,
+) -> AgentAccessGroupCeiling | None:
+ """``None`` when the agent has no access groups attached, so nothing is capped."""
+ agent: Final = await load_agent(agent_id)
+ access_group_ids: Final = tuple(agent.access_group_ids or ()) if agent is not None else ()
+ if not access_group_ids:
+ return None
+
+ loaded: Final = await asyncio.gather(*(load_access_group(group_id) for group_id in access_group_ids))
+ groups: Final = tuple(group for group in loaded if group is not None)
+ return AgentAccessGroupCeiling(
+ access_group_ids=access_group_ids,
+ models=frozenset(model for group in groups for model in group.access_model_names),
+ mcp_server_ids=frozenset(server_id for group in groups for server_id in group.access_mcp_server_ids),
+ agent_ids=frozenset(target_id for group in groups for target_id in group.access_agent_ids),
+ )
diff --git a/litellm/proxy/agent_endpoints/auth/agent_permission_handler.py b/litellm/proxy/agent_endpoints/auth/agent_permission_handler.py
index e4dd77e2f82..11d2a68072c 100644
--- a/litellm/proxy/agent_endpoints/auth/agent_permission_handler.py
+++ b/litellm/proxy/agent_endpoints/auth/agent_permission_handler.py
@@ -66,9 +66,24 @@ class AgentRequestHandler:
Resolve the agents the given user/key may reach.
``UnrestrictedAgentAccess`` is only returned when neither the key nor its team
- carries any grant. Grants that intersect to nothing stay restricted, so
- narrowing a caller can never widen what it reaches.
+ carries any grant and the agent behind the key has no access groups attached.
+ Grants that intersect to nothing stay restricted, so narrowing a caller can
+ never widen what it reaches.
"""
+ key_team_access: Final = await AgentRequestHandler._resolve_key_team_agent_access(user_api_key_auth)
+ agent_ceiling: Final = await AgentRequestHandler._agent_access_group_ceiling(user_api_key_auth)
+ if agent_ceiling is None:
+ return key_team_access
+ match key_team_access:
+ case UnrestrictedAgentAccess():
+ return RestrictedAgentAccess(agent_ceiling)
+ case RestrictedAgentAccess(key_team_ids):
+ return RestrictedAgentAccess(key_team_ids & agent_ceiling)
+
+ @staticmethod
+ async def _resolve_key_team_agent_access(
+ user_api_key_auth: UserAPIKeyAuth | None,
+ ) -> AgentAccess:
try:
key_access: Final = await AgentRequestHandler._get_allowed_agents_for_key(user_api_key_auth)
team_access: Final = await AgentRequestHandler._get_allowed_agents_for_team(user_api_key_auth)
@@ -86,6 +101,20 @@ class AgentRequestHandler:
verbose_logger.warning("Failed to get allowed agents: %s", e)
return UnrestrictedAgentAccess()
+ @staticmethod
+ async def _agent_access_group_ceiling(
+ user_api_key_auth: UserAPIKeyAuth | None,
+ ) -> frozenset[str] | None:
+ """Stable IDs of the agents the calling agent's attached access groups allow; None when none attached."""
+ from litellm.proxy.agent_endpoints.auth.agent_access_groups import resolve_agent_access_group_ceiling
+
+ if user_api_key_auth is None or not user_api_key_auth.agent_id:
+ return None
+ ceiling: Final = await resolve_agent_access_group_ceiling(user_api_key_auth.agent_id)
+ if ceiling is None:
+ return None
+ return _to_stable_ids(ceiling.agent_ids)
+
@staticmethod
async def is_agent_allowed(
agent_id: str,
diff --git a/litellm/proxy/auth/auth_checks.py b/litellm/proxy/auth/auth_checks.py
index 3dd2e2d8eb2..5c53ee49717 100644
--- a/litellm/proxy/auth/auth_checks.py
+++ b/litellm/proxy/auth/auth_checks.py
@@ -1003,6 +1003,9 @@ async def common_checks(
code=status.HTTP_400_BAD_REQUEST,
)
+ # 2.4 If the agent behind the key has access groups attached, they cap the models it can call
+ await _check_agent_access_group_model_access(model=_model, valid_token=valid_token, llm_router=llm_router)
+
## 2.1 If user can call model (if personal key)
if _model and team_object is None and user_object is not None:
with tracer.trace("litellm.proxy.auth.common_checks.can_user_call_model"):
@@ -4126,7 +4129,7 @@ def _can_object_call_model(
models: list[str],
team_model_aliases: dict[str, str] | None = None,
team_id: str | None = None,
- object_type: Literal["user", "team", "key", "org", "project"] = "user",
+ object_type: Literal["user", "team", "key", "org", "project", "agent"] = "user",
fallback_depth: int = 0,
) -> Literal[True]:
"""
@@ -4192,6 +4195,38 @@ def _can_object_call_model(
)
+async def _check_agent_access_group_model_access(
+ model: str | list[str] | None, # mutable-ok: _can_object_call_model and the client message helper take list[str]
+ valid_token: UserAPIKeyAuth | None,
+ llm_router: Router | None,
+) -> Literal[True]:
+ """Raises when the key's agent has access groups attached and none of them names the model.
+ Attached groups that name no model deny every model; ``_can_object_call_model`` would read
+ an empty allowlist as unrestricted."""
+ from litellm.proxy.agent_endpoints.auth.agent_access_groups import resolve_agent_access_group_ceiling
+
+ if not model or valid_token is None or not valid_token.agent_id:
+ return True
+ ceiling: Final = await resolve_agent_access_group_ceiling(valid_token.agent_id)
+ if ceiling is None:
+ return True
+ if not ceiling.models:
+ raise ModelAccessDeniedProxyException(
+ message=model_access_denied_client_message(model=model),
+ internal_message=f"agent {valid_token.agent_id} access groups {ceiling.access_group_ids} grant no models",
+ type=ProxyErrorTypes.agent_model_access_denied,
+ param="model",
+ code=status.HTTP_403_FORBIDDEN,
+ )
+ return _can_object_call_model(
+ model=model,
+ llm_router=llm_router,
+ models=sorted(ceiling.models),
+ team_id=valid_token.team_id,
+ object_type="agent",
+ )
+
+
def _model_in_team_aliases(model: str, team_model_aliases: dict[str, str] | None = None) -> bool:
"""
Returns True if `model` being accessed is an alias of a team model
diff --git a/litellm/proxy/management_endpoints/access_group_endpoints.py b/litellm/proxy/management_endpoints/access_group_endpoints.py
index a6cc5140b15..b4923b0a2dc 100644
--- a/litellm/proxy/management_endpoints/access_group_endpoints.py
+++ b/litellm/proxy/management_endpoints/access_group_endpoints.py
@@ -5,6 +5,7 @@ from types import MappingProxyType
from typing import Final, Protocol
from fastapi import APIRouter, Depends, HTTPException, status
+from typing_extensions import ReadOnly, TypedDict
from litellm._logging import verbose_proxy_logger
from litellm.proxy._experimental.mcp_server.mcp_server_manager import global_mcp_server_manager
@@ -109,6 +110,36 @@ class _KeyTable(Protocol):
async def update(self, where: Mapping[str, object], data: Mapping[str, object]) -> object: ...
+class _AgentRecord(Protocol):
+ @property
+ def agent_id(self) -> str: ...
+
+ @property
+ def access_group_ids(self) -> Sequence[str] | None: ...
+
+
+class _AgentTable(Protocol):
+ async def find_many(self, where: Mapping[str, object]) -> Sequence[_AgentRecord]: ...
+
+ async def update(self, where: Mapping[str, object], data: Mapping[str, object]) -> object: ...
+
+
+class _HasSomeFilter(TypedDict):
+ hasSome: ReadOnly[Sequence[str]]
+
+
+class _AgentAccessGroupsWhere(TypedDict):
+ access_group_ids: ReadOnly[_HasSomeFilter]
+
+
+class _AgentIdWhere(TypedDict):
+ agent_id: ReadOnly[str]
+
+
+class _AgentAccessGroupsData(TypedDict):
+ access_group_ids: ReadOnly[Sequence[str]]
+
+
class _AccessGroupTx(Protocol):
@property
def litellm_accessgrouptable(self) -> _AccessGroupTable: ...
@@ -119,6 +150,9 @@ class _AccessGroupTx(Protocol):
@property
def litellm_verificationtoken(self) -> _KeyTable: ...
+ @property
+ def litellm_agentstable(self) -> _AgentTable: ...
+
def _require_proxy_admin(user_api_key_dict: UserAPIKeyAuth) -> None:
if user_api_key_dict.user_role != LitellmUserRoles.PROXY_ADMIN:
@@ -324,6 +358,41 @@ async def _sync_remove_access_group_from_keys(tx: _AccessGroupTx, key_tokens: li
)
+def _without_access_group(access_group_ids: Sequence[str] | None, access_group_id: str) -> tuple[str, ...]:
+ return tuple(ag for ag in (access_group_ids or ()) if ag != access_group_id)
+
+
+async def _detach_access_group_from_agents(tx: _AccessGroupTx, access_group_id: str) -> tuple[str, ...]:
+ agents_with_group: Final = await tx.litellm_agentstable.find_many(
+ where=_AgentAccessGroupsWhere(access_group_ids=_HasSomeFilter(hasSome=(access_group_id,)))
+ )
+ for agent in agents_with_group:
+ await tx.litellm_agentstable.update(
+ where=_AgentIdWhere(agent_id=agent.agent_id),
+ data=_AgentAccessGroupsData(
+ access_group_ids=_without_access_group(agent.access_group_ids, access_group_id)
+ ),
+ )
+ return tuple(agent.agent_id for agent in agents_with_group)
+
+
+def _detach_access_group_from_agent_registry(agent_ids: Sequence[str], access_group_id: str) -> None:
+ registered: Final = tuple(
+ agent
+ for agent in (global_agent_registry.get_agent_by_id(agent_id) for agent_id in agent_ids)
+ if agent is not None
+ )
+ for agent in registered:
+ global_agent_registry.deregister_agent(agent_name=agent.agent_name)
+ global_agent_registry.register_agent(
+ agent_config=agent.model_copy(
+ update=_AgentAccessGroupsData(
+ access_group_ids=_without_access_group(agent.access_group_ids, access_group_id)
+ )
+ )
+ )
+
+
# ---------------------------------------------------------------------------
# Cache patch helpers
# ---------------------------------------------------------------------------
@@ -705,11 +774,14 @@ async def delete_access_group(
out_of_sync_key_tokens: Final = set(existing.assigned_key_ids or []) - {k.token for k in keys_with_group}
await _sync_remove_access_group_from_keys(tx, list(out_of_sync_key_tokens), access_group_id)
+ detached_agent_ids: Final = await _detach_access_group_from_agents(tx, access_group_id)
+
await tx.litellm_accessgrouptable.delete(where={"access_group_id": access_group_id})
from litellm.proxy.proxy_server import proxy_logging_obj, user_api_key_cache
await invalidate_access_group_cache(access_group_id)
+ _detach_access_group_from_agent_registry(detached_agent_ids, access_group_id)
await _patch_team_caches_remove_access_group(
affected_team_ids, access_group_id, user_api_key_cache, proxy_logging_obj
)
diff --git a/litellm/proxy/schema.prisma b/litellm/proxy/schema.prisma
index 139fb031671..e0b52dd77de 100644
--- a/litellm/proxy/schema.prisma
+++ b/litellm/proxy/schema.prisma
@@ -71,6 +71,7 @@ model LiteLLM_AgentsTable {
static_headers Json? @default("{}")
extra_headers String[] @default([])
agent_access_groups String[] @default([])
+ access_group_ids String[] @default([])
object_permission_id String?
object_permission LiteLLM_ObjectPermissionTable? @relation(fields: [object_permission_id], references: [object_permission_id])
spend Float @default(0.0)
diff --git a/litellm/types/agents.py b/litellm/types/agents.py
index dbaaab62d86..12e60352a97 100644
--- a/litellm/types/agents.py
+++ b/litellm/types/agents.py
@@ -189,6 +189,7 @@ class AgentConfig(TypedDict, total=False):
session_rpm_limit: int | None
static_headers: dict[str, str] | None
extra_headers: list[str] | None
+ access_group_ids: ReadOnly[Sequence[str] | None]
class PatchAgentRequest(TypedDict, total=False):
@@ -202,6 +203,7 @@ class PatchAgentRequest(TypedDict, total=False):
session_rpm_limit: int | None
static_headers: dict[str, str] | None
extra_headers: list[str] | None
+ access_group_ids: ReadOnly[Sequence[str] | None]
# Request/Response models for CRUD endpoints
@@ -226,6 +228,7 @@ class AgentResponse(BaseModel):
session_rpm_limit: int | None = None
static_headers: dict[str, str] | None = None
extra_headers: list[str] | None = None
+ access_group_ids: Sequence[str] | None = None
keys: list[AgentKeySummary] | None = None
search_score: float | None = None
created_at: datetime | None = None
diff --git a/schema.prisma b/schema.prisma
index 139fb031671..e0b52dd77de 100644
--- a/schema.prisma
+++ b/schema.prisma
@@ -71,6 +71,7 @@ model LiteLLM_AgentsTable {
static_headers Json? @default("{}")
extra_headers String[] @default([])
agent_access_groups String[] @default([])
+ access_group_ids String[] @default([])
object_permission_id String?
object_permission LiteLLM_ObjectPermissionTable? @relation(fields: [object_permission_id], references: [object_permission_id])
spend Float @default(0.0)
diff --git a/tests/test_litellm/proxy/_experimental/mcp_server/auth/test_user_api_key_auth_mcp.py b/tests/test_litellm/proxy/_experimental/mcp_server/auth/test_user_api_key_auth_mcp.py
index 90ce821d62e..2f8e3d1cb82 100644
--- a/tests/test_litellm/proxy/_experimental/mcp_server/auth/test_user_api_key_auth_mcp.py
+++ b/tests/test_litellm/proxy/_experimental/mcp_server/auth/test_user_api_key_auth_mcp.py
@@ -4208,6 +4208,65 @@ class TestAgentMCPPermissions:
assert sorted(result) == ["server_1", "server_2"]
mock_agent.assert_called_once_with(user_api_key_auth)
+ @pytest.mark.parametrize(
+ ("group_ceiling", "expected"),
+ [
+ (frozenset({"server_1"}), ["server_1"]),
+ (frozenset({"server_1", "server_2", "server_3"}), ["server_1", "server_2"]),
+ (frozenset(), []),
+ ],
+ )
+ async def test_get_allowed_mcp_servers_agent_access_group_ceiling(self, group_ceiling, expected):
+ """The agent's attached access groups cap the key/team servers; groups naming no server deny all."""
+ user_api_key_auth = UserAPIKeyAuth(api_key="test-key", user_id="test-user", agent_id="agent-ag")
+ with (
+ patch.object(MCPRequestHandler, "_get_allowed_mcp_servers_for_key", return_value=["server_1", "server_2"]),
+ patch.object(MCPRequestHandler, "_get_allowed_mcp_servers_for_team", return_value=[]),
+ patch.object(MCPRequestHandler, "_get_allowed_mcp_servers_for_agent", return_value=[]),
+ patch.object(MCPRequestHandler, "_get_agent_access_group_server_ceiling", return_value=group_ceiling),
+ ):
+ access = await MCPRequestHandler.get_mcp_server_access(user_api_key_auth=user_api_key_auth)
+ assert sorted(access.server_ids) == expected
+ assert access.scope == "scoped"
+
+ async def test_get_allowed_mcp_servers_agent_without_access_groups_is_uncapped(self):
+ user_api_key_auth = UserAPIKeyAuth(api_key="test-key", user_id="test-user", agent_id="agent-ag")
+ with (
+ patch.object(MCPRequestHandler, "_get_allowed_mcp_servers_for_key", return_value=["server_1", "server_2"]),
+ patch.object(MCPRequestHandler, "_get_allowed_mcp_servers_for_team", return_value=[]),
+ patch.object(MCPRequestHandler, "_get_allowed_mcp_servers_for_agent", return_value=[]),
+ patch.object(MCPRequestHandler, "_get_agent_access_group_server_ceiling", return_value=None),
+ ):
+ result = await MCPRequestHandler.get_allowed_mcp_servers(user_api_key_auth=user_api_key_auth)
+ assert sorted(result) == ["server_1", "server_2"]
+
+ async def test_agent_access_group_server_ceiling_expands_group_servers(self):
+ from litellm.proxy.agent_endpoints.auth.agent_access_groups import AgentAccessGroupCeiling
+
+ ceiling = AgentAccessGroupCeiling(
+ access_group_ids=("ag-1",),
+ models=frozenset(),
+ mcp_server_ids=frozenset({"server_1"}),
+ agent_ids=frozenset(),
+ )
+ with (
+ patch(
+ "litellm.proxy.agent_endpoints.auth.agent_access_groups.resolve_agent_access_group_ceiling",
+ new=AsyncMock(return_value=ceiling),
+ ),
+ patch(
+ "litellm.proxy._experimental.mcp_server.mcp_server_manager.global_mcp_server_manager"
+ ) as mock_manager,
+ ):
+ mock_manager.expand_permission_list.return_value = ["server_1"]
+ result = await MCPRequestHandler._get_agent_access_group_server_ceiling(
+ UserAPIKeyAuth(api_key="test-key", agent_id="agent-ag")
+ )
+ assert result == frozenset({"server_1"})
+ mock_manager.expand_permission_list.assert_called_once_with(["server_1"])
+
+ assert await MCPRequestHandler._get_agent_access_group_server_ceiling(UserAPIKeyAuth(api_key="k")) is None
+
async def test_get_allowed_mcp_servers_key_team_agent_intersection(self):
"""Key allows [1, 2], agent allows [2, 3]. Result = [2]."""
user_api_key_auth = UserAPIKeyAuth(
diff --git a/tests/test_litellm/proxy/agent_endpoints/auth/test_agent_access_groups.py b/tests/test_litellm/proxy/agent_endpoints/auth/test_agent_access_groups.py
new file mode 100644
index 00000000000..8c31c9428e7
--- /dev/null
+++ b/tests/test_litellm/proxy/agent_endpoints/auth/test_agent_access_groups.py
@@ -0,0 +1,130 @@
+from typing import Final
+
+import pytest
+from fastapi import HTTPException
+
+from litellm.models.access_group import LiteLLM_AccessGroupTable
+from litellm.proxy.agent_endpoints.auth.agent_access_groups import (
+ AgentAccessGroupCeiling,
+ resolve_agent_access_group_ceiling,
+)
+from litellm.types.agents import AgentResponse
+
+_CARD: Final = {"name": "agent", "url": "http://localhost:9999", "version": "1.0.0"}
+
+
+def _agent(access_group_ids: list[str] | None) -> AgentResponse:
+ return AgentResponse(
+ agent_id="agent-1", agent_name="agent", agent_card_params=_CARD, access_group_ids=access_group_ids
+ )
+
+
+def _group(
+ group_id: str,
+ models: tuple[str, ...] = (),
+ mcp_servers: tuple[str, ...] = (),
+ agents: tuple[str, ...] = (),
+) -> LiteLLM_AccessGroupTable:
+ return LiteLLM_AccessGroupTable(
+ access_group_id=group_id,
+ access_group_name=group_id,
+ access_model_names=list(models),
+ access_mcp_server_ids=list(mcp_servers),
+ access_agent_ids=list(agents),
+ )
+
+
+def _loaders(agent: AgentResponse | None, groups: dict[str, LiteLLM_AccessGroupTable]):
+ async def load_agent(agent_id: str) -> AgentResponse | None:
+ return agent
+
+ async def load_group(group_id: str) -> LiteLLM_AccessGroupTable | None:
+ return groups.get(group_id)
+
+ return load_agent, load_group
+
+
+@pytest.mark.asyncio
+@pytest.mark.parametrize("access_group_ids", [None, []])
+async def test_agent_without_access_groups_has_no_ceiling(access_group_ids: list[str] | None):
+ load_agent, load_group = _loaders(_agent(access_group_ids), {"g1": _group("g1", models=("gpt-5",))})
+
+ assert await resolve_agent_access_group_ceiling("agent-1", load_agent, load_group) is None
+
+
+@pytest.mark.asyncio
+async def test_unknown_agent_has_no_ceiling():
+ load_agent, load_group = _loaders(None, {})
+
+ assert await resolve_agent_access_group_ceiling("missing", load_agent, load_group) is None
+
+
+@pytest.mark.asyncio
+async def test_ceiling_is_the_union_of_every_attached_group():
+ load_agent, load_group = _loaders(
+ _agent(["g1", "g2"]),
+ {
+ "g1": _group("g1", models=("gpt-5",), mcp_servers=("mcp-a",), agents=("agent-b",)),
+ "g2": _group("g2", models=("claude-sonnet",), mcp_servers=("mcp-b",), agents=("agent-c",)),
+ },
+ )
+
+ ceiling: Final = await resolve_agent_access_group_ceiling("agent-1", load_agent, load_group)
+
+ assert ceiling == AgentAccessGroupCeiling(
+ access_group_ids=("g1", "g2"),
+ models=frozenset({"gpt-5", "claude-sonnet"}),
+ mcp_server_ids=frozenset({"mcp-a", "mcp-b"}),
+ agent_ids=frozenset({"agent-b", "agent-c"}),
+ )
+
+
+@pytest.mark.asyncio
+async def test_unloadable_group_contributes_nothing_but_the_ceiling_still_applies():
+ load_agent, load_group = _loaders(_agent(["g1", "gone"]), {"g1": _group("g1", models=("gpt-5",))})
+
+ ceiling: Final = await resolve_agent_access_group_ceiling("agent-1", load_agent, load_group)
+
+ assert ceiling == AgentAccessGroupCeiling(
+ access_group_ids=("g1", "gone"),
+ models=frozenset({"gpt-5"}),
+ mcp_server_ids=frozenset(),
+ agent_ids=frozenset(),
+ )
+
+
+@pytest.mark.asyncio
+async def test_only_unloadable_groups_is_an_empty_ceiling_not_unrestricted():
+ load_agent, load_group = _loaders(_agent(["gone"]), {})
+
+ ceiling: Final = await resolve_agent_access_group_ceiling("agent-1", load_agent, load_group)
+
+ assert ceiling is not None
+ assert ceiling.models == frozenset()
+ assert ceiling.mcp_server_ids == frozenset()
+ assert ceiling.agent_ids == frozenset()
+
+
+@pytest.mark.asyncio
+async def test_default_loader_treats_a_missing_group_as_unreadable(monkeypatch: pytest.MonkeyPatch):
+ from litellm.proxy import proxy_server
+ from litellm.proxy.agent_endpoints.auth.agent_access_groups import _load_access_group
+ from litellm.proxy.auth import auth_checks
+
+ async def missing_group(**_: object) -> LiteLLM_AccessGroupTable:
+ raise HTTPException(status_code=404, detail={"error": "Access group doesn't exist in db."})
+
+ monkeypatch.setattr(proxy_server, "prisma_client", object())
+ monkeypatch.setattr(auth_checks, "get_access_object", missing_group)
+
+ assert await _load_access_group("gone") is None
+
+
+@pytest.mark.asyncio
+async def test_default_loader_returns_nothing_without_a_db(monkeypatch: pytest.MonkeyPatch):
+ from litellm.proxy import proxy_server
+ from litellm.proxy.agent_endpoints.auth.agent_access_groups import _load_access_group
+
+ monkeypatch.setattr(proxy_server, "prisma_client", None)
+
+ assert await _load_access_group("ag-1") is None
diff --git a/tests/test_litellm/proxy/agent_endpoints/auth/test_agent_permission_handler.py b/tests/test_litellm/proxy/agent_endpoints/auth/test_agent_permission_handler.py
index 383b72e5c58..a8a55d332b6 100644
--- a/tests/test_litellm/proxy/agent_endpoints/auth/test_agent_permission_handler.py
+++ b/tests/test_litellm/proxy/agent_endpoints/auth/test_agent_permission_handler.py
@@ -13,6 +13,7 @@ import pytest
from litellm.constants import UI_SESSION_TOKEN_TEAM_ID
from litellm.proxy._types import LitellmUserRoles, UserAPIKeyAuth
from litellm.proxy.agent_endpoints.agent_registry import AgentRegistry
+from litellm.proxy.agent_endpoints.auth.agent_access_groups import AgentAccessGroupCeiling
from litellm.proxy.agent_endpoints.auth.agent_permission_handler import (
AgentAccess,
AgentRequestHandler,
@@ -157,6 +158,88 @@ class TestAgentRequestHandler:
is False
), agent_id
+ @staticmethod
+ def _ceiling(agent_ids: frozenset[str]) -> AgentAccessGroupCeiling:
+ return AgentAccessGroupCeiling(
+ access_group_ids=("ag-1",),
+ models=frozenset(),
+ mcp_server_ids=frozenset(),
+ agent_ids=agent_ids,
+ )
+
+ async def test_agent_access_groups_cap_an_otherwise_unrestricted_key(self):
+ """A key with no agent grant of its own may still only reach the agents its
+ agent's attached access groups name."""
+ agent_key: Final = UserAPIKeyAuth(api_key="test-key", user_id="test-user", agent_id="caller-agent")
+
+ with (
+ patch.object(AgentRequestHandler, "_get_allowed_agents_for_key", return_value=UnrestrictedAgentAccess()),
+ patch.object(AgentRequestHandler, "_get_allowed_agents_for_team", return_value=UnrestrictedAgentAccess()),
+ patch(
+ "litellm.proxy.agent_endpoints.auth.agent_access_groups.resolve_agent_access_group_ceiling",
+ new=AsyncMock(return_value=self._ceiling(frozenset({"agent-beta"}))),
+ ) as mock_ceiling,
+ ):
+ assert await AgentRequestHandler.resolve_agent_access(agent_key) == RestrictedAgentAccess(
+ frozenset({"agent-beta"})
+ )
+ assert await AgentRequestHandler.is_agent_allowed("agent-beta", agent_key) is True
+ assert await AgentRequestHandler.is_agent_allowed("agent-alpha", agent_key) is False
+ mock_ceiling.assert_called_with("caller-agent")
+
+ async def test_agent_access_groups_intersect_with_key_and_team_grants(self):
+ agent_key: Final = UserAPIKeyAuth(
+ api_key="test-key", user_id="test-user", team_id="test-team", agent_id="caller-agent"
+ )
+
+ with (
+ patch.object(
+ AgentRequestHandler,
+ "_get_allowed_agents_for_key",
+ return_value=RestrictedAgentAccess(frozenset({"agent-alpha", "agent-beta"})),
+ ),
+ patch.object(
+ AgentRequestHandler,
+ "_get_allowed_agents_for_team",
+ return_value=RestrictedAgentAccess(frozenset({"agent-alpha", "agent-beta", "agent-gamma"})),
+ ),
+ patch(
+ "litellm.proxy.agent_endpoints.auth.agent_access_groups.resolve_agent_access_group_ceiling",
+ new=AsyncMock(return_value=self._ceiling(frozenset({"agent-beta", "agent-gamma"}))),
+ ),
+ ):
+ assert await AgentRequestHandler.resolve_agent_access(agent_key) == RestrictedAgentAccess(
+ frozenset({"agent-beta"})
+ )
+
+ async def test_agent_access_groups_naming_no_agent_deny_every_agent(self):
+ agent_key: Final = UserAPIKeyAuth(api_key="test-key", user_id="test-user", agent_id="caller-agent")
+
+ with (
+ patch.object(AgentRequestHandler, "_get_allowed_agents_for_key", return_value=UnrestrictedAgentAccess()),
+ patch.object(AgentRequestHandler, "_get_allowed_agents_for_team", return_value=UnrestrictedAgentAccess()),
+ patch(
+ "litellm.proxy.agent_endpoints.auth.agent_access_groups.resolve_agent_access_group_ceiling",
+ new=AsyncMock(return_value=self._ceiling(frozenset())),
+ ),
+ ):
+ assert await AgentRequestHandler.resolve_agent_access(agent_key) == RestrictedAgentAccess(frozenset())
+ assert await AgentRequestHandler.is_agent_allowed("agent-alpha", agent_key) is False
+
+ async def test_key_without_agent_never_consults_agent_access_groups(self):
+ plain_key: Final = UserAPIKeyAuth(api_key="test-key", user_id="test-user")
+
+ with (
+ patch.object(AgentRequestHandler, "_get_allowed_agents_for_key", return_value=UnrestrictedAgentAccess()),
+ patch.object(AgentRequestHandler, "_get_allowed_agents_for_team", return_value=UnrestrictedAgentAccess()),
+ patch(
+ "litellm.proxy.agent_endpoints.auth.agent_access_groups.resolve_agent_access_group_ceiling",
+ new=AsyncMock(return_value=self._ceiling(frozenset())),
+ ) as mock_ceiling,
+ ):
+ assert await AgentRequestHandler.resolve_agent_access(plain_key) == UnrestrictedAgentAccess()
+ mock_ceiling.assert_not_called()
+
async def test_empty_access_group_denies_every_agent(self):
"""LIT-5143: a key restricted to an access group that resolves to no agents is
restricted to nothing, not unrestricted. A failed group lookup still fails open."""
diff --git a/tests/test_litellm/proxy/agent_endpoints/test_agent_registry.py b/tests/test_litellm/proxy/agent_endpoints/test_agent_registry.py
index 231626c7eb5..d15a3adadbd 100644
--- a/tests/test_litellm/proxy/agent_endpoints/test_agent_registry.py
+++ b/tests/test_litellm/proxy/agent_endpoints/test_agent_registry.py
@@ -990,3 +990,136 @@ async def test_patch_agent_in_db_preserves_secret_when_echoed_back_redacted():
stored_params: Final = json.loads(mock_update.call_args.kwargs["data"]["litellm_params"])
assert stored_params["aws_secret_access_key"] == SENTINEL_AWS_SECRET_ACCESS_KEY
assert stored_params["is_public"] is True
+
+
+def _agent_row_mock(access_group_ids: list[str]) -> MagicMock:
+ row: Final = MagicMock()
+ row.model_dump.return_value = {
+ "agent_id": "agent-123",
+ "agent_name": "Test Agent",
+ "agent_card_params": _sample_agent_card_params(),
+ "litellm_params": {},
+ "object_permission": None,
+ "access_group_ids": access_group_ids,
+ }
+ row.object_permission = None
+ return row
+
+
+@pytest.mark.asyncio
+async def test_add_agent_to_db_persists_deduplicated_access_group_ids():
+ registry: Final = AgentRegistry()
+ mock_prisma: Final = MagicMock()
+ mock_create = AsyncMock(return_value=_agent_row_mock(["ag-1", "ag-2"]))
+ mock_prisma.db.litellm_agentstable.create = mock_create
+
+ result: Final = await registry.add_agent_to_db(
+ agent={
+ "agent_name": "Test Agent",
+ "agent_card_params": _sample_agent_card_params(),
+ "access_group_ids": ["ag-1", "ag-2", "ag-1"],
+ },
+ prisma_client=mock_prisma,
+ created_by="test-user",
+ )
+
+ assert tuple(mock_create.call_args.kwargs["data"]["access_group_ids"]) == ("ag-1", "ag-2")
+ assert result.access_group_ids == ["ag-1", "ag-2"]
+
+
+@pytest.mark.asyncio
+async def test_add_agent_to_db_without_access_group_ids_leaves_column_to_its_default():
+ registry: Final = AgentRegistry()
+ mock_prisma: Final = MagicMock()
+ mock_create = AsyncMock(return_value=_agent_row_mock([]))
+ mock_prisma.db.litellm_agentstable.create = mock_create
+
+ await registry.add_agent_to_db(
+ agent={"agent_name": "Test Agent", "agent_card_params": _sample_agent_card_params()},
+ prisma_client=mock_prisma,
+ created_by="test-user",
+ )
+
+ assert "access_group_ids" not in mock_create.call_args.kwargs["data"]
+
+
+@pytest.mark.asyncio
+@pytest.mark.parametrize(
+ ("patch_body", "expected"),
+ [
+ ({"access_group_ids": ["ag-2", "ag-3"]}, ["ag-2", "ag-3"]),
+ ({"access_group_ids": []}, []),
+ ({"access_group_ids": None}, []),
+ ],
+)
+async def test_patch_agent_in_db_replaces_access_group_ids_when_provided(patch_body: dict, expected: list[str]):
+ registry: Final = AgentRegistry()
+ mock_prisma: Final = MagicMock()
+ mock_prisma.db.litellm_agentstable.find_unique = AsyncMock(
+ return_value={
+ "agent_id": "agent-123",
+ "agent_name": "Test Agent",
+ "litellm_params": {},
+ "object_permission_id": None,
+ "access_group_ids": ["ag-1"],
+ }
+ )
+ mock_update = AsyncMock(return_value=_agent_row_mock(expected))
+ mock_prisma.db.litellm_agentstable.update = mock_update
+
+ await registry.patch_agent_in_db(
+ agent_id="agent-123", agent=patch_body, prisma_client=mock_prisma, updated_by="test-user"
+ )
+
+ assert tuple(mock_update.call_args.kwargs["data"]["access_group_ids"]) == tuple(expected)
+
+
+@pytest.mark.asyncio
+async def test_patch_agent_in_db_keeps_access_group_ids_when_omitted():
+ registry: Final = AgentRegistry()
+ mock_prisma: Final = MagicMock()
+ mock_prisma.db.litellm_agentstable.find_unique = AsyncMock(
+ return_value={
+ "agent_id": "agent-123",
+ "agent_name": "Old Name",
+ "litellm_params": {},
+ "object_permission_id": None,
+ "access_group_ids": ["ag-1"],
+ }
+ )
+ mock_update = AsyncMock(return_value=_agent_row_mock(["ag-1"]))
+ mock_prisma.db.litellm_agentstable.update = mock_update
+
+ await registry.patch_agent_in_db(
+ agent_id="agent-123", agent={"agent_name": "New Name"}, prisma_client=mock_prisma, updated_by="test-user"
+ )
+
+ assert "access_group_ids" not in mock_update.call_args.kwargs["data"]
+
+
+@pytest.mark.asyncio
+@pytest.mark.parametrize(
+ ("body_access_group_ids", "expected"),
+ [(["ag-9", "ag-9"], ["ag-9"]), (None, []), ("omitted", [])],
+)
+async def test_update_agent_in_db_always_writes_access_group_ids(body_access_group_ids, expected: list[str]):
+ """PUT is a full replacement: omitting the field clears any previously attached groups."""
+ registry: Final = AgentRegistry()
+ mock_prisma: Final = MagicMock()
+ mock_prisma.db.litellm_agentstable.find_unique = AsyncMock(
+ return_value=SimpleNamespace(litellm_params={}, object_permission_id=None, access_group_ids=["ag-1"])
+ )
+ mock_update = AsyncMock(return_value=_agent_row_mock(expected))
+ mock_prisma.db.litellm_agentstable.update = mock_update
+ body: Final = {
+ "agent_name": "Test Agent",
+ "agent_card_params": _sample_agent_card_params(),
+ "litellm_params": {"model": "bedrock/agentcore/my-agent"},
+ **({} if body_access_group_ids == "omitted" else {"access_group_ids": body_access_group_ids}),
+ }
+
+ await registry.update_agent_in_db(
+ agent_id="agent-123", agent=body, prisma_client=mock_prisma, updated_by="test-user"
+ )
+
+ assert tuple(mock_update.call_args.kwargs["data"]["access_group_ids"]) == tuple(expected)
diff --git a/tests/test_litellm/proxy/auth/test_auth_checks.py b/tests/test_litellm/proxy/auth/test_auth_checks.py
index f480e096081..ad1742db4b7 100644
--- a/tests/test_litellm/proxy/auth/test_auth_checks.py
+++ b/tests/test_litellm/proxy/auth/test_auth_checks.py
@@ -8461,3 +8461,94 @@ def test_route_skips_budget_checks_marks_only_spend_free_routes() -> None:
def test_request_skips_budget_checks_extends_route_rule_with_zero_cost_models() -> None:
assert request_skips_budget_checks(route="/v1/models", model=None, llm_router=None) is True
assert request_skips_budget_checks(route="/v1/chat/completions", model=None, llm_router=None) is False
+
+
+# Agent access group model ceiling
+
+
+def _agent_model_ceiling(models: frozenset[str]):
+ from litellm.proxy.agent_endpoints.auth.agent_access_groups import AgentAccessGroupCeiling
+
+ return AgentAccessGroupCeiling(
+ access_group_ids=("ag-1",), models=models, mcp_server_ids=frozenset(), agent_ids=frozenset()
+ )
+
+
+async def _run_common_checks_for_agent_key(model: str, valid_token: UserAPIKeyAuth):
+ from fastapi import Request
+
+ from litellm.proxy.auth.auth_checks import common_checks
+
+ return await common_checks(
+ request_body={"model": model, "messages": [{"role": "user", "content": "hi"}]},
+ team_object=None,
+ user_object=None,
+ end_user_object=None,
+ global_proxy_spend=None,
+ general_settings={},
+ route="/chat/completions",
+ llm_router=None,
+ proxy_logging_obj=MagicMock(),
+ valid_token=valid_token,
+ request=MagicMock(spec=Request),
+ )
+
+
+@pytest.mark.asyncio
+async def test_common_checks_agent_access_groups_cap_models_even_when_key_allows_them():
+ agent_key: Final = UserAPIKeyAuth(token="agent-token", agent_id="agent-1", models=["gpt-5", "claude-sonnet"])
+
+ with patch(
+ "litellm.proxy.agent_endpoints.auth.agent_access_groups.resolve_agent_access_group_ceiling",
+ new=AsyncMock(return_value=_agent_model_ceiling(frozenset({"gpt-5"}))),
+ ):
+ assert await _run_common_checks_for_agent_key("gpt-5", agent_key) is True
+
+ with pytest.raises(ProxyException) as exc_info:
+ await _run_common_checks_for_agent_key("claude-sonnet", agent_key)
+
+ assert exc_info.value.type == ProxyErrorTypes.agent_model_access_denied
+ assert exc_info.value.code == str(status.HTTP_403_FORBIDDEN)
+
+
+@pytest.mark.asyncio
+async def test_common_checks_agent_access_groups_naming_no_model_deny_every_model():
+ agent_key: Final = UserAPIKeyAuth(token="agent-token", agent_id="agent-1", models=[])
+
+ with (
+ patch(
+ "litellm.proxy.agent_endpoints.auth.agent_access_groups.resolve_agent_access_group_ceiling",
+ new=AsyncMock(return_value=_agent_model_ceiling(frozenset())),
+ ),
+ pytest.raises(ProxyException) as exc_info,
+ ):
+ await _run_common_checks_for_agent_key("gpt-5", agent_key)
+
+ assert exc_info.value.type == ProxyErrorTypes.agent_model_access_denied
+
+
+@pytest.mark.asyncio
+async def test_common_checks_agent_without_access_groups_adds_no_model_ceiling():
+ agent_key: Final = UserAPIKeyAuth(token="agent-token", agent_id="agent-1", models=["gpt-5", "claude-sonnet"])
+
+ with patch(
+ "litellm.proxy.agent_endpoints.auth.agent_access_groups.resolve_agent_access_group_ceiling",
+ new=AsyncMock(return_value=None),
+ ) as mock_ceiling:
+ assert await _run_common_checks_for_agent_key("gpt-5", agent_key) is True
+ assert await _run_common_checks_for_agent_key("claude-sonnet", agent_key) is True
+
+ mock_ceiling.assert_called_with("agent-1")
+
+
+@pytest.mark.asyncio
+async def test_common_checks_key_without_agent_never_consults_agent_access_groups():
+ plain_key: Final = UserAPIKeyAuth(token="plain-token", models=["gpt-5"])
+
+ with patch(
+ "litellm.proxy.agent_endpoints.auth.agent_access_groups.resolve_agent_access_group_ceiling",
+ new=AsyncMock(return_value=_agent_model_ceiling(frozenset())),
+ ) as mock_ceiling:
+ assert await _run_common_checks_for_agent_key("gpt-5", plain_key) is True
+
+ mock_ceiling.assert_not_called()
diff --git a/tests/test_litellm/proxy/management_endpoints/test_access_group_endpoints.py b/tests/test_litellm/proxy/management_endpoints/test_access_group_endpoints.py
index d687f8d1c8d..13e57408bd0 100644
--- a/tests/test_litellm/proxy/management_endpoints/test_access_group_endpoints.py
+++ b/tests/test_litellm/proxy/management_endpoints/test_access_group_endpoints.py
@@ -17,7 +17,9 @@ from litellm.proxy._types import (
LitellmUserRoles,
UserAPIKeyAuth,
)
+from litellm.proxy.agent_endpoints.agent_registry import global_agent_registry
from litellm.proxy.proxy_server import app
+from litellm.types.agents import AgentResponse
def _make_access_group_record(
@@ -126,6 +128,7 @@ def client_and_mocks(monkeypatch):
mock_agents_table = MagicMock()
mock_agents_table.find_many = AsyncMock(return_value=[])
+ mock_agents_table.update = AsyncMock(return_value=None)
@asynccontextmanager
async def mock_tx():
@@ -133,6 +136,7 @@ def client_and_mocks(monkeypatch):
litellm_accessgrouptable=mock_access_group_table,
litellm_teamtable=mock_team_table,
litellm_verificationtoken=mock_key_table,
+ litellm_agentstable=mock_agents_table,
)
yield tx
@@ -158,15 +162,9 @@ def client_and_mocks(monkeypatch):
mock_proxy_logging = MagicMock()
mock_proxy_logging.internal_usage_cache = MagicMock()
mock_proxy_logging.internal_usage_cache.dual_cache = MagicMock()
- mock_proxy_logging.internal_usage_cache.dual_cache.async_delete_cache = AsyncMock(
- return_value=None
- )
- mock_proxy_logging.internal_usage_cache.dual_cache.async_get_cache = AsyncMock(
- return_value=None
- )
- mock_proxy_logging.internal_usage_cache.dual_cache.async_set_cache = AsyncMock(
- return_value=None
- )
+ mock_proxy_logging.internal_usage_cache.dual_cache.async_delete_cache = AsyncMock(return_value=None)
+ mock_proxy_logging.internal_usage_cache.dual_cache.async_get_cache = AsyncMock(return_value=None)
+ mock_proxy_logging.internal_usage_cache.dual_cache.async_set_cache = AsyncMock(return_value=None)
monkeypatch.setattr(ps, "proxy_logging_obj", mock_proxy_logging)
admin_user = UserAPIKeyAuth(
@@ -239,9 +237,7 @@ def test_create_access_group_duplicate_name_conflict(client_and_mocks):
"unique constraint violation",
],
)
-def test_create_access_group_race_condition_returns_409(
- client_and_mocks, error_message
-):
+def test_create_access_group_race_condition_returns_409(client_and_mocks, error_message):
"""Create race condition: Prisma unique constraint surfaces as 409, not 500."""
client, _, mock_table, *_ = client_and_mocks
@@ -288,9 +284,7 @@ def test_create_access_group_500_on_non_constraint_prisma_error(client_and_mocks
# Use raise_server_exceptions=False so unhandled exceptions become 500 responses
test_client = TestClient(app, raise_server_exceptions=False)
- resp = test_client.post(
- "/v1/access_group", json={"access_group_name": "test-group"}
- )
+ resp = test_client.post("/v1/access_group", json={"access_group_name": "test-group"})
assert resp.status_code == 500
@@ -558,9 +552,7 @@ def test_update_access_group_empty_body(client_and_mocks):
"""Update with empty body succeeds; only updated_by is set."""
client, _, mock_table, *_ = client_and_mocks
- existing = _make_access_group_record(
- access_group_id="ag-update", access_group_name="unchanged"
- )
+ existing = _make_access_group_record(access_group_id="ag-update", access_group_name="unchanged")
mock_table.find_unique = AsyncMock(return_value=existing)
resp = client.put("/v1/access_group/ag-update", json={})
@@ -576,14 +568,10 @@ def test_update_access_group_name_success(client_and_mocks):
"""Update access_group_name succeeds when new name is unique."""
client, _, mock_table, *_ = client_and_mocks
- existing = _make_access_group_record(
- access_group_id="ag-update", access_group_name="old-name"
- )
+ existing = _make_access_group_record(access_group_id="ag-update", access_group_name="old-name")
mock_table.find_unique = AsyncMock(return_value=existing)
- resp = client.put(
- "/v1/access_group/ag-update", json={"access_group_name": "new-name"}
- )
+ resp = client.put("/v1/access_group/ag-update", json={"access_group_name": "new-name"})
assert resp.status_code == 200
mock_table.update.assert_awaited_once()
call_kwargs = mock_table.update.call_args.kwargs
@@ -594,19 +582,13 @@ def test_update_access_group_name_duplicate_conflict(client_and_mocks):
"""Update access_group_name to existing name returns 409 (unique constraint)."""
client, _, mock_table, *_ = client_and_mocks
- existing = _make_access_group_record(
- access_group_id="ag-update", access_group_name="old-name"
- )
+ existing = _make_access_group_record(access_group_id="ag-update", access_group_name="old-name")
mock_table.find_unique = AsyncMock(return_value=existing)
mock_table.update = AsyncMock(
- side_effect=Exception(
- "Unique constraint failed on the fields: (`access_group_name`)"
- )
+ side_effect=Exception("Unique constraint failed on the fields: (`access_group_name`)")
)
- resp = client.put(
- "/v1/access_group/ag-update", json={"access_group_name": "taken-name"}
- )
+ resp = client.put("/v1/access_group/ag-update", json={"access_group_name": "taken-name"})
assert resp.status_code == 409
assert "already exists" in resp.json()["detail"]
mock_table.update.assert_awaited_once()
@@ -620,21 +602,15 @@ def test_update_access_group_name_duplicate_conflict(client_and_mocks):
"unique constraint violation",
],
)
-def test_update_access_group_name_unique_constraint_returns_409(
- client_and_mocks, error_message
-):
+def test_update_access_group_name_unique_constraint_returns_409(client_and_mocks, error_message):
"""Update access_group_name: Prisma unique constraint surfaces as 409."""
client, _, mock_table, *_ = client_and_mocks
- existing = _make_access_group_record(
- access_group_id="ag-update", access_group_name="old-name"
- )
+ existing = _make_access_group_record(access_group_id="ag-update", access_group_name="old-name")
mock_table.find_unique = AsyncMock(return_value=existing)
mock_table.update = AsyncMock(side_effect=Exception(error_message))
- resp = client.put(
- "/v1/access_group/ag-update", json={"access_group_name": "race-name"}
- )
+ resp = client.put("/v1/access_group/ag-update", json={"access_group_name": "race-name"})
assert resp.status_code == 409
assert "already exists" in resp.json()["detail"]
@@ -690,9 +666,7 @@ def test_delete_access_group_forbidden_non_admin(client_and_mocks, user_role):
def test_delete_access_group_cleans_up_teams_and_keys(client_and_mocks):
"""Delete removes access_group_id from teams and keys before deleting the group."""
- client, mock_prisma, mock_access_group_table, mock_cache, mock_proxy_logging = (
- client_and_mocks
- )
+ client, mock_prisma, mock_access_group_table, mock_cache, mock_proxy_logging = client_and_mocks
mock_team_table = mock_prisma.db.litellm_teamtable
mock_key_table = mock_prisma.db.litellm_verificationtoken
@@ -722,10 +696,61 @@ def test_delete_access_group_cleans_up_teams_and_keys(client_and_mocks):
where={"token": "key-token-1"},
data={"access_group_ids": []},
)
- mock_access_group_table.delete.assert_awaited_once_with(
- where={"access_group_id": "ag-to-delete"}
+ mock_access_group_table.delete.assert_awaited_once_with(where={"access_group_id": "ag-to-delete"})
+
+
+def test_delete_access_group_detaches_group_from_agents(client_and_mocks):
+ """Delete strips the group from every agent that had it attached, so agents are not left
+ pointing at a group that no longer exists (which would deny them every model, server and agent)."""
+ client, mock_prisma, mock_access_group_table, _mock_cache, _mock_proxy_logging = client_and_mocks
+ mock_agents_table = mock_prisma.db.litellm_agentstable
+
+ existing = _make_access_group_record(access_group_id="ag-to-delete")
+ mock_access_group_table.find_unique = AsyncMock(return_value=existing)
+
+ agent_with_group = MagicMock()
+ agent_with_group.agent_id = "agent-1"
+ agent_with_group.access_group_ids = ["ag-keep", "ag-to-delete"]
+ mock_agents_table.find_many = AsyncMock(return_value=[agent_with_group])
+ global_agent_registry.register_agent(
+ AgentResponse(
+ agent_id="agent-1",
+ agent_name="detach-test-agent",
+ agent_card_params={"name": "detach-test-agent", "url": "http://localhost:9", "version": "1"},
+ access_group_ids=["ag-keep", "ag-to-delete"],
+ )
)
+ try:
+ resp = client.delete("/v1/access_group/ag-to-delete")
+ assert resp.status_code == 204
+
+ mock_agents_table.update.assert_awaited_once_with(
+ where={"agent_id": "agent-1"},
+ data={"access_group_ids": ("ag-keep",)},
+ )
+ mock_access_group_table.delete.assert_awaited_once_with(where={"access_group_id": "ag-to-delete"})
+ registered = global_agent_registry.get_agent_by_id("agent-1")
+ assert registered is not None
+ assert tuple(registered.access_group_ids or ()) == ("ag-keep",)
+ finally:
+ global_agent_registry.deregister_agent("detach-test-agent")
+
+
+def test_delete_access_group_without_attached_agents_leaves_agents_untouched(client_and_mocks):
+ client, mock_prisma, mock_access_group_table, _mock_cache, _mock_proxy_logging = client_and_mocks
+ mock_agents_table = mock_prisma.db.litellm_agentstable
+
+ mock_access_group_table.find_unique = AsyncMock(
+ return_value=_make_access_group_record(access_group_id="ag-to-delete")
+ )
+
+ resp = client.delete("/v1/access_group/ag-to-delete")
+ assert resp.status_code == 204
+
+ mock_agents_table.find_many.assert_awaited_once_with(where={"access_group_ids": {"hasSome": ("ag-to-delete",)}})
+ mock_agents_table.update.assert_not_awaited()
+
@pytest.mark.parametrize(
"team_cache_group_ids,key_cache_group_ids,expected_team_ids_after,expected_key_ids_after",
@@ -792,9 +817,7 @@ def test_delete_access_group_patches_cached_team_and_key(
"""Delete patches cached team/key objects to remove the deleted access_group_id."""
from litellm.proxy._types import LiteLLM_TeamTableCachedObj
- client, mock_prisma, mock_access_group_table, mock_cache, mock_proxy_logging = (
- client_and_mocks
- )
+ client, mock_prisma, mock_access_group_table, mock_cache, mock_proxy_logging = client_and_mocks
mock_team_table = mock_prisma.db.litellm_teamtable
mock_key_table = mock_prisma.db.litellm_verificationtoken
@@ -820,13 +843,9 @@ def test_delete_access_group_patches_cached_team_and_key(
team_id="team-1",
access_group_ids=list(team_cache_group_ids),
)
- mock_proxy_logging.internal_usage_cache.dual_cache.async_get_cache = AsyncMock(
- return_value=cached_team
- )
+ mock_proxy_logging.internal_usage_cache.dual_cache.async_get_cache = AsyncMock(return_value=cached_team)
else:
- mock_proxy_logging.internal_usage_cache.dual_cache.async_get_cache = AsyncMock(
- return_value=None
- )
+ mock_proxy_logging.internal_usage_cache.dual_cache.async_get_cache = AsyncMock(return_value=None)
# user_api_key_cache is queried both for teams (fallback after dual_cache) and
# hashed keys — return the right stub per ``key``. A single AsyncMock(return_value=key)
@@ -834,9 +853,7 @@ def test_delete_access_group_patches_cached_team_and_key(
# Use a synchronous side_effect (not async def): AsyncMock awaits coroutine side_effects
# inconsistently across Python/unittest versions; sync returns are awaited as immediate results.
def user_cache_get_side_effect(*args, **kwargs):
- cache_key = (
- kwargs.get("key") if "key" in kwargs else (args[0] if args else None)
- )
+ cache_key = kwargs.get("key") if "key" in kwargs else (args[0] if args else None)
if cache_key == "team_id:team-1":
if team_cache_group_ids is None:
return None
@@ -868,14 +885,11 @@ def test_delete_access_group_patches_cached_team_and_key(
team_set_calls = [
c
for c in mock_cache.async_set_cache.call_args_list
- if c.kwargs.get("key", "") == "team_id:team-1"
- or (len(c.args) >= 1 and c.args[0] == "team_id:team-1")
+ if c.kwargs.get("key", "") == "team_id:team-1" or (len(c.args) >= 1 and c.args[0] == "team_id:team-1")
]
assert len(team_set_calls) >= 1, "Expected team cache to be patched"
# The cached team object should have the updated access_group_ids
- written_team = (
- team_set_calls[0].kwargs.get("value") or team_set_calls[0].args[1]
- )
+ written_team = team_set_calls[0].kwargs.get("value") or team_set_calls[0].args[1]
if isinstance(written_team, LiteLLM_TeamTableCachedObj):
assert written_team.access_group_ids == expected_team_ids_after
else:
@@ -883,8 +897,7 @@ def test_delete_access_group_patches_cached_team_and_key(
team_set_calls = [
c
for c in mock_cache.async_set_cache.call_args_list
- if c.kwargs.get("key", "") == "team_id:team-1"
- or (len(c.args) >= 1 and c.args[0] == "team_id:team-1")
+ if c.kwargs.get("key", "") == "team_id:team-1" or (len(c.args) >= 1 and c.args[0] == "team_id:team-1")
]
assert len(team_set_calls) == 0, "Should not patch team cache when not cached"
@@ -892,8 +905,7 @@ def test_delete_access_group_patches_cached_team_and_key(
key_set_calls = [
c
for c in mock_cache.async_set_cache.call_args_list
- if c.kwargs.get("key", "") == "hashed-key-1"
- or (len(c.args) >= 1 and c.args[0] == "hashed-key-1")
+ if c.kwargs.get("key", "") == "hashed-key-1" or (len(c.args) >= 1 and c.args[0] == "hashed-key-1")
]
assert len(key_set_calls) >= 1, "Expected key cache to be patched"
written_key = key_set_calls[0].kwargs.get("value") or key_set_calls[0].args[1]
@@ -903,17 +915,14 @@ def test_delete_access_group_patches_cached_team_and_key(
key_set_calls = [
c
for c in mock_cache.async_set_cache.call_args_list
- if c.kwargs.get("key", "") == "hashed-key-1"
- or (len(c.args) >= 1 and c.args[0] == "hashed-key-1")
+ if c.kwargs.get("key", "") == "hashed-key-1" or (len(c.args) >= 1 and c.args[0] == "hashed-key-1")
]
assert len(key_set_calls) == 0, "Should not patch key cache when not cached"
def test_delete_access_group_patches_key_cached_as_dict(client_and_mocks):
"""Delete patches key cache — mock returns UserAPIKeyAuth (what UserApiKeyCache emits after deserialize)."""
- client, mock_prisma, mock_access_group_table, mock_cache, mock_proxy_logging = (
- client_and_mocks
- )
+ client, mock_prisma, mock_access_group_table, mock_cache, mock_proxy_logging = client_and_mocks
mock_team_table = mock_prisma.db.litellm_teamtable
mock_key_table = mock_prisma.db.litellm_verificationtoken
@@ -929,9 +938,7 @@ def test_delete_access_group_patches_key_cached_as_dict(client_and_mocks):
mock_key_table.find_unique = AsyncMock(return_value=key_with_group)
# No team in cache
- mock_proxy_logging.internal_usage_cache.dual_cache.async_get_cache = AsyncMock(
- return_value=None
- )
+ mock_proxy_logging.internal_usage_cache.dual_cache.async_get_cache = AsyncMock(return_value=None)
# Serialized shape from Redis dict; UserApiKeyCache.async_get_cache(model_type=...) yields a model — simulate that.
cached_key_payload = {
@@ -940,18 +947,14 @@ def test_delete_access_group_patches_key_cached_as_dict(client_and_mocks):
}
def user_cache_get_dict_when_key_matches(*args, **kwargs):
- cache_key = (
- kwargs.get("key") if "key" in kwargs else (args[0] if args else None)
- )
+ cache_key = kwargs.get("key") if "key" in kwargs else (args[0] if args else None)
if cache_key == "team_id:team-1":
return None
if cache_key == "hashed-key-dict":
return UserAPIKeyAuth.model_validate(cached_key_payload)
return None
- mock_cache.async_get_cache = AsyncMock(
- side_effect=user_cache_get_dict_when_key_matches
- )
+ mock_cache.async_get_cache = AsyncMock(side_effect=user_cache_get_dict_when_key_matches)
resp = client.delete("/v1/access_group/ag-to-delete")
assert resp.status_code == 204
@@ -960,8 +963,7 @@ def test_delete_access_group_patches_key_cached_as_dict(client_and_mocks):
key_set_calls = [
c
for c in mock_cache.async_set_cache.call_args_list
- if c.kwargs.get("key", "") == "hashed-key-dict"
- or (len(c.args) >= 1 and c.args[0] == "hashed-key-dict")
+ if c.kwargs.get("key", "") == "hashed-key-dict" or (len(c.args) >= 1 and c.args[0] == "hashed-key-dict")
]
assert len(key_set_calls) >= 1, "Expected key cache to be patched"
written_key = key_set_calls[0].kwargs.get("value") or key_set_calls[0].args[1]
@@ -988,9 +990,7 @@ def test_delete_access_group_404_on_p2025_or_record_not_found(client_and_mocks):
existing = _make_access_group_record(access_group_id="ag-to-delete")
mock_table.find_unique = AsyncMock(return_value=existing)
- mock_table.delete = AsyncMock(
- side_effect=Exception("P2025: Record to delete does not exist")
- )
+ mock_table.delete = AsyncMock(side_effect=Exception("P2025: Record to delete does not exist"))
resp = client.delete("/v1/access_group/ag-to-delete")
assert resp.status_code == 404
@@ -1039,9 +1039,7 @@ def test_delete_access_group_500_on_generic_exception(client_and_mocks):
("delete", "/v1/unified_access_group/ag-123", lambda: {}),
],
)
-def test_access_group_endpoints_db_not_connected(
- client_and_mocks, monkeypatch, method, url, factory
-):
+def test_access_group_endpoints_db_not_connected(client_and_mocks, monkeypatch, method, url, factory):
"""All endpoints return 500 when DB is not connected."""
client, *_ = client_and_mocks
@@ -1049,9 +1047,7 @@ def test_access_group_endpoints_db_not_connected(
resp = getattr(client, method)(url, **factory())
assert resp.status_code == 500
- assert (
- resp.json()["detail"]["error"] == CommonProxyErrors.db_not_connected_error.value
- )
+ assert resp.json()["detail"]["error"] == CommonProxyErrors.db_not_connected_error.value
# ---------------------------------------------------------------------------
@@ -1107,9 +1103,7 @@ def test_attached_team_ids_by_group_keeps_column_order_then_appends_unmirrored_t
def test_create_access_group_syncs_assigned_teams(client_and_mocks):
"""Create adds access_group_id to each assigned team's access_group_ids in DB."""
- client, mock_prisma, mock_access_group_table, mock_cache, mock_proxy_logging = (
- client_and_mocks
- )
+ client, mock_prisma, mock_access_group_table, mock_cache, mock_proxy_logging = client_and_mocks
mock_team_table = mock_prisma.db.litellm_teamtable
team_record = _make_team_record("team-1")
@@ -1132,9 +1126,7 @@ def test_create_access_group_syncs_assigned_teams(client_and_mocks):
def test_create_access_group_syncs_assigned_keys(client_and_mocks):
"""Create adds access_group_id to each assigned key's access_group_ids in DB."""
- client, mock_prisma, mock_access_group_table, mock_cache, mock_proxy_logging = (
- client_and_mocks
- )
+ client, mock_prisma, mock_access_group_table, mock_cache, mock_proxy_logging = client_and_mocks
mock_key_table = mock_prisma.db.litellm_verificationtoken
key_record = MagicMock()
@@ -1148,9 +1140,7 @@ def test_create_access_group_syncs_assigned_keys(client_and_mocks):
)
assert resp.status_code == 201
- mock_key_table.find_unique.assert_awaited_once_with(
- where={"token": "hashed-token-1"}
- )
+ mock_key_table.find_unique.assert_awaited_once_with(where={"token": "hashed-token-1"})
mock_key_table.update.assert_awaited_once()
call_kwargs = mock_key_table.update.call_args.kwargs
assert call_kwargs["where"] == {"token": "hashed-token-1"}
@@ -1200,14 +1190,10 @@ def test_create_access_group_idempotent_team_sync(client_and_mocks):
def test_update_access_group_syncs_added_teams(client_and_mocks):
"""Update adds access_group_id to newly assigned teams."""
- client, mock_prisma, mock_access_group_table, mock_cache, mock_proxy_logging = (
- client_and_mocks
- )
+ client, mock_prisma, mock_access_group_table, mock_cache, mock_proxy_logging = client_and_mocks
mock_team_table = mock_prisma.db.litellm_teamtable
- existing = _make_access_group_record(
- access_group_id="ag-update", assigned_team_ids=["team-existing"]
- )
+ existing = _make_access_group_record(access_group_id="ag-update", assigned_team_ids=["team-existing"])
mock_access_group_table.find_unique = AsyncMock(return_value=existing)
team_record = _make_team_record("team-new")
@@ -1248,14 +1234,10 @@ def test_update_access_group_rejects_nonexistent_team(client_and_mocks):
def test_update_access_group_syncs_removed_teams(client_and_mocks):
"""Update removes access_group_id from de-assigned teams."""
- client, mock_prisma, mock_access_group_table, mock_cache, mock_proxy_logging = (
- client_and_mocks
- )
+ client, mock_prisma, mock_access_group_table, mock_cache, mock_proxy_logging = client_and_mocks
mock_team_table = mock_prisma.db.litellm_teamtable
- existing = _make_access_group_record(
- access_group_id="ag-update", assigned_team_ids=["team-keep", "team-remove"]
- )
+ existing = _make_access_group_record(access_group_id="ag-update", assigned_team_ids=["team-keep", "team-remove"])
mock_access_group_table.find_unique = AsyncMock(return_value=existing)
team_to_remove = _make_team_record("team-remove", ["ag-update"])
@@ -1268,9 +1250,7 @@ def test_update_access_group_syncs_removed_teams(client_and_mocks):
)
assert resp.status_code == 200
- mock_team_table.find_unique.assert_awaited_once_with(
- where={"team_id": "team-remove"}
- )
+ mock_team_table.find_unique.assert_awaited_once_with(where={"team_id": "team-remove"})
mock_team_table.update.assert_awaited_once()
call_kwargs = mock_team_table.update.call_args.kwargs
assert call_kwargs["where"] == {"team_id": "team-remove"}
@@ -1296,19 +1276,15 @@ def test_update_access_group_detaches_team_the_mirror_missed(client_and_mocks):
mock_team_table.update.assert_awaited_once()
call_kwargs = mock_team_table.update.call_args.kwargs
assert call_kwargs["where"] == {"team_id": "team-unmirrored"}
- assert call_kwargs["data"]["access_group_ids"] == ["ag-other"]
+ assert tuple(call_kwargs["data"]["access_group_ids"]) == ("ag-other",)
def test_update_access_group_no_team_sync_when_ids_not_in_payload(client_and_mocks):
"""Update does not sync teams when assigned_team_ids is absent from the payload."""
- client, mock_prisma, mock_access_group_table, mock_cache, mock_proxy_logging = (
- client_and_mocks
- )
+ client, mock_prisma, mock_access_group_table, mock_cache, mock_proxy_logging = client_and_mocks
mock_team_table = mock_prisma.db.litellm_teamtable
- existing = _make_access_group_record(
- access_group_id="ag-update", assigned_team_ids=["team-1"]
- )
+ existing = _make_access_group_record(access_group_id="ag-update", assigned_team_ids=["team-1"])
mock_access_group_table.find_unique = AsyncMock(return_value=existing)
resp = client.put("/v1/access_group/ag-update", json={"description": "new desc"})
@@ -1320,14 +1296,10 @@ def test_update_access_group_no_team_sync_when_ids_not_in_payload(client_and_moc
def test_update_access_group_syncs_added_keys(client_and_mocks):
"""Update adds access_group_id to newly assigned keys."""
- client, mock_prisma, mock_access_group_table, mock_cache, mock_proxy_logging = (
- client_and_mocks
- )
+ client, mock_prisma, mock_access_group_table, mock_cache, mock_proxy_logging = client_and_mocks
mock_key_table = mock_prisma.db.litellm_verificationtoken
- existing = _make_access_group_record(
- access_group_id="ag-update", assigned_key_ids=["old-token"]
- )
+ existing = _make_access_group_record(access_group_id="ag-update", assigned_key_ids=["old-token"])
mock_access_group_table.find_unique = AsyncMock(return_value=existing)
key_record = MagicMock()
@@ -1350,14 +1322,10 @@ def test_update_access_group_syncs_added_keys(client_and_mocks):
def test_update_access_group_syncs_removed_keys(client_and_mocks):
"""Update removes access_group_id from de-assigned keys."""
- client, mock_prisma, mock_access_group_table, mock_cache, mock_proxy_logging = (
- client_and_mocks
- )
+ client, mock_prisma, mock_access_group_table, mock_cache, mock_proxy_logging = client_and_mocks
mock_key_table = mock_prisma.db.litellm_verificationtoken
- existing = _make_access_group_record(
- access_group_id="ag-update", assigned_key_ids=["keep-token", "remove-token"]
- )
+ existing = _make_access_group_record(access_group_id="ag-update", assigned_key_ids=["keep-token", "remove-token"])
mock_access_group_table.find_unique = AsyncMock(return_value=existing)
key_to_remove = MagicMock()
@@ -1385,9 +1353,7 @@ def test_update_access_group_syncs_removed_keys(client_and_mocks):
def test_delete_access_group_handles_out_of_sync_assigned_teams(client_and_mocks):
"""Delete includes teams from assigned_team_ids even when not found by hasSome query."""
- client, mock_prisma, mock_access_group_table, mock_cache, mock_proxy_logging = (
- client_and_mocks
- )
+ client, mock_prisma, mock_access_group_table, mock_cache, mock_proxy_logging = client_and_mocks
mock_team_table = mock_prisma.db.litellm_teamtable
# Access group has assigned_team_ids but the team's access_group_ids is not synced
@@ -1409,18 +1375,14 @@ def test_delete_access_group_handles_out_of_sync_assigned_teams(client_and_mocks
assert resp.status_code == 204
# find_unique is called for the out-of-sync team (included via union with assigned_team_ids)
- mock_team_table.find_unique.assert_awaited_once_with(
- where={"team_id": "team-out-of-sync"}
- )
+ mock_team_table.find_unique.assert_awaited_once_with(where={"team_id": "team-out-of-sync"})
# No update needed since team's access_group_ids doesn't contain "ag-to-delete"
mock_team_table.update.assert_not_awaited()
def test_delete_access_group_handles_out_of_sync_assigned_keys(client_and_mocks):
"""Delete includes keys from assigned_key_ids even when not found by hasSome query."""
- client, mock_prisma, mock_access_group_table, mock_cache, mock_proxy_logging = (
- client_and_mocks
- )
+ client, mock_prisma, mock_access_group_table, mock_cache, mock_proxy_logging = client_and_mocks
mock_key_table = mock_prisma.db.litellm_verificationtoken
existing = _make_access_group_record(
@@ -1439,9 +1401,7 @@ def test_delete_access_group_handles_out_of_sync_assigned_keys(client_and_mocks)
resp = client.delete("/v1/access_group/ag-to-delete")
assert resp.status_code == 204
- mock_key_table.find_unique.assert_awaited_once_with(
- where={"token": "token-out-of-sync"}
- )
+ mock_key_table.find_unique.assert_awaited_once_with(where={"token": "token-out-of-sync"})
mock_key_table.update.assert_not_awaited()
@@ -1536,10 +1496,16 @@ def test_list_access_groups_resolves_names_with_one_query_per_table(client_and_m
mock_table.find_many = AsyncMock(
return_value=[
_make_access_group_record(
- access_group_id="ag-1", access_mcp_server_ids=["mcp-a"], access_agent_ids=["agent-a"], assigned_key_ids=["key-a"]
+ access_group_id="ag-1",
+ access_mcp_server_ids=["mcp-a"],
+ access_agent_ids=["agent-a"],
+ assigned_key_ids=["key-a"],
),
_make_access_group_record(
- access_group_id="ag-2", access_mcp_server_ids=["mcp-b"], access_agent_ids=["agent-b"], assigned_key_ids=["key-b"]
+ access_group_id="ag-2",
+ access_mcp_server_ids=["mcp-b"],
+ access_agent_ids=["agent-b"],
+ assigned_key_ids=["key-b"],
),
]
)
@@ -1573,7 +1539,10 @@ def test_list_access_groups_skips_lookups_when_nothing_to_resolve(client_and_moc
"""Groups with no MCP servers, agents or keys must not trigger an empty IN () query per table."""
client, mock_prisma, mock_table, *_ = client_and_mocks
mock_table.find_many = AsyncMock(
- return_value=[_make_access_group_record(access_group_id="ag-1"), _make_access_group_record(access_group_id="ag-2")]
+ return_value=[
+ _make_access_group_record(access_group_id="ag-1"),
+ _make_access_group_record(access_group_id="ag-2"),
+ ]
)
resp = client.get("/v1/access_group")
diff --git a/ui/litellm-dashboard/src/app/(dashboard)/agents/_components/AgentFormKit.tsx b/ui/litellm-dashboard/src/app/(dashboard)/agents/_components/AgentFormKit.tsx
index 15b85001cdc..8e100d0c3ed 100644
--- a/ui/litellm-dashboard/src/app/(dashboard)/agents/_components/AgentFormKit.tsx
+++ b/ui/litellm-dashboard/src/app/(dashboard)/agents/_components/AgentFormKit.tsx
@@ -90,6 +90,7 @@ export interface AgentFormValues {
guardrails?: string[];
entitlement_models?: string[];
entitlement_agents?: string[];
+ access_group_ids?: string[];
allowed_mcp_servers_and_groups?: McpServerSelection;
mcp_tool_permissions?: Record;
defaultInputModes?: string[];
@@ -121,6 +122,7 @@ export interface AgentRequestPayload {
agent_card_params?: Record;
litellm_params?: Record;
object_permission?: Record;
+ access_group_ids?: string[];
}
interface AgentFormFieldProps {
diff --git a/ui/litellm-dashboard/src/app/(dashboard)/agents/_components/add_agent_form.integration.test.tsx b/ui/litellm-dashboard/src/app/(dashboard)/agents/_components/add_agent_form.integration.test.tsx
index ebc97891744..457ee656415 100644
--- a/ui/litellm-dashboard/src/app/(dashboard)/agents/_components/add_agent_form.integration.test.tsx
+++ b/ui/litellm-dashboard/src/app/(dashboard)/agents/_components/add_agent_form.integration.test.tsx
@@ -21,6 +21,9 @@ vi.mock("./agent_card_discovery", () => ({ default: () => ({ default: () =>
}));
vi.mock("@/components/mcp_server_management/MCPToolPermissions", () => ({ default: () =>
}));
vi.mock("@/components/guardrails/GuardrailSelector", () => ({ default: () =>
}));
+vi.mock("@/app/(dashboard)/hooks/accessGroups/useAccessGroups", () => ({
+ useAccessGroups: () => ({ data: [], isLoading: false, isError: false }),
+}));
vi.mock("@/components/common_components/team_dropdown", () => ({ default: () =>
}));
const a2aInfo: AgentCreateInfo = {
diff --git a/ui/litellm-dashboard/src/app/(dashboard)/agents/_components/add_agent_form.test.tsx b/ui/litellm-dashboard/src/app/(dashboard)/agents/_components/add_agent_form.test.tsx
index ccac244f019..b5301d0e6a0 100644
--- a/ui/litellm-dashboard/src/app/(dashboard)/agents/_components/add_agent_form.test.tsx
+++ b/ui/litellm-dashboard/src/app/(dashboard)/agents/_components/add_agent_form.test.tsx
@@ -44,6 +44,14 @@ vi.mock("@/components/mcp_server_management/MCPToolPermissions", () => ({
default: () => null,
}));
+vi.mock("@/components/common_components/AccessGroupSelector", () => ({
+ default: ({ onChange }: { onChange: (value: string[]) => void }) => (
+
onChange(["ag-1", "ag-2"])}>
+ Select access group
+
+ ),
+}));
+
vi.mock("@/components/common_components/team_dropdown", () => ({
default: () => null,
}));
@@ -141,5 +149,27 @@ describe("AddAgentForm logos", () => {
await vi.waitFor(() => expect(networking.createAgentCall).toHaveBeenCalled());
const [, payload] = vi.mocked(networking.createAgentCall).mock.calls[0];
expect(payload.object_permission).toEqual({ mcp_toolsets: ["ts-1"] });
+ expect(payload).not.toHaveProperty("access_group_ids");
+ });
+
+ it("includes selected access groups in the create payload", async () => {
+ const user = userEvent.setup({ pointerEventsCheck: PointerEventsCheckLevel.Never });
+ vi.mocked(networking.createAgentCall).mockReset().mockResolvedValue({
+ agent_id: "agent-1",
+ agent_name: "Test Agent",
+ } as never);
+ vi.mocked(networking.keyListCall).mockResolvedValue({ keys: [] });
+
+ renderForm();
+ await user.click(screen.getByRole("button", { name: "Next →" }));
+ await user.click(screen.getByTestId("select-access-group"));
+ await user.click(screen.getByRole("button", { name: "Next →" }));
+ await user.click(screen.getByRole("button", { name: "Next →" }));
+ await user.click(screen.getByText(/Skip for now/));
+ await user.click(screen.getByRole("button", { name: "Create Agent →" }));
+
+ await vi.waitFor(() => expect(networking.createAgentCall).toHaveBeenCalled());
+ const [, payload] = vi.mocked(networking.createAgentCall).mock.calls[0];
+ expect(payload.access_group_ids).toEqual(["ag-1", "ag-2"]);
});
});
diff --git a/ui/litellm-dashboard/src/app/(dashboard)/agents/_components/add_agent_form.tsx b/ui/litellm-dashboard/src/app/(dashboard)/agents/_components/add_agent_form.tsx
index e71fed40209..5bd6ea9b83a 100644
--- a/ui/litellm-dashboard/src/app/(dashboard)/agents/_components/add_agent_form.tsx
+++ b/ui/litellm-dashboard/src/app/(dashboard)/agents/_components/add_agent_form.tsx
@@ -50,6 +50,7 @@ import {
} from "./AgentFormKit";
import MCPServerSelector from "@/components/mcp_server_management/MCPServerSelector";
import MCPToolPermissions from "@/components/mcp_server_management/MCPToolPermissions";
+import AccessGroupSelector from "@/components/common_components/AccessGroupSelector";
import GuardrailSelector from "@/components/guardrails/GuardrailSelector";
import { Dialog, DialogContent, DialogHeader, DialogTitle } from "@/components/ui/dialog";
@@ -113,6 +114,7 @@ const SHARED_INITIAL_VALUES: AgentFormValues = {
mcp_tool_permissions: {},
entitlement_models: [],
entitlement_agents: [],
+ access_group_ids: [],
guardrails: [],
};
@@ -374,6 +376,9 @@ const AddAgentForm: React.FC
= ({ visible, onClose, accessTok
if (Object.keys(objectPermission).length > 0) {
agentData.object_permission = objectPermission;
}
+ if (values.access_group_ids?.length) {
+ agentData.access_group_ids = values.access_group_ids;
+ }
// Wire trace-id flags and budget controls into agent litellm_params (before create call)
if (requireTraceIdInbound || requireTraceIdOutbound) {
@@ -494,6 +499,22 @@ const AddAgentForm: React.FC = ({ visible, onClose, accessTok
)}
+
+ {({ value, onChange }) => (
+
+ )}
+
+
{
return agentData;
};
+export const parseAccessGroupIdsForForm = (agent: { access_group_ids?: string[] | null }) => ({
+ access_group_ids: agent.access_group_ids ?? [],
+});
+
export const parseMcpPermissionsForForm = (agent: any) => ({
allowed_mcp_servers_and_groups: {
servers: agent.object_permission?.mcp_servers ?? [],
@@ -377,5 +381,6 @@ export const parseAgentForForm = (agent: any) => {
// extra_headers: already an array of strings
extra_headers: agent.extra_headers ?? [],
...parseMcpPermissionsForForm(agent),
+ ...parseAccessGroupIdsForForm(agent),
};
};
diff --git a/ui/litellm-dashboard/src/app/(dashboard)/agents/_components/agent_info.integration.test.tsx b/ui/litellm-dashboard/src/app/(dashboard)/agents/_components/agent_info.integration.test.tsx
index 357f924cbe7..37e00766a75 100644
--- a/ui/litellm-dashboard/src/app/(dashboard)/agents/_components/agent_info.integration.test.tsx
+++ b/ui/litellm-dashboard/src/app/(dashboard)/agents/_components/agent_info.integration.test.tsx
@@ -25,6 +25,10 @@ vi.mock("@/app/(dashboard)/hooks/keys/useKeys", () => ({
vi.mock("./agent_card_discovery", () => ({ default: () =>
}));
+vi.mock("@/app/(dashboard)/hooks/accessGroups/useAccessGroups", () => ({
+ useAccessGroups: () => ({ data: [], isLoading: false, isError: false }),
+}));
+
const A2A_AGENT = {
agent_id: "agent-1",
agent_name: "my-agent",
@@ -176,6 +180,7 @@ describe("AgentInfoView update payload", () => {
session_tpm_limit: 333,
session_rpm_limit: 444,
object_permission: { mcp_servers: [], mcp_access_groups: [], mcp_toolsets: [], mcp_tool_permissions: {} },
+ access_group_ids: [],
});
});
@@ -217,6 +222,7 @@ describe("AgentInfoView update payload", () => {
session_tpm_limit: 333,
session_rpm_limit: 444,
object_permission: { mcp_servers: [], mcp_access_groups: [], mcp_toolsets: [], mcp_tool_permissions: {} },
+ access_group_ids: [],
});
});
@@ -295,6 +301,7 @@ describe("AgentInfoView update payload", () => {
model: "langgraph/asst_1",
},
object_permission: { mcp_servers: [], mcp_access_groups: [], mcp_toolsets: [], mcp_tool_permissions: {} },
+ access_group_ids: [],
});
});
diff --git a/ui/litellm-dashboard/src/app/(dashboard)/agents/_components/agent_info.test.tsx b/ui/litellm-dashboard/src/app/(dashboard)/agents/_components/agent_info.test.tsx
index 29d97a20afe..7e6c7c0e05c 100644
--- a/ui/litellm-dashboard/src/app/(dashboard)/agents/_components/agent_info.test.tsx
+++ b/ui/litellm-dashboard/src/app/(dashboard)/agents/_components/agent_info.test.tsx
@@ -28,6 +28,28 @@ vi.mock("@/app/(dashboard)/hooks/mcpServers/useMCPServers", () => ({
useMCPServers: () => ({ data: [{ server_id: "srv-1", server_name: "github" }] }),
}));
+vi.mock("@/app/(dashboard)/hooks/accessGroups/useAccessGroups", () => ({
+ useAccessGroups: () => ({
+ data: [{ access_group_id: "ag-1", access_group_name: "support-tools" }],
+ isLoading: false,
+ isError: false,
+ }),
+}));
+
+vi.mock("@/components/common_components/AccessGroupSelector", () => ({
+ default: ({ value, onChange }: { value?: string[]; onChange: (value: string[]) => void }) => (
+
+ {(value ?? []).join(",")}
+ onChange(["ag-1"])}>
+ Attach ag-1
+
+ onChange([])}>
+ Detach all access groups
+
+
+ ),
+}));
+
vi.mock("@/components/mcp_server_management/MCPServerSelector", () => ({
default: () =>
,
}));
@@ -76,6 +98,38 @@ describe("AgentInfoView settings", () => {
expect(payload.tpm_limit).toBe(42);
const clearedMcpGrants = { mcp_servers: [], mcp_access_groups: [], mcp_toolsets: [], mcp_tool_permissions: {} };
expect(payload.object_permission).toEqual(clearedMcpGrants);
+ expect(payload.access_group_ids).toEqual([]);
+ });
+
+ it("sends the newly attached access group in the update payload", async () => {
+ render( );
+
+ fireEvent.click(await screen.findByRole("tab", { name: "Settings" }));
+ fireEvent.click(screen.getByRole("button", { name: "Edit Settings" }));
+ fireEvent.click(await screen.findByRole("button", { name: "Attach ag-1" }));
+ expect(screen.getByTestId("selected-access-groups")).toHaveTextContent("ag-1");
+
+ fireEvent.click(screen.getByRole("button", { name: /Save Changes/ }));
+
+ await waitFor(() => expect(networking.patchAgentCall).toHaveBeenCalledTimes(1));
+ const [, , payload] = vi.mocked(networking.patchAgentCall).mock.calls[0];
+ expect(payload.access_group_ids).toEqual(["ag-1"]);
+ });
+
+ it("loads the attached access groups into the editor and sends an empty list once detached", async () => {
+ vi.mocked(networking.getAgentInfo).mockResolvedValue({ ...agent, access_group_ids: ["ag-1"] });
+ render( );
+
+ fireEvent.click(await screen.findByRole("tab", { name: "Settings" }));
+ fireEvent.click(screen.getByRole("button", { name: "Edit Settings" }));
+ expect(await screen.findByTestId("selected-access-groups")).toHaveTextContent("ag-1");
+
+ fireEvent.click(screen.getByRole("button", { name: "Detach all access groups" }));
+ fireEvent.click(screen.getByRole("button", { name: /Save Changes/ }));
+
+ await waitFor(() => expect(networking.patchAgentCall).toHaveBeenCalledTimes(1));
+ const [, , payload] = vi.mocked(networking.patchAgentCall).mock.calls[0];
+ expect(payload.access_group_ids).toEqual([]);
});
it("shows MCP grants with server names on the overview tab", async () => {
@@ -88,4 +142,20 @@ describe("AgentInfoView settings", () => {
expect(await screen.findByText("github (srv-1)")).toBeInTheDocument();
});
+
+ it("shows attached access groups with their names on the overview tab", async () => {
+ vi.mocked(networking.getAgentInfo).mockResolvedValue({ ...agent, access_group_ids: ["ag-1", "ag-unknown"] });
+
+ render( );
+
+ expect(await screen.findByText("support-tools (ag-1)")).toBeInTheDocument();
+ expect(screen.getByText("ag-unknown")).toBeInTheDocument();
+ });
+
+ it("shows None when the agent has no access groups attached", async () => {
+ render( );
+
+ expect(await screen.findByText("Access Groups")).toBeInTheDocument();
+ expect(screen.getByText("None")).toBeInTheDocument();
+ });
});
diff --git a/ui/litellm-dashboard/src/app/(dashboard)/agents/_components/agent_info.tsx b/ui/litellm-dashboard/src/app/(dashboard)/agents/_components/agent_info.tsx
index 6cb99e9692f..adac456f232 100644
--- a/ui/litellm-dashboard/src/app/(dashboard)/agents/_components/agent_info.tsx
+++ b/ui/litellm-dashboard/src/app/(dashboard)/agents/_components/agent_info.tsx
@@ -16,6 +16,8 @@ import { Agent } from "@/components/agents/types";
import { KeyResponse } from "@/components/key_team_helpers/key_list";
import { useKeys } from "@/app/(dashboard)/hooks/keys/useKeys";
import { useMCPServers } from "@/app/(dashboard)/hooks/mcpServers/useMCPServers";
+import { useAccessGroups } from "@/app/(dashboard)/hooks/accessGroups/useAccessGroups";
+import AccessGroupSelector from "@/components/common_components/AccessGroupSelector";
import KeyInfoView from "@/components/templates/key_info_view";
import MCPServerSelector from "@/components/mcp_server_management/MCPServerSelector";
import MCPToolPermissions from "@/components/mcp_server_management/MCPToolPermissions";
@@ -26,6 +28,7 @@ import {
AGENT_FORM_CONFIG,
buildAgentDataFromForm,
buildMcpObjectPermission,
+ parseAccessGroupIdsForForm,
parseAgentForForm,
parseMcpPermissionsForForm,
} from "./agent_config";
@@ -122,7 +125,11 @@ const AgentInfoView: React.FC = ({ agentId, onClose, accessT
} else {
const typeInfo = agentTypeMetadata.find((t) => t.agent_type === agentType);
if (typeInfo) {
- form.reset({ ...parseDynamicAgentForForm(data, typeInfo), ...parseMcpPermissionsForForm(data) });
+ form.reset({
+ ...parseDynamicAgentForForm(data, typeInfo),
+ ...parseMcpPermissionsForForm(data),
+ ...parseAccessGroupIdsForForm(data),
+ });
} else {
form.reset(parseAgentForForm(data));
}
@@ -142,7 +149,11 @@ const AgentInfoView: React.FC = ({ agentId, onClose, accessT
if (agentType !== "a2a") {
const typeInfo = agentTypeMetadata.find((t) => t.agent_type === agentType);
if (typeInfo) {
- form.reset({ ...parseDynamicAgentForForm(agent, typeInfo), ...parseMcpPermissionsForForm(agent) });
+ form.reset({
+ ...parseDynamicAgentForForm(agent, typeInfo),
+ ...parseMcpPermissionsForForm(agent),
+ ...parseAccessGroupIdsForForm(agent),
+ });
}
}
}
@@ -153,12 +164,18 @@ const AgentInfoView: React.FC = ({ agentId, onClose, accessT
const mcpSelection = useWatch({ control: form.control, name: "allowed_mcp_servers_and_groups" });
const mcpToolPermissions = useWatch({ control: form.control, name: "mcp_tool_permissions" });
const { data: mcpServers = [] } = useMCPServers();
+ const { data: accessGroups = [] } = useAccessGroups();
const mcpServerLabel = (serverId: string) => {
const server = mcpServers.find((s) => s.server_id === serverId);
return server?.server_name ? `${server.server_name} (${serverId})` : serverId;
};
+ const accessGroupLabel = (accessGroupId: string) => {
+ const group = accessGroups.find((g) => g.access_group_id === accessGroupId);
+ return group ? `${group.access_group_name} (${accessGroupId})` : accessGroupId;
+ };
+
const discoveryRequest = useMemo(
() => buildDiscoveryRequest(detectedAgentType, watchedFormValues || {}, selectedAgentTypeInfo),
[watchedFormValues, selectedAgentTypeInfo, detectedAgentType],
@@ -221,6 +238,7 @@ const AgentInfoView: React.FC = ({ agentId, onClose, accessT
await patchAgentCall(accessToken, agentId, {
...updateData,
object_permission: buildMcpObjectPermission(values),
+ access_group_ids: values.access_group_ids ?? [],
});
toast.success("Agent updated successfully");
setIsEditing(false);
@@ -350,6 +368,17 @@ const AgentInfoView: React.FC = ({ agentId, onClose, accessT
{agent.rpm_limit ?? "Unlimited"}
{agent.session_tpm_limit ?? "Unlimited"}
{agent.session_rpm_limit ?? "Unlimited"}
+
+ {agent.access_group_ids?.length ? (
+
+ {agent.access_group_ids.map((accessGroupId) => (
+
{accessGroupLabel(accessGroupId)}
+ ))}
+
+ ) : (
+ "None"
+ )}
+
{formatDate(agent.created_at)}
{formatDate(agent.updated_at)}
@@ -489,6 +518,26 @@ const AgentInfoView: React.FC = ({ agentId, onClose, accessT
{rateLimitField("session_rpm_limit", "Session RPM Limit")}
+
+ Access Groups
+
+
+ {({ value, onChange }) => (
+
+ )}
+
+
+
MCP Servers
diff --git a/ui/litellm-dashboard/src/components/agents/types.ts b/ui/litellm-dashboard/src/components/agents/types.ts
index 24ff0c0e12c..6adb3fa9dda 100644
--- a/ui/litellm-dashboard/src/components/agents/types.ts
+++ b/ui/litellm-dashboard/src/components/agents/types.ts
@@ -21,6 +21,7 @@ export interface Agent {
[key: string]: any;
};
object_permission?: AgentObjectPermission;
+ access_group_ids?: string[] | null;
keys?: AgentAttachedKey[] | null;
spend?: number;
tpm_limit?: number | null;
diff --git a/ui/litellm-dashboard/src/components/networking.tsx b/ui/litellm-dashboard/src/components/networking.tsx
index e77c8ba7e41..7714f96804f 100644
--- a/ui/litellm-dashboard/src/components/networking.tsx
+++ b/ui/litellm-dashboard/src/components/networking.tsx
@@ -6267,6 +6267,7 @@ export const patchAgentCall = async (
rpm_limit?: number | null;
session_tpm_limit?: number | null;
session_rpm_limit?: number | null;
+ access_group_ids?: string[];
},
) => {
try {
diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts
index 872875cc535..be193647dca 100644
--- a/ui/litellm-dashboard/src/lib/http/schema.d.ts
+++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts
@@ -23195,6 +23195,8 @@ export interface components {
};
/** AgentConfig */
AgentConfig: {
+ /** Access Group Ids */
+ access_group_ids?: string[] | null;
agent_card_params: components["schemas"]["AgentCard"];
/** Agent Name */
agent_name: string;
@@ -23342,6 +23344,8 @@ export interface components {
};
/** AgentResponse */
AgentResponse: {
+ /** Access Group Ids */
+ access_group_ids?: string[] | null;
/** Agent Card Params */
agent_card_params: {
[key: string]: unknown;
@@ -34111,6 +34115,8 @@ export interface components {
};
/** PatchAgentRequest */
PatchAgentRequest: {
+ /** Access Group Ids */
+ access_group_ids?: string[] | null;
agent_card_params?: components["schemas"]["AgentCard"];
/** Agent Name */
agent_name?: string;
From 433c804a6cfec9c6674872857f96b435f69f4d96 Mon Sep 17 00:00:00 2001
From: yassin
Date: Thu, 17 Sep 2026 19:32:39 +0000
Subject: [PATCH 005/160] refactor(agents): mark Callable loader aliases for
the type discipline gate
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
---
litellm/proxy/agent_endpoints/auth/agent_access_groups.py | 7 ++++---
1 file changed, 4 insertions(+), 3 deletions(-)
diff --git a/litellm/proxy/agent_endpoints/auth/agent_access_groups.py b/litellm/proxy/agent_endpoints/auth/agent_access_groups.py
index 4a579de679e..f5adc897e11 100644
--- a/litellm/proxy/agent_endpoints/auth/agent_access_groups.py
+++ b/litellm/proxy/agent_endpoints/auth/agent_access_groups.py
@@ -18,8 +18,9 @@ from litellm._logging import verbose_proxy_logger
from litellm.proxy._types import LiteLLM_AccessGroupTable
from litellm.types.agents import AgentResponse
-AgentLoader: TypeAlias = Callable[[str], Awaitable[AgentResponse | None]]
-AccessGroupLoader: TypeAlias = Callable[[str], Awaitable[LiteLLM_AccessGroupTable | None]]
+AgentLoader: TypeAlias = Callable[[str], Awaitable[AgentResponse | None]] # mutable-ok: Callable parameter syntax
+LoadedAccessGroup: TypeAlias = LiteLLM_AccessGroupTable | None
+AccessGroupLoader: TypeAlias = Callable[[str], Awaitable[LoadedAccessGroup]] # mutable-ok: Callable parameter syntax
@dataclass(frozen=True, slots=True)
@@ -38,7 +39,7 @@ async def _load_agent(agent_id: str) -> AgentResponse | None:
return await get_agent_with_read_through(agent_id)
-async def _load_access_group(access_group_id: str) -> LiteLLM_AccessGroupTable | None:
+async def _load_access_group(access_group_id: str) -> LoadedAccessGroup:
from litellm.proxy.auth.auth_checks import get_access_object
from litellm.proxy.proxy_server import prisma_client, proxy_logging_obj, user_api_key_cache
From 322262db01f9ba69b1a77ab2ec26268c468a3099 Mon Sep 17 00:00:00 2001
From: yassin
Date: Thu, 17 Sep 2026 19:42:56 +0000
Subject: [PATCH 006/160] style(ui): format add_agent_form test with prettier
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
---
.../agents/_components/add_agent_form.test.tsx | 10 ++++++----
1 file changed, 6 insertions(+), 4 deletions(-)
diff --git a/ui/litellm-dashboard/src/app/(dashboard)/agents/_components/add_agent_form.test.tsx b/ui/litellm-dashboard/src/app/(dashboard)/agents/_components/add_agent_form.test.tsx
index b5301d0e6a0..fbf5cf8c1fb 100644
--- a/ui/litellm-dashboard/src/app/(dashboard)/agents/_components/add_agent_form.test.tsx
+++ b/ui/litellm-dashboard/src/app/(dashboard)/agents/_components/add_agent_form.test.tsx
@@ -154,10 +154,12 @@ describe("AddAgentForm logos", () => {
it("includes selected access groups in the create payload", async () => {
const user = userEvent.setup({ pointerEventsCheck: PointerEventsCheckLevel.Never });
- vi.mocked(networking.createAgentCall).mockReset().mockResolvedValue({
- agent_id: "agent-1",
- agent_name: "Test Agent",
- } as never);
+ vi.mocked(networking.createAgentCall)
+ .mockReset()
+ .mockResolvedValue({
+ agent_id: "agent-1",
+ agent_name: "Test Agent",
+ } as never);
vi.mocked(networking.keyListCall).mockResolvedValue({ keys: [] });
renderForm();
From d743e08432ee738355a144046b57321485401fea Mon Sep 17 00:00:00 2001
From: yassin
Date: Thu, 17 Sep 2026 20:04:23 +0000
Subject: [PATCH 007/160] test(agents): inject the access group ceiling
resolver instead of patching it
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
---
.../mcp_server/auth/user_api_key_auth_mcp.py | 47 ++++---
.../auth/agent_access_groups.py | 3 +
.../auth/agent_permission_handler.py | 15 ++-
litellm/proxy/auth/auth_checks.py | 9 +-
.../auth/test_user_api_key_auth_mcp.py | 109 ++++++++--------
.../auth/test_agent_permission_handler.py | 118 ++++++++----------
.../proxy/auth/test_auth_checks.py | 93 ++++++--------
7 files changed, 193 insertions(+), 201 deletions(-)
diff --git a/litellm/proxy/_experimental/mcp_server/auth/user_api_key_auth_mcp.py b/litellm/proxy/_experimental/mcp_server/auth/user_api_key_auth_mcp.py
index bbb3d30864f..05661584a6b 100644
--- a/litellm/proxy/_experimental/mcp_server/auth/user_api_key_auth_mcp.py
+++ b/litellm/proxy/_experimental/mcp_server/auth/user_api_key_auth_mcp.py
@@ -44,6 +44,10 @@ from litellm.proxy._types import (
UserAPIKeyAuth,
user_api_key_has_admin_view,
)
+from litellm.proxy.agent_endpoints.auth.agent_access_groups import (
+ CeilingResolver,
+ resolve_agent_access_group_ceiling,
+)
from litellm.proxy.auth.ip_address_utils import IPAddressUtils
from litellm.proxy.auth.user_api_key_auth import (
_get_bearer_token_or_received_api_key, # pyright: ignore[reportPrivateUsage] # shared x-litellm-api-key parser lives with user_api_key_auth
@@ -184,6 +188,24 @@ def _has_client_supplied_mcp_auth(
return bool(mcp_auth_header) or bool(mcp_server_auth_headers)
+def _agent_capped_servers(
+ allowed_mcp_servers: Sequence[str],
+ agent_servers: Sequence[str],
+ agent_access_group_servers: frozenset[str] | None,
+) -> tuple[str, ...] | None:
+ """Servers left once the agent's object_permission and attached access groups both cap the
+ key/team result, or None when the agent restricts nothing. An attached group set naming no
+ server is an empty ceiling, not an absent one, so it denies every server."""
+ if not agent_servers and agent_access_group_servers is None:
+ return None
+ return tuple(
+ s
+ for s in allowed_mcp_servers
+ if (not agent_servers or s in agent_servers)
+ and (agent_access_group_servers is None or s in agent_access_group_servers)
+ )
+
+
def _is_mcp_admitted_user_subject(user_api_key_auth: UserAPIKeyAuth | None) -> bool:
"""True when this auth is a keyless subject admitted by the gateway session / bridge user
path, as opposed to a JWT or other keyless auth that merely lacks a ``team_id``.
@@ -1546,21 +1568,14 @@ class MCPRequestHandler:
# Check agent permissions if agent_id is set on the key
#########################################################
if user_api_key_auth and user_api_key_auth.agent_id:
- allowed_mcp_servers_for_agent: Final = await MCPRequestHandler._get_allowed_mcp_servers_for_agent(
- user_api_key_auth
+ agent_capped: Final = _agent_capped_servers(
+ allowed_mcp_servers,
+ await MCPRequestHandler._get_allowed_mcp_servers_for_agent(user_api_key_auth),
+ await MCPRequestHandler._get_agent_access_group_server_ceiling(user_api_key_auth),
)
- agent_access_group_servers: Final = await MCPRequestHandler._get_agent_access_group_server_ceiling(
- user_api_key_auth
- )
- if len(allowed_mcp_servers_for_agent) > 0 or agent_access_group_servers is not None:
+ if agent_capped is not None:
has_lower_level_mcp_restrictions = True
- # Intersect: agent can only use servers allowed by key/team AND agent config AND agent access groups
- allowed_mcp_servers = [
- s
- for s in allowed_mcp_servers
- if (len(allowed_mcp_servers_for_agent) == 0 or s in allowed_mcp_servers_for_agent)
- and (agent_access_group_servers is None or s in agent_access_group_servers)
- ]
+ allowed_mcp_servers = list(agent_capped)
verbose_logger.debug(
"Applied agent intersection filter. Final allowed servers: %s", allowed_mcp_servers
)
@@ -3148,6 +3163,7 @@ class MCPRequestHandler:
@staticmethod
async def _get_agent_access_group_server_ceiling(
user_api_key_auth: UserAPIKeyAuth,
+ resolve_ceiling: CeilingResolver = resolve_agent_access_group_ceiling,
) -> frozenset[str] | None:
"""
Server IDs the agent's attached unified access groups (``LiteLLM_AgentsTable.access_group_ids``)
@@ -3157,13 +3173,10 @@ class MCPRequestHandler:
from litellm.proxy._experimental.mcp_server.mcp_server_manager import (
global_mcp_server_manager,
)
- from litellm.proxy.agent_endpoints.auth.agent_access_groups import (
- resolve_agent_access_group_ceiling,
- )
if not user_api_key_auth.agent_id:
return None
- ceiling: Final = await resolve_agent_access_group_ceiling(user_api_key_auth.agent_id)
+ ceiling: Final = await resolve_ceiling(user_api_key_auth.agent_id)
if ceiling is None:
return None
return frozenset(global_mcp_server_manager.expand_permission_list(sorted(ceiling.mcp_server_ids)))
diff --git a/litellm/proxy/agent_endpoints/auth/agent_access_groups.py b/litellm/proxy/agent_endpoints/auth/agent_access_groups.py
index f5adc897e11..67bb43638e0 100644
--- a/litellm/proxy/agent_endpoints/auth/agent_access_groups.py
+++ b/litellm/proxy/agent_endpoints/auth/agent_access_groups.py
@@ -33,6 +33,9 @@ class AgentAccessGroupCeiling:
agent_ids: frozenset[str]
+CeilingResolver: TypeAlias = Callable[[str], Awaitable[AgentAccessGroupCeiling | None]] # mutable-ok: Callable params
+
+
async def _load_agent(agent_id: str) -> AgentResponse | None:
from litellm.proxy.common_utils.registry_read_through import get_agent_with_read_through
diff --git a/litellm/proxy/agent_endpoints/auth/agent_permission_handler.py b/litellm/proxy/agent_endpoints/auth/agent_permission_handler.py
index 11d2a68072c..1759090a29a 100644
--- a/litellm/proxy/agent_endpoints/auth/agent_permission_handler.py
+++ b/litellm/proxy/agent_endpoints/auth/agent_permission_handler.py
@@ -19,6 +19,10 @@ from litellm.proxy._types import (
LitellmUserRoles,
UserAPIKeyAuth,
)
+from litellm.proxy.agent_endpoints.auth.agent_access_groups import (
+ CeilingResolver,
+ resolve_agent_access_group_ceiling,
+)
from litellm.repositories.table_repositories import AgentsRepository
from litellm.types.agents import AgentResponse
@@ -61,6 +65,7 @@ class AgentRequestHandler:
@staticmethod
async def resolve_agent_access(
user_api_key_auth: UserAPIKeyAuth | None = None,
+ resolve_ceiling: CeilingResolver = resolve_agent_access_group_ceiling,
) -> AgentAccess:
"""
Resolve the agents the given user/key may reach.
@@ -71,7 +76,7 @@ class AgentRequestHandler:
never widen what it reaches.
"""
key_team_access: Final = await AgentRequestHandler._resolve_key_team_agent_access(user_api_key_auth)
- agent_ceiling: Final = await AgentRequestHandler._agent_access_group_ceiling(user_api_key_auth)
+ agent_ceiling: Final = await AgentRequestHandler._agent_access_group_ceiling(user_api_key_auth, resolve_ceiling)
if agent_ceiling is None:
return key_team_access
match key_team_access:
@@ -104,13 +109,12 @@ class AgentRequestHandler:
@staticmethod
async def _agent_access_group_ceiling(
user_api_key_auth: UserAPIKeyAuth | None,
+ resolve_ceiling: CeilingResolver,
) -> frozenset[str] | None:
"""Stable IDs of the agents the calling agent's attached access groups allow; None when none attached."""
- from litellm.proxy.agent_endpoints.auth.agent_access_groups import resolve_agent_access_group_ceiling
-
if user_api_key_auth is None or not user_api_key_auth.agent_id:
return None
- ceiling: Final = await resolve_agent_access_group_ceiling(user_api_key_auth.agent_id)
+ ceiling: Final = await resolve_ceiling(user_api_key_auth.agent_id)
if ceiling is None:
return None
return _to_stable_ids(ceiling.agent_ids)
@@ -119,6 +123,7 @@ class AgentRequestHandler:
async def is_agent_allowed(
agent_id: str,
user_api_key_auth: UserAPIKeyAuth | None = None,
+ resolve_ceiling: CeilingResolver = resolve_agent_access_group_ceiling,
) -> bool:
"""
Check if a specific agent is allowed for the given user/key.
@@ -132,7 +137,7 @@ class AgentRequestHandler:
"""
from litellm.proxy.agent_endpoints.agent_registry import global_agent_registry
- match await AgentRequestHandler.resolve_agent_access(user_api_key_auth):
+ match await AgentRequestHandler.resolve_agent_access(user_api_key_auth, resolve_ceiling):
case UnrestrictedAgentAccess():
return True
case RestrictedAgentAccess(allowed_agent_ids):
diff --git a/litellm/proxy/auth/auth_checks.py b/litellm/proxy/auth/auth_checks.py
index 5c53ee49717..137cf849389 100644
--- a/litellm/proxy/auth/auth_checks.py
+++ b/litellm/proxy/auth/auth_checks.py
@@ -68,6 +68,10 @@ from litellm.proxy._types import (
SpecialModelNames,
UserAPIKeyAuth,
)
+from litellm.proxy.agent_endpoints.auth.agent_access_groups import (
+ CeilingResolver,
+ resolve_agent_access_group_ceiling,
+)
from litellm.proxy.auth.budget_throttle import (
budget_throttle_percentage,
should_throttle_budget_exceeded,
@@ -4199,15 +4203,14 @@ async def _check_agent_access_group_model_access(
model: str | list[str] | None, # mutable-ok: _can_object_call_model and the client message helper take list[str]
valid_token: UserAPIKeyAuth | None,
llm_router: Router | None,
+ resolve_ceiling: CeilingResolver = resolve_agent_access_group_ceiling,
) -> Literal[True]:
"""Raises when the key's agent has access groups attached and none of them names the model.
Attached groups that name no model deny every model; ``_can_object_call_model`` would read
an empty allowlist as unrestricted."""
- from litellm.proxy.agent_endpoints.auth.agent_access_groups import resolve_agent_access_group_ceiling
-
if not model or valid_token is None or not valid_token.agent_id:
return True
- ceiling: Final = await resolve_agent_access_group_ceiling(valid_token.agent_id)
+ ceiling: Final = await resolve_ceiling(valid_token.agent_id)
if ceiling is None:
return True
if not ceiling.models:
diff --git a/tests/test_litellm/proxy/_experimental/mcp_server/auth/test_user_api_key_auth_mcp.py b/tests/test_litellm/proxy/_experimental/mcp_server/auth/test_user_api_key_auth_mcp.py
index 2f8e3d1cb82..f8e648fd383 100644
--- a/tests/test_litellm/proxy/_experimental/mcp_server/auth/test_user_api_key_auth_mcp.py
+++ b/tests/test_litellm/proxy/_experimental/mcp_server/auth/test_user_api_key_auth_mcp.py
@@ -12,6 +12,7 @@ from starlette.datastructures import Headers
from litellm.proxy._experimental.mcp_server.auth.user_api_key_auth_mcp import (
MCPRequestHandler,
UnloadableEntitlementError,
+ _agent_capped_servers,
_is_mcp_admitted_user_subject,
)
from litellm.proxy._types import (
@@ -4169,6 +4170,27 @@ async def test_get_allowed_mcp_servers_for_key_prefers_in_memory_permission():
global_mcp_server_manager.registry.pop("direct-server", None)
+@pytest.mark.parametrize(
+ ("agent_servers", "group_ceiling", "expected"),
+ [
+ ([], frozenset({"server_1"}), ("server_1",)),
+ ([], frozenset({"server_1", "server_2", "server_3"}), ("server_1", "server_2")),
+ ([], frozenset(), ()),
+ (["server_2"], frozenset({"server_1", "server_2"}), ("server_2",)),
+ (["server_1"], frozenset({"server_2"}), ()),
+ (["server_1"], None, ("server_1",)),
+ ],
+)
+def test_agent_capped_servers_intersects_agent_config_and_access_groups(agent_servers, group_ceiling, expected):
+ """The agent's attached access groups cap the key/team servers alongside its own
+ object_permission; groups naming no server deny all."""
+ assert _agent_capped_servers(["server_1", "server_2"], agent_servers, group_ceiling) == expected
+
+
+def test_agent_capped_servers_without_agent_restrictions_is_uncapped():
+ assert _agent_capped_servers(["server_1", "server_2"], [], None) is None
+
+
@pytest.mark.asyncio
class TestAgentMCPPermissions:
"""Test agent-level MCP server and tool permission intersection."""
@@ -4208,64 +4230,45 @@ class TestAgentMCPPermissions:
assert sorted(result) == ["server_1", "server_2"]
mock_agent.assert_called_once_with(user_api_key_auth)
- @pytest.mark.parametrize(
- ("group_ceiling", "expected"),
- [
- (frozenset({"server_1"}), ["server_1"]),
- (frozenset({"server_1", "server_2", "server_3"}), ["server_1", "server_2"]),
- (frozenset(), []),
- ],
- )
- async def test_get_allowed_mcp_servers_agent_access_group_ceiling(self, group_ceiling, expected):
- """The agent's attached access groups cap the key/team servers; groups naming no server deny all."""
- user_api_key_auth = UserAPIKeyAuth(api_key="test-key", user_id="test-user", agent_id="agent-ag")
- with (
- patch.object(MCPRequestHandler, "_get_allowed_mcp_servers_for_key", return_value=["server_1", "server_2"]),
- patch.object(MCPRequestHandler, "_get_allowed_mcp_servers_for_team", return_value=[]),
- patch.object(MCPRequestHandler, "_get_allowed_mcp_servers_for_agent", return_value=[]),
- patch.object(MCPRequestHandler, "_get_agent_access_group_server_ceiling", return_value=group_ceiling),
- ):
- access = await MCPRequestHandler.get_mcp_server_access(user_api_key_auth=user_api_key_auth)
- assert sorted(access.server_ids) == expected
- assert access.scope == "scoped"
-
- async def test_get_allowed_mcp_servers_agent_without_access_groups_is_uncapped(self):
- user_api_key_auth = UserAPIKeyAuth(api_key="test-key", user_id="test-user", agent_id="agent-ag")
- with (
- patch.object(MCPRequestHandler, "_get_allowed_mcp_servers_for_key", return_value=["server_1", "server_2"]),
- patch.object(MCPRequestHandler, "_get_allowed_mcp_servers_for_team", return_value=[]),
- patch.object(MCPRequestHandler, "_get_allowed_mcp_servers_for_agent", return_value=[]),
- patch.object(MCPRequestHandler, "_get_agent_access_group_server_ceiling", return_value=None),
- ):
- result = await MCPRequestHandler.get_allowed_mcp_servers(user_api_key_auth=user_api_key_auth)
- assert sorted(result) == ["server_1", "server_2"]
-
async def test_agent_access_group_server_ceiling_expands_group_servers(self):
+ from litellm.proxy._experimental.mcp_server.mcp_server_manager import global_mcp_server_manager
from litellm.proxy.agent_endpoints.auth.agent_access_groups import AgentAccessGroupCeiling
+ from litellm.types.mcp import MCPTransport
+ from litellm.types.mcp_server.mcp_server_manager import MCPServer
- ceiling = AgentAccessGroupCeiling(
- access_group_ids=("ag-1",),
- models=frozenset(),
- mcp_server_ids=frozenset({"server_1"}),
- agent_ids=frozenset(),
- )
- with (
- patch(
- "litellm.proxy.agent_endpoints.auth.agent_access_groups.resolve_agent_access_group_ceiling",
- new=AsyncMock(return_value=ceiling),
- ),
- patch(
- "litellm.proxy._experimental.mcp_server.mcp_server_manager.global_mcp_server_manager"
- ) as mock_manager,
- ):
- mock_manager.expand_permission_list.return_value = ["server_1"]
- result = await MCPRequestHandler._get_agent_access_group_server_ceiling(
- UserAPIKeyAuth(api_key="test-key", agent_id="agent-ag")
+ asked: list[str] = []
+
+ async def resolve(agent_id: str) -> AgentAccessGroupCeiling | None:
+ asked.append(agent_id)
+ return AgentAccessGroupCeiling(
+ access_group_ids=("ag-1",),
+ models=frozenset(),
+ mcp_server_ids=frozenset({"aliased-server"}),
+ agent_ids=frozenset(),
)
- assert result == frozenset({"server_1"})
- mock_manager.expand_permission_list.assert_called_once_with(["server_1"])
- assert await MCPRequestHandler._get_agent_access_group_server_ceiling(UserAPIKeyAuth(api_key="k")) is None
+ global_mcp_server_manager.registry["ag-server-id"] = MCPServer(
+ server_id="ag-server-id",
+ name="ag-server",
+ server_name="ag-server",
+ alias="aliased-server",
+ url="https://ag-server.example.com",
+ transport=MCPTransport.http,
+ )
+ try:
+ result = await MCPRequestHandler._get_agent_access_group_server_ceiling(
+ UserAPIKeyAuth(api_key="test-key", agent_id="agent-ag"), resolve
+ )
+ finally:
+ global_mcp_server_manager.registry.pop("ag-server-id", None)
+
+ assert result == frozenset({"ag-server-id"})
+ assert asked == ["agent-ag"]
+ assert (
+ await MCPRequestHandler._get_agent_access_group_server_ceiling(UserAPIKeyAuth(api_key="k"), resolve)
+ is None
+ )
+ assert asked == ["agent-ag"]
async def test_get_allowed_mcp_servers_key_team_agent_intersection(self):
"""Key allows [1, 2], agent allows [2, 3]. Result = [2]."""
diff --git a/tests/test_litellm/proxy/agent_endpoints/auth/test_agent_permission_handler.py b/tests/test_litellm/proxy/agent_endpoints/auth/test_agent_permission_handler.py
index a8a55d332b6..2a98e6e4feb 100644
--- a/tests/test_litellm/proxy/agent_endpoints/auth/test_agent_permission_handler.py
+++ b/tests/test_litellm/proxy/agent_endpoints/auth/test_agent_permission_handler.py
@@ -11,9 +11,9 @@ import pytest
from litellm.constants import UI_SESSION_TOKEN_TEAM_ID
-from litellm.proxy._types import LitellmUserRoles, UserAPIKeyAuth
+from litellm.proxy._types import LiteLLM_ObjectPermissionTable, LitellmUserRoles, UserAPIKeyAuth
from litellm.proxy.agent_endpoints.agent_registry import AgentRegistry
-from litellm.proxy.agent_endpoints.auth.agent_access_groups import AgentAccessGroupCeiling
+from litellm.proxy.agent_endpoints.auth.agent_access_groups import AgentAccessGroupCeiling, CeilingResolver
from litellm.proxy.agent_endpoints.auth.agent_permission_handler import (
AgentAccess,
AgentRequestHandler,
@@ -159,86 +159,74 @@ class TestAgentRequestHandler:
), agent_id
@staticmethod
- def _ceiling(agent_ids: frozenset[str]) -> AgentAccessGroupCeiling:
- return AgentAccessGroupCeiling(
- access_group_ids=("ag-1",),
- models=frozenset(),
- mcp_server_ids=frozenset(),
- agent_ids=agent_ids,
+ def _ceiling_resolver(agent_ids: frozenset[str] | None) -> tuple[CeilingResolver, list[str]]:
+ """A resolver that records the agent ids it was asked about and answers with a fixed
+ ceiling, or None when the agent has no access groups attached."""
+ asked: Final[list[str]] = []
+
+ async def resolve(agent_id: str) -> AgentAccessGroupCeiling | None:
+ asked.append(agent_id)
+ if agent_ids is None:
+ return None
+ return AgentAccessGroupCeiling(
+ access_group_ids=("ag-1",), models=frozenset(), mcp_server_ids=frozenset(), agent_ids=agent_ids
+ )
+
+ return resolve, asked
+
+ @staticmethod
+ def _key_granting(agent_ids: list[str], agent_id: str | None) -> UserAPIKeyAuth:
+ return UserAPIKeyAuth(
+ api_key="test-key",
+ user_id="test-user",
+ agent_id=agent_id,
+ object_permission=LiteLLM_ObjectPermissionTable(object_permission_id="obj-1", agents=agent_ids),
)
async def test_agent_access_groups_cap_an_otherwise_unrestricted_key(self):
"""A key with no agent grant of its own may still only reach the agents its
agent's attached access groups name."""
agent_key: Final = UserAPIKeyAuth(api_key="test-key", user_id="test-user", agent_id="caller-agent")
+ resolve, asked = self._ceiling_resolver(frozenset({"agent-beta"}))
- with (
- patch.object(AgentRequestHandler, "_get_allowed_agents_for_key", return_value=UnrestrictedAgentAccess()),
- patch.object(AgentRequestHandler, "_get_allowed_agents_for_team", return_value=UnrestrictedAgentAccess()),
- patch(
- "litellm.proxy.agent_endpoints.auth.agent_access_groups.resolve_agent_access_group_ceiling",
- new=AsyncMock(return_value=self._ceiling(frozenset({"agent-beta"}))),
- ) as mock_ceiling,
- ):
- assert await AgentRequestHandler.resolve_agent_access(agent_key) == RestrictedAgentAccess(
- frozenset({"agent-beta"})
- )
- assert await AgentRequestHandler.is_agent_allowed("agent-beta", agent_key) is True
- assert await AgentRequestHandler.is_agent_allowed("agent-alpha", agent_key) is False
- mock_ceiling.assert_called_with("caller-agent")
-
- async def test_agent_access_groups_intersect_with_key_and_team_grants(self):
- agent_key: Final = UserAPIKeyAuth(
- api_key="test-key", user_id="test-user", team_id="test-team", agent_id="caller-agent"
+ assert await AgentRequestHandler.resolve_agent_access(agent_key, resolve) == RestrictedAgentAccess(
+ frozenset({"agent-beta"})
)
+ assert await AgentRequestHandler.is_agent_allowed("agent-beta", agent_key, resolve) is True
+ assert await AgentRequestHandler.is_agent_allowed("agent-alpha", agent_key, resolve) is False
+ assert asked == ["caller-agent"] * 3
- with (
- patch.object(
- AgentRequestHandler,
- "_get_allowed_agents_for_key",
- return_value=RestrictedAgentAccess(frozenset({"agent-alpha", "agent-beta"})),
- ),
- patch.object(
- AgentRequestHandler,
- "_get_allowed_agents_for_team",
- return_value=RestrictedAgentAccess(frozenset({"agent-alpha", "agent-beta", "agent-gamma"})),
- ),
- patch(
- "litellm.proxy.agent_endpoints.auth.agent_access_groups.resolve_agent_access_group_ceiling",
- new=AsyncMock(return_value=self._ceiling(frozenset({"agent-beta", "agent-gamma"}))),
- ),
- ):
- assert await AgentRequestHandler.resolve_agent_access(agent_key) == RestrictedAgentAccess(
- frozenset({"agent-beta"})
- )
+ async def test_agent_access_groups_intersect_with_key_grants(self):
+ agent_key: Final = self._key_granting(["agent-alpha", "agent-beta"], agent_id="caller-agent")
+ resolve, _ = self._ceiling_resolver(frozenset({"agent-beta", "agent-gamma"}))
+
+ assert await AgentRequestHandler.resolve_agent_access(agent_key, resolve) == RestrictedAgentAccess(
+ frozenset({"agent-beta"})
+ )
+ assert await AgentRequestHandler.is_agent_allowed("agent-gamma", agent_key, resolve) is False
async def test_agent_access_groups_naming_no_agent_deny_every_agent(self):
agent_key: Final = UserAPIKeyAuth(api_key="test-key", user_id="test-user", agent_id="caller-agent")
+ resolve, _ = self._ceiling_resolver(frozenset())
- with (
- patch.object(AgentRequestHandler, "_get_allowed_agents_for_key", return_value=UnrestrictedAgentAccess()),
- patch.object(AgentRequestHandler, "_get_allowed_agents_for_team", return_value=UnrestrictedAgentAccess()),
- patch(
- "litellm.proxy.agent_endpoints.auth.agent_access_groups.resolve_agent_access_group_ceiling",
- new=AsyncMock(return_value=self._ceiling(frozenset())),
- ),
- ):
- assert await AgentRequestHandler.resolve_agent_access(agent_key) == RestrictedAgentAccess(frozenset())
- assert await AgentRequestHandler.is_agent_allowed("agent-alpha", agent_key) is False
+ assert await AgentRequestHandler.resolve_agent_access(agent_key, resolve) == RestrictedAgentAccess(frozenset())
+ assert await AgentRequestHandler.is_agent_allowed("agent-alpha", agent_key, resolve) is False
+
+ async def test_agent_without_access_groups_keeps_key_grants(self):
+ agent_key: Final = self._key_granting(["agent-alpha"], agent_id="caller-agent")
+ resolve, asked = self._ceiling_resolver(None)
+
+ assert await AgentRequestHandler.resolve_agent_access(agent_key, resolve) == RestrictedAgentAccess(
+ frozenset({"agent-alpha"})
+ )
+ assert asked == ["caller-agent"]
async def test_key_without_agent_never_consults_agent_access_groups(self):
plain_key: Final = UserAPIKeyAuth(api_key="test-key", user_id="test-user")
+ resolve, asked = self._ceiling_resolver(frozenset())
- with (
- patch.object(AgentRequestHandler, "_get_allowed_agents_for_key", return_value=UnrestrictedAgentAccess()),
- patch.object(AgentRequestHandler, "_get_allowed_agents_for_team", return_value=UnrestrictedAgentAccess()),
- patch(
- "litellm.proxy.agent_endpoints.auth.agent_access_groups.resolve_agent_access_group_ceiling",
- new=AsyncMock(return_value=self._ceiling(frozenset())),
- ) as mock_ceiling,
- ):
- assert await AgentRequestHandler.resolve_agent_access(plain_key) == UnrestrictedAgentAccess()
- mock_ceiling.assert_not_called()
+ assert await AgentRequestHandler.resolve_agent_access(plain_key, resolve) == UnrestrictedAgentAccess()
+ assert asked == []
async def test_empty_access_group_denies_every_agent(self):
"""LIT-5143: a key restricted to an access group that resolves to no agents is
diff --git a/tests/test_litellm/proxy/auth/test_auth_checks.py b/tests/test_litellm/proxy/auth/test_auth_checks.py
index ad1742db4b7..b9b9a786d93 100644
--- a/tests/test_litellm/proxy/auth/test_auth_checks.py
+++ b/tests/test_litellm/proxy/auth/test_auth_checks.py
@@ -32,11 +32,13 @@ from litellm.proxy._types import (
UserAPIKeyAuth,
WebhookEvent,
)
+from litellm.proxy.agent_endpoints.auth.agent_access_groups import AgentAccessGroupCeiling, CeilingResolver
from litellm.proxy.auth.auth_checks import (
ExperimentalUIJWTToken,
_cache_management_object,
_can_object_call_model,
_can_object_call_vector_stores,
+ _check_agent_access_group_model_access,
_check_end_user_budget,
_check_team_member_budget,
_fetch_key_object_from_db_with_reconnect,
@@ -8466,89 +8468,64 @@ def test_request_skips_budget_checks_extends_route_rule_with_zero_cost_models()
# Agent access group model ceiling
-def _agent_model_ceiling(models: frozenset[str]):
- from litellm.proxy.agent_endpoints.auth.agent_access_groups import AgentAccessGroupCeiling
+def _agent_model_ceiling_resolver(
+ models: frozenset[str] | None,
+) -> tuple[CeilingResolver, list[str]]:
+ """Resolver that records the agent ids it was asked about and answers with a fixed model
+ ceiling, or None when the agent has no access groups attached."""
+ asked: Final[list[str]] = []
- return AgentAccessGroupCeiling(
- access_group_ids=("ag-1",), models=models, mcp_server_ids=frozenset(), agent_ids=frozenset()
- )
+ async def resolve(agent_id: str) -> AgentAccessGroupCeiling | None:
+ asked.append(agent_id)
+ if models is None:
+ return None
+ return AgentAccessGroupCeiling(
+ access_group_ids=("ag-1",), models=models, mcp_server_ids=frozenset(), agent_ids=frozenset()
+ )
-
-async def _run_common_checks_for_agent_key(model: str, valid_token: UserAPIKeyAuth):
- from fastapi import Request
-
- from litellm.proxy.auth.auth_checks import common_checks
-
- return await common_checks(
- request_body={"model": model, "messages": [{"role": "user", "content": "hi"}]},
- team_object=None,
- user_object=None,
- end_user_object=None,
- global_proxy_spend=None,
- general_settings={},
- route="/chat/completions",
- llm_router=None,
- proxy_logging_obj=MagicMock(),
- valid_token=valid_token,
- request=MagicMock(spec=Request),
- )
+ return resolve, asked
@pytest.mark.asyncio
-async def test_common_checks_agent_access_groups_cap_models_even_when_key_allows_them():
+async def test_agent_access_groups_cap_models_even_when_key_allows_them():
agent_key: Final = UserAPIKeyAuth(token="agent-token", agent_id="agent-1", models=["gpt-5", "claude-sonnet"])
+ resolve, asked = _agent_model_ceiling_resolver(frozenset({"gpt-5"}))
- with patch(
- "litellm.proxy.agent_endpoints.auth.agent_access_groups.resolve_agent_access_group_ceiling",
- new=AsyncMock(return_value=_agent_model_ceiling(frozenset({"gpt-5"}))),
- ):
- assert await _run_common_checks_for_agent_key("gpt-5", agent_key) is True
+ assert await _check_agent_access_group_model_access("gpt-5", agent_key, None, resolve) is True
- with pytest.raises(ProxyException) as exc_info:
- await _run_common_checks_for_agent_key("claude-sonnet", agent_key)
+ with pytest.raises(ProxyException) as exc_info:
+ await _check_agent_access_group_model_access("claude-sonnet", agent_key, None, resolve)
assert exc_info.value.type == ProxyErrorTypes.agent_model_access_denied
assert exc_info.value.code == str(status.HTTP_403_FORBIDDEN)
+ assert asked == ["agent-1", "agent-1"]
@pytest.mark.asyncio
-async def test_common_checks_agent_access_groups_naming_no_model_deny_every_model():
+async def test_agent_access_groups_naming_no_model_deny_every_model():
agent_key: Final = UserAPIKeyAuth(token="agent-token", agent_id="agent-1", models=[])
+ resolve, _ = _agent_model_ceiling_resolver(frozenset())
- with (
- patch(
- "litellm.proxy.agent_endpoints.auth.agent_access_groups.resolve_agent_access_group_ceiling",
- new=AsyncMock(return_value=_agent_model_ceiling(frozenset())),
- ),
- pytest.raises(ProxyException) as exc_info,
- ):
- await _run_common_checks_for_agent_key("gpt-5", agent_key)
+ with pytest.raises(ProxyException) as exc_info:
+ await _check_agent_access_group_model_access("gpt-5", agent_key, None, resolve)
assert exc_info.value.type == ProxyErrorTypes.agent_model_access_denied
@pytest.mark.asyncio
-async def test_common_checks_agent_without_access_groups_adds_no_model_ceiling():
+async def test_agent_without_access_groups_adds_no_model_ceiling():
agent_key: Final = UserAPIKeyAuth(token="agent-token", agent_id="agent-1", models=["gpt-5", "claude-sonnet"])
+ resolve, asked = _agent_model_ceiling_resolver(None)
- with patch(
- "litellm.proxy.agent_endpoints.auth.agent_access_groups.resolve_agent_access_group_ceiling",
- new=AsyncMock(return_value=None),
- ) as mock_ceiling:
- assert await _run_common_checks_for_agent_key("gpt-5", agent_key) is True
- assert await _run_common_checks_for_agent_key("claude-sonnet", agent_key) is True
-
- mock_ceiling.assert_called_with("agent-1")
+ assert await _check_agent_access_group_model_access("gpt-5", agent_key, None, resolve) is True
+ assert await _check_agent_access_group_model_access("claude-sonnet", agent_key, None, resolve) is True
+ assert asked == ["agent-1", "agent-1"]
@pytest.mark.asyncio
-async def test_common_checks_key_without_agent_never_consults_agent_access_groups():
+async def test_key_without_agent_never_consults_agent_access_groups():
plain_key: Final = UserAPIKeyAuth(token="plain-token", models=["gpt-5"])
+ resolve, asked = _agent_model_ceiling_resolver(frozenset())
- with patch(
- "litellm.proxy.agent_endpoints.auth.agent_access_groups.resolve_agent_access_group_ceiling",
- new=AsyncMock(return_value=_agent_model_ceiling(frozenset())),
- ) as mock_ceiling:
- assert await _run_common_checks_for_agent_key("gpt-5", plain_key) is True
-
- mock_ceiling.assert_not_called()
+ assert await _check_agent_access_group_model_access("gpt-5", plain_key, None, resolve) is True
+ assert asked == []
From 5b04560997e5838743d852dd6af97749e5979ff9 Mon Sep 17 00:00:00 2001
From: yassin
Date: Thu, 17 Sep 2026 20:30:37 +0000
Subject: [PATCH 008/160] refactor(agents): keep the model listing cap within
the type-discipline budget
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
---
.../auth/agent_access_groups.py | 104 ++++++++++++++----
litellm/proxy/utils.py | 60 +++++++++-
2 files changed, 144 insertions(+), 20 deletions(-)
diff --git a/litellm/proxy/agent_endpoints/auth/agent_access_groups.py b/litellm/proxy/agent_endpoints/auth/agent_access_groups.py
index 67bb43638e0..e0e5193d0a4 100644
--- a/litellm/proxy/agent_endpoints/auth/agent_access_groups.py
+++ b/litellm/proxy/agent_endpoints/auth/agent_access_groups.py
@@ -1,33 +1,41 @@
-"""
-Ceiling that an agent's attached access groups place on requests made with that agent's key.
-
-Keys and teams use access groups as grants. An agent uses them the way it already uses its
-``object_permission``: the union of the attached groups caps what the agent's key can reach,
-on top of whatever the key and team allow. A group that cannot be loaded contributes nothing,
-so a missing or unreadable group can only narrow the agent, never widen it.
-"""
-
import asyncio
-from collections.abc import Awaitable, Callable
+from collections.abc import Awaitable, Callable, Sequence
from dataclasses import dataclass
-from typing import Final, TypeAlias
+from typing import Final, Protocol, TypeAlias
from fastapi import HTTPException
+from pydantic import TypeAdapter, ValidationError
+from typing_extensions import ReadOnly, TypedDict
from litellm._logging import verbose_proxy_logger
+from litellm.caching.dual_cache import DualCache
from litellm.proxy._types import LiteLLM_AccessGroupTable
-from litellm.types.agents import AgentResponse
+from litellm.proxy.common_utils.user_api_key_cache import get_management_object_ttl
-AgentLoader: TypeAlias = Callable[[str], Awaitable[AgentResponse | None]] # mutable-ok: Callable parameter syntax
+
+class _AgentAccessGroupsRecord(Protocol):
+ @property
+ def access_group_ids(self) -> Sequence[str] | None: ...
+
+
+class _AgentIdWhere(TypedDict):
+ agent_id: ReadOnly[str]
+
+
+AccessGroupIds: TypeAlias = tuple[str, ...]
+AccessGroupIdsLoader: TypeAlias = Callable[[str], Awaitable[AccessGroupIds]] # mutable-ok: Callable params
+AgentRecordFinder: TypeAlias = Callable[[str], Awaitable[_AgentAccessGroupsRecord | None]] # mutable-ok: Callable
LoadedAccessGroup: TypeAlias = LiteLLM_AccessGroupTable | None
AccessGroupLoader: TypeAlias = Callable[[str], Awaitable[LoadedAccessGroup]] # mutable-ok: Callable parameter syntax
+_CACHED_IDS: Final = TypeAdapter(list[str])
+
@dataclass(frozen=True, slots=True)
class AgentAccessGroupCeiling:
"""Everything the agent's attached access groups allow. An empty set denies that resource kind."""
- access_group_ids: tuple[str, ...]
+ access_group_ids: AccessGroupIds
models: frozenset[str]
mcp_server_ids: frozenset[str]
agent_ids: frozenset[str]
@@ -36,10 +44,69 @@ class AgentAccessGroupCeiling:
CeilingResolver: TypeAlias = Callable[[str], Awaitable[AgentAccessGroupCeiling | None]] # mutable-ok: Callable params
-async def _load_agent(agent_id: str) -> AgentResponse | None:
+def agent_access_group_ids_cache_key(agent_id: str) -> str:
+ return f"agent_access_group_ids:{agent_id}"
+
+
+def _cached_access_group_ids(cached: object) -> AccessGroupIds | None:
+ if cached is None:
+ return None
+ try:
+ return tuple(_CACHED_IDS.validate_python(cached))
+ except ValidationError:
+ return None
+
+
+async def _registry_access_group_ids(agent_id: str) -> AccessGroupIds:
from litellm.proxy.common_utils.registry_read_through import get_agent_with_read_through
- return await get_agent_with_read_through(agent_id)
+ agent: Final = await get_agent_with_read_through(agent_id)
+ return tuple(agent.access_group_ids or ()) if agent is not None else ()
+
+
+async def load_agent_access_group_ids(
+ agent_id: str,
+ cache: DualCache,
+ find_agent: AgentRecordFinder,
+ fallback: AccessGroupIdsLoader,
+) -> AccessGroupIds:
+ """The agent row's groups, cached for the management-object TTL and evicted on every agent write."""
+ cache_key: Final = agent_access_group_ids_cache_key(agent_id)
+ cached: Final = _cached_access_group_ids(await cache.async_get_cache(key=cache_key))
+ if cached is not None:
+ return cached
+ try:
+ record: Final = await find_agent(agent_id)
+ except Exception as e: # noqa: BLE001 # prisma raises many error types; the registry snapshot answers instead
+ verbose_proxy_logger.warning("Failed to read access groups for agent %r, using registry: %s", agent_id, e)
+ return await fallback(agent_id)
+ access_group_ids: Final = tuple(record.access_group_ids or ()) if record is not None else ()
+ await cache.async_set_cache(key=cache_key, value=access_group_ids, ttl=get_management_object_ttl(cache))
+ return access_group_ids
+
+
+async def _load_agent_access_group_ids(agent_id: str) -> AccessGroupIds:
+ from litellm.proxy.agent_endpoints.agent_registry import agents_table
+ from litellm.proxy.proxy_server import prisma_client, user_api_key_cache
+
+ if prisma_client is None:
+ return await _registry_access_group_ids(agent_id)
+ db: Final = prisma_client
+
+ async def find_agent(row_agent_id: str) -> _AgentAccessGroupsRecord | None:
+ return await agents_table(db).find_unique(where=_AgentIdWhere(agent_id=row_agent_id))
+
+ return await load_agent_access_group_ids(agent_id, user_api_key_cache, find_agent, _registry_access_group_ids)
+
+
+async def evict_agent_access_group_ids(agent_ids: Sequence[str]) -> None:
+ from litellm.proxy.common_utils.auth_cache_invalidation_pubsub import evict_and_broadcast
+ from litellm.proxy.proxy_server import user_api_key_cache
+
+ await evict_and_broadcast(
+ cache_keys=tuple(agent_access_group_ids_cache_key(agent_id) for agent_id in agent_ids),
+ user_api_key_cache=user_api_key_cache,
+ )
async def _load_access_group(access_group_id: str) -> LoadedAccessGroup:
@@ -65,12 +132,11 @@ async def _load_access_group(access_group_id: str) -> LoadedAccessGroup:
async def resolve_agent_access_group_ceiling(
agent_id: str,
- load_agent: AgentLoader = _load_agent,
+ load_access_group_ids: AccessGroupIdsLoader = _load_agent_access_group_ids,
load_access_group: AccessGroupLoader = _load_access_group,
) -> AgentAccessGroupCeiling | None:
"""``None`` when the agent has no access groups attached, so nothing is capped."""
- agent: Final = await load_agent(agent_id)
- access_group_ids: Final = tuple(agent.access_group_ids or ()) if agent is not None else ()
+ access_group_ids: Final = await load_access_group_ids(agent_id)
if not access_group_ids:
return None
diff --git a/litellm/proxy/utils.py b/litellm/proxy/utils.py
index 950ac5e9906..95c221a8f20 100644
--- a/litellm/proxy/utils.py
+++ b/litellm/proxy/utils.py
@@ -122,6 +122,7 @@ from litellm.proxy._types import (
Member,
UserAPIKeyAuth,
)
+from litellm.proxy.agent_endpoints.auth.agent_access_groups import CeilingResolver, resolve_agent_access_group_ceiling
from litellm.proxy.auth.route_checks import RouteChecks
from litellm.proxy.common_utils.config_sync_pubsub import publish_config_param_change
from litellm.proxy.common_utils.user_api_key_cache import UserApiKeyCache
@@ -7980,6 +7981,51 @@ async def _get_access_group_models(
return tuple(dict.fromkeys((*team_group_models, *key_group_models)))
+async def _agent_access_group_visible_models(
+ user_api_key_dict: "UserAPIKeyAuth",
+ llm_router: Optional["Router"],
+ include_model_access_groups: bool,
+ return_wildcard_routes: bool,
+ team_id: str | None,
+ resolve_agent_ceiling: CeilingResolver,
+) -> frozenset[str] | None:
+ """Models an agent key may still list once its attached access groups cap it, ``None`` when
+ nothing caps it, so ``/v1/models`` never advertises a model the same key would be denied on."""
+ from litellm.proxy.auth.model_checks import get_complete_model_list, get_team_models
+
+ if not user_api_key_dict.agent_id:
+ return None
+ ceiling: Final = await resolve_agent_ceiling(user_api_key_dict.agent_id)
+ if ceiling is None:
+ return None
+ if llm_router is None:
+ return ceiling.models
+ proxy_model_list: Final = llm_router.get_model_names()
+ model_access_groups: Final = llm_router.get_model_access_groups()
+ granted: Final = get_team_models(
+ team_models=sorted(ceiling.models),
+ proxy_model_list=proxy_model_list,
+ model_access_groups=model_access_groups,
+ include_model_access_groups=include_model_access_groups,
+ )
+ if not granted:
+ return frozenset()
+ return frozenset(
+ get_complete_model_list(
+ key_models=granted,
+ team_models=(),
+ proxy_model_list=proxy_model_list,
+ user_model=None,
+ infer_model_from_keys=False,
+ return_wildcard_routes=return_wildcard_routes,
+ llm_router=llm_router,
+ model_access_groups=model_access_groups,
+ include_model_access_groups=include_model_access_groups,
+ team_id=team_id,
+ )
+ )
+
+
async def get_available_models_for_user(
user_api_key_dict: "UserAPIKeyAuth",
llm_router: Optional["Router"],
@@ -7992,6 +8038,7 @@ async def get_available_models_for_user(
only_model_access_groups: bool = False,
return_wildcard_routes: bool = False,
user_api_key_cache: Optional["UserApiKeyCache"] = None,
+ resolve_agent_ceiling: CeilingResolver = resolve_agent_access_group_ceiling,
) -> list[str]:
"""
Get the list of models available to a user based on their API key and team permissions.
@@ -8095,7 +8142,18 @@ async def get_available_models_for_user(
team_id=effective_team_id,
)
- return all_models
+ agent_visible: Final = await _agent_access_group_visible_models(
+ user_api_key_dict=user_api_key_dict,
+ llm_router=llm_router,
+ include_model_access_groups=include_model_access_groups,
+ return_wildcard_routes=return_wildcard_routes,
+ team_id=effective_team_id,
+ resolve_agent_ceiling=resolve_agent_ceiling,
+ )
+ if agent_visible is None:
+ return all_models
+ capped: Final = [m for m in all_models if m in agent_visible] # mutable-ok: callers expect the list all_models is
+ return capped
def _safe_get_model_info(model: str, get_model_info: Callable[[str], ModelInfo]) -> ModelInfo | None:
From 3a86567c9d3ffcd12507407215ea14d9d89d2184 Mon Sep 17 00:00:00 2001
From: yassin
Date: Thu, 17 Sep 2026 20:31:03 +0000
Subject: [PATCH 009/160] fix(agents): evict the cached agent access groups on
every agent write and cap the model listing
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
---
.../proxy/agent_endpoints/agent_registry.py | 3 +
litellm/proxy/agent_endpoints/endpoints.py | 4 +
.../access_group_endpoints.py | 2 +
.../auth/test_agent_access_groups.py | 91 ++++++++++++++++++-
.../proxy/utils/helpers/test_model_access.py | 87 ++++++++++++++++--
5 files changed, 177 insertions(+), 10 deletions(-)
diff --git a/litellm/proxy/agent_endpoints/agent_registry.py b/litellm/proxy/agent_endpoints/agent_registry.py
index c8948d8d70e..d6b12e830e1 100644
--- a/litellm/proxy/agent_endpoints/agent_registry.py
+++ b/litellm/proxy/agent_endpoints/agent_registry.py
@@ -67,6 +67,9 @@ class AgentRecord(Protocol):
@property
def object_permission(self) -> AgentObjectPermissionRecord | None: ...
+ @property
+ def access_group_ids(self) -> Sequence[str] | None: ...
+
@property
def spend(self) -> float: ...
diff --git a/litellm/proxy/agent_endpoints/endpoints.py b/litellm/proxy/agent_endpoints/endpoints.py
index aa8979a73c6..62783fb412a 100644
--- a/litellm/proxy/agent_endpoints/endpoints.py
+++ b/litellm/proxy/agent_endpoints/endpoints.py
@@ -44,6 +44,7 @@ from litellm.proxy.agent_endpoints.agent_search import (
global_agent_search_index,
search_agents,
)
+from litellm.proxy.agent_endpoints.auth.agent_access_groups import evict_agent_access_group_ids
from litellm.proxy.agent_endpoints.auth.agent_permission_handler import accessible_agents
from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
from litellm.proxy.common_utils.rbac_utils import check_feature_access_for_user
@@ -696,6 +697,7 @@ async def update_agent(
prisma_client=prisma_client,
updated_by=updated_by,
)
+ await evict_agent_access_group_ids((agent_id,))
# deregister in memory
AGENT_REGISTRY.deregister_agent(agent_name=existing_agent.get("agent_name"))
@@ -799,6 +801,7 @@ async def patch_agent(
prisma_client=prisma_client,
updated_by=updated_by,
)
+ await evict_agent_access_group_ids((agent_id,))
# deregister in memory
AGENT_REGISTRY.deregister_agent(agent_name=existing_agent.get("agent_name"))
@@ -861,6 +864,7 @@ async def delete_agent(
raise HTTPException(status_code=404, detail=f"Agent with ID {agent_id} not found in DB.")
await AGENT_REGISTRY.delete_agent_from_db(agent_id=agent_id, prisma_client=prisma_client)
+ await evict_agent_access_group_ids((agent_id,))
AGENT_REGISTRY.deregister_agent(agent_name=existing_agent.get("agent_name"))
diff --git a/litellm/proxy/management_endpoints/access_group_endpoints.py b/litellm/proxy/management_endpoints/access_group_endpoints.py
index b4923b0a2dc..2694d00b17f 100644
--- a/litellm/proxy/management_endpoints/access_group_endpoints.py
+++ b/litellm/proxy/management_endpoints/access_group_endpoints.py
@@ -16,6 +16,7 @@ from litellm.proxy._types import (
UserAPIKeyAuth,
)
from litellm.proxy.agent_endpoints.agent_registry import global_agent_registry
+from litellm.proxy.agent_endpoints.auth.agent_access_groups import evict_agent_access_group_ids
from litellm.proxy.auth.auth_checks import (
_cache_access_object,
_cache_key_object,
@@ -782,6 +783,7 @@ async def delete_access_group(
await invalidate_access_group_cache(access_group_id)
_detach_access_group_from_agent_registry(detached_agent_ids, access_group_id)
+ await evict_agent_access_group_ids(detached_agent_ids)
await _patch_team_caches_remove_access_group(
affected_team_ids, access_group_id, user_api_key_cache, proxy_logging_obj
)
diff --git a/tests/test_litellm/proxy/agent_endpoints/auth/test_agent_access_groups.py b/tests/test_litellm/proxy/agent_endpoints/auth/test_agent_access_groups.py
index 8c31c9428e7..266ce52e3a9 100644
--- a/tests/test_litellm/proxy/agent_endpoints/auth/test_agent_access_groups.py
+++ b/tests/test_litellm/proxy/agent_endpoints/auth/test_agent_access_groups.py
@@ -1,11 +1,16 @@
+from collections.abc import Sequence
+from dataclasses import dataclass
from typing import Final
import pytest
from fastapi import HTTPException
+from litellm.caching.dual_cache import DualCache
from litellm.models.access_group import LiteLLM_AccessGroupTable
from litellm.proxy.agent_endpoints.auth.agent_access_groups import (
AgentAccessGroupCeiling,
+ agent_access_group_ids_cache_key,
+ load_agent_access_group_ids,
resolve_agent_access_group_ceiling,
)
from litellm.types.agents import AgentResponse
@@ -35,8 +40,8 @@ def _group(
def _loaders(agent: AgentResponse | None, groups: dict[str, LiteLLM_AccessGroupTable]):
- async def load_agent(agent_id: str) -> AgentResponse | None:
- return agent
+ async def load_agent(agent_id: str) -> tuple[str, ...]:
+ return tuple(agent.access_group_ids or ()) if agent is not None else ()
async def load_group(group_id: str) -> LiteLLM_AccessGroupTable | None:
return groups.get(group_id)
@@ -105,6 +110,88 @@ async def test_only_unloadable_groups_is_an_empty_ceiling_not_unrestricted():
assert ceiling.agent_ids == frozenset()
+@dataclass(frozen=True, slots=True)
+class _AgentRow:
+ access_group_ids: Sequence[str] | None
+
+
+class _FakeAgentTable:
+ def __init__(self, rows: dict[str, _AgentRow], failing: bool = False) -> None:
+ self._rows: Final = rows
+ self._failing: Final = failing
+ self.reads = 0
+
+ async def find_agent(self, agent_id: str) -> _AgentRow | None:
+ self.reads += 1
+ if self._failing:
+ raise RuntimeError("db down")
+ return self._rows.get(agent_id)
+
+
+async def _registry_snapshot(agent_id: str) -> tuple[str, ...]:
+ return ("registry-group",)
+
+
+@pytest.mark.asyncio
+async def test_agent_row_is_read_once_then_served_from_cache():
+ cache: Final = DualCache()
+ table: Final = _FakeAgentTable({"agent-1": _AgentRow(["g1", "g2"])})
+
+ first: Final = await load_agent_access_group_ids("agent-1", cache, table.find_agent, _registry_snapshot)
+ second: Final = await load_agent_access_group_ids("agent-1", cache, table.find_agent, _registry_snapshot)
+
+ assert (first, second, table.reads) == (("g1", "g2"), ("g1", "g2"), 1)
+
+
+@pytest.mark.asyncio
+async def test_agent_with_no_row_or_no_groups_caches_an_empty_answer():
+ cache: Final = DualCache()
+ table: Final = _FakeAgentTable({"bare": _AgentRow(None)})
+
+ bare: Final = await load_agent_access_group_ids("bare", cache, table.find_agent, _registry_snapshot)
+ missing: Final = await load_agent_access_group_ids("missing", cache, table.find_agent, _registry_snapshot)
+ again: Final = await load_agent_access_group_ids("missing", cache, table.find_agent, _registry_snapshot)
+
+ assert (bare, missing, again, table.reads) == ((), (), (), 2)
+
+
+@pytest.mark.asyncio
+async def test_evicted_cache_entry_picks_up_the_patched_row():
+ cache: Final = DualCache()
+ rows: Final = {"agent-1": _AgentRow(["g1"])}
+ table: Final = _FakeAgentTable(rows)
+ await load_agent_access_group_ids("agent-1", cache, table.find_agent, _registry_snapshot)
+
+ rows["agent-1"] = _AgentRow(["g2"])
+ stale: Final = await load_agent_access_group_ids("agent-1", cache, table.find_agent, _registry_snapshot)
+ await cache.async_delete_cache(key=agent_access_group_ids_cache_key("agent-1"))
+ fresh: Final = await load_agent_access_group_ids("agent-1", cache, table.find_agent, _registry_snapshot)
+
+ assert (stale, fresh) == (("g1",), ("g2",))
+
+
+@pytest.mark.asyncio
+async def test_unreadable_row_falls_back_to_the_registry_without_caching():
+ cache: Final = DualCache()
+ table: Final = _FakeAgentTable({}, failing=True)
+
+ answer: Final = await load_agent_access_group_ids("agent-1", cache, table.find_agent, _registry_snapshot)
+
+ assert answer == ("registry-group",)
+ assert await cache.async_get_cache(key=agent_access_group_ids_cache_key("agent-1")) is None
+
+
+@pytest.mark.asyncio
+async def test_garbage_in_the_cache_is_treated_as_a_miss():
+ cache: Final = DualCache()
+ await cache.async_set_cache(key=agent_access_group_ids_cache_key("agent-1"), value={"not": "a list"})
+ table: Final = _FakeAgentTable({"agent-1": _AgentRow(["g1"])})
+
+ answer: Final = await load_agent_access_group_ids("agent-1", cache, table.find_agent, _registry_snapshot)
+
+ assert (answer, table.reads) == (("g1",), 1)
+
+
@pytest.mark.asyncio
async def test_default_loader_treats_a_missing_group_as_unreadable(monkeypatch: pytest.MonkeyPatch):
from litellm.proxy import proxy_server
diff --git a/tests/test_litellm/proxy/utils/helpers/test_model_access.py b/tests/test_litellm/proxy/utils/helpers/test_model_access.py
index 5fb4392eec6..7f77f938323 100644
--- a/tests/test_litellm/proxy/utils/helpers/test_model_access.py
+++ b/tests/test_litellm/proxy/utils/helpers/test_model_access.py
@@ -110,8 +110,7 @@ def test_create_model_info_response_happy_path_no_metadata():
"owned_by": result["owned_by"],
"created_is_int": isinstance(result["created"], int),
"metadata_absent": "metadata" not in result,
- "max_input_tokens_positive_int": isinstance(result["max_input_tokens"], int)
- and result["max_input_tokens"] > 0,
+ "max_input_tokens_positive_int": isinstance(result["max_input_tokens"], int) and result["max_input_tokens"] > 0,
"max_output_tokens_positive_int": isinstance(result["max_output_tokens"], int)
and result["max_output_tokens"] > 0,
}
@@ -205,9 +204,7 @@ def test_validate_model_access_happy_path_single_model_in_list():
def test_validate_model_access_happy_path_batch_all_accessible():
summary = {
- "result": validate_model_access(
- "gpt-4o,claude-haiku", ["gpt-4o", "claude-haiku", "gemini"]
- ),
+ "result": validate_model_access("gpt-4o,claude-haiku", ["gpt-4o", "claude-haiku", "gemini"]),
"input": "gpt-4o,claude-haiku",
"available": ["gpt-4o", "claude-haiku", "gemini"],
}
@@ -389,9 +386,7 @@ async def test_get_available_models_for_user_error_path_complete_list_raises(
def _boom(**_kwargs):
raise RuntimeError("downstream failure")
- monkeypatch.setattr(
- "litellm.proxy.auth.model_checks.get_complete_model_list", _boom
- )
+ monkeypatch.setattr("litellm.proxy.auth.model_checks.get_complete_model_list", _boom)
user_api_key_dict = UserAPIKeyAuth(
api_key="sk-test-key",
user_id="user-1",
@@ -481,6 +476,7 @@ async def test_get_available_models_for_user_without_access_groups_grants_nothin
)
assert result == []
+
@pytest.mark.asyncio
async def test_get_available_models_for_user_resolves_key_access_group_models(
monkeypatch,
@@ -521,3 +517,78 @@ async def test_get_available_models_for_user_resolves_key_access_group_models(
user_api_key_cache=MagicMock(),
)
assert result == ["model-b"]
+
+
+def _agent_ceiling(models: frozenset[str] | None):
+ from litellm.proxy.agent_endpoints.auth.agent_access_groups import AgentAccessGroupCeiling
+
+ async def resolve(agent_id: str) -> AgentAccessGroupCeiling | None:
+ if models is None:
+ return None
+ return AgentAccessGroupCeiling(
+ access_group_ids=("ag-agent",), models=models, mcp_server_ids=frozenset(), agent_ids=frozenset()
+ )
+
+ return resolve
+
+
+def _agent_key(models: list[str]) -> UserAPIKeyAuth:
+ return UserAPIKeyAuth(api_key="sk-agent-key", user_id="user-1", agent_id="agent-1", models=models)
+
+
+@pytest.mark.asyncio
+async def test_agent_key_listing_is_capped_to_its_access_groups():
+ result = await get_available_models_for_user(
+ user_api_key_dict=_agent_key(["model-a", "model-b", "model-c"]),
+ llm_router=_router_with_models(["model-a", "model-b", "model-c"]),
+ general_settings={},
+ user_model=None,
+ resolve_agent_ceiling=_agent_ceiling(frozenset({"model-b", "model-d"})),
+ )
+ assert result == ["model-b"]
+
+
+@pytest.mark.asyncio
+async def test_agent_key_listing_is_empty_when_its_groups_grant_no_model():
+ result = await get_available_models_for_user(
+ user_api_key_dict=_agent_key(["model-a"]),
+ llm_router=_router_with_models(["model-a"]),
+ general_settings={},
+ user_model=None,
+ resolve_agent_ceiling=_agent_ceiling(frozenset()),
+ )
+ assert result == []
+
+
+@pytest.mark.asyncio
+async def test_agent_ceiling_expands_a_model_access_group_name_for_listing():
+ router = _router_with_models(["model-a", "model-b"])
+ router.get_model_access_groups.return_value = {"fast-models": ["model-b"]}
+ result = await get_available_models_for_user(
+ user_api_key_dict=_agent_key(["model-a", "model-b"]),
+ llm_router=router,
+ general_settings={},
+ user_model=None,
+ resolve_agent_ceiling=_agent_ceiling(frozenset({"fast-models"})),
+ )
+ assert result == ["model-b"]
+
+
+@pytest.mark.asyncio
+async def test_listing_is_unchanged_without_an_agent_or_without_attached_groups():
+ router = _router_with_models(["model-a", "model-b"])
+ plain_key = await get_available_models_for_user(
+ user_api_key_dict=UserAPIKeyAuth(api_key="sk-plain", user_id="user-1", models=["model-a", "model-b"]),
+ llm_router=router,
+ general_settings={},
+ user_model=None,
+ resolve_agent_ceiling=_agent_ceiling(frozenset({"model-a"})),
+ )
+ agent_without_groups = await get_available_models_for_user(
+ user_api_key_dict=_agent_key(["model-a", "model-b"]),
+ llm_router=router,
+ general_settings={},
+ user_model=None,
+ resolve_agent_ceiling=_agent_ceiling(None),
+ )
+ assert (plain_key, agent_without_groups) == (["model-a", "model-b"], ["model-a", "model-b"])
From 89330cdac6e3b103421114e1385efe719e417534 Mon Sep 17 00:00:00 2001
From: yassin
Date: Thu, 17 Sep 2026 20:34:06 +0000
Subject: [PATCH 010/160] refactor(agents): exhaust the agent access match and
drop routine comments
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
---
.../mcp_server/auth/user_api_key_auth_mcp.py | 3 ---
.../auth/agent_permission_handler.py | 14 ++++----------
litellm/proxy/auth/auth_checks.py | 5 +----
3 files changed, 5 insertions(+), 17 deletions(-)
diff --git a/litellm/proxy/_experimental/mcp_server/auth/user_api_key_auth_mcp.py b/litellm/proxy/_experimental/mcp_server/auth/user_api_key_auth_mcp.py
index 05661584a6b..4bca15190cf 100644
--- a/litellm/proxy/_experimental/mcp_server/auth/user_api_key_auth_mcp.py
+++ b/litellm/proxy/_experimental/mcp_server/auth/user_api_key_auth_mcp.py
@@ -193,9 +193,6 @@ def _agent_capped_servers(
agent_servers: Sequence[str],
agent_access_group_servers: frozenset[str] | None,
) -> tuple[str, ...] | None:
- """Servers left once the agent's object_permission and attached access groups both cap the
- key/team result, or None when the agent restricts nothing. An attached group set naming no
- server is an empty ceiling, not an absent one, so it denies every server."""
if not agent_servers and agent_access_group_servers is None:
return None
return tuple(
diff --git a/litellm/proxy/agent_endpoints/auth/agent_permission_handler.py b/litellm/proxy/agent_endpoints/auth/agent_permission_handler.py
index 1759090a29a..ea73d4634e3 100644
--- a/litellm/proxy/agent_endpoints/auth/agent_permission_handler.py
+++ b/litellm/proxy/agent_endpoints/auth/agent_permission_handler.py
@@ -8,7 +8,7 @@ Follows the same pattern as MCP permission handling.
import asyncio
from collections.abc import Awaitable, Callable, Sequence
from dataclasses import dataclass
-from typing import Final, TypeAlias
+from typing import Final, TypeAlias, assert_never
from litellm._logging import verbose_logger
from litellm.proxy._experimental.mcp_server.ui_session_utils import build_effective_auth_contexts
@@ -67,14 +67,7 @@ class AgentRequestHandler:
user_api_key_auth: UserAPIKeyAuth | None = None,
resolve_ceiling: CeilingResolver = resolve_agent_access_group_ceiling,
) -> AgentAccess:
- """
- Resolve the agents the given user/key may reach.
-
- ``UnrestrictedAgentAccess`` is only returned when neither the key nor its team
- carries any grant and the agent behind the key has no access groups attached.
- Grants that intersect to nothing stay restricted, so narrowing a caller can
- never widen what it reaches.
- """
+ """Agents the key may reach: key and team grants intersected with the agent's access group ceiling."""
key_team_access: Final = await AgentRequestHandler._resolve_key_team_agent_access(user_api_key_auth)
agent_ceiling: Final = await AgentRequestHandler._agent_access_group_ceiling(user_api_key_auth, resolve_ceiling)
if agent_ceiling is None:
@@ -84,6 +77,8 @@ class AgentRequestHandler:
return RestrictedAgentAccess(agent_ceiling)
case RestrictedAgentAccess(key_team_ids):
return RestrictedAgentAccess(key_team_ids & agent_ceiling)
+ case _:
+ assert_never(key_team_access)
@staticmethod
async def _resolve_key_team_agent_access(
@@ -111,7 +106,6 @@ class AgentRequestHandler:
user_api_key_auth: UserAPIKeyAuth | None,
resolve_ceiling: CeilingResolver,
) -> frozenset[str] | None:
- """Stable IDs of the agents the calling agent's attached access groups allow; None when none attached."""
if user_api_key_auth is None or not user_api_key_auth.agent_id:
return None
ceiling: Final = await resolve_ceiling(user_api_key_auth.agent_id)
diff --git a/litellm/proxy/auth/auth_checks.py b/litellm/proxy/auth/auth_checks.py
index 137cf849389..df20358bcce 100644
--- a/litellm/proxy/auth/auth_checks.py
+++ b/litellm/proxy/auth/auth_checks.py
@@ -1007,7 +1007,6 @@ async def common_checks(
code=status.HTTP_400_BAD_REQUEST,
)
- # 2.4 If the agent behind the key has access groups attached, they cap the models it can call
await _check_agent_access_group_model_access(model=_model, valid_token=valid_token, llm_router=llm_router)
## 2.1 If user can call model (if personal key)
@@ -4205,9 +4204,7 @@ async def _check_agent_access_group_model_access(
llm_router: Router | None,
resolve_ceiling: CeilingResolver = resolve_agent_access_group_ceiling,
) -> Literal[True]:
- """Raises when the key's agent has access groups attached and none of them names the model.
- Attached groups that name no model deny every model; ``_can_object_call_model`` would read
- an empty allowlist as unrestricted."""
+ """Attached groups naming no model deny every model, unlike the empty allowlist ``_can_object_call_model`` allows."""
if not model or valid_token is None or not valid_token.agent_id:
return True
ceiling: Final = await resolve_ceiling(valid_token.agent_id)
From 18c31e6fc6cb4f7a5ad5bb658b7f5d6356664922 Mon Sep 17 00:00:00 2001
From: yassin
Date: Thu, 17 Sep 2026 20:40:06 +0000
Subject: [PATCH 011/160] fix(agents): import assert_never from
typing_extensions for Python 3.10
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
---
.../proxy/agent_endpoints/auth/agent_permission_handler.py | 4 +++-
1 file changed, 3 insertions(+), 1 deletion(-)
diff --git a/litellm/proxy/agent_endpoints/auth/agent_permission_handler.py b/litellm/proxy/agent_endpoints/auth/agent_permission_handler.py
index ea73d4634e3..4e022e48bb4 100644
--- a/litellm/proxy/agent_endpoints/auth/agent_permission_handler.py
+++ b/litellm/proxy/agent_endpoints/auth/agent_permission_handler.py
@@ -8,7 +8,9 @@ Follows the same pattern as MCP permission handling.
import asyncio
from collections.abc import Awaitable, Callable, Sequence
from dataclasses import dataclass
-from typing import Final, TypeAlias, assert_never
+from typing import Final, TypeAlias
+
+from typing_extensions import assert_never
from litellm._logging import verbose_logger
from litellm.proxy._experimental.mcp_server.ui_session_utils import build_effective_auth_contexts
From 1206fa802b851fb485cf494e234f6a628f86813c Mon Sep 17 00:00:00 2001
From: yassin
Date: Thu, 17 Sep 2026 20:51:04 +0000
Subject: [PATCH 012/160] refactor(agents): resolve attached access groups from
the agent registry instead of the DB on the request path
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
---
.../auth/agent_access_groups.py | 81 +---------------
litellm/proxy/agent_endpoints/endpoints.py | 4 -
.../access_group_endpoints.py | 2 -
.../auth/test_agent_access_groups.py | 93 +++----------------
4 files changed, 14 insertions(+), 166 deletions(-)
diff --git a/litellm/proxy/agent_endpoints/auth/agent_access_groups.py b/litellm/proxy/agent_endpoints/auth/agent_access_groups.py
index e0e5193d0a4..49e5407ff88 100644
--- a/litellm/proxy/agent_endpoints/auth/agent_access_groups.py
+++ b/litellm/proxy/agent_endpoints/auth/agent_access_groups.py
@@ -1,35 +1,18 @@
import asyncio
-from collections.abc import Awaitable, Callable, Sequence
+from collections.abc import Awaitable, Callable
from dataclasses import dataclass
-from typing import Final, Protocol, TypeAlias
+from typing import Final, TypeAlias
from fastapi import HTTPException
-from pydantic import TypeAdapter, ValidationError
-from typing_extensions import ReadOnly, TypedDict
from litellm._logging import verbose_proxy_logger
-from litellm.caching.dual_cache import DualCache
from litellm.proxy._types import LiteLLM_AccessGroupTable
-from litellm.proxy.common_utils.user_api_key_cache import get_management_object_ttl
-
-
-class _AgentAccessGroupsRecord(Protocol):
- @property
- def access_group_ids(self) -> Sequence[str] | None: ...
-
-
-class _AgentIdWhere(TypedDict):
- agent_id: ReadOnly[str]
-
AccessGroupIds: TypeAlias = tuple[str, ...]
AccessGroupIdsLoader: TypeAlias = Callable[[str], Awaitable[AccessGroupIds]] # mutable-ok: Callable params
-AgentRecordFinder: TypeAlias = Callable[[str], Awaitable[_AgentAccessGroupsRecord | None]] # mutable-ok: Callable
LoadedAccessGroup: TypeAlias = LiteLLM_AccessGroupTable | None
AccessGroupLoader: TypeAlias = Callable[[str], Awaitable[LoadedAccessGroup]] # mutable-ok: Callable parameter syntax
-_CACHED_IDS: Final = TypeAdapter(list[str])
-
@dataclass(frozen=True, slots=True)
class AgentAccessGroupCeiling:
@@ -44,19 +27,6 @@ class AgentAccessGroupCeiling:
CeilingResolver: TypeAlias = Callable[[str], Awaitable[AgentAccessGroupCeiling | None]] # mutable-ok: Callable params
-def agent_access_group_ids_cache_key(agent_id: str) -> str:
- return f"agent_access_group_ids:{agent_id}"
-
-
-def _cached_access_group_ids(cached: object) -> AccessGroupIds | None:
- if cached is None:
- return None
- try:
- return tuple(_CACHED_IDS.validate_python(cached))
- except ValidationError:
- return None
-
-
async def _registry_access_group_ids(agent_id: str) -> AccessGroupIds:
from litellm.proxy.common_utils.registry_read_through import get_agent_with_read_through
@@ -64,51 +34,6 @@ async def _registry_access_group_ids(agent_id: str) -> AccessGroupIds:
return tuple(agent.access_group_ids or ()) if agent is not None else ()
-async def load_agent_access_group_ids(
- agent_id: str,
- cache: DualCache,
- find_agent: AgentRecordFinder,
- fallback: AccessGroupIdsLoader,
-) -> AccessGroupIds:
- """The agent row's groups, cached for the management-object TTL and evicted on every agent write."""
- cache_key: Final = agent_access_group_ids_cache_key(agent_id)
- cached: Final = _cached_access_group_ids(await cache.async_get_cache(key=cache_key))
- if cached is not None:
- return cached
- try:
- record: Final = await find_agent(agent_id)
- except Exception as e: # noqa: BLE001 # prisma raises many error types; the registry snapshot answers instead
- verbose_proxy_logger.warning("Failed to read access groups for agent %r, using registry: %s", agent_id, e)
- return await fallback(agent_id)
- access_group_ids: Final = tuple(record.access_group_ids or ()) if record is not None else ()
- await cache.async_set_cache(key=cache_key, value=access_group_ids, ttl=get_management_object_ttl(cache))
- return access_group_ids
-
-
-async def _load_agent_access_group_ids(agent_id: str) -> AccessGroupIds:
- from litellm.proxy.agent_endpoints.agent_registry import agents_table
- from litellm.proxy.proxy_server import prisma_client, user_api_key_cache
-
- if prisma_client is None:
- return await _registry_access_group_ids(agent_id)
- db: Final = prisma_client
-
- async def find_agent(row_agent_id: str) -> _AgentAccessGroupsRecord | None:
- return await agents_table(db).find_unique(where=_AgentIdWhere(agent_id=row_agent_id))
-
- return await load_agent_access_group_ids(agent_id, user_api_key_cache, find_agent, _registry_access_group_ids)
-
-
-async def evict_agent_access_group_ids(agent_ids: Sequence[str]) -> None:
- from litellm.proxy.common_utils.auth_cache_invalidation_pubsub import evict_and_broadcast
- from litellm.proxy.proxy_server import user_api_key_cache
-
- await evict_and_broadcast(
- cache_keys=tuple(agent_access_group_ids_cache_key(agent_id) for agent_id in agent_ids),
- user_api_key_cache=user_api_key_cache,
- )
-
-
async def _load_access_group(access_group_id: str) -> LoadedAccessGroup:
from litellm.proxy.auth.auth_checks import get_access_object
from litellm.proxy.proxy_server import prisma_client, proxy_logging_obj, user_api_key_cache
@@ -132,7 +57,7 @@ async def _load_access_group(access_group_id: str) -> LoadedAccessGroup:
async def resolve_agent_access_group_ceiling(
agent_id: str,
- load_access_group_ids: AccessGroupIdsLoader = _load_agent_access_group_ids,
+ load_access_group_ids: AccessGroupIdsLoader = _registry_access_group_ids,
load_access_group: AccessGroupLoader = _load_access_group,
) -> AgentAccessGroupCeiling | None:
"""``None`` when the agent has no access groups attached, so nothing is capped."""
diff --git a/litellm/proxy/agent_endpoints/endpoints.py b/litellm/proxy/agent_endpoints/endpoints.py
index 62783fb412a..aa8979a73c6 100644
--- a/litellm/proxy/agent_endpoints/endpoints.py
+++ b/litellm/proxy/agent_endpoints/endpoints.py
@@ -44,7 +44,6 @@ from litellm.proxy.agent_endpoints.agent_search import (
global_agent_search_index,
search_agents,
)
-from litellm.proxy.agent_endpoints.auth.agent_access_groups import evict_agent_access_group_ids
from litellm.proxy.agent_endpoints.auth.agent_permission_handler import accessible_agents
from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
from litellm.proxy.common_utils.rbac_utils import check_feature_access_for_user
@@ -697,7 +696,6 @@ async def update_agent(
prisma_client=prisma_client,
updated_by=updated_by,
)
- await evict_agent_access_group_ids((agent_id,))
# deregister in memory
AGENT_REGISTRY.deregister_agent(agent_name=existing_agent.get("agent_name"))
@@ -801,7 +799,6 @@ async def patch_agent(
prisma_client=prisma_client,
updated_by=updated_by,
)
- await evict_agent_access_group_ids((agent_id,))
# deregister in memory
AGENT_REGISTRY.deregister_agent(agent_name=existing_agent.get("agent_name"))
@@ -864,7 +861,6 @@ async def delete_agent(
raise HTTPException(status_code=404, detail=f"Agent with ID {agent_id} not found in DB.")
await AGENT_REGISTRY.delete_agent_from_db(agent_id=agent_id, prisma_client=prisma_client)
- await evict_agent_access_group_ids((agent_id,))
AGENT_REGISTRY.deregister_agent(agent_name=existing_agent.get("agent_name"))
diff --git a/litellm/proxy/management_endpoints/access_group_endpoints.py b/litellm/proxy/management_endpoints/access_group_endpoints.py
index 2694d00b17f..b4923b0a2dc 100644
--- a/litellm/proxy/management_endpoints/access_group_endpoints.py
+++ b/litellm/proxy/management_endpoints/access_group_endpoints.py
@@ -16,7 +16,6 @@ from litellm.proxy._types import (
UserAPIKeyAuth,
)
from litellm.proxy.agent_endpoints.agent_registry import global_agent_registry
-from litellm.proxy.agent_endpoints.auth.agent_access_groups import evict_agent_access_group_ids
from litellm.proxy.auth.auth_checks import (
_cache_access_object,
_cache_key_object,
@@ -783,7 +782,6 @@ async def delete_access_group(
await invalidate_access_group_cache(access_group_id)
_detach_access_group_from_agent_registry(detached_agent_ids, access_group_id)
- await evict_agent_access_group_ids(detached_agent_ids)
await _patch_team_caches_remove_access_group(
affected_team_ids, access_group_id, user_api_key_cache, proxy_logging_obj
)
diff --git a/tests/test_litellm/proxy/agent_endpoints/auth/test_agent_access_groups.py b/tests/test_litellm/proxy/agent_endpoints/auth/test_agent_access_groups.py
index 266ce52e3a9..e744e84d671 100644
--- a/tests/test_litellm/proxy/agent_endpoints/auth/test_agent_access_groups.py
+++ b/tests/test_litellm/proxy/agent_endpoints/auth/test_agent_access_groups.py
@@ -1,16 +1,11 @@
-from collections.abc import Sequence
-from dataclasses import dataclass
from typing import Final
import pytest
from fastapi import HTTPException
-from litellm.caching.dual_cache import DualCache
from litellm.models.access_group import LiteLLM_AccessGroupTable
from litellm.proxy.agent_endpoints.auth.agent_access_groups import (
AgentAccessGroupCeiling,
- agent_access_group_ids_cache_key,
- load_agent_access_group_ids,
resolve_agent_access_group_ceiling,
)
from litellm.types.agents import AgentResponse
@@ -110,86 +105,20 @@ async def test_only_unloadable_groups_is_an_empty_ceiling_not_unrestricted():
assert ceiling.agent_ids == frozenset()
-@dataclass(frozen=True, slots=True)
-class _AgentRow:
- access_group_ids: Sequence[str] | None
-
-
-class _FakeAgentTable:
- def __init__(self, rows: dict[str, _AgentRow], failing: bool = False) -> None:
- self._rows: Final = rows
- self._failing: Final = failing
- self.reads = 0
-
- async def find_agent(self, agent_id: str) -> _AgentRow | None:
- self.reads += 1
- if self._failing:
- raise RuntimeError("db down")
- return self._rows.get(agent_id)
-
-
-async def _registry_snapshot(agent_id: str) -> tuple[str, ...]:
- return ("registry-group",)
-
-
@pytest.mark.asyncio
-async def test_agent_row_is_read_once_then_served_from_cache():
- cache: Final = DualCache()
- table: Final = _FakeAgentTable({"agent-1": _AgentRow(["g1", "g2"])})
+async def test_default_agent_loader_reads_the_attached_groups_from_the_registry():
+ from litellm.proxy.agent_endpoints.agent_registry import global_agent_registry
- first: Final = await load_agent_access_group_ids("agent-1", cache, table.find_agent, _registry_snapshot)
- second: Final = await load_agent_access_group_ids("agent-1", cache, table.find_agent, _registry_snapshot)
+ _, load_group = _loaders(None, {"g1": _group("g1", models=("gpt-5",))})
+ global_agent_registry.register_agent(_agent(["g1"]))
+ try:
+ ceiling: Final = await resolve_agent_access_group_ceiling("agent-1", load_access_group=load_group)
+ finally:
+ global_agent_registry.deregister_agent("agent")
- assert (first, second, table.reads) == (("g1", "g2"), ("g1", "g2"), 1)
-
-
-@pytest.mark.asyncio
-async def test_agent_with_no_row_or_no_groups_caches_an_empty_answer():
- cache: Final = DualCache()
- table: Final = _FakeAgentTable({"bare": _AgentRow(None)})
-
- bare: Final = await load_agent_access_group_ids("bare", cache, table.find_agent, _registry_snapshot)
- missing: Final = await load_agent_access_group_ids("missing", cache, table.find_agent, _registry_snapshot)
- again: Final = await load_agent_access_group_ids("missing", cache, table.find_agent, _registry_snapshot)
-
- assert (bare, missing, again, table.reads) == ((), (), (), 2)
-
-
-@pytest.mark.asyncio
-async def test_evicted_cache_entry_picks_up_the_patched_row():
- cache: Final = DualCache()
- rows: Final = {"agent-1": _AgentRow(["g1"])}
- table: Final = _FakeAgentTable(rows)
- await load_agent_access_group_ids("agent-1", cache, table.find_agent, _registry_snapshot)
-
- rows["agent-1"] = _AgentRow(["g2"])
- stale: Final = await load_agent_access_group_ids("agent-1", cache, table.find_agent, _registry_snapshot)
- await cache.async_delete_cache(key=agent_access_group_ids_cache_key("agent-1"))
- fresh: Final = await load_agent_access_group_ids("agent-1", cache, table.find_agent, _registry_snapshot)
-
- assert (stale, fresh) == (("g1",), ("g2",))
-
-
-@pytest.mark.asyncio
-async def test_unreadable_row_falls_back_to_the_registry_without_caching():
- cache: Final = DualCache()
- table: Final = _FakeAgentTable({}, failing=True)
-
- answer: Final = await load_agent_access_group_ids("agent-1", cache, table.find_agent, _registry_snapshot)
-
- assert answer == ("registry-group",)
- assert await cache.async_get_cache(key=agent_access_group_ids_cache_key("agent-1")) is None
-
-
-@pytest.mark.asyncio
-async def test_garbage_in_the_cache_is_treated_as_a_miss():
- cache: Final = DualCache()
- await cache.async_set_cache(key=agent_access_group_ids_cache_key("agent-1"), value={"not": "a list"})
- table: Final = _FakeAgentTable({"agent-1": _AgentRow(["g1"])})
-
- answer: Final = await load_agent_access_group_ids("agent-1", cache, table.find_agent, _registry_snapshot)
-
- assert (answer, table.reads) == (("g1",), 1)
+ assert ceiling == AgentAccessGroupCeiling(
+ access_group_ids=("g1",), models=frozenset({"gpt-5"}), mcp_server_ids=frozenset(), agent_ids=frozenset()
+ )
@pytest.mark.asyncio
From 7229fc1952a83387e4126949114e9d253d5caec3 Mon Sep 17 00:00:00 2001
From: yassin
Date: Thu, 17 Sep 2026 21:00:56 +0000
Subject: [PATCH 013/160] fix(agents): return on every branch of the agent
access ceiling so CodeQL sees no fall-through
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
---
.../agent_endpoints/auth/agent_permission_handler.py | 12 +++---------
1 file changed, 3 insertions(+), 9 deletions(-)
diff --git a/litellm/proxy/agent_endpoints/auth/agent_permission_handler.py b/litellm/proxy/agent_endpoints/auth/agent_permission_handler.py
index 4e022e48bb4..fe0a1b13b43 100644
--- a/litellm/proxy/agent_endpoints/auth/agent_permission_handler.py
+++ b/litellm/proxy/agent_endpoints/auth/agent_permission_handler.py
@@ -10,8 +10,6 @@ from collections.abc import Awaitable, Callable, Sequence
from dataclasses import dataclass
from typing import Final, TypeAlias
-from typing_extensions import assert_never
-
from litellm._logging import verbose_logger
from litellm.proxy._experimental.mcp_server.ui_session_utils import build_effective_auth_contexts
from litellm.proxy._types import (
@@ -74,13 +72,9 @@ class AgentRequestHandler:
agent_ceiling: Final = await AgentRequestHandler._agent_access_group_ceiling(user_api_key_auth, resolve_ceiling)
if agent_ceiling is None:
return key_team_access
- match key_team_access:
- case UnrestrictedAgentAccess():
- return RestrictedAgentAccess(agent_ceiling)
- case RestrictedAgentAccess(key_team_ids):
- return RestrictedAgentAccess(key_team_ids & agent_ceiling)
- case _:
- assert_never(key_team_access)
+ if isinstance(key_team_access, UnrestrictedAgentAccess):
+ return RestrictedAgentAccess(agent_ceiling)
+ return RestrictedAgentAccess(key_team_access.agent_ids & agent_ceiling)
@staticmethod
async def _resolve_key_team_agent_access(
From 25729c521fc944b44f0d518f4b6bc45d87e5c68c Mon Sep 17 00:00:00 2001
From: yassin
Date: Thu, 17 Sep 2026 21:22:39 +0000
Subject: [PATCH 014/160] refactor(agents): combine key and team agent grants
without a fall-through match
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
---
.../auth/agent_permission_handler.py | 27 ++++++++++++-------
1 file changed, 17 insertions(+), 10 deletions(-)
diff --git a/litellm/proxy/agent_endpoints/auth/agent_permission_handler.py b/litellm/proxy/agent_endpoints/auth/agent_permission_handler.py
index fe0a1b13b43..b7b7638e478 100644
--- a/litellm/proxy/agent_endpoints/auth/agent_permission_handler.py
+++ b/litellm/proxy/agent_endpoints/auth/agent_permission_handler.py
@@ -48,6 +48,22 @@ def _to_stable_ids(agent_ids: frozenset[str]) -> frozenset[str]:
return frozenset(global_agent_registry.stable_agent_id(agent_id) for agent_id in agent_ids)
+def _restricted_ids(access: AgentAccess) -> frozenset[str] | None:
+ if isinstance(access, UnrestrictedAgentAccess):
+ return None
+ return _to_stable_ids(access.agent_ids)
+
+
+def _intersect_agent_access(key_access: AgentAccess, team_access: AgentAccess) -> AgentAccess:
+ key_ids: Final = _restricted_ids(key_access)
+ team_ids: Final = _restricted_ids(team_access)
+ if key_ids is None:
+ return UnrestrictedAgentAccess() if team_ids is None else RestrictedAgentAccess(team_ids)
+ if team_ids is None:
+ return RestrictedAgentAccess(key_ids)
+ return RestrictedAgentAccess(key_ids & team_ids)
+
+
class AgentRequestHandler:
"""
Class to handle agent permission checking, including:
@@ -83,19 +99,10 @@ class AgentRequestHandler:
try:
key_access: Final = await AgentRequestHandler._get_allowed_agents_for_key(user_api_key_auth)
team_access: Final = await AgentRequestHandler._get_allowed_agents_for_team(user_api_key_auth)
-
- match (key_access, team_access):
- case (UnrestrictedAgentAccess(), UnrestrictedAgentAccess()):
- return UnrestrictedAgentAccess()
- case (UnrestrictedAgentAccess(), RestrictedAgentAccess(team_ids)):
- return RestrictedAgentAccess(_to_stable_ids(team_ids))
- case (RestrictedAgentAccess(key_ids), UnrestrictedAgentAccess()):
- return RestrictedAgentAccess(_to_stable_ids(key_ids))
- case (RestrictedAgentAccess(key_ids), RestrictedAgentAccess(team_ids)):
- return RestrictedAgentAccess(_to_stable_ids(key_ids) & _to_stable_ids(team_ids))
except Exception as e:
verbose_logger.warning("Failed to get allowed agents: %s", e)
return UnrestrictedAgentAccess()
+ return _intersect_agent_access(key_access, team_access)
@staticmethod
async def _agent_access_group_ceiling(
From db17841b3d6ff41900276de6eac8c66fbc924801 Mon Sep 17 00:00:00 2001
From: Yuneng Jiang
Date: Thu, 17 Sep 2026 22:37:16 -0700
Subject: [PATCH 015/160] test(model_management): cover actor edges and
wildcard models
---
.../test_model_management_endpoints.py | 340 +++++++++++++++++-
.../handle_add_model_submit.test.tsx | 19 +
2 files changed, 356 insertions(+), 3 deletions(-)
diff --git a/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py b/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py
index d1fe88df26c..302585e42c4 100644
--- a/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py
+++ b/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py
@@ -1224,11 +1224,11 @@ class TestUpdateModel:
"litellm.proxy.management_endpoints.model_management_endpoints.ModelManagementAuthChecks.can_user_make_model_call",
new=AsyncMock(return_value=None),
),
- patch(
+ patch( # test-quality-ok: [TQ008] isolate persistence from encryption implementation
"litellm.proxy.management_endpoints.model_management_endpoints.encrypt_value_helper",
side_effect=lambda value: value,
),
- patch(
+ patch( # test-quality-ok: [TQ008] isolate persistence from router reload implementation
"litellm.proxy.management_endpoints.model_management_endpoints.clear_cache",
new=AsyncMock(
return_value=ReconcileOutcome(still_desired=None, live_after=None)
@@ -4021,7 +4021,7 @@ class TestPatchModelBlockedAuthGate:
"litellm.proxy.management_endpoints.model_management_endpoints.ModelManagementAuthChecks.can_user_make_model_call",
new=AsyncMock(return_value=None),
),
- patch(
+ patch( # test-quality-ok: [TQ008] isolate persistence from router reload implementation
"litellm.proxy.management_endpoints.model_management_endpoints.clear_cache",
new=AsyncMock(
return_value=ReconcileOutcome(still_desired=None, live_after=None)
@@ -6631,3 +6631,337 @@ class TestTeamMemberAutoRouterWrites:
assert json.loads(written["model_info"])["member_auto_router"] is True
assert appended.await_args.kwargs["data"].models == ["new-personal-router"]
assert appended.await_args.kwargs["data"].team_id == "member-team"
+
+
+class TestModelManagementActorEdges:
+ @pytest.mark.asyncio
+ async def test_add_model_rejects_non_team_internal_user(self):
+ from litellm.proxy._types import ProxyException
+ from litellm.proxy.management_endpoints.model_management_endpoints import add_new_model
+
+ actor: Final = UserAPIKeyAuth(user_id="internal-user", user_role=LitellmUserRoles.INTERNAL_USER)
+ prisma: Final = MagicMock()
+ deployment: Final = Deployment(
+ model_name="internal-model",
+ litellm_params=LiteLLM_Params(model="openai/test-model"),
+ model_info=ModelInfo(id="internal-model-id"),
+ )
+ with (
+ patch("litellm.proxy.proxy_server.prisma_client", prisma), # test-quality-ok: [TQ008] endpoint reads proxy-server state through its only test seam
+ patch("litellm.proxy.proxy_server.store_model_in_db", True), # test-quality-ok: [TQ008] endpoint reads proxy-server state through its only test seam
+ patch("litellm.proxy.proxy_server.premium_user", True), # test-quality-ok: [TQ008] endpoint reads proxy-server state through its only test seam
+ patch("litellm.proxy.proxy_server.general_settings", {}), # test-quality-ok: [TQ008] endpoint reads proxy-server state through its only test seam
+ ):
+ with pytest.raises(ProxyException) as exc_info:
+ await add_new_model(model_params=deployment, user_api_key_dict=actor)
+
+ assert str(exc_info.value.code) == "403"
+ assert "permission" in str(exc_info.value).lower()
+ prisma.db.litellm_proxymodeltable.create.assert_not_called()
+
+ @pytest.mark.asyncio
+ async def test_add_model_rejects_proxy_admin_viewer(self):
+ from litellm.proxy._types import ProxyException
+ from litellm.proxy.management_endpoints.model_management_endpoints import add_new_model
+
+ actor: Final = UserAPIKeyAuth(
+ user_id="view-only-user", user_role=LitellmUserRoles.PROXY_ADMIN_VIEW_ONLY
+ )
+ prisma: Final = MagicMock()
+ deployment: Final = Deployment(
+ model_name="view-only-model",
+ litellm_params=LiteLLM_Params(model="openai/test-model"),
+ model_info=ModelInfo(id="view-only-model-id"),
+ )
+ with (
+ patch("litellm.proxy.proxy_server.prisma_client", prisma), # test-quality-ok: [TQ008] endpoint reads proxy-server state through its only test seam
+ patch("litellm.proxy.proxy_server.store_model_in_db", True), # test-quality-ok: [TQ008] endpoint reads proxy-server state through its only test seam
+ patch("litellm.proxy.proxy_server.premium_user", True), # test-quality-ok: [TQ008] endpoint reads proxy-server state through its only test seam
+ patch("litellm.proxy.proxy_server.general_settings", {}), # test-quality-ok: [TQ008] endpoint reads proxy-server state through its only test seam
+ ):
+ with pytest.raises(ProxyException) as exc_info:
+ await add_new_model(model_params=deployment, user_api_key_dict=actor)
+
+ assert str(exc_info.value.code) == "403"
+ assert "view-only" in str(exc_info.value).lower()
+ prisma.db.litellm_proxymodeltable.create.assert_not_called()
+
+ @pytest.mark.asyncio
+ async def test_add_model_requires_database_storage(self):
+ from litellm.proxy._types import ProxyException
+ from litellm.proxy.management_endpoints.model_management_endpoints import add_new_model
+
+ actor: Final = UserAPIKeyAuth(user_id="admin", user_role=LitellmUserRoles.PROXY_ADMIN)
+ prisma: Final = MagicMock()
+ deployment: Final = Deployment(
+ model_name="database-disabled-model",
+ litellm_params=LiteLLM_Params(model="openai/test-model"),
+ model_info=ModelInfo(id="database-disabled-model-id"),
+ )
+ with (
+ patch("litellm.proxy.proxy_server.prisma_client", prisma), # test-quality-ok: [TQ008] endpoint reads proxy-server state through its only test seam
+ patch("litellm.proxy.proxy_server.store_model_in_db", False), # test-quality-ok: [TQ008] endpoint reads proxy-server state through its only test seam
+ patch("litellm.proxy.proxy_server.premium_user", True), # test-quality-ok: [TQ008] endpoint reads proxy-server state through its only test seam
+ patch("litellm.proxy.proxy_server.general_settings", {}), # test-quality-ok: [TQ008] endpoint reads proxy-server state through its only test seam
+ ):
+ with pytest.raises(ProxyException) as exc_info:
+ await add_new_model(model_params=deployment, user_api_key_dict=actor)
+
+ assert str(exc_info.value.code) == "500"
+ assert "STORE_MODEL_IN_DB" in str(exc_info.value)
+ prisma.db.litellm_proxymodeltable.create.assert_not_called()
+
+ @pytest.mark.asyncio
+ async def test_legacy_model_update_persists_changed_field(self):
+ from litellm.proxy.management_endpoints.model_management_endpoints import update_model
+
+ model_id: Final = "legacy-update-model-id"
+ existing_row: Final = MagicMock()
+ existing_row.litellm_params = {"model": "openai/test-model", "timeout": 30}
+ existing_row.model_dump.return_value = {
+ "model_name": "legacy-update-model",
+ "litellm_params": existing_row.litellm_params,
+ "model_info": {"id": model_id},
+ }
+ existing_row.model_dump_json.return_value = "{}"
+ updated_row: Final = MagicMock()
+ updated_row.model_dump_json.return_value = "{}"
+ prisma: Final = MagicMock()
+ prisma.db.litellm_proxymodeltable.find_unique = AsyncMock(return_value=existing_row)
+ prisma.db.litellm_proxymodeltable.update = AsyncMock(return_value=updated_row)
+ router: Final = MagicMock()
+ router.get_model_ids.return_value = [model_id]
+ actor: Final = UserAPIKeyAuth(user_id="admin", user_role=LitellmUserRoles.PROXY_ADMIN)
+ with (
+ patch("litellm.proxy.proxy_server.prisma_client", prisma), # test-quality-ok: [TQ008] update endpoint reads proxy-server state through its only test seam
+ patch("litellm.proxy.proxy_server.llm_router", router), # test-quality-ok: [TQ008] update endpoint reads proxy-server state through its only test seam
+ patch("litellm.proxy.proxy_server.store_model_in_db", True), # test-quality-ok: [TQ008] update endpoint reads proxy-server state through its only test seam
+ patch("litellm.proxy.proxy_server.premium_user", True), # test-quality-ok: [TQ008] update endpoint reads proxy-server state through its only test seam
+ patch( # test-quality-ok: [TQ008] isolate persistence from encryption implementation
+ "litellm.proxy.management_endpoints.model_management_endpoints.encrypt_value_helper",
+ side_effect=lambda value: value,
+ ),
+ patch( # test-quality-ok: [TQ008] isolate persistence from router reload implementation
+ "litellm.proxy.management_endpoints.model_management_endpoints.clear_cache",
+ new=AsyncMock(return_value=ReconcileOutcome(still_desired=None, live_after=None)),
+ ),
+ ):
+ await update_model(
+ model_params=updateDeployment(
+ litellm_params=updateLiteLLMParams(timeout=42),
+ model_info=ModelInfo(id=model_id),
+ ),
+ user_api_key_dict=actor,
+ )
+
+ written: Final = json.loads(
+ prisma.db.litellm_proxymodeltable.update.await_args.kwargs["data"]["litellm_params"]
+ )
+ assert written["timeout"] == 42
+ assert written["model"] == "openai/test-model"
+
+ @pytest.mark.asyncio
+ async def test_legacy_model_update_explicit_null_preserves_existing_field(self):
+ from litellm.proxy.management_endpoints.model_management_endpoints import update_model
+
+ model_id: Final = "legacy-null-model-id"
+ existing_row: Final = MagicMock()
+ existing_row.litellm_params = {"model": "openai/test-model", "timeout": 30}
+ existing_row.model_dump.return_value = {
+ "model_name": "legacy-null-model",
+ "litellm_params": existing_row.litellm_params,
+ "model_info": {"id": model_id},
+ }
+ existing_row.model_dump_json.return_value = "{}"
+ updated_row: Final = MagicMock()
+ updated_row.model_dump_json.return_value = "{}"
+ prisma: Final = MagicMock()
+ prisma.db.litellm_proxymodeltable.find_unique = AsyncMock(return_value=existing_row)
+ prisma.db.litellm_proxymodeltable.update = AsyncMock(return_value=updated_row)
+ router: Final = MagicMock()
+ router.get_model_ids.return_value = [model_id]
+ actor: Final = UserAPIKeyAuth(user_id="admin", user_role=LitellmUserRoles.PROXY_ADMIN)
+ with (
+ patch("litellm.proxy.proxy_server.prisma_client", prisma), # test-quality-ok: [TQ008] update endpoint reads proxy-server state through its only test seam
+ patch("litellm.proxy.proxy_server.llm_router", router), # test-quality-ok: [TQ008] update endpoint reads proxy-server state through its only test seam
+ patch("litellm.proxy.proxy_server.store_model_in_db", True), # test-quality-ok: [TQ008] update endpoint reads proxy-server state through its only test seam
+ patch("litellm.proxy.proxy_server.premium_user", True), # test-quality-ok: [TQ008] update endpoint reads proxy-server state through its only test seam
+ patch( # test-quality-ok: [TQ008] isolate persistence from encryption implementation
+ "litellm.proxy.management_endpoints.model_management_endpoints.encrypt_value_helper",
+ side_effect=lambda value: value,
+ ),
+ patch( # test-quality-ok: [TQ008] isolate persistence from router reload implementation
+ "litellm.proxy.management_endpoints.model_management_endpoints.clear_cache",
+ new=AsyncMock(return_value=ReconcileOutcome(still_desired=None, live_after=None)),
+ ),
+ ):
+ await update_model(
+ model_params=updateDeployment(
+ litellm_params=updateLiteLLMParams(timeout=None),
+ model_info=ModelInfo(id=model_id),
+ ),
+ user_api_key_dict=actor,
+ )
+
+ written: Final = json.loads(
+ prisma.db.litellm_proxymodeltable.update.await_args.kwargs["data"]["litellm_params"]
+ )
+ assert written["timeout"] == 30
+
+ @pytest.mark.asyncio
+ async def test_patch_model_rejects_config_file_model(self):
+ from litellm.proxy._types import ProxyException
+ from litellm.proxy.management_endpoints.model_management_endpoints import patch_model
+
+ model_id: Final = "config-model-id"
+ prisma: Final = MagicMock()
+ prisma.db.litellm_proxymodeltable.find_unique = AsyncMock(return_value=None)
+ prisma.db.litellm_proxymodeltable.update = AsyncMock()
+ router: Final = MagicMock()
+ router.get_deployment.return_value = Deployment(
+ model_name="config-model",
+ litellm_params=LiteLLM_Params(model="openai/test-model"),
+ model_info=ModelInfo(id=model_id),
+ )
+ actor: Final = UserAPIKeyAuth(user_id="admin", user_role=LitellmUserRoles.PROXY_ADMIN)
+ with (
+ patch("litellm.proxy.proxy_server.prisma_client", prisma), # test-quality-ok: [TQ008] patch endpoint reads proxy-server state through its only test seam
+ patch("litellm.proxy.proxy_server.llm_router", router), # test-quality-ok: [TQ008] patch endpoint reads proxy-server state through its only test seam
+ patch("litellm.proxy.proxy_server.store_model_in_db", True), # test-quality-ok: [TQ008] patch endpoint reads proxy-server state through its only test seam
+ patch("litellm.proxy.proxy_server.premium_user", True), # test-quality-ok: [TQ008] patch endpoint reads proxy-server state through its only test seam
+ ):
+ with pytest.raises(ProxyException) as exc_info:
+ await patch_model(
+ model_id=model_id,
+ patch_data=updateDeployment(
+ litellm_params=updateLiteLLMParams(timeout=42),
+ model_info=ModelInfo(id=model_id),
+ ),
+ user_api_key_dict=actor,
+ )
+
+ assert str(exc_info.value.code) == "400"
+ assert "Cannot edit config-based model" in str(exc_info.value)
+ prisma.db.litellm_proxymodeltable.update.assert_not_awaited()
+
+ @contextlib.contextmanager
+ def _client_for(self, actor: UserAPIKeyAuth) -> Iterator[TestClient]:
+ import litellm.proxy.proxy_server as proxy_server
+ from litellm.proxy.proxy_server import app
+
+ app.dependency_overrides[proxy_server.user_api_key_auth] = lambda: actor
+ try:
+ yield TestClient(app)
+ finally:
+ app.dependency_overrides.pop(proxy_server.user_api_key_auth, None)
+
+ def test_post_model_new_binds_to_actor_guard(self):
+ actor: Final = UserAPIKeyAuth(user_id="internal-user", user_role=LitellmUserRoles.INTERNAL_USER)
+ prisma: Final = MagicMock()
+ with (
+ patch("litellm.proxy.proxy_server.prisma_client", prisma), # test-quality-ok: [TQ008] route reads proxy-server state through its only test seam
+ patch("litellm.proxy.proxy_server.store_model_in_db", True), # test-quality-ok: [TQ008] route reads proxy-server state through its only test seam
+ patch("litellm.proxy.proxy_server.premium_user", True), # test-quality-ok: [TQ008] route reads proxy-server state through its only test seam
+ patch("litellm.proxy.proxy_server.general_settings", {}), # test-quality-ok: [TQ008] route reads proxy-server state through its only test seam
+ self._client_for(actor) as client,
+ ):
+ response: Final = client.post(
+ "/model/new",
+ json={
+ "model_name": "internal-model",
+ "litellm_params": {"model": "openai/test-model"},
+ "model_info": {"id": "internal-model-id"},
+ },
+ )
+
+ assert response.status_code == 403
+ assert "permission" in response.text.lower()
+ prisma.db.litellm_proxymodeltable.create.assert_not_called()
+
+ def test_post_legacy_model_update_binds_to_persistence(self):
+ model_id: Final = "legacy-route-model-id"
+ existing_row: Final = LiteLLM_ProxyModelTable(
+ model_id=model_id,
+ model_name="legacy-route-model",
+ litellm_params={"model": "openai/test-model", "timeout": 30},
+ model_info={"id": model_id},
+ created_by="admin",
+ updated_by="admin",
+ )
+ updated_row: Final = LiteLLM_ProxyModelTable(
+ model_id=model_id,
+ model_name="legacy-route-model",
+ litellm_params={"model": "openai/test-model", "timeout": 42},
+ model_info={"id": model_id},
+ created_by="admin",
+ updated_by="admin",
+ )
+ prisma: Final = MagicMock()
+ prisma.db.litellm_proxymodeltable.find_unique = AsyncMock(return_value=existing_row)
+ prisma.db.litellm_proxymodeltable.update = AsyncMock(return_value=updated_row)
+ router: Final = MagicMock()
+ router.get_model_ids.return_value = [model_id]
+ actor: Final = UserAPIKeyAuth(user_id="admin", user_role=LitellmUserRoles.PROXY_ADMIN)
+ with (
+ patch("litellm.proxy.proxy_server.prisma_client", prisma), # test-quality-ok: [TQ008] route reads proxy-server state through its only test seam
+ patch("litellm.proxy.proxy_server.llm_router", router), # test-quality-ok: [TQ008] route reads proxy-server state through its only test seam
+ patch("litellm.proxy.proxy_server.store_model_in_db", True), # test-quality-ok: [TQ008] route reads proxy-server state through its only test seam
+ patch("litellm.proxy.proxy_server.premium_user", True), # test-quality-ok: [TQ008] route reads proxy-server state through its only test seam
+ patch( # test-quality-ok: [TQ008] isolate persistence from encryption implementation
+ "litellm.proxy.management_endpoints.model_management_endpoints.encrypt_value_helper",
+ side_effect=lambda value: value,
+ ),
+ patch( # test-quality-ok: [TQ008] isolate persistence from router reload implementation
+ "litellm.proxy.management_endpoints.model_management_endpoints.clear_cache",
+ new=AsyncMock(return_value=ReconcileOutcome(still_desired=None, live_after=None)),
+ ),
+ patch( # test-quality-ok: [TQ008] audit logging is outside the persistence contract
+ "litellm.proxy.management_endpoints.model_management_endpoints.create_object_audit_log",
+ new=AsyncMock(return_value=None),
+ ),
+ self._client_for(actor) as client,
+ ):
+ response: Final = client.post(
+ "/model/update",
+ json={
+ "litellm_params": {"timeout": 42},
+ "model_info": {"id": model_id},
+ },
+ )
+
+ assert response.status_code == 200, response.text
+ written: Final = json.loads(
+ prisma.db.litellm_proxymodeltable.update.await_args.kwargs["data"]["litellm_params"]
+ )
+ assert written["timeout"] == 42
+
+ def test_patch_config_model_binds_to_patch_route(self):
+ model_id: Final = "config-route-model-id"
+ prisma: Final = MagicMock()
+ prisma.db.litellm_proxymodeltable.find_unique = AsyncMock(return_value=None)
+ prisma.db.litellm_proxymodeltable.update = AsyncMock()
+ router: Final = MagicMock()
+ router.get_deployment.return_value = Deployment(
+ model_name="config-route-model",
+ litellm_params=LiteLLM_Params(model="openai/test-model"),
+ model_info=ModelInfo(id=model_id),
+ )
+ actor: Final = UserAPIKeyAuth(user_id="admin", user_role=LitellmUserRoles.PROXY_ADMIN)
+ with (
+ patch("litellm.proxy.proxy_server.prisma_client", prisma), # test-quality-ok: [TQ008] route reads proxy-server state through its only test seam
+ patch("litellm.proxy.proxy_server.llm_router", router), # test-quality-ok: [TQ008] route reads proxy-server state through its only test seam
+ patch("litellm.proxy.proxy_server.store_model_in_db", True), # test-quality-ok: [TQ008] route reads proxy-server state through its only test seam
+ patch("litellm.proxy.proxy_server.premium_user", True), # test-quality-ok: [TQ008] route reads proxy-server state through its only test seam
+ self._client_for(actor) as client,
+ ):
+ response: Final = client.patch(
+ f"/model/{model_id}/update",
+ json={
+ "litellm_params": {"timeout": 42},
+ "model_info": {"id": model_id},
+ },
+ )
+
+ assert response.status_code == 400
+ assert "Cannot edit config-based model" in response.text
+ prisma.db.litellm_proxymodeltable.update.assert_not_awaited()
diff --git a/ui/litellm-dashboard/src/components/add_model/handle_add_model_submit.test.tsx b/ui/litellm-dashboard/src/components/add_model/handle_add_model_submit.test.tsx
index 9d792480c9f..923cf2aa0e3 100644
--- a/ui/litellm-dashboard/src/components/add_model/handle_add_model_submit.test.tsx
+++ b/ui/litellm-dashboard/src/components/add_model/handle_add_model_submit.test.tsx
@@ -101,4 +101,23 @@ describe("prepareModelAddRequest", () => {
expect(deployment.litellmParamsObj.litellm_credential_name).toBe("from-json");
expect(deployment.litellmParamsObj.timeout).toBe(5);
});
+
+ it.each([
+ ["OpenAI", "openai/*"],
+ ["Azure_AI_Studio", "azure_ai/*"],
+ ["Petals", "petals/*"],
+ ])("composes wildcard names for the all-model selection", async (custom_llm_provider, wildcardModel) => {
+ const formValues = {
+ model_mappings: [],
+ model: "all-wildcard",
+ custom_llm_provider,
+ };
+
+ const deployments = await prepareModelAddRequest({ ...formValues }, "token", null);
+
+ expect(deployments).toHaveLength(1);
+ const [deployment] = deployments!;
+ expect(deployment.modelName).toBe(wildcardModel);
+ expect(deployment.litellmParamsObj.model).toBe(wildcardModel);
+ });
});
From c34adb4ab2bd1b6579cc3eb9a3922cf9703aae58 Mon Sep 17 00:00:00 2001
From: Yuneng Jiang
Date: Thu, 17 Sep 2026 22:44:03 -0700
Subject: [PATCH 016/160] test(ui): cover narrowed dashboard form journeys
---
tests/e2e/ui/tests/prompts/addPrompt.spec.ts | 55 ++++++++++++++
.../tests/tagManagement/tagManagement.spec.ts | 76 +++++++++++++++++++
...PaginatedSearchSelect.integration.test.tsx | 45 +++++++++++
.../shared/SearchSelect.integration.test.tsx | 33 ++++++++
.../view_logs/RequestLogsFilters.test.tsx | 11 +++
5 files changed, 220 insertions(+)
create mode 100644 tests/e2e/ui/tests/prompts/addPrompt.spec.ts
create mode 100644 tests/e2e/ui/tests/tagManagement/tagManagement.spec.ts
diff --git a/tests/e2e/ui/tests/prompts/addPrompt.spec.ts b/tests/e2e/ui/tests/prompts/addPrompt.spec.ts
new file mode 100644
index 00000000000..cd87e4d3b56
--- /dev/null
+++ b/tests/e2e/ui/tests/prompts/addPrompt.spec.ts
@@ -0,0 +1,55 @@
+import { test, expect } from "@playwright/test";
+
+import { ADMIN_STORAGE_PATH } from "../../constants";
+import { Page as DashboardPage } from "../../fixtures/pages";
+import { navigateToPage } from "../../helpers/navigation";
+import { readBack } from "../../helpers/roundTrip";
+import { masterKey, uniqueSuffix } from "../../helpers/traffic";
+
+test.use({ storageState: ADMIN_STORAGE_PATH });
+
+test.describe("Prompt upload form", () => {
+ test("uploads a prompt file and reads the created prompt back", async ({
+ page,
+ }) => {
+ const promptId = `e2e-prompt-${uniqueSuffix()}`;
+ await navigateToPage(page, DashboardPage.Prompts);
+ await page.getByRole("button", { name: "Upload .prompt File" }).click();
+
+ try {
+ await expect(
+ page.getByRole("dialog", { name: "Add New Prompt" }),
+ ).toBeVisible();
+ await page.getByLabel("Prompt ID").fill(promptId);
+ await page.locator('input[type="file"]').setInputFiles({
+ name: "e2e.prompt",
+ mimeType: "text/plain",
+ buffer: Buffer.from(
+ 'model: fake-openai-gpt-4\ntemplate: "Hello {{name}}"\n',
+ ),
+ });
+ await expect(page.getByText("Selected: e2e.prompt")).toBeVisible();
+ await page.getByRole("button", { name: "Create Prompt" }).click();
+
+ await expect
+ .poll(async () => {
+ const response = await page.request.get(
+ `/prompts/${encodeURIComponent(promptId)}/info?environment=development`,
+ {
+ headers: { Authorization: `Bearer ${masterKey()}` },
+ },
+ );
+ return response.ok();
+ })
+ .toBe(true);
+ await expect(page.getByText(promptId, { exact: true })).toBeVisible();
+ } finally {
+ await page.request.delete(
+ `/prompts/${encodeURIComponent(promptId)}?environment=development`,
+ {
+ headers: { Authorization: `Bearer ${masterKey()}` },
+ },
+ );
+ }
+ });
+});
diff --git a/tests/e2e/ui/tests/tagManagement/tagManagement.spec.ts b/tests/e2e/ui/tests/tagManagement/tagManagement.spec.ts
new file mode 100644
index 00000000000..785211463dd
--- /dev/null
+++ b/tests/e2e/ui/tests/tagManagement/tagManagement.spec.ts
@@ -0,0 +1,76 @@
+import { test, expect } from "@playwright/test";
+
+import { ADMIN_STORAGE_PATH } from "../../constants";
+import { Page as DashboardPage } from "../../fixtures/pages";
+import { navigateToPage } from "../../helpers/navigation";
+import { captureRequestBody, readBack } from "../../helpers/roundTrip";
+import { masterKey, uniqueSuffix } from "../../helpers/traffic";
+
+test.use({ storageState: ADMIN_STORAGE_PATH });
+
+test.describe("Tag management", () => {
+ test("creates, edits, reopens, and reads back a tag", async ({ page }) => {
+ const tagName = `e2e-tag-${uniqueSuffix()}`;
+ const description = "synthetic tag description";
+ const updatedDescription = `${description} updated`;
+
+ await navigateToPage(page, DashboardPage.TagManagement);
+ await page.getByRole("button", { name: "+ Create New Tag" }).click();
+
+ try {
+ await expect(
+ page.getByRole("dialog", { name: "Create New Tag" }),
+ ).toBeVisible();
+ await page.getByLabel("Tag Name").fill(tagName);
+ await page.getByLabel("Description").fill(description);
+ await page.getByRole("button", { name: "Create Tag" }).click();
+
+ await expect
+ .poll(async () => {
+ const response = await readBack<
+ Record>
+ >(page, "/tag/list");
+ return Object.values(response).some((tag) => tag.name === tagName);
+ })
+ .toBe(true);
+ await expect(page.getByText(tagName, { exact: true })).toBeVisible();
+
+ await page.getByText(tagName, { exact: true }).click();
+ await expect(page.getByText("Tag Name:")).toBeVisible();
+ await page.getByRole("button", { name: "Edit Tag" }).click();
+ await page.getByLabel("Description").fill(updatedDescription);
+ const updateBody = await captureRequestBody(
+ page,
+ { method: "POST", urlIncludes: "/tag/update" },
+ () => page.getByRole("button", { name: "Save Changes" }).click(),
+ );
+ expect(updateBody).toMatchObject({
+ name: tagName,
+ description: updatedDescription,
+ });
+
+ await expect
+ .poll(async () => {
+ const infoResponse = await page.request.post("/tag/info", {
+ headers: { Authorization: `Bearer ${masterKey()}` },
+ data: { names: [tagName] },
+ });
+ expect(infoResponse.ok()).toBe(true);
+ const info = (await infoResponse.json()) as Record<
+ string,
+ { description?: string }
+ >;
+ return info[tagName]?.description;
+ })
+ .toBe(updatedDescription);
+ } finally {
+ await page.request.post("/tag/delete", {
+ headers: {
+ Authorization: `Bearer ${masterKey()}`,
+ "Content-Type": "application/json",
+ },
+ data: { name: tagName },
+ });
+ }
+ });
+});
diff --git a/ui/litellm-dashboard/src/components/shared/PaginatedSearchSelect.integration.test.tsx b/ui/litellm-dashboard/src/components/shared/PaginatedSearchSelect.integration.test.tsx
index 2b948ca8420..6d30f1513ef 100644
--- a/ui/litellm-dashboard/src/components/shared/PaginatedSearchSelect.integration.test.tsx
+++ b/ui/litellm-dashboard/src/components/shared/PaginatedSearchSelect.integration.test.tsx
@@ -1,5 +1,6 @@
import { fireEvent, renderWithProviders as render, screen, waitFor } from "../../../tests/test-utils";
import userEvent from "@testing-library/user-event";
+import { useQuery } from "@tanstack/react-query";
import { useState } from "react";
import { describe, expect, it, vi } from "vitest";
@@ -406,4 +407,48 @@ describe("PaginatedSearchSelect", () => {
expect(input).toHaveValue("aliasalpha");
await waitFor(() => expect(onSearchChange).toHaveBeenLastCalledWith("aliasalpha"));
});
+
+ it("keeps the latest query results when an earlier response resolves last", async () => {
+ const pending = new Map void>();
+
+ function QueryBackedSelect() {
+ const [query, setQuery] = useState("");
+ const result = useQuery({
+ queryKey: ["paginated-select-race", query],
+ queryFn: () =>
+ new Promise((resolve) => {
+ pending.set(query, resolve);
+ }),
+ enabled: query.length > 0,
+ });
+ return (
+ <>
+ setQuery("A")}>
+ Search A
+
+ setQuery("B")}>
+ Search B
+
+
+ >
+ );
+ }
+
+ const user = userEvent.setup();
+ render( );
+ await user.click(screen.getByRole("button", { name: "Search A" }));
+ await user.click(screen.getByRole("button", { name: "Search B" }));
+ await waitFor(() => {
+ expect(pending.has("A")).toBe(true);
+ expect(pending.has("B")).toBe(true);
+ });
+
+ pending.get("B")?.([{ label: "B result", value: "b" }]);
+ await user.click(screen.getByRole("combobox"));
+ expect(await screen.findByText("B result")).toBeInTheDocument();
+
+ pending.get("A")?.([{ label: "A result", value: "a" }]);
+ await waitFor(() => expect(screen.queryByText("A result")).not.toBeInTheDocument());
+ expect(screen.getByText("B result")).toBeInTheDocument();
+ });
});
diff --git a/ui/litellm-dashboard/src/components/shared/SearchSelect.integration.test.tsx b/ui/litellm-dashboard/src/components/shared/SearchSelect.integration.test.tsx
index c981010dff9..ed83acef14a 100644
--- a/ui/litellm-dashboard/src/components/shared/SearchSelect.integration.test.tsx
+++ b/ui/litellm-dashboard/src/components/shared/SearchSelect.integration.test.tsx
@@ -111,4 +111,37 @@ describe("SearchSelect", () => {
expect(screen.queryByText("Growth")).not.toBeInTheDocument();
expect(onValueChange).not.toHaveBeenCalled();
});
+
+ it("supports keyboard select, clear, escape, blur, and reopen", async () => {
+ const onValueChange = vi.fn();
+ const user = userEvent.setup();
+ function Controlled() {
+ const [value, setValue] = useState(null);
+ return (
+ {
+ setValue(next);
+ onValueChange(next);
+ }}
+ />
+ );
+ }
+
+ render( );
+ const input = screen.getByRole("combobox");
+ await user.tab();
+ await user.keyboard("{Enter}");
+ await user.keyboard("{ArrowDown}{Enter}");
+ expect(onValueChange).toHaveBeenLastCalledWith("team-1");
+ const clear = screen.getByRole("button", { name: "Clear" });
+ clear.focus();
+ await user.keyboard("{Enter}");
+ expect(onValueChange).toHaveBeenLastCalledWith(null);
+ await user.keyboard("{Escape}");
+ await user.tab();
+ await user.tab({ shift: true });
+ expect(input).toHaveFocus();
+ });
});
diff --git a/ui/litellm-dashboard/src/components/view_logs/RequestLogsFilters.test.tsx b/ui/litellm-dashboard/src/components/view_logs/RequestLogsFilters.test.tsx
index 5a35c7ae16b..8d1847e0121 100644
--- a/ui/litellm-dashboard/src/components/view_logs/RequestLogsFilters.test.tsx
+++ b/ui/litellm-dashboard/src/components/view_logs/RequestLogsFilters.test.tsx
@@ -344,4 +344,15 @@ describe("RequestLogsFilters", () => {
expect(set).toHaveBeenCalledWith(LOG_FILTER_IDS.CACHE_STATUS, undefined);
});
+
+ it("clears the raw Error Code combobox through the undefined filter contract", async () => {
+ const user = userEvent.setup();
+ const { set } = renderFilters({ [LOG_FILTER_IDS.ERROR_CODE]: "429" });
+ const input = await screen.findByPlaceholderText("Select or type an error code");
+
+ await user.click(input);
+ await user.click(screen.getByRole("button", { name: "Clear", hidden: true }));
+
+ expect(set).toHaveBeenCalledWith(LOG_FILTER_IDS.ERROR_CODE, undefined);
+ });
});
From 8d1ca16652056512999a61c87e96ae3aaf9c236d Mon Sep 17 00:00:00 2001
From: Yuneng Jiang
Date: Thu, 17 Sep 2026 23:06:17 -0700
Subject: [PATCH 017/160] test(ui): strengthen dashboard journey assertions
---
tests/e2e/ui/tests/prompts/addPrompt.spec.ts | 14 ++++++++---
.../tests/tagManagement/tagManagement.spec.ts | 3 ++-
...PaginatedSearchSelect.integration.test.tsx | 25 ++++++-------------
.../shared/SearchSelect.integration.test.tsx | 9 +++----
4 files changed, 24 insertions(+), 27 deletions(-)
diff --git a/tests/e2e/ui/tests/prompts/addPrompt.spec.ts b/tests/e2e/ui/tests/prompts/addPrompt.spec.ts
index cd87e4d3b56..f9868f7b04b 100644
--- a/tests/e2e/ui/tests/prompts/addPrompt.spec.ts
+++ b/tests/e2e/ui/tests/prompts/addPrompt.spec.ts
@@ -13,6 +13,7 @@ test.describe("Prompt upload form", () => {
page,
}) => {
const promptId = `e2e-prompt-${uniqueSuffix()}`;
+ const promptContent = "Hello {{name}}";
await navigateToPage(page, DashboardPage.Prompts);
await page.getByRole("button", { name: "Upload .prompt File" }).click();
@@ -25,7 +26,7 @@ test.describe("Prompt upload form", () => {
name: "e2e.prompt",
mimeType: "text/plain",
buffer: Buffer.from(
- 'model: fake-openai-gpt-4\ntemplate: "Hello {{name}}"\n',
+ `model: fake-openai-gpt-4\ntemplate: "${promptContent}"\n`,
),
});
await expect(page.getByText("Selected: e2e.prompt")).toBeVisible();
@@ -39,17 +40,22 @@ test.describe("Prompt upload form", () => {
headers: { Authorization: `Bearer ${masterKey()}` },
},
);
- return response.ok();
+ if (!response.ok()) return undefined;
+ const promptInfo = (await response.json()) as {
+ raw_prompt_template?: { content?: string };
+ };
+ return promptInfo.raw_prompt_template?.content;
})
- .toBe(true);
+ .toContain(promptContent);
await expect(page.getByText(promptId, { exact: true })).toBeVisible();
} finally {
- await page.request.delete(
+ const deleteResponse = await page.request.delete(
`/prompts/${encodeURIComponent(promptId)}?environment=development`,
{
headers: { Authorization: `Bearer ${masterKey()}` },
},
);
+ expect(deleteResponse.ok()).toBe(true);
}
});
});
diff --git a/tests/e2e/ui/tests/tagManagement/tagManagement.spec.ts b/tests/e2e/ui/tests/tagManagement/tagManagement.spec.ts
index 785211463dd..e1d5138ea90 100644
--- a/tests/e2e/ui/tests/tagManagement/tagManagement.spec.ts
+++ b/tests/e2e/ui/tests/tagManagement/tagManagement.spec.ts
@@ -64,13 +64,14 @@ test.describe("Tag management", () => {
})
.toBe(updatedDescription);
} finally {
- await page.request.post("/tag/delete", {
+ const deleteResponse = await page.request.post("/tag/delete", {
headers: {
Authorization: `Bearer ${masterKey()}`,
"Content-Type": "application/json",
},
data: { name: tagName },
});
+ expect(deleteResponse.ok()).toBe(true);
}
});
});
diff --git a/ui/litellm-dashboard/src/components/shared/PaginatedSearchSelect.integration.test.tsx b/ui/litellm-dashboard/src/components/shared/PaginatedSearchSelect.integration.test.tsx
index 6d30f1513ef..905610a77e6 100644
--- a/ui/litellm-dashboard/src/components/shared/PaginatedSearchSelect.integration.test.tsx
+++ b/ui/litellm-dashboard/src/components/shared/PaginatedSearchSelect.integration.test.tsx
@@ -421,27 +421,18 @@ describe("PaginatedSearchSelect", () => {
}),
enabled: query.length > 0,
});
- return (
- <>
- setQuery("A")}>
- Search A
-
- setQuery("B")}>
- Search B
-
-
- >
- );
+ return ;
}
const user = userEvent.setup();
render( );
- await user.click(screen.getByRole("button", { name: "Search A" }));
- await user.click(screen.getByRole("button", { name: "Search B" }));
- await waitFor(() => {
- expect(pending.has("A")).toBe(true);
- expect(pending.has("B")).toBe(true);
- });
+ const input = screen.getByRole("combobox");
+ await user.click(input);
+ await user.type(input, "A");
+ await waitFor(() => expect(pending.has("A")).toBe(true));
+ await user.clear(input);
+ await user.type(input, "B");
+ await waitFor(() => expect(pending.has("B")).toBe(true));
pending.get("B")?.([{ label: "B result", value: "b" }]);
await user.click(screen.getByRole("combobox"));
diff --git a/ui/litellm-dashboard/src/components/shared/SearchSelect.integration.test.tsx b/ui/litellm-dashboard/src/components/shared/SearchSelect.integration.test.tsx
index ed83acef14a..9f320049b09 100644
--- a/ui/litellm-dashboard/src/components/shared/SearchSelect.integration.test.tsx
+++ b/ui/litellm-dashboard/src/components/shared/SearchSelect.integration.test.tsx
@@ -112,7 +112,7 @@ describe("SearchSelect", () => {
expect(onValueChange).not.toHaveBeenCalled();
});
- it("supports keyboard select, clear, escape, blur, and reopen", async () => {
+ it("supports keyboard select, clear, and reselect", async () => {
const onValueChange = vi.fn();
const user = userEvent.setup();
function Controlled() {
@@ -139,9 +139,8 @@ describe("SearchSelect", () => {
clear.focus();
await user.keyboard("{Enter}");
expect(onValueChange).toHaveBeenLastCalledWith(null);
- await user.keyboard("{Escape}");
- await user.tab();
- await user.tab({ shift: true });
- expect(input).toHaveFocus();
+ input.focus();
+ await user.keyboard("{Enter}{ArrowDown}{Enter}");
+ expect(onValueChange).toHaveBeenLastCalledWith("team-1");
});
});
From b4c3adc37d1be33550803bce9c88bc190c8f4ec6 Mon Sep 17 00:00:00 2001
From: Yuneng Jiang
Date: Thu, 17 Sep 2026 23:12:33 -0700
Subject: [PATCH 018/160] test(ui): assert dashboard form cleanup
---
tests/e2e/ui/tests/prompts/addPrompt.spec.ts | 39 ++++++++++++-------
.../tests/tagManagement/tagManagement.spec.ts | 30 +++++++-------
2 files changed, 40 insertions(+), 29 deletions(-)
diff --git a/tests/e2e/ui/tests/prompts/addPrompt.spec.ts b/tests/e2e/ui/tests/prompts/addPrompt.spec.ts
index f9868f7b04b..a4a6cf62b7e 100644
--- a/tests/e2e/ui/tests/prompts/addPrompt.spec.ts
+++ b/tests/e2e/ui/tests/prompts/addPrompt.spec.ts
@@ -17,21 +17,32 @@ test.describe("Prompt upload form", () => {
await navigateToPage(page, DashboardPage.Prompts);
await page.getByRole("button", { name: "Upload .prompt File" }).click();
- try {
- await expect(
- page.getByRole("dialog", { name: "Add New Prompt" }),
- ).toBeVisible();
- await page.getByLabel("Prompt ID").fill(promptId);
- await page.locator('input[type="file"]').setInputFiles({
- name: "e2e.prompt",
- mimeType: "text/plain",
- buffer: Buffer.from(
- `model: fake-openai-gpt-4\ntemplate: "${promptContent}"\n`,
- ),
- });
- await expect(page.getByText("Selected: e2e.prompt")).toBeVisible();
- await page.getByRole("button", { name: "Create Prompt" }).click();
+ await expect(
+ page.getByRole("dialog", { name: "Add New Prompt" }),
+ ).toBeVisible();
+ await page.getByLabel("Prompt ID").fill(promptId);
+ await page.locator('input[type="file"]').setInputFiles({
+ name: "e2e.prompt",
+ mimeType: "text/plain",
+ buffer: Buffer.from(
+ `model: fake-openai-gpt-4\ntemplate: "${promptContent}"\n`,
+ ),
+ });
+ await expect(page.getByText("Selected: e2e.prompt")).toBeVisible();
+ await page.getByRole("button", { name: "Create Prompt" }).click();
+ await expect
+ .poll(async () => {
+ const response = await page.request.get(
+ `/prompts/${encodeURIComponent(promptId)}/info?environment=development`,
+ {
+ headers: { Authorization: `Bearer ${masterKey()}` },
+ },
+ );
+ return response.ok();
+ })
+ .toBe(true);
+ try {
await expect
.poll(async () => {
const response = await page.request.get(
diff --git a/tests/e2e/ui/tests/tagManagement/tagManagement.spec.ts b/tests/e2e/ui/tests/tagManagement/tagManagement.spec.ts
index e1d5138ea90..1104031263e 100644
--- a/tests/e2e/ui/tests/tagManagement/tagManagement.spec.ts
+++ b/tests/e2e/ui/tests/tagManagement/tagManagement.spec.ts
@@ -17,22 +17,22 @@ test.describe("Tag management", () => {
await navigateToPage(page, DashboardPage.TagManagement);
await page.getByRole("button", { name: "+ Create New Tag" }).click();
- try {
- await expect(
- page.getByRole("dialog", { name: "Create New Tag" }),
- ).toBeVisible();
- await page.getByLabel("Tag Name").fill(tagName);
- await page.getByLabel("Description").fill(description);
- await page.getByRole("button", { name: "Create Tag" }).click();
+ await expect(
+ page.getByRole("dialog", { name: "Create New Tag" }),
+ ).toBeVisible();
+ await page.getByLabel("Tag Name").fill(tagName);
+ await page.getByLabel("Description").fill(description);
+ await page.getByRole("button", { name: "Create Tag" }).click();
- await expect
- .poll(async () => {
- const response = await readBack<
- Record>
- >(page, "/tag/list");
- return Object.values(response).some((tag) => tag.name === tagName);
- })
- .toBe(true);
+ await expect
+ .poll(async () => {
+ const response = await readBack<
+ Record>
+ >(page, "/tag/list");
+ return Object.values(response).some((tag) => tag.name === tagName);
+ })
+ .toBe(true);
+ try {
await expect(page.getByText(tagName, { exact: true })).toBeVisible();
await page.getByText(tagName, { exact: true }).click();
From 035271b510d5f4f4053685cf156410775d8e5d46 Mon Sep 17 00:00:00 2001
From: Yuneng Jiang
Date: Thu, 17 Sep 2026 23:15:31 -0700
Subject: [PATCH 019/160] test(ui): preserve cleanup on failed readback
---
tests/e2e/ui/tests/prompts/addPrompt.spec.ts | 53 +++++++++----------
.../tests/tagManagement/tagManagement.spec.ts | 33 ++++++------
2 files changed, 42 insertions(+), 44 deletions(-)
diff --git a/tests/e2e/ui/tests/prompts/addPrompt.spec.ts b/tests/e2e/ui/tests/prompts/addPrompt.spec.ts
index a4a6cf62b7e..2b254c78b10 100644
--- a/tests/e2e/ui/tests/prompts/addPrompt.spec.ts
+++ b/tests/e2e/ui/tests/prompts/addPrompt.spec.ts
@@ -17,32 +17,32 @@ test.describe("Prompt upload form", () => {
await navigateToPage(page, DashboardPage.Prompts);
await page.getByRole("button", { name: "Upload .prompt File" }).click();
- await expect(
- page.getByRole("dialog", { name: "Add New Prompt" }),
- ).toBeVisible();
- await page.getByLabel("Prompt ID").fill(promptId);
- await page.locator('input[type="file"]').setInputFiles({
- name: "e2e.prompt",
- mimeType: "text/plain",
- buffer: Buffer.from(
- `model: fake-openai-gpt-4\ntemplate: "${promptContent}"\n`,
- ),
- });
- await expect(page.getByText("Selected: e2e.prompt")).toBeVisible();
- await page.getByRole("button", { name: "Create Prompt" }).click();
-
- await expect
- .poll(async () => {
- const response = await page.request.get(
- `/prompts/${encodeURIComponent(promptId)}/info?environment=development`,
- {
- headers: { Authorization: `Bearer ${masterKey()}` },
- },
- );
- return response.ok();
- })
- .toBe(true);
try {
+ await expect(
+ page.getByRole("dialog", { name: "Add New Prompt" }),
+ ).toBeVisible();
+ await page.getByLabel("Prompt ID").fill(promptId);
+ await page.locator('input[type="file"]').setInputFiles({
+ name: "e2e.prompt",
+ mimeType: "text/plain",
+ buffer: Buffer.from(
+ `model: fake-openai-gpt-4\ntemplate: "${promptContent}"\n`,
+ ),
+ });
+ await expect(page.getByText("Selected: e2e.prompt")).toBeVisible();
+ await page.getByRole("button", { name: "Create Prompt" }).click();
+
+ await expect
+ .poll(async () => {
+ const response = await page.request.get(
+ `/prompts/${encodeURIComponent(promptId)}/info?environment=development`,
+ {
+ headers: { Authorization: `Bearer ${masterKey()}` },
+ },
+ );
+ return response.ok();
+ })
+ .toBe(true);
await expect
.poll(async () => {
const response = await page.request.get(
@@ -60,13 +60,12 @@ test.describe("Prompt upload form", () => {
.toContain(promptContent);
await expect(page.getByText(promptId, { exact: true })).toBeVisible();
} finally {
- const deleteResponse = await page.request.delete(
+ await page.request.delete(
`/prompts/${encodeURIComponent(promptId)}?environment=development`,
{
headers: { Authorization: `Bearer ${masterKey()}` },
},
);
- expect(deleteResponse.ok()).toBe(true);
}
});
});
diff --git a/tests/e2e/ui/tests/tagManagement/tagManagement.spec.ts b/tests/e2e/ui/tests/tagManagement/tagManagement.spec.ts
index 1104031263e..785211463dd 100644
--- a/tests/e2e/ui/tests/tagManagement/tagManagement.spec.ts
+++ b/tests/e2e/ui/tests/tagManagement/tagManagement.spec.ts
@@ -17,22 +17,22 @@ test.describe("Tag management", () => {
await navigateToPage(page, DashboardPage.TagManagement);
await page.getByRole("button", { name: "+ Create New Tag" }).click();
- await expect(
- page.getByRole("dialog", { name: "Create New Tag" }),
- ).toBeVisible();
- await page.getByLabel("Tag Name").fill(tagName);
- await page.getByLabel("Description").fill(description);
- await page.getByRole("button", { name: "Create Tag" }).click();
-
- await expect
- .poll(async () => {
- const response = await readBack<
- Record>
- >(page, "/tag/list");
- return Object.values(response).some((tag) => tag.name === tagName);
- })
- .toBe(true);
try {
+ await expect(
+ page.getByRole("dialog", { name: "Create New Tag" }),
+ ).toBeVisible();
+ await page.getByLabel("Tag Name").fill(tagName);
+ await page.getByLabel("Description").fill(description);
+ await page.getByRole("button", { name: "Create Tag" }).click();
+
+ await expect
+ .poll(async () => {
+ const response = await readBack<
+ Record>
+ >(page, "/tag/list");
+ return Object.values(response).some((tag) => tag.name === tagName);
+ })
+ .toBe(true);
await expect(page.getByText(tagName, { exact: true })).toBeVisible();
await page.getByText(tagName, { exact: true }).click();
@@ -64,14 +64,13 @@ test.describe("Tag management", () => {
})
.toBe(updatedDescription);
} finally {
- const deleteResponse = await page.request.post("/tag/delete", {
+ await page.request.post("/tag/delete", {
headers: {
Authorization: `Bearer ${masterKey()}`,
"Content-Type": "application/json",
},
data: { name: tagName },
});
- expect(deleteResponse.ok()).toBe(true);
}
});
});
From 43f096dde8ef6c0a0c20036f0a595a7608ea2ea1 Mon Sep 17 00:00:00 2001
From: Yuneng Jiang
Date: Thu, 17 Sep 2026 23:18:43 -0700
Subject: [PATCH 020/160] test(ui): preserve form failure evidence
---
tests/e2e/ui/tests/prompts/addPrompt.spec.ts | 116 +++++++++-------
.../tests/tagManagement/tagManagement.spec.ts | 124 ++++++++++--------
2 files changed, 136 insertions(+), 104 deletions(-)
diff --git a/tests/e2e/ui/tests/prompts/addPrompt.spec.ts b/tests/e2e/ui/tests/prompts/addPrompt.spec.ts
index 2b254c78b10..b601b7e8c06 100644
--- a/tests/e2e/ui/tests/prompts/addPrompt.spec.ts
+++ b/tests/e2e/ui/tests/prompts/addPrompt.spec.ts
@@ -3,7 +3,6 @@ import { test, expect } from "@playwright/test";
import { ADMIN_STORAGE_PATH } from "../../constants";
import { Page as DashboardPage } from "../../fixtures/pages";
import { navigateToPage } from "../../helpers/navigation";
-import { readBack } from "../../helpers/roundTrip";
import { masterKey, uniqueSuffix } from "../../helpers/traffic";
test.use({ storageState: ADMIN_STORAGE_PATH });
@@ -14,58 +13,75 @@ test.describe("Prompt upload form", () => {
}) => {
const promptId = `e2e-prompt-${uniqueSuffix()}`;
const promptContent = "Hello {{name}}";
- await navigateToPage(page, DashboardPage.Prompts);
- await page.getByRole("button", { name: "Upload .prompt File" }).click();
+ const cleanup = async (): Promise => {
+ try {
+ const response = await page.request.delete(
+ `/prompts/${encodeURIComponent(promptId)}?environment=development`,
+ {
+ headers: { Authorization: `Bearer ${masterKey()}` },
+ },
+ );
+ return response.ok();
+ } catch {
+ return false;
+ }
+ };
+ const testOutcome = await (async () => {
+ try {
+ await navigateToPage(page, DashboardPage.Prompts);
+ await page.getByRole("button", { name: "Upload .prompt File" }).click();
+ await expect(
+ page.getByRole("dialog", { name: "Add New Prompt" }),
+ ).toBeVisible();
+ await page.getByLabel("Prompt ID").fill(promptId);
+ await page.locator('input[type="file"]').setInputFiles({
+ name: "e2e.prompt",
+ mimeType: "text/plain",
+ buffer: Buffer.from(
+ `model: fake-openai-gpt-4\ntemplate: "${promptContent}"\n`,
+ ),
+ });
+ await expect(page.getByText("Selected: e2e.prompt")).toBeVisible();
+ await page.getByRole("button", { name: "Create Prompt" }).click();
+
+ await expect
+ .poll(async () => {
+ const response = await page.request.get(
+ `/prompts/${encodeURIComponent(promptId)}/info?environment=development`,
+ {
+ headers: { Authorization: `Bearer ${masterKey()}` },
+ },
+ );
+ return response.ok();
+ })
+ .toBe(true);
+ await expect
+ .poll(async () => {
+ const response = await page.request.get(
+ `/prompts/${encodeURIComponent(promptId)}/info?environment=development`,
+ {
+ headers: { Authorization: `Bearer ${masterKey()}` },
+ },
+ );
+ if (!response.ok()) return undefined;
+ const promptInfo = (await response.json()) as {
+ raw_prompt_template?: { content?: string };
+ };
+ return promptInfo.raw_prompt_template?.content;
+ })
+ .toContain(promptContent);
+ await expect(page.getByText(promptId, { exact: true })).toBeVisible();
+ return { passed: true as const };
+ } catch (error) {
+ return { passed: false as const, error };
+ }
+ })();
try {
- await expect(
- page.getByRole("dialog", { name: "Add New Prompt" }),
- ).toBeVisible();
- await page.getByLabel("Prompt ID").fill(promptId);
- await page.locator('input[type="file"]').setInputFiles({
- name: "e2e.prompt",
- mimeType: "text/plain",
- buffer: Buffer.from(
- `model: fake-openai-gpt-4\ntemplate: "${promptContent}"\n`,
- ),
- });
- await expect(page.getByText("Selected: e2e.prompt")).toBeVisible();
- await page.getByRole("button", { name: "Create Prompt" }).click();
-
- await expect
- .poll(async () => {
- const response = await page.request.get(
- `/prompts/${encodeURIComponent(promptId)}/info?environment=development`,
- {
- headers: { Authorization: `Bearer ${masterKey()}` },
- },
- );
- return response.ok();
- })
- .toBe(true);
- await expect
- .poll(async () => {
- const response = await page.request.get(
- `/prompts/${encodeURIComponent(promptId)}/info?environment=development`,
- {
- headers: { Authorization: `Bearer ${masterKey()}` },
- },
- );
- if (!response.ok()) return undefined;
- const promptInfo = (await response.json()) as {
- raw_prompt_template?: { content?: string };
- };
- return promptInfo.raw_prompt_template?.content;
- })
- .toContain(promptContent);
- await expect(page.getByText(promptId, { exact: true })).toBeVisible();
+ if (!testOutcome.passed) throw testOutcome.error;
} finally {
- await page.request.delete(
- `/prompts/${encodeURIComponent(promptId)}?environment=development`,
- {
- headers: { Authorization: `Bearer ${masterKey()}` },
- },
- );
+ const cleanupSucceeded = await cleanup();
+ if (testOutcome.passed) expect(cleanupSucceeded).toBe(true);
}
});
});
diff --git a/tests/e2e/ui/tests/tagManagement/tagManagement.spec.ts b/tests/e2e/ui/tests/tagManagement/tagManagement.spec.ts
index 785211463dd..4223324ff09 100644
--- a/tests/e2e/ui/tests/tagManagement/tagManagement.spec.ts
+++ b/tests/e2e/ui/tests/tagManagement/tagManagement.spec.ts
@@ -13,64 +13,80 @@ test.describe("Tag management", () => {
const tagName = `e2e-tag-${uniqueSuffix()}`;
const description = "synthetic tag description";
const updatedDescription = `${description} updated`;
+ const cleanup = async (): Promise => {
+ try {
+ const response = await page.request.post("/tag/delete", {
+ headers: {
+ Authorization: `Bearer ${masterKey()}`,
+ "Content-Type": "application/json",
+ },
+ data: { name: tagName },
+ });
+ return response.ok();
+ } catch {
+ return false;
+ }
+ };
+ const testOutcome = await (async () => {
+ try {
+ await navigateToPage(page, DashboardPage.TagManagement);
+ await page.getByRole("button", { name: "+ Create New Tag" }).click();
+ await expect(
+ page.getByRole("dialog", { name: "Create New Tag" }),
+ ).toBeVisible();
+ await page.getByLabel("Tag Name").fill(tagName);
+ await page.getByLabel("Description").fill(description);
+ await page.getByRole("button", { name: "Create Tag" }).click();
- await navigateToPage(page, DashboardPage.TagManagement);
- await page.getByRole("button", { name: "+ Create New Tag" }).click();
+ await expect
+ .poll(async () => {
+ const response = await readBack<
+ Record>
+ >(page, "/tag/list");
+ return Object.values(response).some((tag) => tag.name === tagName);
+ })
+ .toBe(true);
+ await expect(page.getByText(tagName, { exact: true })).toBeVisible();
+
+ await page.getByText(tagName, { exact: true }).click();
+ await expect(page.getByText("Tag Name:")).toBeVisible();
+ await page.getByRole("button", { name: "Edit Tag" }).click();
+ await page.getByLabel("Description").fill(updatedDescription);
+ const updateBody = await captureRequestBody(
+ page,
+ { method: "POST", urlIncludes: "/tag/update" },
+ () => page.getByRole("button", { name: "Save Changes" }).click(),
+ );
+ expect(updateBody).toMatchObject({
+ name: tagName,
+ description: updatedDescription,
+ });
+
+ await expect
+ .poll(async () => {
+ const infoResponse = await page.request.post("/tag/info", {
+ headers: { Authorization: `Bearer ${masterKey()}` },
+ data: { names: [tagName] },
+ });
+ expect(infoResponse.ok()).toBe(true);
+ const info = (await infoResponse.json()) as Record<
+ string,
+ { description?: string }
+ >;
+ return info[tagName]?.description;
+ })
+ .toBe(updatedDescription);
+ return { passed: true as const };
+ } catch (error) {
+ return { passed: false as const, error };
+ }
+ })();
try {
- await expect(
- page.getByRole("dialog", { name: "Create New Tag" }),
- ).toBeVisible();
- await page.getByLabel("Tag Name").fill(tagName);
- await page.getByLabel("Description").fill(description);
- await page.getByRole("button", { name: "Create Tag" }).click();
-
- await expect
- .poll(async () => {
- const response = await readBack<
- Record>
- >(page, "/tag/list");
- return Object.values(response).some((tag) => tag.name === tagName);
- })
- .toBe(true);
- await expect(page.getByText(tagName, { exact: true })).toBeVisible();
-
- await page.getByText(tagName, { exact: true }).click();
- await expect(page.getByText("Tag Name:")).toBeVisible();
- await page.getByRole("button", { name: "Edit Tag" }).click();
- await page.getByLabel("Description").fill(updatedDescription);
- const updateBody = await captureRequestBody(
- page,
- { method: "POST", urlIncludes: "/tag/update" },
- () => page.getByRole("button", { name: "Save Changes" }).click(),
- );
- expect(updateBody).toMatchObject({
- name: tagName,
- description: updatedDescription,
- });
-
- await expect
- .poll(async () => {
- const infoResponse = await page.request.post("/tag/info", {
- headers: { Authorization: `Bearer ${masterKey()}` },
- data: { names: [tagName] },
- });
- expect(infoResponse.ok()).toBe(true);
- const info = (await infoResponse.json()) as Record<
- string,
- { description?: string }
- >;
- return info[tagName]?.description;
- })
- .toBe(updatedDescription);
+ if (!testOutcome.passed) throw testOutcome.error;
} finally {
- await page.request.post("/tag/delete", {
- headers: {
- Authorization: `Bearer ${masterKey()}`,
- "Content-Type": "application/json",
- },
- data: { name: tagName },
- });
+ const cleanupSucceeded = await cleanup();
+ if (testOutcome.passed) expect(cleanupSucceeded).toBe(true);
}
});
});
From 0dc2f0b1c1ff86356d6e9ab9180f708402131df8 Mon Sep 17 00:00:00 2001
From: Yuneng Jiang
Date: Thu, 17 Sep 2026 23:20:54 -0700
Subject: [PATCH 021/160] test(ui): protect dashboard form cleanup
---
tests/e2e/ui/helpers/roundTrip.ts | 28 ++++++++++-
tests/e2e/ui/tests/prompts/addPrompt.spec.ts | 42 ++++++----------
.../tests/tagManagement/tagManagement.spec.ts | 49 ++++++++-----------
3 files changed, 61 insertions(+), 58 deletions(-)
diff --git a/tests/e2e/ui/helpers/roundTrip.ts b/tests/e2e/ui/helpers/roundTrip.ts
index 8d6e264e622..ee484b8d512 100644
--- a/tests/e2e/ui/helpers/roundTrip.ts
+++ b/tests/e2e/ui/helpers/roundTrip.ts
@@ -12,17 +12,41 @@ export async function captureRequestBody(
match: { method: string; urlIncludes: string },
action: () => Promise,
): Promise> {
- const pending = page.waitForRequest((req) => req.method() === match.method && req.url().includes(match.urlIncludes));
+ const pending = page.waitForRequest(
+ (req) =>
+ req.method() === match.method && req.url().includes(match.urlIncludes),
+ );
await action();
const request = await pending;
return JSON.parse(request.postData() ?? "{}") as Record;
}
/** Reads an endpoint as the master key, so a failure is bad data and not an expired UI token. */
-export async function readBack(page: Page, endpoint: string): Promise {
+export async function readBack(
+ page: Page,
+ endpoint: string,
+): Promise {
const res = await page.request.get(endpoint, {
headers: { Authorization: `Bearer ${masterKey()}` },
});
expect(res.ok(), `GET ${endpoint}`).toBe(true);
return (await res.json()) as T;
}
+
+export async function runWithCleanup(
+ action: () => Promise,
+ cleanup: () => Promise,
+): Promise {
+ const outcome = await action().then(
+ () => ({ status: "success" as const }),
+ (error: unknown) => ({ status: "failure" as const, error }),
+ );
+ try {
+ if (outcome.status === "failure") throw outcome.error;
+ } finally {
+ const cleanupSucceeded = await cleanup().catch(() => false);
+ if (outcome.status === "success" && !cleanupSucceeded) {
+ throw new Error("Failed to clean up UI E2E resource");
+ }
+ }
+}
diff --git a/tests/e2e/ui/tests/prompts/addPrompt.spec.ts b/tests/e2e/ui/tests/prompts/addPrompt.spec.ts
index b601b7e8c06..891739fda28 100644
--- a/tests/e2e/ui/tests/prompts/addPrompt.spec.ts
+++ b/tests/e2e/ui/tests/prompts/addPrompt.spec.ts
@@ -3,6 +3,7 @@ import { test, expect } from "@playwright/test";
import { ADMIN_STORAGE_PATH } from "../../constants";
import { Page as DashboardPage } from "../../fixtures/pages";
import { navigateToPage } from "../../helpers/navigation";
+import { runWithCleanup } from "../../helpers/roundTrip";
import { masterKey, uniqueSuffix } from "../../helpers/traffic";
test.use({ storageState: ADMIN_STORAGE_PATH });
@@ -13,21 +14,9 @@ test.describe("Prompt upload form", () => {
}) => {
const promptId = `e2e-prompt-${uniqueSuffix()}`;
const promptContent = "Hello {{name}}";
- const cleanup = async (): Promise => {
- try {
- const response = await page.request.delete(
- `/prompts/${encodeURIComponent(promptId)}?environment=development`,
- {
- headers: { Authorization: `Bearer ${masterKey()}` },
- },
- );
- return response.ok();
- } catch {
- return false;
- }
- };
- const testOutcome = await (async () => {
- try {
+
+ await runWithCleanup(
+ async () => {
await navigateToPage(page, DashboardPage.Prompts);
await page.getByRole("button", { name: "Upload .prompt File" }).click();
await expect(
@@ -71,17 +60,16 @@ test.describe("Prompt upload form", () => {
})
.toContain(promptContent);
await expect(page.getByText(promptId, { exact: true })).toBeVisible();
- return { passed: true as const };
- } catch (error) {
- return { passed: false as const, error };
- }
- })();
-
- try {
- if (!testOutcome.passed) throw testOutcome.error;
- } finally {
- const cleanupSucceeded = await cleanup();
- if (testOutcome.passed) expect(cleanupSucceeded).toBe(true);
- }
+ },
+ async () => {
+ const response = await page.request.delete(
+ `/prompts/${encodeURIComponent(promptId)}?environment=development`,
+ {
+ headers: { Authorization: `Bearer ${masterKey()}` },
+ },
+ );
+ return response.ok();
+ },
+ );
});
});
diff --git a/tests/e2e/ui/tests/tagManagement/tagManagement.spec.ts b/tests/e2e/ui/tests/tagManagement/tagManagement.spec.ts
index 4223324ff09..bf46b5ea161 100644
--- a/tests/e2e/ui/tests/tagManagement/tagManagement.spec.ts
+++ b/tests/e2e/ui/tests/tagManagement/tagManagement.spec.ts
@@ -3,7 +3,11 @@ import { test, expect } from "@playwright/test";
import { ADMIN_STORAGE_PATH } from "../../constants";
import { Page as DashboardPage } from "../../fixtures/pages";
import { navigateToPage } from "../../helpers/navigation";
-import { captureRequestBody, readBack } from "../../helpers/roundTrip";
+import {
+ captureRequestBody,
+ readBack,
+ runWithCleanup,
+} from "../../helpers/roundTrip";
import { masterKey, uniqueSuffix } from "../../helpers/traffic";
test.use({ storageState: ADMIN_STORAGE_PATH });
@@ -13,22 +17,9 @@ test.describe("Tag management", () => {
const tagName = `e2e-tag-${uniqueSuffix()}`;
const description = "synthetic tag description";
const updatedDescription = `${description} updated`;
- const cleanup = async (): Promise => {
- try {
- const response = await page.request.post("/tag/delete", {
- headers: {
- Authorization: `Bearer ${masterKey()}`,
- "Content-Type": "application/json",
- },
- data: { name: tagName },
- });
- return response.ok();
- } catch {
- return false;
- }
- };
- const testOutcome = await (async () => {
- try {
+
+ await runWithCleanup(
+ async () => {
await navigateToPage(page, DashboardPage.TagManagement);
await page.getByRole("button", { name: "+ Create New Tag" }).click();
await expect(
@@ -76,17 +67,17 @@ test.describe("Tag management", () => {
return info[tagName]?.description;
})
.toBe(updatedDescription);
- return { passed: true as const };
- } catch (error) {
- return { passed: false as const, error };
- }
- })();
-
- try {
- if (!testOutcome.passed) throw testOutcome.error;
- } finally {
- const cleanupSucceeded = await cleanup();
- if (testOutcome.passed) expect(cleanupSucceeded).toBe(true);
- }
+ },
+ async () => {
+ const response = await page.request.post("/tag/delete", {
+ headers: {
+ Authorization: `Bearer ${masterKey()}`,
+ "Content-Type": "application/json",
+ },
+ data: { name: tagName },
+ });
+ return response.ok();
+ },
+ );
});
});
From a8ab1187ca67f59432beed8b655e833f622a4055 Mon Sep 17 00:00:00 2001
From: Yuneng Jiang
Date: Thu, 17 Sep 2026 23:23:48 -0700
Subject: [PATCH 022/160] test(ui): clean up synchronous browser failures
---
tests/e2e/ui/helpers/roundTrip.ts | 10 ++++++----
1 file changed, 6 insertions(+), 4 deletions(-)
diff --git a/tests/e2e/ui/helpers/roundTrip.ts b/tests/e2e/ui/helpers/roundTrip.ts
index ee484b8d512..55eb5d6ad9c 100644
--- a/tests/e2e/ui/helpers/roundTrip.ts
+++ b/tests/e2e/ui/helpers/roundTrip.ts
@@ -37,10 +37,12 @@ export async function runWithCleanup(
action: () => Promise,
cleanup: () => Promise,
): Promise {
- const outcome = await action().then(
- () => ({ status: "success" as const }),
- (error: unknown) => ({ status: "failure" as const, error }),
- );
+ const outcome = await Promise.resolve()
+ .then(action)
+ .then(
+ () => ({ status: "success" as const }),
+ (error: unknown) => ({ status: "failure" as const, error }),
+ );
try {
if (outcome.status === "failure") throw outcome.error;
} finally {
From 1c08c78ad598f0fa1277305f5a32aa7ac45a2022 Mon Sep 17 00:00:00 2001
From: Yuneng Jiang
Date: Thu, 17 Sep 2026 23:27:54 -0700
Subject: [PATCH 023/160] test(ui): retain primary cleanup failures
---
tests/e2e/ui/helpers/roundTrip.ts | 4 +++-
1 file changed, 3 insertions(+), 1 deletion(-)
diff --git a/tests/e2e/ui/helpers/roundTrip.ts b/tests/e2e/ui/helpers/roundTrip.ts
index 55eb5d6ad9c..4175ba6ec72 100644
--- a/tests/e2e/ui/helpers/roundTrip.ts
+++ b/tests/e2e/ui/helpers/roundTrip.ts
@@ -46,7 +46,9 @@ export async function runWithCleanup(
try {
if (outcome.status === "failure") throw outcome.error;
} finally {
- const cleanupSucceeded = await cleanup().catch(() => false);
+ const cleanupSucceeded = await Promise.resolve()
+ .then(cleanup)
+ .catch(() => false);
if (outcome.status === "success" && !cleanupSucceeded) {
throw new Error("Failed to clean up UI E2E resource");
}
From eb831d956ccb328411ebb86a0161ac7a23b4aba8 Mon Sep 17 00:00:00 2001
From: Yuneng Jiang
Date: Thu, 17 Sep 2026 23:46:53 -0700
Subject: [PATCH 024/160] test(ui): address review feedback
---
tests/e2e/ui/helpers/roundTrip.ts | 23 +++++++++---
tests/e2e/ui/tests/prompts/addPrompt.spec.ts | 4 +--
.../tests/tagManagement/tagManagement.spec.ts | 9 ++---
...PaginatedSearchSelect.integration.test.tsx | 36 -------------------
4 files changed, 26 insertions(+), 46 deletions(-)
diff --git a/tests/e2e/ui/helpers/roundTrip.ts b/tests/e2e/ui/helpers/roundTrip.ts
index 4175ba6ec72..1fc2d0aec1a 100644
--- a/tests/e2e/ui/helpers/roundTrip.ts
+++ b/tests/e2e/ui/helpers/roundTrip.ts
@@ -46,11 +46,26 @@ export async function runWithCleanup(
try {
if (outcome.status === "failure") throw outcome.error;
} finally {
- const cleanupSucceeded = await Promise.resolve()
+ const cleanupOutcome = await Promise.resolve()
.then(cleanup)
- .catch(() => false);
- if (outcome.status === "success" && !cleanupSucceeded) {
- throw new Error("Failed to clean up UI E2E resource");
+ .then(
+ (succeeded) =>
+ succeeded
+ ? { status: "success" as const }
+ : {
+ status: "failure" as const,
+ error: new Error("Failed to clean up UI E2E resource"),
+ },
+ (error: unknown) => ({ status: "failure" as const, error }),
+ );
+ if (cleanupOutcome.status === "failure") {
+ if (outcome.status === "failure") {
+ throw new AggregateError(
+ [outcome.error, cleanupOutcome.error],
+ "Action and cleanup failed",
+ );
+ }
+ throw cleanupOutcome.error;
}
}
}
diff --git a/tests/e2e/ui/tests/prompts/addPrompt.spec.ts b/tests/e2e/ui/tests/prompts/addPrompt.spec.ts
index 891739fda28..9d85236c4a6 100644
--- a/tests/e2e/ui/tests/prompts/addPrompt.spec.ts
+++ b/tests/e2e/ui/tests/prompts/addPrompt.spec.ts
@@ -27,7 +27,7 @@ test.describe("Prompt upload form", () => {
name: "e2e.prompt",
mimeType: "text/plain",
buffer: Buffer.from(
- `model: fake-openai-gpt-4\ntemplate: "${promptContent}"\n`,
+ `---\nmodel: fake-openai-gpt-4\n---\n${promptContent}\n`,
),
});
await expect(page.getByText("Selected: e2e.prompt")).toBeVisible();
@@ -58,7 +58,7 @@ test.describe("Prompt upload form", () => {
};
return promptInfo.raw_prompt_template?.content;
})
- .toContain(promptContent);
+ .toBe(promptContent);
await expect(page.getByText(promptId, { exact: true })).toBeVisible();
},
async () => {
diff --git a/tests/e2e/ui/tests/tagManagement/tagManagement.spec.ts b/tests/e2e/ui/tests/tagManagement/tagManagement.spec.ts
index bf46b5ea161..fe659080eab 100644
--- a/tests/e2e/ui/tests/tagManagement/tagManagement.spec.ts
+++ b/tests/e2e/ui/tests/tagManagement/tagManagement.spec.ts
@@ -31,10 +31,11 @@ test.describe("Tag management", () => {
await expect
.poll(async () => {
- const response = await readBack<
- Record>
- >(page, "/tag/list");
- return Object.values(response).some((tag) => tag.name === tagName);
+ const response = await readBack>(
+ page,
+ "/tag/list",
+ );
+ return response.some((tag) => tag.name === tagName);
})
.toBe(true);
await expect(page.getByText(tagName, { exact: true })).toBeVisible();
diff --git a/ui/litellm-dashboard/src/components/shared/PaginatedSearchSelect.integration.test.tsx b/ui/litellm-dashboard/src/components/shared/PaginatedSearchSelect.integration.test.tsx
index 905610a77e6..2b948ca8420 100644
--- a/ui/litellm-dashboard/src/components/shared/PaginatedSearchSelect.integration.test.tsx
+++ b/ui/litellm-dashboard/src/components/shared/PaginatedSearchSelect.integration.test.tsx
@@ -1,6 +1,5 @@
import { fireEvent, renderWithProviders as render, screen, waitFor } from "../../../tests/test-utils";
import userEvent from "@testing-library/user-event";
-import { useQuery } from "@tanstack/react-query";
import { useState } from "react";
import { describe, expect, it, vi } from "vitest";
@@ -407,39 +406,4 @@ describe("PaginatedSearchSelect", () => {
expect(input).toHaveValue("aliasalpha");
await waitFor(() => expect(onSearchChange).toHaveBeenLastCalledWith("aliasalpha"));
});
-
- it("keeps the latest query results when an earlier response resolves last", async () => {
- const pending = new Map void>();
-
- function QueryBackedSelect() {
- const [query, setQuery] = useState("");
- const result = useQuery({
- queryKey: ["paginated-select-race", query],
- queryFn: () =>
- new Promise((resolve) => {
- pending.set(query, resolve);
- }),
- enabled: query.length > 0,
- });
- return ;
- }
-
- const user = userEvent.setup();
- render( );
- const input = screen.getByRole("combobox");
- await user.click(input);
- await user.type(input, "A");
- await waitFor(() => expect(pending.has("A")).toBe(true));
- await user.clear(input);
- await user.type(input, "B");
- await waitFor(() => expect(pending.has("B")).toBe(true));
-
- pending.get("B")?.([{ label: "B result", value: "b" }]);
- await user.click(screen.getByRole("combobox"));
- expect(await screen.findByText("B result")).toBeInTheDocument();
-
- pending.get("A")?.([{ label: "A result", value: "a" }]);
- await waitFor(() => expect(screen.queryByText("A result")).not.toBeInTheDocument());
- expect(screen.getByText("B result")).toBeInTheDocument();
- });
});
From 5a8d1f5ecaedac9348f93fd87873e7a16b9fba0f Mon Sep 17 00:00:00 2001
From: Yuneng Jiang
Date: Fri, 18 Sep 2026 00:05:43 -0700
Subject: [PATCH 025/160] feat(proxy): report sources in config read endpoints
---
litellm/proxy/_types.py | 3 +
litellm/proxy/config_resolvers/__init__.py | 11 +-
.../proxy/config_resolvers/settings_store.py | 12 +-
.../router_settings_endpoints.py | 12 +-
litellm/proxy/proxy_server.py | 168 ++++++++++--------
.../proxy_setting_endpoints.py | 65 +++++--
.../test_router_settings_endpoints.py | 30 ++++
.../proxy/proxy_server/test_routes_config.py | 137 ++++++++++++++
.../proxy_server/test_routes_model_metrics.py | 38 ++++
.../test_proxy_setting_endpoints.py | 39 ++++
ui/litellm-dashboard/src/lib/http/schema.d.ts | 29 +++
11 files changed, 450 insertions(+), 94 deletions(-)
diff --git a/litellm/proxy/_types.py b/litellm/proxy/_types.py
index b0d31df92ce..8574f2d8bf4 100644
--- a/litellm/proxy/_types.py
+++ b/litellm/proxy/_types.py
@@ -2410,6 +2410,7 @@ class FieldDetail(BaseModel):
field_description: str
field_default_value: Any = None
stored_in_db: bool | None
+ source: Literal["config", "db", "default", "unset"] = "unset"
class ConfigList(LiteLLMPydanticObjectBase):
@@ -2418,6 +2419,7 @@ class ConfigList(LiteLLMPydanticObjectBase):
field_description: str
field_value: Any
stored_in_db: bool | None
+ source: Literal["config", "db", "default", "unset"] = "unset"
field_default_value: Any
premium_field: bool = False
nested_fields: list[FieldDetail] | None = None # For nested dictionary or Pydantic fields
@@ -3693,6 +3695,7 @@ class InvitationClaim(LiteLLMPydanticObjectBase):
class ConfigFieldInfo(LiteLLMPydanticObjectBase):
field_name: str
field_value: Any
+ source: Literal["config", "db", "default", "unset"] = "unset"
class CallbackOnUI(LiteLLMPydanticObjectBase):
diff --git a/litellm/proxy/config_resolvers/__init__.py b/litellm/proxy/config_resolvers/__init__.py
index ebd339b34c3..77da2c413c2 100644
--- a/litellm/proxy/config_resolvers/__init__.py
+++ b/litellm/proxy/config_resolvers/__init__.py
@@ -5,6 +5,13 @@ from litellm.proxy.config_resolvers._descriptors import (
FieldSource,
resolve_fields,
)
-from litellm.proxy.config_resolvers.settings_store import SettingsStore
+from litellm.proxy.config_resolvers.settings_store import SettingsSource, SettingsStore, source_for
-__all__ = ("FieldDescriptor", "FieldSource", "SettingsStore", "resolve_fields")
+__all__ = (
+ "FieldDescriptor",
+ "FieldSource",
+ "SettingsSource",
+ "SettingsStore",
+ "resolve_fields",
+ "source_for",
+)
diff --git a/litellm/proxy/config_resolvers/settings_store.py b/litellm/proxy/config_resolvers/settings_store.py
index 3fe869ee2ce..8f400853fa9 100644
--- a/litellm/proxy/config_resolvers/settings_store.py
+++ b/litellm/proxy/config_resolvers/settings_store.py
@@ -2,7 +2,7 @@ from __future__ import annotations
from collections.abc import Iterator, Mapping, MutableMapping
from types import MappingProxyType
-from typing import Final
+from typing import Final, Literal, TypeAlias
from litellm.proxy.config_resolvers._descriptors import FieldSource
from litellm.proxy.config_resolvers.settings_rules import (
@@ -19,6 +19,7 @@ from litellm.proxy.config_resolvers.settings_rules import (
_EMPTY_VALUES: Final[Mapping[str, JsonValue]] = MappingProxyType({})
_EMPTY_ROWS: Final[Mapping[DbRow, Mapping[str, JsonValue]]] = MappingProxyType({})
+SettingsSource: TypeAlias = Literal["config", "db", "default", "unset"]
class SettingsStore(MutableMapping[str, JsonValue]):
@@ -109,3 +110,12 @@ class SettingsStore(MutableMapping[str, JsonValue]):
yaml_value: Final[SettingValue] = self._yaml_values.get(key, ABSENT)
db_value: Final[SettingValue] = self._database_rows.get(rule.db_row, _EMPTY_VALUES).get(key, ABSENT)
return resolve(rule, yaml_value, db_value)
+
+
+def source_for(settings: SettingsStore, key: str, default: object = None) -> SettingsSource:
+ source: Final = settings.source(key)
+ if source == "unset":
+ return "default" if default is not None else "unset"
+ if source in ("config", "db", "default"):
+ return source
+ return "unset"
diff --git a/litellm/proxy/management_endpoints/router_settings_endpoints.py b/litellm/proxy/management_endpoints/router_settings_endpoints.py
index fc000b1638b..5d3b6d40601 100644
--- a/litellm/proxy/management_endpoints/router_settings_endpoints.py
+++ b/litellm/proxy/management_endpoints/router_settings_endpoints.py
@@ -8,7 +8,7 @@ GET /router/fields - Get router settings field definitions without values (for U
"""
import inspect
-from typing import Any, Final, get_args
+from typing import Any, Final, cast, get_args
from fastapi import APIRouter, Depends
from pydantic import BaseModel, Field
@@ -16,6 +16,7 @@ from pydantic import BaseModel, Field
from litellm._logging import verbose_proxy_logger
from litellm.proxy._types import UserAPIKeyAuth
from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
+from litellm.proxy.config_resolvers import SettingsSource, source_for
from litellm.router import Router
from litellm.types.management_endpoints import (
ROUTER_SETTINGS_FIELDS,
@@ -30,6 +31,7 @@ class RouterSettingsResponse(BaseModel):
fields: list[RouterSettingsField] = Field(description="List of all configurable router settings with metadata")
current_values: dict[str, Any] = Field(description="Current values of router settings")
routing_strategy_descriptions: dict[str, str] = Field(description="Descriptions for each routing strategy option")
+ source: dict[str, SettingsSource] = Field(description="Source of each current router setting")
class RouterFieldsResponse(BaseModel):
@@ -109,15 +111,21 @@ async def get_router_settings(
# Merge with config values (config takes precedence)
current_values.update(router_settings_from_config)
- # Update field values with current values
for field in router_fields:
if field.field_name in current_values:
field.field_value = current_values[field.field_name]
+ field_defaults: Final[dict[str, object]] = {
+ field.field_name: cast(object, field.field_default) for field in router_fields
+ }
+ source: Final[dict[str, SettingsSource]] = {
+ key: source_for(proxy_config.router_settings, key, field_defaults.get(key)) for key in current_values
+ }
return RouterSettingsResponse(
fields=router_fields,
current_values=current_values,
routing_strategy_descriptions=ROUTING_STRATEGY_DESCRIPTIONS,
+ source=source,
)
except Exception as e:
verbose_proxy_logger.error("Error fetching router settings: %s", e)
diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py
index 6a720a066b4..1131d7bea86 100644
--- a/litellm/proxy/proxy_server.py
+++ b/litellm/proxy/proxy_server.py
@@ -50,6 +50,7 @@ import anyio
import websockets
import websockets.exceptions
from pydantic import BaseModel, Json, JsonValue, TypeAdapter, ValidationError
+from pydantic.fields import FieldInfo, PydanticUndefined
from typing_extensions import NotRequired, ReadOnly, assert_never
from litellm._uuid import uuid
@@ -431,7 +432,7 @@ from litellm.proxy.common_utils.user_api_key_cache import (
model_access_group_spend_counter_key,
tag_cache_key,
)
-from litellm.proxy.config_resolvers import SettingsStore, resolve_fields
+from litellm.proxy.config_resolvers import SettingsStore, resolve_fields, source_for
from litellm.proxy.config_resolvers.alerting import (
EMAIL_DESCRIPTORS,
MS_TEAMS_DESCRIPTORS,
@@ -4816,6 +4817,12 @@ def _as_settings_mapping(value: object) -> Mapping[str, SettingsJsonValue]:
return _SETTINGS_MAPPING.validate_python(value)
+def _get_field_default(field_info: FieldInfo) -> JsonValue:
+ if field_info.default is PydanticUndefined:
+ return None
+ return cast(JsonValue, field_info.default)
+
+
def _bind_general_settings_store(settings: SettingsStore) -> None:
global general_settings
general_settings = settings # pyright: ignore[reportAssignmentType] # legacy global accepts mappings
@@ -15635,17 +15642,16 @@ async def alerting_settings(
where={"param_name": "general_settings"}
)
- if db_general_settings is not None and db_general_settings.param_value is not None:
- db_general_settings_dict: Final = dict(db_general_settings.param_value)
- alerting_args_dict: dict = cast( # cast-ok: ConfigGeneralSettings validates alerting_args as a dict on write
- dict[str, JsonValue], db_general_settings_dict.get("alerting_args", {})
- )
- alerting_values: list | None = cast( # cast-ok: ConfigGeneralSettings validates alerting as a list on write
- list[JsonValue] | None, db_general_settings_dict.get("alerting")
- )
- else:
- alerting_args_dict = {}
- alerting_values = None
+ db_general_settings_dict: Final[Mapping[str, JsonValue]] = (
+ dict(db_general_settings.param_value)
+ if db_general_settings is not None and db_general_settings.param_value is not None
+ else {}
+ )
+ alerting_args_dict: Final = cast(dict[str, JsonValue], db_general_settings_dict.get("alerting_args", {}))
+ alerting_values: Final = cast(list[JsonValue] | None, db_general_settings_dict.get("alerting"))
+
+ settings: Final = proxy_config.settings
+ settings.apply_db_row("general_settings", db_general_settings_dict)
allowed_args: Final = MappingProxyType(
{
@@ -15674,9 +15680,9 @@ async def alerting_settings(
is_slack_enabled = False
- if general_settings.get("alerting") and isinstance(general_settings["alerting"], list):
- if "slack" in general_settings["alerting"]:
- is_slack_enabled = True
+ alerting: Final = settings.get("alerting")
+ if isinstance(alerting, list) and "slack" in alerting:
+ is_slack_enabled = True
_response_obj = ConfigList(
field_name="slack_alerting",
@@ -15684,6 +15690,7 @@ async def alerting_settings(
field_description="Enable slack alerting for monitoring proxy in production: llm outages, budgets, spend tracking failures.",
field_value=is_slack_enabled,
stored_in_db=True if alerting_values is not None else False,
+ source=source_for(settings, "alerting"),
field_default_value=None,
premium_field=False,
)
@@ -15691,6 +15698,7 @@ async def alerting_settings(
for field_name, field_info in SlackAlertingArgs.model_fields.items():
if field_name in allowed_args:
+ field_default: JsonValue = _get_field_default(field_info)
_stored_in_db: bool | None = None
if field_name in alerting_args_dict:
_stored_in_db = True
@@ -15701,9 +15709,10 @@ async def alerting_settings(
field_name=field_name,
field_type=allowed_args[field_name],
field_description=field_info.description or "",
- field_value=_slack_alerting_args_dict.get(field_name, None),
+ field_value=_slack_alerting_args_dict.get(field_name, field_default),
stored_in_db=_stored_in_db,
- field_default_value=field_info.default,
+ source=source_for(settings, "alerting_args", field_default),
+ field_default_value=field_default,
premium_field=(True if field_name == "region_outage_alert_ttl" else False),
)
return_val.append(_response_obj)
@@ -17390,20 +17399,6 @@ async def get_config_general_settings(
field_name: str,
user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth),
):
- global prisma_client
-
- ## VALIDATION ##
- """
- - Check if prisma_client is None
- - Check if user allowed to call this endpoint (admin-only)
- - Check if param in general settings
- """
- if prisma_client is None:
- raise HTTPException(
- status_code=400,
- detail={"error": CommonProxyErrors.db_not_connected_error.value},
- )
-
if not _user_has_admin_view(user_api_key_dict):
raise HTTPException(
status_code=400,
@@ -17416,37 +17411,47 @@ async def get_config_general_settings(
detail={"error": f"Invalid field={field_name} passed in."},
)
- ## get general settings from db
- db_general_settings: Final[_ConfigParamRow | None] = await _config_param_table(prisma_client).find_first(
- where={"param_name": "general_settings"}
- )
- ### pop the value
+ field_info: Final = ConfigGeneralSettings.model_fields[field_name]
+ field_default: JsonValue = _get_field_default(field_info)
+ settings: Final = proxy_config.settings
+ db_values: Mapping[str, JsonValue]
+ if prisma_client is None:
+ db_values = {}
+ else:
+ db_general_settings: Final[_ConfigParamRow | None] = await _config_param_table(prisma_client).find_first(
+ where={"param_name": "general_settings"}
+ )
+ db_values = (
+ dict(db_general_settings.param_value)
+ if db_general_settings is not None and db_general_settings.param_value is not None
+ else {}
+ )
+ settings.apply_db_row("general_settings", db_values)
- if db_general_settings is None or db_general_settings.param_value is None:
+ if field_name not in settings and field_default is None:
raise HTTPException(
status_code=400,
detail={"error": f"Field name={field_name} not in DB"},
)
- else:
- general_settings = dict(db_general_settings.param_value)
- if field_name in general_settings:
- field_value = _redact_general_setting_value(
- field_name,
- general_settings[field_name],
- user_api_key_dict.user_role == LitellmUserRoles.PROXY_ADMIN,
- )
- if field_name == "plugins" and isinstance(field_value, list):
- field_value = [
- ({k: ("***" if k == "plugin_key" else v) for k, v in p.items()} if isinstance(p, dict) else p)
- for p in field_value
- ]
- return ConfigFieldInfo(field_name=field_name, field_value=field_value)
- else:
- raise HTTPException(
- status_code=400,
- detail={"error": f"Field name={field_name} not in DB"},
- )
+ redacted_field_value: Final = _redact_general_setting_value(
+ field_name,
+ settings.get(field_name, field_default),
+ user_api_key_dict.user_role == LitellmUserRoles.PROXY_ADMIN,
+ )
+ field_value: Final = (
+ [
+ ({k: ("***" if k == "plugin_key" else v) for k, v in p.items()} if isinstance(p, dict) else p)
+ for p in redacted_field_value
+ ]
+ if field_name == "plugins" and isinstance(redacted_field_value, list)
+ else redacted_field_value
+ )
+ return ConfigFieldInfo(
+ field_name=field_name,
+ field_value=field_value,
+ source=source_for(settings, field_name, field_default),
+ )
GeneralSettingsUILiteLLMValue = float | bool | str | None
@@ -17600,7 +17605,7 @@ async def get_config_list(
"""
List the available fields + current values for a given type of setting (currently just 'general_settings'user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth),)
"""
- global prisma_client, general_settings
+ global prisma_client
## VALIDATION ##
"""
@@ -17627,10 +17632,16 @@ async def get_config_list(
where={"param_name": "general_settings"}
)
- if db_general_settings is not None and db_general_settings.param_value is not None:
- db_general_settings_dict: Mapping[str, JsonValue] = dict(db_general_settings.param_value)
- else:
- db_general_settings_dict = {}
+ db_general_settings_dict: Final[Mapping[str, JsonValue]] = (
+ dict(db_general_settings.param_value)
+ if db_general_settings is not None and db_general_settings.param_value is not None
+ else {}
+ )
+ settings: Final = proxy_config.settings
+ settings.apply_db_row("general_settings", db_general_settings_dict)
+ runtime_settings: Final[Mapping[str, JsonValue]] = (
+ cast(Mapping[str, JsonValue], general_settings) if not isinstance(general_settings, SettingsStore) else settings
+ )
allowed_args: Final = _GENERAL_SETTINGS_CONFIG_LIST_FIELD_TYPES
@@ -17638,6 +17649,7 @@ async def get_config_list(
for field_name, field_info in ConfigGeneralSettings.model_fields.items():
if field_name in allowed_args:
+ field_default: JsonValue = _get_field_default(field_info)
## HANDLE TYPED DICT
typed_dict_type = allowed_args[field_name]
@@ -17657,10 +17669,11 @@ async def get_config_list(
field_description="", # Add custom logic if descriptions are available
field_default_value=_redact_general_setting_value(
sub_field,
- general_settings.get(sub_field, None),
+ runtime_settings.get(sub_field, None),
is_full_admin,
),
stored_in_db=None,
+ source=source_for(settings, field_name),
)
for sub_field, sub_field_type in pydantic_class.__annotations__.items()
]
@@ -17677,7 +17690,7 @@ async def get_config_list(
_stored_in_db = None
if field_name in db_general_settings_dict:
_stored_in_db = True
- elif field_name in general_settings:
+ elif field_name in runtime_settings:
_stored_in_db = False
_response_obj = ConfigList(
@@ -17686,11 +17699,12 @@ async def get_config_list(
field_description=field_info.description or "",
field_value=_redact_general_setting_value(
field_name,
- general_settings.get(field_name, None),
+ runtime_settings.get(field_name, field_default),
is_full_admin,
),
stored_in_db=_stored_in_db,
- field_default_value=field_info.default,
+ source=source_for(settings, field_name, field_default),
+ field_default_value=field_default,
nested_fields=nested_fields,
)
return_val.append(_response_obj)
@@ -17701,12 +17715,10 @@ async def get_config_list(
_stored_in_db = None
if field_name in db_general_settings_dict:
_stored_in_db = True
- elif field_name in general_settings:
+ elif field_name in runtime_settings:
_stored_in_db = False
- _field_value = general_settings.get(field_name, None)
- if _field_value is None and field_name in db_general_settings_dict:
- _field_value = db_general_settings_dict[field_name]
+ _field_value: JsonValue = runtime_settings.get(field_name, field_default)
_response_obj = ConfigList(
field_name=field_name,
@@ -17714,7 +17726,8 @@ async def get_config_list(
field_description=field_info.description or "",
field_value=_redact_general_setting_value(field_name, _field_value, is_full_admin),
stored_in_db=_stored_in_db,
- field_default_value=field_info.default,
+ source=source_for(settings, field_name, field_default),
+ field_default_value=field_default,
nested_fields=nested_fields,
)
return_val.append(_response_obj)
@@ -17722,18 +17735,24 @@ async def get_config_list(
db_litellm_settings_row: Final[_ConfigParamRow | None] = await _config_param_table(prisma_client).find_first(
where={"param_name": "litellm_settings"}
)
- db_litellm_settings: Final[dict] = (
+ db_litellm_settings: Final[Mapping[str, JsonValue]] = (
dict(db_litellm_settings_row.param_value)
if db_litellm_settings_row is not None and db_litellm_settings_row.param_value is not None
else {}
)
+ litellm_settings_store: Final = proxy_config.litellm_settings
+ litellm_settings_store.apply_db_row("litellm_settings", db_litellm_settings)
for litellm_field_name, spec in _GENERAL_SETTINGS_UI_LITELLM_FIELDS.items():
- current_value: GeneralSettingsUILiteLLMValue = getattr(litellm, litellm_field_name, None)
- default_value = _general_settings_ui_litellm_default(spec)
+ default_value: GeneralSettingsUILiteLLMValue = _general_settings_ui_litellm_default(spec)
+ current_value: GeneralSettingsUILiteLLMValue = cast(
+ GeneralSettingsUILiteLLMValue,
+ litellm_settings_store.get(litellm_field_name, default_value),
+ )
+ source = source_for(litellm_settings_store, litellm_field_name, default_value)
stored_in_db_litellm: bool | None
if litellm_field_name in db_litellm_settings:
stored_in_db_litellm = True
- elif current_value != default_value:
+ elif source == "config":
stored_in_db_litellm = False
else:
stored_in_db_litellm = None
@@ -17744,6 +17763,7 @@ async def get_config_list(
field_description=spec["description"],
field_value=current_value,
stored_in_db=stored_in_db_litellm,
+ source=source,
field_default_value=default_value,
field_options=list(spec.get("options", ())) or None,
field_tab=spec.get("tab"),
diff --git a/litellm/proxy/ui_crud_endpoints/proxy_setting_endpoints.py b/litellm/proxy/ui_crud_endpoints/proxy_setting_endpoints.py
index fd160636d46..4cd031b8780 100644
--- a/litellm/proxy/ui_crud_endpoints/proxy_setting_endpoints.py
+++ b/litellm/proxy/ui_crud_endpoints/proxy_setting_endpoints.py
@@ -14,8 +14,8 @@ from typing import (
from urllib.parse import urlparse
from fastapi import APIRouter, Body, Depends, File, HTTPException, UploadFile
-from pydantic import ConfigDict, JsonValue, TypeAdapter, ValidationError, create_model
-from pydantic.fields import FieldInfo
+from pydantic import BaseModel, ConfigDict, JsonValue, TypeAdapter, ValidationError, create_model
+from pydantic.fields import FieldInfo, PydanticUndefined
from typing_extensions import NotRequired, ReadOnly, TypedDict
import litellm
@@ -24,6 +24,7 @@ from litellm.litellm_core_utils.sensitive_data_masker import mask_sensitive_keys
from litellm.proxy._experimental.mcp_server.tool_search import MCP_TOOL_SEARCH_SETTINGS_KEY
from litellm.proxy._types import *
from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
+from litellm.proxy.config_resolvers import SettingsSource, source_for
from litellm.proxy.config_resolvers.sso import (
SSO_FIELD_ENV_VARS,
SSO_SECRET_FIELDS,
@@ -197,6 +198,11 @@ class SettingsResponse(BaseModel):
"""Schema information including descriptions and property types for UI display"""
+class _SettingsWithSchema(BaseModel):
+ values: dict[str, object]
+ field_schema: dict[str, object]
+
+
class SSOSettingsResponse(SettingsResponse):
"""Response model for SSO settings"""
@@ -327,6 +333,8 @@ class UISettings(BaseModel):
class UISettingsResponse(SettingsResponse):
"""Response model for UI settings"""
+ source: dict[str, SettingsSource]
+
# Allowlist of UI settings that can be stored
ALLOWED_UI_SETTINGS_FIELDS: Final = {
@@ -658,6 +666,13 @@ def _root_schema(settings_class: type[BaseModel]) -> _RootSchema:
)
+def _model_field_default(settings_class: type[BaseModel], field_name: str) -> object:
+ field_info: Final = settings_class.model_fields.get(field_name)
+ if field_info is None or field_info.default is PydanticUndefined:
+ return None
+ return cast(object, field_info.default)
+
+
async def _get_settings_with_schema(
settings_key: str,
settings_class: type[BaseModel],
@@ -1527,7 +1542,7 @@ async def get_ui_settings():
Get UI-specific configuration flags.
All authenticated users can fetch these settings for client-side behavior.
"""
- from litellm.proxy.proxy_server import prisma_client
+ from litellm.proxy.proxy_server import prisma_client, proxy_config
if prisma_client is None:
raise HTTPException(
@@ -1546,26 +1561,46 @@ async def get_ui_settings():
ui_settings: Final = {k: v for k, v in parsed.items() if k in ALLOWED_UI_SETTINGS_FIELDS}
apply_runtime_general_settings_flags(ui_settings)
+ proxy_config.settings.apply_db_row("ui_settings", ui_settings)
# Refresh DualCache so other code paths (e.g. /user/filter/ui) see fresh values
from litellm.proxy.proxy_server import user_api_key_cache
await user_api_key_cache.async_set_cache(key=UI_SETTINGS_CACHE_KEY, value=ui_settings, ttl=UI_SETTINGS_CACHE_TTL)
- # Build config-like object for schema helper
- config: Final[dict[str, object]] = {"litellm_settings": {"ui_settings": ui_settings}}
-
- settings: Final = await _get_settings_with_schema(
- settings_key="ui_settings",
- settings_class=_get_effective_ui_settings_class(),
- config=config,
+ effective_ui_settings: Final = {
+ **{key: proxy_config.settings[key] for key in ALLOWED_UI_SETTINGS_FIELDS if key in proxy_config.settings},
+ **ui_settings,
+ }
+ config: Final[dict[str, object]] = {"litellm_settings": {"ui_settings": effective_ui_settings}}
+ settings_class: Final = _get_effective_ui_settings_class()
+ resolved_settings: Final = _SettingsWithSchema.model_validate(
+ await _get_settings_with_schema(
+ settings_key="ui_settings",
+ settings_class=settings_class,
+ config=config,
+ )
)
+ values: Final = {
+ **resolved_settings.values,
+ ENABLE_PTU_COST_ATTRIBUTION_UI_SETTING: is_ptu_cost_attribution_enabled(),
+ }
+ source: Final[dict[str, SettingsSource]] = {
+ key: (
+ "db"
+ if key in ui_settings
+ else source_for(
+ proxy_config.settings,
+ key,
+ _model_field_default(settings_class, key),
+ )
+ )
+ for key in values
+ }
return UISettingsResponse(
- values={
- **settings["values"],
- ENABLE_PTU_COST_ATTRIBUTION_UI_SETTING: is_ptu_cost_attribution_enabled(),
- },
- field_schema=settings["field_schema"],
+ values=values,
+ field_schema=resolved_settings.field_schema,
+ source=source,
)
diff --git a/tests/test_litellm/proxy/management_endpoints/test_router_settings_endpoints.py b/tests/test_litellm/proxy/management_endpoints/test_router_settings_endpoints.py
index 308f4d88f02..148f30f517b 100644
--- a/tests/test_litellm/proxy/management_endpoints/test_router_settings_endpoints.py
+++ b/tests/test_litellm/proxy/management_endpoints/test_router_settings_endpoints.py
@@ -75,6 +75,36 @@ class TestRouterSettingsEndpoints:
assert isinstance(routing_strategy_field["options"], list)
assert len(routing_strategy_field["options"]) > 0
+ @pytest.mark.asyncio
+ async def test_get_router_settings_reports_sources(self, monkeypatch):
+ from litellm.proxy.config_resolvers import SettingsStore
+
+ store = SettingsStore("router_settings")
+ store.load_yaml({"routing_strategy": "simple-shuffle"})
+ store.apply_db_row("router_settings", {"num_retries": 3})
+ monkeypatch.setattr(proxy_server.proxy_config, "router_settings", store)
+ monkeypatch.setattr(proxy_server, "llm_router", None)
+
+ async def fake_get_config(self, config_file_path=None):
+ return {
+ "router_settings": {
+ "routing_strategy": "simple-shuffle",
+ "num_retries": 3,
+ }
+ }
+
+ monkeypatch.setattr(
+ proxy_server.ProxyConfig, "get_config", fake_get_config, raising=True
+ )
+
+ admin_user = UserAPIKeyAuth(
+ user_role=LitellmUserRoles.PROXY_ADMIN, api_key="sk-x"
+ )
+ response = await get_router_settings(user_api_key_dict=admin_user)
+
+ assert response.source["routing_strategy"] == "config"
+ assert response.source["num_retries"] == "db"
+
@pytest.mark.asyncio
async def test_get_router_settings_includes_routing_groups_from_live_router(
self, monkeypatch
diff --git a/tests/test_litellm/proxy/proxy_server/test_routes_config.py b/tests/test_litellm/proxy/proxy_server/test_routes_config.py
index dd3914e3ad5..d6c63ec9a78 100644
--- a/tests/test_litellm/proxy/proxy_server/test_routes_config.py
+++ b/tests/test_litellm/proxy/proxy_server/test_routes_config.py
@@ -15,10 +15,15 @@ from __future__ import annotations
import asyncio
import json
+from collections.abc import Mapping
+from typing import Final
from unittest.mock import AsyncMock, MagicMock
import pytest
+from litellm.proxy.config_resolvers import SettingsStore
+from litellm.proxy.config_resolvers.settings_rules import JsonValue
+
from .conftest import VOLATILE_KEYS, normalize
@@ -37,6 +42,21 @@ def _install_litellm_config(mock_prisma: MagicMock) -> MagicMock:
return table
+def _install_settings_store(
+ monkeypatch: pytest.MonkeyPatch,
+ config_values: Mapping[str, JsonValue],
+ db_values: Mapping[str, JsonValue],
+) -> SettingsStore:
+ from litellm.proxy import proxy_server
+
+ store: Final = SettingsStore("general_settings")
+ store.load_yaml(config_values)
+ store.apply_db_row("general_settings", db_values)
+ monkeypatch.setattr(proxy_server.proxy_config, "settings", store)
+ monkeypatch.setattr(proxy_server, "general_settings", store)
+ return store
+
+
# ---------------------------------------------------------------------------
# POST /config/update
# ---------------------------------------------------------------------------
@@ -338,6 +358,7 @@ def test_config_field_info_happy_admin(client, auth_as, mock_prisma, monkeypatch
assert normalize(response.json()) == {
"field_name": "max_parallel_requests",
"field_value": 7,
+ "source": "db",
}
@@ -566,6 +587,122 @@ def test_config_list_happy_admin(client, auth_as, mock_prisma, monkeypatch):
}
+def test_config_read_routes_report_effective_values_and_sources(client, auth_as, mock_prisma, monkeypatch):
+ from litellm.proxy import proxy_server as ps
+ from litellm.proxy._types import LitellmUserRoles
+
+ table = _install_litellm_config(mock_prisma)
+ row = MagicMock()
+ row.param_value = {"max_parallel_requests": 7, "max_file_size_mb": 222}
+ table.find_first = AsyncMock(return_value=row)
+ monkeypatch.setattr(ps, "prisma_client", mock_prisma)
+ _install_settings_store(
+ monkeypatch,
+ {
+ "max_parallel_requests": 5,
+ "max_file_size_mb": 111,
+ "pass_through_endpoints": [{"path": "/synthetic"}],
+ },
+ {"max_parallel_requests": 7, "max_file_size_mb": 222},
+ )
+
+ with auth_as(LitellmUserRoles.PROXY_ADMIN):
+ list_response = client.get("/config/list", params={"config_type": "general_settings"})
+ config_only_response = client.get(
+ "/config/field/info", params={"field_name": "max_file_size_mb"}
+ )
+ db_wins_response = client.get(
+ "/config/field/info", params={"field_name": "max_parallel_requests"}
+ )
+
+ assert list_response.status_code == 200
+ by_name: Final = {entry["field_name"]: entry for entry in list_response.json()}
+ assert by_name["max_file_size_mb"]["field_value"] == 111
+ assert by_name["max_file_size_mb"]["source"] == "config"
+ assert by_name["pass_through_endpoints"]["source"] == "config"
+ assert by_name["pass_through_endpoints"]["nested_fields"][0]["source"] == "config"
+ assert by_name["max_parallel_requests"]["field_value"] == 7
+ assert by_name["max_parallel_requests"]["source"] == "db"
+
+ assert config_only_response.status_code == 200
+ assert config_only_response.json() == {
+ "field_name": "max_file_size_mb",
+ "field_value": 111,
+ "source": "config",
+ }
+ assert db_wins_response.status_code == 200
+ assert db_wins_response.json() == {
+ "field_name": "max_parallel_requests",
+ "field_value": 7,
+ "source": "db",
+ }
+
+
+def test_config_read_routes_report_default_source(client, auth_as, mock_prisma, monkeypatch):
+ from litellm.proxy import proxy_server as ps
+ from litellm.proxy._types import LitellmUserRoles
+
+ table = _install_litellm_config(mock_prisma)
+ row = MagicMock()
+ row.param_value = {}
+ table.find_first = AsyncMock(return_value=row)
+ monkeypatch.setattr(ps, "prisma_client", mock_prisma)
+ _install_settings_store(monkeypatch, {}, {})
+
+ with auth_as(LitellmUserRoles.PROXY_ADMIN):
+ list_response = client.get("/config/list", params={"config_type": "general_settings"})
+ field_response = client.get(
+ "/config/field/info", params={"field_name": "proxy_config_reload_interval_seconds"}
+ )
+
+ assert list_response.status_code == 200
+ by_name: Final = {entry["field_name"]: entry for entry in list_response.json()}
+ assert by_name["proxy_config_reload_interval_seconds"]["field_value"] == 30
+ assert by_name["proxy_config_reload_interval_seconds"]["source"] == "default"
+ assert field_response.status_code == 200
+ assert field_response.json() == {
+ "field_name": "proxy_config_reload_interval_seconds",
+ "field_value": 30,
+ "source": "default",
+ }
+
+
+def test_config_field_info_uses_store_without_db(client, auth_as, monkeypatch):
+ from litellm.proxy import proxy_server as ps
+ from litellm.proxy._types import LitellmUserRoles
+
+ monkeypatch.setattr(ps, "prisma_client", None)
+ _install_settings_store(monkeypatch, {"max_file_size_mb": 111}, {})
+
+ with auth_as(LitellmUserRoles.PROXY_ADMIN):
+ response = client.get("/config/field/info", params={"field_name": "max_file_size_mb"})
+
+ assert response.status_code == 200
+ assert response.json() == {
+ "field_name": "max_file_size_mb",
+ "field_value": 111,
+ "source": "config",
+ }
+
+
+def test_config_field_info_unset_source_remains_an_error(client, auth_as, mock_prisma, monkeypatch):
+ from litellm.proxy import proxy_server as ps
+ from litellm.proxy._types import LitellmUserRoles
+
+ table = _install_litellm_config(mock_prisma)
+ row = MagicMock()
+ row.param_value = {}
+ table.find_first = AsyncMock(return_value=row)
+ monkeypatch.setattr(ps, "prisma_client", mock_prisma)
+ _install_settings_store(monkeypatch, {}, {})
+
+ with auth_as(LitellmUserRoles.PROXY_ADMIN):
+ response = client.get("/config/field/info", params={"field_name": "max_parallel_requests"})
+
+ assert response.status_code == 400
+ assert "not in" in response.json()["detail"]["error"]
+
+
def test_config_list_exposes_config_reload_interval(client, auth_as, mock_prisma, monkeypatch):
"""proxy_config_reload_interval_seconds must surface in the admin UI general-settings
list as an Integer field defaulting to 30, so operators can tune multi-pod convergence
diff --git a/tests/test_litellm/proxy/proxy_server/test_routes_model_metrics.py b/tests/test_litellm/proxy/proxy_server/test_routes_model_metrics.py
index 246e2cbba54..6a57ad1636d 100644
--- a/tests/test_litellm/proxy/proxy_server/test_routes_model_metrics.py
+++ b/tests/test_litellm/proxy/proxy_server/test_routes_model_metrics.py
@@ -179,6 +179,44 @@ def test_model_settings_method_not_allowed(client, auth_as):
# ---------------------------------------------------------------------------
+def test_alerting_settings_reports_sources(client, auth_as, monkeypatch):
+ from litellm.proxy.config_resolvers import SettingsStore
+
+ pc = MagicMock()
+ row = MagicMock()
+ row.param_value = {"alerting_args": {"daily_report_frequency": 7}}
+ pc.db.litellm_config.find_first = AsyncMock(return_value=row)
+ monkeypatch.setattr(proxy_server, "prisma_client", pc)
+
+ logging_obj = MagicMock()
+ args_model = MagicMock()
+ args_model.model_dump = MagicMock(return_value={"daily_report_frequency": 7})
+ logging_obj.slack_alerting_instance.alerting_args = args_model
+ monkeypatch.setattr(proxy_server, "proxy_logging_obj", logging_obj)
+
+ store = SettingsStore("general_settings")
+ store.load_yaml(
+ {
+ "alerting": ["slack"],
+ "alerting_args": {"daily_report_frequency": 3},
+ }
+ )
+ store.apply_db_row(
+ "general_settings",
+ {"alerting_args": {"daily_report_frequency": 7}},
+ )
+ monkeypatch.setattr(proxy_server.proxy_config, "settings", store)
+ monkeypatch.setattr(proxy_server, "general_settings", store)
+
+ with auth_as(LitellmUserRoles.PROXY_ADMIN):
+ response = client.get("/alerting/settings")
+
+ assert response.status_code == 200
+ by_name = {entry["field_name"]: entry for entry in response.json()}
+ assert by_name["slack_alerting"]["source"] == "config"
+ assert by_name["daily_report_frequency"]["source"] == "db"
+
+
def test_alerting_settings_no_db_error(client, auth_as, no_prisma):
"""Pins ``GET /alerting/settings`` (error: db not connected)."""
with auth_as(LitellmUserRoles.PROXY_ADMIN):
diff --git a/tests/test_litellm/proxy/ui_crud_endpoints/test_proxy_setting_endpoints.py b/tests/test_litellm/proxy/ui_crud_endpoints/test_proxy_setting_endpoints.py
index 93fb54f84eb..faf5b336410 100644
--- a/tests/test_litellm/proxy/ui_crud_endpoints/test_proxy_setting_endpoints.py
+++ b/tests/test_litellm/proxy/ui_crud_endpoints/test_proxy_setting_endpoints.py
@@ -1342,6 +1342,45 @@ class TestProxySettingEndpoints:
where={"id": "ui_settings"}
)
+ def test_get_ui_settings_reports_sources(self, monkeypatch):
+ from unittest.mock import AsyncMock, MagicMock
+
+ from litellm.proxy import proxy_server
+ from litellm.proxy.config_resolvers import SettingsStore
+
+ mock_prisma = MagicMock()
+ mock_db_record = MagicMock()
+ mock_db_record.ui_settings = {
+ "disable_model_add_for_internal_users": True,
+ }
+ mock_prisma.db.litellm_uisettings.find_unique = AsyncMock(
+ return_value=mock_db_record
+ )
+ monkeypatch.setattr(proxy_server, "prisma_client", mock_prisma)
+
+ store = SettingsStore("general_settings")
+ store.load_yaml(
+ {
+ "disable_model_add_for_internal_users": False,
+ "forward_client_headers_to_llm_api": True,
+ }
+ )
+ store.apply_db_row(
+ "ui_settings",
+ {"disable_model_add_for_internal_users": True},
+ )
+ monkeypatch.setattr(proxy_server.proxy_config, "settings", store)
+ monkeypatch.setattr(proxy_server, "general_settings", store)
+
+ response = client.get("/get/ui_settings")
+
+ assert response.status_code == 200
+ data = response.json()
+ assert data["values"]["disable_model_add_for_internal_users"] is True
+ assert data["values"]["forward_client_headers_to_llm_api"] is True
+ assert data["source"]["disable_model_add_for_internal_users"] == "db"
+ assert data["source"]["forward_client_headers_to_llm_api"] == "config"
+
def test_get_ui_settings_schema_description_preserved_with_extensions(
self, mock_auth, monkeypatch
):
diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts
index fd882937e79..790ddb93546 100644
--- a/ui/litellm-dashboard/src/lib/http/schema.d.ts
+++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts
@@ -26393,6 +26393,12 @@ export interface components {
field_name: string;
/** Field Value */
field_value: unknown;
+ /**
+ * Source
+ * @default unset
+ * @enum {string}
+ */
+ source: "config" | "db" | "default" | "unset";
};
/** ConfigFieldUpdate */
ConfigFieldUpdate: {
@@ -26895,6 +26901,12 @@ export interface components {
* @default false
*/
premium_field: boolean;
+ /**
+ * Source
+ * @default unset
+ * @enum {string}
+ */
+ source: "config" | "db" | "default" | "unset";
/** Stored In Db */
stored_in_db: boolean | null;
};
@@ -28286,6 +28298,12 @@ export interface components {
field_name: string;
/** Field Type */
field_type: string;
+ /**
+ * Source
+ * @default unset
+ * @enum {string}
+ */
+ source: "config" | "db" | "default" | "unset";
/** Stored In Db */
stored_in_db: boolean | null;
};
@@ -36439,6 +36457,13 @@ export interface components {
routing_strategy_descriptions: {
[key: string]: string;
};
+ /**
+ * Source
+ * @description Source of each current router setting
+ */
+ source: {
+ [key: string]: "config" | "db" | "default" | "unset";
+ };
};
/**
* RoutingGroup
@@ -38856,6 +38881,10 @@ export interface components {
field_schema: {
[key: string]: unknown;
};
+ /** Source */
+ source: {
+ [key: string]: "config" | "db" | "default" | "unset";
+ };
/** Values */
values: {
[key: string]: unknown;
From 8dcf9e8b78509e0392eaef34abb43d2754e2a43f Mon Sep 17 00:00:00 2001
From: Yuneng Jiang
Date: Fri, 18 Sep 2026 01:26:10 -0700
Subject: [PATCH 026/160] fix(proxy): correct settings source provenance
---
litellm/proxy/_lazy_openapi_snapshot.json | 2 +-
.../proxy/config_resolvers/settings_store.py | 20 ++++++
.../router_settings_endpoints.py | 25 ++++++-
litellm/proxy/proxy_server.py | 65 +++++++++++++-----
.../proxy_setting_endpoints.py | 30 ++++++---
.../config_resolvers/test_settings_store.py | 33 +++++++++
.../test_router_settings_endpoints.py | 2 +
.../proxy/proxy_server/test_routes_config.py | 44 +++++++++---
.../proxy_server/test_routes_model_metrics.py | 67 ++++++++++++++++---
.../test_proxy_setting_endpoints.py | 42 ++++++++++++
10 files changed, 280 insertions(+), 50 deletions(-)
diff --git a/litellm/proxy/_lazy_openapi_snapshot.json b/litellm/proxy/_lazy_openapi_snapshot.json
index b244678e201..fa046ef0a72 100644
--- a/litellm/proxy/_lazy_openapi_snapshot.json
+++ b/litellm/proxy/_lazy_openapi_snapshot.json
@@ -19346,7 +19346,7 @@
}
}
},
- "description": "\n Unified rate-limit error.\n\n Every rate-limit condition surfaced by litellm \u2014 whether it originated from\n an upstream LLM provider, a vendor batch endpoint, or one of litellm's own\n proxy-side limiters (parallel-requests, dynamic-rate, batch-rate, budget,\n max-iterations, etc.) \u2014 is raised as an instance of this class.\n\n The :attr:`category` attribute lets callers distinguish the source. See\n :class:`RateLimitErrorCategory` for the available values.\n "
+ "description": "\nUnified rate-limit error.\n\nEvery rate-limit condition surfaced by litellm \u2014 whether it originated from\nan upstream LLM provider, a vendor batch endpoint, or one of litellm's own\nproxy-side limiters (parallel-requests, dynamic-rate, batch-rate, budget,\nmax-iterations, etc.) \u2014 is raised as an instance of this class.\n\nThe :attr:`category` attribute lets callers distinguish the source. See\n:class:`RateLimitErrorCategory` for the available values.\n"
},
"500": {
"content": {
diff --git a/litellm/proxy/config_resolvers/settings_store.py b/litellm/proxy/config_resolvers/settings_store.py
index 8f400853fa9..73bca3222ea 100644
--- a/litellm/proxy/config_resolvers/settings_store.py
+++ b/litellm/proxy/config_resolvers/settings_store.py
@@ -28,6 +28,7 @@ class SettingsStore(MutableMapping[str, JsonValue]):
self._yaml_values: Mapping[str, JsonValue] = _EMPTY_VALUES
self._database_rows: Mapping[DbRow, Mapping[str, JsonValue]] = _EMPTY_ROWS
self._runtime_values: Mapping[str, JsonValue] = _EMPTY_VALUES
+ self._runtime_sources: Mapping[str, FieldSource] = MappingProxyType({})
self._deleted_runtime_keys: frozenset[str] = frozenset()
def load_yaml(self, mapping: Mapping[str, JsonValue]) -> None:
@@ -39,11 +40,22 @@ class SettingsStore(MutableMapping[str, JsonValue]):
self._database_rows = MappingProxyType({**self._database_rows, row: MappingProxyType(dict(db_row))})
self._clear_runtime_keys(frozenset((*previous_row, *db_row)))
+ def without_db(self) -> SettingsStore:
+ copy: Final = SettingsStore(self._section)
+ copy.load_yaml(self._yaml_values)
+ runtime_values: Final = {
+ key: value for key, value in self._runtime_values.items() if self._runtime_sources.get(key) != "db"
+ }
+ copy.apply_runtime_values(runtime_values)
+ copy._deleted_runtime_keys = self._deleted_runtime_keys
+ return copy
+
def resolved(self) -> Mapping[str, JsonValue]:
return MappingProxyType(dict(self))
def apply_runtime_values(self, values: Mapping[str, JsonValue]) -> None:
self._runtime_values = MappingProxyType(dict(values))
+ self._runtime_sources = MappingProxyType({key: self.source(key) for key in values})
self._deleted_runtime_keys = frozenset()
def source(self, key: str) -> FieldSource:
@@ -61,6 +73,7 @@ class SettingsStore(MutableMapping[str, JsonValue]):
def __setitem__(self, key: str, value: JsonValue) -> None:
self._runtime_values = MappingProxyType({**self._runtime_values, key: value})
+ self._runtime_sources = MappingProxyType({**self._runtime_sources, key: self.source(key)})
self._deleted_runtime_keys = self._deleted_runtime_keys - frozenset((key,))
def __delitem__(self, key: str) -> None:
@@ -69,6 +82,9 @@ class SettingsStore(MutableMapping[str, JsonValue]):
self._runtime_values = MappingProxyType(
{key_: value for key_, value in self._runtime_values.items() if key_ != key}
)
+ self._runtime_sources = MappingProxyType(
+ {key_: source for key_, source in self._runtime_sources.items() if key_ != key}
+ )
self._deleted_runtime_keys = self._deleted_runtime_keys | frozenset((key,))
def __iter__(self) -> Iterator[str]:
@@ -84,6 +100,7 @@ class SettingsStore(MutableMapping[str, JsonValue]):
def _clear_runtime(self) -> None:
self._runtime_values = _EMPTY_VALUES
+ self._runtime_sources = MappingProxyType({})
self._deleted_runtime_keys = frozenset()
def _clear_runtime_keys(self, keys: frozenset[str]) -> None:
@@ -92,6 +109,9 @@ class SettingsStore(MutableMapping[str, JsonValue]):
self._runtime_values = MappingProxyType(
{key: value for key, value in self._runtime_values.items() if key not in keys}
)
+ self._runtime_sources = MappingProxyType(
+ {key: source for key, source in self._runtime_sources.items() if key not in keys}
+ )
self._deleted_runtime_keys = self._deleted_runtime_keys - keys
def _keys(self) -> tuple[str, ...]:
diff --git a/litellm/proxy/management_endpoints/router_settings_endpoints.py b/litellm/proxy/management_endpoints/router_settings_endpoints.py
index 5d3b6d40601..1869be88d1b 100644
--- a/litellm/proxy/management_endpoints/router_settings_endpoints.py
+++ b/litellm/proxy/management_endpoints/router_settings_endpoints.py
@@ -16,7 +16,7 @@ from pydantic import BaseModel, Field
from litellm._logging import verbose_proxy_logger
from litellm.proxy._types import UserAPIKeyAuth
from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
-from litellm.proxy.config_resolvers import SettingsSource, source_for
+from litellm.proxy.config_resolvers import SettingsSource, SettingsStore, source_for
from litellm.router import Router
from litellm.types.management_endpoints import (
ROUTER_SETTINGS_FIELDS,
@@ -41,6 +41,18 @@ class RouterFieldsResponse(BaseModel):
routing_strategy_descriptions: dict[str, str] = Field(description="Descriptions for each routing strategy option")
+def _router_setting_source(
+ settings: SettingsStore,
+ key: str,
+ current_value: object,
+ field_default: object,
+) -> SettingsSource:
+ source: Final = source_for(settings, key, field_default)
+ if source != "unset":
+ return source
+ return "default" if current_value is not None else "unset"
+
+
def _get_routing_strategies_from_router_class() -> list[str]:
"""
Dynamically extract routing strategies from the Router class __init__ method.
@@ -116,10 +128,17 @@ async def get_router_settings(
field.field_value = current_values[field.field_name]
field_defaults: Final[dict[str, object]] = {
- field.field_name: cast(object, field.field_default) for field in router_fields
+ field.field_name: cast(object, field.field_default) # cast-ok: Pydantic field defaults are untyped
+ for field in router_fields
}
source: Final[dict[str, SettingsSource]] = {
- key: source_for(proxy_config.router_settings, key, field_defaults.get(key)) for key in current_values
+ key: _router_setting_source(
+ proxy_config.router_settings,
+ key,
+ cast(object, current_values[key]), # cast-ok: current values are stored in a typed response map
+ field_defaults.get(key),
+ )
+ for key in current_values
}
return RouterSettingsResponse(
fields=router_fields,
diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py
index 1131d7bea86..3898f522feb 100644
--- a/litellm/proxy/proxy_server.py
+++ b/litellm/proxy/proxy_server.py
@@ -432,7 +432,7 @@ from litellm.proxy.common_utils.user_api_key_cache import (
model_access_group_spend_counter_key,
tag_cache_key,
)
-from litellm.proxy.config_resolvers import SettingsStore, resolve_fields, source_for
+from litellm.proxy.config_resolvers import SettingsSource, SettingsStore, resolve_fields, source_for
from litellm.proxy.config_resolvers.alerting import (
EMAIL_DESCRIPTORS,
MS_TEAMS_DESCRIPTORS,
@@ -4820,7 +4820,7 @@ def _as_settings_mapping(value: object) -> Mapping[str, SettingsJsonValue]:
def _get_field_default(field_info: FieldInfo) -> JsonValue:
if field_info.default is PydanticUndefined:
return None
- return cast(JsonValue, field_info.default)
+ return cast(JsonValue, field_info.default) # cast-ok: Pydantic field defaults are JSON values at runtime
def _bind_general_settings_store(settings: SettingsStore) -> None:
@@ -15605,6 +15605,22 @@ async def model_settings():
#### ALERTING MANAGEMENT ENDPOINTS ####
+def _nested_setting_source(
+ settings: SettingsStore,
+ db_values: Mapping[str, JsonValue],
+ parent_key: str,
+ field_name: str,
+ field_default: JsonValue,
+) -> SettingsSource:
+ db_value: Final = db_values.get(field_name)
+ if db_value is not None and db_value != []:
+ return "db"
+ parent_value: Final = settings.without_db().get(parent_key)
+ if isinstance(parent_value, Mapping) and field_name in parent_value:
+ return "config"
+ return "default" if field_default is not None else "unset"
+
+
@router.get(
"/alerting/settings",
description="Return the configurable alerting param, description, and current value",
@@ -15647,8 +15663,13 @@ async def alerting_settings(
if db_general_settings is not None and db_general_settings.param_value is not None
else {}
)
- alerting_args_dict: Final = cast(dict[str, JsonValue], db_general_settings_dict.get("alerting_args", {}))
- alerting_values: Final = cast(list[JsonValue] | None, db_general_settings_dict.get("alerting"))
+ alerting_args_value: Final = db_general_settings_dict.get("alerting_args")
+ alerting_args_dict: Final[Mapping[str, JsonValue]] = (
+ alerting_args_value if isinstance(alerting_args_value, dict) else {}
+ )
+ alerting_values: Final = cast( # cast-ok: alerting is stored as a JSON list when present
+ list[JsonValue] | None, db_general_settings_dict.get("alerting")
+ )
settings: Final = proxy_config.settings
settings.apply_db_row("general_settings", db_general_settings_dict)
@@ -15711,7 +15732,13 @@ async def alerting_settings(
field_description=field_info.description or "",
field_value=_slack_alerting_args_dict.get(field_name, field_default),
stored_in_db=_stored_in_db,
- source=source_for(settings, "alerting_args", field_default),
+ source=_nested_setting_source(
+ settings,
+ alerting_args_dict,
+ "alerting_args",
+ field_name,
+ field_default,
+ ),
field_default_value=field_default,
premium_field=(True if field_name == "region_outage_alert_ttl" else False),
)
@@ -17414,21 +17441,19 @@ async def get_config_general_settings(
field_info: Final = ConfigGeneralSettings.model_fields[field_name]
field_default: JsonValue = _get_field_default(field_info)
settings: Final = proxy_config.settings
- db_values: Mapping[str, JsonValue]
- if prisma_client is None:
- db_values = {}
- else:
+ if prisma_client is not None:
db_general_settings: Final[_ConfigParamRow | None] = await _config_param_table(prisma_client).find_first(
where={"param_name": "general_settings"}
)
- db_values = (
+ db_values: Final[Mapping[str, JsonValue]] = (
dict(db_general_settings.param_value)
if db_general_settings is not None and db_general_settings.param_value is not None
else {}
)
settings.apply_db_row("general_settings", db_values)
+ effective_settings: Final = settings.without_db() if prisma_client is None else settings
- if field_name not in settings and field_default is None:
+ if field_name not in effective_settings and field_default is None:
raise HTTPException(
status_code=400,
detail={"error": f"Field name={field_name} not in DB"},
@@ -17436,7 +17461,7 @@ async def get_config_general_settings(
redacted_field_value: Final = _redact_general_setting_value(
field_name,
- settings.get(field_name, field_default),
+ effective_settings.get(field_name, field_default),
user_api_key_dict.user_role == LitellmUserRoles.PROXY_ADMIN,
)
field_value: Final = (
@@ -17450,7 +17475,7 @@ async def get_config_general_settings(
return ConfigFieldInfo(
field_name=field_name,
field_value=field_value,
- source=source_for(settings, field_name, field_default),
+ source=source_for(effective_settings, field_name, field_default),
)
@@ -17640,7 +17665,11 @@ async def get_config_list(
settings: Final = proxy_config.settings
settings.apply_db_row("general_settings", db_general_settings_dict)
runtime_settings: Final[Mapping[str, JsonValue]] = (
- cast(Mapping[str, JsonValue], general_settings) if not isinstance(general_settings, SettingsStore) else settings
+ cast( # cast-ok: legacy general_settings remains a mapping at this route boundary
+ Mapping[str, JsonValue], general_settings
+ )
+ if not isinstance(general_settings, SettingsStore)
+ else settings
)
allowed_args: Final = _GENERAL_SETTINGS_CONFIG_LIST_FIELD_TYPES
@@ -17744,9 +17773,11 @@ async def get_config_list(
litellm_settings_store.apply_db_row("litellm_settings", db_litellm_settings)
for litellm_field_name, spec in _GENERAL_SETTINGS_UI_LITELLM_FIELDS.items():
default_value: GeneralSettingsUILiteLLMValue = _general_settings_ui_litellm_default(spec)
- current_value: GeneralSettingsUILiteLLMValue = cast(
- GeneralSettingsUILiteLLMValue,
- litellm_settings_store.get(litellm_field_name, default_value),
+ current_value: GeneralSettingsUILiteLLMValue = (
+ cast( # cast-ok: UI field defaults are validated by the field spec
+ GeneralSettingsUILiteLLMValue,
+ litellm_settings_store.get(litellm_field_name, default_value),
+ )
)
source = source_for(litellm_settings_store, litellm_field_name, default_value)
stored_in_db_litellm: bool | None
diff --git a/litellm/proxy/ui_crud_endpoints/proxy_setting_endpoints.py b/litellm/proxy/ui_crud_endpoints/proxy_setting_endpoints.py
index 4cd031b8780..aba65719f9b 100644
--- a/litellm/proxy/ui_crud_endpoints/proxy_setting_endpoints.py
+++ b/litellm/proxy/ui_crud_endpoints/proxy_setting_endpoints.py
@@ -24,7 +24,7 @@ from litellm.litellm_core_utils.sensitive_data_masker import mask_sensitive_keys
from litellm.proxy._experimental.mcp_server.tool_search import MCP_TOOL_SEARCH_SETTINGS_KEY
from litellm.proxy._types import *
from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
-from litellm.proxy.config_resolvers import SettingsSource, source_for
+from litellm.proxy.config_resolvers import SettingsSource, SettingsStore, source_for
from litellm.proxy.config_resolvers.sso import (
SSO_FIELD_ENV_VARS,
SSO_SECRET_FIELDS,
@@ -34,7 +34,10 @@ from litellm.proxy.management_endpoints.team_admin_field_permissions import (
SUPPORTED_TEAM_ADMIN_EDITABLE_TEAM_FIELDS,
TEAM_ADMIN_EDITABLE_TEAM_FIELDS_SETTING,
)
-from litellm.proxy.spend_tracking.ptu_feature_flag import is_ptu_cost_attribution_enabled
+from litellm.proxy.spend_tracking.ptu_feature_flag import (
+ PTU_COST_ATTRIBUTION_ENV_VAR,
+ is_ptu_cost_attribution_enabled,
+)
from litellm.proxy.utils import invalidate_config_param
from litellm.repositories.config_repository import ConfigRepository
from litellm.repositories.organization_repository import OrganizationRepository
@@ -44,6 +47,7 @@ from litellm.repositories.table_repositories import (
UISettingsRepository,
)
from litellm.repositories.team_repository import TeamRepository
+from litellm.secret_managers.main import get_secret
from litellm.types.mcp import MCPToolSearchSettings
from litellm.types.proxy.management_endpoints.ui_sso import (
DefaultTeamSSOParams,
@@ -670,7 +674,19 @@ def _model_field_default(settings_class: type[BaseModel], field_name: str) -> ob
field_info: Final = settings_class.model_fields.get(field_name)
if field_info is None or field_info.default is PydanticUndefined:
return None
- return cast(object, field_info.default)
+ return cast(object, field_info.default) # cast-ok: Pydantic field defaults are untyped
+
+
+def _ui_setting_source(
+ key: str,
+ value: object,
+ settings: SettingsStore,
+ settings_class: type[BaseModel],
+) -> SettingsSource:
+ if key == ENABLE_PTU_COST_ATTRIBUTION_UI_SETTING:
+ configured_value: Final = get_secret(PTU_COST_ATTRIBUTION_ENV_VAR, None)
+ return "config" if configured_value is not None or value is True else "default"
+ return source_for(settings, key, _model_field_default(settings_class, key))
async def _get_settings_with_schema(
@@ -1587,13 +1603,7 @@ async def get_ui_settings():
}
source: Final[dict[str, SettingsSource]] = {
key: (
- "db"
- if key in ui_settings
- else source_for(
- proxy_config.settings,
- key,
- _model_field_default(settings_class, key),
- )
+ "db" if key in ui_settings else _ui_setting_source(key, values[key], proxy_config.settings, settings_class)
)
for key in values
}
diff --git a/tests/test_litellm/proxy/config_resolvers/test_settings_store.py b/tests/test_litellm/proxy/config_resolvers/test_settings_store.py
index e41c852eacb..efff9ad24c8 100644
--- a/tests/test_litellm/proxy/config_resolvers/test_settings_store.py
+++ b/tests/test_litellm/proxy/config_resolvers/test_settings_store.py
@@ -82,6 +82,39 @@ def test_settings_store_keeps_unaffected_runtime_values_on_a_db_row_refresh() ->
assert store.source("changed") == "db"
+def test_settings_store_without_db_uses_yaml_without_mutating_runtime_values() -> None:
+ store: Final = SettingsStore("general_settings")
+ store.load_yaml({"max_parallel_requests": 5})
+ store.apply_db_row("general_settings", {"max_parallel_requests": 7})
+ store.apply_runtime_values({"max_parallel_requests": 7})
+
+ without_db: Final = store.without_db()
+
+ assert without_db["max_parallel_requests"] == 5
+ assert without_db.source("max_parallel_requests") == "config"
+ assert store["max_parallel_requests"] == 7
+ assert store.source("max_parallel_requests") == "db"
+
+
+def test_settings_store_without_db_preserves_non_db_runtime_values() -> None:
+ store: Final = SettingsStore("general_settings")
+ store.load_yaml({"max_parallel_requests": "os.environ/MAX_PARALLEL_REQUESTS"})
+ store.apply_runtime_values({"max_parallel_requests": 7})
+
+ without_db: Final = store.without_db()
+
+ assert without_db["max_parallel_requests"] == 7
+ assert without_db.source("max_parallel_requests") == "config"
+
+
+def test_settings_store_without_db_preserves_runtime_deletions() -> None:
+ store: Final = SettingsStore("general_settings")
+ store.load_yaml({"deleted": 1})
+ del store["deleted"]
+
+ assert "deleted" not in store.without_db()
+
+
def test_settings_store_removes_only_runtime_values_affected_by_a_cleared_db_row() -> None:
store: Final = SettingsStore("general_settings")
store.load_yaml({"template": "os.environ/SETTING"})
diff --git a/tests/test_litellm/proxy/management_endpoints/test_router_settings_endpoints.py b/tests/test_litellm/proxy/management_endpoints/test_router_settings_endpoints.py
index 148f30f517b..3af7de62abe 100644
--- a/tests/test_litellm/proxy/management_endpoints/test_router_settings_endpoints.py
+++ b/tests/test_litellm/proxy/management_endpoints/test_router_settings_endpoints.py
@@ -146,6 +146,8 @@ class TestRouterSettingsEndpoints:
response = await get_router_settings(user_api_key_dict=admin_user)
assert response.current_values.get("routing_groups") == groups
+ assert response.current_values["timeout"] is not None
+ assert response.source["timeout"] == "default"
rg_field = next(f for f in response.fields if f.field_name == "routing_groups")
assert rg_field.field_value == groups
diff --git a/tests/test_litellm/proxy/proxy_server/test_routes_config.py b/tests/test_litellm/proxy/proxy_server/test_routes_config.py
index d6c63ec9a78..fdaa2476219 100644
--- a/tests/test_litellm/proxy/proxy_server/test_routes_config.py
+++ b/tests/test_litellm/proxy/proxy_server/test_routes_config.py
@@ -15,8 +15,11 @@ from __future__ import annotations
import asyncio
import json
-from collections.abc import Mapping
+from collections.abc import Callable, Mapping
+from contextlib import AbstractContextManager
from typing import Final
+
+from fastapi.testclient import TestClient
from unittest.mock import AsyncMock, MagicMock
import pytest
@@ -362,6 +365,33 @@ def test_config_field_info_happy_admin(client, auth_as, mock_prisma, monkeypatch
}
+def test_config_field_info_clears_stale_db_source_without_connection(
+ client: TestClient,
+ auth_as: Callable[..., AbstractContextManager[None]],
+ monkeypatch: pytest.MonkeyPatch,
+):
+ from litellm.proxy import proxy_server as ps
+ from litellm.proxy._types import LitellmUserRoles
+
+ store = SettingsStore("general_settings")
+ store.load_yaml({"max_parallel_requests": 5})
+ store.apply_db_row("general_settings", {"max_parallel_requests": 7})
+ store.apply_runtime_values({"max_parallel_requests": 7})
+ monkeypatch.setattr(ps.proxy_config, "settings", store)
+ monkeypatch.setattr(ps, "prisma_client", None)
+
+ with auth_as(LitellmUserRoles.PROXY_ADMIN):
+ response = client.get("/config/field/info", params={"field_name": "max_parallel_requests"})
+
+ assert response.status_code == 200
+ assert normalize(response.json()) == {
+ "field_name": "max_parallel_requests",
+ "field_value": 5,
+ "source": "config",
+ }
+ assert store["max_parallel_requests"] == 7
+
+
def test_config_field_info_non_admin_rejected(client, auth_as, mock_prisma, monkeypatch):
"""Non-admin (INTERNAL_USER) is denied — admin-view gate fires."""
from litellm.proxy import proxy_server as ps
@@ -608,12 +638,8 @@ def test_config_read_routes_report_effective_values_and_sources(client, auth_as,
with auth_as(LitellmUserRoles.PROXY_ADMIN):
list_response = client.get("/config/list", params={"config_type": "general_settings"})
- config_only_response = client.get(
- "/config/field/info", params={"field_name": "max_file_size_mb"}
- )
- db_wins_response = client.get(
- "/config/field/info", params={"field_name": "max_parallel_requests"}
- )
+ config_only_response = client.get("/config/field/info", params={"field_name": "max_file_size_mb"})
+ db_wins_response = client.get("/config/field/info", params={"field_name": "max_parallel_requests"})
assert list_response.status_code == 200
by_name: Final = {entry["field_name"]: entry for entry in list_response.json()}
@@ -651,9 +677,7 @@ def test_config_read_routes_report_default_source(client, auth_as, mock_prisma,
with auth_as(LitellmUserRoles.PROXY_ADMIN):
list_response = client.get("/config/list", params={"config_type": "general_settings"})
- field_response = client.get(
- "/config/field/info", params={"field_name": "proxy_config_reload_interval_seconds"}
- )
+ field_response = client.get("/config/field/info", params={"field_name": "proxy_config_reload_interval_seconds"})
assert list_response.status_code == 200
by_name: Final = {entry["field_name"]: entry for entry in list_response.json()}
diff --git a/tests/test_litellm/proxy/proxy_server/test_routes_model_metrics.py b/tests/test_litellm/proxy/proxy_server/test_routes_model_metrics.py
index 6a57ad1636d..97ca3b3dbbe 100644
--- a/tests/test_litellm/proxy/proxy_server/test_routes_model_metrics.py
+++ b/tests/test_litellm/proxy/proxy_server/test_routes_model_metrics.py
@@ -11,13 +11,17 @@ Pins (PR2):
from __future__ import annotations
+from collections.abc import Callable
+from contextlib import AbstractContextManager
from unittest.mock import AsyncMock, MagicMock
import pytest
+from fastapi.testclient import TestClient
import litellm
from litellm.proxy import proxy_server
from litellm.proxy._types import LitellmUserRoles
+from litellm.proxy.config_resolvers.settings_rules import JsonValue
from .conftest import normalize # type: ignore[import-not-found]
@@ -53,9 +57,7 @@ def test_model_streaming_metrics_happy(client, auth_as, prisma_with_query_raw):
pin can rely on the exact response shape.
"""
with auth_as():
- response = client.get(
- "/model/streaming_metrics", params={"_selected_model_group": "gpt-4"}
- )
+ response = client.get("/model/streaming_metrics", params={"_selected_model_group": "gpt-4"})
assert response.status_code == 200
assert normalize(response.json()) == {"data": [], "all_api_bases": []}
@@ -94,9 +96,7 @@ def test_model_metrics_no_prisma_error(client, auth_as, no_prisma):
# ---------------------------------------------------------------------------
-def test_model_metrics_slow_responses_happy(
- client, auth_as, prisma_with_query_raw, monkeypatch
-):
+def test_model_metrics_slow_responses_happy(client, auth_as, prisma_with_query_raw, monkeypatch):
"""Pins ``GET /model/metrics/slow_responses`` (happy: empty list)."""
logging_obj = MagicMock()
logging_obj.slack_alerting_instance.alerting_threshold = 30
@@ -184,7 +184,12 @@ def test_alerting_settings_reports_sources(client, auth_as, monkeypatch):
pc = MagicMock()
row = MagicMock()
- row.param_value = {"alerting_args": {"daily_report_frequency": 7}}
+ row.param_value = {
+ "alerting_args": {
+ "daily_report_frequency": 7,
+ "report_check_interval": None,
+ }
+ }
pc.db.litellm_config.find_first = AsyncMock(return_value=row)
monkeypatch.setattr(proxy_server, "prisma_client", pc)
@@ -198,12 +203,20 @@ def test_alerting_settings_reports_sources(client, auth_as, monkeypatch):
store.load_yaml(
{
"alerting": ["slack"],
- "alerting_args": {"daily_report_frequency": 3},
+ "alerting_args": {
+ "daily_report_frequency": 3,
+ "report_check_interval": 300,
+ },
}
)
store.apply_db_row(
"general_settings",
- {"alerting_args": {"daily_report_frequency": 7}},
+ {
+ "alerting_args": {
+ "daily_report_frequency": 7,
+ "report_check_interval": None,
+ }
+ },
)
monkeypatch.setattr(proxy_server.proxy_config, "settings", store)
monkeypatch.setattr(proxy_server, "general_settings", store)
@@ -215,6 +228,42 @@ def test_alerting_settings_reports_sources(client, auth_as, monkeypatch):
by_name = {entry["field_name"]: entry for entry in response.json()}
assert by_name["slack_alerting"]["source"] == "config"
assert by_name["daily_report_frequency"]["source"] == "db"
+ assert by_name["report_check_interval"]["source"] == "config"
+ assert by_name["budget_alert_ttl"]["source"] == "default"
+
+
+@pytest.mark.parametrize("db_alerting_args", [None, []])
+def test_alerting_settings_handles_empty_db_args(
+ client: TestClient,
+ auth_as: Callable[..., AbstractContextManager[None]],
+ monkeypatch: pytest.MonkeyPatch,
+ db_alerting_args: JsonValue,
+):
+ from litellm.proxy.config_resolvers import SettingsStore
+
+ pc = MagicMock()
+ row = MagicMock()
+ row.param_value = {"alerting_args": db_alerting_args}
+ pc.db.litellm_config.find_first = AsyncMock(return_value=row)
+ monkeypatch.setattr(proxy_server, "prisma_client", pc)
+
+ logging_obj = MagicMock()
+ args_model = MagicMock()
+ args_model.model_dump = MagicMock(return_value={})
+ logging_obj.slack_alerting_instance.alerting_args = args_model
+ monkeypatch.setattr(proxy_server, "proxy_logging_obj", logging_obj)
+
+ store = SettingsStore("general_settings")
+ store.load_yaml({"alerting_args": {"report_check_interval": 300}})
+ monkeypatch.setattr(proxy_server.proxy_config, "settings", store)
+ monkeypatch.setattr(proxy_server, "general_settings", store)
+
+ with auth_as(LitellmUserRoles.PROXY_ADMIN):
+ response = client.get("/alerting/settings")
+
+ assert response.status_code == 200
+ by_name = {entry["field_name"]: entry for entry in response.json()}
+ assert by_name["report_check_interval"]["source"] == "config"
def test_alerting_settings_no_db_error(client, auth_as, no_prisma):
diff --git a/tests/test_litellm/proxy/ui_crud_endpoints/test_proxy_setting_endpoints.py b/tests/test_litellm/proxy/ui_crud_endpoints/test_proxy_setting_endpoints.py
index faf5b336410..d8308d7831f 100644
--- a/tests/test_litellm/proxy/ui_crud_endpoints/test_proxy_setting_endpoints.py
+++ b/tests/test_litellm/proxy/ui_crud_endpoints/test_proxy_setting_endpoints.py
@@ -3198,6 +3198,7 @@ class TestPtuCostAttributionUISetting:
assert response.status_code == 200
assert response.json()["values"]["enable_ptu_cost_attribution"] is False
+ assert response.json()["source"]["enable_ptu_cost_attribution"] == "default"
def test_reported_true_once_the_env_var_is_set(self, mock_auth, monkeypatch):
from litellm.proxy.spend_tracking.ptu_feature_flag import PTU_COST_ATTRIBUTION_ENV_VAR
@@ -3209,6 +3210,47 @@ class TestPtuCostAttributionUISetting:
assert response.status_code == 200
assert response.json()["values"]["enable_ptu_cost_attribution"] is True
+ assert response.json()["source"]["enable_ptu_cost_attribution"] == "config"
+
+ def test_reported_config_when_secret_manager_enables_the_flag(
+ self, mock_auth: None, monkeypatch: pytest.MonkeyPatch
+ ):
+ from litellm.proxy.spend_tracking.ptu_feature_flag import PTU_COST_ATTRIBUTION_ENV_VAR
+
+ monkeypatch.delenv(PTU_COST_ATTRIBUTION_ENV_VAR, raising=False)
+ monkeypatch.setattr(
+ "litellm.proxy.ui_crud_endpoints.proxy_setting_endpoints.is_ptu_cost_attribution_enabled",
+ lambda: True,
+ )
+ self._mock_prisma(monkeypatch)
+
+ response = client.get("/get/ui_settings")
+
+ assert response.status_code == 200
+ assert response.json()["values"]["enable_ptu_cost_attribution"] is True
+ assert response.json()["source"]["enable_ptu_cost_attribution"] == "config"
+
+ def test_reported_config_when_secret_manager_disables_the_flag(
+ self, mock_auth: None, monkeypatch: pytest.MonkeyPatch
+ ):
+ from litellm.proxy.spend_tracking.ptu_feature_flag import PTU_COST_ATTRIBUTION_ENV_VAR
+
+ monkeypatch.delenv(PTU_COST_ATTRIBUTION_ENV_VAR, raising=False)
+ monkeypatch.setattr(
+ "litellm.proxy.ui_crud_endpoints.proxy_setting_endpoints.is_ptu_cost_attribution_enabled",
+ lambda: False,
+ )
+ monkeypatch.setattr(
+ "litellm.proxy.ui_crud_endpoints.proxy_setting_endpoints.get_secret",
+ lambda *_args: False,
+ )
+ self._mock_prisma(monkeypatch)
+
+ response = client.get("/get/ui_settings")
+
+ assert response.status_code == 200
+ assert response.json()["values"]["enable_ptu_cost_attribution"] is False
+ assert response.json()["source"]["enable_ptu_cost_attribution"] == "config"
def test_a_persisted_true_cannot_forge_the_derived_value(self, mock_auth, monkeypatch):
"""A row written before the allowlist existed must not be able to turn the feature on."""
From 066cc1883afe3a69a49638df4c8d7cfbc1fd0f2d Mon Sep 17 00:00:00 2001
From: Yuneng Jiang
Date: Fri, 18 Sep 2026 02:39:58 -0700
Subject: [PATCH 027/160] test(router): cover legacy lowest TPM selection
---
.../router_strategy/test_lowest_tpm_rpm.py | 54 +++++++++++++++++++
1 file changed, 54 insertions(+)
create mode 100644 tests/test_litellm/router_strategy/test_lowest_tpm_rpm.py
diff --git a/tests/test_litellm/router_strategy/test_lowest_tpm_rpm.py b/tests/test_litellm/router_strategy/test_lowest_tpm_rpm.py
new file mode 100644
index 00000000000..7b13b196d5b
--- /dev/null
+++ b/tests/test_litellm/router_strategy/test_lowest_tpm_rpm.py
@@ -0,0 +1,54 @@
+from datetime import datetime, timedelta
+from typing import Final
+
+from litellm import Router
+from litellm.types.router import DeploymentTypedDict, LiteLLMParamsTypedDict
+
+MODEL_GROUP: Final = "lowest-tpm-router"
+HIGH_USAGE_DEPLOYMENT_ID: Final = "highest-usage"
+LOW_USAGE_DEPLOYMENT_ID: Final = "lowest-usage"
+
+
+def _deployment(deployment_id: str) -> DeploymentTypedDict:
+ params: LiteLLMParamsTypedDict = {
+ "model": "gpt-4o",
+ "api_key": "key",
+ "mock_response": f"from {deployment_id}",
+ }
+ return {
+ "model_name": MODEL_GROUP,
+ "litellm_params": params,
+ "model_info": {"id": deployment_id},
+ }
+
+
+def test_usage_based_routing_v1_selects_the_lowest_recorded_tpm() -> None:
+ router: Final = Router(
+ model_list=[
+ _deployment(HIGH_USAGE_DEPLOYMENT_ID),
+ _deployment(LOW_USAGE_DEPLOYMENT_ID),
+ ],
+ routing_strategy="usage-based-routing",
+ num_retries=0,
+ )
+ usage_by_deployment: Final = {
+ HIGH_USAGE_DEPLOYMENT_ID: 100,
+ LOW_USAGE_DEPLOYMENT_ID: 1,
+ }
+ now: Final = datetime.now()
+ cache_keys: Final = tuple(
+ f"{MODEL_GROUP}:tpm:{(now + timedelta(minutes=offset)).strftime('%H-%M')}"
+ for offset in range(60)
+ )
+
+ for cache_key in cache_keys:
+ router.cache.set_cache(
+ key=cache_key, value=usage_by_deployment, ttl=float("inf")
+ )
+
+ deployment: Final = router.get_available_deployment(
+ model=MODEL_GROUP,
+ messages=[{"role": "user", "content": "test"}],
+ )
+
+ assert deployment["model_info"]["id"] == LOW_USAGE_DEPLOYMENT_ID
From da3bd9e31c72e29a2a75ef82107d05e7f91cc4db Mon Sep 17 00:00:00 2001
From: Yuneng Jiang
Date: Fri, 18 Sep 2026 04:08:34 -0700
Subject: [PATCH 028/160] test(ui): model E2E cleanup failures as values
---
tests/e2e/ui/helpers/roundTrip.ts | 111 ++++++++++++++++++++++--------
1 file changed, 81 insertions(+), 30 deletions(-)
diff --git a/tests/e2e/ui/helpers/roundTrip.ts b/tests/e2e/ui/helpers/roundTrip.ts
index 1fc2d0aec1a..91e48b7087c 100644
--- a/tests/e2e/ui/helpers/roundTrip.ts
+++ b/tests/e2e/ui/helpers/roundTrip.ts
@@ -33,39 +33,90 @@ export async function readBack(
return (await res.json()) as T;
}
-export async function runWithCleanup(
- action: () => Promise,
- cleanup: () => Promise,
-): Promise {
- const outcome = await Promise.resolve()
+type OperationOutcome =
+ | { readonly status: "success" }
+ | { readonly status: "failure"; readonly error: unknown };
+
+type RunFailure =
+ | { readonly status: "action_failure"; readonly error: unknown }
+ | { readonly status: "cleanup_failure"; readonly error: unknown }
+ | {
+ readonly status: "action_and_cleanup_failure";
+ readonly actionError: unknown;
+ readonly cleanupError: unknown;
+ };
+
+function toRunFailure(
+ actionOutcome: OperationOutcome,
+ cleanupOutcome: OperationOutcome,
+): RunFailure | null {
+ if (
+ actionOutcome.status === "failure" &&
+ cleanupOutcome.status === "failure"
+ ) {
+ return {
+ status: "action_and_cleanup_failure",
+ actionError: actionOutcome.error,
+ cleanupError: cleanupOutcome.error,
+ };
+ }
+ if (actionOutcome.status === "failure") {
+ return { status: "action_failure", error: actionOutcome.error };
+ }
+ if (cleanupOutcome.status === "failure") {
+ return { status: "cleanup_failure", error: cleanupOutcome.error };
+ }
+ return null;
+}
+
+function raiseRunFailure(failure: RunFailure): never {
+ switch (failure.status) {
+ case "action_failure":
+ throw failure.error;
+ case "cleanup_failure":
+ throw failure.error;
+ case "action_and_cleanup_failure":
+ throw new AggregateError(
+ [failure.actionError, failure.cleanupError],
+ "Action and cleanup failed",
+ );
+ }
+}
+
+async function runAction(
+ action: () => void | Promise,
+): Promise {
+ return Promise.resolve()
.then(action)
.then(
() => ({ status: "success" as const }),
(error: unknown) => ({ status: "failure" as const, error }),
);
- try {
- if (outcome.status === "failure") throw outcome.error;
- } finally {
- const cleanupOutcome = await Promise.resolve()
- .then(cleanup)
- .then(
- (succeeded) =>
- succeeded
- ? { status: "success" as const }
- : {
- status: "failure" as const,
- error: new Error("Failed to clean up UI E2E resource"),
- },
- (error: unknown) => ({ status: "failure" as const, error }),
- );
- if (cleanupOutcome.status === "failure") {
- if (outcome.status === "failure") {
- throw new AggregateError(
- [outcome.error, cleanupOutcome.error],
- "Action and cleanup failed",
- );
- }
- throw cleanupOutcome.error;
- }
- }
+}
+
+async function runCleanup(
+ cleanup: () => boolean | Promise,
+): Promise {
+ return Promise.resolve()
+ .then(cleanup)
+ .then(
+ (succeeded) =>
+ succeeded
+ ? { status: "success" as const }
+ : {
+ status: "failure" as const,
+ error: new Error("Failed to clean up UI E2E resource"),
+ },
+ (error: unknown) => ({ status: "failure" as const, error }),
+ );
+}
+
+export async function runWithCleanup(
+ action: () => void | Promise,
+ cleanup: () => boolean | Promise,
+): Promise {
+ const actionOutcome = await runAction(action);
+ const cleanupOutcome = await runCleanup(cleanup);
+ const failure = toRunFailure(actionOutcome, cleanupOutcome);
+ if (failure !== null) raiseRunFailure(failure);
}
From c1c566db875f31e28e07f662be39207dca9d8344 Mon Sep 17 00:00:00 2001
From: Yucheng He
Date: Tue, 15 Sep 2026 01:18:18 -0700
Subject: [PATCH 029/160] fix(proxy): record aborted outcome when spend-log
cleanup is cancelled at shutdown
cleanup_old_spend_logs only caught Exception, so a run cut short by
CancelledError recorded no outcome and logged nothing. Under uvicorn the
job was never cancelled at all: uvicorn re-raises the captured SIGTERM as
soon as the lifespan shutdown returns, before asyncio cancels outstanding
tasks, so an in-flight scheduler job simply died with the process.
The cleanup now handles CancelledError by logging elapsed time, rows
deleted and batch count at error level, recording outcome="aborted", and
re-raising. The lifespan shutdown stops the scheduler and awaits the jobs
it cancels while the database is still connected, so that handler runs
under uvicorn too, and the pod lock is released instead of orphaned.
Resolves LIT-6990
---
.../db_transaction_queue/spend_log_cleanup.py | 16 +++
litellm/proxy/proxy_server.py | 18 ++-
litellm/proxy/shutdown/scheduled_jobs.py | 70 ++++++++++
.../proxy/shutdown/test_scheduled_jobs.py | 125 ++++++++++++++++++
.../proxy/test_spend_log_cleanup.py | 92 +++++++++++++
5 files changed, 318 insertions(+), 3 deletions(-)
create mode 100644 litellm/proxy/shutdown/scheduled_jobs.py
create mode 100644 tests/test_litellm/proxy/shutdown/test_scheduled_jobs.py
diff --git a/litellm/proxy/db/db_transaction_queue/spend_log_cleanup.py b/litellm/proxy/db/db_transaction_queue/spend_log_cleanup.py
index e97e9f6e683..1a14210dbec 100644
--- a/litellm/proxy/db/db_transaction_queue/spend_log_cleanup.py
+++ b/litellm/proxy/db/db_transaction_queue/spend_log_cleanup.py
@@ -96,6 +96,8 @@ class SpendLogCleanup:
self.general_settings = general_settings or default_settings
self._refresh_bounds()
+ self._run_rows_deleted: int = 0
+ self._run_batches: int = 0
from litellm.proxy.proxy_server import proxy_logging_obj
pod_lock_manager: Final = proxy_logging_obj.db_spend_update_writer.pod_lock_manager
@@ -422,6 +424,8 @@ class SpendLogCleanup:
total_deleted += deleted_count
run_count += 1
+ self._run_rows_deleted += deleted_count
+ self._run_batches += 1
# Add a small sleep to prevent overwhelming the database
await asyncio.sleep(0.1)
@@ -590,6 +594,9 @@ class SpendLogCleanup:
If no pod_lock_manager, runs cleanup without distributed locking.
"""
lock_acquired = False
+ run_started_at: Final = time.monotonic()
+ self._run_rows_deleted = 0
+ self._run_batches = 0
try:
verbose_proxy_logger.info("Cleanup job triggered at %s", datetime.now())
self._refresh_bounds()
@@ -670,6 +677,15 @@ class SpendLogCleanup:
self._run_outcome(spend_log_results + session_results + health_check_results)
)
+ except asyncio.CancelledError:
+ verbose_proxy_logger.error(
+ "Spend log cleanup cancelled after %.2fs (rows_deleted=%d, batches=%d); the next run resumes from here",
+ time.monotonic() - run_started_at,
+ self._run_rows_deleted,
+ self._run_batches,
+ )
+ SpendLogCleanupMetrics.record_run("aborted")
+ raise
except Exception as e:
# .exception() captures the traceback; str(e) alone on a Prisma/DB
# timeout is often empty and gives operators no signal to diagnose.
diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py
index 3b5f4236d22..cab0f4d0733 100644
--- a/litellm/proxy/proxy_server.py
+++ b/litellm/proxy/proxy_server.py
@@ -680,6 +680,10 @@ from litellm.proxy.route_llm_request import route_request
from litellm.proxy.route_priority import hot_routes_first
from litellm.proxy.search_endpoints.endpoints import router as search_router
from litellm.proxy.shutdown.graceful_shutdown_manager import GracefulShutdownManager
+from litellm.proxy.shutdown.scheduled_jobs import (
+ AwaitableAsyncIOExecutor,
+ cancel_in_flight_scheduler_jobs,
+)
from litellm.proxy.spend_tracking.budget_reservation import get_budget_window_start
from litellm.proxy.spend_tracking.spend_counter_batch import (
PendingSpendIncrement,
@@ -1456,6 +1460,13 @@ async def proxy_startup_event(app: FastAPI) -> AsyncGenerator[None, None]:
await proxy_config.stop_auth_cache_invalidation_subscriber()
+ # Shutdown event - cancel and await in-flight scheduled jobs while the DB is still connected
+ if scheduler is not None and scheduler_executor is not None:
+ try:
+ await cancel_in_flight_scheduler_jobs(scheduler, scheduler_executor)
+ except Exception as e:
+ verbose_proxy_logger.error("Error cancelling in-flight scheduled jobs: %s", e)
+
await proxy_shutdown_event(worker_heartbeat=worker_heartbeat)
if prometheus_multiproc_dir:
@@ -2451,6 +2462,7 @@ celery_app_conn: Final = None
celery_fn: Final = None # Redis Queue for handling requests
scheduler = None
+scheduler_executor: AwaitableAsyncIOExecutor | None = None # rebind-ok: bound once the scheduler is built at startup
# Global variable for anthropic beta headers reload scheduling
last_anthropic_beta_headers_reload = None
@@ -9763,7 +9775,7 @@ class ProxyStartupEvent:
proxy_logging_obj: ProxyLogging,
) -> ProxyWorkerHeartbeat:
"""Initializes scheduled background jobs"""
- global heuristic_v1_tuning_baselines, store_model_in_db, scheduler # rebind-ok: startup publishes the one read-only baseline snapshot
+ global heuristic_v1_tuning_baselines, store_model_in_db, scheduler, scheduler_executor # rebind-ok: startup publishes the one read-only baseline snapshot
# MEMORY LEAK FIX: Configure scheduler with optimized settings
# Memray analysis showed APScheduler's normalize() and _apply_jitter() causing
@@ -9772,9 +9784,9 @@ class ProxyStartupEvent:
# 1. Remove/minimize jitter to avoid normalize() memory explosion
# 2. Use larger misfire_grace_time to prevent backlog calculations
# 3. Set replace_existing=True to avoid duplicate jobs
- from apscheduler.executors.asyncio import AsyncIOExecutor
from apscheduler.jobstores.memory import MemoryJobStore
+ scheduler_executor = AwaitableAsyncIOExecutor() # rebind-ok: shutdown awaits the jobs this executor runs
scheduler = AsyncIOScheduler(
job_defaults={
"coalesce": APSCHEDULER_COALESCE,
@@ -9787,7 +9799,7 @@ class ProxyStartupEvent:
jobstores={"default": MemoryJobStore()}, # explicitly use memory job store
# Use simple executor to minimize overhead
executors={
- "default": AsyncIOExecutor(),
+ "default": scheduler_executor,
},
# Disable timezone awareness to reduce computation
timezone=None,
diff --git a/litellm/proxy/shutdown/scheduled_jobs.py b/litellm/proxy/shutdown/scheduled_jobs.py
new file mode 100644
index 00000000000..3c9e791f51c
--- /dev/null
+++ b/litellm/proxy/shutdown/scheduled_jobs.py
@@ -0,0 +1,70 @@
+"""
+Cancel the proxy's in-flight scheduled jobs at shutdown so they can record how they ended.
+
+APScheduler's ``AsyncIOExecutor.shutdown`` cancels the job tasks it has in flight but cannot
+wait for them, because it is not a coroutine, and under uvicorn nothing else ever will: uvicorn
+re-raises the SIGTERM it captured as soon as the ASGI lifespan shutdown returns, so the process
+dies before ``asyncio.run`` reaches its cancel-all-tasks step. A job mid-run at that point is
+killed without ever observing cancellation, which is how a spend-log cleanup interrupted by a
+rolling restart left no outcome metric and no log line behind. Cancelling here and awaiting the
+cancelled tasks while the database is still connected is what lets a job's own
+``CancelledError`` handler run.
+
+The wait is bounded by ``JOB_CANCEL_TIMEOUT_SECONDS``. A job that has just been cancelled has only
+its own cleanup left to do, so the bound is there for a job that swallows cancellation, not one
+that honours it, and it keeps shutdown well inside a Kubernetes termination grace period.
+"""
+
+# pyright: reportMissingTypeStubs=false # apscheduler ships no type information
+
+import asyncio
+from collections.abc import Collection
+from typing import Final, Protocol
+
+from apscheduler.executors.asyncio import AsyncIOExecutor
+
+from litellm._logging import verbose_proxy_logger
+
+JOB_CANCEL_TIMEOUT_SECONDS: Final = 5.0
+
+
+class StoppableScheduler(Protocol):
+ """The slice of ``AsyncIOScheduler`` shutdown uses, which ships no type information"""
+
+ @property
+ def running(self) -> bool: ...
+
+ def shutdown(self, wait: bool = ...) -> None: ...
+
+
+class AwaitableAsyncIOExecutor(AsyncIOExecutor): # pyright: ignore[reportUntypedBaseClass] # apscheduler ships no type information and is absent from the type-check env
+ """``AsyncIOExecutor`` whose in-flight job tasks can be awaited after ``shutdown`` cancels them"""
+
+ _pending_futures: Collection["asyncio.Future[object]"]
+
+ def in_flight_jobs(self) -> tuple["asyncio.Future[object]", ...]:
+ """The job tasks that are running right now, as a snapshot"""
+ return tuple(future for future in self._pending_futures if not future.done())
+
+
+async def cancel_in_flight_scheduler_jobs(scheduler: StoppableScheduler, executor: AwaitableAsyncIOExecutor) -> None:
+ """
+ Stop the scheduler and wait, bounded by JOB_CANCEL_TIMEOUT_SECONDS, for the jobs it cancels.
+
+ Must run before the database is disconnected: a job's cancellation handler is what records
+ the run's outcome, and it needs the connection the job was using.
+ """
+ if not scheduler.running:
+ return
+ in_flight: Final = executor.in_flight_jobs()
+ scheduler.shutdown(wait=False)
+ if not in_flight:
+ return
+ verbose_proxy_logger.info("Cancelling %d in-flight scheduled job(s) for shutdown", len(in_flight))
+ _done, pending = await asyncio.wait(in_flight, timeout=JOB_CANCEL_TIMEOUT_SECONDS)
+ if pending:
+ verbose_proxy_logger.warning(
+ "%d scheduled job(s) did not finish within %ss of cancellation; giving up on them",
+ len(pending),
+ JOB_CANCEL_TIMEOUT_SECONDS,
+ )
diff --git a/tests/test_litellm/proxy/shutdown/test_scheduled_jobs.py b/tests/test_litellm/proxy/shutdown/test_scheduled_jobs.py
new file mode 100644
index 00000000000..b77d7c4ae50
--- /dev/null
+++ b/tests/test_litellm/proxy/shutdown/test_scheduled_jobs.py
@@ -0,0 +1,125 @@
+"""
+Tests for cancelling in-flight scheduled jobs at proxy shutdown.
+
+These drive a real AsyncIOScheduler: the point of the helper is the hand-off
+between APScheduler's fire-and-forget cancellation and the lifespan shutdown
+that has to outlive it, and a mocked scheduler would not exercise that.
+"""
+
+import asyncio
+import logging
+from collections.abc import AsyncIterator
+from contextlib import asynccontextmanager
+from datetime import datetime
+
+import pytest
+from apscheduler.schedulers.asyncio import AsyncIOScheduler
+
+import litellm.proxy.shutdown.scheduled_jobs as scheduled_jobs
+from litellm.proxy.shutdown.scheduled_jobs import (
+ AwaitableAsyncIOExecutor,
+ cancel_in_flight_scheduler_jobs,
+)
+
+
+class _Job:
+ """A scheduled job that blocks until cancelled and records what it observed."""
+
+ def __init__(self, swallow_cancellation: bool = False) -> None:
+ self.started = asyncio.Event()
+ self.events: list[str] = []
+ self.swallow_cancellation = swallow_cancellation
+
+ async def run(self) -> None:
+ self.started.set()
+ try:
+ await asyncio.Event().wait()
+ except asyncio.CancelledError:
+ self.events.append("cancelled")
+ if self.swallow_cancellation:
+ await asyncio.Event().wait()
+ raise
+ finally:
+ self.events.append("finished")
+
+
+@asynccontextmanager
+async def _running_scheduler(*jobs: _Job) -> AsyncIterator[tuple[AsyncIOScheduler, AwaitableAsyncIOExecutor]]:
+ """A started scheduler with every job in flight; stopped on the way out whatever the test did."""
+ executor = AwaitableAsyncIOExecutor()
+ scheduler = AsyncIOScheduler(executors={"default": executor})
+ for index, job in enumerate(jobs):
+ scheduler.add_job(job.run, id=f"job-{index}", next_run_time=datetime.now())
+ scheduler.start()
+ try:
+ for job in jobs:
+ await asyncio.wait_for(job.started.wait(), timeout=5)
+ yield scheduler, executor
+ finally:
+ if scheduler.running:
+ scheduler.shutdown(wait=False)
+ stragglers = executor.in_flight_jobs()
+ for straggler in stragglers:
+ straggler.cancel()
+ await asyncio.gather(*stragglers, return_exceptions=True)
+
+
+@pytest.mark.asyncio
+async def test_in_flight_jobs_observe_cancellation_before_shutdown_returns():
+ """
+ The job's own CancelledError handler is what records how a run ended, so
+ shutdown must not return until that handler has run.
+ """
+ job = _Job()
+ async with _running_scheduler(job) as (scheduler, executor):
+ await cancel_in_flight_scheduler_jobs(scheduler, executor)
+
+ assert job.events == ["cancelled", "finished"]
+ assert scheduler.running is False
+ assert executor.in_flight_jobs() == ()
+
+
+@pytest.mark.asyncio
+async def test_every_in_flight_job_is_cancelled_not_only_the_first():
+ first, second = _Job(), _Job()
+ async with _running_scheduler(first, second) as (scheduler, executor):
+ await cancel_in_flight_scheduler_jobs(scheduler, executor)
+
+ assert first.events == ["cancelled", "finished"]
+ assert second.events == ["cancelled", "finished"]
+
+
+@pytest.mark.asyncio
+async def test_a_job_that_ignores_cancellation_is_abandoned_after_the_timeout(monkeypatch, caplog):
+ """
+ A job that swallows CancelledError must not hold the pod past its
+ termination grace period, so shutdown gives up on it and says so.
+ """
+ monkeypatch.setattr(scheduled_jobs, "JOB_CANCEL_TIMEOUT_SECONDS", 0.05)
+ job = _Job(swallow_cancellation=True)
+ async with _running_scheduler(job) as (scheduler, executor):
+ with caplog.at_level(logging.WARNING, logger="LiteLLM Proxy"):
+ await cancel_in_flight_scheduler_jobs(scheduler, executor)
+
+ assert job.events == ["cancelled"]
+ assert "1 scheduled job(s) did not finish within 0.05s of cancellation" in caplog.text
+
+
+@pytest.mark.asyncio
+async def test_shutdown_with_nothing_in_flight_still_stops_the_scheduler():
+ async with _running_scheduler() as (scheduler, executor):
+ await cancel_in_flight_scheduler_jobs(scheduler, executor)
+ await asyncio.sleep(0)
+
+ assert scheduler.running is False
+
+
+@pytest.mark.asyncio
+async def test_a_scheduler_that_never_started_is_left_alone():
+ """The proxy runs without a scheduler when it has no database; shutdown must not trip on that."""
+ executor = AwaitableAsyncIOExecutor()
+ scheduler = AsyncIOScheduler(executors={"default": executor})
+
+ await cancel_in_flight_scheduler_jobs(scheduler, executor)
+
+ assert scheduler.running is False
diff --git a/tests/test_litellm/proxy/test_spend_log_cleanup.py b/tests/test_litellm/proxy/test_spend_log_cleanup.py
index bf1538183ab..ed35af7ee38 100644
--- a/tests/test_litellm/proxy/test_spend_log_cleanup.py
+++ b/tests/test_litellm/proxy/test_spend_log_cleanup.py
@@ -7,6 +7,7 @@ import math
import time
from contextlib import asynccontextmanager
from datetime import datetime, timedelta, timezone
+from typing import Final
from unittest.mock import AsyncMock, MagicMock
import pytest
@@ -1417,3 +1418,94 @@ def test_the_reported_run_outcome_is_the_most_significant_reason_in_any_order(st
"""
results = tuple(TableCleanupResult(rows_deleted=0, stop_reason=reason) for reason in stop_reasons)
assert SpendLogCleanup._run_outcome(results) == expected
+
+
+_OTHER_OUTCOMES: Final = ("completed", "budget_exhausted", "batch_cap_reached", "skipped_locked", "skipped_disabled")
+
+
+def _runs_recorded(outcome: str) -> float:
+ """The real ``litellm_spend_log_cleanup_runs_total`` sample for one outcome, 0 when unset"""
+ from prometheus_client import REGISTRY
+
+ return REGISTRY.get_sample_value("litellm_spend_log_cleanup_runs_total", {"outcome": outcome}) or 0.0
+
+
+@pytest.mark.asyncio
+async def test_a_cancelled_run_records_aborted_and_logs_its_progress_before_re_raising(monkeypatch):
+ """
+ Shutdown cancels a run by throwing CancelledError into whichever batch is in
+ flight. That is a BaseException, so the Exception handler never saw it and
+ an interrupted run left no outcome metric and no log line; operators could
+ not tell that cleanup stopped early, let alone how far it got.
+ """
+ import litellm.proxy.db.db_transaction_queue.spend_log_cleanup as cleanup_module
+
+ mock_logger = MagicMock()
+ monkeypatch.setattr(cleanup_module, "verbose_proxy_logger", mock_logger)
+ aborted_runs_before = _runs_recorded("aborted")
+ other_runs_before = {outcome: _runs_recorded(outcome) for outcome in _OTHER_OUTCOMES}
+
+ third_batch_reached = asyncio.Event()
+
+ async def _execute_raw(sql, *args):
+ if third_batch_reached.is_set():
+ raise AssertionError("no batch may be issued after the cancelled one")
+ if _execute_raw.calls < 2:
+ _execute_raw.calls += 1
+ return 150
+ third_batch_reached.set()
+ await asyncio.Event().wait()
+
+ _execute_raw.calls = 0
+ mock_prisma_client = MagicMock()
+ _wire_tx(mock_prisma_client.db)
+ mock_prisma_client.db.execute_raw = _execute_raw
+
+ cleaner = SpendLogCleanup(general_settings={"maximum_spend_logs_retention_period": "7d"})
+ cleaner.pod_lock_manager = MagicMock()
+ cleaner.pod_lock_manager.redis_cache = MagicMock()
+ cleaner.pod_lock_manager.acquire_lock = AsyncMock(return_value=True)
+ cleaner.pod_lock_manager.release_lock = AsyncMock()
+
+ run = asyncio.ensure_future(cleaner.cleanup_old_spend_logs(mock_prisma_client))
+ await asyncio.wait_for(third_batch_reached.wait(), timeout=5)
+ run.cancel()
+ with pytest.raises(asyncio.CancelledError):
+ await run
+
+ assert _runs_recorded("aborted") == aborted_runs_before + 1
+ assert {outcome: _runs_recorded(outcome) for outcome in _OTHER_OUTCOMES} == other_runs_before
+ cleaner.pod_lock_manager.release_lock.assert_awaited_once()
+ mock_logger.exception.assert_not_called()
+ (error_call,) = mock_logger.error.call_args_list
+ rendered = error_call[0][0] % error_call[0][1:]
+ assert rendered.startswith("Spend log cleanup cancelled after ")
+ assert "s (rows_deleted=300, batches=2)" in rendered
+
+
+@pytest.mark.asyncio
+async def test_progress_reported_for_a_cancelled_run_is_that_run_only(monkeypatch):
+ """
+ The scheduler holds one cleaner for the life of the process, so the
+ progress counters must start from zero on every run rather than carrying
+ an earlier run's totals into the cancellation line.
+ """
+ import litellm.proxy.db.db_transaction_queue.spend_log_cleanup as cleanup_module
+
+ mock_logger = MagicMock()
+ monkeypatch.setattr(cleanup_module, "verbose_proxy_logger", mock_logger)
+
+ mock_prisma_client = MagicMock()
+ _wire_tx(mock_prisma_client.db)
+ mock_prisma_client.db.execute_raw = AsyncMock(side_effect=[150, 0, 0])
+ cleaner = SpendLogCleanup(general_settings={"maximum_spend_logs_retention_period": "7d"})
+ cleaner.pod_lock_manager = None
+ await cleaner.cleanup_old_spend_logs(mock_prisma_client)
+
+ mock_prisma_client.db.execute_raw = AsyncMock(side_effect=[150, asyncio.CancelledError()])
+ with pytest.raises(asyncio.CancelledError):
+ await cleaner.cleanup_old_spend_logs(mock_prisma_client)
+
+ (error_call,) = mock_logger.error.call_args_list
+ rendered = error_call[0][0] % error_call[0][1:]
+ assert "(rows_deleted=150, batches=1)" in rendered
From 8ce8dd9b3b184a8e8e8884cde29f07e07063f6f7 Mon Sep 17 00:00:00 2001
From: Yucheng He
Date: Tue, 15 Sep 2026 02:25:36 -0700
Subject: [PATCH 030/160] fix(proxy): pause the scheduler at shutdown start and
keep cleanup progress per run
Review follow-ups on #41213:
- Pause the scheduler as the first shutdown step so a job whose fire time
falls inside the shutdown window does not start only to be cancelled.
Jobs already running keep the whole window and are cancelled and
awaited before the database disconnects, as before.
- Keep the cleanup run's progress in a task-scoped ContextVar rather than
on the cleaner instance, so two runs overlapping on one cleaner
(APSCHEDULER_MAX_INSTANCES above 1 without a Redis lock) each report
their own rows and batches on cancellation.
- Drop the module docstrings the repository comment policy does not
allow; the rationale lives in the PR description.
---
.../db_transaction_queue/spend_log_cleanup.py | 37 +++++++++---
litellm/proxy/proxy_server.py | 5 ++
litellm/proxy/shutdown/scheduled_jobs.py | 25 +++-----
.../proxy/shutdown/test_scheduled_jobs.py | 57 ++++++++++++-------
.../proxy/test_spend_log_cleanup.py | 53 +++++++++++++----
5 files changed, 121 insertions(+), 56 deletions(-)
diff --git a/litellm/proxy/db/db_transaction_queue/spend_log_cleanup.py b/litellm/proxy/db/db_transaction_queue/spend_log_cleanup.py
index 1a14210dbec..34213c0d2ce 100644
--- a/litellm/proxy/db/db_transaction_queue/spend_log_cleanup.py
+++ b/litellm/proxy/db/db_transaction_queue/spend_log_cleanup.py
@@ -1,5 +1,6 @@
import asyncio
import time
+from contextvars import ContextVar
from dataclasses import dataclass
from datetime import datetime, timedelta, timezone
from typing import Final, Literal, TypeAlias
@@ -40,6 +41,28 @@ class TableCleanupResult:
stop_reason: StopReason
+class _RunProgress:
+ """How far one cleanup run has got, reported if that run is cancelled"""
+
+ def __init__(self) -> None:
+ self.rows_deleted: int = 0
+ self.batches: int = 0
+
+ def record_batch(self, rows_deleted: int) -> None:
+ self.rows_deleted += rows_deleted
+ self.batches += 1
+
+
+_run_progress: ContextVar[_RunProgress] = ContextVar("spend_log_cleanup_run_progress")
+
+
+def _record_run_batch(rows_deleted: int) -> None:
+ """Count a batch towards the run in progress, if a run is what issued it"""
+ progress: Final = _run_progress.get(None)
+ if progress is not None:
+ progress.record_batch(rows_deleted)
+
+
class _RemainingRow(BaseModel):
"""One row of the capped outstanding-rows probe, validated out of prisma's untyped result."""
@@ -96,8 +119,6 @@ class SpendLogCleanup:
self.general_settings = general_settings or default_settings
self._refresh_bounds()
- self._run_rows_deleted: int = 0
- self._run_batches: int = 0
from litellm.proxy.proxy_server import proxy_logging_obj
pod_lock_manager: Final = proxy_logging_obj.db_spend_update_writer.pod_lock_manager
@@ -424,8 +445,7 @@ class SpendLogCleanup:
total_deleted += deleted_count
run_count += 1
- self._run_rows_deleted += deleted_count
- self._run_batches += 1
+ _record_run_batch(deleted_count)
# Add a small sleep to prevent overwhelming the database
await asyncio.sleep(0.1)
@@ -595,8 +615,8 @@ class SpendLogCleanup:
"""
lock_acquired = False
run_started_at: Final = time.monotonic()
- self._run_rows_deleted = 0
- self._run_batches = 0
+ progress: Final = _RunProgress()
+ progress_token: Final = _run_progress.set(progress)
try:
verbose_proxy_logger.info("Cleanup job triggered at %s", datetime.now())
self._refresh_bounds()
@@ -681,8 +701,8 @@ class SpendLogCleanup:
verbose_proxy_logger.error(
"Spend log cleanup cancelled after %.2fs (rows_deleted=%d, batches=%d); the next run resumes from here",
time.monotonic() - run_started_at,
- self._run_rows_deleted,
- self._run_batches,
+ progress.rows_deleted,
+ progress.batches,
)
SpendLogCleanupMetrics.record_run("aborted")
raise
@@ -697,6 +717,7 @@ class SpendLogCleanup:
SpendLogCleanupMetrics.record_run("aborted")
return # Return after error handling
finally:
+ _run_progress.reset(progress_token)
# Only release the lock if it was actually acquired
if lock_acquired and self.pod_lock_manager and self.pod_lock_manager.redis_cache:
await self.pod_lock_manager.release_lock(cronjob_id=SPEND_LOG_CLEANUP_JOB_NAME)
diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py
index cab0f4d0733..1a7b3a6ccfb 100644
--- a/litellm/proxy/proxy_server.py
+++ b/litellm/proxy/proxy_server.py
@@ -683,6 +683,7 @@ from litellm.proxy.shutdown.graceful_shutdown_manager import GracefulShutdownMan
from litellm.proxy.shutdown.scheduled_jobs import (
AwaitableAsyncIOExecutor,
cancel_in_flight_scheduler_jobs,
+ pause_scheduled_jobs,
)
from litellm.proxy.spend_tracking.budget_reservation import get_budget_window_start
from litellm.proxy.spend_tracking.spend_counter_batch import (
@@ -1419,6 +1420,10 @@ async def proxy_startup_event(app: FastAPI) -> AsyncGenerator[None, None]:
if model_info_scheduler is not scheduler:
model_info_scheduler.shutdown(wait=False)
+ # Shutdown event - stop starting scheduled jobs; the ones already running keep the drain window
+ if scheduler is not None:
+ pause_scheduled_jobs(scheduler)
+
# Shutdown event - drain in-flight requests before tearing down dependencies
# so SIGTERM (rolling update, scale-down, liveness kill) doesn't drop them.
GracefulShutdownManager.start_shutdown()
diff --git a/litellm/proxy/shutdown/scheduled_jobs.py b/litellm/proxy/shutdown/scheduled_jobs.py
index 3c9e791f51c..46c57e1a608 100644
--- a/litellm/proxy/shutdown/scheduled_jobs.py
+++ b/litellm/proxy/shutdown/scheduled_jobs.py
@@ -1,20 +1,3 @@
-"""
-Cancel the proxy's in-flight scheduled jobs at shutdown so they can record how they ended.
-
-APScheduler's ``AsyncIOExecutor.shutdown`` cancels the job tasks it has in flight but cannot
-wait for them, because it is not a coroutine, and under uvicorn nothing else ever will: uvicorn
-re-raises the SIGTERM it captured as soon as the ASGI lifespan shutdown returns, so the process
-dies before ``asyncio.run`` reaches its cancel-all-tasks step. A job mid-run at that point is
-killed without ever observing cancellation, which is how a spend-log cleanup interrupted by a
-rolling restart left no outcome metric and no log line behind. Cancelling here and awaiting the
-cancelled tasks while the database is still connected is what lets a job's own
-``CancelledError`` handler run.
-
-The wait is bounded by ``JOB_CANCEL_TIMEOUT_SECONDS``. A job that has just been cancelled has only
-its own cleanup left to do, so the bound is there for a job that swallows cancellation, not one
-that honours it, and it keeps shutdown well inside a Kubernetes termination grace period.
-"""
-
# pyright: reportMissingTypeStubs=false # apscheduler ships no type information
import asyncio
@@ -34,6 +17,8 @@ class StoppableScheduler(Protocol):
@property
def running(self) -> bool: ...
+ def pause(self) -> None: ...
+
def shutdown(self, wait: bool = ...) -> None: ...
@@ -47,6 +32,12 @@ class AwaitableAsyncIOExecutor(AsyncIOExecutor): # pyright: ignore[reportUntype
return tuple(future for future in self._pending_futures if not future.done())
+def pause_scheduled_jobs(scheduler: StoppableScheduler) -> None:
+ """Stop the scheduler from starting jobs that shutdown would only cancel; running jobs continue"""
+ if scheduler.running:
+ scheduler.pause()
+
+
async def cancel_in_flight_scheduler_jobs(scheduler: StoppableScheduler, executor: AwaitableAsyncIOExecutor) -> None:
"""
Stop the scheduler and wait, bounded by JOB_CANCEL_TIMEOUT_SECONDS, for the jobs it cancels.
diff --git a/tests/test_litellm/proxy/shutdown/test_scheduled_jobs.py b/tests/test_litellm/proxy/shutdown/test_scheduled_jobs.py
index b77d7c4ae50..3301ce34cd6 100644
--- a/tests/test_litellm/proxy/shutdown/test_scheduled_jobs.py
+++ b/tests/test_litellm/proxy/shutdown/test_scheduled_jobs.py
@@ -1,16 +1,8 @@
-"""
-Tests for cancelling in-flight scheduled jobs at proxy shutdown.
-
-These drive a real AsyncIOScheduler: the point of the helper is the hand-off
-between APScheduler's fire-and-forget cancellation and the lifespan shutdown
-that has to outlive it, and a mocked scheduler would not exercise that.
-"""
-
import asyncio
import logging
from collections.abc import AsyncIterator
from contextlib import asynccontextmanager
-from datetime import datetime
+from datetime import datetime, timedelta
import pytest
from apscheduler.schedulers.asyncio import AsyncIOScheduler
@@ -19,11 +11,12 @@ import litellm.proxy.shutdown.scheduled_jobs as scheduled_jobs
from litellm.proxy.shutdown.scheduled_jobs import (
AwaitableAsyncIOExecutor,
cancel_in_flight_scheduler_jobs,
+ pause_scheduled_jobs,
)
class _Job:
- """A scheduled job that blocks until cancelled and records what it observed."""
+ """A scheduled job that blocks until cancelled and records what it observed"""
def __init__(self, swallow_cancellation: bool = False) -> None:
self.started = asyncio.Event()
@@ -45,7 +38,7 @@ class _Job:
@asynccontextmanager
async def _running_scheduler(*jobs: _Job) -> AsyncIterator[tuple[AsyncIOScheduler, AwaitableAsyncIOExecutor]]:
- """A started scheduler with every job in flight; stopped on the way out whatever the test did."""
+ """A started scheduler with every job in flight, stopped on the way out whatever the test did"""
executor = AwaitableAsyncIOExecutor()
scheduler = AsyncIOScheduler(executors={"default": executor})
for index, job in enumerate(jobs):
@@ -66,10 +59,7 @@ async def _running_scheduler(*jobs: _Job) -> AsyncIterator[tuple[AsyncIOSchedule
@pytest.mark.asyncio
async def test_in_flight_jobs_observe_cancellation_before_shutdown_returns():
- """
- The job's own CancelledError handler is what records how a run ended, so
- shutdown must not return until that handler has run.
- """
+ """The job's own CancelledError handler records how a run ended, so shutdown must wait for it"""
job = _Job()
async with _running_scheduler(job) as (scheduler, executor):
await cancel_in_flight_scheduler_jobs(scheduler, executor)
@@ -91,10 +81,7 @@ async def test_every_in_flight_job_is_cancelled_not_only_the_first():
@pytest.mark.asyncio
async def test_a_job_that_ignores_cancellation_is_abandoned_after_the_timeout(monkeypatch, caplog):
- """
- A job that swallows CancelledError must not hold the pod past its
- termination grace period, so shutdown gives up on it and says so.
- """
+ """A job that swallows CancelledError must not hold the pod past its termination grace period"""
monkeypatch.setattr(scheduled_jobs, "JOB_CANCEL_TIMEOUT_SECONDS", 0.05)
job = _Job(swallow_cancellation=True)
async with _running_scheduler(job) as (scheduler, executor):
@@ -116,10 +103,40 @@ async def test_shutdown_with_nothing_in_flight_still_stops_the_scheduler():
@pytest.mark.asyncio
async def test_a_scheduler_that_never_started_is_left_alone():
- """The proxy runs without a scheduler when it has no database; shutdown must not trip on that."""
+ """The proxy runs without a scheduler when it has no database"""
executor = AwaitableAsyncIOExecutor()
scheduler = AsyncIOScheduler(executors={"default": executor})
await cancel_in_flight_scheduler_jobs(scheduler, executor)
assert scheduler.running is False
+
+
+@pytest.mark.asyncio
+async def test_pausing_stops_new_jobs_from_starting_but_leaves_running_ones_alone():
+ """A job due during the shutdown drain would only be cancelled, so it must not start at all"""
+ running = _Job()
+ async with _running_scheduler(running) as (scheduler, executor):
+ late = _Job()
+ scheduler.add_job(late.run, id="late", next_run_time=datetime.now() + timedelta(seconds=0.1))
+
+ pause_scheduled_jobs(scheduler)
+ await asyncio.sleep(0.3)
+
+ assert late.started.is_set() is False
+ assert running.events == []
+ assert scheduler.running is True
+
+ await cancel_in_flight_scheduler_jobs(scheduler, executor)
+
+ assert running.events == ["cancelled", "finished"]
+ assert late.started.is_set() is False
+
+
+@pytest.mark.asyncio
+async def test_pausing_a_scheduler_that_never_started_is_a_no_op():
+ scheduler = AsyncIOScheduler(executors={"default": AwaitableAsyncIOExecutor()})
+
+ pause_scheduled_jobs(scheduler)
+
+ assert scheduler.running is False
diff --git a/tests/test_litellm/proxy/test_spend_log_cleanup.py b/tests/test_litellm/proxy/test_spend_log_cleanup.py
index ed35af7ee38..1691b2d174a 100644
--- a/tests/test_litellm/proxy/test_spend_log_cleanup.py
+++ b/tests/test_litellm/proxy/test_spend_log_cleanup.py
@@ -1432,12 +1432,7 @@ def _runs_recorded(outcome: str) -> float:
@pytest.mark.asyncio
async def test_a_cancelled_run_records_aborted_and_logs_its_progress_before_re_raising(monkeypatch):
- """
- Shutdown cancels a run by throwing CancelledError into whichever batch is in
- flight. That is a BaseException, so the Exception handler never saw it and
- an interrupted run left no outcome metric and no log line; operators could
- not tell that cleanup stopped early, let alone how far it got.
- """
+ """A run cut short by shutdown must leave its outcome and how far it got behind"""
import litellm.proxy.db.db_transaction_queue.spend_log_cleanup as cleanup_module
mock_logger = MagicMock()
@@ -1485,11 +1480,7 @@ async def test_a_cancelled_run_records_aborted_and_logs_its_progress_before_re_r
@pytest.mark.asyncio
async def test_progress_reported_for_a_cancelled_run_is_that_run_only(monkeypatch):
- """
- The scheduler holds one cleaner for the life of the process, so the
- progress counters must start from zero on every run rather than carrying
- an earlier run's totals into the cancellation line.
- """
+ """The scheduler holds one cleaner for the life of the process, so progress must not carry over"""
import litellm.proxy.db.db_transaction_queue.spend_log_cleanup as cleanup_module
mock_logger = MagicMock()
@@ -1509,3 +1500,43 @@ async def test_progress_reported_for_a_cancelled_run_is_that_run_only(monkeypatc
(error_call,) = mock_logger.error.call_args_list
rendered = error_call[0][0] % error_call[0][1:]
assert "(rows_deleted=150, batches=1)" in rendered
+
+
+@pytest.mark.asyncio
+async def test_progress_reported_by_an_overlapping_run_is_its_own(monkeypatch):
+ """With APSCHEDULER_MAX_INSTANCES above one, two runs share the cleaner but not their progress"""
+ import litellm.proxy.db.db_transaction_queue.spend_log_cleanup as cleanup_module
+
+ mock_logger = MagicMock()
+ monkeypatch.setattr(cleanup_module, "verbose_proxy_logger", mock_logger)
+
+ first_batch_done = asyncio.Event()
+ second_run_done = asyncio.Event()
+
+ async def _slow_execute_raw(sql, *args):
+ first_batch_done.set()
+ await second_run_done.wait()
+ return 100
+
+ slow_client = MagicMock()
+ _wire_tx(slow_client.db)
+ slow_client.db.execute_raw = _slow_execute_raw
+ fast_client = MagicMock()
+ _wire_tx(fast_client.db)
+ fast_client.db.execute_raw = AsyncMock(side_effect=[150, 150, 0, 0])
+
+ cleaner = SpendLogCleanup(general_settings={"maximum_spend_logs_retention_period": "7d"})
+ cleaner.pod_lock_manager = None
+
+ slow_run = asyncio.ensure_future(cleaner.cleanup_old_spend_logs(slow_client))
+ await asyncio.wait_for(first_batch_done.wait(), timeout=5)
+ await cleaner.cleanup_old_spend_logs(fast_client)
+ second_run_done.set()
+ await asyncio.sleep(0)
+ slow_run.cancel()
+ with pytest.raises(asyncio.CancelledError):
+ await slow_run
+
+ (error_call,) = mock_logger.error.call_args_list
+ rendered = error_call[0][0] % error_call[0][1:]
+ assert "(rows_deleted=100, batches=1)" in rendered
From 39a199d9a2285a0647fd0d70b3f2a7e2d72120d1 Mon Sep 17 00:00:00 2001
From: Yucheng He
Date: Tue, 15 Sep 2026 03:13:35 -0700
Subject: [PATCH 031/160] fix(proxy): let in-flight scheduled jobs finish
before cancelling them at shutdown
Cancelling every in-flight job the moment shutdown reached the scheduler
dropped the rows a write job had already popped: flush_gateway_requests
drains its accumulator before committing and does not restore it on
CancelledError, and update_spend requeues its batch only after the
shutdown drain had already run.
Shutdown now waits up to JOB_FINISH_TIMEOUT_SECONDS for in-flight jobs
to finish on their own, cancels the ones still running, and does both
before the shutdown flushes so a requeued batch is still written. The
cleanup run never finishes inside the grace, so it is still cancelled
and still records outcome="aborted".
Resolves LIT-6990
---
litellm/proxy/proxy_server.py | 16 ++++----
litellm/proxy/shutdown/scheduled_jobs.py | 22 +++++++----
.../proxy/shutdown/test_scheduled_jobs.py | 39 ++++++++++++++-----
3 files changed, 52 insertions(+), 25 deletions(-)
diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py
index 1a7b3a6ccfb..7505714b418 100644
--- a/litellm/proxy/proxy_server.py
+++ b/litellm/proxy/proxy_server.py
@@ -682,8 +682,8 @@ from litellm.proxy.search_endpoints.endpoints import router as search_router
from litellm.proxy.shutdown.graceful_shutdown_manager import GracefulShutdownManager
from litellm.proxy.shutdown.scheduled_jobs import (
AwaitableAsyncIOExecutor,
- cancel_in_flight_scheduler_jobs,
pause_scheduled_jobs,
+ stop_in_flight_scheduler_jobs,
)
from litellm.proxy.spend_tracking.budget_reservation import get_budget_window_start
from litellm.proxy.spend_tracking.spend_counter_batch import (
@@ -1457,6 +1457,13 @@ async def proxy_startup_event(app: FastAPI) -> AsyncGenerator[None, None]:
await _drain_spend_event_producer_on_shutdown()
+ # Shutdown event - finish or cancel in-flight scheduled jobs before the shutdown flushes and the DB disconnect
+ if scheduler is not None and scheduler_executor is not None:
+ try:
+ await stop_in_flight_scheduler_jobs(scheduler, scheduler_executor)
+ except Exception as e:
+ verbose_proxy_logger.error("Error stopping in-flight scheduled jobs: %s", e)
+
await flush_spend_counters_on_shutdown()
await _flush_spend_logs_queue_on_shutdown()
@@ -1465,13 +1472,6 @@ async def proxy_startup_event(app: FastAPI) -> AsyncGenerator[None, None]:
await proxy_config.stop_auth_cache_invalidation_subscriber()
- # Shutdown event - cancel and await in-flight scheduled jobs while the DB is still connected
- if scheduler is not None and scheduler_executor is not None:
- try:
- await cancel_in_flight_scheduler_jobs(scheduler, scheduler_executor)
- except Exception as e:
- verbose_proxy_logger.error("Error cancelling in-flight scheduled jobs: %s", e)
-
await proxy_shutdown_event(worker_heartbeat=worker_heartbeat)
if prometheus_multiproc_dir:
diff --git a/litellm/proxy/shutdown/scheduled_jobs.py b/litellm/proxy/shutdown/scheduled_jobs.py
index 46c57e1a608..e7625a73b47 100644
--- a/litellm/proxy/shutdown/scheduled_jobs.py
+++ b/litellm/proxy/shutdown/scheduled_jobs.py
@@ -8,6 +8,7 @@ from apscheduler.executors.asyncio import AsyncIOExecutor
from litellm._logging import verbose_proxy_logger
+JOB_FINISH_TIMEOUT_SECONDS: Final = 5.0
JOB_CANCEL_TIMEOUT_SECONDS: Final = 5.0
@@ -38,21 +39,28 @@ def pause_scheduled_jobs(scheduler: StoppableScheduler) -> None:
scheduler.pause()
-async def cancel_in_flight_scheduler_jobs(scheduler: StoppableScheduler, executor: AwaitableAsyncIOExecutor) -> None:
+async def stop_in_flight_scheduler_jobs(scheduler: StoppableScheduler, executor: AwaitableAsyncIOExecutor) -> None:
"""
- Stop the scheduler and wait, bounded by JOB_CANCEL_TIMEOUT_SECONDS, for the jobs it cancels.
+ Let in-flight jobs finish for up to JOB_FINISH_TIMEOUT_SECONDS, then stop the scheduler and
+ wait, bounded by JOB_CANCEL_TIMEOUT_SECONDS, for the jobs it cancels.
- Must run before the database is disconnected: a job's cancellation handler is what records
- the run's outcome, and it needs the connection the job was using.
+ Must run before the database is disconnected: a write job that finishes needs its connection,
+ and a job's cancellation handler is what records the run's outcome.
"""
if not scheduler.running:
return
in_flight: Final = executor.in_flight_jobs()
+ still_running: set[asyncio.Future[object]] = set()
+ if in_flight:
+ verbose_proxy_logger.info(
+ "Waiting up to %ss for %d in-flight scheduled job(s) to finish", JOB_FINISH_TIMEOUT_SECONDS, len(in_flight)
+ )
+ _done, still_running = await asyncio.wait(in_flight, timeout=JOB_FINISH_TIMEOUT_SECONDS)
scheduler.shutdown(wait=False)
- if not in_flight:
+ if not still_running:
return
- verbose_proxy_logger.info("Cancelling %d in-flight scheduled job(s) for shutdown", len(in_flight))
- _done, pending = await asyncio.wait(in_flight, timeout=JOB_CANCEL_TIMEOUT_SECONDS)
+ verbose_proxy_logger.info("Cancelling %d in-flight scheduled job(s) for shutdown", len(still_running))
+ _done, pending = await asyncio.wait(still_running, timeout=JOB_CANCEL_TIMEOUT_SECONDS)
if pending:
verbose_proxy_logger.warning(
"%d scheduled job(s) did not finish within %ss of cancellation; giving up on them",
diff --git a/tests/test_litellm/proxy/shutdown/test_scheduled_jobs.py b/tests/test_litellm/proxy/shutdown/test_scheduled_jobs.py
index 3301ce34cd6..7defd6cef6c 100644
--- a/tests/test_litellm/proxy/shutdown/test_scheduled_jobs.py
+++ b/tests/test_litellm/proxy/shutdown/test_scheduled_jobs.py
@@ -10,23 +10,28 @@ from apscheduler.schedulers.asyncio import AsyncIOScheduler
import litellm.proxy.shutdown.scheduled_jobs as scheduled_jobs
from litellm.proxy.shutdown.scheduled_jobs import (
AwaitableAsyncIOExecutor,
- cancel_in_flight_scheduler_jobs,
+ stop_in_flight_scheduler_jobs,
pause_scheduled_jobs,
)
class _Job:
- """A scheduled job that blocks until cancelled and records what it observed"""
+ """A scheduled job that blocks until cancelled, or for ``work_seconds``, and records what it observed"""
- def __init__(self, swallow_cancellation: bool = False) -> None:
+ def __init__(self, swallow_cancellation: bool = False, work_seconds: float | None = None) -> None:
self.started = asyncio.Event()
self.events: list[str] = []
self.swallow_cancellation = swallow_cancellation
+ self.work_seconds = work_seconds
async def run(self) -> None:
self.started.set()
try:
- await asyncio.Event().wait()
+ if self.work_seconds is None:
+ await asyncio.Event().wait()
+ else:
+ await asyncio.sleep(self.work_seconds)
+ self.events.append("committed")
except asyncio.CancelledError:
self.events.append("cancelled")
if self.swallow_cancellation:
@@ -62,18 +67,32 @@ async def test_in_flight_jobs_observe_cancellation_before_shutdown_returns():
"""The job's own CancelledError handler records how a run ended, so shutdown must wait for it"""
job = _Job()
async with _running_scheduler(job) as (scheduler, executor):
- await cancel_in_flight_scheduler_jobs(scheduler, executor)
+ await stop_in_flight_scheduler_jobs(scheduler, executor)
assert job.events == ["cancelled", "finished"]
assert scheduler.running is False
assert executor.in_flight_jobs() == ()
+@pytest.mark.asyncio
+async def test_a_job_that_is_finishing_is_allowed_to_finish_rather_than_cancelled(monkeypatch):
+ """A spend write cancelled mid-commit drops the rows it popped, so short jobs get to finish first"""
+ monkeypatch.setattr(scheduled_jobs, "JOB_FINISH_TIMEOUT_SECONDS", 2.0)
+ write = _Job(work_seconds=0.2)
+ stuck = _Job()
+ async with _running_scheduler(write, stuck) as (scheduler, executor):
+ await stop_in_flight_scheduler_jobs(scheduler, executor)
+
+ assert write.events == ["committed", "finished"]
+ assert stuck.events == ["cancelled", "finished"]
+ assert scheduler.running is False
+
+
@pytest.mark.asyncio
async def test_every_in_flight_job_is_cancelled_not_only_the_first():
first, second = _Job(), _Job()
async with _running_scheduler(first, second) as (scheduler, executor):
- await cancel_in_flight_scheduler_jobs(scheduler, executor)
+ await stop_in_flight_scheduler_jobs(scheduler, executor)
assert first.events == ["cancelled", "finished"]
assert second.events == ["cancelled", "finished"]
@@ -86,7 +105,7 @@ async def test_a_job_that_ignores_cancellation_is_abandoned_after_the_timeout(mo
job = _Job(swallow_cancellation=True)
async with _running_scheduler(job) as (scheduler, executor):
with caplog.at_level(logging.WARNING, logger="LiteLLM Proxy"):
- await cancel_in_flight_scheduler_jobs(scheduler, executor)
+ await stop_in_flight_scheduler_jobs(scheduler, executor)
assert job.events == ["cancelled"]
assert "1 scheduled job(s) did not finish within 0.05s of cancellation" in caplog.text
@@ -95,7 +114,7 @@ async def test_a_job_that_ignores_cancellation_is_abandoned_after_the_timeout(mo
@pytest.mark.asyncio
async def test_shutdown_with_nothing_in_flight_still_stops_the_scheduler():
async with _running_scheduler() as (scheduler, executor):
- await cancel_in_flight_scheduler_jobs(scheduler, executor)
+ await stop_in_flight_scheduler_jobs(scheduler, executor)
await asyncio.sleep(0)
assert scheduler.running is False
@@ -107,7 +126,7 @@ async def test_a_scheduler_that_never_started_is_left_alone():
executor = AwaitableAsyncIOExecutor()
scheduler = AsyncIOScheduler(executors={"default": executor})
- await cancel_in_flight_scheduler_jobs(scheduler, executor)
+ await stop_in_flight_scheduler_jobs(scheduler, executor)
assert scheduler.running is False
@@ -127,7 +146,7 @@ async def test_pausing_stops_new_jobs_from_starting_but_leaves_running_ones_alon
assert running.events == []
assert scheduler.running is True
- await cancel_in_flight_scheduler_jobs(scheduler, executor)
+ await stop_in_flight_scheduler_jobs(scheduler, executor)
assert running.events == ["cancelled", "finished"]
assert late.started.is_set() is False
From e9109ddf4a563c7d72b30db9bbcc1d334111fec8 Mon Sep 17 00:00:00 2001
From: Tin Chi Lo
Date: Fri, 18 Sep 2026 15:11:12 -0500
Subject: [PATCH 032/160] fix(router): make context-window escalation opt-in
---
.../complexity_router/README.md | 13 ++++
.../complexity_router/config.py | 5 +-
.../router_strategy/test_complexity_router.py | 70 ++++++++++++++-----
.../ContextWindowEscalationConfig.tsx | 5 +-
.../add_model/add_auto_router_tab.test.tsx | 11 +--
.../build_complexity_router_config.test.ts | 19 +++--
.../build_complexity_router_config.ts | 6 +-
...d_updated_complexity_router_config.test.ts | 23 ++++--
.../src/lib/autorouter_presets.test.ts | 6 +-
ui/litellm-dashboard/src/lib/http/schema.d.ts | 4 +-
10 files changed, 111 insertions(+), 51 deletions(-)
diff --git a/litellm/router_strategy/complexity_router/README.md b/litellm/router_strategy/complexity_router/README.md
index 6505746bca1..2c2aea333f9 100644
--- a/litellm/router_strategy/complexity_router/README.md
+++ b/litellm/router_strategy/complexity_router/README.md
@@ -68,6 +68,19 @@ still resolve to a deployment in `model_list`; this configuration does not creat
- abc
```
+### Context-window escalation
+
+Context-window escalation is opt-in. Omit `enable_context_window_escalation` or set it to
+`false` to keep the complexity-selected model without context-window replacement or filtering
+
+Set `enable_context_window_escalation: true` inside `complexity_router_config` to restrict the
+selected tier to models whose declared windows fit the prompt, or move to the lowest configured
+tier with a fitting model when none in the selected tier fit. Unknown windows do not justify
+moving a request. `context_window_escalation_buffer` defaults to `0.95`
+
+Existing saved configurations with explicit `true` keep escalation enabled. Configurations that
+omit the setting now default to disabled; set it to `true` to retain their previous behavior
+
### Capability forecasting
Set `classifier_type: capability` to use
diff --git a/litellm/router_strategy/complexity_router/config.py b/litellm/router_strategy/complexity_router/config.py
index aa39dff8c53..213d3864dc0 100644
--- a/litellm/router_strategy/complexity_router/config.py
+++ b/litellm/router_strategy/complexity_router/config.py
@@ -1305,7 +1305,7 @@ class ComplexityRouterConfig(BaseModel):
)
enable_context_window_escalation: bool = Field(
- default=True,
+ default=False,
description=(
"Escalate a request off a tier whose models provably cannot hold its prompt, before "
"dispatch. The classifier scores complexity and never prompt size, so a long agentic "
@@ -1315,7 +1315,8 @@ class ComplexityRouterConfig(BaseModel):
"moves to the lowest configured tier with a model whose declared window fits; when "
"only some of the tier's models fit, the pick is restricted to those and the tier "
"keeps the request. Models with no resolvable window are never escalated away from "
- "and never escalated onto. Set false to dispatch on complexity alone, as before."
+ "and never escalated onto. Disabled by default: omit or set false to dispatch on "
+ "complexity alone; set true to enable context-window escalation."
),
)
context_window_escalation_buffer: float = Field(
diff --git a/tests/test_litellm/router_strategy/test_complexity_router.py b/tests/test_litellm/router_strategy/test_complexity_router.py
index 9b25c869f1c..10d674f1ab8 100644
--- a/tests/test_litellm/router_strategy/test_complexity_router.py
+++ b/tests/test_litellm/router_strategy/test_complexity_router.py
@@ -13,6 +13,7 @@ import time
from collections.abc import AsyncIterator, Mapping, Sequence
from copy import deepcopy
from functools import partial
+from types import MappingProxyType
from typing import Dict, Final, List, Literal
from unittest.mock import AsyncMock, MagicMock, patch
@@ -90,6 +91,7 @@ from litellm.types.router import (
TaggedPreRoutingStrategy,
)
from litellm.types.llms.openai import ResponsesAPIResponse
+from litellm.types.management_endpoints.auto_router_endpoints import RequestComplexityRouterConfig
from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler
@@ -6558,6 +6560,7 @@ class TestTierModelAffinity:
litellm_router_instance=_windowed_router(_SMALL, _BIG),
complexity_router_config={
"tiers": {"SIMPLE": ["small-model", "big-model"]},
+ "enable_context_window_escalation": True,
"adaptive": adaptive,
"deployment_affinity": True,
"session_affinity": False,
@@ -13723,8 +13726,12 @@ _CJK_TURNS = [
]
-def _tier_config(**overrides) -> Dict:
- return {"tiers": {"SIMPLE": "small-model", "COMPLEX": "big-model"}, **overrides}
+def _tier_config(**overrides: object) -> dict[str, object]:
+ return {
+ "tiers": {"SIMPLE": "small-model", "COMPLEX": "big-model"},
+ "enable_context_window_escalation": True,
+ **overrides,
+ }
class TestContextWindowEscalation:
@@ -13783,7 +13790,7 @@ class TestContextWindowEscalation:
router = ComplexityRouter(
model_name="test-router",
litellm_router_instance=_windowed_router(_SMALL, ("mid-model", "openai/gpt-4o-mini", 200000), _BIG),
- complexity_router_config={"tiers": {"SIMPLE": ["small-model", "mid-model"], "COMPLEX": "big-model"}},
+ complexity_router_config=_tier_config(tiers={"SIMPLE": ["small-model", "mid-model"], "COMPLEX": "big-model"}),
)
result = await router.async_pre_routing_hook(model="test-router", request_kwargs={}, messages=_OVERSIZED_TURNS)
@@ -13820,7 +13827,7 @@ class TestContextWindowEscalation:
},
]
),
- complexity_router_config={"tiers": {"SIMPLE": "mixed-pool", "COMPLEX": "big-model"}},
+ complexity_router_config=_tier_config(tiers={"SIMPLE": "mixed-pool", "COMPLEX": "big-model"}),
)
result = await router.async_pre_routing_hook(model="test-router", request_kwargs={}, messages=_OVERSIZED_TURNS)
@@ -13871,7 +13878,7 @@ class TestContextWindowEscalation:
router = ComplexityRouter(
model_name="test-router",
litellm_router_instance=_windowed_router(*deployments),
- complexity_router_config={"tiers": tiers},
+ complexity_router_config=_tier_config(tiers=tiers),
)
result = await router.async_pre_routing_hook(model="test-router", request_kwargs={}, messages=_OVERSIZED_TURNS)
@@ -13880,19 +13887,37 @@ class TestContextWindowEscalation:
assert result.model == expected_model
@pytest.mark.asyncio
- async def test_the_disabled_gate_dispatches_on_complexity_alone(self):
- """The escape hatch: enable_context_window_escalation false restores today's behavior."""
- router = ComplexityRouter(
+ @pytest.mark.parametrize("enabled", (None, False, True), ids=("omitted", "disabled", "enabled"))
+ @pytest.mark.parametrize("serialized", (False, True), ids=("config", "http-json"))
+ async def test_context_window_escalation_requires_opt_in(self, enabled: bool | None, serialized: bool) -> None:
+ setting: Final = (
+ MappingProxyType({"enable_context_window_escalation": enabled})
+ if enabled is not None
+ else MappingProxyType({})
+ )
+ raw_config: Final = RequestComplexityRouterConfig.model_validate(
+ MappingProxyType(
+ {"tiers": MappingProxyType({"SIMPLE": "small-model", "COMPLEX": "big-model"}), **setting}
+ )
+ )
+ config: Final = (
+ RequestComplexityRouterConfig.model_validate_json(raw_config.model_dump_json())
+ if serialized
+ else raw_config
+ )
+ router: Final = ComplexityRouter(
model_name="test-router",
litellm_router_instance=_windowed_router(_SMALL, _BIG),
- complexity_router_config=_tier_config(enable_context_window_escalation=False),
+ complexity_router_config=config.model_dump(exclude_unset=not serialized, exclude_none=True),
)
- result = await router.async_pre_routing_hook(model="test-router", request_kwargs={}, messages=_OVERSIZED_TURNS)
+ result: Final = await router.async_pre_routing_hook(
+ model="test-router", request_kwargs={}, messages=_OVERSIZED_TURNS
+ )
assert result is not None
- assert result.model == "small-model"
- assert "context_escalated" not in result.routing_decision
+ assert result.model == ("big-model" if enabled else "small-model")
+ assert result.routing_decision.get("context_escalated", False) is (enabled is True)
@pytest.mark.asyncio
async def test_out_of_band_system_and_tools_count_against_the_window(self):
@@ -14001,7 +14026,7 @@ class TestContextWindowEscalation:
},
]
),
- complexity_router_config={"adaptive": True, "tiers": {"SIMPLE": ["small-model", "mid-model"]}},
+ complexity_router_config=_tier_config(adaptive=True, tiers={"SIMPLE": ["small-model", "mid-model"]}),
)
result = await router.async_pre_routing_hook(model="test-router", request_kwargs={}, messages=_OVERSIZED_TURNS)
@@ -14031,7 +14056,7 @@ class TestContextWindowEscalation:
},
]
),
- complexity_router_config={"tiers": {"SIMPLE": "cop-pool", "COMPLEX": "big-model"}},
+ complexity_router_config=_tier_config(tiers={"SIMPLE": "cop-pool", "COMPLEX": "big-model"}),
)
real_get_llm_provider = litellm.get_llm_provider
copilot_resolutions: List = []
@@ -14064,7 +14089,7 @@ class TestContextWindowEscalation:
"model_name": "smart-router",
"litellm_params": {
"model": "auto_router/complexity_router",
- "complexity_router_config": {"tiers": {"SIMPLE": "small-model", "COMPLEX": "big-model"}},
+ "complexity_router_config": _tier_config(),
},
},
{
@@ -14894,7 +14919,12 @@ class TestHealthFallbackDispatch:
) -> None:
from litellm.types.router import RouterRateLimitError
- router: Final = self._router(config={"tiers": {"SIMPLE": "primary", "MEDIUM": "peer", "COMPLEX": "large"}})
+ router: Final = self._router(
+ config={
+ "tiers": {"SIMPLE": "primary", "MEDIUM": "peer", "COMPLEX": "large"},
+ "enable_context_window_escalation": True,
+ }
+ )
router.add_deployment(
Deployment(
model_name="large",
@@ -14971,7 +15001,13 @@ class TestHealthFallbackDispatch:
@pytest.mark.asyncio
@pytest.mark.parametrize("default_fits", [True, False])
async def test_modality_default_must_also_fit_context(self, default_fits: bool) -> None:
- router: Final = self._router(config={"modality_routing": True, "tiers": {"SIMPLE": "primary"}})
+ router: Final = self._router(
+ config={
+ "modality_routing": True,
+ "tiers": {"SIMPLE": "primary"},
+ "enable_context_window_escalation": True,
+ }
+ )
for deployment in router.model_list:
deployment["model_info"]["supports_vision"] = deployment["model_name"] == "fallback"
deployment["model_info"]["max_input_tokens"] = 10000 if default_fits else 10
diff --git a/ui/litellm-dashboard/src/components/add_model/ContextWindowEscalationConfig.tsx b/ui/litellm-dashboard/src/components/add_model/ContextWindowEscalationConfig.tsx
index c0a65076d20..ad09efd8059 100644
--- a/ui/litellm-dashboard/src/components/add_model/ContextWindowEscalationConfig.tsx
+++ b/ui/litellm-dashboard/src/components/add_model/ContextWindowEscalationConfig.tsx
@@ -7,7 +7,7 @@ const ContextWindowEscalationConfig: React.FC<{
value: ComplexityRouterConfigValue;
onChange: (value: ComplexityRouterConfigValue) => void;
}> = ({ value, onChange }) => {
- const enabled = value.enable_context_window_escalation ?? true;
+ const enabled = value.enable_context_window_escalation ?? false;
// A number input renders Number("0.") as "0", so a decimal cannot be typed without a local draft.
const [bufferDraft, setBufferDraft] = React.useState(null);
const commitBuffer = (raw: string) => {
@@ -32,7 +32,8 @@ const ContextWindowEscalationConfig: React.FC<{
When a prompt provably cannot fit the decided tier's context windows, route it to the lowest tier whose
- window holds it instead of letting the provider reject it. Off means requests dispatch on complexity alone.
+ window holds it instead of letting the provider reject it. Disabled by default. Off means requests dispatch on
+ complexity alone.
{enabled && (
diff --git a/ui/litellm-dashboard/src/components/add_model/add_auto_router_tab.test.tsx b/ui/litellm-dashboard/src/components/add_model/add_auto_router_tab.test.tsx
index 48903d585ff..578b455d2b8 100644
--- a/ui/litellm-dashboard/src/components/add_model/add_auto_router_tab.test.tsx
+++ b/ui/litellm-dashboard/src/components/add_model/add_auto_router_tab.test.tsx
@@ -669,7 +669,7 @@ describe("AddAutoRouterTab", () => {
});
});
- it("carries a context-window escalation opt-out through to the create payload", async () => {
+ it("starts context-window escalation disabled and carries an explicit opt-in to the create payload", async () => {
const user = userEvent.setup();
vi.mocked(getMissingTiersError).mockReturnValue(null);
@@ -679,14 +679,15 @@ describe("AddAutoRouterTab", () => {
expandDetailedConfiguration();
await user.click(screen.getByText("Advanced: Context Window Escalation"));
const toggle = await screen.findByRole("switch", { name: "Escalate oversized prompts to a tier that fits" });
- expect(toggle).toBeChecked();
+ expect(toggle).not.toBeChecked();
+ expect(screen.queryByLabelText("Window fit buffer")).not.toBeInTheDocument();
await user.click(toggle);
await user.click(screen.getByRole("button", { name: /add auto router/i }));
await waitFor(() => expect(handleAddAutoRouterSubmit).toHaveBeenCalled());
expect(vi.mocked(handleAddAutoRouterSubmit).mock.calls.at(-1)?.[0].complexity_router_config).toMatchObject({
- enable_context_window_escalation: false,
+ enable_context_window_escalation: true,
});
});
@@ -699,6 +700,7 @@ describe("AddAutoRouterTab", () => {
await user.type(screen.getByPlaceholderText(/smart_router/i), "ctx-buffer-router");
expandDetailedConfiguration();
await user.click(screen.getByText("Advanced: Context Window Escalation"));
+ await user.click(screen.getByRole("switch", { name: "Escalate oversized prompts to a tier that fits" }));
const buffer = await screen.findByLabelText("Window fit buffer");
fireEvent.change(buffer, { target: { value: "1.5" } });
fireEvent.blur(buffer, { target: { value: "1.5" } });
@@ -708,7 +710,7 @@ describe("AddAutoRouterTab", () => {
await waitFor(() => expect(handleAddAutoRouterSubmit).toHaveBeenCalled());
const config = vi.mocked(handleAddAutoRouterSubmit).mock.calls.at(-1)?.[0].complexity_router_config;
expect(config).toMatchObject({ context_window_escalation_buffer: 1 });
- expect(config).not.toHaveProperty("enable_context_window_escalation");
+ expect(config).toHaveProperty("enable_context_window_escalation", true);
});
it("clearing the buffer removes it from the payload so the router tracks the backend default", async () => {
@@ -720,6 +722,7 @@ describe("AddAutoRouterTab", () => {
await user.type(screen.getByPlaceholderText(/smart_router/i), "ctx-clear-router");
expandDetailedConfiguration();
await user.click(screen.getByText("Advanced: Context Window Escalation"));
+ await user.click(screen.getByRole("switch", { name: "Escalate oversized prompts to a tier that fits" }));
const buffer = await screen.findByLabelText("Window fit buffer");
fireEvent.change(buffer, { target: { value: "0.8" } });
fireEvent.blur(buffer, { target: { value: "0.8" } });
diff --git a/ui/litellm-dashboard/src/components/add_model/build_complexity_router_config.test.ts b/ui/litellm-dashboard/src/components/add_model/build_complexity_router_config.test.ts
index 6e6e7a3c6cd..0d70b18cd94 100644
--- a/ui/litellm-dashboard/src/components/add_model/build_complexity_router_config.test.ts
+++ b/ui/litellm-dashboard/src/components/add_model/build_complexity_router_config.test.ts
@@ -49,7 +49,7 @@ const baseParams: BuildComplexityRouterConfigParams = {
describe("buildComplexityRouterConfig", () => {
it.each(["capability", "llm_v2", "heuristic"] as const)(
- "disables the removed overrides only for forecast creates: %s",
+ "preserves explicit context-window opt-in beside forecast restrictions: %s",
(classifierType) => {
const forecast = classifierType !== "heuristic";
const params = {
@@ -61,14 +61,10 @@ describe("buildComplexityRouterConfig", () => {
};
const config = buildComplexityRouterConfig(params);
expect(config.adaptive).toBe(!forecast);
- expect(config.enable_context_window_escalation).toBe(!forecast);
+ expect(config.enable_context_window_escalation).toBe(true);
+ expect(config.context_window_escalation_buffer).toBe(0.9);
expect(config.escalation_keywords).toEqual(forecast ? [] : ["LITELLM ESCALATE"]);
- for (const key of [
- "adaptive_weights",
- "adaptive_eligible",
- "tier_distance_penalty",
- "context_window_escalation_buffer",
- ]) {
+ for (const key of ["adaptive_weights", "adaptive_eligible", "tier_distance_penalty"]) {
expect(Object.hasOwn(config, key)).toBe(!forecast);
}
if (forecast) {
@@ -107,13 +103,14 @@ describe("buildComplexityRouterConfig", () => {
expect(config).toEqual(expected);
});
- it("carries an explicit context-window escalation opt-out and buffer, false included", () => {
+ it.each([undefined, false, true])("preserves the context-window escalation setting: %s", (enabled) => {
const config = buildComplexityRouterConfig({
...baseParams,
- enableContextWindowEscalation: false,
+ enableContextWindowEscalation: enabled,
contextWindowEscalationBuffer: 0.9,
});
- expect(config.enable_context_window_escalation).toBe(false);
+ expect(config.enable_context_window_escalation).toBe(enabled);
+ expect(Object.hasOwn(config, "enable_context_window_escalation")).toBe(enabled !== undefined);
expect(config.context_window_escalation_buffer).toBe(0.9);
});
diff --git a/ui/litellm-dashboard/src/components/add_model/build_complexity_router_config.ts b/ui/litellm-dashboard/src/components/add_model/build_complexity_router_config.ts
index 8a377c17ad7..d1cea6d48a9 100644
--- a/ui/litellm-dashboard/src/components/add_model/build_complexity_router_config.ts
+++ b/ui/litellm-dashboard/src/components/add_model/build_complexity_router_config.ts
@@ -629,6 +629,7 @@ export const buildComplexityRouterConfig = ({
// the form never rewrote. The UI gates the same controls on this, not on the raw value.
const effectiveType: ClassifierType = customTierSet ? "llm" : classifierType;
const forecast = isForecastClassifier(effectiveType);
+ const preserveContextWindowBuffer = !forecast || enableContextWindowEscalation === true;
const supportsOpeningPrompt = !customTierSet && !forecast && usesLlmClassifier(effectiveType);
const payload: ComplexityRouterConfigPayload = {
@@ -682,11 +683,10 @@ export const buildComplexityRouterConfig = ({
adaptive_eligible: adaptiveEligible,
}),
...(returnRawModelName && { return_raw_model_name: true }),
- // Omission enables the backend default, so hidden forecast controls need an explicit opt-out.
...((forecast || enableContextWindowEscalation !== undefined) && {
- enable_context_window_escalation: forecast ? false : enableContextWindowEscalation,
+ enable_context_window_escalation: enableContextWindowEscalation ?? false,
}),
- ...(!forecast &&
+ ...(preserveContextWindowBuffer &&
contextWindowEscalationBuffer !== undefined && {
context_window_escalation_buffer: contextWindowEscalationBuffer,
}),
diff --git a/ui/litellm-dashboard/src/components/edit_auto_router/build_updated_complexity_router_config.test.ts b/ui/litellm-dashboard/src/components/edit_auto_router/build_updated_complexity_router_config.test.ts
index 4ae6efbb12d..31f2b8ef68a 100644
--- a/ui/litellm-dashboard/src/components/edit_auto_router/build_updated_complexity_router_config.test.ts
+++ b/ui/litellm-dashboard/src/components/edit_auto_router/build_updated_complexity_router_config.test.ts
@@ -64,14 +64,10 @@ describe("buildUpdatedComplexityRouterConfig keyword matching", () => {
const saved = buildUpdatedComplexityRouterConfig(stored, value, undefined, keywordState);
const forecast = classifier_type !== "heuristic";
expect(saved.adaptive).toBe(!forecast);
- expect(saved.enable_context_window_escalation).toBe(!forecast);
+ expect(saved.enable_context_window_escalation).toBe(true);
+ expect(saved.context_window_escalation_buffer).toBe(0.9);
expect(saved.escalation_keywords).toEqual(forecast ? [] : stored.escalation_keywords);
- for (const key of [
- "adaptive_weights",
- "adaptive_eligible",
- "tier_distance_penalty",
- "context_window_escalation_buffer",
- ]) {
+ for (const key of ["adaptive_weights", "adaptive_eligible", "tier_distance_penalty"]) {
expect(Object.hasOwn(saved, key)).toBe(!forecast);
}
expect(saved.keyword_tier_rules).toEqual(STORED.keyword_tier_rules);
@@ -83,6 +79,19 @@ describe("buildUpdatedComplexityRouterConfig keyword matching", () => {
},
);
+ it.each([undefined, false, true])("preserves stored context-window escalation on save: %s", (enabled) => {
+ const stored = {
+ ...STORED,
+ ...(enabled !== undefined && { enable_context_window_escalation: enabled }),
+ };
+ const value = hydrateComplexityRouterConfig(stored, undefined);
+ const saved = buildUpdatedComplexityRouterConfig(stored, value, undefined, hydratedState);
+ const serialized: typeof saved = JSON.parse(JSON.stringify(saved));
+ expect(value.enable_context_window_escalation).toBe(enabled);
+ expect(serialized.enable_context_window_escalation).toBe(enabled);
+ expect(Object.hasOwn(serialized, "enable_context_window_escalation")).toBe(enabled !== undefined);
+ });
+
it("round-trips an untouched edit without changing any keyword-matching value", () => {
// Opening the modal hydrates state from STORED; saving with nothing changed must be a
// no-op. These keys are now MANAGED, so a hydration bug silently wipes them.
diff --git a/ui/litellm-dashboard/src/lib/autorouter_presets.test.ts b/ui/litellm-dashboard/src/lib/autorouter_presets.test.ts
index fed11454c23..cda9ba104e5 100644
--- a/ui/litellm-dashboard/src/lib/autorouter_presets.test.ts
+++ b/ui/litellm-dashboard/src/lib/autorouter_presets.test.ts
@@ -708,7 +708,7 @@ describe("autorouter_presets", () => {
expect(prefill.escalationKeywords).toEqual([]);
});
- it("carries a preset's context-window escalation opt-out and buffer through the prefill", () => {
+ it.each([undefined, false, true])("preserves a preset's context-window escalation setting: %s", (enabled) => {
const prefill = buildPresetPrefill(
{
tiers: { SIMPLE: ["gpt-5-nano"], MEDIUM: [], COMPLEX: [], REASONING: [] },
@@ -716,12 +716,12 @@ describe("autorouter_presets", () => {
classification_mode: "every_request",
session_affinity: false,
deployment_affinity: true,
- enable_context_window_escalation: false,
+ enable_context_window_escalation: enabled,
context_window_escalation_buffer: 0.9,
},
groupsOnly(["gpt-5-nano"]),
);
- expect(prefill.complexityRouterConfig.enable_context_window_escalation).toBe(false);
+ expect(prefill.complexityRouterConfig.enable_context_window_escalation).toBe(enabled);
expect(prefill.complexityRouterConfig.context_window_escalation_buffer).toBe(0.9);
});
diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts
index d43adfe1ae4..a18b02646ee 100644
--- a/ui/litellm-dashboard/src/lib/http/schema.d.ts
+++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts
@@ -36502,8 +36502,8 @@ export interface components {
embedding_model?: string | null;
/**
* Enable Context Window Escalation
- * @description Escalate a request off a tier whose models provably cannot hold its prompt, before dispatch. The classifier scores complexity and never prompt size, so a long agentic session whose newest ask is trivial lands on a small-window tier and the provider rejects it with a context-window 400 that nothing retries. When every model of the decided tier has a declared window smaller than the estimated prompt, the request moves to the lowest configured tier with a model whose declared window fits; when only some of the tier's models fit, the pick is restricted to those and the tier keeps the request. Models with no resolvable window are never escalated away from and never escalated onto. Set false to dispatch on complexity alone, as before.
- * @default true
+ * @description Escalate a request off a tier whose models provably cannot hold its prompt, before dispatch. The classifier scores complexity and never prompt size, so a long agentic session whose newest ask is trivial lands on a small-window tier and the provider rejects it with a context-window 400 that nothing retries. When every model of the decided tier has a declared window smaller than the estimated prompt, the request moves to the lowest configured tier with a model whose declared window fits; when only some of the tier's models fit, the pick is restricted to those and the tier keeps the request. Models with no resolvable window are never escalated away from and never escalated onto. Disabled by default: omit or set false to dispatch on complexity alone; set true to enable context-window escalation.
+ * @default false
*/
enable_context_window_escalation: boolean;
/**
From 5f722bc19559a82df160b748f26746dac7b55488 Mon Sep 17 00:00:00 2001
From: Tin
Date: Sat, 19 Sep 2026 10:26:42 -0700
Subject: [PATCH 033/160] chore(router): remove in-repo escalation docs
---
litellm/router_strategy/complexity_router/README.md | 13 -------------
1 file changed, 13 deletions(-)
diff --git a/litellm/router_strategy/complexity_router/README.md b/litellm/router_strategy/complexity_router/README.md
index 2c2aea333f9..6505746bca1 100644
--- a/litellm/router_strategy/complexity_router/README.md
+++ b/litellm/router_strategy/complexity_router/README.md
@@ -68,19 +68,6 @@ still resolve to a deployment in `model_list`; this configuration does not creat
- abc
```
-### Context-window escalation
-
-Context-window escalation is opt-in. Omit `enable_context_window_escalation` or set it to
-`false` to keep the complexity-selected model without context-window replacement or filtering
-
-Set `enable_context_window_escalation: true` inside `complexity_router_config` to restrict the
-selected tier to models whose declared windows fit the prompt, or move to the lowest configured
-tier with a fitting model when none in the selected tier fit. Unknown windows do not justify
-moving a request. `context_window_escalation_buffer` defaults to `0.95`
-
-Existing saved configurations with explicit `true` keep escalation enabled. Configurations that
-omit the setting now default to disabled; set it to `true` to retain their previous behavior
-
### Capability forecasting
Set `classifier_type: capability` to use
From 0068df5a8beac2b137fc9412586048c2f42a0af3 Mon Sep 17 00:00:00 2001
From: Tin Chi Lo
Date: Sat, 19 Sep 2026 14:46:58 -0700
Subject: [PATCH 034/160] feat(ui): add internal-user savings and auto-router
usage
---
.../migration.sql | 42 +++
.../litellm_proxy_extras/schema.prisma | 41 ++
litellm/proxy/db/autorouter_session_rollup.py | 170 ++++++---
litellm/proxy/db/baseline_accounting.py | 40 +-
.../db_transaction_queue/spend_log_cleanup.py | 24 +-
.../auto_router_endpoints.py | 11 +-
litellm/proxy/schema.prisma | 41 ++
schema.prisma | 41 ++
.../spend/test_autorouter_session_rollup.py | 171 ++++++++-
.../spend/test_baseline_accounting.py | 67 +++-
.../db/test_autorouter_session_rollup.py | 115 +++++-
.../test_auto_router_endpoints.py | 42 ++-
.../proxy/test_spend_log_cleanup.py | 13 +-
.../AutoRouterBenchmarksTab.test.tsx | 4 +-
.../_components/AutoRouterBenchmarksTab.tsx | 19 +-
.../_components/useAutoRouterBenchmarks.ts | 9 +-
.../useDailyActivityRange.test.tsx | 2 +
.../_components/useDailyActivityRange.ts | 6 +-
.../user_info_view.integration.test.tsx | 357 +++++++++++++++++-
.../_components/view_users/user_info_view.tsx | 60 ++-
.../components/shared/ScopedSavingsTab.tsx | 133 +++++++
.../components/templates/KeySavingsTab.tsx | 133 +------
ui/litellm-dashboard/src/lib/http/schema.d.ts | 9 +-
23 files changed, 1321 insertions(+), 229 deletions(-)
create mode 100644 litellm-proxy-extras/litellm_proxy_extras/migrations/20260919000000_add_autorouter_user_session_rollup/migration.sql
create mode 100644 ui/litellm-dashboard/src/components/shared/ScopedSavingsTab.tsx
diff --git a/litellm-proxy-extras/litellm_proxy_extras/migrations/20260919000000_add_autorouter_user_session_rollup/migration.sql b/litellm-proxy-extras/litellm_proxy_extras/migrations/20260919000000_add_autorouter_user_session_rollup/migration.sql
new file mode 100644
index 00000000000..2b864131ab2
--- /dev/null
+++ b/litellm-proxy-extras/litellm_proxy_extras/migrations/20260919000000_add_autorouter_user_session_rollup/migration.sql
@@ -0,0 +1,42 @@
+CREATE TABLE IF NOT EXISTS "LiteLLM_AutoRouterUserSession" (
+ "user_id" TEXT NOT NULL,
+ "api_key" TEXT NOT NULL,
+ "session_id" TEXT NOT NULL,
+ "router_name" TEXT NOT NULL,
+ "router_type" TEXT NOT NULL,
+ "first_turn_at" TIMESTAMP(3) NOT NULL,
+ "last_turn_at" TIMESTAMP(3) NOT NULL,
+ "last_model" TEXT NOT NULL,
+ "models" JSONB NOT NULL DEFAULT '{}',
+ "turns" INTEGER NOT NULL DEFAULT 0,
+ "unordered_turns" INTEGER NOT NULL DEFAULT 0,
+ "covered_turns" INTEGER NOT NULL DEFAULT 0,
+ "cache_hits" INTEGER NOT NULL DEFAULT 0,
+ "same_model_turns" INTEGER NOT NULL DEFAULT 0,
+ "same_model_hits" INTEGER NOT NULL DEFAULT 0,
+ "first_visit_turns" INTEGER NOT NULL DEFAULT 0,
+ "first_visit_hits" INTEGER NOT NULL DEFAULT 0,
+ "return_turns" INTEGER NOT NULL DEFAULT 0,
+ "return_hits" INTEGER NOT NULL DEFAULT 0,
+ "return_expired_misses" INTEGER NOT NULL DEFAULT 0,
+ "return_within_ttl_misses" INTEGER NOT NULL DEFAULT 0,
+ "ttl_5m_turns" INTEGER NOT NULL DEFAULT 0,
+ "ttl_1h_turns" INTEGER NOT NULL DEFAULT 0,
+ "total_tokens" BIGINT NOT NULL DEFAULT 0,
+ "spend" DOUBLE PRECISION NOT NULL DEFAULT 0,
+ "saved_spend" DOUBLE PRECISION NOT NULL DEFAULT 0,
+ "savings_estimated_turns" INTEGER NOT NULL DEFAULT 0,
+ "savings_estimated_actual_spend" DOUBLE PRECISION NOT NULL DEFAULT 0,
+ "savings_estimated_saved_spend" DOUBLE PRECISION NOT NULL DEFAULT 0,
+ "savings_estimated_baseline_models" JSONB NOT NULL DEFAULT '{}',
+ "classifier_cost" DOUBLE PRECISION NOT NULL DEFAULT 0,
+ "classifier_cost_recorded_turns" INTEGER NOT NULL DEFAULT 0,
+ "tier_turns" JSONB NOT NULL DEFAULT '{}',
+ "baseline_models" JSONB NOT NULL DEFAULT '{}',
+
+ CONSTRAINT "LiteLLM_AutoRouterUserSession_pkey" PRIMARY KEY ("user_id", "api_key", "session_id", "router_name")
+);
+
+CREATE INDEX IF NOT EXISTS "idx_autorouter_user_session_last_turn" ON "LiteLLM_AutoRouterUserSession"("last_turn_at");
+
+CREATE INDEX IF NOT EXISTS "idx_autorouter_user_session_user_last_turn" ON "LiteLLM_AutoRouterUserSession"("user_id", "last_turn_at");
diff --git a/litellm-proxy-extras/litellm_proxy_extras/schema.prisma b/litellm-proxy-extras/litellm_proxy_extras/schema.prisma
index d2032cec0d0..f4015ed9277 100644
--- a/litellm-proxy-extras/litellm_proxy_extras/schema.prisma
+++ b/litellm-proxy-extras/litellm_proxy_extras/schema.prisma
@@ -1620,6 +1620,47 @@ model LiteLLM_AutoRouterSession {
@@index([last_turn_at], map: "idx_autorouter_session_last_turn")
}
+model LiteLLM_AutoRouterUserSession {
+ user_id String
+ api_key String
+ session_id String
+ router_name String
+ router_type String
+ first_turn_at DateTime
+ last_turn_at DateTime
+ last_model String
+ models Json @default("{}")
+ turns Int @default(0)
+ unordered_turns Int @default(0)
+ covered_turns Int @default(0)
+ cache_hits Int @default(0)
+ same_model_turns Int @default(0)
+ same_model_hits Int @default(0)
+ first_visit_turns Int @default(0)
+ first_visit_hits Int @default(0)
+ return_turns Int @default(0)
+ return_hits Int @default(0)
+ return_expired_misses Int @default(0)
+ return_within_ttl_misses Int @default(0)
+ ttl_5m_turns Int @default(0)
+ ttl_1h_turns Int @default(0)
+ total_tokens BigInt @default(0)
+ spend Float @default(0)
+ saved_spend Float @default(0)
+ savings_estimated_turns Int @default(0)
+ savings_estimated_actual_spend Float @default(0)
+ savings_estimated_saved_spend Float @default(0)
+ savings_estimated_baseline_models Json @default("{}")
+ classifier_cost Float @default(0)
+ classifier_cost_recorded_turns Int @default(0)
+ tier_turns Json @default("{}")
+ baseline_models Json @default("{}")
+
+ @@id([user_id, api_key, session_id, router_name])
+ @@index([last_turn_at], map: "idx_autorouter_user_session_last_turn")
+ @@index([user_id, last_turn_at], map: "idx_autorouter_user_session_user_last_turn")
+}
+
// Shadow eval: evaluation of an auto-router against one or more keys' live traffic, in
// either direction. forward duplicates the requests the keys did not route through the
// router through it, answering whether they should adopt it; reverse duplicates the
diff --git a/litellm/proxy/db/autorouter_session_rollup.py b/litellm/proxy/db/autorouter_session_rollup.py
index 0d812ee812a..dd08cfd1bef 100644
--- a/litellm/proxy/db/autorouter_session_rollup.py
+++ b/litellm/proxy/db/autorouter_session_rollup.py
@@ -4,7 +4,7 @@ Per-session auto-router benchmarks rollup.
At request time the spend writer builds one AutoRouterTurnTransaction per successful
auto-routed request (a request whose metadata carries a routing_decision) and queues it
on the prisma client. The spend-log flush job drains the queue into
-LiteLLM_AutoRouterSession with one conditional upsert per turn: the statement classifies
+key and user session rollups with one atomic statement per turn: each upsert classifies
the turn (same model, first visit, return to a model the session already used, out of
order) against the row's own columns, so nothing is read before the write and concurrent
pods compose. The benchmarks endpoint aggregates these rows and never touches
@@ -35,10 +35,27 @@ if TYPE_CHECKING:
CACHE_TTL_5M_SECONDS: Final = 300
CACHE_TTL_1H_SECONDS: Final = 3600
-AUTOROUTER_BENCHMARKS_SQL: Final = """
+_SESSION_COLUMNS: Final = """
+ api_key, session_id, router_name, router_type, first_turn_at, last_turn_at,
+ last_model, models, turns, unordered_turns, covered_turns, cache_hits,
+ same_model_turns, same_model_hits, first_visit_turns, first_visit_hits,
+ return_turns, return_hits, return_expired_misses, return_within_ttl_misses,
+ ttl_5m_turns, ttl_1h_turns, total_tokens, spend, saved_spend, classifier_cost, classifier_cost_recorded_turns, tier_turns,
+ baseline_models, savings_estimated_turns, savings_estimated_actual_spend, savings_estimated_saved_spend,
+ savings_estimated_baseline_models
+"""
+
+AUTOROUTER_BENCHMARKS_SQL: Final = f"""
WITH windowed AS (
- SELECT * FROM "LiteLLM_AutoRouterSession"
- WHERE last_turn_at >= $1::timestamp
+ SELECT {_SESSION_COLUMNS} FROM "LiteLLM_AutoRouterSession"
+ WHERE $4::text IS NULL
+ AND last_turn_at >= $1::timestamp
+ AND first_turn_at < $2::timestamp
+ AND ($3::text IS NULL OR api_key = $3::text)
+ UNION ALL
+ SELECT {_SESSION_COLUMNS} FROM "LiteLLM_AutoRouterUserSession"
+ WHERE (($4::text IS NOT NULL AND user_id = $4::text) OR ($4::text IS NULL AND api_key = ''))
+ AND last_turn_at >= $1::timestamp
AND first_turn_at < $2::timestamp
AND ($3::text IS NULL OR api_key = $3::text)
),
@@ -53,7 +70,7 @@ tier_maps AS (
)
SELECT
agg.*,
- COALESCE(tier_maps.tier_turns, '{}'::jsonb) AS tier_turns
+ COALESCE(tier_maps.tier_turns, '{{}}'::jsonb) AS tier_turns
FROM (
SELECT
router_name,
@@ -111,6 +128,7 @@ class AutoRouterTurnTransaction:
savings_estimated_turns: int = 0
savings_estimated_actual_spend: float = 0.0
savings_estimated_saved_spend: float = 0.0
+ user_id: str = ""
class TurnCacheFacts(NamedTuple):
@@ -214,10 +232,11 @@ def build_autorouter_turn_transaction(
if not isinstance(routing_decision, Mapping) or not routing_decision:
return None
router_name: Final = routing_decision.get("router_model_name") or payload.get("model_group")
- api_key: Final = payload.get("api_key")
+ api_key: Final = payload.get("api_key") or ""
+ user_id: Final = payload.get("user") or ""
session_id: Final = payload.get("session_id")
model: Final = payload.get("model")
- if not (isinstance(router_name, str) and router_name and api_key and session_id and model):
+ if not (isinstance(router_name, str) and router_name and (api_key or user_id) and session_id and model):
return None
turn_at: Final = _turn_time_utc(str(payload.get("startTime") or ""))
if turn_at is None:
@@ -236,6 +255,7 @@ def build_autorouter_turn_transaction(
estimated_savings: Final = recorded_estimated_autorouter_savings(metadata)
return AutoRouterTurnTransaction(
api_key=api_key,
+ user_id=user_id,
session_id=bounded_session_id(session_id),
router_name=router_name,
router_type=str(routing_decision.get("router_type") or "unknown"),
@@ -293,18 +313,18 @@ _RETURN_MISS: Final = (
_IDLE_SECONDS: Final = f"EXTRACT(EPOCH FROM {_TURN_AT}::timestamp) - (t.models -> {_MODEL} ->> 'at')::float8"
_CACHE_TOUCHED: Final = f"{_TOUCHED}::int = 1"
-UPSERT_AUTOROUTER_SESSION_SQL: Final = f"""
-INSERT INTO "LiteLLM_AutoRouterSession" AS t (
- api_key, session_id, router_name, router_type, first_turn_at, last_turn_at,
- last_model, models, turns, unordered_turns, covered_turns, cache_hits,
- same_model_turns, same_model_hits, first_visit_turns, first_visit_hits,
- return_turns, return_hits, return_expired_misses, return_within_ttl_misses,
- ttl_5m_turns, ttl_1h_turns, total_tokens, spend, saved_spend, classifier_cost, classifier_cost_recorded_turns, tier_turns,
- baseline_models, savings_estimated_turns, savings_estimated_actual_spend, savings_estimated_saved_spend,
- savings_estimated_baseline_models
+
+def _session_upsert_sql(*, user_scoped: bool) -> str:
+ table_name: Final = "LiteLLM_AutoRouterUserSession" if user_scoped else "LiteLLM_AutoRouterSession"
+ user_column: Final = "user_id, " if user_scoped else ""
+ user_value: Final = f"{_p('user_id')}::text, " if user_scoped else ""
+ required_identity: Final = _p("user_id" if user_scoped else "api_key")
+ return f"""
+INSERT INTO "{table_name}" AS t (
+ {user_column}{_SESSION_COLUMNS}
)
-VALUES (
- {_p("api_key")}, {_p("session_id")}, {_p("router_name")}, {_p("router_type")}, {_TURN_AT}::timestamp, {_TURN_AT}::timestamp,
+SELECT
+ {user_value}{_p("api_key")}, {_p("session_id")}, {_p("router_name")}, {_p("router_type")}, {_TURN_AT}::timestamp, {_TURN_AT}::timestamp,
{_MODEL}, jsonb_build_object({_MODEL}, jsonb_build_object('at', EXTRACT(EPOCH FROM {_TURN_AT}::timestamp), 'ttl', {_CACHE_TTL}::int)),
1, 0, {_COVERED}::int, {_CACHE_HIT}::int,
0, 0, 1, {_CACHE_HIT}::int,
@@ -315,8 +335,8 @@ VALUES (
{_p("classifier_cost")}::float8, 1, {_TIER_DELTA}, {_BASELINE_DELTA},
{_p("savings_estimated_turns")}::int, {_p("savings_estimated_actual_spend")}::float8,
{_p("savings_estimated_saved_spend")}::float8, {_ESTIMATED_BASELINE_DELTA}
-)
-ON CONFLICT (api_key, session_id, router_name) DO UPDATE SET
+WHERE {required_identity}::text <> ''
+ON CONFLICT ({user_column}api_key, session_id, router_name) DO UPDATE SET
turns = t.turns + 1,
total_tokens = t.total_tokens + EXCLUDED.total_tokens,
spend = t.spend + EXCLUDED.spend,
@@ -365,6 +385,17 @@ ON CONFLICT (api_key, session_id, router_name) DO UPDATE SET
"""
+UPSERT_AUTOROUTER_SESSION_SQL: Final = f"""
+WITH key_rollup AS (
+ {_session_upsert_sql(user_scoped=False)}
+ RETURNING 1
+)
+{_session_upsert_sql(user_scoped=True)}
+"""
+
+UPSERT_AUTOROUTER_USER_SESSION_SQL: Final = _session_upsert_sql(user_scoped=True)
+
+
def _as_sql_param(value: str | float | bool | datetime | None) -> str | float | None:
if isinstance(value, bool):
return int(value)
@@ -377,18 +408,23 @@ def _upsert_params(transaction: AutoRouterTurnTransaction) -> tuple[str | float
return tuple(_as_sql_param(getattr(transaction, name)) for name in _UPSERT_PARAM_FIELDS)
-async def write_autorouter_turn(db: SupportsExecuteRaw, transaction: AutoRouterTurnTransaction) -> None:
- await db.execute_raw(UPSERT_AUTOROUTER_SESSION_SQL, *_upsert_params(transaction))
+async def write_autorouter_turn(
+ db: SupportsExecuteRaw,
+ transaction: AutoRouterTurnTransaction,
+ statement: str = UPSERT_AUTOROUTER_SESSION_SQL,
+) -> None:
+ await db.execute_raw(statement, *_upsert_params(transaction))
async def _upsert_turn_with_retry(
prisma_client: PrismaClient,
transaction: AutoRouterTurnTransaction,
n_retry_times: int,
+ statement: str,
) -> None:
for attempt in range(n_retry_times + 1):
try:
- await write_autorouter_turn(prisma_client.db, transaction)
+ await write_autorouter_turn(prisma_client.db, transaction, statement)
except DB_RETRY_SAFE_ERROR_TYPES:
if attempt >= n_retry_times:
raise
@@ -397,6 +433,58 @@ async def _upsert_turn_with_retry(
return
+def _session_partition(transaction: AutoRouterTurnTransaction) -> tuple[str, str, str, str]:
+ identity: Final = ("key", transaction.api_key) if transaction.api_key else ("user", transaction.user_id)
+ return (*identity, transaction.session_id, transaction.router_name)
+
+
+async def _drain_session_partition(
+ prisma_client: PrismaClient,
+ transactions: tuple[AutoRouterTurnTransaction, ...],
+ n_retry_times: int,
+ statement: str,
+) -> tuple[AutoRouterTurnTransaction, ...]:
+ for position, transaction in enumerate(transactions):
+ try:
+ await _upsert_turn_with_retry(prisma_client, transaction, n_retry_times, statement)
+ except Exception as flush_err: # noqa: BLE001 # stop dependent turns without retrying an ambiguous write
+ verbose_proxy_logger.error(
+ "Spend tracking - auto-router session rollup flush failed for router %s; "
+ "%s of %s turn writes stopped in this partition: %s",
+ transaction.router_name,
+ len(transactions) - position,
+ len(transactions),
+ flush_err,
+ )
+ return transactions[position:]
+ return ()
+
+
+async def _flush_session_partition(
+ prisma_client: PrismaClient,
+ transactions: tuple[AutoRouterTurnTransaction, ...],
+ n_retry_times: int,
+) -> None:
+ failed_suffix: Final = await _drain_session_partition(
+ prisma_client, transactions, n_retry_times, UPSERT_AUTOROUTER_SESSION_SQL
+ )
+ if not failed_suffix or not failed_suffix[0].api_key:
+ return
+ failed_user: Final = failed_suffix[0].user_id
+ other_users: Final = sorted(
+ (
+ transaction
+ for transaction in failed_suffix[1:]
+ if transaction.user_id and transaction.user_id != failed_user
+ ),
+ key=lambda transaction: transaction.user_id,
+ )
+ for _, user_turns in groupby(other_users, key=lambda transaction: transaction.user_id):
+ await _drain_session_partition(
+ prisma_client, tuple(user_turns), n_retry_times, UPSERT_AUTOROUTER_USER_SESSION_SQL
+ )
+
+
async def flush_autorouter_turn_transactions(
prisma_client: PrismaClient,
transactions: Sequence[AutoRouterTurnTransaction],
@@ -407,38 +495,20 @@ async def flush_autorouter_turn_transactions(
Statements run sequentially in per-session event order: a turn's classification
depends on the turns before it, and Postgres rejects one multi-row INSERT touching
the same key twice. Only ConnectError is retried, per statement, because it proves
- that statement never reached the database. Any other failure drops the remaining
- turns of THAT session only, with an error log, and the flush continues with the
- next session: sessions are independent state machines, so one poisoned statement
- must not discard unrelated sessions, and a repeated increment is worse than an
- undercount. Callers must not add their own retry around this function.
+ that statement never reached the database. A failed write stops its key and user
+ histories for this batch. Other users sharing that key can still advance their
+ independent user histories, with the key projection disabled and the real key
+ identity preserved. The failed turn is never replayed. Callers must not add their
+ own retry around this function.
"""
if not transactions:
return
ordered: Final = sorted(
transactions,
- key=lambda transaction: (
- transaction.api_key,
- transaction.session_id,
- transaction.router_name,
- transaction.turn_at,
- ),
+ key=lambda transaction: (*_session_partition(transaction), transaction.turn_at),
)
- for session_key, session_group in groupby(
+ for _, session_group in groupby(
ordered,
- key=lambda transaction: (transaction.api_key, transaction.session_id, transaction.router_name),
+ key=_session_partition,
):
- session_turns = tuple(session_group)
- for position, transaction in enumerate(session_turns):
- try:
- await _upsert_turn_with_retry(prisma_client, transaction, n_retry_times)
- except Exception as flush_err: # noqa: BLE001 # a statement failure drops only its session's remainder by design
- verbose_proxy_logger.error(
- "Spend tracking - auto-router session rollup flush failed for router %s; "
- "%s of %s turn transactions dropped for one session: %s",
- session_key[2],
- len(session_turns) - position,
- len(session_turns),
- flush_err,
- )
- break
+ await _flush_session_partition(prisma_client, tuple(session_group), n_retry_times)
diff --git a/litellm/proxy/db/baseline_accounting.py b/litellm/proxy/db/baseline_accounting.py
index 8622cb9e481..4219102d9aa 100644
--- a/litellm/proxy/db/baseline_accounting.py
+++ b/litellm/proxy/db/baseline_accounting.py
@@ -171,6 +171,7 @@ class _Change(BaseModel):
request_id: str
publication: BaselinePublication
api_key: str
+ user_id: str = ""
session_id: str
router_name: str
baseline_model: str
@@ -256,42 +257,54 @@ SET publication = x.publication::text
FROM jsonb_to_recordset($1::jsonb) AS x(request_id text, publication jsonb)
WHERE observations.request_id = x.request_id
"""
-_UPDATE_SESSIONS: Final = """
+
+
+def _session_correction_sql(*, user_scoped: bool) -> str:
+ table_name: Final = "LiteLLM_AutoRouterUserSession" if user_scoped else "LiteLLM_AutoRouterSession"
+ identity_columns: Final = ("user_id, " if user_scoped else "") + "api_key, session_id, router_name"
+ user_filter: Final = "WHERE user_id <> ''" if user_scoped else ""
+ user_match: Final = "session.user_id = totals.user_id AND " if user_scoped else ""
+ return f"""
WITH changes AS (
SELECT * FROM jsonb_to_recordset($1::jsonb) AS x(
- api_key text, session_id text, router_name text, baseline_model text,
+ user_id text, api_key text, session_id text, router_name text, baseline_model text,
covered_delta int, actual_delta float8, savings_delta float8
)
+ {user_filter}
), totals AS (
- SELECT api_key, session_id, router_name, SUM(covered_delta)::int AS covered_delta,
+ SELECT {identity_columns}, SUM(covered_delta)::int AS covered_delta,
SUM(actual_delta) AS actual_delta, SUM(savings_delta) AS savings_delta
- FROM changes GROUP BY api_key, session_id, router_name
+ FROM changes GROUP BY {identity_columns}
), models AS (
- SELECT api_key, session_id, router_name, jsonb_object_agg(baseline_model, delta) AS deltas
+ SELECT {identity_columns}, jsonb_object_agg(baseline_model, delta) AS deltas
FROM (
- SELECT api_key, session_id, router_name, baseline_model, SUM(covered_delta)::int AS delta
- FROM changes GROUP BY api_key, session_id, router_name, baseline_model
- ) grouped GROUP BY api_key, session_id, router_name
+ SELECT {identity_columns}, baseline_model, SUM(covered_delta)::int AS delta
+ FROM changes GROUP BY {identity_columns}, baseline_model
+ ) grouped GROUP BY {identity_columns}
)
-UPDATE "LiteLLM_AutoRouterSession" AS session
+UPDATE "{table_name}" AS session
SET saved_spend = session.saved_spend + totals.savings_delta,
savings_estimated_turns = session.savings_estimated_turns + totals.covered_delta,
savings_estimated_actual_spend = session.savings_estimated_actual_spend + totals.actual_delta,
savings_estimated_saved_spend = session.savings_estimated_saved_spend + totals.savings_delta,
savings_estimated_baseline_models = (
- SELECT COALESCE(jsonb_object_agg(key, value), '{}'::jsonb) FROM (
+ SELECT COALESCE(jsonb_object_agg(key, value), '{{}}'::jsonb) FROM (
SELECT key, SUM(value::int)::int AS value FROM (
SELECT * FROM jsonb_each_text(session.savings_estimated_baseline_models)
UNION ALL SELECT * FROM jsonb_each_text(models.deltas)
) combined GROUP BY key HAVING SUM(value::int) > 0
) counts
)
-FROM totals JOIN models USING (api_key, session_id, router_name)
-WHERE session.api_key = totals.api_key AND session.session_id = totals.session_id
+FROM totals JOIN models USING ({identity_columns})
+WHERE {user_match}session.api_key = totals.api_key AND session.session_id = totals.session_id
AND session.router_name = totals.router_name
"""
+_UPDATE_SESSIONS: Final = _session_correction_sql(user_scoped=False)
+_UPDATE_USER_SESSIONS: Final = _session_correction_sql(user_scoped=True)
+
+
def _primary_transaction(client: PrismaClient) -> _TransactionManager:
primary: Final = cast(_TransactionalDatabase, writer_wrapper(client.db))
return primary.tx(timeout=_TRANSACTION_TIMEOUT)
@@ -308,6 +321,7 @@ def _change(record: BaselineAccountingRecord, old: BaselinePublication | None, n
request_id=record.observation.request_id,
publication=new,
api_key=record.api_key,
+ user_id=record.turn.user_id if record.turn is not None else "",
session_id=record.session_id,
router_name=record.router_name,
baseline_model=record.baseline_model,
@@ -357,6 +371,8 @@ async def _publish(db: SupportsRawQueries, changes: Sequence[_Change]) -> None:
serialized: Final = json.dumps(tuple(change.model_dump(mode="json") for change in changes), separators=(",", ":"))
await db.execute_raw(_UPDATE_LOGS, serialized)
await db.execute_raw(_UPDATE_SESSIONS, serialized)
+ if any(change.user_id for change in changes):
+ await db.execute_raw(_UPDATE_USER_SESSIONS, serialized)
for entity, table in DAILY_SPEND_TABLES.items():
if adjustments := tuple(
change.daily.adjustment(target, change.savings_delta, change.request_id)
diff --git a/litellm/proxy/db/db_transaction_queue/spend_log_cleanup.py b/litellm/proxy/db/db_transaction_queue/spend_log_cleanup.py
index b28a653c9aa..db6045071a8 100644
--- a/litellm/proxy/db/db_transaction_queue/spend_log_cleanup.py
+++ b/litellm/proxy/db/db_transaction_queue/spend_log_cleanup.py
@@ -492,6 +492,18 @@ class SpendLogCleanup:
deadline=deadline,
)
+ async def _delete_old_autorouter_user_session_rows(
+ self, prisma_client: PrismaClient, cutoff_date: datetime, deadline: float
+ ) -> TableCleanupResult:
+ return await self._delete_old_rows_batched(
+ prisma_client,
+ cutoff_date,
+ table_name="LiteLLM_AutoRouterUserSession",
+ key_columns=("user_id", "api_key", "session_id", "router_name"),
+ time_column="last_turn_at",
+ deadline=deadline,
+ )
+
async def _delete_old_health_check_rows(
self, prisma_client: PrismaClient, cutoff_date: datetime, deadline: float
) -> TableCleanupResult:
@@ -560,9 +572,17 @@ class SpendLogCleanup:
)
except Exception: # noqa: BLE001 # retained observations are retried by the next cleanup job
verbose_proxy_logger.warning("Auto-router baseline retention remains pending")
- sessions_result: Final = await self._delete_old_autorouter_session_rows(prisma_client, session_cutoff, deadline)
+ sessions_result: Final = await self._delete_old_autorouter_session_rows(
+ prisma_client, session_cutoff, self._group_deadline(deadline, 2)
+ )
verbose_proxy_logger.info("Deleted %s expired auto-router session rollup rows", sessions_result.rows_deleted)
- return (sessions_result,)
+ user_sessions_result: Final = await self._delete_old_autorouter_user_session_rows(
+ prisma_client, session_cutoff, deadline
+ )
+ verbose_proxy_logger.info(
+ "Deleted %s expired auto-router user session rollup rows", user_sessions_result.rows_deleted
+ )
+ return (sessions_result, user_sessions_result)
async def _clean_health_checks(
self, prisma_client: PrismaClient, retention_seconds: int, deadline: float
diff --git a/litellm/proxy/management_endpoints/auto_router_endpoints.py b/litellm/proxy/management_endpoints/auto_router_endpoints.py
index a6d5a17d73e..32f3bf9accb 100644
--- a/litellm/proxy/management_endpoints/auto_router_endpoints.py
+++ b/litellm/proxy/management_endpoints/auto_router_endpoints.py
@@ -746,14 +746,18 @@ async def get_auto_router_benchmarks(
] = None,
end_date: Annotated[str | None, Query(description="YYYY-MM-DD UTC, inclusive (defaults to today)")] = None,
api_key: Annotated[str | None, Query(description="Filter to one virtual key token hash")] = None,
+ user_id: Annotated[
+ str | None, Query(min_length=1, description="Filter to one canonical internal user recorded on each turn")
+ ] = None,
) -> AutoRouterBenchmarksResponse:
"""
Benchmarks for the auto-router dashboard: session shape, savings against the configured
baseline, and prompt-caching behaviour bucketed by what the router did.
- Reads the LiteLLM_AutoRouterSession rollup, folded once per request at spend-write time,
- so this endpoint never scans LiteLLM_SpendLogs. A session is in the window when it
- overlaps it: its last turn is on or after start_date and its first turn is on or before
+ Reads session rollups folded once per request at spend-write time, so this endpoint
+ never scans LiteLLM_SpendLogs. A user filter selects only turns attributed to that
+ internal user when written; older key-only history remains outside user views. A session
+ is in the window when it overlaps it: its last turn is on or after start_date and its first turn is on or before
end_date. Overall hit rate is over telemetry-bearing turns; each bucket's hit rate is
over that bucket's turns.
@@ -783,6 +787,7 @@ async def get_auto_router_benchmarks(
start_day.isoformat(),
(end_day + timedelta(days=1)).isoformat(),
api_key,
+ user_id,
)
rows: Final = _SESSION_AGG_ROWS.validate_python(raw_rows or ())
groups: Final = (
diff --git a/litellm/proxy/schema.prisma b/litellm/proxy/schema.prisma
index d2032cec0d0..f4015ed9277 100644
--- a/litellm/proxy/schema.prisma
+++ b/litellm/proxy/schema.prisma
@@ -1620,6 +1620,47 @@ model LiteLLM_AutoRouterSession {
@@index([last_turn_at], map: "idx_autorouter_session_last_turn")
}
+model LiteLLM_AutoRouterUserSession {
+ user_id String
+ api_key String
+ session_id String
+ router_name String
+ router_type String
+ first_turn_at DateTime
+ last_turn_at DateTime
+ last_model String
+ models Json @default("{}")
+ turns Int @default(0)
+ unordered_turns Int @default(0)
+ covered_turns Int @default(0)
+ cache_hits Int @default(0)
+ same_model_turns Int @default(0)
+ same_model_hits Int @default(0)
+ first_visit_turns Int @default(0)
+ first_visit_hits Int @default(0)
+ return_turns Int @default(0)
+ return_hits Int @default(0)
+ return_expired_misses Int @default(0)
+ return_within_ttl_misses Int @default(0)
+ ttl_5m_turns Int @default(0)
+ ttl_1h_turns Int @default(0)
+ total_tokens BigInt @default(0)
+ spend Float @default(0)
+ saved_spend Float @default(0)
+ savings_estimated_turns Int @default(0)
+ savings_estimated_actual_spend Float @default(0)
+ savings_estimated_saved_spend Float @default(0)
+ savings_estimated_baseline_models Json @default("{}")
+ classifier_cost Float @default(0)
+ classifier_cost_recorded_turns Int @default(0)
+ tier_turns Json @default("{}")
+ baseline_models Json @default("{}")
+
+ @@id([user_id, api_key, session_id, router_name])
+ @@index([last_turn_at], map: "idx_autorouter_user_session_last_turn")
+ @@index([user_id, last_turn_at], map: "idx_autorouter_user_session_user_last_turn")
+}
+
// Shadow eval: evaluation of an auto-router against one or more keys' live traffic, in
// either direction. forward duplicates the requests the keys did not route through the
// router through it, answering whether they should adopt it; reverse duplicates the
diff --git a/schema.prisma b/schema.prisma
index d2032cec0d0..f4015ed9277 100644
--- a/schema.prisma
+++ b/schema.prisma
@@ -1620,6 +1620,47 @@ model LiteLLM_AutoRouterSession {
@@index([last_turn_at], map: "idx_autorouter_session_last_turn")
}
+model LiteLLM_AutoRouterUserSession {
+ user_id String
+ api_key String
+ session_id String
+ router_name String
+ router_type String
+ first_turn_at DateTime
+ last_turn_at DateTime
+ last_model String
+ models Json @default("{}")
+ turns Int @default(0)
+ unordered_turns Int @default(0)
+ covered_turns Int @default(0)
+ cache_hits Int @default(0)
+ same_model_turns Int @default(0)
+ same_model_hits Int @default(0)
+ first_visit_turns Int @default(0)
+ first_visit_hits Int @default(0)
+ return_turns Int @default(0)
+ return_hits Int @default(0)
+ return_expired_misses Int @default(0)
+ return_within_ttl_misses Int @default(0)
+ ttl_5m_turns Int @default(0)
+ ttl_1h_turns Int @default(0)
+ total_tokens BigInt @default(0)
+ spend Float @default(0)
+ saved_spend Float @default(0)
+ savings_estimated_turns Int @default(0)
+ savings_estimated_actual_spend Float @default(0)
+ savings_estimated_saved_spend Float @default(0)
+ savings_estimated_baseline_models Json @default("{}")
+ classifier_cost Float @default(0)
+ classifier_cost_recorded_turns Int @default(0)
+ tier_turns Json @default("{}")
+ baseline_models Json @default("{}")
+
+ @@id([user_id, api_key, session_id, router_name])
+ @@index([last_turn_at], map: "idx_autorouter_user_session_last_turn")
+ @@index([user_id, last_turn_at], map: "idx_autorouter_user_session_user_last_turn")
+}
+
// Shadow eval: evaluation of an auto-router against one or more keys' live traffic, in
// either direction. forward duplicates the requests the keys did not route through the
// router through it, answering whether they should adopt it; reverse duplicates the
diff --git a/tests/proxy_behavior/spend/test_autorouter_session_rollup.py b/tests/proxy_behavior/spend/test_autorouter_session_rollup.py
index f3c68b489a5..77549b527d8 100644
--- a/tests/proxy_behavior/spend/test_autorouter_session_rollup.py
+++ b/tests/proxy_behavior/spend/test_autorouter_session_rollup.py
@@ -6,17 +6,24 @@ tests/test_litellm/proxy/db/test_autorouter_session_rollup.py.
"""
import asyncio
+import time
import uuid
from datetime import datetime, timedelta, timezone
-from typing import Final
+from types import SimpleNamespace
+from typing import Final, TypedDict, cast
import pytest
from prisma import Prisma
+from prisma.errors import RawQueryError
+from typing_extensions import ReadOnly
from litellm.proxy.db.autorouter_session_rollup import (
AUTOROUTER_BENCHMARKS_SQL,
UPSERT_AUTOROUTER_SESSION_SQL,
+ AutoRouterTurnTransaction,
+ flush_autorouter_turn_transactions,
)
+from litellm.proxy.db.db_transaction_queue.spend_log_cleanup import SpendLogCleanup
pytestmark = pytest.mark.asyncio(loop_scope="session")
@@ -45,6 +52,7 @@ async def _turn(
tier: "str | None" = None,
baseline: "str | None" = None,
estimated: bool = True,
+ user_id: str = "",
) -> None:
touched: Final = 1 if (hit or ttl is not None or not covered) else 0
await db.execute_raw(
@@ -68,6 +76,7 @@ async def _turn(
int(estimated),
spend if estimated else 0.0,
saved if estimated else 0.0,
+ user_id,
)
@@ -217,7 +226,7 @@ async def test_subtotal_coverage_survives_legacy_and_rolling_writers(db, writers
assert row["savings_estimated_actual_spend"] == pytest.approx(0.01 * sum(writers))
assert row["savings_estimated_saved_spend"] == pytest.approx(0.02 * sum(writers))
groups: Final = await db.query_raw(
- AUTOROUTER_BENCHMARKS_SQL, T0.isoformat(), (T0 + timedelta(days=1)).isoformat(), key
+ AUTOROUTER_BENCHMARKS_SQL, T0.isoformat(), (T0 + timedelta(days=1)).isoformat(), key, None
)
assert len(groups) == 1
assert groups[0]["classifier_cost"] == row["classifier_cost"]
@@ -242,7 +251,7 @@ async def test_unknown_and_legacy_turns_preserve_actual_spend_without_entering_t
assert row["saved_spend"] == pytest.approx(-0.03)
assert row["savings_estimated_baseline_models"] == {"opus": 1}
groups: Final = await db.query_raw(
- AUTOROUTER_BENCHMARKS_SQL, T0.isoformat(), (T0 + timedelta(days=1)).isoformat(), key
+ AUTOROUTER_BENCHMARKS_SQL, T0.isoformat(), (T0 + timedelta(days=1)).isoformat(), key, None
)
assert len(groups) == 1
for actual in (row, groups[0]):
@@ -277,6 +286,7 @@ async def test_the_benchmarks_aggregate_reads_only_overlapping_sessions(db):
(T0 - timedelta(days=1)).isoformat(),
(T0 + timedelta(days=1)).isoformat(),
None,
+ None,
)
matching = [row for row in rows if row["router_name"] == router]
assert len(matching) == 1
@@ -304,6 +314,7 @@ async def test_the_benchmarks_aggregate_can_filter_to_one_key(db):
(T0 - timedelta(days=1)).isoformat(),
(T0 + timedelta(days=1)).isoformat(),
first_key,
+ None,
)
matching = [row for row in rows if row["router_name"] == router]
assert len(matching) == 1
@@ -317,10 +328,160 @@ async def test_the_benchmarks_aggregate_can_filter_to_one_key(db):
(T0 - timedelta(days=1)).isoformat(),
(T0 + timedelta(days=1)).isoformat(),
f"k-{uuid.uuid4()}",
+ None,
)
assert [row for row in unknown_key_rows if row["router_name"] == router] == []
+class _BenchmarkRow(TypedDict):
+ sessions: ReadOnly[int]
+ turns: ReadOnly[int]
+ same_model_turns: ReadOnly[int]
+ first_visit_turns: ReadOnly[int]
+ spend: ReadOnly[float]
+ saved_spend: ReadOnly[float]
+ tier_turns: ReadOnly[dict[str, int]]
+ cache_hits: ReadOnly[int]
+ savings_estimated_turns: ReadOnly[int]
+ savings_estimated_actual_spend: ReadOnly[float]
+ savings_estimated_saved_spend: ReadOnly[float]
+
+
+async def _scoped_benchmarks(
+ db: Prisma, router: str, user_id: str | None = None, key: str | None = None
+) -> tuple[_BenchmarkRow, ...]:
+ rows: Final = await db.query_raw(
+ AUTOROUTER_BENCHMARKS_SQL,
+ (T0 - timedelta(days=1)).isoformat(),
+ (T0 + timedelta(days=1)).isoformat(),
+ key,
+ user_id,
+ )
+ return tuple(cast(_BenchmarkRow, row) for row in rows if row["router_name"] == router)
+
+
+async def test_users_keep_written_identity_across_shared_keys_and_keyless_sessions(db: Prisma) -> None:
+ router: Final = f"r-{uuid.uuid4()}"
+ alice: Final = f"u-{uuid.uuid4()}"
+ bob: Final = f"u-{uuid.uuid4()}"
+ first_key: Final = f"k-{uuid.uuid4()}"
+ second_key: Final = f"k-{uuid.uuid4()}"
+ await _legacy_turn(db, first_key, T0, router=router)
+ await _turn(db, first_key, "A", T0 + timedelta(seconds=10), router=router, user_id=alice, tier="simple")
+ await _turn(
+ db, first_key, "B", T0 + timedelta(seconds=20), router=router, user_id=bob, spend=0.03, saved=0.06, tier="complex"
+ )
+ await _turn(db, second_key, "C", T0, router=router, user_id=alice, spend=0.02, saved=0.04)
+ await _turn(db, "", "A", T0, router=router, user_id=alice, ttl=300)
+ await _turn(db, "", "A", T0 + timedelta(seconds=1), router=router, user_id=alice, hit=1)
+ await _turn(db, "", "B", T0, router=router, user_id=bob, spend=0.04, saved=0.08)
+ await _turn(db, second_key, "C", T0 - timedelta(days=40), router=router, user_id=alice, session_id="expired")
+
+ alice_rows: Final = await _scoped_benchmarks(db, router, user_id=alice)
+ bob_rows: Final = await _scoped_benchmarks(db, router, user_id=bob)
+ global_rows: Final = await _scoped_benchmarks(db, router)
+ key_rows: Final = await _scoped_benchmarks(db, router, key=first_key)
+ intersection: Final = await _scoped_benchmarks(db, router, user_id=alice, key=first_key)
+ assert len(alice_rows) == len(bob_rows) == len(global_rows) == len(key_rows) == len(intersection) == 1
+ assert (alice_rows[0]["sessions"], alice_rows[0]["turns"], alice_rows[0]["same_model_turns"]) == (3, 4, 1)
+ assert (bob_rows[0]["sessions"], bob_rows[0]["turns"], bob_rows[0]["first_visit_turns"]) == (2, 2, 2)
+ assert alice_rows[0]["spend"] == pytest.approx(0.05)
+ assert bob_rows[0]["spend"] == pytest.approx(0.07)
+ assert alice_rows[0]["tier_turns"] == {"simple": 1}
+ assert bob_rows[0]["tier_turns"] == {"complex": 1}
+ assert (alice_rows[0]["cache_hits"], bob_rows[0]["cache_hits"]) == (1, 0)
+ assert (global_rows[0]["sessions"], global_rows[0]["turns"]) == (4, 7)
+ assert (alice_rows[0]["savings_estimated_turns"], bob_rows[0]["savings_estimated_turns"]) == (4, 2)
+ assert global_rows[0]["savings_estimated_turns"] == 6
+ for scoped in (alice_rows[0], bob_rows[0]):
+ assert scoped["savings_estimated_actual_spend"] == pytest.approx(scoped["spend"])
+ assert scoped["savings_estimated_saved_spend"] == pytest.approx(scoped["saved_spend"])
+ assert global_rows[0]["spend"] == pytest.approx(alice_rows[0]["spend"] + bob_rows[0]["spend"] + 0.01)
+ assert global_rows[0]["saved_spend"] == pytest.approx(alice_rows[0]["saved_spend"] + bob_rows[0]["saved_spend"] + 0.02)
+ assert global_rows[0]["tier_turns"] == {"simple": 1, "complex": 1}
+ assert (key_rows[0]["sessions"], key_rows[0]["turns"]) == (1, 3)
+ assert key_rows[0]["spend"] == pytest.approx(0.05)
+ assert (intersection[0]["sessions"], intersection[0]["turns"]) == (1, 1)
+ assert intersection[0]["spend"] == pytest.approx(0.01)
+ assert await _scoped_benchmarks(db, router, user_id=bob, key=second_key) == ()
+ assert await _scoped_benchmarks(db, router, user_id=f"u-{uuid.uuid4()}") == ()
+ assert await _scoped_benchmarks(db, router, user_id="") == ()
+
+
+async def test_a_failed_user_projection_rolls_back_the_keys_increment(db: Prisma) -> None:
+ key: Final = f"k-{uuid.uuid4()}"
+ user_id: Final = "".join(str(uuid.uuid4()) for _ in range(200))
+ await _turn(db, key, "A", T0)
+ before: Final = await _row(db, key)
+
+ with pytest.raises(RawQueryError, match=r"index row (requires|size)"):
+ await _turn(db, key, "B", T0 + timedelta(seconds=1), user_id=user_id)
+
+ assert await _row(db, key) == before
+ assert await db.query_raw('SELECT user_id FROM "LiteLLM_AutoRouterUserSession" WHERE user_id = $1', user_id) == []
+
+ first_user: Final = f"u-{uuid.uuid4()}"
+ second_user: Final = f"u-{uuid.uuid4()}"
+ turns: Final = tuple(
+ AutoRouterTurnTransaction(
+ api_key=key,
+ user_id=user,
+ session_id="s1",
+ router_name="auto-1",
+ router_type="complexity",
+ model=model,
+ turn_at=T0 + timedelta(seconds=second),
+ total_tokens=100,
+ spend=0.01,
+ saved_spend=0.02,
+ classifier_cost=0.0,
+ covered=True,
+ cache_hit=False,
+ cache_ttl_seconds=None,
+ cache_touched=False,
+ )
+ for user, model, second in (
+ (first_user, "A", 1),
+ (user_id, "B", 2),
+ (first_user, "B", 3),
+ (second_user, "C", 4),
+ (first_user, "B", 5),
+ (second_user, "C", 6),
+ (user_id, "A", 7),
+ )
+ )
+ await flush_autorouter_turn_transactions(SimpleNamespace(db=db), tuple(reversed(turns)), n_retry_times=0)
+
+ key_row: Final = await _row(db, key)
+ assert (key_row["turns"], key_row["last_model"], key_row["unordered_turns"]) == (2, "A", 0)
+ assert key_row["spend"] == pytest.approx(0.02)
+ user_rows: Final = await db.query_raw('SELECT * FROM "LiteLLM_AutoRouterUserSession" WHERE api_key = $1', key)
+ by_user: Final = {row["user_id"]: row for row in user_rows}
+ assert set(by_user) == {first_user, second_user}
+ for user, count, model in ((first_user, 3, "B"), (second_user, 2, "C")):
+ row: Final = by_user[user]
+ assert (row["turns"], row["same_model_turns"], row["unordered_turns"], row["last_model"]) == (count, 1, 0, model)
+ assert row["spend"] == pytest.approx(count * 0.01)
+ assert row["saved_spend"] == pytest.approx(count * 0.02)
+
+
+async def test_user_session_cleanup_keeps_another_users_recent_keyless_session(db: Prisma) -> None:
+ router: Final = f"r-{uuid.uuid4()}"
+ expired_user: Final = f"u-{uuid.uuid4()}"
+ recent_user: Final = f"u-{uuid.uuid4()}"
+ await _turn(db, "", "A", T0 - timedelta(days=1), router=router, user_id=expired_user)
+ await _turn(db, "", "A", T0 + timedelta(days=1), router=router, user_id=recent_user)
+ cleaner: Final = SpendLogCleanup(general_settings={})
+
+ await cleaner._delete_old_autorouter_user_session_rows(
+ SimpleNamespace(db=db), T0.replace(tzinfo=timezone.utc), time.monotonic() + 60
+ )
+
+ assert await db.query_raw(
+ 'SELECT user_id, turns FROM "LiteLLM_AutoRouterUserSession" WHERE router_name = $1', router
+ ) == [{"user_id": recent_user, "turns": 1}]
+
+
async def test_a_reconfigured_alias_reports_each_router_type_as_its_own_group(db):
key = f"k-{uuid.uuid4()}"
router = f"r-{uuid.uuid4()}"
@@ -334,6 +495,7 @@ async def test_a_reconfigured_alias_reports_each_router_type_as_its_own_group(db
(T0 - timedelta(days=1)).isoformat(),
(T0 + timedelta(days=1)).isoformat(),
None,
+ None,
)
matching = sorted(
(row for row in rows if row["router_name"] == router),
@@ -418,6 +580,7 @@ async def test_the_benchmarks_aggregate_sums_tier_turns_across_sessions(db):
(T0 - timedelta(days=1)).isoformat(),
(T0 + timedelta(days=1)).isoformat(),
None,
+ None,
)
grouped = next(row for row in rows if row["router_name"] == router)
assert grouped["tier_turns"] == {"simple": 2, "complex": 1}
@@ -446,6 +609,7 @@ async def test_tier_maps_stay_separate_per_router_type_on_a_reconfigured_alias(d
(T0 - timedelta(days=1)).isoformat(),
(T0 + timedelta(days=1)).isoformat(),
None,
+ None,
)
by_type = {row["router_type"]: row["tier_turns"] for row in rows if row["router_name"] == router}
assert by_type == {"complexity": {"medium": 1}, "quality": {"2": 1}}
@@ -461,6 +625,7 @@ async def test_a_window_with_no_tiered_turns_aggregates_to_an_empty_map(db):
(T0 - timedelta(days=1)).isoformat(),
(T0 + timedelta(days=1)).isoformat(),
None,
+ None,
)
grouped = next(row for row in rows if row["router_name"] == router)
assert grouped["tier_turns"] == {}
diff --git a/tests/proxy_behavior/spend/test_baseline_accounting.py b/tests/proxy_behavior/spend/test_baseline_accounting.py
index e187a44c29d..3504751d132 100644
--- a/tests/proxy_behavior/spend/test_baseline_accounting.py
+++ b/tests/proxy_behavior/spend/test_baseline_accounting.py
@@ -56,7 +56,9 @@ def record() -> Callable[..., BaselineAccountingRecord]:
},
)
- def create(label: str = "first", started: float = 10000.0, identical: bool = True) -> BaselineAccountingRecord:
+ def create(
+ label: str = "first", started: float = 10000.0, identical: bool = True, user_id: str = ""
+ ) -> BaselineAccountingRecord:
return BaselineAccountingRecord(
scope="autorouter-baseline:v3:" + run * 2, api_key=run, session_id=run,
router_name="test-router", baseline_model="anthropic/claude-opus-5",
@@ -76,6 +78,7 @@ def record() -> Callable[..., BaselineAccountingRecord]:
total_tokens=6230, spend=0.17, saved_spend=0.0, classifier_cost=0.0,
covered=True, cache_hit=False, cache_ttl_seconds=3600, cache_touched=True,
baseline_model="anthropic/claude-opus-5",
+ user_id=user_id,
),
daily=DailyBaselineAttribution(
date="2026-09-15", api_key=run, model="claude-opus-5", custom_llm_provider="anthropic",
@@ -99,21 +102,36 @@ async def _session(db: Prisma, record: BaselineAccountingRecord):
return rows[0]
+async def _user_sessions(db: Prisma, record: BaselineAccountingRecord) -> dict[str, dict[str, object]]:
+ rows: Final = await db.query_raw('SELECT * FROM "LiteLLM_AutoRouterUserSession" WHERE api_key=$1', record.api_key)
+ return {str(row["user_id"]): row for row in rows}
+
+
async def test_late_replay_updates_all_projections_without_rebilling(db: Prisma, record: Callable[..., BaselineAccountingRecord]) -> None:
store: Final = _store(db)
- late: Final = record("late", 10001.0)
- early: Final = record("early", identical=False)
+ late: Final = record("late", 10001.0, user_id="late-user")
+ early: Final = record("early", identical=False, user_id="early-user")
await _log(db, late)
assert await store.append(late) == "recorded"
assert await store.project(late.scope) == "published"
before: Final = await _session(db, late)
assert before["savings_estimated_actual_spend"] == before["spend"] == 0.17
assert before["saved_spend"] == 0.0
+ before_users: Final = await _user_sessions(db, late)
+ assert set(before_users) == {"late-user"}
+ assert before_users["late-user"]["savings_estimated_turns"] == 1
+ assert before_users["late-user"]["savings_estimated_baseline_models"] == {late.baseline_model: 1}
await _log(db, early)
assert await store.append(early) == "recorded"
pending: Final = await _session(db, late)
assert pending["spend"] == 0.34 and pending["savings_estimated_turns"] == 0
assert pending["saved_spend"] == pending["savings_estimated_actual_spend"] == 0.0
+ pending_users: Final = await _user_sessions(db, late)
+ assert set(pending_users) == {"late-user", "early-user"}
+ for user in pending_users.values():
+ assert user["turns"] == 1 and user["spend"] == 0.17
+ assert user["savings_estimated_turns"] == user["savings_estimated_actual_spend"] == user["saved_spend"] == 0
+ assert user["savings_estimated_baseline_models"] == {}
waiting: Final = await db.query_raw('SELECT metadata FROM "LiteLLM_SpendLogs" WHERE request_id=$1', late.observation.request_id)
assert waiting[0]["metadata"]["autorouter_savings"] is None
assert waiting[0]["metadata"]["autorouter_savings_estimate"]["reason"] == "pending_projection"
@@ -125,37 +143,69 @@ async def test_late_replay_updates_all_projections_without_rebilling(db: Prisma,
assert logs[0]["spend"] == 0.17
assert logs[0]["metadata"]["autorouter_savings_estimate"]["provenance"] == "modeled"
assert after["saved_spend"] == pytest.approx(logs[0]["metadata"]["autorouter_savings"])
+ after_users: Final = await _user_sessions(db, late)
+ assert after_users["early-user"] == pending_users["early-user"]
+ for field in (
+ "saved_spend", "savings_estimated_turns", "savings_estimated_actual_spend",
+ "savings_estimated_saved_spend", "savings_estimated_baseline_models",
+ ):
+ assert after_users["late-user"][field] == after[field]
+ assert after_users["late-user"]["turns"] == 1 and after_users["late-user"]["spend"] == 0.17
for table in ("DailyUserSpend", "DailyTeamSpend", "DailyOrganizationSpend", "DailyEndUserSpend", "DailyAgentSpend", "DailyTagSpend"):
rows: Final = await db.query_raw(f'SELECT spend,api_requests,autorouter_savings_spend FROM "LiteLLM_{table}" WHERE api_key=$1', late.api_key)
assert rows[0]["spend"] == rows[0]["api_requests"] == 0
assert rows[0]["autorouter_savings_spend"] == pytest.approx(after["saved_spend"])
-async def test_commit_ack_loss_and_concurrent_duplicate_delivery_are_idempotent(db: Prisma, record: Callable[..., BaselineAccountingRecord]) -> None:
- event: Final = record()
+@pytest.mark.parametrize("attributed", [True, False])
+async def test_commit_ack_loss_and_concurrent_duplicate_delivery_are_idempotent(
+ db: Prisma, record: Callable[..., BaselineAccountingRecord], attributed: bool
+) -> None:
+ event: Final = record(user_id="first-user" if attributed else "")
+ other: Final = record("other", 10001.0, user_id="second-user" if attributed else "")
await _log(db, event)
assert await _store(db, after_commit=True).append(event) == "unavailable"
store: Final = _store(db)
assert set(await asyncio.gather(*(store.append(event) for _ in range(4)))) == {"recorded"}
+ await _log(db, other)
+ assert await store.append(other) == "recorded"
+ if not attributed:
+ await db.execute_raw(
+ 'UPDATE "LiteLLM_AutoRouterBaselineObservation" SET data=(data::jsonb #- \'{turn,user_id}\')::text WHERE scope=$1',
+ event.scope,
+ )
assert await store.project(event.scope) == "published"
assert await store.project(event.scope) == "unchanged"
session: Final = await _session(db, event)
- assert session["turns"] == session["savings_estimated_turns"] == 1
- assert session["spend"] == session["savings_estimated_actual_spend"] == 0.17
+ assert session["turns"] == session["savings_estimated_turns"] == 2
+ assert session["spend"] == session["savings_estimated_actual_spend"] == 0.34
+ users: Final = await _user_sessions(db, event)
+ assert set(users) == ({"first-user", "second-user"} if attributed else set())
+ for user in users.values():
+ assert user["turns"] == user["savings_estimated_turns"] == 1
+ assert user["spend"] == user["savings_estimated_actual_spend"] == 0.17
+ assert user["savings_estimated_baseline_models"] == {event.baseline_model: 1}
async def test_publication_rollback_keeps_dirty_revision_for_retry(db: Prisma, record: Callable[..., BaselineAccountingRecord]) -> None:
- event: Final = record()
+ event: Final = record(user_id="rollback-user")
await _log(db, event)
store: Final = _store(db)
assert await store.append(event) == "recorded"
assert await _store(db, before_commit=True).project(event.scope) == "unavailable"
session: Final = await _session(db, event)
assert session["spend"] == 0.17 and session["savings_estimated_turns"] == 0
+ before_users: Final = await _user_sessions(db, event)
+ assert before_users["rollback-user"]["spend"] == 0.17
+ assert before_users["rollback-user"]["savings_estimated_turns"] == 0
+ assert before_users["rollback-user"]["savings_estimated_baseline_models"] == {}
revisions: Final = await db.query_raw('SELECT revision,published_revision FROM "LiteLLM_AutoRouterBaselineComparison" WHERE scope=$1', event.scope)
assert revisions[0]["revision"] > revisions[0]["published_revision"]
assert await store.project(event.scope) == "published"
assert (await _session(db, event))["savings_estimated_turns"] == 1
+ after_users: Final = await _user_sessions(db, event)
+ assert after_users["rollback-user"]["turns"] == after_users["rollback-user"]["savings_estimated_turns"] == 1
+ assert after_users["rollback-user"]["spend"] == after_users["rollback-user"]["savings_estimated_actual_spend"] == 0.17
async def test_conflicting_duplicate_cannot_restore_an_observed_estimate(db: Prisma, record: Callable[..., BaselineAccountingRecord]) -> None:
@@ -196,6 +246,7 @@ async def test_native_observation_enters_spend_pipeline_once_with_shared_daily_a
db: Prisma, record: Callable[..., BaselineAccountingRecord], monkeypatch: pytest.MonkeyPatch,
) -> None:
import os
+
from litellm.proxy.common_utils.user_api_key_cache import UserApiKeyCache
from litellm.proxy.db.db_spend_update_writer import DBSpendUpdateWriter
from litellm.proxy.hooks.autorouter_baseline_cache import CapturedBaselineObservation
diff --git a/tests/test_litellm/proxy/db/test_autorouter_session_rollup.py b/tests/test_litellm/proxy/db/test_autorouter_session_rollup.py
index acd3dc18b54..c61a489f894 100644
--- a/tests/test_litellm/proxy/db/test_autorouter_session_rollup.py
+++ b/tests/test_litellm/proxy/db/test_autorouter_session_rollup.py
@@ -56,6 +56,31 @@ def _build(payload: dict | None = None, metadata: dict | None = None):
class TestBuildTransaction:
+ @pytest.mark.parametrize(
+ "api_key, user_id, included",
+ [
+ ("hashed-key", "canonical-user", True),
+ ("hashed-key", None, True),
+ ("hashed-key", "", True),
+ ("", "canonical-user", True),
+ ("", None, False),
+ ("", "", False),
+ ],
+ )
+ def test_attribution_uses_the_canonical_user_even_without_a_key(
+ self, api_key: str, user_id: str | None, included: bool
+ ) -> None:
+ transaction: Final = _build(
+ payload=_payload(api_key=api_key, user=user_id),
+ metadata=_metadata(user="client-user", user_api_key_user_id="metadata-user"),
+ )
+ if not included:
+ assert transaction is None
+ return
+ assert transaction is not None
+ assert transaction.api_key == api_key
+ assert transaction.user_id == (user_id or "")
+
def test_successful_auto_routed_turn_builds_every_field(self):
transaction = _build(
metadata=_metadata(
@@ -205,23 +230,43 @@ class TestBuildTransaction:
class _FakeDB:
- def __init__(self, failures: "list[Exception] | None" = None, poison_session: str | None = None):
+ def __init__(
+ self,
+ failures: "list[Exception] | None" = None,
+ poison_session: str | None = None,
+ poison_user: str | None = None,
+ commit_then_error_users: frozenset[str] = frozenset(),
+ ):
self.calls: list[tuple] = []
+ self.attempts: list[tuple[str, tuple[object, ...]]] = []
self._failures = list(failures or [])
self._poison_session = poison_session
+ self._poison_user = poison_user
+ self._commit_then_error_users = commit_then_error_users
async def execute_raw(self, sql: str, *params: object) -> int:
+ self.attempts.append((sql, params))
if self._poison_session is not None and params[1] == self._poison_session:
raise RuntimeError("index row size exceeds btree maximum")
+ if self._poison_user is not None and params[19] == self._poison_user:
+ raise RuntimeError("index row size exceeds btree maximum")
if self._failures:
raise self._failures.pop(0)
self.calls.append((sql, params))
+ if params[19] in self._commit_then_error_users:
+ raise RuntimeError("commit succeeded but acknowledgement was lost")
return 1
class _FakeClient:
- def __init__(self, failures: "list[Exception] | None" = None, poison_session: str | None = None):
- self.db = _FakeDB(failures, poison_session)
+ def __init__(
+ self,
+ failures: "list[Exception] | None" = None,
+ poison_session: str | None = None,
+ poison_user: str | None = None,
+ commit_then_error_users: frozenset[str] = frozenset(),
+ ):
+ self.db = _FakeDB(failures, poison_session, poison_user, commit_then_error_users)
def _transaction(
@@ -229,9 +274,11 @@ def _transaction(
at: datetime = datetime(2026, 8, 1, 12, 0, 0),
tier: str | None = "medium",
baseline_model: str | None = "anthropic/claude-opus-5",
+ api_key: str = "k1",
+ user_id: str = "",
) -> AutoRouterTurnTransaction:
return AutoRouterTurnTransaction(
- api_key="k1",
+ api_key=api_key,
session_id=session_id,
router_name="live-auto",
router_type="complexity",
@@ -247,6 +294,7 @@ def _transaction(
cache_touched=False,
tier=tier,
baseline_model=baseline_model,
+ user_id=user_id,
)
@@ -261,7 +309,7 @@ class TestFlush:
def test_params_marshal_in_statement_order(self):
client = _FakeClient()
- asyncio.run(flush_autorouter_turn_transactions(client, [_transaction()]))
+ asyncio.run(flush_autorouter_turn_transactions(client, [_transaction(user_id="canonical-user")]))
sql, params = client.db.calls[0]
assert sql == UPSERT_AUTOROUTER_SESSION_SQL
assert params == (
@@ -284,8 +332,65 @@ class TestFlush:
0,
0.0,
0.0,
+ "canonical-user",
)
+ def test_a_keys_turns_stay_chronological_when_its_canonical_user_changes(self) -> None:
+ client: Final = _FakeClient()
+ earlier: Final = _transaction(user_id="z-user", at=datetime(2026, 8, 1, 12, 0, 0))
+ later: Final = _transaction(user_id="a-user", at=datetime(2026, 8, 1, 12, 0, 10))
+ asyncio.run(flush_autorouter_turn_transactions(client, [later, earlier]))
+ assert [(params[5], params[19]) for _, params in client.db.calls] == [
+ ("2026-08-01T12:00:00", "z-user"),
+ ("2026-08-01T12:00:10", "a-user"),
+ ]
+
+ def test_one_keyless_users_failed_session_does_not_drop_another_users_turn(self) -> None:
+ client: Final = _FakeClient(poison_user="a-user")
+ failed: Final = _transaction(api_key="", user_id="a-user")
+ other: Final = _transaction(api_key="", user_id="b-user", at=datetime(2026, 8, 1, 12, 0, 10))
+ asyncio.run(flush_autorouter_turn_transactions(client, [other, failed]))
+ assert [(params[0], params[1], params[19]) for _, params in client.db.calls] == [("", "s1", "b-user")]
+
+ def test_uncertain_commits_quarantine_only_the_key_and_each_failed_user(self) -> None:
+ client: Final = _FakeClient(commit_then_error_users=frozenset({"a-failed", "c-failed"}))
+ turns: Final = tuple(
+ _transaction(user_id=user, at=datetime(2026, 8, 1, 12, 0, second), api_key=key)
+ for user, second, key in (
+ ("b-healthy", 0, "k1"),
+ ("a-failed", 1, "k1"),
+ ("b-healthy", 2, "k1"),
+ ("c-failed", 3, "k1"),
+ ("b-healthy", 4, "k1"),
+ ("d-healthy", 5, "k1"),
+ ("c-failed", 6, "k1"),
+ ("d-healthy", 7, "k1"),
+ ("a-failed", 8, "k1"),
+ ("", 9, "k1"),
+ ("z-other", 10, "k2"),
+ )
+ )
+ asyncio.run(flush_autorouter_turn_transactions(client, tuple(reversed(turns))))
+
+ assert client.db.attempts == client.db.calls
+ assert [
+ (params[0], params[19], params[5])
+ for sql, params in client.db.calls
+ if sql == UPSERT_AUTOROUTER_SESSION_SQL
+ ] == [
+ ("k1", "b-healthy", "2026-08-01T12:00:00"),
+ ("k1", "a-failed", "2026-08-01T12:00:01"),
+ ("k2", "z-other", "2026-08-01T12:00:10"),
+ ]
+ assert [params[19] for _, params in client.db.attempts].count("a-failed") == 1
+ assert [params[19] for _, params in client.db.attempts].count("c-failed") == 1
+ for user, seconds in (("b-healthy", (2, 4)), ("c-failed", (3,)), ("d-healthy", (5, 7))):
+ assert [
+ (params[0], params[5])
+ for sql, params in client.db.calls
+ if sql != UPSERT_AUTOROUTER_SESSION_SQL and params[19] == user
+ ] == [("k1", f"2026-08-01T12:00:{second:02d}") for second in seconds]
+
def test_a_connect_error_retries_the_same_statement(self):
client = _FakeClient(failures=[httpx.ConnectError("boom")])
asyncio.run(flush_autorouter_turn_transactions(client, [_transaction()]))
diff --git a/tests/test_litellm/proxy/management_endpoints/test_auto_router_endpoints.py b/tests/test_litellm/proxy/management_endpoints/test_auto_router_endpoints.py
index 6ac053f4e15..d5ddd5a78c3 100644
--- a/tests/test_litellm/proxy/management_endpoints/test_auto_router_endpoints.py
+++ b/tests/test_litellm/proxy/management_endpoints/test_auto_router_endpoints.py
@@ -4,6 +4,7 @@ Unit tests for auto router management endpoints
from collections.abc import Mapping, Sequence
from pathlib import Path
+from types import SimpleNamespace
from typing import Final
import pytest
@@ -654,17 +655,43 @@ class TestAutoRouterBenchmarks:
assert _summed_agg_row([complexity, quality]).tier_turns == {}
@pytest.mark.asyncio
- async def test_non_admin_roles_cannot_read_benchmarks(self):
+ @pytest.mark.parametrize("user_id", [None, "own-user", "other-user"])
+ async def test_non_admin_roles_cannot_read_benchmarks(self, user_id: str | None):
from litellm.proxy.management_endpoints.auto_router_endpoints import get_auto_router_benchmarks
with pytest.raises(HTTPException) as err:
await get_auto_router_benchmarks(
- user_api_key_dict=UserAPIKeyAuth(user_role=LitellmUserRoles.INTERNAL_USER, api_key="sk-x"),
+ user_api_key_dict=UserAPIKeyAuth(
+ user_role=LitellmUserRoles.INTERNAL_USER, api_key="sk-x", user_id="own-user"
+ ),
start_date="2026-08-01",
end_date="2026-08-02",
+ user_id=user_id,
)
assert err.value.status_code == 403
+ @pytest.mark.asyncio
+ async def test_an_empty_user_filter_is_rejected_before_querying_deployment_data(
+ self, monkeypatch: pytest.MonkeyPatch
+ ) -> None:
+ import httpx
+ from fastapi import FastAPI
+
+ from litellm.proxy import proxy_server
+ from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
+ from litellm.proxy.management_endpoints.auto_router_endpoints import get_auto_router_benchmarks
+
+ query: Final = AsyncMock(return_value=[])
+ monkeypatch.setattr(proxy_server, "prisma_client", SimpleNamespace(db=SimpleNamespace(query_raw=query)))
+ app: Final = FastAPI()
+ app.get("/auto_router/benchmarks")(get_auto_router_benchmarks)
+ app.dependency_overrides[user_api_key_auth] = lambda: ADMIN
+ async with httpx.AsyncClient(transport=httpx.ASGITransport(app=app), base_url="http://test") as client:
+ response: Final = await client.get("/auto_router/benchmarks", params={"user_id": ""})
+
+ assert response.status_code == 422
+ query.assert_not_awaited()
+
@pytest.mark.asyncio
async def test_a_reversed_window_is_rejected(self, monkeypatch: pytest.MonkeyPatch):
from litellm.proxy import proxy_server
@@ -680,7 +707,11 @@ class TestAutoRouterBenchmarks:
assert err.value.status_code == 400
@pytest.mark.asyncio
- async def test_endpoint_returns_groups_and_totals_from_the_rollup(self, monkeypatch: pytest.MonkeyPatch):
+ @pytest.mark.parametrize("role", [LitellmUserRoles.PROXY_ADMIN, LitellmUserRoles.PROXY_ADMIN_VIEW_ONLY])
+ @pytest.mark.parametrize("user_id", [None, "selected-user"])
+ async def test_endpoint_returns_groups_and_totals_from_the_rollup(
+ self, monkeypatch: pytest.MonkeyPatch, role: LitellmUserRoles, user_id: str | None
+ ):
from litellm.proxy import proxy_server
from litellm.proxy.management_endpoints.auto_router_endpoints import get_auto_router_benchmarks
@@ -695,12 +726,13 @@ class TestAutoRouterBenchmarks:
monkeypatch.setattr(proxy_server, "prisma_client", type("P", (), {"db": _DB()})())
response = await get_auto_router_benchmarks(
- user_api_key_dict=ADMIN,
+ user_api_key_dict=UserAPIKeyAuth(user_role=role, api_key="sk-admin", user_id="viewer"),
start_date="2026-07-01",
end_date="2026-08-01",
api_key="key-hash",
+ user_id=user_id,
)
- assert captured["params"] == ("2026-07-01T00:00:00", "2026-08-02T00:00:00", "key-hash")
+ assert captured["params"] == ("2026-07-01T00:00:00", "2026-08-02T00:00:00", "key-hash", user_id)
assert response.routers_in_scope == 1
assert response.groups[0].router_name == "live-auto"
assert response.groups[0].saved_pct == response.totals.saved_pct == 75.0
diff --git a/tests/test_litellm/proxy/test_spend_log_cleanup.py b/tests/test_litellm/proxy/test_spend_log_cleanup.py
index bf1538183ab..f395c146cf0 100644
--- a/tests/test_litellm/proxy/test_spend_log_cleanup.py
+++ b/tests/test_litellm/proxy/test_spend_log_cleanup.py
@@ -793,18 +793,20 @@ async def test_spend_logs_retention_alone_does_not_touch_the_session_rollup():
tables = [call[0][0] for call in client.db.execute_raw.call_args_list]
assert any('"LiteLLM_SpendLogs"' in sql for sql in tables)
assert not any('"LiteLLM_AutoRouterSession"' in sql for sql in tables)
+ assert not any('"LiteLLM_AutoRouterUserSession"' in sql for sql in tables)
assert not any('"LiteLLM_HealthCheckTable"' in sql for sql in tables)
@pytest.mark.asyncio
-async def test_session_retention_alone_cleans_only_the_session_rollup():
- client = _mock_prisma_for_retention([0])
+async def test_session_retention_alone_cleans_both_session_rollups():
+ client = _mock_prisma_for_retention([0, 0])
cleaner = SpendLogCleanup(general_settings={"maximum_autorouter_session_retention_period": "365d"})
cleaner.pod_lock_manager = None
await cleaner.cleanup_old_spend_logs(client)
tables = [call[0][0] for call in client.db.execute_raw.call_args_list]
- assert len(tables) == 1
+ assert len(tables) == 2
assert '"LiteLLM_AutoRouterSession"' in tables[0]
+ assert '"LiteLLM_AutoRouterUserSession"' in tables[1]
@pytest.mark.asyncio
@@ -825,7 +827,7 @@ async def test_health_check_retention_alone_cleans_only_the_health_check_table()
@pytest.mark.asyncio
async def test_each_retention_key_cuts_off_at_its_own_horizon():
- client = _mock_prisma_for_retention([0, 0, 0, 0])
+ client = _mock_prisma_for_retention([0, 0, 0, 0, 0])
cleaner = SpendLogCleanup(
general_settings={
"maximum_spend_logs_retention_period": "7d",
@@ -839,6 +841,8 @@ async def test_each_retention_key_cuts_off_at_its_own_horizon():
(
"LiteLLM_AutoRouterSession"
if '"LiteLLM_AutoRouterSession"' in call[0][0]
+ else "LiteLLM_AutoRouterUserSession"
+ if '"LiteLLM_AutoRouterUserSession"' in call[0][0]
else "LiteLLM_HealthCheckTable"
if '"LiteLLM_HealthCheckTable"' in call[0][0]
else "logs"
@@ -848,6 +852,7 @@ async def test_each_retention_key_cuts_off_at_its_own_horizon():
now = datetime.now(timezone.utc)
assert (now - cutoffs["logs"]).days == 7
assert (now - cutoffs["LiteLLM_AutoRouterSession"]).days == 365
+ assert cutoffs["LiteLLM_AutoRouterUserSession"] == cutoffs["LiteLLM_AutoRouterSession"]
assert (now - cutoffs["LiteLLM_HealthCheckTable"]).days == 30
diff --git a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/AutoRouterBenchmarksTab.test.tsx b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/AutoRouterBenchmarksTab.test.tsx
index 5c7453c1394..a144630cdd0 100644
--- a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/AutoRouterBenchmarksTab.test.tsx
+++ b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/AutoRouterBenchmarksTab.test.tsx
@@ -434,7 +434,7 @@ describe("AutoRouterBenchmarksTab", () => {
mockHook({ data: response([group()]) });
const { dateValue, onDateChange } = renderTab();
- expect(vi.mocked(useAutoRouterBenchmarks)).toHaveBeenCalledWith("sk-test", dateValue, undefined);
+ expect(vi.mocked(useAutoRouterBenchmarks)).toHaveBeenCalledWith("sk-test", dateValue, undefined, undefined);
expect(screen.getByText("Jul 6 – Aug 5 (UTC)")).toBeInTheDocument();
fireEvent.click(screen.getByTestId("date-picker"));
@@ -460,7 +460,7 @@ describe("AutoRouterBenchmarksTab", () => {
,
);
- expect(vi.mocked(useAutoRouterBenchmarks)).toHaveBeenCalledWith("sk-test", dateValue, "key-hash-1");
+ expect(vi.mocked(useAutoRouterBenchmarks)).toHaveBeenCalledWith("sk-test", dateValue, "key-hash-1", undefined);
expect(screen.getByText("Total estimated savings")).toBeInTheDocument();
expect(screen.queryByRole("tab", { name: "Shadow Evals" })).not.toBeInTheDocument();
});
diff --git a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/AutoRouterBenchmarksTab.tsx b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/AutoRouterBenchmarksTab.tsx
index ce55b633b60..063598bd46e 100644
--- a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/AutoRouterBenchmarksTab.tsx
+++ b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/AutoRouterBenchmarksTab.tsx
@@ -312,8 +312,7 @@ const BenchmarksBody: React.FC = ({ isPending, error, data,
length. Total actual spend includes every turn; savings and baseline spend include only turns with a current
estimate, including turns with zero savings. Savings are net of recorded LLM classification cost. Classification
cost per 1K turns is averaged over all auto-router turns, including those that skip classification. The range
- counts whole sessions that overlap it, so totals can differ slightly from the Overall tab, which buckets savings
- by UTC day.
+ counts whole sessions that overlap it, so totals can differ from savings views that group usage by UTC day.
@@ -333,11 +332,17 @@ interface AutoRouterBenchmarksTabProps {
accessToken: string | null;
activity: Pick
;
apiKey?: string;
+ userId?: string;
}
-export const AutoRouterUsageView: React.FC = ({ accessToken, activity, apiKey }) => {
+export const AutoRouterUsageView: React.FC = ({
+ accessToken,
+ activity,
+ apiKey,
+ userId,
+}) => {
const { dateValue, onDateChange } = activity;
- const { data, isPending, error } = useAutoRouterBenchmarks(accessToken, dateValue, apiKey);
+ const { data, isPending, error } = useAutoRouterBenchmarks(accessToken, dateValue, apiKey, userId);
const [selectedKey, setSelectedKey] = useState(ALL_ROUTERS);
const { data: autoRouters } = useAutoRouters();
@@ -372,6 +377,12 @@ export const AutoRouterUsageView: React.FC = ({ ac
+ {userId && (
+
+ Usage for this user across API keys and JWT-authenticated requests. Older sessions recorded without a user ID
+ are not included.
+
+ )}
+export const useAutoRouterBenchmarks = (
+ accessToken: string | null,
+ range: DateRange,
+ apiKey?: string,
+ userId?: string,
+) =>
$api.useQuery(
"get",
"/auto_router/benchmarks",
- { params: { query: { ...benchmarksWindow(range, new Date()), api_key: apiKey } } },
+ { params: { query: { ...benchmarksWindow(range, new Date()), api_key: apiKey, user_id: userId } } },
{ enabled: Boolean(accessToken && range.from && range.to), retry: false },
);
diff --git a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/useDailyActivityRange.test.tsx b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/useDailyActivityRange.test.tsx
index 4059303d5a5..e501cf00b90 100644
--- a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/useDailyActivityRange.test.tsx
+++ b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/useDailyActivityRange.test.tsx
@@ -15,6 +15,8 @@ vi.mock("@/app/(dashboard)/usage/_components/hooks/usePaginatedDailyActivity", (
isFetchingMore: false,
progress: { currentPage: 4, totalPages: 9 },
cancelled: false,
+ failed: false,
+ coversRange: true,
cancel: mockCancel,
};
},
diff --git a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/useDailyActivityRange.ts b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/useDailyActivityRange.ts
index 92dd24b8d6d..4eb9f257d30 100644
--- a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/useDailyActivityRange.ts
+++ b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/useDailyActivityRange.ts
@@ -67,14 +67,16 @@ export const useScopedDailyActivityRange = (
args: [accessToken, startTime, endTime, userId, true, apiKey],
enabled: !!accessToken && !!startTime && !!endTime,
};
- const { data, loading, isFetchingMore, progress, cancelled, failed, cancel } =
+ const { data, loading, isFetchingMore, progress, cancelled, failed, coversRange, cancel } =
usePaginatedDailyActivity(activityQueryOptions);
+ const readUnavailable = failed || cancelled;
+ const waitingForRange = activityQueryOptions.enabled && !coversRange && !readUnavailable;
return {
dateValue,
onDateChange,
results: data.results as DailyData[],
- loading,
+ loading: loading || waitingForRange,
isFetchingMore,
progress,
cancelled,
diff --git a/ui/litellm-dashboard/src/app/(dashboard)/users/_components/view_users/user_info_view.integration.test.tsx b/ui/litellm-dashboard/src/app/(dashboard)/users/_components/view_users/user_info_view.integration.test.tsx
index 0f1a44851c7..6a0e55a6dda 100644
--- a/ui/litellm-dashboard/src/app/(dashboard)/users/_components/view_users/user_info_view.integration.test.tsx
+++ b/ui/litellm-dashboard/src/app/(dashboard)/users/_components/view_users/user_info_view.integration.test.tsx
@@ -1,7 +1,17 @@
-import { fireEvent, renderWithProviders as render, screen, waitFor } from "../../../../../../tests/test-utils";
+import {
+ act,
+ fireEvent,
+ renderWithProviders as render,
+ screen,
+ testQueryClient,
+ waitFor,
+} from "../../../../../../tests/test-utils";
import userEvent, { PointerEventsCheckLevel } from "@testing-library/user-event";
-import { describe, expect, it, vi, beforeEach } from "vitest";
+import { describe, expect, it, vi, beforeEach, afterEach } from "vitest";
+import { Profiler } from "react";
import UserInfoView from "./user_info_view";
+import type { DailyData, SpendMetrics } from "@/components/UsagePage/types";
+import type { AutoRouterBenchmarksResponse } from "@/app/(dashboard)/cost-optimization/_components/autoRouterBenchmarks";
const mockTeamMemberAddCall = vi.fn();
const mockTeamMemberDeleteCall = vi.fn();
@@ -11,6 +21,8 @@ const mockTeamInfoCall = vi.fn();
const mockUserUpdateUserCall = vi.fn();
const mockFetchMCPServers = vi.fn();
const mockListMCPTools = vi.fn();
+const mockUserDailyActivityCall = vi.fn();
+const mockUserDailyActivityAggregatedCall = vi.fn();
const MCP_SERVER = { server_id: "srv-1", server_name: "GitHub MCP", alias: "GitHub MCP" };
@@ -47,13 +59,18 @@ vi.mock("next/navigation", () => ({
useSearchParams: () => new URLSearchParams(window.location.search),
}));
-vi.mock("@/components/networking", () => {
+vi.mock("@/components/networking", async (importOriginal) => {
+ const original = await importOriginal();
return {
+ formatDate: original.formatDate,
serverRootPath: "/",
userGetInfoV2: (...args: unknown[]) => mockUserGetInfoV2(...args),
+ userDailyActivityCall: (...args: unknown[]) => mockUserDailyActivityCall(...args),
+ userDailyActivityAggregatedCall: (...args: unknown[]) => mockUserDailyActivityAggregatedCall(...args),
userDeleteCall: vi.fn(),
userUpdateUserCall: (...args: unknown[]) => mockUserUpdateUserCall(...args),
modelAvailableCall: vi.fn().mockResolvedValue({ data: [] }),
+ modelInfoCall: vi.fn().mockResolvedValue({ data: [], total_pages: 1 }),
invitationCreateCall: vi.fn(),
teamInfoCall: (...args: unknown[]) => mockTeamInfoCall(...args),
teamListCall: (...args: unknown[]) => mockTeamListCall(...args),
@@ -291,3 +308,337 @@ describe("UserInfoView add-to-team form", () => {
expect(screen.getByText("Add User to Team")).toBeInTheDocument();
});
});
+
+const savingsDay = (date: string, metrics: Partial): DailyData => ({
+ date,
+ metrics: {
+ spend: 0,
+ prompt_tokens: 0,
+ completion_tokens: 0,
+ total_tokens: 0,
+ api_requests: 1,
+ successful_requests: 1,
+ failed_requests: 0,
+ cache_read_input_tokens: 0,
+ cache_creation_input_tokens: 0,
+ ...metrics,
+ },
+ breakdown: { models: {}, model_groups: {}, mcp_servers: {}, providers: {}, api_keys: {}, entities: {} },
+});
+
+const savingsResponse = (results: DailyData[]) => ({
+ results,
+ metadata: { total_pages: 1, has_more: false, page: 1 },
+});
+
+const routerUsageResponse = (saved: number): AutoRouterBenchmarksResponse => ({
+ start_date: "2026-09-01",
+ end_date: "2026-09-19",
+ routers_in_scope: 0,
+ groups: [],
+ totals: {
+ sessions: 2,
+ turns: 2,
+ avg_turns_per_session: 1,
+ avg_session_seconds: 0,
+ avg_tokens_per_session: 100,
+ spend: 10,
+ savings_estimated_turns: 2,
+ savings_estimated_actual_spend: 10,
+ classifier_cost: 0,
+ saved_spend: saved,
+ baseline_spend: 10 + saved,
+ saved_pct: (100 * saved) / (10 + saved),
+ saved_per_session: saved / 2,
+ cache: {
+ coverage_pct: 100,
+ hit_rate_pct: 0,
+ same_model: { turns: 0, hits: 0, hit_rate_pct: 0 },
+ first_visit: { turns: 2, hits: 0, hit_rate_pct: 0 },
+ return_to_tier: { turns: 0, hits: 0, hit_rate_pct: 0 },
+ unordered_turns: 0,
+ return_misses_expired: 0,
+ return_misses_within_ttl: 0,
+ return_misses_unknown: 0,
+ ttl_5m_turns: 0,
+ ttl_1h_turns: 0,
+ },
+ },
+});
+
+describe("UserInfoView auto-router usage", () => {
+ const props = {
+ userId: "user-123",
+ onClose: vi.fn(),
+ accessToken: "admin-token",
+ userRole: "proxy_admin",
+ possibleUIRoles: null,
+ };
+ const mockFetch = vi.fn();
+
+ beforeEach(() => {
+ testQueryClient.clear();
+ vi.clearAllMocks();
+ mockUserGetInfoV2.mockImplementation((_token: string, userId: string) =>
+ Promise.resolve({ ...MOCK_USER_DATA_NO_TEAMS, user_id: userId }),
+ );
+ mockFetch.mockReset().mockResolvedValue(Response.json(routerUsageResponse(42)));
+ vi.stubGlobal("fetch", mockFetch);
+ });
+
+ afterEach(() => {
+ testQueryClient.clear();
+ vi.unstubAllGlobals();
+ });
+
+ it.each(["proxy_admin", "proxy_admin_viewer"])(
+ "loads selected-user usage lazily for %s without a key filter",
+ async (userRole) => {
+ const user = userEvent.setup();
+ render( );
+ const tab = await screen.findByRole("tab", { name: "Auto-router usage" });
+ expect(mockFetch).not.toHaveBeenCalled();
+ await user.click(tab);
+
+ expect(await screen.findByText("$42.00")).toBeInTheDocument();
+ const request = mockFetch.mock.calls[0][0] as Request;
+ const params = new URL(request.url).searchParams;
+ expect(params.get("user_id")).toBe("user-123");
+ expect(params.has("api_key")).toBe(false);
+ expect(screen.getByText(/Older sessions recorded without a user ID are not included/)).toBeInTheDocument();
+ },
+ );
+
+ it("switches query scope without displaying the previous user's usage", async () => {
+ const nextUser = Promise.withResolvers();
+ mockFetch.mockResolvedValueOnce(Response.json(routerUsageResponse(42))).mockReturnValue(nextUser.promise);
+ const user = userEvent.setup();
+ const { rerender } = render( );
+ await user.click(await screen.findByRole("tab", { name: "Auto-router usage" }));
+ expect(await screen.findByText("$42.00")).toBeInTheDocument();
+
+ rerender( );
+ expect(screen.getByText("Loading auto-router usage...")).toBeInTheDocument();
+ expect(screen.queryByText("$42.00")).not.toBeInTheDocument();
+ await act(async () => nextUser.resolve(Response.json(routerUsageResponse(-7))));
+ expect(await screen.findByText("-$7.00")).toBeInTheDocument();
+ expect(
+ mockFetch.mock.calls.map(([request]) => new URL((request as Request).url).searchParams.get("user_id")),
+ ).toEqual(["user-123", "user-456"]);
+ });
+
+ it.each(["internal_user", "org_admin", null])("keeps the admin-only tab unavailable to %s", async (userRole) => {
+ render( );
+ await screen.findByRole("tab", { name: "Overview" });
+ expect(screen.queryByRole("tab", { name: "Auto-router usage" })).not.toBeInTheDocument();
+ expect(mockFetch).not.toHaveBeenCalled();
+ });
+
+ it("never turns an absent user ID into a deployment-wide request", async () => {
+ const user = userEvent.setup();
+ render( );
+ await user.click(await screen.findByRole("tab", { name: "Auto-router usage" }));
+ expect(screen.getByRole("alert")).toHaveTextContent("this user has no ID");
+ expect(mockFetch).not.toHaveBeenCalled();
+ });
+});
+
+describe("UserInfoView savings", () => {
+ const props = {
+ userId: "user-123",
+ onClose: vi.fn(),
+ accessToken: "admin-token",
+ userRole: "proxy_admin",
+ possibleUIRoles: null,
+ };
+
+ beforeEach(() => {
+ vi.clearAllMocks();
+ mockUserGetInfoV2.mockImplementation((_token: string, userId: string) =>
+ Promise.resolve({ ...MOCK_USER_DATA_NO_TEAMS, user_id: userId }),
+ );
+ mockUserDailyActivityAggregatedCall.mockReset().mockResolvedValue(savingsResponse([]));
+ mockUserDailyActivityCall.mockReset().mockResolvedValue(savingsResponse([]));
+ });
+
+ afterEach(() => {
+ vi.unstubAllGlobals();
+ });
+
+ it.each(["internal_user", "org_admin", "team_admin"])(
+ "only offers self savings to %s and stops querying after switching to another user",
+ async (userRole) => {
+ const user = userEvent.setup();
+ const { rerender } = render( );
+ await user.click(await screen.findByRole("tab", { name: "Savings" }));
+ expect(await screen.findByText("No usage recorded for this user in this range.")).toBeInTheDocument();
+ expect(mockUserDailyActivityAggregatedCall.mock.calls[0][3]).toBe("user-1");
+
+ mockUserDailyActivityAggregatedCall.mockClear();
+ mockUserDailyActivityCall.mockClear();
+ rerender( );
+ await screen.findAllByText("another-user");
+ expect(screen.getByRole("tab", { name: "Overview" })).toHaveAttribute("aria-selected", "true");
+ expect(screen.queryByRole("tab", { name: "Savings" })).not.toBeInTheDocument();
+ expect(screen.queryByText("No usage recorded for this user in this range.")).not.toBeInTheDocument();
+ expect(mockUserDailyActivityAggregatedCall).not.toHaveBeenCalled();
+ expect(mockUserDailyActivityCall).not.toHaveBeenCalled();
+ },
+ );
+
+ it("loads selected user savings without a key filter, including losses", async () => {
+ const firstDay: Partial = {
+ compression_savings_spend: 1.5,
+ gateway_injected_caching_savings_spend: 0.1,
+ prompt_caching_savings_spend: 0.25,
+ autorouter_savings_spend: -1,
+ };
+ const secondDay: Partial = {
+ compression_savings_spend: 0.5,
+ gateway_injected_caching_savings_spend: 0.3,
+ prompt_caching_savings_spend: 0.75,
+ autorouter_savings_spend: -2,
+ };
+ mockUserDailyActivityAggregatedCall.mockResolvedValue(
+ savingsResponse([savingsDay("2026-09-18", firstDay), savingsDay("2026-09-19", secondDay)]),
+ );
+ const user = userEvent.setup();
+ render( );
+ const savingsTab = await screen.findByRole("tab", { name: "Savings" });
+ expect(mockUserDailyActivityAggregatedCall).not.toHaveBeenCalled();
+ expect(mockUserDailyActivityCall).not.toHaveBeenCalled();
+
+ await user.click(savingsTab);
+
+ expect(await screen.findByTestId("summary-card-total-recorded-savings")).toHaveTextContent("-$0.6000");
+ expect(screen.getByTestId("summary-card-compression-savings")).toHaveTextContent("$2.00");
+ expect(screen.getByTestId("summary-card-prompt-caching-savings")).toHaveTextContent("$0.4000");
+ expect(screen.getByTestId("summary-card-prompt-caching-savings")).toHaveTextContent("$1.00Total");
+ expect(screen.getByTestId("summary-card-auto-router-savings")).toHaveTextContent("-$3.00");
+ expect(mockUserDailyActivityAggregatedCall).toHaveBeenCalledExactlyOnceWith(
+ "admin-token",
+ expect.any(Date),
+ expect.any(Date),
+ "user-123",
+ true,
+ null,
+ );
+ expect(screen.getByTestId("user-savings-scope-note")).toHaveTextContent("JWT-authenticated requests");
+ await user.click(screen.getByRole("tab", { name: "Per day" }));
+ expect(screen.getByRole("tab", { name: "Per day" })).toHaveAttribute("aria-selected", "true");
+ expect(screen.getByTestId("summary-card-total-recorded-savings")).toHaveTextContent("-$0.6000");
+ });
+
+ it("removes the prior user's savings while the newly selected user's results are loading", async () => {
+ const nextUser = Promise.withResolvers>();
+ mockUserDailyActivityAggregatedCall
+ .mockResolvedValueOnce(savingsResponse([savingsDay("2026-09-19", { compression_savings_spend: 42 })]))
+ .mockReturnValueOnce(nextUser.promise);
+ const user = userEvent.setup();
+ const { rerender } = render( );
+ await user.click(await screen.findByRole("tab", { name: "Savings" }));
+ expect(await screen.findByTestId("summary-card-total-recorded-savings")).toHaveTextContent("$42.00");
+
+ rerender( );
+
+ expect(await screen.findByTestId("user-savings-empty")).toHaveTextContent("Loading savings");
+ expect(screen.queryByTestId("summary-card-total-recorded-savings")).not.toBeInTheDocument();
+ expect(mockUserDailyActivityAggregatedCall).toHaveBeenLastCalledWith(
+ "admin-token",
+ expect.any(Date),
+ expect.any(Date),
+ "user-456",
+ true,
+ null,
+ );
+ await act(async () => {
+ nextUser.resolve(savingsResponse([savingsDay("2026-09-19", { autorouter_savings_spend: -7 })]));
+ });
+ expect(await screen.findByTestId("summary-card-total-recorded-savings")).toHaveTextContent("-$7.00");
+ expect(screen.queryByText("$42.00")).not.toBeInTheDocument();
+ });
+
+ it("never commits the previous range's savings under the newly selected dates", async () => {
+ vi.stubGlobal("requestIdleCallback", (callback: IdleRequestCallback) =>
+ window.setTimeout(() => callback({ didTimeout: false, timeRemaining: () => 0 }), 0),
+ );
+ const nextRange = Promise.withResolvers>();
+ mockUserDailyActivityAggregatedCall
+ .mockResolvedValueOnce(savingsResponse([savingsDay("2026-09-19", { compression_savings_spend: 42 })]))
+ .mockReturnValue(nextRange.promise);
+ const committedTotals: Array = [];
+ const captureNewRange = () => {
+ if (screen.queryByText("Running total saved · Sep 1 – Sep 2 (UTC)")) {
+ committedTotals.push(screen.queryByTestId("summary-card-total-recorded-savings")?.textContent ?? null);
+ }
+ };
+ const user = userEvent.setup();
+ render(
+
+
+ ,
+ );
+ await user.click(await screen.findByRole("tab", { name: "Savings" }));
+ expect(await screen.findByTestId("summary-card-total-recorded-savings")).toHaveTextContent("$42.00");
+
+ await user.click(screen.getByRole("button", { name: / - / }));
+ const [startDateInput, endDateInput] = screen.getAllByDisplayValue(/^\d{4}-\d{2}-\d{2}$/);
+ fireEvent.change(startDateInput, { target: { value: "2026-09-01" } });
+ fireEvent.change(endDateInput, { target: { value: "2026-09-02" } });
+ await user.click(screen.getByRole("button", { name: "Apply" }));
+
+ expect(committedTotals.length).toBeGreaterThan(0);
+ expect(committedTotals.every((total) => total === null)).toBe(true);
+ expect(screen.getByTestId("user-savings-empty")).toHaveTextContent("Loading savings");
+ await act(async () => {
+ nextRange.resolve(savingsResponse([savingsDay("2026-09-02", { autorouter_savings_spend: -7 })]));
+ });
+ expect(await screen.findByTestId("summary-card-total-recorded-savings")).toHaveTextContent("-$7.00");
+ });
+
+ it("reports an incomplete paginated read as unavailable instead of displaying a partial savings total", async () => {
+ mockUserDailyActivityAggregatedCall.mockRejectedValue(new Error("aggregated unavailable"));
+ mockUserDailyActivityCall
+ .mockResolvedValueOnce({
+ results: [savingsDay("2026-09-19", { compression_savings_spend: 42 })],
+ metadata: { total_pages: 2, has_more: true, page: 1 },
+ })
+ .mockRejectedValueOnce(new Error("next page unavailable"));
+ const user = userEvent.setup();
+ render( );
+ await user.click(await screen.findByRole("tab", { name: "Savings" }));
+
+ expect(await screen.findByRole("alert")).toHaveTextContent("Savings are unavailable for this range");
+ expect(mockUserDailyActivityCall).toHaveBeenLastCalledWith(
+ "admin-token",
+ expect.any(Date),
+ expect.any(Date),
+ 2,
+ "user-123",
+ true,
+ null,
+ );
+ expect(screen.queryByTestId("summary-card-total-recorded-savings")).not.toBeInTheDocument();
+ expect(screen.queryByText(/No usage recorded/)).not.toBeInTheDocument();
+ });
+
+ it("distinguishes a user with no usage from an unavailable read", async () => {
+ const user = userEvent.setup();
+ render( );
+ await user.click(await screen.findByRole("tab", { name: "Savings" }));
+
+ expect(await screen.findByTestId("user-savings-empty")).toHaveTextContent("No usage recorded for this user");
+ expect(screen.getByTestId("summary-card-total-recorded-savings")).toHaveTextContent("$0.00");
+ expect(screen.queryByRole("alert")).not.toBeInTheDocument();
+ });
+
+ it.each(["", " "])("never queries an absent selected user ID (%j)", async (userId) => {
+ const user = userEvent.setup();
+ render( );
+ await user.click(await screen.findByRole("tab", { name: "Savings" }));
+
+ expect(screen.getByRole("alert")).toHaveTextContent("this user has no ID");
+ expect(mockUserDailyActivityAggregatedCall).not.toHaveBeenCalled();
+ expect(mockUserDailyActivityCall).not.toHaveBeenCalled();
+ });
+});
diff --git a/ui/litellm-dashboard/src/app/(dashboard)/users/_components/view_users/user_info_view.tsx b/ui/litellm-dashboard/src/app/(dashboard)/users/_components/view_users/user_info_view.tsx
index e083e549552..c95badc587a 100644
--- a/ui/litellm-dashboard/src/app/(dashboard)/users/_components/view_users/user_info_view.tsx
+++ b/ui/litellm-dashboard/src/app/(dashboard)/users/_components/view_users/user_info_view.tsx
@@ -28,7 +28,7 @@ import {
ComboboxList,
} from "@/components/ui/combobox";
import { Select, SelectContent, SelectItem, SelectTrigger, SelectValue } from "@/components/ui/select";
-import { rolesWithWriteAccess } from "@/utils/roles";
+import { hasProxyWideSpendView, rolesWithWriteAccess } from "@/utils/roles";
import { teamDetailHref } from "@/utils/entityLinks";
import { BadgeLink } from "@/components/shared/BadgeLink";
import { UserEditView } from "../user_edit_view";
@@ -44,6 +44,9 @@ import { useMCPServers } from "@/app/(dashboard)/hooks/mcpServers/useMCPServers"
import { useMCPToolsets } from "@/app/(dashboard)/hooks/mcpServers/useMCPToolsets";
import { extractMcpEntitlement } from "@/components/mcp_server_management/mcpEntitlement";
import { Dialog, DialogContent, DialogHeader, DialogTitle } from "@/components/ui/dialog";
+import ScopedSavingsTab from "@/components/shared/ScopedSavingsTab";
+import { AutoRouterUsageView } from "@/app/(dashboard)/cost-optimization/_components/AutoRouterBenchmarksTab";
+import { useActivityDateRange } from "@/app/(dashboard)/cost-optimization/_components/useDailyActivityRange";
interface UserInfoViewProps {
userId: string;
@@ -85,7 +88,10 @@ export default function UserInfoView({
initialTab = 0,
startInEditMode = false,
}: UserInfoViewProps) {
- const { premiumUser } = useAuthorized();
+ const { premiumUser, userId: signedInUserId } = useAuthorized();
+ const canViewAutoRouterUsage = hasProxyWideSpendView(userRole);
+ const canViewSavings = canViewAutoRouterUsage || (Boolean(userId.trim()) && userId === signedInUserId);
+ const activityDateRange = useActivityDateRange();
const [userData, setUserData] = useState(null);
const [teamDetails, setTeamDetails] = useState([]);
const [isDeleteModalOpen, setIsDeleteModalOpen] = useState(false);
@@ -97,6 +103,8 @@ export default function UserInfoView({
const [invitationLinkData, setInvitationLinkData] = useState(null);
const [baseUrl, setBaseUrl] = useState(null);
const [activeTab, setActiveTab] = useState(initialTab === 1 ? "details" : "overview");
+ const hiddenSavingsTab = activeTab === "savings" && !canViewSavings;
+ const hiddenRouterTab = activeTab === "auto-router-usage" && !canViewAutoRouterUsage;
const [copiedStates, setCopiedStates] = useState>({});
const [isTeamsExpanded, setIsTeamsExpanded] = useState(false);
const [isAddTeamModalOpen, setIsAddTeamModalOpen] = useState(false);
@@ -467,7 +475,11 @@ export default function UserInfoView({
confirmLoading={isDeletingUser}
/>
- setActiveTab(String(v))} className="gap-0">
+ setActiveTab(String(v))}
+ className="gap-0"
+ >
Overview
@@ -475,6 +487,16 @@ export default function UserInfoView({
Details
+ {canViewSavings && (
+
+ Savings
+
+ )}
+ {canViewAutoRouterUsage && (
+
+ Auto-router usage
+
+ )}
{/* Overview Panel */}
@@ -685,6 +707,38 @@ export default function UserInfoView({
)}
+ {canViewSavings && (
+
+ {activeTab === "savings" &&
+ (userId.trim() ? (
+
+ ) : (
+ Savings are unavailable because this user has no ID.
+ ))}
+
+ )}
+ {canViewAutoRouterUsage && (
+
+ {activeTab === "auto-router-usage" &&
+ (userId.trim() ? (
+
+ ) : (
+ Auto-router usage is unavailable because this user has no ID.
+ ))}
+
+ )}
{
+ const { dateValue, onDateChange, results, loading, isFetchingMore, failed, cancelled } = useScopedDailyActivityRange(
+ accessToken,
+ scope,
+ activity,
+ );
+ const startTime = dateValue.from;
+ const endTime = dateValue.to;
+
+ const [accumulation, setAccumulation] = useState("cumulative");
+
+ const perInterval = useMemo(() => savingsSeriesOf(results), [results]);
+
+ const overTime = useMemo(() => {
+ if (accumulation !== "cumulative") return perInterval;
+ const startLabel = startTime ? shortDate(localIsoDay(startTime)) : "";
+ return withStartAnchor(toCumulative(perInterval), startLabel);
+ }, [accumulation, perInterval, startTime]);
+
+ const intervalLabel = "Per day";
+ const rangeLabel = formatRangeLabel(startTime, endTime);
+ const savingsSubtitle = [
+ accumulation === "cumulative" ? "Running total saved" : `Saved ${intervalLabel.toLowerCase()}`,
+ rangeLabel && `${rangeLabel} (UTC)`,
+ ]
+ .filter(Boolean)
+ .join(" · ");
+
+ const isLoading = loading || isFetchingMore;
+ const unavailable = failed || cancelled;
+ const showResults = !isLoading && !unavailable;
+ const hasRows = results.length > 0;
+ const showEmpty = !unavailable && (isLoading || !hasRows);
+ const showChart = showResults && hasRows;
+ const chartProps = {
+ data: overTime,
+ index: "date",
+ categories: SAVINGS_SERIES,
+ colors: SAVINGS_COLORS,
+ valueFormatter: usd,
+ showLegend: false,
+ };
+
+ return (
+
+
+
Spend is bucketed by UTC day
+
+
+
+ {scopeNote && (
+
+ {scopeNote}
+
+ )}
+
+ {unavailable && (
+
+ Savings are unavailable for this range. Try another date range or reopen this tab.
+
+ )}
+ {showResults &&
}
+
+
+
+ Savings
+ {savingsSubtitle}
+
+
+ setAccumulation(value as SavingsAccumulation)}>
+
+ Cumulative
+ {intervalLabel}
+
+
+
+
+
+ {showEmpty && (
+
+ {isLoading ? "Loading savings..." : `No usage recorded for this ${entityType} in this range.`}
+
+ )}
+ {showChart &&
+ (accumulation === "cumulative" ? (
+
+ ) : (
+
+ ))}
+
+
+
+ );
+};
+
+export default ScopedSavingsTab;
diff --git a/ui/litellm-dashboard/src/components/templates/KeySavingsTab.tsx b/ui/litellm-dashboard/src/components/templates/KeySavingsTab.tsx
index c33529eb042..d1395e0be51 100644
--- a/ui/litellm-dashboard/src/components/templates/KeySavingsTab.tsx
+++ b/ui/litellm-dashboard/src/components/templates/KeySavingsTab.tsx
@@ -1,132 +1,29 @@
"use client";
-import React, { useMemo, useState } from "react";
-
-import { AreaChart, BarChart, CustomLegend } from "@/components/shared/charts";
-import AdvancedDatePicker from "@/components/shared/advanced_date_picker";
-import SavingsTiles from "@/components/shared/SavingsTiles";
-import { Card, CardAction, CardContent, CardDescription, CardHeader, CardTitle } from "@/components/ui/card";
-import { Tabs, TabsList, TabsTrigger } from "@/components/ui/tabs";
+import ScopedSavingsTab from "@/components/shared/ScopedSavingsTab";
import { hasProxyWideSpendView, spendScopeUserId } from "@/utils/roles";
-import {
- formatRangeLabel,
- localIsoDay,
- MAX_POINTS_WITH_DOTS,
- SAVINGS_COLORS,
- SAVINGS_SERIES,
- SavingsAccumulation,
- SavingsPoint,
- savingsSeriesOf,
- shortDate,
- toCumulative,
- usd,
- withStartAnchor,
-} from "@/app/(dashboard)/cost-optimization/_components/costOptimizationUtils";
-import {
- useScopedDailyActivityRange,
- type ActivityDateRange,
-} from "@/app/(dashboard)/cost-optimization/_components/useDailyActivityRange";
+import type { ActivityDateRange } from "@/app/(dashboard)/cost-optimization/_components/useDailyActivityRange";
interface KeySavingsTabProps {
accessToken: string | null;
- /** The key's token hash — what spend rows are keyed by, not the one-time plaintext secret. */
keyToken: string;
userId: string | null;
userRole: string;
activity: ActivityDateRange;
}
-const KeySavingsTab: React.FC = ({ accessToken, keyToken, userId, userRole, activity }) => {
- // Proxy admins read the whole key. For anyone else the endpoint applies the caller's own user_id
- // alongside the key filter, so the figures cover only that viewer's requests on this key -- said
- // plainly in the scope note below rather than left to be misread as the key's total.
- const readsWholeKey = hasProxyWideSpendView(userRole);
- const { dateValue, onDateChange, results, loading, isFetchingMore } = useScopedDailyActivityRange(
- accessToken,
- { userId: spendScopeUserId(userRole, userId), apiKey: keyToken },
- activity,
- );
- const startTime = dateValue.from ?? null;
- const endTime = dateValue.to ?? null;
-
- const [accumulation, setAccumulation] = useState("cumulative");
-
- const perInterval = useMemo(() => savingsSeriesOf(results), [results]);
-
- const overTime = useMemo(() => {
- if (accumulation !== "cumulative") return perInterval;
- const startLabel = startTime ? shortDate(localIsoDay(startTime)) : "";
- return withStartAnchor(toCumulative(perInterval), startLabel);
- }, [accumulation, perInterval, startTime]);
-
- const intervalLabel = "Per day";
- const rangeLabel = formatRangeLabel(startTime ?? undefined, endTime ?? undefined);
- const savingsSubtitle = [
- accumulation === "cumulative" ? "Running total saved" : `Saved ${intervalLabel.toLowerCase()}`,
- rangeLabel && `${rangeLabel} (UTC)`,
- ]
- .filter(Boolean)
- .join(" · ");
-
- const isLoading = loading || isFetchingMore;
- const hasRows = results.length > 0;
- const chartProps = {
- data: overTime,
- index: "date",
- categories: SAVINGS_SERIES,
- colors: SAVINGS_COLORS,
- valueFormatter: usd,
- showLegend: false,
- };
-
- return (
-
-
-
Spend is bucketed by UTC day
-
-
-
- {!readsWholeKey && (
-
- Showing your own requests on this key. A key shared across a team will have spend from other members that is
- not counted here.
-
- )}
-
-
-
-
-
- Savings
- {savingsSubtitle}
-
-
- setAccumulation(value as SavingsAccumulation)}>
-
- Cumulative
- {intervalLabel}
-
-
-
-
-
- {/* Distinguishes "still fetching" from "this key genuinely had no traffic": an empty
- chart alone reads as a broken panel, and a $0.00 tile reads as a real zero. */}
- {!hasRows && (
-
- {isLoading ? "Loading savings..." : "No usage recorded for this key in this range."}
-
- )}
- {hasRows && accumulation === "cumulative" && (
-
- )}
- {/* Not stacked: auto-router can go negative on a cold-cache write, and stacking would
- draw that below the axis while the rest of the bar still read as the total. */}
- {hasRows && accumulation !== "cumulative" && }
-
-
-
- );
-};
+const KeySavingsTab = ({ accessToken, keyToken, userId, userRole, activity }: KeySavingsTabProps) => (
+
+);
export default KeySavingsTab;
diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts
index 81580c8bfb1..0055ef4337c 100644
--- a/ui/litellm-dashboard/src/lib/http/schema.d.ts
+++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts
@@ -1246,9 +1246,10 @@ export interface paths {
* @description Benchmarks for the auto-router dashboard: session shape, savings against the configured
* baseline, and prompt-caching behaviour bucketed by what the router did.
*
- * Reads the LiteLLM_AutoRouterSession rollup, folded once per request at spend-write time,
- * so this endpoint never scans LiteLLM_SpendLogs. A session is in the window when it
- * overlaps it: its last turn is on or after start_date and its first turn is on or before
+ * Reads session rollups folded once per request at spend-write time, so this endpoint
+ * never scans LiteLLM_SpendLogs. A user filter selects only turns attributed to that
+ * internal user when written; older key-only history remains outside user views. A session
+ * is in the window when it overlaps it: its last turn is on or after start_date and its first turn is on or before
* end_date. Overall hit rate is over telemetry-bearing turns; each bucket's hit rate is
* over that bucket's turns.
*
@@ -43601,6 +43602,8 @@ export interface operations {
end_date?: string | null;
/** @description Filter to one virtual key token hash */
api_key?: string | null;
+ /** @description Filter to one canonical internal user recorded on each turn */
+ user_id?: string | null;
};
header?: never;
path?: never;
From 39a14f39e5961605ba70deacc0475892f50d3cb7 Mon Sep 17 00:00:00 2001
From: yucheng
Date: Sun, 20 Sep 2026 08:20:47 +0000
Subject: [PATCH 035/160] style(proxy): format cleanup shutdown tests
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
---
.../proxy/shutdown/test_scheduled_jobs.py | 4 +-
.../proxy/test_spend_log_cleanup.py | 127 +++++-------------
2 files changed, 37 insertions(+), 94 deletions(-)
diff --git a/tests/test_litellm/proxy/shutdown/test_scheduled_jobs.py b/tests/test_litellm/proxy/shutdown/test_scheduled_jobs.py
index 7defd6cef6c..87adc464608 100644
--- a/tests/test_litellm/proxy/shutdown/test_scheduled_jobs.py
+++ b/tests/test_litellm/proxy/shutdown/test_scheduled_jobs.py
@@ -7,11 +7,11 @@ from datetime import datetime, timedelta
import pytest
from apscheduler.schedulers.asyncio import AsyncIOScheduler
-import litellm.proxy.shutdown.scheduled_jobs as scheduled_jobs
+from litellm.proxy.shutdown import scheduled_jobs
from litellm.proxy.shutdown.scheduled_jobs import (
AwaitableAsyncIOExecutor,
- stop_in_flight_scheduler_jobs,
pause_scheduled_jobs,
+ stop_in_flight_scheduler_jobs,
)
diff --git a/tests/test_litellm/proxy/test_spend_log_cleanup.py b/tests/test_litellm/proxy/test_spend_log_cleanup.py
index 1691b2d174a..b8f28d0c780 100644
--- a/tests/test_litellm/proxy/test_spend_log_cleanup.py
+++ b/tests/test_litellm/proxy/test_spend_log_cleanup.py
@@ -83,10 +83,10 @@ def test_spend_log_cleanup_cron_scheduling():
assert trigger_weekly is not None
# Invalid cron expression should raise ValueError
- with pytest.raises(ValueError, match='Wrong number of fields; got'):
+ with pytest.raises(ValueError, match="Wrong number of fields; got"):
CronTrigger.from_crontab("invalid cron")
- with pytest.raises(ValueError, match='is higher than the maximum value'):
+ with pytest.raises(ValueError, match="is higher than the maximum value"):
CronTrigger.from_crontab("60 25 * * *") # Invalid minute and hour
@@ -99,6 +99,7 @@ def test_spend_log_cleanup_cron_scheduler_integration():
a real database connection.
"""
from unittest.mock import MagicMock
+
from apscheduler.triggers.cron import CronTrigger
# Mock scheduler
@@ -145,15 +146,11 @@ def test_spend_log_cleanup_cron_scheduler_integration():
# No cron, so it should fall back to interval
}
- cleanup_cron_fallback = general_settings_interval.get(
- "maximum_spend_logs_cleanup_cron"
- )
+ cleanup_cron_fallback = general_settings_interval.get("maximum_spend_logs_cleanup_cron")
assert cleanup_cron_fallback is None # No cron configured
# Simulate interval-based scheduling fallback
- retention_interval = general_settings_interval.get(
- "maximum_spend_logs_retention_interval", "1d"
- )
+ retention_interval = general_settings_interval.get("maximum_spend_logs_retention_interval", "1d")
from litellm.litellm_core_utils.duration_parser import duration_in_seconds
interval_seconds = duration_in_seconds(retention_interval)
@@ -181,27 +178,19 @@ async def test_should_delete_spend_logs():
assert cleaner._should_delete_spend_logs() is False
# Test case 2: Valid seconds string
- cleaner = SpendLogCleanup(
- general_settings={"maximum_spend_logs_retention_period": "3600s"}
- )
+ cleaner = SpendLogCleanup(general_settings={"maximum_spend_logs_retention_period": "3600s"})
assert cleaner._should_delete_spend_logs() is True
# Test case 3: Valid days string
- cleaner = SpendLogCleanup(
- general_settings={"maximum_spend_logs_retention_period": "30d"}
- )
+ cleaner = SpendLogCleanup(general_settings={"maximum_spend_logs_retention_period": "30d"})
assert cleaner._should_delete_spend_logs() is True
# Test case 4: Valid hours string
- cleaner = SpendLogCleanup(
- general_settings={"maximum_spend_logs_retention_period": "24h"}
- )
+ cleaner = SpendLogCleanup(general_settings={"maximum_spend_logs_retention_period": "24h"})
assert cleaner._should_delete_spend_logs() is True
# Test case 5: Invalid format
- cleaner = SpendLogCleanup(
- general_settings={"maximum_spend_logs_retention_period": "invalid"}
- )
+ cleaner = SpendLogCleanup(general_settings={"maximum_spend_logs_retention_period": "invalid"})
assert cleaner._should_delete_spend_logs() is False
@@ -288,9 +277,7 @@ async def test_cleanup_old_spend_logs_retention_period_cutoff():
# Verify the cutoff date is correct
cutoff_date = mock_db.execute_raw.call_args[0][1]
expected_cutoff = datetime.now(timezone.utc) - timedelta(seconds=86400)
- assert (
- abs((cutoff_date - expected_cutoff).total_seconds()) < 1
- ) # Allow 1 second difference for test execution time
+ assert abs((cutoff_date - expected_cutoff).total_seconds()) < 1 # Allow 1 second difference for test execution time
@pytest.mark.asyncio
@@ -310,9 +297,7 @@ async def test_cleanup_drops_partitions_when_enabled_and_partitioned():
partition_manager = MagicMock()
partition_manager.is_partitioned = AsyncMock(return_value=True)
partition_manager.ensure_partitions = AsyncMock(return_value=["p1"])
- partition_manager.drop_partitions_older_than = AsyncMock(
- return_value=["LiteLLM_SpendLogs_p20260601"]
- )
+ partition_manager.drop_partitions_older_than = AsyncMock(return_value=["LiteLLM_SpendLogs_p20260601"])
cleaner = SpendLogCleanup(
general_settings={
@@ -450,9 +435,7 @@ async def test_integer_retention_treated_as_days():
An integer value for maximum_spend_logs_retention_period should be treated
as days (e.g., 3 → '3d' → 259200 seconds).
"""
- cleaner = SpendLogCleanup(
- general_settings={"maximum_spend_logs_retention_period": 3}
- )
+ cleaner = SpendLogCleanup(general_settings={"maximum_spend_logs_retention_period": 3})
result = cleaner._should_delete_spend_logs()
assert result is True
assert cleaner.retention_seconds == 3 * 86400 # 3 days in seconds
@@ -469,13 +452,11 @@ def test_string_retention_still_works():
("2w", 2 * 604800),
]
for setting, expected_seconds in cases:
- cleaner = SpendLogCleanup(
- general_settings={"maximum_spend_logs_retention_period": setting}
- )
+ cleaner = SpendLogCleanup(general_settings={"maximum_spend_logs_retention_period": setting})
assert cleaner._should_delete_spend_logs() is True, f"Failed for {setting}"
- assert (
- cleaner.retention_seconds == expected_seconds
- ), f"Expected {expected_seconds} for {setting}, got {cleaner.retention_seconds}"
+ assert cleaner.retention_seconds == expected_seconds, (
+ f"Expected {expected_seconds} for {setting}, got {cleaner.retention_seconds}"
+ )
@pytest.mark.asyncio
@@ -489,9 +470,7 @@ async def test_delete_old_logs_aborts_on_non_int_execute_raw_return():
mock_db.execute_raw = AsyncMock(return_value=None)
mock_prisma_client.db = mock_db
- cleaner = SpendLogCleanup(
- general_settings={"maximum_spend_logs_retention_period": "7d"}
- )
+ cleaner = SpendLogCleanup(general_settings={"maximum_spend_logs_retention_period": "7d"})
cutoff_date = datetime.now(timezone.utc) - timedelta(days=7)
result = await cleaner._delete_old_logs(mock_prisma_client, cutoff_date, _far_deadline())
@@ -510,9 +489,7 @@ async def test_delete_old_logs_continues_on_valid_int_return():
mock_db.execute_raw = AsyncMock(side_effect=[500, 300, 0])
mock_prisma_client.db = mock_db
- cleaner = SpendLogCleanup(
- general_settings={"maximum_spend_logs_retention_period": "7d"}
- )
+ cleaner = SpendLogCleanup(general_settings={"maximum_spend_logs_retention_period": "7d"})
cutoff_date = datetime.now(timezone.utc) - timedelta(days=7)
result = await cleaner._delete_old_logs(mock_prisma_client, cutoff_date, _far_deadline())
@@ -559,9 +536,7 @@ async def test_delete_old_tool_index_rows_deletes_on_composite_key():
mock_db.execute_raw = AsyncMock(side_effect=[5, 0])
mock_prisma_client.db = mock_db
- cleaner = SpendLogCleanup(
- general_settings={"maximum_spend_logs_retention_period": "7d"}
- )
+ cleaner = SpendLogCleanup(general_settings={"maximum_spend_logs_retention_period": "7d"})
cutoff_date = datetime.now(timezone.utc) - timedelta(days=7)
result = await cleaner._delete_old_tool_index_rows(mock_prisma_client, cutoff_date, _far_deadline())
@@ -581,9 +556,7 @@ async def test_delete_old_logs_continues_after_single_batch_failure(monkeypatch)
import litellm.proxy.db.db_transaction_queue.spend_log_cleanup as cleanup_module
# Zero out the failure backoff so the test doesn't take ~0.5s of real sleep.
- monkeypatch.setattr(
- cleanup_module, "SPEND_LOG_CLEANUP_BATCH_FAILURE_BACKOFF_SECONDS", 0.0
- )
+ monkeypatch.setattr(cleanup_module, "SPEND_LOG_CLEANUP_BATCH_FAILURE_BACKOFF_SECONDS", 0.0)
mock_prisma_client = MagicMock()
_wire_tx(mock_prisma_client.db)
@@ -591,14 +564,10 @@ async def test_delete_old_logs_continues_after_single_batch_failure(monkeypatch)
_wire_tx(mock_db)
# batch 1 succeeds, batch 2 raises (one-off DB timeout), batches 3-4 succeed,
# batch 5 returns 0 → loop exits naturally.
- mock_db.execute_raw = AsyncMock(
- side_effect=[100, TimeoutError("simulated DB timeout"), 200, 50, 0]
- )
+ mock_db.execute_raw = AsyncMock(side_effect=[100, TimeoutError("simulated DB timeout"), 200, 50, 0])
mock_prisma_client.db = mock_db
- cleaner = cleanup_module.SpendLogCleanup(
- general_settings={"maximum_spend_logs_retention_period": "7d"}
- )
+ cleaner = cleanup_module.SpendLogCleanup(general_settings={"maximum_spend_logs_retention_period": "7d"})
cutoff_date = datetime.now(timezone.utc) - timedelta(days=7)
result = await cleaner._delete_old_logs(mock_prisma_client, cutoff_date, _far_deadline())
@@ -615,26 +584,18 @@ async def test_delete_old_logs_aborts_after_consecutive_failures(monkeypatch):
import litellm.proxy.db.db_transaction_queue.spend_log_cleanup as cleanup_module
# Lower the threshold so the test is fast and deterministic.
- monkeypatch.setattr(
- cleanup_module, "SPEND_LOG_CLEANUP_MAX_CONSECUTIVE_BATCH_FAILURES", 3
- )
- monkeypatch.setattr(
- cleanup_module, "SPEND_LOG_CLEANUP_BATCH_FAILURE_BACKOFF_SECONDS", 0.0
- )
+ monkeypatch.setattr(cleanup_module, "SPEND_LOG_CLEANUP_MAX_CONSECUTIVE_BATCH_FAILURES", 3)
+ monkeypatch.setattr(cleanup_module, "SPEND_LOG_CLEANUP_BATCH_FAILURE_BACKOFF_SECONDS", 0.0)
mock_prisma_client = MagicMock()
_wire_tx(mock_prisma_client.db)
mock_db = MagicMock()
_wire_tx(mock_db)
# Every batch raises — must abort after exactly 3 attempts, not loop forever.
- mock_db.execute_raw = AsyncMock(
- side_effect=ConnectionError("simulated persistent DB outage")
- )
+ mock_db.execute_raw = AsyncMock(side_effect=ConnectionError("simulated persistent DB outage"))
mock_prisma_client.db = mock_db
- cleaner = cleanup_module.SpendLogCleanup(
- general_settings={"maximum_spend_logs_retention_period": "7d"}
- )
+ cleaner = cleanup_module.SpendLogCleanup(general_settings={"maximum_spend_logs_retention_period": "7d"})
cutoff_date = datetime.now(timezone.utc) - timedelta(days=7)
result = await cleaner._delete_old_logs(mock_prisma_client, cutoff_date, _far_deadline())
@@ -649,12 +610,8 @@ async def test_delete_old_logs_resets_consecutive_failures_on_success(monkeypatc
intermittent timeouts don't trip the abort threshold."""
import litellm.proxy.db.db_transaction_queue.spend_log_cleanup as cleanup_module
- monkeypatch.setattr(
- cleanup_module, "SPEND_LOG_CLEANUP_MAX_CONSECUTIVE_BATCH_FAILURES", 3
- )
- monkeypatch.setattr(
- cleanup_module, "SPEND_LOG_CLEANUP_BATCH_FAILURE_BACKOFF_SECONDS", 0.0
- )
+ monkeypatch.setattr(cleanup_module, "SPEND_LOG_CLEANUP_MAX_CONSECUTIVE_BATCH_FAILURES", 3)
+ monkeypatch.setattr(cleanup_module, "SPEND_LOG_CLEANUP_BATCH_FAILURE_BACKOFF_SECONDS", 0.0)
mock_prisma_client = MagicMock()
_wire_tx(mock_prisma_client.db)
@@ -675,9 +632,7 @@ async def test_delete_old_logs_resets_consecutive_failures_on_success(monkeypatc
)
mock_prisma_client.db = mock_db
- cleaner = cleanup_module.SpendLogCleanup(
- general_settings={"maximum_spend_logs_retention_period": "7d"}
- )
+ cleaner = cleanup_module.SpendLogCleanup(general_settings={"maximum_spend_logs_retention_period": "7d"})
cutoff_date = datetime.now(timezone.utc) - timedelta(days=7)
result = await cleaner._delete_old_logs(mock_prisma_client, cutoff_date, _far_deadline())
@@ -698,9 +653,7 @@ async def test_cleanup_uses_logger_exception_for_full_traceback(monkeypatch):
mock_prisma_client = MagicMock()
_wire_tx(mock_prisma_client.db)
# Force the outer try/except to fire by making _should_delete_spend_logs raise.
- cleaner = cleanup_module.SpendLogCleanup(
- general_settings={"maximum_spend_logs_retention_period": "7d"}
- )
+ cleaner = cleanup_module.SpendLogCleanup(general_settings={"maximum_spend_logs_retention_period": "7d"})
cleaner.pod_lock_manager = None
def boom():
@@ -725,12 +678,8 @@ async def test_cleanup_releases_lock_after_persistent_batch_failures(monkeypatch
must still be released so the next scheduled run isn't permanently blocked."""
import litellm.proxy.db.db_transaction_queue.spend_log_cleanup as cleanup_module
- monkeypatch.setattr(
- cleanup_module, "SPEND_LOG_CLEANUP_MAX_CONSECUTIVE_BATCH_FAILURES", 2
- )
- monkeypatch.setattr(
- cleanup_module, "SPEND_LOG_CLEANUP_BATCH_FAILURE_BACKOFF_SECONDS", 0.0
- )
+ monkeypatch.setattr(cleanup_module, "SPEND_LOG_CLEANUP_MAX_CONSECUTIVE_BATCH_FAILURES", 2)
+ monkeypatch.setattr(cleanup_module, "SPEND_LOG_CLEANUP_BATCH_FAILURE_BACKOFF_SECONDS", 0.0)
mock_prisma_client = MagicMock()
_wire_tx(mock_prisma_client.db)
@@ -744,9 +693,7 @@ async def test_cleanup_releases_lock_after_persistent_batch_failures(monkeypatch
mock_pod_lock_manager.acquire_lock = AsyncMock(return_value=True)
mock_pod_lock_manager.release_lock = AsyncMock()
- cleaner = cleanup_module.SpendLogCleanup(
- general_settings={"maximum_spend_logs_retention_period": "7d"}
- )
+ cleaner = cleanup_module.SpendLogCleanup(general_settings={"maximum_spend_logs_retention_period": "7d"})
cleaner.pod_lock_manager = mock_pod_lock_manager
await cleaner.cleanup_old_spend_logs(mock_prisma_client)
@@ -996,9 +943,7 @@ async def test_each_batch_carries_a_statement_and_lock_timeout():
}
)
- await cleaner._delete_old_logs(
- mock_prisma_client, datetime.now(timezone.utc) - timedelta(days=7), _far_deadline()
- )
+ await cleaner._delete_old_logs(mock_prisma_client, datetime.now(timezone.utc) - timedelta(days=7), _far_deadline())
assert "SET LOCAL statement_timeout = 12000" in recorded
assert "SET LOCAL lock_timeout = 12000" in recorded
@@ -1134,9 +1079,7 @@ async def test_remaining_rows_probe_is_capped_so_it_cannot_scan_the_table():
cleaner = SpendLogCleanup(general_settings={"maximum_spend_logs_retention_period": "7d"})
- await cleaner._delete_old_logs(
- mock_prisma_client, datetime.now(timezone.utc) - timedelta(days=7), _far_deadline()
- )
+ await cleaner._delete_old_logs(mock_prisma_client, datetime.now(timezone.utc) - timedelta(days=7), _far_deadline())
count_sql = mock_db.query_raw.call_args[0][0]
assert "count(*)" in count_sql
From 0b2d52edc29a7098e95314fbf2c45e58b508efc5 Mon Sep 17 00:00:00 2001
From: yucheng
Date: Sun, 20 Sep 2026 08:21:35 +0000
Subject: [PATCH 036/160] Revert "style(proxy): format cleanup shutdown tests"
This reverts commit 39a14f39e5961605ba70deacc0475892f50d3cb7.
---
.../proxy/shutdown/test_scheduled_jobs.py | 4 +-
.../proxy/test_spend_log_cleanup.py | 127 +++++++++++++-----
2 files changed, 94 insertions(+), 37 deletions(-)
diff --git a/tests/test_litellm/proxy/shutdown/test_scheduled_jobs.py b/tests/test_litellm/proxy/shutdown/test_scheduled_jobs.py
index 87adc464608..7defd6cef6c 100644
--- a/tests/test_litellm/proxy/shutdown/test_scheduled_jobs.py
+++ b/tests/test_litellm/proxy/shutdown/test_scheduled_jobs.py
@@ -7,11 +7,11 @@ from datetime import datetime, timedelta
import pytest
from apscheduler.schedulers.asyncio import AsyncIOScheduler
-from litellm.proxy.shutdown import scheduled_jobs
+import litellm.proxy.shutdown.scheduled_jobs as scheduled_jobs
from litellm.proxy.shutdown.scheduled_jobs import (
AwaitableAsyncIOExecutor,
- pause_scheduled_jobs,
stop_in_flight_scheduler_jobs,
+ pause_scheduled_jobs,
)
diff --git a/tests/test_litellm/proxy/test_spend_log_cleanup.py b/tests/test_litellm/proxy/test_spend_log_cleanup.py
index b8f28d0c780..1691b2d174a 100644
--- a/tests/test_litellm/proxy/test_spend_log_cleanup.py
+++ b/tests/test_litellm/proxy/test_spend_log_cleanup.py
@@ -83,10 +83,10 @@ def test_spend_log_cleanup_cron_scheduling():
assert trigger_weekly is not None
# Invalid cron expression should raise ValueError
- with pytest.raises(ValueError, match="Wrong number of fields; got"):
+ with pytest.raises(ValueError, match='Wrong number of fields; got'):
CronTrigger.from_crontab("invalid cron")
- with pytest.raises(ValueError, match="is higher than the maximum value"):
+ with pytest.raises(ValueError, match='is higher than the maximum value'):
CronTrigger.from_crontab("60 25 * * *") # Invalid minute and hour
@@ -99,7 +99,6 @@ def test_spend_log_cleanup_cron_scheduler_integration():
a real database connection.
"""
from unittest.mock import MagicMock
-
from apscheduler.triggers.cron import CronTrigger
# Mock scheduler
@@ -146,11 +145,15 @@ def test_spend_log_cleanup_cron_scheduler_integration():
# No cron, so it should fall back to interval
}
- cleanup_cron_fallback = general_settings_interval.get("maximum_spend_logs_cleanup_cron")
+ cleanup_cron_fallback = general_settings_interval.get(
+ "maximum_spend_logs_cleanup_cron"
+ )
assert cleanup_cron_fallback is None # No cron configured
# Simulate interval-based scheduling fallback
- retention_interval = general_settings_interval.get("maximum_spend_logs_retention_interval", "1d")
+ retention_interval = general_settings_interval.get(
+ "maximum_spend_logs_retention_interval", "1d"
+ )
from litellm.litellm_core_utils.duration_parser import duration_in_seconds
interval_seconds = duration_in_seconds(retention_interval)
@@ -178,19 +181,27 @@ async def test_should_delete_spend_logs():
assert cleaner._should_delete_spend_logs() is False
# Test case 2: Valid seconds string
- cleaner = SpendLogCleanup(general_settings={"maximum_spend_logs_retention_period": "3600s"})
+ cleaner = SpendLogCleanup(
+ general_settings={"maximum_spend_logs_retention_period": "3600s"}
+ )
assert cleaner._should_delete_spend_logs() is True
# Test case 3: Valid days string
- cleaner = SpendLogCleanup(general_settings={"maximum_spend_logs_retention_period": "30d"})
+ cleaner = SpendLogCleanup(
+ general_settings={"maximum_spend_logs_retention_period": "30d"}
+ )
assert cleaner._should_delete_spend_logs() is True
# Test case 4: Valid hours string
- cleaner = SpendLogCleanup(general_settings={"maximum_spend_logs_retention_period": "24h"})
+ cleaner = SpendLogCleanup(
+ general_settings={"maximum_spend_logs_retention_period": "24h"}
+ )
assert cleaner._should_delete_spend_logs() is True
# Test case 5: Invalid format
- cleaner = SpendLogCleanup(general_settings={"maximum_spend_logs_retention_period": "invalid"})
+ cleaner = SpendLogCleanup(
+ general_settings={"maximum_spend_logs_retention_period": "invalid"}
+ )
assert cleaner._should_delete_spend_logs() is False
@@ -277,7 +288,9 @@ async def test_cleanup_old_spend_logs_retention_period_cutoff():
# Verify the cutoff date is correct
cutoff_date = mock_db.execute_raw.call_args[0][1]
expected_cutoff = datetime.now(timezone.utc) - timedelta(seconds=86400)
- assert abs((cutoff_date - expected_cutoff).total_seconds()) < 1 # Allow 1 second difference for test execution time
+ assert (
+ abs((cutoff_date - expected_cutoff).total_seconds()) < 1
+ ) # Allow 1 second difference for test execution time
@pytest.mark.asyncio
@@ -297,7 +310,9 @@ async def test_cleanup_drops_partitions_when_enabled_and_partitioned():
partition_manager = MagicMock()
partition_manager.is_partitioned = AsyncMock(return_value=True)
partition_manager.ensure_partitions = AsyncMock(return_value=["p1"])
- partition_manager.drop_partitions_older_than = AsyncMock(return_value=["LiteLLM_SpendLogs_p20260601"])
+ partition_manager.drop_partitions_older_than = AsyncMock(
+ return_value=["LiteLLM_SpendLogs_p20260601"]
+ )
cleaner = SpendLogCleanup(
general_settings={
@@ -435,7 +450,9 @@ async def test_integer_retention_treated_as_days():
An integer value for maximum_spend_logs_retention_period should be treated
as days (e.g., 3 → '3d' → 259200 seconds).
"""
- cleaner = SpendLogCleanup(general_settings={"maximum_spend_logs_retention_period": 3})
+ cleaner = SpendLogCleanup(
+ general_settings={"maximum_spend_logs_retention_period": 3}
+ )
result = cleaner._should_delete_spend_logs()
assert result is True
assert cleaner.retention_seconds == 3 * 86400 # 3 days in seconds
@@ -452,11 +469,13 @@ def test_string_retention_still_works():
("2w", 2 * 604800),
]
for setting, expected_seconds in cases:
- cleaner = SpendLogCleanup(general_settings={"maximum_spend_logs_retention_period": setting})
- assert cleaner._should_delete_spend_logs() is True, f"Failed for {setting}"
- assert cleaner.retention_seconds == expected_seconds, (
- f"Expected {expected_seconds} for {setting}, got {cleaner.retention_seconds}"
+ cleaner = SpendLogCleanup(
+ general_settings={"maximum_spend_logs_retention_period": setting}
)
+ assert cleaner._should_delete_spend_logs() is True, f"Failed for {setting}"
+ assert (
+ cleaner.retention_seconds == expected_seconds
+ ), f"Expected {expected_seconds} for {setting}, got {cleaner.retention_seconds}"
@pytest.mark.asyncio
@@ -470,7 +489,9 @@ async def test_delete_old_logs_aborts_on_non_int_execute_raw_return():
mock_db.execute_raw = AsyncMock(return_value=None)
mock_prisma_client.db = mock_db
- cleaner = SpendLogCleanup(general_settings={"maximum_spend_logs_retention_period": "7d"})
+ cleaner = SpendLogCleanup(
+ general_settings={"maximum_spend_logs_retention_period": "7d"}
+ )
cutoff_date = datetime.now(timezone.utc) - timedelta(days=7)
result = await cleaner._delete_old_logs(mock_prisma_client, cutoff_date, _far_deadline())
@@ -489,7 +510,9 @@ async def test_delete_old_logs_continues_on_valid_int_return():
mock_db.execute_raw = AsyncMock(side_effect=[500, 300, 0])
mock_prisma_client.db = mock_db
- cleaner = SpendLogCleanup(general_settings={"maximum_spend_logs_retention_period": "7d"})
+ cleaner = SpendLogCleanup(
+ general_settings={"maximum_spend_logs_retention_period": "7d"}
+ )
cutoff_date = datetime.now(timezone.utc) - timedelta(days=7)
result = await cleaner._delete_old_logs(mock_prisma_client, cutoff_date, _far_deadline())
@@ -536,7 +559,9 @@ async def test_delete_old_tool_index_rows_deletes_on_composite_key():
mock_db.execute_raw = AsyncMock(side_effect=[5, 0])
mock_prisma_client.db = mock_db
- cleaner = SpendLogCleanup(general_settings={"maximum_spend_logs_retention_period": "7d"})
+ cleaner = SpendLogCleanup(
+ general_settings={"maximum_spend_logs_retention_period": "7d"}
+ )
cutoff_date = datetime.now(timezone.utc) - timedelta(days=7)
result = await cleaner._delete_old_tool_index_rows(mock_prisma_client, cutoff_date, _far_deadline())
@@ -556,7 +581,9 @@ async def test_delete_old_logs_continues_after_single_batch_failure(monkeypatch)
import litellm.proxy.db.db_transaction_queue.spend_log_cleanup as cleanup_module
# Zero out the failure backoff so the test doesn't take ~0.5s of real sleep.
- monkeypatch.setattr(cleanup_module, "SPEND_LOG_CLEANUP_BATCH_FAILURE_BACKOFF_SECONDS", 0.0)
+ monkeypatch.setattr(
+ cleanup_module, "SPEND_LOG_CLEANUP_BATCH_FAILURE_BACKOFF_SECONDS", 0.0
+ )
mock_prisma_client = MagicMock()
_wire_tx(mock_prisma_client.db)
@@ -564,10 +591,14 @@ async def test_delete_old_logs_continues_after_single_batch_failure(monkeypatch)
_wire_tx(mock_db)
# batch 1 succeeds, batch 2 raises (one-off DB timeout), batches 3-4 succeed,
# batch 5 returns 0 → loop exits naturally.
- mock_db.execute_raw = AsyncMock(side_effect=[100, TimeoutError("simulated DB timeout"), 200, 50, 0])
+ mock_db.execute_raw = AsyncMock(
+ side_effect=[100, TimeoutError("simulated DB timeout"), 200, 50, 0]
+ )
mock_prisma_client.db = mock_db
- cleaner = cleanup_module.SpendLogCleanup(general_settings={"maximum_spend_logs_retention_period": "7d"})
+ cleaner = cleanup_module.SpendLogCleanup(
+ general_settings={"maximum_spend_logs_retention_period": "7d"}
+ )
cutoff_date = datetime.now(timezone.utc) - timedelta(days=7)
result = await cleaner._delete_old_logs(mock_prisma_client, cutoff_date, _far_deadline())
@@ -584,18 +615,26 @@ async def test_delete_old_logs_aborts_after_consecutive_failures(monkeypatch):
import litellm.proxy.db.db_transaction_queue.spend_log_cleanup as cleanup_module
# Lower the threshold so the test is fast and deterministic.
- monkeypatch.setattr(cleanup_module, "SPEND_LOG_CLEANUP_MAX_CONSECUTIVE_BATCH_FAILURES", 3)
- monkeypatch.setattr(cleanup_module, "SPEND_LOG_CLEANUP_BATCH_FAILURE_BACKOFF_SECONDS", 0.0)
+ monkeypatch.setattr(
+ cleanup_module, "SPEND_LOG_CLEANUP_MAX_CONSECUTIVE_BATCH_FAILURES", 3
+ )
+ monkeypatch.setattr(
+ cleanup_module, "SPEND_LOG_CLEANUP_BATCH_FAILURE_BACKOFF_SECONDS", 0.0
+ )
mock_prisma_client = MagicMock()
_wire_tx(mock_prisma_client.db)
mock_db = MagicMock()
_wire_tx(mock_db)
# Every batch raises — must abort after exactly 3 attempts, not loop forever.
- mock_db.execute_raw = AsyncMock(side_effect=ConnectionError("simulated persistent DB outage"))
+ mock_db.execute_raw = AsyncMock(
+ side_effect=ConnectionError("simulated persistent DB outage")
+ )
mock_prisma_client.db = mock_db
- cleaner = cleanup_module.SpendLogCleanup(general_settings={"maximum_spend_logs_retention_period": "7d"})
+ cleaner = cleanup_module.SpendLogCleanup(
+ general_settings={"maximum_spend_logs_retention_period": "7d"}
+ )
cutoff_date = datetime.now(timezone.utc) - timedelta(days=7)
result = await cleaner._delete_old_logs(mock_prisma_client, cutoff_date, _far_deadline())
@@ -610,8 +649,12 @@ async def test_delete_old_logs_resets_consecutive_failures_on_success(monkeypatc
intermittent timeouts don't trip the abort threshold."""
import litellm.proxy.db.db_transaction_queue.spend_log_cleanup as cleanup_module
- monkeypatch.setattr(cleanup_module, "SPEND_LOG_CLEANUP_MAX_CONSECUTIVE_BATCH_FAILURES", 3)
- monkeypatch.setattr(cleanup_module, "SPEND_LOG_CLEANUP_BATCH_FAILURE_BACKOFF_SECONDS", 0.0)
+ monkeypatch.setattr(
+ cleanup_module, "SPEND_LOG_CLEANUP_MAX_CONSECUTIVE_BATCH_FAILURES", 3
+ )
+ monkeypatch.setattr(
+ cleanup_module, "SPEND_LOG_CLEANUP_BATCH_FAILURE_BACKOFF_SECONDS", 0.0
+ )
mock_prisma_client = MagicMock()
_wire_tx(mock_prisma_client.db)
@@ -632,7 +675,9 @@ async def test_delete_old_logs_resets_consecutive_failures_on_success(monkeypatc
)
mock_prisma_client.db = mock_db
- cleaner = cleanup_module.SpendLogCleanup(general_settings={"maximum_spend_logs_retention_period": "7d"})
+ cleaner = cleanup_module.SpendLogCleanup(
+ general_settings={"maximum_spend_logs_retention_period": "7d"}
+ )
cutoff_date = datetime.now(timezone.utc) - timedelta(days=7)
result = await cleaner._delete_old_logs(mock_prisma_client, cutoff_date, _far_deadline())
@@ -653,7 +698,9 @@ async def test_cleanup_uses_logger_exception_for_full_traceback(monkeypatch):
mock_prisma_client = MagicMock()
_wire_tx(mock_prisma_client.db)
# Force the outer try/except to fire by making _should_delete_spend_logs raise.
- cleaner = cleanup_module.SpendLogCleanup(general_settings={"maximum_spend_logs_retention_period": "7d"})
+ cleaner = cleanup_module.SpendLogCleanup(
+ general_settings={"maximum_spend_logs_retention_period": "7d"}
+ )
cleaner.pod_lock_manager = None
def boom():
@@ -678,8 +725,12 @@ async def test_cleanup_releases_lock_after_persistent_batch_failures(monkeypatch
must still be released so the next scheduled run isn't permanently blocked."""
import litellm.proxy.db.db_transaction_queue.spend_log_cleanup as cleanup_module
- monkeypatch.setattr(cleanup_module, "SPEND_LOG_CLEANUP_MAX_CONSECUTIVE_BATCH_FAILURES", 2)
- monkeypatch.setattr(cleanup_module, "SPEND_LOG_CLEANUP_BATCH_FAILURE_BACKOFF_SECONDS", 0.0)
+ monkeypatch.setattr(
+ cleanup_module, "SPEND_LOG_CLEANUP_MAX_CONSECUTIVE_BATCH_FAILURES", 2
+ )
+ monkeypatch.setattr(
+ cleanup_module, "SPEND_LOG_CLEANUP_BATCH_FAILURE_BACKOFF_SECONDS", 0.0
+ )
mock_prisma_client = MagicMock()
_wire_tx(mock_prisma_client.db)
@@ -693,7 +744,9 @@ async def test_cleanup_releases_lock_after_persistent_batch_failures(monkeypatch
mock_pod_lock_manager.acquire_lock = AsyncMock(return_value=True)
mock_pod_lock_manager.release_lock = AsyncMock()
- cleaner = cleanup_module.SpendLogCleanup(general_settings={"maximum_spend_logs_retention_period": "7d"})
+ cleaner = cleanup_module.SpendLogCleanup(
+ general_settings={"maximum_spend_logs_retention_period": "7d"}
+ )
cleaner.pod_lock_manager = mock_pod_lock_manager
await cleaner.cleanup_old_spend_logs(mock_prisma_client)
@@ -943,7 +996,9 @@ async def test_each_batch_carries_a_statement_and_lock_timeout():
}
)
- await cleaner._delete_old_logs(mock_prisma_client, datetime.now(timezone.utc) - timedelta(days=7), _far_deadline())
+ await cleaner._delete_old_logs(
+ mock_prisma_client, datetime.now(timezone.utc) - timedelta(days=7), _far_deadline()
+ )
assert "SET LOCAL statement_timeout = 12000" in recorded
assert "SET LOCAL lock_timeout = 12000" in recorded
@@ -1079,7 +1134,9 @@ async def test_remaining_rows_probe_is_capped_so_it_cannot_scan_the_table():
cleaner = SpendLogCleanup(general_settings={"maximum_spend_logs_retention_period": "7d"})
- await cleaner._delete_old_logs(mock_prisma_client, datetime.now(timezone.utc) - timedelta(days=7), _far_deadline())
+ await cleaner._delete_old_logs(
+ mock_prisma_client, datetime.now(timezone.utc) - timedelta(days=7), _far_deadline()
+ )
count_sql = mock_db.query_raw.call_args[0][0]
assert "count(*)" in count_sql
From 29bd2ceb2b5c8f2a172d6b08e6c54cb20c41daaa Mon Sep 17 00:00:00 2001
From: yucheng
Date: Sun, 20 Sep 2026 08:27:06 +0000
Subject: [PATCH 037/160] fix(proxy): avoid mutable shutdown wait set
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
---
litellm/proxy/shutdown/scheduled_jobs.py | 5 +++--
1 file changed, 3 insertions(+), 2 deletions(-)
diff --git a/litellm/proxy/shutdown/scheduled_jobs.py b/litellm/proxy/shutdown/scheduled_jobs.py
index e7625a73b47..5345d380112 100644
--- a/litellm/proxy/shutdown/scheduled_jobs.py
+++ b/litellm/proxy/shutdown/scheduled_jobs.py
@@ -50,12 +50,13 @@ async def stop_in_flight_scheduler_jobs(scheduler: StoppableScheduler, executor:
if not scheduler.running:
return
in_flight: Final = executor.in_flight_jobs()
- still_running: set[asyncio.Future[object]] = set()
if in_flight:
verbose_proxy_logger.info(
"Waiting up to %ss for %d in-flight scheduled job(s) to finish", JOB_FINISH_TIMEOUT_SECONDS, len(in_flight)
)
- _done, still_running = await asyncio.wait(in_flight, timeout=JOB_FINISH_TIMEOUT_SECONDS)
+ still_running: Final = (
+ (await asyncio.wait(in_flight, timeout=JOB_FINISH_TIMEOUT_SECONDS))[1] if in_flight else frozenset()
+ )
scheduler.shutdown(wait=False)
if not still_running:
return
From 6c13e0912b43e54ba912f8257ce436d28e6e8e01 Mon Sep 17 00:00:00 2001
From: ryan
Date: Sun, 20 Sep 2026 08:59:30 +0000
Subject: [PATCH 038/160] fix(proxy): apply DB-stored callback redaction
settings before logger init
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
---
litellm/constants.py | 6 ++++
tests/test_litellm/proxy/test_proxy_server.py | 36 +++++++++++++++++++
2 files changed, 42 insertions(+)
diff --git a/litellm/constants.py b/litellm/constants.py
index bbeb4846e27..4bc2e4e0d26 100644
--- a/litellm/constants.py
+++ b/litellm/constants.py
@@ -1853,6 +1853,12 @@ LITELLM_SETTINGS_SAFE_DB_OVERRIDES: Final = [
"max_ui_session_budget",
"budget_rollover",
"mcp_tool_search",
+ "turn_off_message_logging",
+ "datadog_params",
+ "datadog_llm_observability_params",
+ "newrelic_params",
+ "pointfive_params",
+ "aws_sqs_callback_params",
]
SPECIAL_LITELLM_AUTH_TOKEN: Final = ["ui-token"]
DEFAULT_MANAGEMENT_OBJECT_IN_MEMORY_CACHE_TTL = int(os.getenv("DEFAULT_MANAGEMENT_OBJECT_IN_MEMORY_CACHE_TTL", 60))
diff --git a/tests/test_litellm/proxy/test_proxy_server.py b/tests/test_litellm/proxy/test_proxy_server.py
index f71f9c20f3b..75741b125cb 100644
--- a/tests/test_litellm/proxy/test_proxy_server.py
+++ b/tests/test_litellm/proxy/test_proxy_server.py
@@ -11766,6 +11766,42 @@ def test_prompt_caching_settings_propagate_on_config_reload(monkeypatch, field_n
assert getattr(litellm, field_name) == db_value
+@pytest.mark.asyncio
+async def test_db_stored_datadog_redaction_settings_apply_before_logger_init(monkeypatch):
+ """A DB-only litellm_settings row that pairs success_callback: ["datadog"] with
+ datadog_params.turn_off_message_logging: true must build the DataDogLogger redacted, the
+ same as the identical block in YAML. Regression for the redaction keys being absent from
+ the safe-override allowlist while the callback half of the row was honoured."""
+ import litellm.proxy.proxy_server as ps
+ from litellm.integrations.datadog.datadog import DataDogLogger
+ from litellm.litellm_core_utils import litellm_logging
+
+ monkeypatch.setenv("DD_API_KEY", "test-key")
+ monkeypatch.setenv("DD_SITE", "us5.datadoghq.com")
+ monkeypatch.setattr(litellm, "datadog_params", None)
+ monkeypatch.setattr(litellm, "turn_off_message_logging", False)
+ monkeypatch.setattr(litellm, "success_callback", [])
+ monkeypatch.setattr(litellm, "_async_success_callback", [])
+ monkeypatch.setattr(litellm, "failure_callback", [])
+ monkeypatch.setattr(litellm, "_async_failure_callback", [])
+ monkeypatch.setattr(litellm, "callbacks", [])
+ monkeypatch.setattr(litellm_logging, "_in_memory_loggers", [])
+
+ db_row = {
+ "success_callback": ["datadog"],
+ "datadog_params": {"turn_off_message_logging": True},
+ "turn_off_message_logging": True,
+ }
+ pc = ps.ProxyConfig()
+ pc._apply_litellm_settings_db_values(pc._prepared_db_settings_values("litellm_settings", db_row))
+ pc._add_callbacks_from_db_config({"litellm_settings": db_row})
+
+ datadog_loggers = [cb for cb in litellm.success_callback if isinstance(cb, DataDogLogger)]
+ assert len(datadog_loggers) == 1
+ assert datadog_loggers[0].turn_off_message_logging is True
+ assert litellm.turn_off_message_logging is True
+
+
def test_get_config_list_marks_untouched_prompt_caching_flag_as_not_set(monkeypatch):
"""The flag defaults to False rather than None, so a plain 'is not None' check would
report the default as 'In Config' and imply an admin had set it."""
From a90d852d378b46e3541fbb76ec519845e5caa785 Mon Sep 17 00:00:00 2001
From: ryan
Date: Sun, 20 Sep 2026 09:26:30 +0000
Subject: [PATCH 039/160] test(proxy): type monkeypatch and cover every
DB-overridable callback params key
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
---
tests/test_litellm/proxy/test_proxy_server.py | 26 ++++++++++++++++++-
1 file changed, 25 insertions(+), 1 deletion(-)
diff --git a/tests/test_litellm/proxy/test_proxy_server.py b/tests/test_litellm/proxy/test_proxy_server.py
index 75741b125cb..c56645313c3 100644
--- a/tests/test_litellm/proxy/test_proxy_server.py
+++ b/tests/test_litellm/proxy/test_proxy_server.py
@@ -11767,7 +11767,7 @@ def test_prompt_caching_settings_propagate_on_config_reload(monkeypatch, field_n
@pytest.mark.asyncio
-async def test_db_stored_datadog_redaction_settings_apply_before_logger_init(monkeypatch):
+async def test_db_stored_datadog_redaction_settings_apply_before_logger_init(monkeypatch: pytest.MonkeyPatch):
"""A DB-only litellm_settings row that pairs success_callback: ["datadog"] with
datadog_params.turn_off_message_logging: true must build the DataDogLogger redacted, the
same as the identical block in YAML. Regression for the redaction keys being absent from
@@ -11802,6 +11802,30 @@ async def test_db_stored_datadog_redaction_settings_apply_before_logger_init(mon
assert litellm.turn_off_message_logging is True
+@pytest.mark.parametrize(
+ "field_name",
+ [
+ "datadog_params",
+ "datadog_llm_observability_params",
+ "newrelic_params",
+ "pointfive_params",
+ "aws_sqs_callback_params",
+ ],
+)
+def test_db_stored_callback_params_propagate_to_litellm_module(monkeypatch: pytest.MonkeyPatch, field_name: str):
+ """Every callback init params block stored in the DB litellm_settings row must land on the
+ litellm module before the matching logger is built, so the DB row behaves like YAML."""
+ import litellm.proxy.proxy_server as ps
+
+ monkeypatch.setattr(litellm, field_name, None)
+ db_value = {"turn_off_message_logging": True}
+
+ pc = ps.ProxyConfig()
+ pc._apply_litellm_settings_db_values(pc._prepared_db_settings_values("litellm_settings", {field_name: db_value}))
+
+ assert getattr(litellm, field_name) == db_value
+
+
def test_get_config_list_marks_untouched_prompt_caching_flag_as_not_set(monkeypatch):
"""The flag defaults to False rather than None, so a plain 'is not None' check would
report the default as 'In Config' and imply an admin had set it."""
From d338d3f2d2f6529de70a273b6f0d622708209c9c Mon Sep 17 00:00:00 2001
From: yassin
Date: Sun, 20 Sep 2026 09:40:46 +0000
Subject: [PATCH 040/160] style(agents): tidy typing and docstring in access
group ceiling helpers
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
---
litellm/proxy/auth/auth_checks.py | 2 +-
litellm/proxy/utils.py | 2 +-
.../proxy/agent_endpoints/test_agent_registry.py | 5 ++++-
tests/test_litellm/proxy/auth/test_auth_checks.py | 3 ---
4 files changed, 6 insertions(+), 6 deletions(-)
diff --git a/litellm/proxy/auth/auth_checks.py b/litellm/proxy/auth/auth_checks.py
index 5412a9d9f6a..b5e7ef73d36 100644
--- a/litellm/proxy/auth/auth_checks.py
+++ b/litellm/proxy/auth/auth_checks.py
@@ -4332,7 +4332,7 @@ async def _check_agent_access_group_model_access(
llm_router: Router | None,
resolve_ceiling: CeilingResolver = resolve_agent_access_group_ceiling,
) -> Literal[True]:
- """Attached groups naming no model deny every model, unlike the empty allowlist ``_can_object_call_model`` allows."""
+ """Attached groups naming no model deny every model; the empty allowlist in ``_can_object_call_model`` allows."""
if not model or valid_token is None or not valid_token.agent_id:
return True
ceiling: Final = await resolve_ceiling(valid_token.agent_id)
diff --git a/litellm/proxy/utils.py b/litellm/proxy/utils.py
index e736f2fa1c4..a8e7d3232eb 100644
--- a/litellm/proxy/utils.py
+++ b/litellm/proxy/utils.py
@@ -8188,7 +8188,7 @@ async def _get_access_group_models(
async def _agent_access_group_visible_models(
user_api_key_dict: "UserAPIKeyAuth",
- llm_router: Optional["Router"],
+ llm_router: "Router | None",
include_model_access_groups: bool,
return_wildcard_routes: bool,
team_id: str | None,
diff --git a/tests/test_litellm/proxy/agent_endpoints/test_agent_registry.py b/tests/test_litellm/proxy/agent_endpoints/test_agent_registry.py
index d15a3adadbd..b036e0dac4d 100644
--- a/tests/test_litellm/proxy/agent_endpoints/test_agent_registry.py
+++ b/tests/test_litellm/proxy/agent_endpoints/test_agent_registry.py
@@ -15,6 +15,7 @@ from litellm.proxy.agent_endpoints.agent_registry import (
_restore_redacted_litellm_params,
redact_sensitive_agent_litellm_params,
)
+from litellm.types.agents import PatchAgentRequest
# Obviously-fake stand-ins for a real AWS credential pair (LIT-6736 regression
# fixtures) -- never a real key shape, and must never appear in any response.
@@ -1052,7 +1053,9 @@ async def test_add_agent_to_db_without_access_group_ids_leaves_column_to_its_def
({"access_group_ids": None}, []),
],
)
-async def test_patch_agent_in_db_replaces_access_group_ids_when_provided(patch_body: dict, expected: list[str]):
+async def test_patch_agent_in_db_replaces_access_group_ids_when_provided(
+ patch_body: PatchAgentRequest, expected: list[str]
+):
registry: Final = AgentRegistry()
mock_prisma: Final = MagicMock()
mock_prisma.db.litellm_agentstable.find_unique = AsyncMock(
diff --git a/tests/test_litellm/proxy/auth/test_auth_checks.py b/tests/test_litellm/proxy/auth/test_auth_checks.py
index d8cd578d265..a1179617718 100644
--- a/tests/test_litellm/proxy/auth/test_auth_checks.py
+++ b/tests/test_litellm/proxy/auth/test_auth_checks.py
@@ -8900,9 +8900,6 @@ def test_request_skips_budget_checks_extends_route_rule_with_zero_cost_models()
assert request_skips_budget_checks(route="/v1/chat/completions", model=None, llm_router=None) is False
-# Agent access group model ceiling
-
-
def _agent_model_ceiling_resolver(
models: frozenset[str] | None,
) -> tuple[CeilingResolver, list[str]]:
From 3fe405a5ccdf568ce16d82bd08c4b98e936909bf Mon Sep 17 00:00:00 2001
From: ryan
Date: Mon, 21 Sep 2026 18:48:35 +0000
Subject: [PATCH 041/160] feat(auth): breached password detection, self-service
change-password and forced password reset
Cherry-pick of merge commit b3882d8e43 (PRs #39321, #39562, #40107), which landed on litellm_internal_staging instead of main.
Co-authored-by: ojensen-berri
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
---
.../migration.sql | 3 +
.../litellm_proxy_extras/schema.prisma | 2 +
litellm/constants.py | 3 +
litellm/models/user.py | 2 +
litellm/proxy/_types.py | 31 +-
litellm/proxy/auth/login_utils.py | 87 ++-
litellm/proxy/auth/password_policy.py | 128 +++-
litellm/proxy/auth/route_checks.py | 27 +-
.../internal_user_endpoints.py | 62 +-
.../password_endpoints.py | 126 ++++
litellm/proxy/management_endpoints/ui_sso.py | 1 +
litellm/proxy/proxy_server.py | 16 +-
litellm/proxy/schema.prisma | 2 +
litellm/types/llms/custom_http.py | 1 +
litellm/types/proxy/ui_sso.py | 3 +-
schema.prisma | 2 +
.../endpointaudit/coverage_allowlist.txt | 1 +
.../proxy/auth/test_login_utils.py | 272 ++++++++
.../proxy/auth/test_onboarding.py | 144 +++-
.../proxy/auth/test_password_policy.py | 209 ++++++
.../proxy/auth/test_route_checks.py | 274 +++++---
.../test_internal_user_endpoints.py | 623 ++++++++----------
.../test_password_endpoints.py | 331 ++++++++++
tests/test_litellm/proxy/test__types.py | 42 ++
tests/unit/models/test_models.py | 6 +-
.../ChangePasswordForm.integration.test.tsx | 110 ++++
.../change-password/ChangePasswordForm.tsx | 120 ++++
.../app/(dashboard)/change-password/page.tsx | 7 +
.../app/(dashboard)/hooks/useAuthorized.ts | 2 +
.../src/app/(dashboard)/layout.test.tsx | 58 +-
.../src/app/(dashboard)/layout.tsx | 13 +-
.../Navbar/UserDropdown/UserDropdown.test.tsx | 52 +-
.../Navbar/UserDropdown/UserDropdown.tsx | 23 +-
.../SidebarAccountMenu.test.tsx | 43 ++
.../SidebarAccountMenu/SidebarAccountMenu.tsx | 24 +-
.../src/components/leftnav.test.tsx | 1 +
.../src/components/networking.tsx | 14 +
.../src/contexts/AuthContext.tsx | 4 +
ui/litellm-dashboard/src/lib/http/schema.d.ts | 90 ++-
39 files changed, 2456 insertions(+), 503 deletions(-)
create mode 100644 litellm-proxy-extras/litellm_proxy_extras/migrations/20260921000000_add_password_reset_columns/migration.sql
create mode 100644 litellm/proxy/management_endpoints/password_endpoints.py
create mode 100644 tests/test_litellm/proxy/management_endpoints/test_password_endpoints.py
create mode 100644 ui/litellm-dashboard/src/app/(dashboard)/change-password/ChangePasswordForm.integration.test.tsx
create mode 100644 ui/litellm-dashboard/src/app/(dashboard)/change-password/ChangePasswordForm.tsx
create mode 100644 ui/litellm-dashboard/src/app/(dashboard)/change-password/page.tsx
diff --git a/litellm-proxy-extras/litellm_proxy_extras/migrations/20260921000000_add_password_reset_columns/migration.sql b/litellm-proxy-extras/litellm_proxy_extras/migrations/20260921000000_add_password_reset_columns/migration.sql
new file mode 100644
index 00000000000..960b0d4d7eb
--- /dev/null
+++ b/litellm-proxy-extras/litellm_proxy_extras/migrations/20260921000000_add_password_reset_columns/migration.sql
@@ -0,0 +1,3 @@
+ALTER TABLE "LiteLLM_UserTable" ADD COLUMN IF NOT EXISTS "password_reset_required" BOOLEAN;
+
+ALTER TABLE "LiteLLM_UserTable" ADD COLUMN IF NOT EXISTS "last_breach_check_at" TIMESTAMP(3);
diff --git a/litellm-proxy-extras/litellm_proxy_extras/schema.prisma b/litellm-proxy-extras/litellm_proxy_extras/schema.prisma
index d2032cec0d0..bcafa6dbd0e 100644
--- a/litellm-proxy-extras/litellm_proxy_extras/schema.prisma
+++ b/litellm-proxy-extras/litellm_proxy_extras/schema.prisma
@@ -246,6 +246,8 @@ model LiteLLM_UserTable {
organization_id String?
object_permission_id String?
password String?
+ password_reset_required Boolean?
+ last_breach_check_at DateTime?
teams String[] @default([])
user_role String?
max_budget Float?
diff --git a/litellm/constants.py b/litellm/constants.py
index bbeb4846e27..b5b3647e525 100644
--- a/litellm/constants.py
+++ b/litellm/constants.py
@@ -2114,3 +2114,6 @@ BATCH_ENQUEUED_TOKEN_LIMIT_METADATA_KEY: Final = "batch_enqueued_token_limit"
# Shared read-only empty mapping, for defaulting optional Mapping parameters without
# constructing a fresh mutable dict at each call site.
EMPTY_MAPPING: Final = MappingProxyType({})
+
+# API endpoint for breached password k-anonymity search
+HIBP_RANGE_API_BASE: Final = "https://api.pwnedpasswords.com/range"
diff --git a/litellm/models/user.py b/litellm/models/user.py
index 82f78c28078..92aca87d303 100644
--- a/litellm/models/user.py
+++ b/litellm/models/user.py
@@ -24,6 +24,8 @@ class LiteLLM_UserTable(LiteLLMPydanticObjectBase):
organization_id: str | None = None
object_permission_id: str | None = None
password: str | None = Field(default=None, exclude=True)
+ password_reset_required: bool | None = None
+ last_breach_check_at: datetime | None = None
teams: list[str] = []
user_role: str | None = None
max_budget: float | None = None
diff --git a/litellm/proxy/_types.py b/litellm/proxy/_types.py
index 3b34440d0bf..16b405d8e02 100644
--- a/litellm/proxy/_types.py
+++ b/litellm/proxy/_types.py
@@ -904,6 +904,7 @@ class LiteLLMRoutes(enum.Enum):
"/claude_code_gateway/v1/traces",
"/user/list", # org admins checked in endpoint; non-admins get 403
"/management/v1/users/bulk_delete", # proxy admins delete anyone, org admins only their orgs' users; others 403
+ "/user/password/change", # endpoint only ever writes the caller's own row
"/model/{model_id}/update",
"/prompt/list",
"/prompt/info",
@@ -1864,6 +1865,17 @@ class NewUserRequest(GenerateRequestBase):
send_invite_email: bool | None = None
sso_user_id: str | None = None
organizations: list[str] | None = None
+ password: str | None = None
+
+ @field_validator("password")
+ @classmethod
+ def password_not_supported(cls, value: str | None) -> str | None:
+ if value is not None:
+ raise ValueError(
+ "password cannot be set via /user/new. Users set their own password through an "
+ "invitation link (POST /invitation/new)."
+ )
+ return value
class NewUserResponse(GenerateKeyResponse):
@@ -1886,7 +1898,8 @@ class NewUserResponse(GenerateKeyResponse):
class UpdateUserRequestNoUserIDorEmail(GenerateRequestBase): # shared with BulkUpdateUserRequest
- password: str | None = None
+ # repr=False keeps the plaintext out of management-endpoint alerts, which str() the request model
+ password: str | None = Field(default=None, repr=False)
spend: float | None = None
metadata: dict | None = None
user_alias: str | None = None
@@ -1916,6 +1929,16 @@ class UpdateUserRequest(UpdateUserRequestNoUserIDorEmail):
return values
+class ChangePasswordRequest(LiteLLMPydanticObjectBase):
+ current_password: str = Field(repr=False)
+ new_password: str = Field(repr=False)
+
+
+class ChangePasswordResponse(LiteLLMPydanticObjectBase):
+ user_id: str
+ message: str
+
+
class DeleteUserRequest(LiteLLMPydanticObjectBase):
user_ids: list[str] # required
@@ -3937,6 +3960,12 @@ class AllCallbacks(LiteLLMPydanticObjectBase):
)
+class HTTPExceptionErrorDetail(TypedDict):
+ """The `{"error": }` shape most proxy endpoints raise as `HTTPException.detail`."""
+
+ error: ReadOnly[str]
+
+
class SpendLogsRouterMetadata(TypedDict):
"""
Router provenance stamped on spend logs for deployments flagged with
diff --git a/litellm/proxy/auth/login_utils.py b/litellm/proxy/auth/login_utils.py
index e0d599b0017..0dddb16b531 100644
--- a/litellm/proxy/auth/login_utils.py
+++ b/litellm/proxy/auth/login_utils.py
@@ -10,14 +10,16 @@ import secrets
from collections.abc import Mapping
from datetime import datetime, timedelta, timezone
from types import MappingProxyType
-from typing import Final, Literal, cast
+from typing import TYPE_CHECKING, Final, Literal, cast
import jwt
from fastapi import HTTPException
import litellm
+from litellm._logging import verbose_proxy_logger
from litellm.constants import LITELLM_PROXY_ADMIN_NAME, LITELLM_UI_SESSION_DURATION
from litellm.litellm_core_utils.duration_parser import duration_in_seconds
+from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler
from litellm.proxy._types import (
LiteLLM_UserTable,
LitellmUserRoles,
@@ -28,6 +30,7 @@ from litellm.proxy._types import (
)
from litellm.proxy.auth.auth_utils import is_sso_provider_fully_configured
from litellm.proxy.auth.login_throttle import LoginAttempt, LoginThrottle
+from litellm.proxy.auth.password_policy import is_breach_check_enabled, is_password_breached
from litellm.proxy.management_endpoints.internal_user_endpoints import user_update
from litellm.proxy.management_endpoints.key_management_endpoints import (
generate_key_helper_fn,
@@ -50,6 +53,56 @@ INVALID_UI_CREDENTIALS_MESSAGE: Final = (
)
INVALID_USER_PASSWORD_MESSAGE: Final = "Invalid credentials used to access UI. Check the password set for your user"
+if TYPE_CHECKING:
+ from prisma import types as prisma_types
+
+BREACH_RECHECK_INTERVAL: Final = timedelta(hours=24)
+PASSWORD_RESET_ALLOWED_ROUTES: Final = ("/user/password/change",)
+
+
+def _breach_recheck_due(last_breach_check_at: datetime | None) -> bool:
+ if last_breach_check_at is None:
+ return True
+ last_checked_utc: Final = (
+ last_breach_check_at
+ if last_breach_check_at.tzinfo is not None
+ else last_breach_check_at.replace(tzinfo=timezone.utc)
+ )
+ return datetime.now(timezone.utc) - last_checked_utc >= BREACH_RECHECK_INTERVAL
+
+
+async def screen_login_password_for_breach(
+ user_id: str,
+ password: str,
+ last_breach_check_at: datetime | None,
+ general_settings: Mapping[str, object],
+ prisma_client: PrismaClient,
+ client: AsyncHTTPHandler | None = None,
+) -> bool:
+ """Screens a successfully verified login password against HIBP, stamps
+ ``password_reset_required`` when breached, and returns whether a breach was
+ found so the login it runs in can restrict the session it is about to mint.
+ Fails open (HIBP or DB trouble never fails the login) and rechecks a given
+ user at most once per ``BREACH_RECHECK_INTERVAL``."""
+ if not is_breach_check_enabled(general_settings):
+ return False
+ if not _breach_recheck_due(last_breach_check_at):
+ return False
+ breached: Final = await is_password_breached(password, general_settings, client)
+ checked_at: Final = datetime.now(timezone.utc)
+ breached_update: Final[prisma_types.LiteLLM_UserTableUpdateInput] = {
+ "last_breach_check_at": checked_at,
+ "password_reset_required": True,
+ }
+ recheck_update: Final[prisma_types.LiteLLM_UserTableUpdateInput] = {"last_breach_check_at": checked_at}
+ update_data: Final = breached_update if breached else recheck_update
+ find_user: Final[prisma_types.LiteLLM_UserTableWhereInput] = {"user_id": user_id}
+ try:
+ await UserRepository(prisma_client).table.update(where=find_user, data=update_data)
+ except Exception as e: # noqa: BLE001 # a failed stamp must never surface into the login
+ verbose_proxy_logger.warning("Login-time breach screening could not update user %s: %s", user_id, e)
+ return breached
+
async def _rehash_password_if_needed(user_id: str, password: str, stored: str) -> None:
"""Rehash legacy password (SHA256) to scrypt on successful login."""
@@ -137,6 +190,7 @@ class LoginResult:
user_email: str | None
user_role: str
login_method: Literal["sso", "username_password"]
+ password_reset_required: bool
def __init__(
self,
@@ -145,12 +199,14 @@ class LoginResult:
user_email: str | None,
user_role: str,
login_method: Literal["sso", "username_password"] = "username_password",
+ password_reset_required: bool = False,
):
self.user_id = user_id
self.key = key
self.user_email = user_email
self.user_role = user_role
self.login_method = login_method
+ self.password_reset_required = password_reset_required
async def authenticate_user(
@@ -356,21 +412,26 @@ async def _sign_in(
if verify_password(password, _password):
await _rehash_password_if_needed(_user_row.user_id, password, _password)
+ breached_now: Final = prisma_client is not None and await screen_login_password_for_breach(
+ user_id=_user_row.user_id,
+ password=password,
+ last_breach_check_at=getattr(_user_row, "last_breach_check_at", None),
+ general_settings=general_settings,
+ prisma_client=prisma_client,
+ )
+ password_reset_required: Final = breached_now or getattr(_user_row, "password_reset_required", None) is True
if os.getenv("DATABASE_URL") is not None:
response = await generate_key_helper_fn(
llm_router=None,
request_type="key",
- **{
- "user_role": user_role,
- "duration": LITELLM_UI_SESSION_DURATION,
- "key_max_budget": litellm.max_ui_session_budget,
- "models": [],
- "aliases": {},
- "config": {},
- "spend": 0,
- "user_id": user_id,
- "team_id": "litellm-dashboard",
- },
+ user_role=user_role,
+ duration=LITELLM_UI_SESSION_DURATION,
+ key_max_budget=litellm.max_ui_session_budget,
+ spend=0,
+ user_id=user_id,
+ team_id="litellm-dashboard",
+ allowed_routes=list(PASSWORD_RESET_ALLOWED_ROUTES) if password_reset_required else None,
+ metadata={"password_reset_required": True} if password_reset_required else {},
)
else:
raise ProxyException(
@@ -390,6 +451,7 @@ async def _sign_in(
user_email=user_email,
user_role=cast(str, user_role),
login_method="username_password",
+ password_reset_required=password_reset_required,
)
else:
await attempt.failed()
@@ -460,4 +522,5 @@ def create_ui_token_object(
auth_header_name=general_settings.get("litellm_key_header_name", "Authorization"),
disabled_non_admin_personal_key_creation=disabled_non_admin_personal_key_creation,
server_root_path=get_server_root_path(),
+ password_reset_required=login_result.password_reset_required,
)
diff --git a/litellm/proxy/auth/password_policy.py b/litellm/proxy/auth/password_policy.py
index ab7a565894a..7f06a0993d3 100644
--- a/litellm/proxy/auth/password_policy.py
+++ b/litellm/proxy/auth/password_policy.py
@@ -4,13 +4,28 @@ Applied at every path that persists a new or changed password for a DB-backed
user (``/user/update``, ``/user/bulk_update``, and the invitation onboarding
claim flow), so the strength bar is configured in one place instead of
per-endpoint.
+
+Also screens new passwords against known data breaches via the
+haveibeenpwned.com (HIBP) k-anonymity range API: only the first 5 characters
+of the password's SHA-1 hash ever leave the proxy, and the check fails open
+(allows the password) when HIBP is unreachable.
"""
-from collections.abc import Mapping
+import asyncio
+import hashlib
+from collections.abc import Mapping, Sequence
from dataclasses import dataclass
+from types import MappingProxyType
from typing import Final
+from litellm._logging import verbose_proxy_logger
+from litellm._version import version
+from litellm.constants import HIBP_RANGE_API_BASE
+from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, get_async_httpx_client
from litellm.proxy._types import ProxyErrorTypes, ProxyException
+from litellm.types.llms.custom_http import httpxSpecialProvider
+
+HIBP_TIMEOUT_SECONDS: Final = 5.0
DEFAULT_MIN_LENGTH: Final = 12
MIN_ALLOWED_LENGTH: Final = 8
@@ -90,3 +105,114 @@ def validate_password_policy(password: str, general_settings: Mapping[str, objec
param="password",
code=400,
)
+
+
+def _hibp_client() -> AsyncHTTPHandler:
+ return get_async_httpx_client(
+ llm_provider=httpxSpecialProvider.PasswordBreachCheck,
+ params={"timeout": HIBP_TIMEOUT_SECONDS}, # mutable-ok: callee takes a bare dict (PEP 589)
+ )
+
+
+def _is_suffix_in_range_response(response_body: str, hash_suffix: str) -> bool:
+ for line in response_body.upper().splitlines():
+ entry_suffix, _, count = line.strip().partition(":")
+ if entry_suffix == hash_suffix:
+ return int(count.strip() or "0") > 0
+ return False
+
+
+async def _is_password_breached(password: str, client: AsyncHTTPHandler) -> bool:
+ # usedforsecurity=False: SHA-1 is only a lookup key into the HIBP dataset, so no security property rests on it
+ sha1_hex: Final = hashlib.sha1(password.encode("utf-8"), usedforsecurity=False).hexdigest().upper()
+ headers: Final = { # mutable-ok: callee takes a bare dict (PEP 589)
+ "Add-Padding": "true",
+ "User-Agent": f"litellm-proxy/{version}",
+ }
+ try:
+ response: Final = await client.get(
+ f"{HIBP_RANGE_API_BASE}/{sha1_hex[:5]}",
+ headers=headers,
+ )
+ response.raise_for_status()
+ breached: Final = _is_suffix_in_range_response(response.text, sha1_hex[5:])
+ except Exception as e: # noqa: BLE001 # fail-open: any HIBP failure skips the check, never breaks the caller
+ verbose_proxy_logger.warning("Breached-password check skipped, HIBP lookup failed: %s", e)
+ return False
+ return breached
+
+
+def is_breach_check_enabled(general_settings: Mapping[str, object]) -> bool:
+ return general_settings.get("password_policy_check_breached_passwords", True) is not False
+
+
+async def is_password_breached(
+ password: str,
+ general_settings: Mapping[str, object],
+ client: AsyncHTTPHandler | None = None,
+) -> bool:
+ """False when the check is disabled, the password is absent from the HIBP
+ corpus, or HIBP is unreachable (fail open)."""
+ if not is_breach_check_enabled(general_settings):
+ return False
+ return await _is_password_breached(password, client if client is not None else _hibp_client())
+
+
+def breached_password_error() -> ProxyException:
+ return ProxyException(
+ message=(
+ "This password appears in known data breaches and cannot be used. Please choose a different password."
+ ),
+ type=ProxyErrorTypes.validation_error,
+ param="password",
+ code=400,
+ )
+
+
+async def validate_password_not_breached(
+ password: str,
+ general_settings: Mapping[str, object],
+ client: AsyncHTTPHandler | None = None,
+) -> None:
+ """Raise ``ProxyException`` (400) if ``password`` appears in a known data breach.
+
+ Fails open: an unreachable or misbehaving HIBP allows the password."""
+ if not await is_password_breached(password, general_settings, client):
+ return
+ raise breached_password_error()
+
+
+def _strength_verdict(password: str, general_settings: Mapping[str, object]) -> ProxyException | None:
+ try:
+ validate_password_policy(password, general_settings)
+ except ProxyException as e:
+ return e
+ return None
+
+
+async def validate_passwords_bulk(
+ passwords: Sequence[str],
+ general_settings: Mapping[str, object],
+ client: AsyncHTTPHandler | None = None,
+) -> Mapping[str, ProxyException | None]:
+ """Per-unique-password policy verdicts for a batch: the ProxyException to
+ surface, or None when the password is acceptable.
+
+ Deduplicates first, then issues every needed HIBP lookup concurrently, so a
+ batch caller pays one HIBP timeout window in the worst case instead of one
+ per password (each lookup still fails open independently)."""
+ unique_passwords: Final = tuple(dict.fromkeys(passwords))
+ strength_verdicts: Final[Mapping[str, ProxyException | None]] = MappingProxyType(
+ {password: _strength_verdict(password, general_settings) for password in unique_passwords}
+ )
+ to_screen: Final = tuple(password for password in unique_passwords if strength_verdicts[password] is None)
+ breached_flags: Final = await asyncio.gather(
+ *(is_password_breached(password, general_settings, client) for password in to_screen)
+ )
+ breached_passwords: Final = frozenset(password for password, breached in zip(to_screen, breached_flags) if breached)
+ return MappingProxyType(
+ {
+ password: breached_password_error() if password in breached_passwords else strength_verdicts[password]
+ for password in unique_passwords
+ }
+ )
diff --git a/litellm/proxy/auth/route_checks.py b/litellm/proxy/auth/route_checks.py
index 1b9fd7c42bf..afddb4866a9 100644
--- a/litellm/proxy/auth/route_checks.py
+++ b/litellm/proxy/auth/route_checks.py
@@ -194,6 +194,16 @@ class RouteChecks:
if denied_auth_enforced_pass_through_route:
raise RouteChecks._auth_pass_through_denied_exception(route=route)
+ if valid_token.metadata.get("password_reset_required") is True:
+ raise HTTPException(
+ status_code=status.HTTP_403_FORBIDDEN,
+ detail=(
+ "This account's password must be changed before the session can be used: "
+ "it was either found in a known data breach or set by an admin. "
+ "Change it via POST /user/password/change (UI: /ui/change-password), then log in again."
+ ),
+ )
+
raise HTTPException(
status_code=status.HTTP_403_FORBIDDEN,
detail=f"Virtual key is not allowed to call this route. Only allowed to call routes: {valid_token.allowed_routes}. Tried to call route: {route}",
@@ -812,7 +822,8 @@ class RouteChecks:
in the codebase is automatically readable by Admin Viewer
without needing to remember to add it to an allowlist.
3. Unsafe HTTP method (POST/PUT/PATCH/DELETE):
- - Allow `/user/update` only when restricted to user_email/password.
+ - Allow `/user/update` only when restricted to user_email.
+ - Allow `/user/password/change` (endpoint only writes the caller's own row).
- Block all explicit writes in `_ADMIN_VIEWER_BLOCKED_WRITE_ROUTES`.
- Otherwise allow only if the route is in admin_viewer_routes /
global_spend_tracking_routes (legacy explicit-allow set).
@@ -832,10 +843,10 @@ class RouteChecks:
if request_data is not None and isinstance(request_data, dict):
_params_updated: Final = request_data.keys()
for param in _params_updated:
- if param not in ["user_email", "password"]:
+ if param != "user_email":
raise HTTPException(
status_code=status.HTTP_403_FORBIDDEN,
- detail=f"user not allowed to access this route, role= {_user_role}. Trying to access: {route} and updating invalid param: {param}. only user_email and password can be updated",
+ detail=f"user not allowed to access this route, role= {_user_role}. Trying to access: {route} and updating invalid param: {param}. only user_email can be updated",
)
elif RouteChecks.check_route_access(route=route, allowed_routes=_PROXY_ADMIN_VIEW_ONLY_BLOCKED_ROUTES) or (
route.startswith("/key/") and route.endswith(_PROXY_ADMIN_VIEW_ONLY_BLOCKED_KEY_SUFFIXES)
@@ -854,21 +865,25 @@ class RouteChecks:
return
# ── Unsafe HTTP method: explicit checks ──────────────────────────
- # Allow `/user/update` for self-service email / password change.
+ # Allow `/user/update` for self-service email change.
if route == "/user/update":
if request_data is not None and isinstance(request_data, dict):
for param in request_data:
- if param not in ["user_email", "password"]:
+ if param != "user_email":
raise HTTPException(
status_code=status.HTTP_403_FORBIDDEN,
detail=(
f"user not allowed to access this route, role= {_user_role}. "
f"Trying to access: {route} and updating invalid param: {param}. "
- "only user_email and password can be updated"
+ "only user_email can be updated"
),
)
return
+ # Self-service password change; the endpoint only writes the caller's own row.
+ if route == "/user/password/change":
+ return
+
# Hard-block known write routes regardless of HTTP method (defensive
# — these are POSTs in practice, but pinning them here protects
# against future GET-shaped writes).
diff --git a/litellm/proxy/management_endpoints/internal_user_endpoints.py b/litellm/proxy/management_endpoints/internal_user_endpoints.py
index 1c986305c21..285be23bdf6 100644
--- a/litellm/proxy/management_endpoints/internal_user_endpoints.py
+++ b/litellm/proxy/management_endpoints/internal_user_endpoints.py
@@ -28,6 +28,7 @@ from typing_extensions import ReadOnly, TypedDict
import litellm
from litellm._logging import verbose_proxy_logger
from litellm._uuid import uuid
+from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler
from litellm.proxy._types import *
from litellm.proxy.auth.auth_checks import (
delete_cache_key_objects,
@@ -35,7 +36,11 @@ from litellm.proxy.auth.auth_checks import (
get_team_object,
get_user_object,
)
-from litellm.proxy.auth.password_policy import validate_password_policy
+from litellm.proxy.auth.password_policy import (
+ validate_password_not_breached,
+ validate_password_policy,
+ validate_passwords_bulk,
+)
from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
from litellm.proxy.common_utils.auth_cache_invalidation_pubsub import evict_and_broadcast
from litellm.proxy.common_utils.user_api_key_cache import (
@@ -173,11 +178,23 @@ def _team_membership_table(
return team_membership_table
-def _hash_password_in_dict(data: dict, general_settings: Mapping[str, object]) -> None:
- """Validate and hash password field in-place if present."""
+async def _hash_password_in_dict(
+ data: dict, general_settings: Mapping[str, object], password_prevalidated: bool = False
+) -> None:
+ """Validate and hash password field in-place if present.
+
+ ``password_prevalidated`` skips the policy checks for callers that already
+ validated the password (the bulk path screens its whole batch upfront).
+
+ An admin-set password is known to whoever set it, so the user is also
+ flagged for a forced password change at next login."""
if "password" in data and data["password"] is not None:
- validate_password_policy(data["password"], general_settings)
+ if not password_prevalidated:
+ validate_password_policy(data["password"], general_settings)
+ await validate_password_not_breached(data["password"], general_settings)
data["password"] = hash_password(data["password"])
+ data["password_reset_required"] = True
+ data["last_breach_check_at"] = None
def _strip_password_from_response(response) -> None:
@@ -505,6 +522,7 @@ async def new_user(
- prompts: Optional[List[str]] - List of allowed prompts for the user. If specified, the user will only be able to use these specific prompts.
- organizations: List[str] - List of organization id's the user is a member of
- budget_limits: Optional[list] - List of concurrent budget windows for the user. Each window specifies a budget_limit, time_period, and optional budget_duration. Example - [{"budget_limit": 10.0, "time_period": "1d"}, {"budget_limit": 50.0, "time_period": "7d"}].
+ - password: Optional[str] - Not supported; any value is rejected with a 422. Users set their own password through an invitation link (POST /invitation/new).
Returns:
- key: (str) The generated api key for the user
- expires: (datetime) Datetime object for when key expires.
@@ -524,7 +542,7 @@ async def new_user(
```
"""
try:
- from litellm.proxy.proxy_server import _license_check, general_settings, prisma_client
+ from litellm.proxy.proxy_server import _license_check, prisma_client
if prisma_client is None:
raise HTTPException(status_code=400, detail=CommonProxyErrors.db_not_connected_error.value)
@@ -572,7 +590,7 @@ async def new_user(
# generate_key_helper_fn only forwards object_permission_id, so without this the entitlement
# the caller sent would be dropped on the floor.
data_json = await _set_object_permission(data_json=data_json, prisma_client=prisma_client)
- _hash_password_in_dict(data_json, general_settings)
+ data_json.pop("password", None)
teams = data.teams
if teams is None:
teams = check_if_default_team_set()
@@ -1438,6 +1456,7 @@ async def _update_single_user_helper(
user_request: UpdateUserRequest,
user_api_key_dict: UserAPIKeyAuth,
litellm_changed_by: str | None = None,
+ password_prevalidated: bool = False,
) -> dict[str, Any]:
"""
Helper function to update a single user.
@@ -1460,7 +1479,7 @@ async def _update_single_user_helper(
data_json: Final[dict] = user_request.model_dump(exclude_unset=True)
non_default_values = _update_internal_user_params(data_json=data_json, data=user_request)
- _hash_password_in_dict(non_default_values, general_settings)
+ await _hash_password_in_dict(non_default_values, general_settings, password_prevalidated=password_prevalidated)
existing_user_row: BaseModel | None = None
if user_request.user_id:
@@ -1641,7 +1660,7 @@ async def user_update(
Parameters:
- user_id: Optional[str] - Specify a user id. If not set, a unique id will be generated.
- user_email: Optional[str] - Specify a user email.
- - password: Optional[str] - Specify a user password.
+ - password: Optional[str] - Set the user's password (admin only). Must satisfy the configured password policy. The user is required to change it at their next login. Users change their own password with POST /user/password/change.
- user_alias: Optional[str] - A descriptive name for you to know who this user id refers to.
- teams: Optional[list] - specify a list of team id's a user belongs to.
- send_invite_email: Optional[bool] - Specify if an invite email should be sent.
@@ -1709,19 +1728,38 @@ async def bulk_update_processed_users(
users_to_update: list[UpdateUserRequest],
user_api_key_dict: UserAPIKeyAuth,
litellm_changed_by: str | None = None,
+ hibp_client: AsyncHTTPHandler | None = None,
) -> BulkUpdateUserResponse:
+ from litellm.proxy.proxy_server import general_settings
+
results: Final[list[UserUpdateResult]] = []
successful_updates = 0
failed_updates = 0
+ # Screen the batch's passwords upfront and concurrently: done per-user
+ # inside the loop below, each HIBP lookup would be awaited serially and a
+ # degraded-slow HIBP could stretch a full batch to minutes, timing out the
+ # request after some updates already persisted.
+ password_verdicts: Final = await validate_passwords_bulk(
+ tuple(u.password for u in users_to_update if u.password is not None),
+ general_settings,
+ client=hibp_client,
+ )
+
# Process each user update independently
try:
for user_request in users_to_update:
try:
+ if (
+ user_request.password is not None
+ and (password_error := password_verdicts.get(user_request.password)) is not None
+ ):
+ raise password_error
response = await _update_single_user_helper(
user_request=user_request,
user_api_key_dict=user_api_key_dict,
litellm_changed_by=litellm_changed_by,
+ password_prevalidated=True,
)
# Record success
results.append(
@@ -1859,6 +1897,14 @@ async def bulk_user_update(
status_code=403,
detail="Only proxy admins can update all users at once.",
)
+ if data.user_updates.password is not None:
+ bulk_password_error: Final[HTTPExceptionErrorDetail] = {
+ "error": (
+ "Setting one password for all users is not supported. "
+ "Use per-user updates via the 'users' list instead."
+ )
+ }
+ raise HTTPException(status_code=400, detail=bulk_password_error)
# Optimized path for updating all users directly in database
all_users_in_db: Final = await _user_table(prisma_client).find_many(order={"created_at": "desc"})
diff --git a/litellm/proxy/management_endpoints/password_endpoints.py b/litellm/proxy/management_endpoints/password_endpoints.py
new file mode 100644
index 00000000000..bd3c9722d4d
--- /dev/null
+++ b/litellm/proxy/management_endpoints/password_endpoints.py
@@ -0,0 +1,126 @@
+"""
+Self-service password management.
+
+/user/password/change
+
+Deliberately NOT wrapped in `management_endpoint_wrapper`: the wrapper emits
+request kwargs to OTEL spans, which would log plaintext passwords. The audit
+signal is emitted by hand below, with field names only, never values.
+"""
+
+from typing import TYPE_CHECKING, Final
+
+from fastapi import APIRouter, Depends, HTTPException
+
+from litellm._logging import verbose_proxy_logger
+from litellm.proxy._types import (
+ ChangePasswordRequest,
+ ChangePasswordResponse,
+ CommonProxyErrors,
+ HTTPExceptionErrorDetail,
+ LitellmTableNames,
+ UserAPIKeyAuth,
+)
+from litellm.proxy.auth.password_policy import validate_password_not_breached, validate_password_policy
+from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
+from litellm.proxy.management_helpers.audit_logs import create_object_audit_log
+from litellm.proxy.utils import hash_password, verify_password
+from litellm.repositories.prisma_protocols import TableActions
+from litellm.repositories.user_repository import UserRepository
+
+if TYPE_CHECKING:
+ from prisma import models as prisma_models
+ from prisma import types as prisma_types
+
+ from litellm.proxy.utils import PrismaClient
+
+router: Final = APIRouter()
+
+_PASSWORD_CHANGED_AUDIT_VALUES: Final = '{"fields_changed": ["password"]}'
+
+
+def _error_detail(message: str) -> HTTPExceptionErrorDetail:
+ detail: Final[HTTPExceptionErrorDetail] = {"error": message}
+ return detail
+
+
+def _user_table(
+ prisma_client: "PrismaClient | None",
+) -> "TableActions[prisma_models.LiteLLM_UserTable]":
+ user_table: Final[TableActions[prisma_models.LiteLLM_UserTable]] = UserRepository(prisma_client).table
+ return user_table
+
+
+@router.post(
+ "/user/password/change",
+ tags=("Internal User management",),
+ dependencies=(Depends(user_api_key_auth),),
+)
+async def change_password(
+ data: ChangePasswordRequest,
+ user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth),
+) -> ChangePasswordResponse:
+ """
+ Change the calling user's own password.
+
+ Requires the current password. The new password must satisfy the
+ configured password policy (`general_settings.password_policy_*`: minimum
+ length, character classes, and, when enabled, breached-password screening
+ via haveibeenpwned.com). A successful change lifts any pending forced
+ password reset (`password_reset_required`) on the account.
+
+ Parameters:
+ - current_password: str - The user's current password.
+ - new_password: str - The password to change to.
+ """
+ from litellm.proxy.proxy_server import general_settings, litellm_proxy_admin_name, prisma_client
+
+ if prisma_client is None:
+ raise HTTPException(
+ status_code=500,
+ detail=_error_detail(CommonProxyErrors.db_not_connected_error.value),
+ )
+
+ user_id: Final = user_api_key_dict.user_id
+ if user_id is None:
+ raise HTTPException(
+ status_code=400,
+ detail=_error_detail("No user is associated with this session, so there is no password to change."),
+ )
+
+ find_user: Final[prisma_types.LiteLLM_UserTableWhereInput] = {"user_id": user_id}
+ user_row: Final = await _user_table(prisma_client).find_first(where=find_user)
+ stored_password: Final = user_row.password if user_row is not None else None
+ if stored_password is None:
+ raise HTTPException(
+ status_code=400,
+ detail=_error_detail(
+ "This account has no password set, so there is no password to change. "
+ "Passwords are set through an invitation link (POST /invitation/new)."
+ ),
+ )
+
+ if not verify_password(data.current_password, stored_password):
+ raise HTTPException(status_code=400, detail=_error_detail("Current password is incorrect."))
+
+ validate_password_policy(data.new_password, general_settings)
+ await validate_password_not_breached(data.new_password, general_settings)
+
+ password_update: Final[prisma_types.LiteLLM_UserTableUpdateInput] = {
+ "password": hash_password(data.new_password),
+ "password_reset_required": False,
+ "last_breach_check_at": None,
+ }
+ await _user_table(prisma_client).update(where=find_user, data=password_update)
+
+ verbose_proxy_logger.info("Password changed via /user/password/change for user_id=%s", user_id)
+ await create_object_audit_log(
+ object_id=user_id,
+ action="updated",
+ litellm_changed_by=None,
+ user_api_key_dict=user_api_key_dict,
+ litellm_proxy_admin_name=litellm_proxy_admin_name,
+ table_name=LitellmTableNames.USER_TABLE_NAME,
+ after_value=_PASSWORD_CHANGED_AUDIT_VALUES,
+ )
+ return ChangePasswordResponse(user_id=user_id, message="Password updated successfully.")
diff --git a/litellm/proxy/management_endpoints/ui_sso.py b/litellm/proxy/management_endpoints/ui_sso.py
index 00cf357d89d..7859c678c07 100644
--- a/litellm/proxy/management_endpoints/ui_sso.py
+++ b/litellm/proxy/management_endpoints/ui_sso.py
@@ -3665,6 +3665,7 @@ class SSOAuthenticationHandler:
auth_header_name=general_settings.get("litellm_key_header_name", "Authorization"),
disabled_non_admin_personal_key_creation=disabled_non_admin_personal_key_creation,
server_root_path=get_server_root_path(),
+ password_reset_required=False,
)
from litellm.proxy.auth.login_utils import encode_ui_session_jwt
diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py
index 3a06753834a..638d685d3e0 100644
--- a/litellm/proxy/proxy_server.py
+++ b/litellm/proxy/proxy_server.py
@@ -359,7 +359,7 @@ from litellm.proxy.auth.model_checks import (
get_mcp_server_ids,
get_team_models,
)
-from litellm.proxy.auth.password_policy import validate_password_policy
+from litellm.proxy.auth.password_policy import validate_password_not_breached, validate_password_policy
from litellm.proxy.auth.user_api_key_auth import (
_fetch_global_spend_with_event_coordination,
user_api_key_auth,
@@ -601,6 +601,9 @@ from litellm.proxy.management_endpoints.model_management_endpoints import (
from litellm.proxy.management_endpoints.organization_endpoints import (
router as organization_router,
)
+from litellm.proxy.management_endpoints.password_endpoints import (
+ router as password_management_router,
+)
from litellm.proxy.management_endpoints.router_settings_endpoints import (
router as router_settings_router,
)
@@ -16614,6 +16617,7 @@ async def onboarding(invite_link: str, request: Request):
auth_header_name=general_settings.get("litellm_key_header_name", "Authorization"),
disabled_non_admin_personal_key_creation=disabled_non_admin_personal_key_creation,
server_root_path=get_server_root_path(),
+ password_reset_required=False,
)
jwt_token: Final = jwt.encode(
cast(dict, returned_ui_token_object),
@@ -16724,6 +16728,7 @@ async def _generate_onboarding_ui_session_token(user_obj: _UserTableRow) -> str:
auth_header_name=general_settings.get("litellm_key_header_name", "Authorization"),
disabled_non_admin_personal_key_creation=disabled_non_admin_personal_key_creation,
server_root_path=get_server_root_path(),
+ password_reset_required=False,
)
assert master_key is not None
return jwt.encode(
@@ -16794,6 +16799,7 @@ async def claim_onboarding_link(data: InvitationClaim, request: Request):
)
validate_password_policy(data.password, general_settings)
+ await validate_password_not_breached(data.password, general_settings)
hashed_pw: Final = hash_password(data.password)
current_time = litellm.utils.get_utc_datetime()
async with prisma_client.db.tx() as tx:
@@ -16813,7 +16819,12 @@ async def claim_onboarding_link(data: InvitationClaim, request: Request):
### UPDATE USER OBJECT ###
user_obj: Final[_UserTableRow | None] = await tx.litellm_usertable.update(
- where={"user_id": invite_obj.user_id}, data={"password": hashed_pw}
+ where={"user_id": invite_obj.user_id},
+ data={
+ "password": hashed_pw,
+ "password_reset_required": False,
+ "last_breach_check_at": None,
+ },
)
if user_obj is None:
@@ -19251,6 +19262,7 @@ app.include_router(pass_through_router)
app.include_router(health_router)
app.include_router(key_management_router)
app.include_router(internal_user_router)
+app.include_router(password_management_router)
app.include_router(team_router)
app.include_router(ui_sso_router)
app.include_router(organization_router)
diff --git a/litellm/proxy/schema.prisma b/litellm/proxy/schema.prisma
index d2032cec0d0..bcafa6dbd0e 100644
--- a/litellm/proxy/schema.prisma
+++ b/litellm/proxy/schema.prisma
@@ -246,6 +246,8 @@ model LiteLLM_UserTable {
organization_id String?
object_permission_id String?
password String?
+ password_reset_required Boolean?
+ last_breach_check_at DateTime?
teams String[] @default([])
user_role String?
max_budget Float?
diff --git a/litellm/types/llms/custom_http.py b/litellm/types/llms/custom_http.py
index d80d7410aae..793893451df 100644
--- a/litellm/types/llms/custom_http.py
+++ b/litellm/types/llms/custom_http.py
@@ -31,6 +31,7 @@ class httpxSpecialProvider(str, Enum):
UI = "ui"
Sandbox = "sandbox"
ModelCostMap = "model_cost_map"
+ PasswordBreachCheck = "password_breach_check"
VerifyTypes = str | bool | ssl.SSLContext
diff --git a/litellm/types/proxy/ui_sso.py b/litellm/types/proxy/ui_sso.py
index 0d7e0b99cf0..03b0b92a4d1 100644
--- a/litellm/types/proxy/ui_sso.py
+++ b/litellm/types/proxy/ui_sso.py
@@ -1,6 +1,6 @@
from typing import Literal
-from typing_extensions import TypedDict
+from typing_extensions import ReadOnly, TypedDict
class ReturnedUITokenObject(TypedDict):
@@ -17,6 +17,7 @@ class ReturnedUITokenObject(TypedDict):
auth_header_name: str
disabled_non_admin_personal_key_creation: bool
server_root_path: str # e.g. `/litellm`
+ password_reset_required: ReadOnly[bool]
class ParsedOpenIDResult(TypedDict, total=False):
diff --git a/schema.prisma b/schema.prisma
index d2032cec0d0..bcafa6dbd0e 100644
--- a/schema.prisma
+++ b/schema.prisma
@@ -246,6 +246,8 @@ model LiteLLM_UserTable {
organization_id String?
object_permission_id String?
password String?
+ password_reset_required Boolean?
+ last_breach_check_at DateTime?
teams String[] @default([])
user_role String?
max_budget Float?
diff --git a/terraform/provider/tools/endpointaudit/coverage_allowlist.txt b/terraform/provider/tools/endpointaudit/coverage_allowlist.txt
index 4ea64b152f1..f8277c83a64 100644
--- a/terraform/provider/tools/endpointaudit/coverage_allowlist.txt
+++ b/terraform/provider/tools/endpointaudit/coverage_allowlist.txt
@@ -86,6 +86,7 @@ POST /team/key/bulk_update
POST /team/permissions_bulk_update
POST /team/{team_id}/disable_logging
POST /user/bulk_update
+POST /user/password/change
# Alternate method or path for functionality the provider already manages elsewhere
GET /credentials/by_model/{model_id}
diff --git a/tests/test_litellm/proxy/auth/test_login_utils.py b/tests/test_litellm/proxy/auth/test_login_utils.py
index 55ece36252d..d41b90fe566 100644
--- a/tests/test_litellm/proxy/auth/test_login_utils.py
+++ b/tests/test_litellm/proxy/auth/test_login_utils.py
@@ -5,12 +5,15 @@ This module tests the refactored login logic that was moved from proxy_server.py
to login_utils.py for better reusability.
"""
+import hashlib
import os
from collections.abc import Mapping
from contextlib import ExitStack
from typing import TYPE_CHECKING, Final
+from datetime import datetime, timedelta, timezone
from unittest.mock import AsyncMock, MagicMock, patch
+import httpx
import pytest
if TYPE_CHECKING:
@@ -34,6 +37,7 @@ def _unlimited_throttle():
from litellm.constants import LITELLM_PROXY_ADMIN_NAME
+from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler
from litellm.proxy._types import (
LiteLLM_UserTable,
LitellmUserRoles,
@@ -46,8 +50,13 @@ from litellm.proxy.auth.login_utils import (
authenticate_user,
get_ui_credentials,
is_env_credential_login_enabled,
+ screen_login_password_for_breach,
)
+# Successful DB-user logins schedule the background HIBP screen; disable it so
+# no test ever does live network I/O to haveibeenpwned.com from CI.
+_POLICY_NO_BREACH_CHECK = {"password_policy_check_breached_passwords": False}
+
def test_get_ui_credentials_prefers_explicit_password():
"""The configured UI password should be returned when available."""
@@ -326,6 +335,7 @@ async def test_authenticate_user_email_case_insensitive_login():
master_key=master_key,
prisma_client=mock_prisma_client,
throttle=_unlimited_throttle(),
+ general_settings=_POLICY_NO_BREACH_CHECK,
)
result_lower = await authenticate_user(
username=stored_email,
@@ -333,6 +343,7 @@ async def test_authenticate_user_email_case_insensitive_login():
master_key=master_key,
prisma_client=mock_prisma_client,
throttle=_unlimited_throttle(),
+ general_settings=_POLICY_NO_BREACH_CHECK,
)
assert result_mixed.user_id == result_lower.user_id == "test-user-123"
@@ -576,6 +587,7 @@ async def test_authenticate_user_database_login_with_non_ascii_password():
master_key=master_key,
prisma_client=mock_prisma_client,
throttle=_unlimited_throttle(),
+ general_settings=_POLICY_NO_BREACH_CHECK,
)
assert isinstance(result, LoginResult)
@@ -2064,3 +2076,263 @@ class TestIsEnvCredentialLoginEnabled:
with ExitStack() as stack:
_patch_sso_configured(stack, configured=False)
assert is_env_credential_login_enabled({"disable_password_login_when_sso_enabled": True}) is True
+
+
+def _db_user_row(*, password: str, password_reset_required: bool | None = None, last_breach_check_at=None):
+ hashed = hash_token(token=password)
+ row = MagicMock()
+ row.user_id = "reset-user-1"
+ row.user_email = "reset@example.com"
+ row.password = hashed
+ row.user_role = LitellmUserRoles.INTERNAL_USER
+ row.password_reset_required = password_reset_required
+ row.last_breach_check_at = last_breach_check_at
+ return row
+
+
+def _prisma_with_user(row) -> MagicMock:
+ mock_prisma_client = MagicMock()
+ mock_prisma_client.db.litellm_usertable.find_first = AsyncMock(return_value=row)
+ mock_prisma_client.db.litellm_usertable.update = AsyncMock(return_value=row)
+ return mock_prisma_client
+
+
+_DB_LOGIN_ENV = {
+ "DATABASE_URL": "postgresql://test:test@localhost/test",
+ "UI_USERNAME": "admin",
+ "UI_PASSWORD": "admin-password",
+}
+
+
+class TestPasswordResetRequiredSessionMinting:
+ """A user flagged `password_reset_required` must receive a UI session key
+ restricted to the change-password endpoint (server-side enforcement, so a
+ script driving the management API with the session key is blocked too);
+ an unflagged user must keep getting an unrestricted key."""
+
+ async def _login(self, mock_prisma_client) -> tuple[LoginResult, dict]:
+ with patch.dict(os.environ, _DB_LOGIN_ENV):
+ with patch( # test-quality-ok: asserting the minted key's restriction requires seeing its kwargs
+ "litellm.proxy.auth.login_utils.generate_key_helper_fn",
+ new_callable=AsyncMock,
+ return_value={"token": "session-token"},
+ ) as mock_generate_key:
+ result = await authenticate_user(
+ username="reset@example.com",
+ password="Str0ng!Passw0rd",
+ master_key="sk-1234",
+ prisma_client=mock_prisma_client,
+ general_settings=_POLICY_NO_BREACH_CHECK,
+ )
+ return result, mock_generate_key.call_args.kwargs
+
+ @pytest.mark.asyncio
+ async def test_flagged_user_gets_key_restricted_to_change_password(self):
+ row = _db_user_row(password="Str0ng!Passw0rd", password_reset_required=True)
+ result, key_kwargs = await self._login(_prisma_with_user(row))
+
+ assert key_kwargs["allowed_routes"] == ["/user/password/change"]
+ assert key_kwargs["metadata"] == {"password_reset_required": True}
+ assert result.password_reset_required is True
+
+ @pytest.mark.asyncio
+ async def test_unflagged_user_gets_unrestricted_key(self):
+ row = _db_user_row(password="Str0ng!Passw0rd", password_reset_required=None)
+ result, key_kwargs = await self._login(_prisma_with_user(row))
+
+ assert key_kwargs["allowed_routes"] is None
+ assert not key_kwargs["metadata"]
+ assert result.password_reset_required is False
+
+ async def _login_with_screen_result(self, mock_prisma_client, breached: bool) -> tuple[LoginResult, dict, dict]:
+ with patch.dict(os.environ, _DB_LOGIN_ENV):
+ with patch( # test-quality-ok: asserting the minted key's restriction requires seeing its kwargs
+ "litellm.proxy.auth.login_utils.generate_key_helper_fn",
+ new_callable=AsyncMock,
+ return_value={"token": "session-token"},
+ ) as mock_generate_key:
+ with (
+ patch( # test-quality-ok: authenticate_user has no HIBP client seam; the screen itself is tested against MockTransport below
+ "litellm.proxy.auth.login_utils.screen_login_password_for_breach",
+ new_callable=AsyncMock,
+ return_value=breached,
+ ) as mock_screen
+ ):
+ result = await authenticate_user(
+ username="reset@example.com",
+ password="Str0ng!Passw0rd",
+ master_key="sk-1234",
+ prisma_client=mock_prisma_client,
+ general_settings=_POLICY_NO_BREACH_CHECK,
+ )
+ return result, mock_generate_key.call_args.kwargs, mock_screen.call_args.kwargs
+
+ @pytest.mark.asyncio
+ async def test_login_screens_with_row_state_before_minting(self):
+ """The login must hand the screen the row's recheck timestamp, or the
+ 24h throttle can never work."""
+ checked_at = datetime.now(timezone.utc) - timedelta(hours=1)
+ row = _db_user_row(password="Str0ng!Passw0rd", last_breach_check_at=checked_at)
+ mock_prisma_client = _prisma_with_user(row)
+
+ _, _, screen_kwargs = await self._login_with_screen_result(mock_prisma_client, breached=False)
+
+ assert screen_kwargs["user_id"] == "reset-user-1"
+ assert screen_kwargs["password"] == "Str0ng!Passw0rd"
+ assert screen_kwargs["last_breach_check_at"] == checked_at
+ assert screen_kwargs["prisma_client"] is mock_prisma_client
+
+ @pytest.mark.asyncio
+ async def test_fresh_breach_hit_restricts_the_current_session(self):
+ """A breach found during THIS login must restrict THIS session, not
+ just the next one."""
+ row = _db_user_row(password="Str0ng!Passw0rd", password_reset_required=None)
+ mock_prisma_client = _prisma_with_user(row)
+
+ result, key_kwargs, _ = await self._login_with_screen_result(mock_prisma_client, breached=True)
+
+ assert key_kwargs["allowed_routes"] == ["/user/password/change"]
+ assert key_kwargs["metadata"] == {"password_reset_required": True}
+ assert result.password_reset_required is True
+
+
+def _sha1_upper(password: str) -> str:
+ return hashlib.sha1(password.encode("utf-8"), usedforsecurity=False).hexdigest().upper()
+
+
+def _client_with_transport(handler) -> AsyncHTTPHandler:
+ http_handler = AsyncHTTPHandler()
+ http_handler.client = httpx.AsyncClient(transport=httpx.MockTransport(handler))
+ return http_handler
+
+
+def _client_returning_breach_hit(password: str) -> AsyncHTTPHandler:
+ body = f"{_sha1_upper(password)[5:]}:42"
+ return _client_with_transport(lambda request: httpx.Response(200, text=body))
+
+
+def _client_returning_no_hit() -> AsyncHTTPHandler:
+ return _client_with_transport(lambda request: httpx.Response(200, text="0000000000000000000000000000000000A:3"))
+
+
+def _client_never_called() -> AsyncHTTPHandler:
+ def handler(request: httpx.Request) -> httpx.Response:
+ raise AssertionError(f"unexpected HTTP call to {request.url}")
+
+ return _client_with_transport(handler)
+
+
+class TestScreenLoginPasswordForBreach:
+ """The awaited login-time screen: flags a breached password for a forced
+ reset, stamps the recheck timestamp, rechecks at most every 24h, returns
+ the breach verdict so the login can restrict the session it is minting,
+ and never raises into the login."""
+
+ @pytest.mark.asyncio
+ async def test_breached_password_sets_reset_flag_and_timestamp(self):
+ password = "Password123!"
+ mock_prisma_client = _prisma_with_user(None)
+
+ breached = await screen_login_password_for_breach(
+ user_id="reset-user-1",
+ password=password,
+ last_breach_check_at=None,
+ general_settings={},
+ prisma_client=mock_prisma_client,
+ client=_client_returning_breach_hit(password),
+ )
+
+ assert breached is True
+ update_kwargs = mock_prisma_client.db.litellm_usertable.update.call_args.kwargs
+ assert update_kwargs["where"] == {"user_id": "reset-user-1"}
+ assert update_kwargs["data"]["password_reset_required"] is True
+ assert isinstance(update_kwargs["data"]["last_breach_check_at"], datetime)
+
+ @pytest.mark.asyncio
+ async def test_clean_password_stamps_timestamp_without_flag(self):
+ mock_prisma_client = _prisma_with_user(None)
+
+ breached = await screen_login_password_for_breach(
+ user_id="reset-user-1",
+ password="Str0ng!Passw0rd",
+ last_breach_check_at=None,
+ general_settings={},
+ prisma_client=mock_prisma_client,
+ client=_client_returning_no_hit(),
+ )
+
+ assert breached is False
+ update_kwargs = mock_prisma_client.db.litellm_usertable.update.call_args.kwargs
+ assert "password_reset_required" not in update_kwargs["data"]
+ assert isinstance(update_kwargs["data"]["last_breach_check_at"], datetime)
+
+ @pytest.mark.asyncio
+ async def test_skips_hibp_when_checked_within_24_hours(self):
+ mock_prisma_client = _prisma_with_user(None)
+
+ breached = await screen_login_password_for_breach(
+ user_id="reset-user-1",
+ password="Password123!",
+ last_breach_check_at=datetime.now(timezone.utc) - timedelta(hours=23),
+ general_settings={},
+ prisma_client=mock_prisma_client,
+ client=_client_never_called(),
+ )
+
+ assert breached is False
+ mock_prisma_client.db.litellm_usertable.update.assert_not_called()
+
+ @pytest.mark.asyncio
+ async def test_rechecks_when_last_check_is_older_than_24_hours(self):
+ password = "Password123!"
+ mock_prisma_client = _prisma_with_user(None)
+
+ breached = await screen_login_password_for_breach(
+ user_id="reset-user-1",
+ password=password,
+ last_breach_check_at=datetime.now(timezone.utc) - timedelta(hours=25),
+ general_settings={},
+ prisma_client=mock_prisma_client,
+ client=_client_returning_breach_hit(password),
+ )
+
+ assert breached is True
+ assert (
+ mock_prisma_client.db.litellm_usertable.update.call_args.kwargs["data"]["password_reset_required"] is True
+ )
+
+ @pytest.mark.asyncio
+ async def test_skips_hibp_when_check_disabled(self):
+ mock_prisma_client = _prisma_with_user(None)
+
+ breached = await screen_login_password_for_breach(
+ user_id="reset-user-1",
+ password="Password123!",
+ last_breach_check_at=None,
+ general_settings=_POLICY_NO_BREACH_CHECK,
+ prisma_client=mock_prisma_client,
+ client=_client_never_called(),
+ )
+
+ assert breached is False
+ mock_prisma_client.db.litellm_usertable.update.assert_not_called()
+
+ @pytest.mark.asyncio
+ async def test_db_failure_never_raises_but_still_reports_the_breach(self):
+ """A failed flag write must not fail the login, but the breach verdict
+ still has to restrict the session being minted right now."""
+ password = "Password123!"
+ mock_prisma_client = _prisma_with_user(None)
+ mock_prisma_client.db.litellm_usertable.update = AsyncMock(side_effect=RuntimeError("db down"))
+
+ assert (
+ await screen_login_password_for_breach(
+ user_id="reset-user-1",
+ password=password,
+ last_breach_check_at=None,
+ general_settings={},
+ prisma_client=mock_prisma_client,
+ client=_client_returning_breach_hit(password),
+ )
+ is True
+ )
diff --git a/tests/test_litellm/proxy/auth/test_onboarding.py b/tests/test_litellm/proxy/auth/test_onboarding.py
index 524b655b465..0454aea1239 100644
--- a/tests/test_litellm/proxy/auth/test_onboarding.py
+++ b/tests/test_litellm/proxy/auth/test_onboarding.py
@@ -8,15 +8,20 @@ Covers the security behavior of:
session key only after the password is written
"""
+import hashlib
from datetime import timedelta
from unittest.mock import AsyncMock, MagicMock, patch
+import httpx
import jwt
import pytest
+import respx
from fastapi import HTTPException
import litellm
-from litellm.proxy._types import InvitationClaim
+from litellm.proxy._types import InvitationClaim, ProxyException
+
+_POLICY_NO_BREACH_CHECK = {"password_policy_check_breached_passwords": False}
# ---------------------------------------------------------------------------
# Helpers
@@ -386,7 +391,9 @@ async def test_claim_token_rejects_concurrent_reuse_before_password_write():
with (
patch("litellm.proxy.proxy_server.prisma_client", prisma),
patch("litellm.proxy.proxy_server.master_key", "sk-test"),
- patch("litellm.proxy.proxy_server.general_settings", {}),
+ patch( # test-quality-ok: claim_onboarding_link reads proxy_server module globals; no injection seam
+ "litellm.proxy.proxy_server.general_settings", _POLICY_NO_BREACH_CHECK
+ ),
patch(
"litellm.proxy.proxy_server.generate_key_helper_fn",
new_callable=AsyncMock,
@@ -426,7 +433,9 @@ async def test_claim_token_sets_accepted_at_after_password_written():
with (
patch("litellm.proxy.proxy_server.prisma_client", prisma),
patch("litellm.proxy.proxy_server.master_key", "sk-test"),
- patch("litellm.proxy.proxy_server.general_settings", {}),
+ patch( # test-quality-ok: claim_onboarding_link reads proxy_server module globals; no injection seam
+ "litellm.proxy.proxy_server.general_settings", _POLICY_NO_BREACH_CHECK
+ ),
patch("litellm.proxy.proxy_server.premium_user", False),
patch(
"litellm.proxy.proxy_server.generate_key_helper_fn",
@@ -454,6 +463,10 @@ async def test_claim_token_sets_accepted_at_after_password_written():
call_kwargs = prisma.db.litellm_usertable.update.call_args
assert call_kwargs.kwargs["where"] == {"user_id": "user-123"}
assert "password" in call_kwargs.kwargs["data"]
+ # A freshly claimed, policy-screened password lifts any pending forced
+ # reset and re-arms the login-time breach screen.
+ assert call_kwargs.kwargs["data"]["password_reset_required"] is False
+ assert call_kwargs.kwargs["data"]["last_breach_check_at"] is None
# is_accepted was flipped to True on the invitation link
prisma.db.litellm_invitationlink.update.assert_called_once()
@@ -483,7 +496,9 @@ async def test_claim_token_rolls_back_invite_when_session_key_mint_fails():
with (
patch("litellm.proxy.proxy_server.prisma_client", prisma),
patch("litellm.proxy.proxy_server.master_key", "sk-test"),
- patch("litellm.proxy.proxy_server.general_settings", {}),
+ patch( # test-quality-ok: claim_onboarding_link reads proxy_server module globals; no injection seam
+ "litellm.proxy.proxy_server.general_settings", _POLICY_NO_BREACH_CHECK
+ ),
patch(
"litellm.proxy.proxy_server.generate_key_helper_fn",
new_callable=AsyncMock,
@@ -505,3 +520,124 @@ async def test_claim_token_rolls_back_invite_when_session_key_mint_fails():
}
assert rollback_kwargs["data"]["accepted_at"] is None
assert rollback_kwargs["data"]["is_accepted"] is False
+
+
+# ---------------------------------------------------------------------------
+# POST /onboarding/claim_token - password policy
+# ---------------------------------------------------------------------------
+
+
+def _hibp_url_for(password: str) -> str:
+ sha1 = hashlib.sha1(password.encode("utf-8"), usedforsecurity=False).hexdigest().upper()
+ return f"https://api.pwnedpasswords.com/range/{sha1[:5]}"
+
+
+def _hibp_suffix_for(password: str) -> str:
+ return hashlib.sha1(password.encode("utf-8"), usedforsecurity=False).hexdigest().upper()[5:]
+
+
+@pytest.mark.asyncio
+async def test_claim_token_rejects_short_password_before_consuming_invite():
+ """Default policy requires 12 characters; the invite must stay claimable."""
+ from litellm.proxy.proxy_server import claim_onboarding_link
+
+ invite = _make_invite(is_accepted=False)
+ prisma = _make_prisma(invite, _make_user())
+ request = _make_claim_request(_make_onboarding_token())
+ data = InvitationClaim(
+ invitation_link="invite-abc",
+ user_id="user-123",
+ password="Sh0rt!pw",
+ )
+
+ with (
+ patch("litellm.proxy.proxy_server.prisma_client", prisma), # test-quality-ok: claim_onboarding_link reads proxy_server module globals; no injection seam
+ patch("litellm.proxy.proxy_server.master_key", "sk-test"), # test-quality-ok: same as above
+ patch("litellm.proxy.proxy_server.general_settings", {}), # test-quality-ok: same as above
+ ):
+ with pytest.raises(ProxyException) as exc_info:
+ await claim_onboarding_link(data=data, request=request)
+
+ assert exc_info.value.code == "400"
+ assert "at least 12 characters" in exc_info.value.message
+ prisma.db.litellm_invitationlink.update_many.assert_not_called()
+ prisma.db.litellm_usertable.update.assert_not_called()
+
+
+@pytest.mark.asyncio
+@respx.mock
+async def test_claim_token_rejects_breached_password_before_consuming_invite():
+ """A password found in the HIBP corpus must be rejected and never stored."""
+ from litellm.proxy.proxy_server import claim_onboarding_link
+
+ password = "P@ssword123456"
+ respx.get(_hibp_url_for(password)).mock(
+ return_value=httpx.Response(200, text=f"{_hibp_suffix_for(password)}:1387")
+ )
+
+ invite = _make_invite(is_accepted=False)
+ prisma = _make_prisma(invite, _make_user())
+ request = _make_claim_request(_make_onboarding_token())
+ data = InvitationClaim(
+ invitation_link="invite-abc",
+ user_id="user-123",
+ password=password,
+ )
+
+ with (
+ patch("litellm.proxy.proxy_server.prisma_client", prisma), # test-quality-ok: claim_onboarding_link reads proxy_server module globals; no injection seam
+ patch("litellm.proxy.proxy_server.master_key", "sk-test"), # test-quality-ok: same as above
+ patch("litellm.proxy.proxy_server.general_settings", {}), # test-quality-ok: same as above
+ ):
+ with pytest.raises(ProxyException) as exc_info:
+ await claim_onboarding_link(data=data, request=request)
+
+ assert exc_info.value.code == "400"
+ assert "data breaches" in exc_info.value.message
+ prisma.db.litellm_invitationlink.update_many.assert_not_called()
+ prisma.db.litellm_usertable.update.assert_not_called()
+
+
+@pytest.mark.asyncio
+@respx.mock
+async def test_claim_token_fails_open_when_hibp_unreachable():
+ """An HIBP outage must never block onboarding: the claim proceeds."""
+ from litellm.proxy.proxy_server import claim_onboarding_link
+
+ password = "NewP@ssw0rd-2026"
+ respx.get(_hibp_url_for(password)).mock(side_effect=httpx.ConnectError("no route to host"))
+
+ invite = _make_invite(is_accepted=False)
+ user = _make_user()
+ prisma = _make_prisma(invite, user)
+ request = _make_claim_request(_make_onboarding_token())
+ data = InvitationClaim(
+ invitation_link="invite-abc",
+ user_id="user-123",
+ password=password,
+ )
+
+ with (
+ patch("litellm.proxy.proxy_server.prisma_client", prisma), # test-quality-ok: claim_onboarding_link reads proxy_server module globals; no injection seam
+ patch("litellm.proxy.proxy_server.master_key", "sk-test"), # test-quality-ok: same as above
+ patch("litellm.proxy.proxy_server.general_settings", {}), # test-quality-ok: same as above
+ patch("litellm.proxy.proxy_server.premium_user", False), # test-quality-ok: same as above
+ patch( # test-quality-ok: same as above
+ "litellm.proxy.proxy_server.generate_key_helper_fn",
+ new_callable=AsyncMock,
+ return_value={"token": "sk-generated-key", "user_id": "user-123"},
+ ),
+ patch( # test-quality-ok: same as above
+ "litellm.proxy.proxy_server.get_custom_url",
+ return_value="http://localhost:4000/",
+ ),
+ patch( # test-quality-ok: same as above
+ "litellm.proxy.proxy_server.get_disabled_non_admin_personal_key_creation",
+ return_value=False,
+ ),
+ patch("litellm.proxy.proxy_server.get_server_root_path", return_value=""), # test-quality-ok: same as above
+ ):
+ result = await claim_onboarding_link(data=data, request=request)
+
+ assert "token" in result
+ prisma.db.litellm_usertable.update.assert_called_once()
diff --git a/tests/test_litellm/proxy/auth/test_password_policy.py b/tests/test_litellm/proxy/auth/test_password_policy.py
index f6e7d443907..f9f5025b57f 100644
--- a/tests/test_litellm/proxy/auth/test_password_policy.py
+++ b/tests/test_litellm/proxy/auth/test_password_policy.py
@@ -2,22 +2,56 @@
Tests for the configurable password-strength policy in
`litellm.proxy.auth.password_policy`, enforced on every path that persists a
new or changed password for a locally-managed user.
+
+The breach-check (HIBP) tests inject a real AsyncHTTPHandler wrapping an
+httpx.MockTransport, so no network is touched and nothing is monkeypatched.
"""
+import asyncio
+import hashlib
+
+import httpx
import pytest
+from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler
from litellm.proxy._types import ProxyErrorTypes, ProxyException
from litellm.proxy.auth.password_policy import (
DEFAULT_MIN_LENGTH,
MIN_ALLOWED_LENGTH,
PasswordPolicy,
get_password_policy,
+ validate_password_not_breached,
validate_password_policy,
+ validate_passwords_bulk,
)
STRONG_PASSWORD = "Str0ng!Passw0rd"
+def _sha1_upper(password: str) -> str:
+ return hashlib.sha1(password.encode("utf-8"), usedforsecurity=False).hexdigest().upper()
+
+
+def _client_with_transport(handler) -> AsyncHTTPHandler:
+ http_handler = AsyncHTTPHandler()
+ http_handler.client = httpx.AsyncClient(transport=httpx.MockTransport(handler))
+ return http_handler
+
+
+def _client_never_called() -> AsyncHTTPHandler:
+ def handler(request: httpx.Request) -> httpx.Response:
+ raise AssertionError(f"unexpected HTTP call to {request.url}")
+
+ return _client_with_transport(handler)
+
+
+def _client_returning(body: str, status_code: int = 200) -> AsyncHTTPHandler:
+ def handler(request: httpx.Request) -> httpx.Response:
+ return httpx.Response(status_code, text=body)
+
+ return _client_with_transport(handler)
+
+
def test_get_password_policy_defaults_to_pif_baseline():
policy = get_password_policy({})
assert policy == PasswordPolicy(
@@ -134,3 +168,178 @@ def test_validate_password_policy_rejects_unicode_letter_as_special_character():
def test_validate_password_policy_accepts_real_special_character_with_unicode_letters():
"""Same base password as the rejection test above, plus an actual symbol."""
assert validate_password_policy("Passwörd1234!", {}) is None
+
+
+@pytest.mark.asyncio
+async def test_breach_check_skipped_when_disabled():
+ result = await validate_password_not_breached(
+ password="password12345", # breached in reality, but the check is off
+ general_settings={"password_policy_check_breached_passwords": False},
+ client=_client_never_called(),
+ )
+ assert result is None
+
+
+@pytest.mark.asyncio
+async def test_rejects_breached_password():
+ password = "correct horse battery staple"
+ sha1 = _sha1_upper(password)
+ body = f"AAAA000000000000000000000000000000A:0\r\n{sha1[5:]}:42\r\nBBBB000000000000000000000000000000B:7"
+
+ with pytest.raises(ProxyException) as exc_info:
+ await validate_password_not_breached(password=password, general_settings={}, client=_client_returning(body))
+ assert exc_info.value.code == "400"
+ assert exc_info.value.type == ProxyErrorTypes.validation_error
+ assert exc_info.value.param == "password"
+ assert "data breaches" in exc_info.value.message
+
+
+@pytest.mark.asyncio
+async def test_only_sha1_prefix_leaves_the_proxy():
+ password = "a very secret password"
+ sha1 = _sha1_upper(password)
+ captured_requests: list[httpx.Request] = []
+
+ def handler(request: httpx.Request) -> httpx.Response:
+ captured_requests.append(request)
+ return httpx.Response(200, text="0000000000000000000000000000000000A:1")
+
+ result = await validate_password_not_breached(
+ password=password, general_settings={}, client=_client_with_transport(handler)
+ )
+ assert result is None
+
+ (request,) = captured_requests
+ assert request.url.path == f"/range/{sha1[:5]}"
+ assert sha1[5:] not in str(request.url)
+ assert request.headers["Add-Padding"] == "true"
+ assert "litellm" in request.headers["User-Agent"]
+
+
+@pytest.mark.asyncio
+async def test_ignores_padding_entries_with_zero_count():
+ """HIBP padding entries (requested via Add-Padding) carry count 0 and must
+ not be treated as breaches when they collide with the password's suffix."""
+ password = "a padded-away password"
+ sha1 = _sha1_upper(password)
+
+ result = await validate_password_not_breached(
+ password=password, general_settings={}, client=_client_returning(f"{sha1[5:]}:0")
+ )
+ assert result is None
+
+
+@pytest.mark.asyncio
+async def test_accepts_password_absent_from_breach_corpus():
+ result = await validate_password_not_breached(
+ password="a genuinely novel password",
+ general_settings={},
+ client=_client_returning("0018A45C4D1DEF81644B54AB7F969B88D65:1\r\n00D4F6E8FA6EECAD2A3AA415EEC418D38EC:2"),
+ )
+ assert result is None
+
+
+@pytest.mark.asyncio
+async def test_breach_check_fails_open_on_network_error():
+ def handler(request: httpx.Request) -> httpx.Response:
+ raise httpx.ConnectError("no route to host")
+
+ result = await validate_password_not_breached(
+ password="password12345", # breached, but HIBP is unreachable
+ general_settings={},
+ client=_client_with_transport(handler),
+ )
+ assert result is None
+
+
+@pytest.mark.asyncio
+async def test_breach_check_fails_open_on_http_error_status():
+ result = await validate_password_not_breached(
+ password="password12345",
+ general_settings={},
+ client=_client_returning("service unavailable", status_code=503),
+ )
+ assert result is None
+
+
+@pytest.mark.asyncio
+async def test_breach_check_fails_open_on_malformed_response_body():
+ result = await validate_password_not_breached(
+ password="password12345",
+ general_settings={},
+ client=_client_returning(f"{_sha1_upper('password12345')[5:]}:not-a-number"),
+ )
+ assert result is None
+
+
+@pytest.mark.asyncio
+async def test_validate_passwords_bulk_screens_concurrently():
+ """All HIBP lookups for a batch must be in flight at once: each handler
+ call stalls until every expected request has arrived, and a handler that
+ gives up waiting reports the password as breached. Serial awaiting (the
+ old per-user behavior) leaves each earlier request waiting forever for the
+ later ones, so every verdict comes back as a breach and the test fails."""
+ passwords = ("Uniqu3!Passw0rd-a", "Uniqu3!Passw0rd-b", "Uniqu3!Passw0rd-c")
+ suffix_by_prefix = {_sha1_upper(p)[:5]: _sha1_upper(p)[5:] for p in passwords}
+ all_arrived = asyncio.Event()
+ arrivals: list[str] = []
+
+ async def handler(request: httpx.Request) -> httpx.Response:
+ arrivals.append(request.url.path)
+ if len(arrivals) == len(passwords):
+ all_arrived.set()
+ try:
+ await asyncio.wait_for(all_arrived.wait(), timeout=5)
+ except TimeoutError:
+ return httpx.Response(200, text=f"{suffix_by_prefix[request.url.path.rsplit('/', 1)[-1]]}:1")
+ return httpx.Response(200, text="0000000000000000000000000000000000A:1")
+
+ verdicts = await validate_passwords_bulk(passwords, {}, client=_client_with_transport(handler))
+ assert set(arrivals) == {f"/range/{prefix}" for prefix in suffix_by_prefix}
+ assert all(verdicts[p] is None for p in passwords)
+
+
+@pytest.mark.asyncio
+async def test_validate_passwords_bulk_deduplicates_lookups():
+ """500 users sharing one password must cost exactly one HIBP lookup."""
+ password = "Sh@red-Passw0rd!"
+ request_count = 0
+
+ def handler(request: httpx.Request) -> httpx.Response:
+ nonlocal request_count
+ request_count += 1
+ return httpx.Response(200, text="0000000000000000000000000000000000A:1")
+
+ verdicts = await validate_passwords_bulk((password,) * 500, {}, client=_client_with_transport(handler))
+ assert request_count == 1
+ assert verdicts == {password: None}
+
+
+@pytest.mark.asyncio
+async def test_validate_passwords_bulk_mixed_verdicts():
+ """Weak passwords are rejected without an HIBP lookup; breached ones get
+ the breach error; acceptable ones map to None."""
+ breached = "Br3ached!Passw0rd"
+ clean = "Cl3an!!Passw0rd42"
+ weak = "short1!"
+ breached_sha1 = _sha1_upper(breached)
+ looked_up_prefixes: list[str] = []
+
+ def handler(request: httpx.Request) -> httpx.Response:
+ looked_up_prefixes.append(request.url.path.rsplit("/", 1)[-1])
+ if request.url.path == f"/range/{breached_sha1[:5]}":
+ return httpx.Response(200, text=f"{breached_sha1[5:]}:99")
+ return httpx.Response(200, text="0000000000000000000000000000000000A:1")
+
+ verdicts = await validate_passwords_bulk((breached, clean, weak), {}, client=_client_with_transport(handler))
+ assert _sha1_upper(weak)[:5] not in looked_up_prefixes
+ assert verdicts[clean] is None
+ assert "data breaches" in verdicts[breached].message
+ assert verdicts[breached].code == "400"
+ assert "12 characters" in verdicts[weak].message
+
+
+@pytest.mark.asyncio
+async def test_validate_passwords_bulk_empty_batch_makes_no_lookups():
+ verdicts = await validate_passwords_bulk((), {}, client=_client_never_called())
+ assert verdicts == {}
diff --git a/tests/test_litellm/proxy/auth/test_route_checks.py b/tests/test_litellm/proxy/auth/test_route_checks.py
index 603a8686692..8382c7842b0 100644
--- a/tests/test_litellm/proxy/auth/test_route_checks.py
+++ b/tests/test_litellm/proxy/auth/test_route_checks.py
@@ -3,7 +3,6 @@ from datetime import datetime
from typing import Final
from unittest.mock import MagicMock, patch
-
import pytest
from fastapi import HTTPException, Request
@@ -39,7 +38,7 @@ def test_non_admin_config_update_route_rejected():
request.query_params = {}
# Test that calling /config/update route raises HTTPException with 403 status
- with pytest.raises(Exception, match='Only proxy admin can be used to generate, delete, update') as exc_info:
+ with pytest.raises(Exception, match="Only proxy admin can be used to generate, delete, update") as exc_info:
RouteChecks.non_proxy_admin_allowed_routes_check(
user_obj=user_obj,
_user_role=LitellmUserRoles.INTERNAL_USER.value,
@@ -50,9 +49,8 @@ def test_non_admin_config_update_route_rejected():
)
# Verify the exception is raised with the correct message
- assert (
- "Only proxy admin can be used to generate, delete, update info for new keys/users/teams"
- in str(exc_info.value)
+ assert "Only proxy admin can be used to generate, delete, update info for new keys/users/teams" in str(
+ exc_info.value
)
assert "Route=/config/update" in str(exc_info.value)
assert "Your role=internal_user" in str(exc_info.value)
@@ -131,7 +129,7 @@ def test_user_banner_update_rejected_for_non_admin():
request = MagicMock(spec=Request)
request.query_params = {}
- with pytest.raises(Exception, match='Only proxy admin can be used to generate, delete, update') as exc_info:
+ with pytest.raises(Exception, match="Only proxy admin can be used to generate, delete, update") as exc_info:
RouteChecks.non_proxy_admin_allowed_routes_check(
user_obj=user_obj,
_user_role=LitellmUserRoles.INTERNAL_USER.value,
@@ -706,9 +704,7 @@ def test_virtual_key_llm_api_route_includes_passthrough_prefix(route):
valid_token = UserAPIKeyAuth(user_id="test_user", allowed_routes=["llm_api_routes"])
- result = RouteChecks.is_virtual_key_allowed_to_call_route(
- route=route, valid_token=valid_token
- )
+ result = RouteChecks.is_virtual_key_allowed_to_call_route(route=route, valid_token=valid_token)
assert result is True
@@ -733,9 +729,7 @@ def test_virtual_key_llm_api_routes_allows_google_routes(route):
valid_token = UserAPIKeyAuth(user_id="test_user", allowed_routes=["llm_api_routes"])
- result = RouteChecks.is_virtual_key_allowed_to_call_route(
- route=route, valid_token=valid_token
- )
+ result = RouteChecks.is_virtual_key_allowed_to_call_route(route=route, valid_token=valid_token)
assert result is True
@@ -805,18 +799,14 @@ def test_google_routes_with_dynamic_model_names_accessible_to_internal_users():
)
# If no exception is raised, the test passes
except Exception as e:
- pytest.fail(
- f"Internal user should be able to access Google generateContent route. Got error: {str(e)}"
- )
+ pytest.fail(f"Internal user should be able to access Google generateContent route. Got error: {e!s}")
def test_virtual_key_allowed_routes_with_multiple_litellm_routes_member_names():
"""Test that virtual key works with multiple LiteLLMRoutes member names in allowed_routes"""
# Create a UserAPIKeyAuth with multiple LiteLLMRoutes member names
- valid_token = UserAPIKeyAuth(
- user_id="test_user", allowed_routes=["openai_routes", "info_routes"]
- )
+ valid_token = UserAPIKeyAuth(user_id="test_user", allowed_routes=["openai_routes", "info_routes"])
# Test that routes from both groups are allowed
result1 = RouteChecks.is_virtual_key_allowed_to_call_route(
@@ -870,13 +860,9 @@ def test_virtual_key_allowed_routes_with_no_member_names_only_explicit():
)
# Test that explicit routes are allowed
- result1 = RouteChecks.is_virtual_key_allowed_to_call_route(
- route="/chat/completions", valid_token=valid_token
- )
+ result1 = RouteChecks.is_virtual_key_allowed_to_call_route(route="/chat/completions", valid_token=valid_token)
- result2 = RouteChecks.is_virtual_key_allowed_to_call_route(
- route="/custom/route", valid_token=valid_token
- )
+ result2 = RouteChecks.is_virtual_key_allowed_to_call_route(route="/custom/route", valid_token=valid_token)
assert result1 is True
assert result2 is True
@@ -1274,9 +1260,7 @@ def test_virtual_key_without_llm_api_routes_cannot_access_pass_through():
)
assert exc_info.value.status_code == 403
- assert "Virtual key is not allowed to call this route" in str(
- exc_info.value.detail
- )
+ assert "Virtual key is not allowed to call this route" in str(exc_info.value.detail)
def test_check_passthrough_route_access_key_metadata_exact_match():
@@ -1735,9 +1719,7 @@ def test_videos_route_accessible_to_internal_users():
)
# If no exception is raised, the test passes
except Exception as e:
- pytest.fail(
- f"Internal user should be able to access /v1/videos route. Got error: {str(e)}"
- )
+ pytest.fail(f"Internal user should be able to access /v1/videos route. Got error: {e!s}")
def test_videos_route_with_virtual_key_llm_api_routes():
@@ -1759,12 +1741,8 @@ def test_videos_route_with_virtual_key_llm_api_routes():
]
for route in test_routes:
- result = RouteChecks.is_virtual_key_allowed_to_call_route(
- route=route, valid_token=valid_token
- )
- assert (
- result is True
- ), f"Virtual key with llm_api_routes should be able to access {route}"
+ result = RouteChecks.is_virtual_key_allowed_to_call_route(route=route, valid_token=valid_token)
+ assert result is True, f"Virtual key with llm_api_routes should be able to access {route}"
def test_non_proxy_admin_wildcard_allowed_routes():
@@ -1835,9 +1813,7 @@ def test_proxy_admin_viewer_can_access_global_spend_tags():
)
# If no exception is raised, the test passes
except Exception as e:
- pytest.fail(
- f"proxy_admin_viewer should be able to access /global/spend/tags route. Got error: {str(e)}"
- )
+ pytest.fail(f"proxy_admin_viewer should be able to access /global/spend/tags route. Got error: {e!s}")
# Routes returning proxy-wide spend across every team / customer / api_key.
@@ -1865,7 +1841,7 @@ def test_internal_user_blocked_from_global_spend_routes(route):
request = MagicMock(spec=Request)
request.query_params = {}
- with pytest.raises(Exception, match='Only proxy admin can be used to generate, delete, update') as exc_info:
+ with pytest.raises(Exception, match="Only proxy admin can be used to generate, delete, update") as exc_info:
RouteChecks.non_proxy_admin_allowed_routes_check(
user_obj=user_obj,
_user_role=LitellmUserRoles.INTERNAL_USER.value,
@@ -1894,7 +1870,7 @@ def test_internal_user_view_only_blocked_from_global_spend_routes(route):
request = MagicMock(spec=Request)
request.query_params = {}
- with pytest.raises(Exception, match='Only proxy admin can be used to generate, delete, update') as exc_info:
+ with pytest.raises(Exception, match="Only proxy admin can be used to generate, delete, update") as exc_info:
RouteChecks.non_proxy_admin_allowed_routes_check(
user_obj=user_obj,
_user_role=LitellmUserRoles.INTERNAL_USER_VIEW_ONLY.value,
@@ -1996,9 +1972,7 @@ def test_proxy_admin_viewer_can_access_audit_logs(route):
request_data={},
)
except Exception as e:
- pytest.fail(
- f"proxy_admin_viewer should be able to access {route} route. Got error: {str(e)}"
- )
+ pytest.fail(f"proxy_admin_viewer should be able to access {route} route. Got error: {e!s}")
# ── Admin Viewer parity: Logs page endpoints ──────────────────────────────────
@@ -2061,9 +2035,7 @@ def test_proxy_admin_viewer_can_access_logs_page_endpoints(route):
request_data={},
)
except Exception as e:
- pytest.fail(
- f"proxy_admin_viewer should be able to access {route}. Got error: {str(e)}"
- )
+ pytest.fail(f"proxy_admin_viewer should be able to access {route}. Got error: {e!s}")
@pytest.mark.parametrize(
@@ -2173,7 +2145,7 @@ def test_internal_user_blocked_from_admin_viewer_logs_routes(route):
if route not in INTERNAL_USER_BLOCKED_SUBSET:
return
- with pytest.raises(Exception, match='Only proxy admin can be used to generate, delete, update') as exc_info:
+ with pytest.raises(Exception, match="Only proxy admin can be used to generate, delete, update") as exc_info:
RouteChecks.non_proxy_admin_allowed_routes_check(
user_obj=user_obj,
_user_role=LitellmUserRoles.INTERNAL_USER.value,
@@ -2249,9 +2221,7 @@ def test_proxy_admin_viewer_can_access_settings_read_endpoints(route):
request_data={},
)
except Exception as e:
- pytest.fail(
- f"proxy_admin_viewer should be able to access {route}. Got error: {str(e)}"
- )
+ pytest.fail(f"proxy_admin_viewer should be able to access {route}. Got error: {e!s}")
# ── Admin Viewer parity: default-allow GET semantics ─────────────────────────
@@ -2450,9 +2420,7 @@ class TestModelsRouteExemptFromDisableLLMEndpoints:
)
local_file = os.path.abspath(local_file)
- spec = importlib.util.spec_from_file_location(
- "local_enterprise_route_checks", local_file
- )
+ spec = importlib.util.spec_from_file_location("local_enterprise_route_checks", local_file)
mod = importlib.util.module_from_spec(spec)
spec.loader.exec_module(mod)
return mod.EnterpriseRouteChecks
@@ -2463,9 +2431,7 @@ class TestModelsRouteExemptFromDisableLLMEndpoints:
EnterpriseRouteChecks = self._get_enterprise_route_checks()
with (
- patch.object(
- EnterpriseRouteChecks, "is_llm_api_route_disabled", return_value=True
- ),
+ patch.object(EnterpriseRouteChecks, "is_llm_api_route_disabled", return_value=True),
patch.object(
EnterpriseRouteChecks,
"is_management_routes_disabled",
@@ -2481,9 +2447,7 @@ class TestModelsRouteExemptFromDisableLLMEndpoints:
EnterpriseRouteChecks = self._get_enterprise_route_checks()
with (
- patch.object(
- EnterpriseRouteChecks, "is_llm_api_route_disabled", return_value=True
- ),
+ patch.object(EnterpriseRouteChecks, "is_llm_api_route_disabled", return_value=True),
patch.object(
EnterpriseRouteChecks,
"is_management_routes_disabled",
@@ -2499,9 +2463,7 @@ class TestModelsRouteExemptFromDisableLLMEndpoints:
EnterpriseRouteChecks = self._get_enterprise_route_checks()
with (
- patch.object(
- EnterpriseRouteChecks, "is_llm_api_route_disabled", return_value=True
- ),
+ patch.object(EnterpriseRouteChecks, "is_llm_api_route_disabled", return_value=True),
patch.object(
EnterpriseRouteChecks,
"is_management_routes_disabled",
@@ -2512,9 +2474,7 @@ class TestModelsRouteExemptFromDisableLLMEndpoints:
EnterpriseRouteChecks.should_call_route("/v1/chat/completions")
assert exc_info.value.status_code == 403
- assert "LLM API routes are disabled for this instance." in str(
- exc_info.value.detail
- )
+ assert "LLM API routes are disabled for this instance." in str(exc_info.value.detail)
@patch("litellm.proxy.proxy_server.premium_user", True)
def test_should_embeddings_still_blocked_when_llm_api_disabled(self):
@@ -2522,9 +2482,7 @@ class TestModelsRouteExemptFromDisableLLMEndpoints:
EnterpriseRouteChecks = self._get_enterprise_route_checks()
with (
- patch.object(
- EnterpriseRouteChecks, "is_llm_api_route_disabled", return_value=True
- ),
+ patch.object(EnterpriseRouteChecks, "is_llm_api_route_disabled", return_value=True),
patch.object(
EnterpriseRouteChecks,
"is_management_routes_disabled",
@@ -2542,9 +2500,7 @@ class TestModelsRouteExemptFromDisableLLMEndpoints:
EnterpriseRouteChecks = self._get_enterprise_route_checks()
with (
- patch.object(
- EnterpriseRouteChecks, "is_llm_api_route_disabled", return_value=False
- ),
+ patch.object(EnterpriseRouteChecks, "is_llm_api_route_disabled", return_value=False),
patch.object(
EnterpriseRouteChecks,
"is_management_routes_disabled",
@@ -2563,9 +2519,7 @@ def test_route_in_additional_public_routes_wildcard_match():
from litellm.proxy.auth.auth_utils import route_in_additonal_public_routes
with (
- patch(
- "litellm.proxy.proxy_server.general_settings", {"public_routes": ["/api/*"]}
- ),
+ patch("litellm.proxy.proxy_server.general_settings", {"public_routes": ["/api/*"]}),
patch("litellm.proxy.proxy_server.premium_user", True),
):
# Wildcard should match subpaths
@@ -2657,7 +2611,7 @@ def test_non_admin_non_team_admin_cannot_access_config_update_but_can_attempt_re
)
# /config/update is still blocked
- with pytest.raises(Exception, match='Only proxy admin can be used to generate, delete, update') as exc_info:
+ with pytest.raises(Exception, match="Only proxy admin can be used to generate, delete, update") as exc_info:
RouteChecks.non_proxy_admin_allowed_routes_check(
user_obj=user_obj,
_user_role=LitellmUserRoles.INTERNAL_USER.value,
@@ -2745,8 +2699,6 @@ def test_available_roles_accessible_to_non_admin_users(user_role):
# ── _user_is_org_admin tests ──────────────────────────────────────────────────
-
-
def _make_org_admin_user(org_id: str) -> LiteLLM_UserTable:
membership = LiteLLM_OrganizationMembershipTable(
user_id="org-admin-user",
@@ -2869,9 +2821,7 @@ async def test_add_team_org_context_noop_when_org_id_already_present():
raise AssertionError("must not resolve when organization_id is present")
body = {"team_id": "team-1", "organization_id": "org-explicit"}
- out = await add_team_org_context_to_request_body(
- route="/team/update", request_body=body, fetch_team_org_id=fetch
- )
+ out = await add_team_org_context_to_request_body(route="/team/update", request_body=body, fetch_team_org_id=fetch)
assert out == body
@@ -2883,9 +2833,7 @@ async def test_add_team_org_context_noop_for_other_routes():
raise AssertionError("must not resolve for a non-opted-in route")
body = {"team_id": "team-1"}
- out = await add_team_org_context_to_request_body(
- route="/team/delete", request_body=body, fetch_team_org_id=fetch
- )
+ out = await add_team_org_context_to_request_body(route="/team/delete", request_body=body, fetch_team_org_id=fetch)
assert out == body
@@ -2898,9 +2846,7 @@ async def test_add_team_org_context_noop_when_team_has_no_org():
return None
body = {"team_id": "team-1"}
- out = await add_team_org_context_to_request_body(
- route="/team/update", request_body=body, fetch_team_org_id=fetch
- )
+ out = await add_team_org_context_to_request_body(route="/team/update", request_body=body, fetch_team_org_id=fetch)
assert out == body
@@ -3171,9 +3117,7 @@ async def test_initialize_pass_through_registers_wildcard_for_auth_subpath():
# Removing the endpoint should clean up openai_routes
# remove_endpoint_routes takes endpoint_id (UUID portion of
# the route key "{id}:exact:{path}:{methods}")
- registered = (
- InitPassThroughEndpointHelpers.get_all_registered_pass_through_routes()
- )
+ registered = InitPassThroughEndpointHelpers.get_all_registered_pass_through_routes()
endpoint_ids = {k.split(":")[0] for k in registered}
for eid in endpoint_ids:
InitPassThroughEndpointHelpers.remove_endpoint_routes(eid)
@@ -3183,9 +3127,7 @@ async def test_initialize_pass_through_registers_wildcard_for_auth_subpath():
LiteLLMRoutes.openai_routes.value[:] = original_routes
# Clean up any routes registered during this test to avoid
# polluting the module-level _registered_pass_through_routes
- registered = (
- InitPassThroughEndpointHelpers.get_all_registered_pass_through_routes()
- )
+ registered = InitPassThroughEndpointHelpers.get_all_registered_pass_through_routes()
for k in registered:
InitPassThroughEndpointHelpers.remove_endpoint_routes(k.split(":")[0])
@@ -3216,8 +3158,7 @@ def test_provider_name_substring_not_classified_as_llm_route(route):
from litellm.proxy.auth.route_checks import RouteChecks
assert RouteChecks.is_llm_api_route(route=route) is False, (
- f"{route!r} should NOT be classified as an LLM API route — "
- "provider-name substring match bypass"
+ f"{route!r} should NOT be classified as an LLM API route — provider-name substring match bypass"
)
@@ -3239,9 +3180,7 @@ def test_legitimate_passthrough_routes_still_classified_as_llm_route(route):
"""Legitimate passthrough routes must still pass is_llm_api_route."""
from litellm.proxy.auth.route_checks import RouteChecks
- assert (
- RouteChecks.is_llm_api_route(route=route) is True
- ), f"{route!r} should be classified as an LLM API route"
+ assert RouteChecks.is_llm_api_route(route=route) is True, f"{route!r} should be classified as an LLM API route"
@pytest.mark.parametrize(
@@ -3299,7 +3238,7 @@ def test_internal_user_blocked_from_search_tool_writes(route):
request = MagicMock(spec=Request)
request.query_params = {}
- with pytest.raises(Exception, match='Only proxy admin can be used to generate, delete, update') as exc_info:
+ with pytest.raises(Exception, match="Only proxy admin can be used to generate, delete, update") as exc_info:
RouteChecks.non_proxy_admin_allowed_routes_check(
user_obj=user_obj,
_user_role=LitellmUserRoles.INTERNAL_USER.value,
@@ -3675,12 +3614,7 @@ def test_agent_inference_routes_stay_llm_api(route):
def test_agent_routes_union_still_covers_both_halves(route):
"""Keys configured with allowed_routes=["agent_routes"] must keep both halves."""
- assert (
- RouteChecks.check_route_access(
- route=route, allowed_routes=LiteLLMRoutes.agent_routes.value
- )
- is True
- )
+ assert RouteChecks.check_route_access(route=route, allowed_routes=LiteLLMRoutes.agent_routes.value) is True
@pytest.mark.parametrize("route", AGENT_MANAGEMENT_ROUTES)
@@ -3734,6 +3668,136 @@ def test_agent_registry_route_gate_open_to_non_admin_roles(user_role, method, ro
valid_token=valid_token,
request_data={},
)
+
+
+def test_proxy_admin_viewer_user_update_password_param_rejected():
+ """The self-service /user/update password carve-out is closed: non-admins
+ change their own password through /user/password/change, which verifies
+ the current password. Admin password sets don't pass through this check."""
+ with pytest.raises(HTTPException) as exc_info:
+ RouteChecks._check_proxy_admin_viewer_access(
+ route="/user/update",
+ _user_role=LitellmUserRoles.PROXY_ADMIN_VIEW_ONLY.value,
+ request_data={"password": "hunter2hunter2"},
+ )
+ assert exc_info.value.status_code == 403
+ assert "password" in str(exc_info.value.detail)
+
+
+def test_proxy_admin_viewer_user_update_user_email_still_allowed():
+ request = MagicMock(spec=Request)
+ request.method = "POST"
+
+ allowed = RouteChecks._check_proxy_admin_viewer_access(
+ route="/user/update",
+ _user_role=LitellmUserRoles.PROXY_ADMIN_VIEW_ONLY.value,
+ request_data={"user_email": "viewer@example.com"},
+ request=request,
+ )
+
+ assert allowed is None
+
+
+def test_proxy_admin_viewer_can_change_own_password():
+ request = MagicMock(spec=Request)
+ request.method = "POST"
+
+ allowed = RouteChecks._check_proxy_admin_viewer_access(
+ route="/user/password/change",
+ _user_role=LitellmUserRoles.PROXY_ADMIN_VIEW_ONLY.value,
+ request_data={"current_password": "a", "new_password": "b"},
+ request=request,
+ )
+
+ assert allowed is None
+
+
+@pytest.mark.parametrize(
+ "user_role",
+ [
+ LitellmUserRoles.INTERNAL_USER.value,
+ LitellmUserRoles.INTERNAL_USER_VIEW_ONLY.value,
+ ],
+)
+def test_non_admin_roles_can_change_own_password(user_role):
+ valid_token = UserAPIKeyAuth(user_id="test_user", user_role=user_role)
+ request = MagicMock(spec=Request)
+ request.method = "POST"
+ request.query_params = {}
+
+ allowed = RouteChecks.non_proxy_admin_allowed_routes_check(
+ user_obj=LiteLLM_UserTable(user_id="test_user", user_role=user_role),
+ _user_role=user_role,
+ route="/user/password/change",
+ request=request,
+ valid_token=valid_token,
+ request_data={"current_password": "a", "new_password": "b"},
+ )
+
+ assert allowed is None
+
+
+def _password_reset_session_token() -> UserAPIKeyAuth:
+ """The UI session key `authenticate_user` mints for a user flagged
+ `password_reset_required`."""
+ return UserAPIKeyAuth(
+ user_id="flagged_user",
+ allowed_routes=["/user/password/change"],
+ metadata={"password_reset_required": True},
+ )
+
+
+def test_password_reset_session_can_reach_change_password():
+ result = RouteChecks.is_virtual_key_allowed_to_call_route(
+ route="/user/password/change",
+ valid_token=_password_reset_session_token(),
+ )
+
+ assert result is True
+
+
+@pytest.mark.parametrize(
+ "route",
+ [
+ "/user/info",
+ "/key/generate",
+ "/user/update",
+ "/chat/completions",
+ ],
+)
+def test_password_reset_session_is_blocked_everywhere_else_with_reset_message(route):
+ """Server-side enforcement of the forced reset: a script that logs in via
+ /v2/login and drives the management API with the session key must get a 403
+ naming the remediation endpoint, on every route but the change-password one."""
+ with pytest.raises(HTTPException) as exc_info:
+ RouteChecks.is_virtual_key_allowed_to_call_route(
+ route=route,
+ valid_token=_password_reset_session_token(),
+ )
+
+ assert exc_info.value.status_code == 403
+ assert "password must be changed" in str(exc_info.value.detail)
+ assert "/user/password/change" in str(exc_info.value.detail)
+
+
+def test_restricted_key_without_reset_marker_keeps_generic_message():
+ """The reset-specific 403 must not leak onto ordinary allowed_routes keys."""
+ valid_token = UserAPIKeyAuth(
+ user_id="test_user",
+ allowed_routes=["/chat/completions"],
+ )
+
+ with pytest.raises(HTTPException) as exc_info:
+ RouteChecks.is_virtual_key_allowed_to_call_route(
+ route="/user/info",
+ valid_token=valid_token,
+ )
+
+ assert exc_info.value.status_code == 403
+ assert "password must be changed" not in str(exc_info.value.detail)
+ assert "not allowed to call this route" in str(exc_info.value.detail)
+
+
TEAM_CALLBACK_ROUTES = (
"/team/06bda574-5ca9-43d3-beb8-3b23c2f17112/callback",
"/team/06bda574-5ca9-43d3-beb8-3b23c2f17112/callback/langfuse",
diff --git a/tests/test_litellm/proxy/management_endpoints/test_internal_user_endpoints.py b/tests/test_litellm/proxy/management_endpoints/test_internal_user_endpoints.py
index b8f1aa0330b..6e11dbb6bae 100644
--- a/tests/test_litellm/proxy/management_endpoints/test_internal_user_endpoints.py
+++ b/tests/test_litellm/proxy/management_endpoints/test_internal_user_endpoints.py
@@ -1,13 +1,18 @@
+import hashlib
import json
from datetime import datetime, timezone
from types import SimpleNamespace
from typing import Final
+import httpx
import pytest
+import respx
from fastapi import HTTPException
from fastapi.testclient import TestClient
from pytest_mock import MockerFixture
+from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler
+
from litellm.proxy._types import (
LiteLLM_UserTableFiltered,
LitellmUserRoles,
@@ -67,9 +72,7 @@ async def test_ui_view_users_with_null_email(mocker, caplog):
# Proxy admin: no org filter, no get_user_object call
response = await ui_view_users(
- user_api_key_dict=UserAPIKeyAuth(
- user_id="test_user", user_role=LitellmUserRoles.PROXY_ADMIN
- ),
+ user_api_key_dict=UserAPIKeyAuth(user_id="test_user", user_role=LitellmUserRoles.PROXY_ADMIN),
user_id="test_user",
user_email=None,
team_id=None,
@@ -77,9 +80,7 @@ async def test_ui_view_users_with_null_email(mocker, caplog):
page_size=50,
)
- assert response == [
- LiteLLM_UserTableFiltered(user_id="test-user-null-email", user_email=None)
- ]
+ assert response == [LiteLLM_UserTableFiltered(user_id="test-user-null-email", user_email=None)]
@pytest.mark.asyncio
@@ -103,9 +104,7 @@ async def test_ui_view_users_proxy_admin_no_org_filter(mocker):
mocker.patch("litellm.proxy.proxy_server.prisma_client", mock_prisma_client)
await ui_view_users(
- user_api_key_dict=UserAPIKeyAuth(
- user_id="admin", user_role=LitellmUserRoles.PROXY_ADMIN
- ),
+ user_api_key_dict=UserAPIKeyAuth(user_id="admin", user_role=LitellmUserRoles.PROXY_ADMIN),
user_id=None,
user_email="foo",
team_id=None,
@@ -128,9 +127,7 @@ async def test_ui_view_users_org_admin_filtered_by_org(mocker):
async def mock_find_many(*args, **kwargs):
where = kwargs.get("where") or {}
assert "organization_memberships" in where
- assert where["organization_memberships"] == {
- "some": {"organization_id": {"in": [org_id]}}
- }
+ assert where["organization_memberships"] == {"some": {"organization_id": {"in": [org_id]}}}
return []
mock_prisma_client.db.litellm_usertable.find_many = mock_find_many
@@ -268,9 +265,7 @@ async def test_ui_view_users_flag_on_team_admin_org_team(mocker):
async def mock_find_many(*args, **kwargs):
where = kwargs.get("where") or {}
assert "organization_memberships" in where
- assert where["organization_memberships"] == {
- "some": {"organization_id": {"in": [org_id]}}
- }
+ assert where["organization_memberships"] == {"some": {"organization_id": {"in": [org_id]}}}
return []
mock_prisma_client.db.litellm_usertable.find_many = mock_find_many
@@ -401,9 +396,7 @@ async def test_ui_view_users_flag_on_team_admin_org_member_no_team_id(mocker):
async def mock_find_many(*args, **kwargs):
where = kwargs.get("where") or {}
assert "organization_memberships" in where
- assert where["organization_memberships"] == {
- "some": {"organization_id": {"in": [org_id]}}
- }
+ assert where["organization_memberships"] == {"some": {"organization_id": {"in": [org_id]}}}
return []
mock_prisma_client.db.litellm_usertable.find_many = mock_find_many
@@ -462,9 +455,7 @@ async def test_ui_view_users_flag_on_team_admin_not_in_org_resolves_via_key_team
async def mock_find_many(*args, **kwargs):
where = kwargs.get("where") or {}
assert "organization_memberships" in where
- assert where["organization_memberships"] == {
- "some": {"organization_id": {"in": [org_id]}}
- }
+ assert where["organization_memberships"] == {"some": {"organization_id": {"in": [org_id]}}}
return []
mock_prisma_client.db.litellm_usertable.find_many = mock_find_many
@@ -507,9 +498,7 @@ async def test_ui_view_users_flag_on_team_admin_not_in_org_resolves_via_key_team
# No team_id query param, but team_id on the API key
response = await ui_view_users(
- user_api_key_dict=UserAPIKeyAuth(
- user_id="team-admin-no-org", user_role=None, team_id=tid
- ),
+ user_api_key_dict=UserAPIKeyAuth(user_id="team-admin-no-org", user_role=None, team_id=tid),
user_id=None,
user_email="u",
team_id=None,
@@ -538,13 +527,9 @@ def test_user_daily_activity_types():
# Assert all fields in SpendMetrics are reported in DailySpendMetadata as "total_"
for field in spend_metrics.__dict__:
if field.startswith("total_"):
- assert hasattr(
- daily_spend_metadata, field
- ), f"Field {field} is not reported in DailySpendMetadata"
+ assert hasattr(daily_spend_metadata, field), f"Field {field} is not reported in DailySpendMetadata"
else:
- assert not hasattr(
- daily_spend_metadata, field
- ), f"Field {field} is reported in DailySpendMetadata"
+ assert not hasattr(daily_spend_metadata, field), f"Field {field} is reported in DailySpendMetadata"
@pytest.mark.asyncio
@@ -591,9 +576,7 @@ async def test_get_users_includes_timestamps(mocker):
# Call get_users function directly with proxy admin auth
admin_key = UserAPIKeyAuth(user_id="admin", user_role=LitellmUserRoles.PROXY_ADMIN)
- response = await get_users(
- page=1, page_size=1, user_api_key_dict=admin_key, organization_ids=None
- )
+ response = await get_users(page=1, page_size=1, user_api_key_dict=admin_key, organization_ids=None)
print("user /list response: ", response)
@@ -654,14 +637,10 @@ async def test_get_users_redacts_scim_enterprise_metadata(mocker):
)
admin_key = UserAPIKeyAuth(user_id="admin", user_role=LitellmUserRoles.PROXY_ADMIN)
- response = await get_users(
- page=1, page_size=1, user_api_key_dict=admin_key, organization_ids=None
- )
+ response = await get_users(page=1, page_size=1, user_api_key_dict=admin_key, organization_ids=None)
listed = response["users"][0]
- assert listed.metadata == {
- "scim_metadata": {"givenName": "Jane", "familyName": "Doe"}
- }
+ assert listed.metadata == {"scim_metadata": {"givenName": "Jane", "familyName": "Doe"}}
assert "scim_enterprise" not in (listed.metadata or {})
@@ -853,9 +832,7 @@ async def test_new_user_license_over_limit(mocker):
mocker.patch("litellm.proxy.proxy_server._license_check", mock_license_check)
# Create test request data
- user_request = NewUserRequest(
- user_email="test@example.com", user_role="internal_user"
- )
+ user_request = NewUserRequest(user_email="test@example.com", user_role="internal_user")
# Mock user_api_key_dict
mock_user_api_key_dict = UserAPIKeyAuth(user_id="test_admin")
@@ -916,9 +893,7 @@ async def test_new_user_license_gate_counts_only_billable_users(mocker):
request = NewUserRequest(user_role="internal_user")
# 2 active + 3 deactivated -> billable 2, not over max_users 2: gate passes
- mocker.patch(
- "litellm.proxy.proxy_server.prisma_client", _prisma(total=5, deactivated=3)
- )
+ mocker.patch("litellm.proxy.proxy_server.prisma_client", _prisma(total=5, deactivated=3))
with pytest.raises(ProxyException) as passed:
await new_user(data=request, user_api_key_dict=admin)
assert key_gen.call_count == 1
@@ -926,9 +901,7 @@ async def test_new_user_license_gate_counts_only_billable_users(mocker):
# 3 active, 0 deactivated -> billable 3, over max_users 2: gate blocks
key_gen.reset_mock()
- mocker.patch(
- "litellm.proxy.proxy_server.prisma_client", _prisma(total=3, deactivated=0)
- )
+ mocker.patch("litellm.proxy.proxy_server.prisma_client", _prisma(total=3, deactivated=0))
with pytest.raises(ProxyException) as blocked:
await new_user(data=request, user_api_key_dict=admin)
assert blocked.value.code == 403 or blocked.value.code == "403"
@@ -978,14 +951,10 @@ async def test_new_user_non_admin_cannot_create_admin(mocker):
mocker.patch("litellm.proxy.proxy_server._license_check", mock_license_check)
# Test Case 1: INTERNAL_USER trying to create PROXY_ADMIN
- user_request = NewUserRequest(
- user_email="admin@example.com", user_role=LitellmUserRoles.PROXY_ADMIN
- )
+ user_request = NewUserRequest(user_email="admin@example.com", user_role=LitellmUserRoles.PROXY_ADMIN)
# Mock user_api_key_dict with non-admin role
- mock_user_api_key_dict = UserAPIKeyAuth(
- user_id="test_internal_user", user_role=LitellmUserRoles.INTERNAL_USER
- )
+ mock_user_api_key_dict = UserAPIKeyAuth(user_id="test_internal_user", user_role=LitellmUserRoles.INTERNAL_USER)
# Call new_user function and expect ProxyException
with pytest.raises(ProxyException) as exc_info:
@@ -993,9 +962,7 @@ async def test_new_user_non_admin_cannot_create_admin(mocker):
# Verify the exception details
assert exc_info.value.code == 403 or exc_info.value.code == "403"
- assert "Only proxy admins can create administrative users" in str(
- exc_info.value.message
- )
+ assert "Only proxy admins can create administrative users" in str(exc_info.value.message)
assert "proxy_admin" in str(exc_info.value.message)
assert "proxy_admin_viewer" in str(exc_info.value.message)
assert str(LitellmUserRoles.PROXY_ADMIN) in str(exc_info.value.message)
@@ -1008,15 +975,11 @@ async def test_new_user_non_admin_cannot_create_admin(mocker):
)
with pytest.raises(ProxyException) as exc_info2:
- await new_user(
- data=user_request_viewer, user_api_key_dict=mock_user_api_key_dict
- )
+ await new_user(data=user_request_viewer, user_api_key_dict=mock_user_api_key_dict)
# Verify the exception details
assert exc_info2.value.code == 403 or exc_info2.value.code == "403"
- assert "Only proxy admins can create administrative users" in str(
- exc_info2.value.message
- )
+ assert "Only proxy admins can create administrative users" in str(exc_info2.value.message)
assert str(LitellmUserRoles.PROXY_ADMIN_VIEW_ONLY) in str(exc_info2.value.message)
@@ -1055,9 +1018,7 @@ async def test_new_user_non_admin_permissions_non_empty_rejected(mocker):
user_role=LitellmUserRoles.INTERNAL_USER,
permissions={"get_spend_routes": True},
)
- caller = UserAPIKeyAuth(
- user_id="org-admin", user_role=LitellmUserRoles.ORG_ADMIN
- )
+ caller = UserAPIKeyAuth(user_id="org-admin", user_role=LitellmUserRoles.ORG_ADMIN)
with pytest.raises(ProxyException) as exc_info:
await new_user(data=data, user_api_key_dict=caller)
@@ -1101,9 +1062,7 @@ async def test_new_user_non_admin_permissions_explicit_empty_rejected(mocker):
permissions={},
)
assert "permissions" in data.model_fields_set
- caller = UserAPIKeyAuth(
- user_id="org-admin", user_role=LitellmUserRoles.ORG_ADMIN
- )
+ caller = UserAPIKeyAuth(user_id="org-admin", user_role=LitellmUserRoles.ORG_ADMIN)
with pytest.raises(ProxyException) as exc_info:
await new_user(data=data, user_api_key_dict=caller)
@@ -1156,9 +1115,7 @@ async def test_new_user_non_admin_omits_permissions_succeeds(mocker):
user_role=LitellmUserRoles.INTERNAL_USER,
)
assert "permissions" not in data.model_fields_set
- caller = UserAPIKeyAuth(
- user_id="org-admin", user_role=LitellmUserRoles.ORG_ADMIN
- )
+ caller = UserAPIKeyAuth(user_id="org-admin", user_role=LitellmUserRoles.ORG_ADMIN)
result = await new_user(data=data, user_api_key_dict=caller)
assert result is not None
@@ -1232,14 +1189,10 @@ async def test_update_single_user_non_admin_permissions_rejected(mocker):
user_id="alice",
permissions={"get_spend_routes": True},
)
- caller = UserAPIKeyAuth(
- user_id="org-admin", user_role=LitellmUserRoles.ORG_ADMIN
- )
+ caller = UserAPIKeyAuth(user_id="org-admin", user_role=LitellmUserRoles.ORG_ADMIN)
with pytest.raises(HTTPException) as exc_info:
- await _update_single_user_helper(
- user_request=data, user_api_key_dict=caller
- )
+ await _update_single_user_helper(user_request=data, user_api_key_dict=caller)
assert exc_info.value.status_code == 403
assert "permissions" in str(exc_info.value.detail)
@@ -1261,14 +1214,10 @@ async def test_update_single_user_non_admin_permissions_explicit_empty_rejected(
data = UpdateUserRequest(user_id="alice", permissions={})
assert "permissions" in data.model_fields_set
- caller = UserAPIKeyAuth(
- user_id="org-admin", user_role=LitellmUserRoles.ORG_ADMIN
- )
+ caller = UserAPIKeyAuth(user_id="org-admin", user_role=LitellmUserRoles.ORG_ADMIN)
with pytest.raises(HTTPException) as exc_info:
- await _update_single_user_helper(
- user_request=data, user_api_key_dict=caller
- )
+ await _update_single_user_helper(user_request=data, user_api_key_dict=caller)
assert exc_info.value.status_code == 403
assert "permissions" in str(exc_info.value.detail)
@@ -1324,15 +1273,11 @@ async def test_user_info_url_encoding_plus_character(mocker):
mock_request.url.query = "user_id=machine-user+alp-air-admin-b58-b@tempus.com"
# Mock user_api_key_dict
- mock_user_api_key_dict = UserAPIKeyAuth(
- user_id="test_admin", user_role="proxy_admin"
- )
+ mock_user_api_key_dict = UserAPIKeyAuth(user_id="test_admin", user_role="proxy_admin")
# Call user_info function with the URL-decoded user_id (as FastAPI would pass it)
# FastAPI would normally convert + to space, but our fix should handle this
- decoded_user_id = (
- "machine-user alp-air-admin-b58-b@tempus.com" # What FastAPI gives us
- )
+ decoded_user_id = "machine-user alp-air-admin-b58-b@tempus.com" # What FastAPI gives us
expected_user_id = "machine-user+alp-air-admin-b58-b@tempus.com"
response = await user_info(
@@ -1383,9 +1328,7 @@ async def test_user_info_nonexistent_user(mocker):
mock_request = mocker.MagicMock(spec=Request)
# Mock user_api_key_dict
- mock_user_api_key_dict = UserAPIKeyAuth(
- user_id="test_admin", user_role="proxy_admin"
- )
+ mock_user_api_key_dict = UserAPIKeyAuth(user_id="test_admin", user_role="proxy_admin")
# Call user_info function with a non-existent user_id
nonexistent_user_id = "nonexistent-user@example.com"
@@ -1423,14 +1366,10 @@ async def test_user_info_no_user_id_view_only_admin_gets_proxy_admin_payload(moc
mock_get_user_info_for_proxy_admin,
)
- viewer = UserAPIKeyAuth(
- user_id="viewer", user_role=LitellmUserRoles.PROXY_ADMIN_VIEW_ONLY.value
- )
+ viewer = UserAPIKeyAuth(user_id="viewer", user_role=LitellmUserRoles.PROXY_ADMIN_VIEW_ONLY.value)
mock_request = mocker.MagicMock(spec=Request)
- response = await user_info(
- user_id=None, user_api_key_dict=viewer, request=mock_request
- )
+ response = await user_info(user_id=None, user_api_key_dict=viewer, request=mock_request)
mock_get_user_info_for_proxy_admin.assert_awaited_once_with(user_api_key_dict=viewer)
assert response is admin_payload
@@ -1457,9 +1396,7 @@ async def test_new_user_default_teams_flow(mocker):
mock_prisma_client.db.litellm_usertable.count = mock_count
persisted_user_row = mocker.MagicMock()
persisted_user_row.teams = ["96fed65b-0182-4ff4-8429-2721cd7d42af"]
- mock_prisma_client.db.litellm_usertable.find_unique = mocker.AsyncMock(
- return_value=persisted_user_row
- )
+ mock_prisma_client.db.litellm_usertable.find_unique = mocker.AsyncMock(return_value=persisted_user_row)
# Mock duplicate checks to pass
async def mock_check_duplicate_user_email(*args, **kwargs):
@@ -1527,26 +1464,20 @@ async def test_new_user_default_teams_flow(mocker):
)
# Create test request data WITHOUT teams (teams should come from defaults)
- user_request = NewUserRequest(
- user_email="test@example.com", user_role="internal_user"
- )
+ user_request = NewUserRequest(user_email="test@example.com", user_role="internal_user")
# Mock user_api_key_dict
mock_user_api_key_dict = UserAPIKeyAuth(user_id="test_admin")
# Call new_user function
- response = await new_user(
- data=user_request, user_api_key_dict=mock_user_api_key_dict
- )
+ response = await new_user(data=user_request, user_api_key_dict=mock_user_api_key_dict)
# Verify generate_key_helper_fn was called WITHOUT teams
mock_generate_key_helper_fn.assert_called_once()
call_kwargs = mock_generate_key_helper_fn.call_args.kwargs
# Teams should be removed from the data passed to generate_key_helper_fn
- assert (
- "teams" not in call_kwargs
- ), "Teams should not be passed to generate_key_helper_fn"
+ assert "teams" not in call_kwargs, "Teams should not be passed to generate_key_helper_fn"
assert call_kwargs["request_type"] == "user"
assert call_kwargs["user_email"] == "test@example.com"
assert call_kwargs["user_role"] == "internal_user"
@@ -1591,24 +1522,16 @@ def test_update_internal_new_user_params_proxy_admin_role():
try:
# Create test data with PROXY_ADMIN role
- data = NewUserRequest(
- user_email="admin@example.com", user_role=LitellmUserRoles.PROXY_ADMIN.value
- )
+ data = NewUserRequest(user_email="admin@example.com", user_role=LitellmUserRoles.PROXY_ADMIN.value)
data_json = data.model_dump(exclude_unset=True)
# Call the function
result = _update_internal_new_user_params(data_json=data_json, data=data)
# Assertions - default params should NOT be applied for PROXY_ADMIN
- assert (
- "max_budget" not in result
- ), "Default max_budget should NOT be applied to PROXY_ADMIN"
- assert (
- "models" not in result
- ), "Default models should NOT be applied to PROXY_ADMIN"
- assert (
- "tpm_limit" not in result
- ), "Default tpm_limit should NOT be applied to PROXY_ADMIN"
+ assert "max_budget" not in result, "Default max_budget should NOT be applied to PROXY_ADMIN"
+ assert "models" not in result, "Default models should NOT be applied to PROXY_ADMIN"
+ assert "tpm_limit" not in result, "Default tpm_limit should NOT be applied to PROXY_ADMIN"
# These should still work
assert result["user_email"] == "admin@example.com"
@@ -1722,15 +1645,9 @@ async def test_check_duplicate_user_email_case_insensitive(mocker):
user_email_clause = where_clause.get("user_email", {})
# Check that the query structure is correct for case insensitive search
- assert (
- "equals" in user_email_clause
- ), "Query should use 'equals' for case insensitive search"
- assert (
- user_email_clause.get("mode") == "insensitive"
- ), "Query should use 'insensitive' mode"
- assert (
- user_email_clause.get("equals") == "user@example.com"
- ), "Query should search for the provided email"
+ assert "equals" in user_email_clause, "Query should use 'equals' for case insensitive search"
+ assert user_email_clause.get("mode") == "insensitive", "Query should use 'insensitive' mode"
+ assert user_email_clause.get("equals") == "user@example.com", "Query should search for the provided email"
return mock_existing_user # Return existing user to simulate duplicate
@@ -1741,9 +1658,7 @@ async def test_check_duplicate_user_email_case_insensitive(mocker):
await _check_duplicate_user_email("user@example.com", mock_prisma_client)
assert exc_info.value.status_code == 409
- assert "User with email User@Example.com already exists" in str(
- exc_info.value.detail
- )
+ assert "User with email User@Example.com already exists" in str(exc_info.value.detail)
# Test Case 2: No duplicate found
async def mock_find_first_no_duplicate(*args, **kwargs):
@@ -1768,9 +1683,7 @@ async def test_check_duplicate_user_email_case_insensitive(mocker):
pytest.fail(f"Should not raise exception when no duplicate found, but got: {e}")
# Test Case 3: None email should not cause issues
- await _check_duplicate_user_email(
- None, mock_prisma_client
- ) # Should not raise exception
+ await _check_duplicate_user_email(None, mock_prisma_client) # Should not raise exception
@pytest.mark.asyncio
@@ -1880,9 +1793,7 @@ def test_process_keys_for_user_info_filters_dashboard_keys(monkeypatch):
# Verify dashboard key is not in results
result_team_ids = [key.get("team_id") for key in result]
- assert (
- UI_SESSION_TOKEN_TEAM_ID not in result_team_ids
- ), "Dashboard key should be filtered out"
+ assert UI_SESSION_TOKEN_TEAM_ID not in result_team_ids, "Dashboard key should be filtered out"
# Verify regular keys are included
assert "regular-team" in result_team_ids, "Regular team key should be included"
@@ -1892,9 +1803,7 @@ def test_process_keys_for_user_info_filters_dashboard_keys(monkeypatch):
result_tokens = [key.get("token") for key in result]
assert "sk-regular-token" in result_tokens, "Regular key should be included"
assert "sk-no-team-token" in result_tokens, "No-team key should be included"
- assert (
- "sk-dashboard-token" not in result_tokens
- ), "Dashboard key should not be included"
+ assert "sk-dashboard-token" not in result_tokens, "Dashboard key should not be included"
def test_process_keys_for_user_info_handles_none_keys(monkeypatch):
@@ -2419,9 +2328,7 @@ async def test_get_user_daily_activity_non_admin_cannot_view_other_users(monkeyp
)
assert exc_info.value.status_code == 403
- assert "Non-admin users can only view their own spend data" in str(
- exc_info.value.detail
- )
+ assert "Non-admin users can only view their own spend data" in str(exc_info.value.detail)
# Case 2: Non-admin omits user_id — should default to their own user_id
mock_response = MagicMock()
@@ -2708,14 +2615,10 @@ async def test_delete_user_cleans_up_created_by_invitation_links(mocker):
async def mock_find_unique(*args, **kwargs):
return mock_user_row
- mock_prisma_client.db.litellm_usertable.find_unique = mocker.AsyncMock(
- side_effect=mock_find_unique
- )
+ mock_prisma_client.db.litellm_usertable.find_unique = mocker.AsyncMock(side_effect=mock_find_unique)
# Mock find_many for teams (no teams)
- mock_prisma_client.db.litellm_teamtable.find_many = mocker.AsyncMock(
- return_value=[]
- )
+ mock_prisma_client.db.litellm_teamtable.find_many = mocker.AsyncMock(return_value=[])
# Mock all delete_many calls
mock_prisma_client.db.litellm_verificationtoken.find_many = mocker.AsyncMock(
@@ -2741,9 +2644,7 @@ async def test_delete_user_cleans_up_created_by_invitation_links(mocker):
# Call delete_user
data = DeleteUserRequest(user_ids=["admin-creator"])
- user_api_key_dict = UserAPIKeyAuth(
- user_id="proxy-admin", user_role=LitellmUserRoles.PROXY_ADMIN
- )
+ user_api_key_dict = UserAPIKeyAuth(user_id="proxy-admin", user_role=LitellmUserRoles.PROXY_ADMIN)
await delete_user(data=data, user_api_key_dict=user_api_key_dict)
@@ -2752,9 +2653,7 @@ async def test_delete_user_cleans_up_created_by_invitation_links(mocker):
call_kwargs = mock_prisma_client.db.litellm_invitationlink.delete_many.call_args
where_clause = call_kwargs.kwargs.get("where") or call_kwargs[1].get("where")
- assert (
- "OR" in where_clause
- ), "Should use OR to match user_id, created_by, and updated_by"
+ assert "OR" in where_clause, "Should use OR to match user_id, created_by, and updated_by"
or_conditions = where_clause["OR"]
assert len(or_conditions) == 3, "Should have 3 OR conditions"
@@ -2875,9 +2774,7 @@ async def test_delete_user_rejects_org_admin_deleting_outside_scope(mocker):
async def mock_find_unique(*args, **kwargs):
return mock_target_user
- mock_prisma_client.db.litellm_usertable.find_unique = mocker.AsyncMock(
- side_effect=mock_find_unique
- )
+ mock_prisma_client.db.litellm_usertable.find_unique = mocker.AsyncMock(side_effect=mock_find_unique)
# Caller (org_admin_user) administers org-A.
caller_membership = mocker.MagicMock()
@@ -2903,16 +2800,12 @@ async def test_delete_user_rejects_org_admin_deleting_outside_scope(mocker):
return [caller_membership]
return []
- mock_prisma_client.db.litellm_organizationmembership.find_many = mocker.AsyncMock(
- side_effect=mock_find_memberships
- )
+ mock_prisma_client.db.litellm_organizationmembership.find_many = mocker.AsyncMock(side_effect=mock_find_memberships)
mocker.patch("litellm.proxy.proxy_server.prisma_client", mock_prisma_client)
data = DeleteUserRequest(user_ids=["victim"])
- user_api_key_dict = UserAPIKeyAuth(
- user_id="org_admin_user", user_role=LitellmUserRoles.ORG_ADMIN
- )
+ user_api_key_dict = UserAPIKeyAuth(user_id="org_admin_user", user_role=LitellmUserRoles.ORG_ADMIN)
with pytest.raises(HTTPException) as exc:
await delete_user(data=data, user_api_key_dict=user_api_key_dict)
@@ -2920,11 +2813,8 @@ async def test_delete_user_rejects_org_admin_deleting_outside_scope(mocker):
# Critical: no delete_many calls should have executed.
assert (
- not hasattr(
- mock_prisma_client.db.litellm_verificationtoken.delete_many, "mock_calls"
- )
- or len(mock_prisma_client.db.litellm_verificationtoken.delete_many.mock_calls)
- == 0
+ not hasattr(mock_prisma_client.db.litellm_verificationtoken.delete_many, "mock_calls")
+ or len(mock_prisma_client.db.litellm_verificationtoken.delete_many.mock_calls) == 0
)
@@ -2943,9 +2833,7 @@ async def test_user_update_rejects_silent_create_for_non_proxy_admin(mocker):
mock_prisma_client = mocker.MagicMock()
# user_email lookup yields None → would silently create pre-fix.
- mock_prisma_client.db.litellm_usertable.find_first = mocker.AsyncMock(
- return_value=None
- )
+ mock_prisma_client.db.litellm_usertable.find_first = mocker.AsyncMock(return_value=None)
mocker.patch("litellm.proxy.proxy_server.prisma_client", mock_prisma_client)
user_request = UpdateUserRequest(
@@ -2959,9 +2847,7 @@ async def test_user_update_rejects_silent_create_for_non_proxy_admin(mocker):
)
with pytest.raises(HTTPException) as exc:
- await _update_single_user_helper(
- user_request=user_request, user_api_key_dict=org_admin
- )
+ await _update_single_user_helper(user_request=user_request, user_api_key_dict=org_admin)
assert exc.value.status_code == 404
@@ -3005,17 +2891,13 @@ async def test_user_info_v2_proxy_admin_can_query_any_user(mocker):
return mock_user_row
return None
- mock_prisma_client.db.litellm_usertable.find_unique = mocker.AsyncMock(
- side_effect=mock_find_unique
- )
+ mock_prisma_client.db.litellm_usertable.find_unique = mocker.AsyncMock(side_effect=mock_find_unique)
mocker.patch("litellm.proxy.proxy_server.prisma_client", mock_prisma_client)
mock_request = mocker.MagicMock(spec=Request)
- admin_key = UserAPIKeyAuth(
- user_id="admin-user", user_role=LitellmUserRoles.PROXY_ADMIN
- )
+ admin_key = UserAPIKeyAuth(user_id="admin-user", user_role=LitellmUserRoles.PROXY_ADMIN)
response = await user_info_v2(
request=mock_request,
@@ -3069,17 +2951,13 @@ async def test_user_info_v2_redacts_scim_enterprise_metadata(mocker):
return mock_user_row
return None
- mock_prisma_client.db.litellm_usertable.find_unique = mocker.AsyncMock(
- side_effect=mock_find_unique
- )
+ mock_prisma_client.db.litellm_usertable.find_unique = mocker.AsyncMock(side_effect=mock_find_unique)
mocker.patch("litellm.proxy.proxy_server.prisma_client", mock_prisma_client)
mock_request = mocker.MagicMock(spec=Request)
- admin_key = UserAPIKeyAuth(
- user_id="admin-user", user_role=LitellmUserRoles.PROXY_ADMIN
- )
+ admin_key = UserAPIKeyAuth(user_id="admin-user", user_role=LitellmUserRoles.PROXY_ADMIN)
response = await user_info_v2(
request=mock_request,
@@ -3088,9 +2966,7 @@ async def test_user_info_v2_redacts_scim_enterprise_metadata(mocker):
)
assert isinstance(response, UserInfoV2Response)
- assert response.metadata == {
- "scim_metadata": {"givenName": "Jane", "familyName": "Doe"}
- }
+ assert response.metadata == {"scim_metadata": {"givenName": "Jane", "familyName": "Doe"}}
assert "scim_enterprise" not in (response.metadata or {})
@@ -3159,17 +3035,13 @@ async def test_user_info_v2_internal_user_can_query_self(mocker):
return mock_user_row
return None
- mock_prisma_client.db.litellm_usertable.find_unique = mocker.AsyncMock(
- side_effect=mock_find_unique
- )
+ mock_prisma_client.db.litellm_usertable.find_unique = mocker.AsyncMock(side_effect=mock_find_unique)
mocker.patch("litellm.proxy.proxy_server.prisma_client", mock_prisma_client)
mock_request = mocker.MagicMock(spec=Request)
- user_key = UserAPIKeyAuth(
- user_id="self-user", user_role=LitellmUserRoles.INTERNAL_USER
- )
+ user_key = UserAPIKeyAuth(user_id="self-user", user_role=LitellmUserRoles.INTERNAL_USER)
response = await user_info_v2(
request=mock_request,
@@ -3204,17 +3076,13 @@ async def test_user_info_v2_internal_user_cannot_query_other(mocker):
return mock_caller_row
return None
- mock_prisma_client.db.litellm_usertable.find_unique = mocker.AsyncMock(
- side_effect=mock_find_unique
- )
+ mock_prisma_client.db.litellm_usertable.find_unique = mocker.AsyncMock(side_effect=mock_find_unique)
mocker.patch("litellm.proxy.proxy_server.prisma_client", mock_prisma_client)
mock_request = mocker.MagicMock(spec=Request)
- user_key = UserAPIKeyAuth(
- user_id="caller-user", user_role=LitellmUserRoles.INTERNAL_USER
- )
+ user_key = UserAPIKeyAuth(user_id="caller-user", user_role=LitellmUserRoles.INTERNAL_USER)
with pytest.raises(ProxyException) as exc_info:
await user_info_v2(
@@ -3261,17 +3129,13 @@ async def test_user_info_v2_no_user_id_defaults_to_self(mocker):
return mock_user_row
return None
- mock_prisma_client.db.litellm_usertable.find_unique = mocker.AsyncMock(
- side_effect=mock_find_unique
- )
+ mock_prisma_client.db.litellm_usertable.find_unique = mocker.AsyncMock(side_effect=mock_find_unique)
mocker.patch("litellm.proxy.proxy_server.prisma_client", mock_prisma_client)
mock_request = mocker.MagicMock(spec=Request)
- user_key = UserAPIKeyAuth(
- user_id="my-user-id", user_role=LitellmUserRoles.INTERNAL_USER
- )
+ user_key = UserAPIKeyAuth(user_id="my-user-id", user_role=LitellmUserRoles.INTERNAL_USER)
# Call without user_id
response = await user_info_v2(
@@ -3299,17 +3163,13 @@ async def test_user_info_v2_nonexistent_user_returns_404(mocker):
async def mock_find_unique(*args, **kwargs):
return None
- mock_prisma_client.db.litellm_usertable.find_unique = mocker.AsyncMock(
- side_effect=mock_find_unique
- )
+ mock_prisma_client.db.litellm_usertable.find_unique = mocker.AsyncMock(side_effect=mock_find_unique)
mocker.patch("litellm.proxy.proxy_server.prisma_client", mock_prisma_client)
mock_request = mocker.MagicMock(spec=Request)
- admin_key = UserAPIKeyAuth(
- user_id="admin-user", user_role=LitellmUserRoles.PROXY_ADMIN
- )
+ admin_key = UserAPIKeyAuth(user_id="admin-user", user_role=LitellmUserRoles.PROXY_ADMIN)
with pytest.raises(ProxyException) as exc_info:
await user_info_v2(
@@ -3357,17 +3217,13 @@ async def test_user_info_v2_response_shape(mocker):
async def mock_find_unique(*args, **kwargs):
return mock_user_row
- mock_prisma_client.db.litellm_usertable.find_unique = mocker.AsyncMock(
- side_effect=mock_find_unique
- )
+ mock_prisma_client.db.litellm_usertable.find_unique = mocker.AsyncMock(side_effect=mock_find_unique)
mocker.patch("litellm.proxy.proxy_server.prisma_client", mock_prisma_client)
mock_request = mocker.MagicMock(spec=Request)
- admin_key = UserAPIKeyAuth(
- user_id="admin-user", user_role=LitellmUserRoles.PROXY_ADMIN
- )
+ admin_key = UserAPIKeyAuth(user_id="admin-user", user_role=LitellmUserRoles.PROXY_ADMIN)
response = await user_info_v2(
request=mock_request,
@@ -3402,9 +3258,7 @@ async def test_user_info_v2_response_shape(mocker):
# The dashboard's user edit form hydrates its per-model budget rows from
# these two, so dropping them makes a save replace the user's budgets.
- assert response_dict["model_max_budget"] == {
- "gpt-3.5-turbo": {"budget_limit": 5.0, "time_period": "30d"}
- }
+ assert response_dict["model_max_budget"] == {"gpt-3.5-turbo": {"budget_limit": 5.0, "time_period": "30d"}}
assert response_dict["model_max_budget_usage"] == {
"gpt-3.5-turbo": {"current_spend": 0.0, "budget_limit": 5.0, "time_period": "30d"}
}
@@ -3463,9 +3317,7 @@ async def test_user_info_v2_team_admin_can_query_team_member(mocker):
return mock_target
return None
- mock_prisma_client.db.litellm_usertable.find_unique = mocker.AsyncMock(
- side_effect=mock_find_unique
- )
+ mock_prisma_client.db.litellm_usertable.find_unique = mocker.AsyncMock(side_effect=mock_find_unique)
# Mock team with caller as admin
mock_team = mocker.MagicMock()
@@ -3482,17 +3334,13 @@ async def test_user_info_v2_team_admin_can_query_team_member(mocker):
async def mock_find_many_teams(*args, **kwargs):
return [mock_team]
- mock_prisma_client.db.litellm_teamtable.find_many = mocker.AsyncMock(
- side_effect=mock_find_many_teams
- )
+ mock_prisma_client.db.litellm_teamtable.find_many = mocker.AsyncMock(side_effect=mock_find_many_teams)
mocker.patch("litellm.proxy.proxy_server.prisma_client", mock_prisma_client)
mock_request = mocker.MagicMock(spec=Request)
- team_admin_key = UserAPIKeyAuth(
- user_id="team-admin-user", user_role=LitellmUserRoles.INTERNAL_USER
- )
+ team_admin_key = UserAPIKeyAuth(user_id="team-admin-user", user_role=LitellmUserRoles.INTERNAL_USER)
response = await user_info_v2(
request=mock_request,
@@ -3532,9 +3380,7 @@ async def test_user_info_v2_team_admin_cannot_query_non_team_member(mocker):
return mock_target
return None
- mock_prisma_client.db.litellm_usertable.find_unique = mocker.AsyncMock(
- side_effect=mock_find_unique
- )
+ mock_prisma_client.db.litellm_usertable.find_unique = mocker.AsyncMock(side_effect=mock_find_unique)
# Mock team where caller is admin
mock_team = mocker.MagicMock()
@@ -3550,17 +3396,13 @@ async def test_user_info_v2_team_admin_cannot_query_non_team_member(mocker):
async def mock_find_many_teams(*args, **kwargs):
return [mock_team]
- mock_prisma_client.db.litellm_teamtable.find_many = mocker.AsyncMock(
- side_effect=mock_find_many_teams
- )
+ mock_prisma_client.db.litellm_teamtable.find_many = mocker.AsyncMock(side_effect=mock_find_many_teams)
mocker.patch("litellm.proxy.proxy_server.prisma_client", mock_prisma_client)
mock_request = mocker.MagicMock(spec=Request)
- team_admin_key = UserAPIKeyAuth(
- user_id="team-admin-user", user_role=LitellmUserRoles.INTERNAL_USER
- )
+ team_admin_key = UserAPIKeyAuth(user_id="team-admin-user", user_role=LitellmUserRoles.INTERNAL_USER)
with pytest.raises(ProxyException) as exc_info:
await user_info_v2(
@@ -3610,18 +3452,14 @@ async def test_user_info_v2_url_encoding_plus_character(mocker):
return mock_user_row
return None
- mock_prisma_client.db.litellm_usertable.find_unique = mocker.AsyncMock(
- side_effect=mock_find_unique
- )
+ mock_prisma_client.db.litellm_usertable.find_unique = mocker.AsyncMock(side_effect=mock_find_unique)
mocker.patch("litellm.proxy.proxy_server.prisma_client", mock_prisma_client)
mock_request = mocker.MagicMock(spec=Request)
mock_request.url.query = f"user_id={expected_user_id}"
- admin_key = UserAPIKeyAuth(
- user_id="admin-user", user_role=LitellmUserRoles.PROXY_ADMIN
- )
+ admin_key = UserAPIKeyAuth(user_id="admin-user", user_role=LitellmUserRoles.PROXY_ADMIN)
# Simulate FastAPI converting + to space
decoded_user_id = "machine-user admin@example.com"
@@ -3718,9 +3556,7 @@ def test_enforce_user_info_access_admin_bypass():
_enforce_user_info_access,
)
- admin = UserAPIKeyAuth(
- user_id="admin", user_role=LitellmUserRoles.PROXY_ADMIN.value
- )
+ admin = UserAPIKeyAuth(user_id="admin", user_role=LitellmUserRoles.PROXY_ADMIN.value)
# Should not raise even when querying a different user
_enforce_user_info_access(user_id="someone_else", user_api_key_dict=admin)
@@ -3759,9 +3595,7 @@ def test_enforce_user_info_access_owner_allowed():
_enforce_user_info_access,
)
- user = UserAPIKeyAuth(
- user_id="alice", user_role=LitellmUserRoles.INTERNAL_USER.value
- )
+ user = UserAPIKeyAuth(user_id="alice", user_role=LitellmUserRoles.INTERNAL_USER.value)
_enforce_user_info_access(user_id="alice", user_api_key_dict=user)
@@ -3773,9 +3607,7 @@ def test_enforce_user_info_access_no_user_id_allowed():
_enforce_user_info_access,
)
- user = UserAPIKeyAuth(
- user_id="alice", user_role=LitellmUserRoles.INTERNAL_USER.value
- )
+ user = UserAPIKeyAuth(user_id="alice", user_role=LitellmUserRoles.INTERNAL_USER.value)
_enforce_user_info_access(user_id=None, user_api_key_dict=user)
@@ -3832,9 +3664,7 @@ async def test_ghsa_wvg4_non_admin_cannot_self_escalate_max_budget(mocker, budge
"max_budget": 100,
}
existing_user.user_id = "user-1"
- mock_prisma_client.db.litellm_usertable.find_first = mocker.AsyncMock(
- return_value=existing_user
- )
+ mock_prisma_client.db.litellm_usertable.find_first = mocker.AsyncMock(return_value=existing_user)
mocker.patch("litellm.proxy.proxy_server.prisma_client", mock_prisma_client)
user_request = UpdateUserRequest.model_validate({"user_id": "user-1", budget_field: budget_value})
@@ -3844,9 +3674,7 @@ async def test_ghsa_wvg4_non_admin_cannot_self_escalate_max_budget(mocker, budge
)
with pytest.raises(HTTPException) as exc:
- await _update_single_user_helper(
- user_request=user_request, user_api_key_dict=caller
- )
+ await _update_single_user_helper(user_request=user_request, user_api_key_dict=caller)
assert exc.value.status_code == 403
assert budget_field in str(exc.value.detail)
mock_prisma_client.update_data.assert_not_called()
@@ -3868,9 +3696,7 @@ async def test_ghsa_wvg4_non_admin_cannot_self_escalate_spend(mocker):
"spend": 50.0,
}
existing_user.user_id = "user-1"
- mock_prisma_client.db.litellm_usertable.find_first = mocker.AsyncMock(
- return_value=existing_user
- )
+ mock_prisma_client.db.litellm_usertable.find_first = mocker.AsyncMock(return_value=existing_user)
mocker.patch("litellm.proxy.proxy_server.prisma_client", mock_prisma_client)
user_request = UpdateUserRequest(
@@ -3883,9 +3709,7 @@ async def test_ghsa_wvg4_non_admin_cannot_self_escalate_spend(mocker):
)
with pytest.raises(HTTPException) as exc:
- await _update_single_user_helper(
- user_request=user_request, user_api_key_dict=caller
- )
+ await _update_single_user_helper(user_request=user_request, user_api_key_dict=caller)
assert exc.value.status_code == 403
assert "spend" in str(exc.value.detail)
@@ -3904,12 +3728,8 @@ async def test_ghsa_wvg4_proxy_admin_can_update_user_budget(mocker):
"max_budget": 100,
}
existing_user.user_id = "target-user"
- mock_prisma_client.db.litellm_usertable.find_first = mocker.AsyncMock(
- return_value=existing_user
- )
- mock_prisma_client.update_data = mocker.AsyncMock(
- return_value={"user_id": "target-user", "max_budget": 500}
- )
+ mock_prisma_client.db.litellm_usertable.find_first = mocker.AsyncMock(return_value=existing_user)
+ mock_prisma_client.update_data = mocker.AsyncMock(return_value={"user_id": "target-user", "max_budget": 500})
mock_prisma_client.jsonify_object = mocker.MagicMock(side_effect=lambda x: x)
mocker.patch("litellm.proxy.proxy_server.prisma_client", mock_prisma_client)
mocker.patch("litellm.proxy.proxy_server.litellm_proxy_admin_name", "admin")
@@ -3923,9 +3743,7 @@ async def test_ghsa_wvg4_proxy_admin_can_update_user_budget(mocker):
user_role=LitellmUserRoles.PROXY_ADMIN,
)
- result = await _update_single_user_helper(
- user_request=user_request, user_api_key_dict=admin_caller
- )
+ result = await _update_single_user_helper(user_request=user_request, user_api_key_dict=admin_caller)
assert result is not None
@@ -3941,12 +3759,8 @@ async def test_admin_user_update_spend_invalidates_counter(mocker):
existing_user = mocker.MagicMock()
existing_user.model_dump.return_value = {"user_id": "target-user", "spend": 50.0}
existing_user.user_id = "target-user"
- mock_prisma_client.db.litellm_usertable.find_first = mocker.AsyncMock(
- return_value=existing_user
- )
- mock_prisma_client.update_data = mocker.AsyncMock(
- return_value={"user_id": "target-user", "spend": -25.0}
- )
+ mock_prisma_client.db.litellm_usertable.find_first = mocker.AsyncMock(return_value=existing_user)
+ mock_prisma_client.update_data = mocker.AsyncMock(return_value={"user_id": "target-user", "spend": -25.0})
mock_prisma_client.jsonify_object = mocker.MagicMock(side_effect=lambda x: x)
mocker.patch("litellm.proxy.proxy_server.prisma_client", mock_prisma_client)
mocker.patch("litellm.proxy.proxy_server.litellm_proxy_admin_name", "admin")
@@ -3961,13 +3775,9 @@ async def test_admin_user_update_spend_invalidates_counter(mocker):
# without raising the recurring budget ceiling. Future changes should
# continue allowing negative spend counters.
user_request = UpdateUserRequest(user_id="target-user", spend=-25)
- admin_caller = UserAPIKeyAuth(
- user_id="admin-1", user_role=LitellmUserRoles.PROXY_ADMIN
- )
+ admin_caller = UserAPIKeyAuth(user_id="admin-1", user_role=LitellmUserRoles.PROXY_ADMIN)
- await _update_single_user_helper(
- user_request=user_request, user_api_key_dict=admin_caller
- )
+ await _update_single_user_helper(user_request=user_request, user_api_key_dict=admin_caller)
mock_invalidate.assert_awaited_once_with(counter_key="spend:user:target-user")
@@ -3984,9 +3794,7 @@ async def test_user_update_rejects_non_finite_spend(mocker):
existing_user = mocker.MagicMock()
existing_user.model_dump.return_value = {"user_id": "target-user", "spend": 50.0}
existing_user.user_id = "target-user"
- mock_prisma_client.db.litellm_usertable.find_first = mocker.AsyncMock(
- return_value=existing_user
- )
+ mock_prisma_client.db.litellm_usertable.find_first = mocker.AsyncMock(return_value=existing_user)
mock_prisma_client.update_data = mocker.AsyncMock()
mocker.patch("litellm.proxy.proxy_server.prisma_client", mock_prisma_client)
mocker.patch("litellm.proxy.proxy_server.litellm_proxy_admin_name", "admin")
@@ -3996,14 +3804,10 @@ async def test_user_update_rejects_non_finite_spend(mocker):
)
user_request = UpdateUserRequest(user_id="target-user", spend=float("nan"))
- admin_caller = UserAPIKeyAuth(
- user_id="admin-1", user_role=LitellmUserRoles.PROXY_ADMIN
- )
+ admin_caller = UserAPIKeyAuth(user_id="admin-1", user_role=LitellmUserRoles.PROXY_ADMIN)
with pytest.raises(HTTPException) as exc:
- await _update_single_user_helper(
- user_request=user_request, user_api_key_dict=admin_caller
- )
+ await _update_single_user_helper(user_request=user_request, user_api_key_dict=admin_caller)
assert exc.value.status_code == 400
mock_prisma_client.update_data.assert_not_called()
mock_invalidate.assert_not_awaited()
@@ -4023,9 +3827,7 @@ async def test_resolve_user_email_metadata_maps_page_user_ids_to_email(mocker):
mock_prisma_client = mocker.MagicMock()
find_many = mocker.AsyncMock(
return_value=[
- SimpleNamespace(
- user_id="u1", user_email="alice@example.com", user_alias="Alice"
- ),
+ SimpleNamespace(user_id="u1", user_email="alice@example.com", user_alias="Alice"),
SimpleNamespace(user_id="u2", user_email=None, user_alias="bob-alias"),
]
)
@@ -4224,19 +4026,13 @@ def _object_permission_mocks(mocker, existing_object_permission_id=None):
}
existing_user.user_id = "target-user"
existing_user.object_permission_id = existing_object_permission_id
- mock_prisma_client.db.litellm_usertable.find_first = mocker.AsyncMock(
- return_value=existing_user
- )
- mock_prisma_client.db.litellm_objectpermissiontable.find_unique = mocker.AsyncMock(
- return_value=None
- )
+ mock_prisma_client.db.litellm_usertable.find_first = mocker.AsyncMock(return_value=existing_user)
+ mock_prisma_client.db.litellm_objectpermissiontable.find_unique = mocker.AsyncMock(return_value=None)
mock_prisma_client.db.litellm_objectpermissiontable.upsert = mocker.AsyncMock(
return_value=SimpleNamespace(object_permission_id="perm-new")
)
mock_prisma_client.db.litellm_mcpservertable.find_many = mocker.AsyncMock(return_value=[])
- mock_prisma_client.update_data = mocker.AsyncMock(
- return_value={"user_id": "target-user"}
- )
+ mock_prisma_client.update_data = mocker.AsyncMock(return_value={"user_id": "target-user"})
mock_prisma_client.jsonify_object = mocker.MagicMock(side_effect=lambda x: x)
mocker.patch("litellm.proxy.proxy_server.prisma_client", mock_prisma_client)
mocker.patch("litellm.proxy.proxy_server.litellm_proxy_admin_name", "admin")
@@ -4272,9 +4068,7 @@ async def test_user_update_persists_mcp_entitlement_and_links_it(mocker):
"mcp_tool_permissions": {"github": ["list_issues"]},
},
),
- user_api_key_dict=UserAPIKeyAuth(
- user_id="admin-1", user_role=LitellmUserRoles.PROXY_ADMIN
- ),
+ user_api_key_dict=UserAPIKeyAuth(user_id="admin-1", user_role=LitellmUserRoles.PROXY_ADMIN),
)
upsert_kwargs = mock_prisma_client.db.litellm_objectpermissiontable.upsert.call_args.kwargs
@@ -4308,9 +4102,7 @@ async def test_user_update_invalidates_the_cached_entitlement(mocker):
user_id="target-user",
object_permission={"mcp_tool_permissions": {"github": []}},
),
- user_api_key_dict=UserAPIKeyAuth(
- user_id="admin-1", user_role=LitellmUserRoles.PROXY_ADMIN
- ),
+ user_api_key_dict=UserAPIKeyAuth(user_id="admin-1", user_role=LitellmUserRoles.PROXY_ADMIN),
)
deleted = {call.kwargs["key"] for call in cache.async_delete_cache.call_args_list}
@@ -4343,9 +4135,7 @@ async def test_admin_can_clear_a_users_mcp_entitlement(mocker):
await _update_single_user_helper(
user_request=UpdateUserRequest(user_id="target-user", object_permission={}),
- user_api_key_dict=UserAPIKeyAuth(
- user_id="admin-1", user_role=LitellmUserRoles.PROXY_ADMIN
- ),
+ user_api_key_dict=UserAPIKeyAuth(user_id="admin-1", user_role=LitellmUserRoles.PROXY_ADMIN),
)
written = mock_prisma_client.update_data.call_args.kwargs["data"]
@@ -4382,9 +4172,7 @@ async def test_user_update_invalidates_both_the_old_and_new_permission_rows(mock
user_id="target-user",
object_permission={"mcp_tool_permissions": {"github": ["list_issues"]}},
),
- user_api_key_dict=UserAPIKeyAuth(
- user_id="admin-1", user_role=LitellmUserRoles.PROXY_ADMIN
- ),
+ user_api_key_dict=UserAPIKeyAuth(user_id="admin-1", user_role=LitellmUserRoles.PROXY_ADMIN),
)
deleted = {call.kwargs["key"] for call in cache.async_delete_cache.call_args_list}
@@ -4415,9 +4203,7 @@ async def test_non_admin_cannot_clear_their_own_mcp_entitlement(mocker):
with pytest.raises(HTTPException) as exc:
await _update_single_user_helper(
user_request=UpdateUserRequest(user_id="target-user", object_permission={}),
- user_api_key_dict=UserAPIKeyAuth(
- user_id="target-user", user_role=LitellmUserRoles.INTERNAL_USER
- ),
+ user_api_key_dict=UserAPIKeyAuth(user_id="target-user", user_role=LitellmUserRoles.INTERNAL_USER),
)
assert exc.value.status_code == 403
@@ -4445,9 +4231,7 @@ async def test_non_admin_cannot_rewrite_their_own_mcp_entitlement(mocker):
user_id="target-user",
object_permission={"mcp_servers": [], "mcp_tool_permissions": {}},
),
- user_api_key_dict=UserAPIKeyAuth(
- user_id="target-user", user_role=LitellmUserRoles.INTERNAL_USER
- ),
+ user_api_key_dict=UserAPIKeyAuth(user_id="target-user", user_role=LitellmUserRoles.INTERNAL_USER),
)
assert exc.value.status_code == 403
@@ -4464,9 +4248,7 @@ async def test_new_user_persists_the_requested_mcp_entitlement(mocker):
return_value=SimpleNamespace(object_permission_id="perm-created")
)
mock_prisma_client.db.litellm_mcpservertable.find_many = mocker.AsyncMock(return_value=[])
- mock_prisma_client.db.litellm_usertable.find_first = mocker.AsyncMock(
- return_value=None
- )
+ mock_prisma_client.db.litellm_usertable.find_first = mocker.AsyncMock(return_value=None)
mock_prisma_client.db.litellm_usertable.count = mocker.AsyncMock(return_value=0)
mocker.patch("litellm.proxy.proxy_server.prisma_client", mock_prisma_client)
mocker.patch(
@@ -4475,9 +4257,7 @@ async def test_new_user_persists_the_requested_mcp_entitlement(mocker):
)
mock_generate = mocker.patch(
"litellm.proxy.management_endpoints.internal_user_endpoints.generate_key_helper_fn",
- new=mocker.AsyncMock(
- return_value={"user_id": "new-human", "token": "sk-x", "expires": None}
- ),
+ new=mocker.AsyncMock(return_value={"user_id": "new-human", "token": "sk-x", "expires": None}),
)
mocker.patch(
"litellm.proxy.hooks.user_management_event_hooks.UserManagementEventHooks.async_user_created_hook",
@@ -4489,9 +4269,7 @@ async def test_new_user_persists_the_requested_mcp_entitlement(mocker):
user_id="new-human",
object_permission={"mcp_tool_permissions": {"github": ["list_issues"]}},
),
- user_api_key_dict=UserAPIKeyAuth(
- user_id="admin-1", user_role=LitellmUserRoles.PROXY_ADMIN
- ),
+ user_api_key_dict=UserAPIKeyAuth(user_id="admin-1", user_role=LitellmUserRoles.PROXY_ADMIN),
)
created = mock_prisma_client.db.litellm_objectpermissiontable.create.call_args.kwargs["data"]
@@ -4533,16 +4311,12 @@ async def test_user_info_v2_returns_the_mcp_entitlement(mocker):
response = await user_info_v2(
request=SimpleNamespace(query_params={}),
user_id="human-1",
- user_api_key_dict=UserAPIKeyAuth(
- user_id="admin-1", user_role=LitellmUserRoles.PROXY_ADMIN
- ),
+ user_api_key_dict=UserAPIKeyAuth(user_id="admin-1", user_role=LitellmUserRoles.PROXY_ADMIN),
)
assert response.object_permission is not None
assert response.object_permission.mcp_servers == ["github"]
- assert response.object_permission.mcp_tool_permissions == {
- "github": ["list_issues"]
- }
+ assert response.object_permission.mcp_tool_permissions == {"github": ["list_issues"]}
@pytest.mark.asyncio
@@ -4558,9 +4332,7 @@ async def test_user_info_v2_returns_the_mcp_entitlement(mocker):
],
ids=["supplied", "omitted", "empty"],
)
-async def test_user_new_persists_model_max_budget(
- monkeypatch, model_max_budget, expected_written
-):
+async def test_user_new_persists_model_max_budget(monkeypatch, model_max_budget, expected_written):
"""
/user/new used to echo model_max_budget back while writing {} to the user row,
so a per-model budget looked configured and was read by nothing.
@@ -4674,6 +4446,11 @@ async def test_user_update_hashes_and_persists_strong_password(_admin_prisma, mo
_update_single_user_helper,
)
+ mocker.patch( # test-quality-ok: same module-global mocking every test in this file already uses
+ "litellm.proxy.proxy_server.general_settings",
+ {"password_policy_check_breached_passwords": False},
+ )
+
mock_prisma_client = _admin_prisma
existing_user = mocker.MagicMock()
existing_user.model_dump.return_value = {"user_id": "target-user"}
@@ -4691,3 +4468,149 @@ async def test_user_update_hashes_and_persists_strong_password(_admin_prisma, mo
written_data = mock_prisma_client.update_data.call_args.kwargs["data"]
assert written_data.get("password") is not None
assert written_data["password"] != strong_password
+ # An admin-set password is known to the admin, so the user must be forced
+ # to change it at next login and the breach screen re-armed.
+ assert written_data["password_reset_required"] is True
+ assert written_data["last_breach_check_at"] is None
+
+
+@pytest.mark.asyncio
+@respx.mock
+async def test_user_update_rejects_breached_password(_admin_prisma):
+ """A strength-passing password found in the HIBP corpus must be rejected
+ before it ever reaches the DB write."""
+ from litellm.proxy.management_endpoints.internal_user_endpoints import (
+ _update_single_user_helper,
+ )
+
+ password = "Str0ng!Passw0rd"
+ sha1 = hashlib.sha1(password.encode("utf-8"), usedforsecurity=False).hexdigest().upper()
+ respx.get(f"https://api.pwnedpasswords.com/range/{sha1[:5]}").mock(
+ return_value=httpx.Response(200, text=f"{sha1[5:]}:1387")
+ )
+
+ user_request = UpdateUserRequest(user_id="target-user", password=password)
+ admin_caller = UserAPIKeyAuth(user_id="admin-1", user_role=LitellmUserRoles.PROXY_ADMIN)
+
+ with pytest.raises(ProxyException) as exc_info:
+ await _update_single_user_helper(user_request=user_request, user_api_key_dict=admin_caller)
+
+ assert exc_info.value.code == "400"
+ assert "data breaches" in exc_info.value.message
+ _admin_prisma.db.litellm_usertable.find_first.assert_not_called()
+
+
+@pytest.mark.asyncio
+async def test_bulk_update_all_users_rejects_a_password(_admin_prisma):
+ """The all_users fast path writes user_updates straight to update_many,
+ bypassing _update_single_user_helper. A password riding along would be
+ stored as unvalidated plaintext on every row, so it must be rejected
+ before any DB access."""
+ from fastapi import HTTPException
+
+ from litellm.proxy._types import UpdateUserRequestNoUserIDorEmail
+ from litellm.proxy.management_endpoints.internal_user_endpoints import bulk_user_update
+ from litellm.types.proxy.management_endpoints.internal_user_endpoints import (
+ BulkUpdateUserRequest,
+ )
+
+ data = BulkUpdateUserRequest(
+ all_users=True,
+ user_updates=UpdateUserRequestNoUserIDorEmail(password="Str0ng!Passw0rd"),
+ )
+ admin_caller = UserAPIKeyAuth(user_id="admin-1", user_role=LitellmUserRoles.PROXY_ADMIN)
+
+ with pytest.raises(HTTPException) as exc_info:
+ await bulk_user_update(data=data, user_api_key_dict=admin_caller)
+
+ assert exc_info.value.status_code == 400
+ assert "not supported" in str(exc_info.value.detail)
+ _admin_prisma.db.litellm_usertable.find_many.assert_not_called()
+ _admin_prisma.db.litellm_usertable.update_many.assert_not_called()
+
+
+def _hibp_client_with_handler(handler) -> AsyncHTTPHandler:
+ """A real AsyncHTTPHandler over httpx.MockTransport (the DI seam used
+ throughout test_password_policy.py), so no network is touched."""
+ http_handler = AsyncHTTPHandler()
+ http_handler.client = httpx.AsyncClient(transport=httpx.MockTransport(handler))
+ return http_handler
+
+
+@pytest.mark.asyncio
+async def test_bulk_update_breached_password_fails_only_that_user(_admin_prisma, mocker):
+ """In a bulk batch, a breached password fails only its own entry, before
+ any DB write for it; sibling entries with acceptable passwords persist."""
+ from litellm.proxy.management_endpoints.internal_user_endpoints import (
+ bulk_update_processed_users,
+ )
+
+ breached = "Br3ached!Passw0rd"
+ clean = "NewP@ssw0rd123"
+ breached_sha1 = hashlib.sha1(breached.encode("utf-8"), usedforsecurity=False).hexdigest().upper()
+
+ def handler(request: httpx.Request) -> httpx.Response:
+ if request.url.path == f"/range/{breached_sha1[:5]}":
+ return httpx.Response(200, text=f"{breached_sha1[5:]}:1387")
+ return httpx.Response(200, text="0000000000000000000000000000000000A:1")
+
+ mock_prisma_client = _admin_prisma
+ existing_user = mocker.MagicMock()
+ existing_user.model_dump.return_value = {"user_id": "user-clean"}
+ existing_user.user_id = "user-clean"
+ mock_prisma_client.db.litellm_usertable.find_first = mocker.AsyncMock(return_value=existing_user)
+ mock_prisma_client.update_data = mocker.AsyncMock(return_value={"user_id": "user-clean"})
+ mock_prisma_client.jsonify_object = mocker.MagicMock(side_effect=lambda x: x)
+
+ admin_caller = UserAPIKeyAuth(user_id="admin-1", user_role=LitellmUserRoles.PROXY_ADMIN)
+ response = await bulk_update_processed_users(
+ users_to_update=[
+ UpdateUserRequest(user_id="user-breached", password=breached),
+ UpdateUserRequest(user_id="user-clean", password=clean),
+ ],
+ user_api_key_dict=admin_caller,
+ hibp_client=_hibp_client_with_handler(handler),
+ )
+
+ assert response.successful_updates == 1
+ assert response.failed_updates == 1
+ by_user = {r.user_id: r for r in response.results}
+ assert by_user["user-breached"].success is False
+ assert "data breaches" in by_user["user-breached"].error
+ assert by_user["user-clean"].success is True
+ (write_call,) = mock_prisma_client.update_data.call_args_list
+ assert write_call.kwargs["user_id"] == "user-clean"
+
+
+@pytest.mark.asyncio
+async def test_bulk_update_screens_shared_password_with_single_lookup(_admin_prisma, mocker):
+ """A batch where every user gets the same password costs one HIBP lookup,
+ not one per user (the serial per-user checks this regresses against)."""
+ from litellm.proxy.management_endpoints.internal_user_endpoints import (
+ bulk_update_processed_users,
+ )
+
+ lookup_count = 0
+
+ def handler(request: httpx.Request) -> httpx.Response:
+ nonlocal lookup_count
+ lookup_count += 1
+ return httpx.Response(200, text="0000000000000000000000000000000000A:1")
+
+ mock_prisma_client = _admin_prisma
+ existing_user = mocker.MagicMock()
+ existing_user.model_dump.return_value = {"user_id": "user-0"}
+ existing_user.user_id = "user-0"
+ mock_prisma_client.db.litellm_usertable.find_first = mocker.AsyncMock(return_value=existing_user)
+ mock_prisma_client.update_data = mocker.AsyncMock(return_value={"user_id": "user-0"})
+ mock_prisma_client.jsonify_object = mocker.MagicMock(side_effect=lambda x: x)
+
+ admin_caller = UserAPIKeyAuth(user_id="admin-1", user_role=LitellmUserRoles.PROXY_ADMIN)
+ response = await bulk_update_processed_users(
+ users_to_update=[UpdateUserRequest(user_id=f"user-{i}", password="NewP@ssw0rd123") for i in range(5)],
+ user_api_key_dict=admin_caller,
+ hibp_client=_hibp_client_with_handler(handler),
+ )
+
+ assert response.successful_updates == 5
+ assert lookup_count == 1
diff --git a/tests/test_litellm/proxy/management_endpoints/test_password_endpoints.py b/tests/test_litellm/proxy/management_endpoints/test_password_endpoints.py
new file mode 100644
index 00000000000..bd154ebab41
--- /dev/null
+++ b/tests/test_litellm/proxy/management_endpoints/test_password_endpoints.py
@@ -0,0 +1,331 @@
+"""
+Tests for POST /user/password/change (litellm/proxy/management_endpoints/password_endpoints.py).
+
+HIBP traffic is intercepted with respx; no test here touches the network.
+"""
+
+import hashlib
+from unittest.mock import AsyncMock, MagicMock, patch
+
+import httpx
+import pytest
+import respx
+from fastapi import HTTPException
+
+from litellm.proxy._types import LitellmTableNames, ProxyErrorTypes, ProxyException, UserAPIKeyAuth
+from litellm.proxy.management_endpoints.password_endpoints import change_password
+from litellm.proxy.utils import hash_password, verify_password
+
+CURRENT_PASSWORD = "OldP@ssw0rd-2026"
+NEW_PASSWORD = "NewP@ssw0rd-2026"
+
+_POLICY_NO_BREACH_CHECK = {"password_policy_check_breached_passwords": False}
+
+
+def _make_user_row(password: str | None) -> MagicMock:
+ user = MagicMock()
+ user.user_id = "user-123"
+ user.password = password
+ return user
+
+
+def _make_prisma(user: MagicMock | None) -> MagicMock:
+ prisma = MagicMock()
+ prisma.db.litellm_usertable.find_first = AsyncMock(return_value=user)
+ prisma.db.litellm_usertable.update = AsyncMock(return_value=user)
+ return prisma
+
+
+def _caller(user_id: str | None = "user-123") -> UserAPIKeyAuth:
+ return UserAPIKeyAuth(user_id=user_id)
+
+
+def _hibp_url_for(password: str) -> str:
+ sha1 = hashlib.sha1(password.encode(), usedforsecurity=False).hexdigest().upper()
+ return f"https://api.pwnedpasswords.com/range/{sha1[:5]}"
+
+
+def _hibp_suffix_for(password: str) -> str:
+ return hashlib.sha1(password.encode(), usedforsecurity=False).hexdigest().upper()[5:]
+
+
+@pytest.mark.asyncio
+async def test_change_password_success_writes_new_scrypt_hash():
+ from litellm.proxy._types import ChangePasswordRequest
+
+ prisma = _make_prisma(_make_user_row(hash_password(CURRENT_PASSWORD)))
+
+ with (
+ patch( # test-quality-ok: change_password reads proxy_server module globals; no injection seam
+ "litellm.proxy.proxy_server.prisma_client", prisma
+ ),
+ patch( # test-quality-ok: change_password reads proxy_server module globals; no injection seam
+ "litellm.proxy.proxy_server.general_settings", _POLICY_NO_BREACH_CHECK
+ ),
+ ):
+ response = await change_password(
+ data=ChangePasswordRequest(current_password=CURRENT_PASSWORD, new_password=NEW_PASSWORD),
+ user_api_key_dict=_caller(),
+ )
+
+ assert response.user_id == "user-123"
+ update_kwargs = prisma.db.litellm_usertable.update.call_args.kwargs
+ assert update_kwargs["where"] == {"user_id": "user-123"}
+ stored = update_kwargs["data"]["password"]
+ assert stored != NEW_PASSWORD
+ assert verify_password(NEW_PASSWORD, stored)
+ # A successful change lifts any pending forced reset and re-arms the
+ # login-time breach screen for the new password.
+ assert update_kwargs["data"]["password_reset_required"] is False
+ assert update_kwargs["data"]["last_breach_check_at"] is None
+
+
+@pytest.mark.asyncio
+async def test_change_password_rejects_wrong_current_password():
+ from litellm.proxy._types import ChangePasswordRequest
+
+ prisma = _make_prisma(_make_user_row(hash_password(CURRENT_PASSWORD)))
+
+ with (
+ patch( # test-quality-ok: change_password reads proxy_server module globals; no injection seam
+ "litellm.proxy.proxy_server.prisma_client", prisma
+ ),
+ patch( # test-quality-ok: change_password reads proxy_server module globals; no injection seam
+ "litellm.proxy.proxy_server.general_settings", _POLICY_NO_BREACH_CHECK
+ ),
+ ):
+ with pytest.raises(HTTPException) as exc_info:
+ await change_password(
+ data=ChangePasswordRequest(current_password="not-the-password", new_password=NEW_PASSWORD),
+ user_api_key_dict=_caller(),
+ )
+
+ assert exc_info.value.status_code == 400
+ assert "Current password is incorrect" in exc_info.value.detail["error"]
+ prisma.db.litellm_usertable.update.assert_not_called()
+
+
+@pytest.mark.asyncio
+async def test_change_password_rejects_session_without_user():
+ from litellm.proxy._types import ChangePasswordRequest
+
+ prisma = _make_prisma(user=None)
+
+ with (
+ patch( # test-quality-ok: change_password reads proxy_server module globals; no injection seam
+ "litellm.proxy.proxy_server.prisma_client", prisma
+ ),
+ patch( # test-quality-ok: change_password reads proxy_server module globals; no injection seam
+ "litellm.proxy.proxy_server.general_settings", _POLICY_NO_BREACH_CHECK
+ ),
+ ):
+ with pytest.raises(HTTPException) as exc_info:
+ await change_password(
+ data=ChangePasswordRequest(current_password=CURRENT_PASSWORD, new_password=NEW_PASSWORD),
+ user_api_key_dict=_caller(user_id=None),
+ )
+
+ assert exc_info.value.status_code == 400
+ prisma.db.litellm_usertable.find_first.assert_not_called()
+ prisma.db.litellm_usertable.update.assert_not_called()
+
+
+@pytest.mark.asyncio
+async def test_change_password_rejects_account_without_password():
+ """SSO users and the env-credential admin have no DB password row to change."""
+ from litellm.proxy._types import ChangePasswordRequest
+
+ prisma = _make_prisma(_make_user_row(password=None))
+
+ with (
+ patch( # test-quality-ok: change_password reads proxy_server module globals; no injection seam
+ "litellm.proxy.proxy_server.prisma_client", prisma
+ ),
+ patch( # test-quality-ok: change_password reads proxy_server module globals; no injection seam
+ "litellm.proxy.proxy_server.general_settings", _POLICY_NO_BREACH_CHECK
+ ),
+ ):
+ with pytest.raises(HTTPException) as exc_info:
+ await change_password(
+ data=ChangePasswordRequest(current_password=CURRENT_PASSWORD, new_password=NEW_PASSWORD),
+ user_api_key_dict=_caller(),
+ )
+
+ assert exc_info.value.status_code == 400
+ assert "no password set" in exc_info.value.detail["error"]
+ prisma.db.litellm_usertable.update.assert_not_called()
+
+
+@pytest.mark.asyncio
+async def test_change_password_enforces_min_length():
+ from litellm.proxy._types import ChangePasswordRequest
+
+ prisma = _make_prisma(_make_user_row(hash_password(CURRENT_PASSWORD)))
+
+ with (
+ patch( # test-quality-ok: change_password reads proxy_server module globals; no injection seam
+ "litellm.proxy.proxy_server.prisma_client", prisma
+ ),
+ patch( # test-quality-ok: change_password reads proxy_server module globals; no injection seam
+ "litellm.proxy.proxy_server.general_settings", _POLICY_NO_BREACH_CHECK
+ ),
+ ):
+ with pytest.raises(ProxyException) as exc_info:
+ await change_password(
+ data=ChangePasswordRequest(current_password=CURRENT_PASSWORD, new_password="Short1!"),
+ user_api_key_dict=_caller(),
+ )
+
+ assert exc_info.value.code == "400"
+ assert exc_info.value.type == ProxyErrorTypes.validation_error
+ assert exc_info.value.param == "password"
+ assert "at least 12 characters" in exc_info.value.message
+ prisma.db.litellm_usertable.update.assert_not_called()
+
+
+@pytest.mark.asyncio
+@respx.mock
+async def test_change_password_rejects_breached_password():
+ """With the default policy, the new password is screened against HIBP."""
+ from litellm.proxy._types import ChangePasswordRequest
+
+ breached_password = "Password123!"
+ respx.get(_hibp_url_for(breached_password)).mock(
+ return_value=httpx.Response(200, text=f"{_hibp_suffix_for(breached_password)}:1")
+ )
+ prisma = _make_prisma(_make_user_row(hash_password(CURRENT_PASSWORD)))
+
+ with (
+ patch( # test-quality-ok: change_password reads proxy_server module globals; no injection seam
+ "litellm.proxy.proxy_server.prisma_client", prisma
+ ),
+ patch( # test-quality-ok: change_password reads proxy_server module globals; no injection seam
+ "litellm.proxy.proxy_server.general_settings", {}
+ ),
+ ):
+ with pytest.raises(ProxyException) as exc_info:
+ await change_password(
+ data=ChangePasswordRequest(current_password=CURRENT_PASSWORD, new_password=breached_password),
+ user_api_key_dict=_caller(),
+ )
+
+ assert exc_info.value.code == "400"
+ assert exc_info.value.type == ProxyErrorTypes.validation_error
+ assert exc_info.value.param == "password"
+ assert "data breaches" in exc_info.value.message
+ prisma.db.litellm_usertable.update.assert_not_called()
+
+
+@pytest.mark.asyncio
+@respx.mock
+async def test_change_password_verifies_current_password_before_hibp_lookup():
+ """A caller who fails current-password verification must not trigger any
+ HIBP traffic. The HIBP check fails open on errors, so an unmocked lookup
+ could not prove ordering; instead the route is registered and asserted
+ uncalled."""
+ from litellm.proxy._types import ChangePasswordRequest
+
+ hibp_route = respx.get(_hibp_url_for(NEW_PASSWORD)).mock(return_value=httpx.Response(200, text=""))
+ prisma = _make_prisma(_make_user_row(hash_password(CURRENT_PASSWORD)))
+
+ with (
+ patch( # test-quality-ok: change_password reads proxy_server module globals; no injection seam
+ "litellm.proxy.proxy_server.prisma_client", prisma
+ ),
+ patch( # test-quality-ok: change_password reads proxy_server module globals; no injection seam
+ "litellm.proxy.proxy_server.general_settings", {}
+ ),
+ ):
+ with pytest.raises(HTTPException) as exc_info:
+ await change_password(
+ data=ChangePasswordRequest(current_password="not-the-password", new_password=NEW_PASSWORD),
+ user_api_key_dict=_caller(),
+ )
+
+ assert exc_info.value.status_code == 400
+ assert "Current password is incorrect" in exc_info.value.detail["error"]
+ assert not hibp_route.called
+ prisma.db.litellm_usertable.update.assert_not_called()
+
+
+@pytest.mark.asyncio
+async def test_change_password_success_emits_redacted_audit_log():
+ """A successful change must land in the audit trail as field names only;
+ the plaintext passwords must never reach the audit call."""
+ from litellm.proxy._types import ChangePasswordRequest
+
+ prisma = _make_prisma(_make_user_row(hash_password(CURRENT_PASSWORD)))
+ audit_mock = AsyncMock()
+
+ with (
+ patch( # test-quality-ok: change_password reads proxy_server module globals; no injection seam
+ "litellm.proxy.proxy_server.prisma_client", prisma
+ ),
+ patch( # test-quality-ok: change_password reads proxy_server module globals; no injection seam
+ "litellm.proxy.proxy_server.general_settings", _POLICY_NO_BREACH_CHECK
+ ),
+ patch( # test-quality-ok: audit sink is a module-level import; no injection seam
+ "litellm.proxy.management_endpoints.password_endpoints.create_object_audit_log", audit_mock
+ ),
+ ):
+ await change_password(
+ data=ChangePasswordRequest(current_password=CURRENT_PASSWORD, new_password=NEW_PASSWORD),
+ user_api_key_dict=_caller(),
+ )
+
+ audit_mock.assert_awaited_once()
+ audit_kwargs = audit_mock.await_args.kwargs
+ assert audit_kwargs["object_id"] == "user-123"
+ assert audit_kwargs["action"] == "updated"
+ assert audit_kwargs["table_name"] == LitellmTableNames.USER_TABLE_NAME
+ assert audit_kwargs["after_value"] == '{"fields_changed": ["password"]}'
+ assert CURRENT_PASSWORD not in str(audit_kwargs)
+ assert NEW_PASSWORD not in str(audit_kwargs)
+
+
+@pytest.mark.asyncio
+async def test_change_password_failure_emits_no_audit_log():
+ from litellm.proxy._types import ChangePasswordRequest
+
+ prisma = _make_prisma(_make_user_row(hash_password(CURRENT_PASSWORD)))
+ audit_mock = AsyncMock()
+
+ with (
+ patch( # test-quality-ok: change_password reads proxy_server module globals; no injection seam
+ "litellm.proxy.proxy_server.prisma_client", prisma
+ ),
+ patch( # test-quality-ok: change_password reads proxy_server module globals; no injection seam
+ "litellm.proxy.proxy_server.general_settings", _POLICY_NO_BREACH_CHECK
+ ),
+ patch( # test-quality-ok: audit sink is a module-level import; no injection seam
+ "litellm.proxy.management_endpoints.password_endpoints.create_object_audit_log", audit_mock
+ ),
+ ):
+ with pytest.raises(HTTPException):
+ await change_password(
+ data=ChangePasswordRequest(current_password="not-the-password", new_password=NEW_PASSWORD),
+ user_api_key_dict=_caller(),
+ )
+
+ audit_mock.assert_not_awaited()
+
+
+@pytest.mark.asyncio
+async def test_change_password_requires_db():
+ from litellm.proxy._types import ChangePasswordRequest
+
+ with (
+ patch( # test-quality-ok: change_password reads proxy_server module globals; no injection seam
+ "litellm.proxy.proxy_server.prisma_client", None
+ ),
+ patch( # test-quality-ok: change_password reads proxy_server module globals; no injection seam
+ "litellm.proxy.proxy_server.general_settings", _POLICY_NO_BREACH_CHECK
+ ),
+ ):
+ with pytest.raises(HTTPException) as exc_info:
+ await change_password(
+ data=ChangePasswordRequest(current_password=CURRENT_PASSWORD, new_password=NEW_PASSWORD),
+ user_api_key_dict=_caller(),
+ )
+
+ assert exc_info.value.status_code == 500
diff --git a/tests/test_litellm/proxy/test__types.py b/tests/test_litellm/proxy/test__types.py
index 9e1486ce90f..f5abe0561db 100644
--- a/tests/test_litellm/proxy/test__types.py
+++ b/tests/test_litellm/proxy/test__types.py
@@ -5,11 +5,13 @@ from pydantic import ValidationError
from litellm.proxy._types import (
ROLES_WITHIN_ORG,
+ ChangePasswordRequest,
GenerateKeyRequest,
KeyRequest,
LiteLLM_AuditLogs,
LiteLLM_TeamMembership,
LitellmUserRoles,
+ NewUserRequest,
OrganizationMemberUpdateRequest,
ResetSpendRequest,
UpdateKeyRequest,
@@ -335,3 +337,43 @@ def test_virtual_key_mapping_counts_as_configured_when_any_issuer_sets_the_claim
)
assert jwt_auth.is_virtual_key_mapping_configured() is is_configured
+
+
+def test_new_user_request_loudly_rejects_a_password():
+ """
+ /user/new has never persisted a password (the field used to be silently
+ dropped). Sending one must now fail visibly so the dead path cannot be
+ revived without going through the password policy.
+ """
+ with pytest.raises(ValidationError, match="invitation link"):
+ NewUserRequest(user_email="alice@example.com", password="hunter2hunter2")
+
+
+def test_new_user_request_without_password_still_works():
+ request = NewUserRequest(user_email="alice@example.com")
+ assert request.password is None
+
+
+def test_update_user_request_accepts_a_password():
+ """Admins set user passwords through /user/update; the value must survive
+ model validation so the endpoint can policy-check and hash it."""
+ request = UpdateUserRequest(user_id="user-123", password="hunter2hunter2")
+ assert request.password == "hunter2hunter2"
+
+
+def test_update_user_request_password_hidden_from_repr():
+ """management_endpoint_wrapper string-formats endpoint kwargs into Slack
+ alerts, so the model's repr/str must never contain the plaintext password."""
+ request = UpdateUserRequest(user_id="user-123", password="hunter2hunter2")
+ assert "hunter2hunter2" not in repr(request)
+ assert "hunter2hunter2" not in str(request)
+
+
+def test_change_password_request_passwords_hidden_from_repr():
+ """Any accidental str()/repr() of the request model (debug logs, exception
+ handlers, a future management_endpoint_wrapper) must never contain either
+ plaintext password."""
+ request = ChangePasswordRequest(current_password="hunter2hunter2", new_password="NewP@ssw0rd-2026")
+ for rendered in (repr(request), str(request)):
+ assert "hunter2hunter2" not in rendered
+ assert "NewP@ssw0rd-2026" not in rendered
diff --git a/tests/unit/models/test_models.py b/tests/unit/models/test_models.py
index b8bf55f1b4a..ab456bb1624 100644
--- a/tests/unit/models/test_models.py
+++ b/tests/unit/models/test_models.py
@@ -324,6 +324,8 @@ class TestUser:
assert user_no_models.has_model_access("any-model")
def test_password_hash_excluded_from_serialization(self):
+ import json
+
from litellm.proxy._types import LiteLLM_UserTableWithKeyCount
secret = "$2b$12$abcdefghijklmnopqrstuv"
@@ -331,12 +333,12 @@ class TestUser:
assert user.password == secret
assert "password" not in user.model_dump()
- assert "password" not in user.model_dump_json()
+ assert "password" not in json.loads(user.model_dump_json())
with_keys = LiteLLM_UserTableWithKeyCount(user_id="u1", user_email="a@b.c", password=secret, key_count=2)
assert with_keys.password == secret
assert "password" not in with_keys.model_dump()
- assert "password" not in with_keys.model_dump_json()
+ assert "password" not in json.loads(with_keys.model_dump_json())
class TestVerificationToken:
diff --git a/ui/litellm-dashboard/src/app/(dashboard)/change-password/ChangePasswordForm.integration.test.tsx b/ui/litellm-dashboard/src/app/(dashboard)/change-password/ChangePasswordForm.integration.test.tsx
new file mode 100644
index 00000000000..c4cf8ebcf9d
--- /dev/null
+++ b/ui/litellm-dashboard/src/app/(dashboard)/change-password/ChangePasswordForm.integration.test.tsx
@@ -0,0 +1,110 @@
+import { fireEvent, render, screen, waitFor } from "@testing-library/react";
+import { beforeEach, describe, expect, it, vi } from "vitest";
+import ChangePasswordForm from "./ChangePasswordForm";
+
+const mockChangePasswordCall = vi.fn();
+const mockToastSuccess = vi.fn();
+const mockClearTokenCookies = vi.fn();
+let mockPasswordResetRequired = false;
+
+vi.mock("@/components/networking", () => ({
+ changePasswordCall: (...args: unknown[]) => mockChangePasswordCall(...args),
+ getProxyBaseUrl: () => "",
+}));
+
+vi.mock("@/app/(dashboard)/hooks/useAuthorized", () => ({
+ default: () => ({ accessToken: "sk-session-token", passwordResetRequired: mockPasswordResetRequired }),
+}));
+
+vi.mock("@/lib/toast", () => ({
+ toast: {
+ success: (...args: unknown[]) => mockToastSuccess(...args),
+ fromError: vi.fn(),
+ },
+}));
+
+vi.mock("@/utils/cookieUtils", () => ({
+ clearTokenCookies: (...args: unknown[]) => mockClearTokenCookies(...args),
+}));
+
+const fillForm = (values: { current: string; next: string; confirm: string }) => {
+ fireEvent.change(screen.getByLabelText("Current Password"), { target: { value: values.current } });
+ fireEvent.change(screen.getByLabelText("New Password"), { target: { value: values.next } });
+ fireEvent.change(screen.getByLabelText("Confirm New Password"), { target: { value: values.confirm } });
+};
+
+const submit = () => fireEvent.click(screen.getByRole("button", { name: "Change Password" }));
+
+describe("ChangePasswordForm", () => {
+ beforeEach(() => {
+ vi.clearAllMocks();
+ mockPasswordResetRequired = false;
+ });
+
+ it("sends the current and new password to the change endpoint and resets on success", async () => {
+ mockChangePasswordCall.mockResolvedValue({ user_id: "user-123", message: "Password updated successfully." });
+ render( );
+
+ fillForm({ current: "OldP@ssw0rd-2026", next: "NewP@ssw0rd-2026", confirm: "NewP@ssw0rd-2026" });
+ submit();
+
+ expect(await screen.findByLabelText("Current Password")).toHaveValue("");
+ expect(mockChangePasswordCall).toHaveBeenCalledWith("sk-session-token", "OldP@ssw0rd-2026", "NewP@ssw0rd-2026");
+ expect(mockToastSuccess).toHaveBeenCalled();
+ });
+
+ it("blocks submission when the confirmation does not match", async () => {
+ render( );
+
+ fillForm({ current: "OldP@ssw0rd-2026", next: "NewP@ssw0rd-2026", confirm: "Different-2026" });
+ submit();
+
+ expect(await screen.findByText("New passwords do not match")).toBeInTheDocument();
+ expect(mockChangePasswordCall).not.toHaveBeenCalled();
+ });
+
+ it("shows the proxy's rejection message unwrapped", async () => {
+ mockChangePasswordCall.mockRejectedValue(new Error("{'error': 'Current password is incorrect.'}"));
+ render( );
+
+ fillForm({ current: "wrong-password", next: "NewP@ssw0rd-2026", confirm: "NewP@ssw0rd-2026" });
+ submit();
+
+ expect(await screen.findByText("Current password is incorrect.")).toBeInTheDocument();
+ expect(mockToastSuccess).not.toHaveBeenCalled();
+ });
+
+ describe("forced password reset", () => {
+ it("shows the forced-reset warning only when the session is flagged", () => {
+ mockPasswordResetRequired = true;
+ render( );
+
+ expect(screen.getByText(/must be changed before you can use the dashboard/)).toBeInTheDocument();
+ });
+
+ it("hides the forced-reset warning for a normal session", () => {
+ render( );
+
+ expect(screen.queryByText(/must be changed before you can use the dashboard/)).not.toBeInTheDocument();
+ });
+
+ it("signs the user out to re-login after a successful forced change", async () => {
+ mockPasswordResetRequired = true;
+ mockChangePasswordCall.mockResolvedValue({ user_id: "user-123", message: "Password updated successfully." });
+ const replaceMock = vi.fn();
+ const realLocation = window.location;
+ Object.defineProperty(window, "location", { configurable: true, value: { replace: replaceMock } });
+
+ try {
+ render( );
+ fillForm({ current: "OldP@ssw0rd-2026", next: "NewP@ssw0rd-2026", confirm: "NewP@ssw0rd-2026" });
+ submit();
+
+ await waitFor(() => expect(replaceMock).toHaveBeenCalledWith("/ui/login/"));
+ expect(mockClearTokenCookies).toHaveBeenCalled();
+ } finally {
+ Object.defineProperty(window, "location", { configurable: true, value: realLocation });
+ }
+ });
+ });
+});
diff --git a/ui/litellm-dashboard/src/app/(dashboard)/change-password/ChangePasswordForm.tsx b/ui/litellm-dashboard/src/app/(dashboard)/change-password/ChangePasswordForm.tsx
new file mode 100644
index 00000000000..05a6bf3ae94
--- /dev/null
+++ b/ui/litellm-dashboard/src/app/(dashboard)/change-password/ChangePasswordForm.tsx
@@ -0,0 +1,120 @@
+"use client";
+
+import React, { useState } from "react";
+import { CircleAlert } from "lucide-react";
+import { z } from "zod/v4";
+import useAuthorized from "@/app/(dashboard)/hooks/useAuthorized";
+import { Alert, AlertTitle } from "@/components/shared/Alert";
+import { PasswordInput } from "@/components/shared/PasswordInput";
+import { FormField } from "@/components/shared/form/FormField";
+import { Button } from "@/components/ui/button";
+import { Card, CardContent } from "@/components/ui/card";
+import { FieldGroup } from "@/components/ui/field";
+import { UiLoadingSpinner } from "@/components/ui/ui-loading-spinner";
+import { changePasswordCall, getProxyBaseUrl } from "@/components/networking";
+import { extractProxyErrorMessage } from "@/lib/http/client";
+import { useZodForm } from "@/lib/forms/useZodForm";
+import { toast } from "@/lib/toast";
+import { clearTokenCookies } from "@/utils/cookieUtils";
+import { getLoginUrl } from "@/utils/returnUrlUtils";
+
+const changePasswordSchema = z
+ .object({
+ currentPassword: z.string().min(1, "Current password is required"),
+ newPassword: z.string().min(1, "New password is required"),
+ confirmNewPassword: z.string().min(1, "Confirm your new password"),
+ })
+ .refine((values) => values.newPassword === values.confirmNewPassword, {
+ message: "New passwords do not match",
+ path: ["confirmNewPassword"],
+ });
+
+type ChangePasswordValues = z.infer;
+
+export function ChangePasswordForm() {
+ const { accessToken, passwordResetRequired } = useAuthorized();
+ const form = useZodForm(changePasswordSchema, {
+ defaultValues: { currentPassword: "", newPassword: "", confirmNewPassword: "" },
+ });
+ const [isPending, setIsPending] = useState(false);
+ const [submitError, setSubmitError] = useState(null);
+
+ const handleSubmit = async (values: ChangePasswordValues) => {
+ if (!accessToken) return;
+ setSubmitError(null);
+ setIsPending(true);
+ try {
+ await changePasswordCall(accessToken, values.currentPassword, values.newPassword);
+ if (passwordResetRequired) {
+ // The session key was minted restricted; only a fresh login lifts it.
+ toast.success("Password updated. Please log in with your new password.");
+ clearTokenCookies();
+ window.location.replace(getLoginUrl(getProxyBaseUrl()));
+ return;
+ }
+ toast.success("Password updated");
+ form.reset();
+ } catch (error) {
+ setSubmitError(extractProxyErrorMessage(error));
+ } finally {
+ setIsPending(false);
+ }
+ };
+
+ return (
+
+
+
+ Change Password
+
+ Enter your current password and choose a new one. The new password must meet this proxy's password
+ policy.
+
+
+ {passwordResetRequired && (
+
+
+
+ Your password must be changed before you can use the dashboard: it was either found in a known data
+ breach or set by an administrator as a temporary password. After updating it, you will be signed out to
+ log in again.
+
+
+ )}
+
+
+
+
+
+ );
+}
+
+export default ChangePasswordForm;
diff --git a/ui/litellm-dashboard/src/app/(dashboard)/change-password/page.tsx b/ui/litellm-dashboard/src/app/(dashboard)/change-password/page.tsx
new file mode 100644
index 00000000000..0a6ae926ceb
--- /dev/null
+++ b/ui/litellm-dashboard/src/app/(dashboard)/change-password/page.tsx
@@ -0,0 +1,7 @@
+"use client";
+
+import ChangePasswordForm from "./ChangePasswordForm";
+
+export default function ChangePasswordPage() {
+ return ;
+}
diff --git a/ui/litellm-dashboard/src/app/(dashboard)/hooks/useAuthorized.ts b/ui/litellm-dashboard/src/app/(dashboard)/hooks/useAuthorized.ts
index 40d1ec09d1f..581ee8b2580 100644
--- a/ui/litellm-dashboard/src/app/(dashboard)/hooks/useAuthorized.ts
+++ b/ui/litellm-dashboard/src/app/(dashboard)/hooks/useAuthorized.ts
@@ -50,7 +50,9 @@ const useAuthorized = () => {
isViewOnly: isViewOnlySessionRole(decoded?.user_role),
premiumUser: decoded?.premium_user ?? null,
disabledPersonalKeyCreation: decoded?.disabled_non_admin_personal_key_creation ?? null,
+ loginMethod: decoded?.login_method ?? null,
showSSOBanner: decoded?.login_method === "username_password",
+ passwordResetRequired: decoded?.password_reset_required === true,
};
};
diff --git a/ui/litellm-dashboard/src/app/(dashboard)/layout.test.tsx b/ui/litellm-dashboard/src/app/(dashboard)/layout.test.tsx
index 3fe34610260..ae6575e2349 100644
--- a/ui/litellm-dashboard/src/app/(dashboard)/layout.test.tsx
+++ b/ui/litellm-dashboard/src/app/(dashboard)/layout.test.tsx
@@ -1,4 +1,4 @@
-import { describe, it, expect, vi, beforeEach } from "vitest";
+import { describe, it, expect, vi, beforeEach, afterEach } from "vitest";
import { render, screen, waitFor } from "@testing-library/react";
import { AuthProvider } from "@/contexts/AuthContext";
import Layout from "./layout";
@@ -117,4 +117,60 @@ describe("(dashboard) Layout", () => {
expect(screen.queryByTestId("dashboard-header")).not.toBeInTheDocument();
expect(screen.queryByTestId("sidebar")).not.toBeInTheDocument();
});
+
+ describe("forced password reset routing", () => {
+ const sessionCookie = (claims: Record) => {
+ const encode = (part: Record) =>
+ btoa(JSON.stringify(part)).replace(/\+/g, "-").replace(/\//g, "_").replace(/=+$/, "");
+ const exp = Math.floor(Date.now() / 1000) + 3600;
+ return `${encode({ alg: "HS256", typ: "JWT" })}.${encode({ ...claims, exp })}.sig`;
+ };
+
+ afterEach(() => {
+ document.cookie = "token=; Max-Age=0; Path=/";
+ });
+
+ it("routes a session flagged password_reset_required to the change-password page", async () => {
+ const flaggedClaims = {
+ user_id: "flagged-user",
+ key: "sk-session",
+ login_method: "username_password",
+ password_reset_required: true,
+ };
+ document.cookie = `token=${sessionCookie(flaggedClaims)}; Path=/`;
+
+ render(
+
+
+
+
+ ,
+ );
+
+ pendingUiConfig.resolve();
+
+ await waitFor(() => expect(replaceMock).toHaveBeenCalledWith(expect.stringContaining("/change-password")));
+ });
+
+ it("does not reroute an unflagged session", async () => {
+ document.cookie = `token=${sessionCookie({
+ user_id: "normal-user",
+ key: "sk-session",
+ login_method: "username_password",
+ })}; Path=/`;
+
+ render(
+
+
+
+
+ ,
+ );
+
+ pendingUiConfig.resolve();
+
+ expect(await screen.findByTestId("page-content")).toBeInTheDocument();
+ expect(replaceMock).not.toHaveBeenCalled();
+ });
+ });
});
diff --git a/ui/litellm-dashboard/src/app/(dashboard)/layout.tsx b/ui/litellm-dashboard/src/app/(dashboard)/layout.tsx
index fa6df7f176a..309d7dd1ac2 100644
--- a/ui/litellm-dashboard/src/app/(dashboard)/layout.tsx
+++ b/ui/litellm-dashboard/src/app/(dashboard)/layout.tsx
@@ -7,7 +7,7 @@ import LoadingScreen from "@/components/common_components/LoadingScreen";
import { ThemeProvider } from "@/contexts/ThemeContext";
import { useAuth } from "@/contexts/AuthContext";
import SidebarProvider from "@/app/(dashboard)/components/SidebarProvider";
-import { useRouter, useSearchParams } from "next/navigation";
+import { usePathname, useRouter, useSearchParams } from "next/navigation";
import { DebugWarningBanner } from "@/components/DebugWarningBanner";
import { NoRedisWarningBanner } from "@/components/NoRedisWarningBanner";
import { EnvCredentialLoginWarningBanner } from "@/components/EnvCredentialLoginWarningBanner";
@@ -146,7 +146,8 @@ function DashboardShell({ children }: { children: React.ReactNode }) {
function LayoutContent({ children }: { children: React.ReactNode }) {
const router = useRouter();
const searchParams = useSearchParams();
- const { accessToken, authLoading } = useAuth();
+ const pathname = usePathname();
+ const { accessToken, authLoading, passwordResetRequired } = useAuth();
const isInvitationFlow = Boolean(searchParams.get("invitation_id"));
// Legacy invitation links point at /ui/?invitation_id=; the onboarding form now lives at its own
@@ -157,6 +158,14 @@ function LayoutContent({ children }: { children: React.ReactNode }) {
}
}, [authLoading, isInvitationFlow, router, searchParams]);
+ // A session flagged for a forced password reset can only reach the change-password
+ // endpoint server-side; keep the UI on the matching page.
+ useEffect(() => {
+ if (!authLoading && passwordResetRequired && !pathname?.endsWith("/change-password")) {
+ router.replace(uiHref("change-password"));
+ }
+ }, [authLoading, passwordResetRequired, pathname, router]);
+
if (authLoading || isInvitationFlow) {
return ;
}
diff --git a/ui/litellm-dashboard/src/components/Navbar/UserDropdown/UserDropdown.test.tsx b/ui/litellm-dashboard/src/components/Navbar/UserDropdown/UserDropdown.test.tsx
index cad5ced340e..4bdf0da3407 100644
--- a/ui/litellm-dashboard/src/components/Navbar/UserDropdown/UserDropdown.test.tsx
+++ b/ui/litellm-dashboard/src/components/Navbar/UserDropdown/UserDropdown.test.tsx
@@ -3,13 +3,25 @@ import { beforeEach, describe, expect, it, vi } from "vitest";
import { renderWithProviders, screen, waitFor } from "../../../../tests/test-utils";
import UserDropdown from "./UserDropdown";
-let mockUseAuthorizedImpl = () => ({
+let mockUseAuthorizedImpl: () => {
+ userId: string | null;
+ userEmail: string | null;
+ userRoleLabel: string;
+ premiumUser: boolean;
+ loginMethod?: string | null;
+} = () => ({
userId: "test-user-id",
userEmail: "test@example.com",
userRoleLabel: "Admin",
premiumUser: false,
});
+const mockRouterPush = vi.fn();
+
+vi.mock("next/navigation", () => ({
+ useRouter: () => ({ push: mockRouterPush }),
+}));
+
let mockUseDisableShowPromptsImpl = () => false;
let mockGetLocalStorageItemImpl = (key: string): string | null => {
@@ -143,6 +155,44 @@ describe("UserDropdown", () => {
expect(mockOnLogout).toHaveBeenCalledTimes(1);
});
+ it("should navigate to the change-password page for username/password sessions", async () => {
+ mockUseAuthorizedImpl = () => ({
+ userId: "test-user-id",
+ userEmail: "test@example.com",
+ userRoleLabel: "Admin",
+ premiumUser: false,
+ loginMethod: "username_password",
+ });
+ const user = userEvent.setup();
+ renderWithProviders( );
+
+ await user.click(getAccountTrigger());
+
+ await user.click(await screen.findByText("Change Password"));
+
+ expect(mockRouterPush).toHaveBeenCalledWith(expect.stringContaining("change-password"));
+ });
+
+ it("should hide the change-password entry for SSO sessions", async () => {
+ mockUseAuthorizedImpl = () => ({
+ userId: "test-user-id",
+ userEmail: "test@example.com",
+ userRoleLabel: "Admin",
+ premiumUser: false,
+ loginMethod: "sso",
+ });
+ const user = userEvent.setup();
+ renderWithProviders( );
+
+ await user.click(getAccountTrigger());
+
+ await waitFor(() => {
+ expect(screen.getAllByText("test@example.com").length).toBeGreaterThan(0);
+ });
+
+ expect(screen.queryByText("Change Password")).not.toBeInTheDocument();
+ });
+
it("should toggle hide new feature indicators switch", async () => {
const user = userEvent.setup();
renderWithProviders( );
diff --git a/ui/litellm-dashboard/src/components/Navbar/UserDropdown/UserDropdown.tsx b/ui/litellm-dashboard/src/components/Navbar/UserDropdown/UserDropdown.tsx
index 95f76dbb2cc..9c02defc778 100644
--- a/ui/litellm-dashboard/src/components/Navbar/UserDropdown/UserDropdown.tsx
+++ b/ui/litellm-dashboard/src/components/Navbar/UserDropdown/UserDropdown.tsx
@@ -9,7 +9,9 @@ import {
setLocalStorageItem,
} from "@/utils/localStorageUtils";
import { navAccountDisplayName } from "@/components/Navbar/navDisplayName";
-import { ChevronDown, ChevronsUpDown, Crown, LogOut, Mail, ShieldCheck, User } from "lucide-react";
+import { uiHref } from "@/utils/uiHref";
+import { ChevronDown, ChevronsUpDown, Crown, KeyRound, LogOut, Mail, ShieldCheck, User } from "lucide-react";
+import { useRouter } from "next/navigation";
import { Avatar, AvatarFallback } from "@/components/ui/avatar";
import { Badge } from "@/components/ui/badge";
import { Popover, PopoverContent, PopoverTrigger } from "@/components/ui/popover";
@@ -63,7 +65,9 @@ interface UserDropdownProps {
}
const UserDropdown: React.FC = ({ onLogout, variant = "navbar", collapsed = false }) => {
- const { userId, userEmail, userRoleLabel: userRole, premiumUser } = useAuthorized();
+ const { userId, userEmail, userRoleLabel: userRole, premiumUser, loginMethod } = useAuthorized();
+ const router = useRouter();
+ const [open, setOpen] = useState(false);
const disableShowPrompts = useDisableShowPrompts();
const disableBlogPosts = useDisableBlogPosts();
const disableBouncingIcon = useDisableBouncingIcon();
@@ -197,7 +201,7 @@ const UserDropdown: React.FC = ({ onLogout, variant = "navbar
const displayName = navAccountDisplayName(userEmail, userId);
return (
-
+
{variant === "sidebar" ? (
= ({ onLogout, variant = "navbar
>
{renderUserInfoSection()}
+ {loginMethod === "username_password" && (
+ {
+ setOpen(false);
+ router.push(uiHref("change-password"));
+ }}
+ className="flex w-full items-center gap-2 rounded-sm px-2 py-1.5 text-sm hover:bg-accent"
+ >
+
+ Change Password
+
+ )}
AuthMock = () => ({
@@ -19,6 +20,12 @@ let mockUseAuthorizedImpl: () => AuthMock = () => ({
accessToken: "test-token",
});
+const mockRouterPush = vi.fn();
+
+vi.mock("next/navigation", () => ({
+ useRouter: () => ({ push: mockRouterPush }),
+}));
+
let mockUseDisableShowPromptsImpl = () => false;
let mockUseDisableBouncingIconImpl = () => false;
let mockHealthDataImpl = (): { litellm_version?: string } | undefined => ({ litellm_version: "1.99.0" });
@@ -201,6 +208,42 @@ describe("SidebarAccountMenu", () => {
expect(mockOnLogout).toHaveBeenCalledTimes(1);
});
+ it("should navigate to the change-password page for username/password sessions", async () => {
+ mockUseAuthorizedImpl = () => ({
+ userId: "test-user-id",
+ userEmail: "test@example.com",
+ userRoleLabel: "Admin",
+ premiumUser: false,
+ accessToken: "test-token",
+ loginMethod: "username_password",
+ });
+ const user = userEvent.setup();
+ renderWithProviders( );
+
+ await openMenu(user);
+
+ await user.click(screen.getByRole("button", { name: /change password/i }));
+
+ expect(mockRouterPush).toHaveBeenCalledWith(expect.stringContaining("change-password"));
+ });
+
+ it("should hide the change-password entry for SSO sessions", async () => {
+ mockUseAuthorizedImpl = () => ({
+ userId: "test-user-id",
+ userEmail: "test@example.com",
+ userRoleLabel: "Admin",
+ premiumUser: false,
+ accessToken: "test-token",
+ loginMethod: "sso",
+ });
+ const user = userEvent.setup();
+ renderWithProviders( );
+
+ await openMenu(user);
+
+ expect(screen.queryByRole("button", { name: /change password/i })).not.toBeInTheDocument();
+ });
+
it("should toggle hide new feature indicators on", async () => {
const user = userEvent.setup();
renderWithProviders( );
diff --git a/ui/litellm-dashboard/src/components/SidebarAccountMenu/SidebarAccountMenu.tsx b/ui/litellm-dashboard/src/components/SidebarAccountMenu/SidebarAccountMenu.tsx
index ea0b82869cd..e697cc6f34a 100644
--- a/ui/litellm-dashboard/src/components/SidebarAccountMenu/SidebarAccountMenu.tsx
+++ b/ui/litellm-dashboard/src/components/SidebarAccountMenu/SidebarAccountMenu.tsx
@@ -14,7 +14,9 @@ import { Popover, PopoverContent, PopoverTrigger } from "@/components/ui/popover
import { Separator } from "@/components/ui/separator";
import { Switch } from "@/components/ui/switch";
import { cn } from "@/lib/cva.config";
-import { ChevronsUpDown, Crown, IdCard, LogOut, Mail, ShieldCheck } from "lucide-react";
+import { uiHref } from "@/utils/uiHref";
+import { ChevronsUpDown, Crown, IdCard, KeyRound, LogOut, Mail, ShieldCheck } from "lucide-react";
+import { useRouter } from "next/navigation";
import React from "react";
const RELEASE_NOTES_URL = "https://docs.litellm.ai/release_notes";
@@ -81,7 +83,9 @@ interface SidebarAccountMenuProps {
}
const SidebarAccountMenu: React.FC = ({ onLogout, collapsed = false }) => {
- const { userId, userEmail, userRoleLabel: userRole, premiumUser, accessToken } = useAuthorized();
+ const { userId, userEmail, userRoleLabel: userRole, premiumUser, accessToken, loginMethod } = useAuthorized();
+ const router = useRouter();
+ const [open, setOpen] = React.useState(false);
const { data: healthData } = useHealthReadinessDetails(accessToken);
const version = healthData?.litellm_version;
const disableShowPrompts = useDisableShowPrompts();
@@ -136,7 +140,7 @@ const SidebarAccountMenu: React.FC = ({ onLogout, colla
const triggerLabel = `Account menu — ${userRole ?? "Unknown role"} — signed in as ${userEmail || userId || "unknown"}`;
return (
-
+
= ({ onLogout, colla
+ {loginMethod === "username_password" && (
+ {
+ setOpen(false);
+ router.push(uiHref("change-password"));
+ }}
+ className="h-[42px] w-full justify-start gap-2.5 rounded-none px-3 text-sm font-medium text-foreground"
+ >
+
+ Change Password
+
+ )}
+
({ pathname: "/ui/api-keys" }));
vi.mock("next/navigation", () => ({
usePathname: () => navState.pathname,
+ useRouter: () => ({ push: vi.fn() }),
}));
const { mockUseAuthorized, mockUseOrganizations } = vi.hoisted(() => {
diff --git a/ui/litellm-dashboard/src/components/networking.tsx b/ui/litellm-dashboard/src/components/networking.tsx
index ab1203cf440..178bbeb947f 100644
--- a/ui/litellm-dashboard/src/components/networking.tsx
+++ b/ui/litellm-dashboard/src/components/networking.tsx
@@ -1595,6 +1595,20 @@ export const claimOnboardingToken = async (
}
};
+export const changePasswordCall = async (
+ accessToken: string,
+ currentPassword: string,
+ newPassword: string,
+): Promise<{ user_id: string; message: string }> => {
+ return await apiClient.post(`/user/password/change`, {
+ accessToken,
+ body: {
+ current_password: currentPassword,
+ new_password: newPassword,
+ },
+ });
+};
+
export const regenerateKeyCall = async (accessToken: string, keyToRegenerate: string, formData: any) => {
try {
const url = proxyBaseUrl
diff --git a/ui/litellm-dashboard/src/contexts/AuthContext.tsx b/ui/litellm-dashboard/src/contexts/AuthContext.tsx
index 123feb18a6c..c68c8b9d81a 100644
--- a/ui/litellm-dashboard/src/contexts/AuthContext.tsx
+++ b/ui/litellm-dashboard/src/contexts/AuthContext.tsx
@@ -24,6 +24,7 @@ type AuthContextValue = {
premiumUser: boolean;
disabledPersonalKeyCreation: boolean;
showSSOBanner: boolean;
+ passwordResetRequired: boolean;
setToken: React.Dispatch>;
setUserID: React.Dispatch>;
@@ -46,6 +47,7 @@ export function AuthProvider({ children }: { children: React.ReactNode }) {
const [premiumUser, setPremiumUser] = useState(false);
const [disabledPersonalKeyCreation, setDisabledPersonalKeyCreation] = useState(false);
const [showSSOBanner, setShowSSOBanner] = useState(true);
+ const [passwordResetRequired, setPasswordResetRequired] = useState(false);
// Load runtime UI config (populates proxyBaseUrl etc.) before clearing
// authLoading, so any consumer that builds proxy-rooted URLs from authLoading=false
@@ -124,6 +126,7 @@ export function AuthProvider({ children }: { children: React.ReactNode }) {
if (decoded.user_id) {
setUserID(decoded.user_id);
}
+ setPasswordResetRequired(decoded.password_reset_required === true);
}, [token]);
const value: AuthContextValue = {
@@ -136,6 +139,7 @@ export function AuthProvider({ children }: { children: React.ReactNode }) {
premiumUser,
disabledPersonalKeyCreation,
showSSOBanner,
+ passwordResetRequired,
setToken,
setUserID,
setUserRole,
diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts
index 29b7cd206eb..b12e5b68c49 100644
--- a/ui/litellm-dashboard/src/lib/http/schema.d.ts
+++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts
@@ -17312,6 +17312,7 @@ export interface paths {
* - prompts: Optional[List[str]] - List of allowed prompts for the user. If specified, the user will only be able to use these specific prompts.
* - organizations: List[str] - List of organization id's the user is a member of
* - budget_limits: Optional[list] - List of concurrent budget windows for the user. Each window specifies a budget_limit, time_period, and optional budget_duration. Example - [{"budget_limit": 10.0, "time_period": "1d"}, {"budget_limit": 50.0, "time_period": "7d"}].
+ * - password: Optional[str] - Not supported; any value is rejected with a 422. Users set their own password through an invitation link (POST /invitation/new).
* Returns:
* - key: (str) The generated api key for the user
* - expires: (datetime) Datetime object for when key expires.
@@ -17334,6 +17335,36 @@ export interface paths {
patch?: never;
trace?: never;
};
+ "/user/password/change": {
+ parameters: {
+ query?: never;
+ header?: never;
+ path?: never;
+ cookie?: never;
+ };
+ get?: never;
+ put?: never;
+ /**
+ * Change Password
+ * @description Change the calling user's own password.
+ *
+ * Requires the current password. The new password must satisfy the
+ * configured password policy (`general_settings.password_policy_*`: minimum
+ * length, character classes, and, when enabled, breached-password screening
+ * via haveibeenpwned.com). A successful change lifts any pending forced
+ * password reset (`password_reset_required`) on the account.
+ *
+ * Parameters:
+ * - current_password: str - The user's current password.
+ * - new_password: str - The password to change to.
+ */
+ post: operations["change_password_user_password_change_post"];
+ delete?: never;
+ options?: never;
+ head?: never;
+ patch?: never;
+ trace?: never;
+ };
"/user/spend/report": {
parameters: {
query?: never;
@@ -17381,7 +17412,7 @@ export interface paths {
* Parameters:
* - user_id: Optional[str] - Specify a user id. If not set, a unique id will be generated.
* - user_email: Optional[str] - Specify a user email.
- * - password: Optional[str] - Specify a user password.
+ * - password: Optional[str] - Set the user's password (admin only). Must satisfy the configured password policy. The user is required to change it at their next login. Users change their own password with POST /user/password/change.
* - user_alias: Optional[str] - A descriptive name for you to know who this user id refers to.
* - teams: Optional[list] - specify a list of team id's a user belongs to.
* - send_invite_email: Optional[bool] - Specify if an invite email should be sent.
@@ -25901,6 +25932,20 @@ export interface components {
*/
threshold_step: number;
};
+ /** ChangePasswordRequest */
+ ChangePasswordRequest: {
+ /** Current Password */
+ current_password: string;
+ /** New Password */
+ new_password: string;
+ };
+ /** ChangePasswordResponse */
+ ChangePasswordResponse: {
+ /** Message */
+ message: string;
+ /** User Id */
+ user_id: string;
+ };
/** ChatCompletionAnnotation */
ChatCompletionAnnotation: {
/**
@@ -31604,6 +31649,8 @@ export interface components {
budget_reset_at?: string | null;
/** Created At */
created_at?: string | null;
+ /** Last Breach Check At */
+ last_breach_check_at?: string | null;
/** Max Budget */
max_budget?: number | null;
/** Max Parallel Requests */
@@ -31638,6 +31685,8 @@ export interface components {
organization_id?: string | null;
/** Organization Memberships */
organization_memberships?: components["schemas"]["LiteLLM_OrganizationMembershipTable"][] | null;
+ /** Password Reset Required */
+ password_reset_required?: boolean | null;
/**
* Policies
* @default []
@@ -31697,6 +31746,8 @@ export interface components {
* @default 0
*/
key_count: number;
+ /** Last Breach Check At */
+ last_breach_check_at?: string | null;
/** Max Budget */
max_budget?: number | null;
/** Max Parallel Requests */
@@ -31731,6 +31782,8 @@ export interface components {
organization_id?: string | null;
/** Organization Memberships */
organization_memberships?: components["schemas"]["LiteLLM_OrganizationMembershipTable"][] | null;
+ /** Password Reset Required */
+ password_reset_required?: boolean | null;
/**
* Policies
* @default []
@@ -34327,6 +34380,8 @@ export interface components {
object_permission?: components["schemas"]["LiteLLM_ObjectPermissionBase"] | null;
/** Organizations */
organizations?: string[] | null;
+ /** Password */
+ password?: string | null;
/**
* Permissions
* @default {}
@@ -63569,6 +63624,39 @@ export interface operations {
};
};
};
+ change_password_user_password_change_post: {
+ parameters: {
+ query?: never;
+ header?: never;
+ path?: never;
+ cookie?: never;
+ };
+ requestBody: {
+ content: {
+ "application/json": components["schemas"]["ChangePasswordRequest"];
+ };
+ };
+ responses: {
+ /** @description Successful Response */
+ 200: {
+ headers: {
+ [name: string]: unknown;
+ };
+ content: {
+ "application/json": components["schemas"]["ChangePasswordResponse"];
+ };
+ };
+ /** @description Validation Error */
+ 422: {
+ headers: {
+ [name: string]: unknown;
+ };
+ content: {
+ "application/json": components["schemas"]["HTTPValidationError"];
+ };
+ };
+ };
+ };
get_user_spend_report_user_spend_report_get: {
parameters: {
query?: {
From 07c9355f4ea1650cad904932fa90fe42eb48ebb9 Mon Sep 17 00:00:00 2001
From: ryan
Date: Mon, 21 Sep 2026 18:49:31 +0000
Subject: [PATCH 042/160] fix(ui): regenerate schema.d.ts against main
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
---
ui/litellm-dashboard/src/lib/http/schema.d.ts | 2 ++
1 file changed, 2 insertions(+)
diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts
index b12e5b68c49..e2165bdc021 100644
--- a/ui/litellm-dashboard/src/lib/http/schema.d.ts
+++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts
@@ -25283,6 +25283,8 @@ export interface components {
object_permission?: components["schemas"]["LiteLLM_ObjectPermissionBase"] | null;
/** Organizations */
organizations?: string[] | null;
+ /** Password */
+ password?: string | null;
/**
* Permissions
* @default {}
From 6bfbc7dba4cbf70417ba98d7792bc18005d44486 Mon Sep 17 00:00:00 2001
From: ryan
Date: Mon, 21 Sep 2026 18:53:39 +0000
Subject: [PATCH 043/160] test(auth): pass throttle to authenticate_user in
reset-required tests
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
---
tests/test_litellm/proxy/auth/test_login_utils.py | 9 ++++++++-
1 file changed, 8 insertions(+), 1 deletion(-)
diff --git a/tests/test_litellm/proxy/auth/test_login_utils.py b/tests/test_litellm/proxy/auth/test_login_utils.py
index d41b90fe566..ada231a3e95 100644
--- a/tests/test_litellm/proxy/auth/test_login_utils.py
+++ b/tests/test_litellm/proxy/auth/test_login_utils.py
@@ -733,7 +733,12 @@ async def _db_login(throttle, username: str, password: str, *, correct: bool):
),
):
return await authenticate_user(
- username=username, password=password, master_key="sk-master", prisma_client=MagicMock(), throttle=throttle
+ username=username,
+ password=password,
+ master_key="sk-master",
+ prisma_client=MagicMock(),
+ throttle=throttle,
+ general_settings=_POLICY_NO_BREACH_CHECK,
)
@@ -2122,6 +2127,7 @@ class TestPasswordResetRequiredSessionMinting:
password="Str0ng!Passw0rd",
master_key="sk-1234",
prisma_client=mock_prisma_client,
+ throttle=_unlimited_throttle(),
general_settings=_POLICY_NO_BREACH_CHECK,
)
return result, mock_generate_key.call_args.kwargs
@@ -2163,6 +2169,7 @@ class TestPasswordResetRequiredSessionMinting:
password="Str0ng!Passw0rd",
master_key="sk-1234",
prisma_client=mock_prisma_client,
+ throttle=_unlimited_throttle(),
general_settings=_POLICY_NO_BREACH_CHECK,
)
return result, mock_generate_key.call_args.kwargs, mock_screen.call_args.kwargs
From ec55ab1ddb1ff656498314887510890c2320b812 Mon Sep 17 00:00:00 2001
From: ryan
Date: Mon, 21 Sep 2026 18:57:48 +0000
Subject: [PATCH 044/160] refactor(auth): use Annotated dependency in
change_password to satisfy strict gate
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
---
litellm/proxy/management_endpoints/password_endpoints.py | 4 ++--
1 file changed, 2 insertions(+), 2 deletions(-)
diff --git a/litellm/proxy/management_endpoints/password_endpoints.py b/litellm/proxy/management_endpoints/password_endpoints.py
index bd3c9722d4d..a2466c83ef9 100644
--- a/litellm/proxy/management_endpoints/password_endpoints.py
+++ b/litellm/proxy/management_endpoints/password_endpoints.py
@@ -8,7 +8,7 @@ request kwargs to OTEL spans, which would log plaintext passwords. The audit
signal is emitted by hand below, with field names only, never values.
"""
-from typing import TYPE_CHECKING, Final
+from typing import TYPE_CHECKING, Annotated, Final
from fastapi import APIRouter, Depends, HTTPException
@@ -58,7 +58,7 @@ def _user_table(
)
async def change_password(
data: ChangePasswordRequest,
- user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth),
+ user_api_key_dict: Annotated[UserAPIKeyAuth, Depends(user_api_key_auth)],
) -> ChangePasswordResponse:
"""
Change the calling user's own password.
From 2074273faf21d58a681fb34c6a4039d86f3ffa72 Mon Sep 17 00:00:00 2001
From: ryan
Date: Mon, 21 Sep 2026 19:06:00 +0000
Subject: [PATCH 045/160] test(proxy): give the faked login user the
breach-check columns
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
---
tests/test_litellm/proxy/proxy_server/test_routes_login_sso.py | 3 +++
1 file changed, 3 insertions(+)
diff --git a/tests/test_litellm/proxy/proxy_server/test_routes_login_sso.py b/tests/test_litellm/proxy/proxy_server/test_routes_login_sso.py
index 88f8be4e49a..8daea8d0ad2 100644
--- a/tests/test_litellm/proxy/proxy_server/test_routes_login_sso.py
+++ b/tests/test_litellm/proxy/proxy_server/test_routes_login_sso.py
@@ -10,6 +10,7 @@ Routes covered:
from __future__ import annotations
+from datetime import datetime, timezone
from unittest.mock import AsyncMock, MagicMock
from .conftest import normalize
@@ -510,6 +511,8 @@ def _db_user(monkeypatch, email: str):
user.user_email = email
user.user_role = "internal_user"
user.password = "scrypt:stored"
+ user.password_reset_required = None
+ user.last_breach_check_at = datetime.now(timezone.utc)
repo = MagicMock()
repo.return_value.table.find_first = AsyncMock(return_value=user)
monkeypatch.setattr(ps, "prisma_client", MagicMock())
From 1098604ed661ee54b66ec7a378a6163f589b10df Mon Sep 17 00:00:00 2001
From: Joshua Valluru <326636767+joshua-berri@users.noreply.github.com>
Date: Mon, 21 Sep 2026 12:24:15 -0700
Subject: [PATCH 046/160] refactor(mcp): extract explicit operation context and
dispatch
---
.github/workflows/test-linting.yml | 4 +
Makefile | 1 +
.../_experimental/mcp_server/contracts.py | 95 +
.../mcp_server/legacy_callbacks.py | 83 +
.../mcp_server/mcp_server_manager.py | 129 +-
.../_experimental/mcp_server/operations.py | 3102 +++++++++++++++++
.../mcp_server/rest_endpoints.py | 13 +-
.../proxy/_experimental/mcp_server/server.py | 3059 ++--------------
.../_experimental/mcp_server/tool_search.py | 11 +-
litellm/proxy/_types.py | 2 +
scripts/check_mcp_operation_boundary.py | 65 +
scripts/pre_commit_lint.sh | 3 +
.../mcp_server/test_byok_oauth_endpoints.py | 19 +-
.../mcp_server/test_contracts.py | 60 +
.../mcp_server/test_mcp_block_recording.py | 5 +-
.../mcp_server/test_mcp_hook_extra_headers.py | 6 +-
.../test_mcp_oauth_passthrough_tools.py | 61 +-
.../mcp_server/test_mcp_proxy_mode.py | 8 +-
.../mcp_server/test_mcp_server.py | 385 +-
.../mcp_server/test_mcp_server_manager.py | 174 +-
.../mcp_server/test_mcp_stale_session.py | 40 +-
.../mcp_server/test_mcp_tool_search.py | 40 +-
.../mcp_server/test_mcp_toolset_scope.py | 2 +-
.../mcp_server/test_openapi_tool_auth.py | 75 +-
.../mcp_server/test_operations.py | 343 ++
.../mcp_server/test_rest_endpoints.py | 13 +-
.../test_check_mcp_operation_boundary.py | 52 +
27 files changed, 4624 insertions(+), 3226 deletions(-)
create mode 100644 litellm/proxy/_experimental/mcp_server/contracts.py
create mode 100644 litellm/proxy/_experimental/mcp_server/legacy_callbacks.py
create mode 100644 litellm/proxy/_experimental/mcp_server/operations.py
create mode 100644 scripts/check_mcp_operation_boundary.py
create mode 100644 tests/test_litellm/proxy/_experimental/mcp_server/test_contracts.py
create mode 100644 tests/test_litellm/proxy/_experimental/mcp_server/test_operations.py
create mode 100644 tests/test_litellm/test_check_mcp_operation_boundary.py
diff --git a/.github/workflows/test-linting.yml b/.github/workflows/test-linting.yml
index 06d369eabcd..592d8edf6b8 100644
--- a/.github/workflows/test-linting.yml
+++ b/.github/workflows/test-linting.yml
@@ -130,6 +130,10 @@ jobs:
echo "File content around line 43:"
head -50 litellm/litellm_core_utils/custom_logger_registry.py | tail -10
+ - name: Check MCP operation boundary
+ if: steps.changes.outputs.decision != 'skip'
+ run: uv run --no-sync python scripts/check_mcp_operation_boundary.py
+
- name: Run Ruff linting
if: steps.changes.outputs.decision != 'skip'
run: |
diff --git a/Makefile b/Makefile
index 0e9d2bbf82c..ab7fab6aa99 100644
--- a/Makefile
+++ b/Makefile
@@ -164,6 +164,7 @@ lint-format-check-changed: $(LINT_DEP_INSTALL) $(LINT_DEP_BASE)
# Linting targets
lint-ruff: $(LINT_DEP_INSTALL)
+ $(UV_RUN) python scripts/check_mcp_operation_boundary.py
cd litellm && $(UV_RUN) ruff check . && cd ..
$(UV_RUN) ruff check --config ruff-tests.toml tests
diff --git a/litellm/proxy/_experimental/mcp_server/contracts.py b/litellm/proxy/_experimental/mcp_server/contracts.py
new file mode 100644
index 00000000000..c3129d171ad
--- /dev/null
+++ b/litellm/proxy/_experimental/mcp_server/contracts.py
@@ -0,0 +1,95 @@
+from collections.abc import Mapping
+from copy import deepcopy
+from dataclasses import dataclass, field
+from datetime import datetime
+from types import MappingProxyType
+from typing import Final, Protocol
+
+from litellm.proxy._types import UserAPIKeyAuth
+from litellm.types.mcp_server.mcp_server_manager import MCPServer
+
+
+def copy_caller(auth: UserAPIKeyAuth | None) -> UserAPIKeyAuth | None:
+ if auth is None:
+ return None
+ span: Final = auth.parent_otel_span
+ return deepcopy(auth, {id(span): span} if span is not None else None) # mutable-ok: deepcopy mutates its memo
+
+
+@dataclass(frozen=True, slots=True)
+class OperationContext:
+ _caller: UserAPIKeyAuth | None = field(repr=False)
+ mcp_auth_header: str | None = field(default=None, repr=False)
+ mcp_servers: tuple[str, ...] | None = None
+ mcp_server_auth_headers: Mapping[str, Mapping[str, str]] | None = field(default=None, repr=False)
+ oauth2_headers: Mapping[str, str] | None = field(default=None, repr=False)
+ raw_headers: Mapping[str, str] | None = field(default=None, repr=False)
+ client_ip: str | None = None
+ mcp_proxy_mode: bool = False
+
+ def __post_init__(self) -> None:
+ object.__setattr__(self, "_caller", copy_caller(self._caller))
+ object.__setattr__(self, "mcp_servers", tuple(self.mcp_servers) if self.mcp_servers is not None else None)
+ object.__setattr__(
+ self,
+ "oauth2_headers",
+ MappingProxyType(dict(self.oauth2_headers)) if self.oauth2_headers is not None else None,
+ )
+ object.__setattr__(
+ self, "raw_headers", MappingProxyType(dict(self.raw_headers)) if self.raw_headers is not None else None
+ )
+ object.__setattr__(
+ self,
+ "mcp_server_auth_headers",
+ MappingProxyType(
+ {key: MappingProxyType(dict(value)) for key, value in self.mcp_server_auth_headers.items()}
+ )
+ if self.mcp_server_auth_headers is not None
+ else None,
+ )
+
+ @property
+ def user_api_key_auth(self) -> UserAPIKeyAuth | None:
+ return copy_caller(self._caller)
+
+ def legacy_auth(
+ self,
+ ) -> tuple[
+ UserAPIKeyAuth | None,
+ str | None,
+ list[str] | None, # mutable-ok: detached legacy server-list payload
+ dict[str, dict[str, str]] | None, # mutable-ok: legacy auth dispatch requires concrete dict headers
+ dict[str, str] | None, # mutable-ok: detached legacy header payload
+ dict[str, str] | None, # mutable-ok: detached legacy header payload
+ str | None,
+ ]:
+ return (
+ self.user_api_key_auth,
+ self.mcp_auth_header,
+ list(self.mcp_servers) if self.mcp_servers is not None else None, # mutable-ok: legacy policy list input
+ {
+ key: dict(value) for key, value in self.mcp_server_auth_headers.items()
+ } # mutable-ok: legacy auth dispatch checks concrete dict headers
+ if self.mcp_server_auth_headers is not None
+ else None,
+ dict(self.oauth2_headers)
+ if self.oauth2_headers is not None
+ else None, # mutable-ok: legacy OAuth header input
+ dict(self.raw_headers) if self.raw_headers is not None else None, # mutable-ok: legacy request header input
+ self.client_ip,
+ )
+
+
+class ProgressCallback(Protocol):
+ async def __call__(self, progress: float, total: float | None, /) -> None: ...
+
+
+@dataclass(frozen=True, slots=True)
+class AuthorizedToolCall:
+ name: str
+ arguments: Mapping[str, object]
+ allowed_mcp_servers: tuple[MCPServer, ...]
+ start_time: datetime
+ host_progress_callback: ProgressCallback | None
+ guardrail_context: Mapping[str, object] | None
+ logging_data: Mapping[str, object]
diff --git a/litellm/proxy/_experimental/mcp_server/legacy_callbacks.py b/litellm/proxy/_experimental/mcp_server/legacy_callbacks.py
new file mode 100644
index 00000000000..4424907c28a
--- /dev/null
+++ b/litellm/proxy/_experimental/mcp_server/legacy_callbacks.py
@@ -0,0 +1,83 @@
+from collections.abc import Mapping
+from typing import Final, Protocol
+
+from mcp.client.session import ClientRequestContext
+from mcp.types import (
+ CreateMessageRequestParams,
+ CreateMessageResult,
+ CreateMessageResultWithTools,
+ ElicitRequestParams,
+ ElicitResult,
+ ErrorData,
+)
+
+from litellm.proxy._experimental.mcp_server.contracts import OperationContext
+from litellm.proxy._types import UserAPIKeyAuth
+
+
+class SamplingCallback(Protocol):
+ async def __call__(
+ self, context: ClientRequestContext, params: CreateMessageRequestParams, /
+ ) -> CreateMessageResult | CreateMessageResultWithTools | ErrorData: ...
+
+
+class ElicitationCallback(Protocol):
+ async def __call__(self, context: object, params: ElicitRequestParams, /) -> ElicitResult | ErrorData: ...
+
+
+def create_sampling_callback(
+ user_api_key_auth: UserAPIKeyAuth | None = None,
+ raw_headers: Mapping[str, str] | None = None,
+ client_ip: str | None = None,
+ operation_context: OperationContext | None = None,
+) -> SamplingCallback:
+ from litellm.proxy._experimental.mcp_server.server import get_active_auth_context
+
+ auth: Final = get_active_auth_context() if operation_context is None else None
+ captured: Final = (
+ operation_context
+ if operation_context is not None
+ else OperationContext(
+ _caller=user_api_key_auth if user_api_key_auth is not None else (auth.user_api_key_auth if auth else None),
+ raw_headers=raw_headers if raw_headers is not None else (auth.raw_headers if auth else None),
+ client_ip=client_ip if client_ip is not None else (auth.client_ip if auth else None),
+ )
+ )
+
+ async def callback(
+ context: ClientRequestContext, params: CreateMessageRequestParams
+ ) -> CreateMessageResult | CreateMessageResultWithTools | ErrorData:
+ import litellm
+ from litellm.proxy._experimental.mcp_server.sampling_handler import handle_sampling_create_message
+
+ return await handle_sampling_create_message(
+ context=context,
+ params=params,
+ default_model=getattr(litellm, "default_mcp_sampling_model", None),
+ user_api_key_auth=captured.user_api_key_auth,
+ raw_headers=dict(captured.raw_headers)
+ if captured.raw_headers is not None
+ else None, # mutable-ok: handler consumes an owned request header dict
+ client_ip=captured.client_ip,
+ )
+
+ return callback
+
+
+def create_elicitation_callback() -> ElicitationCallback:
+ from litellm.proxy._experimental.mcp_server.server import get_active_mcp_session
+
+ downstream_session: Final = get_active_mcp_session()
+ downstream_capabilities: Final = getattr(downstream_session, "capabilities", None)
+
+ async def callback(context: object, params: ElicitRequestParams) -> ElicitResult | ErrorData:
+ from litellm.proxy._experimental.mcp_server.elicitation_handler import handle_elicitation_request
+
+ return await handle_elicitation_request(
+ context=context,
+ params=params,
+ downstream_session=downstream_session,
+ downstream_capabilities=downstream_capabilities,
+ )
+
+ return callback
diff --git a/litellm/proxy/_experimental/mcp_server/mcp_server_manager.py b/litellm/proxy/_experimental/mcp_server/mcp_server_manager.py
index b293ab5a206..264d7cf19aa 100644
--- a/litellm/proxy/_experimental/mcp_server/mcp_server_manager.py
+++ b/litellm/proxy/_experimental/mcp_server/mcp_server_manager.py
@@ -73,6 +73,7 @@ from litellm.proxy._experimental.mcp_server.auth.user_api_key_auth_mcp import (
MCPServerAccess,
_is_mcp_admitted_user_subject,
)
+from litellm.proxy._experimental.mcp_server.contracts import OperationContext
from litellm.proxy._experimental.mcp_server.elicitation_handler import (
MCP_ELICITATION_AVAILABLE,
)
@@ -195,9 +196,6 @@ from litellm.types.mcp_server.mcp_server_manager import (
from litellm.types.utils import CallTypes
if TYPE_CHECKING:
- from mcp.client.session import ClientRequestContext
- from mcp.types import CreateMessageRequestParams
-
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
from litellm.types.mcp_server.mcp_toolset import MCPToolset
@@ -1218,7 +1216,7 @@ async def _resolve_byok_mcp_auth_header(
if not mcp_server.is_byok:
return mcp_auth_header
- from litellm.proxy._experimental.mcp_server.server import (
+ from litellm.proxy._experimental.mcp_server.operations import (
_check_byok_credential,
_get_byok_credential,
)
@@ -1577,77 +1575,25 @@ def _normalize_mcp_server_cost_info(mcp_info: MCPInfo) -> None:
mcp_info["mcp_server_cost_info"] = normalized
-def _create_sampling_callback(user_api_key_auth: UserAPIKeyAuth | None = None):
- """
- Create a sampling callback for MCP ClientSession.
- Returns a callable that handles sampling/createMessage requests from
- upstream MCP servers by routing them through litellm.acompletion().
- """
+def _create_sampling_callback(
+ user_api_key_auth: UserAPIKeyAuth | None = None,
+ raw_headers: Mapping[str, str] | None = None,
+ client_ip: str | None = None,
+ operation_context: OperationContext | None = None,
+):
if not MCP_SAMPLING_AVAILABLE:
return None
+ from litellm.proxy._experimental.mcp_server.legacy_callbacks import create_sampling_callback
- async def _sampling_callback(
- context: "ClientRequestContext",
- params: "CreateMessageRequestParams",
- ):
- import litellm
- from litellm.proxy._experimental.mcp_server.sampling_handler import (
- handle_sampling_create_message,
- )
- from litellm.proxy._experimental.mcp_server.server import (
- get_active_auth_context,
- )
-
- auth_context: Final = get_active_auth_context()
- resolved_auth: Final = user_api_key_auth or (auth_context.user_api_key_auth if auth_context else None)
- # Forward original HTTP headers and client IP so that
- # header-dependent guardrails, tag-based routing, trace
- # correlation, and forward_llm_provider_auth_headers work
- # correctly for sampling sub-calls.
- _raw_headers: Final = getattr(auth_context, "raw_headers", None)
- _client_ip: Final = getattr(auth_context, "client_ip", None)
-
- return await handle_sampling_create_message(
- context=context,
- params=params,
- default_model=getattr(litellm, "default_mcp_sampling_model", None),
- user_api_key_auth=resolved_auth,
- raw_headers=_raw_headers,
- client_ip=_client_ip,
- )
-
- return _sampling_callback
+ return create_sampling_callback(user_api_key_auth, raw_headers, client_ip, operation_context)
def _create_elicitation_callback():
- """
- Create an elicitation callback for MCP ClientSession.
- Returns a callable that handles elicitation/create requests from
- upstream MCP servers. In gateway mode, this relays to the downstream
- client; in tool bridge mode, it returns a decline response.
- """
if not MCP_ELICITATION_AVAILABLE:
return None
+ from litellm.proxy._experimental.mcp_server.legacy_callbacks import create_elicitation_callback
- async def _elicitation_callback(context, params):
- from litellm.proxy._experimental.mcp_server.elicitation_handler import (
- handle_elicitation_request,
- )
- from litellm.proxy._experimental.mcp_server.server import get_active_mcp_session
-
- # In Gateway mode, we relay the elicitation request to the downstream client
- # that triggered the current operation.
- downstream_session: Final = get_active_mcp_session()
- downstream_capabilities = getattr(downstream_session, "capabilities", None) if downstream_session else None
-
- return await handle_elicitation_request(
- context=context,
- params=params,
- downstream_session=downstream_session,
- downstream_capabilities=downstream_capabilities,
- )
-
- return _elicitation_callback
+ return create_elicitation_callback()
def _record_mcp_guardrail_evaluations(
@@ -3373,17 +3319,13 @@ class MCPServerManager:
listable but uninvokable.
Empty inside a toolset scope: toolset_mcp_route / dynamic_mcp_route set
- ``_mcp_active_toolset_id`` before calling the handler, pinning the request to the toolset's
+ the caller's server-only ``mcp_toolset_id`` before calling the handler, pinning the request to the toolset's
own servers (checking op.mcp_toolsets==[] instead would false-positive on DB-default rows
where Postgres initialises the column to ARRAY[]::TEXT[]).
``allow_all_server_ids`` / ``submitted_server_ids`` are injectable so the server union,
which precomputes both for its fallback path, does not compute them twice."""
- from litellm.proxy._experimental.mcp_server.mcp_context import ( # noqa: PLC0415
- _mcp_active_toolset_id,
- )
-
- if _mcp_active_toolset_id.get() is not None:
+ if user_api_key_auth is not None and user_api_key_auth.mcp_toolset_id is not None:
return set()
if allow_all_server_ids is None:
allow_all_server_ids = self.get_allow_all_keys_server_ids()
@@ -4151,6 +4093,8 @@ class MCPServerManager:
subject_token: str | None = None,
user_api_key_auth: UserAPIKeyAuth | None = None,
cred_provider: UpstreamCredentialProvider | None = None,
+ raw_headers: Mapping[str, str] | None = None,
+ client_ip: str | None = None,
) -> MCPClient:
"""
Create an MCPClient instance for the given server.
@@ -4199,7 +4143,13 @@ class MCPServerManager:
# Create sampling and elicitation callbacks for this client
sampling_cb = (
- _create_sampling_callback(user_api_key_auth=user_api_key_auth) if resolved_server.allow_sampling else None
+ _create_sampling_callback(
+ operation_context=OperationContext(
+ _caller=user_api_key_auth, raw_headers=raw_headers, client_ip=client_ip
+ )
+ )
+ if resolved_server.allow_sampling
+ else None
)
elicitation_cb: Final = _create_elicitation_callback() if resolved_server.allow_elicitation else None
@@ -4344,6 +4294,7 @@ class MCPServerManager:
raw_headers: dict[str, str] | None = None,
user_api_key_auth: UserAPIKeyAuth | None = None,
oauth2_headers: dict[str, str] | None = None,
+ client_ip: str | None = None,
) -> list[MCPTool]:
"""
Helper method to get tools from a single MCP server with prefixed names.
@@ -4433,6 +4384,8 @@ class MCPServerManager:
stdio_env=stdio_env,
subject_token=subject_token,
user_api_key_auth=user_api_key_auth,
+ raw_headers=raw_headers,
+ client_ip=client_ip,
)
## HANDLE OPENAPI TOOLS
@@ -4543,6 +4496,7 @@ class MCPServerManager:
extra_headers: dict[str, str] | None = None,
add_prefix: bool = True,
raw_headers: dict[str, str] | None = None,
+ client_ip: str | None = None,
) -> list[Prompt]:
try:
headers: Final = (
@@ -4563,6 +4517,8 @@ class MCPServerManager:
stdio_env=stdio_env,
subject_token=subject_token,
user_api_key_auth=user_api_key_auth,
+ raw_headers=raw_headers,
+ client_ip=client_ip,
)
credential_fingerprint: Final = await client.discovery_auth_fingerprint()
key: Final = self._discovery_key(
@@ -4586,6 +4542,7 @@ class MCPServerManager:
extra_headers: dict[str, str] | None = None,
add_prefix: bool = True,
raw_headers: dict[str, str] | None = None,
+ client_ip: str | None = None,
) -> list[Resource]:
try:
headers: Final = (
@@ -4606,6 +4563,8 @@ class MCPServerManager:
stdio_env=stdio_env,
subject_token=subject_token,
user_api_key_auth=user_api_key_auth,
+ raw_headers=raw_headers,
+ client_ip=client_ip,
)
credential_fingerprint: Final = await client.discovery_auth_fingerprint()
key: Final = self._discovery_key(
@@ -4629,6 +4588,7 @@ class MCPServerManager:
extra_headers: dict[str, str] | None = None,
add_prefix: bool = True,
raw_headers: dict[str, str] | None = None,
+ client_ip: str | None = None,
) -> list[ResourceTemplate]:
try:
headers: Final = (
@@ -4649,6 +4609,8 @@ class MCPServerManager:
stdio_env=stdio_env,
subject_token=subject_token,
user_api_key_auth=user_api_key_auth,
+ raw_headers=raw_headers,
+ client_ip=client_ip,
)
credential_fingerprint: Final = await client.discovery_auth_fingerprint()
key: Final = self._discovery_key(
@@ -4672,6 +4634,7 @@ class MCPServerManager:
mcp_auth_header: str | dict[str, str] | None = None,
extra_headers: dict[str, str] | None = None,
raw_headers: dict[str, str] | None = None,
+ client_ip: str | None = None,
) -> ReadResourceResult:
"""Read resource contents from a specific MCP server."""
@@ -4692,6 +4655,9 @@ class MCPServerManager:
extra_headers=extra_headers,
stdio_env=stdio_env,
subject_token=subject_token,
+ raw_headers=raw_headers,
+ client_ip=client_ip,
+ user_api_key_auth=user_api_key_auth,
)
return await client.read_resource(url)
@@ -4705,6 +4671,7 @@ class MCPServerManager:
mcp_auth_header: str | dict[str, str] | None = None,
extra_headers: dict[str, str] | None = None,
raw_headers: dict[str, str] | None = None,
+ client_ip: str | None = None,
) -> GetPromptResult:
"""Fetch a specific prompt definition from a single MCP server."""
@@ -4725,6 +4692,9 @@ class MCPServerManager:
extra_headers=extra_headers,
stdio_env=stdio_env,
subject_token=subject_token,
+ raw_headers=raw_headers,
+ client_ip=client_ip,
+ user_api_key_auth=user_api_key_auth,
)
get_prompt_request_params: Final = GetPromptRequestParams(
@@ -5805,6 +5775,8 @@ class MCPServerManager:
stdio_env: dict[str, str] | None,
subject_token: str | None,
user_api_key_auth: UserAPIKeyAuth | None,
+ raw_headers: Mapping[str, str] | None = None,
+ client_ip: str | None = None,
) -> CallToolResult:
"""Call a token_exchange (OBO) tool; on an upstream 401/403 re-mint the token once and retry.
@@ -5830,6 +5802,8 @@ class MCPServerManager:
stdio_env=stdio_env,
subject_token=subject_token,
user_api_key_auth=user_api_key_auth,
+ raw_headers=raw_headers,
+ client_ip=client_ip,
)
return await retry_client.call_tool(call_tool_params, host_progress_callback=host_progress_callback)
@@ -5847,6 +5821,7 @@ class MCPServerManager:
host_progress_callback: Callable | None = None,
hook_extra_headers: dict[str, str] | None = None,
user_api_key_auth: UserAPIKeyAuth | None = None,
+ client_ip: str | None = None,
) -> CallToolResult:
"""
Call a regular MCP tool using the MCP client.
@@ -5991,6 +5966,8 @@ class MCPServerManager:
stdio_env=stdio_env,
subject_token=subject_token,
user_api_key_auth=user_api_key_auth,
+ raw_headers=raw_headers,
+ client_ip=client_ip,
)
call_tool_params: Final = MCPCallToolRequestParams(
@@ -6014,6 +5991,8 @@ class MCPServerManager:
stdio_env=stdio_env,
subject_token=subject_token,
user_api_key_auth=user_api_key_auth,
+ raw_headers=raw_headers,
+ client_ip=client_ip,
)
tool_call_coro = _obo_call_tool_limited()
@@ -6189,7 +6168,7 @@ class MCPServerManager:
return oauth2_headers
try:
- from litellm.proxy._experimental.mcp_server.server import ( # noqa: PLC0415
+ from litellm.proxy._experimental.mcp_server.operations import ( # noqa: PLC0415
_get_user_oauth_extra_headers_from_db,
)
@@ -6295,6 +6274,7 @@ class MCPServerManager:
host_progress_callback: Callable | None = None,
litellm_logging_obj: "LiteLLMLoggingObj | None" = None,
guardrail_context: Mapping[str, object] | None = None,
+ client_ip: str | None = None,
) -> CallToolResult:
"""
Call a tool with the given name and arguments
@@ -6421,6 +6401,7 @@ class MCPServerManager:
mcp_server_auth_headers=mcp_server_auth_headers,
oauth2_headers=oauth2_headers,
raw_headers=raw_headers,
+ client_ip=client_ip,
proxy_logging_obj=proxy_logging_obj,
host_progress_callback=host_progress_callback,
hook_extra_headers=hook_result.get("extra_headers"),
diff --git a/litellm/proxy/_experimental/mcp_server/operations.py b/litellm/proxy/_experimental/mcp_server/operations.py
new file mode 100644
index 00000000000..3ee4f37add7
--- /dev/null
+++ b/litellm/proxy/_experimental/mcp_server/operations.py
@@ -0,0 +1,3102 @@
+"""Shared MCP operation policy and dispatch."""
+
+import asyncio
+import traceback
+import types
+import uuid
+from collections.abc import Mapping, Sequence
+from datetime import datetime
+from typing import Any, Final, NoReturn, TypeAlias, assert_never, overload
+
+from fastapi import HTTPException
+from mcp import ReadResourceResult, Resource
+from mcp.types import (
+ CallToolRequest,
+ CallToolRequestParams,
+ CallToolResult,
+ GetPromptRequest,
+ GetPromptRequestParams,
+ GetPromptResult,
+ ListPromptsRequest,
+ ListPromptsResult,
+ ListResourcesRequest,
+ ListResourcesResult,
+ ListResourceTemplatesRequest,
+ ListResourceTemplatesResult,
+ ListToolsRequest,
+ ListToolsResult,
+ PaginatedRequestParams,
+ Prompt,
+ ReadResourceRequest,
+ ReadResourceRequestParams,
+ ResourceTemplate,
+ TextContent,
+)
+from mcp.types import Tool as MCPTool
+from pydantic import AnyUrl, ConfigDict, Field, TypeAdapter
+from typing_extensions import ReadOnly, TypedDict
+
+from litellm._logging import verbose_logger
+from litellm.constants import (
+ MAXIMUM_TRACEBACK_LINES_TO_LOG,
+)
+from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
+from litellm.proxy._experimental.mcp_server.auth.user_api_key_auth_mcp import (
+ MCPRequestHandler,
+)
+from litellm.proxy._experimental.mcp_server.byok_credential_cache import (
+ byok_credential_cache,
+ byok_credential_cache_key,
+ cache_byok_credential,
+ get_cached_byok_credential,
+)
+from litellm.proxy._experimental.mcp_server.contracts import (
+ AuthorizedToolCall,
+ OperationContext,
+ ProgressCallback,
+)
+from litellm.proxy._experimental.mcp_server.db import OAuthCredentialPayload
+from litellm.proxy._experimental.mcp_server.exceptions import (
+ MCPToolResultError,
+ MCPUpstreamAuthError,
+)
+from litellm.proxy._experimental.mcp_server.faults.list_outcomes import (
+ SERVER_OUTCOMES_META_KEY,
+ AggregateToolListing,
+ ServerListOk,
+ ServerOutcome,
+ classify_list_exception,
+ outcome_wire_value,
+)
+from litellm.proxy._experimental.mcp_server.mcp_server_manager import (
+ MCPServerManager,
+ _caller_authorization_fans_out,
+ _client_forwarded_authorization_headers,
+ _resolve_openapi_tool_auth,
+ _should_strip_caller_authorization,
+ global_mcp_server_manager,
+)
+from litellm.proxy._experimental.mcp_server.oauth_utils import (
+ _redact_mcp_resource_url,
+ get_byok_www_authenticate,
+)
+from litellm.proxy._experimental.mcp_server.openapi_to_mcp_generator import (
+ _request_auth_header,
+ _request_extra_headers,
+ _request_resolved_auth_headers,
+)
+from litellm.proxy._experimental.mcp_server.tool_registry import (
+ global_mcp_tool_registry,
+)
+from litellm.proxy._experimental.mcp_server.utils import (
+ MCP_TOOL_PREFIX_SEPARATOR,
+ MCPMissingUserEnvVarsError,
+ add_server_prefix_to_name,
+ build_synthetic_mcp_request,
+ extract_mcp_tool_result_error_message,
+ get_server_prefix,
+ is_tool_name_prefixed,
+ iter_known_server_prefixes,
+ logging_safe_mcp_headers,
+ match_known_tool_name,
+ normalize_server_name,
+ split_server_prefix_from_name,
+ strip_known_server_prefix,
+)
+from litellm.proxy._types import (
+ UserAPIKeyAuth,
+)
+from litellm.proxy.common_utils.auth_cache_invalidation_pubsub import (
+ publish_auth_cache_invalidation,
+)
+from litellm.proxy.litellm_pre_call_utils import (
+ LiteLLMProxyRequestSetup,
+ get_chain_id_from_headers,
+)
+from litellm.types.mcp import (
+ DEFAULT_CREDENTIAL_HEADER,
+ MCPAuth,
+ without_header,
+)
+from litellm.types.mcp_server.mcp_server_manager import MCPInfo, MCPServer
+from litellm.types.utils import CallTypes, StandardLoggingMCPToolCall
+from litellm.utils import Rules, client, function_setup
+
+__all__ = (
+ "_MCP_CREDENTIAL_REQUEST_FIELDS",
+ "ListMCPToolsRestAPIResponseObject",
+ "MCPInfo",
+ "MCPServer",
+ "_McpDeniedDetail",
+ "_aggregate_server_key",
+ "_build_virtual_call_logging_obj",
+ "_check_byok_credential",
+ "_client_has_passthrough_authorization",
+ "_client_has_per_server_auth_header",
+ "_dispatch_virtual_mcp_tool",
+ "_fire_mcp_tool_call_logging",
+ "_get_allowed_mcp_servers",
+ "_get_allowed_mcp_servers_from_mcp_server_names",
+ "_get_byok_credential",
+ "_get_prompts_from_mcp_servers",
+ "_get_resource_templates_from_mcp_servers",
+ "_get_resources_from_mcp_servers",
+ "_get_standard_logging_mcp_tool_call",
+ "_get_tools_from_mcp_servers",
+ "_get_user_oauth_extra_headers_from_db",
+ "_handle_local_mcp_tool",
+ "_handle_managed_mcp_tool",
+ "_http_detail_message",
+ "_invalidate_byok_cred_cache",
+ "_list_mcp_prompts",
+ "_list_mcp_resource_templates",
+ "_list_mcp_resources",
+ "_list_mcp_tools",
+ "_list_tools_before_first_call",
+ "_mcp_session_id_from_headers",
+ "_merge_gateway_initialize_instructions",
+ "_prefetch_oauth_creds_for_user",
+ "_prepare_mcp_server_headers",
+ "_raise_if_initialize_grants_no_mcp_servers",
+ "_resolve_display_name_to_original",
+ "_run_post_mcp_call_guardrails",
+ "_server_answers_to",
+ "_tool_name_matches",
+ "apply_tool_overrides",
+ "call_mcp_tool",
+ "execute_mcp_tool",
+ "filter_tools_by_allowed_tools",
+ "filter_tools_by_key_team_permissions",
+ "fire_mcp_tool_call_failure_logging",
+ "mcp_get_prompt",
+ "mcp_read_resource",
+ "raise_denied_scoped_mcp_access",
+)
+
+
+async def _invalidate_byok_cred_cache(user_id: str, server_id: str) -> None:
+ """Drop a stored-or-deleted BYOK credential from this worker's cache and from every peer worker's."""
+ cache_key: Final = byok_credential_cache_key(user_id, server_id)
+ byok_credential_cache.delete_cache(cache_key)
+ await publish_auth_cache_invalidation(cache_key=cache_key)
+
+
+def _mcp_session_id_from_headers(
+ raw_headers: dict[str, str] | None,
+) -> str | None:
+ """The ``mcp-session-id`` of a stateful MCP session, read case-insensitively
+ from the request headers. ``None`` for stateless calls (no such header)."""
+ if not raw_headers:
+ return None
+ for key, value in raw_headers.items():
+ if isinstance(key, str) and key.lower() == "mcp-session-id":
+ return value or None
+ return None
+
+
+class ListMCPToolsRestAPIResponseObject(MCPTool):
+ """
+ Object returned by the /tools/list REST API route.
+ """
+
+ mcp_info: MCPInfo | None = Field(default=None, alias="mcp_info")
+ model_config = ConfigDict(arbitrary_types_allowed=True)
+
+
+async def _build_virtual_call_logging_obj(
+ name: str,
+ arguments: dict[str, object],
+ user_api_key_auth: UserAPIKeyAuth,
+ raw_headers: Mapping[str, str] | None = None,
+ client_ip: str | None = None,
+) -> LiteLLMLoggingObj | None:
+ """Run the pre-call pipeline (guardrails + logging setup) for a virtual
+ mcp_tool_call so the SSE path spend-logs like the REST path."""
+ from litellm.proxy.common_request_processing import (
+ ProxyBaseLLMRequestProcessing,
+ )
+ from litellm.proxy.proxy_server import (
+ general_settings,
+ proxy_config,
+ proxy_logging_obj,
+ )
+
+ request: Final = build_synthetic_mcp_request(
+ path="/mcp/tools/call",
+ raw_headers=raw_headers,
+ client_ip=client_ip,
+ )
+ _, virtual_logging_obj = await ProxyBaseLLMRequestProcessing(
+ data={"name": name, "arguments": arguments}
+ ).common_processing_pre_call_logic(
+ request=request,
+ user_api_key_dict=user_api_key_auth,
+ proxy_config=proxy_config,
+ route_type=CallTypes.call_mcp_tool.value,
+ proxy_logging_obj=proxy_logging_obj,
+ general_settings=general_settings,
+ )
+ return virtual_logging_obj
+
+
+async def _dispatch_virtual_mcp_tool(
+ name: str,
+ arguments: dict[str, object] | None,
+ user_api_key_auth: UserAPIKeyAuth | None,
+ client_ip: str | None,
+ mcp_servers: list[str] | None = None,
+ mcp_auth_header: str | None = None,
+ mcp_server_auth_headers: dict[str, dict[str, str]] | None = None,
+ oauth2_headers: dict[str, str] | None = None,
+ raw_headers: dict[str, str] | None = None,
+ mcp_proxy_mode: bool = False,
+) -> CallToolResult | None:
+ """Handle the mcp_tool_search / mcp_tool_call virtual tools.
+
+ Returns a CallToolResult when ``name`` is a virtual tool, else ``None`` so
+ the caller falls through to normal tool routing.
+ """
+ from litellm.llms.litellm_proxy.skills.skill_search import DEFAULT_SKILL_SEARCH_TOP_K
+ from litellm.proxy._experimental.mcp_server.tool_search import (
+ AGENT_SEARCH_TOOL_NAME,
+ DEFAULT_AGENT_SEARCH_TOP_K,
+ MCP_PROXY_CALL_TOOL_NAME,
+ MCP_PROXY_TOOL_NAMES,
+ MCP_TOOL_SEARCH_TOOL_NAME,
+ SKILL_SEARCH_TOOL_NAME,
+ VIRTUAL_TOOL_NAMES,
+ coerce_top_k,
+ handle_agent_search,
+ handle_mcp_proxy_tool,
+ handle_mcp_tool_call,
+ handle_mcp_tool_search,
+ handle_skill_search,
+ )
+
+ if mcp_proxy_mode and name not in MCP_PROXY_TOOL_NAMES:
+ return CallToolResult(
+ content=[ # mutable-ok: MCP result content
+ TextContent(type="text", text=f"Tool {name} is unavailable on /mcp/proxy")
+ ],
+ is_error=True,
+ )
+
+ if mcp_proxy_mode and name in MCP_PROXY_TOOL_NAMES:
+ assert user_api_key_auth is not None
+ proxy_call_start: Final = datetime.now() # noqa: DTZ005 # logging pipeline uses naive datetimes
+ proxy_logging_obj: Final = (
+ await _build_virtual_call_logging_obj(
+ name=name,
+ arguments=arguments or {}, # mutable-ok: logging pipeline payload
+ user_api_key_auth=user_api_key_auth,
+ raw_headers=raw_headers,
+ client_ip=client_ip,
+ )
+ if name == MCP_PROXY_CALL_TOOL_NAME
+ else None
+ )
+ try:
+ proxy_result: Final = await handle_mcp_proxy_tool(
+ name=name,
+ arguments=arguments or {}, # mutable-ok: proxy handler payload
+ user_api_key_dict=user_api_key_auth,
+ client_ip=client_ip,
+ mcp_servers=mcp_servers,
+ mcp_auth_header=mcp_auth_header,
+ mcp_server_auth_headers=mcp_server_auth_headers,
+ oauth2_headers=oauth2_headers,
+ raw_headers=raw_headers,
+ litellm_logging_obj=proxy_logging_obj,
+ )
+ except Exception as exc:
+ if proxy_logging_obj is not None:
+ from litellm.proxy.proxy_server import proxy_logging_obj as request_logging_obj
+
+ failure_end: Final = datetime.now() # noqa: DTZ005 # matches the logging pipeline start time
+ failure_traceback: Final = traceback.format_exc(limit=MAXIMUM_TRACEBACK_LINES_TO_LOG)
+ try:
+ proxy_logging_obj.failure_handler(exc, failure_traceback, proxy_call_start, failure_end)
+ await proxy_logging_obj.async_failure_handler(exc, failure_traceback, proxy_call_start, failure_end)
+ if not isinstance(exc, MCPUpstreamAuthError):
+ await request_logging_obj.post_call_failure_hook(
+ request_data={ # mutable-ok: failure hook mutates its request payload
+ "name": name,
+ "arguments": arguments,
+ "litellm_logging_obj": proxy_logging_obj,
+ },
+ original_exception=exc,
+ user_api_key_dict=user_api_key_auth,
+ route="/mcp/call_tool",
+ traceback_str=failure_traceback,
+ )
+ except Exception: # noqa: BLE001 # a failing failure hook must not mask the tool call's own error
+ verbose_logger.exception("Error logging failed MCP proxy tool call")
+ raise
+ if proxy_logging_obj is not None:
+ return await _fire_mcp_tool_call_logging(
+ logging_obj=proxy_logging_obj,
+ result=proxy_result,
+ start_time=proxy_call_start,
+ end_time=datetime.now(), # noqa: DTZ005 # matches the logging pipeline start time
+ user_api_key_auth=user_api_key_auth,
+ request_data=types.MappingProxyType({"name": name, "arguments": arguments}),
+ )
+ return proxy_result
+
+ if name not in VIRTUAL_TOOL_NAMES:
+ return None
+
+ if not getattr(
+ getattr(user_api_key_auth, "object_permission", None),
+ "mcp_tool_search_enabled",
+ False,
+ ):
+ return CallToolResult(
+ content=[
+ TextContent(
+ type="text",
+ text=f"Tool {name} requires mcp_tool_search_enabled on the key",
+ )
+ ],
+ is_error=True,
+ )
+
+ args: Final = arguments or {}
+ if name == MCP_TOOL_SEARCH_TOOL_NAME:
+ return await handle_mcp_tool_search(
+ query=TypeAdapter(str).validate_python(args.get("query", "")),
+ top_k=coerce_top_k(args.get("top_k", 5)),
+ user_api_key_dict=user_api_key_auth,
+ client_ip=client_ip,
+ mcp_servers=mcp_servers,
+ mcp_auth_header=mcp_auth_header,
+ mcp_server_auth_headers=mcp_server_auth_headers,
+ oauth2_headers=oauth2_headers,
+ raw_headers=raw_headers,
+ )
+
+ assert user_api_key_auth is not None # guaranteed by the flag check above
+ if name == AGENT_SEARCH_TOOL_NAME:
+ return await handle_agent_search(
+ query=str(args.get("query", "")),
+ top_k=coerce_top_k(args.get("top_k", DEFAULT_AGENT_SEARCH_TOP_K), default=DEFAULT_AGENT_SEARCH_TOP_K),
+ user_api_key_dict=user_api_key_auth,
+ )
+ if name == SKILL_SEARCH_TOOL_NAME:
+ return await handle_skill_search(
+ query=str(args.get("query", "")),
+ top_k=coerce_top_k(args.get("top_k", DEFAULT_SKILL_SEARCH_TOP_K), default=DEFAULT_SKILL_SEARCH_TOP_K),
+ user_api_key_dict=user_api_key_auth,
+ )
+ virtual_logging_obj: Final = await _build_virtual_call_logging_obj(
+ name=name,
+ arguments=args,
+ user_api_key_auth=user_api_key_auth,
+ raw_headers=raw_headers,
+ client_ip=client_ip,
+ )
+ tool_request: Final = CallToolRequestParams.model_validate(
+ types.MappingProxyType({"name": args.get("tool_name", ""), "arguments": args.get("arguments") or {}})
+ )
+ return await handle_mcp_tool_call(
+ tool_name=tool_request.name,
+ arguments=tool_request.arguments or {},
+ user_api_key_dict=user_api_key_auth,
+ client_ip=client_ip,
+ mcp_servers=mcp_servers,
+ mcp_auth_header=mcp_auth_header,
+ mcp_server_auth_headers=mcp_server_auth_headers,
+ oauth2_headers=oauth2_headers,
+ raw_headers=raw_headers,
+ litellm_logging_obj=virtual_logging_obj,
+ )
+
+
+async def _get_allowed_mcp_servers_from_mcp_server_names(
+ mcp_servers: Sequence[str] | None,
+ allowed_mcp_servers: list[MCPServer],
+) -> list[MCPServer]:
+ """
+ Get the filtered MCP servers from the MCP server names.
+
+ Fails closed when ``mcp_servers`` is explicitly provided (path- or
+ header-derived) but none of the names resolve to a server alias or
+ access group the caller can access. The previous behavior returned
+ the full ``allowed_mcp_servers`` set, which silently widened scope
+ when a client targeted ``/mcp//`` and made URL/header
+ namespacing appear to work when it did not.
+ """
+
+ filtered_server: Final[dict[str, MCPServer]] = {}
+ # Filter servers based on mcp_servers parameter if provided
+ if mcp_servers is not None:
+ for server_or_group in mcp_servers:
+ server_name_matched = False
+
+ for server in allowed_mcp_servers:
+ if server and _server_answers_to(server, server_or_group):
+ filtered_server[server.server_id] = server
+ server_name_matched = True
+ break
+
+ if not server_name_matched:
+ try:
+ access_group_server_ids = await MCPRequestHandler._get_mcp_servers_from_access_groups(
+ [server_or_group]
+ )
+ # Only include servers that the user has access to
+ for server_id in access_group_server_ids:
+ for server in allowed_mcp_servers:
+ if server_id == server.server_id:
+ filtered_server[server.server_id] = server
+ except Exception as e:
+ verbose_logger.debug("Could not resolve '%s' as access group: %s", server_or_group, e)
+
+ if filtered_server:
+ return list(filtered_server.values())
+
+ if mcp_servers is not None:
+ # Caller asked for a specific scope but nothing resolved. Fail
+ # closed so URL/header namespacing cannot silently fall back to
+ # the caller's full allowed-server set.
+ verbose_logger.debug(
+ "MCP scope filter resolved to no servers for requested names %s; returning empty list (fail-closed).",
+ mcp_servers,
+ )
+ return []
+
+ return allowed_mcp_servers
+
+
+def _http_detail_message(detail: object) -> str:
+ return str(detail.get("error")) if isinstance(detail, dict) and detail.get("error") else str(detail)
+
+
+def _server_answers_to(server: MCPServer, name: str) -> bool:
+ requested: Final = name.lower()
+ return any(requested == known.lower() for known in iter_known_server_prefixes(server) if known)
+
+
+async def raise_denied_scoped_mcp_access(
+ requested_names: Sequence[str],
+ user_api_key_auth: UserAPIKeyAuth | None,
+ client_ip: str | None = None,
+) -> None:
+ """A scoped request (``/mcp/`` path or ``x-mcp-servers`` header) resolved to zero
+ allowed servers, so the denial must be loud: a silent 200 with no tools reads as a healthy
+ server with no tools. Unknown, unauthorized, and access-group names all share one generic
+ error so scoping cannot probe which servers exist; the agent variant fires only when the
+ same request resolves once the agent binding is stripped, proving the binding caused the veto."""
+ agent_id: Final = user_api_key_auth.agent_id if user_api_key_auth else None
+ if user_api_key_auth is not None and agent_id:
+ resolved_without_agent: Final = await _get_allowed_mcp_servers(
+ user_api_key_auth=user_api_key_auth.model_copy(update=types.MappingProxyType({"agent_id": None})),
+ mcp_servers=requested_names,
+ client_ip=client_ip,
+ )
+
+ def _resolved_to_server(name: str) -> bool:
+ return any(_server_answers_to(server, name) for server in resolved_without_agent)
+
+ vetoed_server: Final = next((name for name in requested_names if _resolved_to_server(name)), None)
+ if vetoed_server is not None:
+ agent_denial: Final[_McpDeniedDetail] = {
+ "error": (
+ f"MCP server '{vetoed_server}' is not available to this key: the key is bound to "
+ f"agent '{agent_id}', whose MCP grants do not include this server. Add the server "
+ f"to the agent's object_permission.mcp_servers (edit the agent in the Admin UI or "
+ f"PATCH /v1/agents/{agent_id}), or use a key that is not bound to the agent."
+ )
+ }
+ raise HTTPException(status_code=403, detail=agent_denial)
+ vetoed_group: Final = next(
+ (
+ name
+ for name in requested_names
+ if not _resolved_to_server(name)
+ and any(name in (server.access_groups or ()) for server in resolved_without_agent)
+ ),
+ None,
+ )
+ if vetoed_group is not None:
+ group_denial: Final[_McpDeniedDetail] = {
+ "error": (
+ f"MCP access group '{vetoed_group}' is not available to this key: the key is bound to "
+ f"agent '{agent_id}', whose MCP grants do not include it. Add the group to the "
+ f"agent's object_permission.mcp_access_groups (edit the agent in the Admin UI or "
+ f"PATCH /v1/agents/{agent_id}), or use a key that is not bound to the agent."
+ )
+ }
+ raise HTTPException(status_code=403, detail=group_denial)
+ generic_denial: Final[_McpDeniedDetail] = {
+ "error": f"The key is not allowed to access the requested MCP servers: {', '.join(requested_names)}"
+ }
+ raise HTTPException(status_code=403, detail=generic_denial)
+
+
+def _tool_name_matches(tool_name: str, filter_list: list[str], mcp_server: MCPServer) -> bool:
+ """
+ Check if a tool name matches any name in the filter list.
+
+ Reads the same owner the server-level permission checks use, so discovery hides
+ exactly what dispatch refuses. ``mcp_server`` is required: guessing the boundary
+ at the first separator mismatches every tool on a server whose prefix contains
+ the separator.
+ """
+ bare_name: Final = strip_known_server_prefix(tool_name, mcp_server)
+ return match_known_tool_name(bare_name, mcp_server, filter_list) is not None
+
+
+def filter_tools_by_allowed_tools(
+ tools: list[MCPTool],
+ mcp_server: MCPServer,
+) -> list[MCPTool]:
+ """
+ Filter tools by allowed/disallowed tools configuration.
+
+ If allowed_tools is set, only tools in that list are returned.
+ If disallowed_tools is set, tools in that list are excluded.
+ Tool names are matched with and without server prefixes for flexibility.
+
+ Args:
+ tools: List of tools to filter
+ mcp_server: Server configuration with allowed_tools/disallowed_tools
+
+ Returns:
+ Filtered list of tools
+ """
+ from litellm.proxy._experimental.mcp_server.utils import (
+ server_applies_tool_allowlist,
+ )
+
+ tools_to_return = tools
+
+ # Filter by allowed_tools (whitelist)
+ if server_applies_tool_allowlist(mcp_server):
+ if not mcp_server.allowed_tools:
+ return []
+ tools_to_return = [
+ tool for tool in tools if _tool_name_matches(tool.name, mcp_server.allowed_tools, mcp_server)
+ ]
+
+ # Filter by disallowed_tools (blacklist)
+ if mcp_server.disallowed_tools:
+ tools_to_return = [
+ tool
+ for tool in tools_to_return
+ if not _tool_name_matches(tool.name, mcp_server.disallowed_tools, mcp_server)
+ ]
+
+ return tools_to_return
+
+
+def apply_tool_overrides(
+ tools: list[MCPTool],
+ mcp_server: MCPServer,
+) -> list[MCPTool]:
+ """Apply admin-configured display name/description overrides to tools.
+
+ Overrides are keyed by the unprefixed tool name, same convention as
+ allowed_tools configuration.
+ """
+ display_name_map: Final = mcp_server.tool_name_to_display_name or {}
+ description_map: Final = mcp_server.tool_name_to_description or {}
+ if not display_name_map and not description_map:
+ return tools
+
+ for tool in tools:
+ unprefixed = strip_known_server_prefix(tool.name, mcp_server)
+ lookup_key = unprefixed or tool.name
+ if lookup_key in display_name_map:
+ tool.name = display_name_map[lookup_key]
+ if lookup_key in description_map:
+ tool.description = description_map[lookup_key]
+ return tools
+
+
+async def _get_allowed_mcp_servers(
+ user_api_key_auth: UserAPIKeyAuth | None,
+ mcp_servers: Sequence[str] | None,
+ client_ip: str | None = None,
+) -> list[MCPServer]:
+ """Return allowed MCP servers for a request after applying filters.
+
+ Args:
+ user_api_key_auth: The authenticated user's API key info.
+ mcp_servers: Optional list of server names to filter to.
+ client_ip: Client IP for IP-based access control. If None, falls back to
+ auth context. Pass explicitly from request handlers for safety.
+ Note: If client_ip is None and auth context is not set, IP filtering is skipped.
+ This is intentional for internal callers but may indicate a bug if called
+ from a request handler without proper context setup.
+ """
+ allowed_mcp_server_ids = await global_mcp_server_manager.get_allowed_mcp_servers(user_api_key_auth)
+ (
+ allowed_mcp_server_ids,
+ _ip_blocked,
+ ) = global_mcp_server_manager.filter_server_ids_by_ip_with_info(allowed_mcp_server_ids, client_ip)
+ verbose_logger.debug(
+ "MCP IP filter: client_ip=%s, allowed_server_ids=%s",
+ client_ip,
+ allowed_mcp_server_ids,
+ )
+ if _ip_blocked > 0:
+ verbose_logger.debug(
+ "MCP IP filtering: %d server(s) are not accessible from client IP %s "
+ "because they are restricted to internal networks. "
+ "No tools from those servers will be returned. "
+ "To expose a server externally, set 'available_on_public_internet: true' "
+ "in its configuration.",
+ _ip_blocked,
+ client_ip,
+ )
+ allowed_mcp_servers: list[MCPServer] = []
+ for allowed_mcp_server_id in allowed_mcp_server_ids:
+ mcp_server = global_mcp_server_manager.get_mcp_server_by_id(allowed_mcp_server_id)
+ if mcp_server is not None:
+ # Apply the request-time oauth2_flow backstop for legacy null rows.
+ mcp_server = MCPServerManager.resolve_oauth2_flow_for_request(mcp_server)
+ allowed_mcp_servers.append(mcp_server)
+
+ if mcp_servers is not None:
+ allowed_mcp_servers = await _get_allowed_mcp_servers_from_mcp_server_names(
+ mcp_servers=mcp_servers,
+ allowed_mcp_servers=allowed_mcp_servers,
+ )
+
+ return allowed_mcp_servers
+
+
+def _client_has_per_server_auth_header(
+ server: MCPServer,
+ mcp_server_auth_headers: dict[str, dict[str, str]] | None,
+) -> bool:
+ """True if the request carries a per-server ``x-mcp-{alias}-authorization``
+ header for this server. This is the multi-server binding: it names one
+ upstream, so it is unambiguously the caller's upstream token regardless of
+ auth mode (never the LiteLLM admission credential).
+
+ Resolves through the same ``lookup_mcp_server_auth_in_headers`` egress uses, so
+ the connect gate and egress agree on which per-server header names match: a
+ dashboard client sends ``x-mcp-{sanitize_mcp_alias_for_header(alias)}-authorization``,
+ and matching only the raw alias here would 401 a token egress would forward.
+ """
+ if not mcp_server_auth_headers:
+ return False
+ from litellm.proxy._experimental.mcp_server.utils import (
+ lookup_mcp_server_auth_in_headers,
+ )
+
+ server_headers: Final = lookup_mcp_server_auth_in_headers(
+ mcp_server_auth_headers,
+ alias=server.alias,
+ server_name=server.server_name,
+ access_groups=server.access_groups,
+ )
+ if isinstance(server_headers, str):
+ return bool(server_headers.strip())
+ if isinstance(server_headers, dict):
+ return any(isinstance(hk, str) and hk.lower() == "authorization" for hk in server_headers)
+ return False
+
+
+def _client_has_passthrough_authorization(
+ server: MCPServer,
+ oauth2_headers: dict[str, str] | None,
+ mcp_server_auth_headers: dict[str, dict[str, str]] | None,
+) -> bool:
+ """True if the incoming request already carries an ``Authorization``
+ header the gateway will forward to this pass-through server.
+
+ The client may supply the bearer as either the top-level
+ ``Authorization`` header (surfaced via ``oauth2_headers``) or a
+ per-server ``x-mcp-auth-`` style header (surfaced via
+ ``mcp_server_auth_headers``). Either form skips the pre-emptive 401.
+ """
+ if oauth2_headers:
+ for k in oauth2_headers:
+ if k.lower() == "authorization":
+ return True
+ return _client_has_per_server_auth_header(server, mcp_server_auth_headers)
+
+
+async def _get_user_oauth_extra_headers_from_db(
+ server: MCPServer,
+ user_api_key_auth: UserAPIKeyAuth | None,
+ prefetched_creds: 'Mapping[str, "OAuthCredentialPayload"] | None' = None,
+) -> dict[str, str] | None:
+ """Stored OAuth2 token for (user, server) as an ``Authorization: Bearer`` header, or None.
+
+ Thin wrapper over ``resolve_user_oauth_access_token`` (Redis cache, else DB + refresh);
+ ``prefetched_creds`` skips the per-server Redis/DB lookups for the batch path.
+ """
+ if server.auth_type != MCPAuth.oauth2 or user_api_key_auth is None:
+ return None
+ from litellm.proxy._experimental.mcp_server.db import ( # noqa: PLC0415
+ resolve_user_oauth_access_token,
+ )
+
+ token: Final = await resolve_user_oauth_access_token(
+ getattr(user_api_key_auth, "user_id", None), server, prefetched_creds
+ )
+ return {"Authorization": f"Bearer {token}"} if token else None
+
+
+async def _prefetch_oauth_creds_for_user(
+ user_api_key_auth: UserAPIKeyAuth | None,
+) -> dict[str, "OAuthCredentialPayload"]:
+ """Fetch all OAuth2 credentials for the user in one DB query.
+
+ Returns a dict keyed by server_id to avoid N+1 queries in asyncio.gather loops.
+ """
+ user_id: Final[str | None] = getattr(user_api_key_auth, "user_id", None) if user_api_key_auth else None
+ if not user_id:
+ return {}
+ try:
+ from litellm.proxy._experimental.mcp_server.db import ( # noqa: PLC0415
+ list_user_oauth_credentials,
+ )
+ from litellm.proxy.utils import get_prisma_client_or_throw # noqa: PLC0415
+
+ prisma_client: Final = get_prisma_client_or_throw(
+ "Database not connected. Connect a database to use OAuth2 MCP tools."
+ )
+ creds: Final = await list_user_oauth_credentials(prisma_client, user_id)
+ return {c["server_id"]: c for c in creds if "server_id" in c}
+ except Exception as e:
+ verbose_logger.warning("_prefetch_oauth_creds_for_user: failed to prefetch for user=%s: %s", user_id, e)
+ return {}
+
+
+def _prepare_mcp_server_headers(
+ server: MCPServer,
+ mcp_server_auth_headers: dict[str, dict[str, str]] | None,
+ mcp_auth_header: str | None,
+ oauth2_headers: dict[str, str] | None,
+ raw_headers: dict[str, str] | None,
+ user_api_key_auth: UserAPIKeyAuth | None = None,
+ scope_servers: list[MCPServer] | None = None,
+) -> tuple[dict[str, str] | str | None, dict[str, str] | None]:
+ """Build auth and extra headers for a server.
+
+ ``scope_servers`` is the full server list a fan-out handler iterates. Passing it lets the
+ client-forwarded token modes withhold the caller's request-wide ``Authorization`` when
+ another server in the scope would also receive it (``_caller_authorization_fans_out``);
+ explicitly-addressed operations leave it None. Per-server ``x-mcp-{alias}-authorization``
+ headers are unaffected — they bind one token to one server and are the multi-server shape.
+ """
+ server_auth_header: dict[str, str] | str | None = None
+ if mcp_server_auth_headers:
+ from litellm.proxy._experimental.mcp_server.utils import (
+ lookup_mcp_server_auth_in_headers,
+ )
+
+ server_auth_header = lookup_mcp_server_auth_in_headers(
+ mcp_server_auth_headers,
+ alias=server.alias,
+ server_name=server.server_name,
+ access_groups=server.access_groups,
+ )
+
+ extra_headers: dict[str, str] | None = None
+ is_client_forwarded_mode: Final = server.is_client_forwarded_token
+ # In a multi-server listing scope the request-wide Authorization can only carry one token,
+ # so it is withheld from a client-forwarded server when another server in scope also consumes
+ # it (RFC 9700 cross-resource replay); such scopes must bind per-server via
+ # x-mcp-{alias}-authorization. The decision is computed once so BOTH the forwarding branch and
+ # the extra_headers copy loop below honor it — otherwise a server that lists Authorization in
+ # extra_headers would re-copy the withheld bearer from raw_headers and replay it anyway.
+ withhold_forwarded_authorization: Final = is_client_forwarded_mode and _caller_authorization_fans_out(
+ server, scope_servers
+ )
+ if server.auth_type == MCPAuth.oauth2:
+ # For OAuth2 M2M servers, upstream Authorization must come from
+ # client_credentials token fetch, never from caller headers.
+ if server.has_client_credentials:
+ extra_headers = None
+ else:
+ # Copy to avoid mutating the original dict (important for parallel fetching)
+ extra_headers = oauth2_headers.copy() if oauth2_headers else None
+ # Migrated authorization_code: the v2 resolver injects the stored per-user
+ # token, so drop the caller-forwarded Authorization (apply-if-absent would
+ # otherwise let it shadow the resolved token). Delegate keeps it. Centralized
+ # via _should_strip_caller_authorization to match _call_regular_mcp_tool.
+ if extra_headers and _should_strip_caller_authorization(
+ mcp_server=server,
+ raw_headers=raw_headers,
+ user_api_key_auth=user_api_key_auth,
+ ):
+ extra_headers = without_header(extra_headers, DEFAULT_CREDENTIAL_HEADER)
+ elif is_client_forwarded_mode:
+ if not withhold_forwarded_authorization:
+ extra_headers = _client_forwarded_authorization_headers(
+ mcp_server=server,
+ oauth2_headers=oauth2_headers,
+ raw_headers=raw_headers,
+ user_api_key_auth=user_api_key_auth,
+ )
+
+ if server.extra_headers and raw_headers:
+ if extra_headers is None:
+ extra_headers = {}
+
+ normalized_raw_headers: Final = {str(k).lower(): v for k, v in raw_headers.items() if isinstance(k, str)}
+
+ # Centralized strip decision shared with
+ # ``MCPServerManager._call_regular_mcp_tool`` so the two
+ # code paths cannot drift on this security-sensitive choice.
+ # See ``_should_strip_caller_authorization`` for the rules.
+ strip_caller_authorization: Final = _should_strip_caller_authorization(
+ mcp_server=server,
+ raw_headers=raw_headers,
+ user_api_key_auth=user_api_key_auth,
+ )
+
+ for header in server.extra_headers:
+ if not isinstance(header, str):
+ continue
+ if header.lower() == "authorization" and (strip_caller_authorization or withhold_forwarded_authorization):
+ continue
+ header_value = normalized_raw_headers.get(header.lower())
+ if header_value is None:
+ continue
+ extra_headers[header] = header_value
+
+ # Reset to None if no headers were actually added
+ if extra_headers is not None and len(extra_headers) == 0:
+ extra_headers = None
+
+ if server_auth_header is None:
+ server_auth_header = mcp_auth_header
+
+ return server_auth_header, extra_headers
+
+
+def _merge_gateway_initialize_instructions(
+ allowed_mcp_servers: list[MCPServer],
+) -> str | None:
+ """YAML/DB override, else upstream text (prefetch on init, or list_tools / health_check / call_tool cache)."""
+ if not allowed_mcp_servers:
+ return None
+
+ texts: Final[list[tuple[str, str]]] = []
+ for server in allowed_mcp_servers:
+ label = server.alias or server.server_name or server.name or server.server_id or "mcp"
+ if server.instructions and server.instructions.strip():
+ texts.append((label, server.instructions.strip()))
+ continue
+ if server.spec_path:
+ continue
+ cached = global_mcp_server_manager._upstream_initialize_instructions_by_server_id.get(server.server_id)
+ if cached and cached.strip():
+ texts.append((label, cached.strip()))
+
+ if not texts:
+ return None
+ if len(texts) == 1:
+ return texts[0][1]
+ return "\n\n---\n\n".join(f"[{lbl}]\n{txt}" for lbl, txt in texts)
+
+
+async def _raise_if_initialize_grants_no_mcp_servers(
+ allowed: Sequence[MCPServer],
+ user_api_key_auth: UserAPIKeyAuth | None,
+ mcp_servers: Sequence[str] | None,
+ client_ip: str | None,
+) -> None:
+ if allowed or user_api_key_auth is None or not user_api_key_auth.api_key:
+ return
+ if mcp_servers:
+ await raise_denied_scoped_mcp_access(
+ requested_names=mcp_servers,
+ user_api_key_auth=user_api_key_auth,
+ client_ip=client_ip,
+ )
+ no_servers_denial: Final[_McpDeniedDetail] = {
+ "error": (
+ "The key has no MCP servers granted, or none of its granted servers is loaded and allowed for "
+ "this client IP. Grant servers or access groups to the key, its team, or its organization "
+ "(object_permission.mcp_servers), check the server's allowed IPs, and reconnect."
+ )
+ }
+ raise HTTPException(status_code=403, detail=no_servers_denial)
+
+
+def _aggregate_server_key(server: MCPServer) -> str:
+ """The client-visible key for a server in listing outcomes and spend metadata: the same
+ display prefix (alias, or the short prefix when that mode is enabled) the caller already
+ sees on the tool names. Canonical internal server names never key a caller-readable
+ surface; when the display naming deliberately hides them, the outcome keys must too."""
+ return get_server_prefix(server) or "unknown"
+
+
+async def _get_tools_from_mcp_servers(
+ user_api_key_auth: UserAPIKeyAuth | None,
+ mcp_auth_header: str | None,
+ mcp_servers: list[str] | None,
+ mcp_server_auth_headers: dict[str, dict[str, str]] | None = None,
+ oauth2_headers: dict[str, str] | None = None,
+ raw_headers: dict[str, str] | None = None,
+ log_list_tools_to_spendlogs: bool = False,
+ list_tools_log_source: str | None = None,
+ litellm_trace_id: str | None = None,
+ request_tags: list[str] | None = None,
+ client_ip: str | None = None,
+ mcp_proxy_mode: bool = False,
+) -> AggregateToolListing:
+ """
+ Helper method to fetch tools from MCP servers based on server filtering criteria.
+
+ Args:
+ user_api_key_auth: User authentication info for access control
+ mcp_auth_header: Optional auth header for MCP server (deprecated)
+ mcp_servers: Optional list of server names/aliases to filter by
+ mcp_server_auth_headers: Optional dict of server-specific auth headers
+ oauth2_headers: Optional dict of oauth2 headers
+
+ Returns:
+ AggregateToolListing: Combined tools from filtered servers plus each server's
+ classified listing outcome
+ """
+
+ list_tools_start_time: Final = datetime.now()
+ litellm_logging_obj: LiteLLMLoggingObj | None = None
+ list_tools_request_data: dict[str, object] = {}
+
+ if log_list_tools_to_spendlogs:
+ # This is intentionally minimal: only async_success_handler / post_call_failure_hook
+ rules_obj: Final = Rules()
+ list_tools_call_id: Final = str(uuid.uuid4())
+ # Derive trace_id from raw_headers when not explicitly passed (same as A2A / MCP call_tool)
+ effective_litellm_trace_id: Final = litellm_trace_id or get_chain_id_from_headers(raw_headers)
+ spend_logs_metadata: Final[dict[str, object]] = {
+ "mcp_operation": "list_tools",
+ }
+ if isinstance(list_tools_log_source, str):
+ spend_logs_metadata["source"] = list_tools_log_source
+ if isinstance(mcp_servers, list):
+ spend_logs_metadata["requested_mcp_servers"] = mcp_servers
+
+ list_tools_request_data = {
+ "model": "MCP: list_tools",
+ "call_type": CallTypes.list_mcp_tools.value,
+ "litellm_call_id": list_tools_call_id,
+ "litellm_trace_id": effective_litellm_trace_id,
+ "metadata": {
+ "spend_logs_metadata": spend_logs_metadata,
+ "headers": logging_safe_mcp_headers(raw_headers),
+ **({"tags": request_tags} if request_tags else {}),
+ },
+ # Provide a small input payload for standard logging
+ "input": [
+ {
+ "role": "system",
+ "content": {
+ "mcp_operation": "list_tools",
+ "requested_mcp_servers": mcp_servers,
+ },
+ }
+ ],
+ }
+
+ # Attach user identifiers using the standard helper
+ if user_api_key_auth is not None:
+ LiteLLMProxyRequestSetup.add_user_api_key_auth_to_request_metadata(
+ data=list_tools_request_data,
+ user_api_key_dict=user_api_key_auth,
+ _metadata_variable_name="metadata",
+ )
+
+ user_identifier: Final = getattr(user_api_key_auth, "end_user_id", None) or getattr(
+ user_api_key_auth, "user_id", None
+ )
+ if user_identifier:
+ list_tools_request_data["user"] = user_identifier
+
+ try:
+ litellm_logging_obj, _ = function_setup(
+ original_function="list_mcp_tools",
+ is_async_call=False,
+ rules_obj=rules_obj,
+ start_time=list_tools_start_time,
+ **list_tools_request_data,
+ )
+ if litellm_logging_obj:
+ litellm_logging_obj.call_type = CallTypes.list_mcp_tools.value
+ litellm_logging_obj.model = "MCP: list_tools"
+ except Exception as logging_error:
+ verbose_logger.debug("Failed to initialize logging for MCP list_tools: %s", logging_error)
+ litellm_logging_obj = None
+
+ try:
+ allowed_mcp_servers: Final = await _get_allowed_mcp_servers(
+ user_api_key_auth=user_api_key_auth,
+ mcp_servers=mcp_servers,
+ client_ip=client_ip,
+ )
+ if mcp_servers and not allowed_mcp_servers:
+ await raise_denied_scoped_mcp_access(
+ requested_names=mcp_servers,
+ user_api_key_auth=user_api_key_auth,
+ client_ip=client_ip,
+ )
+
+ # Pre-fetch OAuth credentials only when at least one server uses OAuth2,
+ # to avoid an unnecessary DB round-trip on requests with no OAuth2 MCP servers.
+ _has_oauth2_server = any(getattr(s, "auth_type", None) == MCPAuth.oauth2 for s in allowed_mcp_servers)
+ _prefetched_oauth_creds: Final = (
+ await _prefetch_oauth_creds_for_user(user_api_key_auth) if _has_oauth2_server else {}
+ )
+
+ async def _fetch_and_filter_server_tools(
+ server: MCPServer,
+ ) -> "tuple[list[MCPTool], ServerOutcome]":
+ """Fetch and filter tools from a single server, classifying any failure into that
+ server's outcome so the aggregate can keep serving the healthy subset without a
+ broken server masquerading as an empty one."""
+ if server is None:
+ return [], ServerListOk(tool_count=0)
+
+ server_auth_header, extra_headers = _prepare_mcp_server_headers(
+ server=server,
+ mcp_server_auth_headers=mcp_server_auth_headers,
+ mcp_auth_header=mcp_auth_header,
+ oauth2_headers=oauth2_headers,
+ raw_headers=raw_headers,
+ user_api_key_auth=user_api_key_auth,
+ scope_servers=allowed_mcp_servers,
+ )
+
+ # Prefer server-stored per-user OAuth when configured, so a stale
+ # Authorization header from the MCP client cannot override Redis/DB
+ # (same issue as call_tool in mcp_server_manager: VS Code caches tokens).
+ from litellm.proxy._experimental.mcp_server.outbound_credentials.adapter import ( # noqa: PLC0415
+ to_server_spec,
+ )
+
+ # A server migrated to the v2 resolver gets its token from the resolver at connect
+ # time; building it here would double-resolve and be shadowed by the v2 graft. The
+ # preemptive 401 already challenged a missing token, so one exists for the connect.
+ migrated_to_v2: Final = to_server_spec(server) is not None
+ if (
+ not migrated_to_v2
+ and server.auth_type == MCPAuth.oauth2
+ and getattr(server, "needs_user_oauth_token", False)
+ and user_api_key_auth is not None
+ ):
+ db_headers: Final = await _get_user_oauth_extra_headers_from_db(
+ server,
+ user_api_key_auth,
+ prefetched_creds=_prefetched_oauth_creds,
+ )
+ if db_headers:
+ extra_headers = db_headers
+
+ # If still no OAuth2 token, fall back to pre-fetched creds (non-stale-client path)
+ elif not migrated_to_v2 and extra_headers is None and server.auth_type == MCPAuth.oauth2:
+ extra_headers = await _get_user_oauth_extra_headers_from_db(
+ server,
+ user_api_key_auth,
+ prefetched_creds=_prefetched_oauth_creds,
+ )
+
+ if server.is_byok and server.auth_type != MCPAuth.oauth2 and server_auth_header is None:
+ server_auth_header = await _get_byok_credential(server, user_api_key_auth)
+
+ try:
+ tools: Final = await global_mcp_server_manager._get_tools_from_server(
+ server=server,
+ mcp_auth_header=server_auth_header,
+ extra_headers=extra_headers,
+ add_prefix=True, # Always add server prefix
+ raw_headers=raw_headers,
+ client_ip=client_ip,
+ user_api_key_auth=user_api_key_auth,
+ oauth2_headers=oauth2_headers,
+ )
+ filtered_tools = filter_tools_by_allowed_tools(tools, server)
+
+ filtered_tools = await filter_tools_by_key_team_permissions(
+ tools=filtered_tools,
+ server_id=server.server_id,
+ user_api_key_auth=user_api_key_auth,
+ )
+
+ if mcp_proxy_mode:
+ from litellm.proxy._experimental.mcp_server.tool_search import with_mcp_proxy_identity
+
+ filtered_tools = [ # mutable-ok: MCP tool pipeline
+ with_mcp_proxy_identity(tool, server.server_id) for tool in filtered_tools
+ ]
+ else:
+ filtered_tools = apply_tool_overrides(filtered_tools, server)
+
+ verbose_logger.debug(
+ "Successfully fetched %s tools from server %s, %s after filtering",
+ len(tools),
+ server.name,
+ len(filtered_tools),
+ )
+ return filtered_tools, ServerListOk(tool_count=len(filtered_tools))
+ except MCPUpstreamAuthError as e:
+ # Absorb so one unauthenticated server does not empty every other server's
+ # tools. Surfacing the upstream 401 to the client as a re-auth challenge is
+ # intentionally not done here: raising from this list handler cannot produce a
+ # 401 + WWW-Authenticate (the MCP session manager serializes it as a JSON-RPC
+ # error). Single-server routes surface it via the request-scope preemptive
+ # check in _raise_preemptive_401_for_unauthenticated_servers instead.
+ verbose_logger.debug("MCP list_tools: omitting %s; it needs upstream auth", server.name)
+ return [], classify_list_exception(e)
+ except Exception as e:
+ verbose_logger.exception("Error getting tools from server %s: %s", server.name, e)
+ return [], classify_list_exception(e)
+
+ # Fetch tools from all servers in parallel
+ tasks: Final = [_fetch_and_filter_server_tools(server) for server in allowed_mcp_servers]
+ results: Final = await asyncio.gather(*tasks)
+
+ # Flatten results into single list
+ all_tools: Final[list[MCPTool]] = [tool for tools, _ in results for tool in tools]
+ server_outcomes: Final[dict[str, ServerOutcome]] = {
+ _aggregate_server_key(server): outcome
+ for server, (_, outcome) in zip(allowed_mcp_servers, results)
+ if server is not None
+ }
+
+ # If logging is enabled, enrich spend_logs_metadata with counts
+ if litellm_logging_obj:
+ per_server_tool_counts: Final[dict[str, int]] = {
+ _aggregate_server_key(server): len(server_tools)
+ for server, (server_tools, _) in zip(allowed_mcp_servers, results)
+ if server is not None
+ }
+
+ metadata_dict: Final = litellm_logging_obj.model_call_details.get("metadata")
+ if isinstance(metadata_dict, dict):
+ spend_meta = metadata_dict.get("spend_logs_metadata")
+ if not isinstance(spend_meta, dict):
+ spend_meta = {}
+ metadata_dict["spend_logs_metadata"] = spend_meta
+ spend_meta["allowed_server_count"] = len(allowed_mcp_servers)
+ spend_meta["tool_count_total"] = len(all_tools)
+ spend_meta["per_server_tool_counts"] = per_server_tool_counts
+ spend_meta["per_server_list_outcomes"] = {
+ key: outcome_wire_value(outcome) for key, outcome in server_outcomes.items()
+ }
+
+ end_time: Final = datetime.now()
+ try:
+ await litellm_logging_obj.async_success_handler(
+ result=[tool.model_dump(mode="json") if isinstance(tool, MCPTool) else tool for tool in all_tools],
+ start_time=list_tools_start_time,
+ end_time=end_time,
+ )
+ except Exception as log_exc:
+ # list_tools responses must not be dropped due to non-blocking
+ # observability/serialization failures.
+ verbose_logger.warning(
+ "MCP list_tools success logging failed (continuing): %s",
+ log_exc,
+ )
+
+ verbose_logger.info("Successfully fetched %s tools total from all MCP servers", len(all_tools))
+
+ return AggregateToolListing(tools=all_tools, outcomes=server_outcomes)
+ except Exception as e:
+ # Only fire failure hook if logging was requested for this list-tools execution
+ if log_list_tools_to_spendlogs and user_api_key_auth is not None:
+ try:
+ from litellm.proxy.proxy_server import proxy_logging_obj
+
+ if proxy_logging_obj:
+ traceback_str: Final = traceback.format_exc(limit=MAXIMUM_TRACEBACK_LINES_TO_LOG)
+ await proxy_logging_obj.post_call_failure_hook(
+ request_data=list_tools_request_data or {},
+ original_exception=e,
+ user_api_key_dict=user_api_key_auth,
+ route="/mcp/list_tools",
+ traceback_str=traceback_str,
+ )
+ except Exception:
+ verbose_logger.debug("Failed to log MCP list_tools failure via post_call_failure_hook")
+ raise
+
+
+async def _get_prompts_from_mcp_servers(
+ user_api_key_auth: UserAPIKeyAuth | None,
+ mcp_auth_header: str | None,
+ mcp_servers: list[str] | None,
+ mcp_server_auth_headers: dict[str, dict[str, str]] | None = None,
+ oauth2_headers: dict[str, str] | None = None,
+ raw_headers: dict[str, str] | None = None,
+ client_ip: str | None = None,
+) -> list[Prompt]:
+ """
+ Helper method to fetch prompt from MCP servers based on server filtering criteria.
+
+ Args:
+ user_api_key_auth: User authentication info for access control
+ mcp_auth_header: Optional auth header for MCP server (deprecated)
+ mcp_servers: Optional list of server names/aliases to filter by
+ mcp_server_auth_headers: Optional dict of server-specific auth headers
+ oauth2_headers: Optional dict of oauth2 headers
+
+ Returns:
+ List[Prompt]: Combined list of prompts from filtered servers
+ """
+
+ allowed_mcp_servers: Final = await _get_allowed_mcp_servers(
+ user_api_key_auth=user_api_key_auth,
+ mcp_servers=mcp_servers,
+ client_ip=client_ip,
+ )
+
+ # Get prompts from each allowed server
+ all_prompts: Final = []
+ for server in allowed_mcp_servers:
+ if server is None:
+ continue
+
+ server_auth_header, extra_headers = _prepare_mcp_server_headers(
+ server=server,
+ mcp_server_auth_headers=mcp_server_auth_headers,
+ mcp_auth_header=mcp_auth_header,
+ oauth2_headers=oauth2_headers,
+ raw_headers=raw_headers,
+ user_api_key_auth=user_api_key_auth,
+ scope_servers=allowed_mcp_servers,
+ )
+
+ try:
+ prompts = await global_mcp_server_manager.get_prompts_from_server(
+ server=server,
+ user_api_key_auth=user_api_key_auth,
+ mcp_auth_header=server_auth_header,
+ extra_headers=extra_headers,
+ add_prefix=True, # Always add server prefix
+ raw_headers=raw_headers,
+ client_ip=client_ip,
+ )
+
+ all_prompts.extend(prompts)
+
+ verbose_logger.debug("Successfully fetched %s prompts from server %s", len(prompts), server.name)
+ except Exception as e:
+ verbose_logger.exception("Error getting prompts from server %s: %s", server.name, e)
+ # Continue with other servers instead of failing completely
+
+ verbose_logger.info("Successfully fetched %s prompts total from all MCP servers", len(all_prompts))
+
+ return all_prompts
+
+
+async def _get_resources_from_mcp_servers(
+ user_api_key_auth: UserAPIKeyAuth | None,
+ mcp_auth_header: str | None,
+ mcp_servers: list[str] | None,
+ mcp_server_auth_headers: dict[str, dict[str, str]] | None = None,
+ oauth2_headers: dict[str, str] | None = None,
+ raw_headers: dict[str, str] | None = None,
+ client_ip: str | None = None,
+) -> list[Resource]:
+ """Fetch resources from allowed MCP servers."""
+
+ allowed_mcp_servers: Final = await _get_allowed_mcp_servers(
+ user_api_key_auth=user_api_key_auth,
+ mcp_servers=mcp_servers,
+ client_ip=client_ip,
+ )
+
+ all_resources: Final[list[Resource]] = []
+ for server in allowed_mcp_servers:
+ if server is None:
+ continue
+
+ server_auth_header, extra_headers = _prepare_mcp_server_headers(
+ server=server,
+ mcp_server_auth_headers=mcp_server_auth_headers,
+ mcp_auth_header=mcp_auth_header,
+ oauth2_headers=oauth2_headers,
+ raw_headers=raw_headers,
+ user_api_key_auth=user_api_key_auth,
+ scope_servers=allowed_mcp_servers,
+ )
+
+ try:
+ resources = await global_mcp_server_manager.get_resources_from_server(
+ server=server,
+ user_api_key_auth=user_api_key_auth,
+ mcp_auth_header=server_auth_header,
+ extra_headers=extra_headers,
+ add_prefix=True, # Always add server prefix
+ raw_headers=raw_headers,
+ client_ip=client_ip,
+ )
+ all_resources.extend(resources)
+
+ verbose_logger.debug("Successfully fetched %s resources from server %s", len(resources), server.name)
+ except Exception as e:
+ verbose_logger.exception("Error getting resources from server %s: %s", server.name, e)
+
+ verbose_logger.info("Successfully fetched %s resources total from all MCP servers", len(all_resources))
+
+ return all_resources
+
+
+async def _get_resource_templates_from_mcp_servers(
+ user_api_key_auth: UserAPIKeyAuth | None,
+ mcp_auth_header: str | None,
+ mcp_servers: list[str] | None,
+ mcp_server_auth_headers: dict[str, dict[str, str]] | None = None,
+ oauth2_headers: dict[str, str] | None = None,
+ raw_headers: dict[str, str] | None = None,
+ client_ip: str | None = None,
+) -> list[ResourceTemplate]:
+ """Fetch resource templates from allowed MCP servers."""
+
+ allowed_mcp_servers: Final = await _get_allowed_mcp_servers(
+ user_api_key_auth=user_api_key_auth,
+ mcp_servers=mcp_servers,
+ client_ip=client_ip,
+ )
+
+ all_resource_templates: Final[list[ResourceTemplate]] = []
+ for server in allowed_mcp_servers:
+ if server is None:
+ continue
+
+ server_auth_header, extra_headers = _prepare_mcp_server_headers(
+ server=server,
+ mcp_server_auth_headers=mcp_server_auth_headers,
+ mcp_auth_header=mcp_auth_header,
+ oauth2_headers=oauth2_headers,
+ raw_headers=raw_headers,
+ user_api_key_auth=user_api_key_auth,
+ scope_servers=allowed_mcp_servers,
+ )
+
+ try:
+ resource_templates = await global_mcp_server_manager.get_resource_templates_from_server(
+ server=server,
+ user_api_key_auth=user_api_key_auth,
+ mcp_auth_header=server_auth_header,
+ extra_headers=extra_headers,
+ add_prefix=True, # Always add server prefix
+ raw_headers=raw_headers,
+ client_ip=client_ip,
+ )
+ all_resource_templates.extend(resource_templates)
+ verbose_logger.debug(
+ "Successfully fetched %s resource templates from server %s",
+ len(resource_templates),
+ server.name,
+ )
+ except Exception as e:
+ verbose_logger.exception(
+ "Error getting resource templates from server %s: %s",
+ server.name,
+ str(e),
+ )
+
+ verbose_logger.info(
+ "Successfully fetched %s resource templates total from all MCP servers",
+ len(all_resource_templates),
+ )
+
+ return all_resource_templates
+
+
+async def filter_tools_by_key_team_permissions(
+ tools: list[MCPTool],
+ server_id: str,
+ user_api_key_auth: UserAPIKeyAuth | None,
+) -> list[MCPTool]:
+ """
+ Filter tools based on key/team mcp_tool_permissions.
+
+ Note: Tool names in the DB are stored without server prefixes,
+ but tool names from MCP servers are prefixed. We need to strip
+ the prefix before comparing.
+ """
+ # Filter by key/team tool-level permissions
+ allowed_tool_names: Final = await MCPRequestHandler.get_allowed_tools_for_server(
+ server_id=server_id,
+ user_api_key_auth=user_api_key_auth,
+ )
+
+ # Tools arrive prefixed with the server's own prefix; strip exactly that
+ # prefix (resolved from the server) rather than the first separator, so a
+ # prefix containing the separator still reduces to the stored bare name.
+ server: Final = global_mcp_server_manager.get_mcp_server_by_id(server_id)
+ return [
+ t
+ for t in tools
+ if MCPRequestHandler.tool_is_granted(strip_known_server_prefix(t.name, server), allowed_tool_names)
+ ]
+
+
+async def _list_mcp_tools(
+ user_api_key_auth: UserAPIKeyAuth | None = None,
+ mcp_auth_header: str | None = None,
+ mcp_servers: list[str] | None = None,
+ mcp_server_auth_headers: dict[str, dict[str, str]] | None = None,
+ oauth2_headers: dict[str, str] | None = None,
+ raw_headers: dict[str, str] | None = None,
+ log_list_tools_to_spendlogs: bool = False,
+ list_tools_log_source: str | None = None,
+ client_ip: str | None = None,
+ mcp_proxy_mode: bool = False,
+) -> AggregateToolListing:
+ """
+ List all available MCP tools.
+
+ Args:
+ user_api_key_auth: User authentication info for access control
+ mcp_auth_header: Optional auth header for MCP server (deprecated)
+ mcp_servers: Optional list of server names/aliases to filter by
+ mcp_server_auth_headers: Optional dict of server-specific auth headers {server_alias: auth_value}
+ client_ip: Client IP for IP-based server access control
+
+ Returns:
+ AggregateToolListing: Combined tools from all accessible servers plus each server's
+ classified listing outcome
+ """
+
+ try:
+ listing: Final = await _get_tools_from_mcp_servers(
+ user_api_key_auth=user_api_key_auth,
+ mcp_auth_header=mcp_auth_header,
+ mcp_servers=mcp_servers,
+ mcp_server_auth_headers=mcp_server_auth_headers,
+ oauth2_headers=oauth2_headers,
+ raw_headers=raw_headers,
+ log_list_tools_to_spendlogs=log_list_tools_to_spendlogs,
+ list_tools_log_source=list_tools_log_source,
+ client_ip=client_ip,
+ mcp_proxy_mode=mcp_proxy_mode,
+ )
+ verbose_logger.debug("Successfully fetched %s tools from managed MCP servers", len(listing.tools))
+ return listing
+ except HTTPException:
+ raise
+ except Exception as e:
+ verbose_logger.exception("Error getting tools from managed MCP servers: %s", e)
+ # Continue with an empty listing instead of failing completely
+ return AggregateToolListing(tools=[], outcomes={})
+
+
+async def _list_mcp_prompts(
+ user_api_key_auth: UserAPIKeyAuth | None = None,
+ mcp_auth_header: str | None = None,
+ mcp_servers: list[str] | None = None,
+ mcp_server_auth_headers: dict[str, dict[str, str]] | None = None,
+ oauth2_headers: dict[str, str] | None = None,
+ raw_headers: dict[str, str] | None = None,
+ client_ip: str | None = None,
+) -> list[Prompt]:
+ """
+ List all available MCP prompts.
+
+ Args:
+ user_api_key_auth: User authentication info for access control
+ mcp_auth_header: Optional auth header for MCP server (deprecated)
+ mcp_servers: Optional list of server names/aliases to filter by
+ mcp_server_auth_headers: Optional dict of server-specific auth headers {server_alias: auth_value}
+
+ Returns:
+ List[Prompt]: Combined list of tools from all accessible servers
+ """
+ # Get tools from managed MCP servers with error handling
+ managed_prompts = []
+ try:
+ managed_prompts = await _get_prompts_from_mcp_servers(
+ user_api_key_auth=user_api_key_auth,
+ mcp_auth_header=mcp_auth_header,
+ mcp_servers=mcp_servers,
+ mcp_server_auth_headers=mcp_server_auth_headers,
+ oauth2_headers=oauth2_headers,
+ raw_headers=raw_headers,
+ client_ip=client_ip,
+ )
+ verbose_logger.debug("Successfully fetched %s prompts from managed MCP servers", len(managed_prompts))
+ except Exception as e:
+ verbose_logger.exception("Error getting tools from managed MCP servers: %s", e)
+ # Continue with empty managed tools list instead of failing completely
+
+ return managed_prompts
+
+
+async def _list_mcp_resources(
+ user_api_key_auth: UserAPIKeyAuth | None = None,
+ mcp_auth_header: str | None = None,
+ mcp_servers: list[str] | None = None,
+ mcp_server_auth_headers: dict[str, dict[str, str]] | None = None,
+ oauth2_headers: dict[str, str] | None = None,
+ raw_headers: dict[str, str] | None = None,
+ client_ip: str | None = None,
+) -> list[Resource]:
+ """List all available MCP resources."""
+
+ managed_resources: list[Resource] = []
+ try:
+ managed_resources = await _get_resources_from_mcp_servers(
+ user_api_key_auth=user_api_key_auth,
+ mcp_auth_header=mcp_auth_header,
+ mcp_servers=mcp_servers,
+ mcp_server_auth_headers=mcp_server_auth_headers,
+ oauth2_headers=oauth2_headers,
+ raw_headers=raw_headers,
+ client_ip=client_ip,
+ )
+ verbose_logger.debug("Successfully fetched %s resources from managed MCP servers", len(managed_resources))
+ except Exception as e:
+ verbose_logger.exception("Error getting resources from managed MCP servers: %s", e)
+
+ return managed_resources
+
+
+async def _list_mcp_resource_templates(
+ user_api_key_auth: UserAPIKeyAuth | None = None,
+ mcp_auth_header: str | None = None,
+ mcp_servers: list[str] | None = None,
+ mcp_server_auth_headers: dict[str, dict[str, str]] | None = None,
+ oauth2_headers: dict[str, str] | None = None,
+ raw_headers: dict[str, str] | None = None,
+ client_ip: str | None = None,
+) -> list[ResourceTemplate]:
+ """List all available MCP resource templates."""
+
+ managed_resource_templates: list[ResourceTemplate] = []
+ try:
+ managed_resource_templates = await _get_resource_templates_from_mcp_servers(
+ user_api_key_auth=user_api_key_auth,
+ mcp_auth_header=mcp_auth_header,
+ mcp_servers=mcp_servers,
+ mcp_server_auth_headers=mcp_server_auth_headers,
+ oauth2_headers=oauth2_headers,
+ raw_headers=raw_headers,
+ client_ip=client_ip,
+ )
+ verbose_logger.debug(
+ "Successfully fetched %s resource templates from managed MCP servers",
+ len(managed_resource_templates),
+ )
+ except Exception as e:
+ verbose_logger.exception(
+ "Error getting resource templates from managed MCP servers: %s",
+ str(e),
+ )
+
+ return managed_resource_templates
+
+
+def _resolve_display_name_to_original(
+ name: str,
+ allowed_mcp_servers: list[MCPServer],
+) -> str:
+ """Translate a display-name override back to the original prefixed tool name.
+
+ When a client received a customised display name from tools/list (e.g.
+ "Get Pet") it will call tools/call with that same string. We need to
+ reverse-map it to the original prefixed name (e.g.
+ "petstore_mcp-getPetById") before any routing or permission logic runs.
+ """
+ for server in allowed_mcp_servers:
+ display_map = server.tool_name_to_display_name or {}
+ for unprefixed_name, display_name in display_map.items():
+ if display_name == name:
+ return add_server_prefix_to_name(unprefixed_name, get_server_prefix(server))
+ return name
+
+
+async def _get_byok_credential(
+ mcp_server: MCPServer,
+ user_api_key_auth: UserAPIKeyAuth | None,
+) -> str | None:
+ """Retrieve the stored BYOK credential for a user+server pair, served from the worker cache within its TTL."""
+ if not mcp_server.is_byok:
+ return None
+ user_id: Final = (user_api_key_auth.user_id if user_api_key_auth else None) or ""
+ if not user_id:
+ return None
+
+ cached: Final = get_cached_byok_credential(user_id, mcp_server.server_id)
+ if cached is not None:
+ return cached.credential
+
+ from litellm.proxy._experimental.mcp_server.db import get_user_credential
+ from litellm.proxy.proxy_server import prisma_client
+
+ if prisma_client is None:
+ return None
+ credential: Final = await get_user_credential(
+ prisma_client=prisma_client,
+ user_id=user_id,
+ server_id=mcp_server.server_id,
+ )
+ cache_byok_credential(user_id, mcp_server.server_id, credential)
+ return credential
+
+
+async def _check_byok_credential(
+ mcp_server: MCPServer,
+ user_api_key_auth: UserAPIKeyAuth | None,
+) -> None:
+ """
+ If the MCP server is BYOK-enabled, verify that the requesting user has a
+ stored credential. When no credential is found, raise an HTTP 401 with a
+ WWW-Authenticate header that points the MCP client to our OAuth metadata
+ endpoint so it can drive the authorization flow.
+ """
+ if not mcp_server.is_byok:
+ return
+
+ user_id: Final = (user_api_key_auth.user_id if user_api_key_auth else None) or ""
+ if not user_id:
+ raise HTTPException(
+ status_code=401,
+ detail={
+ "error": "byok_auth_required",
+ "server_id": mcp_server.server_id,
+ "server_name": mcp_server.server_name or mcp_server.name,
+ "message": "User identity is required for BYOK servers",
+ },
+ headers={"WWW-Authenticate": get_byok_www_authenticate()},
+ )
+
+ cached: Final = get_cached_byok_credential(user_id, mcp_server.server_id)
+ if cached is not None:
+ if cached.credential is None:
+ raise HTTPException(
+ status_code=401,
+ detail={
+ "error": "byok_auth_required",
+ "server_id": mcp_server.server_id,
+ "server_name": mcp_server.server_name or mcp_server.name,
+ "message": (
+ "No stored credential found for this BYOK server. "
+ "Complete the OAuth authorization flow to provide your API key."
+ ),
+ },
+ headers={"WWW-Authenticate": get_byok_www_authenticate()},
+ )
+ return
+
+ from litellm.proxy._experimental.mcp_server.db import get_user_credential
+ from litellm.proxy.proxy_server import prisma_client
+
+ if prisma_client is None:
+ # Fail closed on DB unavailability: returning here previously
+ # bypassed the ownership check and let any proxy-authenticated
+ # caller invoke BYOK tools during outage windows.
+ raise HTTPException(
+ status_code=503,
+ detail={
+ "error": "byok_auth_unavailable",
+ "server_id": mcp_server.server_id,
+ "server_name": mcp_server.server_name or mcp_server.name,
+ "message": "BYOK credential check requires a database connection.",
+ },
+ )
+
+ credential: Final = await get_user_credential(
+ prisma_client=prisma_client,
+ user_id=user_id,
+ server_id=mcp_server.server_id,
+ )
+ cache_byok_credential(user_id, mcp_server.server_id, credential)
+ if credential is None:
+ raise HTTPException(
+ status_code=401,
+ detail={
+ "error": "byok_auth_required",
+ "server_id": mcp_server.server_id,
+ "server_name": mcp_server.server_name or mcp_server.name,
+ "message": (
+ "No stored credential found for this BYOK server. "
+ "Complete the OAuth authorization flow to provide your API key."
+ ),
+ },
+ headers={"WWW-Authenticate": get_byok_www_authenticate()},
+ )
+
+
+async def _list_tools_before_first_call(
+ server: MCPServer | None,
+ tool_name: str,
+ allowed_mcp_servers: list[MCPServer],
+ user_api_key_auth: UserAPIKeyAuth | None,
+ mcp_auth_header: str | None,
+ mcp_server_auth_headers: dict[str, dict[str, str]] | None,
+ oauth2_headers: dict[str, str] | None,
+ raw_headers: dict[str, str] | None,
+ client_ip: str | None = None,
+) -> None:
+ """List ``server`` with the caller's own credentials when it does not yet expose ``tool_name`` here.
+
+ The startup fill skips a server whose upstream wants the caller's token, and mcp 2 no
+ longer lists before an uncached tools/call, so a worker that has not served tools/list
+ for this caller would otherwise answer 404 for a tool the caller can see. Gating on the
+ requested tool, not on any prior listing, keeps callers with different upstream catalogs
+ from masking each other.
+ """
+ if server is None or global_mcp_server_manager.server_exposes_tool(server, tool_name):
+ return
+ if all(allowed.server_id != server.server_id for allowed in allowed_mcp_servers):
+ return
+ try:
+ await _get_tools_from_mcp_servers(
+ user_api_key_auth=user_api_key_auth,
+ mcp_auth_header=mcp_auth_header,
+ mcp_servers=[server.server_id],
+ mcp_server_auth_headers=mcp_server_auth_headers,
+ oauth2_headers=oauth2_headers,
+ raw_headers=raw_headers,
+ client_ip=client_ip,
+ )
+ except Exception as e: # noqa: BLE001 # best effort: resolution below answers as it did before
+ verbose_logger.debug("MCP tools/call: listing %s before its first call failed: %s", server.name, e)
+
+
+async def execute_mcp_tool(
+ name: str,
+ arguments: dict[str, object],
+ allowed_mcp_servers: list[MCPServer],
+ start_time: datetime,
+ user_api_key_auth: UserAPIKeyAuth | None = None,
+ mcp_auth_header: str | None = None,
+ mcp_server_auth_headers: dict[str, dict[str, str]] | None = None,
+ oauth2_headers: dict[str, str] | None = None,
+ raw_headers: dict[str, str] | None = None,
+ host_progress_callback: ProgressCallback | None = None,
+ guardrail_context: Mapping[str, object] | None = None,
+ client_ip: str | None = None,
+ **kwargs: object, # kwargs-ok: preserves the existing REST and decorated logging call contract
+) -> CallToolResult:
+ context: Final = prepare_context(
+ user_api_key_auth=user_api_key_auth,
+ mcp_auth_header=mcp_auth_header,
+ mcp_server_auth_headers=mcp_server_auth_headers,
+ oauth2_headers=oauth2_headers,
+ raw_headers=raw_headers,
+ client_ip=client_ip,
+ )
+ operation: Final = AuthorizedToolCall(
+ name=name,
+ arguments=arguments,
+ allowed_mcp_servers=tuple(allowed_mcp_servers),
+ start_time=start_time,
+ host_progress_callback=host_progress_callback,
+ guardrail_context=guardrail_context,
+ logging_data=types.MappingProxyType(kwargs),
+ )
+ return await GatewayOperations().execute(operation, context)
+
+
+async def _execute_mcp_tool(
+ name: str,
+ arguments: dict[str, object],
+ allowed_mcp_servers: list[MCPServer],
+ start_time: datetime,
+ user_api_key_auth: UserAPIKeyAuth | None = None,
+ mcp_auth_header: str | None = None,
+ mcp_server_auth_headers: dict[str, dict[str, str]] | None = None,
+ oauth2_headers: dict[str, str] | None = None,
+ raw_headers: dict[str, str] | None = None,
+ host_progress_callback: ProgressCallback | None = None,
+ guardrail_context: Mapping[str, object] | None = None,
+ client_ip: str | None = None,
+ **kwargs: Any,
+) -> CallToolResult:
+ """
+ Execute MCP tool.
+
+ This function assumes permission checks have already been performed.
+
+ Args:
+ name: Tool name (may include server prefix)
+ arguments: Tool arguments
+ allowed_mcp_servers: Pre-validated list of servers the user can access
+ start_time: Start time for logging
+ user_api_key_auth: Optional user API key auth for logging
+ mcp_auth_header: Optional MCP auth header
+ mcp_server_auth_headers: Optional server-specific auth headers
+ oauth2_headers: Optional OAuth2 headers
+ raw_headers: Optional raw HTTP headers
+ **kwargs: Additional arguments (e.g., litellm_logging_obj)
+
+ Returns:
+ CallToolResult: Tool execution result
+ """
+ # Track resolved MCP server for both permission checks and dispatch
+ mcp_server: MCPServer | None = None
+ requested_server_id: Final[str | None] = kwargs.get("requested_server_id")
+
+ # If the client called with a display-name override (e.g. "Get Pet"),
+ # translate it back to the original prefixed name before any routing.
+ name = _resolve_display_name_to_original(name, allowed_mcp_servers)
+
+ # Remove prefix from tool name for logging and processing
+ original_tool_name, server_name = split_server_prefix_from_name(name)
+
+ requested_server: MCPServer | None = None
+ if requested_server_id:
+ requested_server = next(
+ (s for s in allowed_mcp_servers if s.server_id == requested_server_id),
+ None,
+ )
+
+ name_is_prefixed = False
+ if requested_server is not None and MCP_TOOL_PREFIX_SEPARATOR in name:
+ all_registry_prefixes: Final[set[str]] = set()
+ for registry_server in global_mcp_server_manager.get_registry().values():
+ for known_prefix in iter_known_server_prefixes(registry_server):
+ all_registry_prefixes.add(normalize_server_name(known_prefix))
+ name_is_prefixed = is_tool_name_prefixed(name, known_server_prefixes=all_registry_prefixes)
+
+ first_call_target: Final = (
+ requested_server
+ if requested_server is not None and not name_is_prefixed
+ else global_mcp_server_manager.server_owning_tool_name_prefix(name)
+ )
+ first_call_tool_name: Final = (
+ name
+ if first_call_target is None or (requested_server is not None and not name_is_prefixed)
+ else strip_known_server_prefix(name, first_call_target)
+ )
+ await _list_tools_before_first_call(
+ server=first_call_target,
+ tool_name=first_call_tool_name,
+ allowed_mcp_servers=allowed_mcp_servers,
+ user_api_key_auth=user_api_key_auth,
+ mcp_auth_header=mcp_auth_header,
+ mcp_server_auth_headers=mcp_server_auth_headers,
+ oauth2_headers=oauth2_headers,
+ raw_headers=raw_headers,
+ client_ip=client_ip,
+ )
+
+ if requested_server is not None and not name_is_prefixed:
+ # REST callers may pass server_id with the upstream tool name (no
+ # LiteLLM prefix). The first segment is not a registered server
+ # prefix, so the whole string is the upstream tool name and may
+ # legitimately contain the separator (e.g. "text-to-speech").
+ # server_id is authoritative for routing and auth.
+ mcp_server = requested_server
+ server_name = requested_server.name
+ original_tool_name = name
+ else:
+ # Resolve from tool name (MCP JSON-RPC or prefixed REST tool names).
+ mcp_server = global_mcp_server_manager._get_mcp_server_from_tool_name(name)
+ if mcp_server is None and requested_server is not None:
+ for known_prefix in iter_known_server_prefixes(requested_server):
+ candidate = global_mcp_server_manager._get_mcp_server_from_tool_name(
+ add_server_prefix_to_name(name, known_prefix)
+ )
+ if candidate is not None:
+ mcp_server = candidate
+ break
+ if mcp_server is not None:
+ server_name = mcp_server.name
+ original_tool_name = strip_known_server_prefix(name, mcp_server)
+
+ if requested_server is not None:
+ if mcp_server is not None and mcp_server.server_id != requested_server.server_id:
+ raise HTTPException(
+ status_code=403,
+ detail={
+ "error": "tool_server_mismatch",
+ "message": (
+ f"Tool '{name}' belongs to MCP server "
+ f"'{mcp_server.name}' but request specified "
+ f"server_id for '{requested_server.name}'."
+ ),
+ },
+ )
+ if mcp_server is None:
+ mcp_server = requested_server
+ server_name = requested_server.name
+ original_tool_name = strip_known_server_prefix(name, requested_server)
+
+ # Only enforce server-level permissions when we can resolve a server
+ if server_name:
+ if not MCPRequestHandler.is_tool_allowed(
+ allowed_mcp_servers=[server.name for server in allowed_mcp_servers],
+ server_name=server_name,
+ ):
+ raise HTTPException(
+ status_code=403,
+ detail="User not allowed to call this tool.",
+ )
+
+ standard_logging_mcp_tool_call: Final[StandardLoggingMCPToolCall] = _get_standard_logging_mcp_tool_call(
+ name=original_tool_name, # Use original name for logging
+ arguments=arguments,
+ server_name=server_name,
+ session_id=_mcp_session_id_from_headers(raw_headers),
+ )
+ litellm_logging_obj: Final[LiteLLMLoggingObj | None] = kwargs.get("litellm_logging_obj", None)
+ if litellm_logging_obj:
+ litellm_logging_obj.model_call_details["mcp_tool_call_metadata"] = standard_logging_mcp_tool_call
+ litellm_logging_obj.model = f"MCP: {name}"
+ litellm_logging_obj.model_call_details["model"] = f"MCP: {name}"
+ # Resolve the MCP server early so BYOK checks and credential injection
+ # apply to ALL dispatch paths (local tool registry AND managed MCP server).
+ if mcp_server is None:
+ mcp_server = global_mcp_server_manager._get_mcp_server_from_tool_name(name)
+
+ if mcp_server:
+ standard_logging_mcp_tool_call["mcp_server_cost_info"] = (mcp_server.mcp_info or {}).get("mcp_server_cost_info")
+ if litellm_logging_obj:
+ litellm_logging_obj.model_call_details["mcp_tool_call_metadata"] = standard_logging_mcp_tool_call
+
+ # BYOK: retrieve the stored per-user credential. A single DB call
+ # both checks existence and fetches the value, avoiding a double query.
+ if mcp_server.is_byok and not mcp_auth_header:
+ byok_cred: Final = await _get_byok_credential(mcp_server, user_api_key_auth)
+ if byok_cred is None:
+ raise HTTPException(
+ status_code=401,
+ detail={
+ "error": "byok_auth_required",
+ "server_id": mcp_server.server_id,
+ "server_name": mcp_server.server_name or mcp_server.name,
+ "message": (
+ "No stored credential found for this BYOK server. "
+ "Complete the OAuth authorization flow to provide your API key."
+ ),
+ },
+ headers={"WWW-Authenticate": get_byok_www_authenticate()},
+ )
+ mcp_auth_header = byok_cred
+ elif mcp_server.is_byok:
+ # External auth header supplied; still enforce user-identity check.
+ await _check_byok_credential(mcp_server, user_api_key_auth)
+
+ # Check if tool exists in local registry first (for OpenAPI-based tools)
+ # These tools are registered with their prefixed names
+ #########################################################
+ local_tool: Final = global_mcp_tool_registry.get_tool(name)
+ if local_tool:
+ # OpenAPI-backed tools used to bypass `pre_call_tool_check` —
+ # only the managed path ran allowed/banned-tool checks, key/team
+ # tool permissions, and parameter validation. Run the same checks
+ # before dispatching to the local registry. Refuse the call if
+ # we cannot resolve a server: tools registered via
+ # openapi_to_mcp_generator are always tied to a server, so a
+ # missing mcp_server here means the tool->server mapping has
+ # not finished initializing or the registry entry is orphaned.
+ # Skipping the check would re-open the same authorization gap.
+ if mcp_server is None:
+ raise HTTPException(
+ status_code=503,
+ detail=(
+ f"MCP server for tool '{name}' is not available; "
+ "refusing to dispatch without authorization checks. "
+ "Retry once the server is registered."
+ ),
+ )
+
+ # `pre_call_tool_check` calls into `proxy_logging_obj` for the
+ # pre-call guardrail hooks, so source it from the canonical
+ # `proxy_server` module the same way `_handle_managed_mcp_tool`
+ # does. `kwargs.get("proxy_logging_obj")` is None on the MCP
+ # entry path and would crash with AttributeError after the
+ # security checks pass.
+ from litellm.proxy.proxy_server import proxy_logging_obj
+
+ hook_result = await global_mcp_server_manager.pre_call_tool_check(
+ name=original_tool_name,
+ arguments=arguments or {},
+ server_name=server_name or mcp_server.name,
+ user_api_key_auth=user_api_key_auth,
+ proxy_logging_obj=proxy_logging_obj,
+ server=mcp_server,
+ raw_headers=raw_headers,
+ litellm_logging_obj=litellm_logging_obj,
+ guardrail_context=guardrail_context,
+ )
+ # `pre_call_tool_check` may return guardrail-modified
+ # arguments; honor them on the local path too.
+ if isinstance(hook_result, dict) and "arguments" in hook_result:
+ arguments = hook_result["arguments"]
+
+ verbose_logger.debug("Executing local registry tool: %s", name)
+ # The credential rides ContextVars because the tool function has its
+ # headers baked into the closure at registration time.
+ auth_header_value, openapi_forwarded_headers, upstream_credential = _resolve_openapi_tool_auth(
+ mcp_server=mcp_server,
+ mcp_auth_header=mcp_auth_header,
+ mcp_server_auth_headers=mcp_server_auth_headers,
+ raw_headers=raw_headers,
+ user_api_key_auth=user_api_key_auth,
+ )
+ (
+ resolved_auth_headers,
+ forwarded_headers,
+ ) = await global_mcp_server_manager.resolve_openapi_upstream_auth(
+ mcp_server=mcp_server,
+ oauth2_headers=oauth2_headers,
+ raw_headers=raw_headers,
+ mcp_auth_header=upstream_credential,
+ user_api_key_auth=user_api_key_auth,
+ forwarded_headers=openapi_forwarded_headers,
+ )
+
+ _auth_token: Final = _request_auth_header.set(auth_header_value)
+ _extra_token: Final = _request_extra_headers.set(forwarded_headers)
+ _resolved_token: Final = _request_resolved_auth_headers.set(resolved_auth_headers)
+ try:
+ response = await _handle_local_mcp_tool(name, arguments)
+ finally:
+ _request_auth_header.reset(_auth_token)
+ _request_extra_headers.reset(_extra_token)
+ _request_resolved_auth_headers.reset(_resolved_token)
+
+ # Try managed MCP server tool (the name is bare; the prefix boundary was
+ # already resolved above against this server's registered prefixes)
+ # Primary and recommended way to use external MCP servers
+ #########################################################
+ elif mcp_server:
+ response = await _handle_managed_mcp_tool(
+ server_name=server_name,
+ name=original_tool_name,
+ arguments=arguments,
+ user_api_key_auth=user_api_key_auth,
+ mcp_auth_header=mcp_auth_header,
+ mcp_server_auth_headers=mcp_server_auth_headers,
+ oauth2_headers=oauth2_headers,
+ raw_headers=raw_headers,
+ client_ip=client_ip,
+ litellm_logging_obj=litellm_logging_obj,
+ guardrail_context=guardrail_context,
+ host_progress_callback=host_progress_callback,
+ )
+
+ # Fall back to local tool registry with original name (legacy support)
+ #########################################################
+ # Deprecated: Local MCP Server Tool
+ #########################################################
+ else:
+ # Gate only what can actually dispatch. When the unprefixed name is
+ # not in the registry either, `_handle_local_mcp_tool` below reports
+ # 404 and nothing runs, so demanding a server here would turn every
+ # unknown tool name into a misleading 503.
+ if global_mcp_tool_registry.get_tool(original_tool_name) is not None:
+ # `mcp_server` is None here because the tool name is not in the
+ # tool -> server mapping, but the name still carries a prefix
+ # that the server-level check above compared against the
+ # caller's `allowed_mcp_servers` by exact `name`. So the named
+ # server is in that list and can carry the tool-level checks,
+ # even with the mapping cold. Resolve it from
+ # `allowed_mcp_servers` rather than the registry: the registry
+ # would happily return a server the caller holds no grant for,
+ # and matching anything other than `name` would accept a server
+ # the check never validated.
+ prefix_server: Final = next(
+ (candidate for candidate in allowed_mcp_servers if candidate.name == server_name),
+ None,
+ )
+ if prefix_server is None:
+ # A non-empty prefix that passed the server-level check
+ # always matches here, so this arm only fires when the
+ # prefix was empty, which is exactly the case that check
+ # skips. Fail closed rather than dispatch with no server to
+ # evaluate a tool ceiling against.
+ raise HTTPException(
+ status_code=503,
+ detail=(
+ f"MCP server for tool '{original_tool_name}' is not available; "
+ "refusing to dispatch without authorization checks. "
+ "Retry once the server is registered."
+ ),
+ )
+
+ from litellm.proxy.proxy_server import proxy_logging_obj
+
+ hook_result = await global_mcp_server_manager.pre_call_tool_check(
+ name=original_tool_name,
+ arguments=arguments,
+ server_name=server_name,
+ user_api_key_auth=user_api_key_auth,
+ proxy_logging_obj=proxy_logging_obj,
+ server=prefix_server,
+ raw_headers=raw_headers,
+ litellm_logging_obj=litellm_logging_obj,
+ guardrail_context=guardrail_context,
+ )
+ if "arguments" in hook_result:
+ arguments = hook_result["arguments"] # pyright: ignore[reportAny] # hook returns untyped args
+
+ response = await _handle_local_mcp_tool(original_tool_name, arguments)
+
+ return await _run_post_mcp_call_guardrails(
+ result=response,
+ litellm_logging_obj=litellm_logging_obj,
+ user_api_key_auth=user_api_key_auth,
+ request_data=kwargs,
+ )
+
+
+async def _run_post_mcp_call_guardrails(
+ result: CallToolResult,
+ litellm_logging_obj: LiteLLMLoggingObj | None,
+ user_api_key_auth: UserAPIKeyAuth | None,
+ request_data: Mapping[str, object],
+) -> CallToolResult:
+ """Run ``post_mcp_call`` guardrails over an executed tool result.
+
+ Lives on ``execute_mcp_tool``'s return path rather than inside
+ ``_fire_mcp_tool_call_logging`` so enforcement never depends on logging
+ being configured, and so every dispatch route gets it: the MCP protocol
+ handler, the REST endpoint, and tool search all funnel through here.
+ A guardrail that rejects the result raises, matching ``pre_mcp_call``.
+ """
+ from litellm.proxy.proxy_server import proxy_logging_obj
+
+ if proxy_logging_obj is None:
+ return result
+ return await proxy_logging_obj.post_mcp_call_hook(
+ response=result,
+ request_data=(
+ litellm_logging_obj.model_call_details if litellm_logging_obj is not None else dict(request_data)
+ ),
+ user_api_key_dict=user_api_key_auth,
+ )
+
+
+async def _fire_mcp_tool_call_logging(
+ logging_obj: LiteLLMLoggingObj,
+ result: CallToolResult,
+ start_time: datetime,
+ end_time: datetime,
+ user_api_key_auth: UserAPIKeyAuth | None = None,
+ request_data: Mapping[str, object] | None = None,
+) -> CallToolResult:
+ """Fire post-call logging for an executed MCP tool call, returning the result to send.
+
+ The returned result is what the caller must forward to the client: a
+ ``post_mcp_call`` guardrail may rewrite the tool output (e.g. mask
+ sensitive values) or reject it, in which case its exception propagates.
+ Guardrails run before the success/failure logging so the masked text, not
+ the raw one, is what gets logged.
+
+ A result with ``is_error=True`` is logged as a failure (``status="failure"``
+ payload, so OTel marks the span ERROR) while the HTTP wire behavior stays
+ 200 + ``isError: true`` per the MCP spec. The error check runs after
+ ``async_post_mcp_tool_call_hook`` because guardrails may flip the result
+ to ``is_error=True`` in that hook. Raised exceptions never reach here (the
+ ``@client`` wrapper and ``call_mcp_tool``'s except path log those), so
+ this cannot double-log a failure.
+
+ ``request_data`` may carry credential-bearing fields (the REST path puts
+ ``raw_headers``, ``mcp_auth_header``, ``mcp_server_auth_headers``, and
+ ``oauth2_headers`` at the top level of its data dict), so those are
+ stripped before the dict is handed to ``post_call_failure_hook``
+ callbacks.
+ """
+ from litellm.proxy.proxy_server import proxy_logging_obj
+
+ logging_obj.post_call(original_response=result)
+ await logging_obj.async_post_mcp_tool_call_hook(
+ kwargs=logging_obj.model_call_details,
+ response_obj=result,
+ start_time=start_time,
+ end_time=end_time,
+ )
+ logging_obj.call_type = CallTypes.call_mcp_tool.value
+ error_message: Final = extract_mcp_tool_result_error_message(result)
+ if error_message is None:
+ await logging_obj.async_success_handler(result=result, start_time=start_time, end_time=end_time)
+ return result
+
+ logging_obj.has_run_logging(event_type="sync_success")
+ logging_obj.has_run_logging(event_type="async_success")
+ tool_error: Final = MCPToolResultError(error_message)
+ logging_obj.failure_handler(tool_error, "", start_time, end_time)
+ await logging_obj.async_failure_handler(tool_error, "", start_time, end_time)
+
+ if user_api_key_auth is None:
+ return result
+
+ if proxy_logging_obj:
+ sanitized_request_data: Final = {
+ key: value for key, value in (request_data or {}).items() if key not in _MCP_CREDENTIAL_REQUEST_FIELDS
+ }
+ await proxy_logging_obj.post_call_failure_hook(
+ request_data=sanitized_request_data,
+ original_exception=tool_error,
+ user_api_key_dict=user_api_key_auth,
+ route="/mcp/call_tool",
+ )
+ return result
+
+
+async def fire_mcp_tool_call_failure_logging(
+ logging_obj: LiteLLMLoggingObj | None,
+ exception: Exception,
+ start_time: datetime,
+ user_api_key_auth: UserAPIKeyAuth | None,
+ request_data: Mapping[str, object],
+) -> None:
+ """Failure logging shared by the ``/mcp`` path and the REST endpoint. Call from
+ inside the ``except`` block so the traceback is still available.
+
+ The failure handlers run first because ``_ProxyDBLogger.async_post_call_failure_hook``
+ builds the failure spend-log row from the ``standard_logging_object`` they produce;
+ both gate on ``should_run_logging``, so the ``@client`` wrapper does not log twice.
+ A relayed upstream 401 (``MCPUpstreamAuthError``) is an expected caller-must-reauth
+ signal and skips ``post_call_failure_hook``, which fires the ``llm_exceptions`` alert.
+ """
+ from litellm.proxy.proxy_server import proxy_logging_obj
+
+ traceback_str: Final = traceback.format_exc(limit=MAXIMUM_TRACEBACK_LINES_TO_LOG)
+ if logging_obj is not None:
+ end_time: Final = datetime.now() # noqa: DTZ005 # naive to match `start_time`, which it is subtracted from
+ logging_obj.failure_handler(exception, traceback_str, start_time, end_time)
+ await logging_obj.async_failure_handler(exception, traceback_str, start_time, end_time)
+
+ if isinstance(exception, MCPUpstreamAuthError) or not proxy_logging_obj or user_api_key_auth is None:
+ return
+ sanitized_request_data: Final = {
+ key: value for key, value in request_data.items() if key not in _MCP_CREDENTIAL_REQUEST_FIELDS
+ }
+ await proxy_logging_obj.post_call_failure_hook(
+ request_data=sanitized_request_data,
+ original_exception=exception,
+ user_api_key_dict=user_api_key_auth,
+ route="/mcp/call_tool",
+ traceback_str=traceback_str,
+ )
+
+
+@client
+async def call_mcp_tool(
+ name: str,
+ arguments: dict[str, object] | None = None,
+ user_api_key_auth: UserAPIKeyAuth | None = None,
+ mcp_auth_header: str | None = None,
+ mcp_servers: list[str] | None = None,
+ mcp_server_auth_headers: dict[str, dict[str, str]] | None = None,
+ oauth2_headers: dict[str, str] | None = None,
+ raw_headers: dict[str, str] | None = None,
+ client_ip: str | None = None,
+ **kwargs: Any,
+) -> CallToolResult:
+ """
+ Call a specific tool with the provided arguments (handles prefixed tool names).
+ """
+ start_time: Final = datetime.now()
+ litellm_logging_obj: Final[LiteLLMLoggingObj | None] = kwargs.get("litellm_logging_obj", None)
+
+ try:
+ if arguments is None:
+ raise HTTPException(status_code=400, detail="Request arguments are required")
+
+ ## CHECK IF USER IS ALLOWED TO CALL THIS TOOL
+ allowed_mcp_server_ids: Final = await global_mcp_server_manager.get_allowed_mcp_servers(
+ user_api_key_auth=user_api_key_auth,
+ )
+
+ allowed_mcp_servers: list[MCPServer] = []
+ for allowed_mcp_server_id in allowed_mcp_server_ids:
+ allowed_server = global_mcp_server_manager.get_mcp_server_by_id(allowed_mcp_server_id)
+ if allowed_server is not None:
+ # Same request-time oauth2_flow backstop the listing path applies,
+ # so a null-flow M2M-shape row is treated as M2M on tool calls too.
+ allowed_server = MCPServerManager.resolve_oauth2_flow_for_request(allowed_server)
+ allowed_mcp_servers.append(allowed_server)
+
+ allowed_mcp_servers = await _get_allowed_mcp_servers_from_mcp_server_names(
+ mcp_servers=mcp_servers,
+ allowed_mcp_servers=allowed_mcp_servers,
+ )
+ if mcp_servers and not allowed_mcp_servers:
+ await raise_denied_scoped_mcp_access(
+ requested_names=mcp_servers,
+ user_api_key_auth=user_api_key_auth,
+ client_ip=client_ip,
+ )
+ if not allowed_mcp_servers:
+ raise HTTPException(
+ status_code=403,
+ detail="User not allowed to call this tool.",
+ )
+
+ # Delegate to execute_mcp_tool for execution
+ response = await execute_mcp_tool(
+ name=name,
+ arguments=arguments,
+ allowed_mcp_servers=allowed_mcp_servers,
+ start_time=start_time,
+ user_api_key_auth=user_api_key_auth,
+ mcp_auth_header=mcp_auth_header,
+ mcp_server_auth_headers=mcp_server_auth_headers,
+ oauth2_headers=oauth2_headers,
+ raw_headers=raw_headers,
+ client_ip=client_ip,
+ **kwargs,
+ )
+ except Exception as e:
+ await fire_mcp_tool_call_failure_logging(litellm_logging_obj, e, start_time, user_api_key_auth, kwargs)
+ raise
+
+ if litellm_logging_obj:
+ response = await _fire_mcp_tool_call_logging(
+ logging_obj=litellm_logging_obj,
+ result=response,
+ start_time=start_time,
+ end_time=datetime.now(),
+ user_api_key_auth=user_api_key_auth,
+ request_data=kwargs,
+ )
+ return response
+
+
+async def mcp_get_prompt(
+ name: str,
+ arguments: dict[str, str] | None = None,
+ user_api_key_auth: UserAPIKeyAuth | None = None,
+ mcp_auth_header: str | None = None,
+ mcp_servers: list[str] | None = None,
+ mcp_server_auth_headers: dict[str, dict[str, str]] | None = None,
+ oauth2_headers: dict[str, str] | None = None,
+ raw_headers: dict[str, str] | None = None,
+ client_ip: str | None = None,
+) -> GetPromptResult:
+ """
+ Fetch a specific MCP prompt, handling both prefixed and unprefixed names.
+ """
+ allowed_mcp_servers: Final = await _get_allowed_mcp_servers(
+ user_api_key_auth=user_api_key_auth,
+ mcp_servers=mcp_servers,
+ client_ip=client_ip,
+ )
+
+ if not allowed_mcp_servers:
+ raise HTTPException(
+ status_code=403,
+ detail="User not allowed to get this prompt.",
+ )
+
+ # Extract server name from prefixed prompt name
+ original_prompt_name, server_name = split_server_prefix_from_name(name)
+
+ server: Final = next((s for s in allowed_mcp_servers if s.name == server_name), None)
+ if server is None:
+ raise HTTPException(
+ status_code=403,
+ detail="User not allowed to get this prompt.",
+ )
+
+ server_auth_header, extra_headers = _prepare_mcp_server_headers(
+ server=server,
+ mcp_server_auth_headers=mcp_server_auth_headers,
+ mcp_auth_header=mcp_auth_header,
+ oauth2_headers=oauth2_headers,
+ raw_headers=raw_headers,
+ user_api_key_auth=user_api_key_auth,
+ )
+
+ return await global_mcp_server_manager.get_prompt_from_server(
+ server=server,
+ user_api_key_auth=user_api_key_auth,
+ prompt_name=original_prompt_name,
+ arguments=arguments,
+ mcp_auth_header=server_auth_header,
+ extra_headers=extra_headers,
+ raw_headers=raw_headers,
+ client_ip=client_ip,
+ )
+
+
+async def mcp_read_resource(
+ url: AnyUrl,
+ user_api_key_auth: UserAPIKeyAuth | None = None,
+ mcp_auth_header: str | None = None,
+ mcp_servers: list[str] | None = None,
+ mcp_server_auth_headers: dict[str, dict[str, str]] | None = None,
+ oauth2_headers: dict[str, str] | None = None,
+ raw_headers: dict[str, str] | None = None,
+ client_ip: str | None = None,
+) -> ReadResourceResult:
+ """Read resource contents from upstream MCP servers."""
+
+ allowed_mcp_servers: Final = await _get_allowed_mcp_servers(
+ user_api_key_auth=user_api_key_auth,
+ mcp_servers=mcp_servers,
+ client_ip=client_ip,
+ )
+
+ if not allowed_mcp_servers:
+ raise HTTPException(
+ status_code=403,
+ detail="User not allowed to read this resource.",
+ )
+
+ if len(allowed_mcp_servers) != 1:
+ raise HTTPException(
+ status_code=400,
+ detail=("Multiple MCP servers configured; read_resource currently supports exactly one allowed server."),
+ )
+
+ server: Final = allowed_mcp_servers[0]
+
+ server_auth_header, extra_headers = _prepare_mcp_server_headers(
+ server=server,
+ mcp_server_auth_headers=mcp_server_auth_headers,
+ mcp_auth_header=mcp_auth_header,
+ oauth2_headers=oauth2_headers,
+ raw_headers=raw_headers,
+ user_api_key_auth=user_api_key_auth,
+ )
+
+ return await global_mcp_server_manager.read_resource_from_server(
+ server=server,
+ user_api_key_auth=user_api_key_auth,
+ url=url,
+ mcp_auth_header=server_auth_header,
+ extra_headers=extra_headers,
+ raw_headers=raw_headers,
+ client_ip=client_ip,
+ )
+
+
+def _get_standard_logging_mcp_tool_call(
+ name: str,
+ arguments: dict[str, object],
+ server_name: str | None,
+ session_id: str | None = None,
+) -> StandardLoggingMCPToolCall:
+ mcp_server: Final = global_mcp_server_manager._get_mcp_server_from_tool_name(
+ add_server_prefix_to_name(name, server_name) if server_name else name
+ )
+ namespaced_tool_name: Final = f"{server_name}/{name}" if server_name else name
+ if mcp_server:
+ mcp_info: Final = mcp_server.mcp_info or {}
+ return StandardLoggingMCPToolCall(
+ name=name,
+ arguments=arguments,
+ mcp_server_name=mcp_info.get("server_name"),
+ mcp_server_logo_url=mcp_info.get("logo_url"),
+ namespaced_tool_name=namespaced_tool_name,
+ mcp_session_id=session_id,
+ mcp_auth_mode=mcp_server.auth_type,
+ mcp_server_resource=_redact_mcp_resource_url(mcp_server.url),
+ )
+ else:
+ return StandardLoggingMCPToolCall(
+ name=name,
+ arguments=arguments,
+ namespaced_tool_name=namespaced_tool_name,
+ mcp_session_id=session_id,
+ )
+
+
+async def _handle_managed_mcp_tool(
+ server_name: str,
+ name: str,
+ arguments: dict[str, object],
+ user_api_key_auth: UserAPIKeyAuth | None = None,
+ mcp_auth_header: str | None = None,
+ mcp_server_auth_headers: dict[str, dict[str, str]] | None = None,
+ oauth2_headers: dict[str, str] | None = None,
+ raw_headers: dict[str, str] | None = None,
+ litellm_logging_obj: LiteLLMLoggingObj | None = None,
+ host_progress_callback: ProgressCallback | None = None,
+ guardrail_context: Mapping[str, object] | None = None,
+ client_ip: str | None = None,
+) -> CallToolResult:
+ """Handle tool execution for managed server tools"""
+ # Import here to avoid circular import
+ from litellm.proxy.proxy_server import proxy_logging_obj
+
+ call_tool_result: Final = await global_mcp_server_manager.call_tool(
+ server_name=server_name,
+ name=name,
+ arguments=arguments,
+ user_api_key_auth=user_api_key_auth,
+ mcp_auth_header=mcp_auth_header,
+ mcp_server_auth_headers=mcp_server_auth_headers,
+ oauth2_headers=oauth2_headers,
+ raw_headers=raw_headers,
+ client_ip=client_ip,
+ proxy_logging_obj=proxy_logging_obj,
+ host_progress_callback=host_progress_callback,
+ litellm_logging_obj=litellm_logging_obj,
+ guardrail_context=guardrail_context,
+ )
+ verbose_logger.debug("CALL TOOL RESULT: %s", call_tool_result)
+ return call_tool_result
+
+
+async def _handle_local_mcp_tool(name: str, arguments: dict[str, object]) -> CallToolResult:
+ """Execute a local-registry tool and report whether it succeeded.
+
+ Returns the result rather than bare content because the verdict is part of it: the content
+ alone cannot say whether the handler failed, so callers used to stamp is_error=False on every
+ outcome and an upstream rejection was served as tool output.
+
+ A failure is reported as ``is_error=True`` here rather than raised, because the REST surface
+ turns an unrecognized exception into a 500 and an upstream 403 or 429 is not a gateway crash.
+ ``MCPUpstreamAuthError`` is the exception: it propagates so the caller is told to
+ re-authenticate, which both renderers already know how to say.
+
+ Note: Local tools don't use prefixes, so we use the original name
+ """
+ import inspect
+
+ tool: Final = global_mcp_tool_registry.get_tool(name)
+ if not tool:
+ raise HTTPException(status_code=404, detail=f"Tool '{name}' not found")
+
+ try:
+ if inspect.iscoroutinefunction(tool.handler):
+ result = await tool.handler(**arguments)
+ else:
+ result = tool.handler(**arguments)
+ except MCPUpstreamAuthError:
+ raise
+ except Exception as e:
+ verbose_logger.exception("Error executing local tool %s: %s", name, e)
+ return CallToolResult(
+ content=[TextContent(text=f"Error: {e}", type="text")], # mutable-ok: MCP result content
+ is_error=True,
+ )
+ return CallToolResult(
+ content=[TextContent(text=str(result), type="text")], # mutable-ok: MCP result content
+ is_error=False,
+ )
+
+
+_MCP_CREDENTIAL_REQUEST_FIELDS: Final = frozenset(
+ {
+ "raw_headers",
+ "mcp_auth_header",
+ "mcp_server_auth_headers",
+ "oauth2_headers",
+ "user_api_key_auth",
+ }
+)
+
+
+class _McpDeniedDetail(TypedDict):
+ error: ReadOnly[str]
+
+
+async def _execute_handle_list_tools(
+ context: OperationContext, params: PaginatedRequestParams, host_progress_callback: ProgressCallback | None = None
+) -> ListToolsResult:
+ try:
+ (
+ user_api_key_auth,
+ mcp_auth_header,
+ mcp_servers,
+ mcp_server_auth_headers,
+ oauth2_headers,
+ raw_headers,
+ _client_ip,
+ ) = context.legacy_auth()
+ verbose_logger.debug("MCP list_tools - User API Key Auth from context: %s", user_api_key_auth)
+ verbose_logger.debug("MCP list_tools - MCP servers from context: %s", mcp_servers)
+ verbose_logger.debug(
+ "MCP list_tools - MCP server auth headers: %s",
+ list(mcp_server_auth_headers.keys()) if mcp_server_auth_headers else None,
+ )
+ from mcp.types import Tool
+
+ from litellm.proxy._experimental.mcp_server.tool_search import (
+ get_mcp_proxy_tool_definitions,
+ get_virtual_tool_definitions,
+ )
+
+ if context.mcp_proxy_mode:
+ return ListToolsResult(tools=[Tool.model_validate(d) for d in get_mcp_proxy_tool_definitions()])
+ if getattr(
+ getattr(user_api_key_auth, "object_permission", None),
+ "mcp_tool_search_enabled",
+ False,
+ ):
+ return ListToolsResult(tools=[Tool.model_validate(d) for d in get_virtual_tool_definitions()])
+
+ # Get mcp_servers from context variable
+ verbose_logger.debug("MCP list_tools - Calling _list_mcp_tools")
+ listing: Final = await _list_mcp_tools(
+ user_api_key_auth=user_api_key_auth,
+ mcp_auth_header=mcp_auth_header,
+ mcp_servers=mcp_servers,
+ mcp_server_auth_headers=mcp_server_auth_headers,
+ oauth2_headers=oauth2_headers,
+ raw_headers=raw_headers,
+ log_list_tools_to_spendlogs=True,
+ list_tools_log_source="mcp_protocol",
+ client_ip=_client_ip,
+ )
+ verbose_logger.info("MCP list_tools - Successfully returned %s tools", len(listing.tools))
+ if not listing.outcomes:
+ return ListToolsResult(tools=listing.tools)
+ outcome_meta: Final = {
+ SERVER_OUTCOMES_META_KEY: {key: outcome_wire_value(outcome) for key, outcome in listing.outcomes.items()}
+ }
+ return ListToolsResult.model_validate({"tools": listing.tools, "_meta": outcome_meta})
+ except HTTPException as e:
+ from mcp.shared.exceptions import MCPError
+ from mcp.types import INVALID_REQUEST
+
+ raise MCPError(code=INVALID_REQUEST, message=_http_detail_message(e.detail)) from e
+ except Exception as e:
+ verbose_logger.exception("Error in list_tools endpoint: %s", e)
+ # Return empty list instead of failing completely
+ # This prevents the HTTP stream from failing and allows the client to get a response
+ return ListToolsResult(tools=[]) # mutable-ok: MCP result payload
+
+
+async def _execute_mcp_server_tool_call(
+ context: OperationContext, params: CallToolRequestParams, host_progress_callback: ProgressCallback | None = None
+) -> CallToolResult:
+ from mcp.types import CallToolResult
+
+ from litellm.exceptions import BlockedPiiEntityError, GuardrailRaisedException
+ from litellm.proxy.litellm_pre_call_utils import add_litellm_data_to_request
+ from litellm.proxy.proxy_server import proxy_config
+
+ (
+ user_api_key_auth,
+ mcp_auth_header,
+ mcp_servers,
+ mcp_server_auth_headers,
+ oauth2_headers,
+ raw_headers,
+ _client_ip,
+ ) = context.legacy_auth()
+ verbose_logger.debug(
+ "MCP mcp_server_tool_call - user_api_key_auth=%s, user_role=%s",
+ user_api_key_auth,
+ getattr(user_api_key_auth, "user_role", "N/A"),
+ )
+
+ verbose_logger.debug("MCP mcp_server_tool_call - User API Key Auth from context: %s", user_api_key_auth)
+
+ try:
+ # Inside this try so virtual-tool errors convert to isError
+ # CallToolResult instead of raising out of the protocol handler.
+ virtual_tool_result: Final = await _dispatch_virtual_mcp_tool(
+ name=params.name,
+ arguments=params.arguments,
+ user_api_key_auth=user_api_key_auth,
+ client_ip=_client_ip,
+ mcp_servers=mcp_servers,
+ mcp_auth_header=mcp_auth_header,
+ mcp_server_auth_headers=mcp_server_auth_headers,
+ oauth2_headers=oauth2_headers,
+ raw_headers=raw_headers,
+ mcp_proxy_mode=context.mcp_proxy_mode,
+ )
+ if virtual_tool_result is not None:
+ return virtual_tool_result
+
+ # Create a body date for logging
+ body_data: Final = {"name": params.name, "arguments": params.arguments} # mutable-ok: logging payload
+ # Set trace/session id from raw_headers so spend logs and logging_obj stay consistent (same as A2A)
+ chain_id: Final = get_chain_id_from_headers(raw_headers)
+ if chain_id:
+ body_data["litellm_trace_id"] = chain_id
+ body_data["litellm_session_id"] = chain_id
+
+ request: Final = build_synthetic_mcp_request(
+ path="/mcp/tools/call",
+ raw_headers=raw_headers,
+ client_ip=_client_ip,
+ )
+ if user_api_key_auth is not None:
+ data = await add_litellm_data_to_request(
+ data=body_data,
+ request=request,
+ # Bill a team-derived call to the team that granted it. A keyless admitted
+ # subject carries no team_id, so spend skipped team updates entirely and
+ # charged the user's PRIMARY org — the granting team's budget never
+ # accumulated (so it could never begin to block) and, cross-org, the wrong
+ # organization was charged. This is the ACCOUNTING half; the enforcement
+ # half (an already-over-budget team stops granting) lives in the source gate.
+ # Authorization is unaffected: it ran before this, and the union is resolved
+ # from the untouched auth object passed to call_mcp_tool below.
+ user_api_key_dict=await MCPRequestHandler.billing_auth_for_tool_call(
+ user_api_key_auth, tool_name=params.name
+ ),
+ proxy_config=proxy_config,
+ )
+ else:
+ data = body_data
+
+ response: Final = await call_mcp_tool(
+ user_api_key_auth=user_api_key_auth,
+ mcp_auth_header=mcp_auth_header,
+ mcp_servers=mcp_servers,
+ mcp_server_auth_headers=mcp_server_auth_headers,
+ oauth2_headers=oauth2_headers,
+ raw_headers=raw_headers,
+ client_ip=_client_ip,
+ host_progress_callback=host_progress_callback,
+ **data, # for logging
+ )
+ except MCPMissingUserEnvVarsError as e:
+ verbose_logger.info(
+ "MCP mcp_server_tool_call missing per-user env vars: server_id=%s missing=%s",
+ e.server_id,
+ e.missing,
+ )
+ return CallToolResult(
+ content=[TextContent(text=str(e), type="text")],
+ is_error=True,
+ )
+ except BlockedPiiEntityError as e:
+ verbose_logger.error("BlockedPiiEntityError in MCP tool call: %s", e)
+ return CallToolResult(
+ content=[
+ TextContent(
+ text=f"Error: Blocked PII entity detected - {e}",
+ type="text",
+ )
+ ],
+ is_error=True,
+ )
+ except GuardrailRaisedException as e:
+ verbose_logger.error("GuardrailRaisedException in MCP tool call: %s", e)
+ return CallToolResult(
+ content=[TextContent(text=f"Error: Guardrail violation - {e}", type="text")],
+ is_error=True,
+ )
+ except HTTPException as e:
+ verbose_logger.error("HTTPException in MCP tool call: %s", e)
+ return CallToolResult(
+ content=[TextContent(text=f"Error: {_http_detail_message(e.detail)}", type="text")],
+ is_error=True,
+ )
+ except MCPUpstreamAuthError as e:
+ # The MCP session manager serializes handler exceptions as JSON-RPC errors, so a
+ # mid-session tool call cannot emit a raw 401 + WWW-Authenticate the way the REST
+ # call path and the connect-time preemptive check do. Return an explicit isError
+ # naming the upstream status (at info level, not a traceback) so the client still
+ # learns it must re-authenticate upstream and expected pass-through 401s don't spam.
+ verbose_logger.info("Upstream auth failure calling MCP tool: HTTP %s", e.status_code)
+ return CallToolResult(
+ content=[
+ TextContent(
+ text=f"Error: upstream authentication required (HTTP {e.status_code})",
+ type="text",
+ )
+ ],
+ is_error=True,
+ )
+ except Exception as e:
+ verbose_logger.exception("MCP mcp_server_tool_call - error: %s", e)
+ return CallToolResult(
+ content=[TextContent(text=f"Error: {e}", type="text")],
+ is_error=True,
+ )
+
+ return response
+
+
+async def _execute_list_prompts(
+ context: OperationContext, params: PaginatedRequestParams, host_progress_callback: ProgressCallback | None = None
+) -> ListPromptsResult:
+ if context.mcp_proxy_mode:
+ _reject_mcp_proxy_operation()
+ try:
+ (
+ user_api_key_auth,
+ mcp_auth_header,
+ mcp_servers,
+ mcp_server_auth_headers,
+ oauth2_headers,
+ raw_headers,
+ _client_ip,
+ ) = context.legacy_auth()
+ verbose_logger.debug("MCP list_prompts - User API Key Auth from context: %s", user_api_key_auth)
+ verbose_logger.debug("MCP list_prompts - MCP servers from context: %s", mcp_servers)
+ verbose_logger.debug(
+ "MCP list_prompts - MCP server auth headers: %s",
+ list(mcp_server_auth_headers.keys()) if mcp_server_auth_headers else None,
+ )
+ # Get mcp_servers from context variable
+ verbose_logger.debug("MCP list_prompts - Calling _list_prompts")
+ prompts: Final = await _list_mcp_prompts(
+ user_api_key_auth=user_api_key_auth,
+ mcp_auth_header=mcp_auth_header,
+ mcp_servers=mcp_servers,
+ mcp_server_auth_headers=mcp_server_auth_headers,
+ oauth2_headers=oauth2_headers,
+ raw_headers=raw_headers,
+ client_ip=_client_ip,
+ )
+ verbose_logger.info("MCP list_prompts - Successfully returned %s prompts", len(prompts))
+ return ListPromptsResult(prompts=prompts)
+ except Exception as e:
+ verbose_logger.exception("Error in list_prompts endpoint: %s", e)
+ # Return empty list instead of failing completely
+ # This prevents the HTTP stream from failing and allows the client to get a response
+ return ListPromptsResult(prompts=[]) # mutable-ok: MCP result payload
+
+
+async def _execute_get_prompt(
+ context: OperationContext, params: GetPromptRequestParams, host_progress_callback: ProgressCallback | None = None
+) -> GetPromptResult:
+ if context.mcp_proxy_mode:
+ _reject_mcp_proxy_operation()
+ (
+ user_api_key_auth,
+ mcp_auth_header,
+ mcp_servers,
+ mcp_server_auth_headers,
+ oauth2_headers,
+ raw_headers,
+ _client_ip,
+ ) = context.legacy_auth()
+
+ verbose_logger.debug("MCP mcp_server_tool_call - User API Key Auth from context: %s", user_api_key_auth)
+ return await mcp_get_prompt(
+ name=params.name,
+ arguments=params.arguments,
+ user_api_key_auth=user_api_key_auth,
+ mcp_auth_header=mcp_auth_header,
+ mcp_servers=mcp_servers,
+ mcp_server_auth_headers=mcp_server_auth_headers,
+ oauth2_headers=oauth2_headers,
+ raw_headers=raw_headers,
+ client_ip=_client_ip,
+ )
+
+
+async def _execute_list_resources(
+ context: OperationContext, params: PaginatedRequestParams, host_progress_callback: ProgressCallback | None = None
+) -> ListResourcesResult:
+ if context.mcp_proxy_mode:
+ _reject_mcp_proxy_operation()
+ try:
+ (
+ user_api_key_auth,
+ mcp_auth_header,
+ mcp_servers,
+ mcp_server_auth_headers,
+ oauth2_headers,
+ raw_headers,
+ _client_ip,
+ ) = context.legacy_auth()
+ verbose_logger.debug("MCP list_resources - User API Key Auth from context: %s", user_api_key_auth)
+ verbose_logger.debug("MCP list_resources - MCP servers from context: %s", mcp_servers)
+ verbose_logger.debug(
+ "MCP list_resources - MCP server auth headers: %s",
+ list(mcp_server_auth_headers.keys()) if mcp_server_auth_headers else None,
+ )
+
+ resources: Final = await _list_mcp_resources(
+ user_api_key_auth=user_api_key_auth,
+ mcp_auth_header=mcp_auth_header,
+ mcp_servers=mcp_servers,
+ mcp_server_auth_headers=mcp_server_auth_headers,
+ oauth2_headers=oauth2_headers,
+ raw_headers=raw_headers,
+ client_ip=_client_ip,
+ )
+ verbose_logger.info("MCP list_resources - Successfully returned %s resources", len(resources))
+ return ListResourcesResult(resources=resources)
+ except Exception as e:
+ verbose_logger.exception("Error in list_resources endpoint: %s", e)
+ return ListResourcesResult(resources=[]) # mutable-ok: MCP result payload
+
+
+async def _execute_list_resource_templates(
+ context: OperationContext, params: PaginatedRequestParams, host_progress_callback: ProgressCallback | None = None
+) -> ListResourceTemplatesResult:
+ if context.mcp_proxy_mode:
+ _reject_mcp_proxy_operation()
+ try:
+ (
+ user_api_key_auth,
+ mcp_auth_header,
+ mcp_servers,
+ mcp_server_auth_headers,
+ oauth2_headers,
+ raw_headers,
+ _client_ip,
+ ) = context.legacy_auth()
+ verbose_logger.debug("MCP list_resource_templates - User API Key Auth from context: %s", user_api_key_auth)
+ verbose_logger.debug("MCP list_resource_templates - MCP servers from context: %s", mcp_servers)
+ verbose_logger.debug(
+ "MCP list_resource_templates - MCP server auth headers: %s",
+ list(mcp_server_auth_headers.keys()) if mcp_server_auth_headers else None,
+ )
+
+ resource_templates: Final = await _list_mcp_resource_templates(
+ user_api_key_auth=user_api_key_auth,
+ mcp_auth_header=mcp_auth_header,
+ mcp_servers=mcp_servers,
+ mcp_server_auth_headers=mcp_server_auth_headers,
+ oauth2_headers=oauth2_headers,
+ raw_headers=raw_headers,
+ client_ip=_client_ip,
+ )
+ verbose_logger.info(
+ "MCP list_resource_templates - Successfully returned %s resource templates", len(resource_templates)
+ )
+ return ListResourceTemplatesResult(resource_templates=resource_templates)
+ except Exception as e:
+ verbose_logger.exception("Error in list_resource_templates endpoint: %s", e)
+ return ListResourceTemplatesResult(resource_templates=[]) # mutable-ok: MCP result payload
+
+
+async def _execute_read_resource(
+ context: OperationContext, params: ReadResourceRequestParams, host_progress_callback: ProgressCallback | None = None
+) -> ReadResourceResult:
+ if context.mcp_proxy_mode:
+ _reject_mcp_proxy_operation()
+ (
+ user_api_key_auth,
+ mcp_auth_header,
+ mcp_servers,
+ mcp_server_auth_headers,
+ oauth2_headers,
+ raw_headers,
+ _client_ip,
+ ) = context.legacy_auth()
+
+ read_resource_result: Final = await mcp_read_resource(
+ url=params.uri,
+ user_api_key_auth=user_api_key_auth,
+ mcp_auth_header=mcp_auth_header,
+ mcp_servers=mcp_servers,
+ mcp_server_auth_headers=mcp_server_auth_headers,
+ oauth2_headers=oauth2_headers,
+ raw_headers=raw_headers,
+ client_ip=_client_ip,
+ )
+
+ return read_resource_result
+
+
+def _reject_mcp_proxy_operation() -> NoReturn:
+ from mcp.shared.exceptions import MCPError
+ from mcp.types import METHOD_NOT_FOUND
+
+ raise MCPError(code=METHOD_NOT_FOUND, message="Operation unavailable on /mcp/proxy")
+
+
+def prepare_context(
+ user_api_key_auth: UserAPIKeyAuth | None = None,
+ mcp_auth_header: str | None = None,
+ mcp_servers: Sequence[str] | None = None,
+ mcp_server_auth_headers: Mapping[str, Mapping[str, str]] | None = None,
+ oauth2_headers: Mapping[str, str] | None = None,
+ raw_headers: Mapping[str, str] | None = None,
+ client_ip: str | None = None,
+ mcp_proxy_mode: bool = False,
+) -> OperationContext:
+ return OperationContext(
+ _caller=user_api_key_auth,
+ mcp_auth_header=mcp_auth_header,
+ mcp_servers=tuple(mcp_servers) if mcp_servers is not None else None,
+ mcp_server_auth_headers=mcp_server_auth_headers,
+ oauth2_headers=oauth2_headers,
+ raw_headers=raw_headers,
+ client_ip=client_ip,
+ mcp_proxy_mode=mcp_proxy_mode,
+ )
+
+
+GatewayOperation: TypeAlias = (
+ AuthorizedToolCall
+ | ListToolsRequest
+ | CallToolRequest
+ | ListPromptsRequest
+ | GetPromptRequest
+ | ListResourcesRequest
+ | ListResourceTemplatesRequest
+ | ReadResourceRequest
+)
+GatewayResult: TypeAlias = (
+ ListToolsResult
+ | CallToolResult
+ | ListPromptsResult
+ | GetPromptResult
+ | ListResourcesResult
+ | ListResourceTemplatesResult
+ | ReadResourceResult
+)
+
+
+class GatewayOperations:
+ def __init__(self, host_progress_callback: ProgressCallback | None = None) -> None:
+ self._host_progress_callback = host_progress_callback
+
+ @overload
+ async def execute(self, operation: AuthorizedToolCall, context: OperationContext) -> CallToolResult: ...
+
+ @overload
+ async def execute(self, operation: ListToolsRequest, context: OperationContext) -> ListToolsResult: ...
+
+ @overload
+ async def execute(self, operation: CallToolRequest, context: OperationContext) -> CallToolResult: ...
+
+ @overload
+ async def execute(self, operation: ListPromptsRequest, context: OperationContext) -> ListPromptsResult: ...
+
+ @overload
+ async def execute(self, operation: GetPromptRequest, context: OperationContext) -> GetPromptResult: ...
+
+ @overload
+ async def execute(self, operation: ListResourcesRequest, context: OperationContext) -> ListResourcesResult: ...
+
+ @overload
+ async def execute(
+ self, operation: ListResourceTemplatesRequest, context: OperationContext
+ ) -> ListResourceTemplatesResult: ...
+
+ @overload
+ async def execute(self, operation: ReadResourceRequest, context: OperationContext) -> ReadResourceResult: ...
+
+ async def execute(self, operation: GatewayOperation, context: OperationContext) -> GatewayResult:
+ match operation:
+ case AuthorizedToolCall():
+ auth, token, _servers, server_headers, oauth_headers, headers, _client_ip = context.legacy_auth()
+ return await _execute_mcp_tool(
+ name=operation.name,
+ arguments=dict(operation.arguments), # mutable-ok: existing tool hooks own mutable argument data
+ allowed_mcp_servers=list(
+ operation.allowed_mcp_servers
+ ), # mutable-ok: legacy dispatch list contract
+ start_time=operation.start_time,
+ user_api_key_auth=auth,
+ mcp_auth_header=token,
+ mcp_server_auth_headers=server_headers,
+ oauth2_headers=oauth_headers,
+ raw_headers=headers,
+ client_ip=_client_ip,
+ host_progress_callback=operation.host_progress_callback,
+ guardrail_context=operation.guardrail_context,
+ **operation.logging_data,
+ )
+ case ListToolsRequest(params=params):
+ return await _execute_handle_list_tools(
+ context, params or PaginatedRequestParams(), self._host_progress_callback
+ )
+ case CallToolRequest(params=params):
+ return await _execute_mcp_server_tool_call(context, params, self._host_progress_callback)
+ case ListPromptsRequest(params=params):
+ return await _execute_list_prompts(
+ context, params or PaginatedRequestParams(), self._host_progress_callback
+ )
+ case GetPromptRequest(params=params):
+ return await _execute_get_prompt(context, params, self._host_progress_callback)
+ case ListResourcesRequest(params=params):
+ return await _execute_list_resources(
+ context, params or PaginatedRequestParams(), self._host_progress_callback
+ )
+ case ListResourceTemplatesRequest(params=params):
+ return await _execute_list_resource_templates(
+ context, params or PaginatedRequestParams(), self._host_progress_callback
+ )
+ case ReadResourceRequest(params=params):
+ return await _execute_read_resource(context, params, self._host_progress_callback)
+ case _:
+ assert_never(operation)
diff --git a/litellm/proxy/_experimental/mcp_server/rest_endpoints.py b/litellm/proxy/_experimental/mcp_server/rest_endpoints.py
index 15f97a15b73..c2f7bf7d531 100644
--- a/litellm/proxy/_experimental/mcp_server/rest_endpoints.py
+++ b/litellm/proxy/_experimental/mcp_server/rest_endpoints.py
@@ -203,17 +203,19 @@ if MCP_AVAILABLE:
from litellm.proxy._experimental.mcp_server.oauth_utils import (
get_request_base_url,
)
- from litellm.proxy._experimental.mcp_server.server import (
+ from litellm.proxy._experimental.mcp_server.operations import (
ListMCPToolsRestAPIResponseObject,
MCPInfo,
MCPServer,
- _aggregate_server_key, # pyright: ignore[reportPrivateUsage] # same per-server key as the tools/list _meta outcomes
- _apply_toolset_scope,
+ _aggregate_server_key,
_fire_mcp_tool_call_logging,
execute_mcp_tool,
filter_tools_by_allowed_tools,
filter_tools_by_key_team_permissions,
fire_mcp_tool_call_failure_logging,
+ )
+ from litellm.proxy._experimental.mcp_server.server import (
+ _apply_toolset_scope,
reject_disallowed_mcp_client,
)
@@ -670,6 +672,7 @@ if MCP_AVAILABLE:
user_api_key_auth: UserAPIKeyAuth | None = None,
extra_headers: dict[str, str] | None = None,
apply_tool_filters: bool = True,
+ client_ip: str | None = None,
):
"""Helper function to get tools for a single server.
@@ -684,6 +687,7 @@ if MCP_AVAILABLE:
extra_headers=extra_headers,
add_prefix=False,
raw_headers=raw_headers,
+ client_ip=client_ip,
user_api_key_auth=user_api_key_auth,
)
@@ -797,6 +801,7 @@ if MCP_AVAILABLE:
user_api_key_dict,
extra_headers=user_oauth_extra_headers,
apply_tool_filters=apply_tool_filters,
+ client_ip=rest_client_ip,
)
except MCPUpstreamAuthError:
# Surface the upstream 401/403 to the caller so it can emit the
@@ -1016,6 +1021,7 @@ if MCP_AVAILABLE:
user_api_key_dict,
extra_headers=user_oauth_extra_headers,
apply_tool_filters=apply_tool_filters,
+ client_ip=_rest_client_ip,
)
except Exception as e:
verbose_logger.warning(
@@ -1193,6 +1199,7 @@ if MCP_AVAILABLE:
mcp_server_auth_headers=data.get("mcp_server_auth_headers"),
oauth2_headers=user_oauth_extra_headers or data.get("oauth2_headers"),
raw_headers=data.get("raw_headers"),
+ client_ip=IPAddressUtils.get_mcp_client_ip(request),
litellm_logging_obj=data.get("litellm_logging_obj"),
guardrail_context=MCPRequestContext.resolve_guardrail_context(data),
requested_server_id=canonical_server_id,
diff --git a/litellm/proxy/_experimental/mcp_server/server.py b/litellm/proxy/_experimental/mcp_server/server.py
index 3a9bca926b0..33978bb9182 100644
--- a/litellm/proxy/_experimental/mcp_server/server.py
+++ b/litellm/proxy/_experimental/mcp_server/server.py
@@ -11,28 +11,22 @@ import hashlib
import json
import os
import time
-import traceback
import types
-import uuid
from collections import Counter
-from collections.abc import AsyncIterator, Callable, Iterable, Mapping, Sequence
-from datetime import datetime
-from typing import TYPE_CHECKING, Any, Final, NoReturn, Protocol
+from collections.abc import AsyncGenerator, AsyncIterator, Callable, Iterable, Mapping, Sequence
+from typing import TYPE_CHECKING, Final, NoReturn, Protocol
import httpx
from fastapi import FastAPI, HTTPException
-from pydantic import AnyUrl, ConfigDict, Field, TypeAdapter, ValidationError
+from pydantic import ConfigDict, TypeAdapter, ValidationError
from starlette.requests import Request as StarletteRequest
from starlette.responses import JSONResponse
from starlette.types import Message, Receive, Scope, Send
-from typing_extensions import ReadOnly, TypedDict
from litellm._logging import verbose_logger
from litellm.constants import (
- MAXIMUM_TRACEBACK_LINES_TO_LOG,
MCP_GATEWAY_SESSION_ID_PREFIX_LENGTH,
)
-from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
from litellm.llms.custom_httpx.http_handler import (
get_async_httpx_client,
httpxSpecialProvider,
@@ -41,12 +35,6 @@ from litellm.proxy._experimental.mcp_server.auth.user_api_key_auth_mcp import (
MCPRequestHandler,
_is_mcp_admitted_user_subject,
)
-from litellm.proxy._experimental.mcp_server.byok_credential_cache import (
- byok_credential_cache,
- byok_credential_cache_key,
- cache_byok_credential,
- get_cached_byok_credential,
-)
from litellm.proxy._experimental.mcp_server.client_allowlist import (
MCPClientAllowlist,
check_mcp_client_allowed,
@@ -56,7 +44,6 @@ from litellm.proxy._experimental.mcp_server.discoverable_endpoints import (
get_request_base_url,
)
from litellm.proxy._experimental.mcp_server.exceptions import (
- MCPToolResultError,
MCPUpstreamAuthError,
)
from litellm.proxy._experimental.mcp_server.mcp_context import (
@@ -74,7 +61,6 @@ from litellm.proxy._experimental.mcp_server.mcp_debug import (
)
from litellm.proxy._experimental.mcp_server.oauth_utils import (
_redact_mcp_resource_url,
- get_byok_www_authenticate,
get_passthrough_www_authenticate,
get_route_relative_request_path,
well_known_root_suffix,
@@ -84,14 +70,6 @@ from litellm.proxy._experimental.mcp_server.utils import (
LITELLM_MCP_SERVER_DESCRIPTION,
LITELLM_MCP_SERVER_NAME,
LITELLM_MCP_SERVER_VERSION,
- MCPMissingUserEnvVarsError,
- add_server_prefix_to_name,
- build_synthetic_mcp_request,
- extract_mcp_tool_result_error_message,
- get_server_prefix,
- iter_known_server_prefixes,
- logging_safe_mcp_headers,
- match_known_tool_name,
)
from litellm.proxy._types import (
ProxyException,
@@ -99,13 +77,6 @@ from litellm.proxy._types import (
UserAPIKeyAuth,
)
from litellm.proxy.auth.ip_address_utils import IPAddressUtils
-from litellm.proxy.common_utils.auth_cache_invalidation_pubsub import (
- publish_auth_cache_invalidation,
-)
-from litellm.proxy.litellm_pre_call_utils import (
- LiteLLMProxyRequestSetup,
- get_chain_id_from_headers,
-)
from litellm.types.mcp import (
MCPAuth,
MCPGatewaySession,
@@ -114,14 +85,11 @@ from litellm.types.mcp import (
MCPGatewaySessionsTerminateResponse,
MCPSpecVersion,
)
-from litellm.types.mcp_server.mcp_server_manager import MCPInfo, MCPServer
-from litellm.types.utils import CallTypes, StandardLoggingMCPToolCall
-from litellm.utils import Rules, client, function_setup
+from litellm.types.mcp_server.mcp_server_manager import MCPServer
if TYPE_CHECKING:
from mcp.server.session import ServerSession as _McpServerSession
- from litellm.proxy._experimental.mcp_server.db import OAuthCredentialPayload
_STATEFUL_SESSION_IDLE_TIMEOUT_SECONDS: Final = 30 * 60
# Upper bound on concurrent stateful sessions a single caller may hold. Each
@@ -159,13 +127,6 @@ def unsupported_protocol_version(scope: Scope) -> str | None:
return None
-async def _invalidate_byok_cred_cache(user_id: str, server_id: str) -> None:
- """Drop a stored-or-deleted BYOK credential from this worker's cache and from every peer worker's."""
- cache_key: Final = byok_credential_cache_key(user_id, server_id)
- byok_credential_cache.delete_cache(cache_key)
- await publish_auth_cache_invalidation(cache_key=cache_key)
-
-
# Check if MCP is available
# "mcp" requires python 3.10 or higher, but several litellm users use python 3.8
# We're making this conditional import to avoid breaking users who use python 3.8.
@@ -210,19 +171,6 @@ _SESSION_MANAGERS_INITIALIZED = False
_INITIALIZATION_LOCK: Final = asyncio.Lock()
-def _mcp_session_id_from_headers(
- raw_headers: dict[str, str] | None,
-) -> str | None:
- """The ``mcp-session-id`` of a stateful MCP session, read case-insensitively
- from the request headers. ``None`` for stateless calls (no such header)."""
- if not raw_headers:
- return None
- for key, value in raw_headers.items():
- if isinstance(key, str) and key.lower() == "mcp-session-id":
- return value or None
- return None
-
-
def _jsonrpc_text_has_top_level_method(text: str) -> bool:
"""Whether a (possibly truncated) JSON-RPC envelope has a ``method`` key at
the root object's top level.
@@ -466,6 +414,56 @@ def _proxy_exception_to_http_exception(exc: ProxyException) -> HTTPException:
if MCP_AVAILABLE:
+ __all__ = (
+ "_MCP_CREDENTIAL_REQUEST_FIELDS",
+ "ListMCPToolsRestAPIResponseObject",
+ "_McpDeniedDetail",
+ "_aggregate_server_key",
+ "_build_virtual_call_logging_obj",
+ "_check_byok_credential",
+ "_client_has_passthrough_authorization",
+ "_client_has_per_server_auth_header",
+ "_dispatch_virtual_mcp_tool",
+ "_fire_mcp_tool_call_logging",
+ "_get_allowed_mcp_servers",
+ "_get_allowed_mcp_servers_from_mcp_server_names",
+ "_get_byok_credential",
+ "_get_prompts_from_mcp_servers",
+ "_get_resource_templates_from_mcp_servers",
+ "_get_resources_from_mcp_servers",
+ "_get_standard_logging_mcp_tool_call",
+ "_get_tools_from_mcp_servers",
+ "_get_user_oauth_extra_headers_from_db",
+ "_handle_local_mcp_tool",
+ "_handle_managed_mcp_tool",
+ "_http_detail_message",
+ "_invalidate_byok_cred_cache",
+ "_list_mcp_prompts",
+ "_list_mcp_resource_templates",
+ "_list_mcp_resources",
+ "_list_mcp_tools",
+ "_list_tools_before_first_call",
+ "_mcp_session_id_from_headers",
+ "_merge_gateway_initialize_instructions",
+ "_prefetch_oauth_creds_for_user",
+ "_prepare_mcp_server_headers",
+ "_raise_if_initialize_grants_no_mcp_servers",
+ "_redact_mcp_resource_url",
+ "_resolve_display_name_to_original",
+ "_run_post_mcp_call_guardrails",
+ "_server_answers_to",
+ "_tool_name_matches",
+ "apply_tool_overrides",
+ "call_mcp_tool",
+ "execute_mcp_tool",
+ "filter_tools_by_allowed_tools",
+ "filter_tools_by_key_team_permissions",
+ "fire_mcp_tool_call_failure_logging",
+ "global_mcp_server_manager",
+ "mcp_get_prompt",
+ "mcp_read_resource",
+ "raise_denied_scoped_mcp_access",
+ )
from mcp.server import Server
# Import auth context variables and middleware
@@ -476,6 +474,23 @@ if MCP_AVAILABLE:
from mcp.server.context import ServerRequestContext
from mcp.server.lowlevel.server import NotificationOptions
from mcp.server.models import InitializationOptions
+ from mcp.shared.exceptions import MCPError
+ from mcp.types import (
+ CallToolRequest,
+ GetPromptRequest,
+ ListPromptsRequest,
+ ListResourcesRequest,
+ ListResourceTemplatesRequest,
+ ListToolsRequest,
+ ReadResourceRequest,
+ )
+
+ from litellm.proxy._experimental.mcp_server import operations
+ from litellm.proxy._experimental.mcp_server.contracts import OperationContext
+ from litellm.proxy._experimental.mcp_server.operations import (
+ _invalidate_byok_cred_cache,
+ _mcp_session_id_from_headers,
+ )
try:
from mcp.server.streamable_http_manager import StreamableHTTPSessionManager
@@ -493,62 +508,27 @@ if MCP_AVAILABLE:
ListResourceTemplatesResult,
ListToolsResult,
PaginatedRequestParams,
- Prompt,
ReadResourceRequestParams,
- TextContent,
)
- from mcp.types import Tool as MCPTool
from mcp_types.version import HANDSHAKE_PROTOCOL_VERSIONS
from litellm.proxy._experimental.mcp_server.auth.litellm_auth_handler import (
MCPAuthenticatedUser,
)
- from litellm.proxy._experimental.mcp_server.faults.list_outcomes import (
- SERVER_OUTCOMES_META_KEY,
- AggregateToolListing,
- ServerListOk,
- ServerOutcome,
- classify_list_exception,
- outcome_wire_value,
- )
from litellm.proxy._experimental.mcp_server.mcp_server_manager import (
MCPServerManager,
- _caller_authorization_fans_out,
- _client_forwarded_authorization_headers,
- _resolve_openapi_tool_auth,
- _should_strip_caller_authorization,
global_mcp_server_manager,
)
- from litellm.proxy._experimental.mcp_server.openapi_to_mcp_generator import (
- _request_auth_header,
- _request_extra_headers,
- _request_resolved_auth_headers,
- )
- from litellm.proxy._experimental.mcp_server.sse_transport import SseServerTransport
- from litellm.proxy._experimental.mcp_server.tool_registry import (
- global_mcp_tool_registry,
- )
- from litellm.proxy._experimental.mcp_server.utils import (
- MCP_TOOL_PREFIX_SEPARATOR,
- is_tool_name_prefixed,
- normalize_server_name,
- split_server_prefix_from_name,
- strip_known_server_prefix,
- )
- from litellm.types.mcp import DEFAULT_CREDENTIAL_HEADER, without_header
######################################################
############ MCP Tools List REST API Response Object #
# Defined here because we don't want to add `mcp` as a
# required dependency for `litellm` pip package
######################################################
- class ListMCPToolsRestAPIResponseObject(MCPTool):
- """
- Object returned by the /tools/list REST API route.
- """
-
- mcp_info: MCPInfo | None = Field(default=None, alias="mcp_info")
- model_config = ConfigDict(arbitrary_types_allowed=True)
+ from litellm.proxy._experimental.mcp_server.operations import (
+ ListMCPToolsRestAPIResponseObject,
+ )
+ from litellm.proxy._experimental.mcp_server.sse_transport import SseServerTransport
def _gateway_create_initialization_options(
self,
@@ -818,94 +798,45 @@ if MCP_AVAILABLE:
############### MCP Server Routes #######################
########################################################
- async def handle_list_tools(ctx: ServerRequestContext, params: PaginatedRequestParams) -> ListToolsResult:
- """
- List all available tools, with each server's listing outcome attached to the result's
- ``_meta`` (SERVER_OUTCOMES_META_KEY) so a broken upstream is distinguishable from a healthy
- server with no tools. Returning a ListToolsResult (rather than a bare list) makes the MCP SDK
- pass the result through unwrapped, which is what lets the ``_meta`` survive to the client.
- Also captures the active session for propagation to callbacks.
- """
- req_ctx: Final = ctx
- _ctx_reset_token: Final = active_mcp_request_ctx_var.set(ctx)
- _session_reset_token: Final = active_mcp_session_var.set(ctx.session)
- _trace_token = None
- _transport_token = None
- _destinations_token = None
-
- try:
- _trace_token = _otel_set_mcp_trace_carrier(_mcp_meta_trace_carrier(req_ctx))
- _transport_token = _otel_set_mcp_transport_span(_otel_transport_span_from_message(req_ctx))
- _destinations_token = _otel_set_mcp_request_destinations(req_ctx)
- # Get user authentication from context variable
+ @contextlib.asynccontextmanager
+ async def _legacy_operation_context(ctx: ServerRequestContext, *, trace: bool) -> AsyncGenerator[OperationContext]:
+ with contextlib.ExitStack() as cleanup:
+ cleanup.callback(active_mcp_request_ctx_var.reset, active_mcp_request_ctx_var.set(ctx))
+ cleanup.callback(active_mcp_session_var.reset, active_mcp_session_var.set(ctx.session))
+ if trace:
+ cleanup.callback(
+ _otel_reset_mcp_trace_carrier, _otel_set_mcp_trace_carrier(_mcp_meta_trace_carrier(ctx))
+ )
+ cleanup.callback(
+ _otel_reset_mcp_transport_span, _otel_set_mcp_transport_span(_otel_transport_span_from_message(ctx))
+ )
+ cleanup.callback(_otel_reset_mcp_request_destinations, _otel_set_mcp_request_destinations(ctx))
(
- user_api_key_auth,
- mcp_auth_header,
- mcp_servers,
- mcp_server_auth_headers,
- oauth2_headers,
- raw_headers,
- _client_ip,
+ auth,
+ token,
+ servers,
+ server_headers,
+ oauth_headers,
+ headers,
+ client_ip,
) = await get_or_extract_auth_context()
- verbose_logger.debug("MCP list_tools - User API Key Auth from context: %s", user_api_key_auth)
- verbose_logger.debug("MCP list_tools - MCP servers from context: %s", mcp_servers)
- verbose_logger.debug(
- "MCP list_tools - MCP server auth headers: %s",
- list(mcp_server_auth_headers.keys()) if mcp_server_auth_headers else None,
- )
- from mcp.types import Tool
-
- from litellm.proxy._experimental.mcp_server.tool_search import (
- get_mcp_proxy_tool_definitions,
- get_virtual_tool_definitions,
+ yield operations.prepare_context(
+ auth, token, servers, server_headers, oauth_headers, headers, client_ip, _mcp_proxy_mode.get()
)
- if _mcp_proxy_mode.get():
- return ListToolsResult(tools=[Tool.model_validate(d) for d in get_mcp_proxy_tool_definitions()])
- if getattr(
- getattr(user_api_key_auth, "object_permission", None),
- "mcp_tool_search_enabled",
- False,
- ):
- return ListToolsResult(tools=[Tool.model_validate(d) for d in get_virtual_tool_definitions()])
-
- # Get mcp_servers from context variable
- verbose_logger.debug("MCP list_tools - Calling _list_mcp_tools")
- listing: Final = await _list_mcp_tools(
- user_api_key_auth=user_api_key_auth,
- mcp_auth_header=mcp_auth_header,
- mcp_servers=mcp_servers,
- mcp_server_auth_headers=mcp_server_auth_headers,
- oauth2_headers=oauth2_headers,
- raw_headers=raw_headers,
- log_list_tools_to_spendlogs=True,
- list_tools_log_source="mcp_protocol",
- )
- verbose_logger.info("MCP list_tools - Successfully returned %s tools", len(listing.tools))
- if not listing.outcomes:
- return ListToolsResult(tools=listing.tools)
- outcome_meta: Final = {
- SERVER_OUTCOMES_META_KEY: {
- key: outcome_wire_value(outcome) for key, outcome in listing.outcomes.items()
- }
- }
- return ListToolsResult.model_validate({"tools": listing.tools, "_meta": outcome_meta})
- except HTTPException as e:
- from mcp.shared.exceptions import MCPError
- from mcp.types import INVALID_REQUEST
-
- raise MCPError(code=INVALID_REQUEST, message=_http_detail_message(e.detail)) from e
- except Exception as e:
- verbose_logger.exception("Error in list_tools endpoint: %s", e)
- # Return empty list instead of failing completely
- # This prevents the HTTP stream from failing and allows the client to get a response
- return ListToolsResult(tools=[]) # mutable-ok: MCP result payload
- finally:
- _otel_reset_mcp_request_destinations(_destinations_token)
- _otel_reset_mcp_transport_span(_transport_token)
- _otel_reset_mcp_trace_carrier(_trace_token)
- active_mcp_session_var.reset(_session_reset_token)
- active_mcp_request_ctx_var.reset(_ctx_reset_token)
+ async def handle_list_tools(ctx: ServerRequestContext, params: PaginatedRequestParams) -> ListToolsResult:
+ try:
+ async with _legacy_operation_context(ctx, trace=True) as context:
+ return await operations.GatewayOperations(_capture_host_progress_callback(ctx)).execute(
+ ListToolsRequest(params=params), context
+ )
+ except MCPError:
+ raise
+ except HTTPException as exc:
+ raise MCPError(code=INVALID_REQUEST, message=operations._http_detail_message(exc.detail)) from exc
+ except Exception as exc: # noqa: BLE001 # preserve native listing fallback for ingress failures
+ verbose_logger.exception("Error in list_tools endpoint: %s", exc)
+ return ListToolsResult(tools=[])
def _capture_host_progress_callback(ctx: ServerRequestContext) -> Callable | None:
"""Return a progress-forwarding callback bound to the host MCP session.
@@ -942,581 +873,71 @@ if MCP_AVAILABLE:
raise MCPError(code=METHOD_NOT_FOUND, message="Operation unavailable on /mcp/proxy")
- async def _build_virtual_call_logging_obj(
- name: str,
- arguments: dict[str, object],
- user_api_key_auth: UserAPIKeyAuth,
- raw_headers: Mapping[str, str] | None = None,
- client_ip: str | None = None,
- ) -> LiteLLMLoggingObj | None:
- """Run the pre-call pipeline (guardrails + logging setup) for a virtual
- mcp_tool_call so the SSE path spend-logs like the REST path."""
- from litellm.proxy.common_request_processing import (
- ProxyBaseLLMRequestProcessing,
- )
- from litellm.proxy.proxy_server import (
- general_settings,
- proxy_config,
- proxy_logging_obj,
- )
-
- request: Final = build_synthetic_mcp_request(
- path="/mcp/tools/call",
- raw_headers=raw_headers,
- client_ip=client_ip,
- )
- _, virtual_logging_obj = await ProxyBaseLLMRequestProcessing(
- data={"name": name, "arguments": arguments}
- ).common_processing_pre_call_logic(
- request=request,
- user_api_key_dict=user_api_key_auth,
- proxy_config=proxy_config,
- route_type=CallTypes.call_mcp_tool.value,
- proxy_logging_obj=proxy_logging_obj,
- general_settings=general_settings,
- )
- return virtual_logging_obj
-
- async def _dispatch_virtual_mcp_tool(
- name: str,
- arguments: dict[str, object] | None,
- user_api_key_auth: UserAPIKeyAuth | None,
- client_ip: str | None,
- mcp_servers: list[str] | None = None,
- mcp_auth_header: str | None = None,
- mcp_server_auth_headers: dict[str, dict[str, str]] | None = None,
- oauth2_headers: dict[str, str] | None = None,
- raw_headers: dict[str, str] | None = None,
- ) -> CallToolResult | None:
- """Handle the mcp_tool_search / mcp_tool_call virtual tools.
-
- Returns a CallToolResult when ``name`` is a virtual tool, else ``None`` so
- the caller falls through to normal tool routing.
- """
- from litellm.llms.litellm_proxy.skills.skill_search import DEFAULT_SKILL_SEARCH_TOP_K
- from litellm.proxy._experimental.mcp_server.tool_search import (
- AGENT_SEARCH_TOOL_NAME,
- DEFAULT_AGENT_SEARCH_TOP_K,
- MCP_PROXY_CALL_TOOL_NAME,
- MCP_PROXY_TOOL_NAMES,
- MCP_TOOL_SEARCH_TOOL_NAME,
- SKILL_SEARCH_TOOL_NAME,
- VIRTUAL_TOOL_NAMES,
- coerce_top_k,
- handle_agent_search,
- handle_mcp_proxy_tool,
- handle_mcp_tool_call,
- handle_mcp_tool_search,
- handle_skill_search,
- )
-
- if _mcp_proxy_mode.get() and name not in MCP_PROXY_TOOL_NAMES:
- return CallToolResult(
- content=[ # mutable-ok: MCP result content
- TextContent(type="text", text=f"Tool {name} is unavailable on /mcp/proxy")
- ],
- is_error=True,
- )
-
- if _mcp_proxy_mode.get() and name in MCP_PROXY_TOOL_NAMES:
- assert user_api_key_auth is not None
- proxy_call_start: Final = datetime.now() # noqa: DTZ005 # logging pipeline uses naive datetimes
- proxy_logging_obj: Final = (
- await _build_virtual_call_logging_obj(
- name=name,
- arguments=arguments or {}, # mutable-ok: logging pipeline payload
- user_api_key_auth=user_api_key_auth,
- raw_headers=raw_headers,
- client_ip=client_ip,
- )
- if name == MCP_PROXY_CALL_TOOL_NAME
- else None
- )
- try:
- proxy_result: Final = await handle_mcp_proxy_tool(
- name=name,
- arguments=arguments or {}, # mutable-ok: proxy handler payload
- user_api_key_dict=user_api_key_auth,
- client_ip=client_ip,
- mcp_servers=mcp_servers,
- mcp_auth_header=mcp_auth_header,
- mcp_server_auth_headers=mcp_server_auth_headers,
- oauth2_headers=oauth2_headers,
- raw_headers=raw_headers,
- litellm_logging_obj=proxy_logging_obj,
- )
- except Exception as exc:
- if proxy_logging_obj is not None:
- from litellm.proxy.proxy_server import proxy_logging_obj as request_logging_obj
-
- failure_end: Final = datetime.now() # noqa: DTZ005 # matches the logging pipeline start time
- failure_traceback: Final = traceback.format_exc(limit=MAXIMUM_TRACEBACK_LINES_TO_LOG)
- try:
- proxy_logging_obj.failure_handler(exc, failure_traceback, proxy_call_start, failure_end)
- await proxy_logging_obj.async_failure_handler(
- exc, failure_traceback, proxy_call_start, failure_end
- )
- if not isinstance(exc, MCPUpstreamAuthError):
- await request_logging_obj.post_call_failure_hook(
- request_data={ # mutable-ok: failure hook mutates its request payload
- "name": name,
- "arguments": arguments,
- "litellm_logging_obj": proxy_logging_obj,
- },
- original_exception=exc,
- user_api_key_dict=user_api_key_auth,
- route="/mcp/call_tool",
- traceback_str=failure_traceback,
- )
- except Exception: # noqa: BLE001 # a failing failure hook must not mask the tool call's own error
- verbose_logger.exception("Error logging failed MCP proxy tool call")
- raise
- if proxy_logging_obj is not None:
- return await _fire_mcp_tool_call_logging(
- logging_obj=proxy_logging_obj,
- result=proxy_result,
- start_time=proxy_call_start,
- end_time=datetime.now(), # noqa: DTZ005 # matches the logging pipeline start time
- user_api_key_auth=user_api_key_auth,
- request_data=types.MappingProxyType({"name": name, "arguments": arguments}),
- )
- return proxy_result
-
- if name not in VIRTUAL_TOOL_NAMES:
- return None
-
- if not getattr(
- getattr(user_api_key_auth, "object_permission", None),
- "mcp_tool_search_enabled",
- False,
- ):
- return CallToolResult(
- content=[
- TextContent(
- type="text",
- text=f"Tool {name} requires mcp_tool_search_enabled on the key",
- )
- ],
- is_error=True,
- )
-
- args: Final = arguments or {}
- if name == MCP_TOOL_SEARCH_TOOL_NAME:
- return await handle_mcp_tool_search(
- query=args.get("query", ""),
- top_k=coerce_top_k(args.get("top_k", 5)),
- user_api_key_dict=user_api_key_auth,
- client_ip=client_ip,
- mcp_servers=mcp_servers,
- mcp_auth_header=mcp_auth_header,
- mcp_server_auth_headers=mcp_server_auth_headers,
- oauth2_headers=oauth2_headers,
- raw_headers=raw_headers,
- )
-
- assert user_api_key_auth is not None # guaranteed by the flag check above
- if name == AGENT_SEARCH_TOOL_NAME:
- return await handle_agent_search(
- query=str(args.get("query", "")),
- top_k=coerce_top_k(args.get("top_k", DEFAULT_AGENT_SEARCH_TOP_K), default=DEFAULT_AGENT_SEARCH_TOP_K),
- user_api_key_dict=user_api_key_auth,
- )
- if name == SKILL_SEARCH_TOOL_NAME:
- return await handle_skill_search(
- query=str(args.get("query", "")),
- top_k=coerce_top_k(args.get("top_k", DEFAULT_SKILL_SEARCH_TOP_K), default=DEFAULT_SKILL_SEARCH_TOP_K),
- user_api_key_dict=user_api_key_auth,
- )
- virtual_logging_obj: Final = await _build_virtual_call_logging_obj(
- name=name,
- arguments=args,
- user_api_key_auth=user_api_key_auth,
- raw_headers=raw_headers,
- client_ip=client_ip,
- )
- return await handle_mcp_tool_call(
- tool_name=args.get("tool_name", ""),
- arguments=args.get("arguments") or {},
- user_api_key_dict=user_api_key_auth,
- client_ip=client_ip,
- mcp_servers=mcp_servers,
- mcp_auth_header=mcp_auth_header,
- mcp_server_auth_headers=mcp_server_auth_headers,
- oauth2_headers=oauth2_headers,
- raw_headers=raw_headers,
- litellm_logging_obj=virtual_logging_obj,
- )
+ from litellm.proxy._experimental.mcp_server.operations import (
+ _build_virtual_call_logging_obj,
+ _dispatch_virtual_mcp_tool,
+ )
async def mcp_server_tool_call(ctx: ServerRequestContext, params: CallToolRequestParams) -> CallToolResult:
- """
- Call a specific tool with the provided arguments
- Args:
- ctx: SDK request context carrying the client session and HTTP request
- params (CallToolRequestParams): Tool name and arguments
- Returns:
- CallToolResult: Tool execution results
- """
- from mcp.types import CallToolResult
-
- from litellm.exceptions import BlockedPiiEntityError, GuardrailRaisedException
- from litellm.proxy.litellm_pre_call_utils import add_litellm_data_to_request
- from litellm.proxy.proxy_server import proxy_config
-
- req_ctx: Final = ctx
- _ctx_reset_token: Final = active_mcp_request_ctx_var.set(ctx)
- _session_reset_token: Final = active_mcp_session_var.set(ctx.session)
- _trace_token = None
- _transport_token = None
- _destinations_token = None
-
- try:
- _trace_token = _otel_set_mcp_trace_carrier(_mcp_meta_trace_carrier(req_ctx))
- _transport_token = _otel_set_mcp_transport_span(_otel_transport_span_from_message(req_ctx))
- _destinations_token = _otel_set_mcp_request_destinations(req_ctx)
- # Validate arguments
- (
- user_api_key_auth,
- mcp_auth_header,
- mcp_servers,
- mcp_server_auth_headers,
- oauth2_headers,
- raw_headers,
- _client_ip,
- ) = await get_or_extract_auth_context()
- verbose_logger.debug(
- "MCP mcp_server_tool_call - user_api_key_auth=%s, user_role=%s",
- user_api_key_auth,
- getattr(user_api_key_auth, "user_role", "N/A"),
+ async with _legacy_operation_context(ctx, trace=True) as context:
+ return await operations.GatewayOperations(_capture_host_progress_callback(ctx)).execute(
+ CallToolRequest(params=params), context
)
- verbose_logger.debug("MCP mcp_server_tool_call - User API Key Auth from context: %s", user_api_key_auth)
-
- try:
- # Inside this try so virtual-tool errors convert to isError
- # CallToolResult instead of raising out of the protocol handler.
- virtual_tool_result: Final = await _dispatch_virtual_mcp_tool(
- name=params.name,
- arguments=params.arguments,
- user_api_key_auth=user_api_key_auth,
- client_ip=_client_ip,
- mcp_servers=mcp_servers,
- mcp_auth_header=mcp_auth_header,
- mcp_server_auth_headers=mcp_server_auth_headers,
- oauth2_headers=oauth2_headers,
- raw_headers=raw_headers,
- )
- if virtual_tool_result is not None:
- return virtual_tool_result
-
- host_progress_callback: Final = _capture_host_progress_callback(ctx)
- # Create a body date for logging
- body_data: Final = {"name": params.name, "arguments": params.arguments} # mutable-ok: logging payload
- # Set trace/session id from raw_headers so spend logs and logging_obj stay consistent (same as A2A)
- chain_id: Final = get_chain_id_from_headers(raw_headers)
- if chain_id:
- body_data["litellm_trace_id"] = chain_id
- body_data["litellm_session_id"] = chain_id
-
- request: Final = build_synthetic_mcp_request(
- path="/mcp/tools/call",
- raw_headers=raw_headers,
- client_ip=_client_ip,
- )
- if user_api_key_auth is not None:
- data = await add_litellm_data_to_request(
- data=body_data,
- request=request,
- # Bill a team-derived call to the team that granted it. A keyless admitted
- # subject carries no team_id, so spend skipped team updates entirely and
- # charged the user's PRIMARY org — the granting team's budget never
- # accumulated (so it could never begin to block) and, cross-org, the wrong
- # organization was charged. This is the ACCOUNTING half; the enforcement
- # half (an already-over-budget team stops granting) lives in the source gate.
- # Authorization is unaffected: it ran before this, and the union is resolved
- # from the untouched auth object passed to call_mcp_tool below.
- user_api_key_dict=await MCPRequestHandler.billing_auth_for_tool_call(
- user_api_key_auth, tool_name=params.name
- ),
- proxy_config=proxy_config,
- )
- else:
- data = body_data
-
- response: Final = await call_mcp_tool(
- user_api_key_auth=user_api_key_auth,
- mcp_auth_header=mcp_auth_header,
- mcp_servers=mcp_servers,
- mcp_server_auth_headers=mcp_server_auth_headers,
- oauth2_headers=oauth2_headers,
- raw_headers=raw_headers,
- client_ip=_client_ip,
- host_progress_callback=host_progress_callback,
- **data, # for logging
- )
- except MCPMissingUserEnvVarsError as e:
- verbose_logger.info(
- "MCP mcp_server_tool_call missing per-user env vars: server_id=%s missing=%s",
- e.server_id,
- e.missing,
- )
- return CallToolResult(
- content=[TextContent(text=str(e), type="text")],
- is_error=True,
- )
- except BlockedPiiEntityError as e:
- verbose_logger.error("BlockedPiiEntityError in MCP tool call: %s", e)
- return CallToolResult(
- content=[
- TextContent(
- text=f"Error: Blocked PII entity detected - {e}",
- type="text",
- )
- ],
- is_error=True,
- )
- except GuardrailRaisedException as e:
- verbose_logger.error("GuardrailRaisedException in MCP tool call: %s", e)
- return CallToolResult(
- content=[TextContent(text=f"Error: Guardrail violation - {e}", type="text")],
- is_error=True,
- )
- except HTTPException as e:
- verbose_logger.error("HTTPException in MCP tool call: %s", e)
- return CallToolResult(
- content=[TextContent(text=f"Error: {_http_detail_message(e.detail)}", type="text")],
- is_error=True,
- )
- except MCPUpstreamAuthError as e:
- # The MCP session manager serializes handler exceptions as JSON-RPC errors, so a
- # mid-session tool call cannot emit a raw 401 + WWW-Authenticate the way the REST
- # call path and the connect-time preemptive check do. Return an explicit isError
- # naming the upstream status (at info level, not a traceback) so the client still
- # learns it must re-authenticate upstream and expected pass-through 401s don't spam.
- verbose_logger.info("Upstream auth failure calling MCP tool: HTTP %s", e.status_code)
- return CallToolResult(
- content=[
- TextContent(
- text=f"Error: upstream authentication required (HTTP {e.status_code})",
- type="text",
- )
- ],
- is_error=True,
- )
- except Exception as e:
- verbose_logger.exception("MCP mcp_server_tool_call - error: %s", e)
- return CallToolResult(
- content=[TextContent(text=f"Error: {e}", type="text")],
- is_error=True,
- )
-
- return response
- finally:
- _otel_reset_mcp_request_destinations(_destinations_token)
- _otel_reset_mcp_transport_span(_transport_token)
- _otel_reset_mcp_trace_carrier(_trace_token)
- active_mcp_session_var.reset(_session_reset_token)
- active_mcp_request_ctx_var.reset(_ctx_reset_token)
-
async def list_prompts(ctx: ServerRequestContext, params: PaginatedRequestParams) -> ListPromptsResult:
- """
- List all available prompts
- """
if _mcp_proxy_mode.get():
_reject_mcp_proxy_operation()
- _ctx_reset_token: Final = active_mcp_request_ctx_var.set(ctx)
- _session_reset_token: Final = active_mcp_session_var.set(ctx.session)
-
try:
- # Get user authentication from context variable
- (
- user_api_key_auth,
- mcp_auth_header,
- mcp_servers,
- mcp_server_auth_headers,
- oauth2_headers,
- raw_headers,
- _client_ip,
- ) = await get_or_extract_auth_context()
- verbose_logger.debug("MCP list_prompts - User API Key Auth from context: %s", user_api_key_auth)
- verbose_logger.debug("MCP list_prompts - MCP servers from context: %s", mcp_servers)
- verbose_logger.debug(
- "MCP list_prompts - MCP server auth headers: %s",
- list(mcp_server_auth_headers.keys()) if mcp_server_auth_headers else None,
- )
- # Get mcp_servers from context variable
- verbose_logger.debug("MCP list_prompts - Calling _list_prompts")
- prompts: Final = await _list_mcp_prompts(
- user_api_key_auth=user_api_key_auth,
- mcp_auth_header=mcp_auth_header,
- mcp_servers=mcp_servers,
- mcp_server_auth_headers=mcp_server_auth_headers,
- oauth2_headers=oauth2_headers,
- raw_headers=raw_headers,
- )
- verbose_logger.info("MCP list_prompts - Successfully returned %s prompts", len(prompts))
- return ListPromptsResult(prompts=prompts)
- except Exception as e:
- verbose_logger.exception("Error in list_prompts endpoint: %s", e)
- # Return empty list instead of failing completely
- # This prevents the HTTP stream from failing and allows the client to get a response
- return ListPromptsResult(prompts=[]) # mutable-ok: MCP result payload
- finally:
- active_mcp_session_var.reset(_session_reset_token)
- active_mcp_request_ctx_var.reset(_ctx_reset_token)
+ async with _legacy_operation_context(ctx, trace=False) as context:
+ return await operations.GatewayOperations(_capture_host_progress_callback(ctx)).execute(
+ ListPromptsRequest(params=params), context
+ )
+ except Exception as exc: # noqa: BLE001 # preserve native listing fallback for ingress failures
+ verbose_logger.exception("Error in list_prompts endpoint: %s", exc)
+ return ListPromptsResult(prompts=[])
async def get_prompt(ctx: ServerRequestContext, params: GetPromptRequestParams) -> GetPromptResult:
- """
- Get a specific prompt with the provided arguments
- """
if _mcp_proxy_mode.get():
_reject_mcp_proxy_operation()
- _ctx_reset_token: Final = active_mcp_request_ctx_var.set(ctx)
- _session_reset_token: Final = active_mcp_session_var.set(ctx.session)
-
- try:
- (
- user_api_key_auth,
- mcp_auth_header,
- mcp_servers,
- mcp_server_auth_headers,
- oauth2_headers,
- raw_headers,
- _client_ip,
- ) = await get_or_extract_auth_context()
-
- verbose_logger.debug("MCP mcp_server_tool_call - User API Key Auth from context: %s", user_api_key_auth)
- return await mcp_get_prompt(
- name=params.name,
- arguments=params.arguments,
- user_api_key_auth=user_api_key_auth,
- mcp_auth_header=mcp_auth_header,
- mcp_servers=mcp_servers,
- mcp_server_auth_headers=mcp_server_auth_headers,
- oauth2_headers=oauth2_headers,
- raw_headers=raw_headers,
+ async with _legacy_operation_context(ctx, trace=False) as context:
+ return await operations.GatewayOperations(_capture_host_progress_callback(ctx)).execute(
+ GetPromptRequest(params=params), context
)
- finally:
- active_mcp_session_var.reset(_session_reset_token)
- active_mcp_request_ctx_var.reset(_ctx_reset_token)
async def list_resources(ctx: ServerRequestContext, params: PaginatedRequestParams) -> ListResourcesResult:
- """List all available resources."""
if _mcp_proxy_mode.get():
_reject_mcp_proxy_operation()
- _ctx_reset_token: Final = active_mcp_request_ctx_var.set(ctx)
- _session_reset_token: Final = active_mcp_session_var.set(ctx.session)
-
try:
- (
- user_api_key_auth,
- mcp_auth_header,
- mcp_servers,
- mcp_server_auth_headers,
- oauth2_headers,
- raw_headers,
- _client_ip,
- ) = await get_or_extract_auth_context()
- verbose_logger.debug("MCP list_resources - User API Key Auth from context: %s", user_api_key_auth)
- verbose_logger.debug("MCP list_resources - MCP servers from context: %s", mcp_servers)
- verbose_logger.debug(
- "MCP list_resources - MCP server auth headers: %s",
- list(mcp_server_auth_headers.keys()) if mcp_server_auth_headers else None,
- )
-
- resources: Final = await _list_mcp_resources(
- user_api_key_auth=user_api_key_auth,
- mcp_auth_header=mcp_auth_header,
- mcp_servers=mcp_servers,
- mcp_server_auth_headers=mcp_server_auth_headers,
- oauth2_headers=oauth2_headers,
- raw_headers=raw_headers,
- )
- verbose_logger.info("MCP list_resources - Successfully returned %s resources", len(resources))
- return ListResourcesResult(resources=resources)
- except Exception as e:
- verbose_logger.exception("Error in list_resources endpoint: %s", e)
- return ListResourcesResult(resources=[]) # mutable-ok: MCP result payload
- finally:
- active_mcp_session_var.reset(_session_reset_token)
- active_mcp_request_ctx_var.reset(_ctx_reset_token)
+ async with _legacy_operation_context(ctx, trace=False) as context:
+ return await operations.GatewayOperations(_capture_host_progress_callback(ctx)).execute(
+ ListResourcesRequest(params=params), context
+ )
+ except Exception as exc: # noqa: BLE001 # preserve native listing fallback for ingress failures
+ verbose_logger.exception("Error in list_resources endpoint: %s", exc)
+ return ListResourcesResult(resources=[])
async def list_resource_templates(
ctx: ServerRequestContext, params: PaginatedRequestParams
) -> ListResourceTemplatesResult:
- """List all available resource templates."""
if _mcp_proxy_mode.get():
_reject_mcp_proxy_operation()
- _ctx_reset_token: Final = active_mcp_request_ctx_var.set(ctx)
- _session_reset_token: Final = active_mcp_session_var.set(ctx.session)
-
try:
- (
- user_api_key_auth,
- mcp_auth_header,
- mcp_servers,
- mcp_server_auth_headers,
- oauth2_headers,
- raw_headers,
- _client_ip,
- ) = await get_or_extract_auth_context()
- verbose_logger.debug("MCP list_resource_templates - User API Key Auth from context: %s", user_api_key_auth)
- verbose_logger.debug("MCP list_resource_templates - MCP servers from context: %s", mcp_servers)
- verbose_logger.debug(
- "MCP list_resource_templates - MCP server auth headers: %s",
- list(mcp_server_auth_headers.keys()) if mcp_server_auth_headers else None,
- )
-
- resource_templates: Final = await _list_mcp_resource_templates(
- user_api_key_auth=user_api_key_auth,
- mcp_auth_header=mcp_auth_header,
- mcp_servers=mcp_servers,
- mcp_server_auth_headers=mcp_server_auth_headers,
- oauth2_headers=oauth2_headers,
- raw_headers=raw_headers,
- )
- verbose_logger.info(
- "MCP list_resource_templates - Successfully returned %s resource templates", len(resource_templates)
- )
- return ListResourceTemplatesResult(resource_templates=resource_templates)
- except Exception as e:
- verbose_logger.exception("Error in list_resource_templates endpoint: %s", e)
- return ListResourceTemplatesResult(resource_templates=[]) # mutable-ok: MCP result payload
- finally:
- active_mcp_session_var.reset(_session_reset_token)
- active_mcp_request_ctx_var.reset(_ctx_reset_token)
+ async with _legacy_operation_context(ctx, trace=False) as context:
+ return await operations.GatewayOperations(_capture_host_progress_callback(ctx)).execute(
+ ListResourceTemplatesRequest(params=params), context
+ )
+ except Exception as exc: # noqa: BLE001 # preserve native listing fallback for ingress failures
+ verbose_logger.exception("Error in list_resource_templates endpoint: %s", exc)
+ return ListResourceTemplatesResult(resource_templates=[])
async def read_resource(ctx: ServerRequestContext, params: ReadResourceRequestParams) -> ReadResourceResult:
if _mcp_proxy_mode.get():
_reject_mcp_proxy_operation()
- _ctx_reset_token: Final = active_mcp_request_ctx_var.set(ctx)
- _session_reset_token: Final = active_mcp_session_var.set(ctx.session)
-
- try:
- (
- user_api_key_auth,
- mcp_auth_header,
- mcp_servers,
- mcp_server_auth_headers,
- oauth2_headers,
- raw_headers,
- _client_ip,
- ) = await get_or_extract_auth_context()
-
- read_resource_result: Final = await mcp_read_resource(
- url=params.uri,
- user_api_key_auth=user_api_key_auth,
- mcp_auth_header=mcp_auth_header,
- mcp_servers=mcp_servers,
- mcp_server_auth_headers=mcp_server_auth_headers,
- oauth2_headers=oauth2_headers,
- raw_headers=raw_headers,
+ async with _legacy_operation_context(ctx, trace=False) as context:
+ return await operations.GatewayOperations(_capture_host_progress_callback(ctx)).execute(
+ ReadResourceRequest(params=params), context
)
- return read_resource_result
- finally:
- active_mcp_session_var.reset(_session_reset_token)
- active_mcp_request_ctx_var.reset(_ctx_reset_token)
-
server.add_request_handler("tools/list", PaginatedRequestParams, handle_list_tools)
server.add_request_handler("tools/call", CallToolRequestParams, mcp_server_tool_call)
server.add_request_handler("prompts/list", PaginatedRequestParams, list_prompts)
@@ -1533,527 +954,24 @@ if MCP_AVAILABLE:
############ Helper Functions ##########################
########################################################
- async def _get_allowed_mcp_servers_from_mcp_server_names(
- mcp_servers: Sequence[str] | None,
- allowed_mcp_servers: list[MCPServer],
- ) -> list[MCPServer]:
- """
- Get the filtered MCP servers from the MCP server names.
-
- Fails closed when ``mcp_servers`` is explicitly provided (path- or
- header-derived) but none of the names resolve to a server alias or
- access group the caller can access. The previous behavior returned
- the full ``allowed_mcp_servers`` set, which silently widened scope
- when a client targeted ``/mcp//`` and made URL/header
- namespacing appear to work when it did not.
- """
-
- filtered_server: Final[dict[str, MCPServer]] = {}
- # Filter servers based on mcp_servers parameter if provided
- if mcp_servers is not None:
- for server_or_group in mcp_servers:
- server_name_matched = False
-
- for server in allowed_mcp_servers:
- if server and _server_answers_to(server, server_or_group):
- filtered_server[server.server_id] = server
- server_name_matched = True
- break
-
- if not server_name_matched:
- try:
- access_group_server_ids = await MCPRequestHandler._get_mcp_servers_from_access_groups(
- [server_or_group]
- )
- # Only include servers that the user has access to
- for server_id in access_group_server_ids:
- for server in allowed_mcp_servers:
- if server_id == server.server_id:
- filtered_server[server.server_id] = server
- except Exception as e:
- verbose_logger.debug("Could not resolve '%s' as access group: %s", server_or_group, e)
-
- if filtered_server:
- return list(filtered_server.values())
-
- if mcp_servers is not None:
- # Caller asked for a specific scope but nothing resolved. Fail
- # closed so URL/header namespacing cannot silently fall back to
- # the caller's full allowed-server set.
- verbose_logger.debug(
- "MCP scope filter resolved to no servers for requested names %s; returning empty list (fail-closed).",
- mcp_servers,
- )
- return []
-
- return allowed_mcp_servers
-
- def _http_detail_message(detail: object) -> str:
- return str(detail.get("error")) if isinstance(detail, dict) and detail.get("error") else str(detail)
-
- def _server_answers_to(server: MCPServer, name: str) -> bool:
- requested: Final = name.lower()
- return any(requested == known.lower() for known in iter_known_server_prefixes(server) if known)
-
- class _McpDeniedDetail(TypedDict):
- error: ReadOnly[str]
-
- async def raise_denied_scoped_mcp_access(
- requested_names: Sequence[str],
- user_api_key_auth: UserAPIKeyAuth | None,
- client_ip: str | None = None,
- ) -> None:
- """A scoped request (``/mcp/`` path or ``x-mcp-servers`` header) resolved to zero
- allowed servers, so the denial must be loud: a silent 200 with no tools reads as a healthy
- server with no tools. Unknown, unauthorized, and access-group names all share one generic
- error so scoping cannot probe which servers exist; the agent variant fires only when the
- same request resolves once the agent binding is stripped, proving the binding caused the veto."""
- agent_id: Final = user_api_key_auth.agent_id if user_api_key_auth else None
- if user_api_key_auth is not None and agent_id:
- resolved_without_agent: Final = await _get_allowed_mcp_servers(
- user_api_key_auth=user_api_key_auth.model_copy(update=types.MappingProxyType({"agent_id": None})),
- mcp_servers=requested_names,
- client_ip=client_ip,
- )
-
- def _resolved_to_server(name: str) -> bool:
- return any(_server_answers_to(server, name) for server in resolved_without_agent)
-
- vetoed_server: Final = next((name for name in requested_names if _resolved_to_server(name)), None)
- if vetoed_server is not None:
- agent_denial: Final[_McpDeniedDetail] = {
- "error": (
- f"MCP server '{vetoed_server}' is not available to this key: the key is bound to "
- f"agent '{agent_id}', whose MCP grants do not include this server. Add the server "
- f"to the agent's object_permission.mcp_servers (edit the agent in the Admin UI or "
- f"PATCH /v1/agents/{agent_id}), or use a key that is not bound to the agent."
- )
- }
- raise HTTPException(status_code=403, detail=agent_denial)
- vetoed_group: Final = next(
- (
- name
- for name in requested_names
- if not _resolved_to_server(name)
- and any(name in (server.access_groups or ()) for server in resolved_without_agent)
- ),
- None,
- )
- if vetoed_group is not None:
- group_denial: Final[_McpDeniedDetail] = {
- "error": (
- f"MCP access group '{vetoed_group}' is not available to this key: the key is bound to "
- f"agent '{agent_id}', whose MCP grants do not include it. Add the group to the "
- f"agent's object_permission.mcp_access_groups (edit the agent in the Admin UI or "
- f"PATCH /v1/agents/{agent_id}), or use a key that is not bound to the agent."
- )
- }
- raise HTTPException(status_code=403, detail=group_denial)
- generic_denial: Final[_McpDeniedDetail] = {
- "error": f"The key is not allowed to access the requested MCP servers: {', '.join(requested_names)}"
- }
- raise HTTPException(status_code=403, detail=generic_denial)
-
- def _tool_name_matches(tool_name: str, filter_list: list[str], mcp_server: MCPServer) -> bool:
- """
- Check if a tool name matches any name in the filter list.
-
- Reads the same owner the server-level permission checks use, so discovery hides
- exactly what dispatch refuses. ``mcp_server`` is required: guessing the boundary
- at the first separator mismatches every tool on a server whose prefix contains
- the separator.
- """
- bare_name: Final = strip_known_server_prefix(tool_name, mcp_server)
- return match_known_tool_name(bare_name, mcp_server, filter_list) is not None
-
- def filter_tools_by_allowed_tools(
- tools: list[MCPTool],
- mcp_server: MCPServer,
- ) -> list[MCPTool]:
- """
- Filter tools by allowed/disallowed tools configuration.
-
- If allowed_tools is set, only tools in that list are returned.
- If disallowed_tools is set, tools in that list are excluded.
- Tool names are matched with and without server prefixes for flexibility.
-
- Args:
- tools: List of tools to filter
- mcp_server: Server configuration with allowed_tools/disallowed_tools
-
- Returns:
- Filtered list of tools
- """
- from litellm.proxy._experimental.mcp_server.utils import (
- server_applies_tool_allowlist,
- )
-
- tools_to_return = tools
-
- # Filter by allowed_tools (whitelist)
- if server_applies_tool_allowlist(mcp_server):
- if not mcp_server.allowed_tools:
- return []
- tools_to_return = [
- tool for tool in tools if _tool_name_matches(tool.name, mcp_server.allowed_tools, mcp_server)
- ]
-
- # Filter by disallowed_tools (blacklist)
- if mcp_server.disallowed_tools:
- tools_to_return = [
- tool
- for tool in tools_to_return
- if not _tool_name_matches(tool.name, mcp_server.disallowed_tools, mcp_server)
- ]
-
- return tools_to_return
-
- def apply_tool_overrides(
- tools: list[MCPTool],
- mcp_server: MCPServer,
- ) -> list[MCPTool]:
- """Apply admin-configured display name/description overrides to tools.
-
- Overrides are keyed by the unprefixed tool name, same convention as
- allowed_tools configuration.
- """
- display_name_map: Final = mcp_server.tool_name_to_display_name or {}
- description_map: Final = mcp_server.tool_name_to_description or {}
- if not display_name_map and not description_map:
- return tools
-
- for tool in tools:
- unprefixed = strip_known_server_prefix(tool.name, mcp_server)
- lookup_key = unprefixed or tool.name
- if lookup_key in display_name_map:
- tool.name = display_name_map[lookup_key]
- if lookup_key in description_map:
- tool.description = description_map[lookup_key]
- return tools
-
- def _get_client_ip_from_context() -> str | None:
- """
- Extract client_ip from auth context.
- Returns None if context not set (caller should handle this as "no IP filtering").
- """
- try:
- auth_user: Final = auth_context_var.get()
- if auth_user and isinstance(auth_user, MCPAuthenticatedUser):
- return auth_user.client_ip
- except Exception:
- pass
- return None
-
- async def _get_allowed_mcp_servers(
- user_api_key_auth: UserAPIKeyAuth | None,
- mcp_servers: Sequence[str] | None,
- client_ip: str | None = None,
- ) -> list[MCPServer]:
- """Return allowed MCP servers for a request after applying filters.
-
- Args:
- user_api_key_auth: The authenticated user's API key info.
- mcp_servers: Optional list of server names to filter to.
- client_ip: Client IP for IP-based access control. If None, falls back to
- auth context. Pass explicitly from request handlers for safety.
- Note: If client_ip is None and auth context is not set, IP filtering is skipped.
- This is intentional for internal callers but may indicate a bug if called
- from a request handler without proper context setup.
- """
- # Use explicit client_ip if provided, otherwise try auth context
- if client_ip is None:
- client_ip = _get_client_ip_from_context()
- if client_ip is None:
- verbose_logger.debug(
- "MCP _get_allowed_mcp_servers called without client_ip and no auth context. "
- "IP filtering will be skipped. This is expected for internal calls."
- )
-
- allowed_mcp_server_ids = await global_mcp_server_manager.get_allowed_mcp_servers(user_api_key_auth)
- (
- allowed_mcp_server_ids,
- _ip_blocked,
- ) = global_mcp_server_manager.filter_server_ids_by_ip_with_info(allowed_mcp_server_ids, client_ip)
- verbose_logger.debug(
- "MCP IP filter: client_ip=%s, allowed_server_ids=%s",
- client_ip,
- allowed_mcp_server_ids,
- )
- if _ip_blocked > 0:
- verbose_logger.debug(
- "MCP IP filtering: %d server(s) are not accessible from client IP %s "
- "because they are restricted to internal networks. "
- "No tools from those servers will be returned. "
- "To expose a server externally, set 'available_on_public_internet: true' "
- "in its configuration.",
- _ip_blocked,
- client_ip,
- )
- allowed_mcp_servers: list[MCPServer] = []
- for allowed_mcp_server_id in allowed_mcp_server_ids:
- mcp_server = global_mcp_server_manager.get_mcp_server_by_id(allowed_mcp_server_id)
- if mcp_server is not None:
- # Apply the request-time oauth2_flow backstop for legacy null rows.
- mcp_server = MCPServerManager.resolve_oauth2_flow_for_request(mcp_server)
- allowed_mcp_servers.append(mcp_server)
-
- if mcp_servers is not None:
- allowed_mcp_servers = await _get_allowed_mcp_servers_from_mcp_server_names(
- mcp_servers=mcp_servers,
- allowed_mcp_servers=allowed_mcp_servers,
- )
-
- return allowed_mcp_servers
-
- def _client_has_per_server_auth_header(
- server: MCPServer,
- mcp_server_auth_headers: dict[str, dict[str, str]] | None,
- ) -> bool:
- """True if the request carries a per-server ``x-mcp-{alias}-authorization``
- header for this server. This is the multi-server binding: it names one
- upstream, so it is unambiguously the caller's upstream token regardless of
- auth mode (never the LiteLLM admission credential).
-
- Resolves through the same ``lookup_mcp_server_auth_in_headers`` egress uses, so
- the connect gate and egress agree on which per-server header names match: a
- dashboard client sends ``x-mcp-{sanitize_mcp_alias_for_header(alias)}-authorization``,
- and matching only the raw alias here would 401 a token egress would forward.
- """
- if not mcp_server_auth_headers:
- return False
- from litellm.proxy._experimental.mcp_server.utils import (
- lookup_mcp_server_auth_in_headers,
- )
-
- server_headers: Final = lookup_mcp_server_auth_in_headers(
- mcp_server_auth_headers,
- alias=server.alias,
- server_name=server.server_name,
- access_groups=server.access_groups,
- )
- if isinstance(server_headers, str):
- return bool(server_headers.strip())
- if isinstance(server_headers, dict):
- return any(isinstance(hk, str) and hk.lower() == "authorization" for hk in server_headers)
- return False
-
- def _client_has_passthrough_authorization(
- server: MCPServer,
- oauth2_headers: dict[str, str] | None,
- mcp_server_auth_headers: dict[str, dict[str, str]] | None,
- ) -> bool:
- """True if the incoming request already carries an ``Authorization``
- header the gateway will forward to this pass-through server.
-
- The client may supply the bearer as either the top-level
- ``Authorization`` header (surfaced via ``oauth2_headers``) or a
- per-server ``x-mcp-auth-`` style header (surfaced via
- ``mcp_server_auth_headers``). Either form skips the pre-emptive 401.
- """
- if oauth2_headers:
- for k in oauth2_headers:
- if k.lower() == "authorization":
- return True
- return _client_has_per_server_auth_header(server, mcp_server_auth_headers)
-
- async def _get_user_oauth_extra_headers_from_db(
- server: MCPServer,
- user_api_key_auth: UserAPIKeyAuth | None,
- prefetched_creds: 'Mapping[str, "OAuthCredentialPayload"] | None' = None,
- ) -> dict[str, str] | None:
- """Stored OAuth2 token for (user, server) as an ``Authorization: Bearer`` header, or None.
-
- Thin wrapper over ``resolve_user_oauth_access_token`` (Redis cache, else DB + refresh);
- ``prefetched_creds`` skips the per-server Redis/DB lookups for the batch path.
- """
- if server.auth_type != MCPAuth.oauth2 or user_api_key_auth is None:
- return None
- from litellm.proxy._experimental.mcp_server.db import ( # noqa: PLC0415
- resolve_user_oauth_access_token,
- )
-
- token: Final = await resolve_user_oauth_access_token(
- getattr(user_api_key_auth, "user_id", None), server, prefetched_creds
- )
- return {"Authorization": f"Bearer {token}"} if token else None
-
- async def _prefetch_oauth_creds_for_user(
- user_api_key_auth: UserAPIKeyAuth | None,
- ) -> dict[str, "OAuthCredentialPayload"]:
- """Fetch all OAuth2 credentials for the user in one DB query.
-
- Returns a dict keyed by server_id to avoid N+1 queries in asyncio.gather loops.
- """
- user_id: Final[str | None] = getattr(user_api_key_auth, "user_id", None) if user_api_key_auth else None
- if not user_id:
- return {}
- try:
- from litellm.proxy._experimental.mcp_server.db import ( # noqa: PLC0415
- list_user_oauth_credentials,
- )
- from litellm.proxy.utils import get_prisma_client_or_throw # noqa: PLC0415
-
- prisma_client: Final = get_prisma_client_or_throw(
- "Database not connected. Connect a database to use OAuth2 MCP tools."
- )
- creds: Final = await list_user_oauth_credentials(prisma_client, user_id)
- return {c["server_id"]: c for c in creds if "server_id" in c}
- except Exception as e:
- verbose_logger.warning("_prefetch_oauth_creds_for_user: failed to prefetch for user=%s: %s", user_id, e)
- return {}
-
- def _prepare_mcp_server_headers(
- server: MCPServer,
- mcp_server_auth_headers: dict[str, dict[str, str]] | None,
- mcp_auth_header: str | None,
- oauth2_headers: dict[str, str] | None,
- raw_headers: dict[str, str] | None,
- user_api_key_auth: UserAPIKeyAuth | None = None,
- scope_servers: list[MCPServer] | None = None,
- ) -> tuple[dict[str, str] | str | None, dict[str, str] | None]:
- """Build auth and extra headers for a server.
-
- ``scope_servers`` is the full server list a fan-out handler iterates. Passing it lets the
- client-forwarded token modes withhold the caller's request-wide ``Authorization`` when
- another server in the scope would also receive it (``_caller_authorization_fans_out``);
- explicitly-addressed operations leave it None. Per-server ``x-mcp-{alias}-authorization``
- headers are unaffected — they bind one token to one server and are the multi-server shape.
- """
- server_auth_header: dict[str, str] | str | None = None
- if mcp_server_auth_headers:
- from litellm.proxy._experimental.mcp_server.utils import (
- lookup_mcp_server_auth_in_headers,
- )
-
- server_auth_header = lookup_mcp_server_auth_in_headers(
- mcp_server_auth_headers,
- alias=server.alias,
- server_name=server.server_name,
- access_groups=server.access_groups,
- )
-
- extra_headers: dict[str, str] | None = None
- is_client_forwarded_mode: Final = server.is_client_forwarded_token
- # In a multi-server listing scope the request-wide Authorization can only carry one token,
- # so it is withheld from a client-forwarded server when another server in scope also consumes
- # it (RFC 9700 cross-resource replay); such scopes must bind per-server via
- # x-mcp-{alias}-authorization. The decision is computed once so BOTH the forwarding branch and
- # the extra_headers copy loop below honor it — otherwise a server that lists Authorization in
- # extra_headers would re-copy the withheld bearer from raw_headers and replay it anyway.
- withhold_forwarded_authorization: Final = is_client_forwarded_mode and _caller_authorization_fans_out(
- server, scope_servers
- )
- if server.auth_type == MCPAuth.oauth2:
- # For OAuth2 M2M servers, upstream Authorization must come from
- # client_credentials token fetch, never from caller headers.
- if server.has_client_credentials:
- extra_headers = None
- else:
- # Copy to avoid mutating the original dict (important for parallel fetching)
- extra_headers = oauth2_headers.copy() if oauth2_headers else None
- # Migrated authorization_code: the v2 resolver injects the stored per-user
- # token, so drop the caller-forwarded Authorization (apply-if-absent would
- # otherwise let it shadow the resolved token). Delegate keeps it. Centralized
- # via _should_strip_caller_authorization to match _call_regular_mcp_tool.
- if extra_headers and _should_strip_caller_authorization(
- mcp_server=server,
- raw_headers=raw_headers,
- user_api_key_auth=user_api_key_auth,
- ):
- extra_headers = without_header(extra_headers, DEFAULT_CREDENTIAL_HEADER)
- elif is_client_forwarded_mode:
- if not withhold_forwarded_authorization:
- extra_headers = _client_forwarded_authorization_headers(
- mcp_server=server,
- oauth2_headers=oauth2_headers,
- raw_headers=raw_headers,
- user_api_key_auth=user_api_key_auth,
- )
-
- if server.extra_headers and raw_headers:
- if extra_headers is None:
- extra_headers = {}
-
- normalized_raw_headers: Final = {str(k).lower(): v for k, v in raw_headers.items() if isinstance(k, str)}
-
- # Centralized strip decision shared with
- # ``MCPServerManager._call_regular_mcp_tool`` so the two
- # code paths cannot drift on this security-sensitive choice.
- # See ``_should_strip_caller_authorization`` for the rules.
- strip_caller_authorization: Final = _should_strip_caller_authorization(
- mcp_server=server,
- raw_headers=raw_headers,
- user_api_key_auth=user_api_key_auth,
- )
-
- for header in server.extra_headers:
- if not isinstance(header, str):
- continue
- if header.lower() == "authorization" and (
- strip_caller_authorization or withhold_forwarded_authorization
- ):
- continue
- header_value = normalized_raw_headers.get(header.lower())
- if header_value is None:
- continue
- extra_headers[header] = header_value
-
- # Reset to None if no headers were actually added
- if extra_headers is not None and len(extra_headers) == 0:
- extra_headers = None
-
- if server_auth_header is None:
- server_auth_header = mcp_auth_header
-
- return server_auth_header, extra_headers
-
- def _merge_gateway_initialize_instructions(
- allowed_mcp_servers: list[MCPServer],
- ) -> str | None:
- """YAML/DB override, else upstream text (prefetch on init, or list_tools / health_check / call_tool cache)."""
- if not allowed_mcp_servers:
- return None
-
- texts: Final[list[tuple[str, str]]] = []
- for server in allowed_mcp_servers:
- label = server.alias or server.server_name or server.name or server.server_id or "mcp"
- if server.instructions and server.instructions.strip():
- texts.append((label, server.instructions.strip()))
- continue
- if server.spec_path:
- continue
- cached = global_mcp_server_manager._upstream_initialize_instructions_by_server_id.get(server.server_id)
- if cached and cached.strip():
- texts.append((label, cached.strip()))
-
- if not texts:
- return None
- if len(texts) == 1:
- return texts[0][1]
- return "\n\n---\n\n".join(f"[{lbl}]\n{txt}" for lbl, txt in texts)
-
- async def _raise_if_initialize_grants_no_mcp_servers(
- allowed: Sequence[MCPServer],
- user_api_key_auth: UserAPIKeyAuth | None,
- mcp_servers: Sequence[str] | None,
- client_ip: str | None,
- ) -> None:
- if allowed or user_api_key_auth is None or not user_api_key_auth.api_key:
- return
- if mcp_servers:
- await raise_denied_scoped_mcp_access(
- requested_names=mcp_servers,
- user_api_key_auth=user_api_key_auth,
- client_ip=client_ip,
- )
- no_servers_denial: Final[_McpDeniedDetail] = {
- "error": (
- "The key has no MCP servers granted, or none of its granted servers is loaded and allowed for "
- "this client IP. Grant servers or access groups to the key, its team, or its organization "
- "(object_permission.mcp_servers), check the server's allowed IPs, and reconnect."
- )
- }
- raise HTTPException(status_code=403, detail=no_servers_denial)
+ from litellm.proxy._experimental.mcp_server.operations import (
+ _client_has_passthrough_authorization,
+ _client_has_per_server_auth_header,
+ _get_allowed_mcp_servers,
+ _get_allowed_mcp_servers_from_mcp_server_names,
+ _get_user_oauth_extra_headers_from_db,
+ _http_detail_message,
+ _McpDeniedDetail,
+ _merge_gateway_initialize_instructions,
+ _prefetch_oauth_creds_for_user,
+ _prepare_mcp_server_headers,
+ _raise_if_initialize_grants_no_mcp_servers,
+ _server_answers_to,
+ _tool_name_matches,
+ apply_tool_overrides,
+ filter_tools_by_allowed_tools,
+ raise_denied_scoped_mcp_access,
+ )
@contextlib.asynccontextmanager
async def _gateway_initialize_instructions_request_scope(
@@ -2063,26 +981,28 @@ if MCP_AVAILABLE:
scoped_server_endpoint: bool = False,
is_initialize: bool = False,
) -> AsyncIterator[None]:
- allowed: Final = await _get_allowed_mcp_servers(
+ allowed: Final = await operations._get_allowed_mcp_servers(
user_api_key_auth=user_api_key_auth,
mcp_servers=mcp_servers,
client_ip=client_ip,
)
if is_initialize:
- await _raise_if_initialize_grants_no_mcp_servers(allowed, user_api_key_auth, mcp_servers, client_ip)
+ await operations._raise_if_initialize_grants_no_mcp_servers(
+ allowed, user_api_key_auth, mcp_servers, client_ip
+ )
if allowed:
# return_exceptions=True: a per-server probe failure (incl. CancelledError
# bubbled from anyio task group teardown on connection refused) must not
# cancel sibling probes or 500 the gateway initialize request.
await asyncio.gather(
*[
- global_mcp_server_manager._ensure_upstream_initialize_instructions_cached(s)
+ operations.global_mcp_server_manager._ensure_upstream_initialize_instructions_cached(s)
for s in allowed
if s is not None
],
return_exceptions=True,
)
- merged: Final = _merge_gateway_initialize_instructions(allowed_mcp_servers=allowed)
+ merged: Final = operations._merge_gateway_initialize_instructions(allowed_mcp_servers=allowed)
scoped_server_name = None
if scoped_server_endpoint and len(allowed) == 1:
scoped_server: Final = allowed[0]
@@ -2097,1599 +1017,34 @@ if MCP_AVAILABLE:
_mcp_gateway_initialize_instructions.reset(instructions_token)
_mcp_gateway_server_name.reset(server_name_token)
- def _aggregate_server_key(server: MCPServer) -> str:
- """The client-visible key for a server in listing outcomes and spend metadata: the same
- display prefix (alias, or the short prefix when that mode is enabled) the caller already
- sees on the tool names. Canonical internal server names never key a caller-readable
- surface; when the display naming deliberately hides them, the outcome keys must too."""
- return get_server_prefix(server) or "unknown"
-
- async def _get_tools_from_mcp_servers(
- user_api_key_auth: UserAPIKeyAuth | None,
- mcp_auth_header: str | None,
- mcp_servers: list[str] | None,
- mcp_server_auth_headers: dict[str, dict[str, str]] | None = None,
- oauth2_headers: dict[str, str] | None = None,
- raw_headers: dict[str, str] | None = None,
- log_list_tools_to_spendlogs: bool = False,
- list_tools_log_source: str | None = None,
- litellm_trace_id: str | None = None,
- request_tags: list[str] | None = None,
- client_ip: str | None = None,
- mcp_proxy_mode: bool = False,
- ) -> AggregateToolListing:
- """
- Helper method to fetch tools from MCP servers based on server filtering criteria.
-
- Args:
- user_api_key_auth: User authentication info for access control
- mcp_auth_header: Optional auth header for MCP server (deprecated)
- mcp_servers: Optional list of server names/aliases to filter by
- mcp_server_auth_headers: Optional dict of server-specific auth headers
- oauth2_headers: Optional dict of oauth2 headers
-
- Returns:
- AggregateToolListing: Combined tools from filtered servers plus each server's
- classified listing outcome
- """
- if not MCP_AVAILABLE:
- return AggregateToolListing(tools=[], outcomes={})
-
- list_tools_start_time: Final = datetime.now()
- litellm_logging_obj: LiteLLMLoggingObj | None = None
- list_tools_request_data: dict[str, object] = {}
-
- if log_list_tools_to_spendlogs:
- # This is intentionally minimal: only async_success_handler / post_call_failure_hook
- rules_obj: Final = Rules()
- list_tools_call_id: Final = str(uuid.uuid4())
- # Derive trace_id from raw_headers when not explicitly passed (same as A2A / MCP call_tool)
- effective_litellm_trace_id: Final = litellm_trace_id or get_chain_id_from_headers(raw_headers)
- spend_logs_metadata: Final[dict[str, object]] = {
- "mcp_operation": "list_tools",
- }
- if isinstance(list_tools_log_source, str):
- spend_logs_metadata["source"] = list_tools_log_source
- if isinstance(mcp_servers, list):
- spend_logs_metadata["requested_mcp_servers"] = mcp_servers
-
- list_tools_request_data = {
- "model": "MCP: list_tools",
- "call_type": CallTypes.list_mcp_tools.value,
- "litellm_call_id": list_tools_call_id,
- "litellm_trace_id": effective_litellm_trace_id,
- "metadata": {
- "spend_logs_metadata": spend_logs_metadata,
- "headers": logging_safe_mcp_headers(raw_headers),
- **({"tags": request_tags} if request_tags else {}),
- },
- # Provide a small input payload for standard logging
- "input": [
- {
- "role": "system",
- "content": {
- "mcp_operation": "list_tools",
- "requested_mcp_servers": mcp_servers,
- },
- }
- ],
- }
-
- # Attach user identifiers using the standard helper
- if user_api_key_auth is not None:
- LiteLLMProxyRequestSetup.add_user_api_key_auth_to_request_metadata(
- data=list_tools_request_data,
- user_api_key_dict=user_api_key_auth,
- _metadata_variable_name="metadata",
- )
-
- user_identifier: Final = getattr(user_api_key_auth, "end_user_id", None) or getattr(
- user_api_key_auth, "user_id", None
- )
- if user_identifier:
- list_tools_request_data["user"] = user_identifier
-
- try:
- litellm_logging_obj, _ = function_setup(
- original_function="list_mcp_tools",
- rules_obj=rules_obj,
- start_time=list_tools_start_time,
- **list_tools_request_data,
- )
- if litellm_logging_obj:
- litellm_logging_obj.call_type = CallTypes.list_mcp_tools.value
- litellm_logging_obj.model = "MCP: list_tools"
- except Exception as logging_error:
- verbose_logger.debug("Failed to initialize logging for MCP list_tools: %s", logging_error)
- litellm_logging_obj = None
-
- try:
- allowed_mcp_servers: Final = await _get_allowed_mcp_servers(
- user_api_key_auth=user_api_key_auth,
- mcp_servers=mcp_servers,
- client_ip=client_ip,
- )
- if mcp_servers and not allowed_mcp_servers:
- await raise_denied_scoped_mcp_access(
- requested_names=mcp_servers,
- user_api_key_auth=user_api_key_auth,
- client_ip=client_ip,
- )
-
- # Pre-fetch OAuth credentials only when at least one server uses OAuth2,
- # to avoid an unnecessary DB round-trip on requests with no OAuth2 MCP servers.
- _has_oauth2_server = any(getattr(s, "auth_type", None) == MCPAuth.oauth2 for s in allowed_mcp_servers)
- _prefetched_oauth_creds: Final = (
- await _prefetch_oauth_creds_for_user(user_api_key_auth) if _has_oauth2_server else {}
- )
-
- async def _fetch_and_filter_server_tools(
- server: MCPServer,
- ) -> "tuple[list[MCPTool], ServerOutcome]":
- """Fetch and filter tools from a single server, classifying any failure into that
- server's outcome so the aggregate can keep serving the healthy subset without a
- broken server masquerading as an empty one."""
- if server is None:
- return [], ServerListOk(tool_count=0)
-
- server_auth_header, extra_headers = _prepare_mcp_server_headers(
- server=server,
- mcp_server_auth_headers=mcp_server_auth_headers,
- mcp_auth_header=mcp_auth_header,
- oauth2_headers=oauth2_headers,
- raw_headers=raw_headers,
- user_api_key_auth=user_api_key_auth,
- scope_servers=allowed_mcp_servers,
- )
-
- # Prefer server-stored per-user OAuth when configured, so a stale
- # Authorization header from the MCP client cannot override Redis/DB
- # (same issue as call_tool in mcp_server_manager: VS Code caches tokens).
- from litellm.proxy._experimental.mcp_server.outbound_credentials.adapter import ( # noqa: PLC0415
- to_server_spec,
- )
-
- # A server migrated to the v2 resolver gets its token from the resolver at connect
- # time; building it here would double-resolve and be shadowed by the v2 graft. The
- # preemptive 401 already challenged a missing token, so one exists for the connect.
- migrated_to_v2: Final = to_server_spec(server) is not None
- if (
- not migrated_to_v2
- and server.auth_type == MCPAuth.oauth2
- and getattr(server, "needs_user_oauth_token", False)
- and user_api_key_auth is not None
- ):
- db_headers: Final = await _get_user_oauth_extra_headers_from_db(
- server,
- user_api_key_auth,
- prefetched_creds=_prefetched_oauth_creds,
- )
- if db_headers:
- extra_headers = db_headers
-
- # If still no OAuth2 token, fall back to pre-fetched creds (non-stale-client path)
- elif not migrated_to_v2 and extra_headers is None and server.auth_type == MCPAuth.oauth2:
- extra_headers = await _get_user_oauth_extra_headers_from_db(
- server,
- user_api_key_auth,
- prefetched_creds=_prefetched_oauth_creds,
- )
-
- if server.is_byok and server.auth_type != MCPAuth.oauth2 and server_auth_header is None:
- server_auth_header = await _get_byok_credential(server, user_api_key_auth)
-
- try:
- tools: Final = await global_mcp_server_manager._get_tools_from_server(
- server=server,
- mcp_auth_header=server_auth_header,
- extra_headers=extra_headers,
- add_prefix=True, # Always add server prefix
- raw_headers=raw_headers,
- user_api_key_auth=user_api_key_auth,
- oauth2_headers=oauth2_headers,
- )
- filtered_tools = filter_tools_by_allowed_tools(tools, server)
-
- filtered_tools = await filter_tools_by_key_team_permissions(
- tools=filtered_tools,
- server_id=server.server_id,
- user_api_key_auth=user_api_key_auth,
- )
-
- if mcp_proxy_mode:
- from litellm.proxy._experimental.mcp_server.tool_search import with_mcp_proxy_identity
-
- filtered_tools = [ # mutable-ok: MCP tool pipeline
- with_mcp_proxy_identity(tool, server.server_id) for tool in filtered_tools
- ]
- else:
- filtered_tools = apply_tool_overrides(filtered_tools, server)
-
- verbose_logger.debug(
- "Successfully fetched %s tools from server %s, %s after filtering",
- len(tools),
- server.name,
- len(filtered_tools),
- )
- return filtered_tools, ServerListOk(tool_count=len(filtered_tools))
- except MCPUpstreamAuthError as e:
- # Absorb so one unauthenticated server does not empty every other server's
- # tools. Surfacing the upstream 401 to the client as a re-auth challenge is
- # intentionally not done here: raising from this list handler cannot produce a
- # 401 + WWW-Authenticate (the MCP session manager serializes it as a JSON-RPC
- # error). Single-server routes surface it via the request-scope preemptive
- # check in _raise_preemptive_401_for_unauthenticated_servers instead.
- verbose_logger.debug("MCP list_tools: omitting %s; it needs upstream auth", server.name)
- return [], classify_list_exception(e)
- except Exception as e:
- verbose_logger.exception("Error getting tools from server %s: %s", server.name, e)
- return [], classify_list_exception(e)
-
- # Fetch tools from all servers in parallel
- tasks: Final = [_fetch_and_filter_server_tools(server) for server in allowed_mcp_servers]
- results: Final = await asyncio.gather(*tasks)
-
- # Flatten results into single list
- all_tools: Final[list[MCPTool]] = [tool for tools, _ in results for tool in tools]
- server_outcomes: Final[dict[str, ServerOutcome]] = {
- _aggregate_server_key(server): outcome
- for server, (_, outcome) in zip(allowed_mcp_servers, results)
- if server is not None
- }
-
- # If logging is enabled, enrich spend_logs_metadata with counts
- if litellm_logging_obj:
- per_server_tool_counts: Final[dict[str, int]] = {
- _aggregate_server_key(server): len(server_tools)
- for server, (server_tools, _) in zip(allowed_mcp_servers, results)
- if server is not None
- }
-
- metadata_dict: Final = litellm_logging_obj.model_call_details.get("metadata")
- if isinstance(metadata_dict, dict):
- spend_meta = metadata_dict.get("spend_logs_metadata")
- if not isinstance(spend_meta, dict):
- spend_meta = {}
- metadata_dict["spend_logs_metadata"] = spend_meta
- spend_meta["allowed_server_count"] = len(allowed_mcp_servers)
- spend_meta["tool_count_total"] = len(all_tools)
- spend_meta["per_server_tool_counts"] = per_server_tool_counts
- spend_meta["per_server_list_outcomes"] = {
- key: outcome_wire_value(outcome) for key, outcome in server_outcomes.items()
- }
-
- end_time: Final = datetime.now()
- try:
- await litellm_logging_obj.async_success_handler(
- result=[
- tool.model_dump(mode="json") if isinstance(tool, MCPTool) else tool for tool in all_tools
- ],
- start_time=list_tools_start_time,
- end_time=end_time,
- )
- except Exception as log_exc:
- # list_tools responses must not be dropped due to non-blocking
- # observability/serialization failures.
- verbose_logger.warning(
- "MCP list_tools success logging failed (continuing): %s",
- log_exc,
- )
-
- verbose_logger.info("Successfully fetched %s tools total from all MCP servers", len(all_tools))
-
- return AggregateToolListing(tools=all_tools, outcomes=server_outcomes)
- except Exception as e:
- # Only fire failure hook if logging was requested for this list-tools execution
- if log_list_tools_to_spendlogs and user_api_key_auth is not None:
- try:
- from litellm.proxy.proxy_server import proxy_logging_obj
-
- if proxy_logging_obj:
- traceback_str: Final = traceback.format_exc(limit=MAXIMUM_TRACEBACK_LINES_TO_LOG)
- await proxy_logging_obj.post_call_failure_hook(
- request_data=list_tools_request_data or {},
- original_exception=e,
- user_api_key_dict=user_api_key_auth,
- route="/mcp/list_tools",
- traceback_str=traceback_str,
- )
- except Exception:
- verbose_logger.debug("Failed to log MCP list_tools failure via post_call_failure_hook")
- raise
-
- async def _get_prompts_from_mcp_servers(
- user_api_key_auth: UserAPIKeyAuth | None,
- mcp_auth_header: str | None,
- mcp_servers: list[str] | None,
- mcp_server_auth_headers: dict[str, dict[str, str]] | None = None,
- oauth2_headers: dict[str, str] | None = None,
- raw_headers: dict[str, str] | None = None,
- ) -> list[Prompt]:
- """
- Helper method to fetch prompt from MCP servers based on server filtering criteria.
-
- Args:
- user_api_key_auth: User authentication info for access control
- mcp_auth_header: Optional auth header for MCP server (deprecated)
- mcp_servers: Optional list of server names/aliases to filter by
- mcp_server_auth_headers: Optional dict of server-specific auth headers
- oauth2_headers: Optional dict of oauth2 headers
-
- Returns:
- List[Prompt]: Combined list of prompts from filtered servers
- """
- if not MCP_AVAILABLE:
- return []
-
- allowed_mcp_servers: Final = await _get_allowed_mcp_servers(
- user_api_key_auth=user_api_key_auth,
- mcp_servers=mcp_servers,
- )
-
- # Get prompts from each allowed server
- all_prompts: Final = []
- for server in allowed_mcp_servers:
- if server is None:
- continue
-
- server_auth_header, extra_headers = _prepare_mcp_server_headers(
- server=server,
- mcp_server_auth_headers=mcp_server_auth_headers,
- mcp_auth_header=mcp_auth_header,
- oauth2_headers=oauth2_headers,
- raw_headers=raw_headers,
- user_api_key_auth=user_api_key_auth,
- scope_servers=allowed_mcp_servers,
- )
-
- try:
- prompts = await global_mcp_server_manager.get_prompts_from_server(
- server=server,
- user_api_key_auth=user_api_key_auth,
- mcp_auth_header=server_auth_header,
- extra_headers=extra_headers,
- add_prefix=True, # Always add server prefix
- raw_headers=raw_headers,
- )
-
- all_prompts.extend(prompts)
-
- verbose_logger.debug("Successfully fetched %s prompts from server %s", len(prompts), server.name)
- except Exception as e:
- verbose_logger.exception("Error getting prompts from server %s: %s", server.name, e)
- # Continue with other servers instead of failing completely
-
- verbose_logger.info("Successfully fetched %s prompts total from all MCP servers", len(all_prompts))
-
- return all_prompts
-
- async def _get_resources_from_mcp_servers(
- user_api_key_auth: UserAPIKeyAuth | None,
- mcp_auth_header: str | None,
- mcp_servers: list[str] | None,
- mcp_server_auth_headers: dict[str, dict[str, str]] | None = None,
- oauth2_headers: dict[str, str] | None = None,
- raw_headers: dict[str, str] | None = None,
- ) -> list[Resource]:
- """Fetch resources from allowed MCP servers."""
-
- if not MCP_AVAILABLE:
- return []
-
- allowed_mcp_servers: Final = await _get_allowed_mcp_servers(
- user_api_key_auth=user_api_key_auth,
- mcp_servers=mcp_servers,
- )
-
- all_resources: Final[list[Resource]] = []
- for server in allowed_mcp_servers:
- if server is None:
- continue
-
- server_auth_header, extra_headers = _prepare_mcp_server_headers(
- server=server,
- mcp_server_auth_headers=mcp_server_auth_headers,
- mcp_auth_header=mcp_auth_header,
- oauth2_headers=oauth2_headers,
- raw_headers=raw_headers,
- user_api_key_auth=user_api_key_auth,
- scope_servers=allowed_mcp_servers,
- )
-
- try:
- resources = await global_mcp_server_manager.get_resources_from_server(
- server=server,
- user_api_key_auth=user_api_key_auth,
- mcp_auth_header=server_auth_header,
- extra_headers=extra_headers,
- add_prefix=True, # Always add server prefix
- raw_headers=raw_headers,
- )
- all_resources.extend(resources)
-
- verbose_logger.debug("Successfully fetched %s resources from server %s", len(resources), server.name)
- except Exception as e:
- verbose_logger.exception("Error getting resources from server %s: %s", server.name, e)
-
- verbose_logger.info("Successfully fetched %s resources total from all MCP servers", len(all_resources))
-
- return all_resources
-
- async def _get_resource_templates_from_mcp_servers(
- user_api_key_auth: UserAPIKeyAuth | None,
- mcp_auth_header: str | None,
- mcp_servers: list[str] | None,
- mcp_server_auth_headers: dict[str, dict[str, str]] | None = None,
- oauth2_headers: dict[str, str] | None = None,
- raw_headers: dict[str, str] | None = None,
- ) -> list[ResourceTemplate]:
- """Fetch resource templates from allowed MCP servers."""
-
- if not MCP_AVAILABLE:
- return []
-
- allowed_mcp_servers: Final = await _get_allowed_mcp_servers(
- user_api_key_auth=user_api_key_auth,
- mcp_servers=mcp_servers,
- )
-
- all_resource_templates: Final[list[ResourceTemplate]] = []
- for server in allowed_mcp_servers:
- if server is None:
- continue
-
- server_auth_header, extra_headers = _prepare_mcp_server_headers(
- server=server,
- mcp_server_auth_headers=mcp_server_auth_headers,
- mcp_auth_header=mcp_auth_header,
- oauth2_headers=oauth2_headers,
- raw_headers=raw_headers,
- user_api_key_auth=user_api_key_auth,
- scope_servers=allowed_mcp_servers,
- )
-
- try:
- resource_templates = await global_mcp_server_manager.get_resource_templates_from_server(
- server=server,
- user_api_key_auth=user_api_key_auth,
- mcp_auth_header=server_auth_header,
- extra_headers=extra_headers,
- add_prefix=True, # Always add server prefix
- raw_headers=raw_headers,
- )
- all_resource_templates.extend(resource_templates)
- verbose_logger.debug(
- "Successfully fetched %s resource templates from server %s",
- len(resource_templates),
- server.name,
- )
- except Exception as e:
- verbose_logger.exception(
- "Error getting resource templates from server %s: %s",
- server.name,
- str(e),
- )
-
- verbose_logger.info(
- "Successfully fetched %s resource templates total from all MCP servers",
- len(all_resource_templates),
- )
-
- return all_resource_templates
-
- async def filter_tools_by_key_team_permissions(
- tools: list[MCPTool],
- server_id: str,
- user_api_key_auth: UserAPIKeyAuth | None,
- ) -> list[MCPTool]:
- """
- Filter tools based on key/team mcp_tool_permissions.
-
- Note: Tool names in the DB are stored without server prefixes,
- but tool names from MCP servers are prefixed. We need to strip
- the prefix before comparing.
- """
- # Filter by key/team tool-level permissions
- allowed_tool_names: Final = await MCPRequestHandler.get_allowed_tools_for_server(
- server_id=server_id,
- user_api_key_auth=user_api_key_auth,
- )
-
- # Tools arrive prefixed with the server's own prefix; strip exactly that
- # prefix (resolved from the server) rather than the first separator, so a
- # prefix containing the separator still reduces to the stored bare name.
- server: Final = global_mcp_server_manager.get_mcp_server_by_id(server_id)
- return [
- t
- for t in tools
- if MCPRequestHandler.tool_is_granted(strip_known_server_prefix(t.name, server), allowed_tool_names)
- ]
-
- async def _list_mcp_tools(
- user_api_key_auth: UserAPIKeyAuth | None = None,
- mcp_auth_header: str | None = None,
- mcp_servers: list[str] | None = None,
- mcp_server_auth_headers: dict[str, dict[str, str]] | None = None,
- oauth2_headers: dict[str, str] | None = None,
- raw_headers: dict[str, str] | None = None,
- log_list_tools_to_spendlogs: bool = False,
- list_tools_log_source: str | None = None,
- client_ip: str | None = None,
- mcp_proxy_mode: bool = False,
- ) -> AggregateToolListing:
- """
- List all available MCP tools.
-
- Args:
- user_api_key_auth: User authentication info for access control
- mcp_auth_header: Optional auth header for MCP server (deprecated)
- mcp_servers: Optional list of server names/aliases to filter by
- mcp_server_auth_headers: Optional dict of server-specific auth headers {server_alias: auth_value}
- client_ip: Client IP for IP-based server access control
-
- Returns:
- AggregateToolListing: Combined tools from all accessible servers plus each server's
- classified listing outcome
- """
- if not MCP_AVAILABLE:
- return AggregateToolListing(tools=[], outcomes={})
-
- try:
- listing: Final = await _get_tools_from_mcp_servers(
- user_api_key_auth=user_api_key_auth,
- mcp_auth_header=mcp_auth_header,
- mcp_servers=mcp_servers,
- mcp_server_auth_headers=mcp_server_auth_headers,
- oauth2_headers=oauth2_headers,
- raw_headers=raw_headers,
- log_list_tools_to_spendlogs=log_list_tools_to_spendlogs,
- list_tools_log_source=list_tools_log_source,
- client_ip=client_ip,
- mcp_proxy_mode=mcp_proxy_mode,
- )
- verbose_logger.debug("Successfully fetched %s tools from managed MCP servers", len(listing.tools))
- return listing
- except HTTPException:
- raise
- except Exception as e:
- verbose_logger.exception("Error getting tools from managed MCP servers: %s", e)
- # Continue with an empty listing instead of failing completely
- return AggregateToolListing(tools=[], outcomes={})
-
- async def _list_mcp_prompts(
- user_api_key_auth: UserAPIKeyAuth | None = None,
- mcp_auth_header: str | None = None,
- mcp_servers: list[str] | None = None,
- mcp_server_auth_headers: dict[str, dict[str, str]] | None = None,
- oauth2_headers: dict[str, str] | None = None,
- raw_headers: dict[str, str] | None = None,
- ) -> list[Prompt]:
- """
- List all available MCP prompts.
-
- Args:
- user_api_key_auth: User authentication info for access control
- mcp_auth_header: Optional auth header for MCP server (deprecated)
- mcp_servers: Optional list of server names/aliases to filter by
- mcp_server_auth_headers: Optional dict of server-specific auth headers {server_alias: auth_value}
-
- Returns:
- List[Prompt]: Combined list of tools from all accessible servers
- """
- if not MCP_AVAILABLE:
- return []
- # Get tools from managed MCP servers with error handling
- managed_prompts = []
- try:
- managed_prompts = await _get_prompts_from_mcp_servers(
- user_api_key_auth=user_api_key_auth,
- mcp_auth_header=mcp_auth_header,
- mcp_servers=mcp_servers,
- mcp_server_auth_headers=mcp_server_auth_headers,
- oauth2_headers=oauth2_headers,
- raw_headers=raw_headers,
- )
- verbose_logger.debug("Successfully fetched %s prompts from managed MCP servers", len(managed_prompts))
- except Exception as e:
- verbose_logger.exception("Error getting tools from managed MCP servers: %s", e)
- # Continue with empty managed tools list instead of failing completely
-
- return managed_prompts
-
- async def _list_mcp_resources(
- user_api_key_auth: UserAPIKeyAuth | None = None,
- mcp_auth_header: str | None = None,
- mcp_servers: list[str] | None = None,
- mcp_server_auth_headers: dict[str, dict[str, str]] | None = None,
- oauth2_headers: dict[str, str] | None = None,
- raw_headers: dict[str, str] | None = None,
- ) -> list[Resource]:
- """List all available MCP resources."""
-
- if not MCP_AVAILABLE:
- return []
-
- managed_resources: list[Resource] = []
- try:
- managed_resources = await _get_resources_from_mcp_servers(
- user_api_key_auth=user_api_key_auth,
- mcp_auth_header=mcp_auth_header,
- mcp_servers=mcp_servers,
- mcp_server_auth_headers=mcp_server_auth_headers,
- oauth2_headers=oauth2_headers,
- raw_headers=raw_headers,
- )
- verbose_logger.debug("Successfully fetched %s resources from managed MCP servers", len(managed_resources))
- except Exception as e:
- verbose_logger.exception("Error getting resources from managed MCP servers: %s", e)
-
- return managed_resources
-
- async def _list_mcp_resource_templates(
- user_api_key_auth: UserAPIKeyAuth | None = None,
- mcp_auth_header: str | None = None,
- mcp_servers: list[str] | None = None,
- mcp_server_auth_headers: dict[str, dict[str, str]] | None = None,
- oauth2_headers: dict[str, str] | None = None,
- raw_headers: dict[str, str] | None = None,
- ) -> list[ResourceTemplate]:
- """List all available MCP resource templates."""
-
- if not MCP_AVAILABLE:
- return []
-
- managed_resource_templates: list[ResourceTemplate] = []
- try:
- managed_resource_templates = await _get_resource_templates_from_mcp_servers(
- user_api_key_auth=user_api_key_auth,
- mcp_auth_header=mcp_auth_header,
- mcp_servers=mcp_servers,
- mcp_server_auth_headers=mcp_server_auth_headers,
- oauth2_headers=oauth2_headers,
- raw_headers=raw_headers,
- )
- verbose_logger.debug(
- "Successfully fetched %s resource templates from managed MCP servers",
- len(managed_resource_templates),
- )
- except Exception as e:
- verbose_logger.exception(
- "Error getting resource templates from managed MCP servers: %s",
- str(e),
- )
-
- return managed_resource_templates
-
- def _resolve_display_name_to_original(
- name: str,
- allowed_mcp_servers: list[MCPServer],
- ) -> str:
- """Translate a display-name override back to the original prefixed tool name.
-
- When a client received a customised display name from tools/list (e.g.
- "Get Pet") it will call tools/call with that same string. We need to
- reverse-map it to the original prefixed name (e.g.
- "petstore_mcp-getPetById") before any routing or permission logic runs.
- """
- for server in allowed_mcp_servers:
- display_map = server.tool_name_to_display_name or {}
- for unprefixed_name, display_name in display_map.items():
- if display_name == name:
- return add_server_prefix_to_name(unprefixed_name, get_server_prefix(server))
- return name
-
- async def _get_byok_credential(
- mcp_server: MCPServer,
- user_api_key_auth: UserAPIKeyAuth | None,
- ) -> str | None:
- """Retrieve the stored BYOK credential for a user+server pair, served from the worker cache within its TTL."""
- if not mcp_server.is_byok:
- return None
- user_id: Final = (user_api_key_auth.user_id if user_api_key_auth else None) or ""
- if not user_id:
- return None
-
- cached: Final = get_cached_byok_credential(user_id, mcp_server.server_id)
- if cached is not None:
- return cached.credential
-
- from litellm.proxy._experimental.mcp_server.db import get_user_credential
- from litellm.proxy.proxy_server import prisma_client
-
- if prisma_client is None:
- return None
- credential: Final = await get_user_credential(
- prisma_client=prisma_client,
- user_id=user_id,
- server_id=mcp_server.server_id,
- )
- cache_byok_credential(user_id, mcp_server.server_id, credential)
- return credential
-
- async def _check_byok_credential(
- mcp_server: MCPServer,
- user_api_key_auth: UserAPIKeyAuth | None,
- ) -> None:
- """
- If the MCP server is BYOK-enabled, verify that the requesting user has a
- stored credential. When no credential is found, raise an HTTP 401 with a
- WWW-Authenticate header that points the MCP client to our OAuth metadata
- endpoint so it can drive the authorization flow.
- """
- if not mcp_server.is_byok:
- return
-
- user_id: Final = (user_api_key_auth.user_id if user_api_key_auth else None) or ""
- if not user_id:
- raise HTTPException(
- status_code=401,
- detail={
- "error": "byok_auth_required",
- "server_id": mcp_server.server_id,
- "server_name": mcp_server.server_name or mcp_server.name,
- "message": "User identity is required for BYOK servers",
- },
- headers={"WWW-Authenticate": get_byok_www_authenticate()},
- )
-
- cached: Final = get_cached_byok_credential(user_id, mcp_server.server_id)
- if cached is not None:
- if cached.credential is None:
- raise HTTPException(
- status_code=401,
- detail={
- "error": "byok_auth_required",
- "server_id": mcp_server.server_id,
- "server_name": mcp_server.server_name or mcp_server.name,
- "message": (
- "No stored credential found for this BYOK server. "
- "Complete the OAuth authorization flow to provide your API key."
- ),
- },
- headers={"WWW-Authenticate": get_byok_www_authenticate()},
- )
- return
-
- from litellm.proxy._experimental.mcp_server.db import get_user_credential
- from litellm.proxy.proxy_server import prisma_client
-
- if prisma_client is None:
- # Fail closed on DB unavailability: returning here previously
- # bypassed the ownership check and let any proxy-authenticated
- # caller invoke BYOK tools during outage windows.
- raise HTTPException(
- status_code=503,
- detail={
- "error": "byok_auth_unavailable",
- "server_id": mcp_server.server_id,
- "server_name": mcp_server.server_name or mcp_server.name,
- "message": "BYOK credential check requires a database connection.",
- },
- )
-
- credential: Final = await get_user_credential(
- prisma_client=prisma_client,
- user_id=user_id,
- server_id=mcp_server.server_id,
- )
- cache_byok_credential(user_id, mcp_server.server_id, credential)
- if credential is None:
- raise HTTPException(
- status_code=401,
- detail={
- "error": "byok_auth_required",
- "server_id": mcp_server.server_id,
- "server_name": mcp_server.server_name or mcp_server.name,
- "message": (
- "No stored credential found for this BYOK server. "
- "Complete the OAuth authorization flow to provide your API key."
- ),
- },
- headers={"WWW-Authenticate": get_byok_www_authenticate()},
- )
-
- async def _list_tools_before_first_call(
- server: MCPServer | None,
- tool_name: str,
- allowed_mcp_servers: list[MCPServer],
- user_api_key_auth: UserAPIKeyAuth | None,
- mcp_auth_header: str | None,
- mcp_server_auth_headers: dict[str, dict[str, str]] | None,
- oauth2_headers: dict[str, str] | None,
- raw_headers: dict[str, str] | None,
- ) -> None:
- """List ``server`` with the caller's own credentials when it does not yet expose ``tool_name`` here.
-
- The startup fill skips a server whose upstream wants the caller's token, and mcp 2 no
- longer lists before an uncached tools/call, so a worker that has not served tools/list
- for this caller would otherwise answer 404 for a tool the caller can see. Gating on the
- requested tool, not on any prior listing, keeps callers with different upstream catalogs
- from masking each other.
- """
- if server is None or global_mcp_server_manager.server_exposes_tool(server, tool_name):
- return
- if all(allowed.server_id != server.server_id for allowed in allowed_mcp_servers):
- return
- try:
- await _get_tools_from_mcp_servers(
- user_api_key_auth=user_api_key_auth,
- mcp_auth_header=mcp_auth_header,
- mcp_servers=[server.server_id],
- mcp_server_auth_headers=mcp_server_auth_headers,
- oauth2_headers=oauth2_headers,
- raw_headers=raw_headers,
- )
- except Exception as e: # noqa: BLE001 # best effort: resolution below answers as it did before
- verbose_logger.debug("MCP tools/call: listing %s before its first call failed: %s", server.name, e)
-
- async def execute_mcp_tool(
- name: str,
- arguments: dict[str, object],
- allowed_mcp_servers: list[MCPServer],
- start_time: datetime,
- user_api_key_auth: UserAPIKeyAuth | None = None,
- mcp_auth_header: str | None = None,
- mcp_server_auth_headers: dict[str, dict[str, str]] | None = None,
- oauth2_headers: dict[str, str] | None = None,
- raw_headers: dict[str, str] | None = None,
- host_progress_callback: Callable | None = None,
- guardrail_context: Mapping[str, object] | None = None,
- **kwargs: Any,
- ) -> CallToolResult:
- """
- Execute MCP tool.
-
- This function assumes permission checks have already been performed.
-
- Args:
- name: Tool name (may include server prefix)
- arguments: Tool arguments
- allowed_mcp_servers: Pre-validated list of servers the user can access
- start_time: Start time for logging
- user_api_key_auth: Optional user API key auth for logging
- mcp_auth_header: Optional MCP auth header
- mcp_server_auth_headers: Optional server-specific auth headers
- oauth2_headers: Optional OAuth2 headers
- raw_headers: Optional raw HTTP headers
- **kwargs: Additional arguments (e.g., litellm_logging_obj)
-
- Returns:
- CallToolResult: Tool execution result
- """
- # Track resolved MCP server for both permission checks and dispatch
- mcp_server: MCPServer | None = None
- requested_server_id: Final[str | None] = kwargs.get("requested_server_id")
-
- # If the client called with a display-name override (e.g. "Get Pet"),
- # translate it back to the original prefixed name before any routing.
- name = _resolve_display_name_to_original(name, allowed_mcp_servers)
-
- # Remove prefix from tool name for logging and processing
- original_tool_name, server_name = split_server_prefix_from_name(name)
-
- requested_server: MCPServer | None = None
- if requested_server_id:
- requested_server = next(
- (s for s in allowed_mcp_servers if s.server_id == requested_server_id),
- None,
- )
-
- name_is_prefixed = False
- if requested_server is not None and MCP_TOOL_PREFIX_SEPARATOR in name:
- all_registry_prefixes: Final[set[str]] = set()
- for registry_server in global_mcp_server_manager.get_registry().values():
- for known_prefix in iter_known_server_prefixes(registry_server):
- all_registry_prefixes.add(normalize_server_name(known_prefix))
- name_is_prefixed = is_tool_name_prefixed(name, known_server_prefixes=all_registry_prefixes)
-
- first_call_target: Final = (
- requested_server
- if requested_server is not None and not name_is_prefixed
- else global_mcp_server_manager.server_owning_tool_name_prefix(name)
- )
- first_call_tool_name: Final = (
- name
- if first_call_target is None or (requested_server is not None and not name_is_prefixed)
- else strip_known_server_prefix(name, first_call_target)
- )
- await _list_tools_before_first_call(
- server=first_call_target,
- tool_name=first_call_tool_name,
- allowed_mcp_servers=allowed_mcp_servers,
- user_api_key_auth=user_api_key_auth,
- mcp_auth_header=mcp_auth_header,
- mcp_server_auth_headers=mcp_server_auth_headers,
- oauth2_headers=oauth2_headers,
- raw_headers=raw_headers,
- )
-
- if requested_server is not None and not name_is_prefixed:
- # REST callers may pass server_id with the upstream tool name (no
- # LiteLLM prefix). The first segment is not a registered server
- # prefix, so the whole string is the upstream tool name and may
- # legitimately contain the separator (e.g. "text-to-speech").
- # server_id is authoritative for routing and auth.
- mcp_server = requested_server
- server_name = requested_server.name
- original_tool_name = name
- else:
- # Resolve from tool name (MCP JSON-RPC or prefixed REST tool names).
- mcp_server = global_mcp_server_manager._get_mcp_server_from_tool_name(name)
- if mcp_server is None and requested_server is not None:
- for known_prefix in iter_known_server_prefixes(requested_server):
- candidate = global_mcp_server_manager._get_mcp_server_from_tool_name(
- add_server_prefix_to_name(name, known_prefix)
- )
- if candidate is not None:
- mcp_server = candidate
- break
- if mcp_server is not None:
- server_name = mcp_server.name
- original_tool_name = strip_known_server_prefix(name, mcp_server)
-
- if requested_server is not None:
- if mcp_server is not None and mcp_server.server_id != requested_server.server_id:
- raise HTTPException(
- status_code=403,
- detail={
- "error": "tool_server_mismatch",
- "message": (
- f"Tool '{name}' belongs to MCP server "
- f"'{mcp_server.name}' but request specified "
- f"server_id for '{requested_server.name}'."
- ),
- },
- )
- if mcp_server is None:
- mcp_server = requested_server
- server_name = requested_server.name
- original_tool_name = strip_known_server_prefix(name, requested_server)
-
- # Only enforce server-level permissions when we can resolve a server
- if server_name:
- if not MCPRequestHandler.is_tool_allowed(
- allowed_mcp_servers=[server.name for server in allowed_mcp_servers],
- server_name=server_name,
- ):
- raise HTTPException(
- status_code=403,
- detail="User not allowed to call this tool.",
- )
-
- standard_logging_mcp_tool_call: Final[StandardLoggingMCPToolCall] = _get_standard_logging_mcp_tool_call(
- name=original_tool_name, # Use original name for logging
- arguments=arguments,
- server_name=server_name,
- session_id=_mcp_session_id_from_headers(raw_headers),
- )
- litellm_logging_obj: Final[LiteLLMLoggingObj | None] = kwargs.get("litellm_logging_obj", None)
- if litellm_logging_obj:
- litellm_logging_obj.model_call_details["mcp_tool_call_metadata"] = standard_logging_mcp_tool_call
- litellm_logging_obj.model = f"MCP: {name}"
- litellm_logging_obj.model_call_details["model"] = f"MCP: {name}"
- # Resolve the MCP server early so BYOK checks and credential injection
- # apply to ALL dispatch paths (local tool registry AND managed MCP server).
- if mcp_server is None:
- mcp_server = global_mcp_server_manager._get_mcp_server_from_tool_name(name)
-
- if mcp_server:
- standard_logging_mcp_tool_call["mcp_server_cost_info"] = (mcp_server.mcp_info or {}).get(
- "mcp_server_cost_info"
- )
- if litellm_logging_obj:
- litellm_logging_obj.model_call_details["mcp_tool_call_metadata"] = standard_logging_mcp_tool_call
-
- # BYOK: retrieve the stored per-user credential. A single DB call
- # both checks existence and fetches the value, avoiding a double query.
- if mcp_server.is_byok and not mcp_auth_header:
- byok_cred: Final = await _get_byok_credential(mcp_server, user_api_key_auth)
- if byok_cred is None:
- raise HTTPException(
- status_code=401,
- detail={
- "error": "byok_auth_required",
- "server_id": mcp_server.server_id,
- "server_name": mcp_server.server_name or mcp_server.name,
- "message": (
- "No stored credential found for this BYOK server. "
- "Complete the OAuth authorization flow to provide your API key."
- ),
- },
- headers={"WWW-Authenticate": get_byok_www_authenticate()},
- )
- mcp_auth_header = byok_cred
- elif mcp_server.is_byok:
- # External auth header supplied; still enforce user-identity check.
- await _check_byok_credential(mcp_server, user_api_key_auth)
-
- # Check if tool exists in local registry first (for OpenAPI-based tools)
- # These tools are registered with their prefixed names
- #########################################################
- local_tool: Final = global_mcp_tool_registry.get_tool(name)
- if local_tool:
- # OpenAPI-backed tools used to bypass `pre_call_tool_check` —
- # only the managed path ran allowed/banned-tool checks, key/team
- # tool permissions, and parameter validation. Run the same checks
- # before dispatching to the local registry. Refuse the call if
- # we cannot resolve a server: tools registered via
- # openapi_to_mcp_generator are always tied to a server, so a
- # missing mcp_server here means the tool->server mapping has
- # not finished initializing or the registry entry is orphaned.
- # Skipping the check would re-open the same authorization gap.
- if mcp_server is None:
- raise HTTPException(
- status_code=503,
- detail=(
- f"MCP server for tool '{name}' is not available; "
- "refusing to dispatch without authorization checks. "
- "Retry once the server is registered."
- ),
- )
-
- # `pre_call_tool_check` calls into `proxy_logging_obj` for the
- # pre-call guardrail hooks, so source it from the canonical
- # `proxy_server` module the same way `_handle_managed_mcp_tool`
- # does. `kwargs.get("proxy_logging_obj")` is None on the MCP
- # entry path and would crash with AttributeError after the
- # security checks pass.
- from litellm.proxy.proxy_server import proxy_logging_obj
-
- hook_result = await global_mcp_server_manager.pre_call_tool_check(
- name=original_tool_name,
- arguments=arguments or {},
- server_name=server_name or mcp_server.name,
- user_api_key_auth=user_api_key_auth,
- proxy_logging_obj=proxy_logging_obj,
- server=mcp_server,
- raw_headers=raw_headers,
- litellm_logging_obj=litellm_logging_obj,
- guardrail_context=guardrail_context,
- )
- # `pre_call_tool_check` may return guardrail-modified
- # arguments; honor them on the local path too.
- if isinstance(hook_result, dict) and "arguments" in hook_result:
- arguments = hook_result["arguments"]
-
- verbose_logger.debug("Executing local registry tool: %s", name)
- # The credential rides ContextVars because the tool function has its
- # headers baked into the closure at registration time.
- auth_header_value, openapi_forwarded_headers, upstream_credential = _resolve_openapi_tool_auth(
- mcp_server=mcp_server,
- mcp_auth_header=mcp_auth_header,
- mcp_server_auth_headers=mcp_server_auth_headers,
- raw_headers=raw_headers,
- user_api_key_auth=user_api_key_auth,
- )
- (
- resolved_auth_headers,
- forwarded_headers,
- ) = await global_mcp_server_manager.resolve_openapi_upstream_auth(
- mcp_server=mcp_server,
- oauth2_headers=oauth2_headers,
- raw_headers=raw_headers,
- mcp_auth_header=upstream_credential,
- user_api_key_auth=user_api_key_auth,
- forwarded_headers=openapi_forwarded_headers,
- )
-
- _auth_token: Final = _request_auth_header.set(auth_header_value)
- _extra_token: Final = _request_extra_headers.set(forwarded_headers)
- _resolved_token: Final = _request_resolved_auth_headers.set(resolved_auth_headers)
- try:
- response = await _handle_local_mcp_tool(name, arguments)
- finally:
- _request_auth_header.reset(_auth_token)
- _request_extra_headers.reset(_extra_token)
- _request_resolved_auth_headers.reset(_resolved_token)
-
- # Try managed MCP server tool (the name is bare; the prefix boundary was
- # already resolved above against this server's registered prefixes)
- # Primary and recommended way to use external MCP servers
- #########################################################
- elif mcp_server:
- response = await _handle_managed_mcp_tool(
- server_name=server_name,
- name=original_tool_name,
- arguments=arguments,
- user_api_key_auth=user_api_key_auth,
- mcp_auth_header=mcp_auth_header,
- mcp_server_auth_headers=mcp_server_auth_headers,
- oauth2_headers=oauth2_headers,
- raw_headers=raw_headers,
- litellm_logging_obj=litellm_logging_obj,
- guardrail_context=guardrail_context,
- host_progress_callback=host_progress_callback,
- )
-
- # Fall back to local tool registry with original name (legacy support)
- #########################################################
- # Deprecated: Local MCP Server Tool
- #########################################################
- else:
- # Gate only what can actually dispatch. When the unprefixed name is
- # not in the registry either, `_handle_local_mcp_tool` below reports
- # 404 and nothing runs, so demanding a server here would turn every
- # unknown tool name into a misleading 503.
- if global_mcp_tool_registry.get_tool(original_tool_name) is not None:
- # `mcp_server` is None here because the tool name is not in the
- # tool -> server mapping, but the name still carries a prefix
- # that the server-level check above compared against the
- # caller's `allowed_mcp_servers` by exact `name`. So the named
- # server is in that list and can carry the tool-level checks,
- # even with the mapping cold. Resolve it from
- # `allowed_mcp_servers` rather than the registry: the registry
- # would happily return a server the caller holds no grant for,
- # and matching anything other than `name` would accept a server
- # the check never validated.
- prefix_server: Final = next(
- (candidate for candidate in allowed_mcp_servers if candidate.name == server_name),
- None,
- )
- if prefix_server is None:
- # A non-empty prefix that passed the server-level check
- # always matches here, so this arm only fires when the
- # prefix was empty, which is exactly the case that check
- # skips. Fail closed rather than dispatch with no server to
- # evaluate a tool ceiling against.
- raise HTTPException(
- status_code=503,
- detail=(
- f"MCP server for tool '{original_tool_name}' is not available; "
- "refusing to dispatch without authorization checks. "
- "Retry once the server is registered."
- ),
- )
-
- from litellm.proxy.proxy_server import proxy_logging_obj
-
- hook_result = await global_mcp_server_manager.pre_call_tool_check(
- name=original_tool_name,
- arguments=arguments,
- server_name=server_name,
- user_api_key_auth=user_api_key_auth,
- proxy_logging_obj=proxy_logging_obj,
- server=prefix_server,
- raw_headers=raw_headers,
- litellm_logging_obj=litellm_logging_obj,
- guardrail_context=guardrail_context,
- )
- if "arguments" in hook_result:
- arguments = hook_result["arguments"] # pyright: ignore[reportAny] # hook returns untyped args
-
- response = await _handle_local_mcp_tool(original_tool_name, arguments)
-
- return await _run_post_mcp_call_guardrails(
- result=response,
- litellm_logging_obj=litellm_logging_obj,
- user_api_key_auth=user_api_key_auth,
- request_data=kwargs,
- )
-
- async def _run_post_mcp_call_guardrails(
- result: CallToolResult,
- litellm_logging_obj: LiteLLMLoggingObj | None,
- user_api_key_auth: UserAPIKeyAuth | None,
- request_data: Mapping[str, object],
- ) -> CallToolResult:
- """Run ``post_mcp_call`` guardrails over an executed tool result.
-
- Lives on ``execute_mcp_tool``'s return path rather than inside
- ``_fire_mcp_tool_call_logging`` so enforcement never depends on logging
- being configured, and so every dispatch route gets it: the MCP protocol
- handler, the REST endpoint, and tool search all funnel through here.
- A guardrail that rejects the result raises, matching ``pre_mcp_call``.
- """
- from litellm.proxy.proxy_server import proxy_logging_obj
-
- if proxy_logging_obj is None:
- return result
- return await proxy_logging_obj.post_mcp_call_hook(
- response=result,
- request_data=(
- litellm_logging_obj.model_call_details if litellm_logging_obj is not None else dict(request_data)
- ),
- user_api_key_dict=user_api_key_auth,
- )
-
- _MCP_CREDENTIAL_REQUEST_FIELDS: Final = frozenset(
- {
- "raw_headers",
- "mcp_auth_header",
- "mcp_server_auth_headers",
- "oauth2_headers",
- "user_api_key_auth",
- }
+ from litellm.proxy._experimental.mcp_server.operations import (
+ _MCP_CREDENTIAL_REQUEST_FIELDS,
+ _aggregate_server_key,
+ _check_byok_credential,
+ _fire_mcp_tool_call_logging,
+ _get_byok_credential,
+ _get_prompts_from_mcp_servers,
+ _get_resource_templates_from_mcp_servers,
+ _get_resources_from_mcp_servers,
+ _get_standard_logging_mcp_tool_call,
+ _get_tools_from_mcp_servers,
+ _handle_local_mcp_tool,
+ _handle_managed_mcp_tool,
+ _list_mcp_prompts,
+ _list_mcp_resource_templates,
+ _list_mcp_resources,
+ _list_mcp_tools,
+ _list_tools_before_first_call,
+ _resolve_display_name_to_original,
+ _run_post_mcp_call_guardrails,
+ call_mcp_tool,
+ execute_mcp_tool,
+ filter_tools_by_key_team_permissions,
+ fire_mcp_tool_call_failure_logging,
+ mcp_get_prompt,
+ mcp_read_resource,
)
- async def _fire_mcp_tool_call_logging(
- logging_obj: LiteLLMLoggingObj,
- result: CallToolResult,
- start_time: datetime,
- end_time: datetime,
- user_api_key_auth: UserAPIKeyAuth | None = None,
- request_data: Mapping[str, object] | None = None,
- ) -> CallToolResult:
- """Fire post-call logging for an executed MCP tool call, returning the result to send.
-
- The returned result is what the caller must forward to the client: a
- ``post_mcp_call`` guardrail may rewrite the tool output (e.g. mask
- sensitive values) or reject it, in which case its exception propagates.
- Guardrails run before the success/failure logging so the masked text, not
- the raw one, is what gets logged.
-
- A result with ``is_error=True`` is logged as a failure (``status="failure"``
- payload, so OTel marks the span ERROR) while the HTTP wire behavior stays
- 200 + ``isError: true`` per the MCP spec. The error check runs after
- ``async_post_mcp_tool_call_hook`` because guardrails may flip the result
- to ``is_error=True`` in that hook. Raised exceptions never reach here (the
- ``@client`` wrapper and ``call_mcp_tool``'s except path log those), so
- this cannot double-log a failure.
-
- ``request_data`` may carry credential-bearing fields (the REST path puts
- ``raw_headers``, ``mcp_auth_header``, ``mcp_server_auth_headers``, and
- ``oauth2_headers`` at the top level of its data dict), so those are
- stripped before the dict is handed to ``post_call_failure_hook``
- callbacks.
- """
- from litellm.proxy.proxy_server import proxy_logging_obj
-
- logging_obj.post_call(original_response=result)
- await logging_obj.async_post_mcp_tool_call_hook(
- kwargs=logging_obj.model_call_details,
- response_obj=result,
- start_time=start_time,
- end_time=end_time,
- )
- logging_obj.call_type = CallTypes.call_mcp_tool.value
- error_message: Final = extract_mcp_tool_result_error_message(result)
- if error_message is None:
- await logging_obj.async_success_handler(result=result, start_time=start_time, end_time=end_time)
- return result
-
- logging_obj.has_run_logging(event_type="sync_success")
- logging_obj.has_run_logging(event_type="async_success")
- tool_error: Final = MCPToolResultError(error_message)
- logging_obj.failure_handler(tool_error, "", start_time, end_time)
- await logging_obj.async_failure_handler(tool_error, "", start_time, end_time)
-
- if user_api_key_auth is None:
- return result
-
- if proxy_logging_obj:
- sanitized_request_data: Final = {
- key: value for key, value in (request_data or {}).items() if key not in _MCP_CREDENTIAL_REQUEST_FIELDS
- }
- await proxy_logging_obj.post_call_failure_hook(
- request_data=sanitized_request_data,
- original_exception=tool_error,
- user_api_key_dict=user_api_key_auth,
- route="/mcp/call_tool",
- )
- return result
-
- async def fire_mcp_tool_call_failure_logging(
- logging_obj: LiteLLMLoggingObj | None,
- exception: Exception,
- start_time: datetime,
- user_api_key_auth: UserAPIKeyAuth | None,
- request_data: Mapping[str, object],
- ) -> None:
- """Failure logging shared by the ``/mcp`` path and the REST endpoint. Call from
- inside the ``except`` block so the traceback is still available.
-
- The failure handlers run first because ``_ProxyDBLogger.async_post_call_failure_hook``
- builds the failure spend-log row from the ``standard_logging_object`` they produce;
- both gate on ``should_run_logging``, so the ``@client`` wrapper does not log twice.
- A relayed upstream 401 (``MCPUpstreamAuthError``) is an expected caller-must-reauth
- signal and skips ``post_call_failure_hook``, which fires the ``llm_exceptions`` alert.
- """
- from litellm.proxy.proxy_server import proxy_logging_obj
-
- traceback_str: Final = traceback.format_exc(limit=MAXIMUM_TRACEBACK_LINES_TO_LOG)
- if logging_obj is not None:
- end_time: Final = datetime.now() # noqa: DTZ005 # naive to match `start_time`, which it is subtracted from
- logging_obj.failure_handler(exception, traceback_str, start_time, end_time)
- await logging_obj.async_failure_handler(exception, traceback_str, start_time, end_time)
-
- if isinstance(exception, MCPUpstreamAuthError) or not proxy_logging_obj or user_api_key_auth is None:
- return
- sanitized_request_data: Final = {
- key: value for key, value in request_data.items() if key not in _MCP_CREDENTIAL_REQUEST_FIELDS
- }
- await proxy_logging_obj.post_call_failure_hook(
- request_data=sanitized_request_data,
- original_exception=exception,
- user_api_key_dict=user_api_key_auth,
- route="/mcp/call_tool",
- traceback_str=traceback_str,
- )
-
- @client
- async def call_mcp_tool(
- name: str,
- arguments: dict[str, object] | None = None,
- user_api_key_auth: UserAPIKeyAuth | None = None,
- mcp_auth_header: str | None = None,
- mcp_servers: list[str] | None = None,
- mcp_server_auth_headers: dict[str, dict[str, str]] | None = None,
- oauth2_headers: dict[str, str] | None = None,
- raw_headers: dict[str, str] | None = None,
- client_ip: str | None = None,
- **kwargs: Any,
- ) -> CallToolResult:
- """
- Call a specific tool with the provided arguments (handles prefixed tool names).
- """
- start_time: Final = datetime.now()
- litellm_logging_obj: Final[LiteLLMLoggingObj | None] = kwargs.get("litellm_logging_obj", None)
-
- try:
- if arguments is None:
- raise HTTPException(status_code=400, detail="Request arguments are required")
-
- ## CHECK IF USER IS ALLOWED TO CALL THIS TOOL
- allowed_mcp_server_ids: Final = await global_mcp_server_manager.get_allowed_mcp_servers(
- user_api_key_auth=user_api_key_auth,
- )
-
- allowed_mcp_servers: list[MCPServer] = []
- for allowed_mcp_server_id in allowed_mcp_server_ids:
- allowed_server = global_mcp_server_manager.get_mcp_server_by_id(allowed_mcp_server_id)
- if allowed_server is not None:
- # Same request-time oauth2_flow backstop the listing path applies,
- # so a null-flow M2M-shape row is treated as M2M on tool calls too.
- allowed_server = MCPServerManager.resolve_oauth2_flow_for_request(allowed_server)
- allowed_mcp_servers.append(allowed_server)
-
- allowed_mcp_servers = await _get_allowed_mcp_servers_from_mcp_server_names(
- mcp_servers=mcp_servers,
- allowed_mcp_servers=allowed_mcp_servers,
- )
- if mcp_servers and not allowed_mcp_servers:
- await raise_denied_scoped_mcp_access(
- requested_names=mcp_servers,
- user_api_key_auth=user_api_key_auth,
- client_ip=client_ip,
- )
- if not allowed_mcp_servers:
- raise HTTPException(
- status_code=403,
- detail="User not allowed to call this tool.",
- )
-
- # Delegate to execute_mcp_tool for execution
- response = await execute_mcp_tool(
- name=name,
- arguments=arguments,
- allowed_mcp_servers=allowed_mcp_servers,
- start_time=start_time,
- user_api_key_auth=user_api_key_auth,
- mcp_auth_header=mcp_auth_header,
- mcp_server_auth_headers=mcp_server_auth_headers,
- oauth2_headers=oauth2_headers,
- raw_headers=raw_headers,
- **kwargs,
- )
- except Exception as e:
- await fire_mcp_tool_call_failure_logging(litellm_logging_obj, e, start_time, user_api_key_auth, kwargs)
- raise
-
- if litellm_logging_obj:
- response = await _fire_mcp_tool_call_logging(
- logging_obj=litellm_logging_obj,
- result=response,
- start_time=start_time,
- end_time=datetime.now(),
- user_api_key_auth=user_api_key_auth,
- request_data=kwargs,
- )
- return response
-
- async def mcp_get_prompt(
- name: str,
- arguments: dict[str, object] | None = None,
- user_api_key_auth: UserAPIKeyAuth | None = None,
- mcp_auth_header: str | None = None,
- mcp_servers: list[str] | None = None,
- mcp_server_auth_headers: dict[str, dict[str, str]] | None = None,
- oauth2_headers: dict[str, str] | None = None,
- raw_headers: dict[str, str] | None = None,
- ) -> GetPromptResult:
- """
- Fetch a specific MCP prompt, handling both prefixed and unprefixed names.
- """
- allowed_mcp_servers: Final = await _get_allowed_mcp_servers(
- user_api_key_auth=user_api_key_auth,
- mcp_servers=mcp_servers,
- )
-
- if not allowed_mcp_servers:
- raise HTTPException(
- status_code=403,
- detail="User not allowed to get this prompt.",
- )
-
- # Extract server name from prefixed prompt name
- original_prompt_name, server_name = split_server_prefix_from_name(name)
-
- server: Final = next((s for s in allowed_mcp_servers if s.name == server_name), None)
- if server is None:
- raise HTTPException(
- status_code=403,
- detail="User not allowed to get this prompt.",
- )
-
- server_auth_header, extra_headers = _prepare_mcp_server_headers(
- server=server,
- mcp_server_auth_headers=mcp_server_auth_headers,
- mcp_auth_header=mcp_auth_header,
- oauth2_headers=oauth2_headers,
- raw_headers=raw_headers,
- user_api_key_auth=user_api_key_auth,
- )
-
- return await global_mcp_server_manager.get_prompt_from_server(
- server=server,
- user_api_key_auth=user_api_key_auth,
- prompt_name=original_prompt_name,
- arguments=arguments,
- mcp_auth_header=server_auth_header,
- extra_headers=extra_headers,
- raw_headers=raw_headers,
- )
-
- async def mcp_read_resource(
- url: AnyUrl,
- user_api_key_auth: UserAPIKeyAuth | None = None,
- mcp_auth_header: str | None = None,
- mcp_servers: list[str] | None = None,
- mcp_server_auth_headers: dict[str, dict[str, str]] | None = None,
- oauth2_headers: dict[str, str] | None = None,
- raw_headers: dict[str, str] | None = None,
- ) -> ReadResourceResult:
- """Read resource contents from upstream MCP servers."""
-
- allowed_mcp_servers: Final = await _get_allowed_mcp_servers(
- user_api_key_auth=user_api_key_auth,
- mcp_servers=mcp_servers,
- )
-
- if not allowed_mcp_servers:
- raise HTTPException(
- status_code=403,
- detail="User not allowed to read this resource.",
- )
-
- if len(allowed_mcp_servers) != 1:
- raise HTTPException(
- status_code=400,
- detail=(
- "Multiple MCP servers configured; read_resource currently supports exactly one allowed server."
- ),
- )
-
- server: Final = allowed_mcp_servers[0]
-
- server_auth_header, extra_headers = _prepare_mcp_server_headers(
- server=server,
- mcp_server_auth_headers=mcp_server_auth_headers,
- mcp_auth_header=mcp_auth_header,
- oauth2_headers=oauth2_headers,
- raw_headers=raw_headers,
- user_api_key_auth=user_api_key_auth,
- )
-
- return await global_mcp_server_manager.read_resource_from_server(
- server=server,
- user_api_key_auth=user_api_key_auth,
- url=url,
- mcp_auth_header=server_auth_header,
- extra_headers=extra_headers,
- raw_headers=raw_headers,
- )
-
- def _get_standard_logging_mcp_tool_call(
- name: str,
- arguments: dict[str, object],
- server_name: str | None,
- session_id: str | None = None,
- ) -> StandardLoggingMCPToolCall:
- mcp_server: Final = global_mcp_server_manager._get_mcp_server_from_tool_name(
- add_server_prefix_to_name(name, server_name) if server_name else name
- )
- namespaced_tool_name: Final = f"{server_name}/{name}" if server_name else name
- if mcp_server:
- mcp_info: Final = mcp_server.mcp_info or {}
- return StandardLoggingMCPToolCall(
- name=name,
- arguments=arguments,
- mcp_server_name=mcp_info.get("server_name"),
- mcp_server_logo_url=mcp_info.get("logo_url"),
- namespaced_tool_name=namespaced_tool_name,
- mcp_session_id=session_id,
- mcp_auth_mode=mcp_server.auth_type,
- mcp_server_resource=_redact_mcp_resource_url(mcp_server.url),
- )
- else:
- return StandardLoggingMCPToolCall(
- name=name,
- arguments=arguments,
- namespaced_tool_name=namespaced_tool_name,
- mcp_session_id=session_id,
- )
-
- async def _handle_managed_mcp_tool(
- server_name: str,
- name: str,
- arguments: dict[str, object],
- user_api_key_auth: UserAPIKeyAuth | None = None,
- mcp_auth_header: str | None = None,
- mcp_server_auth_headers: dict[str, dict[str, str]] | None = None,
- oauth2_headers: dict[str, str] | None = None,
- raw_headers: dict[str, str] | None = None,
- litellm_logging_obj: LiteLLMLoggingObj | None = None,
- host_progress_callback: Callable | None = None,
- guardrail_context: Mapping[str, object] | None = None,
- ) -> CallToolResult:
- """Handle tool execution for managed server tools"""
- # Import here to avoid circular import
- from litellm.proxy.proxy_server import proxy_logging_obj
-
- call_tool_result: Final = await global_mcp_server_manager.call_tool(
- server_name=server_name,
- name=name,
- arguments=arguments,
- user_api_key_auth=user_api_key_auth,
- mcp_auth_header=mcp_auth_header,
- mcp_server_auth_headers=mcp_server_auth_headers,
- oauth2_headers=oauth2_headers,
- raw_headers=raw_headers,
- proxy_logging_obj=proxy_logging_obj,
- host_progress_callback=host_progress_callback,
- litellm_logging_obj=litellm_logging_obj,
- guardrail_context=guardrail_context,
- )
- verbose_logger.debug("CALL TOOL RESULT: %s", call_tool_result)
- return call_tool_result
-
- async def _handle_local_mcp_tool(name: str, arguments: dict[str, object]) -> CallToolResult:
- """Execute a local-registry tool and report whether it succeeded.
-
- Returns the result rather than bare content because the verdict is part of it: the content
- alone cannot say whether the handler failed, so callers used to stamp is_error=False on every
- outcome and an upstream rejection was served as tool output.
-
- A failure is reported as ``is_error=True`` here rather than raised, because the REST surface
- turns an unrecognized exception into a 500 and an upstream 403 or 429 is not a gateway crash.
- ``MCPUpstreamAuthError`` is the exception: it propagates so the caller is told to
- re-authenticate, which both renderers already know how to say.
-
- Note: Local tools don't use prefixes, so we use the original name
- """
- import inspect
-
- tool: Final = global_mcp_tool_registry.get_tool(name)
- if not tool:
- raise HTTPException(status_code=404, detail=f"Tool '{name}' not found")
-
- try:
- if inspect.iscoroutinefunction(tool.handler):
- result = await tool.handler(**arguments)
- else:
- result = tool.handler(**arguments)
- except MCPUpstreamAuthError:
- raise
- except Exception as e:
- verbose_logger.exception("Error executing local tool %s: %s", name, e)
- return CallToolResult(
- content=[TextContent(text=f"Error: {e}", type="text")], # mutable-ok: MCP result content
- is_error=True,
- )
- return CallToolResult(
- content=[TextContent(text=str(result), type="text")], # mutable-ok: MCP result content
- is_error=False,
- )
-
def _get_mcp_servers_in_path(path: str) -> list[str] | None:
"""
Get the MCP servers from the path
@@ -4178,7 +1533,9 @@ if MCP_AVAILABLE:
detail=f"API key does not have access to toolset '{toolset_id}'.",
)
- tool_permissions = await global_mcp_server_manager.resolve_toolset_tool_permissions(toolset_ids=[toolset_id])
+ tool_permissions = await operations.global_mcp_server_manager.resolve_toolset_tool_permissions(
+ toolset_ids=[toolset_id]
+ )
server_ids: Final = list(tool_permissions.keys())
existing_op: Final = user_api_key_auth.object_permission
if existing_op is not None:
@@ -4197,7 +1554,7 @@ if MCP_AVAILABLE:
mcp_servers=server_ids,
mcp_tool_permissions=tool_permissions,
)
- return user_api_key_auth.model_copy(update={"object_permission": updated_op})
+ return user_api_key_auth.model_copy(update={"object_permission": updated_op, "mcp_toolset_id": toolset_id})
async def _raise_preemptive_401_for_unauthenticated_servers(
scope: Scope,
@@ -4221,7 +1578,7 @@ if MCP_AVAILABLE:
a server it will be 403'd on immediately after authentication.
"""
for server_name in mcp_servers or []:
- server = global_mcp_server_manager.get_mcp_server_by_name(server_name, client_ip=client_ip)
+ server = operations.global_mcp_server_manager.get_mcp_server_by_name(server_name, client_ip=client_ip)
if server is not None and allowed_server_ids is not None and server.server_id not in allowed_server_ids:
# Caller's narrowed scope excludes this server — skip the
# preemptive challenge and let downstream authorization
@@ -4234,7 +1591,7 @@ if MCP_AVAILABLE:
# authorization_url/token_url can change their inferred flow.
continue
if server is not None:
- server = await global_mcp_server_manager.ensure_oauth_metadata_discovered(server)
+ server = await operations.global_mcp_server_manager.ensure_oauth_metadata_discovered(server)
if server and server.auth_type == MCPAuth.oauth2:
# The challenge decision is per oauth2 sub-mode, not per header:
# gateway-managed modes (M2M and interactive authorization_code)
@@ -4262,7 +1619,7 @@ if MCP_AVAILABLE:
# authorization server is the gateway itself, vaulting via the
# authorize interlude); the per-server relay advertised below
# cannot vault without a litellm key on its token request.
- if await global_mcp_server_manager.has_user_oauth_token(server, user_api_key_auth):
+ if await operations.global_mcp_server_manager.has_user_oauth_token(server, user_api_key_auth):
continue
if _is_mcp_admitted_user_subject(user_api_key_auth):
@@ -4345,12 +1702,12 @@ if MCP_AVAILABLE:
and server.server_id
in frozenset(
allowed.server_id
- for allowed in await _get_allowed_mcp_servers(
+ for allowed in await operations._get_allowed_mcp_servers(
user_api_key_auth=user_api_key_auth, mcp_servers=mcp_servers, client_ip=client_ip
)
)
):
- await global_mcp_server_manager.preflight_token_exchange(
+ await operations.global_mcp_server_manager.preflight_token_exchange(
server=server,
oauth2_headers=oauth2_headers,
user_api_key_auth=user_api_key_auth,
@@ -4366,7 +1723,9 @@ if MCP_AVAILABLE:
if (
server
and server.is_oauth_passthrough
- and not _client_has_passthrough_authorization(server, oauth2_headers, mcp_server_auth_headers)
+ and not operations._client_has_passthrough_authorization(
+ server, oauth2_headers, mcp_server_auth_headers
+ )
):
www_authenticate = get_passthrough_www_authenticate(
scope=scope,
@@ -4383,7 +1742,7 @@ if MCP_AVAILABLE:
and server.is_oauth_delegate
and len(mcp_servers or []) == 1
and _get_forwarded_auth_from_scope(scope) is None
- and not _client_has_per_server_auth_header(server, mcp_server_auth_headers)
+ and not operations._client_has_per_server_auth_header(server, mcp_server_auth_headers)
):
www_authenticate = get_passthrough_www_authenticate(
scope=scope,
@@ -4400,7 +1759,7 @@ if MCP_AVAILABLE:
and server.is_true_passthrough
and len(mcp_servers or []) == 1
and not _scope_has_authorization_header(scope)
- and not _client_has_per_server_auth_header(server, mcp_server_auth_headers)
+ and not operations._client_has_per_server_auth_header(server, mcp_server_auth_headers)
):
if server.is_dcr_bridge:
raise HTTPException(
@@ -4528,7 +1887,7 @@ if MCP_AVAILABLE:
# Use the authorized server set, not the raw user-supplied names, so that
# a caller cannot force a probe to a server their key is not allowed to use.
- allowed_servers: Final = await _get_allowed_mcp_servers(
+ allowed_servers: Final = await operations._get_allowed_mcp_servers(
user_api_key_auth=user_api_key_auth,
mcp_servers=mcp_servers,
client_ip=client_ip,
diff --git a/litellm/proxy/_experimental/mcp_server/tool_search.py b/litellm/proxy/_experimental/mcp_server/tool_search.py
index a482d02c31d..3650c722103 100644
--- a/litellm/proxy/_experimental/mcp_server/tool_search.py
+++ b/litellm/proxy/_experimental/mcp_server/tool_search.py
@@ -463,8 +463,8 @@ async def handle_mcp_tool_search(
oauth2_headers: dict[str, str] | None = None,
raw_headers: dict[str, str] | None = None,
) -> CallToolResult:
- from litellm.proxy._experimental.mcp_server.server import (
- _list_mcp_tools, # pyright: ignore[reportPrivateUsage] # shared catalog owner
+ from litellm.proxy._experimental.mcp_server.operations import (
+ _list_mcp_tools,
)
from litellm.proxy.proxy_server import llm_router, proxy_logging_obj
@@ -519,8 +519,8 @@ async def handle_mcp_proxy_tool(
from jsonschema import validate
from litellm.proxy import proxy_server
- from litellm.proxy._experimental.mcp_server.server import ( # pyright: ignore[reportPrivateUsage] # shared catalog owner
- _list_mcp_tools, # pyright: ignore[reportPrivateUsage] # shared catalog owner
+ from litellm.proxy._experimental.mcp_server.operations import (
+ _list_mcp_tools,
)
listing: Final = await _list_mcp_tools(
@@ -607,7 +607,7 @@ async def handle_mcp_tool_call(
requested_server_id: str | None = None,
guardrail_context: Mapping[str, object] | None = None,
) -> CallToolResult:
- from litellm.proxy._experimental.mcp_server.server import (
+ from litellm.proxy._experimental.mcp_server.operations import (
_get_allowed_mcp_servers,
execute_mcp_tool,
raise_denied_scoped_mcp_access,
@@ -643,6 +643,7 @@ async def handle_mcp_tool_call(
mcp_server_auth_headers=mcp_server_auth_headers,
oauth2_headers=oauth2_headers,
raw_headers=raw_headers,
+ client_ip=client_ip,
litellm_logging_obj=litellm_logging_obj,
requested_server_id=requested_server_id,
guardrail_context=guardrail_context,
diff --git a/litellm/proxy/_types.py b/litellm/proxy/_types.py
index 3b34440d0bf..07338bcff52 100644
--- a/litellm/proxy/_types.py
+++ b/litellm/proxy/_types.py
@@ -3238,6 +3238,7 @@ class UserAPIKeyAuth(LiteLLM_VerificationTokenView): # the expected response ob
# above; a forged value could at most narrow, but the stripping keeps the field's provenance
# single-owner so its meaning stays trustworthy.
mcp_session_resource_server_id: str | None = Field(default=None, exclude=True)
+ mcp_toolset_id: str | None = Field(default=None, exclude=True)
via_virtual_key: bool = Field(
default=False,
exclude=True,
@@ -3279,6 +3280,7 @@ class UserAPIKeyAuth(LiteLLM_VerificationTokenView): # the expected response ob
values.pop("mcp_admitted_user_subject", None)
values.pop("mcp_source_team_rpm_limits", None)
values.pop("mcp_session_resource_server_id", None)
+ values.pop("mcp_toolset_id", None)
values.pop("via_virtual_key", None)
if values.get("api_key") is not None:
values.update({"token": cls._safe_hash_litellm_api_key(values.get("api_key"))})
diff --git a/scripts/check_mcp_operation_boundary.py b/scripts/check_mcp_operation_boundary.py
new file mode 100644
index 00000000000..b6c9dcefabf
--- /dev/null
+++ b/scripts/check_mcp_operation_boundary.py
@@ -0,0 +1,65 @@
+import ast
+import sys
+from pathlib import Path
+from typing import Final
+
+PACKAGE: Final = Path("litellm/proxy/_experimental/mcp_server")
+LEGACY_ADAPTERS: Final = frozenset({"server.py", "legacy_callbacks.py", "mcp_context.py", "mcp_debug.py"})
+CONFINED_NAMES: Final = frozenset(
+ {
+ "auth_context_var",
+ "active_mcp_session_var",
+ "active_mcp_request_ctx_var",
+ "get_active_auth_context",
+ "get_active_mcp_session",
+ "get_active_mcp_request_ctx",
+ "get_or_extract_auth_context",
+ "_session_obj_auth_storage",
+ "WeakKeyDictionary",
+ "_mcp_active_toolset_id",
+ "_mcp_gateway_initialize_instructions",
+ "_mcp_gateway_server_name",
+ "_mcp_proxy_mode",
+ }
+)
+
+
+def is_confined(name: str) -> bool:
+ return name in CONFINED_NAMES or name.startswith("_stateful_session_")
+
+
+def violations(path: Path, source: str) -> tuple[str, ...]:
+ if path.name in LEGACY_ADAPTERS:
+ return ()
+ tree: Final = ast.parse(source, filename=str(path))
+ return tuple(
+ f"{path}:{node.lineno}: MCP request/session state belongs in a legacy adapter"
+ for node in ast.walk(tree)
+ if (
+ isinstance(node, ast.ImportFrom)
+ and (
+ (node.module or "").endswith(".mcp_context")
+ or any(is_confined(alias.name) for alias in node.names)
+ or (path.name in {"operations.py", "contracts.py"} and (node.module or "").endswith(".server"))
+ )
+ or isinstance(node, ast.Name)
+ and is_confined(node.id)
+ or isinstance(node, ast.Attribute)
+ and is_confined(node.attr)
+ )
+ )
+
+
+def main() -> int:
+ findings: Final = tuple(
+ finding for path in sorted(PACKAGE.rglob("*.py")) for finding in violations(path, path.read_text())
+ )
+ if findings:
+ print("\n".join(findings), file=sys.stderr)
+ return 1
+ print("MCP operation boundary: passed")
+ return 0
+
+
+if __name__ == "__main__":
+ sys.exit(main())
diff --git a/scripts/pre_commit_lint.sh b/scripts/pre_commit_lint.sh
index 1abd415d237..22cc38f841c 100755
--- a/scripts/pre_commit_lint.sh
+++ b/scripts/pre_commit_lint.sh
@@ -102,6 +102,9 @@ ui_prettier_pattern='^ui/litellm-dashboard/.*\.(js|jsx|ts|tsx|mjs|cjs|json|css|s
ui_eslint_pattern='^ui/litellm-dashboard/.*\.(js|jsx|ts|tsx|mjs|cjs)$'
litellm_py_files=$(scope_match "$litellm_py_pattern")
+if [ -n "$(scope_match '^(litellm/proxy/_experimental/mcp_server/|scripts/check_mcp_operation_boundary\.py)')" ]; then
+ uv run --no-sync python scripts/check_mcp_operation_boundary.py || exit 1
+fi
e2e_py_files=$(scope_match "$e2e_py_pattern")
test_tree_files=$(scope_match "$test_tree_pattern")
# ruff format (and CI's format step) skip enterprise; the rest of make lint covers it.
diff --git a/tests/test_litellm/proxy/_experimental/mcp_server/test_byok_oauth_endpoints.py b/tests/test_litellm/proxy/_experimental/mcp_server/test_byok_oauth_endpoints.py
index 87e23893616..a77b4c8d565 100644
--- a/tests/test_litellm/proxy/_experimental/mcp_server/test_byok_oauth_endpoints.py
+++ b/tests/test_litellm/proxy/_experimental/mcp_server/test_byok_oauth_endpoints.py
@@ -1,3 +1,4 @@
+from litellm.proxy._experimental.mcp_server import operations as mcp_operations
"""
Unit tests for the BYOK OAuth 2.1 authorization server endpoints.
@@ -592,7 +593,7 @@ async def test_check_byok_credential_missing_credential(monkeypatch):
monkeypatch.delenv("PROXY_BASE_URL", raising=False)
monkeypatch.delenv("SERVER_ROOT_PATH", raising=False)
- server_module.byok_credential_cache.flush_cache()
+ mcp_operations.byok_credential_cache.flush_cache()
mock_prisma = MagicMock()
with (
@@ -628,13 +629,13 @@ async def test_execute_byok_tool_missing_credential_advertises_api_key_flow(monk
from litellm.types.mcp_server.mcp_server_manager import MCPServer
monkeypatch.setenv("PROXY_BASE_URL", "https://gateway.example.com/proxy")
- mcp_module.byok_credential_cache.flush_cache()
+ mcp_operations.byok_credential_cache.flush_cache()
server = MCPServer(server_id="byok-discovery", name="byok-discovery", transport=MCPTransport.http, is_byok=True)
prisma = MagicMock()
prisma.db.litellm_mcpusercredentials.find_unique = AsyncMock(return_value=None)
monkeypatch.setattr(proxy_server, "prisma_client", prisma)
with pytest.raises(HTTPException) as exc_info:
- await mcp_module.execute_mcp_tool(
+ await mcp_operations.execute_mcp_tool(
name="list_regions",
arguments={},
allowed_mcp_servers=[server],
@@ -687,7 +688,7 @@ async def test_invalidate_byok_cred_cache_evicts_locally_and_broadcasts_the_same
server = MCPServer(server_id="byok-revoke", name="byok-server", transport=MCPTransport.http, is_byok=True)
user_auth = UserAPIKeyAuth(user_id="mallory", api_key="sk-test")
- server_module.byok_credential_cache.flush_cache()
+ mcp_operations.byok_credential_cache.flush_cache()
db_lookup = AsyncMock(side_effect=["sk-before-revoke", None])
publish = AsyncMock()
@@ -699,13 +700,13 @@ async def test_invalidate_byok_cred_cache_evicts_locally_and_broadcasts_the_same
"litellm.proxy.proxy_server.prisma_client", MagicMock()
),
patch.object( # test-quality-ok: the redis publisher is module-level; asserting the broadcast without a redis
- server_module, "publish_auth_cache_invalidation", new=publish
+ mcp_operations, "publish_auth_cache_invalidation", new=publish
),
):
- assert await server_module._get_byok_credential(server, user_auth) == "sk-before-revoke"
- assert await server_module._get_byok_credential(server, user_auth) == "sk-before-revoke"
- await server_module._invalidate_byok_cred_cache("mallory", "byok-revoke")
- assert await server_module._get_byok_credential(server, user_auth) is None
+ assert await mcp_operations._get_byok_credential(server, user_auth) == "sk-before-revoke"
+ assert await mcp_operations._get_byok_credential(server, user_auth) == "sk-before-revoke"
+ await mcp_operations._invalidate_byok_cred_cache("mallory", "byok-revoke")
+ assert await mcp_operations._get_byok_credential(server, user_auth) is None
assert db_lookup.await_count == 2
publish.assert_awaited_once_with(cache_key=byok_credential_cache_key("mallory", "byok-revoke"))
diff --git a/tests/test_litellm/proxy/_experimental/mcp_server/test_contracts.py b/tests/test_litellm/proxy/_experimental/mcp_server/test_contracts.py
new file mode 100644
index 00000000000..e13ecdfcce9
--- /dev/null
+++ b/tests/test_litellm/proxy/_experimental/mcp_server/test_contracts.py
@@ -0,0 +1,60 @@
+from dataclasses import FrozenInstanceError
+
+import pytest
+
+from litellm.proxy._experimental.mcp_server.operations import prepare_context
+from litellm.proxy._types import UserAPIKeyAuth
+
+
+def test_operation_context_isolates_nested_headers_and_caller_permissions():
+ caller = UserAPIKeyAuth(user_id="alpha", models=["allowed"])
+ caller.mcp_admitted_user_subject = True
+ caller.mcp_session_resource_server_id = "alpha-server"
+ caller.mcp_toolset_id = "toolset-alpha"
+ caller.mcp_source_team_rpm_limits = {"team": {"alpha-server": 2}}
+ headers = {"x-caller": "alpha"}
+ server_headers = {"alpha-server": {"authorization": "alpha-token"}}
+ context = prepare_context(caller, raw_headers=headers, mcp_server_auth_headers=server_headers)
+
+ caller.models.append("forbidden")
+ caller.mcp_source_team_rpm_limits["team"]["alpha-server"] = 999
+ headers["x-caller"] = "bravo"
+ server_headers["alpha-server"]["authorization"] = "bravo-token"
+ captured = context.user_api_key_auth
+ assert captured is not None
+ assert captured.models == ["allowed"]
+ assert captured.mcp_admitted_user_subject is True
+ assert captured.mcp_session_resource_server_id == "alpha-server"
+ assert captured.mcp_toolset_id == "toolset-alpha"
+ assert captured.mcp_source_team_rpm_limits == {"team": {"alpha-server": 2}}
+ captured.models.append("also-forbidden")
+ assert context.user_api_key_auth.models == ["allowed"]
+ assert context.raw_headers == {"x-caller": "alpha"}
+ assert context.mcp_server_auth_headers == {"alpha-server": {"authorization": "alpha-token"}}
+ with pytest.raises(TypeError):
+ context.raw_headers["x-caller"] = "changed"
+ with pytest.raises(TypeError):
+ context.mcp_server_auth_headers["alpha-server"]["authorization"] = "changed"
+ with pytest.raises(FrozenInstanceError):
+ context.client_ip = "untrusted"
+
+
+def test_operation_context_preserves_missing_and_empty_inputs():
+ missing = prepare_context()
+ empty = prepare_context(mcp_servers=[], raw_headers={}, oauth2_headers={}, mcp_server_auth_headers={})
+ assert missing.user_api_key_auth is None
+ assert missing.mcp_servers is None
+ assert missing.raw_headers is None
+ assert missing.oauth2_headers is None
+ assert missing.mcp_server_auth_headers is None
+ assert empty.mcp_servers == ()
+ assert empty.raw_headers == {}
+ assert empty.oauth2_headers == {}
+ assert empty.mcp_server_auth_headers == {}
+
+
+def test_toolset_request_marker_cannot_be_supplied_by_caller_or_serialized():
+ auth = UserAPIKeyAuth.model_validate({"user_id": "alpha", "mcp_toolset_id": "forged"})
+ assert auth.mcp_toolset_id is None
+ auth.mcp_toolset_id = "server-resolved"
+ assert "mcp_toolset_id" not in auth.model_dump()
diff --git a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_block_recording.py b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_block_recording.py
index 64d926bc5e3..b8aadef430f 100644
--- a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_block_recording.py
+++ b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_block_recording.py
@@ -1,5 +1,6 @@
+from litellm.proxy._experimental.mcp_server import operations as mcp_operations
"""Tests for guardrail-block recording in
-``litellm.proxy._experimental.mcp_server.server.call_mcp_tool``.
+``litellm.proxy._experimental.mcp_server.operations.call_mcp_tool``.
A pre-call MCP guardrail block *raises* into ``call_mcp_tool``'s
``except Exception``. The failure spend-log row that the Guardrails Monitor's
@@ -70,7 +71,7 @@ async def _call_block(logging_obj, order: list, *, user_api_key_auth=mock.sentin
with mock.patch.dict(sys.modules, {"litellm.proxy.proxy_server": fake_proxy_server}):
with contextlib.suppress(HTTPException):
- await server.call_mcp_tool.__wrapped__(
+ await mcp_operations.call_mcp_tool.__wrapped__(
name="t",
arguments=None,
user_api_key_auth=user_api_key_auth,
diff --git a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_hook_extra_headers.py b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_hook_extra_headers.py
index 28faf375ab8..9659eb1cbc2 100644
--- a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_hook_extra_headers.py
+++ b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_hook_extra_headers.py
@@ -1229,7 +1229,7 @@ class TestResolveByokMcpAuthHeader:
user_auth = UserAPIKeyAuth(user_id="user-1", api_key="sk-dashboard")
with patch(
- "litellm.proxy._experimental.mcp_server.server._get_byok_credential",
+ "litellm.proxy._experimental.mcp_server.operations._get_byok_credential",
new=AsyncMock(return_value="stored-cred"),
):
result = await _resolve_byok_mcp_auth_header(server, user_auth, None)
@@ -1249,7 +1249,7 @@ class TestResolveByokMcpAuthHeader:
user_auth = UserAPIKeyAuth(user_id="user-1", api_key="sk-dashboard")
with patch(
- "litellm.proxy._experimental.mcp_server.server._get_byok_credential",
+ "litellm.proxy._experimental.mcp_server.operations._get_byok_credential",
new=AsyncMock(return_value=None),
):
with pytest.raises(HTTPException) as exc_info:
@@ -1272,7 +1272,7 @@ class TestResolveByokMcpAuthHeader:
check_mock = AsyncMock(return_value=None)
with patch(
- "litellm.proxy._experimental.mcp_server.server._check_byok_credential",
+ "litellm.proxy._experimental.mcp_server.operations._check_byok_credential",
new=check_mock,
):
result = await _resolve_byok_mcp_auth_header(server, user_auth, "caller-header")
diff --git a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_oauth_passthrough_tools.py b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_oauth_passthrough_tools.py
index 3f5d4ad83ea..1909e3306a2 100644
--- a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_oauth_passthrough_tools.py
+++ b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_oauth_passthrough_tools.py
@@ -1,3 +1,4 @@
+from litellm.proxy._experimental.mcp_server import operations as mcp_operations
"""Unit tests for MCP OAuth passthrough tool-fetch behavior."""
import logging
@@ -339,16 +340,16 @@ async def test_aggregate_list_tools_absorbs_one_unauthenticated_server():
raise MCPUpstreamAuthError(status_code=401, www_authenticate=None, server_name=server.name)
return [good_tool]
- with patch.object(mcp_server, "_get_allowed_mcp_servers", AsyncMock(return_value=[delegate, working])), patch.object(
- mcp_server, "_prefetch_oauth_creds_for_user", AsyncMock(return_value={})
- ), patch.object(mcp_server, "_prepare_mcp_server_headers", MagicMock(return_value=(None, None))), patch.object(
- mcp_server, "_get_user_oauth_extra_headers_from_db", AsyncMock(return_value=None)
+ with patch.object(mcp_operations, "_get_allowed_mcp_servers", AsyncMock(return_value=[delegate, working])), patch.object(
+ mcp_operations, "_prefetch_oauth_creds_for_user", AsyncMock(return_value={})
+ ), patch.object(mcp_operations, "_prepare_mcp_server_headers", MagicMock(return_value=(None, None))), patch.object(
+ mcp_operations, "_get_user_oauth_extra_headers_from_db", AsyncMock(return_value=None)
), patch.object(
- mcp_server, "filter_tools_by_key_team_permissions", AsyncMock(side_effect=lambda tools, **k: tools)
+ mcp_operations, "filter_tools_by_key_team_permissions", AsyncMock(side_effect=lambda tools, **k: tools)
), patch.object(
- mcp_server.global_mcp_server_manager, "_get_tools_from_server", AsyncMock(side_effect=fake_get_tools)
+ mcp_operations.global_mcp_server_manager, "_get_tools_from_server", AsyncMock(side_effect=fake_get_tools)
):
- listing = await mcp_server._get_tools_from_mcp_servers(
+ listing = await mcp_operations._get_tools_from_mcp_servers(
user_api_key_auth=UserAPIKeyAuth(token="h", user_id="u1"),
mcp_auth_header=None,
mcp_servers=None,
@@ -382,14 +383,14 @@ async def test_single_server_route_also_absorbs_upstream_auth_error():
# //mcp sets the path-derived single-server scope; absorption must hold even then.
token = _mcp_gateway_server_name.set("delegate_docs")
try:
- with patch.object(mcp_server, "_get_allowed_mcp_servers", AsyncMock(return_value=[delegate])), patch.object(
- mcp_server, "_prefetch_oauth_creds_for_user", AsyncMock(return_value={})
- ), patch.object(mcp_server, "_prepare_mcp_server_headers", MagicMock(return_value=(None, None))), patch.object(
- mcp_server, "_get_user_oauth_extra_headers_from_db", AsyncMock(return_value=None)
+ with patch.object(mcp_operations, "_get_allowed_mcp_servers", AsyncMock(return_value=[delegate])), patch.object(
+ mcp_operations, "_prefetch_oauth_creds_for_user", AsyncMock(return_value={})
+ ), patch.object(mcp_operations, "_prepare_mcp_server_headers", MagicMock(return_value=(None, None))), patch.object(
+ mcp_operations, "_get_user_oauth_extra_headers_from_db", AsyncMock(return_value=None)
), patch.object(
- mcp_server.global_mcp_server_manager, "_get_tools_from_server", AsyncMock(side_effect=fake_get_tools)
+ mcp_operations.global_mcp_server_manager, "_get_tools_from_server", AsyncMock(side_effect=fake_get_tools)
):
- listing = await mcp_server._get_tools_from_mcp_servers(
+ listing = await mcp_operations._get_tools_from_mcp_servers(
user_api_key_auth=UserAPIKeyAuth(token="h", user_id="u1"),
mcp_auth_header=None,
mcp_servers=["delegate_docs"],
@@ -419,15 +420,15 @@ async def test_aggregate_with_single_accessible_server_still_absorbs():
async def fake_get_tools(server, **kwargs):
raise MCPUpstreamAuthError(status_code=401, www_authenticate=None, server_name=server.name)
- with patch.object(mcp_server, "_get_allowed_mcp_servers", AsyncMock(return_value=[delegate])), patch.object(
- mcp_server, "_prefetch_oauth_creds_for_user", AsyncMock(return_value={})
- ), patch.object(mcp_server, "_prepare_mcp_server_headers", MagicMock(return_value=(None, None))), patch.object(
- mcp_server, "_get_user_oauth_extra_headers_from_db", AsyncMock(return_value=None)
+ with patch.object(mcp_operations, "_get_allowed_mcp_servers", AsyncMock(return_value=[delegate])), patch.object(
+ mcp_operations, "_prefetch_oauth_creds_for_user", AsyncMock(return_value={})
+ ), patch.object(mcp_operations, "_prepare_mcp_server_headers", MagicMock(return_value=(None, None))), patch.object(
+ mcp_operations, "_get_user_oauth_extra_headers_from_db", AsyncMock(return_value=None)
), patch.object(
- mcp_server.global_mcp_server_manager, "_get_tools_from_server", AsyncMock(side_effect=fake_get_tools)
+ mcp_operations.global_mcp_server_manager, "_get_tools_from_server", AsyncMock(side_effect=fake_get_tools)
):
# Aggregate route: no explicit server filter, even though only one server is accessible.
- listing = await mcp_server._get_tools_from_mcp_servers(
+ listing = await mcp_operations._get_tools_from_mcp_servers(
user_api_key_auth=UserAPIKeyAuth(token="h", user_id="u1"),
mcp_auth_header=None,
mcp_servers=None,
@@ -475,3 +476,25 @@ async def test_client_creation_failure_logs_sanitized_exchange(monkeypatch, capl
await manager._get_tools_from_server(server)
assert "POST https://upstream/ -> HTTP 500" in caplog.text
assert "missing_scope" in caplog.text and "query-secret" not in caplog.text
+
+
+@pytest.mark.parametrize(
+ "oauth_headers,server_headers,authorized",
+ [
+ ({"Authorization": "Bearer upstream"}, None, True),
+ ({"AUTHORIZATION": "Bearer upstream"}, None, True),
+ ({"x-unrelated": "present"}, None, False),
+ (None, {"catalog": {"Authorization": "Bearer scoped"}}, True),
+ (None, {"other-server": {"Authorization": "Bearer unrelated"}}, False),
+ (None, {"catalog": {"x-unrelated": "present"}}, False),
+ (None, {"catalog": "Bearer legacy"}, True),
+ (None, {"catalog": " "}, False),
+ ],
+)
+def test_passthrough_admission_recognizes_only_matching_authorization(oauth_headers, server_headers, authorized):
+ from litellm.proxy._experimental.mcp_server.operations import _client_has_passthrough_authorization
+ from litellm.types.mcp import MCPTransport
+ from litellm.types.mcp_server.mcp_server_manager import MCPServer
+
+ server = MCPServer(server_id="catalog", name="catalog", alias="catalog", transport=MCPTransport.http)
+ assert _client_has_passthrough_authorization(server, oauth_headers, server_headers) is authorized
diff --git a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_proxy_mode.py b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_proxy_mode.py
index 84d4f1fd083..ed5d67164bd 100644
--- a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_proxy_mode.py
+++ b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_proxy_mode.py
@@ -1,3 +1,4 @@
+from litellm.proxy._experimental.mcp_server import operations as mcp_operations
import json
from datetime import datetime
@@ -27,8 +28,8 @@ def proxy_mode():
@pytest.mark.asyncio
@pytest.mark.usefixtures("proxy_mode")
async def test_proxy_call_rejects_non_proxy_tool_names() -> None:
- result = await server._dispatch_virtual_mcp_tool(
- name="math_stdio-add", arguments={"a": 1, "b": 2}, user_api_key_auth=AUTH, client_ip=None
+ result = await mcp_operations._dispatch_virtual_mcp_tool(
+ name="math_stdio-add", arguments={"a": 1, "b": 2}, user_api_key_auth=AUTH, client_ip=None, mcp_proxy_mode=True
)
assert result is not None
@@ -105,12 +106,13 @@ async def test_proxy_scope_exception_emits_failure_log(monkeypatch: pytest.Monke
arguments = {"tool_id": "denied-scope", "arguments": {}}
with pytest.raises(HTTPException) as denied:
- await server._dispatch_virtual_mcp_tool(
+ await mcp_operations._dispatch_virtual_mcp_tool(
name="call_tool",
arguments=arguments,
user_api_key_auth=auth,
client_ip=None,
mcp_servers=["ungranted"],
+ mcp_proxy_mode=True,
raw_headers={"authorization": "Bearer raw-scope-secret", "x-litellm-call-id": "scope-denial"},
)
diff --git a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_server.py b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_server.py
index 3668a06203c..169de9095cf 100644
--- a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_server.py
+++ b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_server.py
@@ -1,3 +1,4 @@
+from litellm.proxy._experimental.mcp_server import operations as mcp_operations
import asyncio
import contextlib
import contextvars
@@ -138,7 +139,7 @@ async def test_mcp_server_tool_call_body_contains_request_data(_mcp_request_ctx)
mock_add_litellm_data_to_request,
):
with patch(
- "litellm.proxy._experimental.mcp_server.server.call_mcp_tool",
+ "litellm.proxy._experimental.mcp_server.operations.call_mcp_tool",
mock_call_mcp_tool,
):
with patch(
@@ -194,7 +195,7 @@ async def test_mcp_server_tool_call_forwards_client_headers_to_logging(_mcp_requ
mock_add_litellm_data_to_request,
):
with patch(
- "litellm.proxy._experimental.mcp_server.server.call_mcp_tool",
+ "litellm.proxy._experimental.mcp_server.operations.call_mcp_tool",
mock_call_mcp_tool,
):
with patch("litellm.proxy.proxy_server.proxy_config", MagicMock()):
@@ -241,7 +242,7 @@ async def test_mcp_server_tool_call_strips_custom_litellm_key_header(_mcp_reques
capturing_add_litellm_data_to_request,
):
with patch(
- "litellm.proxy._experimental.mcp_server.server.call_mcp_tool",
+ "litellm.proxy._experimental.mcp_server.operations.call_mcp_tool",
mock_call_mcp_tool,
):
with patch("litellm.proxy.proxy_server.proxy_config", MagicMock()):
@@ -287,11 +288,11 @@ async def test_mcp_server_tool_call_relays_upstream_auth_error_as_iserror(_mcp_r
mock_add_litellm_data_to_request,
):
with patch(
- "litellm.proxy._experimental.mcp_server.server.call_mcp_tool",
+ "litellm.proxy._experimental.mcp_server.operations.call_mcp_tool",
mock_call_mcp_tool,
):
with patch("litellm.proxy.proxy_server.proxy_config", MagicMock()):
- with patch("litellm.proxy._experimental.mcp_server.server.verbose_logger", mock_logger):
+ with patch("litellm.proxy._experimental.mcp_server.operations.verbose_logger", mock_logger):
result = await mcp_server_tool_call(_mcp_request_ctx(), _call_tool_params("test_tool", {"param": "value"}))
assert result.is_error is True
@@ -867,15 +868,15 @@ async def test_get_prompts_from_mcp_servers_success():
with (
patch(
- "litellm.proxy._experimental.mcp_server.server._get_allowed_mcp_servers",
+ "litellm.proxy._experimental.mcp_server.operations._get_allowed_mcp_servers",
AsyncMock(return_value=[server_a, server_b]),
) as mock_allowed,
patch(
- "litellm.proxy._experimental.mcp_server.server._prepare_mcp_server_headers",
+ "litellm.proxy._experimental.mcp_server.operations._prepare_mcp_server_headers",
return_value=(None, None),
) as mock_headers,
patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager",
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager",
) as mock_manager,
):
mock_manager.get_prompts_from_server = AsyncMock(
@@ -927,15 +928,15 @@ async def test_get_resources_from_mcp_servers_success():
with (
patch(
- "litellm.proxy._experimental.mcp_server.server._get_allowed_mcp_servers",
+ "litellm.proxy._experimental.mcp_server.operations._get_allowed_mcp_servers",
AsyncMock(return_value=[server_a, server_b]),
) as mock_allowed,
patch(
- "litellm.proxy._experimental.mcp_server.server._prepare_mcp_server_headers",
+ "litellm.proxy._experimental.mcp_server.operations._prepare_mcp_server_headers",
return_value=(None, None),
) as mock_headers,
patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager",
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager",
) as mock_manager,
):
mock_manager.get_resources_from_server = AsyncMock(
@@ -992,15 +993,15 @@ async def test_get_resource_templates_from_mcp_servers_success():
with (
patch(
- "litellm.proxy._experimental.mcp_server.server._get_allowed_mcp_servers",
+ "litellm.proxy._experimental.mcp_server.operations._get_allowed_mcp_servers",
AsyncMock(return_value=[server]),
) as mock_allowed,
patch(
- "litellm.proxy._experimental.mcp_server.server._prepare_mcp_server_headers",
+ "litellm.proxy._experimental.mcp_server.operations._prepare_mcp_server_headers",
return_value=(None, None),
) as mock_headers,
patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager",
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager",
) as mock_manager,
):
mock_manager.get_resource_templates_from_server = AsyncMock(
@@ -1042,15 +1043,15 @@ async def test_mcp_get_prompt_success():
with (
patch(
- "litellm.proxy._experimental.mcp_server.server._get_allowed_mcp_servers",
+ "litellm.proxy._experimental.mcp_server.operations._get_allowed_mcp_servers",
AsyncMock(return_value=[server]),
) as mock_allowed,
patch(
- "litellm.proxy._experimental.mcp_server.server._prepare_mcp_server_headers",
+ "litellm.proxy._experimental.mcp_server.operations._prepare_mcp_server_headers",
return_value=({"Authorization": "token"}, {"X-Test": "1"}),
) as mock_headers,
patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager",
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager",
) as mock_manager,
):
mock_manager.get_prompt_from_server = AsyncMock(return_value=prompt_result)
@@ -1078,6 +1079,7 @@ async def test_mcp_get_prompt_success():
mcp_auth_header={"Authorization": "token"},
extra_headers={"X-Test": "1"},
raw_headers=None,
+ client_ip=None,
)
assert result is prompt_result
@@ -1106,15 +1108,15 @@ async def test_mcp_read_resource_success():
with (
patch(
- "litellm.proxy._experimental.mcp_server.server._get_allowed_mcp_servers",
+ "litellm.proxy._experimental.mcp_server.operations._get_allowed_mcp_servers",
AsyncMock(return_value=[server]),
) as mock_allowed,
patch(
- "litellm.proxy._experimental.mcp_server.server._prepare_mcp_server_headers",
+ "litellm.proxy._experimental.mcp_server.operations._prepare_mcp_server_headers",
return_value=({"Authorization": "token"}, {"X-Test": "1"}),
) as mock_headers,
patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager",
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager",
) as mock_manager,
):
mock_manager.read_resource_from_server = AsyncMock(return_value=read_result)
@@ -1140,6 +1142,7 @@ async def test_mcp_read_resource_success():
mcp_auth_header={"Authorization": "token"},
extra_headers={"X-Test": "1"},
raw_headers=None,
+ client_ip=None,
)
assert result is read_result
@@ -1264,7 +1267,7 @@ async def test_mcp_read_resource_multiple_servers_error():
server_b.name = "server_b"
with patch(
- "litellm.proxy._experimental.mcp_server.server._get_allowed_mcp_servers",
+ "litellm.proxy._experimental.mcp_server.operations._get_allowed_mcp_servers",
AsyncMock(return_value=[server_a, server_b]),
) as mock_allowed:
with pytest.raises(HTTPException) as exc_info:
@@ -1354,11 +1357,11 @@ async def test_get_tools_from_mcp_servers_continues_when_one_server_fails():
mock_manager._get_tools_from_server = mock_get_tools_from_server
with patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager",
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager",
mock_manager,
):
with patch(
- "litellm.proxy._experimental.mcp_server.server.verbose_logger",
+ "litellm.proxy._experimental.mcp_server.operations.verbose_logger",
) as mock_logger:
# Test with server-specific auth headers
mcp_server_auth_headers = {
@@ -1450,11 +1453,11 @@ async def test_get_tools_from_mcp_servers_handles_all_servers_failing():
mock_manager._get_tools_from_server = mock_get_tools_from_server
with patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager",
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager",
mock_manager,
):
with patch(
- "litellm.proxy._experimental.mcp_server.server.verbose_logger",
+ "litellm.proxy._experimental.mcp_server.operations.verbose_logger",
) as mock_logger:
# Test with server-specific auth headers
mcp_server_auth_headers = {
@@ -1524,11 +1527,11 @@ async def _denied_scoped_list(
with (
patch( # test-quality-ok: the permission resolver is a module-level function; the suite's only seam
- "litellm.proxy._experimental.mcp_server.server._get_allowed_mcp_servers",
+ "litellm.proxy._experimental.mcp_server.operations._get_allowed_mcp_servers",
resolver,
),
patch( # test-quality-ok: the server registry is a module-level singleton; the suite's only seam
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager",
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager",
mock_manager,
),
):
@@ -1575,11 +1578,11 @@ async def test_empty_scope_lists_nothing_instead_of_raising_a_nameless_denial():
with (
patch( # test-quality-ok: the permission resolver is a module-level function; the suite's only seam
- "litellm.proxy._experimental.mcp_server.server._get_allowed_mcp_servers",
+ "litellm.proxy._experimental.mcp_server.operations._get_allowed_mcp_servers",
resolver,
),
patch( # test-quality-ok: the server registry is a module-level singleton; the suite's only seam
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager",
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager",
_denied_scope_manager({"github": "srv-github"}),
),
):
@@ -1721,7 +1724,8 @@ async def test_scoped_list_agent_veto_attributed_for_differently_cased_server_na
@pytest.mark.asyncio
-async def test_handle_list_tools_converts_permission_httpexception_to_mcp_error(_mcp_request_ctx):
+@pytest.mark.parametrize("denial_at_auth", [False, True])
+async def test_handle_list_tools_converts_permission_httpexception_to_mcp_error(_mcp_request_ctx, denial_at_auth):
"""The MCP protocol handler surfaces a permission HTTPException as a clean JSON-RPC error
(MCPError, INVALID_REQUEST) carrying the denial message, instead of a raw 500."""
try:
@@ -1738,10 +1742,10 @@ async def test_handle_list_tools_converts_permission_httpexception_to_mcp_error(
with (
patch( # test-quality-ok: the protocol handler reads auth from module context; no injection seam
"litellm.proxy._experimental.mcp_server.server.get_or_extract_auth_context",
- new=AsyncMock(return_value=(None, None, None, None, None, None, None)),
+ new=AsyncMock(return_value=(None, None, None, None, None, None, None), side_effect=denial if denial_at_auth else None),
),
patch( # test-quality-ok: the listing helper is the handler's only collaborator; the suite's seam
- "litellm.proxy._experimental.mcp_server.server._list_mcp_tools",
+ "litellm.proxy._experimental.mcp_server.operations._list_mcp_tools",
new=AsyncMock(side_effect=denial),
),
):
@@ -1768,7 +1772,7 @@ async def test_mcp_server_tool_call_renders_denial_message_not_detail_dict(_mcp_
new=AsyncMock(return_value=(None, None, None, None, None, None, None)),
),
patch( # test-quality-ok: the tool-call helper is the handler's only collaborator; the suite's seam
- "litellm.proxy._experimental.mcp_server.server.call_mcp_tool",
+ "litellm.proxy._experimental.mcp_server.operations.call_mcp_tool",
new=AsyncMock(side_effect=denial),
),
):
@@ -1819,7 +1823,7 @@ async def test_mcp_server_tool_call_body_with_none_arguments(_mcp_request_ctx):
mock_add_litellm_data_to_request,
):
with patch(
- "litellm.proxy._experimental.mcp_server.server.call_mcp_tool",
+ "litellm.proxy._experimental.mcp_server.operations.call_mcp_tool",
mock_call_mcp_tool,
):
with patch(
@@ -1893,7 +1897,7 @@ async def test_concurrent_initialize_session_managers():
"run",
return_value=mock_cm_sse,
) as mock_sse_run,
- patch("litellm.proxy._experimental.mcp_server.server.verbose_logger"),
+ patch("litellm.proxy._experimental.mcp_server.operations.verbose_logger"),
):
# Create multiple concurrent tasks that call initialize_session_managers
async def init_task():
@@ -2333,7 +2337,7 @@ async def test_mcp_routing_chunked_initialize_to_stateful():
"litellm.proxy._experimental.mcp_server.server.set_auth_context",
),
patch( # test-quality-ok: registry is empty in unit tests; key owns one server
- "litellm.proxy._experimental.mcp_server.server._get_allowed_mcp_servers",
+ "litellm.proxy._experimental.mcp_server.operations._get_allowed_mcp_servers",
new_callable=AsyncMock,
return_value=[MagicMock()],
),
@@ -2823,7 +2827,7 @@ async def test_initialize_request_tracks_active_session_after_response_header():
return_value=(owner_auth, None, None, None, None, None),
),
patch( # test-quality-ok: registry is empty in unit tests; key owns one server
- "litellm.proxy._experimental.mcp_server.server._get_allowed_mcp_servers",
+ "litellm.proxy._experimental.mcp_server.operations._get_allowed_mcp_servers",
new_callable=AsyncMock,
return_value=[MagicMock()],
),
@@ -2976,7 +2980,7 @@ async def test_initialize_request_records_client_name_in_gateway_sessions_report
return_value=(owner_auth, None, None, None, None, None),
),
patch( # test-quality-ok: registry is empty in unit tests; key owns one server
- "litellm.proxy._experimental.mcp_server.server._get_allowed_mcp_servers",
+ "litellm.proxy._experimental.mcp_server.operations._get_allowed_mcp_servers",
new_callable=AsyncMock,
return_value=[MagicMock()],
),
@@ -3451,7 +3455,7 @@ async def test_initialize_request_with_existing_session_tracks_new_session():
),
),
patch( # test-quality-ok: registry is empty in unit tests; key owns one server
- "litellm.proxy._experimental.mcp_server.server._get_allowed_mcp_servers",
+ "litellm.proxy._experimental.mcp_server.operations._get_allowed_mcp_servers",
new_callable=AsyncMock,
return_value=[MagicMock()],
),
@@ -4164,7 +4168,7 @@ async def test_mcp_routing_with_conflicting_alias_and_group_name():
with (
patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager.get_allowed_mcp_servers",
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager.get_allowed_mcp_servers",
mock_get_allowed,
),
patch(
@@ -4172,7 +4176,7 @@ async def test_mcp_routing_with_conflicting_alias_and_group_name():
mock_db_lookup,
),
patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager._get_tools_from_server",
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager._get_tools_from_server",
mock_get_tools_spy,
),
):
@@ -4281,16 +4285,16 @@ async def test_oauth2_caller_headers_not_forwarded_for_migrated_server():
side_effect=mock_fetch_tools_with_timeout,
),
patch(
- "litellm.proxy._experimental.mcp_server.server._get_allowed_mcp_servers",
+ "litellm.proxy._experimental.mcp_server.operations._get_allowed_mcp_servers",
AsyncMock(return_value=[oauth2_server]),
),
patch(
- "litellm.proxy._experimental.mcp_server.server._prefetch_oauth_creds_for_user",
+ "litellm.proxy._experimental.mcp_server.operations._prefetch_oauth_creds_for_user",
new_callable=AsyncMock,
return_value={},
),
patch(
- "litellm.proxy._experimental.mcp_server.server._get_user_oauth_extra_headers_from_db",
+ "litellm.proxy._experimental.mcp_server.operations._get_user_oauth_extra_headers_from_db",
new_callable=AsyncMock,
return_value=None,
),
@@ -4372,7 +4376,7 @@ async def test_list_tools_single_server_unprefixed_names():
mock_manager._get_tools_from_server = mock_get_tools_from_server
with patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager",
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager",
mock_manager,
):
listing = await _get_tools_from_mcp_servers(
@@ -4451,7 +4455,7 @@ async def test_list_tools_multiple_servers_prefixed_names():
mock_manager._get_tools_from_server = mock_get_tools_from_server
with patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager",
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager",
mock_manager,
):
listing = await _get_tools_from_mcp_servers(
@@ -4631,7 +4635,7 @@ async def test_call_mcp_tool_user_unauthorized_access():
AsyncMock(return_value=["allowed_server", "another_server"]),
),
patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager.get_mcp_server_by_id",
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager.get_mcp_server_by_id",
side_effect=mock_get_server_by_id,
),
):
@@ -4661,11 +4665,11 @@ async def test_call_mcp_tool_scoped_denial_names_the_binding_agent():
with (
patch( # test-quality-ok: the server registry is a module-level singleton; the suite's only seam
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager.get_allowed_mcp_servers",
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager.get_allowed_mcp_servers",
AsyncMock(return_value=[]),
),
patch( # test-quality-ok: the permission resolver is a module-level function; the suite's only seam
- "litellm.proxy._experimental.mcp_server.server._get_allowed_mcp_servers",
+ "litellm.proxy._experimental.mcp_server.operations._get_allowed_mcp_servers",
_scope_resolver({"github": "srv-github"}),
),
):
@@ -4737,7 +4741,7 @@ async def test_call_mcp_tool_unauthorized_403_does_not_leak_server_credentials()
AsyncMock(return_value=["allowed_server"]),
),
patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager.get_mcp_server_by_id",
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager.get_mcp_server_by_id",
side_effect=mock_get_server_by_id,
),
):
@@ -4880,7 +4884,7 @@ async def test_list_tools_filters_by_key_team_permissions():
mock_manager._get_tools_from_server = mock_get_tools_from_server
with patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager",
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager",
mock_manager,
):
listing = await _get_tools_from_mcp_servers(
@@ -4991,7 +4995,7 @@ async def test_list_tools_with_team_tool_permissions_inheritance():
mock_manager._get_tools_from_server = mock_get_tools_from_server
with patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager",
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager",
mock_manager,
):
# Mock the team object permission retrieval
@@ -5083,7 +5087,7 @@ async def test_list_tools_with_no_tool_permissions_shows_all():
mock_manager._get_tools_from_server = mock_get_tools_from_server
with patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager",
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager",
mock_manager,
):
listing = await _get_tools_from_mcp_servers(
@@ -5189,7 +5193,7 @@ async def test_list_tools_strips_prefix_when_matching_permissions():
mock_manager._get_tools_from_server = mock_get_tools_from_server
with patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager",
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager",
mock_manager,
):
listing = await _get_tools_from_mcp_servers(
@@ -5631,12 +5635,12 @@ async def test_call_mcp_tool_logs_failure_via_post_call_failure_hook():
return_value=mock_server,
),
patch(
- "litellm.proxy._experimental.mcp_server.server._get_allowed_mcp_servers_from_mcp_server_names",
+ "litellm.proxy._experimental.mcp_server.operations._get_allowed_mcp_servers_from_mcp_server_names",
new_callable=AsyncMock,
return_value=[mock_server],
),
patch(
- "litellm.proxy._experimental.mcp_server.server.execute_mcp_tool",
+ "litellm.proxy._experimental.mcp_server.operations.execute_mcp_tool",
new_callable=AsyncMock,
side_effect=Exception("boom"),
),
@@ -5700,26 +5704,26 @@ async def test_get_tools_from_mcp_servers_logs_list_tools_to_spendlogs_when_enab
with (
patch(
- "litellm.proxy._experimental.mcp_server.server._get_allowed_mcp_servers",
+ "litellm.proxy._experimental.mcp_server.operations._get_allowed_mcp_servers",
new=AsyncMock(return_value=[server_a]),
),
patch(
- "litellm.proxy._experimental.mcp_server.server._prepare_mcp_server_headers",
+ "litellm.proxy._experimental.mcp_server.operations._prepare_mcp_server_headers",
return_value=(None, None),
),
patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager",
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager",
) as mock_manager,
patch(
- "litellm.proxy._experimental.mcp_server.server.filter_tools_by_allowed_tools",
+ "litellm.proxy._experimental.mcp_server.operations.filter_tools_by_allowed_tools",
side_effect=lambda tools, _server: tools,
),
patch(
- "litellm.proxy._experimental.mcp_server.server.filter_tools_by_key_team_permissions",
+ "litellm.proxy._experimental.mcp_server.operations.filter_tools_by_key_team_permissions",
new=AsyncMock(side_effect=lambda tools, **_: tools),
),
patch(
- "litellm.proxy._experimental.mcp_server.server.function_setup",
+ "litellm.proxy._experimental.mcp_server.operations.function_setup",
side_effect=_capture_function_setup,
),
):
@@ -5782,26 +5786,26 @@ async def test_get_tools_from_mcp_servers_returns_tools_when_success_logging_fai
with (
patch(
- "litellm.proxy._experimental.mcp_server.server._get_allowed_mcp_servers",
+ "litellm.proxy._experimental.mcp_server.operations._get_allowed_mcp_servers",
new=AsyncMock(return_value=[server_a]),
),
patch(
- "litellm.proxy._experimental.mcp_server.server._prepare_mcp_server_headers",
+ "litellm.proxy._experimental.mcp_server.operations._prepare_mcp_server_headers",
return_value=(None, None),
),
patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager",
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager",
) as mock_manager,
patch(
- "litellm.proxy._experimental.mcp_server.server.filter_tools_by_allowed_tools",
+ "litellm.proxy._experimental.mcp_server.operations.filter_tools_by_allowed_tools",
side_effect=lambda tools, _server: tools,
),
patch(
- "litellm.proxy._experimental.mcp_server.server.filter_tools_by_key_team_permissions",
+ "litellm.proxy._experimental.mcp_server.operations.filter_tools_by_key_team_permissions",
new=AsyncMock(side_effect=lambda tools, **_: tools),
),
patch(
- "litellm.proxy._experimental.mcp_server.server.function_setup",
+ "litellm.proxy._experimental.mcp_server.operations.function_setup",
return_value=(dummy_logging_obj, None),
),
):
@@ -6102,23 +6106,23 @@ async def test_get_tools_from_mcp_servers_injects_stored_oauth2_token():
with (
patch(
- "litellm.proxy._experimental.mcp_server.server._get_allowed_mcp_servers",
+ "litellm.proxy._experimental.mcp_server.operations._get_allowed_mcp_servers",
new=AsyncMock(return_value=[oauth2_server]),
),
patch(
# Patch the bulk prefetch so no real DB connection is needed
- "litellm.proxy._experimental.mcp_server.server._prefetch_oauth_creds_for_user",
+ "litellm.proxy._experimental.mcp_server.operations._prefetch_oauth_creds_for_user",
new=AsyncMock(return_value=prefetched_creds),
) as mock_prefetch,
patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager",
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager",
) as mock_manager,
patch(
- "litellm.proxy._experimental.mcp_server.server.filter_tools_by_allowed_tools",
+ "litellm.proxy._experimental.mcp_server.operations.filter_tools_by_allowed_tools",
side_effect=lambda tools, _server: tools,
),
patch(
- "litellm.proxy._experimental.mcp_server.server.filter_tools_by_key_team_permissions",
+ "litellm.proxy._experimental.mcp_server.operations.filter_tools_by_key_team_permissions",
new=AsyncMock(side_effect=lambda tools, **_: tools),
),
):
@@ -6450,7 +6454,7 @@ class TestGatewayCreateInitializationOptions:
with (
patch(
- "litellm.proxy._experimental.mcp_server.server._get_allowed_mcp_servers",
+ "litellm.proxy._experimental.mcp_server.operations._get_allowed_mcp_servers",
new_callable=AsyncMock,
return_value=[scoped_server],
),
@@ -6480,7 +6484,7 @@ class TestGatewayCreateInitializationOptions:
from litellm.proxy._types import UserAPIKeyAuth
with patch( # test-quality-ok: grant resolution is the input under test
- "litellm.proxy._experimental.mcp_server.server._get_allowed_mcp_servers",
+ "litellm.proxy._experimental.mcp_server.operations._get_allowed_mcp_servers",
new_callable=AsyncMock,
return_value=[],
):
@@ -6506,7 +6510,7 @@ class TestGatewayCreateInitializationOptions:
from litellm.proxy._types import UserAPIKeyAuth
with patch( # test-quality-ok: grant resolution is the input under test
- "litellm.proxy._experimental.mcp_server.server._get_allowed_mcp_servers",
+ "litellm.proxy._experimental.mcp_server.operations._get_allowed_mcp_servers",
new_callable=AsyncMock,
return_value=[],
):
@@ -6531,7 +6535,7 @@ class TestGatewayCreateInitializationOptions:
from litellm.proxy._types import UserAPIKeyAuth
with patch( # test-quality-ok: grant resolution is the input under test
- "litellm.proxy._experimental.mcp_server.server._get_allowed_mcp_servers",
+ "litellm.proxy._experimental.mcp_server.operations._get_allowed_mcp_servers",
new_callable=AsyncMock,
return_value=[],
):
@@ -6587,7 +6591,7 @@ class TestGatewayCreateInitializationOptions:
),
),
patch(
- "litellm.proxy._experimental.mcp_server.server._get_allowed_mcp_servers",
+ "litellm.proxy._experimental.mcp_server.operations._get_allowed_mcp_servers",
new_callable=AsyncMock,
return_value=[scoped_server],
),
@@ -6722,14 +6726,14 @@ async def test_list_tools_with_legacy_db_m2m_server_resolves_oauth2_flow():
with (
patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager",
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager",
) as mock_manager,
patch(
- "litellm.proxy._experimental.mcp_server.server.filter_tools_by_allowed_tools",
+ "litellm.proxy._experimental.mcp_server.operations.filter_tools_by_allowed_tools",
side_effect=lambda tools, _server: tools,
),
patch(
- "litellm.proxy._experimental.mcp_server.server.filter_tools_by_key_team_permissions",
+ "litellm.proxy._experimental.mcp_server.operations.filter_tools_by_key_team_permissions",
new=AsyncMock(side_effect=lambda tools, **_: tools),
),
):
@@ -6992,7 +6996,7 @@ def _patch_delegate_resolver(server: MCPServer, *resolvable_names: str):
return server if name in resolvable_names else None
return patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager.get_mcp_server_by_name",
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager.get_mcp_server_by_name",
side_effect=_resolve,
)
@@ -7011,7 +7015,7 @@ async def test_legacy_delegate_bare_token_is_not_probed_upstream(): # test-qual
with (
_patch_delegate_resolver(server, "delegate_test"),
patch(
- "litellm.proxy._experimental.mcp_server.server._get_allowed_mcp_servers",
+ "litellm.proxy._experimental.mcp_server.operations._get_allowed_mcp_servers",
new=AsyncMock(return_value=[server]),
),
patch(
@@ -7047,7 +7051,7 @@ async def test_legacy_delegate_dual_credentials_are_not_probed_upstream(): # te
with (
patch( # test-quality-ok: isolate authorized-server resolution so this test targets the preflight boundary
- "litellm.proxy._experimental.mcp_server.server._get_allowed_mcp_servers",
+ "litellm.proxy._experimental.mcp_server.operations._get_allowed_mcp_servers",
new=AsyncMock(return_value=[server]),
),
patch( # test-quality-ok: the removed probe call is the security regression under test
@@ -7094,7 +7098,7 @@ async def test_oauth_passthrough_preflight_preserves_status_contract(probe_statu
with (
patch( # test-quality-ok: isolate authorized-server resolution so this test exercises the preflight contract
- "litellm.proxy._experimental.mcp_server.server._get_allowed_mcp_servers",
+ "litellm.proxy._experimental.mcp_server.operations._get_allowed_mcp_servers",
new=AsyncMock(return_value=[server]),
),
patch( # test-quality-ok: the upstream transport boundary is the behavior being mapped to an HTTP response
@@ -7140,7 +7144,7 @@ async def test_delegate_tokenless_request_not_probed():
with (
_patch_delegate_resolver(server, "delegate_test"),
patch(
- "litellm.proxy._experimental.mcp_server.server._get_allowed_mcp_servers",
+ "litellm.proxy._experimental.mcp_server.operations._get_allowed_mcp_servers",
new=AsyncMock(return_value=[server]),
),
patch(
@@ -7173,7 +7177,7 @@ async def test_delegate_preflight_skipped_on_multi_server_routes():
with (
_patch_delegate_resolver(servers[0], "delegate_test", "other_server"),
patch(
- "litellm.proxy._experimental.mcp_server.server._get_allowed_mcp_servers",
+ "litellm.proxy._experimental.mcp_server.operations._get_allowed_mcp_servers",
new=AsyncMock(return_value=servers),
),
patch(
@@ -7216,7 +7220,7 @@ async def test_bare_authorization_never_probes_passthrough_servers():
with (
_patch_delegate_resolver(passthrough_server, "pt_server"),
patch(
- "litellm.proxy._experimental.mcp_server.server._get_allowed_mcp_servers",
+ "litellm.proxy._experimental.mcp_server.operations._get_allowed_mcp_servers",
new=AsyncMock(return_value=[passthrough_server]),
),
patch(
@@ -7262,7 +7266,7 @@ async def test_delegate_not_probed_when_named_only_via_server_id():
with (
_patch_delegate_resolver(server, "delegate_test"),
patch(
- "litellm.proxy._experimental.mcp_server.server._get_allowed_mcp_servers",
+ "litellm.proxy._experimental.mcp_server.operations._get_allowed_mcp_servers",
new=AsyncMock(return_value=[server]),
),
patch(
@@ -7295,7 +7299,7 @@ async def test_delegate_probe_not_fanned_out_to_access_group_members():
with (
_patch_delegate_resolver(group_member, "delegate_test"),
patch(
- "litellm.proxy._experimental.mcp_server.server._get_allowed_mcp_servers",
+ "litellm.proxy._experimental.mcp_server.operations._get_allowed_mcp_servers",
new=AsyncMock(return_value=[group_member]),
),
patch(
@@ -7391,11 +7395,11 @@ async def test_execute_mcp_tool_rest_server_id_authoritative_for_unprefixed_tool
with (
patch.dict(
- mcp_module.global_mcp_server_manager.tool_name_to_mcp_server_name_mapping,
+ mcp_operations.global_mcp_server_manager.tool_name_to_mcp_server_name_mapping,
{"echo": oauth_server.name},
),
patch.object(
- mcp_module.global_mcp_server_manager,
+ mcp_operations.global_mcp_server_manager,
"get_registry",
return_value={
api_key_server.server_id: api_key_server,
@@ -7403,13 +7407,12 @@ async def test_execute_mcp_tool_rest_server_id_authoritative_for_unprefixed_tool
},
),
patch.object(
- mcp_module.global_mcp_server_manager,
+ mcp_operations.global_mcp_server_manager,
"_get_mcp_server_from_tool_name",
return_value=oauth_server,
),
patch.object(
- mcp_module,
- "_handle_managed_mcp_tool",
+ mcp_operations, "_handle_managed_mcp_tool",
new=fake_handle_managed_mcp_tool,
),
patch.object(
@@ -7418,12 +7421,12 @@ async def test_execute_mcp_tool_rest_server_id_authoritative_for_unprefixed_tool
return_value=True,
),
patch.object(
- mcp_module.global_mcp_tool_registry,
+ mcp_operations.global_mcp_tool_registry,
"get_tool",
return_value=None,
),
):
- await mcp_module.execute_mcp_tool(
+ await mcp_operations.execute_mcp_tool(
name="echo",
arguments={"message": "hello"},
allowed_mcp_servers=[api_key_server, oauth_server],
@@ -7456,7 +7459,7 @@ def _worker_that_never_listed(server: MCPServer, upstream_tools: tuple[str, ...]
from litellm.proxy._experimental.mcp_server import server as mcp_module
- mcp_module.global_mcp_server_manager.registry[server.server_id] = server
+ mcp_operations.global_mcp_server_manager.registry[server.server_id] = server
dispatched: dict[str, object] = {}
async def fake_handle_managed_mcp_tool(**kwargs):
@@ -7468,17 +7471,17 @@ def _worker_that_never_listed(server: MCPServer, upstream_tools: tuple[str, ...]
with (
patch.object( # test-quality-ok: the upstream MCP session is the boundary; a real one needs an initialize handshake over a live server
- mcp_module.global_mcp_server_manager,
+ mcp_operations.global_mcp_server_manager,
"_create_mcp_client",
new=AsyncMock(return_value=MagicMock()),
) as create_client,
patch.object( # test-quality-ok: same boundary, this is the tools/list answer the upstream would give
- mcp_module.global_mcp_server_manager,
+ mcp_operations.global_mcp_server_manager,
"_fetch_tools_with_timeout",
side_effect=fake_fetch_tools,
) as fetch_tools,
patch.object( # test-quality-ok: records the resolved server and bare name the managed call would forward upstream
- mcp_module, "_handle_managed_mcp_tool", new=fake_handle_managed_mcp_tool
+ mcp_operations, "_handle_managed_mcp_tool", new=fake_handle_managed_mcp_tool
),
):
yield SimpleNamespace(create_client=create_client, fetch_tools=fetch_tools, dispatched=dispatched)
@@ -7492,7 +7495,7 @@ async def test_execute_mcp_tool_lists_never_listed_passthrough_server_with_calle
server = _never_listed_passthrough_server()
with _worker_that_never_listed(server, upstream_tools=("add",)) as worker:
- await mcp_module.execute_mcp_tool(
+ await mcp_operations.execute_mcp_tool(
name="lazy_map-add",
arguments={"a": 1, "b": 2},
allowed_mcp_servers=[server],
@@ -7513,7 +7516,7 @@ async def test_execute_mcp_tool_rest_server_id_lists_never_listed_server_first()
server = _never_listed_passthrough_server()
with _worker_that_never_listed(server, upstream_tools=("add",)) as worker:
- await mcp_module.execute_mcp_tool(
+ await mcp_operations.execute_mcp_tool(
name="add",
arguments={"a": 1, "b": 2},
allowed_mcp_servers=[server],
@@ -7536,7 +7539,7 @@ async def test_execute_mcp_tool_unknown_tool_on_never_listed_server_lists_once_t
_worker_that_never_listed(server, upstream_tools=("add",)) as worker,
pytest.raises(HTTPException) as exc_info,
):
- await mcp_module.execute_mcp_tool(
+ await mcp_operations.execute_mcp_tool(
name="lazy_map-nope",
arguments={},
allowed_mcp_servers=[server],
@@ -7557,8 +7560,8 @@ async def test_execute_mcp_tool_does_not_relist_a_server_this_worker_already_lis
server = _never_listed_passthrough_server()
with _worker_that_never_listed(server, upstream_tools=("add",)) as worker:
- mcp_module.global_mcp_server_manager._create_prefixed_tools([MCPTool(name="add", inputSchema={})], server)
- await mcp_module.execute_mcp_tool(
+ mcp_operations.global_mcp_server_manager._create_prefixed_tools([MCPTool(name="add", inputSchema={})], server)
+ await mcp_operations.execute_mcp_tool(
name="lazy_map-add",
arguments={"a": 1, "b": 2},
allowed_mcp_servers=[server],
@@ -7580,8 +7583,8 @@ async def test_execute_mcp_tool_lists_a_tool_this_worker_has_not_yet_seen_on_a_l
server = _never_listed_passthrough_server()
with _worker_that_never_listed(server, upstream_tools=("add", "multiply")) as worker:
- mcp_module.global_mcp_server_manager._create_prefixed_tools([MCPTool(name="add", inputSchema={})], server)
- await mcp_module.execute_mcp_tool(
+ mcp_operations.global_mcp_server_manager._create_prefixed_tools([MCPTool(name="add", inputSchema={})], server)
+ await mcp_operations.execute_mcp_tool(
name="lazy_map-multiply",
arguments={"a": 1, "b": 2},
allowed_mcp_servers=[server],
@@ -7604,7 +7607,7 @@ async def test_execute_mcp_tool_never_lists_a_server_the_caller_cannot_access():
_worker_that_never_listed(server, upstream_tools=("add",)) as worker,
pytest.raises(HTTPException) as exc_info,
):
- await mcp_module.execute_mcp_tool(
+ await mcp_operations.execute_mcp_tool(
name="lazy_map-add",
arguments={"a": 1, "b": 2},
allowed_mcp_servers=[other_server],
@@ -7650,13 +7653,12 @@ async def test_execute_mcp_tool_strips_a_prefix_that_contains_the_separator():
with (
patch.object(
- mcp_module.global_mcp_server_manager,
+ mcp_operations.global_mcp_server_manager,
"_get_mcp_server_from_tool_name",
return_value=alias_less_server,
),
patch.object(
- mcp_module,
- "_handle_managed_mcp_tool",
+ mcp_operations, "_handle_managed_mcp_tool",
new=fake_handle_managed_mcp_tool,
),
patch.object(
@@ -7665,12 +7667,12 @@ async def test_execute_mcp_tool_strips_a_prefix_that_contains_the_separator():
return_value=True,
),
patch.object(
- mcp_module.global_mcp_tool_registry,
+ mcp_operations.global_mcp_tool_registry,
"get_tool",
return_value=None,
),
):
- await mcp_module.execute_mcp_tool(
+ await mcp_operations.execute_mcp_tool(
name=f"{server_id}-read_wiki_contents",
arguments={"repoName": "acme/wiki"},
allowed_mcp_servers=[alias_less_server],
@@ -7724,11 +7726,11 @@ async def test_execute_mcp_tool_rest_server_id_injects_requested_server_credenti
with (
patch.dict(
- mcp_module.global_mcp_server_manager.tool_name_to_mcp_server_name_mapping,
+ mcp_operations.global_mcp_server_manager.tool_name_to_mcp_server_name_mapping,
{"echo": collision_server.name, "echo_requested-echo": requested_server.name},
),
patch.object(
- mcp_module.global_mcp_server_manager,
+ mcp_operations.global_mcp_server_manager,
"get_registry",
return_value={
requested_server.server_id: requested_server,
@@ -7736,7 +7738,7 @@ async def test_execute_mcp_tool_rest_server_id_injects_requested_server_credenti
},
),
patch.object(
- mcp_module.global_mcp_server_manager,
+ mcp_operations.global_mcp_server_manager,
"_create_mcp_client",
new=fake_create_mcp_client,
),
@@ -7746,13 +7748,13 @@ async def test_execute_mcp_tool_rest_server_id_injects_requested_server_credenti
return_value=True,
),
patch.object(
- mcp_module.global_mcp_tool_registry,
+ mcp_operations.global_mcp_tool_registry,
"get_tool",
return_value=None,
),
patch("litellm.proxy.proxy_server.proxy_logging_obj", None),
):
- await mcp_module.execute_mcp_tool(
+ await mcp_operations.execute_mcp_tool(
name="echo",
arguments={"message": "hello"},
allowed_mcp_servers=[requested_server, collision_server],
@@ -7795,7 +7797,7 @@ async def test_execute_mcp_tool_rest_prefixed_tool_still_validates_server_id():
with (
patch.object(
- mcp_module.global_mcp_server_manager,
+ mcp_operations.global_mcp_server_manager,
"get_registry",
return_value={
api_key_server.server_id: api_key_server,
@@ -7803,7 +7805,7 @@ async def test_execute_mcp_tool_rest_prefixed_tool_still_validates_server_id():
},
),
patch.object(
- mcp_module.global_mcp_server_manager,
+ mcp_operations.global_mcp_server_manager,
"_get_mcp_server_from_tool_name",
return_value=oauth_server,
),
@@ -7813,13 +7815,13 @@ async def test_execute_mcp_tool_rest_prefixed_tool_still_validates_server_id():
return_value=True,
),
patch.object(
- mcp_module.global_mcp_tool_registry,
+ mcp_operations.global_mcp_tool_registry,
"get_tool",
return_value=None,
),
pytest.raises(HTTPException) as exc_info,
):
- await mcp_module.execute_mcp_tool(
+ await mcp_operations.execute_mcp_tool(
name="echo_oauth_m2m-echo",
arguments={"message": "hello"},
allowed_mcp_servers=[api_key_server, oauth_server],
@@ -7857,7 +7859,7 @@ async def test_execute_mcp_tool_rest_unauthorized_prefix_still_mismatches():
with (
patch.object(
- mcp_module.global_mcp_server_manager,
+ mcp_operations.global_mcp_server_manager,
"get_registry",
return_value={
api_key_server.server_id: api_key_server,
@@ -7865,7 +7867,7 @@ async def test_execute_mcp_tool_rest_unauthorized_prefix_still_mismatches():
},
),
patch.object(
- mcp_module.global_mcp_server_manager,
+ mcp_operations.global_mcp_server_manager,
"_get_mcp_server_from_tool_name",
return_value=restricted_server,
),
@@ -7875,13 +7877,13 @@ async def test_execute_mcp_tool_rest_unauthorized_prefix_still_mismatches():
return_value=True,
),
patch.object(
- mcp_module.global_mcp_tool_registry,
+ mcp_operations.global_mcp_tool_registry,
"get_tool",
return_value=None,
),
pytest.raises(HTTPException) as exc_info,
):
- await mcp_module.execute_mcp_tool(
+ await mcp_operations.execute_mcp_tool(
name="restricted_server-echo",
arguments={"message": "hello"},
allowed_mcp_servers=[api_key_server],
@@ -7921,18 +7923,17 @@ async def test_execute_mcp_tool_rest_hyphenated_upstream_tool_name_routes_to_req
with (
patch.object(
- mcp_module.global_mcp_server_manager,
+ mcp_operations.global_mcp_server_manager,
"get_registry",
return_value={api_key_server.server_id: api_key_server},
),
patch.object(
- mcp_module.global_mcp_server_manager,
+ mcp_operations.global_mcp_server_manager,
"_get_mcp_server_from_tool_name",
return_value=None,
),
patch.object(
- mcp_module,
- "_handle_managed_mcp_tool",
+ mcp_operations, "_handle_managed_mcp_tool",
new=fake_handle_managed_mcp_tool,
),
patch.object(
@@ -7941,12 +7942,12 @@ async def test_execute_mcp_tool_rest_hyphenated_upstream_tool_name_routes_to_req
return_value=True,
),
patch.object(
- mcp_module.global_mcp_tool_registry,
+ mcp_operations.global_mcp_tool_registry,
"get_tool",
return_value=None,
),
):
- await mcp_module.execute_mcp_tool(
+ await mcp_operations.execute_mcp_tool(
name="text-to-speech",
arguments={"message": "hello"},
allowed_mcp_servers=[api_key_server],
@@ -8005,22 +8006,22 @@ async def test_execute_mcp_tool_sets_model_in_model_call_details():
with (
patch.object(
- mcp_module.global_mcp_server_manager,
+ mcp_operations.global_mcp_server_manager,
"_get_mcp_server_from_tool_name",
return_value=fake_server,
),
patch.object(
- mcp_module.global_mcp_server_manager,
+ mcp_operations.global_mcp_server_manager,
"pre_call_tool_check",
new=AsyncMock(return_value={}),
),
patch.object(
- mcp_module.global_mcp_tool_registry,
+ mcp_operations.global_mcp_tool_registry,
"get_tool",
return_value=fake_tool,
),
patch(
- "litellm.proxy._experimental.mcp_server.server._handle_local_mcp_tool",
+ "litellm.proxy._experimental.mcp_server.operations._handle_local_mcp_tool",
new=AsyncMock(return_value=[]),
),
patch(
@@ -8028,7 +8029,7 @@ async def test_execute_mcp_tool_sets_model_in_model_call_details():
return_value=True,
),
):
- await mcp_module.execute_mcp_tool(
+ await mcp_operations.execute_mcp_tool(
name="list_pets",
arguments={"limit": 10},
allowed_mcp_servers=[fake_server],
@@ -8084,7 +8085,7 @@ async def test_execute_mcp_tool_rest_unresolved_prefixed_name_routes_to_requeste
with (
patch.object(
- mcp_module.global_mcp_server_manager,
+ mcp_operations.global_mcp_server_manager,
"get_registry",
return_value={
requested_server.server_id: requested_server,
@@ -8092,13 +8093,12 @@ async def test_execute_mcp_tool_rest_unresolved_prefixed_name_routes_to_requeste
},
),
patch.object(
- mcp_module.global_mcp_server_manager,
+ mcp_operations.global_mcp_server_manager,
"_get_mcp_server_from_tool_name",
return_value=None,
),
patch.object(
- mcp_module,
- "_handle_managed_mcp_tool",
+ mcp_operations, "_handle_managed_mcp_tool",
new=fake_handle_managed_mcp_tool,
),
patch.object(
@@ -8107,12 +8107,12 @@ async def test_execute_mcp_tool_rest_unresolved_prefixed_name_routes_to_requeste
return_value=True,
),
patch.object(
- mcp_module.global_mcp_tool_registry,
+ mcp_operations.global_mcp_tool_registry,
"get_tool",
return_value=None,
),
):
- await mcp_module.execute_mcp_tool(
+ await mcp_operations.execute_mcp_tool(
name="known_prefix-list_things",
arguments={"message": "hello"},
allowed_mcp_servers=[requested_server, prefix_owner],
@@ -8164,7 +8164,7 @@ async def test_execute_mcp_tool_rest_prefix_retry_resolution_still_enforces_serv
with (
patch.object(
- mcp_module.global_mcp_server_manager,
+ mcp_operations.global_mcp_server_manager,
"get_registry",
return_value={
requested_server.server_id: requested_server,
@@ -8172,7 +8172,7 @@ async def test_execute_mcp_tool_rest_prefix_retry_resolution_still_enforces_serv
},
),
patch.object(
- mcp_module.global_mcp_server_manager,
+ mcp_operations.global_mcp_server_manager,
"_get_mcp_server_from_tool_name",
side_effect=resolve_only_when_requested_prefix_added,
),
@@ -8182,13 +8182,13 @@ async def test_execute_mcp_tool_rest_prefix_retry_resolution_still_enforces_serv
return_value=True,
),
patch.object(
- mcp_module.global_mcp_tool_registry,
+ mcp_operations.global_mcp_tool_registry,
"get_tool",
return_value=None,
),
pytest.raises(HTTPException) as exc_info,
):
- await mcp_module.execute_mcp_tool(
+ await mcp_operations.execute_mcp_tool(
name="known_prefix-echo",
arguments={"message": "hello"},
allowed_mcp_servers=[requested_server, prefix_owner],
@@ -9091,14 +9091,14 @@ async def test_call_tool_with_legacy_db_m2m_server_resolves_oauth2_flow():
with (
patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager",
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager",
) as mock_manager,
patch(
- "litellm.proxy._experimental.mcp_server.server.execute_mcp_tool",
+ "litellm.proxy._experimental.mcp_server.operations.execute_mcp_tool",
side_effect=capture_execute,
),
patch(
- "litellm.proxy._experimental.mcp_server.server._get_allowed_mcp_servers_from_mcp_server_names",
+ "litellm.proxy._experimental.mcp_server.operations._get_allowed_mcp_servers_from_mcp_server_names",
new=AsyncMock(side_effect=lambda mcp_servers, allowed_mcp_servers: allowed_mcp_servers),
),
):
@@ -9176,12 +9176,12 @@ async def test_call_mcp_tool_skips_failure_hook_for_upstream_auth_error():
),
patch.object(global_mcp_server_manager, "get_mcp_server_by_id", return_value=mock_server),
patch(
- "litellm.proxy._experimental.mcp_server.server._get_allowed_mcp_servers_from_mcp_server_names",
+ "litellm.proxy._experimental.mcp_server.operations._get_allowed_mcp_servers_from_mcp_server_names",
new_callable=AsyncMock,
return_value=[mock_server],
),
patch(
- "litellm.proxy._experimental.mcp_server.server.execute_mcp_tool",
+ "litellm.proxy._experimental.mcp_server.operations.execute_mcp_tool",
new_callable=AsyncMock,
side_effect=MCPUpstreamAuthError(status_code=401, www_authenticate="Bearer", server_name="test_server"),
),
@@ -9261,7 +9261,7 @@ async def test_aggregate_listing_reports_per_server_outcomes():
mock_manager._get_tools_from_server = mock_get_tools_from_server
with patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager",
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager",
mock_manager,
):
listing = await _get_tools_from_mcp_servers(
@@ -9335,7 +9335,7 @@ async def test_handle_list_tools_attaches_outcome_meta(_mcp_request_ctx):
new=AsyncMock(return_value=(None, None, None, None, None, None, None)),
),
patch(
- "litellm.proxy._experimental.mcp_server.server._list_mcp_tools",
+ "litellm.proxy._experimental.mcp_server.operations._list_mcp_tools",
new=AsyncMock(return_value=listing),
),
):
@@ -9401,12 +9401,12 @@ class TestPreemptive401ModeAware:
with (
patch.object(
- server_module.global_mcp_server_manager,
+ mcp_operations.global_mcp_server_manager,
"get_mcp_server_by_name",
return_value=server,
),
patch.object(
- server_module.global_mcp_server_manager,
+ mcp_operations.global_mcp_server_manager,
"has_user_oauth_token",
new_callable=AsyncMock,
return_value=has_stored_token,
@@ -9425,7 +9425,7 @@ class TestPreemptive401ModeAware:
async def test_deferred_discovery_runs_before_delegate_challenge(self):
from litellm.proxy._experimental.mcp_server import server as server_module
- manager = server_module.global_mcp_server_manager
+ manager = mcp_operations.global_mcp_server_manager
server = _make_oauth2_server(
"lazy_delegate",
oauth2_flow="authorization_code",
@@ -9457,7 +9457,7 @@ class TestPreemptive401ModeAware:
async def test_stamped_m2m_challenge_skips_deferred_discovery(self):
from litellm.proxy._experimental.mcp_server import server as server_module
- manager = server_module.global_mcp_server_manager
+ manager = mcp_operations.global_mcp_server_manager
server = _make_oauth2_server("stamped_m2m", oauth2_flow="client_credentials")
with patch.object(
@@ -9500,12 +9500,12 @@ class TestPreemptive401ModeAware:
with (
patch.dict(os.environ, {"SERVER_ROOT_PATH": "/litellm"}),
patch.object(
- server_module.global_mcp_server_manager,
+ mcp_operations.global_mcp_server_manager,
"get_mcp_server_by_name",
return_value=server,
),
patch.object(
- server_module.global_mcp_server_manager,
+ mcp_operations.global_mcp_server_manager,
"has_user_oauth_token",
new_callable=AsyncMock,
return_value=False,
@@ -9607,17 +9607,17 @@ class TestSingleServerPreflightReachesIdJag:
with (
patch.object( # test-quality-ok: route wiring must use the manager's configured server
- server_module.global_mcp_server_manager,
+ mcp_operations.global_mcp_server_manager,
"get_mcp_server_by_name",
return_value=server,
),
patch.object( # test-quality-ok: route wiring must invoke the manager preflight
- server_module.global_mcp_server_manager,
+ mcp_operations.global_mcp_server_manager,
"preflight_token_exchange",
preflight,
),
patch.object( # test-quality-ok: allowed-set resolution needs the DB; the test controls its answer
- server_module, "_get_allowed_mcp_servers", AsyncMock(return_value=[server])
+ mcp_operations, "_get_allowed_mcp_servers", AsyncMock(return_value=[server])
),
):
await server_module._raise_preemptive_401_for_unauthenticated_servers(
@@ -9667,12 +9667,12 @@ class TestSingleServerPreflightReachesIdJag:
with (
patch.object( # test-quality-ok: route wiring must use the manager's configured server
- server_module.global_mcp_server_manager,
+ mcp_operations.global_mcp_server_manager,
"get_mcp_server_by_name",
return_value=token_exchange,
),
patch.object( # test-quality-ok: route wiring must invoke the manager preflight
- server_module.global_mcp_server_manager,
+ mcp_operations.global_mcp_server_manager,
"preflight_token_exchange",
preflight,
),
@@ -9733,13 +9733,13 @@ class TestOboPreflightScopedToAllowedServers:
preflight = AsyncMock()
with (
patch.object( # test-quality-ok: route handler reads the module-level manager, no injection seam
- server_module.global_mcp_server_manager, "get_mcp_server_by_name", return_value=requested
+ mcp_operations.global_mcp_server_manager, "get_mcp_server_by_name", return_value=requested
),
patch.object( # test-quality-ok: the exchanger is the observable; a real one would call an IdP
- server_module.global_mcp_server_manager, "preflight_token_exchange", preflight
+ mcp_operations.global_mcp_server_manager, "preflight_token_exchange", preflight
),
patch.object( # test-quality-ok: allowed-set resolution needs the DB; the test controls its answer
- server_module, "_get_allowed_mcp_servers", allowed_lookup
+ mcp_operations, "_get_allowed_mcp_servers", allowed_lookup
),
):
await server_module._raise_preemptive_401_for_unauthenticated_servers(
@@ -10048,7 +10048,7 @@ class TestListFiltersHonorThePrefixBoundary:
with (
patch.object(MCPRequestHandler, "get_allowed_tools_for_server", AsyncMock(return_value=grants)),
- patch("litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager") as mock_manager,
+ patch("litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager") as mock_manager,
):
mock_manager.get_mcp_server_by_id.return_value = server
@@ -10111,11 +10111,11 @@ async def test_list_tools_injects_byok_credential_for_non_oauth2_auth_types(auth
with (
patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager",
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager",
mock_manager,
),
patch(
- "litellm.proxy._experimental.mcp_server.server._get_byok_credential",
+ "litellm.proxy._experimental.mcp_server.operations._get_byok_credential",
AsyncMock(return_value="personal-api-key"),
),
):
@@ -10208,3 +10208,44 @@ async def test_streamable_http_rejects_modern_protocol_version(header_value: str
assert header_value in body["error"]["message"]
for version in body["error"]["message"].split("supported: ")[1].split(", "):
assert version in HANDSHAKE_PROTOCOL_VERSIONS
+
+
+@pytest.mark.asyncio
+@pytest.mark.parametrize("handler_name,field", [
+ ("handle_list_tools", "tools"),
+ ("list_prompts", "prompts"),
+ ("list_resources", "resources"),
+ ("list_resource_templates", "resource_templates"),
+])
+async def test_native_listing_preserves_empty_result_on_auth_failure(_mcp_request_ctx, handler_name, field):
+ from litellm.proxy._experimental.mcp_server import server
+
+ with patch.object(server, "get_or_extract_auth_context", AsyncMock(side_effect=RuntimeError("auth failure"))):
+ result = await getattr(server, handler_name)(_mcp_request_ctx(), _paged_params())
+ assert getattr(result, field) == []
+
+
+@pytest.mark.asyncio
+@pytest.mark.parametrize("failure_hook_raises", [False, True])
+async def test_tool_listing_preserves_permission_denial_when_failure_logging_fails(failure_hook_raises):
+ from litellm.proxy._experimental.mcp_server import operations
+ from litellm.proxy import proxy_server
+
+ auth = UserAPIKeyAuth(user_id="denied-caller")
+ denial = HTTPException(status_code=403, detail="scope denied")
+ logger = MagicMock()
+ logger.post_call_failure_hook = AsyncMock(side_effect=RuntimeError("log unavailable") if failure_hook_raises else None)
+ upstream = AsyncMock()
+ with (
+ patch.object(operations, "_get_allowed_mcp_servers", AsyncMock(side_effect=denial)),
+ patch.object(operations, "function_setup", return_value=(None, None)),
+ patch.object(proxy_server, "proxy_logging_obj", logger),
+ patch.object(operations.global_mcp_server_manager, "_get_tools_from_server", upstream),
+ ):
+ with pytest.raises(HTTPException) as rejected:
+ await operations._get_tools_from_mcp_servers(user_api_key_auth=auth, mcp_auth_header=None, mcp_servers=["catalog"], log_list_tools_to_spendlogs=True)
+ assert rejected.value is denial
+ upstream.assert_not_awaited()
+ logger.post_call_failure_hook.assert_awaited_once()
+ assert logger.post_call_failure_hook.await_args.kwargs["original_exception"] is denial
+ assert logger.post_call_failure_hook.await_args.kwargs["user_api_key_dict"] == auth
diff --git a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_server_manager.py b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_server_manager.py
index 9140ac61f1a..7be79f9b514 100644
--- a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_server_manager.py
+++ b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_server_manager.py
@@ -71,6 +71,135 @@ from litellm.proxy.utils import ProxyLogging
from litellm.types.guardrails import GuardrailEventHooks
+@pytest.mark.asyncio
+async def test_manager_sampling_preserves_explicit_headers_without_ambient_context():
+ from litellm.proxy._experimental.mcp_server import server as legacy_server
+
+ caller = UserAPIKeyAuth(user_id="sampling-caller")
+ upstream = MCPServer(
+ server_id="sampling-context",
+ name="sampling_context",
+ url="https://example.invalid/mcp",
+ transport=MCPTransport.http,
+ allow_sampling=True,
+ )
+ sampling = AsyncMock()
+ client = MagicMock()
+ client.call_tool = AsyncMock(return_value=CallToolResult(content=[]))
+ assert legacy_server.get_active_auth_context() is None
+ with (
+ patch("litellm.proxy._experimental.mcp_server.mcp_server_manager.MCPClient", return_value=client) as factory,
+ patch("litellm.proxy._experimental.mcp_server.sampling_handler.handle_sampling_create_message", sampling),
+ ):
+ await MCPServerManager()._call_regular_mcp_tool(
+ mcp_server=upstream,
+ original_tool_name="probe",
+ arguments={},
+ tasks=[],
+ mcp_auth_header=None,
+ mcp_server_auth_headers=None,
+ oauth2_headers=None,
+ raw_headers={"x-test-caller": "sampling-caller"},
+ proxy_logging_obj=None,
+ user_api_key_auth=caller,
+ )
+ callback = factory.call_args.kwargs["sampling_callback"]
+ await callback(None, None)
+ assert sampling.await_args.kwargs["user_api_key_auth"].user_id == "sampling-caller"
+ assert sampling.await_args.kwargs["raw_headers"] == {"x-test-caller": "sampling-caller"}
+
+
+
+@pytest.mark.asyncio
+async def test_sampling_callback_keeps_creation_context_after_caller_switch():
+ from mcp.server.auth.middleware.auth_context import auth_context_var
+
+ from litellm.proxy._experimental.mcp_server import server as legacy_server
+ from litellm.proxy._experimental.mcp_server.mcp_server_manager import _create_sampling_callback
+
+ token = auth_context_var.set(None)
+ recorder = AsyncMock()
+ try:
+ original = UserAPIKeyAuth(user_id="alpha", models=["alpha-model"])
+ original.mcp_admitted_user_subject = True
+ headers = {"x-caller": "alpha"}
+ legacy_server.set_auth_context(original, raw_headers=headers, client_ip="192.0.2.1")
+ callback = _create_sampling_callback()
+ original.models.append("bravo-model")
+ headers["x-caller"] = "bravo"
+ legacy_server.set_auth_context(UserAPIKeyAuth(user_id="bravo"), raw_headers={"x-caller": "bravo"})
+ with patch("litellm.proxy._experimental.mcp_server.sampling_handler.handle_sampling_create_message", recorder):
+ await callback(None, None)
+ observed = recorder.await_args.kwargs
+ assert observed["user_api_key_auth"].user_id == "alpha"
+ assert observed["user_api_key_auth"].models == ["alpha-model"]
+ assert observed["user_api_key_auth"].mcp_admitted_user_subject is True
+ assert observed["raw_headers"] == {"x-caller": "alpha"}
+ assert observed["client_ip"] == "192.0.2.1"
+ finally:
+ auth_context_var.reset(token)
+
+
+@pytest.mark.asyncio
+async def test_elicitation_callback_keeps_initiating_session():
+ from litellm.proxy._experimental.mcp_server import server as legacy_server
+ from litellm.proxy._experimental.mcp_server.mcp_server_manager import _create_elicitation_callback
+
+ initiating = MagicMock()
+ replacement = MagicMock()
+ recorder = AsyncMock()
+ token = legacy_server.active_mcp_session_var.set(initiating)
+ try:
+ callback = _create_elicitation_callback()
+ legacy_server.active_mcp_session_var.set(replacement)
+ with patch("litellm.proxy._experimental.mcp_server.elicitation_handler.handle_elicitation_request", recorder):
+ await callback(None, None)
+ assert recorder.await_args.kwargs["downstream_session"] is initiating
+ assert recorder.await_args.kwargs["downstream_capabilities"] is initiating.capabilities
+ finally:
+ legacy_server.active_mcp_session_var.reset(token)
+
+
+@pytest.mark.asyncio
+async def test_sampling_callbacks_isolate_callers_and_cancellation():
+ from mcp.types import ErrorData
+
+ from litellm.proxy._experimental.mcp_server.mcp_server_manager import _create_sampling_callback
+
+ started = asyncio.Event()
+ cancelled = asyncio.Event()
+ observed = {}
+
+ async def record_sampling(*, user_api_key_auth, raw_headers, **kwargs):
+ label = user_api_key_auth.user_id
+ if label == "cancelled":
+ started.set()
+ try:
+ await asyncio.Event().wait()
+ except asyncio.CancelledError:
+ cancelled.set()
+ raise
+ await asyncio.sleep(0)
+ observed[label] = raw_headers["x-caller"]
+ return ErrorData(code=-1, message=label)
+
+ callbacks = tuple(
+ _create_sampling_callback(UserAPIKeyAuth(user_id=label), raw_headers={"x-caller": label})
+ for label in ("alpha", "bravo", "cancelled")
+ )
+ with patch(
+ "litellm.proxy._experimental.mcp_server.sampling_handler.handle_sampling_create_message", record_sampling
+ ):
+ tasks = tuple(asyncio.create_task(callback(None, None)) for callback in callbacks)
+ await asyncio.wait_for(started.wait(), timeout=2)
+ tasks[2].cancel()
+ results = await asyncio.gather(*tasks, return_exceptions=True)
+ assert observed == {"alpha": "alpha", "bravo": "bravo"}
+ assert [result.message for result in results[:2]] == ["alpha", "bravo"]
+ assert isinstance(results[2], asyncio.CancelledError)
+ assert cancelled.is_set()
+
+
def _reload_mcp_manager_module():
utils_module = sys.modules["litellm.proxy._experimental.mcp_server.utils"]
manager_module = sys.modules["litellm.proxy._experimental.mcp_server.mcp_server_manager"]
@@ -82,6 +211,9 @@ def _reload_mcp_manager_module():
server_module = sys.modules.get("litellm.proxy._experimental.mcp_server.server")
if server_module is not None and hasattr(server_module, "global_mcp_server_manager"):
server_module.global_mcp_server_manager = reloaded.global_mcp_server_manager
+ operations_module = sys.modules.get("litellm.proxy._experimental.mcp_server.operations")
+ if operations_module is not None:
+ operations_module.global_mcp_server_manager = reloaded.global_mcp_server_manager
return reloaded
@@ -3921,6 +4053,7 @@ class TestMCPServerManager:
result = await manager.get_resource_templates_from_server(
server=server,
user_api_key_auth=None,
+ raw_headers=None,
mcp_auth_header="auth",
extra_headers=None,
add_prefix=False,
@@ -3933,6 +4066,8 @@ class TestMCPServerManager:
stdio_env=None,
subject_token=None,
user_api_key_auth=None,
+ raw_headers=None,
+ client_ip=None,
)
mock_client.list_resource_templates.assert_awaited_once()
assert result == expected_templates
@@ -5847,7 +5982,7 @@ class TestMCPServerManager:
stored = {"Authorization": "Bearer stored-user-token"}
with patch(
- "litellm.proxy._experimental.mcp_server.server._get_user_oauth_extra_headers_from_db",
+ "litellm.proxy._experimental.mcp_server.operations._get_user_oauth_extra_headers_from_db",
new=AsyncMock(return_value=stored),
) as mock_lookup:
result = await manager._resolve_oauth2_headers_for_tool_call(
@@ -5874,7 +6009,7 @@ class TestMCPServerManager:
user_auth = UserAPIKeyAuth(api_key="sk-test", user_id="alice")
with patch(
- "litellm.proxy._experimental.mcp_server.server._get_user_oauth_extra_headers_from_db",
+ "litellm.proxy._experimental.mcp_server.operations._get_user_oauth_extra_headers_from_db",
new=AsyncMock(return_value={"Authorization": "Bearer should-not-be-used"}),
) as mock_lookup:
result = await manager._resolve_oauth2_headers_for_tool_call(
@@ -5900,7 +6035,7 @@ class TestMCPServerManager:
user_auth = UserAPIKeyAuth(api_key="sk-test", user_id="alice")
with patch(
- "litellm.proxy._experimental.mcp_server.server._get_user_oauth_extra_headers_from_db",
+ "litellm.proxy._experimental.mcp_server.operations._get_user_oauth_extra_headers_from_db",
new=AsyncMock(side_effect=RuntimeError("redis down")),
):
result = await manager._resolve_oauth2_headers_for_tool_call(
@@ -6056,7 +6191,7 @@ class TestMCPServerManager:
user_auth = UserAPIKeyAuth(api_key="sk-test")
with patch(
- "litellm.proxy._experimental.mcp_server.server._get_user_oauth_extra_headers_from_db",
+ "litellm.proxy._experimental.mcp_server.operations._get_user_oauth_extra_headers_from_db",
new=AsyncMock(return_value={"Authorization": "Bearer x"}),
) as mock_lookup:
result = await manager._resolve_oauth2_headers_for_tool_call(
@@ -6860,7 +6995,8 @@ class TestMCPServerManager:
}
user_api_key_auth = UserAPIKeyAuth(api_key="sk-test", user_id="user-123")
- token = _mcp_active_toolset_id.set("toolset-abc")
+ user_api_key_auth.mcp_toolset_id = "toolset-abc"
+ token = _mcp_active_toolset_id.set("unrelated-ambient-toolset")
try:
with (
patch.object(proxy_server_module, "user_api_key_cache", cache),
@@ -14075,3 +14211,31 @@ async def test_request_selected_during_guardrail_runs_concurrently_with_tool(mon
assert guardrail_started.is_set() is selected
assert result.is_error is False
assert result.content[0].text == "executed"
+
+
+@pytest.mark.asyncio
+@pytest.mark.parametrize("with_caller", [True, False])
+async def test_client_sampling_does_not_fill_explicit_context_from_another_ambient_caller(with_caller):
+ from mcp.server.auth.middleware.auth_context import auth_context_var
+ from litellm.proxy._experimental.mcp_server import server as legacy_server
+
+ upstream = MCPServer(server_id="explicit-empty", name="explicit_empty", url="https://example.invalid/mcp", transport=MCPTransport.http, allow_sampling=True)
+ token = auth_context_var.set(None)
+ sampling = AsyncMock()
+ try:
+ legacy_server.set_auth_context(UserAPIKeyAuth(user_id="unrelated"), raw_headers={"authorization": "unrelated-credential"}, client_ip="192.0.2.99")
+ with (
+ patch("litellm.proxy._experimental.mcp_server.mcp_server_manager.MCPClient") as factory,
+ patch("litellm.proxy._experimental.mcp_server.sampling_handler.handle_sampling_create_message", sampling),
+ ):
+ await MCPServerManager()._create_mcp_client(upstream, user_api_key_auth=UserAPIKeyAuth(user_id="explicit") if with_caller else None)
+ await factory.call_args.kwargs["sampling_callback"](None, None)
+ captured = sampling.await_args.kwargs
+ if with_caller:
+ assert captured["user_api_key_auth"].user_id == "explicit"
+ else:
+ assert captured["user_api_key_auth"] is None
+ assert captured["raw_headers"] is None
+ assert captured["client_ip"] is None
+ finally:
+ auth_context_var.reset(token)
diff --git a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_stale_session.py b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_stale_session.py
index 9420eecd222..ec6fdef69ee 100644
--- a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_stale_session.py
+++ b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_stale_session.py
@@ -639,12 +639,12 @@ async def test_per_user_oauth_missing_stored_token_returns_preemptive_401():
return_value=False,
),
patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager.has_user_oauth_token",
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager.has_user_oauth_token",
new_callable=AsyncMock,
return_value=False,
) as mock_has_token,
patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager.get_mcp_server_by_name",
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager.get_mcp_server_by_name",
return_value=oauth_server,
),
patch.object(
@@ -727,12 +727,12 @@ async def test_admitted_subject_missing_stored_token_challenged_with_resource_me
return_value=False,
),
patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager.has_user_oauth_token",
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager.has_user_oauth_token",
new_callable=AsyncMock,
return_value=False,
) as mock_has_token,
patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager.get_mcp_server_by_name",
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager.get_mcp_server_by_name",
return_value=oauth_server,
),
patch.object(
@@ -833,11 +833,11 @@ async def test_client_credentials_server_is_not_preemptively_challenged(m2m_fiel
return_value=False,
),
patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager.has_user_oauth_token",
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager.has_user_oauth_token",
new_callable=AsyncMock,
) as mock_has_token,
patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager.get_mcp_server_by_name",
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager.get_mcp_server_by_name",
return_value=m2m_server,
),
patch.object(session_manager_stateless, "handle_request", new_callable=AsyncMock) as mock_handle_request,
@@ -929,16 +929,16 @@ async def test_handle_streamable_http_mcp_delegated_server_surfaces_upstream_cha
return_value=False,
),
patch(
- "litellm.proxy._experimental.mcp_server.server._get_user_oauth_extra_headers_from_db",
+ "litellm.proxy._experimental.mcp_server.operations._get_user_oauth_extra_headers_from_db",
new_callable=AsyncMock,
return_value=None,
),
patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager.get_mcp_server_by_name",
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager.get_mcp_server_by_name",
return_value=delegated_server,
),
patch( # test-quality-ok: registry is empty in unit tests; key owns the delegated server
- "litellm.proxy._experimental.mcp_server.server._get_allowed_mcp_servers",
+ "litellm.proxy._experimental.mcp_server.operations._get_allowed_mcp_servers",
new_callable=AsyncMock,
return_value=[delegated_server],
),
@@ -1022,12 +1022,12 @@ async def test_per_user_oauth_with_stored_token_skips_preemptive_401():
return_value=False,
),
patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager.has_user_oauth_token",
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager.has_user_oauth_token",
new_callable=AsyncMock,
return_value=True,
) as mock_has_token,
patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager.get_mcp_server_by_name",
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager.get_mcp_server_by_name",
return_value=oauth_server,
),
patch.object(
@@ -1126,11 +1126,11 @@ async def test_handle_streamable_http_mcp_delegated_server_without_token_returns
return_value=False,
),
patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager.has_user_oauth_token",
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager.has_user_oauth_token",
new_callable=AsyncMock,
) as mock_has_token,
patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager.get_mcp_server_by_name",
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager.get_mcp_server_by_name",
return_value=delegated_server,
),
patch.object(
@@ -1218,7 +1218,7 @@ async def test_handle_streamable_http_mcp_token_exchange_without_subject_returns
return_value=False,
),
patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager.get_mcp_server_by_name",
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager.get_mcp_server_by_name",
return_value=obo_server,
),
patch.object(
@@ -1317,7 +1317,7 @@ async def test_handle_streamable_http_mcp_oauth_delegate_without_token_returns_g
True,
),
patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager.get_mcp_server_by_name",
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager.get_mcp_server_by_name",
return_value=od_server,
),
patch.object(
@@ -1391,7 +1391,7 @@ async def test_handle_streamable_http_mcp_oauth_delegate_with_forwarded_token_sk
new_callable=AsyncMock,
),
patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager.get_mcp_server_by_name",
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager.get_mcp_server_by_name",
return_value=od_server,
),
patch.object(
@@ -1453,7 +1453,7 @@ async def _run_passthrough_connect(
new_callable=AsyncMock,
),
patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager.get_mcp_server_by_name",
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager.get_mcp_server_by_name",
return_value=server,
),
patch.object(session_manager_stateless, "handle_request", new_callable=AsyncMock) as mock_handle_request,
@@ -1574,7 +1574,7 @@ async def test_handle_streamable_http_mcp_true_passthrough_without_token_surface
return_value=probe_client,
),
patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager.get_mcp_server_by_name",
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager.get_mcp_server_by_name",
return_value=tp_server,
),
patch.object(
@@ -1642,7 +1642,7 @@ async def test_handle_streamable_http_mcp_true_passthrough_dcr_bridge_challenges
return_value=probe_client,
),
patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager.get_mcp_server_by_name",
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager.get_mcp_server_by_name",
return_value=bridge_server,
),
patch.object(
@@ -1720,7 +1720,7 @@ async def test_handle_streamable_http_mcp_true_passthrough_with_token_skips_prob
return_value=probe_client,
),
patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager.get_mcp_server_by_name",
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager.get_mcp_server_by_name",
return_value=tp_server,
),
patch.object(
diff --git a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_tool_search.py b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_tool_search.py
index cb43d2c2592..4575741aa8b 100644
--- a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_tool_search.py
+++ b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_tool_search.py
@@ -1,3 +1,4 @@
+from litellm.proxy._experimental.mcp_server import operations as mcp_operations
"""
Tests for MCP tool search feature.
@@ -572,7 +573,7 @@ class TestCallToolRestApiVirtualTools:
mock_tool.input_schema = {"type": "object", "properties": {}}
with patch(
- "litellm.proxy._experimental.mcp_server.server._list_mcp_tools",
+ "litellm.proxy._experimental.mcp_server.operations._list_mcp_tools",
new_callable=AsyncMock,
return_value=AggregateToolListing(tools=[mock_tool], outcomes={}),
):
@@ -616,12 +617,12 @@ class TestCallToolRestApiVirtualTools:
with (
patch(
- "litellm.proxy._experimental.mcp_server.server._get_allowed_mcp_servers",
+ "litellm.proxy._experimental.mcp_server.operations._get_allowed_mcp_servers",
new_callable=AsyncMock,
return_value=[MagicMock()],
),
patch(
- "litellm.proxy._experimental.mcp_server.server.execute_mcp_tool",
+ "litellm.proxy._experimental.mcp_server.operations.execute_mcp_tool",
new_callable=AsyncMock,
return_value=fake_result,
) as mock_execute,
@@ -669,12 +670,12 @@ class TestCallToolRestApiVirtualTools:
return_value="203.0.113.7",
),
patch(
- "litellm.proxy._experimental.mcp_server.server._get_allowed_mcp_servers",
+ "litellm.proxy._experimental.mcp_server.operations._get_allowed_mcp_servers",
new_callable=AsyncMock,
return_value=[MagicMock()],
) as mock_allowed,
patch(
- "litellm.proxy._experimental.mcp_server.server.execute_mcp_tool",
+ "litellm.proxy._experimental.mcp_server.operations.execute_mcp_tool",
new_callable=AsyncMock,
return_value=fake_result,
),
@@ -699,7 +700,7 @@ class TestCallToolRestApiVirtualTools:
return_value="203.0.113.7",
),
patch(
- "litellm.proxy._experimental.mcp_server.server._list_mcp_tools",
+ "litellm.proxy._experimental.mcp_server.operations._list_mcp_tools",
new_callable=AsyncMock,
return_value=AggregateToolListing(tools=[], outcomes={}),
) as mock_list,
@@ -832,7 +833,7 @@ class TestCallToolRestApiVirtualTools:
"litellm.proxy.proxy_server.proxy_logging_obj", key_limits
),
patch( # test-quality-ok: the authorized catalog is the seam every virtual tool shares; the ranking under test stays real
- "litellm.proxy._experimental.mcp_server.server._list_mcp_tools",
+ "litellm.proxy._experimental.mcp_server.operations._list_mcp_tools",
new_callable=AsyncMock,
return_value=AggregateToolListing(tools=list(CATALOG), outcomes={}),
) as mock_list,
@@ -939,7 +940,7 @@ class TestDispatchVirtualMcpTool:
new_callable=AsyncMock,
return_value="SEARCH_RESULT",
) as mock_search:
- result = await srv._dispatch_virtual_mcp_tool(
+ result = await mcp_operations._dispatch_virtual_mcp_tool(
name=MCP_TOOL_SEARCH_TOOL_NAME,
arguments={"query": "q", "top_k": 3},
user_api_key_auth=uak,
@@ -961,7 +962,7 @@ class TestDispatchVirtualMcpTool:
new_callable=AsyncMock,
return_value="AGENT_RESULT",
) as mock_agent_search:
- result = await srv._dispatch_virtual_mcp_tool(
+ result = await mcp_operations._dispatch_virtual_mcp_tool(
name=AGENT_SEARCH_TOOL_NAME,
arguments={"query": "translate a document", "top_k": "2"},
user_api_key_auth=uak,
@@ -996,7 +997,7 @@ class TestDispatchVirtualMcpTool:
new_callable=AsyncMock,
return_value="CALL_RESULT",
) as mock_call:
- result = await srv._dispatch_virtual_mcp_tool(
+ result = await mcp_operations._dispatch_virtual_mcp_tool(
name=MCP_TOOL_CALL_TOOL_NAME,
arguments={"tool_name": "math-add", "arguments": {"a": 1, "b": 2}},
user_api_key_auth=uak,
@@ -1027,8 +1028,7 @@ class TestDispatchVirtualMcpTool:
sentinel_logging_obj = object()
with (
patch.object(
- srv,
- "_build_virtual_call_logging_obj",
+ mcp_operations, "_build_virtual_call_logging_obj",
new_callable=AsyncMock,
return_value=sentinel_logging_obj,
) as mock_build,
@@ -1038,7 +1038,7 @@ class TestDispatchVirtualMcpTool:
return_value="CALL_RESULT",
) as mock_call,
):
- await srv._dispatch_virtual_mcp_tool(
+ await mcp_operations._dispatch_virtual_mcp_tool(
name=MCP_TOOL_CALL_TOOL_NAME,
arguments={"tool_name": "math-add", "arguments": {"a": 1}},
user_api_key_auth=uak,
@@ -1060,7 +1060,7 @@ class TestDispatchVirtualMcpTool:
new_callable=AsyncMock,
return_value="SEARCH_RESULT",
) as mock_search:
- await srv._dispatch_virtual_mcp_tool(
+ await mcp_operations._dispatch_virtual_mcp_tool(
name=MCP_TOOL_SEARCH_TOOL_NAME,
arguments={"query": "issue", "top_k": "not-a-number"},
user_api_key_auth=uak,
@@ -1083,12 +1083,12 @@ class TestDispatchVirtualMcpTool:
fake = CallToolResult(content=[TextContent(type="text", text="ok")], isError=False)
with (
patch(
- "litellm.proxy._experimental.mcp_server.server._get_allowed_mcp_servers",
+ "litellm.proxy._experimental.mcp_server.operations._get_allowed_mcp_servers",
new_callable=AsyncMock,
return_value=[MagicMock()],
) as mock_allowed,
patch(
- "litellm.proxy._experimental.mcp_server.server.execute_mcp_tool",
+ "litellm.proxy._experimental.mcp_server.operations.execute_mcp_tool",
new_callable=AsyncMock,
return_value=fake,
) as mock_exec,
@@ -1130,12 +1130,12 @@ class TestDispatchVirtualMcpTool:
uak = UserAPIKeyAuth(api_key="k", object_permission=_make_perm(mcp_tool_search_enabled=True))
with (
patch(
- "litellm.proxy._experimental.mcp_server.server._get_allowed_mcp_servers",
+ "litellm.proxy._experimental.mcp_server.operations._get_allowed_mcp_servers",
new_callable=AsyncMock,
return_value=[],
),
patch(
- "litellm.proxy._experimental.mcp_server.server.execute_mcp_tool",
+ "litellm.proxy._experimental.mcp_server.operations.execute_mcp_tool",
new_callable=AsyncMock,
) as mock_exec,
):
@@ -1217,7 +1217,7 @@ class TestMcpServerToolCallErrorHandling:
return_value=(uak, None, None, None, None, None, None),
),
patch(
- "litellm.proxy._experimental.mcp_server.server._dispatch_virtual_mcp_tool",
+ "litellm.proxy._experimental.mcp_server.operations._dispatch_virtual_mcp_tool",
new_callable=AsyncMock,
side_effect=HTTPException(status_code=403, detail="User not allowed to call this tool"),
),
@@ -1254,7 +1254,7 @@ async def test_handle_mcp_tool_call_scoped_denial_names_the_binding_agent() -> N
]
with patch( # test-quality-ok: the permission resolver is a module-level function; the suite's only seam
- "litellm.proxy._experimental.mcp_server.server._get_allowed_mcp_servers",
+ "litellm.proxy._experimental.mcp_server.operations._get_allowed_mcp_servers",
new=AsyncMock(side_effect=resolve),
):
with pytest.raises(HTTPException) as exc_info:
diff --git a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_toolset_scope.py b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_toolset_scope.py
index 519acc241c6..60098b1f656 100644
--- a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_toolset_scope.py
+++ b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_toolset_scope.py
@@ -564,7 +564,7 @@ class TestMCPActiveToolsetContextVar:
MagicMock(get_mcp_client_ip=MagicMock(return_value="127.0.0.1")),
),
patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager",
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager",
MagicMock(get_mcp_server_by_name=MagicMock(return_value=None)),
),
patch(
diff --git a/tests/test_litellm/proxy/_experimental/mcp_server/test_openapi_tool_auth.py b/tests/test_litellm/proxy/_experimental/mcp_server/test_openapi_tool_auth.py
index ac716bace3c..bb70f38285c 100644
--- a/tests/test_litellm/proxy/_experimental/mcp_server/test_openapi_tool_auth.py
+++ b/tests/test_litellm/proxy/_experimental/mcp_server/test_openapi_tool_auth.py
@@ -1,3 +1,4 @@
+from litellm.proxy._experimental.mcp_server import operations as mcp_operations
"""
VERIA-7 regression: OpenAPI-backed (local-registry) MCP tools must run
through `pre_call_tool_check` before dispatch, the same as managed
@@ -49,22 +50,22 @@ async def test_openapi_local_tool_runs_pre_call_tool_check():
with (
patch.object(
- mcp_module.global_mcp_server_manager,
+ mcp_operations.global_mcp_server_manager,
"_get_mcp_server_from_tool_name",
return_value=fake_server,
),
patch.object(
- mcp_module.global_mcp_server_manager,
+ mcp_operations.global_mcp_server_manager,
"pre_call_tool_check",
new=pre_call,
),
patch.object(
- mcp_module.global_mcp_tool_registry,
+ mcp_operations.global_mcp_tool_registry,
"get_tool",
return_value=fake_tool,
),
patch(
- "litellm.proxy._experimental.mcp_server.server._handle_local_mcp_tool",
+ "litellm.proxy._experimental.mcp_server.operations._handle_local_mcp_tool",
new=handle_local,
),
patch(
@@ -72,7 +73,7 @@ async def test_openapi_local_tool_runs_pre_call_tool_check():
return_value=True,
),
):
- await mcp_module.execute_mcp_tool(
+ await mcp_operations.execute_mcp_tool(
name="list_pets",
arguments={"limit": 10},
allowed_mcp_servers=[fake_server],
@@ -92,7 +93,7 @@ async def test_openapi_local_tool_runs_pre_call_tool_check():
assert pre_call_kwargs["guardrail_context"] == {"metadata": {"guardrails": ("block-all",)}}
assert pre_call_kwargs["name"] == "list_pets"
assert pre_call_kwargs["server"] is fake_server
- assert pre_call_kwargs["user_api_key_auth"] is user
+ assert pre_call_kwargs["user_api_key_auth"] == user
# `proxy_logging_obj` must be sourced from the canonical proxy_server
# module (same as the managed path) — passing None would crash the
# downstream `_create_mcp_request_object_from_kwargs` call with
@@ -134,22 +135,22 @@ async def test_openapi_local_tool_blocked_when_pre_call_check_raises():
with (
patch.object(
- mcp_module.global_mcp_server_manager,
+ mcp_operations.global_mcp_server_manager,
"_get_mcp_server_from_tool_name",
return_value=fake_server,
),
patch.object(
- mcp_module.global_mcp_server_manager,
+ mcp_operations.global_mcp_server_manager,
"pre_call_tool_check",
new=pre_call,
),
patch.object(
- mcp_module.global_mcp_tool_registry,
+ mcp_operations.global_mcp_tool_registry,
"get_tool",
return_value=fake_tool,
),
patch(
- "litellm.proxy._experimental.mcp_server.server._handle_local_mcp_tool",
+ "litellm.proxy._experimental.mcp_server.operations._handle_local_mcp_tool",
new=handle_local,
),
patch(
@@ -158,7 +159,7 @@ async def test_openapi_local_tool_blocked_when_pre_call_check_raises():
),
):
with pytest.raises(HTTPException) as exc:
- await mcp_module.execute_mcp_tool(
+ await mcp_operations.execute_mcp_tool(
name="delete_pet",
arguments={},
allowed_mcp_servers=[fake_server],
@@ -195,24 +196,24 @@ async def test_openapi_local_tool_denied_when_server_not_resolvable():
# `_get_mcp_server_from_tool_name` returns None — no server context.
with (
- patch.object(mcp_module, "_resolve_openapi_tool_auth", new=resolve_auth),
+ patch.object(mcp_operations, "_resolve_openapi_tool_auth", new=resolve_auth),
patch.object(
- mcp_module.global_mcp_server_manager,
+ mcp_operations.global_mcp_server_manager,
"_get_mcp_server_from_tool_name",
return_value=None,
),
patch.object(
- mcp_module.global_mcp_server_manager,
+ mcp_operations.global_mcp_server_manager,
"pre_call_tool_check",
new=pre_call,
),
patch.object(
- mcp_module.global_mcp_tool_registry,
+ mcp_operations.global_mcp_tool_registry,
"get_tool",
return_value=fake_tool,
),
patch(
- "litellm.proxy._experimental.mcp_server.server._handle_local_mcp_tool",
+ "litellm.proxy._experimental.mcp_server.operations._handle_local_mcp_tool",
new=handle_local,
),
patch(
@@ -221,7 +222,7 @@ async def test_openapi_local_tool_denied_when_server_not_resolvable():
),
):
with pytest.raises(HTTPException) as exc:
- await mcp_module.execute_mcp_tool(
+ await mcp_operations.execute_mcp_tool(
name="list_pets",
arguments={},
allowed_mcp_servers=[],
@@ -280,27 +281,27 @@ async def test_openapi_local_tool_injects_resolved_oauth_token():
with (
patch.object(
- mcp_module.global_mcp_server_manager,
+ mcp_operations.global_mcp_server_manager,
"_get_mcp_server_from_tool_name",
return_value=oauth_server,
),
patch.object(
- mcp_module.global_mcp_server_manager,
+ mcp_operations.global_mcp_server_manager,
"pre_call_tool_check",
new=AsyncMock(return_value={}),
),
patch.object(
- mcp_module.global_mcp_tool_registry,
+ mcp_operations.global_mcp_tool_registry,
"get_tool",
return_value=fake_tool,
),
patch.object(
- mcp_module.global_mcp_server_manager._cred_provider,
+ mcp_operations.global_mcp_server_manager._cred_provider,
"resolve_credentials",
new=AsyncMock(return_value=Ok(StaticHeaderAuth("Bearer stored-user-token"))),
),
patch(
- "litellm.proxy._experimental.mcp_server.server._handle_local_mcp_tool",
+ "litellm.proxy._experimental.mcp_server.operations._handle_local_mcp_tool",
new=handle_local,
),
patch(
@@ -308,7 +309,7 @@ async def test_openapi_local_tool_injects_resolved_oauth_token():
return_value=True,
),
):
- await mcp_module.execute_mcp_tool(
+ await mcp_operations.execute_mcp_tool(
name="get_values",
arguments={},
allowed_mcp_servers=[oauth_server],
@@ -417,7 +418,7 @@ async def test_legacy_local_tool_fallback_refuses_unentitled_caller(legacy_local
)
with pytest.raises(HTTPException) as exc:
- await mcp_module.execute_mcp_tool(
+ await mcp_operations.execute_mcp_tool(
name=f"{LEGACY_SERVER_NAME}-{LEGACY_TOOL}",
arguments={},
allowed_mcp_servers=[server],
@@ -451,7 +452,7 @@ async def test_legacy_local_tool_fallback_still_dispatches_entitled_caller(
server, executed = legacy_local_tool
user = _caller_entitled_to([LEGACY_TOOL])
- result = await mcp_module.execute_mcp_tool(
+ result = await mcp_operations.execute_mcp_tool(
name=f"{LEGACY_SERVER_NAME}-{LEGACY_TOOL}",
arguments={},
allowed_mcp_servers=[server],
@@ -481,7 +482,7 @@ async def test_legacy_local_tool_fallback_fails_closed_on_empty_prefix(
_server, executed = legacy_local_tool
with pytest.raises(HTTPException) as exc:
- await mcp_module.execute_mcp_tool(
+ await mcp_operations.execute_mcp_tool(
name=f"-{LEGACY_TOOL}",
arguments={},
allowed_mcp_servers=[],
@@ -523,7 +524,7 @@ async def test_legacy_local_tool_fallback_fails_closed_when_prefix_names_no_serv
return_value=True,
):
with pytest.raises(HTTPException) as exc:
- await mcp_module.execute_mcp_tool(
+ await mcp_operations.execute_mcp_tool(
name=f"{LEGACY_SERVER_NAME}-{LEGACY_TOOL}",
arguments={},
allowed_mcp_servers=[other_server],
@@ -546,7 +547,7 @@ async def test_unknown_tool_name_still_reports_not_found():
from litellm.proxy._experimental.mcp_server import server as mcp_module
with pytest.raises(HTTPException) as exc:
- await mcp_module.execute_mcp_tool(
+ await mcp_operations.execute_mcp_tool(
name="tool_no_registry_knows",
arguments={},
allowed_mcp_servers=[],
@@ -610,7 +611,7 @@ async def test_per_server_auth_header_reaches_both_openapi_dispatch_arms(dispatc
captured["injected"] = _request_auth_header.get()
return []
- manager = mcp_module.global_mcp_server_manager
+ manager = mcp_operations.global_mcp_server_manager
with (
patch.object(manager, "resolve_openapi_upstream_auth", new=fake_resolver),
patch.object(manager, "pre_call_tool_check", new=AsyncMock(return_value={})),
@@ -620,9 +621,9 @@ async def test_per_server_auth_header_reaches_both_openapi_dispatch_arms(dispatc
fake_tool.name = "list_reports"
with (
patch.object(manager, "_get_mcp_server_from_tool_name", return_value=server),
- patch.object(mcp_module.global_mcp_tool_registry, "get_tool", return_value=fake_tool),
+ patch.object(mcp_operations.global_mcp_tool_registry, "get_tool", return_value=fake_tool),
patch(
- "litellm.proxy._experimental.mcp_server.server._handle_local_mcp_tool",
+ "litellm.proxy._experimental.mcp_server.operations._handle_local_mcp_tool",
new=capture_local,
),
patch(
@@ -630,7 +631,7 @@ async def test_per_server_auth_header_reaches_both_openapi_dispatch_arms(dispatc
return_value=True,
),
):
- await mcp_module.execute_mcp_tool(
+ await mcp_operations.execute_mcp_tool(
name="list_reports",
arguments={},
allowed_mcp_servers=[server],
@@ -702,11 +703,11 @@ async def test_local_dispatch_reports_the_outcome_instead_of_success(failure: st
user = UserAPIKeyAuth(api_key="sk-user", user_id="alice", user_role=LitellmUserRoles.INTERNAL_USER.value)
with (
- patch.object(mcp_module.global_mcp_server_manager, "_get_mcp_server_from_tool_name", return_value=server),
- patch.object(mcp_module.global_mcp_server_manager, "pre_call_tool_check", new=AsyncMock(return_value={})),
- patch.object(mcp_module.global_mcp_tool_registry, "get_tool", return_value=fake_tool),
+ patch.object(mcp_operations.global_mcp_server_manager, "_get_mcp_server_from_tool_name", return_value=server),
+ patch.object(mcp_operations.global_mcp_server_manager, "pre_call_tool_check", new=AsyncMock(return_value={})),
+ patch.object(mcp_operations.global_mcp_tool_registry, "get_tool", return_value=fake_tool),
patch.object(
- mcp_module.global_mcp_server_manager,
+ mcp_operations.global_mcp_server_manager,
"resolve_openapi_upstream_auth",
new=AsyncMock(return_value=(None, None)),
),
@@ -715,7 +716,7 @@ async def test_local_dispatch_reports_the_outcome_instead_of_success(failure: st
return_value=True,
),
):
- call = mcp_module.execute_mcp_tool(
+ call = mcp_operations.execute_mcp_tool(
name="list_reports",
arguments={},
allowed_mcp_servers=[server],
diff --git a/tests/test_litellm/proxy/_experimental/mcp_server/test_operations.py b/tests/test_litellm/proxy/_experimental/mcp_server/test_operations.py
new file mode 100644
index 00000000000..870f2cd5f8f
--- /dev/null
+++ b/tests/test_litellm/proxy/_experimental/mcp_server/test_operations.py
@@ -0,0 +1,343 @@
+import asyncio
+from unittest.mock import AsyncMock, patch
+
+import pytest
+from mcp.types import GetPromptRequest, GetPromptRequestParams, GetPromptResult
+
+from litellm.proxy._experimental.mcp_server.operations import GatewayOperations, prepare_context
+from litellm.proxy._types import UserAPIKeyAuth
+
+
+@pytest.mark.asyncio
+async def test_dispatch_uses_explicit_context_when_ambient_caller_differs():
+ from mcp.server.auth.middleware.auth_context import auth_context_var
+ from litellm.proxy._experimental.mcp_server.server import set_auth_context
+
+ context = prepare_context(
+ UserAPIKeyAuth(user_id="alpha"),
+ raw_headers={"x-caller": "alpha"},
+ mcp_servers=["alpha-server"],
+ client_ip="192.0.2.1",
+ )
+ token = auth_context_var.set(None)
+ handler = AsyncMock(return_value=GetPromptResult(messages=[]))
+ try:
+ set_auth_context(UserAPIKeyAuth(user_id="bravo"), raw_headers={"x-caller": "bravo"})
+ with patch("litellm.proxy._experimental.mcp_server.operations.mcp_get_prompt", handler):
+ result = await GatewayOperations().execute(
+ GetPromptRequest(params=GetPromptRequestParams(name="alpha-prompt")), context
+ )
+ assert result.messages == []
+ assert handler.await_args.kwargs["name"] == "alpha-prompt"
+ assert handler.await_args.kwargs["user_api_key_auth"].user_id == "alpha"
+ assert handler.await_args.kwargs["raw_headers"] == {"x-caller": "alpha"}
+ assert handler.await_args.kwargs["mcp_servers"] == ["alpha-server"]
+ assert handler.await_args.kwargs["client_ip"] == "192.0.2.1"
+ finally:
+ auth_context_var.reset(token)
+
+
+@pytest.mark.asyncio
+async def test_legacy_adapter_cleans_context_after_cancelled_operation():
+ from types import SimpleNamespace
+ from litellm.proxy._experimental.mcp_server import server
+ from litellm.proxy._experimental.mcp_server.mcp_context import active_mcp_request_ctx_var
+
+ previous_session = server.active_mcp_session_var.get()
+ previous_request = active_mcp_request_ctx_var.get()
+ request = SimpleNamespace(session=object())
+ auth = (None, None, None, None, None, None, None)
+
+ async def cancelled_operation():
+ async with server._legacy_operation_context(request, trace=False):
+ assert server.active_mcp_session_var.get() is request.session
+ assert active_mcp_request_ctx_var.get() is request
+ raise asyncio.CancelledError
+
+ with patch(
+ "litellm.proxy._experimental.mcp_server.server.get_or_extract_auth_context", AsyncMock(return_value=auth)
+ ):
+ with pytest.raises(asyncio.CancelledError):
+ await cancelled_operation()
+ assert server.active_mcp_session_var.get() is previous_session
+ assert active_mcp_request_ctx_var.get() is previous_request
+
+
+@pytest.mark.asyncio
+async def test_legacy_adapter_cleans_context_when_trace_setup_fails():
+ from types import SimpleNamespace
+ from litellm.proxy._experimental.mcp_server import server
+ from litellm.proxy._experimental.mcp_server.mcp_context import active_mcp_request_ctx_var
+
+ previous_session = server.active_mcp_session_var.get()
+ previous_request = active_mcp_request_ctx_var.get()
+ request = SimpleNamespace(session=object())
+
+ async def enter_operation():
+ async with server._legacy_operation_context(request, trace=True):
+ pytest.fail("Trace setup failure must prevent dispatch")
+
+ with patch.object(server, "_otel_set_mcp_transport_span", side_effect=RuntimeError("trace failure")):
+ with pytest.raises(RuntimeError, match="trace failure"):
+ await enter_operation()
+ assert server.active_mcp_session_var.get() is previous_session
+ assert active_mcp_request_ctx_var.get() is previous_request
+
+
+@pytest.mark.asyncio
+async def test_prompt_sampling_receives_explicit_operation_caller_headers_and_ip():
+ from unittest.mock import MagicMock
+ from litellm.proxy._experimental.mcp_server import operations
+ from litellm.types.mcp import MCPTransport
+ from litellm.types.mcp_server.mcp_server_manager import MCPServer
+
+ upstream = MCPServer(
+ server_id="explicit-prompt",
+ name="explicit_prompt",
+ url="https://example.invalid/mcp",
+ transport=MCPTransport.http,
+ allow_sampling=True,
+ )
+ context = prepare_context(
+ UserAPIKeyAuth(user_id="prompt-caller"),
+ raw_headers={"x-caller": "prompt-caller"},
+ client_ip="192.0.2.41",
+ )
+ client = MagicMock()
+ client.get_prompt = AsyncMock(return_value=GetPromptResult(messages=[]))
+ sampling = AsyncMock()
+ with (
+ patch.object(operations, "_get_allowed_mcp_servers", AsyncMock(return_value=[upstream])),
+ patch("litellm.proxy._experimental.mcp_server.mcp_server_manager.MCPClient", return_value=client) as factory,
+ patch("litellm.proxy._experimental.mcp_server.sampling_handler.handle_sampling_create_message", sampling),
+ ):
+ result = await GatewayOperations().execute(
+ GetPromptRequest(params=GetPromptRequestParams(name="explicit_prompt-prompt")), context
+ )
+ assert result.messages == []
+ await factory.call_args.kwargs["sampling_callback"](None, None)
+ captured = sampling.await_args.kwargs
+ assert captured["user_api_key_auth"] is not None
+ assert captured["user_api_key_auth"].user_id == "prompt-caller"
+ assert captured["raw_headers"] == {"x-caller": "prompt-caller"}
+ assert captured["client_ip"] == "192.0.2.41"
+
+
+def _catalog_case(method):
+ from mcp import types
+
+ cases = {
+ "prompts/list": (
+ types.ListPromptsRequest(),
+ "list_prompts",
+ "get_prompts_from_server",
+ [types.Prompt(name="catalog-prompt")],
+ "prompts",
+ ),
+ "prompts/get": (
+ types.GetPromptRequest(
+ params=types.GetPromptRequestParams(name="catalog-prompt", arguments={"topic": "test"})
+ ),
+ "get_prompt",
+ "get_prompt_from_server",
+ types.GetPromptResult(messages=[]),
+ None,
+ ),
+ "resources/list": (
+ types.ListResourcesRequest(),
+ "list_resources",
+ "get_resources_from_server",
+ [types.Resource(name="document", uri="https://example.com/document")],
+ "resources",
+ ),
+ "resources/templates/list": (
+ types.ListResourceTemplatesRequest(),
+ "list_resource_templates",
+ "get_resource_templates_from_server",
+ [types.ResourceTemplate(name="document", uri_template="https://example.com/{name}")],
+ "resource_templates",
+ ),
+ "resources/read": (
+ types.ReadResourceRequest(params=types.ReadResourceRequestParams(uri="https://example.com/document")),
+ "read_resource",
+ "read_resource_from_server",
+ types.ReadResourceResult(
+ contents=[types.TextResourceContents(uri="https://example.com/document", text="document body")]
+ ),
+ None,
+ ),
+ }
+ return cases[method]
+
+
+@pytest.mark.asyncio
+@pytest.mark.parametrize(
+ "method", ["prompts/list", "prompts/get", "resources/list", "resources/templates/list", "resources/read"]
+)
+@pytest.mark.parametrize("state", ["success", "denied", "upstream_failure", "scope_failure"])
+async def test_native_catalog_operations_preserve_context_results_and_failure_policy(method, state):
+ from types import SimpleNamespace
+ from fastapi import HTTPException
+ from mcp.server.context import ServerRequestContext
+ from mcp.types import PaginatedRequestParams
+ from litellm.proxy._experimental.mcp_server import operations, server
+ from litellm.types.mcp import MCPTransport
+ from litellm.types.mcp_server.mcp_server_manager import MCPServer
+
+ operation, handler_name, manager_method, payload, collection = _catalog_case(method)
+ caller = UserAPIKeyAuth(user_id="catalog-caller")
+ headers = {"x-caller": "catalog-caller"}
+ upstream_server = MCPServer(server_id="catalog", name="catalog", transport=MCPTransport.http)
+ allowed = AsyncMock(
+ return_value=[] if state == "denied" else [upstream_server],
+ side_effect=HTTPException(status_code=403, detail="scope denied") if state == "scope_failure" else None,
+ )
+ upstream = AsyncMock(
+ return_value=payload, side_effect=RuntimeError("upstream unavailable") if state == "upstream_failure" else None
+ )
+ ctx = ServerRequestContext(
+ session=SimpleNamespace(), lifespan_context={}, protocol_version="2025-06-18", method=method
+ )
+ auth = (caller, None, ["catalog"], None, None, headers, "192.0.2.41")
+ with (
+ patch.object(server, "get_or_extract_auth_context", AsyncMock(return_value=auth)),
+ patch.object(operations, "_get_allowed_mcp_servers", allowed),
+ patch.object(operations.global_mcp_server_manager, manager_method, upstream),
+ ):
+ if collection is None and state != "success":
+ expected_error = RuntimeError if state == "upstream_failure" else HTTPException
+ with pytest.raises(expected_error):
+ await getattr(server, handler_name)(ctx, operation.params)
+ else:
+ result = await getattr(server, handler_name)(ctx, operation.params or PaginatedRequestParams())
+ if collection:
+ assert getattr(result, collection) == (payload if state == "success" else [])
+ else:
+ assert result == payload
+ assert allowed.await_args.kwargs == {
+ "user_api_key_auth": caller,
+ "mcp_servers": ["catalog"],
+ "client_ip": "192.0.2.41",
+ }
+ if state in ("denied", "scope_failure"):
+ upstream.assert_not_awaited()
+ else:
+ upstream.assert_awaited_once()
+ forwarded = upstream.await_args.kwargs
+ assert forwarded["user_api_key_auth"] == caller
+ assert forwarded["raw_headers"] == headers
+ assert forwarded["client_ip"] == "192.0.2.41"
+
+
+@pytest.mark.asyncio
+@pytest.mark.parametrize(
+ "method", ["prompts/list", "prompts/get", "resources/list", "resources/templates/list", "resources/read"]
+)
+async def test_explicit_proxy_context_rejects_catalog_operations_before_upstream_access(method):
+ from mcp.shared.exceptions import MCPError
+ from mcp.types import METHOD_NOT_FOUND
+ from litellm.proxy._experimental.mcp_server import operations
+
+ operation, _, manager_method, _, _ = _catalog_case(method)
+ upstream = AsyncMock()
+ with patch.object(operations.global_mcp_server_manager, manager_method, upstream):
+ with pytest.raises(MCPError) as rejected:
+ await GatewayOperations().execute(operation, prepare_context(mcp_proxy_mode=True))
+ assert rejected.value.error.code == METHOD_NOT_FOUND
+ assert rejected.value.error.message == "Operation unavailable on /mcp/proxy"
+ upstream.assert_not_awaited()
+
+
+@pytest.mark.asyncio
+@pytest.mark.parametrize("failure", ["missing_env", "pii", "guardrail", "unexpected"])
+async def test_tool_operation_preserves_failure_messages_and_request_trace(failure):
+ from mcp.types import CallToolRequest, CallToolRequestParams
+ from litellm.exceptions import BlockedPiiEntityError, GuardrailRaisedException
+ from litellm.proxy._experimental.mcp_server import operations
+ from litellm.proxy._experimental.mcp_server.utils import MCPMissingUserEnvVarsError
+
+ failures = {
+ "missing_env": (
+ MCPMissingUserEnvVarsError(
+ server_id="server", server_name="server", missing=["TOKEN"], setup_url="https://example.com/setup"
+ ),
+ "https://example.com/setup",
+ ),
+ "pii": (
+ BlockedPiiEntityError(entity_type="EMAIL_ADDRESS", guardrail_name="test"),
+ "Blocked PII entity detected",
+ ),
+ "guardrail": (GuardrailRaisedException(message="request denied"), "Guardrail violation"),
+ "unexpected": (RuntimeError("upstream unavailable"), "Error: upstream unavailable"),
+ }
+ error, expected = failures[failure]
+ dispatch = AsyncMock(side_effect=error)
+ context = prepare_context(
+ raw_headers={"x-litellm-trace-id": "operation-trace", "authorization": "private-test-header"}
+ )
+ with patch.object(operations, "call_mcp_tool", dispatch):
+ result = await GatewayOperations().execute(
+ CallToolRequest(params=CallToolRequestParams(name="catalog-tool", arguments={})), context
+ )
+ assert result.is_error is True
+ assert expected in result.content[0].text
+ assert "private-test-header" not in result.content[0].text
+ dispatch.assert_awaited_once()
+ assert dispatch.await_args.kwargs["litellm_trace_id"] == "operation-trace"
+ assert dispatch.await_args.kwargs["litellm_session_id"] == "operation-trace"
+
+
+@pytest.mark.asyncio
+@pytest.mark.parametrize(
+ "method,helper",
+ [
+ ("prompts/list", "_list_mcp_prompts"),
+ ("resources/list", "_list_mcp_resources"),
+ ("resources/templates/list", "_list_mcp_resource_templates"),
+ ],
+)
+async def test_catalog_operation_preserves_empty_result_for_malformed_upstream_items(method, helper):
+ from litellm.proxy._experimental.mcp_server import operations
+
+ operation, _, _, _, collection = _catalog_case(method)
+ with patch.object(operations, helper, AsyncMock(return_value=[{"unexpected": "item"}])):
+ result = await GatewayOperations().execute(operation, prepare_context())
+ assert getattr(result, collection) == []
+
+
+@pytest.mark.asyncio
+@pytest.mark.parametrize("catalog_unavailable", [False, True])
+async def test_tool_listing_returns_empty_result_without_dispatch_for_unavailable_catalog(catalog_unavailable):
+ from mcp.types import ListToolsRequest
+ from litellm.proxy._experimental.mcp_server import operations
+
+ allowed = AsyncMock(
+ return_value=[], side_effect=RuntimeError("catalog unavailable") if catalog_unavailable else None
+ )
+ upstream = AsyncMock()
+ with (
+ patch.object(operations, "_get_allowed_mcp_servers", allowed),
+ patch.object(operations.global_mcp_server_manager, "_get_tools_from_server", upstream),
+ ):
+ result = await GatewayOperations().execute(ListToolsRequest(), prepare_context())
+ assert result.tools == []
+ allowed.assert_awaited_once()
+ upstream.assert_not_awaited()
+
+
+@pytest.mark.asyncio
+async def test_explicit_proxy_context_lists_builtin_tools_and_blocks_direct_tool_dispatch():
+ from mcp.types import CallToolRequest, CallToolRequestParams, ListToolsRequest
+ from litellm.proxy._experimental.mcp_server import operations
+
+ context = prepare_context(mcp_proxy_mode=True)
+ allowed = AsyncMock()
+ with patch.object(operations, "_get_allowed_mcp_servers", allowed):
+ listing = await GatewayOperations().execute(ListToolsRequest(), context)
+ denied = await GatewayOperations().execute(
+ CallToolRequest(params=CallToolRequestParams(name="catalog-tool", arguments={})), context
+ )
+ assert {tool.name for tool in listing.tools} == {"search_tools", "get_tool_schema", "call_tool"}
+ assert denied.is_error is True
+ assert "unavailable on /mcp/proxy" in denied.content[0].text
+ allowed.assert_not_awaited()
diff --git a/tests/test_litellm/proxy/_experimental/mcp_server/test_rest_endpoints.py b/tests/test_litellm/proxy/_experimental/mcp_server/test_rest_endpoints.py
index 13af58c15c0..233a8cc96ba 100644
--- a/tests/test_litellm/proxy/_experimental/mcp_server/test_rest_endpoints.py
+++ b/tests/test_litellm/proxy/_experimental/mcp_server/test_rest_endpoints.py
@@ -1,3 +1,4 @@
+from litellm.proxy._experimental.mcp_server import operations as mcp_operations
import asyncio
import inspect
import json
@@ -1253,6 +1254,7 @@ class TestListToolsRestAPI:
user_api_key_auth=None,
extra_headers=None,
apply_tool_filters=True,
+ client_ip=None,
):
captured["called"] = True
captured["server"] = server
@@ -1338,6 +1340,7 @@ class TestListToolsRestAPI:
user_api_key_auth=None,
extra_headers=None,
apply_tool_filters=True,
+ client_ip=None,
):
captured["user_api_key_auth"] = user_api_key_auth
return ["tool-1"]
@@ -1891,6 +1894,7 @@ class TestListToolsRestAPI:
user_api_key_auth=None,
extra_headers=None,
apply_tool_filters=True,
+ client_ip=None,
):
captured["called"] = True
captured["server_arg"] = server
@@ -2027,6 +2031,7 @@ class TestListToolsRestAPI:
user_api_key_auth=None,
extra_headers=None,
apply_tool_filters=True,
+ client_ip=None,
):
captured["called"] = True
captured["server_arg"] = server
@@ -2112,6 +2117,7 @@ class TestListToolsRestAPI:
user_api_key_auth=None,
extra_headers=None,
apply_tool_filters=True,
+ client_ip=None,
):
return ["scoped-tool"]
@@ -2319,6 +2325,7 @@ class TestListToolsRestAPI:
user_api_key_auth=None,
extra_headers=None,
apply_tool_filters=True,
+ client_ip=None,
):
captured["server"] = server
captured["auth_header"] = server_auth_header
@@ -3145,10 +3152,10 @@ async def test_request_selected_tool_specific_guardrail_applies_to_virtual_execu
monkeypatch.setattr(litellm, "callbacks", [guardrail])
monkeypatch.setattr(tool_registry, "global_mcp_tool_registry", registry)
monkeypatch.setattr(mcp_server_manager, "global_mcp_server_manager", manager)
- monkeypatch.setattr(server, "global_mcp_tool_registry", registry)
- monkeypatch.setattr(server, "global_mcp_server_manager", manager)
+ monkeypatch.setattr(mcp_operations, "global_mcp_tool_registry", registry)
+ monkeypatch.setattr(mcp_operations, "global_mcp_server_manager", manager)
monkeypatch.setattr(rest_endpoints, "global_mcp_server_manager", manager)
- monkeypatch.setattr(server, "_get_allowed_mcp_servers", AsyncMock(return_value=[managed_server]))
+ monkeypatch.setattr(mcp_operations, "_get_allowed_mcp_servers", AsyncMock(return_value=[managed_server]))
monkeypatch.setattr(proxy_server, "proxy_logging_obj", ProxyLogging(user_api_key_cache=DualCache()))
monkeypatch.setattr(proxy_server, "add_litellm_data_to_request", passthrough_request_data)
monkeypatch.setattr(proxy_server, "proxy_config", {})
diff --git a/tests/test_litellm/test_check_mcp_operation_boundary.py b/tests/test_litellm/test_check_mcp_operation_boundary.py
new file mode 100644
index 00000000000..d7ac72de9f0
--- /dev/null
+++ b/tests/test_litellm/test_check_mcp_operation_boundary.py
@@ -0,0 +1,52 @@
+from pathlib import Path
+
+import pytest
+
+from scripts.check_mcp_operation_boundary import main, violations
+
+
+@pytest.mark.parametrize(
+ "source",
+ (
+ "from mcp.server.auth.middleware.auth_context import auth_context_var as hidden",
+ "from litellm.proxy._experimental.mcp_server.mcp_context import _mcp_proxy_mode as mode",
+ "caller = legacy.get_active_auth_context()",
+ "owners = transport._stateful_session_owners",
+ "from weakref import WeakKeyDictionary",
+ "from litellm.proxy._experimental.mcp_server.server import get_auth_context",
+ ),
+)
+def test_shared_operation_boundary_rejects_ambient_state(source):
+ assert violations(Path("operations.py"), source)
+
+
+def test_legacy_adapter_may_resolve_context_but_policy_must_receive_it():
+ source = "from litellm.proxy._experimental.mcp_server.mcp_context import _mcp_proxy_mode"
+ assert violations(Path("server.py"), source) == ()
+ assert violations(Path("legacy_callbacks.py"), source) == ()
+ assert violations(Path("operations.py"), "def execute(context):\n return context.client_ip") == ()
+ assert violations(Path("mcp_server_manager.py"), "def _mcp_registry_key(server):\n return server.name") == ()
+
+
+def test_boundary_command_rejects_shared_state_and_accepts_explicit_context(tmp_path, monkeypatch, capsys):
+ import subprocess
+ import sys
+
+ package = tmp_path / "litellm/proxy/_experimental/mcp_server"
+ package.mkdir(parents=True)
+ module = package / "operations.py"
+ module.write_text("from mcp.server.auth.middleware.auth_context import auth_context_var as hidden\n")
+ command = [sys.executable, str(Path(__file__).resolve().parents[2] / "scripts/check_mcp_operation_boundary.py")]
+ monkeypatch.chdir(tmp_path)
+ assert main() == 1
+ assert "operations.py:1:" in capsys.readouterr().err
+ rejected = subprocess.run(command, cwd=tmp_path, capture_output=True, text=True, check=False)
+ assert rejected.returncode == 1
+ assert "operations.py:1: MCP request/session state belongs in a legacy adapter" in rejected.stderr
+
+ module.write_text("def execute(context):\n return context.client_ip\n")
+ assert main() == 0
+ assert "MCP operation boundary: passed" in capsys.readouterr().out
+ accepted = subprocess.run(command, cwd=tmp_path, capture_output=True, text=True, check=False)
+ assert accepted.returncode == 0
+ assert "MCP operation boundary: passed" in accepted.stdout
From bf4fccc937175999d9327051479274bfb8c6d5fd Mon Sep 17 00:00:00 2001
From: yuneng
Date: Mon, 21 Sep 2026 19:25:52 +0000
Subject: [PATCH 047/160] feat(ui): expose remaining complexity router advanced
settings
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
---
.../add_model/ClassificationMethodConfig.tsx | 40 ++++++++-
.../add_model/ComplexityRouterConfig.tsx | 36 ++++++++
.../add_model/HeuristicKeywordOverrides.tsx | 42 +++++++++
.../add_model/HousekeepingRoutingControls.tsx | 44 ++++++++++
.../add_model/PlanModeOverrideControls.tsx | 18 ++++
.../components/add_model/ReminderMarkers.tsx | 77 +++++++++++++++++
.../add_model/ResponseFormatControls.tsx | 12 +++
.../add_model/add_auto_router_tab.tsx | 14 +++
.../build_complexity_router_config.test.ts | 58 +++++++++++++
.../build_complexity_router_config.ts | 85 +++++++++++++++++++
.../components/add_model/classifier_types.ts | 3 +-
...d_updated_complexity_router_config.test.ts | 17 ++++
.../edit_auto_router_modal.tsx | 54 ++++++++++++
13 files changed, 498 insertions(+), 2 deletions(-)
create mode 100644 ui/litellm-dashboard/src/components/add_model/HeuristicKeywordOverrides.tsx
create mode 100644 ui/litellm-dashboard/src/components/add_model/HousekeepingRoutingControls.tsx
create mode 100644 ui/litellm-dashboard/src/components/add_model/ReminderMarkers.tsx
diff --git a/ui/litellm-dashboard/src/components/add_model/ClassificationMethodConfig.tsx b/ui/litellm-dashboard/src/components/add_model/ClassificationMethodConfig.tsx
index b72b29a29f4..2564137fb02 100644
--- a/ui/litellm-dashboard/src/components/add_model/ClassificationMethodConfig.tsx
+++ b/ui/litellm-dashboard/src/components/add_model/ClassificationMethodConfig.tsx
@@ -19,7 +19,10 @@ import HeuristicScoringConfig from "./HeuristicScoringConfig";
import ClassifierReasoningEffortSelect from "./ClassifierReasoningEffortSelect";
import ClassifierCircuitBreakerConfig from "./ClassifierCircuitBreakerConfig";
import ClassifierVisionConfig from "./ClassifierVisionConfig";
-import { getHeuristicV2SuccessThresholdError } from "./build_complexity_router_config";
+import {
+ getClassifierPluginTimeoutError,
+ getHeuristicV2SuccessThresholdError,
+} from "./build_complexity_router_config";
import type { ReasoningEffort } from "./complexity_router_tiers";
import { useComplexityScorerDefaults } from "@/app/(dashboard)/hooks/autoRouter/useComplexityScorerDefaults";
import {
@@ -61,6 +64,7 @@ const CLASSIFIER_CONTEXT_WINDOW_SIZE_ID = "classifier-context-window-size";
const CLASSIFIER_CONTEXT_BUDGET_CHARS_ID = "classifier-context-budget-chars";
const HYBRID_BOUNDARY_MARGIN_ID = "hybrid-boundary-margin";
const HEURISTIC_V2_SUCCESS_THRESHOLD_ID = "heuristic-v2-success-threshold";
+const CLASSIFIER_PLUGIN_TIMEOUT_ID = "classifier-plugin-timeout-ms";
const CUSTOM_PROMPT_WITH_HEURISTIC_FALLBACK =
"This router classifies with your own prompt, so the tier comes from whatever rubric it states. The four tier " +
@@ -467,6 +471,40 @@ const ClassificationMethodConfig: React.FC = ({
<>
+ {classifierType === "custom" && (
+
+
+ This router uses a custom classifier plugin set in config.yaml. Pick a classifier below to replace it.
+
+
+ Classifier plugin timeout (ms)
+
+
+ onChange({
+ ...value,
+ classifier_plugin_timeout_ms: event.target.value.trim() === "" ? undefined : Number(event.target.value),
+ })
+ }
+ aria-invalid={Boolean(
+ showValidationErrors && getClassifierPluginTimeoutError(classifierType, value.classifier_plugin_timeout_ms),
+ )}
+ />
+
+ Time budget for the plugin call. On expiry the fallback path decides the tier.
+
+ {showValidationErrors && getClassifierPluginTimeoutError(classifierType, value.classifier_plugin_timeout_ms) && (
+
+ {getClassifierPluginTimeoutError(classifierType, value.classifier_plugin_timeout_ms)}
+
+ )}
+
+ )}
+
{classifierType === "heuristic_v2" && (
diff --git a/ui/litellm-dashboard/src/components/add_model/ComplexityRouterConfig.tsx b/ui/litellm-dashboard/src/components/add_model/ComplexityRouterConfig.tsx
index 9fa4e762015..a97e7a77a21 100644
--- a/ui/litellm-dashboard/src/components/add_model/ComplexityRouterConfig.tsx
+++ b/ui/litellm-dashboard/src/components/add_model/ComplexityRouterConfig.tsx
@@ -63,9 +63,14 @@ import { type DimensionWeights, type TierBoundaries, type TokenThresholds } from
import { type CustomDimensionRow } from "./custom_dimensions";
import CompressionControls from "./CompressionControls";
import { type AutoRouterCompressionState, DEFAULT_AUTO_ROUTER_COMPRESSION } from "./buildAutoRouterCompression";
+import HeuristicKeywordOverrides from "./HeuristicKeywordOverrides";
+import HousekeepingRoutingControls from "./HousekeepingRoutingControls";
+import ReminderMarkers from "./ReminderMarkers";
+import { type ReminderMarkerPair } from "./build_complexity_router_config";
export type { DimensionWeights, TierBoundaries, TokenThresholds };
export type { CustomTierSet, TierRow } from "./tier_rows";
+export type { ReminderMarkerPair } from "./build_complexity_router_config";
export const DEFAULT_CLASSIFIER_TIMEOUT_MS = 3000;
export const DEFAULT_TIER_DISTANCE_PENALTY = 0.5;
@@ -429,6 +434,16 @@ export interface ComplexityRouterConfigValue {
* edit round-trip.
*/
tier_model_params?: TierModelParamsByTier;
+ code_keywords?: string[];
+ reasoning_keywords?: string[];
+ technical_keywords?: string[];
+ simple_keywords?: string[];
+ plan_mode_patterns?: string[];
+ route_housekeeping_to_cheapest_tier?: boolean;
+ housekeeping_patterns?: string[];
+ reminder_markers?: ReminderMarkerPair[];
+ max_tokens_from_tier_model?: boolean;
+ classifier_plugin_timeout_ms?: number;
}
/** Session affinity wins where a hand-authored config sets both, matching the backend's own `or`. */
@@ -786,6 +801,17 @@ const ComplexityRouterConfig: React.FC = ({
/>
),
},
+ ]
+ : []),
+ ...(!forecast
+ ? [
+ {
+ key: "keyword-overrides",
+ label: (
+ Advanced: Heuristic Keyword Overrides
+ ),
+ children: ,
+ },
]
: []),
{
@@ -814,6 +840,16 @@ const ComplexityRouterConfig: React.FC = ({
),
},
+ {
+ key: "housekeeping",
+ label: Advanced: Housekeeping Routing ,
+ children: ,
+ },
+ {
+ key: "reminder-markers",
+ label: Advanced: Reminder Markers ,
+ children: ,
+ },
{
key: "context-window",
label: Advanced: Context Window Escalation ,
diff --git a/ui/litellm-dashboard/src/components/add_model/HeuristicKeywordOverrides.tsx b/ui/litellm-dashboard/src/components/add_model/HeuristicKeywordOverrides.tsx
new file mode 100644
index 00000000000..696924f4e77
--- /dev/null
+++ b/ui/litellm-dashboard/src/components/add_model/HeuristicKeywordOverrides.tsx
@@ -0,0 +1,42 @@
+import React from "react";
+import { MultiSelect } from "@/components/shared/MultiSelect";
+import type { ComplexityRouterConfigValue } from "./ComplexityRouterConfig";
+
+const fields = [
+ ["code_keywords", "Code keywords"],
+ ["reasoning_keywords", "Reasoning keywords"],
+ ["technical_keywords", "Technical keywords"],
+ ["simple_keywords", "Simple keywords"],
+] as const;
+
+const HeuristicKeywordOverrides: React.FC<{
+ value: ComplexityRouterConfigValue;
+ onChange: (value: ComplexityRouterConfigValue) => void;
+}> = ({ value, onChange }) => (
+
+
+ Each list replaces the built-in keyword list of the same name for the heuristic scorer. Leave a list empty to
+ keep the built-in one. To add technical terms without replacing the list, use custom technical keywords under
+ Classification Method.
+
+ {fields.map(([key, label]) => {
+ const keywords = value[key] ?? [];
+ return (
+
+ {label}
+ ({ label: keyword, value: keyword }))}
+ value={keywords}
+ onValueChange={(next) => onChange({ ...value, [key]: next.length > 0 ? next : undefined })}
+ placeholder={`Add ${label.toLowerCase()}`}
+ emptyText="Type to add a keyword"
+ allowCustomValues
+ className="w-full"
+ />
+
+ );
+ })}
+
+);
+
+export default HeuristicKeywordOverrides;
diff --git a/ui/litellm-dashboard/src/components/add_model/HousekeepingRoutingControls.tsx b/ui/litellm-dashboard/src/components/add_model/HousekeepingRoutingControls.tsx
new file mode 100644
index 00000000000..2b6caba6291
--- /dev/null
+++ b/ui/litellm-dashboard/src/components/add_model/HousekeepingRoutingControls.tsx
@@ -0,0 +1,44 @@
+import React from "react";
+import { MultiSelect } from "@/components/shared/MultiSelect";
+import { Switch } from "@/components/ui/switch";
+import type { ComplexityRouterConfigValue } from "./ComplexityRouterConfig";
+
+const HousekeepingRoutingControls: React.FC<{
+ value: ComplexityRouterConfigValue;
+ onChange: (value: ComplexityRouterConfigValue) => void;
+}> = ({ value, onChange }) => {
+ const enabled = value.route_housekeeping_to_cheapest_tier ?? true;
+ const patterns = value.housekeeping_patterns ?? [];
+ return (
+ <>
+
+ onChange({ ...value, route_housekeeping_to_cheapest_tier: next })}
+ aria-label="Route housekeeping calls to the cheapest tier"
+ />
+ Route housekeeping calls to the cheapest tier
+
+
+ Conversation-title style calls skip the classifier and go to the cheapest tier.
+
+ Additional housekeeping sentinels
+ ({ label: pattern, value: pattern }))}
+ value={patterns}
+ onValueChange={(next) => onChange({ ...value, housekeeping_patterns: next.length > 0 ? next : undefined })}
+ placeholder="e.g., conversation title"
+ emptyText="Type to add a sentinel"
+ allowCustomValues
+ disabled={!enabled}
+ className="w-full"
+ />
+
+ Case-sensitive literal strings added to the built-in conversation-title sentinels.
+ {!enabled && " Turn housekeeping routing on for these to take effect."}
+
+ >
+ );
+};
+
+export default HousekeepingRoutingControls;
diff --git a/ui/litellm-dashboard/src/components/add_model/PlanModeOverrideControls.tsx b/ui/litellm-dashboard/src/components/add_model/PlanModeOverrideControls.tsx
index 83fd0c42d16..e195fb62222 100644
--- a/ui/litellm-dashboard/src/components/add_model/PlanModeOverrideControls.tsx
+++ b/ui/litellm-dashboard/src/components/add_model/PlanModeOverrideControls.tsx
@@ -1,5 +1,6 @@
import React from "react";
import { Switch } from "@/components/ui/switch";
+import { MultiSelect } from "@/components/shared/MultiSelect";
import TierRowSelect from "./TierRowSelect";
import type { ComplexityRouterConfigValue } from "./ComplexityRouterConfig";
@@ -38,6 +39,23 @@ const PlanModeOverrideControls: React.FC<{
/>
)}
+
+ Additional plan-mode sentinels
+ ({ label: pattern, value: pattern }))}
+ value={value.plan_mode_patterns ?? []}
+ onValueChange={(patterns) =>
+ onChange({ ...value, plan_mode_patterns: patterns.length > 0 ? patterns : undefined })
+ }
+ placeholder="e.g., enter plan mode"
+ emptyText="Type to add a sentinel"
+ allowCustomValues
+ className="w-full"
+ />
+
+ Case-sensitive literal strings added to the built-in Claude Code and Copilot plan-mode markers.
+
+
>
);
diff --git a/ui/litellm-dashboard/src/components/add_model/ReminderMarkers.tsx b/ui/litellm-dashboard/src/components/add_model/ReminderMarkers.tsx
new file mode 100644
index 00000000000..452f4dc727f
--- /dev/null
+++ b/ui/litellm-dashboard/src/components/add_model/ReminderMarkers.tsx
@@ -0,0 +1,77 @@
+import React from "react";
+import { Plus, Trash2 } from "lucide-react";
+import { Button } from "@/components/ui/button";
+import { Input } from "@/components/ui/input";
+import type { ComplexityRouterConfigValue } from "./ComplexityRouterConfig";
+import { getReminderMarkersError, type ReminderMarkerPair } from "./build_complexity_router_config";
+
+const ReminderMarkers: React.FC<{
+ value: ComplexityRouterConfigValue;
+ onChange: (value: ComplexityRouterConfigValue) => void;
+ showValidationErrors?: boolean;
+}> = ({ value, onChange, showValidationErrors = false }) => {
+ const markers = value.reminder_markers ?? [];
+ const update = (index: number, patch: Partial) =>
+ onChange({
+ ...value,
+ reminder_markers: markers.map((marker, markerIndex) => (markerIndex === index ? { ...marker, ...patch } : marker)),
+ });
+ const remove = (index: number) => {
+ const next = markers.filter((_, markerIndex) => markerIndex !== index);
+ onChange({ ...value, reminder_markers: next.length > 0 ? next : undefined });
+ };
+ const error = getReminderMarkersError(value.reminder_markers);
+ return (
+
+
+ Delimiter pairs that wrap harness-injected reminder blocks, which are stripped before classification. Setting any
+ pair replaces the built-in pairs, so list every pair your harness emits. Matching is case-insensitive and values
+ are saved lowercased.
+
+
+ {markers.map((marker, index) => (
+
+
+
+ Opening delimiter
+
+ update(index, { open: event.target.value })}
+ />
+
+
+
+ Closing delimiter
+
+ update(index, { close: event.target.value })}
+ />
+
+
remove(index)}>
+
+
+
+ ))}
+
+
onChange({ ...value, reminder_markers: [...markers, { open: "", close: "" }] })}
+ >
+
+ Add marker pair
+
+ {showValidationErrors && error &&
{error}
}
+
+ );
+};
+
+export default ReminderMarkers;
diff --git a/ui/litellm-dashboard/src/components/add_model/ResponseFormatControls.tsx b/ui/litellm-dashboard/src/components/add_model/ResponseFormatControls.tsx
index 68dd880a684..9edbf204a6d 100644
--- a/ui/litellm-dashboard/src/components/add_model/ResponseFormatControls.tsx
+++ b/ui/litellm-dashboard/src/components/add_model/ResponseFormatControls.tsx
@@ -18,6 +18,18 @@ const ResponseFormatControls: React.FC<{
Return the resolved underlying model name in responses instead of the autorouter alias.
+
+ onChange({ ...value, max_tokens_from_tier_model: enabled })}
+ aria-label="Cap max_tokens at the tier model's output ceiling"
+ />
+ Cap max_tokens at the tier model's output ceiling
+
+
+ Replace the caller's max_tokens with the routed tier model's output ceiling so one client value fits every
+ tier. Off forwards the caller's value unchanged.
+
>
);
diff --git a/ui/litellm-dashboard/src/components/add_model/add_auto_router_tab.tsx b/ui/litellm-dashboard/src/components/add_model/add_auto_router_tab.tsx
index 84e44fee9c3..2fa28f761e3 100644
--- a/ui/litellm-dashboard/src/components/add_model/add_auto_router_tab.tsx
+++ b/ui/litellm-dashboard/src/components/add_model/add_auto_router_tab.tsx
@@ -48,6 +48,8 @@ import {
getKeywordTierRulesError,
getClassifierModelError,
getHeuristicV2SuccessThresholdError,
+ getReminderMarkersError,
+ getClassifierPluginTimeoutError,
getClassifierReasoningEffortError,
getMissingTiersError,
getPlanModeTierError,
@@ -152,6 +154,8 @@ export const getSubmitBlockedReason = (
getKeywordTierRulesError(keywordTierRules, activeTierRows(config)) ??
getClassifierModelError(config) ??
getHeuristicV2SuccessThresholdError(config.heuristic_v2_success_threshold) ??
+ getReminderMarkersError(config.reminder_markers) ??
+ getClassifierPluginTimeoutError(config.classifier_type, config.classifier_plugin_timeout_ms) ??
(heuristicScoringRole(config) === "decides" ? customDimensionsError(config.custom_dimensions) : null) ??
getClassifierReasoningEffortError(config, modelInfo) ??
getReferencedModelsError(referencedModelsParams, availability)
@@ -448,6 +452,16 @@ const AddAutoRouterTab: React.FC = ({
enableContextWindowEscalation: complexityRouterConfig.enable_context_window_escalation,
contextWindowEscalationBuffer: complexityRouterConfig.context_window_escalation_buffer,
sessionAffinityTtlSeconds: complexityRouterConfig.session_affinity_ttl_seconds,
+ codeKeywords: complexityRouterConfig.code_keywords,
+ reasoningKeywords: complexityRouterConfig.reasoning_keywords,
+ technicalKeywords: complexityRouterConfig.technical_keywords,
+ simpleKeywords: complexityRouterConfig.simple_keywords,
+ planModePatterns: complexityRouterConfig.plan_mode_patterns,
+ routeHousekeepingToCheapestTier: complexityRouterConfig.route_housekeeping_to_cheapest_tier,
+ housekeepingPatterns: complexityRouterConfig.housekeeping_patterns,
+ reminderMarkers: complexityRouterConfig.reminder_markers,
+ maxTokensFromTierModel: complexityRouterConfig.max_tokens_from_tier_model,
+ classifierPluginTimeoutMs: complexityRouterConfig.classifier_plugin_timeout_ms,
};
const submitRecommendedRouter = async (name: string) => {
diff --git a/ui/litellm-dashboard/src/components/add_model/build_complexity_router_config.test.ts b/ui/litellm-dashboard/src/components/add_model/build_complexity_router_config.test.ts
index 054aba7a6aa..cf5164d91c4 100644
--- a/ui/litellm-dashboard/src/components/add_model/build_complexity_router_config.test.ts
+++ b/ui/litellm-dashboard/src/components/add_model/build_complexity_router_config.test.ts
@@ -6,6 +6,8 @@ import {
getKeywordTierRulesError,
getClassifierModelError,
getHeuristicV2SuccessThresholdError,
+ getReminderMarkersError,
+ getClassifierPluginTimeoutError,
getClassifierReasoningEffortError,
getMissingTiersError,
hydrateCustomTierSet,
@@ -1482,3 +1484,59 @@ describe("classifier vision wire payload", () => {
expect(payload.classifier_llm_config).not.toHaveProperty("vision");
});
});
+
+describe("advanced complexity router fields", () => {
+ it("normalizes lists, reminder markers, and explicit false values", () => {
+ const payload = buildComplexityRouterConfig({
+ ...baseParams,
+ codeKeywords: [" async ", " "],
+ reasoningKeywords: ["prove"],
+ technicalKeywords: ["api"],
+ simpleKeywords: ["hello"],
+ planModePatterns: [" plan "],
+ routeHousekeepingToCheapestTier: false,
+ housekeepingPatterns: [" title "],
+ reminderMarkers: [{ open: " ", close: " " }],
+ maxTokensFromTierModel: false,
+ classifierType: "custom",
+ classifierPluginTimeoutMs: 3000,
+ });
+ expect(payload).toMatchObject({
+ code_keywords: ["async"],
+ reasoning_keywords: ["prove"],
+ technical_keywords: ["api"],
+ simple_keywords: ["hello"],
+ plan_mode_patterns: ["plan"],
+ route_housekeeping_to_cheapest_tier: false,
+ housekeeping_patterns: ["title"],
+ reminder_markers: [{ open: "", close: " " }],
+ max_tokens_from_tier_model: false,
+ classifier_plugin_timeout_ms: 3000,
+ });
+ });
+
+ it("omits defaults, empty lists, and timeout values for non-custom classifiers", () => {
+ const payload = buildComplexityRouterConfig({
+ ...baseParams,
+ codeKeywords: [" ", ""],
+ reminderMarkers: [],
+ routeHousekeepingToCheapestTier: true,
+ maxTokensFromTierModel: true,
+ classifierPluginTimeoutMs: 3000,
+ });
+ expect(payload).not.toHaveProperty("code_keywords");
+ expect(payload).not.toHaveProperty("reminder_markers");
+ expect(payload).not.toHaveProperty("route_housekeeping_to_cheapest_tier");
+ expect(payload).not.toHaveProperty("max_tokens_from_tier_model");
+ expect(payload).not.toHaveProperty("classifier_plugin_timeout_ms");
+ });
+
+ it("validates marker pairs and custom classifier timeout", () => {
+ expect(getReminderMarkersError([{ open: " ", close: " " }])).toContain("different");
+ expect(getReminderMarkersError([{ open: "", close: " " }])).toContain("needs both");
+ expect(getReminderMarkersError([{ open: "", close: " " }])).toBeNull();
+ expect(getClassifierPluginTimeoutError("custom", 0)).toContain("whole number");
+ expect(getClassifierPluginTimeoutError("custom", 3000)).toBeNull();
+ expect(getClassifierPluginTimeoutError("heuristic", 0)).toBeNull();
+ });
+});
diff --git a/ui/litellm-dashboard/src/components/add_model/build_complexity_router_config.ts b/ui/litellm-dashboard/src/components/add_model/build_complexity_router_config.ts
index 15ae2b4c0b0..96fbe7f2c2e 100644
--- a/ui/litellm-dashboard/src/components/add_model/build_complexity_router_config.ts
+++ b/ui/litellm-dashboard/src/components/add_model/build_complexity_router_config.ts
@@ -54,6 +54,10 @@ import {
export type ClassifierVisionConfig = { enabled?: boolean; max_images?: number };
export type ClassifierLLMConfigWire = ClassifierLLMConfig & { vision?: ClassifierVisionConfig };
+export interface ReminderMarkerPair {
+ open: string;
+ close: string;
+}
/**
* Drop an empty system_prompt so the payload carries an override only when there is one. The
@@ -181,6 +185,16 @@ export interface StoredComplexityRouterConfig {
stall_escalation_enabled?: unknown;
stall_escalation_window?: unknown;
stall_escalation_repeat_threshold?: unknown;
+ code_keywords?: unknown;
+ reasoning_keywords?: unknown;
+ technical_keywords?: unknown;
+ simple_keywords?: unknown;
+ plan_mode_patterns?: unknown;
+ route_housekeeping_to_cheapest_tier?: unknown;
+ housekeeping_patterns?: unknown;
+ reminder_markers?: unknown;
+ max_tokens_from_tier_model?: unknown;
+ classifier_plugin_timeout_ms?: unknown;
}
export interface BuildComplexityRouterConfigParams {
@@ -233,6 +247,16 @@ export interface BuildComplexityRouterConfigParams {
enableContextWindowEscalation?: boolean;
contextWindowEscalationBuffer?: number;
sessionAffinityTtlSeconds?: number;
+ codeKeywords?: string[];
+ reasoningKeywords?: string[];
+ technicalKeywords?: string[];
+ simpleKeywords?: string[];
+ planModePatterns?: string[];
+ routeHousekeepingToCheapestTier?: boolean;
+ housekeepingPatterns?: string[];
+ reminderMarkers?: ReminderMarkerPair[];
+ maxTokensFromTierModel?: boolean;
+ classifierPluginTimeoutMs?: number;
}
/**
@@ -302,6 +326,16 @@ export interface ComplexityRouterConfigPayload {
enable_context_window_escalation?: boolean;
context_window_escalation_buffer?: number;
tier_model_configs?: Record;
+ code_keywords?: string[];
+ reasoning_keywords?: string[];
+ technical_keywords?: string[];
+ simple_keywords?: string[];
+ plan_mode_patterns?: string[];
+ route_housekeeping_to_cheapest_tier?: boolean;
+ housekeeping_patterns?: string[];
+ reminder_markers?: ReminderMarkerPair[];
+ max_tokens_from_tier_model?: boolean;
+ classifier_plugin_timeout_ms?: number;
}
export const serializeTierLabels = (tierLabels: ComplexityTierLabels | undefined): ComplexityTierLabels | undefined => {
@@ -376,6 +410,26 @@ export const getHeuristicV2SuccessThresholdError = (threshold: number | undefine
return validProbability ? null : "Success threshold must be a number between 0 and 1";
};
+export const getReminderMarkersError = (pairs: ReminderMarkerPair[] | undefined): string | null => {
+ for (const [index, pair] of (pairs ?? []).entries()) {
+ const open = pair.open.trim().toLowerCase();
+ const close = pair.close.trim().toLowerCase();
+ if (!open || !close) return `Reminder marker pair ${index + 1} needs both an opening and a closing delimiter`;
+ if (open === close) return `Reminder marker pair ${index + 1} must use different opening and closing delimiters`;
+ }
+ return null;
+};
+
+export const getClassifierPluginTimeoutError = (
+ classifierType: ClassifierType,
+ timeoutMs: number | undefined,
+): string | null => {
+ if (classifierType !== "custom" || timeoutMs === undefined) return null;
+ return Number.isInteger(timeoutMs) && timeoutMs > 0
+ ? null
+ : "Classifier plugin timeout must be a whole number of milliseconds greater than 0";
+};
+
export const getClassifierModelError = (
config: Pick<
ComplexityRouterConfigValue,
@@ -640,6 +694,16 @@ export const buildComplexityRouterConfig = ({
enableContextWindowEscalation,
contextWindowEscalationBuffer,
sessionAffinityTtlSeconds,
+ codeKeywords,
+ reasoningKeywords,
+ technicalKeywords,
+ simpleKeywords,
+ planModePatterns,
+ routeHousekeepingToCheapestTier,
+ housekeepingPatterns,
+ reminderMarkers,
+ maxTokensFromTierModel,
+ classifierPluginTimeoutMs,
}: BuildComplexityRouterConfigParams): ComplexityRouterConfigPayload => {
const serializedTierModelConfigs = customTierSet
? serializeTierModelConfigs(
@@ -672,6 +736,14 @@ export const buildComplexityRouterConfig = ({
};
const effectiveType = effectiveClassifierType({ custom_tier_set: customTierSet, classifier_type: classifierType });
const forecast = isForecastClassifier(effectiveType);
+ const cleanList = (items: string[] | undefined): string[] | undefined => {
+ const cleaned = (items ?? []).map((item) => item.trim()).filter(Boolean);
+ return cleaned.length > 0 ? cleaned : undefined;
+ };
+ const cleanedReminderMarkers = reminderMarkers?.map(({ open, close }) => ({
+ open: open.trim().toLowerCase(),
+ close: close.trim().toLowerCase(),
+ }));
const supportsOpeningPrompt = !customTierSet && !forecast && usesLlmClassifier(effectiveType);
const payload: ComplexityRouterConfigPayload = {
@@ -740,6 +812,19 @@ export const buildComplexityRouterConfig = ({
...(sessionAffinityTtlSeconds !== undefined && {
session_affinity_ttl_seconds: sessionAffinityTtlSeconds,
}),
+ ...(cleanList(codeKeywords) && { code_keywords: cleanList(codeKeywords) }),
+ ...(cleanList(reasoningKeywords) && { reasoning_keywords: cleanList(reasoningKeywords) }),
+ ...(cleanList(technicalKeywords) && { technical_keywords: cleanList(technicalKeywords) }),
+ ...(cleanList(simpleKeywords) && { simple_keywords: cleanList(simpleKeywords) }),
+ ...(cleanList(planModePatterns) && { plan_mode_patterns: cleanList(planModePatterns) }),
+ ...(routeHousekeepingToCheapestTier === false && { route_housekeeping_to_cheapest_tier: false }),
+ ...(cleanList(housekeepingPatterns) && { housekeeping_patterns: cleanList(housekeepingPatterns) }),
+ ...(cleanedReminderMarkers && cleanedReminderMarkers.length > 0 && { reminder_markers: cleanedReminderMarkers }),
+ ...(maxTokensFromTierModel === false && { max_tokens_from_tier_model: false }),
+ ...(classifierType === "custom" &&
+ classifierPluginTimeoutMs !== undefined &&
+ Number.isInteger(classifierPluginTimeoutMs) &&
+ classifierPluginTimeoutMs > 0 && { classifier_plugin_timeout_ms: classifierPluginTimeoutMs }),
...scorerKnobs,
};
if (!customTierSet) return payload;
diff --git a/ui/litellm-dashboard/src/components/add_model/classifier_types.ts b/ui/litellm-dashboard/src/components/add_model/classifier_types.ts
index ec88166ed2e..aa9d5619052 100644
--- a/ui/litellm-dashboard/src/components/add_model/classifier_types.ts
+++ b/ui/litellm-dashboard/src/components/add_model/classifier_types.ts
@@ -6,7 +6,8 @@ export type ClassifierType =
| "heuristic_first"
| "hybrid"
| "capability"
- | "llm_v2";
+ | "llm_v2"
+ | "custom";
export const usesLlmClassifier = (classifierType: ClassifierType): boolean =>
(["llm", "heuristic_first", "hybrid", "capability", "llm_v2"] as const).some((type) => type === classifierType);
diff --git a/ui/litellm-dashboard/src/components/edit_auto_router/build_updated_complexity_router_config.test.ts b/ui/litellm-dashboard/src/components/edit_auto_router/build_updated_complexity_router_config.test.ts
index 2450f7bce27..eaa6595d8a3 100644
--- a/ui/litellm-dashboard/src/components/edit_auto_router/build_updated_complexity_router_config.test.ts
+++ b/ui/litellm-dashboard/src/components/edit_auto_router/build_updated_complexity_router_config.test.ts
@@ -854,6 +854,16 @@ describe("managed keys survive an untouched open-and-save", () => {
reasoning_override_min_score: 0.3,
enable_context_window_escalation: false,
context_window_escalation_buffer: 0.9,
+ code_keywords: ["async", "await"],
+ reasoning_keywords: ["prove"],
+ technical_keywords: ["api"],
+ simple_keywords: ["hello"],
+ plan_mode_patterns: ["plan now"],
+ route_housekeeping_to_cheapest_tier: false,
+ housekeeping_patterns: ["conversation title"],
+ reminder_markers: [{ open: "", close: " " }],
+ max_tokens_from_tier_model: false,
+ classifier_plugin_timeout_ms: 3000,
};
// tier_definitions and fallback_tier cannot sit beside heuristic_first, which this fixture uses,
@@ -864,6 +874,7 @@ describe("managed keys survive an untouched open-and-save", () => {
"fallback_tier",
"hybrid_boundary_margin",
"jev_classifier_config",
+ "classifier_plugin_timeout_ms",
]);
// The stall keys are rejected beside the session pinning and user-turn classification this
@@ -894,6 +905,12 @@ describe("managed keys survive an untouched open-and-save", () => {
expect(dropped).toEqual([]);
});
+ it("keeps the custom classifier plugin timeout through an untouched save", () => {
+ const stored = { ...STORED_ALL_MANAGED, classifier_type: "custom", classifier_plugin_timeout_ms: 3000 };
+ const hydrated = hydrateComplexityRouterConfig(stored, undefined);
+ expect(buildUpdatedComplexityRouterConfig(stored, hydrated).classifier_plugin_timeout_ms).toBe(3000);
+ });
+
it("carries an enabled non-reasoning tier and its models through their own round trip", () => {
// `tiers` is rewritten wholesale on save, so this is the regression that matters: opening an
// enabled router and saving an unrelated edit must not delete the tier or its pool.
diff --git a/ui/litellm-dashboard/src/components/edit_auto_router/edit_auto_router_modal.tsx b/ui/litellm-dashboard/src/components/edit_auto_router/edit_auto_router_modal.tsx
index c88bbb101f7..476207553f7 100644
--- a/ui/litellm-dashboard/src/components/edit_auto_router/edit_auto_router_modal.tsx
+++ b/ui/litellm-dashboard/src/components/edit_auto_router/edit_auto_router_modal.tsx
@@ -45,6 +45,8 @@ import {
buildComplexityRouterConfig,
getClassifierModelError,
getHeuristicV2SuccessThresholdError,
+ getReminderMarkersError,
+ getClassifierPluginTimeoutError,
getClassifierReasoningEffortError,
getKeywordTierRulesError,
getMissingTiersError,
@@ -113,6 +115,8 @@ export const hydrateComplexityRouterConfig = (
parsedConfig: StoredComplexityRouterConfig,
complexityRouterDefaultModel: string | null | undefined,
): ComplexityRouterConfigValue => {
+ const stringList = (input: unknown): string[] | undefined =>
+ Array.isArray(input) ? input.filter((item): item is string => typeof item === "string") : undefined;
const builtIn = hydrateBuiltInTiers(parsedConfig.tiers, parsedConfig.enable_non_reasoning_tier);
const { tiers: hydratedTiers, enable_non_reasoning_tier } = builtIn;
const custom_tier_set = hydrateCustomTierSet(parsedConfig);
@@ -219,6 +223,31 @@ export const hydrateComplexityRouterConfig = (
typeof parsedConfig.stall_escalation_repeat_threshold === "number"
? parsedConfig.stall_escalation_repeat_threshold
: undefined,
+ code_keywords: stringList(parsedConfig.code_keywords),
+ reasoning_keywords: stringList(parsedConfig.reasoning_keywords),
+ technical_keywords: stringList(parsedConfig.technical_keywords),
+ simple_keywords: stringList(parsedConfig.simple_keywords),
+ plan_mode_patterns: stringList(parsedConfig.plan_mode_patterns),
+ route_housekeeping_to_cheapest_tier:
+ typeof parsedConfig.route_housekeeping_to_cheapest_tier === "boolean"
+ ? parsedConfig.route_housekeeping_to_cheapest_tier
+ : undefined,
+ housekeeping_patterns: stringList(parsedConfig.housekeeping_patterns),
+ reminder_markers: Array.isArray(parsedConfig.reminder_markers)
+ ? parsedConfig.reminder_markers.filter(
+ (pair): pair is { open: string; close: string } =>
+ typeof pair === "object" &&
+ pair !== null &&
+ typeof (pair as { open?: unknown }).open === "string" &&
+ typeof (pair as { close?: unknown }).close === "string",
+ )
+ : undefined,
+ max_tokens_from_tier_model:
+ typeof parsedConfig.max_tokens_from_tier_model === "boolean" ? parsedConfig.max_tokens_from_tier_model : undefined,
+ classifier_plugin_timeout_ms:
+ typeof parsedConfig.classifier_plugin_timeout_ms === "number" && Number.isFinite(parsedConfig.classifier_plugin_timeout_ms)
+ ? parsedConfig.classifier_plugin_timeout_ms
+ : undefined,
};
};
@@ -266,6 +295,16 @@ export const MANAGED_COMPLEXITY_ROUTER_KEYS = new Set([
"stall_escalation_enabled",
"stall_escalation_window",
"stall_escalation_repeat_threshold",
+ "code_keywords",
+ "reasoning_keywords",
+ "technical_keywords",
+ "simple_keywords",
+ "plan_mode_patterns",
+ "route_housekeeping_to_cheapest_tier",
+ "housekeeping_patterns",
+ "reminder_markers",
+ "max_tokens_from_tier_model",
+ "classifier_plugin_timeout_ms",
]);
// Managed only when the caller passes the corresponding state. A caller that does not render
@@ -387,6 +426,16 @@ export const buildUpdatedComplexityRouterConfig = (
stallEscalationEnabled: value.stall_escalation_enabled,
stallEscalationWindow: value.stall_escalation_window,
stallEscalationRepeatThreshold: value.stall_escalation_repeat_threshold,
+ codeKeywords: value.code_keywords,
+ reasoningKeywords: value.reasoning_keywords,
+ technicalKeywords: value.technical_keywords,
+ simpleKeywords: value.simple_keywords,
+ planModePatterns: value.plan_mode_patterns,
+ routeHousekeepingToCheapestTier: value.route_housekeeping_to_cheapest_tier,
+ housekeepingPatterns: value.housekeeping_patterns,
+ reminderMarkers: value.reminder_markers,
+ maxTokensFromTierModel: value.max_tokens_from_tier_model,
+ classifierPluginTimeoutMs: value.classifier_plugin_timeout_ms,
};
const built = buildComplexityRouterConfig(builderParams);
@@ -585,6 +634,11 @@ const EditAutoRouterModal: React.FC = ({
const classifierError =
getClassifierModelError(complexityRouterConfig) ??
getHeuristicV2SuccessThresholdError(complexityRouterConfig.heuristic_v2_success_threshold) ??
+ getReminderMarkersError(complexityRouterConfig.reminder_markers) ??
+ getClassifierPluginTimeoutError(
+ complexityRouterConfig.classifier_type,
+ complexityRouterConfig.classifier_plugin_timeout_ms,
+ ) ??
getForecastConfigError(complexityRouterConfig) ??
(heuristicScoringRole(complexityRouterConfig) === "decides"
? customDimensionsError(complexityRouterConfig.custom_dimensions)
From d2f30a77fded3a39c32b5dfc2f7f464bcd8f0409 Mon Sep 17 00:00:00 2001
From: Joshua Valluru <326636767+joshua-berri@users.noreply.github.com>
Date: Mon, 21 Sep 2026 12:31:48 -0700
Subject: [PATCH 048/160] test(mcp): preserve toolset scope across explicit
context
---
.../mcp_server/test_mcp_toolset_scope.py | 16 ++++++++++++++++
1 file changed, 16 insertions(+)
diff --git a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_toolset_scope.py b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_toolset_scope.py
index 60098b1f656..1398884783e 100644
--- a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_toolset_scope.py
+++ b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_toolset_scope.py
@@ -58,6 +58,22 @@ class TestApplyToolsetScope:
assert set(op.mcp_servers or []) == {"server-a", "server-b"}
assert op.mcp_tool_permissions == toolset_perms
+ from litellm.proxy._experimental.mcp_server.mcp_server_manager import MCPServerManager
+ from litellm.proxy._experimental.mcp_server.operations import prepare_context
+
+ manager = MCPServerManager()
+ unscoped_open = await manager.operator_open_server_ids(
+ auth, allow_all_server_ids=["operator-open-outside-toolset"], submitted_server_ids=[]
+ )
+ scoped_open = await manager.operator_open_server_ids(
+ prepare_context(result).user_api_key_auth,
+ allow_all_server_ids=["operator-open-outside-toolset"],
+ submitted_server_ids=[],
+ )
+ assert unscoped_open == {"operator-open-outside-toolset"}
+ assert scoped_open == set()
+ assert auth.mcp_toolset_id is None
+
@pytest.mark.asyncio
async def test_admin_creates_object_permission_when_none(self):
"""Admin key with object_permission=None can access any toolset."""
From 9c411dd6f2e569ea7ac54c08f03847d5c70cddba Mon Sep 17 00:00:00 2001
From: yucheng
Date: Mon, 21 Sep 2026 19:35:29 +0000
Subject: [PATCH 049/160] refactor(proxy): make scheduled job shutdown timeouts
configurable via env
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
---
litellm/constants.py | 2 ++
litellm/proxy/shutdown/scheduled_jobs.py | 23 +++++++++++--------
.../proxy/shutdown/test_scheduled_jobs.py | 15 ++++++------
3 files changed, 24 insertions(+), 16 deletions(-)
diff --git a/litellm/constants.py b/litellm/constants.py
index bbeb4846e27..842adf62f6b 100644
--- a/litellm/constants.py
+++ b/litellm/constants.py
@@ -1742,6 +1742,8 @@ SPEND_LOG_CLEANUP_BATCH_FAILURE_BACKOFF_SECONDS: Final = float(
SPEND_LOG_CLEANUP_RUN_BUDGET_SECONDS: Final = float(os.getenv("SPEND_LOG_CLEANUP_RUN_BUDGET_SECONDS", "300"))
SPEND_LOG_CLEANUP_BATCH_TIMEOUT_SECONDS: Final = float(os.getenv("SPEND_LOG_CLEANUP_BATCH_TIMEOUT_SECONDS", "30"))
SPEND_LOG_CLEANUP_REMAINING_COUNT_CAP: Final = int(os.getenv("SPEND_LOG_CLEANUP_REMAINING_COUNT_CAP", "100000"))
+SCHEDULED_JOB_SHUTDOWN_FINISH_TIMEOUT_SECONDS: Final = float(os.getenv("SCHEDULED_JOB_SHUTDOWN_FINISH_TIMEOUT_SECONDS", "5"))
+SCHEDULED_JOB_SHUTDOWN_CANCEL_TIMEOUT_SECONDS: Final = float(os.getenv("SCHEDULED_JOB_SHUTDOWN_CANCEL_TIMEOUT_SECONDS", "5"))
TOOL_SPEND_TOP_TOOLS: Final = 100
SPEND_LOG_PARTITION_INTERVAL: Final = os.getenv("SPEND_LOG_PARTITION_INTERVAL", "day")
SPEND_LOG_PARTITION_PRECREATE_AHEAD: Final = int(os.getenv("SPEND_LOG_PARTITION_PRECREATE_AHEAD", 7))
diff --git a/litellm/proxy/shutdown/scheduled_jobs.py b/litellm/proxy/shutdown/scheduled_jobs.py
index 5345d380112..cf4937780b8 100644
--- a/litellm/proxy/shutdown/scheduled_jobs.py
+++ b/litellm/proxy/shutdown/scheduled_jobs.py
@@ -7,9 +7,10 @@ from typing import Final, Protocol
from apscheduler.executors.asyncio import AsyncIOExecutor
from litellm._logging import verbose_proxy_logger
-
-JOB_FINISH_TIMEOUT_SECONDS: Final = 5.0
-JOB_CANCEL_TIMEOUT_SECONDS: Final = 5.0
+from litellm.constants import (
+ SCHEDULED_JOB_SHUTDOWN_CANCEL_TIMEOUT_SECONDS,
+ SCHEDULED_JOB_SHUTDOWN_FINISH_TIMEOUT_SECONDS,
+)
class StoppableScheduler(Protocol):
@@ -41,8 +42,8 @@ def pause_scheduled_jobs(scheduler: StoppableScheduler) -> None:
async def stop_in_flight_scheduler_jobs(scheduler: StoppableScheduler, executor: AwaitableAsyncIOExecutor) -> None:
"""
- Let in-flight jobs finish for up to JOB_FINISH_TIMEOUT_SECONDS, then stop the scheduler and
- wait, bounded by JOB_CANCEL_TIMEOUT_SECONDS, for the jobs it cancels.
+ Let in-flight jobs finish for up to SCHEDULED_JOB_SHUTDOWN_FINISH_TIMEOUT_SECONDS, then stop the scheduler and
+ wait, bounded by SCHEDULED_JOB_SHUTDOWN_CANCEL_TIMEOUT_SECONDS, for the jobs it cancels.
Must run before the database is disconnected: a write job that finishes needs its connection,
and a job's cancellation handler is what records the run's outcome.
@@ -52,19 +53,23 @@ async def stop_in_flight_scheduler_jobs(scheduler: StoppableScheduler, executor:
in_flight: Final = executor.in_flight_jobs()
if in_flight:
verbose_proxy_logger.info(
- "Waiting up to %ss for %d in-flight scheduled job(s) to finish", JOB_FINISH_TIMEOUT_SECONDS, len(in_flight)
+ "Waiting up to %ss for %d in-flight scheduled job(s) to finish",
+ SCHEDULED_JOB_SHUTDOWN_FINISH_TIMEOUT_SECONDS,
+ len(in_flight),
)
still_running: Final = (
- (await asyncio.wait(in_flight, timeout=JOB_FINISH_TIMEOUT_SECONDS))[1] if in_flight else frozenset()
+ (await asyncio.wait(in_flight, timeout=SCHEDULED_JOB_SHUTDOWN_FINISH_TIMEOUT_SECONDS))[1]
+ if in_flight
+ else frozenset()
)
scheduler.shutdown(wait=False)
if not still_running:
return
verbose_proxy_logger.info("Cancelling %d in-flight scheduled job(s) for shutdown", len(still_running))
- _done, pending = await asyncio.wait(still_running, timeout=JOB_CANCEL_TIMEOUT_SECONDS)
+ _done, pending = await asyncio.wait(still_running, timeout=SCHEDULED_JOB_SHUTDOWN_CANCEL_TIMEOUT_SECONDS)
if pending:
verbose_proxy_logger.warning(
"%d scheduled job(s) did not finish within %ss of cancellation; giving up on them",
len(pending),
- JOB_CANCEL_TIMEOUT_SECONDS,
+ SCHEDULED_JOB_SHUTDOWN_CANCEL_TIMEOUT_SECONDS,
)
diff --git a/tests/test_litellm/proxy/shutdown/test_scheduled_jobs.py b/tests/test_litellm/proxy/shutdown/test_scheduled_jobs.py
index 7defd6cef6c..1f94cee04ed 100644
--- a/tests/test_litellm/proxy/shutdown/test_scheduled_jobs.py
+++ b/tests/test_litellm/proxy/shutdown/test_scheduled_jobs.py
@@ -7,11 +7,11 @@ from datetime import datetime, timedelta
import pytest
from apscheduler.schedulers.asyncio import AsyncIOScheduler
-import litellm.proxy.shutdown.scheduled_jobs as scheduled_jobs
+from litellm.constants import SCHEDULED_JOB_SHUTDOWN_CANCEL_TIMEOUT_SECONDS
from litellm.proxy.shutdown.scheduled_jobs import (
AwaitableAsyncIOExecutor,
- stop_in_flight_scheduler_jobs,
pause_scheduled_jobs,
+ stop_in_flight_scheduler_jobs,
)
@@ -75,9 +75,8 @@ async def test_in_flight_jobs_observe_cancellation_before_shutdown_returns():
@pytest.mark.asyncio
-async def test_a_job_that_is_finishing_is_allowed_to_finish_rather_than_cancelled(monkeypatch):
+async def test_a_job_that_is_finishing_is_allowed_to_finish_rather_than_cancelled():
"""A spend write cancelled mid-commit drops the rows it popped, so short jobs get to finish first"""
- monkeypatch.setattr(scheduled_jobs, "JOB_FINISH_TIMEOUT_SECONDS", 2.0)
write = _Job(work_seconds=0.2)
stuck = _Job()
async with _running_scheduler(write, stuck) as (scheduler, executor):
@@ -99,16 +98,18 @@ async def test_every_in_flight_job_is_cancelled_not_only_the_first():
@pytest.mark.asyncio
-async def test_a_job_that_ignores_cancellation_is_abandoned_after_the_timeout(monkeypatch, caplog):
+async def test_a_job_that_ignores_cancellation_is_abandoned_after_the_timeout(caplog):
"""A job that swallows CancelledError must not hold the pod past its termination grace period"""
- monkeypatch.setattr(scheduled_jobs, "JOB_CANCEL_TIMEOUT_SECONDS", 0.05)
job = _Job(swallow_cancellation=True)
async with _running_scheduler(job) as (scheduler, executor):
with caplog.at_level(logging.WARNING, logger="LiteLLM Proxy"):
await stop_in_flight_scheduler_jobs(scheduler, executor)
assert job.events == ["cancelled"]
- assert "1 scheduled job(s) did not finish within 0.05s of cancellation" in caplog.text
+ assert (
+ f"1 scheduled job(s) did not finish within {SCHEDULED_JOB_SHUTDOWN_CANCEL_TIMEOUT_SECONDS}s of cancellation"
+ in caplog.text
+ )
@pytest.mark.asyncio
From 4e388e6aea52ad6c9c2939997f860689e69716aa Mon Sep 17 00:00:00 2001
From: yucheng
Date: Mon, 21 Sep 2026 19:37:28 +0000
Subject: [PATCH 050/160] refactor(proxy): inject scheduled job shutdown
timeouts
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
---
litellm/proxy/shutdown/scheduled_jobs.py | 20 ++++++++++++-------
.../proxy/shutdown/test_scheduled_jobs.py | 10 +++-------
2 files changed, 16 insertions(+), 14 deletions(-)
diff --git a/litellm/proxy/shutdown/scheduled_jobs.py b/litellm/proxy/shutdown/scheduled_jobs.py
index cf4937780b8..e920ce19eb9 100644
--- a/litellm/proxy/shutdown/scheduled_jobs.py
+++ b/litellm/proxy/shutdown/scheduled_jobs.py
@@ -40,10 +40,16 @@ def pause_scheduled_jobs(scheduler: StoppableScheduler) -> None:
scheduler.pause()
-async def stop_in_flight_scheduler_jobs(scheduler: StoppableScheduler, executor: AwaitableAsyncIOExecutor) -> None:
+async def stop_in_flight_scheduler_jobs(
+ scheduler: StoppableScheduler,
+ executor: AwaitableAsyncIOExecutor,
+ *,
+ finish_timeout_seconds: float = SCHEDULED_JOB_SHUTDOWN_FINISH_TIMEOUT_SECONDS,
+ cancel_timeout_seconds: float = SCHEDULED_JOB_SHUTDOWN_CANCEL_TIMEOUT_SECONDS,
+) -> None:
"""
- Let in-flight jobs finish for up to SCHEDULED_JOB_SHUTDOWN_FINISH_TIMEOUT_SECONDS, then stop the scheduler and
- wait, bounded by SCHEDULED_JOB_SHUTDOWN_CANCEL_TIMEOUT_SECONDS, for the jobs it cancels.
+ Let in-flight jobs finish for up to finish_timeout_seconds, then stop the scheduler and wait, bounded by
+ cancel_timeout_seconds, for the jobs it cancels.
Must run before the database is disconnected: a write job that finishes needs its connection,
and a job's cancellation handler is what records the run's outcome.
@@ -54,11 +60,11 @@ async def stop_in_flight_scheduler_jobs(scheduler: StoppableScheduler, executor:
if in_flight:
verbose_proxy_logger.info(
"Waiting up to %ss for %d in-flight scheduled job(s) to finish",
- SCHEDULED_JOB_SHUTDOWN_FINISH_TIMEOUT_SECONDS,
+ finish_timeout_seconds,
len(in_flight),
)
still_running: Final = (
- (await asyncio.wait(in_flight, timeout=SCHEDULED_JOB_SHUTDOWN_FINISH_TIMEOUT_SECONDS))[1]
+ (await asyncio.wait(in_flight, timeout=finish_timeout_seconds))[1]
if in_flight
else frozenset()
)
@@ -66,10 +72,10 @@ async def stop_in_flight_scheduler_jobs(scheduler: StoppableScheduler, executor:
if not still_running:
return
verbose_proxy_logger.info("Cancelling %d in-flight scheduled job(s) for shutdown", len(still_running))
- _done, pending = await asyncio.wait(still_running, timeout=SCHEDULED_JOB_SHUTDOWN_CANCEL_TIMEOUT_SECONDS)
+ _done, pending = await asyncio.wait(still_running, timeout=cancel_timeout_seconds)
if pending:
verbose_proxy_logger.warning(
"%d scheduled job(s) did not finish within %ss of cancellation; giving up on them",
len(pending),
- SCHEDULED_JOB_SHUTDOWN_CANCEL_TIMEOUT_SECONDS,
+ cancel_timeout_seconds,
)
diff --git a/tests/test_litellm/proxy/shutdown/test_scheduled_jobs.py b/tests/test_litellm/proxy/shutdown/test_scheduled_jobs.py
index 1f94cee04ed..fbce38db39f 100644
--- a/tests/test_litellm/proxy/shutdown/test_scheduled_jobs.py
+++ b/tests/test_litellm/proxy/shutdown/test_scheduled_jobs.py
@@ -7,7 +7,6 @@ from datetime import datetime, timedelta
import pytest
from apscheduler.schedulers.asyncio import AsyncIOScheduler
-from litellm.constants import SCHEDULED_JOB_SHUTDOWN_CANCEL_TIMEOUT_SECONDS
from litellm.proxy.shutdown.scheduled_jobs import (
AwaitableAsyncIOExecutor,
pause_scheduled_jobs,
@@ -80,7 +79,7 @@ async def test_a_job_that_is_finishing_is_allowed_to_finish_rather_than_cancelle
write = _Job(work_seconds=0.2)
stuck = _Job()
async with _running_scheduler(write, stuck) as (scheduler, executor):
- await stop_in_flight_scheduler_jobs(scheduler, executor)
+ await stop_in_flight_scheduler_jobs(scheduler, executor, finish_timeout_seconds=2.0)
assert write.events == ["committed", "finished"]
assert stuck.events == ["cancelled", "finished"]
@@ -103,13 +102,10 @@ async def test_a_job_that_ignores_cancellation_is_abandoned_after_the_timeout(ca
job = _Job(swallow_cancellation=True)
async with _running_scheduler(job) as (scheduler, executor):
with caplog.at_level(logging.WARNING, logger="LiteLLM Proxy"):
- await stop_in_flight_scheduler_jobs(scheduler, executor)
+ await stop_in_flight_scheduler_jobs(scheduler, executor, cancel_timeout_seconds=0.05)
assert job.events == ["cancelled"]
- assert (
- f"1 scheduled job(s) did not finish within {SCHEDULED_JOB_SHUTDOWN_CANCEL_TIMEOUT_SECONDS}s of cancellation"
- in caplog.text
- )
+ assert "1 scheduled job(s) did not finish within 0.05s of cancellation" in caplog.text
@pytest.mark.asyncio
From 9644032cb806a8bef55d1bcf4219ddec4886b758 Mon Sep 17 00:00:00 2001
From: yuneng
Date: Mon, 21 Sep 2026 19:37:45 +0000
Subject: [PATCH 051/160] refactor(ui): split complexity router form files and
cover advanced fields
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
---
.../add_model/ClassificationMethodConfig.tsx | 118 +--------
.../ClassifierPluginTimeoutField.tsx | 54 ++++
.../add_model/ClassifierTypeRadios.tsx | 89 +++++++
.../ComplexityRouterAdvancedSections.tsx | 240 +++++++++++++++++
.../add_model/ComplexityRouterConfig.test.tsx | 58 +++++
.../add_model/ComplexityRouterConfig.tsx | 210 +++------------
.../components/add_model/ReminderMarkers.tsx | 4 +-
.../add_model/add_auto_router_tab.tsx | 59 +----
.../build_complexity_router_config.test.ts | 16 ++
.../build_complexity_router_config.ts | 17 +-
.../complexity_router_builder_params.ts | 74 ++++++
...dit_auto_router_modal.integration.test.tsx | 71 +++++
.../edit_auto_router_modal.tsx | 243 +-----------------
.../hydrate_complexity_router_config.ts | 183 +++++++++++++
14 files changed, 839 insertions(+), 597 deletions(-)
create mode 100644 ui/litellm-dashboard/src/components/add_model/ClassifierPluginTimeoutField.tsx
create mode 100644 ui/litellm-dashboard/src/components/add_model/ClassifierTypeRadios.tsx
create mode 100644 ui/litellm-dashboard/src/components/add_model/ComplexityRouterAdvancedSections.tsx
create mode 100644 ui/litellm-dashboard/src/components/add_model/complexity_router_builder_params.ts
create mode 100644 ui/litellm-dashboard/src/components/edit_auto_router/hydrate_complexity_router_config.ts
diff --git a/ui/litellm-dashboard/src/components/add_model/ClassificationMethodConfig.tsx b/ui/litellm-dashboard/src/components/add_model/ClassificationMethodConfig.tsx
index 2564137fb02..b7a0fd67443 100644
--- a/ui/litellm-dashboard/src/components/add_model/ClassificationMethodConfig.tsx
+++ b/ui/litellm-dashboard/src/components/add_model/ClassificationMethodConfig.tsx
@@ -19,10 +19,9 @@ import HeuristicScoringConfig from "./HeuristicScoringConfig";
import ClassifierReasoningEffortSelect from "./ClassifierReasoningEffortSelect";
import ClassifierCircuitBreakerConfig from "./ClassifierCircuitBreakerConfig";
import ClassifierVisionConfig from "./ClassifierVisionConfig";
-import {
- getClassifierPluginTimeoutError,
- getHeuristicV2SuccessThresholdError,
-} from "./build_complexity_router_config";
+import { getHeuristicV2SuccessThresholdError } from "./build_complexity_router_config";
+import ClassifierPluginTimeoutField from "./ClassifierPluginTimeoutField";
+import ClassifierTypeRadios from "./ClassifierTypeRadios";
import type { ReasoningEffort } from "./complexity_router_tiers";
import { useComplexityScorerDefaults } from "@/app/(dashboard)/hooks/autoRouter/useComplexityScorerDefaults";
import {
@@ -64,7 +63,6 @@ const CLASSIFIER_CONTEXT_WINDOW_SIZE_ID = "classifier-context-window-size";
const CLASSIFIER_CONTEXT_BUDGET_CHARS_ID = "classifier-context-budget-chars";
const HYBRID_BOUNDARY_MARGIN_ID = "hybrid-boundary-margin";
const HEURISTIC_V2_SUCCESS_THRESHOLD_ID = "heuristic-v2-success-threshold";
-const CLASSIFIER_PLUGIN_TIMEOUT_ID = "classifier-plugin-timeout-ms";
const CUSTOM_PROMPT_WITH_HEURISTIC_FALLBACK =
"This router classifies with your own prompt, so the tier comes from whatever rubric it states. The four tier " +
@@ -208,84 +206,6 @@ export const InactiveHeuristicV2Threshold: React.FC void;
-}> = ({ value, classifierType, onTypeChange }) => {
- const scorerLocked = Boolean(value.custom_tier_set);
- const scorerLockedReason = restrictedBy(value, "heuristicClassifier")?.reason;
- return (
- onTypeChange(classifierType as ClassifierType)}
- className="w-full"
- >
-
-
-
-
-
- Heuristic {" "}
-
- (default), rule-based scoring with no API calls and <1ms latency
-
-
-
-
-
-
-
-
- Heuristic v2 {" "}
-
- uses bundled calibrated four-tier probabilities with no API call
-
-
-
-
-
-
-
- LLM Classifier {" "}
- calls a model to decide the tier (e.g. a small/fast model)
-
-
-
-
-
- JEV Classifier {" "}
- uses TypeSafe System One Choice to decide the tier
-
-
-
-
-
-
- Heuristic first {" "}
-
- scores locally, and only pays for the classifier when the score does not confidently land a cheap tier
-
-
-
-
-
-
-
-
- Hybrid {" "}
-
- keeps the local score at any tier, and only pays for the classifier when that score lands near a tier
- boundary
-
-
-
-
-
-
- );
-};
-
const ClassificationMethodConfig: React.FC = ({
value,
onChange,
@@ -472,37 +392,7 @@ const ClassificationMethodConfig: React.FC = ({
{classifierType === "custom" && (
-
-
- This router uses a custom classifier plugin set in config.yaml. Pick a classifier below to replace it.
-
-
- Classifier plugin timeout (ms)
-
-
- onChange({
- ...value,
- classifier_plugin_timeout_ms: event.target.value.trim() === "" ? undefined : Number(event.target.value),
- })
- }
- aria-invalid={Boolean(
- showValidationErrors && getClassifierPluginTimeoutError(classifierType, value.classifier_plugin_timeout_ms),
- )}
- />
-
- Time budget for the plugin call. On expiry the fallback path decides the tier.
-
- {showValidationErrors && getClassifierPluginTimeoutError(classifierType, value.classifier_plugin_timeout_ms) && (
-
- {getClassifierPluginTimeoutError(classifierType, value.classifier_plugin_timeout_ms)}
-
- )}
-
+
)}
{classifierType === "heuristic_v2" && (
diff --git a/ui/litellm-dashboard/src/components/add_model/ClassifierPluginTimeoutField.tsx b/ui/litellm-dashboard/src/components/add_model/ClassifierPluginTimeoutField.tsx
new file mode 100644
index 00000000000..45decf09a09
--- /dev/null
+++ b/ui/litellm-dashboard/src/components/add_model/ClassifierPluginTimeoutField.tsx
@@ -0,0 +1,54 @@
+import React from "react";
+import { Input } from "@/components/ui/input";
+import { Label } from "@/components/ui/label";
+import type { ComplexityRouterConfigValue } from "./ComplexityRouterConfig";
+import { getClassifierPluginTimeoutError } from "./build_complexity_router_config";
+
+const CLASSIFIER_PLUGIN_TIMEOUT_ID = "classifier-plugin-timeout-ms";
+
+interface ClassifierPluginTimeoutFieldProps {
+ value: ComplexityRouterConfigValue;
+ onChange: (value: ComplexityRouterConfigValue) => void;
+ showValidationErrors?: boolean;
+}
+
+const ClassifierPluginTimeoutField: React.FC = ({
+ value,
+ onChange,
+ showValidationErrors = false,
+}) => {
+ const error = getClassifierPluginTimeoutError("custom", value.classifier_plugin_timeout_ms);
+ return (
+
+
+ This router uses a custom classifier plugin set in config.yaml. Pick a classifier below to replace it.
+
+
+ Classifier plugin timeout (ms)
+
+
+ onChange({
+ ...value,
+ classifier_plugin_timeout_ms: event.target.value.trim() === "" ? undefined : Number(event.target.value),
+ })
+ }
+ aria-invalid={Boolean(showValidationErrors && error)}
+ />
+
+ Time budget for the plugin call. On expiry the fallback path decides the tier.
+
+ {showValidationErrors && error && (
+
+ {error}
+
+ )}
+
+ );
+};
+
+export default ClassifierPluginTimeoutField;
diff --git a/ui/litellm-dashboard/src/components/add_model/ClassifierTypeRadios.tsx b/ui/litellm-dashboard/src/components/add_model/ClassifierTypeRadios.tsx
new file mode 100644
index 00000000000..1602e19069a
--- /dev/null
+++ b/ui/litellm-dashboard/src/components/add_model/ClassifierTypeRadios.tsx
@@ -0,0 +1,89 @@
+import React from "react";
+import { Label } from "@/components/ui/label";
+import { RadioGroup, RadioGroupItem } from "@/components/ui/radio-group";
+import { SimpleTooltip } from "@/components/ui/tooltip";
+import type { ClassifierType } from "./classifier_types";
+import type { ComplexityRouterConfigValue } from "./ComplexityRouterConfig";
+import { restrictedBy } from "./TierRestrictions";
+
+interface ClassifierTypeRadiosProps {
+ value: ComplexityRouterConfigValue;
+ classifierType: ClassifierType;
+ onTypeChange: (classifierType: ClassifierType) => void;
+}
+
+const ClassifierTypeRadios: React.FC = ({ value, classifierType, onTypeChange }) => {
+ const scorerLocked = Boolean(value.custom_tier_set);
+ const scorerLockedReason = restrictedBy(value, "heuristicClassifier")?.reason;
+ return (
+ onTypeChange(nextType as ClassifierType)}
+ className="w-full"
+ >
+
+
+
+
+
+ Heuristic {" "}
+
+ (default), rule-based scoring with no API calls and <1ms latency
+
+
+
+
+
+
+
+
+ Heuristic v2 {" "}
+
+ uses bundled calibrated four-tier probabilities with no API call
+
+
+
+
+
+
+
+ LLM Classifier {" "}
+ calls a model to decide the tier (e.g. a small/fast model)
+
+
+
+
+
+ JEV Classifier {" "}
+ uses TypeSafe System One Choice to decide the tier
+
+
+
+
+
+
+ Heuristic first {" "}
+
+ scores locally, and only pays for the classifier when the score does not confidently land a cheap tier
+
+
+
+
+
+
+
+
+ Hybrid {" "}
+
+ keeps the local score at any tier, and only pays for the classifier when that score lands near a tier
+ boundary
+
+
+
+
+
+
+ );
+};
+
+export default ClassifierTypeRadios;
diff --git a/ui/litellm-dashboard/src/components/add_model/ComplexityRouterAdvancedSections.tsx b/ui/litellm-dashboard/src/components/add_model/ComplexityRouterAdvancedSections.tsx
new file mode 100644
index 00000000000..dd822907735
--- /dev/null
+++ b/ui/litellm-dashboard/src/components/add_model/ComplexityRouterAdvancedSections.tsx
@@ -0,0 +1,240 @@
+import React from "react";
+import { ChevronRight } from "lucide-react";
+import { Collapsible, CollapsibleContent, CollapsibleTrigger } from "@/components/ui/collapsible";
+import { Separator } from "@/components/ui/separator";
+import type { ModelGroup } from "@/components/llm_calls/fetch_models";
+import type { ComplexityRouterConfigValue } from "./ComplexityRouterConfig";
+import AdaptiveRoutingConfig from "./AdaptiveRoutingConfig";
+import ClassificationMethodConfig from "./ClassificationMethodConfig";
+import ContextWindowEscalationConfig from "./ContextWindowEscalationConfig";
+import ResponseFormatControls from "./ResponseFormatControls";
+import StallEscalationConfig from "./StallEscalationConfig";
+import { Restricted, restrictedBy } from "./TierRestrictions";
+import EscalationKeywords from "./EscalationKeywords";
+import KeywordTierRules, { type KeywordTierRule } from "./KeywordTierRules";
+import SemanticKeywordMatching from "./SemanticKeywordMatching";
+import CompressionControls from "./CompressionControls";
+import PlanModeOverrideControls from "./PlanModeOverrideControls";
+import { AffinityControls } from "./AffinityControls";
+import { ModalityRoutingControls } from "./ModalityRoutingControls";
+import HeuristicKeywordOverrides from "./HeuristicKeywordOverrides";
+import HousekeepingRoutingControls from "./HousekeepingRoutingControls";
+import ReminderMarkers from "./ReminderMarkers";
+import type { AutoRouterCompressionState } from "./buildAutoRouterCompression";
+import { activeTierName, type TierRow } from "./tier_rows";
+
+interface ComplexityRouterAdvancedSectionsProps {
+ value: ComplexityRouterConfigValue;
+ onChange: (value: ComplexityRouterConfigValue) => void;
+ forecast: boolean;
+ modelOptions: { value: string; label: string }[];
+ classifierEffortOptionsByModel: Record;
+ customTechnicalKeywords?: string[];
+ onCustomTechnicalKeywordsChange?: (keywords: string[]) => void;
+ showValidationErrors: boolean;
+ defaultModel?: string;
+ planModeTierOptions: { value: string; label: string }[];
+ keywordTierRules: KeywordTierRule[];
+ onKeywordTierRulesChange?: (rules: KeywordTierRule[]) => void;
+ semanticMatchingEnabled: boolean;
+ onSemanticMatchingEnabledChange?: (enabled: boolean) => void;
+ embeddingModel?: string;
+ onEmbeddingModelChange: (model: string) => void;
+ matchThreshold: number;
+ onMatchThresholdChange: (threshold: number) => void;
+ escalationKeywords: string[];
+ onEscalationKeywordsChange?: (keywords: string[]) => void;
+ autoRouterCompression: AutoRouterCompressionState;
+ onAutoRouterCompressionChange?: (state: AutoRouterCompressionState) => void;
+ modelInfo: ModelGroup[];
+ tierRows: TierRow[];
+ customTierSet: ComplexityRouterConfigValue["custom_tier_set"];
+}
+
+const ComplexityRouterAdvancedSections: React.FC = ({
+ value,
+ onChange,
+ forecast,
+ modelOptions,
+ classifierEffortOptionsByModel,
+ customTechnicalKeywords,
+ onCustomTechnicalKeywordsChange,
+ showValidationErrors,
+ defaultModel,
+ planModeTierOptions,
+ keywordTierRules,
+ onKeywordTierRulesChange,
+ semanticMatchingEnabled,
+ onSemanticMatchingEnabledChange,
+ embeddingModel,
+ onEmbeddingModelChange,
+ matchThreshold,
+ onMatchThresholdChange,
+ escalationKeywords,
+ onEscalationKeywordsChange,
+ autoRouterCompression,
+ onAutoRouterCompressionChange,
+ modelInfo,
+ tierRows,
+ customTierSet,
+}) => {
+ const sections = [
+ ...(!forecast
+ ? [
+ {
+ key: "classifier",
+ label: Advanced: Classification Method ,
+ children: (
+
+ ),
+ },
+ ]
+ : []),
+ ...(!forecast
+ ? [
+ {
+ key: "keyword-overrides",
+ label: Advanced: Heuristic Keyword Overrides ,
+ children: ,
+ },
+ ]
+ : []),
+ {
+ key: "adaptive",
+ label: Advanced: Adaptive Routing ,
+ children: (
+
+
+
+ ),
+ },
+ {
+ key: "affinity",
+ label: Advanced: Affinity ,
+ children: ,
+ },
+ {
+ key: "modality",
+ label: Advanced: Modality Routing ,
+ children: ,
+ },
+ {
+ key: "plan-mode",
+ label: Advanced: Plan-Mode Override ,
+ children: ,
+ },
+ {
+ key: "housekeeping",
+ label: Advanced: Housekeeping Routing ,
+ children: ,
+ },
+ {
+ key: "reminder-markers",
+ label: Advanced: Reminder Markers ,
+ children: ,
+ },
+ {
+ key: "context-window",
+ label: Advanced: Context Window Escalation ,
+ children: ,
+ },
+ {
+ key: "stall-escalation",
+ label: Advanced: Stalled Task Escalation ,
+ children: (
+
+
+
+ ),
+ },
+ {
+ key: "response",
+ label: Advanced: Response Format ,
+ children: ,
+ },
+ ...(onEscalationKeywordsChange
+ ? [
+ {
+ key: "escalation",
+ label: Advanced: Escalation Keywords ,
+ children: (
+
+
+
+ ),
+ },
+ ]
+ : []),
+ ...(onAutoRouterCompressionChange
+ ? [
+ {
+ key: "compression",
+ label: Advanced: Compression ,
+ children: ,
+ },
+ ]
+ : []),
+ ...(onKeywordTierRulesChange || onSemanticMatchingEnabledChange
+ ? [
+ {
+ key: "keyword-semantic",
+ label: Advanced: Keyword/Semantic Matching ,
+ children: (
+ <>
+ {onKeywordTierRulesChange && (
+
+ )}
+ {onKeywordTierRulesChange && onSemanticMatchingEnabledChange && }
+ {onSemanticMatchingEnabledChange && (
+
+ )}
+ >
+ ),
+ },
+ ]
+ : []),
+ ];
+
+ return (
+ <>
+ {sections
+ .filter(({ key }) => !forecast || !["adaptive", "context-window", "escalation"].includes(key))
+ .map(({ key, label, children }) => (
+
+
+
+ {label}
+
+ {children}
+
+ ))}
+ >
+ );
+};
+
+export default ComplexityRouterAdvancedSections;
diff --git a/ui/litellm-dashboard/src/components/add_model/ComplexityRouterConfig.test.tsx b/ui/litellm-dashboard/src/components/add_model/ComplexityRouterConfig.test.tsx
index 70658b787f0..9a8577100df 100644
--- a/ui/litellm-dashboard/src/components/add_model/ComplexityRouterConfig.test.tsx
+++ b/ui/litellm-dashboard/src/components/add_model/ComplexityRouterConfig.test.tsx
@@ -96,6 +96,64 @@ describe("ComplexityRouterConfig", () => {
expect(screen.queryByText("Classifier Model")).not.toBeInTheDocument();
});
+ it("shows heuristic advanced sections and hides keyword overrides for capability classifiers", () => {
+ const { rerender } = renderWithProviders( );
+
+ expect(screen.getByText("Advanced: Heuristic Keyword Overrides")).toBeInTheDocument();
+ expect(screen.getByText("Advanced: Housekeeping Routing")).toBeInTheDocument();
+ expect(screen.getByText("Advanced: Reminder Markers")).toBeInTheDocument();
+
+ const capabilityValue = { ...defaultValue, classifier_type: "capability" as const };
+ rerender( );
+ expect(screen.queryByText("Advanced: Heuristic Keyword Overrides")).not.toBeInTheDocument();
+
+ });
+
+ it.each([
+ ["custom", true],
+ ["heuristic", false],
+ ] as const)("shows plugin timeout only for %s classifiers", (classifierType, visible) => {
+ renderWithProviders(
+ ,
+ );
+ fireEvent.click(screen.getByText("Advanced: Classification Method"));
+ if (visible) {
+ expect(screen.getByLabelText("Classifier plugin timeout (ms)")).toBeInTheDocument();
+ } else {
+ expect(screen.queryByLabelText("Classifier plugin timeout (ms)")).not.toBeInTheDocument();
+ }
+ });
+
+ it.each([true, false])("shows reminder marker validation only when requested: %s", (showValidationErrors) => {
+ const value = { ...defaultValue, reminder_markers: [{ open: "", close: "x" }] };
+ renderWithProviders(
+ ,
+ );
+ fireEvent.click(screen.getByText("Advanced: Reminder Markers"));
+ const validation = screen.queryByText(/needs both/i);
+ if (showValidationErrors) {
+ expect(validation).toBeInTheDocument();
+ } else {
+ expect(validation).not.toBeInTheDocument();
+ }
+ });
+
+ it("disables housekeeping sentinels when cheapest-tier routing is off", () => {
+ renderWithProviders(
+ ,
+ );
+ fireEvent.click(screen.getByText("Advanced: Housekeeping Routing"));
+ const sentinelInput = screen.getByRole("combobox", { name: "e.g., conversation title" });
+ expect(sentinelInput).toBeDisabled();
+ });
+
it("should toggle returning the raw model name", async () => {
const user = userEvent.setup();
const onChange = vi.fn();
diff --git a/ui/litellm-dashboard/src/components/add_model/ComplexityRouterConfig.tsx b/ui/litellm-dashboard/src/components/add_model/ComplexityRouterConfig.tsx
index a97e7a77a21..acf6b62a95a 100644
--- a/ui/litellm-dashboard/src/components/add_model/ComplexityRouterConfig.tsx
+++ b/ui/litellm-dashboard/src/components/add_model/ComplexityRouterConfig.tsx
@@ -2,21 +2,17 @@ import RoutingOptions from "./RoutingOptions";
import type { JevClassifierConfig } from "./jev_classifier_config";
import { type ClassifierType } from "./classifier_types";
export { type ClassifierType, usesLlmClassifier, usesClassifierContext } from "./classifier_types";
-import PlanModeOverrideControls from "./PlanModeOverrideControls";
import ForecastClassifierConfig, { ForecastSolverModels } from "./ForecastClassifierConfig";
import { isForecastClassifier, type CapabilitySettings, type FuseSettings } from "./forecast_classifier_config";
import { SimpleTooltip } from "@/components/ui/tooltip";
import { MultiSelect } from "@/components/shared/MultiSelect";
import DefaultModelField from "./DefaultModelField";
-import { ChevronRight, Info, Plus, Trash2, X } from "lucide-react";
+import { Info, Plus, Trash2, X } from "lucide-react";
-import { AffinityControls } from "./AffinityControls";
import NonReasoningTierToggle from "./NonReasoningTierToggle";
import TierConfigIntro from "./TierConfigIntro";
import TierRowSelect from "./TierRowSelect";
-import { ModalityRoutingControls } from "./ModalityRoutingControls";
import { Card, CardContent } from "@/components/ui/card";
-import { Collapsible, CollapsibleContent, CollapsibleTrigger } from "@/components/ui/collapsible";
import { InputGroup, InputGroupAddon, InputGroupButton, InputGroupInput } from "@/components/ui/input-group";
import { Separator } from "@/components/ui/separator";
import { Button } from "@/components/ui/button";
@@ -39,12 +35,8 @@ import {
} from "./tier_rows";
import React from "react";
import { ModelGroup } from "@/components/llm_calls/fetch_models";
-import AdaptiveRoutingConfig from "./AdaptiveRoutingConfig";
-import ClassificationMethodConfig, { InactiveHeuristicV2Threshold } from "./ClassificationMethodConfig";
-import ContextWindowEscalationConfig from "./ContextWindowEscalationConfig";
-import ResponseFormatControls from "./ResponseFormatControls";
-import StallEscalationConfig from "./StallEscalationConfig";
-import { Restricted, restrictedBy } from "./TierRestrictions";
+import { InactiveHeuristicV2Threshold } from "./ClassificationMethodConfig";
+import ComplexityRouterAdvancedSections from "./ComplexityRouterAdvancedSections";
import { type TierSetAction, applyTierSetAction, setFallbackTier } from "./tier_set_actions";
import {
ReasoningEffort,
@@ -56,16 +48,10 @@ import {
tierRowLabel,
} from "./complexity_router_tiers";
import TierModelEffortRows from "./TierModelEffortRows";
-import EscalationKeywords from "./EscalationKeywords";
-import KeywordTierRules, { KeywordTierRule } from "./KeywordTierRules";
-import SemanticKeywordMatching from "./SemanticKeywordMatching";
+import { KeywordTierRule } from "./KeywordTierRules";
import { type DimensionWeights, type TierBoundaries, type TokenThresholds } from "./heuristic_scoring_knobs";
import { type CustomDimensionRow } from "./custom_dimensions";
-import CompressionControls from "./CompressionControls";
import { type AutoRouterCompressionState, DEFAULT_AUTO_ROUTER_COMPRESSION } from "./buildAutoRouterCompression";
-import HeuristicKeywordOverrides from "./HeuristicKeywordOverrides";
-import HousekeepingRoutingControls from "./HousekeepingRoutingControls";
-import ReminderMarkers from "./ReminderMarkers";
import { type ReminderMarkerPair } from "./build_complexity_router_config";
export type { DimensionWeights, TierBoundaries, TokenThresholds };
@@ -782,167 +768,33 @@ const ComplexityRouterConfig: React.FC = ({
>
)}
- {[
- ...(!forecast
- ? [
- {
- key: "classifier",
- label:
Advanced: Classification Method ,
- children: (
-
- ),
- },
- ]
- : []),
- ...(!forecast
- ? [
- {
- key: "keyword-overrides",
- label: (
-
Advanced: Heuristic Keyword Overrides
- ),
- children:
,
- },
- ]
- : []),
- {
- key: "adaptive",
- label:
Advanced: Adaptive Routing ,
- children: (
-
-
-
- ),
- },
- {
- key: "affinity",
- label:
Advanced: Affinity ,
- children:
,
- },
- {
- key: "modality",
- label:
Advanced: Modality Routing ,
- children:
,
- },
- {
- key: "plan-mode",
- label:
Advanced: Plan-Mode Override ,
- children: (
-
- ),
- },
- {
- key: "housekeeping",
- label:
Advanced: Housekeeping Routing ,
- children:
,
- },
- {
- key: "reminder-markers",
- label:
Advanced: Reminder Markers ,
- children:
,
- },
- {
- key: "context-window",
- label:
Advanced: Context Window Escalation ,
- children:
,
- },
- {
- key: "stall-escalation",
- label:
Advanced: Stalled Task Escalation ,
- children: (
-
-
-
- ),
- },
- {
- key: "response",
- label:
Advanced: Response Format ,
- children:
,
- },
- ...(onEscalationKeywordsChange
- ? [
- {
- key: "escalation",
- label:
Advanced: Escalation Keywords ,
- children: (
-
-
-
- ),
- },
- ]
- : []),
- ...(onAutoRouterCompressionChange
- ? [
- {
- key: "compression",
- label:
Advanced: Compression ,
- children: (
-
- ),
- },
- ]
- : []),
- ...(onKeywordTierRulesChange || onSemanticMatchingEnabledChange
- ? [
- {
- key: "keyword-semantic",
- label: (
-
Advanced: Keyword/Semantic Matching
- ),
- children: (
- <>
- {onKeywordTierRulesChange && (
-
- )}
- {onKeywordTierRulesChange && onSemanticMatchingEnabledChange &&
}
- {onSemanticMatchingEnabledChange && (
-
- )}
- >
- ),
- },
- ]
- : []),
- ]
- .filter(({ key }) => !forecast || !["adaptive", "context-window", "escalation"].includes(key))
- .map(({ key, label, children }) => (
-
-
-
- {label}
-
- {children}
-
- ))}
+
diff --git a/ui/litellm-dashboard/src/components/add_model/ReminderMarkers.tsx b/ui/litellm-dashboard/src/components/add_model/ReminderMarkers.tsx
index 452f4dc727f..970394291d4 100644
--- a/ui/litellm-dashboard/src/components/add_model/ReminderMarkers.tsx
+++ b/ui/litellm-dashboard/src/components/add_model/ReminderMarkers.tsx
@@ -30,14 +30,13 @@ const ReminderMarkers: React.FC<{
{markers.map((marker, index) => (
-
+
Opening delimiter
update(index, { open: event.target.value })}
@@ -49,7 +48,6 @@ const ReminderMarkers: React.FC<{
update(index, { close: event.target.value })}
diff --git a/ui/litellm-dashboard/src/components/add_model/add_auto_router_tab.tsx b/ui/litellm-dashboard/src/components/add_model/add_auto_router_tab.tsx
index 2fa28f761e3..f805a5b5511 100644
--- a/ui/litellm-dashboard/src/components/add_model/add_auto_router_tab.tsx
+++ b/ui/litellm-dashboard/src/components/add_model/add_auto_router_tab.tsx
@@ -28,10 +28,6 @@ import ComplexityRouterConfig, {
effectiveClassifierType,
usesLlmClassifier,
heuristicScoringRole,
- DEFAULT_ADAPTIVE_WEIGHTS,
- DEFAULT_SESSION_AFFINITY,
- DEFAULT_DEPLOYMENT_AFFINITY,
- DEFAULT_TIER_DISTANCE_PENALTY,
} from "./ComplexityRouterConfig";
import { KeywordTierRule } from "./KeywordTierRules";
import { customDimensionsError } from "./custom_dimensions";
@@ -57,6 +53,7 @@ import {
getTierLabelsError,
dryRunRejection,
} from "./build_complexity_router_config";
+import { builderParamsFromValue } from "./complexity_router_builder_params";
import { activeTierName, activeTierRows, getCustomTierRowsError, resolveComplexityDefaultModel } from "./tier_rows";
import { tierRowLabel } from "./complexity_router_tiers";
import { buildAutoRouterTestTargets, AutoRouterTestTarget } from "./build_auto_router_test_targets";
@@ -403,65 +400,13 @@ const AddAutoRouterTab: React.FC
= ({
);
const complexityRouterConfigParams: BuildComplexityRouterConfigParams = {
- tiers: complexityRouterConfig.tiers,
- enableNonReasoningTier: complexityRouterConfig.enable_non_reasoning_tier,
- customTierSet: complexityRouterConfig.custom_tier_set,
- defaultModel: complexityRouterConfig.default_model,
- planModeMinTier: complexityRouterConfig.plan_mode_min_tier,
- classificationPrompt: complexityRouterConfig.classification_prompt,
- classificationExamples: complexityRouterConfig.classification_examples,
- heuristicFirstMaxTier: complexityRouterConfig.heuristic_first_max_tier,
- hybridBoundaryMargin: complexityRouterConfig.hybrid_boundary_margin,
- classificationMode: complexityRouterConfig.classification_mode,
- tierLabels: complexityRouterConfig.tier_labels,
- classifierType: complexityRouterConfig.classifier_type,
- jevClassifierConfig: complexityRouterConfig.jev_classifier_config,
- heuristicV2SuccessThreshold: complexityRouterConfig.heuristic_v2_success_threshold,
- capabilityClassifierConfig: complexityRouterConfig.capability_classifier_config,
- llmV2Config: complexityRouterConfig.llm_v2_config,
- classifierLlmConfig: complexityRouterConfig.classifier_llm_config,
- classifierContextWindowSize: complexityRouterConfig.classifier_context_window_size,
- classifierContextBudgetChars: complexityRouterConfig.classifier_context_budget_chars,
- classifierContextPerTurnChars: complexityRouterConfig.classifier_context_per_turn_chars,
- classifierContextIncludeAssistantTurns: complexityRouterConfig.classifier_context_include_assistant_turns,
- classifierFallback: complexityRouterConfig.classifier_fallback,
- sessionAffinity: complexityRouterConfig.session_affinity ?? DEFAULT_SESSION_AFFINITY,
- modalityRouting: complexityRouterConfig.modality_routing ?? false,
- modalityPinOverride: complexityRouterConfig.modality_pin_override ?? false,
- deploymentAffinity: complexityRouterConfig.deployment_affinity ?? DEFAULT_DEPLOYMENT_AFFINITY,
+ ...builderParamsFromValue(complexityRouterConfig),
customTechnicalKeywords,
keywordTierRules,
semanticMatchingEnabled,
embeddingModel,
matchThreshold,
escalationKeywords,
- stallEscalationEnabled: complexityRouterConfig.stall_escalation_enabled,
- stallEscalationWindow: complexityRouterConfig.stall_escalation_window,
- stallEscalationRepeatThreshold: complexityRouterConfig.stall_escalation_repeat_threshold,
- adaptive: complexityRouterConfig.adaptive ?? false,
- adaptiveWeights: complexityRouterConfig.adaptive_weights ?? DEFAULT_ADAPTIVE_WEIGHTS,
- tierDistancePenalty: complexityRouterConfig.tier_distance_penalty ?? DEFAULT_TIER_DISTANCE_PENALTY,
- adaptiveEligible: complexityRouterConfig.adaptive_eligible ?? "all",
- returnRawModelName: complexityRouterConfig.return_raw_model_name ?? false,
- tierModelParams: complexityRouterConfig.tier_model_params,
- tierBoundaries: complexityRouterConfig.tier_boundaries,
- tokenThresholds: complexityRouterConfig.token_thresholds,
- dimensionWeights: complexityRouterConfig.dimension_weights,
- customDimensions: complexityRouterConfig.custom_dimensions,
- reasoningOverrideMinScore: complexityRouterConfig.reasoning_override_min_score,
- enableContextWindowEscalation: complexityRouterConfig.enable_context_window_escalation,
- contextWindowEscalationBuffer: complexityRouterConfig.context_window_escalation_buffer,
- sessionAffinityTtlSeconds: complexityRouterConfig.session_affinity_ttl_seconds,
- codeKeywords: complexityRouterConfig.code_keywords,
- reasoningKeywords: complexityRouterConfig.reasoning_keywords,
- technicalKeywords: complexityRouterConfig.technical_keywords,
- simpleKeywords: complexityRouterConfig.simple_keywords,
- planModePatterns: complexityRouterConfig.plan_mode_patterns,
- routeHousekeepingToCheapestTier: complexityRouterConfig.route_housekeeping_to_cheapest_tier,
- housekeepingPatterns: complexityRouterConfig.housekeeping_patterns,
- reminderMarkers: complexityRouterConfig.reminder_markers,
- maxTokensFromTierModel: complexityRouterConfig.max_tokens_from_tier_model,
- classifierPluginTimeoutMs: complexityRouterConfig.classifier_plugin_timeout_ms,
};
const submitRecommendedRouter = async (name: string) => {
diff --git a/ui/litellm-dashboard/src/components/add_model/build_complexity_router_config.test.ts b/ui/litellm-dashboard/src/components/add_model/build_complexity_router_config.test.ts
index cf5164d91c4..6db2b8213d1 100644
--- a/ui/litellm-dashboard/src/components/add_model/build_complexity_router_config.test.ts
+++ b/ui/litellm-dashboard/src/components/add_model/build_complexity_router_config.test.ts
@@ -1531,6 +1531,22 @@ describe("advanced complexity router fields", () => {
expect(payload).not.toHaveProperty("classifier_plugin_timeout_ms");
});
+ it.each([
+ "code_keywords",
+ "reasoning_keywords",
+ "technical_keywords",
+ "simple_keywords",
+ "plan_mode_patterns",
+ "route_housekeeping_to_cheapest_tier",
+ "housekeeping_patterns",
+ "reminder_markers",
+ "max_tokens_from_tier_model",
+ "classifier_plugin_timeout_ms",
+ ])("omits unset advanced field %s", (key) => {
+ const payload = buildComplexityRouterConfig(baseParams);
+ expect(payload).not.toHaveProperty(key);
+ });
+
it("validates marker pairs and custom classifier timeout", () => {
expect(getReminderMarkersError([{ open: " ", close: " " }])).toContain("different");
expect(getReminderMarkersError([{ open: "", close: " " }])).toContain("needs both");
diff --git a/ui/litellm-dashboard/src/components/add_model/build_complexity_router_config.ts b/ui/litellm-dashboard/src/components/add_model/build_complexity_router_config.ts
index 96fbe7f2c2e..ab71b6b3cce 100644
--- a/ui/litellm-dashboard/src/components/add_model/build_complexity_router_config.ts
+++ b/ui/litellm-dashboard/src/components/add_model/build_complexity_router_config.ts
@@ -744,6 +744,16 @@ export const buildComplexityRouterConfig = ({
open: open.trim().toLowerCase(),
close: close.trim().toLowerCase(),
}));
+ const cleanedLists = Object.fromEntries(
+ Object.entries({
+ code_keywords: cleanList(codeKeywords),
+ reasoning_keywords: cleanList(reasoningKeywords),
+ technical_keywords: cleanList(technicalKeywords),
+ simple_keywords: cleanList(simpleKeywords),
+ plan_mode_patterns: cleanList(planModePatterns),
+ housekeeping_patterns: cleanList(housekeepingPatterns),
+ }).filter(([, list]) => list !== undefined),
+ );
const supportsOpeningPrompt = !customTierSet && !forecast && usesLlmClassifier(effectiveType);
const payload: ComplexityRouterConfigPayload = {
@@ -812,13 +822,8 @@ export const buildComplexityRouterConfig = ({
...(sessionAffinityTtlSeconds !== undefined && {
session_affinity_ttl_seconds: sessionAffinityTtlSeconds,
}),
- ...(cleanList(codeKeywords) && { code_keywords: cleanList(codeKeywords) }),
- ...(cleanList(reasoningKeywords) && { reasoning_keywords: cleanList(reasoningKeywords) }),
- ...(cleanList(technicalKeywords) && { technical_keywords: cleanList(technicalKeywords) }),
- ...(cleanList(simpleKeywords) && { simple_keywords: cleanList(simpleKeywords) }),
- ...(cleanList(planModePatterns) && { plan_mode_patterns: cleanList(planModePatterns) }),
+ ...cleanedLists,
...(routeHousekeepingToCheapestTier === false && { route_housekeeping_to_cheapest_tier: false }),
- ...(cleanList(housekeepingPatterns) && { housekeeping_patterns: cleanList(housekeepingPatterns) }),
...(cleanedReminderMarkers && cleanedReminderMarkers.length > 0 && { reminder_markers: cleanedReminderMarkers }),
...(maxTokensFromTierModel === false && { max_tokens_from_tier_model: false }),
...(classifierType === "custom" &&
diff --git a/ui/litellm-dashboard/src/components/add_model/complexity_router_builder_params.ts b/ui/litellm-dashboard/src/components/add_model/complexity_router_builder_params.ts
new file mode 100644
index 00000000000..124a85ce9a3
--- /dev/null
+++ b/ui/litellm-dashboard/src/components/add_model/complexity_router_builder_params.ts
@@ -0,0 +1,74 @@
+import type { BuildComplexityRouterConfigParams } from "./build_complexity_router_config";
+import type { ComplexityRouterConfigValue } from "./ComplexityRouterConfig";
+import {
+ DEFAULT_ADAPTIVE_WEIGHTS,
+ DEFAULT_DEPLOYMENT_AFFINITY,
+ DEFAULT_SESSION_AFFINITY,
+ DEFAULT_TIER_DISTANCE_PENALTY,
+} from "./ComplexityRouterConfig";
+
+export const builderParamsFromValue = (
+ value: ComplexityRouterConfigValue,
+): Omit<
+ BuildComplexityRouterConfigParams,
+ | "customTechnicalKeywords"
+ | "keywordTierRules"
+ | "semanticMatchingEnabled"
+ | "embeddingModel"
+ | "matchThreshold"
+ | "escalationKeywords"
+> => ({
+ tiers: value.tiers,
+ enableNonReasoningTier: value.enable_non_reasoning_tier,
+ customTierSet: value.custom_tier_set,
+ defaultModel: value.default_model,
+ planModeMinTier: value.plan_mode_min_tier,
+ classificationPrompt: value.classification_prompt,
+ classificationExamples: value.classification_examples,
+ heuristicFirstMaxTier: value.heuristic_first_max_tier,
+ hybridBoundaryMargin: value.hybrid_boundary_margin,
+ classificationMode: value.classification_mode,
+ tierLabels: value.tier_labels,
+ classifierType: value.classifier_type,
+ jevClassifierConfig: value.jev_classifier_config,
+ heuristicV2SuccessThreshold: value.heuristic_v2_success_threshold,
+ capabilityClassifierConfig: value.capability_classifier_config,
+ llmV2Config: value.llm_v2_config,
+ classifierLlmConfig: value.classifier_llm_config,
+ classifierContextWindowSize: value.classifier_context_window_size,
+ classifierContextBudgetChars: value.classifier_context_budget_chars,
+ classifierContextPerTurnChars: value.classifier_context_per_turn_chars,
+ classifierContextIncludeAssistantTurns: value.classifier_context_include_assistant_turns,
+ classifierFallback: value.classifier_fallback,
+ sessionAffinity: value.session_affinity ?? DEFAULT_SESSION_AFFINITY,
+ sessionAffinityTtlSeconds: value.session_affinity_ttl_seconds,
+ modalityRouting: value.modality_routing ?? false,
+ modalityPinOverride: value.modality_pin_override ?? false,
+ deploymentAffinity: value.deployment_affinity ?? DEFAULT_DEPLOYMENT_AFFINITY,
+ adaptive: value.adaptive ?? false,
+ adaptiveWeights: value.adaptive_weights ?? DEFAULT_ADAPTIVE_WEIGHTS,
+ tierDistancePenalty: value.tier_distance_penalty ?? DEFAULT_TIER_DISTANCE_PENALTY,
+ adaptiveEligible: value.adaptive_eligible ?? "all",
+ returnRawModelName: value.return_raw_model_name ?? false,
+ tierBoundaries: value.tier_boundaries,
+ tokenThresholds: value.token_thresholds,
+ dimensionWeights: value.dimension_weights,
+ customDimensions: value.custom_dimensions,
+ reasoningOverrideMinScore: value.reasoning_override_min_score,
+ tierModelParams: value.tier_model_params,
+ enableContextWindowEscalation: value.enable_context_window_escalation,
+ contextWindowEscalationBuffer: value.context_window_escalation_buffer,
+ stallEscalationEnabled: value.stall_escalation_enabled,
+ stallEscalationWindow: value.stall_escalation_window,
+ stallEscalationRepeatThreshold: value.stall_escalation_repeat_threshold,
+ codeKeywords: value.code_keywords,
+ reasoningKeywords: value.reasoning_keywords,
+ technicalKeywords: value.technical_keywords,
+ simpleKeywords: value.simple_keywords,
+ planModePatterns: value.plan_mode_patterns,
+ routeHousekeepingToCheapestTier: value.route_housekeeping_to_cheapest_tier,
+ housekeepingPatterns: value.housekeeping_patterns,
+ reminderMarkers: value.reminder_markers,
+ maxTokensFromTierModel: value.max_tokens_from_tier_model,
+ classifierPluginTimeoutMs: value.classifier_plugin_timeout_ms,
+});
diff --git a/ui/litellm-dashboard/src/components/edit_auto_router/edit_auto_router_modal.integration.test.tsx b/ui/litellm-dashboard/src/components/edit_auto_router/edit_auto_router_modal.integration.test.tsx
index 34db61483cf..aa6cf92ceb9 100644
--- a/ui/litellm-dashboard/src/components/edit_auto_router/edit_auto_router_modal.integration.test.tsx
+++ b/ui/litellm-dashboard/src/components/edit_auto_router/edit_auto_router_modal.integration.test.tsx
@@ -347,6 +347,77 @@ describe("EditAutoRouterModal keyword matching", () => {
});
});
+describe("EditAutoRouterModal advanced field round trips", () => {
+ const storedAdvancedConfig = {
+ ...STORED_CONFIG,
+ route_housekeeping_to_cheapest_tier: false,
+ housekeeping_patterns: ["conversation title"],
+ reminder_markers: [{ open: "", close: " " }],
+ max_tokens_from_tier_model: false,
+ };
+
+ const renderAdvancedModal = (props: Partial> = {}) =>
+ renderModal({
+ modelData: {
+ ...MODEL_DATA,
+ litellm_params: { ...MODEL_DATA.litellm_params, complexity_router_config: storedAdvancedConfig },
+ },
+ ...props,
+ });
+
+ beforeEach(() => {
+ modelPatchUpdateCall.mockClear();
+ });
+
+ it("hydrates housekeeping and reminder fields, then omits the default max-token value after editing", async () => {
+ const user = userEvent.setup();
+ renderAdvancedModal();
+
+ await user.click(await screen.findByText("Advanced: Housekeeping Routing"));
+ expect(screen.getByRole("switch", { name: "Route housekeeping calls to the cheapest tier" })).not.toBeChecked();
+ expect(screen.getByRole("combobox", { name: "e.g., conversation title" })).toHaveValue("");
+
+ await user.click(screen.getByText("Advanced: Reminder Markers"));
+ expect(screen.getByLabelText("Opening delimiter")).toHaveValue("");
+ expect(screen.getByLabelText("Closing delimiter")).toHaveValue(" ");
+
+ await user.click(screen.getByText("Advanced: Response Format"));
+ const maxTokensSwitch = screen.getByRole("switch", { name: "Cap max_tokens at the tier model's output ceiling" });
+ await user.click(maxTokensSwitch);
+ await user.click(screen.getByRole("button", { name: /save changes/i }));
+ await waitFor(() => expect(modelPatchUpdateCall).toHaveBeenCalledOnce());
+
+ expect(savedConfig()).not.toHaveProperty("max_tokens_from_tier_model");
+ expect(savedConfig()).toMatchObject({
+ route_housekeeping_to_cheapest_tier: false,
+ housekeeping_patterns: ["conversation title"],
+ reminder_markers: [{ open: "", close: " " }],
+ });
+ });
+
+ it("does not PATCH when the edit is cancelled", async () => {
+ const user = userEvent.setup();
+ const onCancel = vi.fn();
+ renderAdvancedModal({ onCancel });
+ await user.click(screen.getByRole("button", { name: /cancel/i }));
+ expect(onCancel).toHaveBeenCalledOnce();
+ expect(modelPatchUpdateCall).not.toHaveBeenCalled();
+ });
+
+ it("preserves all stored advanced fields through an untouched save", async () => {
+ const user = userEvent.setup();
+ renderAdvancedModal();
+ await user.click(screen.getByRole("button", { name: /save changes/i }));
+ await waitFor(() => expect(modelPatchUpdateCall).toHaveBeenCalledOnce());
+ expect(savedConfig()).toMatchObject({
+ route_housekeeping_to_cheapest_tier: false,
+ housekeeping_patterns: ["conversation title"],
+ reminder_markers: [{ open: "", close: " " }],
+ max_tokens_from_tier_model: false,
+ });
+ });
+});
+
describe("EditAutoRouterModal classifier context window", () => {
beforeEach(() => {
modelPatchUpdateCall.mockClear();
diff --git a/ui/litellm-dashboard/src/components/edit_auto_router/edit_auto_router_modal.tsx b/ui/litellm-dashboard/src/components/edit_auto_router/edit_auto_router_modal.tsx
index 476207553f7..e3a2df39b88 100644
--- a/ui/litellm-dashboard/src/components/edit_auto_router/edit_auto_router_modal.tsx
+++ b/ui/litellm-dashboard/src/components/edit_auto_router/edit_auto_router_modal.tsx
@@ -1,13 +1,9 @@
import AutoRouterClassifierTabs from "../add_model/AutoRouterClassifierTabs";
import { usesClassifierContext } from "../add_model/classifier_types";
-import { defaultJevClassifierConfig, jevClassifierConfigSchema } from "../add_model/jev_classifier_config";
-import type { StoredComplexityRouterConfig } from "../add_model/build_complexity_router_config";
export type { StoredComplexityRouterConfig } from "../add_model/build_complexity_router_config";
import {
getForecastConfigError,
isForecastClassifier,
- capabilitySettingsSchema,
- fuseSettingsSchema,
} from "../add_model/forecast_classifier_config";
import React, { useEffect, useMemo, useState } from "react";
import {
@@ -30,13 +26,10 @@ import ModelChoiceCombobox, { type ModelChoice } from "../add_model/ModelChoiceC
import { modelAvailableCall, modelPatchUpdateCall, validateAutoRouterConfig } from "../networking";
import { fetchAutoRouterModels, fetchAvailableModels, ModelGroup } from "@/components/llm_calls/fetch_models";
import RouterConfigBuilder, { type RouterConfig, serializeRouterConfig } from "../add_model/RouterConfigBuilder";
-import { hydrateTierModelParams } from "../add_model/complexity_router_tiers";
import {
- type ActiveTierSet,
CUSTOM_TIER_OMITTED_KEYS,
activeTierRows,
getCustomTierRowsError,
- tierParamsByRowId,
resolveComplexityDefaultModel,
} from "../add_model/tier_rows";
import { isComplexityRouter } from "../add_model/auto_router_strategies";
@@ -53,10 +46,6 @@ import {
getSemanticConfigError,
getPlanModeTierError,
getTierLabelsError,
- hydrateBuiltInTiers,
- hydrateCustomTierSet,
- hydratePlanModeMinTier,
- hydrateTierLabels,
dryRunRejection,
} from "../add_model/build_complexity_router_config";
import { KeywordTierRule } from "../add_model/KeywordTierRules";
@@ -68,22 +57,14 @@ import {
hydrateAutoRouterCompression,
} from "../add_model/buildAutoRouterCompression";
import { hydrateKeywordTierRules } from "../add_model/complexity_router_keywords";
-import { customDimensionsError, hydrateCustomDimensions } from "../add_model/custom_dimensions";
-import {
- hydrateDimensionWeights,
- hydrateReasoningOverrideMinScore,
- hydrateTierBoundaries,
- hydrateTokenThresholds,
-} from "../add_model/heuristic_scoring_knobs";
+import { customDimensionsError } from "../add_model/custom_dimensions";
import ComplexityRouterConfig, {
ComplexityRouterConfigValue,
effectiveClassifierType,
heuristicScoringRole,
- DEFAULT_ADAPTIVE_WEIGHTS,
- DEFAULT_SESSION_AFFINITY,
- DEFAULT_DEPLOYMENT_AFFINITY,
- DEFAULT_TIER_DISTANCE_PENALTY,
} from "../add_model/ComplexityRouterConfig";
+import { builderParamsFromValue } from "../add_model/complexity_router_builder_params";
+import { hydrateComplexityRouterConfig, hydratePinnedDefaultModel } from "./hydrate_complexity_router_config";
import {
Dialog,
DialogContent,
@@ -106,151 +87,7 @@ interface EditAutoRouterModalProps {
// Keys this modal rewrites from its own form state on save. Anything absent from this set is
// carried through untouched from the stored config, so a key only belongs here once the modal
// actually renders a control that can set it.
-
-/**
- * The stored complexity_router_config as form state. Every key in MANAGED_COMPLEXITY_ROUTER_KEYS is
- * rewritten from this state on save, so a key missing here is silently dropped from the saved config.
- */
-export const hydrateComplexityRouterConfig = (
- parsedConfig: StoredComplexityRouterConfig,
- complexityRouterDefaultModel: string | null | undefined,
-): ComplexityRouterConfigValue => {
- const stringList = (input: unknown): string[] | undefined =>
- Array.isArray(input) ? input.filter((item): item is string => typeof item === "string") : undefined;
- const builtIn = hydrateBuiltInTiers(parsedConfig.tiers, parsedConfig.enable_non_reasoning_tier);
- const { tiers: hydratedTiers, enable_non_reasoning_tier } = builtIn;
- const custom_tier_set = hydrateCustomTierSet(parsedConfig);
- const activeTiers = { ...builtIn, custom_tier_set };
-
- return {
- tiers: hydratedTiers,
- enable_non_reasoning_tier,
- custom_tier_set,
- tier_model_params: tierParamsByRowId(
- hydrateTierModelParams(parsedConfig.tiers, parsedConfig.tier_model_configs),
- activeTierRows(activeTiers),
- ),
- default_model: hydratePinnedDefaultModel(parsedConfig.default_model, complexityRouterDefaultModel, activeTiers),
- plan_mode_min_tier: hydratePlanModeMinTier(parsedConfig.plan_mode_min_tier, custom_tier_set),
- tier_labels: hydrateTierLabels(parsedConfig.tier_labels),
- classifier_type: parsedConfig.classifier_type || "heuristic",
- heuristic_v2_success_threshold:
- typeof parsedConfig.heuristic_v2_success_threshold === "number"
- ? parsedConfig.heuristic_v2_success_threshold
- : undefined,
- capability_classifier_config: capabilitySettingsSchema.safeParse(parsedConfig.capability_classifier_config).data,
- llm_v2_config: fuseSettingsSchema.safeParse(parsedConfig.llm_v2_config).data,
- classifier_llm_config: parsedConfig.classifier_type === "jev" ? undefined : parsedConfig.classifier_llm_config,
- jev_classifier_config:
- parsedConfig.classifier_type === "jev"
- ? jevClassifierConfigSchema.safeParse(parsedConfig.jev_classifier_config ?? {}).data ??
- defaultJevClassifierConfig()
- : undefined,
- classifier_context_window_size:
- typeof parsedConfig.classifier_context_window_size === "number"
- ? parsedConfig.classifier_context_window_size
- : undefined,
- classifier_context_budget_chars:
- typeof parsedConfig.classifier_context_budget_chars === "number"
- ? parsedConfig.classifier_context_budget_chars
- : undefined,
- classifier_context_per_turn_chars:
- typeof parsedConfig.classifier_context_per_turn_chars === "number"
- ? parsedConfig.classifier_context_per_turn_chars
- : undefined,
- classifier_context_include_assistant_turns:
- typeof parsedConfig.classifier_context_include_assistant_turns === "boolean"
- ? parsedConfig.classifier_context_include_assistant_turns
- : undefined,
- classifier_fallback:
- parsedConfig.classifier_fallback === "default_model" || parsedConfig.classifier_fallback === "heuristic"
- ? parsedConfig.classifier_fallback
- : undefined,
- classification_prompt:
- typeof parsedConfig.classification_prompt === "string" && parsedConfig.classification_prompt.trim() !== ""
- ? parsedConfig.classification_prompt
- : undefined,
- classification_examples:
- typeof parsedConfig.classification_examples === "string" && parsedConfig.classification_examples.trim() !== ""
- ? parsedConfig.classification_examples
- : undefined,
- heuristic_first_max_tier:
- typeof parsedConfig.heuristic_first_max_tier === "string" && parsedConfig.heuristic_first_max_tier.trim() !== ""
- ? parsedConfig.heuristic_first_max_tier
- : undefined,
- hybrid_boundary_margin:
- typeof parsedConfig.hybrid_boundary_margin === "number" ? parsedConfig.hybrid_boundary_margin : undefined,
- classification_mode:
- parsedConfig.classification_mode === "user_turn" || parsedConfig.classification_mode === "every_request"
- ? parsedConfig.classification_mode
- : undefined,
- tier_boundaries: hydrateTierBoundaries(parsedConfig.tier_boundaries),
- token_thresholds: hydrateTokenThresholds(parsedConfig.token_thresholds),
- dimension_weights: hydrateDimensionWeights(parsedConfig.dimension_weights),
- custom_dimensions: hydrateCustomDimensions(parsedConfig.custom_dimensions),
- reasoning_override_min_score: hydrateReasoningOverrideMinScore(parsedConfig.reasoning_override_min_score),
- session_affinity:
- typeof parsedConfig.session_affinity === "boolean" ? parsedConfig.session_affinity : DEFAULT_SESSION_AFFINITY,
- session_affinity_ttl_seconds:
- typeof parsedConfig.session_affinity_ttl_seconds === "number" &&
- Number.isFinite(parsedConfig.session_affinity_ttl_seconds)
- ? parsedConfig.session_affinity_ttl_seconds
- : undefined,
- modality_routing: typeof parsedConfig.modality_routing === "boolean" ? parsedConfig.modality_routing : false,
- modality_pin_override:
- typeof parsedConfig.modality_pin_override === "boolean" ? parsedConfig.modality_pin_override : false,
- deployment_affinity:
- typeof parsedConfig.deployment_affinity === "boolean"
- ? parsedConfig.deployment_affinity
- : DEFAULT_DEPLOYMENT_AFFINITY,
- adaptive: parsedConfig.adaptive || false,
- adaptive_weights: parsedConfig.adaptive_weights,
- tier_distance_penalty: parsedConfig.tier_distance_penalty,
- adaptive_eligible: parsedConfig.adaptive_eligible || "all",
- return_raw_model_name: parsedConfig.return_raw_model_name || false,
- enable_context_window_escalation:
- typeof parsedConfig.enable_context_window_escalation === "boolean"
- ? parsedConfig.enable_context_window_escalation
- : undefined,
- context_window_escalation_buffer:
- typeof parsedConfig.context_window_escalation_buffer === "number"
- ? parsedConfig.context_window_escalation_buffer
- : undefined,
- stall_escalation_enabled: parsedConfig.stall_escalation_enabled === true || undefined,
- stall_escalation_window:
- typeof parsedConfig.stall_escalation_window === "number" ? parsedConfig.stall_escalation_window : undefined,
- stall_escalation_repeat_threshold:
- typeof parsedConfig.stall_escalation_repeat_threshold === "number"
- ? parsedConfig.stall_escalation_repeat_threshold
- : undefined,
- code_keywords: stringList(parsedConfig.code_keywords),
- reasoning_keywords: stringList(parsedConfig.reasoning_keywords),
- technical_keywords: stringList(parsedConfig.technical_keywords),
- simple_keywords: stringList(parsedConfig.simple_keywords),
- plan_mode_patterns: stringList(parsedConfig.plan_mode_patterns),
- route_housekeeping_to_cheapest_tier:
- typeof parsedConfig.route_housekeeping_to_cheapest_tier === "boolean"
- ? parsedConfig.route_housekeeping_to_cheapest_tier
- : undefined,
- housekeeping_patterns: stringList(parsedConfig.housekeeping_patterns),
- reminder_markers: Array.isArray(parsedConfig.reminder_markers)
- ? parsedConfig.reminder_markers.filter(
- (pair): pair is { open: string; close: string } =>
- typeof pair === "object" &&
- pair !== null &&
- typeof (pair as { open?: unknown }).open === "string" &&
- typeof (pair as { close?: unknown }).close === "string",
- )
- : undefined,
- max_tokens_from_tier_model:
- typeof parsedConfig.max_tokens_from_tier_model === "boolean" ? parsedConfig.max_tokens_from_tier_model : undefined,
- classifier_plugin_timeout_ms:
- typeof parsedConfig.classifier_plugin_timeout_ms === "number" && Number.isFinite(parsedConfig.classifier_plugin_timeout_ms)
- ? parsedConfig.classifier_plugin_timeout_ms
- : undefined,
- };
-};
-
+export { hydrateComplexityRouterConfig, hydratePinnedDefaultModel };
export const MANAGED_COMPLEXITY_ROUTER_KEYS = new Set([
"tiers",
"enable_non_reasoning_tier",
@@ -324,24 +161,6 @@ const toRecord = (value: unknown): Record => {
: {};
};
-// A pin lives in two places: complexity_router_config.default_model (this UI's own marker, added
-// by PR #36615) and litellm_params.complexity_router_default_model (what the backend reads). Only
-// the marker proves an operator picked it, because before #36615 every save wrote a tier-derived
-// value into litellm_params. So with no marker, a litellm_params value counts as a pin only when
-// it diverges from what the tiers alone derive; a match stays unpinned and keeps tracking tiers.
-export const hydratePinnedDefaultModel = (
- storedConfigDefaultModel: unknown,
- litellmParamsDefaultModel: string | null | undefined,
- activeTiers: ActiveTierSet,
-): string | undefined => {
- if (typeof storedConfigDefaultModel === "string" && storedConfigDefaultModel.trim()) {
- return storedConfigDefaultModel;
- }
- const tierDerived = resolveComplexityDefaultModel(activeTiers);
- const externalOverride = litellmParamsDefaultModel?.trim();
- return externalOverride && externalOverride !== tierDerived ? externalOverride : undefined;
-};
-
export interface KeywordMatchingState {
keywordTierRules: KeywordTierRule[];
escalationKeywords: string[];
@@ -377,65 +196,13 @@ export const buildUpdatedComplexityRouterConfig = (
);
const builderParams: BuildComplexityRouterConfigParams = {
- tiers: value.tiers,
- enableNonReasoningTier: value.enable_non_reasoning_tier,
- customTierSet: value.custom_tier_set,
- defaultModel: value.default_model,
- planModeMinTier: value.plan_mode_min_tier,
- classificationPrompt: value.classification_prompt,
- classificationExamples: value.classification_examples,
- heuristicFirstMaxTier: value.heuristic_first_max_tier,
- hybridBoundaryMargin: value.hybrid_boundary_margin,
- classificationMode: value.classification_mode,
- tierLabels: value.tier_labels,
- classifierType: value.classifier_type,
- jevClassifierConfig: value.jev_classifier_config,
- heuristicV2SuccessThreshold: value.heuristic_v2_success_threshold,
- capabilityClassifierConfig: value.capability_classifier_config,
- llmV2Config: value.llm_v2_config,
- classifierLlmConfig: value.classifier_llm_config,
- classifierContextWindowSize: value.classifier_context_window_size,
- classifierContextBudgetChars: value.classifier_context_budget_chars,
- classifierContextPerTurnChars: value.classifier_context_per_turn_chars,
- classifierContextIncludeAssistantTurns: value.classifier_context_include_assistant_turns,
- classifierFallback: value.classifier_fallback,
- sessionAffinity: value.session_affinity ?? DEFAULT_SESSION_AFFINITY,
- sessionAffinityTtlSeconds: value.session_affinity_ttl_seconds,
- modalityRouting: value.modality_routing ?? false,
- modalityPinOverride: value.modality_pin_override ?? false,
- deploymentAffinity: value.deployment_affinity ?? DEFAULT_DEPLOYMENT_AFFINITY,
+ ...builderParamsFromValue(value),
customTechnicalKeywords: customTechnicalKeywords ?? [],
keywordTierRules: keywordMatching?.keywordTierRules ?? [],
semanticMatchingEnabled: keywordMatching?.semanticMatchingEnabled ?? false,
embeddingModel: keywordMatching?.embeddingModel,
matchThreshold: keywordMatching?.matchThreshold ?? DEFAULT_MATCH_THRESHOLD,
escalationKeywords: keywordMatching?.escalationKeywords ?? [],
- adaptive: value.adaptive ?? false,
- adaptiveWeights: value.adaptive_weights ?? DEFAULT_ADAPTIVE_WEIGHTS,
- tierDistancePenalty: value.tier_distance_penalty ?? DEFAULT_TIER_DISTANCE_PENALTY,
- adaptiveEligible: value.adaptive_eligible ?? "all",
- returnRawModelName: value.return_raw_model_name ?? false,
- tierBoundaries: value.tier_boundaries,
- tokenThresholds: value.token_thresholds,
- dimensionWeights: value.dimension_weights,
- customDimensions: value.custom_dimensions,
- reasoningOverrideMinScore: value.reasoning_override_min_score,
- tierModelParams: value.tier_model_params,
- enableContextWindowEscalation: value.enable_context_window_escalation,
- contextWindowEscalationBuffer: value.context_window_escalation_buffer,
- stallEscalationEnabled: value.stall_escalation_enabled,
- stallEscalationWindow: value.stall_escalation_window,
- stallEscalationRepeatThreshold: value.stall_escalation_repeat_threshold,
- codeKeywords: value.code_keywords,
- reasoningKeywords: value.reasoning_keywords,
- technicalKeywords: value.technical_keywords,
- simpleKeywords: value.simple_keywords,
- planModePatterns: value.plan_mode_patterns,
- routeHousekeepingToCheapestTier: value.route_housekeeping_to_cheapest_tier,
- housekeepingPatterns: value.housekeeping_patterns,
- reminderMarkers: value.reminder_markers,
- maxTokensFromTierModel: value.max_tokens_from_tier_model,
- classifierPluginTimeoutMs: value.classifier_plugin_timeout_ms,
};
const built = buildComplexityRouterConfig(builderParams);
diff --git a/ui/litellm-dashboard/src/components/edit_auto_router/hydrate_complexity_router_config.ts b/ui/litellm-dashboard/src/components/edit_auto_router/hydrate_complexity_router_config.ts
new file mode 100644
index 00000000000..c7c7bcb3382
--- /dev/null
+++ b/ui/litellm-dashboard/src/components/edit_auto_router/hydrate_complexity_router_config.ts
@@ -0,0 +1,183 @@
+import { defaultJevClassifierConfig, jevClassifierConfigSchema } from "../add_model/jev_classifier_config";
+import { capabilitySettingsSchema, fuseSettingsSchema } from "../add_model/forecast_classifier_config";
+import type { StoredComplexityRouterConfig } from "../add_model/build_complexity_router_config";
+import {
+ hydrateBuiltInTiers,
+ hydrateCustomTierSet,
+ hydratePlanModeMinTier,
+ hydrateTierLabels,
+} from "../add_model/build_complexity_router_config";
+import { hydrateTierModelParams } from "../add_model/complexity_router_tiers";
+import { hydrateCustomDimensions } from "../add_model/custom_dimensions";
+import {
+ hydrateDimensionWeights,
+ hydrateReasoningOverrideMinScore,
+ hydrateTierBoundaries,
+ hydrateTokenThresholds,
+} from "../add_model/heuristic_scoring_knobs";
+import type { ComplexityRouterConfigValue } from "../add_model/ComplexityRouterConfig";
+import { DEFAULT_DEPLOYMENT_AFFINITY, DEFAULT_SESSION_AFFINITY } from "../add_model/ComplexityRouterConfig";
+import {
+ type ActiveTierSet,
+ activeTierRows,
+ tierParamsByRowId,
+ resolveComplexityDefaultModel,
+} from "../add_model/tier_rows";
+
+const isReminderMarkerPair = (
+ input: unknown,
+): input is { open: string; close: string } =>
+ typeof input === "object" &&
+ input !== null &&
+ "open" in input &&
+ "close" in input &&
+ typeof input.open === "string" &&
+ typeof input.close === "string";
+
+const stringList = (input: unknown): string[] | undefined =>
+ Array.isArray(input) ? input.filter((item): item is string => typeof item === "string") : undefined;
+
+export const hydratePinnedDefaultModel = (
+ storedConfigDefaultModel: unknown,
+ litellmParamsDefaultModel: string | null | undefined,
+ activeTiers: ActiveTierSet,
+): string | undefined => {
+ if (typeof storedConfigDefaultModel === "string" && storedConfigDefaultModel.trim()) {
+ return storedConfigDefaultModel;
+ }
+ const tierDerived = resolveComplexityDefaultModel(activeTiers);
+ const externalOverride = litellmParamsDefaultModel?.trim();
+ return externalOverride && externalOverride !== tierDerived ? externalOverride : undefined;
+};
+
+export const hydrateComplexityRouterConfig = (
+ parsedConfig: StoredComplexityRouterConfig,
+ complexityRouterDefaultModel: string | null | undefined,
+): ComplexityRouterConfigValue => {
+ const builtIn = hydrateBuiltInTiers(parsedConfig.tiers, parsedConfig.enable_non_reasoning_tier);
+ const { tiers: hydratedTiers, enable_non_reasoning_tier } = builtIn;
+ const custom_tier_set = hydrateCustomTierSet(parsedConfig);
+ const activeTiers = { ...builtIn, custom_tier_set };
+
+ return {
+ tiers: hydratedTiers,
+ enable_non_reasoning_tier,
+ custom_tier_set,
+ tier_model_params: tierParamsByRowId(
+ hydrateTierModelParams(parsedConfig.tiers, parsedConfig.tier_model_configs),
+ activeTierRows(activeTiers),
+ ),
+ default_model: hydratePinnedDefaultModel(parsedConfig.default_model, complexityRouterDefaultModel, activeTiers),
+ plan_mode_min_tier: hydratePlanModeMinTier(parsedConfig.plan_mode_min_tier, custom_tier_set),
+ tier_labels: hydrateTierLabels(parsedConfig.tier_labels),
+ classifier_type: parsedConfig.classifier_type || "heuristic",
+ heuristic_v2_success_threshold:
+ typeof parsedConfig.heuristic_v2_success_threshold === "number"
+ ? parsedConfig.heuristic_v2_success_threshold
+ : undefined,
+ capability_classifier_config: capabilitySettingsSchema.safeParse(parsedConfig.capability_classifier_config).data,
+ llm_v2_config: fuseSettingsSchema.safeParse(parsedConfig.llm_v2_config).data,
+ classifier_llm_config: parsedConfig.classifier_type === "jev" ? undefined : parsedConfig.classifier_llm_config,
+ jev_classifier_config:
+ parsedConfig.classifier_type === "jev"
+ ? jevClassifierConfigSchema.safeParse(parsedConfig.jev_classifier_config ?? {}).data ??
+ defaultJevClassifierConfig()
+ : undefined,
+ classifier_context_window_size:
+ typeof parsedConfig.classifier_context_window_size === "number"
+ ? parsedConfig.classifier_context_window_size
+ : undefined,
+ classifier_context_budget_chars:
+ typeof parsedConfig.classifier_context_budget_chars === "number"
+ ? parsedConfig.classifier_context_budget_chars
+ : undefined,
+ classifier_context_per_turn_chars:
+ typeof parsedConfig.classifier_context_per_turn_chars === "number"
+ ? parsedConfig.classifier_context_per_turn_chars
+ : undefined,
+ classifier_context_include_assistant_turns:
+ typeof parsedConfig.classifier_context_include_assistant_turns === "boolean"
+ ? parsedConfig.classifier_context_include_assistant_turns
+ : undefined,
+ classifier_fallback:
+ parsedConfig.classifier_fallback === "default_model" || parsedConfig.classifier_fallback === "heuristic"
+ ? parsedConfig.classifier_fallback
+ : undefined,
+ classification_prompt:
+ typeof parsedConfig.classification_prompt === "string" && parsedConfig.classification_prompt.trim() !== ""
+ ? parsedConfig.classification_prompt
+ : undefined,
+ classification_examples:
+ typeof parsedConfig.classification_examples === "string" && parsedConfig.classification_examples.trim() !== ""
+ ? parsedConfig.classification_examples
+ : undefined,
+ heuristic_first_max_tier:
+ typeof parsedConfig.heuristic_first_max_tier === "string" && parsedConfig.heuristic_first_max_tier.trim() !== ""
+ ? parsedConfig.heuristic_first_max_tier
+ : undefined,
+ hybrid_boundary_margin:
+ typeof parsedConfig.hybrid_boundary_margin === "number" ? parsedConfig.hybrid_boundary_margin : undefined,
+ classification_mode:
+ parsedConfig.classification_mode === "user_turn" || parsedConfig.classification_mode === "every_request"
+ ? parsedConfig.classification_mode
+ : undefined,
+ tier_boundaries: hydrateTierBoundaries(parsedConfig.tier_boundaries),
+ token_thresholds: hydrateTokenThresholds(parsedConfig.token_thresholds),
+ dimension_weights: hydrateDimensionWeights(parsedConfig.dimension_weights),
+ custom_dimensions: hydrateCustomDimensions(parsedConfig.custom_dimensions),
+ reasoning_override_min_score: hydrateReasoningOverrideMinScore(parsedConfig.reasoning_override_min_score),
+ session_affinity:
+ typeof parsedConfig.session_affinity === "boolean" ? parsedConfig.session_affinity : DEFAULT_SESSION_AFFINITY,
+ session_affinity_ttl_seconds:
+ typeof parsedConfig.session_affinity_ttl_seconds === "number" &&
+ Number.isFinite(parsedConfig.session_affinity_ttl_seconds)
+ ? parsedConfig.session_affinity_ttl_seconds
+ : undefined,
+ modality_routing: typeof parsedConfig.modality_routing === "boolean" ? parsedConfig.modality_routing : false,
+ modality_pin_override:
+ typeof parsedConfig.modality_pin_override === "boolean" ? parsedConfig.modality_pin_override : false,
+ deployment_affinity:
+ typeof parsedConfig.deployment_affinity === "boolean"
+ ? parsedConfig.deployment_affinity
+ : DEFAULT_DEPLOYMENT_AFFINITY,
+ adaptive: parsedConfig.adaptive || false,
+ adaptive_weights: parsedConfig.adaptive_weights,
+ tier_distance_penalty: parsedConfig.tier_distance_penalty,
+ adaptive_eligible: parsedConfig.adaptive_eligible || "all",
+ return_raw_model_name: parsedConfig.return_raw_model_name || false,
+ enable_context_window_escalation:
+ typeof parsedConfig.enable_context_window_escalation === "boolean"
+ ? parsedConfig.enable_context_window_escalation
+ : undefined,
+ context_window_escalation_buffer:
+ typeof parsedConfig.context_window_escalation_buffer === "number"
+ ? parsedConfig.context_window_escalation_buffer
+ : undefined,
+ stall_escalation_enabled: parsedConfig.stall_escalation_enabled === true || undefined,
+ stall_escalation_window:
+ typeof parsedConfig.stall_escalation_window === "number" ? parsedConfig.stall_escalation_window : undefined,
+ stall_escalation_repeat_threshold:
+ typeof parsedConfig.stall_escalation_repeat_threshold === "number"
+ ? parsedConfig.stall_escalation_repeat_threshold
+ : undefined,
+ code_keywords: stringList(parsedConfig.code_keywords),
+ reasoning_keywords: stringList(parsedConfig.reasoning_keywords),
+ technical_keywords: stringList(parsedConfig.technical_keywords),
+ simple_keywords: stringList(parsedConfig.simple_keywords),
+ plan_mode_patterns: stringList(parsedConfig.plan_mode_patterns),
+ route_housekeeping_to_cheapest_tier:
+ typeof parsedConfig.route_housekeeping_to_cheapest_tier === "boolean"
+ ? parsedConfig.route_housekeeping_to_cheapest_tier
+ : undefined,
+ housekeeping_patterns: stringList(parsedConfig.housekeeping_patterns),
+ reminder_markers: Array.isArray(parsedConfig.reminder_markers)
+ ? parsedConfig.reminder_markers.filter(isReminderMarkerPair)
+ : undefined,
+ max_tokens_from_tier_model:
+ typeof parsedConfig.max_tokens_from_tier_model === "boolean" ? parsedConfig.max_tokens_from_tier_model : undefined,
+ classifier_plugin_timeout_ms:
+ typeof parsedConfig.classifier_plugin_timeout_ms === "number" && Number.isFinite(parsedConfig.classifier_plugin_timeout_ms)
+ ? parsedConfig.classifier_plugin_timeout_ms
+ : undefined,
+ };
+};
From 68074da1d1658e2c1a8337e090ceccf8488994c4 Mon Sep 17 00:00:00 2001
From: Joshua Valluru <326636767+joshua-berri@users.noreply.github.com>
Date: Mon, 21 Sep 2026 12:39:47 -0700
Subject: [PATCH 052/160] test(mcp): cover stable server ordering and sort
priorities
---
.../test_mcp_management_endpoints.py | 83 ++++++++++---------
.../_components/mcp_servers.test.tsx | 72 +++++++++++++++-
2 files changed, 112 insertions(+), 43 deletions(-)
diff --git a/tests/test_litellm/proxy/management_endpoints/test_mcp_management_endpoints.py b/tests/test_litellm/proxy/management_endpoints/test_mcp_management_endpoints.py
index a526bfc90aa..b745c35ba1f 100644
--- a/tests/test_litellm/proxy/management_endpoints/test_mcp_management_endpoints.py
+++ b/tests/test_litellm/proxy/management_endpoints/test_mcp_management_endpoints.py
@@ -6,7 +6,7 @@ import logging
from contextlib import ExitStack
from datetime import datetime, timedelta
from types import SimpleNamespace
-from typing import List, Optional
+from typing import Final, List, Optional
from unittest.mock import AsyncMock, MagicMock, patch
import pytest
@@ -1539,54 +1539,57 @@ class TestTeamScopedMCPServerAccess:
class TestFetchAllMCPServersOrdering:
- def test_display_order_is_case_insensitive_name_then_id(self):
- servers = [
- generate_mock_mcp_server_db_record(server_id="s-2", alias="github"),
- generate_mock_mcp_server_db_record(server_id="s-1", alias="github"),
- generate_mock_mcp_server_db_record(server_id="s-0", alias="Slack"),
- generate_mock_mcp_server_db_record(server_id="s-3", alias="confluence"),
- ]
+ def test_display_order_is_case_insensitive_name_then_id(self) -> None:
+ servers: Final = (
+ LiteLLM_MCPServerTable(server_id="s-2", server_name="GitHub", alias="aaa", transport=MCPTransport.http),
+ LiteLLM_MCPServerTable(server_id="s-1", alias="github", transport=MCPTransport.http),
+ LiteLLM_MCPServerTable(server_id="s-0", server_name="Slack", alias="zzz", transport=MCPTransport.http),
+ LiteLLM_MCPServerTable(server_id="confluence", server_name="", alias="", transport=MCPTransport.http),
+ )
- ordered = sorted(servers, key=mgmt_endpoints._mcp_server_display_order)
- assert [s.server_id for s in ordered] == ["s-3", "s-1", "s-2", "s-0"]
+ ordered: Final = sorted(servers, key=mgmt_endpoints._mcp_server_display_order)
+ assert [s.server_id for s in ordered] == ["confluence", "s-1", "s-2", "s-0"]
+ @pytest.mark.parametrize("team_id", [None, "team-1"])
+ @pytest.mark.parametrize("reverse", [False, True])
@pytest.mark.asyncio
- async def test_list_is_sorted_by_display_name_regardless_of_resolution_order(self):
- """The registry resolves ids through a set, so the response must impose its own order."""
- mock_user_auth = generate_mock_user_api_key_auth(
+ async def test_list_is_sorted_by_display_name_regardless_of_resolution_order(
+ self, team_id: str | None, reverse: bool
+ ) -> None:
+ mock_user_auth: Final = generate_mock_user_api_key_auth(
user_role=LitellmUserRoles.PROXY_ADMIN,
user_id="admin_user",
)
- first_order = [
+ servers: Final = (
generate_mock_mcp_server_db_record(server_id="s-zeta", alias="zeta"),
generate_mock_mcp_server_db_record(server_id="s-alpha", alias="Alpha"),
generate_mock_mcp_server_db_record(server_id="s-mid", alias="mid"),
- ]
- second_order = list(reversed(first_order))
-
- for resolved in (first_order, second_order):
- mock_manager = MagicMock()
- mock_manager.get_all_allowed_mcp_servers = AsyncMock(return_value=resolved)
- with (
- patch( # test-quality-ok: the route reads a module-global manager with no injection seam
- "litellm.proxy.management_endpoints.mcp_management_endpoints.global_mcp_server_manager",
- mock_manager,
- ),
- patch( # test-quality-ok: admin view is derived from module-global proxy settings
- "litellm.proxy.management_endpoints.mcp_management_endpoints._user_has_admin_view",
- return_value=True,
- ),
- patch( # test-quality-ok: auth contexts need a live prisma client
- "litellm.proxy.management_endpoints.mcp_management_endpoints.build_effective_auth_contexts",
- AsyncMock(return_value=[mock_user_auth]),
- ),
- ):
- from litellm.proxy.management_endpoints.mcp_management_endpoints import (
- fetch_all_mcp_servers,
- )
-
- result = await fetch_all_mcp_servers(user_api_key_dict=mock_user_auth)
- assert [s.server_id for s in result] == ["s-alpha", "s-mid", "s-zeta"]
+ )
+ resolved: Final = list(reversed(servers) if reverse else servers)
+ mock_manager: Final = MagicMock()
+ mock_manager.get_all_allowed_mcp_servers = AsyncMock(return_value=resolved)
+ with (
+ patch( # test-quality-ok: the route reads a module-global manager with no injection seam
+ "litellm.proxy.management_endpoints.mcp_management_endpoints.global_mcp_server_manager",
+ mock_manager,
+ ),
+ patch( # test-quality-ok: admin view is derived from module-global proxy settings
+ "litellm.proxy.management_endpoints.mcp_management_endpoints._user_has_admin_view",
+ return_value=True,
+ ),
+ patch( # test-quality-ok: auth contexts need a live prisma client
+ "litellm.proxy.management_endpoints.mcp_management_endpoints.build_effective_auth_contexts",
+ AsyncMock(return_value=[mock_user_auth]),
+ ),
+ patch( # test-quality-ok: isolate the route's ordering from team database resolution
+ "litellm.proxy.management_endpoints.mcp_management_endpoints._get_team_scoped_mcp_server_list",
+ AsyncMock(return_value=resolved),
+ ),
+ ):
+ result: Final = await mgmt_endpoints.fetch_all_mcp_servers(
+ user_api_key_dict=mock_user_auth, team_id=team_id
+ )
+ assert [s.server_id for s in result] == ["s-alpha", "s-mid", "s-zeta"]
@pytest.mark.asyncio
async def test_restricted_virtual_key_cannot_use_team_id_filter(self):
diff --git a/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/mcp_servers.test.tsx b/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/mcp_servers.test.tsx
index 94a2c75016d..61880363234 100644
--- a/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/mcp_servers.test.tsx
+++ b/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/mcp_servers.test.tsx
@@ -3,7 +3,7 @@ import { render, waitFor, screen, act, within } from "@testing-library/react";
import userEvent from "@testing-library/user-event";
import { describe, it, expect, vi, beforeEach } from "vitest";
import { QueryClient, QueryClientProvider } from "@tanstack/react-query";
-import MCPServers, { compareServers } from "./mcp_servers";
+import MCPServers, { compareServers, type SortKey } from "./mcp_servers";
import type { MCPServer } from "@/components/mcp_tools/types";
import * as networking from "@/components/networking";
@@ -33,8 +33,14 @@ const createQueryClient = () =>
});
describe("compareServers", () => {
- const server = (server_id: string, name: string, created_at = ""): MCPServer =>
- ({ server_id, server_name: name, created_at, updated_at: created_at }) as MCPServer;
+ const server = (server_id: string, name: string, created_at = ""): MCPServer => ({
+ server_id,
+ server_name: name,
+ created_at,
+ updated_at: created_at,
+ created_by: "user",
+ updated_by: "user",
+ });
const shuffled = [server("c", "github"), server("a", "slack"), server("b", "Jira")];
@@ -56,6 +62,66 @@ describe("compareServers", () => {
"old",
]);
});
+
+ it.each(["created_desc", "updated_desc", "name_asc", "health"])(
+ "breaks equal timestamps and names by ID for %s regardless of input order",
+ (sort) => {
+ const servers = [
+ server("b", "GitHub", "2026-01-01T00:00:00Z"),
+ server("c", "Slack", "2026-01-01T00:00:00Z"),
+ server("a", "github", "2026-01-01T00:00:00Z"),
+ ];
+ for (const input of [servers, [...servers].reverse()]) {
+ expect([...input].sort((a, b) => compareServers(a, b, sort)).map((s) => s.server_id)).toEqual(["a", "b", "c"]);
+ }
+ },
+ );
+
+ it("uses the display name before alias, then falls back to alias and ID", () => {
+ const servers: MCPServer[] = [
+ { ...server("s-slack", "Slack"), alias: "aaa" },
+ { ...server("s-github", ""), server_name: null, alias: "GitHub" },
+ { ...server("confluence", ""), alias: "" },
+ ];
+ for (const input of [servers, [...servers].reverse()]) {
+ expect([...input].sort((a, b) => compareServers(a, b, "name_asc")).map((s) => s.server_id)).toEqual([
+ "confluence",
+ "s-github",
+ "s-slack",
+ ]);
+ }
+ });
+
+ it.each(["created_desc", "updated_desc", "health"])(
+ "keeps timestamped servers before missing timestamps for %s",
+ (sort) => {
+ const servers = [
+ server("config", "aaa"),
+ server("older", "bbb", "2026-01-01T00:00:00Z"),
+ server("newer", "zzz", "2026-02-01T00:00:00Z"),
+ ];
+ for (const input of [servers, [...servers].reverse()]) {
+ expect([...input].sort((a, b) => compareServers(a, b, sort)).map((s) => s.server_id)).toEqual([
+ "newer",
+ "older",
+ "config",
+ ]);
+ }
+ },
+ );
+
+ it("sorts health before recency and display name", () => {
+ const servers: MCPServer[] = [
+ { ...server("healthy", "aaa", "2026-03-01T00:00:00Z"), status: "healthy" },
+ { ...server("unknown", "bbb", "2026-02-01T00:00:00Z"), status: "unknown" },
+ { ...server("unhealthy", "zzz", "2026-01-01T00:00:00Z"), status: "unhealthy" },
+ ];
+ expect(servers.sort((a, b) => compareServers(a, b, "health")).map((s) => s.server_id)).toEqual([
+ "unhealthy",
+ "unknown",
+ "healthy",
+ ]);
+ });
});
describe("MCPServers", () => {
From e208b4e89ed41c6e4239c638f8e51a1239560959 Mon Sep 17 00:00:00 2001
From: Joshua Valluru <326636767+joshua-berri@users.noreply.github.com>
Date: Mon, 21 Sep 2026 12:40:12 -0700
Subject: [PATCH 053/160] fix(mcp): keep explicit legacy sampling callers
isolated
---
.../_experimental/mcp_server/legacy_callbacks.py | 2 +-
.../mcp_server/test_mcp_server_manager.py | 13 +++++++++----
2 files changed, 10 insertions(+), 5 deletions(-)
diff --git a/litellm/proxy/_experimental/mcp_server/legacy_callbacks.py b/litellm/proxy/_experimental/mcp_server/legacy_callbacks.py
index 4424907c28a..9e321062643 100644
--- a/litellm/proxy/_experimental/mcp_server/legacy_callbacks.py
+++ b/litellm/proxy/_experimental/mcp_server/legacy_callbacks.py
@@ -33,7 +33,7 @@ def create_sampling_callback(
) -> SamplingCallback:
from litellm.proxy._experimental.mcp_server.server import get_active_auth_context
- auth: Final = get_active_auth_context() if operation_context is None else None
+ auth: Final = get_active_auth_context() if operation_context is None and user_api_key_auth is None else None
captured: Final = (
operation_context
if operation_context is not None
diff --git a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_server_manager.py b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_server_manager.py
index 7be79f9b514..2dc7e01966a 100644
--- a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_server_manager.py
+++ b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_server_manager.py
@@ -14214,10 +14214,11 @@ async def test_request_selected_during_guardrail_runs_concurrently_with_tool(mon
@pytest.mark.asyncio
-@pytest.mark.parametrize("with_caller", [True, False])
-async def test_client_sampling_does_not_fill_explicit_context_from_another_ambient_caller(with_caller):
+@pytest.mark.parametrize("with_caller,legacy_factory", [(True, False), (False, False), (True, True)])
+async def test_client_sampling_does_not_fill_explicit_context_from_another_ambient_caller(with_caller, legacy_factory):
from mcp.server.auth.middleware.auth_context import auth_context_var
from litellm.proxy._experimental.mcp_server import server as legacy_server
+ from litellm.proxy._experimental.mcp_server.mcp_server_manager import _create_sampling_callback
upstream = MCPServer(server_id="explicit-empty", name="explicit_empty", url="https://example.invalid/mcp", transport=MCPTransport.http, allow_sampling=True)
token = auth_context_var.set(None)
@@ -14228,8 +14229,12 @@ async def test_client_sampling_does_not_fill_explicit_context_from_another_ambie
patch("litellm.proxy._experimental.mcp_server.mcp_server_manager.MCPClient") as factory,
patch("litellm.proxy._experimental.mcp_server.sampling_handler.handle_sampling_create_message", sampling),
):
- await MCPServerManager()._create_mcp_client(upstream, user_api_key_auth=UserAPIKeyAuth(user_id="explicit") if with_caller else None)
- await factory.call_args.kwargs["sampling_callback"](None, None)
+ if legacy_factory:
+ callback = _create_sampling_callback(user_api_key_auth=UserAPIKeyAuth(user_id="explicit"))
+ else:
+ await MCPServerManager()._create_mcp_client(upstream, user_api_key_auth=UserAPIKeyAuth(user_id="explicit") if with_caller else None)
+ callback = factory.call_args.kwargs["sampling_callback"]
+ await callback(None, None)
captured = sampling.await_args.kwargs
if with_caller:
assert captured["user_api_key_auth"].user_id == "explicit"
From ee07f710630bdadb7035f08cfb9cd728ec4425db Mon Sep 17 00:00:00 2001
From: yucheng
Date: Mon, 21 Sep 2026 19:43:11 +0000
Subject: [PATCH 054/160] style(proxy): format scheduled job timeout
configuration
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
---
litellm/constants.py | 8 ++++++--
litellm/proxy/shutdown/scheduled_jobs.py | 4 +---
2 files changed, 7 insertions(+), 5 deletions(-)
diff --git a/litellm/constants.py b/litellm/constants.py
index 842adf62f6b..1971c336a96 100644
--- a/litellm/constants.py
+++ b/litellm/constants.py
@@ -1742,8 +1742,12 @@ SPEND_LOG_CLEANUP_BATCH_FAILURE_BACKOFF_SECONDS: Final = float(
SPEND_LOG_CLEANUP_RUN_BUDGET_SECONDS: Final = float(os.getenv("SPEND_LOG_CLEANUP_RUN_BUDGET_SECONDS", "300"))
SPEND_LOG_CLEANUP_BATCH_TIMEOUT_SECONDS: Final = float(os.getenv("SPEND_LOG_CLEANUP_BATCH_TIMEOUT_SECONDS", "30"))
SPEND_LOG_CLEANUP_REMAINING_COUNT_CAP: Final = int(os.getenv("SPEND_LOG_CLEANUP_REMAINING_COUNT_CAP", "100000"))
-SCHEDULED_JOB_SHUTDOWN_FINISH_TIMEOUT_SECONDS: Final = float(os.getenv("SCHEDULED_JOB_SHUTDOWN_FINISH_TIMEOUT_SECONDS", "5"))
-SCHEDULED_JOB_SHUTDOWN_CANCEL_TIMEOUT_SECONDS: Final = float(os.getenv("SCHEDULED_JOB_SHUTDOWN_CANCEL_TIMEOUT_SECONDS", "5"))
+SCHEDULED_JOB_SHUTDOWN_FINISH_TIMEOUT_SECONDS: Final = float(
+ os.getenv("SCHEDULED_JOB_SHUTDOWN_FINISH_TIMEOUT_SECONDS", "5")
+)
+SCHEDULED_JOB_SHUTDOWN_CANCEL_TIMEOUT_SECONDS: Final = float(
+ os.getenv("SCHEDULED_JOB_SHUTDOWN_CANCEL_TIMEOUT_SECONDS", "5")
+)
TOOL_SPEND_TOP_TOOLS: Final = 100
SPEND_LOG_PARTITION_INTERVAL: Final = os.getenv("SPEND_LOG_PARTITION_INTERVAL", "day")
SPEND_LOG_PARTITION_PRECREATE_AHEAD: Final = int(os.getenv("SPEND_LOG_PARTITION_PRECREATE_AHEAD", 7))
diff --git a/litellm/proxy/shutdown/scheduled_jobs.py b/litellm/proxy/shutdown/scheduled_jobs.py
index e920ce19eb9..7889c35cf4e 100644
--- a/litellm/proxy/shutdown/scheduled_jobs.py
+++ b/litellm/proxy/shutdown/scheduled_jobs.py
@@ -64,9 +64,7 @@ async def stop_in_flight_scheduler_jobs(
len(in_flight),
)
still_running: Final = (
- (await asyncio.wait(in_flight, timeout=finish_timeout_seconds))[1]
- if in_flight
- else frozenset()
+ (await asyncio.wait(in_flight, timeout=finish_timeout_seconds))[1] if in_flight else frozenset()
)
scheduler.shutdown(wait=False)
if not still_running:
From f70683ae92748a9e43c5c0faade2e81275e23a8c Mon Sep 17 00:00:00 2001
From: yuneng
Date: Mon, 21 Sep 2026 19:56:30 +0000
Subject: [PATCH 055/160] fix(ui): satisfy complexity router CI lint budgets
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
---
.../ComplexityRouterAdvancedSections.tsx | 2 +-
.../add_model/add_auto_router_tab.tsx | 26 +++++++++---------
.../build_complexity_router_config.ts | 27 ++++++++++---------
.../edit_auto_router_modal.tsx | 13 +++++----
.../hydrate_complexity_router_config.ts | 16 ++++++-----
5 files changed, 45 insertions(+), 39 deletions(-)
diff --git a/ui/litellm-dashboard/src/components/add_model/ComplexityRouterAdvancedSections.tsx b/ui/litellm-dashboard/src/components/add_model/ComplexityRouterAdvancedSections.tsx
index dd822907735..0412298ccdd 100644
--- a/ui/litellm-dashboard/src/components/add_model/ComplexityRouterAdvancedSections.tsx
+++ b/ui/litellm-dashboard/src/components/add_model/ComplexityRouterAdvancedSections.tsx
@@ -28,7 +28,7 @@ interface ComplexityRouterAdvancedSectionsProps {
onChange: (value: ComplexityRouterConfigValue) => void;
forecast: boolean;
modelOptions: { value: string; label: string }[];
- classifierEffortOptionsByModel: Record;
+ classifierEffortOptionsByModel: Record;
customTechnicalKeywords?: string[];
onCustomTechnicalKeywordsChange?: (keywords: string[]) => void;
showValidationErrors: boolean;
diff --git a/ui/litellm-dashboard/src/components/add_model/add_auto_router_tab.tsx b/ui/litellm-dashboard/src/components/add_model/add_auto_router_tab.tsx
index f805a5b5511..9be49edf08d 100644
--- a/ui/litellm-dashboard/src/components/add_model/add_auto_router_tab.tsx
+++ b/ui/litellm-dashboard/src/components/add_model/add_auto_router_tab.tsx
@@ -408,6 +408,17 @@ const AddAutoRouterTab: React.FC = ({
matchThreshold,
escalationKeywords,
};
+ const jevRequestParams =
+ effectiveClassifierType(complexityRouterConfig) === "jev"
+ ? {
+ prompt: JEV_CONNECTION_TEST_PROMPT,
+ config: buildComplexityRouterConfig(complexityRouterConfigParams),
+ defaultModel: resolveComplexityDefaultModel(complexityRouterConfig, complexityRouterConfig.default_model),
+ routerName: watchedName,
+ teamId: requiresTeamScope ? watchedTeamId ?? undefined : undefined,
+ }
+ : undefined;
+ const jevRequest = jevRequestParams ? buildAutoRouterRoutingTestRequest(jevRequestParams) : undefined;
const submitRecommendedRouter = async (name: string) => {
// The one answer the submit button reads, so a disabled button and a refused submit cannot
@@ -816,20 +827,7 @@ const AddAutoRouterTab: React.FC = ({
testId={connectionTestId}
accessToken={accessToken}
targets={testTargets}
- jevRequest={
- effectiveClassifierType(complexityRouterConfig) === "jev"
- ? buildAutoRouterRoutingTestRequest({
- prompt: JEV_CONNECTION_TEST_PROMPT,
- config: buildComplexityRouterConfig(complexityRouterConfigParams),
- defaultModel: resolveComplexityDefaultModel(
- complexityRouterConfig,
- complexityRouterConfig.default_model,
- ),
- routerName: watchedName,
- teamId: requiresTeamScope ? watchedTeamId ?? undefined : undefined,
- })
- : undefined
- }
+ jevRequest={jevRequest}
onTestComplete={() => setIsTestingConnection(false)}
/>
diff --git a/ui/litellm-dashboard/src/components/add_model/build_complexity_router_config.ts b/ui/litellm-dashboard/src/components/add_model/build_complexity_router_config.ts
index ab71b6b3cce..4782a58513d 100644
--- a/ui/litellm-dashboard/src/components/add_model/build_complexity_router_config.ts
+++ b/ui/litellm-dashboard/src/components/add_model/build_complexity_router_config.ts
@@ -744,16 +744,22 @@ export const buildComplexityRouterConfig = ({
open: open.trim().toLowerCase(),
close: close.trim().toLowerCase(),
}));
+ const cleanedListValues = {
+ code_keywords: cleanList(codeKeywords),
+ reasoning_keywords: cleanList(reasoningKeywords),
+ technical_keywords: cleanList(technicalKeywords),
+ simple_keywords: cleanList(simpleKeywords),
+ plan_mode_patterns: cleanList(planModePatterns),
+ housekeeping_patterns: cleanList(housekeepingPatterns),
+ };
const cleanedLists = Object.fromEntries(
- Object.entries({
- code_keywords: cleanList(codeKeywords),
- reasoning_keywords: cleanList(reasoningKeywords),
- technical_keywords: cleanList(technicalKeywords),
- simple_keywords: cleanList(simpleKeywords),
- plan_mode_patterns: cleanList(planModePatterns),
- housekeeping_patterns: cleanList(housekeepingPatterns),
- }).filter(([, list]) => list !== undefined),
+ Object.entries(cleanedListValues).filter(([, list]) => list !== undefined),
);
+ const hasValidCustomClassifierTimeout =
+ classifierType === "custom" &&
+ classifierPluginTimeoutMs !== undefined &&
+ Number.isInteger(classifierPluginTimeoutMs) &&
+ classifierPluginTimeoutMs > 0;
const supportsOpeningPrompt = !customTierSet && !forecast && usesLlmClassifier(effectiveType);
const payload: ComplexityRouterConfigPayload = {
@@ -826,10 +832,7 @@ export const buildComplexityRouterConfig = ({
...(routeHousekeepingToCheapestTier === false && { route_housekeeping_to_cheapest_tier: false }),
...(cleanedReminderMarkers && cleanedReminderMarkers.length > 0 && { reminder_markers: cleanedReminderMarkers }),
...(maxTokensFromTierModel === false && { max_tokens_from_tier_model: false }),
- ...(classifierType === "custom" &&
- classifierPluginTimeoutMs !== undefined &&
- Number.isInteger(classifierPluginTimeoutMs) &&
- classifierPluginTimeoutMs > 0 && { classifier_plugin_timeout_ms: classifierPluginTimeoutMs }),
+ ...(hasValidCustomClassifierTimeout && { classifier_plugin_timeout_ms: classifierPluginTimeoutMs }),
...scorerKnobs,
};
if (!customTierSet) return payload;
diff --git a/ui/litellm-dashboard/src/components/edit_auto_router/edit_auto_router_modal.tsx b/ui/litellm-dashboard/src/components/edit_auto_router/edit_auto_router_modal.tsx
index e3a2df39b88..e6f1ccd0e17 100644
--- a/ui/litellm-dashboard/src/components/edit_auto_router/edit_auto_router_modal.tsx
+++ b/ui/litellm-dashboard/src/components/edit_auto_router/edit_auto_router_modal.tsx
@@ -373,12 +373,13 @@ const EditAutoRouterModal: React.FC = ({
setRouterConfig(parsedConfig);
// Set form values
- form.reset({
+ const routerFormValues = {
auto_router_name: modelData.model_name,
auto_router_default_model: modelData.litellm_params?.auto_router_default_model || null,
auto_router_embedding_model: modelData.litellm_params?.auto_router_embedding_model || null,
model_access_group: modelData.model_info?.access_groups || [],
- });
+ };
+ form.reset(routerFormValues);
} catch (error) {
console.error("Error parsing auto router config:", error);
toast.fromError("Error loading auto router configuration");
@@ -456,11 +457,12 @@ const EditAutoRouterModal: React.FC = ({
// Dual write: complexity_router_config.default_model (the pin marker hydratePinnedDefaultModel
// reads back) and complexity_router_default_model (what the backend routes on) must always be
// written together from the same value. Same pairing in add_auto_router_tab.tsx.
+ const keywordMatching = { keywordTierRules, escalationKeywords, semanticMatchingEnabled, embeddingModel, matchThreshold };
const updatedConfig = buildUpdatedComplexityRouterConfig(
modelData.litellm_params?.complexity_router_config,
complexityRouterConfig,
customTechnicalKeywords,
- { keywordTierRules, escalationKeywords, semanticMatchingEnabled, embeddingModel, matchThreshold },
+ keywordMatching,
);
const serverVerdict = await validateAutoRouterConfig(accessToken, updatedConfig, modelData?.model_info?.team_id);
const dryRunError = dryRunRejection(serverVerdict);
@@ -497,12 +499,13 @@ const EditAutoRouterModal: React.FC = ({
);
toast.success("Auto router configuration updated successfully");
- onSuccess({
+ const updatedModelData = {
...modelData,
model_name: values.auto_router_name,
litellm_params: updatedLitellmParams,
model_info: updatedModelInfo,
- });
+ };
+ onSuccess(updatedModelData);
onCancel();
return;
}
diff --git a/ui/litellm-dashboard/src/components/edit_auto_router/hydrate_complexity_router_config.ts b/ui/litellm-dashboard/src/components/edit_auto_router/hydrate_complexity_router_config.ts
index c7c7bcb3382..b3a9ae0a504 100644
--- a/ui/litellm-dashboard/src/components/edit_auto_router/hydrate_complexity_router_config.ts
+++ b/ui/litellm-dashboard/src/components/edit_auto_router/hydrate_complexity_router_config.ts
@@ -26,13 +26,15 @@ import {
const isReminderMarkerPair = (
input: unknown,
-): input is { open: string; close: string } =>
- typeof input === "object" &&
- input !== null &&
- "open" in input &&
- "close" in input &&
- typeof input.open === "string" &&
- typeof input.close === "string";
+): input is { open: string; close: string } => {
+ if (typeof input !== "object" || input === null) {
+ return false;
+ }
+ if (!("open" in input) || !("close" in input)) {
+ return false;
+ }
+ return typeof input.open === "string" && typeof input.close === "string";
+};
const stringList = (input: unknown): string[] | undefined =>
Array.isArray(input) ? input.filter((item): item is string => typeof item === "string") : undefined;
From 9002749e29bcff3de2f8991ee988f923ffc5a3c5 Mon Sep 17 00:00:00 2001
From: Joshua Valluru <326636767+joshua-berri@users.noreply.github.com>
Date: Mon, 21 Sep 2026 12:58:06 -0700
Subject: [PATCH 056/160] fix(mcp): preserve Python 3.10 imports and
integration test seams
---
.../_experimental/mcp_server/operations.py | 4 ++--
tests/mcp_tests/test_mcp_logging.py | 6 ++---
tests/mcp_tests/test_mcp_server.py | 24 ++++++++++---------
3 files changed, 18 insertions(+), 16 deletions(-)
diff --git a/litellm/proxy/_experimental/mcp_server/operations.py b/litellm/proxy/_experimental/mcp_server/operations.py
index 3ee4f37add7..a579c33e702 100644
--- a/litellm/proxy/_experimental/mcp_server/operations.py
+++ b/litellm/proxy/_experimental/mcp_server/operations.py
@@ -6,7 +6,7 @@ import types
import uuid
from collections.abc import Mapping, Sequence
from datetime import datetime
-from typing import Any, Final, NoReturn, TypeAlias, assert_never, overload
+from typing import Any, Final, NoReturn, TypeAlias, overload
from fastapi import HTTPException
from mcp import ReadResourceResult, Resource
@@ -34,7 +34,7 @@ from mcp.types import (
)
from mcp.types import Tool as MCPTool
from pydantic import AnyUrl, ConfigDict, Field, TypeAdapter
-from typing_extensions import ReadOnly, TypedDict
+from typing_extensions import ReadOnly, TypedDict, assert_never
from litellm._logging import verbose_logger
from litellm.constants import (
diff --git a/tests/mcp_tests/test_mcp_logging.py b/tests/mcp_tests/test_mcp_logging.py
index ed8829945e5..41d0e2cb59b 100644
--- a/tests/mcp_tests/test_mcp_logging.py
+++ b/tests/mcp_tests/test_mcp_logging.py
@@ -142,7 +142,7 @@ async def test_mcp_cost_tracking():
local_mcp_server_manager,
),
patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager",
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager",
local_mcp_server_manager,
),
):
@@ -293,7 +293,7 @@ async def test_mcp_cost_tracking_per_tool():
local_mcp_server_manager,
),
patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager",
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager",
local_mcp_server_manager,
),
):
@@ -451,7 +451,7 @@ async def test_mcp_tool_call_hook():
local_mcp_server_manager,
),
patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager",
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager",
local_mcp_server_manager,
),
):
diff --git a/tests/mcp_tests/test_mcp_server.py b/tests/mcp_tests/test_mcp_server.py
index 94cf35b675d..2b92367f186 100644
--- a/tests/mcp_tests/test_mcp_server.py
+++ b/tests/mcp_tests/test_mcp_server.py
@@ -922,7 +922,7 @@ async def test_get_tools_from_mcp_servers():
)
with patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager",
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager",
mock_manager,
):
# Test with specific servers
@@ -950,6 +950,7 @@ async def test_get_tools_from_mcp_servers():
extra_headers=None,
add_prefix=False,
raw_headers=None,
+ client_ip=None,
user_api_key_auth=None,
oauth2_headers=None,
):
@@ -966,7 +967,7 @@ async def test_get_tools_from_mcp_servers():
)
with patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager",
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager",
mock_manager_2,
):
result = await _get_tools_from_mcp_servers(
@@ -998,7 +999,7 @@ async def test_get_tools_from_mcp_servers():
)
with patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager",
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager",
mock_manager,
):
with patch(
@@ -1981,6 +1982,7 @@ async def test_get_tools_for_single_server():
extra_headers=None,
add_prefix=False,
raw_headers=None,
+ client_ip=None,
user_api_key_auth=None,
)
@@ -2076,7 +2078,7 @@ async def test_rest_listing_hides_key_grants_dispatch_would_refuse():
with patch(
"litellm.proxy._experimental.mcp_server.rest_endpoints.global_mcp_server_manager"
) as mock_manager, patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager"
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager"
) as mock_server_manager, patch.object(
MCPRequestHandler,
"get_allowed_tools_for_server",
@@ -2473,7 +2475,7 @@ async def test_filter_tools_by_allowed_tools_integration():
# Mock the global MCP server manager
with patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager"
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager"
) as mock_manager:
# Mock manager methods
mock_manager.get_allowed_mcp_servers = AsyncMock(
@@ -2588,7 +2590,7 @@ async def test_filter_tools_by_disallowed_tools_integration():
# Mock the global MCP server manager
with patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager"
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager"
) as mock_manager:
# Mock manager methods
mock_manager.get_allowed_mcp_servers = AsyncMock(
@@ -2689,7 +2691,7 @@ async def test_filter_tools_no_restrictions_integration():
# Mock the global MCP server manager
with patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager"
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_server_manager"
) as mock_manager:
# Mock manager methods
mock_manager.get_allowed_mcp_servers = AsyncMock(
@@ -2970,10 +2972,10 @@ async def test_call_mcp_tool_uses_manager_permission_lookup():
return_value=mock_server,
) as mock_get_server,
patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_tool_registry"
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_tool_registry"
) as mock_tool_registry,
patch(
- "litellm.proxy._experimental.mcp_server.server._handle_managed_mcp_tool",
+ "litellm.proxy._experimental.mcp_server.operations._handle_managed_mcp_tool",
new_callable=AsyncMock,
) as mock_handle_managed,
patch(
@@ -3046,10 +3048,10 @@ async def test_call_mcp_tool_resolves_unprefixed_tool_name_and_checks_permission
return_value=mock_server,
) as mock_get_server,
patch(
- "litellm.proxy._experimental.mcp_server.server.global_mcp_tool_registry"
+ "litellm.proxy._experimental.mcp_server.operations.global_mcp_tool_registry"
) as mock_tool_registry,
patch(
- "litellm.proxy._experimental.mcp_server.server._handle_managed_mcp_tool",
+ "litellm.proxy._experimental.mcp_server.operations._handle_managed_mcp_tool",
new_callable=AsyncMock,
) as mock_handle_managed,
patch(
From 4117d9f7860bc7bba6250682654f5cddeca4cd3f Mon Sep 17 00:00:00 2001
From: yuneng
Date: Mon, 21 Sep 2026 20:01:08 +0000
Subject: [PATCH 057/160] style(ui): format complexity router files
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
---
.../ComplexityRouterAdvancedSections.tsx | 8 ++++----
.../add_model/ComplexityRouterConfig.test.tsx | 12 ++----------
.../add_model/HeuristicKeywordOverrides.tsx | 4 ++--
.../components/add_model/ReminderMarkers.tsx | 17 ++++++++++++-----
.../add_model/ResponseFormatControls.tsx | 4 ++--
.../add_model/build_complexity_router_config.ts | 4 +---
.../edit_auto_router/edit_auto_router_modal.tsx | 13 ++++++++-----
.../hydrate_complexity_router_config.ts | 11 ++++++-----
8 files changed, 37 insertions(+), 36 deletions(-)
diff --git a/ui/litellm-dashboard/src/components/add_model/ComplexityRouterAdvancedSections.tsx b/ui/litellm-dashboard/src/components/add_model/ComplexityRouterAdvancedSections.tsx
index 0412298ccdd..71f8bce76b5 100644
--- a/ui/litellm-dashboard/src/components/add_model/ComplexityRouterAdvancedSections.tsx
+++ b/ui/litellm-dashboard/src/components/add_model/ComplexityRouterAdvancedSections.tsx
@@ -130,7 +130,9 @@ const ComplexityRouterAdvancedSections: React.FCAdvanced: Plan-Mode Override,
- children: ,
+ children: (
+
+ ),
},
{
key: "housekeeping",
@@ -195,9 +197,7 @@ const ComplexityRouterAdvancedSections: React.FC
)}
{onKeywordTierRulesChange && onSemanticMatchingEnabledChange && }
diff --git a/ui/litellm-dashboard/src/components/add_model/ComplexityRouterConfig.test.tsx b/ui/litellm-dashboard/src/components/add_model/ComplexityRouterConfig.test.tsx
index 9a8577100df..56b37b92af3 100644
--- a/ui/litellm-dashboard/src/components/add_model/ComplexityRouterConfig.test.tsx
+++ b/ui/litellm-dashboard/src/components/add_model/ComplexityRouterConfig.test.tsx
@@ -106,7 +106,6 @@ describe("ComplexityRouterConfig", () => {
const capabilityValue = { ...defaultValue, classifier_type: "capability" as const };
rerender( );
expect(screen.queryByText("Advanced: Heuristic Keyword Overrides")).not.toBeInTheDocument();
-
});
it.each([
@@ -127,11 +126,7 @@ describe("ComplexityRouterConfig", () => {
it.each([true, false])("shows reminder marker validation only when requested: %s", (showValidationErrors) => {
const value = { ...defaultValue, reminder_markers: [{ open: "", close: "x" }] };
renderWithProviders(
- ,
+ ,
);
fireEvent.click(screen.getByText("Advanced: Reminder Markers"));
const validation = screen.queryByText(/needs both/i);
@@ -144,10 +139,7 @@ describe("ComplexityRouterConfig", () => {
it("disables housekeeping sentinels when cheapest-tier routing is off", () => {
renderWithProviders(
- ,
+ ,
);
fireEvent.click(screen.getByText("Advanced: Housekeeping Routing"));
const sentinelInput = screen.getByRole("combobox", { name: "e.g., conversation title" });
diff --git a/ui/litellm-dashboard/src/components/add_model/HeuristicKeywordOverrides.tsx b/ui/litellm-dashboard/src/components/add_model/HeuristicKeywordOverrides.tsx
index 696924f4e77..185f187bbfc 100644
--- a/ui/litellm-dashboard/src/components/add_model/HeuristicKeywordOverrides.tsx
+++ b/ui/litellm-dashboard/src/components/add_model/HeuristicKeywordOverrides.tsx
@@ -15,8 +15,8 @@ const HeuristicKeywordOverrides: React.FC<{
}> = ({ value, onChange }) => (
- Each list replaces the built-in keyword list of the same name for the heuristic scorer. Leave a list empty to
- keep the built-in one. To add technical terms without replacing the list, use custom technical keywords under
+ Each list replaces the built-in keyword list of the same name for the heuristic scorer. Leave a list empty to keep
+ the built-in one. To add technical terms without replacing the list, use custom technical keywords under
Classification Method.
{fields.map(([key, label]) => {
diff --git a/ui/litellm-dashboard/src/components/add_model/ReminderMarkers.tsx b/ui/litellm-dashboard/src/components/add_model/ReminderMarkers.tsx
index 970394291d4..c7f9ee48e0b 100644
--- a/ui/litellm-dashboard/src/components/add_model/ReminderMarkers.tsx
+++ b/ui/litellm-dashboard/src/components/add_model/ReminderMarkers.tsx
@@ -14,7 +14,9 @@ const ReminderMarkers: React.FC<{
const update = (index: number, patch: Partial
) =>
onChange({
...value,
- reminder_markers: markers.map((marker, markerIndex) => (markerIndex === index ? { ...marker, ...patch } : marker)),
+ reminder_markers: markers.map((marker, markerIndex) =>
+ markerIndex === index ? { ...marker, ...patch } : marker,
+ ),
});
const remove = (index: number) => {
const next = markers.filter((_, markerIndex) => markerIndex !== index);
@@ -24,9 +26,9 @@ const ReminderMarkers: React.FC<{
return (
- Delimiter pairs that wrap harness-injected reminder blocks, which are stripped before classification. Setting any
- pair replaces the built-in pairs, so list every pair your harness emits. Matching is case-insensitive and values
- are saved lowercased.
+ Delimiter pairs that wrap harness-injected reminder blocks, which are stripped before classification. Setting
+ any pair replaces the built-in pairs, so list every pair your harness emits. Matching is case-insensitive and
+ values are saved lowercased.
{markers.map((marker, index) => (
@@ -53,7 +55,12 @@ const ReminderMarkers: React.FC<{
onChange={(event) => update(index, { close: event.target.value })}
/>
-
remove(index)}>
+ remove(index)}
+ >
diff --git a/ui/litellm-dashboard/src/components/add_model/ResponseFormatControls.tsx b/ui/litellm-dashboard/src/components/add_model/ResponseFormatControls.tsx
index 9edbf204a6d..6e7b1489a4c 100644
--- a/ui/litellm-dashboard/src/components/add_model/ResponseFormatControls.tsx
+++ b/ui/litellm-dashboard/src/components/add_model/ResponseFormatControls.tsx
@@ -27,8 +27,8 @@ const ResponseFormatControls: React.FC<{
Cap max_tokens at the tier model's output ceiling
- Replace the caller's max_tokens with the routed tier model's output ceiling so one client value fits every
- tier. Off forwards the caller's value unchanged.
+ Replace the caller's max_tokens with the routed tier model's output ceiling so one client value fits
+ every tier. Off forwards the caller's value unchanged.
>
);
diff --git a/ui/litellm-dashboard/src/components/add_model/build_complexity_router_config.ts b/ui/litellm-dashboard/src/components/add_model/build_complexity_router_config.ts
index 4782a58513d..ca46e171970 100644
--- a/ui/litellm-dashboard/src/components/add_model/build_complexity_router_config.ts
+++ b/ui/litellm-dashboard/src/components/add_model/build_complexity_router_config.ts
@@ -752,9 +752,7 @@ export const buildComplexityRouterConfig = ({
plan_mode_patterns: cleanList(planModePatterns),
housekeeping_patterns: cleanList(housekeepingPatterns),
};
- const cleanedLists = Object.fromEntries(
- Object.entries(cleanedListValues).filter(([, list]) => list !== undefined),
- );
+ const cleanedLists = Object.fromEntries(Object.entries(cleanedListValues).filter(([, list]) => list !== undefined));
const hasValidCustomClassifierTimeout =
classifierType === "custom" &&
classifierPluginTimeoutMs !== undefined &&
diff --git a/ui/litellm-dashboard/src/components/edit_auto_router/edit_auto_router_modal.tsx b/ui/litellm-dashboard/src/components/edit_auto_router/edit_auto_router_modal.tsx
index e6f1ccd0e17..0dafa9b330a 100644
--- a/ui/litellm-dashboard/src/components/edit_auto_router/edit_auto_router_modal.tsx
+++ b/ui/litellm-dashboard/src/components/edit_auto_router/edit_auto_router_modal.tsx
@@ -1,10 +1,7 @@
import AutoRouterClassifierTabs from "../add_model/AutoRouterClassifierTabs";
import { usesClassifierContext } from "../add_model/classifier_types";
export type { StoredComplexityRouterConfig } from "../add_model/build_complexity_router_config";
-import {
- getForecastConfigError,
- isForecastClassifier,
-} from "../add_model/forecast_classifier_config";
+import { getForecastConfigError, isForecastClassifier } from "../add_model/forecast_classifier_config";
import React, { useEffect, useMemo, useState } from "react";
import {
complexityRouterSchema,
@@ -457,7 +454,13 @@ const EditAutoRouterModal: React.FC = ({
// Dual write: complexity_router_config.default_model (the pin marker hydratePinnedDefaultModel
// reads back) and complexity_router_default_model (what the backend routes on) must always be
// written together from the same value. Same pairing in add_auto_router_tab.tsx.
- const keywordMatching = { keywordTierRules, escalationKeywords, semanticMatchingEnabled, embeddingModel, matchThreshold };
+ const keywordMatching = {
+ keywordTierRules,
+ escalationKeywords,
+ semanticMatchingEnabled,
+ embeddingModel,
+ matchThreshold,
+ };
const updatedConfig = buildUpdatedComplexityRouterConfig(
modelData.litellm_params?.complexity_router_config,
complexityRouterConfig,
diff --git a/ui/litellm-dashboard/src/components/edit_auto_router/hydrate_complexity_router_config.ts b/ui/litellm-dashboard/src/components/edit_auto_router/hydrate_complexity_router_config.ts
index b3a9ae0a504..6dbd2b19b52 100644
--- a/ui/litellm-dashboard/src/components/edit_auto_router/hydrate_complexity_router_config.ts
+++ b/ui/litellm-dashboard/src/components/edit_auto_router/hydrate_complexity_router_config.ts
@@ -24,9 +24,7 @@ import {
resolveComplexityDefaultModel,
} from "../add_model/tier_rows";
-const isReminderMarkerPair = (
- input: unknown,
-): input is { open: string; close: string } => {
+const isReminderMarkerPair = (input: unknown): input is { open: string; close: string } => {
if (typeof input !== "object" || input === null) {
return false;
}
@@ -176,9 +174,12 @@ export const hydrateComplexityRouterConfig = (
? parsedConfig.reminder_markers.filter(isReminderMarkerPair)
: undefined,
max_tokens_from_tier_model:
- typeof parsedConfig.max_tokens_from_tier_model === "boolean" ? parsedConfig.max_tokens_from_tier_model : undefined,
+ typeof parsedConfig.max_tokens_from_tier_model === "boolean"
+ ? parsedConfig.max_tokens_from_tier_model
+ : undefined,
classifier_plugin_timeout_ms:
- typeof parsedConfig.classifier_plugin_timeout_ms === "number" && Number.isFinite(parsedConfig.classifier_plugin_timeout_ms)
+ typeof parsedConfig.classifier_plugin_timeout_ms === "number" &&
+ Number.isFinite(parsedConfig.classifier_plugin_timeout_ms)
? parsedConfig.classifier_plugin_timeout_ms
: undefined,
};
From ef67412e50c9f3250eb473a0a14fbf00617e36ea Mon Sep 17 00:00:00 2001
From: Joshua Valluru <326636767+joshua-berri@users.noreply.github.com>
Date: Mon, 21 Sep 2026 13:13:20 -0700
Subject: [PATCH 058/160] fix(mcp): keep OAuth prefetch failure logs free of
caller data
---
.../_experimental/mcp_server/operations.py | 6 ++---
.../proxy/_experimental/mcp_server/server.py | 3 +++
.../mcp_server/test_operations.py | 22 +++++++++++++++++++
3 files changed, 28 insertions(+), 3 deletions(-)
diff --git a/litellm/proxy/_experimental/mcp_server/operations.py b/litellm/proxy/_experimental/mcp_server/operations.py
index a579c33e702..fcee3483e15 100644
--- a/litellm/proxy/_experimental/mcp_server/operations.py
+++ b/litellm/proxy/_experimental/mcp_server/operations.py
@@ -763,8 +763,8 @@ async def _prefetch_oauth_creds_for_user(
)
creds: Final = await list_user_oauth_credentials(prisma_client, user_id)
return {c["server_id"]: c for c in creds if "server_id" in c}
- except Exception as e:
- verbose_logger.warning("_prefetch_oauth_creds_for_user: failed to prefetch for user=%s: %s", user_id, e)
+ except Exception:
+ verbose_logger.warning("_prefetch_oauth_creds_for_user: failed to prefetch OAuth credentials")
return {}
@@ -3099,4 +3099,4 @@ class GatewayOperations:
case ReadResourceRequest(params=params):
return await _execute_read_resource(context, params, self._host_progress_callback)
case _:
- assert_never(operation)
+ return assert_never(operation)
diff --git a/litellm/proxy/_experimental/mcp_server/server.py b/litellm/proxy/_experimental/mcp_server/server.py
index 33978bb9182..a57eed04419 100644
--- a/litellm/proxy/_experimental/mcp_server/server.py
+++ b/litellm/proxy/_experimental/mcp_server/server.py
@@ -416,7 +416,10 @@ def _proxy_exception_to_http_exception(exc: ProxyException) -> HTTPException:
if MCP_AVAILABLE:
__all__ = (
"_MCP_CREDENTIAL_REQUEST_FIELDS",
+ "BlobResourceContents",
"ListMCPToolsRestAPIResponseObject",
+ "ResourceTemplate",
+ "TextResourceContents",
"_McpDeniedDetail",
"_aggregate_server_key",
"_build_virtual_call_logging_obj",
diff --git a/tests/test_litellm/proxy/_experimental/mcp_server/test_operations.py b/tests/test_litellm/proxy/_experimental/mcp_server/test_operations.py
index 870f2cd5f8f..abb925ddc77 100644
--- a/tests/test_litellm/proxy/_experimental/mcp_server/test_operations.py
+++ b/tests/test_litellm/proxy/_experimental/mcp_server/test_operations.py
@@ -8,6 +8,28 @@ from litellm.proxy._experimental.mcp_server.operations import GatewayOperations,
from litellm.proxy._types import UserAPIKeyAuth
+@pytest.mark.asyncio
+async def test_oauth_prefetch_failure_does_not_log_caller_or_exception_text(caplog):
+ from litellm.proxy._experimental.mcp_server.operations import _prefetch_oauth_creds_for_user
+
+ user_id = "caller\nFORGED-USER-LINE"
+ fetch = AsyncMock(side_effect=RuntimeError("database\nFORGED-ERROR-LINE"))
+ database = object()
+ with (
+ patch("litellm.proxy.utils.get_prisma_client_or_throw", return_value=database),
+ patch("litellm.proxy._experimental.mcp_server.db.list_user_oauth_credentials", fetch),
+ caplog.at_level("WARNING", logger="LiteLLM"),
+ ):
+ result = await _prefetch_oauth_creds_for_user(UserAPIKeyAuth(user_id=user_id))
+ assert result == {}
+ fetch.assert_awaited_once_with(database, user_id)
+ warnings = [record.getMessage() for record in caplog.records if "prefetch" in record.getMessage()]
+ assert len(warnings) == 1
+ assert "failed" in warnings[0]
+ assert "\n" not in warnings[0]
+ assert "FORGED" not in warnings[0]
+
+
@pytest.mark.asyncio
async def test_dispatch_uses_explicit_context_when_ambient_caller_differs():
from mcp.server.auth.middleware.auth_context import auth_context_var
From d2f457f144a430ad2848b33839d298d9312c8d4e Mon Sep 17 00:00:00 2001
From: Yujong Lee
Date: Mon, 21 Sep 2026 20:20:31 +0000
Subject: [PATCH 059/160] feat(cache): add semantic cache context and
unsupported operation error
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
---
litellm-rust/crates/cache/src/base_cache.rs | 50 +++++++++++++++++++++
litellm-rust/crates/cache/src/error.rs | 2 +
litellm-rust/crates/cache/src/lib.rs | 2 +-
3 files changed, 53 insertions(+), 1 deletion(-)
diff --git a/litellm-rust/crates/cache/src/base_cache.rs b/litellm-rust/crates/cache/src/base_cache.rs
index 8bd69ba5ad6..5c10e7fd5c3 100644
--- a/litellm-rust/crates/cache/src/base_cache.rs
+++ b/litellm-rust/crates/cache/src/base_cache.rs
@@ -32,6 +32,28 @@ impl CacheContext for ExactCacheContext {
}
}
+#[derive(Clone, Debug, Default, PartialEq)]
+pub struct SemanticCacheContext {
+ pub input: Option,
+ pub messages: Option,
+ pub metadata: Option,
+ pub scope: Option,
+ pub ttl: Option,
+}
+
+impl CacheContext for SemanticCacheContext {
+ fn ttl(&self) -> Option {
+ self.ttl
+ }
+
+ fn with_ttl(&self, ttl: Option) -> Self {
+ Self {
+ ttl,
+ ..self.clone()
+ }
+ }
+}
+
#[derive(Clone, Copy, Debug, Deserialize, Serialize, PartialEq, Eq)]
#[serde(rename_all = "lowercase")]
pub enum CacheConnectionStatus {
@@ -105,3 +127,31 @@ pub trait BaseCache: Send + Sync {
fn test_connection(&self) -> impl Future> + Send;
}
+
+#[cfg(test)]
+mod tests {
+ use std::time::Duration;
+
+ use serde_json::json;
+
+ use super::{CacheContext, SemanticCacheContext};
+
+ #[test]
+ fn semantic_context_with_ttl_only_replaces_ttl() {
+ let context = SemanticCacheContext {
+ input: Some(json!({"input": "hello"})),
+ messages: Some(json!([{"role": "user", "content": "hello"}])),
+ metadata: Some(json!({"tenant": "team"})),
+ scope: Some("scope".into()),
+ ttl: Some(Duration::from_secs(10)),
+ };
+
+ let updated = context.with_ttl(Some(Duration::from_secs(20)));
+
+ assert_eq!(updated.ttl, Some(Duration::from_secs(20)));
+ assert_eq!(updated.input, context.input);
+ assert_eq!(updated.messages, context.messages);
+ assert_eq!(updated.metadata, context.metadata);
+ assert_eq!(updated.scope, context.scope);
+ }
+}
diff --git a/litellm-rust/crates/cache/src/error.rs b/litellm-rust/crates/cache/src/error.rs
index ff3ff6572d4..51e4fe2d66a 100644
--- a/litellm-rust/crates/cache/src/error.rs
+++ b/litellm-rust/crates/cache/src/error.rs
@@ -6,4 +6,6 @@ pub enum Error {
InvalidEntry,
#[error("flushing Redis requires an explicit namespace")]
UnscopedFlush,
+ #[error("cache operation is not supported by this backend")]
+ UnsupportedOperation,
}
diff --git a/litellm-rust/crates/cache/src/lib.rs b/litellm-rust/crates/cache/src/lib.rs
index ce9f93b6dc4..8364c635e3a 100644
--- a/litellm-rust/crates/cache/src/lib.rs
+++ b/litellm-rust/crates/cache/src/lib.rs
@@ -8,7 +8,7 @@ mod error;
pub use base_cache::{
BaseCache, BatchEntry, CacheConnectionResult, CacheConnectionStatus, CacheContext,
- ExactCacheContext,
+ ExactCacheContext, SemanticCacheContext,
};
pub use cache_type::CacheType;
pub use caching::{Cache, CacheBackend, get_cache, set_cache};
From df97b274fc78ab261064f0a691016488c6c709dc Mon Sep 17 00:00:00 2001
From: Yujong Lee
Date: Mon, 21 Sep 2026 20:21:19 +0000
Subject: [PATCH 060/160] refactor(cache-response): generalize ResponseCache
over the backend context
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
---
litellm-rust/crates/cache-response/README.md | 2 +-
.../crates/cache-response/src/response.rs | 43 +++++++++++--------
.../crates/python-bridge/src/cache/request.rs | 3 +-
3 files changed, 28 insertions(+), 20 deletions(-)
diff --git a/litellm-rust/crates/cache-response/README.md b/litellm-rust/crates/cache-response/README.md
index 56c1646d343..46e561ddad1 100644
--- a/litellm-rust/crates/cache-response/README.md
+++ b/litellm-rust/crates/cache-response/README.md
@@ -58,4 +58,4 @@ Verify typed values, TTL precedence, missing entries, serialization failures, na
Public SDK, Router, and proxy activation still need constructor parity, stream replay, embedding partial-batch integration, response reconstruction, callback scheduling, and failure-policy integration. This foundation does not switch those request paths
-Redis cluster, disk, cloud stores, and semantic caching remain follow-ups. The generic dual cache takes read, write, and remote-failure policies, runs its async operations through the async L2 methods, and provides L2-first counters and atomic affinity claims. Errors propagate by default, and `RemoteFailurePolicy::UseLocal` opts key-value operations and claims into the local tier when L2 is unavailable. Claims compare decoded values, so a pin written by Python still matches. Public Router integration remains follow-up work. Reservations and pubsub still need explicit capabilities owned by their consuming features. Adding a cache backend does not establish those guarantees
+Redis cluster, disk, and cloud stores remain follow-ups. Semantic backends plug in through `SemanticCacheContext`, which carries the prompt inputs and metadata alongside the cache TTL. The generic dual cache takes read, write, and remote-failure policies, runs its async operations through the async L2 methods, and provides L2-first counters and atomic affinity claims. Errors propagate by default, and `RemoteFailurePolicy::UseLocal` opts key-value operations and claims into the local tier when L2 is unavailable. Claims compare decoded values, so a pin written by Python still matches. Public Router integration remains follow-up work. Reservations and pubsub still need explicit capabilities owned by their consuming features. Adding a cache backend does not establish those guarantees
diff --git a/litellm-rust/crates/cache-response/src/response.rs b/litellm-rust/crates/cache-response/src/response.rs
index e50e68cdabb..e70e07a5d26 100644
--- a/litellm-rust/crates/cache-response/src/response.rs
+++ b/litellm-rust/crates/cache-response/src/response.rs
@@ -1,21 +1,21 @@
use std::{sync::Arc, time::Duration};
use litellm_cache::{
- BaseCache, BatchCache, BatchEntry, CacheConnectionResult, Error, ExactCacheContext, FlushCache,
+ BaseCache, BatchCache, BatchEntry, CacheConnectionResult, CacheContext, Error, FlushCache,
};
use serde_json::Value;
use crate::{CacheControls, CacheEntry, CacheKeyInput, PartialHits, cache_key};
#[derive(Clone)]
-pub struct ResponseCacheRequest {
+pub struct ResponseCacheRequest {
pub key: CacheKeyInput,
pub controls: CacheControls,
- pub context: ExactCacheContext,
+ pub context: C,
pub max_age: Option,
}
-impl ResponseCacheRequest {
+impl ResponseCacheRequest {
pub fn new(key: CacheKeyInput) -> Self {
Self {
key,
@@ -26,17 +26,24 @@ impl ResponseCacheRequest {
default_on: true,
..Default::default()
},
- context: ExactCacheContext::default(),
+ context: C::default(),
max_age: None,
}
}
}
-pub struct ResponseCache> {
+pub struct ResponseCache>
+where
+ B::Context: Default + PartialEq,
+{
backend: Arc,
}
-impl> ResponseCache {
+impl ResponseCache
+where
+ B: BaseCache,
+ B::Context: Default + PartialEq,
+{
pub fn new(backend: Arc) -> Self {
Self { backend }
}
@@ -46,7 +53,7 @@ impl> ResponseCach
}
pub fn default_ttl(&self) -> Option {
- self.backend.get_ttl(&ExactCacheContext::default())
+ self.backend.get_ttl(&B::Context::default())
}
pub async fn async_flush(&self) -> Result<(), Error>
@@ -62,7 +69,7 @@ impl> ResponseCach
pub fn lookup(
&self,
- request: &ResponseCacheRequest,
+ request: &ResponseCacheRequest,
now: Duration,
) -> Result, Error> {
if !request.controls.reads() {
@@ -81,7 +88,7 @@ impl> ResponseCach
pub async fn async_lookup(
&self,
- request: &ResponseCacheRequest,
+ request: &ResponseCacheRequest,
now: Duration,
) -> Result, Error> {
if !request.controls.reads() {
@@ -101,7 +108,7 @@ impl> ResponseCach
pub fn lookup_batch(
&self,
- requests: &[ResponseCacheRequest],
+ requests: &[ResponseCacheRequest],
now: Duration,
) -> Result
where
@@ -126,7 +133,7 @@ impl> ResponseCach
pub async fn async_lookup_batch(
&self,
- requests: &[ResponseCacheRequest],
+ requests: &[ResponseCacheRequest],
now: Duration,
) -> Result
where
@@ -153,7 +160,7 @@ impl> ResponseCach
pub fn store(
&self,
- request: &ResponseCacheRequest,
+ request: &ResponseCacheRequest,
response: Value,
now: Duration,
) -> Result<(), Error> {
@@ -172,7 +179,7 @@ impl> ResponseCach
pub async fn async_store(
&self,
- request: &ResponseCacheRequest,
+ request: &ResponseCacheRequest,
response: Value,
now: Duration,
) -> Result<(), Error> {
@@ -193,7 +200,7 @@ impl> ResponseCach
pub async fn async_store_batch(
&self,
- entries: Vec<(ResponseCacheRequest, Value)>,
+ entries: Vec<(ResponseCacheRequest, Value)>,
now: Duration,
) -> Result<(), Error> {
self.async_store_entries(
@@ -209,7 +216,7 @@ impl> ResponseCach
/// the freshness of its original response.
pub async fn async_store_entries(
&self,
- entries: Vec<(ResponseCacheRequest, Value, Duration)>,
+ entries: Vec<(ResponseCacheRequest, Value, Duration)>,
) -> Result<(), Error> {
let writable = entries
.into_iter()
@@ -249,8 +256,8 @@ impl> ResponseCach
}
fn partial_hits(
- requests: &[ResponseCacheRequest],
- readable: Vec<(usize, &ResponseCacheRequest)>,
+ requests: &[ResponseCacheRequest],
+ readable: Vec<(usize, &ResponseCacheRequest)>,
entries: Vec>,
now: Duration,
) -> Result {
diff --git a/litellm-rust/crates/python-bridge/src/cache/request.rs b/litellm-rust/crates/python-bridge/src/cache/request.rs
index 0c5343a63d0..52a5f7d9055 100644
--- a/litellm-rust/crates/python-bridge/src/cache/request.rs
+++ b/litellm-rust/crates/python-bridge/src/cache/request.rs
@@ -1,5 +1,6 @@
use std::time::{Duration, SystemTime, UNIX_EPOCH};
+use litellm_cache::ExactCacheContext;
use litellm_cache_response::{CacheControls, CacheKeyInput, ResponseCacheRequest};
use litellm_host_python::from_py;
use pyo3::{exceptions::PyValueError, prelude::*};
@@ -20,7 +21,7 @@ pub(super) fn request(value: &Bound<'_, PyAny>) -> PyResult PyResult {
- let mut request = ResponseCacheRequest::new(input.key);
+ let mut request = ResponseCacheRequest::::new(input.key);
if let Some(controls) = input.controls {
request.controls = controls;
}
From 8dc960c928614c2276c8f5e7cc8d1658009ba796 Mon Sep 17 00:00:00 2001
From: Yujong Lee
Date: Mon, 21 Sep 2026 20:22:13 +0000
Subject: [PATCH 061/160] feat(cache): add SemanticCacheContext
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
---
litellm-rust/crates/cache/src/base_cache.rs | 25 +++++++++++++++++++++
litellm-rust/crates/cache/src/lib.rs | 2 +-
litellm-rust/crates/cache/tests/caching.rs | 24 +++++++++++++++++++-
3 files changed, 49 insertions(+), 2 deletions(-)
diff --git a/litellm-rust/crates/cache/src/base_cache.rs b/litellm-rust/crates/cache/src/base_cache.rs
index 8bd69ba5ad6..6f961798795 100644
--- a/litellm-rust/crates/cache/src/base_cache.rs
+++ b/litellm-rust/crates/cache/src/base_cache.rs
@@ -32,6 +32,31 @@ impl CacheContext for ExactCacheContext {
}
}
+#[derive(Clone, Debug, Default, PartialEq)]
+pub struct SemanticCacheContext {
+ pub input: Option,
+ pub messages: Vec,
+ pub metadata: serde_json::Map,
+ pub scope: Option,
+ pub ttl: Option,
+}
+
+impl CacheContext for SemanticCacheContext {
+ fn ttl(&self) -> Option {
+ self.ttl
+ }
+
+ fn with_ttl(&self, ttl: Option) -> Self {
+ Self {
+ input: self.input.clone(),
+ messages: self.messages.clone(),
+ metadata: self.metadata.clone(),
+ scope: self.scope.clone(),
+ ttl,
+ }
+ }
+}
+
#[derive(Clone, Copy, Debug, Deserialize, Serialize, PartialEq, Eq)]
#[serde(rename_all = "lowercase")]
pub enum CacheConnectionStatus {
diff --git a/litellm-rust/crates/cache/src/lib.rs b/litellm-rust/crates/cache/src/lib.rs
index ce9f93b6dc4..8364c635e3a 100644
--- a/litellm-rust/crates/cache/src/lib.rs
+++ b/litellm-rust/crates/cache/src/lib.rs
@@ -8,7 +8,7 @@ mod error;
pub use base_cache::{
BaseCache, BatchEntry, CacheConnectionResult, CacheConnectionStatus, CacheContext,
- ExactCacheContext,
+ ExactCacheContext, SemanticCacheContext,
};
pub use cache_type::CacheType;
pub use caching::{Cache, CacheBackend, get_cache, set_cache};
diff --git a/litellm-rust/crates/cache/tests/caching.rs b/litellm-rust/crates/cache/tests/caching.rs
index 9180ee9d0dc..2e65b4eeae5 100644
--- a/litellm-rust/crates/cache/tests/caching.rs
+++ b/litellm-rust/crates/cache/tests/caching.rs
@@ -1,7 +1,8 @@
use std::{sync::Mutex, time::Duration};
use litellm_cache::{
- BaseCache, CacheConnectionResult, CacheContext, Error, ExactCacheContext, get_cache,
+ BaseCache, CacheConnectionResult, CacheContext, Error, ExactCacheContext,
+ SemanticCacheContext, get_cache,
};
struct TestCache {
@@ -126,6 +127,27 @@ fn associated_context_preserves_backend_specific_lookup_inputs() {
);
}
+#[test]
+fn semantic_context_with_ttl_preserves_lookup_inputs() {
+ let context = SemanticCacheContext {
+ input: Some(serde_json::json!("text")),
+ messages: vec![serde_json::json!({"role": "user", "content": "hi"})],
+ metadata: serde_json::Map::from_iter([(
+ "key".into(),
+ serde_json::json!("value"),
+ )]),
+ scope: Some("scope".into()),
+ ttl: None,
+ };
+ let updated = context.with_ttl(Some(Duration::from_secs(30)));
+ assert_eq!(updated.ttl(), Some(Duration::from_secs(30)));
+ assert_eq!(updated.input, context.input);
+ assert_eq!(updated.messages, context.messages);
+ assert_eq!(updated.metadata, context.metadata);
+ assert_eq!(updated.scope, context.scope);
+ assert_eq!(context.with_ttl(None).ttl(), None);
+}
+
#[tokio::test]
async fn default_batch_operations_use_async_writes_and_stop_on_failure() {
let cache = TestCache {
From 1320eeeb41fbcd2a5842889e39f768d446fa6a05 Mon Sep 17 00:00:00 2001
From: Yujong Lee
Date: Mon, 21 Sep 2026 20:23:29 +0000
Subject: [PATCH 062/160] refactor(cache-response): generalize ResponseCache
over the backend context
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
---
.../crates/cache-response/src/response.rs | 50 +++++++++++--------
1 file changed, 30 insertions(+), 20 deletions(-)
diff --git a/litellm-rust/crates/cache-response/src/response.rs b/litellm-rust/crates/cache-response/src/response.rs
index e50e68cdabb..a27cc1967d5 100644
--- a/litellm-rust/crates/cache-response/src/response.rs
+++ b/litellm-rust/crates/cache-response/src/response.rs
@@ -1,21 +1,22 @@
use std::{sync::Arc, time::Duration};
use litellm_cache::{
- BaseCache, BatchCache, BatchEntry, CacheConnectionResult, Error, ExactCacheContext, FlushCache,
+ BaseCache, BatchCache, BatchEntry, CacheConnectionResult, CacheContext, Error,
+ ExactCacheContext, FlushCache,
};
use serde_json::Value;
use crate::{CacheControls, CacheEntry, CacheKeyInput, PartialHits, cache_key};
#[derive(Clone)]
-pub struct ResponseCacheRequest {
+pub struct ResponseCacheRequest {
pub key: CacheKeyInput,
pub controls: CacheControls,
- pub context: ExactCacheContext,
+ pub context: C,
pub max_age: Option,
}
-impl ResponseCacheRequest {
+impl ResponseCacheRequest {
pub fn new(key: CacheKeyInput) -> Self {
Self {
key,
@@ -32,11 +33,11 @@ impl ResponseCacheRequest {
}
}
-pub struct ResponseCache> {
+pub struct ResponseCache> {
backend: Arc,
}
-impl> ResponseCache {
+impl> ResponseCache {
pub fn new(backend: Arc) -> Self {
Self { backend }
}
@@ -45,8 +46,11 @@ impl> ResponseCach
&self.backend
}
- pub fn default_ttl(&self) -> Option {
- self.backend.get_ttl(&ExactCacheContext::default())
+ pub fn default_ttl(&self) -> Option
+ where
+ B::Context: Default,
+ {
+ self.backend.get_ttl(&B::Context::default())
}
pub async fn async_flush(&self) -> Result<(), Error>
@@ -62,7 +66,7 @@ impl> ResponseCach
pub fn lookup(
&self,
- request: &ResponseCacheRequest,
+ request: &ResponseCacheRequest,
now: Duration,
) -> Result, Error> {
if !request.controls.reads() {
@@ -81,7 +85,7 @@ impl> ResponseCach
pub async fn async_lookup(
&self,
- request: &ResponseCacheRequest,
+ request: &ResponseCacheRequest,
now: Duration,
) -> Result, Error> {
if !request.controls.reads() {
@@ -101,7 +105,7 @@ impl> ResponseCach
pub fn lookup_batch(
&self,
- requests: &[ResponseCacheRequest],
+ requests: &[ResponseCacheRequest],
now: Duration,
) -> Result
where
@@ -126,7 +130,7 @@ impl> ResponseCach
pub async fn async_lookup_batch(
&self,
- requests: &[ResponseCacheRequest],
+ requests: &[ResponseCacheRequest],
now: Duration,
) -> Result
where
@@ -153,7 +157,7 @@ impl> ResponseCach
pub fn store(
&self,
- request: &ResponseCacheRequest,
+ request: &ResponseCacheRequest,
response: Value,
now: Duration,
) -> Result<(), Error> {
@@ -172,7 +176,7 @@ impl> ResponseCach
pub async fn async_store(
&self,
- request: &ResponseCacheRequest,
+ request: &ResponseCacheRequest,
response: Value,
now: Duration,
) -> Result<(), Error> {
@@ -193,9 +197,12 @@ impl> ResponseCach
pub async fn async_store_batch(
&self,
- entries: Vec<(ResponseCacheRequest, Value)>,
+ entries: Vec<(ResponseCacheRequest, Value)>,
now: Duration,
- ) -> Result<(), Error> {
+ ) -> Result<(), Error>
+ where
+ B::Context: PartialEq,
+ {
self.async_store_entries(
entries
.into_iter()
@@ -209,8 +216,11 @@ impl> ResponseCach
/// the freshness of its original response.
pub async fn async_store_entries(
&self,
- entries: Vec<(ResponseCacheRequest, Value, Duration)>,
- ) -> Result<(), Error> {
+ entries: Vec<(ResponseCacheRequest, Value, Duration)>,
+ ) -> Result<(), Error>
+ where
+ B::Context: PartialEq,
+ {
let writable = entries
.into_iter()
.filter(|(request, _, _)| request.controls.writes())
@@ -249,8 +259,8 @@ impl> ResponseCach
}
fn partial_hits(
- requests: &[ResponseCacheRequest],
- readable: Vec<(usize, &ResponseCacheRequest)>,
+ requests: &[ResponseCacheRequest],
+ readable: Vec<(usize, &ResponseCacheRequest)>,
entries: Vec>,
now: Duration,
) -> Result {
From 8d9ab9eeaafe88efdf19fe67809e5fd21b2adb03 Mon Sep 17 00:00:00 2001
From: Yujong Lee
Date: Mon, 21 Sep 2026 20:24:23 +0000
Subject: [PATCH 063/160] feat(cache-redis): expose the pooled connection
handling for reuse
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
---
litellm-rust/crates/cache-redis/src/cache.rs | 93 +++++++++++--------
.../cache-redis/src/cache/operations.rs | 28 +++---
litellm-rust/crates/cache-redis/src/lib.rs | 4 +
3 files changed, 72 insertions(+), 53 deletions(-)
diff --git a/litellm-rust/crates/cache-redis/src/cache.rs b/litellm-rust/crates/cache-redis/src/cache.rs
index a960c383bf4..6388448accc 100644
--- a/litellm-rust/crates/cache-redis/src/cache.rs
+++ b/litellm-rust/crates/cache-redis/src/cache.rs
@@ -19,7 +19,7 @@ const DEFAULT_TTL: Duration = Duration::from_secs(600);
const REDIS_TIMEOUT: Duration = Duration::from_secs(5);
const REDIS_POOL_SIZE: u32 = 16;
-struct PooledConnection {
+pub struct PooledConnection {
connection: redis::Connection,
failed: bool,
}
@@ -27,16 +27,19 @@ struct PooledConnection {
/// Pools connections without a checkout PING, which would double every operation's round trips.
/// A timed-out command leaves its reply on the socket while redis still reports the connection
/// open, so any connection whose operation failed is discarded instead of being reused.
-struct ConnectionManager(redis::Client);
+pub struct ConnectionManager {
+ client: redis::Client,
+ timeout: Duration,
+}
impl r2d2::ManageConnection for ConnectionManager {
type Connection = PooledConnection;
type Error = redis::RedisError;
fn connect(&self) -> Result {
- let connection = self.0.get_connection()?;
- connection.set_read_timeout(Some(REDIS_TIMEOUT))?;
- connection.set_write_timeout(Some(REDIS_TIMEOUT))?;
+ let connection = self.client.get_connection()?;
+ connection.set_read_timeout(Some(self.timeout))?;
+ connection.set_write_timeout(Some(self.timeout))?;
Ok(PooledConnection {
connection,
failed: false,
@@ -68,12 +71,12 @@ const CLAIM_SCRIPT: &str = concat!(
);
const CLAIM_ATTEMPTS: usize = 8;
-enum Connections {
+pub enum Connections {
Pool(r2d2::Pool),
Fixed(Mutex),
}
-struct ConnectionRef<'a>(&'a mut dyn redis::ConnectionLike);
+pub struct ConnectionRef<'a>(&'a mut dyn redis::ConnectionLike);
impl redis::ConnectionLike for ConnectionRef<'_> {
fn req_packed_command(&mut self, cmd: &[u8]) -> redis::RedisResult {
@@ -110,7 +113,23 @@ impl Connections
where
C: redis::ConnectionLike + Send + 'static,
{
- fn execute(
+ pub fn pooled(url: &str, timeout: Duration, pool_size: u32) -> Result {
+ let client = redis::Client::open(url).map_err(|_| Error::Unavailable)?;
+ let pool = r2d2::Pool::builder()
+ .max_size(pool_size)
+ .min_idle(Some(0))
+ .connection_timeout(timeout)
+ .test_on_check_out(false)
+ .build(ConnectionManager { client, timeout })
+ .map_err(|_| Error::Unavailable)?;
+ Ok(Self::Pool(pool))
+ }
+
+ pub fn fixed(connection: C) -> Self {
+ Self::Fixed(Mutex::new(connection))
+ }
+
+ pub fn execute(
&self,
operation: impl FnOnce(&mut ConnectionRef<'_>) -> Result,
) -> Result {
@@ -127,6 +146,16 @@ where
}
}
}
+
+ pub async fn run_blocking(connections: Arc, operation: F) -> Result
+ where
+ T: Send + 'static,
+ F: FnOnce(&mut ConnectionRef<'_>) -> Result + Send + 'static,
+ {
+ tokio::task::spawn_blocking(move || connections.execute(operation))
+ .await
+ .map_err(|_| Error::Unavailable)?
+ }
}
pub struct RedisCache {
@@ -138,16 +167,8 @@ pub struct RedisCache {
impl RedisCache {
pub fn new(url: &str, default_ttl: Option, codec: S) -> Result {
- let client = redis::Client::open(url).map_err(|_| Error::Unavailable)?;
- let pool = r2d2::Pool::builder()
- .max_size(REDIS_POOL_SIZE)
- .min_idle(Some(0))
- .connection_timeout(REDIS_TIMEOUT)
- .test_on_check_out(false)
- .build(ConnectionManager(client))
- .map_err(|_| Error::Unavailable)?;
Ok(Self {
- connections: Arc::new(Connections::Pool(pool)),
+ connections: Arc::new(Connections::pooled(url, REDIS_TIMEOUT, REDIS_POOL_SIZE)?),
default_ttl: default_ttl.unwrap_or(DEFAULT_TTL),
codec,
namespace: None,
@@ -162,7 +183,7 @@ where
{
pub fn with_connection(connection: C, default_ttl: Option, codec: S) -> Self {
Self {
- connections: Arc::new(Connections::Fixed(Mutex::new(connection))),
+ connections: Arc::new(Connections::fixed(connection)),
default_ttl: default_ttl.unwrap_or(DEFAULT_TTL),
codec,
namespace: None,
@@ -241,20 +262,14 @@ where
}
fn ttl_seconds(ttl: Duration) -> u64 {
- ttl.as_secs()
- .saturating_add(u64::from(ttl.subsec_nanos() > 0))
- .max(1)
+ ttl_seconds(ttl)
}
+}
- async fn run_blocking(connections: Arc>, operation: F) -> Result
- where
- T: Send + 'static,
- F: FnOnce(&mut ConnectionRef<'_>) -> Result + Send + 'static,
- {
- tokio::task::spawn_blocking(move || connections.execute(operation))
- .await
- .map_err(|_| Error::Unavailable)?
- }
+pub fn ttl_seconds(ttl: Duration) -> u64 {
+ ttl.as_secs()
+ .saturating_add(u64::from(ttl.subsec_nanos() > 0))
+ .max(1)
}
fn namespaced_key(namespace: Option<&str>, key: &str) -> String {
@@ -313,7 +328,7 @@ where
let payload = self.codec.encode(&value)?;
let key = self.namespaced_key(key);
let ttl = Self::ttl_seconds(self.get_ttl(&context).unwrap_or(self.default_ttl));
- Self::run_blocking(Arc::clone(&self.connections), move |connection| {
+ Connections::run_blocking(Arc::clone(&self.connections), move |connection| {
connection
.set_ex::<_, _, ()>(key, payload, ttl)
.map_err(|_| Error::Unavailable)
@@ -327,7 +342,7 @@ where
_: &ExactCacheContext,
) -> Result, Error> {
let key = self.namespaced_key(key);
- let value = Self::run_blocking(Arc::clone(&self.connections), move |connection| {
+ let value = Connections::run_blocking(Arc::clone(&self.connections), move |connection| {
connection
.get::<_, redis::Value>(key)
.map_err(|_| Error::Unavailable)
@@ -350,7 +365,7 @@ where
})
.collect::, _>>()?;
let ttl = Self::ttl_seconds(self.get_ttl(&context).unwrap_or(self.default_ttl));
- Self::run_blocking(Arc::clone(&self.connections), move |connection| {
+ Connections::run_blocking(Arc::clone(&self.connections), move |connection| {
let mut pipeline = redis::pipe();
for (key, payload) in entries {
pipeline
@@ -372,7 +387,7 @@ where
}
async fn test_connection(&self) -> Result {
- match Self::run_blocking(Arc::clone(&self.connections), |connection| {
+ match Connections::run_blocking(Arc::clone(&self.connections), |connection| {
Ok(match redis::cmd("PING").query::(connection) {
Ok(_) => CacheConnectionResult {
status: CacheConnectionStatus::Success,
@@ -433,7 +448,7 @@ where
.iter()
.map(|key| self.namespaced_key(key))
.collect::>();
- let values = Self::run_blocking(Arc::clone(&self.connections), move |connection| {
+ let values = Connections::run_blocking(Arc::clone(&self.connections), move |connection| {
redis::cmd("MGET")
.arg(keys)
.query::>(connection)
@@ -460,7 +475,7 @@ where
async fn async_delete_cache(&self, key: &str) -> Result<(), Error> {
let key = self.namespaced_key(key);
- Self::run_blocking(Arc::clone(&self.connections), move |connection| {
+ Connections::run_blocking(Arc::clone(&self.connections), move |connection| {
connection.del::<_, ()>(key).map_err(|_| Error::Unavailable)
})
.await
@@ -480,7 +495,7 @@ where
async fn async_flush_cache(&self) -> Result<(), Error> {
let pattern = self.namespaced_pattern()?;
- Self::run_blocking(Arc::clone(&self.connections), move |connection| {
+ Connections::run_blocking(Arc::clone(&self.connections), move |connection| {
Self::flush_matching(connection, &pattern)
})
.await
@@ -512,7 +527,7 @@ where
) -> Result {
let key = self.namespaced_key(key);
let ttl = Self::ttl_seconds(self.get_ttl(&context).unwrap_or(self.default_ttl));
- Self::run_blocking(Arc::clone(&self.connections), move |connection| {
+ Connections::run_blocking(Arc::clone(&self.connections), move |connection| {
increment(connection, key, amount, ttl)
})
.await
@@ -623,7 +638,7 @@ where
let key = self.namespaced_key(key);
let ttl = Self::ttl_seconds(self.get_ttl(&context).unwrap_or(self.default_ttl));
let codec = self.codec.clone();
- Self::run_blocking(Arc::clone(&self.connections), move |connection| {
+ Connections::run_blocking(Arc::clone(&self.connections), move |connection| {
claim(connection, &codec, &key, candidate, &eligible, ttl)
})
.await
diff --git a/litellm-rust/crates/cache-redis/src/cache/operations.rs b/litellm-rust/crates/cache-redis/src/cache/operations.rs
index d8d9ae24c4c..f27a7802bab 100644
--- a/litellm-rust/crates/cache-redis/src/cache/operations.rs
+++ b/litellm-rust/crates/cache-redis/src/cache/operations.rs
@@ -144,7 +144,7 @@ where
.into_iter()
.map(|key| self.namespaced_key(&key))
.collect::>();
- Self::run_blocking(Arc::clone(&self.connections), move |connection| {
+ Connections::run_blocking(Arc::clone(&self.connections), move |connection| {
connection.del(keys).map_err(|_| Error::Unavailable)
})
.await
@@ -172,7 +172,7 @@ where
.iter()
.map(|key| self.namespaced_key(key))
.collect::>();
- let values = Self::run_blocking(Arc::clone(&self.connections), move |connection| {
+ let values = Connections::run_blocking(Arc::clone(&self.connections), move |connection| {
redis::cmd("MGET")
.arg(keys)
.query::>(connection)
@@ -192,7 +192,7 @@ where
}
pub async fn ping(&self) -> Result {
- Self::run_blocking(Arc::clone(&self.connections), |connection| {
+ Connections::run_blocking(Arc::clone(&self.connections), |connection| {
redis::cmd("PING")
.query::(connection)
.map(|response| response == "PONG")
@@ -203,7 +203,7 @@ where
pub async fn async_get_ttl(&self, key: &str) -> Result, Error> {
let key = self.namespaced_key(key);
- let ttl = Self::run_blocking(Arc::clone(&self.connections), move |connection| {
+ let ttl = Connections::run_blocking(Arc::clone(&self.connections), move |connection| {
redis::cmd("TTL")
.arg(key)
.query::(connection)
@@ -215,7 +215,7 @@ where
pub async fn async_scan_iter(&self, pattern: &str, count: usize) -> Result, Error> {
let pattern = format!("{}*", self.namespaced_key(pattern));
- Self::run_blocking(Arc::clone(&self.connections), move |connection| {
+ Connections::run_blocking(Arc::clone(&self.connections), move |connection| {
let mut cursor = 0u64;
let mut matches = Vec::new();
loop {
@@ -249,7 +249,7 @@ where
}
let key = self.namespaced_key(key);
let ttl = Self::ttl_seconds(ttl.unwrap_or(self.default_ttl));
- Self::run_blocking(Arc::clone(&self.connections), move |connection| {
+ Connections::run_blocking(Arc::clone(&self.connections), move |connection| {
let mut pipeline = redis::pipe();
pipeline.cmd("SADD").arg(&key).arg(values);
pipeline.cmd("EXPIRE").arg(&key).arg(ttl).ignore();
@@ -266,7 +266,7 @@ where
return Err(Error::InvalidEntry);
}
let key = self.namespaced_key(key);
- Self::run_blocking(Arc::clone(&self.connections), move |connection| {
+ Connections::run_blocking(Arc::clone(&self.connections), move |connection| {
redis::cmd("RPUSH")
.arg(key)
.arg(values)
@@ -292,7 +292,7 @@ where
if operations.is_empty() {
return Ok(Vec::new());
}
- Self::run_blocking(Arc::clone(&self.connections), move |connection| {
+ Connections::run_blocking(Arc::clone(&self.connections), move |connection| {
let mut pipeline = redis::pipe();
for (key, values) in operations {
pipeline.cmd("RPUSH").arg(key).arg(values);
@@ -309,7 +309,7 @@ where
) -> Result {
let key = self.namespaced_key(key);
let multiple = count.is_some();
- let value = Self::run_blocking(Arc::clone(&self.connections), move |connection| {
+ let value = Connections::run_blocking(Arc::clone(&self.connections), move |connection| {
let mut command = redis::cmd("LPOP");
command.arg(key);
if let Some(count) = count {
@@ -338,7 +338,7 @@ where
.iter()
.map(|(_, count)| count.is_some())
.collect::>();
- let values = Self::run_blocking(Arc::clone(&self.connections), move |connection| {
+ let values = Connections::run_blocking(Arc::clone(&self.connections), move |connection| {
let mut pipeline = redis::pipe();
for (key, count) in operations {
let command = pipeline.cmd("LPOP").arg(key);
@@ -368,7 +368,7 @@ where
.into_iter()
.map(|key| self.namespaced_key(&key))
.collect::>();
- Self::run_blocking(Arc::clone(&self.connections), move |connection| {
+ Connections::run_blocking(Arc::clone(&self.connections), move |connection| {
redis::cmd("EVAL")
.arg(script)
.arg(keys.len())
@@ -440,7 +440,7 @@ where
if operations.is_empty() {
return Ok(Vec::new());
}
- Self::run_blocking(Arc::clone(&self.connections), move |connection| {
+ Connections::run_blocking(Arc::clone(&self.connections), move |connection| {
let mut pipeline = redis::pipe();
for (key, amount, ttl) in operations {
pipeline.cmd("INCRBYFLOAT").arg(&key).arg(amount);
@@ -461,7 +461,7 @@ where
) -> Result {
let key = self.namespaced_key(key);
let ttl = Self::ttl_seconds(ttl);
- Self::run_blocking(Arc::clone(&self.connections), move |connection| {
+ Connections::run_blocking(Arc::clone(&self.connections), move |connection| {
increment_with_floor(connection, key, amount, ttl)
})
.await
@@ -475,7 +475,7 @@ where
) -> Result {
let key = self.namespaced_key(key);
let ttl = Self::ttl_seconds(ttl.unwrap_or(self.default_ttl));
- Self::run_blocking(Arc::clone(&self.connections), move |connection| {
+ Connections::run_blocking(Arc::clone(&self.connections), move |connection| {
redis::cmd("EVAL")
.arg(SET_MAX_SCRIPT)
.arg(1)
diff --git a/litellm-rust/crates/cache-redis/src/lib.rs b/litellm-rust/crates/cache-redis/src/lib.rs
index 98f6bfd8ce5..ea75906e9c9 100644
--- a/litellm-rust/crates/cache-redis/src/lib.rs
+++ b/litellm-rust/crates/cache-redis/src/lib.rs
@@ -1,6 +1,10 @@
mod cache;
mod topology;
+pub mod connection {
+ pub use crate::cache::{ConnectionRef, Connections, ttl_seconds};
+}
+
pub use cache::{
RedisArg, RedisCache, RedisLpopOperation, RedisLpopResult, RedisRpushOperation, RedisScript,
};
From 1f86bb8e4640fd7e758e10f66106fd7c1bda01de Mon Sep 17 00:00:00 2001
From: Yujong Lee
Date: Mon, 21 Sep 2026 20:24:34 +0000
Subject: [PATCH 064/160] feat(cache-valkey-semantic): add native Valkey
semantic cache backend
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
---
litellm-rust/Cargo.lock | 16 +
.../crates/cache-valkey-semantic/Cargo.toml | 20 +
.../crates/cache-valkey-semantic/src/lib.rs | 844 ++++++++++++++++++
3 files changed, 880 insertions(+)
create mode 100644 litellm-rust/crates/cache-valkey-semantic/Cargo.toml
create mode 100644 litellm-rust/crates/cache-valkey-semantic/src/lib.rs
diff --git a/litellm-rust/Cargo.lock b/litellm-rust/Cargo.lock
index ed4ae4e3353..5daf691d4c3 100644
--- a/litellm-rust/Cargo.lock
+++ b/litellm-rust/Cargo.lock
@@ -2502,6 +2502,22 @@ dependencies = [
"tokio",
]
+[[package]]
+name = "litellm-cache-valkey-semantic"
+version = "0.1.0"
+dependencies = [
+ "litellm-cache",
+ "litellm-cache-response",
+ "r2d2",
+ "redis",
+ "redis-test",
+ "rstest",
+ "serde_json",
+ "sha2 0.10.9",
+ "tokio",
+ "uuid",
+]
+
[[package]]
name = "litellm-callbacks-legacy-python"
version = "0.1.0"
diff --git a/litellm-rust/crates/cache-valkey-semantic/Cargo.toml b/litellm-rust/crates/cache-valkey-semantic/Cargo.toml
new file mode 100644
index 00000000000..9a0a566ca3b
--- /dev/null
+++ b/litellm-rust/crates/cache-valkey-semantic/Cargo.toml
@@ -0,0 +1,20 @@
+[package]
+name = "litellm-cache-valkey-semantic"
+version = "0.1.0"
+edition.workspace = true
+license.workspace = true
+repository.workspace = true
+
+[dependencies]
+litellm-cache.workspace = true
+litellm-cache-response.workspace = true
+r2d2 = "0.8.10"
+redis = { version = "1.7.0", features = ["tls-rustls"] }
+serde_json.workspace = true
+sha2.workspace = true
+tokio.workspace = true
+uuid = { version = "1", features = ["v4"] }
+
+[dev-dependencies]
+redis-test = "1.0.4"
+rstest.workspace = true
diff --git a/litellm-rust/crates/cache-valkey-semantic/src/lib.rs b/litellm-rust/crates/cache-valkey-semantic/src/lib.rs
new file mode 100644
index 00000000000..85c4c9af15c
--- /dev/null
+++ b/litellm-rust/crates/cache-valkey-semantic/src/lib.rs
@@ -0,0 +1,844 @@
+use std::{
+ future::Future,
+ sync::{Arc, Mutex},
+ time::Duration,
+};
+
+use litellm_cache::{BaseCache, CacheCodec, CacheConnectionResult, Error, SemanticCacheContext};
+use litellm_cache_response::CacheEntry;
+use serde_json::Value;
+use sha2::{Digest, Sha256};
+use uuid::Uuid;
+
+pub trait Embedder: Send + Sync + 'static {
+ fn embed(&self, prompt: &str, metadata: Option<&Value>) -> Result, Error>;
+
+ fn async_embed(
+ &self,
+ prompt: &str,
+ metadata: Option<&Value>,
+ ) -> impl Future, Error>> + Send;
+}
+
+#[derive(Clone, Debug, PartialEq)]
+pub struct ValkeySemanticConfig {
+ pub similarity_threshold: f64,
+ pub index_name: String,
+}
+
+pub const DEFAULT_INDEX_NAME: &str = "litellm_semantic_cache_index";
+
+struct PooledConnection {
+ connection: redis::Connection,
+ failed: bool,
+}
+
+struct ConnectionManager(redis::Client);
+
+impl r2d2::ManageConnection for ConnectionManager {
+ type Connection = PooledConnection;
+ type Error = redis::RedisError;
+
+ fn connect(&self) -> Result {
+ let connection = self.0.get_connection()?;
+ Ok(PooledConnection {
+ connection,
+ failed: false,
+ })
+ }
+
+ fn is_valid(&self, connection: &mut Self::Connection) -> Result<(), Self::Error> {
+ redis::cmd("PING").query::(&mut connection.connection)?;
+ Ok(())
+ }
+
+ fn has_broken(&self, connection: &mut Self::Connection) -> bool {
+ connection.failed || !redis::ConnectionLike::is_open(&connection.connection)
+ }
+}
+
+enum Connections {
+ Pool(r2d2::Pool),
+ Fixed(Mutex),
+}
+
+struct ConnectionRef<'a>(&'a mut dyn redis::ConnectionLike);
+
+impl redis::ConnectionLike for ConnectionRef<'_> {
+ fn req_packed_command(&mut self, cmd: &[u8]) -> redis::RedisResult {
+ self.0.req_packed_command(cmd)
+ }
+
+ fn req_packed_commands(
+ &mut self,
+ cmd: &[u8],
+ offset: usize,
+ count: usize,
+ ) -> redis::RedisResult> {
+ self.0.req_packed_commands(cmd, offset, count)
+ }
+
+ fn get_db(&self) -> i64 {
+ self.0.get_db()
+ }
+
+ fn supports_pipelining(&self) -> bool {
+ self.0.supports_pipelining()
+ }
+
+ fn check_connection(&mut self) -> bool {
+ self.0.check_connection()
+ }
+
+ fn is_open(&self) -> bool {
+ self.0.is_open()
+ }
+}
+
+impl Connections
+where
+ C: redis::ConnectionLike + Send + 'static,
+{
+ fn execute(
+ &self,
+ operation: impl FnOnce(&mut ConnectionRef<'_>) -> Result,
+ ) -> Result {
+ match self {
+ Self::Pool(pool) => {
+ let mut pooled = pool.get().map_err(|_| Error::Unavailable)?;
+ let result = operation(&mut ConnectionRef(&mut pooled.connection));
+ pooled.failed = matches!(result, Err(Error::Unavailable));
+ result
+ }
+ Self::Fixed(connection) => {
+ let mut connection = connection.lock().map_err(|_| Error::Unavailable)?;
+ operation(&mut ConnectionRef(&mut *connection))
+ }
+ }
+ }
+}
+
+pub struct ValkeySemanticCache<
+ E: Embedder,
+ S: CacheCodec,
+ C = redis::Connection,
+> {
+ connections: Arc>,
+ embedder: E,
+ codec: S,
+ config: ValkeySemanticConfig,
+ index_dimension: Arc>>,
+}
+
+impl ValkeySemanticCache
+where
+ E: Embedder,
+ S: CacheCodec,
+{
+ pub fn new(
+ url: &str,
+ embedder: E,
+ codec: S,
+ config: ValkeySemanticConfig,
+ ) -> Result {
+ let client = redis::Client::open(url).map_err(|_| Error::Unavailable)?;
+ let pool = r2d2::Pool::builder()
+ .max_size(16)
+ .min_idle(Some(0))
+ .test_on_check_out(false)
+ .build(ConnectionManager(client))
+ .map_err(|_| Error::Unavailable)?;
+ Ok(Self {
+ connections: Arc::new(Connections::Pool(pool)),
+ embedder,
+ codec,
+ config,
+ index_dimension: Arc::new(Mutex::new(None)),
+ })
+ }
+}
+
+impl ValkeySemanticCache
+where
+ E: Embedder,
+ S: CacheCodec,
+ C: redis::ConnectionLike + Send + 'static,
+{
+ pub fn with_connection(
+ connection: C,
+ embedder: E,
+ codec: S,
+ config: ValkeySemanticConfig,
+ ) -> Self {
+ Self {
+ connections: Arc::new(Connections::Fixed(Mutex::new(connection))),
+ embedder,
+ codec,
+ config,
+ index_dimension: Arc::new(Mutex::new(None)),
+ }
+ }
+
+ pub fn similarity_threshold(&self) -> f64 {
+ self.config.similarity_threshold
+ }
+
+ pub fn index_name(&self) -> &str {
+ &self.config.index_name
+ }
+
+ fn key_prefix(&self) -> String {
+ format!("{}:", self.config.index_name)
+ }
+
+ fn ensure_index(&self, dimension: usize) -> Result<(), Error> {
+ ensure_index(
+ &self.connections,
+ &self.config.index_name,
+ &self.key_prefix(),
+ &self.index_dimension,
+ dimension,
+ )
+ }
+}
+
+impl BaseCache for ValkeySemanticCache
+where
+ E: Embedder,
+ S: CacheCodec,
+ C: redis::ConnectionLike + Send + 'static,
+{
+ type Value = CacheEntry;
+ type Context = SemanticCacheContext;
+
+ fn get_ttl(&self, context: &Self::Context) -> Option {
+ context.ttl
+ }
+
+ fn set_cache(
+ &self,
+ key: &str,
+ value: Self::Value,
+ context: &Self::Context,
+ ) -> Result<(), Error> {
+ let Some(prompt) = prompt_from_context(context) else {
+ return Ok(());
+ };
+ let embedding = self.embedder.embed(&prompt, context.metadata.as_ref())?;
+ self.ensure_index(embedding.len())?;
+ let scope = scope_tag(key);
+ let document = format!("{}{}:{}", self.key_prefix(), scope, Uuid::new_v4());
+ let response = self.codec.encode(&value)?;
+ let vector = embedding_bytes(&embedding);
+ let ttl = self.get_ttl(context);
+ self.connections.execute(|connection| {
+ let mut pipeline = redis::pipe();
+ pipeline
+ .cmd("HSET")
+ .arg(&document)
+ .arg("litellm_cache_key")
+ .arg(&scope)
+ .arg("prompt")
+ .arg(prompt)
+ .arg("response")
+ .arg(response)
+ .arg("embedding")
+ .arg(vector)
+ .ignore();
+ if let Some(ttl) = ttl {
+ pipeline
+ .cmd("EXPIRE")
+ .arg(&document)
+ .arg(ttl.as_secs())
+ .ignore();
+ }
+ pipeline
+ .query::<()>(connection)
+ .map_err(|_| Error::Unavailable)
+ })
+ }
+
+ fn get_cache(&self, key: &str, context: &Self::Context) -> Result, Error> {
+ let Some(prompt) = prompt_from_context(context) else {
+ return Ok(None);
+ };
+ let embedding = self.embedder.embed(&prompt, context.metadata.as_ref())?;
+ self.ensure_index(embedding.len())?;
+ let scope = scope_tag(key);
+ let query =
+ format!("(@litellm_cache_key:{{{scope}}})=>[KNN 1 @embedding $vec AS vector_distance]");
+ let vector = embedding_bytes(&embedding);
+ let response = self.connections.execute(|connection| {
+ redis::cmd("FT.SEARCH")
+ .arg(&self.config.index_name)
+ .arg(query)
+ .arg("PARAMS")
+ .arg(2)
+ .arg("vec")
+ .arg(vector)
+ .arg("RETURN")
+ .arg(2)
+ .arg("response")
+ .arg("vector_distance")
+ .arg("DIALECT")
+ .arg(2)
+ .query::(connection)
+ .map_err(|_| Error::Unavailable)
+ })?;
+ let Some(fields) = search_fields(response)? else {
+ return Ok(None);
+ };
+ let response = fields
+ .iter()
+ .find_map(|(name, value)| (name == "response").then(|| value.clone()))
+ .ok_or(Error::InvalidEntry)?;
+ let distance = fields
+ .iter()
+ .find_map(|(name, value)| (name == "vector_distance").then(|| value.clone()))
+ .ok_or(Error::InvalidEntry)?;
+ let distance = parse_f64(&distance)?;
+ if 1.0 - distance < self.config.similarity_threshold {
+ return Ok(None);
+ }
+ self.codec.decode(&response).map(Some)
+ }
+
+ fn async_set_cache(
+ &self,
+ key: &str,
+ value: Self::Value,
+ context: Self::Context,
+ ) -> impl Future> + Send {
+ let key = key.to_owned();
+ let prompt = prompt_from_context(&context);
+ let metadata = context.metadata.clone();
+ async move {
+ let Some(prompt) = prompt else {
+ return Ok(());
+ };
+ let embedding = self
+ .embedder
+ .async_embed(&prompt, metadata.as_ref())
+ .await?;
+ let connections = Arc::clone(&self.connections);
+ let config = self.config.clone();
+ let index_dimension = Arc::clone(&self.index_dimension);
+ let response = self.codec.encode(&value)?;
+ let vector = embedding_bytes(&embedding);
+ let prefix = format!("{}:", config.index_name);
+ let scope = scope_tag(&key);
+ let document = format!("{prefix}{scope}:{}", Uuid::new_v4());
+ let ttl = context.ttl;
+ tokio::task::spawn_blocking(move || {
+ ensure_index(
+ &connections,
+ &config.index_name,
+ &prefix,
+ &index_dimension,
+ embedding.len(),
+ )?;
+ connections.execute(|connection| {
+ let mut pipeline = redis::pipe();
+ pipeline
+ .cmd("HSET")
+ .arg(&document)
+ .arg("litellm_cache_key")
+ .arg(&scope)
+ .arg("prompt")
+ .arg(prompt)
+ .arg("response")
+ .arg(response)
+ .arg("embedding")
+ .arg(vector)
+ .ignore();
+ if let Some(ttl) = ttl {
+ pipeline
+ .cmd("EXPIRE")
+ .arg(&document)
+ .arg(ttl.as_secs())
+ .ignore();
+ }
+ pipeline
+ .query::<()>(connection)
+ .map_err(|_| Error::Unavailable)
+ })
+ })
+ .await
+ .map_err(|_| Error::Unavailable)?
+ }
+ }
+
+ fn async_get_cache(
+ &self,
+ key: &str,
+ context: &Self::Context,
+ ) -> impl Future, Error>> + Send {
+ let key = key.to_owned();
+ let prompt = prompt_from_context(context);
+ let metadata = context.metadata.clone();
+ async move {
+ let Some(prompt) = prompt else {
+ return Ok(None);
+ };
+ let embedding = self
+ .embedder
+ .async_embed(&prompt, metadata.as_ref())
+ .await?;
+ let connections = Arc::clone(&self.connections);
+ let config = self.config.clone();
+ let index_dimension = Arc::clone(&self.index_dimension);
+ let threshold = config.similarity_threshold;
+ tokio::task::spawn_blocking(move || {
+ let prefix = format!("{}:", config.index_name);
+ ensure_index(
+ &connections,
+ &config.index_name,
+ &prefix,
+ &index_dimension,
+ embedding.len(),
+ )?;
+ let scope = scope_tag(&key);
+ let query = format!(
+ "(@litellm_cache_key:{{{scope}}})=>[KNN 1 @embedding $vec AS vector_distance]"
+ );
+ let vector = embedding_bytes(&embedding);
+ let response = connections.execute(|connection| {
+ redis::cmd("FT.SEARCH")
+ .arg(&config.index_name)
+ .arg(query)
+ .arg("PARAMS")
+ .arg(2)
+ .arg("vec")
+ .arg(vector)
+ .arg("RETURN")
+ .arg(2)
+ .arg("response")
+ .arg("vector_distance")
+ .arg("DIALECT")
+ .arg(2)
+ .query::(connection)
+ .map_err(|_| Error::Unavailable)
+ })?;
+ let Some(fields) = search_fields(response)? else {
+ return Ok(None);
+ };
+ let response = fields
+ .iter()
+ .find_map(|(name, value)| (name == "response").then(|| value.clone()))
+ .ok_or(Error::InvalidEntry)?;
+ let distance = fields
+ .iter()
+ .find_map(|(name, value)| (name == "vector_distance").then(|| value.clone()))
+ .ok_or(Error::InvalidEntry)?;
+ let distance = parse_f64(&distance)?;
+ if 1.0 - distance < threshold {
+ return Ok(None);
+ }
+ Ok(Some(response))
+ })
+ .await
+ .map_err(|_| Error::Unavailable)?
+ .and_then(|response| response.map(|bytes| self.codec.decode(&bytes)).transpose())
+ }
+ }
+
+ async fn disconnect(&self) -> Result<(), Error> {
+ Ok(())
+ }
+
+ async fn test_connection(&self) -> Result {
+ Err(Error::UnsupportedOperation)
+ }
+}
+
+pub fn prompt_from_context(context: &SemanticCacheContext) -> Option {
+ if let Some(Value::Array(messages)) = context.messages.as_ref()
+ && !messages.is_empty()
+ {
+ return Some(
+ messages
+ .iter()
+ .filter_map(Value::as_object)
+ .map(message_text)
+ .collect(),
+ );
+ }
+ let input = context.input.as_ref()?;
+ let mut parts = Vec::new();
+ collect_input_text(input, &mut parts);
+ let prompt = parts.join("\n").trim().to_owned();
+ (!prompt.is_empty()).then_some(prompt)
+}
+
+fn message_text(message: &serde_json::Map) -> String {
+ let content = match message.get("content") {
+ Some(Value::String(value)) => value.clone(),
+ Some(Value::Array(parts)) => parts
+ .iter()
+ .filter_map(Value::as_object)
+ .filter_map(|part| part.get("text").and_then(Value::as_str))
+ .filter(|text| !text.is_empty())
+ .collect(),
+ _ => String::new(),
+ };
+ format!(
+ "{content}{}",
+ search_results_text(message.get("search_results"))
+ )
+}
+
+fn search_results_text(value: Option<&Value>) -> String {
+ let Some(Value::Array(results)) = value else {
+ return String::new();
+ };
+ results
+ .iter()
+ .filter_map(Value::as_object)
+ .map(|result| {
+ let source = result.get("source").and_then(Value::as_str).unwrap_or("");
+ let title = result.get("title").and_then(Value::as_str).unwrap_or("");
+ let content = result
+ .get("content")
+ .and_then(Value::as_array)
+ .map(|blocks| {
+ blocks
+ .iter()
+ .filter_map(Value::as_object)
+ .filter_map(|block| block.get("text").and_then(Value::as_str))
+ .collect::()
+ })
+ .unwrap_or_default();
+ let citations = result
+ .get("citations")
+ .filter(|value| !value.is_null())
+ .and_then(|value| serde_json::to_string(value).ok())
+ .unwrap_or_default();
+ format!("{source}{title}{content}{citations}")
+ })
+ .collect()
+}
+
+fn collect_input_text(value: &Value, parts: &mut Vec) {
+ match value {
+ Value::String(value) => {
+ let value = value.trim();
+ if !value.is_empty() {
+ parts.push(value.to_owned());
+ }
+ }
+ Value::Array(values) => values
+ .iter()
+ .for_each(|value| collect_input_text(value, parts)),
+ Value::Object(object) => {
+ if let Some(content) = object.get("content").filter(|value| !value.is_null()) {
+ collect_input_text(content, parts);
+ return;
+ }
+ for key in ["text", "output", "input_text", "output_text"] {
+ if let Some(Value::String(value)) = object.get(key) {
+ let value = value.trim();
+ if !value.is_empty() {
+ parts.push(value.to_owned());
+ return;
+ }
+ }
+ }
+ }
+ _ => {}
+ }
+}
+
+fn scope_tag(key: &str) -> String {
+ let digest = Sha256::digest(key.as_bytes());
+ digest.iter().map(|byte| format!("{byte:02x}")).collect()
+}
+
+fn embedding_bytes(embedding: &[f32]) -> Vec {
+ embedding
+ .iter()
+ .flat_map(|value| value.to_le_bytes())
+ .collect()
+}
+
+fn ensure_index(
+ connections: &Connections,
+ index_name: &str,
+ prefix: &str,
+ index_dimension: &Mutex>,
+ dimension: usize,
+) -> Result<(), Error>
+where
+ C: redis::ConnectionLike + Send + 'static,
+{
+ if index_dimension
+ .lock()
+ .map_err(|_| Error::Unavailable)?
+ .is_some_and(|existing| existing == dimension)
+ {
+ return Ok(());
+ }
+ let create = connections.execute(|connection| {
+ Ok(redis::cmd("FT.CREATE")
+ .arg(index_name)
+ .arg("ON")
+ .arg("HASH")
+ .arg("PREFIX")
+ .arg(1)
+ .arg(prefix)
+ .arg("SCHEMA")
+ .arg("litellm_cache_key")
+ .arg("TAG")
+ .arg("embedding")
+ .arg("VECTOR")
+ .arg("HNSW")
+ .arg(6)
+ .arg("TYPE")
+ .arg("FLOAT32")
+ .arg("DIM")
+ .arg(dimension)
+ .arg("DISTANCE_METRIC")
+ .arg("COSINE")
+ .query::(connection)
+ .map(|_| ())
+ .map_err(|error| error.to_string()))
+ })?;
+ if let Err(message) = create {
+ if !message.to_ascii_lowercase().contains("already exists") {
+ return Err(Error::Unavailable);
+ }
+ let info = connections.execute(|connection| {
+ redis::cmd("FT.INFO")
+ .arg(index_name)
+ .query::(connection)
+ .map_err(|_| Error::Unavailable)
+ })?;
+ let existing = index_dimension_from_info(&info).ok_or(Error::Unavailable)?;
+ if existing != dimension {
+ return Err(Error::Unavailable);
+ }
+ }
+ *index_dimension.lock().map_err(|_| Error::Unavailable)? = Some(dimension);
+ Ok(())
+}
+
+fn index_dimension_from_info(value: &redis::Value) -> Option {
+ let redis::Value::Array(values) = value else {
+ return None;
+ };
+ let attributes = values.windows(2).find_map(|pair| {
+ (value_text(&pair[0]).as_deref() == Some("attributes")).then_some(&pair[1])
+ })?;
+ let redis::Value::Array(fields) = attributes else {
+ return None;
+ };
+ fields.iter().find_map(|field| {
+ let redis::Value::Array(values) = field else {
+ return None;
+ };
+ let flattened = values.iter().flat_map(|value| match value {
+ redis::Value::Array(values) => values.as_slice(),
+ _ => std::slice::from_ref(value),
+ });
+ let values = flattened.collect::>();
+ values.windows(2).find_map(|pair| {
+ if value_text(pair[0]).as_deref() == Some("dimensions") {
+ return value_text(pair[1]).and_then(|value| value.parse().ok());
+ }
+ None
+ })
+ })
+}
+
+type SearchFields = Vec<(String, Vec)>;
+
+fn search_fields(value: redis::Value) -> Result, Error> {
+ let redis::Value::Array(values) = value else {
+ return Err(Error::InvalidEntry);
+ };
+ let total = parse_i64(values.first().ok_or(Error::InvalidEntry)?)?;
+ if total <= 0 || values.len() < 3 {
+ return Ok(None);
+ }
+ let redis::Value::Array(fields) = &values[2] else {
+ return Err(Error::InvalidEntry);
+ };
+ let (pairs, remainder) = fields.as_chunks::<2>();
+ if !remainder.is_empty() {
+ return Err(Error::InvalidEntry);
+ }
+ let pairs = pairs
+ .iter()
+ .map(|pair| {
+ Ok((
+ value_text(&pair[0]).ok_or(Error::InvalidEntry)?,
+ value_bytes(&pair[1])?,
+ ))
+ })
+ .collect::, Error>>()?;
+ Ok(Some(pairs))
+}
+
+fn parse_i64(value: &redis::Value) -> Result {
+ value_text(value)
+ .ok_or(Error::InvalidEntry)?
+ .parse()
+ .map_err(|_| Error::InvalidEntry)
+}
+
+fn parse_f64(value: &[u8]) -> Result {
+ std::str::from_utf8(value)
+ .map_err(|_| Error::InvalidEntry)?
+ .parse()
+ .map_err(|_| Error::InvalidEntry)
+}
+
+fn value_text(value: &redis::Value) -> Option {
+ match value {
+ redis::Value::BulkString(bytes) => String::from_utf8(bytes.clone()).ok(),
+ redis::Value::SimpleString(value) => Some(value.clone()),
+ redis::Value::Int(value) => Some(value.to_string()),
+ _ => None,
+ }
+}
+
+fn value_bytes(value: &redis::Value) -> Result, Error> {
+ match value {
+ redis::Value::BulkString(bytes) => Ok(bytes.clone()),
+ redis::Value::SimpleString(value) => Ok(value.as_bytes().to_vec()),
+ redis::Value::Int(value) => Ok(value.to_string().into_bytes()),
+ _ => Err(Error::InvalidEntry),
+ }
+}
+
+#[cfg(test)]
+mod tests {
+ use std::sync::{Arc, Mutex};
+
+ use litellm_cache::BaseCache;
+ use litellm_cache_response::ResponseCacheCodec;
+ use redis_test::MockRedisConnection;
+ use rstest::rstest;
+ use serde_json::{Value, json};
+
+ use super::{
+ Embedder, ValkeySemanticCache, ValkeySemanticConfig, index_dimension_from_info,
+ prompt_from_context, scope_tag,
+ };
+
+ #[derive(Clone)]
+ struct FixedEmbedder {
+ vector: Vec,
+ calls: EmbedderCalls,
+ }
+
+ type EmbedderCalls = Arc)>>>;
+
+ impl Embedder for FixedEmbedder {
+ fn embed(&self, prompt: &str, metadata: Option<&Value>) -> Result, super::Error> {
+ self.calls
+ .lock()
+ .unwrap()
+ .push((prompt.into(), metadata.cloned()));
+ Ok(self.vector.clone())
+ }
+
+ async fn async_embed(
+ &self,
+ prompt: &str,
+ metadata: Option<&Value>,
+ ) -> Result, super::Error> {
+ self.embed(prompt, metadata)
+ }
+ }
+
+ fn context(
+ messages: Option,
+ input: Option,
+ ) -> litellm_cache::SemanticCacheContext {
+ litellm_cache::SemanticCacheContext {
+ messages,
+ input,
+ ..Default::default()
+ }
+ }
+
+ #[rstest]
+ #[case(json!([{"content": "hello"}]), None, Some("hello"))]
+ #[case(json!([{"content": [{"text": "hello"}, {"text": " world"}]}]), None, Some("hello world"))]
+ #[case(json!([{"search_results": [{"source": "s", "title": "t", "content": [{"text": "c"}], "citations": ["x"]}]}]), None, Some(r#"stc["x"]"#))]
+ #[case(Value::Array(vec![]), Some(json!(" hello ")), Some("hello"))]
+ #[case(Value::Array(vec![]), Some(json!([{"content": "first"}, {"text": "second"}])), Some("first\nsecond"))]
+ #[case(Value::Array(vec![]), Some(json!(" ")), None)]
+ fn prompt_shapes(
+ #[case] messages: Value,
+ #[case] input: Option,
+ #[case] expected: Option<&str>,
+ ) {
+ assert_eq!(
+ prompt_from_context(&context(Some(messages), input)),
+ expected.map(str::to_owned)
+ );
+ }
+
+ #[test]
+ fn scope_tags_are_lowercase_sha256() {
+ assert_eq!(
+ scope_tag("key"),
+ "2c70e12b7a0646f92279f427c7b38e7334d8e5389cff167a1dc30e73f826b683"
+ );
+ }
+
+ #[test]
+ fn existing_index_dimension_is_read_from_attributes() {
+ let info = redis::Value::Array(vec![
+ redis::Value::SimpleString("attributes".into()),
+ redis::Value::Array(vec![redis::Value::Array(vec![
+ redis::Value::SimpleString("identifier".into()),
+ redis::Value::SimpleString("embedding".into()),
+ redis::Value::Array(vec![
+ redis::Value::SimpleString("dimensions".into()),
+ redis::Value::SimpleString("2".into()),
+ ]),
+ ])]),
+ ]);
+ assert_eq!(index_dimension_from_info(&info), Some(2));
+ }
+
+ #[tokio::test]
+ async fn unsupported_connection_test_is_reported() {
+ let cache = ValkeySemanticCache::with_connection(
+ MockRedisConnection::new([]).assert_all_commands_consumed(),
+ FixedEmbedder {
+ vector: vec![1.0, 0.0],
+ calls: Arc::default(),
+ },
+ ResponseCacheCodec,
+ ValkeySemanticConfig {
+ similarity_threshold: 0.8,
+ index_name: "test".into(),
+ },
+ );
+ assert_eq!(
+ cache.test_connection().await,
+ Err(super::Error::UnsupportedOperation)
+ );
+ }
+
+ #[test]
+ fn missing_prompt_does_not_touch_redis() {
+ let cache = ValkeySemanticCache::with_connection(
+ MockRedisConnection::new([]).assert_all_commands_consumed(),
+ FixedEmbedder {
+ vector: vec![1.0, 0.0],
+ calls: Arc::default(),
+ },
+ ResponseCacheCodec,
+ ValkeySemanticConfig {
+ similarity_threshold: 0.8,
+ index_name: "test".into(),
+ },
+ );
+ assert_eq!(cache.get_cache("key", &context(None, None)).unwrap(), None);
+ assert_eq!(cache.get_ttl(&context(None, None)), None);
+ }
+}
From be2f0d081b6c7ac41090ad9300b18402f226ef5f Mon Sep 17 00:00:00 2001
From: Yuneng Jiang
Date: Mon, 21 Sep 2026 13:25:39 -0700
Subject: [PATCH 065/160] fix(proxy): report sources only on the read endpoints
main does not cover
/config/field/info and /config/list already report per-key source on main,
so this drops the branch's versions of those and keeps /alerting/settings,
/get/ui_settings and /router/settings.
Read endpoints no longer write the freshly read database row back into the
shared settings store; the reload path already keeps it current, and a GET
that mutates global state leaks across callers.
Regenerates the lazy OpenAPI snapshot on Python 3.12, matching CI, and the
dashboard API types for the two new response fields.
---
litellm/proxy/_lazy_openapi_snapshot.json | 2 +-
.../router_settings_endpoints.py | 31 +++++++-------
litellm/proxy/proxy_server.py | 9 ++--
.../proxy_setting_endpoints.py | 41 +++++++++++--------
.../test_router_settings_endpoints.py | 41 ++++++++++---------
ui/litellm-dashboard/src/lib/http/schema.d.ts | 11 +++++
6 files changed, 80 insertions(+), 55 deletions(-)
diff --git a/litellm/proxy/_lazy_openapi_snapshot.json b/litellm/proxy/_lazy_openapi_snapshot.json
index 391f0042ed0..06e157498aa 100644
--- a/litellm/proxy/_lazy_openapi_snapshot.json
+++ b/litellm/proxy/_lazy_openapi_snapshot.json
@@ -19632,7 +19632,7 @@
}
}
},
- "description": "\nUnified rate-limit error.\n\nEvery rate-limit condition surfaced by litellm \u2014 whether it originated from\nan upstream LLM provider, a vendor batch endpoint, or one of litellm's own\nproxy-side limiters (parallel-requests, dynamic-rate, batch-rate, budget,\nmax-iterations, etc.) \u2014 is raised as an instance of this class.\n\nThe :attr:`category` attribute lets callers distinguish the source. See\n:class:`RateLimitErrorCategory` for the available values.\n"
+ "description": "\n Unified rate-limit error.\n\n Every rate-limit condition surfaced by litellm \u2014 whether it originated from\n an upstream LLM provider, a vendor batch endpoint, or one of litellm's own\n proxy-side limiters (parallel-requests, dynamic-rate, batch-rate, budget,\n max-iterations, etc.) \u2014 is raised as an instance of this class.\n\n The :attr:`category` attribute lets callers distinguish the source. See\n :class:`RateLimitErrorCategory` for the available values.\n "
},
"500": {
"content": {
diff --git a/litellm/proxy/management_endpoints/router_settings_endpoints.py b/litellm/proxy/management_endpoints/router_settings_endpoints.py
index 019ac68ae23..d6d74ada35a 100644
--- a/litellm/proxy/management_endpoints/router_settings_endpoints.py
+++ b/litellm/proxy/management_endpoints/router_settings_endpoints.py
@@ -8,7 +8,9 @@ GET /router/fields - Get router settings field definitions without values (for U
"""
import inspect
-from typing import Any, Final, cast, get_args
+from collections.abc import Mapping
+from types import MappingProxyType
+from typing import Any, Final, get_args
from fastapi import APIRouter, Depends
from pydantic import BaseModel, Field
@@ -127,19 +129,20 @@ async def get_router_settings(
if field.field_name in current_values:
field.field_value = current_values[field.field_name]
- field_defaults: Final[dict[str, object]] = {
- field.field_name: cast(object, field.field_default) # cast-ok: Pydantic field defaults are untyped
- for field in router_fields
- }
- source: Final[dict[str, FieldSource]] = {
- key: _router_setting_source(
- proxy_config.router_settings,
- key,
- cast(object, current_values[key]), # cast-ok: current values are stored in a typed response map
- field_defaults.get(key),
- )
- for key in current_values
- }
+ field_defaults: Final[Mapping[str, object]] = MappingProxyType(
+ {field.field_name: field.field_default for field in router_fields}
+ )
+ source: Final[Mapping[str, FieldSource]] = MappingProxyType(
+ {
+ key: _router_setting_source(
+ proxy_config.router_settings,
+ key,
+ current_values[key],
+ field_defaults.get(key),
+ )
+ for key in current_values
+ }
+ )
return RouterSettingsResponse(
fields=router_fields,
current_values=current_values,
diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py
index 4b2781e031b..24af6d7f7d3 100644
--- a/litellm/proxy/proxy_server.py
+++ b/litellm/proxy/proxy_server.py
@@ -15991,7 +15991,7 @@ def _nested_setting_source(
field_default: JsonValue,
) -> FieldSource:
db_value: Final = db_values.get(field_name)
- if db_value is not None and db_value != []:
+ if db_value is not None and not (isinstance(db_value, list) and len(db_value) == 0):
return "db"
parent_value: Final = settings.config_value(parent_key)
if isinstance(parent_value, Mapping) and field_name in parent_value:
@@ -16036,13 +16036,13 @@ async def alerting_settings(
where={"param_name": "general_settings"}
)
- db_general_settings_dict: Final[Mapping[str, JsonValue]] = (
- dict(db_general_settings.param_value)
+ db_general_settings_dict: Final[Mapping[str, JsonValue]] = MappingProxyType(
+ dict(db_general_settings.param_value) # mutable-ok: Prisma returns the JSON column as a plain dict
if db_general_settings is not None and db_general_settings.param_value is not None
else {}
)
alerting_args_value: Final = db_general_settings_dict.get("alerting_args")
- alerting_args_dict: Final[Mapping[str, JsonValue]] = (
+ alerting_args_dict: Final[Mapping[str, JsonValue]] = MappingProxyType(
alerting_args_value if isinstance(alerting_args_value, dict) else {}
)
alerting_values: Final = cast( # cast-ok: alerting is stored as a JSON list when present
@@ -16050,7 +16050,6 @@ async def alerting_settings(
)
settings: Final = proxy_config.settings
- settings.apply_db_row("general_settings", db_general_settings_dict)
allowed_args: Final = MappingProxyType(
{
diff --git a/litellm/proxy/ui_crud_endpoints/proxy_setting_endpoints.py b/litellm/proxy/ui_crud_endpoints/proxy_setting_endpoints.py
index 4505f3144ee..ed626bdb624 100644
--- a/litellm/proxy/ui_crud_endpoints/proxy_setting_endpoints.py
+++ b/litellm/proxy/ui_crud_endpoints/proxy_setting_endpoints.py
@@ -1755,18 +1755,21 @@ async def get_ui_settings():
ui_settings: Final = {k: v for k, v in parsed.items() if k in ALLOWED_UI_SETTINGS_FIELDS}
apply_runtime_general_settings_flags(ui_settings)
- proxy_config.settings.apply_db_row("ui_settings", ui_settings)
# Refresh DualCache so other code paths (e.g. /user/filter/ui) see fresh values
from litellm.proxy.proxy_server import user_api_key_cache
await user_api_key_cache.async_set_cache(key=UI_SETTINGS_CACHE_KEY, value=ui_settings, ttl=UI_SETTINGS_CACHE_TTL)
- effective_ui_settings: Final = {
- **{key: proxy_config.settings[key] for key in ALLOWED_UI_SETTINGS_FIELDS if key in proxy_config.settings},
- **ui_settings,
- }
- config: Final[dict[str, object]] = {"litellm_settings": {"ui_settings": effective_ui_settings}}
+ effective_ui_settings: Final[Mapping[str, object]] = MappingProxyType(
+ {
+ **{key: proxy_config.settings[key] for key in ALLOWED_UI_SETTINGS_FIELDS if key in proxy_config.settings},
+ **ui_settings,
+ }
+ )
+ config: Final[Mapping[str, object]] = MappingProxyType(
+ {"litellm_settings": MappingProxyType({"ui_settings": effective_ui_settings})}
+ )
settings_class: Final = _get_effective_ui_settings_class()
resolved_settings: Final = _SettingsWithSchema.model_validate(
await _get_settings_with_schema(
@@ -1775,16 +1778,22 @@ async def get_ui_settings():
config=config,
)
)
- values: Final = {
- **resolved_settings.values,
- ENABLE_PTU_COST_ATTRIBUTION_UI_SETTING: is_ptu_cost_attribution_enabled(),
- }
- source: Final[dict[str, FieldSource]] = {
- key: (
- "db" if key in ui_settings else _ui_setting_source(key, values[key], proxy_config.settings, settings_class)
- )
- for key in values
- }
+ values: Final[Mapping[str, object]] = MappingProxyType(
+ {
+ **resolved_settings.values,
+ ENABLE_PTU_COST_ATTRIBUTION_UI_SETTING: is_ptu_cost_attribution_enabled(),
+ }
+ )
+ source: Final[Mapping[str, FieldSource]] = MappingProxyType(
+ {
+ key: (
+ "db"
+ if key in ui_settings
+ else _ui_setting_source(key, values[key], proxy_config.settings, settings_class)
+ )
+ for key in values
+ }
+ )
return UISettingsResponse(
values=values,
field_schema=resolved_settings.field_schema,
diff --git a/tests/test_litellm/proxy/management_endpoints/test_router_settings_endpoints.py b/tests/test_litellm/proxy/management_endpoints/test_router_settings_endpoints.py
index 3af7de62abe..51c8679e89e 100644
--- a/tests/test_litellm/proxy/management_endpoints/test_router_settings_endpoints.py
+++ b/tests/test_litellm/proxy/management_endpoints/test_router_settings_endpoints.py
@@ -15,12 +15,24 @@ from litellm.proxy._types import LitellmUserRoles, UserAPIKeyAuth
from litellm.proxy.management_endpoints.router_settings_endpoints import (
get_router_settings,
)
+from litellm.proxy.config_resolvers import SettingsStore
from litellm.proxy.proxy_server import app
from litellm.router import Router
client = TestClient(app)
+def _stub_proxy_config(router_settings, config_router_settings):
+ class _StubProxyConfig:
+ def __init__(self):
+ self.router_settings = router_settings
+
+ async def get_config(self, config_file_path=None):
+ return {"router_settings": dict(config_router_settings)}
+
+ return _StubProxyConfig()
+
+
class TestRouterSettingsEndpoints:
"""Test suite for router settings endpoints"""
@@ -77,25 +89,18 @@ class TestRouterSettingsEndpoints:
@pytest.mark.asyncio
async def test_get_router_settings_reports_sources(self, monkeypatch):
- from litellm.proxy.config_resolvers import SettingsStore
-
store = SettingsStore("router_settings")
store.load_yaml({"routing_strategy": "simple-shuffle"})
store.apply_db_row("router_settings", {"num_retries": 3})
- monkeypatch.setattr(proxy_server.proxy_config, "router_settings", store)
- monkeypatch.setattr(proxy_server, "llm_router", None)
-
- async def fake_get_config(self, config_file_path=None):
- return {
- "router_settings": {
- "routing_strategy": "simple-shuffle",
- "num_retries": 3,
- }
- }
-
monkeypatch.setattr(
- proxy_server.ProxyConfig, "get_config", fake_get_config, raising=True
+ proxy_server,
+ "proxy_config",
+ _stub_proxy_config(
+ store,
+ {"routing_strategy": "simple-shuffle", "num_retries": 3},
+ ),
)
+ monkeypatch.setattr(proxy_server, "llm_router", None)
admin_user = UserAPIKeyAuth(
user_role=LitellmUserRoles.PROXY_ADMIN, api_key="sk-x"
@@ -132,12 +137,10 @@ class TestRouterSettingsEndpoints:
)
monkeypatch.setattr(proxy_server, "llm_router", llm_router)
-
- async def fake_get_config(self, config_file_path=None):
- return {}
-
monkeypatch.setattr(
- proxy_server.ProxyConfig, "get_config", fake_get_config, raising=True
+ proxy_server,
+ "proxy_config",
+ _stub_proxy_config(SettingsStore("router_settings"), {}),
)
admin_user = UserAPIKeyAuth(
diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts
index 6211eeaf962..ac89d676921 100644
--- a/ui/litellm-dashboard/src/lib/http/schema.d.ts
+++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts
@@ -37069,6 +37069,13 @@ export interface components {
routing_strategy_descriptions: {
[key: string]: string;
};
+ /**
+ * Source
+ * @description Source of each current router setting
+ */
+ source: {
+ [key: string]: "config" | "db" | "env" | "default" | "unset";
+ };
};
/**
* RoutingGroup
@@ -39532,6 +39539,10 @@ export interface components {
field_schema: {
[key: string]: unknown;
};
+ /** Source */
+ source: {
+ [key: string]: "config" | "db" | "env" | "default" | "unset";
+ };
/** Values */
values: {
[key: string]: unknown;
From 0a88658227f8e0d7e2d4928df7c5ee63bd83fd1d Mon Sep 17 00:00:00 2001
From: yucheng
Date: Mon, 21 Sep 2026 20:25:56 +0000
Subject: [PATCH 066/160] chore: retrigger ci after docs merge
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
From 3ba4a60d5ed5eeeb217b63ac7898746927f0db12 Mon Sep 17 00:00:00 2001
From: Yujong Lee
Date: Mon, 21 Sep 2026 20:26:28 +0000
Subject: [PATCH 067/160] feat(rust): add HashiCorp Vault secret manager crate
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
---
.github/workflows/test-rust.yml | 2 +-
litellm-rust/Cargo.lock | 19 +
litellm-rust/Cargo.toml | 1 +
.../crates/secrets-hashicorp/Cargo.toml | 23 +
.../crates/secrets-hashicorp/src/config.rs | 169 +++++++
.../crates/secrets-hashicorp/src/error.rs | 32 ++
.../crates/secrets-hashicorp/src/lib.rs | 9 +
.../secrets-hashicorp/src/secret_manager.rs | 333 ++++++++++++++
.../secrets-hashicorp/tests/secret_manager.rs | 424 ++++++++++++++++++
litellm-rust/crates/secrets/Cargo.toml | 2 +
litellm-rust/crates/secrets/README.md | 2 +
litellm-rust/crates/secrets/src/error.rs | 3 +
litellm-rust/crates/secrets/src/handler.rs | 10 +
litellm-rust/crates/secrets/src/lib.rs | 2 +
litellm-rust/crates/secrets/tests/handler.rs | 128 ++++++
.../hashicorp_secret_manager.py | 16 +-
.../hashicorp_vault_parity.json | 97 ++++
.../test_hashicorp_secret_manager.py | 50 +++
18 files changed, 1314 insertions(+), 8 deletions(-)
create mode 100644 litellm-rust/crates/secrets-hashicorp/Cargo.toml
create mode 100644 litellm-rust/crates/secrets-hashicorp/src/config.rs
create mode 100644 litellm-rust/crates/secrets-hashicorp/src/error.rs
create mode 100644 litellm-rust/crates/secrets-hashicorp/src/lib.rs
create mode 100644 litellm-rust/crates/secrets-hashicorp/src/secret_manager.rs
create mode 100644 litellm-rust/crates/secrets-hashicorp/tests/secret_manager.rs
create mode 100644 tests/test_litellm/secret_managers/hashicorp_vault_parity.json
diff --git a/.github/workflows/test-rust.yml b/.github/workflows/test-rust.yml
index 278fa7c425f..56bb9a568fc 100644
--- a/.github/workflows/test-rust.yml
+++ b/.github/workflows/test-rust.yml
@@ -130,7 +130,7 @@ jobs:
- name: Test secret manager feature combinations
run: |
cargo test -p litellm-auth-gcp --locked --no-default-features
- for features in '' aws google aws,google; do
+ for features in '' aws google hashicorp aws,google,hashicorp; do
cargo test -p litellm-secrets --locked --no-default-features --features "$features"
done
diff --git a/litellm-rust/Cargo.lock b/litellm-rust/Cargo.lock
index ed4ae4e3353..6f05bba9417 100644
--- a/litellm-rust/Cargo.lock
+++ b/litellm-rust/Cargo.lock
@@ -2698,6 +2698,7 @@ dependencies = [
"litellm-core-utils",
"litellm-secrets-aws",
"litellm-secrets-google",
+ "litellm-secrets-hashicorp",
"litellm-secrets-types",
"moka",
"reqwest 0.12.28",
@@ -2755,6 +2756,24 @@ dependencies = [
"wiremock",
]
+[[package]]
+name = "litellm-secrets-hashicorp"
+version = "0.1.0"
+dependencies = [
+ "litellm-core-utils",
+ "litellm-secrets-types",
+ "moka",
+ "reqwest 0.12.28",
+ "rstest",
+ "serde",
+ "serde_json",
+ "tempfile",
+ "thiserror 2.0.19",
+ "tokio",
+ "veil",
+ "wiremock",
+]
+
[[package]]
name = "litellm-secrets-types"
version = "0.1.0"
diff --git a/litellm-rust/Cargo.toml b/litellm-rust/Cargo.toml
index 570d0dd3568..5030a94b140 100644
--- a/litellm-rust/Cargo.toml
+++ b/litellm-rust/Cargo.toml
@@ -22,6 +22,7 @@ litellm-secrets = { path = "crates/secrets" }
litellm-secrets-types = { path = "crates/secrets-types" }
litellm-secrets-aws = { path = "crates/secrets-aws" }
litellm-secrets-google = { path = "crates/secrets-google" }
+litellm-secrets-hashicorp = { path = "crates/secrets-hashicorp" }
litellm-http = { path = "crates/http" }
litellm-llms = { path = "crates/llms" }
litellm-types = { path = "crates/types" }
diff --git a/litellm-rust/crates/secrets-hashicorp/Cargo.toml b/litellm-rust/crates/secrets-hashicorp/Cargo.toml
new file mode 100644
index 00000000000..0656980a033
--- /dev/null
+++ b/litellm-rust/crates/secrets-hashicorp/Cargo.toml
@@ -0,0 +1,23 @@
+[package]
+name = "litellm-secrets-hashicorp"
+version = "0.1.0"
+edition.workspace = true
+license.workspace = true
+repository.workspace = true
+
+[dependencies]
+litellm-core-utils.workspace = true
+litellm-secrets-types.workspace = true
+moka.workspace = true
+reqwest.workspace = true
+serde.workspace = true
+serde_json.workspace = true
+thiserror.workspace = true
+tokio.workspace = true
+veil.workspace = true
+
+[dev-dependencies]
+rstest.workspace = true
+tempfile = "3"
+tokio.workspace = true
+wiremock = "0.6.5"
diff --git a/litellm-rust/crates/secrets-hashicorp/src/config.rs b/litellm-rust/crates/secrets-hashicorp/src/config.rs
new file mode 100644
index 00000000000..f9491f71afb
--- /dev/null
+++ b/litellm-rust/crates/secrets-hashicorp/src/config.rs
@@ -0,0 +1,169 @@
+use std::{path::PathBuf, time::Duration};
+
+use litellm_core_utils::settings::Lookup;
+use litellm_secrets_types::KeyManagementSettings;
+use litellm_secrets_types::SecretValue;
+
+use crate::Error;
+
+const DEFAULT_ADDRESS: &str = "http://127.0.0.1:8200";
+const DEFAULT_MOUNT: &str = "secret";
+const DEFAULT_APPROLE_MOUNT_PATH: &str = "approle";
+const DEFAULT_REFRESH_INTERVAL: Duration = Duration::from_secs(86400);
+const HCP_VAULT_ADDR: &str = "HCP_VAULT_ADDR";
+const HCP_VAULT_TOKEN: &str = "HCP_VAULT_TOKEN";
+const HCP_VAULT_NAMESPACE: &str = "HCP_VAULT_NAMESPACE";
+const HCP_VAULT_LOGIN_NAMESPACE: &str = "HCP_VAULT_LOGIN_NAMESPACE";
+const HCP_VAULT_SECRET_NAMESPACE: &str = "HCP_VAULT_SECRET_NAMESPACE";
+const HCP_VAULT_MOUNT_NAME: &str = "HCP_VAULT_MOUNT_NAME";
+const HCP_VAULT_PATH_PREFIX: &str = "HCP_VAULT_PATH_PREFIX";
+const HCP_VAULT_APPROLE_ROLE_ID: &str = "HCP_VAULT_APPROLE_ROLE_ID";
+const HCP_VAULT_APPROLE_SECRET_ID: &str = "HCP_VAULT_APPROLE_SECRET_ID";
+const HCP_VAULT_APPROLE_MOUNT_PATH: &str = "HCP_VAULT_APPROLE_MOUNT_PATH";
+const HCP_VAULT_CLIENT_CERT: &str = "HCP_VAULT_CLIENT_CERT";
+const HCP_VAULT_CLIENT_KEY: &str = "HCP_VAULT_CLIENT_KEY";
+const HCP_VAULT_CERT_ROLE: &str = "HCP_VAULT_CERT_ROLE";
+const HCP_VAULT_REFRESH_INTERVAL: &str = "HCP_VAULT_REFRESH_INTERVAL";
+const SECRET_MANAGER_REFRESH_INTERVAL: &str = "SECRET_MANAGER_REFRESH_INTERVAL";
+
+#[derive(Clone, Debug)]
+pub struct AppRoleAuth {
+ pub role_id: String,
+ pub secret_id: SecretValue,
+ pub mount_path: String,
+}
+
+#[derive(Clone, Debug)]
+pub struct TlsCertAuth {
+ pub cert_path: PathBuf,
+ pub key_path: PathBuf,
+ pub role: Option,
+}
+
+#[derive(Clone, Debug)]
+pub struct HashicorpVaultConfig {
+ pub address: String,
+ pub token: Option,
+ pub namespace: Option,
+ pub login_namespace: Option,
+ pub secret_namespace: Option,
+ pub mount: String,
+ pub path_prefix: Option,
+ pub approle: Option,
+ pub tls_cert: Option,
+ pub refresh_interval: Duration,
+}
+
+impl HashicorpVaultConfig {
+ pub fn from_environment(environment: &dyn Lookup) -> Result {
+ let address: String = environment
+ .get(HCP_VAULT_ADDR)
+ .and_then(|value| nonempty(value.trim()))
+ .map(|value| value.trim_end_matches('/').to_owned())
+ .filter(|value| !value.is_empty())
+ .unwrap_or_else(|| DEFAULT_ADDRESS.to_owned());
+ let token: Option = environment
+ .get(HCP_VAULT_TOKEN)
+ .and_then(nonempty)
+ .map(SecretValue::new);
+ let namespace: Option = path_component(environment.get(HCP_VAULT_NAMESPACE));
+ let login_namespace: Option =
+ path_component(environment.get(HCP_VAULT_LOGIN_NAMESPACE));
+ let secret_namespace: Option =
+ path_component(environment.get(HCP_VAULT_SECRET_NAMESPACE));
+ let mount: String = path_component(environment.get(HCP_VAULT_MOUNT_NAME))
+ .unwrap_or_else(|| DEFAULT_MOUNT.to_owned());
+ let path_prefix: Option = path_component(environment.get(HCP_VAULT_PATH_PREFIX));
+ let approle: Option = match (
+ environment
+ .get(HCP_VAULT_APPROLE_ROLE_ID)
+ .and_then(nonempty),
+ environment
+ .get(HCP_VAULT_APPROLE_SECRET_ID)
+ .and_then(nonempty)
+ .map(SecretValue::new),
+ ) {
+ (Some(role_id), Some(secret_id)) => Some(AppRoleAuth {
+ role_id,
+ secret_id,
+ mount_path: path_component(environment.get(HCP_VAULT_APPROLE_MOUNT_PATH))
+ .unwrap_or_else(|| DEFAULT_APPROLE_MOUNT_PATH.to_owned()),
+ }),
+ _ => None,
+ };
+ let tls_cert: Option = match (
+ environment.get(HCP_VAULT_CLIENT_CERT).and_then(nonempty),
+ environment.get(HCP_VAULT_CLIENT_KEY).and_then(nonempty),
+ ) {
+ (Some(cert_path), Some(key_path)) => Some(TlsCertAuth {
+ cert_path: PathBuf::from(cert_path),
+ key_path: PathBuf::from(key_path),
+ role: environment.get(HCP_VAULT_CERT_ROLE).and_then(nonempty),
+ }),
+ _ => None,
+ };
+ let refresh_interval: Duration = refresh_interval(environment)?;
+ Ok(Self {
+ address,
+ token,
+ namespace,
+ login_namespace,
+ secret_namespace,
+ mount,
+ path_prefix,
+ approle,
+ tls_cert,
+ refresh_interval,
+ })
+ }
+
+ pub fn from_settings(
+ _settings: &KeyManagementSettings,
+ environment: &dyn Lookup,
+ ) -> Result {
+ Self::from_environment(environment)
+ }
+
+ pub fn login_namespace(&self) -> Option<&str> {
+ self.login_namespace
+ .as_deref()
+ .or(self.namespace.as_deref())
+ }
+
+ pub fn secret_namespace(&self) -> Option<&str> {
+ self.secret_namespace
+ .as_deref()
+ .or(self.namespace.as_deref())
+ }
+}
+
+fn nonempty(value: impl AsRef