From 28c1e431968d61ea7caf1d82e351aca765ec83a1 Mon Sep 17 00:00:00 2001
From: Yuneng Jiang
Date: Fri, 14 Aug 2026 00:01:16 -0700
Subject: [PATCH 01/24] feat(ui): standardize the Teams page header
---
.../_components/AccessGroupsPage.tsx | 4 +-
.../budgets/_components/budget_panel.tsx | 4 +-
.../projects/_components/ProjectsPage.tsx | 4 +-
.../src/components/Teams.test.tsx | 37 ++++++---
ui/litellm-dashboard/src/components/Teams.tsx | 47 +++++------
.../VirtualKeysPage/VirtualKeysTable.tsx | 4 +-
.../shared/LegacyPageHeader.test.tsx | 33 ++++++++
.../components/shared/LegacyPageHeader.tsx | 25 ++++++
.../src/components/shared/PageHeader.test.tsx | 80 +++++++++++++++----
.../src/components/shared/PageHeader.tsx | 63 +++++++++++----
10 files changed, 224 insertions(+), 77 deletions(-)
create mode 100644 ui/litellm-dashboard/src/components/shared/LegacyPageHeader.test.tsx
create mode 100644 ui/litellm-dashboard/src/components/shared/LegacyPageHeader.tsx
diff --git a/ui/litellm-dashboard/src/app/(dashboard)/access-groups/_components/AccessGroupsPage.tsx b/ui/litellm-dashboard/src/app/(dashboard)/access-groups/_components/AccessGroupsPage.tsx
index f37acb3d85a..aeff249fdd3 100644
--- a/ui/litellm-dashboard/src/app/(dashboard)/access-groups/_components/AccessGroupsPage.tsx
+++ b/ui/litellm-dashboard/src/app/(dashboard)/access-groups/_components/AccessGroupsPage.tsx
@@ -3,7 +3,7 @@ import { useDeleteAccessGroup } from "@/app/(dashboard)/hooks/accessGroups/useDe
import { Plus, SearchIcon, X } from "lucide-react";
import { useMemo, useState } from "react";
import DeleteResourceModal from "@/components/common_components/DeleteResourceModal";
-import { PageHeader } from "@/components/shared/PageHeader";
+import { LegacyPageHeader } from "@/components/shared/LegacyPageHeader";
import { Button } from "@/components/ui/button";
import { InputGroup, InputGroupAddon, InputGroupButton, InputGroupInput } from "@/components/ui/input-group";
import { AccessGroupDetail } from "./AccessGroupsDetailsPage";
@@ -61,7 +61,7 @@ export function AccessGroupsPage() {
return (
-
= ({ accessToken }) => {
return (
-
}
title="Budgets"
subtitle="Spend, TPM and RPM limits you can assign to customers."
diff --git a/ui/litellm-dashboard/src/app/(dashboard)/projects/_components/ProjectsPage.tsx b/ui/litellm-dashboard/src/app/(dashboard)/projects/_components/ProjectsPage.tsx
index 91ad9f847f5..dc18a05edca 100644
--- a/ui/litellm-dashboard/src/app/(dashboard)/projects/_components/ProjectsPage.tsx
+++ b/ui/litellm-dashboard/src/app/(dashboard)/projects/_components/ProjectsPage.tsx
@@ -3,7 +3,7 @@ import { useTeams } from "@/app/(dashboard)/hooks/teams/useTeams";
import { Plus, SearchIcon, X } from "lucide-react";
import { parseAsString, useQueryState } from "nuqs";
import { useMemo, useState } from "react";
-import { PageHeader } from "@/components/shared/PageHeader";
+import { LegacyPageHeader } from "@/components/shared/LegacyPageHeader";
import { Button } from "@/components/ui/button";
import { InputGroup, InputGroupAddon, InputGroupButton, InputGroupInput } from "@/components/ui/input-group";
import { CreateProjectModal } from "./ProjectModals/CreateProjectModal";
@@ -56,7 +56,7 @@ export function ProjectsPage() {
return (
-
{
expect(onUrlUpdate.mock.calls.at(-1)![0].searchParams.has("team")).toBe(false);
await waitFor(() => expect(screen.queryByTestId("team-info-view")).not.toBeInTheDocument());
});
+
+ it("should preserve the legacy inset for the team detail view", async () => {
+ renderWithQueryClient(, {
+ searchParams: "?team=team-from-url",
+ });
+
+ await waitFor(() => expect(mockTeamInfoView).toHaveBeenCalled());
+ expect(screen.getByRole("main")).toHaveClass("px-12", "py-6");
+ });
});
describe("Teams - Create Team CTA is grouped with the tabs on the left", () => {
@@ -521,22 +530,28 @@ describe("Teams - Create Team CTA is grouped with the tabs on the left", () => {
mockUseOrganizations.mockReturnValue({ data: [] });
});
- it("renders the Create Team button inside the tab bar, ahead of the tabs", () => {
- const { container } = renderWithQueryClient();
+ it("should render the Create Team button inside the tab bar, ahead of the tabs", () => {
+ renderWithQueryClient();
- const createButton = screen.getByTestId("create-team-button");
- const tabNav = container.querySelector(".ant-tabs-nav");
+ const tabNav = screen.getByRole("tablist");
+ const createButton = within(tabNav).getByTestId("create-team-button");
+ const firstTab = within(tabNav).getByRole("tab", { name: "Your Teams" });
+ const tabs = tabNav.closest(".ant-tabs");
- // The CTA lives in the tab bar's left slot, not the standalone page header.
- expect(tabNav).not.toBeNull();
- expect(tabNav!.contains(createButton)).toBe(true);
-
- // It reads as the left end of the cluster: it precedes the first tab in DOM order.
- const firstTab = screen.getByRole("tab", { name: "Your Teams" });
+ expect(screen.getByRole("main")).toHaveClass("p-8");
+ expect(within(tabNav).getByRole("separator")).toBeInTheDocument();
expect(createButton.compareDocumentPosition(firstTab) & Node.DOCUMENT_POSITION_FOLLOWING).toBeTruthy();
+ expect(tabs).toHaveClass(
+ "[&>.ant-tabs-nav]:!mb-6",
+ "[&>.ant-tabs-nav]:before:!border-b-0",
+ "[&_.ant-tabs-ink-bar]:!h-0.5",
+ "[&_.ant-tabs-tab]:!py-[7px]",
+ "[&_.ant-tabs-tab+_.ant-tabs-tab]:!ml-[22px]",
+ "[&_.ant-tabs-tab-active]:font-semibold",
+ );
});
- it("omits the Create Team CTA for a role that cannot manage teams", () => {
+ it("should omit the Create Team CTA for a role that cannot manage teams", () => {
renderWithQueryClient();
expect(screen.queryByTestId("create-team-button")).not.toBeInTheDocument();
});
diff --git a/ui/litellm-dashboard/src/components/Teams.tsx b/ui/litellm-dashboard/src/components/Teams.tsx
index becbe0e2b48..5b79b067412 100644
--- a/ui/litellm-dashboard/src/components/Teams.tsx
+++ b/ui/litellm-dashboard/src/components/Teams.tsx
@@ -6,7 +6,7 @@ import TeamSSOSettings from "@/components/TeamSSOSettings";
import { isProxyAdminRole } from "@/utils/roles";
import { InfoCircleOutlined } from "@ant-design/icons";
import { Accordion, AccordionBody, AccordionHeader, TextInput } from "@tremor/react";
-import { Button, Form, Input, Layout, Modal, Select, Switch, Tabs, theme, Tooltip, Typography } from "antd";
+import { Button, Form, Input, Layout, Modal, Select, Switch, Tabs, Tooltip, Typography } from "antd";
import { Plus, Users } from "lucide-react";
import React, { useEffect, useState } from "react";
import { useQuery, useQueryClient } from "@tanstack/react-query";
@@ -403,7 +403,6 @@ const Teams: React.FC = ({ accessToken, userID, userRole, premiumUser
return false;
};
- const { token } = theme.useToken();
const { Text } = Typography;
const { Content } = Layout;
@@ -474,7 +473,7 @@ const Teams: React.FC = ({ accessToken, userID, userRole, premiumUser
];
return (
-
+
{selectedTeamId ? (
= ({ accessToken, userID, userRole, premiumUser
premiumUser={premiumUser}
/>
) : (
- <>
-
-
}
- title="Teams"
- subtitle="Manage teams, members, and their access to models and budgets"
+
}
+ title="Teams"
+ subtitle="Manage teams, members, and their access to models and budgets"
+ primaryAction={
+ canCreateOrManageTeams(userRole, userID, organizations) ? (
+
setIsTeamModalVisible(true)} data-testid="create-team-button">
+
+ Create Team
+
+ ) : undefined
+ }
+ tabs={({ leadingControls }) => (
+
-
-
-
- setIsTeamModalVisible(true)} data-testid="create-team-button">
-
- Create Team
-
-
-
- ) : undefined,
- }}
- />
- >
+ )}
+ />
)}
{canCreateOrManageTeams(userRole, userID, organizations) && (
diff --git a/ui/litellm-dashboard/src/components/VirtualKeysPage/VirtualKeysTable.tsx b/ui/litellm-dashboard/src/components/VirtualKeysPage/VirtualKeysTable.tsx
index fa0360c0dda..b278b4f675d 100644
--- a/ui/litellm-dashboard/src/components/VirtualKeysPage/VirtualKeysTable.tsx
+++ b/ui/litellm-dashboard/src/components/VirtualKeysPage/VirtualKeysTable.tsx
@@ -12,7 +12,7 @@ import {
DataTableToolbar,
} from "@/components/shared/DataTable";
import { SearchSelect } from "@/components/shared/SearchSelect";
-import { PageHeader } from "@/components/shared/PageHeader";
+import { LegacyPageHeader } from "@/components/shared/LegacyPageHeader";
import { Input } from "@/components/ui/input";
import { useDebouncedValue } from "@tanstack/react-pacer/debouncer";
import { ColumnFiltersState, OnChangeFn, PaginationState, SortingState } from "@tanstack/react-table";
@@ -172,7 +172,7 @@ export function VirtualKeysTable({ headerActions }: VirtualKeysTableProps) {
return (
-
}
title="Virtual Keys"
subtitle="Every key that authenticates requests to the gateway."
diff --git a/ui/litellm-dashboard/src/components/shared/LegacyPageHeader.test.tsx b/ui/litellm-dashboard/src/components/shared/LegacyPageHeader.test.tsx
new file mode 100644
index 00000000000..a0081c1c7f1
--- /dev/null
+++ b/ui/litellm-dashboard/src/components/shared/LegacyPageHeader.test.tsx
@@ -0,0 +1,33 @@
+import { renderWithProviders, screen } from "@/../tests/test-utils";
+import { describe, expect, it } from "vitest";
+
+import { LegacyPageHeader } from "./LegacyPageHeader";
+
+describe("LegacyPageHeader", () => {
+ it("should render the title as a heading", () => {
+ renderWithProviders(
);
+
+ expect(screen.getByRole("heading", { name: "Virtual Keys" })).toBeInTheDocument();
+ });
+
+ it("should render the optional identity and actions", () => {
+ renderWithProviders(
+
Key icon}
+ actions={}
+ />,
+ );
+
+ expect(screen.getByText("Every key that authenticates requests")).toBeInTheDocument();
+ expect(screen.getByText("Key icon")).toBeInTheDocument();
+ expect(screen.getByRole("button", { name: "Create New Key" })).toBeInTheDocument();
+ });
+
+ it("should omit optional actions when none are provided", () => {
+ renderWithProviders();
+
+ expect(screen.queryByRole("button")).not.toBeInTheDocument();
+ });
+});
diff --git a/ui/litellm-dashboard/src/components/shared/LegacyPageHeader.tsx b/ui/litellm-dashboard/src/components/shared/LegacyPageHeader.tsx
new file mode 100644
index 00000000000..43979ad00b4
--- /dev/null
+++ b/ui/litellm-dashboard/src/components/shared/LegacyPageHeader.tsx
@@ -0,0 +1,25 @@
+"use client";
+
+import * as React from "react";
+
+interface LegacyPageHeaderProps {
+ title: React.ReactNode;
+ subtitle?: React.ReactNode;
+ icon?: React.ReactNode;
+ actions?: React.ReactNode;
+}
+
+export function LegacyPageHeader({ title, subtitle, icon, actions }: LegacyPageHeaderProps) {
+ return (
+
+
+ {icon != null &&
{icon}}
+
+
{title}
+ {subtitle != null &&
{subtitle}
}
+
+
+ {actions != null &&
{actions}
}
+
+ );
+}
diff --git a/ui/litellm-dashboard/src/components/shared/PageHeader.test.tsx b/ui/litellm-dashboard/src/components/shared/PageHeader.test.tsx
index f7a313271da..3741542abad 100644
--- a/ui/litellm-dashboard/src/components/shared/PageHeader.test.tsx
+++ b/ui/litellm-dashboard/src/components/shared/PageHeader.test.tsx
@@ -1,31 +1,77 @@
-import { render, screen } from "@testing-library/react";
+import { renderWithProviders, screen, within } from "@/../tests/test-utils";
import { describe, expect, it } from "vitest";
import { PageHeader } from "./PageHeader";
+const identity = {
+ icon: Teams icon,
+ title: "Teams",
+ subtitle: "Manage teams, members, and their access to models and budgets",
+};
+
describe("PageHeader", () => {
- it("renders the title as a heading", () => {
- render();
- expect(screen.getByRole("heading", { name: "Virtual Keys" })).toBeInTheDocument();
+ it("should render the page identity", () => {
+ renderWithProviders();
+
+ expect(screen.getByRole("heading", { name: "Teams" })).toBeInTheDocument();
+ expect(screen.getByText("Teams icon").parentElement).toHaveAttribute("aria-hidden", "true");
+ expect(screen.getByText(identity.subtitle)).toBeInTheDocument();
});
- it("renders the subtitle, icon, and actions when provided", () => {
- render(
+ it("should apply the standard title and subtext typography", () => {
+ renderWithProviders();
+
+ const icon = screen.getByText("Teams icon").parentElement;
+ expect(screen.getByRole("heading", { name: "Teams" })).toHaveClass("text-2xl", "font-semibold", "tracking-tight");
+ expect(screen.getByText(identity.subtitle)).toHaveClass("mt-1.5", "text-sm", "text-muted-foreground");
+ expect(icon).toHaveClass("size-5", "[&_svg]:size-5", "[&_svg]:stroke-[1.75]");
+ expect(icon?.parentElement).toHaveClass("gap-2.5");
+ });
+
+ it("should render the primary action, divider, tabs, and utilities in the standard control row", () => {
+ renderWithProviders(
}
- actions={}
+ {...identity}
+ primaryAction={}
+ tabs={
+
+
+
+ }
+ utilities={}
/>,
);
- expect(screen.getByText("Every key that authenticates requests")).toBeInTheDocument();
- expect(screen.getByTestId("icon")).toBeInTheDocument();
- expect(screen.getByRole("button", { name: "Create New Key" })).toBeInTheDocument();
+
+ const controls = screen.getByRole("group", { name: "Page controls" });
+ expect(controls).toHaveClass("mt-5", "h-9");
+ expect(within(controls).getByRole("separator")).toHaveClass("mx-4", "h-6");
+ expect(controls).toHaveTextContent("Create TeamYour TeamsRefresh");
});
- it("omits the optional slots when not provided", () => {
- render();
- expect(screen.queryByRole("button")).not.toBeInTheDocument();
- expect(document.querySelector("p")).toBeNull();
+ it("should omit the divider when tabs are absent", () => {
+ renderWithProviders(Create Team} />);
+
+ expect(screen.queryByRole("separator")).not.toBeInTheDocument();
+ });
+
+ it("should provide standard controls to an embedded tab shell", () => {
+ renderWithProviders(
+ Create Team}
+ tabs={({ leadingControls, utilities }) => (
+
+ {leadingControls}
+
+ {utilities}
+
+ )}
+ utilities={}
+ />,
+ );
+
+ const tabs = screen.getByRole("tablist");
+ expect(within(tabs).getByRole("separator")).toBeInTheDocument();
+ expect(tabs).toHaveTextContent("Create TeamYour TeamsRefresh");
});
});
diff --git a/ui/litellm-dashboard/src/components/shared/PageHeader.tsx b/ui/litellm-dashboard/src/components/shared/PageHeader.tsx
index e314e8e8bc2..81092821efc 100644
--- a/ui/litellm-dashboard/src/components/shared/PageHeader.tsx
+++ b/ui/litellm-dashboard/src/components/shared/PageHeader.tsx
@@ -2,24 +2,57 @@
import * as React from "react";
-interface PageHeaderProps {
- title: React.ReactNode;
- subtitle?: React.ReactNode;
- icon?: React.ReactNode;
- actions?: React.ReactNode;
+import { ToolbarSeparator } from "./ToolbarSeparator";
+
+interface EmbeddedTabsSlots {
+ leadingControls: React.ReactNode;
+ utilities: React.ReactNode;
}
-export function PageHeader({ title, subtitle, icon, actions }: PageHeaderProps) {
- return (
-
-
- {icon != null &&
{icon}}
-
-
{title}
- {subtitle != null &&
{subtitle}
}
-
+interface PageHeaderProps {
+ title: React.ReactNode;
+ subtitle: React.ReactNode;
+ icon: React.ReactNode;
+ primaryAction?: React.ReactNode;
+ tabs?: React.ReactNode | ((slots: EmbeddedTabsSlots) => React.ReactNode);
+ utilities?: React.ReactNode;
+}
+
+export function PageHeader({ title, subtitle, icon, primaryAction, tabs, utilities }: PageHeaderProps) {
+ const leadingControls =
+ primaryAction == null ? null : (
+
+ {primaryAction}
+ {tabs != null && }
- {actions != null &&
{actions}
}
+ );
+ const utilityControls = utilities == null ? null :
{utilities}
;
+ const hasControlRow = primaryAction != null || tabs != null || utilities != null;
+
+ return (
+
+
+
+ {icon}
+
+
{title}
+
+
{subtitle}
+
+ {typeof tabs === "function" ? (
+
{tabs({ leadingControls, utilities: utilityControls })}
+ ) : (
+ hasControlRow && (
+
+ {leadingControls}
+ {tabs}
+ {utilityControls != null &&
{utilityControls}
}
+
+ )
+ )}
);
}
From 20a3a16c2f6ba7a7d5755cf56740d5180973ede0 Mon Sep 17 00:00:00 2001
From: mateo-berri <277851410+mateo-berri@users.noreply.github.com>
Date: Wed, 19 Aug 2026 14:25:43 -0700
Subject: [PATCH 02/24] fix(proxy): populate deployment fields on
failed-request spend logs from the standard logging payload
---
.../spend_tracking/spend_tracking_utils.py | 30 ++++++--
.../test_spend_tracking_utils.py | 77 +++++++++++++++++++
2 files changed, 102 insertions(+), 5 deletions(-)
diff --git a/litellm/proxy/spend_tracking/spend_tracking_utils.py b/litellm/proxy/spend_tracking/spend_tracking_utils.py
index 3146d8bccfb..0b56f0d8246 100644
--- a/litellm/proxy/spend_tracking/spend_tracking_utils.py
+++ b/litellm/proxy/spend_tracking/spend_tracking_utils.py
@@ -216,6 +216,15 @@ def _extract_usage_for_ocr_call(response_obj: Any, response_obj_dict: dict) -> d
return {}
+def _sl_attribution_fallback(
+ standard_logging_payload: StandardLoggingPayload | None,
+ field: Literal["model_id", "model_group", "api_base", "custom_llm_provider"],
+) -> str:
+ if standard_logging_payload is None:
+ return ""
+ return standard_logging_payload.get(field) or ""
+
+
def get_logging_payload(kwargs, response_obj, start_time, end_time) -> SpendLogsPayload:
if kwargs is None:
kwargs = {}
@@ -288,8 +297,15 @@ def get_logging_payload(kwargs, response_obj, start_time, end_time) -> SpendLogs
): # use 'tags' from standard logging payload instead
request_tags = safe_dumps(standard_logging_payload["request_tags"])
- _model_id: Final = metadata.get("model_info", {}).get("id", "")
- _model_group: Final = metadata.get("model_group", "")
+ _model_id: Final = metadata.get("model_info", {}).get("id", "") or _sl_attribution_fallback(
+ standard_logging_payload, "model_id"
+ )
+ _model_group: Final = metadata.get("model_group", "") or _sl_attribution_fallback(
+ standard_logging_payload, "model_group"
+ )
+ _api_base: Final = litellm_params.get("api_base", "") or _sl_attribution_fallback(
+ standard_logging_payload, "api_base"
+ )
# Extract overhead from hidden_params if available
litellm_overhead_time_ms = None
@@ -389,7 +405,11 @@ def get_logging_payload(kwargs, response_obj, start_time, end_time) -> SpendLogs
# Extract agent_id for A2A requests (set directly on model_call_details)
agent_id: Final[str | None] = kwargs.get("agent_id") or metadata.get("agent_id")
- custom_llm_provider: Final = kwargs.get("custom_llm_provider")
+ custom_llm_provider: Final = (
+ kwargs.get("custom_llm_provider")
+ or _sl_attribution_fallback(standard_logging_payload, "custom_llm_provider")
+ or None
+ )
raw_model: Final = cast(str, kwargs.get("model") or "")
model_name: Final = reconstruct_model_name(raw_model, custom_llm_provider, metadata or {})
@@ -414,13 +434,13 @@ def get_logging_payload(kwargs, response_obj, start_time, end_time) -> SpendLogs
completion_tokens=usage.get("completion_tokens", standard_logging_completion_tokens),
request_tags=request_tags,
end_user=end_user_id or "",
- api_base=litellm_params.get("api_base", ""),
+ api_base=_api_base,
model_group=_model_group,
model_id=_model_id,
mcp_namespaced_tool_name=mcp_namespaced_tool_name,
agent_id=agent_id,
requester_ip_address=clean_metadata.get("requester_ip_address", None),
- custom_llm_provider=kwargs.get("custom_llm_provider", ""),
+ custom_llm_provider=custom_llm_provider or "",
messages=_get_messages_for_spend_logs_payload(
standard_logging_payload=standard_logging_payload, metadata=metadata
),
diff --git a/tests/test_litellm/proxy/spend_tracking/test_spend_tracking_utils.py b/tests/test_litellm/proxy/spend_tracking/test_spend_tracking_utils.py
index e5add059260..9710dc44e99 100644
--- a/tests/test_litellm/proxy/spend_tracking/test_spend_tracking_utils.py
+++ b/tests/test_litellm/proxy/spend_tracking/test_spend_tracking_utils.py
@@ -3164,3 +3164,80 @@ def test_batch_cost_row_id_is_stable_across_repeated_accounting():
]
assert ids[0] == ids[1] == "batch_same_batch_cost"
+
+
+def _make_failed_request_standard_logging_payload() -> StandardLoggingPayload:
+ base: Final = _make_standard_logging_payload_with_usage_object(usage_object={})
+ return cast(
+ StandardLoggingPayload,
+ {
+ **base,
+ "status": "failure",
+ "call_type": "aresponses",
+ "model_id": "mid-123",
+ "model_group": "group-x",
+ "api_base": "https://api.openai.com/v1/responses",
+ "custom_llm_provider": "openai",
+ },
+ )
+
+
+def test_get_logging_payload_failed_request_falls_back_to_standard_logging_payload():
+ """Failed-request kwargs from the proxy failure hook carry no deployment info
+ (LIT-5795), so the attribution columns must come from the failure-time
+ standard_logging_object."""
+ payload = get_logging_payload(
+ kwargs={
+ "model": "group-x",
+ "litellm_params": {"metadata": {"user_api_key": "test-key", "status": "failure"}},
+ "standard_logging_object": _make_failed_request_standard_logging_payload(),
+ },
+ response_obj={},
+ start_time=datetime.datetime.now(timezone.utc),
+ end_time=datetime.datetime.now(timezone.utc),
+ )
+ assert payload["model_id"] == "mid-123"
+ assert payload["model_group"] == "group-x"
+ assert payload["api_base"] == "https://api.openai.com/v1/responses"
+ assert payload["custom_llm_provider"] == "openai"
+
+
+def test_get_logging_payload_request_kwargs_win_over_standard_logging_payload():
+ payload = get_logging_payload(
+ kwargs={
+ "model": "group-y",
+ "custom_llm_provider": "anthropic",
+ "litellm_params": {
+ "api_base": "https://kwargs.example.com",
+ "metadata": {
+ "user_api_key": "test-key",
+ "model_group": "kwargs-group",
+ "model_info": {"id": "kwargs-mid"},
+ },
+ },
+ "standard_logging_object": _make_failed_request_standard_logging_payload(),
+ },
+ response_obj={},
+ start_time=datetime.datetime.now(timezone.utc),
+ end_time=datetime.datetime.now(timezone.utc),
+ )
+ assert payload["model_id"] == "kwargs-mid"
+ assert payload["model_group"] == "kwargs-group"
+ assert payload["api_base"] == "https://kwargs.example.com"
+ assert payload["custom_llm_provider"] == "anthropic"
+
+
+def test_get_logging_payload_failed_request_without_standard_logging_payload_leaves_fields_empty():
+ payload = get_logging_payload(
+ kwargs={
+ "model": "group-x",
+ "litellm_params": {"metadata": {"user_api_key": "test-key", "status": "failure"}},
+ },
+ response_obj={},
+ start_time=datetime.datetime.now(timezone.utc),
+ end_time=datetime.datetime.now(timezone.utc),
+ )
+ assert payload["model_id"] == ""
+ assert payload["model_group"] == ""
+ assert payload["api_base"] == ""
+ assert payload["custom_llm_provider"] == ""
From b39a339b7d9e67efcd8fddde8c289c59410d9160 Mon Sep 17 00:00:00 2001
From: mateo-berri <277851410+mateo-berri@users.noreply.github.com>
Date: Wed, 19 Aug 2026 15:21:06 -0700
Subject: [PATCH 03/24] fix(vertex_ai): apply regional endpoint uplift to cost
tracking
---
basedpyright-code-budget.json | 8 +-
ci_cd/generate_model_prices_schema.py | 5 +
litellm/cost_calculator.py | 15 ++
litellm/litellm_core_utils/litellm_logging.py | 28 +++-
.../litellm_core_utils/llm_cost_calc/utils.py | 48 +++++++
litellm/llms/vertex_ai/cost_calculator.py | 19 ++-
litellm/llms/vertex_ai/vertex_llm_base.py | 17 ++-
...odel_prices_and_context_window_backup.json | 25 ++++
litellm/proxy/spend_tracking/savings.py | 8 +-
litellm/types/utils.py | 17 ++-
litellm/utils.py | 1 +
model_prices_and_context_window.json | 25 ++++
model_prices_and_context_window.schema.json | 5 +
.../llm_cost_calc/test_llm_cost_calc_utils.py | 132 ++++++++++++++++++
.../test_litellm_logging.py | 77 ++++++++++
.../proxy/spend_tracking/test_savings.py | 38 +++++
tests/test_litellm/test_cost_calculator.py | 69 +++++++++
type-discipline-budget.json | 2 +-
ui/litellm-dashboard/src/lib/http/schema.d.ts | 4 +
19 files changed, 516 insertions(+), 27 deletions(-)
diff --git a/basedpyright-code-budget.json b/basedpyright-code-budget.json
index 1ce71c5bd2c..32c77146b42 100644
--- a/basedpyright-code-budget.json
+++ b/basedpyright-code-budget.json
@@ -57,7 +57,7 @@
"limit": 5663
},
"reportMissingTypeArgument": {
- "limit": 15557
+ "limit": 15556
},
"reportMissingTypeStubs": {
"limit": 40
@@ -105,13 +105,13 @@
"limit": 109
},
"reportUnknownMemberType": {
- "limit": 39043
+ "limit": 39030
},
"reportUnknownParameterType": {
- "limit": 19887
+ "limit": 19886
},
"reportUnknownVariableType": {
- "limit": 30574
+ "limit": 30573
},
"reportUnnecessaryCast": {
"limit": 117
diff --git a/ci_cd/generate_model_prices_schema.py b/ci_cd/generate_model_prices_schema.py
index 153fbc0fdc2..252e3675329 100644
--- a/ci_cd/generate_model_prices_schema.py
+++ b/ci_cd/generate_model_prices_schema.py
@@ -145,6 +145,11 @@ NUMBER_KEYS: dict[str, JsonSchema] = {
"minimum": 1,
"description": "Multiplier applied to all token costs for US data residency (e.g. 1.10 = +10%).",
},
+ "regional_endpoint_uplift_multiplier": {
+ "type": "number",
+ "minimum": 1,
+ "description": "Multiplier applied to all token costs when served from a non-global Vertex AI endpoint (e.g. 1.10 = +10%).",
+ },
}
COST_DESCRIPTIONS: dict[str, str] = {
diff --git a/litellm/cost_calculator.py b/litellm/cost_calculator.py
index 7d7380665d3..8f7cd09d364 100644
--- a/litellm/cost_calculator.py
+++ b/litellm/cost_calculator.py
@@ -327,6 +327,8 @@ def cost_per_token(
service_tier: str | None = None, # for OpenAI service tier pricing
### DATA RESIDENCY ###
data_residency: str | None = None, # for OpenAI regional-processing uplift (e.g. "eu", "us")
+ ### VERTEX LOCATION ###
+ vertex_location: str | None = None, # for Vertex AI regional-endpoint uplift (e.g. "us-east5", "global")
response: Any | None = None,
### REQUEST MODEL ###
request_model: str | None = None, # original request model for router detection
@@ -587,6 +589,7 @@ def cost_per_token(
prompt_characters=prompt_characters,
completion_characters=completion_characters,
usage=usage_block,
+ vertex_location=vertex_location,
)
elif cost_router == "cost_per_token":
return google_cost_per_token(
@@ -594,6 +597,7 @@ def cost_per_token(
custom_llm_provider=custom_llm_provider,
usage=usage_block,
service_tier=service_tier,
+ vertex_location=vertex_location,
)
elif custom_llm_provider == "anthropic":
return anthropic_cost_per_token(model=model, usage=usage_block, service_tier=service_tier)
@@ -1071,6 +1075,7 @@ def _store_cost_breakdown_in_logging_obj(
reasoning_cost: float | None = None,
service_tier: str | None = None,
data_residency: str | None = None,
+ vertex_location: str | None = None,
) -> None:
"""
Helper function to store cost breakdown in the logging object.
@@ -1090,6 +1095,7 @@ def _store_cost_breakdown_in_logging_obj(
margin_total_amount: Total margin added in USD
service_tier: Tier the costs above were priced on, already resolved
data_residency: Region uplift the costs above were priced on, already resolved
+ vertex_location: Vertex AI location the costs above were priced on, already resolved
"""
if litellm_logging_obj is None:
return
@@ -1113,6 +1119,7 @@ def _store_cost_breakdown_in_logging_obj(
reasoning_cost=reasoning_cost,
service_tier=service_tier,
data_residency=data_residency,
+ vertex_location=vertex_location,
)
except Exception as breakdown_error:
@@ -1149,6 +1156,8 @@ def completion_cost(
service_tier: str | None = None, # for OpenAI service tier pricing
### DATA RESIDENCY ###
data_residency: str | None = None, # for OpenAI regional-processing uplift (e.g. "eu", "us")
+ ### VERTEX LOCATION ###
+ vertex_location: str | None = None, # for Vertex AI regional-endpoint uplift (e.g. "us-east5", "global")
) -> float:
"""
Calculate the cost of a given completion call fot GPT-3.5-turbo, llama2, any litellm supported llm.
@@ -1577,6 +1586,7 @@ def completion_cost(
rerank_billed_units=rerank_billed_units,
service_tier=service_tier,
data_residency=data_residency,
+ vertex_location=vertex_location,
response=completion_response,
request_model=request_model_for_cost,
)
@@ -1664,6 +1674,7 @@ def completion_cost(
usage=cost_per_token_usage_object,
service_tier=service_tier,
data_residency=data_residency,
+ vertex_location=vertex_location,
)
_reasoning_cost = _token_type_breakdown.reasoning_cost
_cache_read_cost = _token_type_breakdown.cache_read_cost
@@ -1686,6 +1697,7 @@ def completion_cost(
reasoning_cost=_reasoning_cost,
service_tier=service_tier,
data_residency=data_residency,
+ vertex_location=vertex_location,
)
return _final_cost
@@ -1765,6 +1777,8 @@ def response_cost_calculator(
service_tier: str | None = None, # for OpenAI service tier pricing
### DATA RESIDENCY ###
data_residency: str | None = None, # for OpenAI regional-processing uplift (e.g. "eu", "us")
+ ### VERTEX LOCATION ###
+ vertex_location: str | None = None, # for Vertex AI regional-endpoint uplift (e.g. "us-east5", "global")
) -> float:
"""
Returns
@@ -1797,6 +1811,7 @@ def response_cost_calculator(
litellm_logging_obj=litellm_logging_obj,
service_tier=service_tier,
data_residency=data_residency,
+ vertex_location=vertex_location,
)
return response_cost
except Exception as e:
diff --git a/litellm/litellm_core_utils/litellm_logging.py b/litellm/litellm_core_utils/litellm_logging.py
index 946110abf9e..e97abcf0af6 100644
--- a/litellm/litellm_core_utils/litellm_logging.py
+++ b/litellm/litellm_core_utils/litellm_logging.py
@@ -13,7 +13,7 @@ import traceback
from collections.abc import Callable, Mapping, Sequence
from datetime import datetime as dt_object
from functools import lru_cache
-from types import TracebackType
+from types import MappingProxyType, TracebackType
from typing import TYPE_CHECKING, Any, Final, Literal, Optional, Union, cast
from httpx import Response
@@ -372,6 +372,24 @@ def _published_pricing(deployment_model: str | None) -> ModelInfo | None:
return None
+def _resolve_vertex_location_for_cost(
+ custom_llm_provider: str | None,
+ litellm_params: Mapping[str, object] | None,
+ model: str,
+) -> str | None:
+ """
+ The Vertex AI location a request was served from, resolved the same way
+ dispatch resolves it, so regional deployments price with the
+ regional-endpoint uplift. None for non-Vertex providers.
+ """
+ if custom_llm_provider is None or not custom_llm_provider.startswith("vertex_ai"):
+ return None
+ from litellm.llms.vertex_ai.vertex_llm_base import VertexBase
+
+ configured_location: Final = VertexBase.safe_get_vertex_ai_location(litellm_params or MappingProxyType({}))
+ return VertexBase.get_vertex_region(configured_location, model)
+
+
class Logging(LiteLLMLoggingBaseClass):
global \
supabaseClient, \
@@ -1432,6 +1450,7 @@ class Logging(LiteLLMLoggingBaseClass):
reasoning_cost: float | None = None,
service_tier: str | None = None,
data_residency: str | None = None,
+ vertex_location: str | None = None,
) -> None:
"""
Helper method to store cost breakdown in the logging object.
@@ -1450,6 +1469,7 @@ class Logging(LiteLLMLoggingBaseClass):
margin_total_amount: Total margin added in USD
service_tier: Tier the costs above were priced on, already resolved
data_residency: Region uplift the costs above were priced on, already resolved
+ vertex_location: Vertex AI location the costs above were priced on, already resolved
"""
self.cost_breakdown = CostBreakdown(
@@ -1459,6 +1479,7 @@ class Logging(LiteLLMLoggingBaseClass):
tool_usage_cost=cost_for_built_in_tools_cost_usd_dollar,
service_tier=service_tier,
data_residency=data_residency,
+ vertex_location=vertex_location,
)
if cache_read_cost is not None and cache_read_cost > 0:
self.cost_breakdown["cache_read_cost"] = cache_read_cost
@@ -1574,6 +1595,11 @@ class Logging(LiteLLMLoggingBaseClass):
if hasattr(self, "litellm_params") and self.litellm_params
else None
),
+ "vertex_location": _resolve_vertex_location_for_cost(
+ custom_llm_provider=self.model_call_details.get("custom_llm_provider", None),
+ litellm_params=(self.litellm_params if hasattr(self, "litellm_params") else None),
+ model=litellm_model_name or self.model,
+ ),
}
except Exception as e: # error creating kwargs for cost calculation
debug_info = StandardLoggingModelCostFailureDebugInformation(
diff --git a/litellm/litellm_core_utils/llm_cost_calc/utils.py b/litellm/litellm_core_utils/llm_cost_calc/utils.py
index f73c4942a1c..dec35d16ea0 100644
--- a/litellm/litellm_core_utils/llm_cost_calc/utils.py
+++ b/litellm/litellm_core_utils/llm_cost_calc/utils.py
@@ -757,6 +757,33 @@ def _get_regional_uplift_multiplier(model_info: ModelInfo, data_residency: str |
return 1.0
+def get_vertex_regional_endpoint_uplift(model_info: ModelInfo, vertex_location: str | None) -> float:
+ """
+ Resolve the per-model uplift multiplier for Vertex AI non-global (regional and
+ multi-region) endpoints.
+
+ Google prices every non-global endpoint at a flat premium over the global
+ endpoint (e.g. 1.10 = +10%) on all token types for the models that carry
+ regional pricing. The multiplier is stored on the model entry as
+ ``regional_endpoint_uplift_multiplier``.
+
+ Returns 1.0 (no uplift) when ``vertex_location`` is ``None`` or ``"global"``,
+ or when the model has no multiplier configured.
+ """
+ if vertex_location is None or vertex_location.lower() == "global":
+ return 1.0
+ multiplier: Final = model_info.get("regional_endpoint_uplift_multiplier")
+ if multiplier is None:
+ return 1.0
+ try:
+ return float(cast(float, multiplier))
+ except (TypeError, ValueError):
+ verbose_logger.exception(
+ "Invalid regional_endpoint_uplift_multiplier for model; defaulting to 1.0",
+ )
+ return 1.0
+
+
def get_provider_specific_geo_multiplier(model_info: ModelInfo, usage: Usage) -> float:
"""
Resolve the provider-specific regional pricing multiplier for the geo the
@@ -798,6 +825,7 @@ def generic_cost_per_token(
service_tier: str | None = None,
data_residency: str | None = None,
model_info: ModelInfo | None = None,
+ vertex_location: str | None = None,
) -> tuple[float, float]:
"""
Calculates the cost per token for a given model, prompt tokens, and completion tokens.
@@ -809,6 +837,9 @@ def generic_cost_per_token(
- usage: LiteLLM Usage block, containing anthropic caching information
- data_residency: optional OpenAI data-residency region (e.g. "eu", "us"),
used to apply the per-model regional-processing uplift multiplier.
+ - vertex_location: optional Vertex AI location the request was served from
+ (e.g. "us-east5", "global"), used to apply the per-model
+ regional-endpoint uplift multiplier when non-global.
Returns:
Tuple[float, float] - prompt_cost_in_usd, completion_cost_in_usd
@@ -968,6 +999,14 @@ def generic_cost_per_token(
prompt_cost *= uplift
completion_cost *= uplift
+ ## VERTEX REGIONAL-ENDPOINT UPLIFT
+ # Applied as a flat multiplier across all token costs for the request
+ # when the Vertex AI endpoint serving it is non-global.
+ vertex_uplift: Final = get_vertex_regional_endpoint_uplift(model_info, vertex_location)
+ if vertex_uplift != 1.0:
+ prompt_cost *= vertex_uplift
+ completion_cost *= vertex_uplift
+
return prompt_cost, completion_cost
@@ -988,6 +1027,7 @@ def get_token_type_cost_breakdown(
usage: Usage,
service_tier: str | None = None,
data_residency: str | None = None,
+ vertex_location: str | None = None,
) -> TokenTypeCostBreakdown:
"""
Provider-agnostic cost of reasoning and cache tokens, derived from the usage
@@ -1069,6 +1109,14 @@ def get_token_type_cost_breakdown(
cache_read_cost *= uplift
cache_creation_cost *= uplift
+ # Same flat uplift for Vertex AI non-global endpoints, keeping per-type
+ # costs reconciled with the totals for regional Vertex deployments.
+ vertex_uplift: Final = get_vertex_regional_endpoint_uplift(model_info, vertex_location)
+ if vertex_uplift != 1.0:
+ reasoning_cost *= vertex_uplift
+ cache_read_cost *= vertex_uplift
+ cache_creation_cost *= vertex_uplift
+
# Mirror the provider-specific geo uplift (e.g. Anthropic us: 1.1) the totals
# apply, so cache and reasoning line items stay reconciled with them.
geo_multiplier: Final = get_provider_specific_geo_multiplier(model_info=model_info, usage=usage)
diff --git a/litellm/llms/vertex_ai/cost_calculator.py b/litellm/llms/vertex_ai/cost_calculator.py
index 86a5bb207ec..a9f5d77350c 100644
--- a/litellm/llms/vertex_ai/cost_calculator.py
+++ b/litellm/llms/vertex_ai/cost_calculator.py
@@ -7,6 +7,7 @@ from litellm import verbose_logger
from litellm.litellm_core_utils.llm_cost_calc.utils import (
_is_above_128k,
generic_cost_per_token,
+ get_vertex_regional_endpoint_uplift,
)
from litellm.types.utils import ModelInfo, Usage
@@ -63,6 +64,7 @@ def cost_per_character(
usage: Usage,
prompt_characters: float | None = None,
completion_characters: float | None = None,
+ vertex_location: str | None = None,
) -> tuple[float, float]:
"""
Calculates the cost per character for a given VertexAI model, input messages, and response object.
@@ -72,6 +74,8 @@ def cost_per_character(
- custom_llm_provider: str, "vertex_ai-*"
- prompt_characters: float, the number of input characters
- completion_characters: float, the number of output characters
+ - vertex_location: the Vertex AI location serving the request; non-global
+ locations apply the model's regional-endpoint uplift multiplier
Returns:
Tuple[float, float] - prompt_cost_in_usd, completion_cost_in_usd
@@ -79,8 +83,6 @@ def cost_per_character(
Raises:
Exception if model requires >128k pricing, but model cost not mapped
"""
- model_info = litellm.get_model_info(model=model, custom_llm_provider=custom_llm_provider)
-
## GET MODEL INFO
model_info = litellm.get_model_info(model=model, custom_llm_provider=custom_llm_provider)
@@ -162,7 +164,10 @@ def cost_per_character(
usage=usage,
)
- return prompt_cost, completion_cost
+ # Applied once here; the cost_per_token fallbacks above are called without
+ # vertex_location so the uplift can never compound.
+ vertex_uplift: Final = get_vertex_regional_endpoint_uplift(model_info, vertex_location)
+ return prompt_cost * vertex_uplift, completion_cost * vertex_uplift
def _handle_128k_pricing(
@@ -196,6 +201,7 @@ def cost_per_token(
custom_llm_provider: str,
usage: Usage,
service_tier: str | None = None,
+ vertex_location: str | None = None,
) -> tuple[float, float]:
"""
Calculates the cost per token for a given model, prompt tokens, and completion tokens.
@@ -207,6 +213,8 @@ def cost_per_token(
- completion_tokens: float, the number of output tokens
- service_tier: optional tier derived from Gemini trafficType
("priority" for ON_DEMAND_PRIORITY, "flex" for FLEX/batch).
+ - vertex_location: the Vertex AI location serving the request; non-global
+ locations apply the model's regional-endpoint uplift multiplier
Returns:
Tuple[float, float] - prompt_cost_in_usd, completion_cost_in_usd
@@ -222,14 +230,17 @@ def cost_per_token(
input_cost_per_token_above_128k_tokens: Final = model_info.get("input_cost_per_token_above_128k_tokens")
output_cost_per_token_above_128k_tokens: Final = model_info.get("output_cost_per_token_above_128k_tokens")
if input_cost_per_token_above_128k_tokens is not None or output_cost_per_token_above_128k_tokens is not None:
- return _handle_128k_pricing(
+ prompt_cost_128k, completion_cost_128k = _handle_128k_pricing(
model_info=model_info,
usage=usage,
)
+ vertex_uplift: Final = get_vertex_regional_endpoint_uplift(model_info, vertex_location)
+ return prompt_cost_128k * vertex_uplift, completion_cost_128k * vertex_uplift
return generic_cost_per_token(
model=model,
custom_llm_provider=custom_llm_provider,
usage=usage,
service_tier=service_tier,
+ vertex_location=vertex_location,
)
diff --git a/litellm/llms/vertex_ai/vertex_llm_base.py b/litellm/llms/vertex_ai/vertex_llm_base.py
index 445e34966a9..0b3c003a60a 100644
--- a/litellm/llms/vertex_ai/vertex_llm_base.py
+++ b/litellm/llms/vertex_ai/vertex_llm_base.py
@@ -8,6 +8,7 @@ import asyncio
import json
import os
import threading
+from collections.abc import Mapping
from typing import TYPE_CHECKING, Any, Final, Literal
from urllib.parse import urlparse
@@ -68,7 +69,8 @@ class VertexBase:
# re-acquire it without deadlocking the current thread.
self._sync_refresh_lock = threading.RLock()
- def get_vertex_region(self, vertex_region: str | None, model: str) -> str:
+ @staticmethod
+ def get_vertex_region(vertex_region: str | None, model: str) -> str:
import litellm
# Try to get supported_regions directly from model_cost
@@ -1191,7 +1193,7 @@ class VertexBase:
)
@staticmethod
- def safe_get_vertex_ai_location(litellm_params: dict) -> str | None:
+ def safe_get_vertex_ai_location(litellm_params: Mapping[str, object]) -> str | None:
"""
Safely get Vertex AI location without mutating the litellm_params dict.
@@ -1204,10 +1206,7 @@ class VertexBase:
Returns:
Vertex AI location/region or None
"""
- return (
- litellm_params.get("vertex_location")
- or litellm_params.get("vertex_ai_location")
- or litellm.vertex_location
- or get_secret_str("VERTEXAI_LOCATION")
- or get_secret_str("VERTEX_LOCATION")
- )
+ for configured in (litellm_params.get("vertex_location"), litellm_params.get("vertex_ai_location")):
+ if isinstance(configured, str) and configured:
+ return configured
+ return litellm.vertex_location or get_secret_str("VERTEXAI_LOCATION") or get_secret_str("VERTEX_LOCATION")
diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json
index 07f9027313b..52f1763a5ba 100644
--- a/litellm/model_prices_and_context_window_backup.json
+++ b/litellm/model_prices_and_context_window_backup.json
@@ -19624,6 +19624,7 @@
"mode": "chat",
"output_cost_per_reasoning_token": 9e-06,
"output_cost_per_token": 9e-06,
+ "regional_endpoint_uplift_multiplier": 1.1,
"source": "https://cloud.google.com/vertex-ai/generative-ai/pricing",
"supported_endpoints": [
"/v1/chat/completions",
@@ -19679,6 +19680,7 @@
"output_cost_per_token": 3.75e-06,
"output_cost_per_token_batches": 1.875e-06,
"output_cost_per_token_flex": 1.875e-06,
+ "regional_endpoint_uplift_multiplier": 1.1,
"source": "https://cloud.google.com/vertex-ai/generative-ai/pricing",
"supported_endpoints": [
"/v1/chat/completions",
@@ -19733,6 +19735,7 @@
"output_cost_per_token": 3.75e-06,
"output_cost_per_token_batches": 1.875e-06,
"output_cost_per_token_flex": 1.875e-06,
+ "regional_endpoint_uplift_multiplier": 1.1,
"source": "https://cloud.google.com/vertex-ai/generative-ai/pricing",
"supported_endpoints": [
"/v1/chat/completions",
@@ -38685,6 +38688,7 @@
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 5e-06,
+ "regional_endpoint_uplift_multiplier": 1.1,
"source": "https://cloud.google.com/vertex-ai/generative-ai/docs/partner-models/claude/haiku-4-5",
"supports_assistant_prefill": true,
"supports_function_calling": true,
@@ -38708,6 +38712,7 @@
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 5e-06,
+ "regional_endpoint_uplift_multiplier": 1.1,
"source": "https://cloud.google.com/vertex-ai/generative-ai/docs/partner-models/claude/haiku-4-5",
"supports_assistant_prefill": true,
"supports_function_calling": true,
@@ -38923,6 +38928,7 @@
"max_tokens": 64000,
"mode": "chat",
"output_cost_per_token": 2.5e-05,
+ "regional_endpoint_uplift_multiplier": 1.1,
"search_context_cost_per_query": {
"search_context_size_high": 0.01,
"search_context_size_low": 0.01,
@@ -38951,6 +38957,7 @@
"max_tokens": 64000,
"mode": "chat",
"output_cost_per_token": 2.5e-05,
+ "regional_endpoint_uplift_multiplier": 1.1,
"search_context_cost_per_query": {
"search_context_size_high": 0.01,
"search_context_size_low": 0.01,
@@ -38970,6 +38977,7 @@
"prompt_cache_min_tokens": 4096
},
"vertex_ai/claude-opus-4-6": {
+ "regional_endpoint_uplift_multiplier": 1.1,
"supports_adaptive_thinking": true,
"cache_creation_input_token_cost": 6.25e-06,
"cache_creation_input_token_cost_above_1hr": 1e-05,
@@ -39000,6 +39008,7 @@
"prompt_cache_min_tokens": 4096
},
"vertex_ai/claude-opus-4-6@default": {
+ "regional_endpoint_uplift_multiplier": 1.1,
"supports_adaptive_thinking": true,
"cache_creation_input_token_cost": 6.25e-06,
"cache_creation_input_token_cost_above_1hr": 1e-05,
@@ -39030,6 +39039,7 @@
"prompt_cache_min_tokens": 4096
},
"vertex_ai/claude-opus-4-7": {
+ "regional_endpoint_uplift_multiplier": 1.1,
"supports_adaptive_thinking": true,
"cache_creation_input_token_cost": 6.25e-06,
"cache_creation_input_token_cost_above_1hr": 1e-05,
@@ -39061,6 +39071,7 @@
"prompt_cache_min_tokens": 2048
},
"vertex_ai/claude-opus-4-7@default": {
+ "regional_endpoint_uplift_multiplier": 1.1,
"supports_adaptive_thinking": true,
"cache_creation_input_token_cost": 6.25e-06,
"cache_creation_input_token_cost_above_1hr": 1e-05,
@@ -39092,6 +39103,7 @@
"prompt_cache_min_tokens": 2048
},
"vertex_ai/claude-fable-5": {
+ "regional_endpoint_uplift_multiplier": 1.1,
"supports_mid_conversation_system": true,
"cache_creation_input_token_cost": 1.25e-05,
"cache_creation_input_token_cost_above_1hr": 2e-05,
@@ -39123,6 +39135,7 @@
"supports_max_reasoning_effort": true
},
"vertex_ai/claude-fable-5@default": {
+ "regional_endpoint_uplift_multiplier": 1.1,
"supports_mid_conversation_system": true,
"cache_creation_input_token_cost": 1.25e-05,
"cache_creation_input_token_cost_above_1hr": 2e-05,
@@ -39154,6 +39167,7 @@
"supports_max_reasoning_effort": true
},
"vertex_ai/claude-opus-5": {
+ "regional_endpoint_uplift_multiplier": 1.1,
"supports_mid_conversation_system": true,
"supports_adaptive_thinking": true,
"cache_creation_input_token_cost": 6.25e-06,
@@ -39186,6 +39200,7 @@
"prompt_cache_min_tokens": 512
},
"vertex_ai/claude-opus-5@default": {
+ "regional_endpoint_uplift_multiplier": 1.1,
"supports_mid_conversation_system": true,
"supports_adaptive_thinking": true,
"cache_creation_input_token_cost": 6.25e-06,
@@ -39218,6 +39233,7 @@
"prompt_cache_min_tokens": 512
},
"vertex_ai/claude-opus-4-8": {
+ "regional_endpoint_uplift_multiplier": 1.1,
"supports_mid_conversation_system": true,
"supports_adaptive_thinking": true,
"cache_creation_input_token_cost": 6.25e-06,
@@ -39250,6 +39266,7 @@
"prompt_cache_min_tokens": 1024
},
"vertex_ai/claude-opus-4-8@default": {
+ "regional_endpoint_uplift_multiplier": 1.1,
"supports_mid_conversation_system": true,
"supports_adaptive_thinking": true,
"cache_creation_input_token_cost": 6.25e-06,
@@ -39298,6 +39315,7 @@
"mode": "chat",
"output_cost_per_token": 1.5e-05,
"output_cost_per_token_batches": 7.5e-06,
+ "regional_endpoint_uplift_multiplier": 1.1,
"supports_assistant_prefill": true,
"supports_computer_use": true,
"supports_function_calling": true,
@@ -39310,6 +39328,7 @@
"prompt_cache_min_tokens": 1024
},
"vertex_ai/claude-sonnet-5": {
+ "regional_endpoint_uplift_multiplier": 1.1,
"supports_mid_conversation_system": true,
"cache_creation_input_token_cost": 2.5e-06,
"cache_creation_input_token_cost_above_1hr": 4e-06,
@@ -39342,6 +39361,7 @@
"prompt_cache_min_tokens": 1024
},
"vertex_ai/claude-sonnet-4-6": {
+ "regional_endpoint_uplift_multiplier": 1.1,
"supports_adaptive_thinking": true,
"cache_creation_input_token_cost": 3.75e-06,
"cache_creation_input_token_cost_above_1hr": 6e-06,
@@ -39388,6 +39408,7 @@
"mode": "chat",
"output_cost_per_token": 1.5e-05,
"output_cost_per_token_batches": 7.5e-06,
+ "regional_endpoint_uplift_multiplier": 1.1,
"supports_assistant_prefill": true,
"supports_computer_use": true,
"supports_function_calling": true,
@@ -39794,6 +39815,7 @@
"output_cost_per_token_batches": 7.5e-07,
"output_cost_per_token_flex": 7.5e-07,
"output_cost_per_token_priority": 2.7e-06,
+ "regional_endpoint_uplift_multiplier": 1.1,
"source": "https://cloud.google.com/vertex-ai/generative-ai/pricing#gemini-models",
"supported_endpoints": [
"/v1/chat/completions",
@@ -39849,6 +39871,7 @@
"output_cost_per_token_batches": 1.25e-06,
"output_cost_per_token_flex": 1.25e-06,
"output_cost_per_token_priority": 4.5e-06,
+ "regional_endpoint_uplift_multiplier": 1.1,
"source": "https://cloud.google.com/vertex-ai/generative-ai/pricing#gemini-models",
"supported_endpoints": [
"/v1/chat/completions",
@@ -47014,6 +47037,7 @@
}
},
"vertex_ai/claude-sonnet-5@default": {
+ "regional_endpoint_uplift_multiplier": 1.1,
"supports_mid_conversation_system": true,
"cache_creation_input_token_cost": 2.5e-06,
"cache_creation_input_token_cost_above_1hr": 4e-06,
@@ -47046,6 +47070,7 @@
"prompt_cache_min_tokens": 1024
},
"vertex_ai/claude-sonnet-4-6@default": {
+ "regional_endpoint_uplift_multiplier": 1.1,
"supports_adaptive_thinking": true,
"cache_creation_input_token_cost": 3.75e-06,
"cache_creation_input_token_cost_above_1hr": 6e-06,
diff --git a/litellm/proxy/spend_tracking/savings.py b/litellm/proxy/spend_tracking/savings.py
index 448723ab3bc..997180efdde 100644
--- a/litellm/proxy/spend_tracking/savings.py
+++ b/litellm/proxy/spend_tracking/savings.py
@@ -130,6 +130,7 @@ class PricingBasis(NamedTuple):
service_tier: str | None = None
data_residency: str | None = None
+ vertex_location: str | None = None
_STANDARD_RATES: Final = PricingBasis()
@@ -141,8 +142,8 @@ def _pricing_basis(cost_breakdown: Mapping[str, object] | None) -> PricingBasis:
Rows written before this field shipped carry neither key, and there is no backfill:
they price at standard rates, which is what they already did.
- Both values survive a JSON round trip on the way here, so neither is guaranteed to be
- a string. `generic_cost_per_token` calls `.lower()` on both without a type check, and
+ These values survive a JSON round trip on the way here, so none is guaranteed to be
+ a string. `generic_cost_per_token` calls `.lower()` on them without a type check, and
the resulting `AttributeError` would be swallowed into a silent zero by the caller's
`except`, so anything that is not a string is dropped here instead.
"""
@@ -150,9 +151,11 @@ def _pricing_basis(cost_breakdown: Mapping[str, object] | None) -> PricingBasis:
return _STANDARD_RATES
service_tier: Final = cost_breakdown.get("service_tier")
data_residency: Final = cost_breakdown.get("data_residency")
+ vertex_location: Final = cost_breakdown.get("vertex_location")
return PricingBasis(
service_tier=service_tier if isinstance(service_tier, str) else None,
data_residency=data_residency if isinstance(data_residency, str) else None,
+ vertex_location=vertex_location if isinstance(vertex_location, str) else None,
)
@@ -193,6 +196,7 @@ def _cost_of_usage(
service_tier=basis.service_tier,
data_residency=basis.data_residency,
model_info=model_info,
+ vertex_location=basis.vertex_location,
)
except Exception as e: # noqa: BLE001 # get_model_info raises bare Exception for unmapped models; degrade to zero savings
verbose_proxy_logger.debug(
diff --git a/litellm/types/utils.py b/litellm/types/utils.py
index 07005d7f9ad..5f629eb129f 100644
--- a/litellm/types/utils.py
+++ b/litellm/types/utils.py
@@ -248,6 +248,9 @@ class ModelInfoBase(ProviderSpecificModelInfo, total=False):
regional_processing_uplift_multiplier_us: (
float | None
) # OpenAI US data-residency uplift multiplier applied to all token costs (e.g. 1.10 = +10%)
+ regional_endpoint_uplift_multiplier: ReadOnly[
+ float | None
+ ] # Vertex AI non-global (regional) endpoint uplift multiplier applied to all token costs (e.g. 1.10 = +10%)
output_cost_per_character: float | None # only for vertex ai models
output_cost_per_audio_token: float | None
output_cost_per_token_above_128k_tokens: float | None # only for vertex ai models
@@ -3113,16 +3116,17 @@ class CostBreakdown(TypedDict, total=False):
"""
Detailed cost breakdown for a request.
- ``service_tier`` and ``data_residency`` record the pricing basis the cost was
- computed on, not what the caller asked for. A consumer that has to price a
- counterfactual against this request (what another model would have charged for
- it) needs the same basis to compare like with like, and re-deriving it from the
- request is not possible after the fact: the tier the biller used comes from
- ``optional_params``, which no log record carries.
+ ``service_tier``, ``data_residency``, and ``vertex_location`` record the pricing
+ basis the cost was computed on, not what the caller asked for. A consumer that has
+ to price a counterfactual against this request (what another model would have
+ charged for it) needs the same basis to compare like with like, and re-deriving it
+ from the request is not possible after the fact: the tier the biller used comes
+ from ``optional_params``, which no log record carries.
"""
service_tier: str | None
data_residency: str | None
+ vertex_location: ReadOnly[str | None]
input_cost: float # Cost of raw (non-cached) input tokens only
cache_read_cost: float # Cost of cache-read tokens (discounted rate)
cache_creation_cost: float # Cost of cache-write tokens (premium rate)
@@ -3388,6 +3392,7 @@ class CustomPricingLiteLLMParams(MirroredPricingParams):
annotation_cost_per_page: float | None = None
regional_processing_uplift_multiplier_eu: float | None = None
regional_processing_uplift_multiplier_us: float | None = None
+ regional_endpoint_uplift_multiplier: float | None = None
@classmethod
def strip_custom_pricing_fields(cls, model_info: dict[str, Any]) -> dict[str, Any]:
diff --git a/litellm/utils.py b/litellm/utils.py
index a7b70c4129a..8be2c98fc68 100644
--- a/litellm/utils.py
+++ b/litellm/utils.py
@@ -5662,6 +5662,7 @@ def _get_model_info_helper(
regional_processing_uplift_multiplier_us=_model_info.get(
"regional_processing_uplift_multiplier_us", None
),
+ regional_endpoint_uplift_multiplier=_model_info.get("regional_endpoint_uplift_multiplier", None),
output_cost_per_audio_token=_model_info.get("output_cost_per_audio_token", None),
output_cost_per_character=_model_info.get("output_cost_per_character", None),
output_cost_per_reasoning_token=_model_info.get("output_cost_per_reasoning_token", None),
diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json
index 07f9027313b..52f1763a5ba 100644
--- a/model_prices_and_context_window.json
+++ b/model_prices_and_context_window.json
@@ -19624,6 +19624,7 @@
"mode": "chat",
"output_cost_per_reasoning_token": 9e-06,
"output_cost_per_token": 9e-06,
+ "regional_endpoint_uplift_multiplier": 1.1,
"source": "https://cloud.google.com/vertex-ai/generative-ai/pricing",
"supported_endpoints": [
"/v1/chat/completions",
@@ -19679,6 +19680,7 @@
"output_cost_per_token": 3.75e-06,
"output_cost_per_token_batches": 1.875e-06,
"output_cost_per_token_flex": 1.875e-06,
+ "regional_endpoint_uplift_multiplier": 1.1,
"source": "https://cloud.google.com/vertex-ai/generative-ai/pricing",
"supported_endpoints": [
"/v1/chat/completions",
@@ -19733,6 +19735,7 @@
"output_cost_per_token": 3.75e-06,
"output_cost_per_token_batches": 1.875e-06,
"output_cost_per_token_flex": 1.875e-06,
+ "regional_endpoint_uplift_multiplier": 1.1,
"source": "https://cloud.google.com/vertex-ai/generative-ai/pricing",
"supported_endpoints": [
"/v1/chat/completions",
@@ -38685,6 +38688,7 @@
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 5e-06,
+ "regional_endpoint_uplift_multiplier": 1.1,
"source": "https://cloud.google.com/vertex-ai/generative-ai/docs/partner-models/claude/haiku-4-5",
"supports_assistant_prefill": true,
"supports_function_calling": true,
@@ -38708,6 +38712,7 @@
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 5e-06,
+ "regional_endpoint_uplift_multiplier": 1.1,
"source": "https://cloud.google.com/vertex-ai/generative-ai/docs/partner-models/claude/haiku-4-5",
"supports_assistant_prefill": true,
"supports_function_calling": true,
@@ -38923,6 +38928,7 @@
"max_tokens": 64000,
"mode": "chat",
"output_cost_per_token": 2.5e-05,
+ "regional_endpoint_uplift_multiplier": 1.1,
"search_context_cost_per_query": {
"search_context_size_high": 0.01,
"search_context_size_low": 0.01,
@@ -38951,6 +38957,7 @@
"max_tokens": 64000,
"mode": "chat",
"output_cost_per_token": 2.5e-05,
+ "regional_endpoint_uplift_multiplier": 1.1,
"search_context_cost_per_query": {
"search_context_size_high": 0.01,
"search_context_size_low": 0.01,
@@ -38970,6 +38977,7 @@
"prompt_cache_min_tokens": 4096
},
"vertex_ai/claude-opus-4-6": {
+ "regional_endpoint_uplift_multiplier": 1.1,
"supports_adaptive_thinking": true,
"cache_creation_input_token_cost": 6.25e-06,
"cache_creation_input_token_cost_above_1hr": 1e-05,
@@ -39000,6 +39008,7 @@
"prompt_cache_min_tokens": 4096
},
"vertex_ai/claude-opus-4-6@default": {
+ "regional_endpoint_uplift_multiplier": 1.1,
"supports_adaptive_thinking": true,
"cache_creation_input_token_cost": 6.25e-06,
"cache_creation_input_token_cost_above_1hr": 1e-05,
@@ -39030,6 +39039,7 @@
"prompt_cache_min_tokens": 4096
},
"vertex_ai/claude-opus-4-7": {
+ "regional_endpoint_uplift_multiplier": 1.1,
"supports_adaptive_thinking": true,
"cache_creation_input_token_cost": 6.25e-06,
"cache_creation_input_token_cost_above_1hr": 1e-05,
@@ -39061,6 +39071,7 @@
"prompt_cache_min_tokens": 2048
},
"vertex_ai/claude-opus-4-7@default": {
+ "regional_endpoint_uplift_multiplier": 1.1,
"supports_adaptive_thinking": true,
"cache_creation_input_token_cost": 6.25e-06,
"cache_creation_input_token_cost_above_1hr": 1e-05,
@@ -39092,6 +39103,7 @@
"prompt_cache_min_tokens": 2048
},
"vertex_ai/claude-fable-5": {
+ "regional_endpoint_uplift_multiplier": 1.1,
"supports_mid_conversation_system": true,
"cache_creation_input_token_cost": 1.25e-05,
"cache_creation_input_token_cost_above_1hr": 2e-05,
@@ -39123,6 +39135,7 @@
"supports_max_reasoning_effort": true
},
"vertex_ai/claude-fable-5@default": {
+ "regional_endpoint_uplift_multiplier": 1.1,
"supports_mid_conversation_system": true,
"cache_creation_input_token_cost": 1.25e-05,
"cache_creation_input_token_cost_above_1hr": 2e-05,
@@ -39154,6 +39167,7 @@
"supports_max_reasoning_effort": true
},
"vertex_ai/claude-opus-5": {
+ "regional_endpoint_uplift_multiplier": 1.1,
"supports_mid_conversation_system": true,
"supports_adaptive_thinking": true,
"cache_creation_input_token_cost": 6.25e-06,
@@ -39186,6 +39200,7 @@
"prompt_cache_min_tokens": 512
},
"vertex_ai/claude-opus-5@default": {
+ "regional_endpoint_uplift_multiplier": 1.1,
"supports_mid_conversation_system": true,
"supports_adaptive_thinking": true,
"cache_creation_input_token_cost": 6.25e-06,
@@ -39218,6 +39233,7 @@
"prompt_cache_min_tokens": 512
},
"vertex_ai/claude-opus-4-8": {
+ "regional_endpoint_uplift_multiplier": 1.1,
"supports_mid_conversation_system": true,
"supports_adaptive_thinking": true,
"cache_creation_input_token_cost": 6.25e-06,
@@ -39250,6 +39266,7 @@
"prompt_cache_min_tokens": 1024
},
"vertex_ai/claude-opus-4-8@default": {
+ "regional_endpoint_uplift_multiplier": 1.1,
"supports_mid_conversation_system": true,
"supports_adaptive_thinking": true,
"cache_creation_input_token_cost": 6.25e-06,
@@ -39298,6 +39315,7 @@
"mode": "chat",
"output_cost_per_token": 1.5e-05,
"output_cost_per_token_batches": 7.5e-06,
+ "regional_endpoint_uplift_multiplier": 1.1,
"supports_assistant_prefill": true,
"supports_computer_use": true,
"supports_function_calling": true,
@@ -39310,6 +39328,7 @@
"prompt_cache_min_tokens": 1024
},
"vertex_ai/claude-sonnet-5": {
+ "regional_endpoint_uplift_multiplier": 1.1,
"supports_mid_conversation_system": true,
"cache_creation_input_token_cost": 2.5e-06,
"cache_creation_input_token_cost_above_1hr": 4e-06,
@@ -39342,6 +39361,7 @@
"prompt_cache_min_tokens": 1024
},
"vertex_ai/claude-sonnet-4-6": {
+ "regional_endpoint_uplift_multiplier": 1.1,
"supports_adaptive_thinking": true,
"cache_creation_input_token_cost": 3.75e-06,
"cache_creation_input_token_cost_above_1hr": 6e-06,
@@ -39388,6 +39408,7 @@
"mode": "chat",
"output_cost_per_token": 1.5e-05,
"output_cost_per_token_batches": 7.5e-06,
+ "regional_endpoint_uplift_multiplier": 1.1,
"supports_assistant_prefill": true,
"supports_computer_use": true,
"supports_function_calling": true,
@@ -39794,6 +39815,7 @@
"output_cost_per_token_batches": 7.5e-07,
"output_cost_per_token_flex": 7.5e-07,
"output_cost_per_token_priority": 2.7e-06,
+ "regional_endpoint_uplift_multiplier": 1.1,
"source": "https://cloud.google.com/vertex-ai/generative-ai/pricing#gemini-models",
"supported_endpoints": [
"/v1/chat/completions",
@@ -39849,6 +39871,7 @@
"output_cost_per_token_batches": 1.25e-06,
"output_cost_per_token_flex": 1.25e-06,
"output_cost_per_token_priority": 4.5e-06,
+ "regional_endpoint_uplift_multiplier": 1.1,
"source": "https://cloud.google.com/vertex-ai/generative-ai/pricing#gemini-models",
"supported_endpoints": [
"/v1/chat/completions",
@@ -47014,6 +47037,7 @@
}
},
"vertex_ai/claude-sonnet-5@default": {
+ "regional_endpoint_uplift_multiplier": 1.1,
"supports_mid_conversation_system": true,
"cache_creation_input_token_cost": 2.5e-06,
"cache_creation_input_token_cost_above_1hr": 4e-06,
@@ -47046,6 +47070,7 @@
"prompt_cache_min_tokens": 1024
},
"vertex_ai/claude-sonnet-4-6@default": {
+ "regional_endpoint_uplift_multiplier": 1.1,
"supports_adaptive_thinking": true,
"cache_creation_input_token_cost": 3.75e-06,
"cache_creation_input_token_cost_above_1hr": 6e-06,
diff --git a/model_prices_and_context_window.schema.json b/model_prices_and_context_window.schema.json
index cd02fde595f..82854a3b717 100644
--- a/model_prices_and_context_window.schema.json
+++ b/model_prices_and_context_window.schema.json
@@ -514,6 +514,11 @@
"type": "object",
"description": "Provider-internal routing hints (e.g. bedrock_invocation_schema)."
},
+ "regional_endpoint_uplift_multiplier": {
+ "type": "number",
+ "minimum": 1,
+ "description": "Multiplier applied to all token costs when served from a non-global Vertex AI endpoint (e.g. 1.10 = +10%)."
+ },
"regional_processing_uplift_multiplier_eu": {
"type": "number",
"minimum": 1,
diff --git a/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py b/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py
index 1826f56d667..06be96fefdf 100644
--- a/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py
+++ b/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py
@@ -2324,6 +2324,87 @@ def test_data_residency_composes_with_service_tier(_local_model_cost_map):
assert priority_eu_total == pytest.approx(priority_base_total * 1.10, rel=1e-9)
+@pytest.mark.parametrize("model", ["gemini-3.5-flash", "claude-haiku-4-5@20251001"])
+@pytest.mark.parametrize("vertex_location", ["us-central1", "us-east5", "europe-west1", "asia-southeast1"])
+def test_vertex_regional_location_applies_uplift(vertex_location, model, _local_model_cost_map):
+ """Google bills every non-global Vertex endpoint at 1.1x the global rate for GA
+ Gemini 3+ and regional-pricing Claude models, so a request served from a regional
+ location must cost 1.1x what the same usage costs on the global endpoint."""
+ from litellm.types.utils import Usage
+
+ usage = Usage(prompt_tokens=1000, completion_tokens=500, total_tokens=1500)
+
+ base = generic_cost_per_token(model=model, usage=usage, custom_llm_provider="vertex_ai")
+ regional = generic_cost_per_token(
+ model=model,
+ usage=usage,
+ custom_llm_provider="vertex_ai",
+ vertex_location=vertex_location,
+ )
+
+ base_total = base[0] + base[1]
+ regional_total = regional[0] + regional[1]
+
+ assert base_total > 0
+ assert regional_total == pytest.approx(base_total * 1.10, rel=1e-9)
+ assert regional[0] == pytest.approx(base[0] * 1.10, rel=1e-9)
+ assert regional[1] == pytest.approx(base[1] * 1.10, rel=1e-9)
+
+
+@pytest.mark.parametrize("vertex_location", [None, "global", "GLOBAL"])
+def test_vertex_global_or_absent_location_no_uplift(vertex_location, _local_model_cost_map):
+ """The global endpoint prices at the base rate, whatever the casing, and an
+ unresolved location must never uplift."""
+ from litellm.types.utils import Usage
+
+ usage = Usage(prompt_tokens=1000, completion_tokens=500, total_tokens=1500)
+
+ base = generic_cost_per_token(
+ model="claude-haiku-4-5@20251001", usage=usage, custom_llm_provider="vertex_ai"
+ )
+ located = generic_cost_per_token(
+ model="claude-haiku-4-5@20251001",
+ usage=usage,
+ custom_llm_provider="vertex_ai",
+ vertex_location=vertex_location,
+ )
+
+ assert base == located
+
+
+@pytest.mark.parametrize("model", ["claude-opus-4-1", "gemini-2.0-flash-001"])
+def test_vertex_location_no_uplift_for_uniformly_priced_model(model, _local_model_cost_map):
+ """Models Google prices uniformly across endpoints (Gemini 2.x, Claude Opus 4.1
+ and older) carry no multiplier and must not move with the location."""
+ from litellm.types.utils import Usage
+
+ usage = Usage(prompt_tokens=1000, completion_tokens=500, total_tokens=1500)
+
+ base = generic_cost_per_token(model=model, usage=usage, custom_llm_provider="vertex_ai")
+ regional = generic_cost_per_token(
+ model=model,
+ usage=usage,
+ custom_llm_provider="vertex_ai",
+ vertex_location="us-east5",
+ )
+
+ assert base == regional, f"{model} should not have a regional-endpoint uplift"
+
+
+def test_vertex_uplift_invalid_multiplier_defaults_to_one():
+ """A malformed multiplier in the cost map degrades to base pricing, never raises."""
+ from litellm.litellm_core_utils.llm_cost_calc.utils import (
+ get_vertex_regional_endpoint_uplift,
+ )
+
+ assert (
+ get_vertex_regional_endpoint_uplift(
+ {"regional_endpoint_uplift_multiplier": "not-a-number"}, "us-east5"
+ )
+ == 1.0
+ )
+
+
def test_priority_service_tier_above_threshold_uses_priority_tier_rates_for_cached_tokens(
_local_model_cost_map,
):
@@ -2877,6 +2958,57 @@ def test_token_type_cost_breakdown_applies_regional_uplift():
assert text_input_cost + eu.cache_read_cost == pytest.approx(prompt_cost)
+def test_token_type_cost_breakdown_applies_vertex_regional_uplift():
+ """
+ Non-global Vertex endpoints apply a flat 1.1x uplift to every token cost. The
+ per-type breakdown must apply the same uplift via vertex_location so it stays
+ reconciled with the uplifted input_cost/output_cost totals, instead of being
+ logged at the global rate.
+ """
+ os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
+ litellm.model_cost = litellm.get_model_cost_map(url="")
+
+ model = "claude-haiku-4-5@20251001"
+ custom_llm_provider = "vertex_ai"
+ usage = Usage(
+ prompt_tokens=1000,
+ completion_tokens=500,
+ total_tokens=1500,
+ prompt_tokens_details=PromptTokensDetailsWrapper(
+ cached_tokens=400, text_tokens=600
+ ),
+ )
+
+ model_info = litellm.get_model_info(
+ model=model, custom_llm_provider=custom_llm_provider
+ )
+ uplift = model_info["regional_endpoint_uplift_multiplier"]
+ assert uplift > 1.0
+
+ base = get_token_type_cost_breakdown(
+ model=model, custom_llm_provider=custom_llm_provider, usage=usage
+ )
+ regional = get_token_type_cost_breakdown(
+ model=model,
+ custom_llm_provider=custom_llm_provider,
+ usage=usage,
+ vertex_location="us-east5",
+ )
+
+ assert base.cache_read_cost > 0
+ assert regional.cache_read_cost == pytest.approx(base.cache_read_cost * uplift)
+
+ # The uplifted breakdown must still reconcile with the uplifted totals.
+ prompt_cost, _completion_cost = generic_cost_per_token(
+ model=model,
+ usage=usage,
+ custom_llm_provider=custom_llm_provider,
+ vertex_location="us-east5",
+ )
+ text_input_cost = 600 * model_info["input_cost_per_token"] * uplift
+ assert text_input_cost + regional.cache_read_cost == pytest.approx(prompt_cost)
+
+
def test_token_type_cost_breakdown_applies_anthropic_geo_multiplier(monkeypatch):
"""
Anthropic's regional (geo) uplift lives in provider_specific_entry and is
diff --git a/tests/test_litellm/litellm_core_utils/test_litellm_logging.py b/tests/test_litellm/litellm_core_utils/test_litellm_logging.py
index 54016470f8b..01e5ec26b1e 100644
--- a/tests/test_litellm/litellm_core_utils/test_litellm_logging.py
+++ b/tests/test_litellm/litellm_core_utils/test_litellm_logging.py
@@ -4958,3 +4958,80 @@ def test_pre_call_redacts_and_masks_raw_request(logging_obj):
raw_api_base = logging_obj.model_call_details["raw_request_typed_dict"]["raw_request_api_base"]
assert _GEMINI_KEY not in raw_api_base
assert "key=*****" in raw_api_base
+
+
+def test_resolve_vertex_location_for_cost():
+ """Vertex requests resolve the serving location the way dispatch does; other providers get None."""
+ from litellm.litellm_core_utils.litellm_logging import (
+ _resolve_vertex_location_for_cost,
+ )
+
+ assert _resolve_vertex_location_for_cost("openai", {"vertex_location": "us-east5"}, "gpt-4o") is None
+ assert _resolve_vertex_location_for_cost(None, {}, "gemini-3.5-flash") is None
+ assert (
+ _resolve_vertex_location_for_cost("vertex_ai", {"vertex_location": "us-east5"}, "gemini-3.5-flash")
+ == "us-east5"
+ )
+ assert (
+ _resolve_vertex_location_for_cost("vertex_ai", {"vertex_location": "global"}, "gemini-3.5-flash") == "global"
+ )
+ assert (
+ _resolve_vertex_location_for_cost(
+ "vertex_ai_beta", {"vertex_ai_location": "europe-west1"}, "claude-haiku-4-5@20251001"
+ )
+ == "europe-west1"
+ )
+
+
+def test_resolve_vertex_location_for_cost_default_region(monkeypatch):
+ """With no location configured anywhere, resolution lands on the dispatch default us-central1."""
+ from litellm.litellm_core_utils.litellm_logging import (
+ _resolve_vertex_location_for_cost,
+ )
+
+ monkeypatch.delenv("VERTEXAI_LOCATION", raising=False)
+ monkeypatch.delenv("VERTEX_LOCATION", raising=False)
+ monkeypatch.setattr(litellm, "vertex_location", None)
+
+ assert _resolve_vertex_location_for_cost("vertex_ai", {}, "gemini-3.5-flash") == "us-central1"
+ assert _resolve_vertex_location_for_cost("vertex_ai", None, "gemini-3.5-flash") == "us-central1"
+
+
+def test_set_cost_breakdown_stores_vertex_location():
+ """vertex_location is recorded in the pricing basis, None for non-vertex requests."""
+ from datetime import datetime
+
+ logging_obj = LitellmLogging(
+ model="vertex_ai/claude-haiku-4-5@20251001",
+ messages=[{"role": "user", "content": "hi"}],
+ stream=False,
+ call_type="completion",
+ start_time=datetime.now(),
+ litellm_call_id="vertex-location-set",
+ function_id="f",
+ )
+ logging_obj.set_cost_breakdown(
+ input_cost=0.001,
+ output_cost=0.002,
+ total_cost=0.003,
+ cost_for_built_in_tools_cost_usd_dollar=0.0,
+ vertex_location="us-east5",
+ )
+ assert logging_obj.cost_breakdown["vertex_location"] == "us-east5"
+
+ no_location = LitellmLogging(
+ model="gpt-4o",
+ messages=[{"role": "user", "content": "hi"}],
+ stream=False,
+ call_type="completion",
+ start_time=datetime.now(),
+ litellm_call_id="vertex-location-absent",
+ function_id="f",
+ )
+ no_location.set_cost_breakdown(
+ input_cost=0.001,
+ output_cost=0.002,
+ total_cost=0.003,
+ cost_for_built_in_tools_cost_usd_dollar=0.0,
+ )
+ assert no_location.cost_breakdown.get("vertex_location") is None
diff --git a/tests/test_litellm/proxy/spend_tracking/test_savings.py b/tests/test_litellm/proxy/spend_tracking/test_savings.py
index 1435547c434..9006288bdae 100644
--- a/tests/test_litellm/proxy/spend_tracking/test_savings.py
+++ b/tests/test_litellm/proxy/spend_tracking/test_savings.py
@@ -841,6 +841,44 @@ def test_the_baseline_is_priced_on_the_basis_the_request_was_billed_at(basis, ex
assert reported == pytest.approx(expected_multiplier * baseline - served)
+def test_the_baseline_is_priced_on_the_vertex_location_the_request_was_billed_at(monkeypatch):
+ """A request served from a regional Vertex endpoint was billed with the
+ regional-endpoint uplift, so the counterfactual single-model operator would
+ have paid it too. The served model carries no uplift field, so only the
+ baseline moves with the recorded location."""
+ monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
+ monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
+
+ gemini = litellm.get_model_info("gemini-3.5-flash", "vertex_ai")
+ haiku = litellm.get_model_info("claude-haiku-4-5", "anthropic")
+ assert gemini.get("regional_endpoint_uplift_multiplier") == 1.1
+ assert haiku.get("regional_endpoint_uplift_multiplier") is None, "served model must not move with the basis"
+
+ usage = _usage(fresh=20_000, cached=0, written=0, out=1_000)
+ served = 20_000 * haiku["input_cost_per_token"] + 1_000 * haiku["output_cost_per_token"]
+ baseline = 20_000 * gemini["input_cost_per_token"] + 1_000 * gemini["output_cost_per_token"]
+
+ regional = compute_autorouter_savings(
+ baseline_model="vertex_ai/gemini-3.5-flash",
+ selected_model="claude-haiku-4-5",
+ selected_provider="anthropic",
+ usage=usage,
+ conversation_continuing=False,
+ cost_breakdown=_breakdown(served, vertex_location="us-east5"),
+ )
+ global_endpoint = compute_autorouter_savings(
+ baseline_model="vertex_ai/gemini-3.5-flash",
+ selected_model="claude-haiku-4-5",
+ selected_provider="anthropic",
+ usage=usage,
+ conversation_continuing=False,
+ cost_breakdown=_breakdown(served, vertex_location="global"),
+ )
+
+ assert regional == pytest.approx(1.1 * baseline - served)
+ assert global_endpoint == pytest.approx(baseline - served)
+
+
def test_a_baseline_recorded_on_the_decision_turns_the_driver_on():
"""An operator who configures nothing still sees the driver work."""
result = compute_savings_spend(
diff --git a/tests/test_litellm/test_cost_calculator.py b/tests/test_litellm/test_cost_calculator.py
index 75c90d793fe..6deaf5479e0 100644
--- a/tests/test_litellm/test_cost_calculator.py
+++ b/tests/test_litellm/test_cost_calculator.py
@@ -1742,6 +1742,75 @@ def test_azure_ai_cache_cost_calculation():
), f"Output cost mismatch: got {output_cost}, expected {expected_output_cost}"
+def test_vertex_regional_deployment_costs_uplift_over_global(monkeypatch):
+ """
+ Regression for https://github.com/BerriAI/litellm/issues/34393: two Vertex
+ deployments differing only in vertex_location must not price identically.
+ Google bills non-global endpoints at 1.1x for regional-pricing models, so the
+ regional request costs 1.1x the global one for the exact same usage, through
+ both vertex cost routes (Claude via cost_per_token, Gemini via
+ cost_per_character's token fallback).
+ """
+ monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
+ monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
+
+ usage = Usage(prompt_tokens=15, completion_tokens=5, total_tokens=20)
+ for model in ("claude-haiku-4-5@20251001", "gemini-3.5-flash"):
+ global_prompt, global_completion = cost_per_token(
+ model=model,
+ custom_llm_provider="vertex_ai",
+ usage_object=usage,
+ vertex_location="global",
+ )
+ regional_prompt, regional_completion = cost_per_token(
+ model=model,
+ custom_llm_provider="vertex_ai",
+ usage_object=usage,
+ vertex_location="us-east5",
+ )
+ global_total = global_prompt + global_completion
+ regional_total = regional_prompt + regional_completion
+ assert global_total > 0
+ assert regional_total == pytest.approx(global_total * 1.10, rel=1e-9), (
+ f"{model}: regional Vertex request must cost 1.1x the global one"
+ )
+
+
+def test_vertex_uplift_composes_with_above_128k_pricing(monkeypatch):
+ """The regional-endpoint uplift multiplies whatever rate the request priced at,
+ including the above-128k dynamic rates, so a synthetic model carrying both keys
+ prices regional above-128k usage at 1.1x the above-128k rate."""
+ monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
+ monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
+ litellm.model_cost["vertex_ai/fake-regional-128k-model"] = {
+ "litellm_provider": "vertex_ai",
+ "mode": "chat",
+ "input_cost_per_token": 1e-06,
+ "output_cost_per_token": 2e-06,
+ "input_cost_per_token_above_128k_tokens": 2e-06,
+ "output_cost_per_token_above_128k_tokens": 4e-06,
+ "regional_endpoint_uplift_multiplier": 1.1,
+ }
+
+ usage = Usage(prompt_tokens=200_000, completion_tokens=10, total_tokens=200_010)
+ global_prompt, global_completion = cost_per_token(
+ model="fake-regional-128k-model",
+ custom_llm_provider="vertex_ai",
+ usage_object=usage,
+ vertex_location="global",
+ )
+ regional_prompt, regional_completion = cost_per_token(
+ model="fake-regional-128k-model",
+ custom_llm_provider="vertex_ai",
+ usage_object=usage,
+ vertex_location="europe-west1",
+ )
+
+ assert global_prompt == pytest.approx(200_000 * 2e-06, rel=1e-9)
+ assert regional_prompt == pytest.approx(global_prompt * 1.10, rel=1e-9)
+ assert regional_completion == pytest.approx(global_completion * 1.10, rel=1e-9)
+
+
def test_cost_discount_vertex_ai():
"""
Test that cost discount is applied correctly for Vertex AI provider
diff --git a/type-discipline-budget.json b/type-discipline-budget.json
index 6f77a621a9e..56ffbe9fde8 100644
--- a/type-discipline-budget.json
+++ b/type-discipline-budget.json
@@ -1,6 +1,6 @@
{
"LIT001": {
- "limit": 22809
+ "limit": 22808
},
"LIT002": {
"limit": 26878
diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts
index 5794c60e97b..3cc7397f5bb 100644
--- a/ui/litellm-dashboard/src/lib/http/schema.d.ts
+++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts
@@ -27559,6 +27559,8 @@ export interface components {
quality_router_default_model?: string | null;
/** Region Name */
region_name?: string | null;
+ /** Regional Endpoint Uplift Multiplier */
+ regional_endpoint_uplift_multiplier?: number | null;
/** Regional Processing Uplift Multiplier Eu */
regional_processing_uplift_multiplier_eu?: number | null;
/** Regional Processing Uplift Multiplier Us */
@@ -36689,6 +36691,8 @@ export interface components {
quality_router_default_model?: string | null;
/** Region Name */
region_name?: string | null;
+ /** Regional Endpoint Uplift Multiplier */
+ regional_endpoint_uplift_multiplier?: number | null;
/** Regional Processing Uplift Multiplier Eu */
regional_processing_uplift_multiplier_eu?: number | null;
/** Regional Processing Uplift Multiplier Us */
From b477d0967a4bf1f4a869c70ddf5a740f1762ae86 Mon Sep 17 00:00:00 2001
From: mateo-berri <277851410+mateo-berri@users.noreply.github.com>
Date: Wed, 19 Aug 2026 15:34:38 -0700
Subject: [PATCH 04/24] fix(proxy): lift standard_logging_object onto
request_data before the logging object is popped
---
litellm/proxy/utils.py | 52 ++++++++++---------
tests/test_litellm/proxy/test_proxy_utils.py | 53 ++++++++++++++++++++
2 files changed, 82 insertions(+), 23 deletions(-)
diff --git a/litellm/proxy/utils.py b/litellm/proxy/utils.py
index a743526e975..58ac7882cb8 100644
--- a/litellm/proxy/utils.py
+++ b/litellm/proxy/utils.py
@@ -517,6 +517,34 @@ def _failure_usage_to_lift(
return estimated_usage, 0.0
+_EMPTY_LIFT: Final = MappingProxyType({})
+
+
+def _failure_fields_to_lift(request_data: Mapping[str, object]) -> Mapping[str, object]:
+ """Failure-path callbacks run after ``litellm_logging_obj`` is popped from
+ request_data (it is not serialisable), so the caller merges these fields
+ onto request_data first: the first-handoff instant for preprocessing
+ latency, recovered or estimated usage for token counts, and the standard
+ logging object for deployment attribution on failed-request spend logs."""
+ _logging_obj: Final = request_data.get("litellm_logging_obj")
+ if _logging_obj is None:
+ return _EMPTY_LIFT
+ _model_call_details: Final = getattr(_logging_obj, "model_call_details", {})
+ _first_handoff: Final = _model_call_details.get("first_api_call_start_time")
+ _usage_to_lift: Final = _failure_usage_to_lift(
+ model_call_details=_model_call_details,
+ request_body=request_data,
+ dispatched=_first_handoff is not None,
+ )
+ _entries: Final = (
+ ("first_api_call_start_time", _first_handoff),
+ ("combined_usage_object", None if _usage_to_lift is None else _usage_to_lift[0]),
+ ("response_cost", None if _usage_to_lift is None else _usage_to_lift[1]),
+ ("standard_logging_object", _model_call_details.get("standard_logging_object")),
+ )
+ return MappingProxyType({key: value for key, value in _entries if value is not None})
+
+
@dataclass(frozen=True)
class _CallbackCapabilities:
"""Cached per-hook capability flags derived from ``litellm.callbacks``.
@@ -2294,29 +2322,7 @@ class ProxyLogging:
original_exception=original_exception,
)
- # Lift the first-handoff instant onto request_data (top-level
- # internal key, not metadata) so failure-path callbacks can still
- # compute preprocessing latency after the logging object is popped.
- _logging_obj: Final = request_data.get("litellm_logging_obj")
- if _logging_obj is not None:
- _model_call_details: Final = getattr(_logging_obj, "model_call_details", {})
- _first_handoff: Final = _model_call_details.get("first_api_call_start_time")
- if _first_handoff is not None:
- request_data["first_api_call_start_time"] = _first_handoff
-
- # Lift recovered partial-stream usage, or an estimated input-side
- # usage for a dispatched failure, onto request_data so the
- # failure-path spend callbacks (which run after the logging object
- # is popped) record real token counts instead of zero.
- _usage_to_lift: Final = _failure_usage_to_lift(
- model_call_details=_model_call_details,
- request_body=request_data,
- dispatched=_first_handoff is not None,
- )
- if _usage_to_lift is not None:
- _lifted_usage, _lifted_cost = _usage_to_lift
- request_data["combined_usage_object"] = _lifted_usage
- request_data["response_cost"] = _lifted_cost
+ request_data.update(_failure_fields_to_lift(request_data))
# Remove before callbacks iterate — not serialisable
request_data.pop("litellm_logging_obj", None)
diff --git a/tests/test_litellm/proxy/test_proxy_utils.py b/tests/test_litellm/proxy/test_proxy_utils.py
index d6cf0e30139..c80130b44da 100644
--- a/tests/test_litellm/proxy/test_proxy_utils.py
+++ b/tests/test_litellm/proxy/test_proxy_utils.py
@@ -478,6 +478,59 @@ class TestPostCallFailureHookLiftsRecoveredPartialSpend:
assert "response_cost" not in request_data
+class TestPostCallFailureHookLiftsStandardLoggingObject:
+ """Failure callbacks read standard_logging_object from request_data, but
+ post_call_failure_hook pops litellm_logging_obj before they run. The hook
+ must lift the logging obj's standard_logging_object onto request_data so
+ failed-request spend logs keep deployment attribution (LIT-5795).
+ """
+
+ async def _run(self, request_data):
+ from unittest.mock import AsyncMock, patch
+
+ from litellm.proxy._types import UserAPIKeyAuth
+
+ proxy_logging_obj = ProxyLogging(user_api_key_cache=DualCache())
+ proxy_logging_obj.alert_types = []
+ with patch.object(proxy_logging_obj, "update_request_status", new=AsyncMock()):
+ await proxy_logging_obj.post_call_failure_hook(
+ request_data=request_data,
+ original_exception=Exception("boom"),
+ user_api_key_dict=UserAPIKeyAuth(),
+ )
+
+ @pytest.mark.asyncio
+ async def test_lifts_standard_logging_object(self):
+ sl_object = {"model_id": "mid-123", "model_group": "group-x"}
+ logging_obj = MagicMock()
+ logging_obj.model_call_details = {"standard_logging_object": sl_object}
+ request_data = {"litellm_logging_obj": logging_obj, "metadata": {}}
+ await self._run(request_data)
+ assert request_data["standard_logging_object"] is sl_object
+ assert "litellm_logging_obj" not in request_data
+
+ @pytest.mark.asyncio
+ async def test_logging_obj_value_overwrites_preexisting_key(self):
+ authoritative = {"model_id": "from-logging-obj"}
+ logging_obj = MagicMock()
+ logging_obj.model_call_details = {"standard_logging_object": authoritative}
+ request_data = {
+ "litellm_logging_obj": logging_obj,
+ "standard_logging_object": {"model_id": "client-injected"},
+ "metadata": {},
+ }
+ await self._run(request_data)
+ assert request_data["standard_logging_object"] is authoritative
+
+ @pytest.mark.asyncio
+ async def test_no_standard_logging_object_is_noop(self):
+ logging_obj = MagicMock()
+ logging_obj.model_call_details = {}
+ request_data = {"litellm_logging_obj": logging_obj, "metadata": {}}
+ await self._run(request_data)
+ assert "standard_logging_object" not in request_data
+
+
class TestPostCallFailureHookEstimatesDispatchedInputTokens:
"""A non-stream request that failed after dispatch (timeout, provider
error) consumed provider-billed input tokens but recovered no usage.
From 8494a4deee8bcc693de97e11ac23a0e6d062374b Mon Sep 17 00:00:00 2001
From: mateo-berri <277851410+mateo-berri@users.noreply.github.com>
Date: Wed, 19 Aug 2026 15:58:12 -0700
Subject: [PATCH 05/24] fix(vertex_ai): read the served location from
optional_params when pricing proxy calls
---
litellm/litellm_core_utils/litellm_logging.py | 14 ++-
litellm/llms/vertex_ai/vertex_llm_base.py | 21 +++-
.../test_litellm_logging.py | 108 +++++++++++++++---
3 files changed, 119 insertions(+), 24 deletions(-)
diff --git a/litellm/litellm_core_utils/litellm_logging.py b/litellm/litellm_core_utils/litellm_logging.py
index e97abcf0af6..91a312b4f45 100644
--- a/litellm/litellm_core_utils/litellm_logging.py
+++ b/litellm/litellm_core_utils/litellm_logging.py
@@ -375,18 +375,29 @@ def _published_pricing(deployment_model: str | None) -> ModelInfo | None:
def _resolve_vertex_location_for_cost(
custom_llm_provider: str | None,
litellm_params: Mapping[str, object] | None,
+ optional_params: Mapping[str, object] | None,
model: str,
) -> str | None:
"""
The Vertex AI location a request was served from, resolved the same way
dispatch resolves it, so regional deployments price with the
regional-endpoint uplift. None for non-Vertex providers.
+
+ Chat dispatch reads the location from request kwargs, which reach this
+ logging object through optional_params: on the proxy the logging object is
+ created before the router picks a deployment, so the deployment's location
+ never lands in litellm_params.
"""
if custom_llm_provider is None or not custom_llm_provider.startswith("vertex_ai"):
return None
from litellm.llms.vertex_ai.vertex_llm_base import VertexBase
- configured_location: Final = VertexBase.safe_get_vertex_ai_location(litellm_params or MappingProxyType({}))
+ empty: Final[Mapping[str, object]] = MappingProxyType({})
+ configured_location: Final = (
+ VertexBase.explicit_vertex_ai_location(optional_params or empty)
+ or VertexBase.explicit_vertex_ai_location(litellm_params or empty)
+ or VertexBase.safe_get_vertex_ai_location(empty)
+ )
return VertexBase.get_vertex_region(configured_location, model)
@@ -1598,6 +1609,7 @@ class Logging(LiteLLMLoggingBaseClass):
"vertex_location": _resolve_vertex_location_for_cost(
custom_llm_provider=self.model_call_details.get("custom_llm_provider", None),
litellm_params=(self.litellm_params if hasattr(self, "litellm_params") else None),
+ optional_params=self.optional_params,
model=litellm_model_name or self.model,
),
}
diff --git a/litellm/llms/vertex_ai/vertex_llm_base.py b/litellm/llms/vertex_ai/vertex_llm_base.py
index 0b3c003a60a..75098515deb 100644
--- a/litellm/llms/vertex_ai/vertex_llm_base.py
+++ b/litellm/llms/vertex_ai/vertex_llm_base.py
@@ -1192,6 +1192,17 @@ class VertexBase:
or get_secret_str("VERTEXAI_CREDENTIALS")
)
+ @staticmethod
+ def explicit_vertex_ai_location(params: Mapping[str, object]) -> str | None:
+ """
+ The location explicitly configured in the given params, without any
+ module-level or environment fallback. None when not configured.
+ """
+ for configured in (params.get("vertex_location"), params.get("vertex_ai_location")):
+ if isinstance(configured, str) and configured:
+ return configured
+ return None
+
@staticmethod
def safe_get_vertex_ai_location(litellm_params: Mapping[str, object]) -> str | None:
"""
@@ -1206,7 +1217,9 @@ class VertexBase:
Returns:
Vertex AI location/region or None
"""
- for configured in (litellm_params.get("vertex_location"), litellm_params.get("vertex_ai_location")):
- if isinstance(configured, str) and configured:
- return configured
- return litellm.vertex_location or get_secret_str("VERTEXAI_LOCATION") or get_secret_str("VERTEX_LOCATION")
+ return (
+ VertexBase.explicit_vertex_ai_location(litellm_params)
+ or litellm.vertex_location
+ or get_secret_str("VERTEXAI_LOCATION")
+ or get_secret_str("VERTEX_LOCATION")
+ )
diff --git a/tests/test_litellm/litellm_core_utils/test_litellm_logging.py b/tests/test_litellm/litellm_core_utils/test_litellm_logging.py
index 01e5ec26b1e..c2d73ea467d 100644
--- a/tests/test_litellm/litellm_core_utils/test_litellm_logging.py
+++ b/tests/test_litellm/litellm_core_utils/test_litellm_logging.py
@@ -4960,41 +4960,111 @@ def test_pre_call_redacts_and_masks_raw_request(logging_obj):
assert "key=*****" in raw_api_base
-def test_resolve_vertex_location_for_cost():
- """Vertex requests resolve the serving location the way dispatch does; other providers get None."""
+def _resolve(custom_llm_provider, litellm_params, optional_params, model):
from litellm.litellm_core_utils.litellm_logging import (
_resolve_vertex_location_for_cost,
)
- assert _resolve_vertex_location_for_cost("openai", {"vertex_location": "us-east5"}, "gpt-4o") is None
- assert _resolve_vertex_location_for_cost(None, {}, "gemini-3.5-flash") is None
- assert (
- _resolve_vertex_location_for_cost("vertex_ai", {"vertex_location": "us-east5"}, "gemini-3.5-flash")
- == "us-east5"
+ return _resolve_vertex_location_for_cost(
+ custom_llm_provider=custom_llm_provider,
+ litellm_params=litellm_params,
+ optional_params=optional_params,
+ model=model,
)
+
+
+def test_resolve_vertex_location_for_cost():
+ """Vertex requests resolve the serving location the way dispatch does; other providers get None."""
+ assert _resolve("openai", {"vertex_location": "us-east5"}, None, "gpt-4o") is None
+ assert _resolve(None, {}, None, "gemini-3.5-flash") is None
+ assert _resolve("vertex_ai", {"vertex_location": "us-east5"}, None, "gemini-3.5-flash") == "us-east5"
+ assert _resolve("vertex_ai", {"vertex_location": "global"}, None, "gemini-3.5-flash") == "global"
assert (
- _resolve_vertex_location_for_cost("vertex_ai", {"vertex_location": "global"}, "gemini-3.5-flash") == "global"
- )
- assert (
- _resolve_vertex_location_for_cost(
- "vertex_ai_beta", {"vertex_ai_location": "europe-west1"}, "claude-haiku-4-5@20251001"
- )
+ _resolve("vertex_ai_beta", {"vertex_ai_location": "europe-west1"}, None, "claude-haiku-4-5@20251001")
== "europe-west1"
)
+def test_resolve_vertex_location_for_cost_reads_optional_params(monkeypatch):
+ """
+ On the proxy the logging object predates deployment selection, so the deployment's
+ configured location only reaches it through optional_params. A configured global
+ location must beat the environment fallback, or every proxy call gets the regional uplift.
+ """
+ monkeypatch.setenv("VERTEXAI_LOCATION", "us-east5")
+ monkeypatch.setattr(litellm, "vertex_location", None)
+
+ assert _resolve("vertex_ai", {}, {"vertex_location": "global"}, "gemini-3.5-flash") == "global"
+ assert _resolve("vertex_ai", None, {"vertex_location": "europe-west1"}, "gemini-3.5-flash") == "europe-west1"
+ assert (
+ _resolve(
+ "vertex_ai",
+ {"vertex_location": "us-east5"},
+ {"vertex_location": "global"},
+ "gemini-3.5-flash",
+ )
+ == "global"
+ )
+ assert _resolve("vertex_ai", {"vertex_location": "global"}, {}, "gemini-3.5-flash") == "global"
+ assert _resolve("vertex_ai", {}, {}, "gemini-3.5-flash") == "us-east5"
+
+
def test_resolve_vertex_location_for_cost_default_region(monkeypatch):
"""With no location configured anywhere, resolution lands on the dispatch default us-central1."""
- from litellm.litellm_core_utils.litellm_logging import (
- _resolve_vertex_location_for_cost,
- )
-
monkeypatch.delenv("VERTEXAI_LOCATION", raising=False)
monkeypatch.delenv("VERTEX_LOCATION", raising=False)
monkeypatch.setattr(litellm, "vertex_location", None)
- assert _resolve_vertex_location_for_cost("vertex_ai", {}, "gemini-3.5-flash") == "us-central1"
- assert _resolve_vertex_location_for_cost("vertex_ai", None, "gemini-3.5-flash") == "us-central1"
+ assert _resolve("vertex_ai", {}, None, "gemini-3.5-flash") == "us-central1"
+ assert _resolve("vertex_ai", None, None, "gemini-3.5-flash") == "us-central1"
+
+
+def test_response_cost_calculator_prices_proxy_vertex_calls_on_the_configured_location(monkeypatch):
+ """
+ Proxy-shaped logging objects (created before the router picks a deployment) carry the
+ deployment's vertex_location only in optional_params. A global deployment must price at
+ base rates even when the environment points at a regional location, and a regional one
+ must price with the uplift.
+ """
+ from datetime import datetime
+
+ from litellm.litellm_core_utils.get_model_cost_map import get_model_cost_map
+
+ monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
+ monkeypatch.setattr(litellm, "model_cost", get_model_cost_map(url=""))
+ monkeypatch.setenv("VERTEXAI_LOCATION", "us-east5")
+ monkeypatch.setattr(litellm, "vertex_location", None)
+
+ def cost_at(location):
+ logging_obj = LitellmLogging(
+ model="gemini-3.5-flash",
+ messages=[{"role": "user", "content": "hi"}],
+ stream=False,
+ call_type="completion",
+ start_time=datetime.now(),
+ litellm_call_id=f"vertex-loc-{location}",
+ function_id="f",
+ )
+ logging_obj.update_environment_variables(
+ model="gemini-3.5-flash",
+ user="",
+ optional_params={"vertex_location": location},
+ litellm_params={"api_base": ""},
+ custom_llm_provider="vertex_ai",
+ )
+ response = ModelResponse(
+ id="resp-1",
+ model="gemini-3.5-flash",
+ choices=[{"message": {"role": "assistant", "content": "hello"}, "index": 0, "finish_reason": "stop"}],
+ usage={"prompt_tokens": 10, "completion_tokens": 5, "total_tokens": 15},
+ )
+ return logging_obj._response_cost_calculator(result=response)
+
+ info = litellm.model_cost["vertex_ai/gemini-3.5-flash"]
+ expected_global = 10 * info["input_cost_per_token"] + 5 * info["output_cost_per_token"]
+
+ assert cost_at("global") == pytest.approx(expected_global)
+ assert cost_at("us-east5") == pytest.approx(info["regional_endpoint_uplift_multiplier"] * expected_global)
def test_set_cost_breakdown_stores_vertex_location():
From dce207add43a6ece53a5a0a12458dd7edf03ed3e Mon Sep 17 00:00:00 2001
From: mateo-berri <277851410+mateo-berri@users.noreply.github.com>
Date: Wed, 19 Aug 2026 16:19:26 -0700
Subject: [PATCH 06/24] fix(proxy): strip client standard_logging_object and
zero-fill unknown recovered cost on the failure path
Auth and pass-through failures reach post_call_failure_hook with the raw request body unstripped, so a client-supplied standard_logging_object could feed the new attribution fallback when the logging object carries none. Pop the key before the lift so only the logging object may supply it. Also coalesce a None recovered cost to 0.0 so the lift always overwrites any client-supplied response_cost, matching the merge base's clobber semantics.
---
litellm/proxy/utils.py | 5 ++-
tests/test_litellm/proxy/test_proxy_utils.py | 34 ++++++++++++++++++++
2 files changed, 38 insertions(+), 1 deletion(-)
diff --git a/litellm/proxy/utils.py b/litellm/proxy/utils.py
index 58ac7882cb8..acafc600b80 100644
--- a/litellm/proxy/utils.py
+++ b/litellm/proxy/utils.py
@@ -539,7 +539,7 @@ def _failure_fields_to_lift(request_data: Mapping[str, object]) -> Mapping[str,
_entries: Final = (
("first_api_call_start_time", _first_handoff),
("combined_usage_object", None if _usage_to_lift is None else _usage_to_lift[0]),
- ("response_cost", None if _usage_to_lift is None else _usage_to_lift[1]),
+ ("response_cost", None if _usage_to_lift is None else (_usage_to_lift[1] or 0.0)),
("standard_logging_object", _model_call_details.get("standard_logging_object")),
)
return MappingProxyType({key: value for key, value in _entries if value is not None})
@@ -2322,6 +2322,9 @@ class ProxyLogging:
original_exception=original_exception,
)
+ # Auth and pass-through failures reach this hook with the raw request
+ # body unstripped, so only the logging object may supply this key.
+ request_data.pop("standard_logging_object", None)
request_data.update(_failure_fields_to_lift(request_data))
# Remove before callbacks iterate — not serialisable
diff --git a/tests/test_litellm/proxy/test_proxy_utils.py b/tests/test_litellm/proxy/test_proxy_utils.py
index c80130b44da..71af44d6682 100644
--- a/tests/test_litellm/proxy/test_proxy_utils.py
+++ b/tests/test_litellm/proxy/test_proxy_utils.py
@@ -468,6 +468,23 @@ class TestPostCallFailureHookLiftsRecoveredPartialSpend:
assert request_data["response_cost"] == 3.5e-05
assert "litellm_logging_obj" not in request_data
+ @pytest.mark.asyncio
+ async def test_recovered_usage_without_cost_clobbers_client_cost_with_zero(self):
+ from litellm.types.utils import Usage
+
+ recovered_usage = Usage(prompt_tokens=30, completion_tokens=1, total_tokens=31)
+ logging_obj = MagicMock()
+ logging_obj.model_call_details = {"combined_usage_object": recovered_usage}
+ request_data = {
+ "litellm_logging_obj": logging_obj,
+ "response_cost": 999.0,
+ "metadata": {},
+ }
+ await self._run(request_data)
+
+ assert request_data["combined_usage_object"] is recovered_usage
+ assert request_data["response_cost"] == 0.0
+
@pytest.mark.asyncio
async def test_no_recovered_usage_is_noop(self):
logging_obj = MagicMock()
@@ -522,6 +539,23 @@ class TestPostCallFailureHookLiftsStandardLoggingObject:
await self._run(request_data)
assert request_data["standard_logging_object"] is authoritative
+ @pytest.mark.asyncio
+ async def test_client_supplied_key_is_stripped_when_logging_obj_supplies_none(self):
+ spoofed = {"model_id": "client-injected"}
+ request_data = {"standard_logging_object": spoofed, "metadata": {}}
+ await self._run(request_data)
+ assert "standard_logging_object" not in request_data
+
+ logging_obj = MagicMock()
+ logging_obj.model_call_details = {}
+ request_data_with_obj = {
+ "litellm_logging_obj": logging_obj,
+ "standard_logging_object": spoofed,
+ "metadata": {},
+ }
+ await self._run(request_data_with_obj)
+ assert "standard_logging_object" not in request_data_with_obj
+
@pytest.mark.asyncio
async def test_no_standard_logging_object_is_noop(self):
logging_obj = MagicMock()
From 7b6f537855bf684a6b118ba1591c34475647b79a Mon Sep 17 00:00:00 2001
From: mateo-berri <277851410+mateo-berri@users.noreply.github.com>
Date: Wed, 19 Aug 2026 16:26:34 -0700
Subject: [PATCH 07/24] fix(anthropic_messages): price native /v1/messages
vertex calls on the deployment location
The proxy pre-creates the logging object before the router picks a deployment,
and the native /v1/messages handler never copied the deployment's
vertex_location into the logging params it updates, so cost resolution fell
back to the environment or the default region and priced every call on this
surface with the regional endpoint uplift. Copy the explicitly configured
location from the request's litellm params, the same source dispatch builds
the request URL from, and register the new regional_endpoint_uplift_multiplier
field in the cost map schema test.
---
litellm/llms/custom_httpx/llm_http_handler.py | 11 ++-
.../custom_httpx/test_llm_http_handler.py | 75 +++++++++++++++++++
tests/test_litellm/test_utils.py | 1 +
3 files changed, 86 insertions(+), 1 deletion(-)
diff --git a/litellm/llms/custom_httpx/llm_http_handler.py b/litellm/llms/custom_httpx/llm_http_handler.py
index 3ee803646a9..e03942100b0 100644
--- a/litellm/llms/custom_httpx/llm_http_handler.py
+++ b/litellm/llms/custom_httpx/llm_http_handler.py
@@ -5,7 +5,7 @@ import ssl
from collections.abc import AsyncIterator, Coroutine, Iterator, Mapping, Sequence
from contextlib import asynccontextmanager
from functools import lru_cache
-from types import ModuleType
+from types import MappingProxyType, ModuleType
from typing import TYPE_CHECKING, Any, Final, Literal, Optional, TypedDict, TypeVar, Union, cast, get_type_hints
from urllib.parse import parse_qs, urlencode, urlparse, urlunparse
@@ -2069,6 +2069,14 @@ class BaseLLMHTTPHandler:
if anthropic_messages_provider_config.should_filter_anthropic_beta_headers():
headers = update_headers_with_filtered_beta(headers=headers, provider=custom_llm_provider)
+ from litellm.llms.vertex_ai.vertex_llm_base import VertexBase
+
+ explicit_vertex_location: Final = VertexBase.explicit_vertex_ai_location(MappingProxyType(dict(litellm_params)))
+ vertex_location_params: Final = (
+ MappingProxyType({"vertex_location": explicit_vertex_location})
+ if explicit_vertex_location
+ else MappingProxyType({})
+ )
logging_obj.update_from_kwargs(
kwargs=kwargs,
model=model,
@@ -2077,6 +2085,7 @@ class BaseLLMHTTPHandler:
"preset_cache_key": None,
"stream_response": {},
"model_info": kwargs.get("model_info"),
+ **vertex_location_params,
**anthropic_messages_optional_request_params,
},
custom_llm_provider=custom_llm_provider,
diff --git a/tests/test_litellm/llms/custom_httpx/test_llm_http_handler.py b/tests/test_litellm/llms/custom_httpx/test_llm_http_handler.py
index 69f2312f203..9e9242137e6 100644
--- a/tests/test_litellm/llms/custom_httpx/test_llm_http_handler.py
+++ b/tests/test_litellm/llms/custom_httpx/test_llm_http_handler.py
@@ -2226,3 +2226,78 @@ def test_direct_vector_store_search_debug_log_omits_stored_credentials(caplog, i
logged = "\n".join(record.getMessage() for record in caplog.records)
assert "sup3r-s3cret-valkey-pw" not in logged
assert "sk-embedding-s3cret" not in logged
+
+
+@pytest.mark.asyncio
+async def test_async_anthropic_messages_handler_carries_deployment_vertex_location_for_pricing(monkeypatch):
+ """
+ The proxy pre-creates the logging object before the router picks a deployment, so the
+ native /v1/messages path must copy the deployment's vertex_location into the logging
+ params it updates; otherwise cost resolution falls back to the environment and every
+ call on this surface prices with the regional uplift (#34393).
+ """
+ import contextlib
+ from datetime import datetime
+
+ from litellm.litellm_core_utils.litellm_logging import (
+ Logging,
+ _resolve_vertex_location_for_cost,
+ )
+
+ monkeypatch.setenv("VERTEXAI_LOCATION", "us-east5")
+ monkeypatch.setattr(litellm, "vertex_location", None)
+
+ handler = BaseLLMHTTPHandler()
+
+ async def logging_obj_after_handler(generic_params):
+ logging_obj = Logging(
+ model="vertex_ai/claude-haiku-4-5@20251001",
+ messages=[{"role": "user", "content": "hi"}],
+ stream=False,
+ call_type="anthropic_messages",
+ start_time=datetime.now(),
+ litellm_call_id="vertex-messages-location",
+ function_id="f",
+ )
+ logging_obj.update_environment_variables(
+ model="vertex_ai/claude-haiku-4-5@20251001",
+ user="",
+ optional_params={},
+ litellm_params={"api_base": ""},
+ custom_llm_provider="vertex_ai",
+ )
+ mock_config = Mock()
+ mock_config.validate_anthropic_messages_environment = Mock(
+ return_value=({"authorization": "Bearer t"}, "https://us-east5-aiplatform.googleapis.com")
+ )
+ mock_config.transform_anthropic_messages_request = Mock(
+ return_value={"model": "claude-haiku-4-5@20251001", "messages": []}
+ )
+ with contextlib.suppress(Exception):
+ await handler.async_anthropic_messages_handler(
+ model="claude-haiku-4-5@20251001",
+ messages=[{"role": "user", "content": "hi"}],
+ anthropic_messages_provider_config=mock_config,
+ anthropic_messages_optional_request_params={"max_tokens": 10},
+ custom_llm_provider="vertex_ai",
+ litellm_params=generic_params,
+ logging_obj=logging_obj,
+ client=AsyncMock(),
+ kwargs={},
+ )
+ return logging_obj
+
+ global_deployment = await logging_obj_after_handler(GenericLiteLLMParams(vertex_location="global"))
+ assert global_deployment.litellm_params["vertex_location"] == "global"
+ assert (
+ _resolve_vertex_location_for_cost(
+ custom_llm_provider="vertex_ai",
+ litellm_params=global_deployment.litellm_params,
+ optional_params=global_deployment.optional_params,
+ model="claude-haiku-4-5@20251001",
+ )
+ == "global"
+ )
+
+ unconfigured_deployment = await logging_obj_after_handler(GenericLiteLLMParams())
+ assert "vertex_location" not in unconfigured_deployment.litellm_params
diff --git a/tests/test_litellm/test_utils.py b/tests/test_litellm/test_utils.py
index 1d9477432ee..60453931595 100644
--- a/tests/test_litellm/test_utils.py
+++ b/tests/test_litellm/test_utils.py
@@ -832,6 +832,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
"output_cost_per_token_above_200k_tokens_priority": {"type": "number"},
"output_cost_per_token_above_272k_tokens_priority": {"type": "number"},
"output_cost_per_token_above_272k_tokens_flex": {"type": "number"},
+ "regional_endpoint_uplift_multiplier": {"type": "number"},
"regional_processing_uplift_multiplier_eu": {"type": "number"},
"regional_processing_uplift_multiplier_us": {"type": "number"},
"input_cost_per_pixel": {"type": "number"},
From dc70c144d7d60bbbe09562fa3006feb2bc1c613b Mon Sep 17 00:00:00 2001
From: mateo
Date: Wed, 19 Aug 2026 23:35:54 +0000
Subject: [PATCH 08/24] chore(codeowners): require @mateo-berri approval for
the model prices jsons
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
---
.github/CODEOWNERS | 2 ++
1 file changed, 2 insertions(+)
diff --git a/.github/CODEOWNERS b/.github/CODEOWNERS
index 51d489459d9..118e5491939 100644
--- a/.github/CODEOWNERS
+++ b/.github/CODEOWNERS
@@ -1,3 +1,5 @@
/ui/ @yuneng-jiang @ryan-crabbe-berri
/litellm/proxy/_experimental/out/ @yuneng-jiang @ryan-crabbe-berri
/ui/litellm-dashboard/src/lib/http/schema.d.ts
+/model_prices_and_context_window.json @mateo-berri
+/litellm/model_prices_and_context_window_backup.json @mateo-berri
From c549cddada10190a3a593c6c92850fceaa10e009 Mon Sep 17 00:00:00 2001
From: mateo-berri <277851410+mateo-berri@users.noreply.github.com>
Date: Wed, 19 Aug 2026 16:44:52 -0700
Subject: [PATCH 09/24] fix(vertex_ai): price passthrough calls on the URL's
serving location
---
.../litellm_core_utils/llm_cost_calc/utils.py | 5 -
litellm/llms/vertex_ai/cost_calculator.py | 2 -
.../vertex_passthrough_logging_handler.py | 11 +++
.../test_llm_pass_through_endpoints.py | 92 +++++++++++++++++++
tests/test_litellm/test_cost_calculator.py | 26 ++++--
5 files changed, 119 insertions(+), 17 deletions(-)
diff --git a/litellm/litellm_core_utils/llm_cost_calc/utils.py b/litellm/litellm_core_utils/llm_cost_calc/utils.py
index dec35d16ea0..0793fe20b21 100644
--- a/litellm/litellm_core_utils/llm_cost_calc/utils.py
+++ b/litellm/litellm_core_utils/llm_cost_calc/utils.py
@@ -999,9 +999,6 @@ def generic_cost_per_token(
prompt_cost *= uplift
completion_cost *= uplift
- ## VERTEX REGIONAL-ENDPOINT UPLIFT
- # Applied as a flat multiplier across all token costs for the request
- # when the Vertex AI endpoint serving it is non-global.
vertex_uplift: Final = get_vertex_regional_endpoint_uplift(model_info, vertex_location)
if vertex_uplift != 1.0:
prompt_cost *= vertex_uplift
@@ -1109,8 +1106,6 @@ def get_token_type_cost_breakdown(
cache_read_cost *= uplift
cache_creation_cost *= uplift
- # Same flat uplift for Vertex AI non-global endpoints, keeping per-type
- # costs reconciled with the totals for regional Vertex deployments.
vertex_uplift: Final = get_vertex_regional_endpoint_uplift(model_info, vertex_location)
if vertex_uplift != 1.0:
reasoning_cost *= vertex_uplift
diff --git a/litellm/llms/vertex_ai/cost_calculator.py b/litellm/llms/vertex_ai/cost_calculator.py
index a9f5d77350c..23cb1e5b580 100644
--- a/litellm/llms/vertex_ai/cost_calculator.py
+++ b/litellm/llms/vertex_ai/cost_calculator.py
@@ -164,8 +164,6 @@ def cost_per_character(
usage=usage,
)
- # Applied once here; the cost_per_token fallbacks above are called without
- # vertex_location so the uplift can never compound.
vertex_uplift: Final = get_vertex_regional_endpoint_uplift(model_info, vertex_location)
return prompt_cost * vertex_uplift, completion_cost * vertex_uplift
diff --git a/litellm/proxy/pass_through_endpoints/llm_provider_handlers/vertex_passthrough_logging_handler.py b/litellm/proxy/pass_through_endpoints/llm_provider_handlers/vertex_passthrough_logging_handler.py
index 621b3ff9c83..f8e521410bf 100644
--- a/litellm/proxy/pass_through_endpoints/llm_provider_handlers/vertex_passthrough_logging_handler.py
+++ b/litellm/proxy/pass_through_endpoints/llm_provider_handlers/vertex_passthrough_logging_handler.py
@@ -10,6 +10,7 @@ import litellm
from litellm._logging import verbose_proxy_logger
from litellm.constants import VERTEX_BATCH_PREDICTION_JOBS_ROUTE
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
+from litellm.llms.vertex_ai.common_utils import get_vertex_location_from_url
from litellm.llms.vertex_ai.gemini.vertex_and_google_ai_studio_gemini import (
ModelResponseIterator as VertexModelResponseIterator,
)
@@ -60,6 +61,7 @@ class VertexPassthroughLoggingHandler:
request_body: dict | None = None,
**kwargs,
) -> PassThroughEndpointLoggingTypedDict:
+ vertex_location: Final = get_vertex_location_from_url(url_route)
if "predictLongRunning" in url_route:
model = VertexPassthroughLoggingHandler.extract_model_from_url(url_route)
@@ -82,6 +84,7 @@ class VertexPassthroughLoggingHandler:
model=model,
custom_llm_provider="vertex_ai",
call_type="create_video",
+ vertex_location=vertex_location,
)
# Set response_cost in _hidden_params to prevent recalculation
@@ -123,6 +126,7 @@ class VertexPassthroughLoggingHandler:
end_time=end_time,
logging_obj=logging_obj,
custom_llm_provider=VertexPassthroughLoggingHandler._get_custom_llm_provider_from_url(url_route),
+ vertex_location=vertex_location,
)
return {
@@ -190,6 +194,7 @@ class VertexPassthroughLoggingHandler:
end_time=end_time,
logging_obj=logging_obj,
custom_llm_provider="vertex_ai",
+ vertex_location=vertex_location,
)
return {
@@ -206,6 +211,7 @@ class VertexPassthroughLoggingHandler:
model="vertex_ai/search_api",
custom_llm_provider="vertex_ai",
call_type="vector_store_search",
+ vertex_location=vertex_location,
)
standard_pass_through_response_object: Final[StandardPassThroughResponseObject] = {
@@ -302,6 +308,7 @@ class VertexPassthroughLoggingHandler:
completion_response=litellm_prediction_response,
model=model,
custom_llm_provider="vertex_ai",
+ vertex_location=get_vertex_location_from_url(url_route),
)
kwargs["response_cost"] = response_cost
@@ -381,6 +388,7 @@ class VertexPassthroughLoggingHandler:
completion_response=litellm_embedding_response,
model=model,
custom_llm_provider=custom_llm_provider,
+ vertex_location=get_vertex_location_from_url(url_route),
)
kwargs["response_cost"] = response_cost
@@ -438,6 +446,7 @@ class VertexPassthroughLoggingHandler:
end_time=end_time,
logging_obj=litellm_logging_obj,
custom_llm_provider=VertexPassthroughLoggingHandler._get_custom_llm_provider_from_url(url_route),
+ vertex_location=get_vertex_location_from_url(url_route),
)
return {
@@ -591,6 +600,7 @@ class VertexPassthroughLoggingHandler:
end_time: datetime,
logging_obj: LiteLLMLoggingObj,
custom_llm_provider: str,
+ vertex_location: str | None,
) -> dict:
"""
Create the standard logging object for Vertex passthrough generateContent (streaming and non-streaming)
@@ -601,6 +611,7 @@ class VertexPassthroughLoggingHandler:
completion_response=litellm_model_response,
model=model,
custom_llm_provider="vertex_ai",
+ vertex_location=vertex_location,
)
kwargs["response_cost"] = response_cost
diff --git a/tests/test_litellm/proxy/pass_through_endpoints/test_llm_pass_through_endpoints.py b/tests/test_litellm/proxy/pass_through_endpoints/test_llm_pass_through_endpoints.py
index b56a8da7c66..8da6d0fe81d 100644
--- a/tests/test_litellm/proxy/pass_through_endpoints/test_llm_pass_through_endpoints.py
+++ b/tests/test_litellm/proxy/pass_through_endpoints/test_llm_pass_through_endpoints.py
@@ -1109,6 +1109,98 @@ class TestVertexAIPassThroughHandler:
assert result["kwargs"].get("model") == "gemini-embedding-2-preview"
mock_completion_cost.assert_called_once()
+ @pytest.mark.parametrize("streaming", [False, True])
+ def test_vertex_passthrough_handler_prices_regional_endpoint_with_uplift(self, monkeypatch, streaming):
+ """
+ Passthrough cost is computed inside the handler and stored as response_cost before the
+ logging cost resolver runs, so the handler itself must read the serving location out of
+ the passthrough URL; otherwise regional Vertex passthrough traffic bills at the global
+ rate (#34393).
+ """
+ import datetime
+
+ from litellm.proxy.pass_through_endpoints.llm_provider_handlers.vertex_passthrough_logging_handler import (
+ VertexPassthroughLoggingHandler,
+ )
+
+ monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
+ monkeypatch.setattr(
+ litellm,
+ "model_cost",
+ {
+ **litellm.get_model_cost_map(url=""),
+ "vertex_ai/gemini-fake-regional": {
+ "litellm_provider": "vertex_ai",
+ "mode": "chat",
+ "input_cost_per_token": 1e-06,
+ "output_cost_per_token": 2e-06,
+ "regional_endpoint_uplift_multiplier": 1.1,
+ },
+ },
+ )
+
+ response_body: Final = {
+ "candidates": [
+ {
+ "content": {"parts": [{"text": "hello"}], "role": "model"},
+ "finishReason": "STOP",
+ }
+ ],
+ "usageMetadata": {
+ "promptTokenCount": 10,
+ "candidatesTokenCount": 20,
+ "totalTokenCount": 30,
+ },
+ }
+
+ def cost_for(location: str) -> float:
+ url_route: Final = (
+ f"https://{location}-aiplatform.googleapis.com/v1/projects/p/locations/{location}"
+ "/publishers/google/models/gemini-fake-regional:"
+ f"{'streamGenerateContent' if streaming else 'generateContent'}"
+ )
+ mock_logging_obj: Final = Mock()
+ mock_logging_obj.litellm_call_id = "call-id"
+ mock_logging_obj.model_call_details = {}
+ mock_logging_obj.optional_params = {}
+ start_time: Final = datetime.datetime.now()
+ end_time: Final = datetime.datetime.now()
+ if streaming:
+ result = VertexPassthroughLoggingHandler._handle_logging_vertex_collected_chunks(
+ litellm_logging_obj=mock_logging_obj,
+ passthrough_success_handler_obj=Mock(),
+ url_route=url_route,
+ request_body={},
+ endpoint_type="vertex_ai",
+ start_time=start_time,
+ all_chunks=[json.dumps(response_body)],
+ model=None,
+ end_time=end_time,
+ )
+ else:
+ mock_httpx_response: Final = Mock()
+ mock_httpx_response.json.return_value = response_body
+ mock_httpx_response.headers = {}
+ mock_httpx_response.status_code = 200
+ result = VertexPassthroughLoggingHandler.vertex_passthrough_handler(
+ httpx_response=mock_httpx_response,
+ logging_obj=mock_logging_obj,
+ url_route=url_route,
+ result="test-result",
+ start_time=start_time,
+ end_time=end_time,
+ cache_hit=False,
+ )
+ return result["kwargs"]["response_cost"]
+
+ global_cost: Final = cost_for("global")
+ regional_cost: Final = cost_for("us-east5")
+
+ assert global_cost == pytest.approx(10 * 1e-06 + 20 * 2e-06, rel=1e-9)
+ assert regional_cost == pytest.approx(global_cost * 1.10, rel=1e-9), (
+ "regional Vertex passthrough traffic must bill at 1.1x the global rate"
+ )
+
class TestVertexAIDiscoveryPassThroughHandler:
"""
diff --git a/tests/test_litellm/test_cost_calculator.py b/tests/test_litellm/test_cost_calculator.py
index 6deaf5479e0..98938dee62e 100644
--- a/tests/test_litellm/test_cost_calculator.py
+++ b/tests/test_litellm/test_cost_calculator.py
@@ -1781,16 +1781,22 @@ def test_vertex_uplift_composes_with_above_128k_pricing(monkeypatch):
including the above-128k dynamic rates, so a synthetic model carrying both keys
prices regional above-128k usage at 1.1x the above-128k rate."""
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
- monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
- litellm.model_cost["vertex_ai/fake-regional-128k-model"] = {
- "litellm_provider": "vertex_ai",
- "mode": "chat",
- "input_cost_per_token": 1e-06,
- "output_cost_per_token": 2e-06,
- "input_cost_per_token_above_128k_tokens": 2e-06,
- "output_cost_per_token_above_128k_tokens": 4e-06,
- "regional_endpoint_uplift_multiplier": 1.1,
- }
+ monkeypatch.setattr(
+ litellm,
+ "model_cost",
+ {
+ **litellm.get_model_cost_map(url=""),
+ "vertex_ai/fake-regional-128k-model": {
+ "litellm_provider": "vertex_ai",
+ "mode": "chat",
+ "input_cost_per_token": 1e-06,
+ "output_cost_per_token": 2e-06,
+ "input_cost_per_token_above_128k_tokens": 2e-06,
+ "output_cost_per_token_above_128k_tokens": 4e-06,
+ "regional_endpoint_uplift_multiplier": 1.1,
+ },
+ },
+ )
usage = Usage(prompt_tokens=200_000, completion_tokens=10, total_tokens=200_010)
global_prompt, global_completion = cost_per_token(
From 919bf1a09785c020dd859605cb1074bfa6c42383 Mon Sep 17 00:00:00 2001
From: mateo-berri <277851410+mateo-berri@users.noreply.github.com>
Date: Wed, 19 Aug 2026 16:53:02 -0700
Subject: [PATCH 10/24] fix(proxy): strip client standard_logging_object before
the failure logging handler
---
litellm/proxy/utils.py | 8 +++--
tests/test_litellm/proxy/test_proxy_utils.py | 35 ++++++++++++++++++++
2 files changed, 40 insertions(+), 3 deletions(-)
diff --git a/litellm/proxy/utils.py b/litellm/proxy/utils.py
index acafc600b80..19dcd64c58d 100644
--- a/litellm/proxy/utils.py
+++ b/litellm/proxy/utils.py
@@ -2309,6 +2309,11 @@ class ProxyLogging:
)
)
+ # Auth and pass-through failure bodies are unstripped client input, and
+ # the logging handler below flattens body keys into model_call_details,
+ # so drop the key before it can masquerade as the built payload.
+ request_data.pop("standard_logging_object", None)
+
### LOGGING ###
if self._is_proxy_only_llm_api_error(
original_exception=original_exception,
@@ -2322,9 +2327,6 @@ class ProxyLogging:
original_exception=original_exception,
)
- # Auth and pass-through failures reach this hook with the raw request
- # body unstripped, so only the logging object may supply this key.
- request_data.pop("standard_logging_object", None)
request_data.update(_failure_fields_to_lift(request_data))
# Remove before callbacks iterate — not serialisable
diff --git a/tests/test_litellm/proxy/test_proxy_utils.py b/tests/test_litellm/proxy/test_proxy_utils.py
index 71af44d6682..07877514b69 100644
--- a/tests/test_litellm/proxy/test_proxy_utils.py
+++ b/tests/test_litellm/proxy/test_proxy_utils.py
@@ -556,6 +556,41 @@ class TestPostCallFailureHookLiftsStandardLoggingObject:
await self._run(request_data_with_obj)
assert "standard_logging_object" not in request_data_with_obj
+ @pytest.mark.asyncio
+ async def test_pass_through_failure_never_relifts_client_supplied_key(self):
+ from datetime import datetime
+ from unittest.mock import AsyncMock, patch
+
+ from fastapi import HTTPException
+
+ from litellm.litellm_core_utils.litellm_logging import Logging
+ from litellm.proxy._types import UserAPIKeyAuth
+
+ logging_obj = Logging(
+ model="claude-haiku-4-5",
+ messages=[{"role": "user", "content": "hi"}],
+ stream=False,
+ call_type="pass_through_endpoint",
+ start_time=datetime.now(),
+ litellm_call_id="test-call-id",
+ function_id="test-function-id",
+ )
+ request_data = {
+ "litellm_logging_obj": logging_obj,
+ "standard_logging_object": {"model_id": "client-injected"},
+ "metadata": {},
+ }
+ proxy_logging_obj = ProxyLogging(user_api_key_cache=DualCache())
+ proxy_logging_obj.alert_types = []
+ with patch.object(proxy_logging_obj, "update_request_status", new=AsyncMock()):
+ await proxy_logging_obj.post_call_failure_hook(
+ request_data=request_data,
+ original_exception=HTTPException(status_code=401, detail="unauthorized"),
+ user_api_key_dict=UserAPIKeyAuth(request_route="/v1/chat/completions"),
+ )
+ assert "standard_logging_object" not in request_data
+ assert "standard_logging_object" not in logging_obj.model_call_details
+
@pytest.mark.asyncio
async def test_no_standard_logging_object_is_noop(self):
logging_obj = MagicMock()
From f22eeb2ce05427083318c04a91e7281a56268249 Mon Sep 17 00:00:00 2001
From: Yassin Kortam
Date: Wed, 19 Aug 2026 17:00:26 -0700
Subject: [PATCH 11/24] fix(proxy): initialize the secret manager before
resolving os.environ config references (#37544)
`ProxyConfig.get_config()` walked the parsed config and replaced every
`os.environ/` string with `get_secret(value)` before anything initialized
the secret manager, so a key held only by the manager resolved to `None` and
that `None` was written back into the config. The later fallback in
`load_config` could not recover it, because the key now existed with a `None`
value.
Hoist the initialization into `get_config()`, ahead of the resolution pass, so
every entrypoint gets it: the CLI already did this itself, but the microservice
entrypoints (`gateway/main.py`, `backend/main.py`) uvicorn the app directly and
bypass the CLI. `load_config`'s own call is now redundant and is dropped, so
startup builds the manager once instead of building one and discarding it.
`get_config()` also runs on management-endpoint request paths, so this returns
early once a manager exists rather than rebuilding the client per request.
Also warn when a reference the manager would have been asked for resolves to
`None`. The reporter had no log line at all to work from. `get_secret` only
reaches the manager when reads are enabled and the name is in `hosted_keys`, so
`secret_manager_would_be_consulted` mirrors that gate and keeps the warning off
env-only references, which are expected rather than an error.
---
litellm/proxy/proxy_server.py | 62 +++++-
litellm/proxy/read_model_list.py | 3 +-
litellm/secret_managers/main.py | 28 ++-
.../proxy/proxy_server/test_proxy_config.py | 208 ++++++++++++++++++
.../test_secret_managers_main.py | 69 +++++-
5 files changed, 352 insertions(+), 18 deletions(-)
diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py
index 5adee476f33..0f39ef2e18e 100644
--- a/litellm/proxy/proxy_server.py
+++ b/litellm/proxy/proxy_server.py
@@ -639,6 +639,7 @@ from litellm.secret_managers.main import (
get_secret_bool,
get_secret_str,
normalize_nonempty_secret_str,
+ secret_manager_would_be_consulted,
str_to_bool,
)
from litellm.types.integrations.slack_alerting import AlertType, SlackAlertingArgs
@@ -4380,9 +4381,55 @@ class ProxyConfig:
item = self._check_for_os_environ_vars(config=item, depth=depth + 1, max_depth=max_depth)
# if the value is a string and starts with "os.environ/" - then it's an environment variable
elif isinstance(value, str) and value.startswith("os.environ/"):
- config[key] = get_secret(value)
+ resolved = get_secret(value)
+ if resolved is None and secret_manager_would_be_consulted(value):
+ verbose_proxy_logger.warning("%s is absent from the configured secret manager", value)
+ config[key] = resolved
return config
+ def _initialize_secret_manager_from_raw_config(
+ self, config: Mapping[str, object], config_file_path: str | None
+ ) -> None:
+ """
+ Bring the secret manager up before `os.environ/` references are resolved.
+
+ `_check_for_os_environ_vars` writes whatever it resolves back into the config, so a key
+ held only by the secret manager would otherwise become a permanent `None` that the later
+ fallbacks in `load_config` can no longer recover from.
+
+ `get_config` also runs on management-endpoint request paths, so this returns early once a
+ manager exists rather than rebuilding the client on every request.
+
+ The manager's own settings can only come from real environment variables, so they are
+ resolved against a throwaway copy and the config is left untouched for the main pass.
+ """
+ if litellm.secret_manager_client is not None:
+ return
+
+ general_settings: Final = config.get("general_settings")
+ if not isinstance(general_settings, dict):
+ return
+
+ raw_system: Final = general_settings.get("key_management_system")
+ key_management_system: Final = (
+ get_secret(raw_system)
+ if isinstance(raw_system, str) and raw_system.startswith("os.environ/")
+ else raw_system
+ )
+ if not isinstance(key_management_system, str):
+ return
+
+ raw_settings: Final = general_settings.get("key_management_settings")
+ if isinstance(raw_settings, dict):
+ litellm._key_management_settings = KeyManagementSettings(
+ **self._check_for_os_environ_vars(config=copy.deepcopy(raw_settings))
+ )
+
+ self.initialize_secret_manager(
+ key_management_system=key_management_system,
+ config_file_path=config_file_path,
+ )
+
def _get_team_config(self, team_id: str, all_teams_config: list[dict]) -> dict:
team_config: dict = {}
for team in all_teams_config:
@@ -4553,6 +4600,8 @@ class ProxyConfig:
printed_yaml: Final = copy.deepcopy(config)
printed_yaml.pop("environment_variables", None)
+ self._initialize_secret_manager_from_raw_config(config=config, config_file_path=config_file_path)
+
config = self._check_for_os_environ_vars(config=config)
self.update_config_state(config=config)
@@ -5123,17 +5172,14 @@ class ProxyConfig:
key: general_settings[key] for key in SPEND_LOG_CLEANUP_BOUND_SETTINGS if key in general_settings
}
- ### LOAD KEY MANAGEMENT SETTINGS FIRST (needed for custom secret manager) ###
+ ### LOAD KEY MANAGEMENT SETTINGS ###
+ # The secret manager itself is brought up by get_config(), which runs before the
+ # `os.environ/` references in this config were resolved. Re-reading the settings here
+ # picks up any of them that were themselves secret-manager backed.
key_management_settings: Final = general_settings.get("key_management_settings", None)
if key_management_settings is not None:
litellm._key_management_settings = KeyManagementSettings(**key_management_settings)
- ### LOAD SECRET MANAGER ###
- key_management_system: Final = general_settings.get("key_management_system", None)
- self.initialize_secret_manager(
- key_management_system=key_management_system,
- config_file_path=config_file_path,
- )
### [DEPRECATED] LOAD FROM GOOGLE KMS ### old way of loading from google kms
use_google_kms: Final = general_settings.get("use_google_kms", False)
load_google_kms(use_google_kms=use_google_kms)
diff --git a/litellm/proxy/read_model_list.py b/litellm/proxy/read_model_list.py
index cdd6680aa40..a1830e7f2bc 100644
--- a/litellm/proxy/read_model_list.py
+++ b/litellm/proxy/read_model_list.py
@@ -9,7 +9,8 @@ effects.
Instead we reuse ``ProxyConfig.get_config`` — the actual config reader — so the
gateway inherits the same heavy lifting the proxy does: ``include:`` merging,
``os.environ/`` + secret-manager resolution, and DB-stored models (when a DB is
-configured). It has no proxy-setup side effects. Returns the resolved
+configured). Its only proxy-setup side effect is bringing up the configured
+secret manager, which is what makes that resolution work. Returns the resolved
``model_list``; the Rust side deserializes each entry into its ``Deployment``.
"""
diff --git a/litellm/secret_managers/main.py b/litellm/secret_managers/main.py
index d1e4b3bb2ce..e89fbbdab65 100644
--- a/litellm/secret_managers/main.py
+++ b/litellm/secret_managers/main.py
@@ -365,6 +365,22 @@ def get_secret(
raise e
+def secret_manager_would_be_consulted(secret_name: str) -> bool:
+ """
+ Returns True if a `get_secret` read for `secret_name` would actually reach the hosted manager.
+
+ Mirrors the gating `get_secret` applies below: the manager has to be up and readable, and
+ `hosted_keys`, when set, is an allowlist of the names it is consulted for. Callers use this to
+ tell "the manager does not have this key" apart from "the manager was never asked".
+ """
+ if not _should_read_secret_from_secret_manager():
+ return False
+ key_management_settings: Final = litellm._key_management_settings
+ if key_management_settings is None or key_management_settings.hosted_keys is None:
+ return True
+ return secret_name.removeprefix("os.environ/") in key_management_settings.hosted_keys
+
+
def _should_read_secret_from_secret_manager() -> bool:
"""
Returns True if the secret manager should be used to read the secret, False otherwise
@@ -373,11 +389,7 @@ def _should_read_secret_from_secret_manager() -> bool:
- If the `_key_management_settings` access mode is "read_only" or "read_and_write", return True
- Otherwise, return False
"""
- if litellm.secret_manager_client is not None:
- if litellm._key_management_settings is not None:
- if (
- litellm._key_management_settings.access_mode == "read_only"
- or litellm._key_management_settings.access_mode == "read_and_write"
- ):
- return True
- return False
+ key_management_settings: Final = litellm._key_management_settings
+ if litellm.secret_manager_client is None or key_management_settings is None:
+ return False
+ return key_management_settings.access_mode in ("read_only", "read_and_write")
diff --git a/tests/test_litellm/proxy/proxy_server/test_proxy_config.py b/tests/test_litellm/proxy/proxy_server/test_proxy_config.py
index 37a03617a65..58465772b3b 100644
--- a/tests/test_litellm/proxy/proxy_server/test_proxy_config.py
+++ b/tests/test_litellm/proxy/proxy_server/test_proxy_config.py
@@ -769,6 +769,214 @@ async def test_ProxyConfig_get_config_missing_file_raises(monkeypatch):
await pc.get_config(config_file_path="/no/such/path.yaml")
+# ---------------------------------------------------------------------------
+# ProxyConfig._initialize_secret_manager_from_raw_config
+# ---------------------------------------------------------------------------
+
+VAULT_SECRET_MANAGER_MODULE = '''
+import os
+
+from litellm.integrations.custom_secret_manager import CustomSecretManager
+
+VAULT = {"LITELLM_MASTER_KEY": "master-from-vault", "MY_PROVIDER_KEY": "provider-from-vault"}
+
+
+class VaultSecretManager(CustomSecretManager):
+ def __init__(self):
+ super().__init__()
+ # The loader re-executes this module on every construction, so an in-module counter
+ # would reset. Append to a file instead, to count constructions across the whole load.
+ with open(os.environ["VAULT_CONSTRUCTION_LOG"], "a") as f:
+ f.write("constructed\\n")
+
+ def sync_read_secret(self, secret_name, optional_params=None, timeout=None, **kwargs):
+ return VAULT.get(secret_name)
+
+ async def async_read_secret(self, secret_name, optional_params=None, timeout=None, **kwargs):
+ return VAULT.get(secret_name)
+'''
+
+VAULT_BACKED_CONFIG = """
+model_list:
+ - model_name: my-model
+ litellm_params:
+ model: openai/gpt-4o-mini
+ api_key: os.environ/MY_PROVIDER_KEY
+
+general_settings:
+ master_key: os.environ/LITELLM_MASTER_KEY
+ key_management_system: custom
+ key_management_settings:
+ custom_secret_manager: vault_secret_manager.VaultSecretManager
+ hosted_keys:
+ - LITELLM_MASTER_KEY
+ - MY_PROVIDER_KEY
+"""
+
+
+def _write_vault_backed_config(tmp_path, monkeypatch, config_yaml: str) -> str:
+ """Write a config whose secrets live only in a custom secret manager, never in the env."""
+ (tmp_path / "vault_secret_manager.py").write_text(VAULT_SECRET_MANAGER_MODULE)
+ config_file = tmp_path / "c.yaml"
+ config_file.write_text(config_yaml)
+ monkeypatch.setattr("litellm.proxy.proxy_server.prisma_client", None)
+ monkeypatch.setattr("litellm.proxy.proxy_server.store_model_in_db", False)
+ monkeypatch.delenv("LITELLM_CONFIG_BUCKET_NAME", raising=False)
+ monkeypatch.delenv("LITELLM_MASTER_KEY", raising=False)
+ monkeypatch.delenv("MY_PROVIDER_KEY", raising=False)
+ monkeypatch.setenv("VAULT_CONSTRUCTION_LOG", str(tmp_path / "constructions.log"))
+ monkeypatch.setattr(litellm, "secret_manager_client", None)
+ return str(config_file)
+
+
+def _construction_count(tmp_path) -> int:
+ log = tmp_path / "constructions.log"
+ return len(log.read_text().splitlines()) if log.exists() else 0
+
+
+@pytest.mark.asyncio
+async def test_ProxyConfig_get_config_resolves_keys_held_only_by_the_secret_manager(tmp_path, monkeypatch):
+ """Regression for GH #35239.
+
+ get_config() used to resolve every ``os.environ/`` reference and write the result
+ back into the config before the secret manager was initialized, so any key that lived
+ only in the manager became a permanent ``None``.
+ """
+ config_file_path = _write_vault_backed_config(tmp_path, monkeypatch, VAULT_BACKED_CONFIG)
+
+ cfg = await ProxyConfig().get_config(config_file_path=config_file_path)
+
+ assert {
+ "master_key": cfg["general_settings"]["master_key"],
+ "api_key": cfg["model_list"][0]["litellm_params"]["api_key"],
+ "hosted_keys": litellm._key_management_settings.hosted_keys,
+ } == {
+ "master_key": "master-from-vault",
+ "api_key": "provider-from-vault",
+ "hosted_keys": ["LITELLM_MASTER_KEY", "MY_PROVIDER_KEY"],
+ }
+
+
+@pytest.mark.asyncio
+async def test_ProxyConfig_load_config_builds_the_secret_manager_exactly_once(tmp_path, monkeypatch):
+ """The full startup path must not build the manager, then throw it away and build another.
+
+ A discarded client costs a Vault/CyberArk re-auth and leaks a gRPC channel on Google KMS.
+ """
+ config_file_path = _write_vault_backed_config(tmp_path, monkeypatch, VAULT_BACKED_CONFIG)
+
+ _router, _model_list, general_settings = await ProxyConfig().load_config(
+ router=None, config_file_path=config_file_path
+ )
+
+ assert {
+ "constructions": _construction_count(tmp_path),
+ "master_key": general_settings["master_key"],
+ } == {"constructions": 1, "master_key": "master-from-vault"}
+
+
+@pytest.mark.asyncio
+async def test_ProxyConfig_get_config_reuses_an_already_initialized_secret_manager(tmp_path, monkeypatch):
+ """get_config() also runs on management-endpoint request paths.
+
+ Rebuilding the client on every call would re-execute the custom manager module, drop the
+ Vault/CyberArk token caches, and leak a gRPC channel per request on Google KMS.
+ """
+ config_file_path = _write_vault_backed_config(tmp_path, monkeypatch, VAULT_BACKED_CONFIG)
+
+ await ProxyConfig().get_config(config_file_path=config_file_path)
+ first_client = litellm.secret_manager_client
+ second = await ProxyConfig().get_config(config_file_path=config_file_path)
+
+ assert {
+ "client_reused": litellm.secret_manager_client is first_client,
+ "master_key": second["general_settings"]["master_key"],
+ } == {"client_reused": True, "master_key": "master-from-vault"}
+
+
+@pytest.mark.asyncio
+async def test_ProxyConfig_get_config_without_key_management_system_leaves_secret_manager_unset(
+ tmp_path, monkeypatch
+):
+ """No ``key_management_system`` means no manager, an unresolvable reference stays None, and
+ nothing is warned about: with no manager there is nothing to have been absent from."""
+ config_yaml = VAULT_BACKED_CONFIG.replace(" key_management_system: custom\n", "")
+ config_file_path = _write_vault_backed_config(tmp_path, monkeypatch, config_yaml)
+ warn = MagicMock()
+ monkeypatch.setattr("litellm.proxy.proxy_server.verbose_proxy_logger.warning", warn)
+
+ cfg = await ProxyConfig().get_config(config_file_path=config_file_path)
+
+ assert {
+ "master_key": cfg["general_settings"]["master_key"],
+ "api_key": cfg["model_list"][0]["litellm_params"]["api_key"],
+ "client": litellm.secret_manager_client,
+ "warned_about": [call.args[1] for call in warn.call_args_list],
+ } == {"master_key": None, "api_key": None, "client": None, "warned_about": []}
+
+
+@pytest.mark.asyncio
+async def test_ProxyConfig_get_config_warns_when_a_reference_is_missing_from_the_secret_manager(
+ tmp_path, monkeypatch
+):
+ """A reference the manager cannot resolve is logged, instead of silently becoming None."""
+ config_yaml = VAULT_BACKED_CONFIG.replace("MY_PROVIDER_KEY", "NOT_IN_VAULT")
+ config_file_path = _write_vault_backed_config(tmp_path, monkeypatch, config_yaml)
+ warn = MagicMock()
+ monkeypatch.setattr("litellm.proxy.proxy_server.verbose_proxy_logger.warning", warn)
+
+ cfg = await ProxyConfig().get_config(config_file_path=config_file_path)
+
+ assert {
+ "api_key": cfg["model_list"][0]["litellm_params"]["api_key"],
+ "warned_about": [call.args[1] for call in warn.call_args_list],
+ } == {"api_key": None, "warned_about": ["os.environ/NOT_IN_VAULT"]}
+
+
+@pytest.mark.asyncio
+async def test_ProxyConfig_get_config_does_not_warn_for_a_name_outside_hosted_keys(tmp_path, monkeypatch):
+ """``hosted_keys`` is an allowlist, so a name outside it is never looked up in the manager.
+
+ Warning about it would claim a lookup that never happened, on every optional env-only
+ reference, on every config reload.
+ """
+ config_yaml = VAULT_BACKED_CONFIG.replace("api_key: os.environ/MY_PROVIDER_KEY", "api_key: os.environ/ENV_ONLY")
+ config_file_path = _write_vault_backed_config(tmp_path, monkeypatch, config_yaml)
+ warn = MagicMock()
+ monkeypatch.setattr("litellm.proxy.proxy_server.verbose_proxy_logger.warning", warn)
+
+ cfg = await ProxyConfig().get_config(config_file_path=config_file_path)
+
+ assert {
+ "api_key": cfg["model_list"][0]["litellm_params"]["api_key"],
+ "client_is_up": litellm.secret_manager_client is not None,
+ "warned_about": [call.args[1] for call in warn.call_args_list],
+ } == {"api_key": None, "client_is_up": True, "warned_about": []}
+
+
+@pytest.mark.asyncio
+async def test_ProxyConfig_get_config_does_not_warn_under_write_only_access_mode(tmp_path, monkeypatch):
+ """``write_only`` means reads never reach the manager, so an absent name is not its fault.
+
+ That mode exists so the manager can store virtual keys while config secrets stay in the
+ environment, which makes env-only references the expected state rather than an error.
+ """
+ config_yaml = VAULT_BACKED_CONFIG.replace(
+ " key_management_settings:\n", " key_management_settings:\n access_mode: write_only\n"
+ )
+ config_file_path = _write_vault_backed_config(tmp_path, monkeypatch, config_yaml)
+ warn = MagicMock()
+ monkeypatch.setattr("litellm.proxy.proxy_server.verbose_proxy_logger.warning", warn)
+
+ cfg = await ProxyConfig().get_config(config_file_path=config_file_path)
+
+ assert {
+ "master_key": cfg["general_settings"]["master_key"],
+ "client_is_up": litellm.secret_manager_client is not None,
+ "warned_about": [call.args[1] for call in warn.call_args_list],
+ } == {"master_key": None, "client_is_up": True, "warned_about": []}
+
+
# ---------------------------------------------------------------------------
# ProxyConfig.update_config_state / get_config_state
# ---------------------------------------------------------------------------
diff --git a/tests/test_litellm/secret_managers/test_secret_managers_main.py b/tests/test_litellm/secret_managers/test_secret_managers_main.py
index 3631f640136..acc91b691d4 100644
--- a/tests/test_litellm/secret_managers/test_secret_managers_main.py
+++ b/tests/test_litellm/secret_managers/test_secret_managers_main.py
@@ -7,7 +7,14 @@ from unittest.mock import Mock, patch
import pytest
-from litellm.secret_managers.main import get_secret, normalize_nonempty_secret_str
+import litellm
+from litellm.integrations.custom_secret_manager import CustomSecretManager
+from litellm.secret_managers.main import (
+ get_secret,
+ normalize_nonempty_secret_str,
+ secret_manager_would_be_consulted,
+)
+from litellm.types.secret_managers.main import KeyManagementSettings, KeyManagementSystem
# Set up logging for debugging
logging.basicConfig(level=logging.DEBUG)
@@ -364,3 +371,63 @@ def test_unsupported_oidc_provider():
)
def test_normalize_nonempty_secret_str(raw, expected):
assert normalize_nonempty_secret_str(raw) == expected
+
+
+class _SpySecretManager(CustomSecretManager):
+ """Records every name the manager is actually asked for."""
+
+ def __init__(self, asked):
+ self.asked = asked
+
+ def sync_read_secret(self, secret_name, optional_params=None, timeout=None, **kwargs):
+ self.asked.append(secret_name)
+ return "a-value"
+
+ async def async_read_secret(self, secret_name, optional_params=None, timeout=None, **kwargs):
+ self.asked.append(secret_name)
+ return "a-value"
+
+
+@pytest.mark.parametrize(
+ ("access_mode", "hosted_keys", "secret_name", "expected"),
+ [
+ ("read_only", None, "ANY_NAME", True),
+ ("read_only", ["ALLOWED"], "ALLOWED", True),
+ ("read_only", ["ALLOWED"], "NOT_ALLOWED", False),
+ ("read_and_write", ["ALLOWED"], "ALLOWED", True),
+ ("write_only", None, "ANY_NAME", False),
+ ("write_only", ["ALLOWED"], "ALLOWED", False),
+ ],
+)
+def test_secret_manager_would_be_consulted_matches_get_secret(
+ monkeypatch, access_mode, hosted_keys, secret_name, expected
+):
+ """The predicate must agree with what get_secret actually does, not with a reading of it.
+
+ Callers use it to tell "the manager does not have this key" apart from "the manager was
+ never asked", so a predicate that drifts from get_secret's gating makes them state a
+ lookup that never happened.
+ """
+ asked = []
+ monkeypatch.setattr(litellm, "secret_manager_client", _SpySecretManager(asked))
+ monkeypatch.setattr(litellm, "_key_management_system", KeyManagementSystem.CUSTOM)
+ monkeypatch.setattr(
+ litellm,
+ "_key_management_settings",
+ KeyManagementSettings(access_mode=access_mode, hosted_keys=hosted_keys),
+ )
+ monkeypatch.delenv(secret_name, raising=False)
+
+ predicted = secret_manager_would_be_consulted(f"os.environ/{secret_name}")
+ get_secret(f"os.environ/{secret_name}")
+
+ assert {"predicted": predicted, "actually_consulted": bool(asked)} == {
+ "predicted": expected,
+ "actually_consulted": expected,
+ }
+
+
+def test_secret_manager_would_be_consulted_is_false_without_a_client(monkeypatch):
+ monkeypatch.setattr(litellm, "secret_manager_client", None)
+
+ assert secret_manager_would_be_consulted("os.environ/ANY_NAME") is False
From 629d7683f2be588ab69af2e6832e310b3c38b33e Mon Sep 17 00:00:00 2001
From: ryan-crabbe-berri
Date: Wed, 19 Aug 2026 17:04:14 -0700
Subject: [PATCH 12/24] refactor(ui): swap @ant-design/icons for lucide-react
(#37553)
The dashboard drew its icons from two libraries at once: lucide-react,
which shadcn/ui ships with, and @ant-design/icons, left over from antd.
This moves the last 39 files onto lucide and drops the dependency, so
the icon set matches the component library everywhere.
antd icons sized themselves from the inherited font-size and rendered as
role="img" with an aria-label, neither of which a lucide svg does, so the
swap carries explicit size classes and gives the two icon-only plugin
buttons real accessible names.
---
ui/litellm-dashboard/eslint.config.mjs | 4 ++
ui/litellm-dashboard/package-lock.json | 1 -
ui/litellm-dashboard/package.json | 1 -
.../agents/_components/add_agent_form.tsx | 16 ++---
.../_components/GuardrailConfig.tsx | 18 ++----
.../guardrail_garden_card.test.tsx | 4 --
.../_components/guardrail_info.test.tsx | 2 +-
.../guardrails/_components/guardrail_info.tsx | 7 +--
.../_components/AwsSigV4Fields.tsx | 4 +-
.../_components/CreateMCPServer.tsx | 10 +--
.../_components/DcrBridgeToggle.tsx | 4 +-
.../_components/EnvVarsSection.tsx | 12 ++--
.../_components/IdJagFormFields.tsx | 4 +-
.../_components/MCPPermissionManagement.tsx | 23 ++++---
.../_components/OAuthFormFields.tsx | 4 +-
.../_components/OpenAPIFormSection.tsx | 4 +-
.../_components/OpenApiByokFields.tsx | 12 ++--
.../_components/StdioConfiguration.tsx | 4 +-
.../TokenEndpointAuthMethodField.tsx | 4 +-
.../_components/TokenExchangeFormFields.tsx | 4 +-
.../_components/mcp_server_edit.tsx | 23 ++++---
.../LoggingSettings/LoggingSettings.tsx | 5 +-
.../PluginSettings.integration.test.tsx | 6 +-
.../PluginSettings/PluginSettings.tsx | 20 +++---
.../VirtualKeysPage/keyTableColumns.tsx | 4 +-
.../add_model/ClassificationMethodConfig.tsx | 8 +--
.../add_model/ComplexityRouterConfig.tsx | 8 +--
.../add_model/EscalationKeywords.tsx | 4 +-
.../components/add_model/KeywordTierRules.tsx | 8 +--
.../add_model/SemanticKeywordMatching.tsx | 4 +-
.../add_model/advanced_settings.tsx | 7 +--
.../add_model/provider_specific_fields.tsx | 4 +-
.../claude_code_plugins/skill_detail.tsx | 10 +--
.../common_components/AccessGroupSelector.tsx | 6 +-
.../check_openapi_schema.tsx | 4 +-
.../common_components/user_search_modal.tsx | 5 +-
.../mcp_tools/ByokCredentialModal.tsx | 28 +++------
.../src/components/model_info_view.tsx | 5 +-
.../organisms/RegenerateKeyModal.tsx | 7 +--
.../organisms/create_key_button.tsx | 61 +++++++++----------
.../src/components/routing_groups/index.tsx | 8 +--
.../shared/advanced_date_picker.test.tsx | 6 +-
.../shared/advanced_date_picker.tsx | 6 +-
.../src/components/team/LoggingSettings.tsx | 7 +--
.../src/components/team/TeamInfo.tsx | 17 +++---
45 files changed, 195 insertions(+), 218 deletions(-)
diff --git a/ui/litellm-dashboard/eslint.config.mjs b/ui/litellm-dashboard/eslint.config.mjs
index 35603123736..3cbb93f9a48 100644
--- a/ui/litellm-dashboard/eslint.config.mjs
+++ b/ui/litellm-dashboard/eslint.config.mjs
@@ -62,6 +62,10 @@ const eslintConfig = [
message:
"antd is being phased out; build new UI with shadcn/ui primitives instead of adding antd imports.",
},
+ {
+ group: ["@ant-design/icons", "@ant-design/icons/*"],
+ message: "@ant-design/icons is gone from the dashboard; use lucide-react instead.",
+ },
],
},
],
diff --git a/ui/litellm-dashboard/package-lock.json b/ui/litellm-dashboard/package-lock.json
index 7069450a75c..154a19da7f3 100644
--- a/ui/litellm-dashboard/package-lock.json
+++ b/ui/litellm-dashboard/package-lock.json
@@ -9,7 +9,6 @@
"version": "0.1.0",
"dependencies": {
"@ant-design/cssinjs": "1.24.0",
- "@ant-design/icons": "5.6.1",
"@anthropic-ai/sdk": "0.92.0",
"@base-ui/react": "^1.6.0",
"@headlessui/tailwindcss": "0.2.2",
diff --git a/ui/litellm-dashboard/package.json b/ui/litellm-dashboard/package.json
index 962cabcba7b..3f9c29d2afe 100644
--- a/ui/litellm-dashboard/package.json
+++ b/ui/litellm-dashboard/package.json
@@ -25,7 +25,6 @@
},
"dependencies": {
"@ant-design/cssinjs": "1.24.0",
- "@ant-design/icons": "5.6.1",
"@anthropic-ai/sdk": "0.92.0",
"@base-ui/react": "^1.6.0",
"@headlessui/tailwindcss": "0.2.2",
diff --git a/ui/litellm-dashboard/src/app/(dashboard)/agents/_components/add_agent_form.tsx b/ui/litellm-dashboard/src/app/(dashboard)/agents/_components/add_agent_form.tsx
index 22471ee66f9..09737c3151b 100644
--- a/ui/litellm-dashboard/src/app/(dashboard)/agents/_components/add_agent_form.tsx
+++ b/ui/litellm-dashboard/src/app/(dashboard)/agents/_components/add_agent_form.tsx
@@ -3,7 +3,7 @@ import { Select, Steps, Tag } from "antd";
import { FormProvider, useForm, useWatch } from "react-hook-form";
import { toast } from "@/lib/toast";
import { Logo } from "@/components/molecules/logo/Logo";
-import { CheckCircleFilled, KeyOutlined, RobotOutlined, AppstoreOutlined } from "@ant-design/icons";
+import { Bot, CircleCheck, Key, LayoutGrid } from "lucide-react";
import CreatedKeyDisplay from "@/components/shared/CreatedKeyDisplay";
import { Button } from "@/components/ui/button";
import { UiLoadingSpinner } from "@/components/ui/ui-loading-spinner";
@@ -707,7 +707,7 @@ const AddAgentForm: React.FC = ({ visible, onClose, accessTok
}`}
onClick={() => handleAgentTypeChange(CUSTOM_AGENT_TYPE)}
>
-
+
Custom / Other
@@ -846,7 +846,8 @@ const AddAgentForm: React.FC
= ({ visible, onClose, accessTok
{/* Agent name chip */}
- } color="purple" className="px-3 py-1 text-sm">
+
+
{agentName}
@@ -884,7 +885,7 @@ const AddAgentForm: React.FC
= ({ visible, onClose, accessTok
-
+
Create a new key for this agent
A dedicated key scoped to this agent.
@@ -920,7 +921,7 @@ const AddAgentForm: React.FC
= ({ visible, onClose, accessTok
-
+
Assign an existing key
Re-assign a key you already have to this agent.
@@ -958,10 +959,11 @@ const AddAgentForm: React.FC
= ({ visible, onClose, accessTok
const renderReadyStep = () => (
-
+
Agent Created!
- } color="purple" className="px-3 py-1 text-sm">
+
+
{createdAgentName}
diff --git a/ui/litellm-dashboard/src/app/(dashboard)/guardrails-monitor/_components/GuardrailConfig.tsx b/ui/litellm-dashboard/src/app/(dashboard)/guardrails-monitor/_components/GuardrailConfig.tsx
index 619ccd8d974..271de78c272 100644
--- a/ui/litellm-dashboard/src/app/(dashboard)/guardrails-monitor/_components/GuardrailConfig.tsx
+++ b/ui/litellm-dashboard/src/app/(dashboard)/guardrails-monitor/_components/GuardrailConfig.tsx
@@ -1,10 +1,4 @@
-import {
- CheckCircleOutlined,
- CodeOutlined,
- PlayCircleOutlined,
- RollbackOutlined,
- SaveOutlined,
-} from "@ant-design/icons";
+import { CircleCheck, CirclePlay, Code, Save, Undo2 } from "lucide-react";
import { Select, SelectContent, SelectItem, SelectTrigger, SelectValue } from "@/components/ui/select";
import { Button } from "@/components/ui/button";
import { Input } from "@/components/ui/input";
@@ -100,11 +94,11 @@ export function GuardrailConfig({ guardrailName, guardrailType, provider }: Guar
@@ -214,7 +208,7 @@ export function GuardrailConfig({ guardrailName, guardrailType, provider }: Guar
-
+
Custom Code Override
Replace the built-in guardrail with custom evaluation code
@@ -247,13 +241,13 @@ export function GuardrailConfig({ guardrailName, guardrailType, provider }: Guar
{rerunStatus === "success" && (
- 7/10 would now pass with new config
+ 7/10 would now pass with new config
)}
diff --git a/ui/litellm-dashboard/src/app/(dashboard)/guardrails/_components/guardrail_garden_card.test.tsx b/ui/litellm-dashboard/src/app/(dashboard)/guardrails/_components/guardrail_garden_card.test.tsx
index d81eaa5a4fc..25f63ca2902 100644
--- a/ui/litellm-dashboard/src/app/(dashboard)/guardrails/_components/guardrail_garden_card.test.tsx
+++ b/ui/litellm-dashboard/src/app/(dashboard)/guardrails/_components/guardrail_garden_card.test.tsx
@@ -4,10 +4,6 @@ import userEvent from "@testing-library/user-event";
import GuardrailCard from "./guardrail_garden_card";
import type { GuardrailCardInfo } from "./guardrail_garden_data";
-vi.mock("@ant-design/icons", () => ({
- CheckCircleFilled: ({ style, ...props }: any) =>
,
-}));
-
const baseCard: GuardrailCardInfo = {
id: "test-guard",
name: "Test Guardrail",
diff --git a/ui/litellm-dashboard/src/app/(dashboard)/guardrails/_components/guardrail_info.test.tsx b/ui/litellm-dashboard/src/app/(dashboard)/guardrails/_components/guardrail_info.test.tsx
index b6ee130d50a..4f1ac3e7d7a 100644
--- a/ui/litellm-dashboard/src/app/(dashboard)/guardrails/_components/guardrail_info.test.tsx
+++ b/ui/litellm-dashboard/src/app/(dashboard)/guardrails/_components/guardrail_info.test.tsx
@@ -153,7 +153,7 @@ describe("Guardrail Info", () => {
expect(getByText("Guardrail Settings")).toBeInTheDocument();
});
- await userEvent.hover(within(container).getByRole("img", { name: "info-circle" }));
+ await userEvent.hover(within(container).getByRole("img", { name: "Config guardrail details" }));
expect(await findByText("Guardrail is defined in the config file and cannot be edited.")).toBeInTheDocument();
});
diff --git a/ui/litellm-dashboard/src/app/(dashboard)/guardrails/_components/guardrail_info.tsx b/ui/litellm-dashboard/src/app/(dashboard)/guardrails/_components/guardrail_info.tsx
index f89e8a3248c..8f3ec6185fd 100644
--- a/ui/litellm-dashboard/src/app/(dashboard)/guardrails/_components/guardrail_info.tsx
+++ b/ui/litellm-dashboard/src/app/(dashboard)/guardrails/_components/guardrail_info.tsx
@@ -5,9 +5,8 @@ import {
updateGuardrailCall,
} from "@/components/networking";
import { copyToClipboard as utilCopyToClipboard } from "@/utils/dataUtils";
-import { EyeInvisibleOutlined, InfoCircleOutlined, StopOutlined } from "@ant-design/icons";
-import { ArrowLeft, CheckIcon, Code, CopyIcon } from "lucide-react";
+import { ArrowLeft, Ban, CheckIcon, Code, CopyIcon, EyeOff, Info } from "lucide-react";
import { Badge } from "@/components/ui/badge";
import { Card } from "@/components/ui/card";
import { Tabs, TabsContent, TabsList, TabsTrigger } from "@/components/ui/tabs";
@@ -607,7 +606,7 @@ const GuardrailInfoView: React.FC
= ({ guardrailId, onClose,
value === "MASK" ? "text-blue-600" : "text-red-600"
}`}
>
- {value === "MASK" ? : }
+ {value === "MASK" ? : }
{String(value)}
@@ -667,7 +666,7 @@ const GuardrailInfoView: React.FC = ({ guardrailId, onClose,
Guardrail Settings
{isConfigGuardrail && (
-
+
)}
{!isEditing &&
diff --git a/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/AwsSigV4Fields.tsx b/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/AwsSigV4Fields.tsx
index 09016bda266..8137e551028 100644
--- a/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/AwsSigV4Fields.tsx
+++ b/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/AwsSigV4Fields.tsx
@@ -1,6 +1,6 @@
import React from "react";
import { Input, Tooltip } from "antd";
-import { InfoCircleOutlined } from "@ant-design/icons";
+import { Info } from "lucide-react";
import { MountedFormField } from "@/components/common_components/MountedFormField";
import { antdRequired } from "@/components/common_components/antdFormRules";
@@ -12,7 +12,7 @@ const FieldLabel: React.FC<{ label: string; tooltip: string }> = ({ label, toolt
{label}
-
+
);
diff --git a/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/CreateMCPServer.tsx b/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/CreateMCPServer.tsx
index 5233081fc39..f5acd116f33 100644
--- a/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/CreateMCPServer.tsx
+++ b/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/CreateMCPServer.tsx
@@ -1,7 +1,7 @@
import React, { useState } from "react";
import { Tooltip, Select, Input as AntdInput, InputNumber, Collapse } from "antd";
import { FormProvider, useForm, useWatch } from "react-hook-form";
-import { InfoCircleOutlined } from "@ant-design/icons";
+import { Info } from "lucide-react";
import { Button } from "@/components/ui/button";
import { Input } from "@/components/ui/input";
import { UiLoadingSpinner } from "@/components/ui/ui-loading-spinner";
@@ -688,7 +688,7 @@ const CreateMCPServer: React.FC = ({
MCP Server Name
-
+
}
@@ -709,7 +709,7 @@ const CreateMCPServer: React.FC = ({
Alias
-
+
}
@@ -827,7 +827,7 @@ const CreateMCPServer: React.FC = ({
Max Concurrent Requests (optional)
-
+
}
@@ -910,7 +910,7 @@ const CreateMCPServer: React.FC = ({
Authentication Value
-
+
}
diff --git a/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/DcrBridgeToggle.tsx b/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/DcrBridgeToggle.tsx
index 35b23f9873c..a18c4592838 100644
--- a/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/DcrBridgeToggle.tsx
+++ b/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/DcrBridgeToggle.tsx
@@ -1,6 +1,6 @@
import React from "react";
import { Switch, Tooltip } from "antd";
-import { InfoCircleOutlined } from "@ant-design/icons";
+import { Info } from "lucide-react";
import { MountedFormField } from "@/components/common_components/MountedFormField";
import { isClientForwardedTokenMode } from "@/components/mcp_tools/types";
@@ -29,7 +29,7 @@ export default function DcrBridgeToggle({
Gateway-hosted sign-in (DCR bridge)
-
+
}
diff --git a/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/EnvVarsSection.tsx b/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/EnvVarsSection.tsx
index 23a55d2dc1a..01f68d2a531 100644
--- a/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/EnvVarsSection.tsx
+++ b/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/EnvVarsSection.tsx
@@ -1,7 +1,7 @@
import React from "react";
import { Input, Select, Tooltip, Typography } from "antd";
import { Button } from "@/components/ui/button";
-import { InfoCircleOutlined, MinusCircleOutlined, PlusOutlined } from "@ant-design/icons";
+import { CircleMinus, Info, Plus } from "lucide-react";
import { useFieldArray, useFormContext, useWatch } from "react-hook-form";
import {
@@ -55,7 +55,7 @@ const EnvVarsSection: React.FC = () => {
>
}
>
-
+
@@ -100,15 +100,15 @@ const EnvVarsSection: React.FC = () => {
{(control) =>
))}
@@ -130,7 +130,7 @@ const ScopedValueOrDescription: React.FC<{ index: number }> = ({ index }) => {
addonBefore={
-
+
Hint
diff --git a/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/IdJagFormFields.tsx b/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/IdJagFormFields.tsx
index 9a6f4ab571f..534f7bdf99b 100644
--- a/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/IdJagFormFields.tsx
+++ b/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/IdJagFormFields.tsx
@@ -1,6 +1,6 @@
import React from "react";
import { Input, Select, Tooltip } from "antd";
-import { InfoCircleOutlined } from "@ant-design/icons";
+import { Info } from "lucide-react";
import { MountedFormField } from "@/components/common_components/MountedFormField";
import { antdRequired } from "@/components/common_components/antdFormRules";
@@ -16,7 +16,7 @@ const FieldLabel: React.FC<{ label: string; tooltip: string }> = ({ label, toolt
{label}
-
+
);
diff --git a/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/MCPPermissionManagement.tsx b/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/MCPPermissionManagement.tsx
index 80a82dc22b8..e5c91a971c7 100644
--- a/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/MCPPermissionManagement.tsx
+++ b/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/MCPPermissionManagement.tsx
@@ -1,9 +1,8 @@
import React, { useEffect } from "react";
import { Select, Tooltip, Collapse, Input, Space, Switch } from "antd";
-import { TriangleAlert } from "lucide-react";
+import { CircleMinus, Info, Plus, TriangleAlert } from "lucide-react";
import { Alert, AlertDescription, AlertTitle } from "@/components/shared/Alert";
import { Button } from "@/components/ui/button";
-import { InfoCircleOutlined, MinusCircleOutlined, PlusOutlined } from "@ant-design/icons";
import { useFieldArray, useFormContext, useWatch } from "react-hook-form";
import { MCPServer, AUTH_TYPE } from "@/components/mcp_tools/types";
import {
@@ -74,14 +73,14 @@ const StaticHeadersFieldArray: React.FC = () => {
/>
)}
- remove(index)}
- className="text-gray-500 hover:text-red-500 cursor-pointer"
+ className="size-4 text-gray-500 hover:text-red-500 cursor-pointer"
/>
))}
@@ -196,7 +195,7 @@ const MCPPermissionManagement: React.FC = ({
Allow All LiteLLM Keys
-
+
@@ -213,7 +212,7 @@ const MCPPermissionManagement: React.FC = ({
Internal network only
-
+
@@ -231,7 +230,7 @@ const MCPPermissionManagement: React.FC = ({
Delegate auth to upstream (PKCE passthrough)
-
+
@@ -254,7 +253,7 @@ const MCPPermissionManagement: React.FC = ({
OAuth pass-through
-
+
@@ -289,7 +288,7 @@ const MCPPermissionManagement: React.FC = ({
MCP Access Groups
-
+
}
@@ -318,7 +317,7 @@ const MCPPermissionManagement: React.FC = ({
Extra Headers
-
+
{mcpServer?.extra_headers && mcpServer.extra_headers.length > 0 && (
@@ -351,7 +350,7 @@ const MCPPermissionManagement: React.FC = ({
Static Headers
-
+
diff --git a/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/OAuthFormFields.tsx b/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/OAuthFormFields.tsx
index 545efb910c4..549a2d71842 100644
--- a/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/OAuthFormFields.tsx
+++ b/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/OAuthFormFields.tsx
@@ -1,6 +1,6 @@
import React from "react";
import { Input as AntdInput, InputNumber, Select, Tooltip } from "antd";
-import { InfoCircleOutlined } from "@ant-design/icons";
+import { Info } from "lucide-react";
import { Button } from "@/components/ui/button";
import { Input } from "@/components/ui/input";
import { OAUTH_FLOW } from "@/components/mcp_tools/types";
@@ -38,7 +38,7 @@ const FieldLabel: React.FC<{ label: string; tooltip: string }> = ({ label, toolt
{label}
-
+
);
diff --git a/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/OpenAPIFormSection.tsx b/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/OpenAPIFormSection.tsx
index 78c8bbe73a9..9d4dced9759 100644
--- a/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/OpenAPIFormSection.tsx
+++ b/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/OpenAPIFormSection.tsx
@@ -1,6 +1,6 @@
import React, { useState } from "react";
import { Input, Tooltip } from "antd";
-import { InfoCircleOutlined } from "@ant-design/icons";
+import { Info } from "lucide-react";
import { AUTH_TYPE, OAUTH_FLOW } from "@/components/mcp_tools/types";
import { MountedFormField } from "@/components/common_components/MountedFormField";
import { antdRequired } from "@/components/common_components/antdFormRules";
@@ -69,7 +69,7 @@ const OpenAPIFormSection: React.FC = ({
OpenAPI Spec URL
-
+
}
diff --git a/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/OpenApiByokFields.tsx b/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/OpenApiByokFields.tsx
index 83c841439f0..f46040ff831 100644
--- a/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/OpenApiByokFields.tsx
+++ b/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/OpenApiByokFields.tsx
@@ -1,6 +1,6 @@
import React from "react";
import { Input, Select, Switch, Tooltip } from "antd";
-import { InfoCircleOutlined } from "@ant-design/icons";
+import { Info } from "lucide-react";
import { useWatch } from "react-hook-form";
import { MountedFormField } from "@/components/common_components/MountedFormField";
@@ -26,7 +26,7 @@ const OpenApiByokFields: React.FC = () => {
BYOK (Bring Your Own Key)
-
+
}
@@ -39,7 +39,7 @@ const OpenApiByokFields: React.FC = () => {
<>
{hasAuthType && (
-
+
User keys will be sent as:{" "}
@@ -50,7 +50,7 @@ const OpenApiByokFields: React.FC = () => {
)}
{!authType && (
-
+
Set the Authentication Type below to specify how user keys are sent (e.g., Bearer
Token, API Key header).
@@ -62,7 +62,7 @@ const OpenApiByokFields: React.FC = () => {
Access Description
-
+
}
@@ -84,7 +84,7 @@ const OpenApiByokFields: React.FC = () => {
API Key Help URL
-
+
}
diff --git a/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/StdioConfiguration.tsx b/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/StdioConfiguration.tsx
index bb410aa19dc..0efed153432 100644
--- a/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/StdioConfiguration.tsx
+++ b/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/StdioConfiguration.tsx
@@ -1,6 +1,6 @@
import React from "react";
import { Input, Tooltip } from "antd";
-import { InfoCircleOutlined } from "@ant-design/icons";
+import { Info } from "lucide-react";
import { MountedFormField } from "@/components/common_components/MountedFormField";
import { antdRequired } from "@/components/common_components/antdFormRules";
@@ -37,7 +37,7 @@ const StdioConfiguration: React.FC = ({ isVisible, requ
Stdio Configuration (JSON)
-
+
}
diff --git a/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/TokenEndpointAuthMethodField.tsx b/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/TokenEndpointAuthMethodField.tsx
index 38fa3573079..9b84afd4fd3 100644
--- a/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/TokenEndpointAuthMethodField.tsx
+++ b/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/TokenEndpointAuthMethodField.tsx
@@ -1,6 +1,6 @@
import React from "react";
import { Select, Tooltip } from "antd";
-import { InfoCircleOutlined } from "@ant-design/icons";
+import { Info } from "lucide-react";
import { MountedFormField } from "@/components/common_components/MountedFormField";
import { selectControl } from "./mcpFieldRules";
@@ -20,7 +20,7 @@ const TokenEndpointAuthMethodField: React.FC
Token Endpoint Auth Method (optional)
-
+
}
diff --git a/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/TokenExchangeFormFields.tsx b/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/TokenExchangeFormFields.tsx
index ba213b20655..7a9f8e70e4b 100644
--- a/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/TokenExchangeFormFields.tsx
+++ b/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/TokenExchangeFormFields.tsx
@@ -1,6 +1,6 @@
import React from "react";
import { Input, Select, Tooltip } from "antd";
-import { InfoCircleOutlined } from "@ant-design/icons";
+import { Info } from "lucide-react";
import { useWatch } from "react-hook-form";
import { MountedFormField } from "@/components/common_components/MountedFormField";
@@ -17,7 +17,7 @@ const FieldLabel: React.FC<{ label: string; tooltip: string }> = ({ label, toolt
{label}
-
+
);
diff --git a/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/mcp_server_edit.tsx b/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/mcp_server_edit.tsx
index 4b092aa3f80..adf8f8c078e 100644
--- a/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/mcp_server_edit.tsx
+++ b/ui/litellm-dashboard/src/app/(dashboard)/mcp-servers/_components/mcp_server_edit.tsx
@@ -1,9 +1,8 @@
import React, { useState, useEffect } from "react";
import { Select, Tooltip, Input, InputNumber } from "antd";
-import { TriangleAlert } from "lucide-react";
+import { Info, TriangleAlert } from "lucide-react";
import { Alert, AlertDescription, AlertTitle } from "@/components/shared/Alert";
import { FormProvider, useForm } from "react-hook-form";
-import { InfoCircleOutlined } from "@ant-design/icons";
import { Button } from "@/components/ui/button";
import { Tabs, TabsContent, TabsList, TabsTrigger } from "@/components/ui/tabs";
import {
@@ -875,7 +874,7 @@ const MCPServerEdit: React.FC = ({
OpenAPI Spec URL
-
+
}
@@ -898,7 +897,7 @@ const MCPServerEdit: React.FC = ({
Max Concurrent Requests (optional)
-
+
}
@@ -1026,7 +1025,7 @@ const MCPServerEdit: React.FC = ({
Authentication Value
-
+
}
@@ -1091,7 +1090,7 @@ const MCPServerEdit: React.FC = ({
AWS Region
-
+
}
@@ -1110,7 +1109,7 @@ const MCPServerEdit: React.FC = ({
AWS Service Name
-
+
}
@@ -1129,7 +1128,7 @@ const MCPServerEdit: React.FC = ({
AWS Access Key ID
-
+
}
@@ -1148,7 +1147,7 @@ const MCPServerEdit: React.FC = ({
AWS Secret Access Key
-
+
}
@@ -1167,7 +1166,7 @@ const MCPServerEdit: React.FC = ({
AWS Session Token
-
+
}
@@ -1186,7 +1185,7 @@ const MCPServerEdit: React.FC = ({
AWS Role ARN
-
+
}
@@ -1205,7 +1204,7 @@ const MCPServerEdit: React.FC = ({
AWS Session Name
-
+
}
diff --git a/ui/litellm-dashboard/src/components/Settings/AdminSettings/LoggingSettings/LoggingSettings.tsx b/ui/litellm-dashboard/src/components/Settings/AdminSettings/LoggingSettings/LoggingSettings.tsx
index 20dfb3d96fb..a9702451035 100644
--- a/ui/litellm-dashboard/src/components/Settings/AdminSettings/LoggingSettings/LoggingSettings.tsx
+++ b/ui/litellm-dashboard/src/components/Settings/AdminSettings/LoggingSettings/LoggingSettings.tsx
@@ -20,10 +20,9 @@ import { InputGroup, InputGroupAddon, InputGroupInput } from "@/components/ui/in
import { Switch } from "@/components/ui/switch";
import { Tooltip, TooltipContent, TooltipProvider, TooltipTrigger } from "@/components/ui/tooltip";
import { UiLoadingSpinner } from "@/components/ui/ui-loading-spinner";
-import { ClockCircleOutlined } from "@ant-design/icons";
import { Skeleton } from "@/components/ui/skeleton";
import { Card, CardContent, CardHeader, CardTitle } from "@/components/ui/card";
-import { CircleHelp } from "lucide-react";
+import { CircleHelp, Clock } from "lucide-react";
import React, { useCallback, useMemo } from "react";
import { useForm } from "react-hook-form";
@@ -199,7 +198,7 @@ const LoggingSettingsForm: React.FC = ({
placeholder={field.placeholder}
/>
-
+
) : (
diff --git a/ui/litellm-dashboard/src/components/Settings/AdminSettings/PluginSettings/PluginSettings.integration.test.tsx b/ui/litellm-dashboard/src/components/Settings/AdminSettings/PluginSettings/PluginSettings.integration.test.tsx
index 9f23fc4897c..a91e54b058b 100644
--- a/ui/litellm-dashboard/src/components/Settings/AdminSettings/PluginSettings/PluginSettings.integration.test.tsx
+++ b/ui/litellm-dashboard/src/components/Settings/AdminSettings/PluginSettings/PluginSettings.integration.test.tsx
@@ -62,7 +62,7 @@ describe("PluginSettings config payload", () => {
render();
expect(await screen.findByText("Alpha")).toBeInTheDocument();
- await user.click(screen.getByRole("button", { name: "edit" }));
+ await user.click(screen.getByRole("button", { name: "Edit alpha" }));
expect(await screen.findByLabelText(/Plugin Key/)).toHaveValue("");
await user.click(screen.getByRole("button", { name: "Save" }));
@@ -88,7 +88,7 @@ describe("PluginSettings config payload", () => {
render();
expect(await screen.findByText("Alpha")).toBeInTheDocument();
- await user.click(screen.getByRole("button", { name: "edit" }));
+ await user.click(screen.getByRole("button", { name: "Edit alpha" }));
fireEvent.change(await screen.findByLabelText(/Plugin Key/), { target: { value: "sk-brand-new" } });
await user.click(screen.getByRole("button", { name: "Save" }));
@@ -119,7 +119,7 @@ describe("PluginSettings plugin key reveal (post-migration shadcn affordance)",
render();
expect(await screen.findByText("Alpha")).toBeInTheDocument();
- await user.click(screen.getByRole("button", { name: "edit" }));
+ await user.click(screen.getByRole("button", { name: "Edit alpha" }));
const keyInput = await screen.findByLabelText(/Plugin Key/);
expect(keyInput).toHaveAttribute("type", "password");
diff --git a/ui/litellm-dashboard/src/components/Settings/AdminSettings/PluginSettings/PluginSettings.tsx b/ui/litellm-dashboard/src/components/Settings/AdminSettings/PluginSettings/PluginSettings.tsx
index 8a567f9dd21..7628dfbc13a 100644
--- a/ui/litellm-dashboard/src/components/Settings/AdminSettings/PluginSettings/PluginSettings.tsx
+++ b/ui/litellm-dashboard/src/components/Settings/AdminSettings/PluginSettings/PluginSettings.tsx
@@ -3,8 +3,7 @@
import { useState, useEffect } from "react";
import { Card, Space, Table, Typography } from "antd";
import { Button } from "@/components/ui/button";
-import { DeleteOutlined, EditOutlined, PlusOutlined } from "@ant-design/icons";
-import { Eye, EyeOff } from "lucide-react";
+import { Eye, EyeOff, Pencil, Plus, Trash2 } from "lucide-react";
import { getConfigFieldSetting, updateConfigFieldSetting } from "@/components/networking";
import useAuthorized from "@/app/(dashboard)/hooks/useAuthorized";
import { FieldGroup } from "@/components/shared/form/field";
@@ -113,13 +112,18 @@ export default function PluginSettings() {
{
title: "Actions",
key: "actions",
- render: (_: unknown, __: Plugin, idx: number) => (
+ render: (_: unknown, plugin: Plugin, idx: number) => (
-
),
@@ -138,7 +142,7 @@ export default function PluginSettings() {
-
+
Add Plugin
diff --git a/ui/litellm-dashboard/src/components/VirtualKeysPage/keyTableColumns.tsx b/ui/litellm-dashboard/src/components/VirtualKeysPage/keyTableColumns.tsx
index 8c500607ec4..2fecdb01547 100644
--- a/ui/litellm-dashboard/src/components/VirtualKeysPage/keyTableColumns.tsx
+++ b/ui/litellm-dashboard/src/components/VirtualKeysPage/keyTableColumns.tsx
@@ -1,6 +1,6 @@
"use client";
-import { InfoCircleOutlined } from "@ant-design/icons";
+import { Info } from "lucide-react";
import { ColumnDef } from "@tanstack/react-table";
import { DataTableMultiSortHeader, DataTableSortHeader, type DataTableSortField } from "@/components/shared/DataTable";
@@ -119,7 +119,7 @@ const InfoHeader = ({ label, tooltip }: { label: string; tooltip: string }) => (
{label}
- } />
+ } />
{tooltip}
diff --git a/ui/litellm-dashboard/src/components/add_model/ClassificationMethodConfig.tsx b/ui/litellm-dashboard/src/components/add_model/ClassificationMethodConfig.tsx
index be4573f8a85..019df7d4fa9 100644
--- a/ui/litellm-dashboard/src/components/add_model/ClassificationMethodConfig.tsx
+++ b/ui/litellm-dashboard/src/components/add_model/ClassificationMethodConfig.tsx
@@ -1,4 +1,4 @@
-import { InfoCircleOutlined } from "@ant-design/icons";
+import { Info } from "lucide-react";
import { SimpleTooltip } from "@/components/ui/tooltip";
import { Select as AntdSelect, Card, InputNumber, Radio, Space, Switch, Typography } from "antd";
import React from "react";
@@ -303,7 +303,7 @@ const ClassificationMethodConfig: React.FC = ({
Classification Rubric
-
+
= ({
/>
Include Assistant Turns
-
+
@@ -431,7 +431,7 @@ const ClassificationMethodConfig: React.FC = ({
Custom Technical Keywords
-
+
diff --git a/ui/litellm-dashboard/src/components/add_model/ComplexityRouterConfig.tsx b/ui/litellm-dashboard/src/components/add_model/ComplexityRouterConfig.tsx
index 744c74cecb0..ad5260c2c9c 100644
--- a/ui/litellm-dashboard/src/components/add_model/ComplexityRouterConfig.tsx
+++ b/ui/litellm-dashboard/src/components/add_model/ComplexityRouterConfig.tsx
@@ -1,4 +1,4 @@
-import { InfoCircleOutlined } from "@ant-design/icons";
+import { Info } from "lucide-react";
import { SimpleTooltip } from "@/components/ui/tooltip";
import { Select as AntdSelect, Card, Collapse, Divider, Input, Space, Switch, Typography } from "antd";
import React from "react";
@@ -256,7 +256,7 @@ const ComplexityRouterConfig: React.FC = ({
Complexity Tier Configuration
-
+
@@ -286,7 +286,7 @@ const ComplexityRouterConfig: React.FC = ({
{label} Tier
-
+
Tier {index + 1} of {TIER_KEYS.length} · {tier}
@@ -336,7 +336,7 @@ const ComplexityRouterConfig: React.FC = ({
Default Model
-
+
= ({ keywords, onCha
Escalation Keywords
-
+
diff --git a/ui/litellm-dashboard/src/components/add_model/KeywordTierRules.tsx b/ui/litellm-dashboard/src/components/add_model/KeywordTierRules.tsx
index 24e4c33a49d..08d00606987 100644
--- a/ui/litellm-dashboard/src/components/add_model/KeywordTierRules.tsx
+++ b/ui/litellm-dashboard/src/components/add_model/KeywordTierRules.tsx
@@ -1,4 +1,4 @@
-import { DeleteOutlined, InfoCircleOutlined, PlusOutlined } from "@ant-design/icons";
+import { Info, Plus, Trash2 } from "lucide-react";
import { SimpleTooltip } from "@/components/ui/tooltip";
import { Card, Empty, Select as AntdSelect, Typography } from "antd";
import { Button } from "@/components/ui/button";
@@ -73,11 +73,11 @@ const KeywordTierRules: React.FC = ({ rules, onChange, ti
Keyword Tier Overrides
-
+
-
+
Add keyword rule
@@ -139,7 +139,7 @@ const KeywordTierRules: React.FC
= ({ rules, onChange, ti
aria-label={`Remove keyword rule ${index + 1}`}
onClick={() => removeRule(rule.id)}
>
-
+
diff --git a/ui/litellm-dashboard/src/components/add_model/SemanticKeywordMatching.tsx b/ui/litellm-dashboard/src/components/add_model/SemanticKeywordMatching.tsx
index 867f6a901e8..7f8a71b1b8b 100644
--- a/ui/litellm-dashboard/src/components/add_model/SemanticKeywordMatching.tsx
+++ b/ui/litellm-dashboard/src/components/add_model/SemanticKeywordMatching.tsx
@@ -1,4 +1,4 @@
-import { InfoCircleOutlined } from "@ant-design/icons";
+import { Info } from "lucide-react";
import { SimpleTooltip } from "@/components/ui/tooltip";
import { InputNumber, Select as AntdSelect, Switch, Typography } from "antd";
import React from "react";
@@ -43,7 +43,7 @@ const SemanticKeywordMatching: React.FC = ({
Semantic keyword matching
-
+
diff --git a/ui/litellm-dashboard/src/components/add_model/advanced_settings.tsx b/ui/litellm-dashboard/src/components/add_model/advanced_settings.tsx
index 140b4363327..2d35a772a63 100644
--- a/ui/litellm-dashboard/src/components/add_model/advanced_settings.tsx
+++ b/ui/litellm-dashboard/src/components/add_model/advanced_settings.tsx
@@ -1,11 +1,10 @@
import React from "react";
import { Switch, Select, Tooltip, DatePicker } from "antd";
-import { ChevronDown } from "lucide-react";
+import { ChevronDown, Info } from "lucide-react";
import { Collapsible, CollapsibleContent, CollapsibleTrigger } from "@/components/ui/collapsible";
import { Input } from "@/components/ui/input";
import { Row, Col, Typography } from "antd";
import TextArea from "antd/es/input/TextArea";
-import { InfoCircleOutlined } from "@ant-design/icons";
import { Team } from "../key_team_helpers/key_list";
import { antdRules } from "../common_components/antdFormRules";
import { labelWithHint } from "@/components/shared/form/LabelWithHint";
@@ -115,7 +114,7 @@ const AdvancedSettings: React.FC = ({
rel="noopener noreferrer"
onClick={(e) => e.stopPropagation()}
>
-
+
@@ -145,7 +144,7 @@ const AdvancedSettings: React.FC = ({
rel="noopener noreferrer"
onClick={(e) => e.stopPropagation()} // Prevent accordion from collapsing when clicking link
>
-
+
diff --git a/ui/litellm-dashboard/src/components/add_model/provider_specific_fields.tsx b/ui/litellm-dashboard/src/components/add_model/provider_specific_fields.tsx
index e5f1f108b4f..ab132ff88a1 100644
--- a/ui/litellm-dashboard/src/components/add_model/provider_specific_fields.tsx
+++ b/ui/litellm-dashboard/src/components/add_model/provider_specific_fields.tsx
@@ -1,8 +1,8 @@
import { useProviderFields } from "@/app/(dashboard)/hooks/providers/useProviderFields";
-import { UploadOutlined } from "@ant-design/icons";
import { Input } from "@/components/ui/input";
import { Col, Input as AntdInput, Row, Select, Typography, Upload, UploadProps } from "antd";
import { Button } from "@/components/ui/button";
+import { Upload as UploadIcon } from "lucide-react";
import React from "react";
import { useFormContext } from "react-hook-form";
import { antdRequired } from "../common_components/antdFormRules";
@@ -257,7 +257,7 @@ const ProviderSpecificFields: React.FC = ({ selecte
}}
>
-
+
Click to Upload
diff --git a/ui/litellm-dashboard/src/components/claude_code_plugins/skill_detail.tsx b/ui/litellm-dashboard/src/components/claude_code_plugins/skill_detail.tsx
index fe001641135..c4f692a4d8d 100644
--- a/ui/litellm-dashboard/src/components/claude_code_plugins/skill_detail.tsx
+++ b/ui/litellm-dashboard/src/components/claude_code_plugins/skill_detail.tsx
@@ -1,5 +1,5 @@
import React, { useState } from "react";
-import { ArrowLeftOutlined, CopyOutlined, CheckOutlined, LinkOutlined } from "@ant-design/icons";
+import { ArrowLeft, Check, Copy, Link2 } from "lucide-react";
import { buildMarketplaceSettingsSnippet, formatInstallCommand } from "./helpers";
import { Plugin } from "./types";
@@ -64,7 +64,7 @@ const SkillDetail: React.FC = ({ skill, onBack }) => {
marginBottom: 24,
}}
>
-
+
Skills
@@ -163,7 +163,7 @@ const SkillDetail: React.FC
= ({ skill, onBack }) => {
}}
>
{sourceUrl.replace("https://", "")}
-
+
)}
@@ -243,7 +243,7 @@ const SkillDetail: React.FC = ({ skill, onBack }) => {
padding: 0,
}}
>
- {copiedKey === "install" ? : }
+ {copiedKey === "install" ? : }
{copiedKey === "install" ? "Copied" : "Copy"}
@@ -315,7 +315,7 @@ const SkillDetail: React.FC
= ({ skill, onBack }) => {
padding: 0,
}}
>
- {copiedKey === "settings" ? : }
+ {copiedKey === "settings" ? : }
{copiedKey === "settings" ? "Copied" : "Copy"}
diff --git a/ui/litellm-dashboard/src/components/common_components/AccessGroupSelector.tsx b/ui/litellm-dashboard/src/components/common_components/AccessGroupSelector.tsx
index fb8629dc8fa..96f21f30385 100644
--- a/ui/litellm-dashboard/src/components/common_components/AccessGroupSelector.tsx
+++ b/ui/litellm-dashboard/src/components/common_components/AccessGroupSelector.tsx
@@ -1,6 +1,6 @@
import React from "react";
import { Select, Skeleton } from "antd";
-import { TeamOutlined } from "@ant-design/icons";
+import { Users } from "lucide-react";
import { useAccessGroups, AccessGroupResponse } from "@/app/(dashboard)/hooks/accessGroups/useAccessGroups";
export interface AccessGroupSelectorProps {
@@ -42,7 +42,7 @@ const AccessGroupSelector: React.FC
= ({
{showLabel && (
- {labelText}
+ {labelText}
)}
@@ -68,7 +68,7 @@ const AccessGroupSelector: React.FC
= ({
{showLabel && (
- {labelText}
+ {labelText}
)}
-
+
@@ -105,7 +97,7 @@ export const ByokCredentialModal: React.FC = ({ server
L
-
+
{firstLetter}
@@ -148,7 +140,7 @@ export const ByokCredentialModal: React.FC = ({ server
{server.byok_description.map((item, i) => (
-
-
+
{item}
))}
@@ -160,7 +152,7 @@ export const ByokCredentialModal: React.FC = ({ server
onClick={() => setStep(2)}
className="w-full bg-gray-900 hover:bg-gray-700 text-white font-medium py-3 px-6 rounded-xl flex items-center justify-center gap-2 transition-colors"
>
- Continue to Authentication
+ Continue to Authentication
Cancel
@@ -170,7 +162,7 @@ export const ByokCredentialModal: React.FC = ({ server
{/* Key icon */}
-
+
Provide API Key
@@ -192,7 +184,7 @@ export const ByokCredentialModal: React.FC
= ({ server
rel="noopener noreferrer"
className="text-blue-500 hover:text-blue-700 text-sm mt-2 flex items-center gap-1"
>
- Where do I find my API key?
+ Where do I find my API key?
)}
@@ -213,7 +205,7 @@ export const ByokCredentialModal: React.FC = ({ server
{/* Security note */}
-
+
Your key is stored securely and transmitted over HTTPS. It is never shared with third parties.
@@ -224,7 +216,7 @@ export const ByokCredentialModal: React.FC
= ({ server
disabled={loading}
className="w-full bg-blue-500 hover:bg-blue-600 disabled:opacity-60 text-white font-medium py-3 px-6 rounded-xl flex items-center justify-center gap-2 transition-colors"
>
- Connect & Authorize
+ Connect & Authorize
)}
diff --git a/ui/litellm-dashboard/src/components/model_info_view.tsx b/ui/litellm-dashboard/src/components/model_info_view.tsx
index 58d7a7bca6e..407144d1a52 100644
--- a/ui/litellm-dashboard/src/components/model_info_view.tsx
+++ b/ui/litellm-dashboard/src/components/model_info_view.tsx
@@ -2,7 +2,6 @@ import { useModelCostMap } from "@/app/(dashboard)/hooks/models/useModelCostMap"
import { useModelHub, useModelsInfo } from "@/app/(dashboard)/hooks/models/useModels";
import { useQueryClient } from "@tanstack/react-query";
import { transformModelData } from "@/app/(dashboard)/models-and-endpoints/utils/modelDataTransformer";
-import { InfoCircleOutlined } from "@ant-design/icons";
import { KeyIcon, RefreshIcon, TrashIcon } from "@heroicons/react/outline";
import { Button } from "@/components/ui/button";
import { Card } from "@/components/ui/card";
@@ -10,7 +9,7 @@ import { Tabs, TabsContent, TabsList, TabsTrigger } from "@/components/ui/tabs";
import { SimpleTooltip } from "@/components/ui/tooltip";
import { applyPtuModelInfo } from "../utils/ptuModelInfo";
import { usePtuCostAttributionEnabled } from "@/app/(dashboard)/hooks/uiSettings/usePtuCostAttributionEnabled";
-import { ArrowLeft, CheckIcon, CopyIcon } from "lucide-react";
+import { ArrowLeft, CheckIcon, CopyIcon, Info } from "lucide-react";
import { useEffect, useMemo, useState } from "react";
import { copyToClipboard as utilCopyToClipboard } from "../utils/dataUtils";
import { stripMaskedSecrets } from "../utils/maskedSecretUtils";
@@ -757,7 +756,7 @@ export default function ModelInfoView({
)
) : (
-
+
)}
diff --git a/ui/litellm-dashboard/src/components/organisms/RegenerateKeyModal.tsx b/ui/litellm-dashboard/src/components/organisms/RegenerateKeyModal.tsx
index 5158d8612ab..81d3dcc61a2 100644
--- a/ui/litellm-dashboard/src/components/organisms/RegenerateKeyModal.tsx
+++ b/ui/litellm-dashboard/src/components/organisms/RegenerateKeyModal.tsx
@@ -1,9 +1,8 @@
import useAuthorized from "@/app/(dashboard)/hooks/useAuthorized";
-import { CheckOutlined, CopyOutlined, SyncOutlined } from "@ant-design/icons";
import { Alert, AlertTitle } from "@/components/shared/Alert";
import { Button } from "@/components/ui/button";
import { Dialog, DialogContent, DialogFooter, DialogHeader, DialogTitle } from "@/components/ui/dialog";
-import { CircleHelp, TriangleAlert } from "lucide-react";
+import { Check, CircleHelp, Copy, RefreshCw, TriangleAlert } from "lucide-react";
import React, { useEffect, useMemo, useState } from "react";
import { useWatch } from "react-hook-form";
import { CopyToClipboard } from "react-copy-to-clipboard";
@@ -264,7 +263,7 @@ export function RegenerateKeyModal({ selectedToken, visible, onClose, onKeyUpdat
- {copied ? : }
+ {copied ? : }
{copied ? "Copied" : "Copy Key"}
@@ -275,7 +274,7 @@ export function RegenerateKeyModal({ selectedToken, visible, onClose, onKeyUpdat
Cancel
-
+
Regenerate
>
diff --git a/ui/litellm-dashboard/src/components/organisms/create_key_button.tsx b/ui/litellm-dashboard/src/components/organisms/create_key_button.tsx
index eab5abb9936..168b7f6cf25 100644
--- a/ui/litellm-dashboard/src/components/organisms/create_key_button.tsx
+++ b/ui/litellm-dashboard/src/components/organisms/create_key_button.tsx
@@ -7,14 +7,13 @@ import { useUISettings } from "@/app/(dashboard)/hooks/uiSettings/useUISettings"
import useAuthorized from "@/app/(dashboard)/hooks/useAuthorized";
import useCan from "@/app/(dashboard)/hooks/useCan";
import { formatNumberWithCommas } from "@/utils/dataUtils";
-import { InfoCircleOutlined } from "@ant-design/icons";
import { useQueryClient } from "@tanstack/react-query";
import { Button } from "@/components/ui/button";
import { Collapsible, CollapsibleContent, CollapsibleTrigger } from "@/components/ui/collapsible";
import { Input } from "@/components/ui/input";
import { Field, FieldLabel } from "@/components/shared/form/field";
import { Input as AntdInput, Radio, Select, Switch, Tag, Tooltip, Typography } from "antd";
-import { ChevronDown } from "lucide-react";
+import { ChevronDown, Info } from "lucide-react";
import { useDebouncedCallback } from "@tanstack/react-pacer/debouncer";
import { DEBOUNCE_WAIT_MS } from "@/utils/debounceConstants";
import React, { useEffect, useMemo, useRef, useState } from "react";
@@ -647,7 +646,7 @@ const CreateKey: React.FC = ({ team, teams, data, addKey, autoOp
Owned By{" "}
-
+
@@ -667,7 +666,7 @@ const CreateKey: React.FC = ({ team, teams, data, addKey, autoOp
User ID{" "}
-
+
}
@@ -740,7 +739,7 @@ const CreateKey: React.FC = ({ team, teams, data, addKey, autoOp
Organization{" "}
-
+
}
@@ -763,7 +762,7 @@ const CreateKey: React.FC = ({ team, teams, data, addKey, autoOp
Team{" "}
-
+
}
@@ -790,7 +789,7 @@ const CreateKey: React.FC = ({ team, teams, data, addKey, autoOp
Project{" "}
-
+
}
@@ -836,7 +835,7 @@ const CreateKey: React.FC = ({ team, teams, data, addKey, autoOp
: "Unique identifier for this service account"
}
>
-
+
}
@@ -856,7 +855,7 @@ const CreateKey: React.FC = ({ team, teams, data, addKey, autoOp
Models{" "}
-
+
}
@@ -909,7 +908,7 @@ const CreateKey: React.FC = ({ team, teams, data, addKey, autoOp
Key Type{" "}
-
+
}
@@ -972,7 +971,7 @@ const CreateKey: React.FC = ({ team, teams, data, addKey, autoOp
Max Budget (USD){" "}
-
+
}
@@ -999,7 +998,7 @@ const CreateKey: React.FC = ({ team, teams, data, addKey, autoOp
Reset Budget{" "}
-
+
}
@@ -1021,7 +1020,7 @@ const CreateKey: React.FC = ({ team, teams, data, addKey, autoOp
Budget Windows{" "}
-
+
@@ -1032,7 +1031,7 @@ const CreateKey: React.FC = ({ team, teams, data, addKey, autoOp
Budget Fallbacks{" "}
-
+
@@ -1049,7 +1048,7 @@ const CreateKey: React.FC = ({ team, teams, data, addKey, autoOp
Tokens per minute Limit (TPM){" "}
-
+
}
@@ -1090,7 +1089,7 @@ const CreateKey: React.FC = ({ team, teams, data, addKey, autoOp
Requests per minute Limit (RPM){" "}
-
+
}
@@ -1130,7 +1129,7 @@ const CreateKey: React.FC = ({ team, teams, data, addKey, autoOp
Per-Tag Rate Limits{" "}
-
+
@@ -1142,7 +1141,7 @@ const CreateKey: React.FC = ({ team, teams, data, addKey, autoOp
Throttle on budget exceeded{" "}
-
+
}
@@ -1164,7 +1163,7 @@ const CreateKey: React.FC = ({ team, teams, data, addKey, autoOp
Enable Prompt Caching{" "}
-
+
}
@@ -1191,7 +1190,7 @@ const CreateKey: React.FC = ({ team, teams, data, addKey, autoOp
rel="noopener noreferrer"
onClick={(e) => e.stopPropagation()} // Prevent accordion from collapsing when clicking link
>
-
+
@@ -1231,7 +1230,7 @@ const CreateKey: React.FC = ({ team, teams, data, addKey, autoOp
rel="noopener noreferrer"
onClick={(e) => e.stopPropagation()} // Prevent accordion from collapsing when clicking link
>
-
+
@@ -1267,7 +1266,7 @@ const CreateKey: React.FC = ({ team, teams, data, addKey, autoOp
rel="noopener noreferrer"
onClick={(e) => e.stopPropagation()} // Prevent accordion from collapsing when clicking link
>
-
+
@@ -1309,7 +1308,7 @@ const CreateKey: React.FC = ({ team, teams, data, addKey, autoOp
rel="noopener noreferrer"
onClick={(e) => e.stopPropagation()} // Prevent accordion from collapsing when clicking link
>
-
+
@@ -1344,7 +1343,7 @@ const CreateKey: React.FC = ({ team, teams, data, addKey, autoOp
Access Groups{" "}
-
+
}
@@ -1371,7 +1370,7 @@ const CreateKey: React.FC = ({ team, teams, data, addKey, autoOp
rel="noopener noreferrer"
onClick={(e) => e.stopPropagation()} // Prevent accordion from collapsing when clicking link
>
-
+
@@ -1404,7 +1403,7 @@ const CreateKey: React.FC = ({ team, teams, data, addKey, autoOp
Allowed Vector Stores{" "}
-
+
}
@@ -1426,7 +1425,7 @@ const CreateKey: React.FC = ({ team, teams, data, addKey, autoOp
Metadata{" "}
-
+
}
@@ -1447,7 +1446,7 @@ const CreateKey: React.FC = ({ team, teams, data, addKey, autoOp
Tags{" "}
-
+
}
@@ -1478,7 +1477,7 @@ const CreateKey: React.FC = ({ team, teams, data, addKey, autoOp
Allowed MCP Servers{" "}
-
+
}
@@ -1521,7 +1520,7 @@ const CreateKey: React.FC = ({ team, teams, data, addKey, autoOp
Allowed Agents{" "}
-
+
}
@@ -1688,7 +1687,7 @@ const CreateKey: React.FC = ({ team, teams, data, addKey, autoOp
}
>
-
+
diff --git a/ui/litellm-dashboard/src/components/routing_groups/index.tsx b/ui/litellm-dashboard/src/components/routing_groups/index.tsx
index 560523972ca..9f94fbf6c98 100644
--- a/ui/litellm-dashboard/src/components/routing_groups/index.tsx
+++ b/ui/litellm-dashboard/src/components/routing_groups/index.tsx
@@ -3,7 +3,7 @@
import React, { useMemo, useState } from "react";
import { Card, Flex, Input, Space, Typography } from "antd";
import { Button } from "@/components/ui/button";
-import { PlusOutlined, ReloadOutlined, SearchOutlined } from "@ant-design/icons";
+import { Plus, RefreshCw, Search } from "lucide-react";
import { useRoutingGroups, useSaveRoutingGroups } from "@/app/(dashboard)/hooks/routingGroups/useRoutingGroups";
import { useRouterFields } from "@/app/(dashboard)/hooks/router/useRouterFields";
import { useModelHub } from "@/app/(dashboard)/hooks/models/useModels";
@@ -107,7 +107,7 @@ const RoutingGroups: React.FC = () => {
}
+ prefix={}
placeholder="Search groups..."
value={searchQuery}
onChange={(e) => setSearchQuery(e.target.value)}
@@ -120,11 +120,11 @@ const RoutingGroups: React.FC = () => {
disabled={isFetching && !isLoading}
aria-busy={isFetching && !isLoading}
>
-
+
Refresh
-
+
Create Group
diff --git a/ui/litellm-dashboard/src/components/shared/advanced_date_picker.test.tsx b/ui/litellm-dashboard/src/components/shared/advanced_date_picker.test.tsx
index c907aeae035..bb4e0865f3f 100644
--- a/ui/litellm-dashboard/src/components/shared/advanced_date_picker.test.tsx
+++ b/ui/litellm-dashboard/src/components/shared/advanced_date_picker.test.tsx
@@ -49,10 +49,8 @@ describe("AdvancedDatePicker", () => {
});
it("should display formatted date range", () => {
- render();
- // The component displays date range in the format "D MMM, HH:mm - D MMM, HH:mm"
- // Just check that the clock icon is present
- expect(screen.getByLabelText("clock-circle")).toBeInTheDocument();
+ const { container } = render();
+ expect(getTrigger(container)).toHaveTextContent(/\d{1,2} \w{3}, \d{2}:\d{2} - \d{1,2} \w{3}, \d{2}:\d{2}/);
});
it("should open dropdown when clicked", () => {
diff --git a/ui/litellm-dashboard/src/components/shared/advanced_date_picker.tsx b/ui/litellm-dashboard/src/components/shared/advanced_date_picker.tsx
index b9a4d28539b..0013f702362 100644
--- a/ui/litellm-dashboard/src/components/shared/advanced_date_picker.tsx
+++ b/ui/litellm-dashboard/src/components/shared/advanced_date_picker.tsx
@@ -1,4 +1,4 @@
-import { CalendarOutlined, ClockCircleOutlined } from "@ant-design/icons";
+import { Calendar, Clock } from "lucide-react";
import { Button } from "@/components/ui/button";
import { cn } from "@/lib/cva.config";
import type { DateRangePickerValue } from "./date_picker_types";
@@ -291,7 +291,7 @@ const AdvancedDatePicker: React.FC = ({
>
-
+
{formatDisplayRange(value.from, value.to)}
diff --git a/ui/litellm-dashboard/src/components/team/TeamInfo.tsx b/ui/litellm-dashboard/src/components/team/TeamInfo.tsx
index 7c6f03e91aa..fa73f1d8a43 100644
--- a/ui/litellm-dashboard/src/components/team/TeamInfo.tsx
+++ b/ui/litellm-dashboard/src/components/team/TeamInfo.tsx
@@ -20,7 +20,6 @@ import { formatNumberWithCommas } from "@/utils/dataUtils";
import { mapEmptyStringToNull } from "@/utils/keyUpdateUtils";
import type { ObjectPermission } from "@/components/object_permission_types";
import { isProxyAdminRole } from "@/utils/roles";
-import { EditOutlined, InfoCircleOutlined } from "@ant-design/icons";
import { ArrowLeftIcon } from "@heroicons/react/outline";
import { StatusBadge, type StatusTone } from "@/components/shared/table_cells/status_badge";
import { Badge } from "@/components/ui/badge";
@@ -42,7 +41,7 @@ import { TagsInput } from "@/app/(dashboard)/guardrails/_components/content_filt
import { Tabs, TabsContent, TabsList, TabsTrigger } from "@/components/ui/tabs";
import { useVisitedTabs } from "@/hooks/useVisitedTabs";
import { toast } from "@/lib/toast";
-import { CheckIcon, ChevronDown, CircleMinus, CopyIcon, Plus, Save } from "lucide-react";
+import { CheckIcon, ChevronDown, CircleMinus, CopyIcon, Info, Pencil, Plus, Save } from "lucide-react";
import React, { useEffect, useMemo, useState } from "react";
import { useFieldArray } from "react-hook-form";
import { z } from "zod/v4";
@@ -1119,7 +1118,7 @@ const TeamInfoView: React.FC = ({
startEditing();
}}
>
-
+
Edit Settings
)}
@@ -1809,7 +1808,7 @@ const TeamInfoView: React.FC = ({
Team Member Settings{" "}
-
+
Max Budget: {info.team_member_budget_table?.max_budget || "No Limit"}
@@ -1970,7 +1969,7 @@ const TeamInfoView: React.FC = ({
Team Member Budget (USD){" "}
-
+
),
@@ -1985,7 +1984,7 @@ const TeamInfoView: React.FC = ({
Budget Reset Period{" "}
-
+
),
@@ -1997,7 +1996,7 @@ const TeamInfoView: React.FC = ({
Team Member TPM Limit{" "}
-
+
),
@@ -2012,7 +2011,7 @@ const TeamInfoView: React.FC = ({
Team Member RPM Limit{" "}
-
+
),
@@ -2027,7 +2026,7 @@ const TeamInfoView: React.FC = ({
Allowed Models{" "}
-
+
),
From 1140366beeadf659ed4214e7004af63490e6e4dd Mon Sep 17 00:00:00 2001
From: mateo-berri <277851410+mateo-berri@users.noreply.github.com>
Date: Wed, 19 Aug 2026 17:24:44 -0700
Subject: [PATCH 13/24] fix(vertex_ai): resolve passthrough serving location in
the logging cost recompute
---
.../vertex_passthrough_logging_handler.py | 7 ++-
.../test_llm_pass_through_endpoints.py | 56 +++++++++++++------
2 files changed, 46 insertions(+), 17 deletions(-)
diff --git a/litellm/proxy/pass_through_endpoints/llm_provider_handlers/vertex_passthrough_logging_handler.py b/litellm/proxy/pass_through_endpoints/llm_provider_handlers/vertex_passthrough_logging_handler.py
index f8e521410bf..ddcca1d372b 100644
--- a/litellm/proxy/pass_through_endpoints/llm_provider_handlers/vertex_passthrough_logging_handler.py
+++ b/litellm/proxy/pass_through_endpoints/llm_provider_handlers/vertex_passthrough_logging_handler.py
@@ -62,6 +62,8 @@ class VertexPassthroughLoggingHandler:
**kwargs,
) -> PassThroughEndpointLoggingTypedDict:
vertex_location: Final = get_vertex_location_from_url(url_route)
+ if vertex_location is not None:
+ logging_obj.optional_params["vertex_location"] = vertex_location
if "predictLongRunning" in url_route:
model = VertexPassthroughLoggingHandler.extract_model_from_url(url_route)
@@ -421,6 +423,9 @@ class VertexPassthroughLoggingHandler:
- Logs in litellm callbacks
"""
kwargs: dict[str, Any] = {}
+ vertex_location: Final = get_vertex_location_from_url(url_route)
+ if vertex_location is not None:
+ litellm_logging_obj.optional_params["vertex_location"] = vertex_location
model = model or VertexPassthroughLoggingHandler.extract_model_from_url(url_route)
complete_streaming_response: Final = VertexPassthroughLoggingHandler._build_complete_streaming_response(
all_chunks=all_chunks,
@@ -446,7 +451,7 @@ class VertexPassthroughLoggingHandler:
end_time=end_time,
logging_obj=litellm_logging_obj,
custom_llm_provider=VertexPassthroughLoggingHandler._get_custom_llm_provider_from_url(url_route),
- vertex_location=get_vertex_location_from_url(url_route),
+ vertex_location=vertex_location,
)
return {
diff --git a/tests/test_litellm/proxy/pass_through_endpoints/test_llm_pass_through_endpoints.py b/tests/test_litellm/proxy/pass_through_endpoints/test_llm_pass_through_endpoints.py
index 8da6d0fe81d..a9454854948 100644
--- a/tests/test_litellm/proxy/pass_through_endpoints/test_llm_pass_through_endpoints.py
+++ b/tests/test_litellm/proxy/pass_through_endpoints/test_llm_pass_through_endpoints.py
@@ -719,6 +719,7 @@ class TestVertexAIPassThroughHandler:
# Create mock logging object
mock_logging_obj = Mock(spec=LiteLLMLoggingObj)
+ mock_logging_obj.optional_params = {}
mock_logging_obj.litellm_call_id = "test-call-id-123"
mock_logging_obj.model_call_details = {}
@@ -895,6 +896,7 @@ class TestVertexAIPassThroughHandler:
# Create mock logging object
mock_logging_obj = Mock(spec=LiteLLMLoggingObj)
+ mock_logging_obj.optional_params = {}
mock_logging_obj.litellm_call_id = "test-call-id-123"
mock_logging_obj.model_call_details = {}
@@ -965,6 +967,7 @@ class TestVertexAIPassThroughHandler:
mock_httpx_response.status_code = 200
mock_logging_obj = Mock(spec=LiteLLMLoggingObj)
+ mock_logging_obj.optional_params = {}
mock_logging_obj.litellm_call_id = "test-call-id-embed"
mock_logging_obj.model_call_details = {}
@@ -1023,6 +1026,7 @@ class TestVertexAIPassThroughHandler:
mock_httpx_response.status_code = 200
mock_logging_obj = Mock(spec=LiteLLMLoggingObj)
+ mock_logging_obj.optional_params = {}
mock_logging_obj.litellm_call_id = "test-call-id-batch"
mock_logging_obj.model_call_details = {}
@@ -1079,6 +1083,7 @@ class TestVertexAIPassThroughHandler:
mock_httpx_response.status_code = 200
mock_logging_obj = Mock(spec=LiteLLMLoggingObj)
+ mock_logging_obj.optional_params = {}
mock_logging_obj.litellm_call_id = "test-call-id-gemini-studio"
mock_logging_obj.model_call_details = {}
@@ -1112,13 +1117,14 @@ class TestVertexAIPassThroughHandler:
@pytest.mark.parametrize("streaming", [False, True])
def test_vertex_passthrough_handler_prices_regional_endpoint_with_uplift(self, monkeypatch, streaming):
"""
- Passthrough cost is computed inside the handler and stored as response_cost before the
- logging cost resolver runs, so the handler itself must read the serving location out of
- the passthrough URL; otherwise regional Vertex passthrough traffic bills at the global
- rate (#34393).
+ Both cost computations for a passthrough call must price on the URL's serving location:
+ the handler-computed cost, and the async success recompute, which re-resolves the
+ location from the logging object and previously fell through empty optional_params to
+ the us-central1 default, billing the regional uplift on global traffic too (#34393).
"""
import datetime
+ from litellm.litellm_core_utils.litellm_logging import Logging
from litellm.proxy.pass_through_endpoints.llm_provider_handlers.vertex_passthrough_logging_handler import (
VertexPassthroughLoggingHandler,
)
@@ -1153,21 +1159,33 @@ class TestVertexAIPassThroughHandler:
},
}
- def cost_for(location: str) -> float:
+ def costs_for(location: str) -> tuple[float, float]:
url_route: Final = (
f"https://{location}-aiplatform.googleapis.com/v1/projects/p/locations/{location}"
"/publishers/google/models/gemini-fake-regional:"
f"{'streamGenerateContent' if streaming else 'generateContent'}"
)
- mock_logging_obj: Final = Mock()
- mock_logging_obj.litellm_call_id = "call-id"
- mock_logging_obj.model_call_details = {}
- mock_logging_obj.optional_params = {}
start_time: Final = datetime.datetime.now()
end_time: Final = datetime.datetime.now()
+ logging_obj: Final = Logging(
+ model="gemini-fake-regional",
+ messages=[{"role": "user", "content": "hi"}],
+ stream=streaming,
+ call_type="pass_through_endpoint",
+ start_time=start_time,
+ litellm_call_id="call-id",
+ function_id="fn-id",
+ )
+ logging_obj.update_environment_variables(
+ model="gemini-fake-regional",
+ user="unknown",
+ optional_params={},
+ litellm_params={},
+ call_type="pass_through_endpoint",
+ )
if streaming:
result = VertexPassthroughLoggingHandler._handle_logging_vertex_collected_chunks(
- litellm_logging_obj=mock_logging_obj,
+ litellm_logging_obj=logging_obj,
passthrough_success_handler_obj=Mock(),
url_route=url_route,
request_body={},
@@ -1184,22 +1202,28 @@ class TestVertexAIPassThroughHandler:
mock_httpx_response.status_code = 200
result = VertexPassthroughLoggingHandler.vertex_passthrough_handler(
httpx_response=mock_httpx_response,
- logging_obj=mock_logging_obj,
+ logging_obj=logging_obj,
url_route=url_route,
result="test-result",
start_time=start_time,
end_time=end_time,
cache_hit=False,
)
- return result["kwargs"]["response_cost"]
+ recomputed: Final = logging_obj._response_cost_calculator(result=result["result"])
+ return result["kwargs"]["response_cost"], recomputed
- global_cost: Final = cost_for("global")
- regional_cost: Final = cost_for("us-east5")
+ global_handler_cost, global_recomputed_cost = costs_for("global")
+ regional_handler_cost, regional_recomputed_cost = costs_for("us-east5")
- assert global_cost == pytest.approx(10 * 1e-06 + 20 * 2e-06, rel=1e-9)
- assert regional_cost == pytest.approx(global_cost * 1.10, rel=1e-9), (
+ plain_cost: Final = 10 * 1e-06 + 20 * 2e-06
+ assert global_handler_cost == pytest.approx(plain_cost, rel=1e-9)
+ assert regional_handler_cost == pytest.approx(plain_cost * 1.10, rel=1e-9), (
"regional Vertex passthrough traffic must bill at 1.1x the global rate"
)
+ assert global_recomputed_cost == pytest.approx(plain_cost, rel=1e-9), (
+ "the logging recompute must not price global passthrough traffic as regional"
+ )
+ assert regional_recomputed_cost == pytest.approx(plain_cost * 1.10, rel=1e-9)
class TestVertexAIDiscoveryPassThroughHandler:
From fc8a6b2a8d8537c3a564fa863a71f001855b5901 Mon Sep 17 00:00:00 2001
From: ryan-crabbe-berri
Date: Wed, 19 Aug 2026 17:25:09 -0700
Subject: [PATCH 14/24] refactor(ui): migrate shared primitives and common
components off antd (#37521)
* refactor(ui): migrate shared primitives and common components off antd
Adds the success variant to the shared Alert plus success, warning and
info variants to Badge, introduces UtcDateTimeInput to replace antd's
DatePicker, and converts the common components and key/team helpers onto
the shadcn primitives.
* fix(ui): keep MultiSelect and budget input faithful to their antd behaviour
Restore the clear-all control MultiSelect lost, split comma-separated
custom entries into one value per token, and stop rounding the budget
input on every keystroke so a fractional amount survives typing.
* test(ui): drive the access group picker through the migrated MultiSelect
AccessGroupSelector no longer renders an antd Select, so the placeholder
is an input label rather than a text node and the popup inerts the page
until it closes.
---
ui/litellm-dashboard/eslint-suppressions.json | 58 +-------------
.../components/UsagePageView.test.tsx | 4 +-
.../common_components/AccessGroupSelector.tsx | 55 ++++---------
.../RateLimitTypeFormItem.test.tsx | 34 ++------
.../check_openapi_schema.tsx | 50 ++++++++----
.../user_search_modal.test.tsx | 2 +-
.../BudgetFallbacksEditor.tsx | 46 ++++-------
.../key_team_helpers/BudgetWindowsEditor.tsx | 53 ++++++++----
.../create_key_button.integration.test.tsx | 9 +--
.../src/components/shared/Alert.test.tsx | 80 +++++++++++++++++++
.../src/components/shared/Alert.tsx | 11 ++-
.../components/shared/MultiSelect.test.tsx | 25 ++++++
.../src/components/shared/MultiSelect.tsx | 23 +++++-
.../shared/PaginatedSearchSelect.tsx | 4 +-
.../src/components/shared/SearchSelect.tsx | 7 ++
.../src/components/ui/badge.tsx | 4 +-
.../src/contexts/AntdGlobalProvider.tsx | 7 +-
17 files changed, 266 insertions(+), 206 deletions(-)
create mode 100644 ui/litellm-dashboard/src/components/shared/Alert.test.tsx
diff --git a/ui/litellm-dashboard/eslint-suppressions.json b/ui/litellm-dashboard/eslint-suppressions.json
index 31e8f81f3b3..d625cca0c60 100644
--- a/ui/litellm-dashboard/eslint-suppressions.json
+++ b/ui/litellm-dashboard/eslint-suppressions.json
@@ -542,11 +542,6 @@
"count": 1
}
},
- "src/app/(dashboard)/mcp-servers/_components/EnvVarsSection.tsx": {
- "no-restricted-imports": {
- "count": 1
- }
- },
"src/app/(dashboard)/mcp-servers/_components/IdJagFormFields.tsx": {
"no-restricted-imports": {
"count": 1
@@ -647,9 +642,6 @@
"src/app/(dashboard)/mcp-servers/_components/UserEnvVarsModal.tsx": {
"no-nested-ternary": {
"count": 2
- },
- "no-restricted-imports": {
- "count": 1
}
},
"src/app/(dashboard)/mcp-servers/_components/index.tsx": {
@@ -661,9 +653,6 @@
"local/filename-pascal-case": {
"count": 1
},
- "no-restricted-imports": {
- "count": 1
- },
"react-hooks/static-components": {
"count": 4
}
@@ -1475,11 +1464,6 @@
"count": 1
}
},
- "src/components/Settings/AdminSettings/LoggingSettings/LoggingSettings.tsx": {
- "no-restricted-imports": {
- "count": 1
- }
- },
"src/components/Settings/AdminSettings/MCPSemanticFilterSettings/MCPSemanticFilterSettings.tsx": {
"no-nested-ternary": {
"count": 1
@@ -1506,11 +1490,6 @@
"count": 1
}
},
- "src/components/Settings/AdminSettings/SSOSettings/RoleMappings.tsx": {
- "no-restricted-imports": {
- "count": 1
- }
- },
"src/components/Settings/AdminSettings/UISettings/PageVisibilitySettings.tsx": {
"react-hooks/set-state-in-render": {
"count": 1
@@ -1590,9 +1569,6 @@
"src/components/VirtualKeysPage/keyTableColumns.tsx": {
"local/filename-pascal-case": {
"count": 1
- },
- "no-restricted-imports": {
- "count": 1
}
},
"src/components/activity_metrics.tsx": {
@@ -1603,11 +1579,6 @@
"count": 1
}
},
- "src/components/add_model/AdaptiveRoutingConfig.tsx": {
- "no-restricted-imports": {
- "count": 1
- }
- },
"src/components/add_model/AddModelForm.test.tsx": {
"no-restricted-imports": {
"count": 1
@@ -1620,26 +1591,6 @@
"no-nested-ternary": {
"count": 1
},
- "no-restricted-imports": {
- "count": 2
- }
- },
- "src/components/add_model/ClassificationMethodConfig.tsx": {
- "no-restricted-imports": {
- "count": 1
- }
- },
- "src/components/add_model/ComplexityRouterConfig.tsx": {
- "no-restricted-imports": {
- "count": 1
- }
- },
- "src/components/add_model/EscalationKeywords.tsx": {
- "no-restricted-imports": {
- "count": 1
- }
- },
- "src/components/add_model/KeywordTierRules.tsx": {
"no-restricted-imports": {
"count": 1
}
@@ -1649,11 +1600,6 @@
"count": 1
}
},
- "src/components/add_model/SemanticKeywordMatching.tsx": {
- "no-restricted-imports": {
- "count": 1
- }
- },
"src/components/add_model/add_auto_router_tab.tsx": {
"local/filename-pascal-case": {
"count": 1
@@ -1672,7 +1618,7 @@
"count": 1
},
"no-restricted-imports": {
- "count": 3
+ "count": 1
}
},
"src/components/add_model/auto_router_connection_test.tsx": {
@@ -2078,7 +2024,7 @@
},
"src/components/model_add/CredentialModal.tsx": {
"no-restricted-imports": {
- "count": 2
+ "count": 1
}
},
"src/components/model_add/reuse_credentials.tsx": {
diff --git a/ui/litellm-dashboard/src/app/(dashboard)/usage/_components/components/UsagePageView.test.tsx b/ui/litellm-dashboard/src/app/(dashboard)/usage/_components/components/UsagePageView.test.tsx
index a6f46ee8242..118371aa9ac 100644
--- a/ui/litellm-dashboard/src/app/(dashboard)/usage/_components/components/UsagePageView.test.tsx
+++ b/ui/litellm-dashboard/src/app/(dashboard)/usage/_components/components/UsagePageView.test.tsx
@@ -971,8 +971,8 @@ describe("UsagePage", () => {
expect(mockUserDailyActivityCall).toHaveBeenCalled();
});
- // Should still render the data from the paginated fallback
- expect(screen.getByText("1,500")).toBeInTheDocument();
+ // Should still render the data from the paginated fallback, which lands a render after the call
+ expect(await screen.findByText("1,500")).toBeInTheDocument();
});
it("should stop showing the previous range's paginated pages while a new range is in flight", async () => {
diff --git a/ui/litellm-dashboard/src/components/common_components/AccessGroupSelector.tsx b/ui/litellm-dashboard/src/components/common_components/AccessGroupSelector.tsx
index 96f21f30385..e522e7c7e62 100644
--- a/ui/litellm-dashboard/src/components/common_components/AccessGroupSelector.tsx
+++ b/ui/litellm-dashboard/src/components/common_components/AccessGroupSelector.tsx
@@ -1,6 +1,7 @@
import React from "react";
-import { Select, Skeleton } from "antd";
import { Users } from "lucide-react";
+import { Skeleton } from "@/components/ui/skeleton";
+import { MultiSelect, type MultiSelectOption } from "@/components/shared/MultiSelect";
import { useAccessGroups, AccessGroupResponse } from "@/app/(dashboard)/hooks/accessGroups/useAccessGroups";
export interface AccessGroupSelectorProps {
@@ -12,8 +13,6 @@ export interface AccessGroupSelectorProps {
className?: string;
showLabel?: boolean;
labelText?: string;
- /** Allow clearing the selection */
- allowClear?: boolean;
}
/**
@@ -32,7 +31,6 @@ const AccessGroupSelector: React.FC = ({
className,
showLabel = false,
labelText = "Access Group",
- allowClear = true,
}) => {
const { data: accessGroups, isLoading, isError } = useAccessGroups();
@@ -45,22 +43,16 @@ const AccessGroupSelector: React.FC = ({
{labelText}
)}
-
+
);
}
// ── Build options ────────────────────────────────────────────────────────
- const options = (accessGroups ?? []).map((group: AccessGroupResponse) => ({
- label: (
-
- {group.access_group_name}{" "}
- ({group.access_group_id})
-
- ),
+ const options: MultiSelectOption[] = (accessGroups ?? []).map((group: AccessGroupResponse) => ({
+ label: group.access_group_name,
value: group.access_group_id,
- selectedLabel: group.access_group_name,
- searchText: `${group.access_group_name} ${group.access_group_id}`,
+ description: group.access_group_id,
}));
// ── Render ───────────────────────────────────────────────────────────────
@@ -71,30 +63,17 @@ const AccessGroupSelector: React.FC = ({
{labelText}
)}
-