diff --git a/ui/litellm-dashboard/src/components/templates/KeyBudgetsTableColumns.test.ts b/ui/litellm-dashboard/src/components/templates/KeyBudgetsTableColumns.test.ts index 23f4268e344..bad06814347 100644 --- a/ui/litellm-dashboard/src/components/templates/KeyBudgetsTableColumns.test.ts +++ b/ui/litellm-dashboard/src/components/templates/KeyBudgetsTableColumns.test.ts @@ -63,11 +63,12 @@ const USER_ON_TEAM_KEY_NOTE = { text: noteText("user_budget_not_applied_to_team_key"), } as const; -// Transient, not dead: the counter is merely cold and the cap blocks once one request warms it. -const COLD_MODEL_NOTE = { - code: "model_budget_fails_open", - severity: "info", - text: noteText("model_budget_fails_open"), +// Fires on every row when a custom auth callable skips the read-time checks. Not dead: the +// reservation layer still enforces the scopes it covers, so it cannot condemn a row on its own. +const CUSTOM_AUTH_SKIPS_NOTE = { + code: "custom_auth_skips_read_time_checks", + severity: "warning", + text: noteText("custom_auth_skips_read_time_checks"), } as const; const TEAM_MEMBER_AT_LIMIT: KeyBudgetEntry = { @@ -203,16 +204,25 @@ describe("cannotTrip", () => { }); }); -describe("cold per-model counters", () => { - it("stays live, because the cap blocks as soon as one request warms the counter", () => { - expect(COLD_MODEL_NOTE.severity).toBe("info"); - expect(cannotTrip({ ...HEALTHY, scope: "key_model", notes: [COLD_MODEL_NOTE] })).toBe(false); +describe("custom auth skipping read-time checks", () => { + it("does not condemn a row, because the reservation layer still covers most scopes", () => { + expect(cannotTrip({ ...HEALTHY, scope: "team", notes: [CUSTOM_AUTH_SKIPS_NOTE] })).toBe(false); }); - it("is not lumped in with a budget that is permanently inert", () => { - expect(cannotTrip({ ...HEALTHY, notes: [USER_ON_TEAM_KEY_NOTE] })).toBe(true); - expect(rowRank({ ...HEALTHY, scope: "key_model", notes: [COLD_MODEL_NOTE] })).toBeLessThan( - rowRank({ ...HEALTHY, notes: [USER_ON_TEAM_KEY_NOTE] }), + it("cannot flatten the table, since it rides every row and would otherwise kill all of them", () => { + const rows: KeyBudgetEntry[] = [ + { ...BLOCKING, notes: [CUSTOM_AUTH_SKIPS_NOTE] }, + { ...HEALTHY, notes: [CUSTOM_AUTH_SKIPS_NOTE] }, + { ...UNCONFIGURED_BUDGET, notes: [CUSTOM_AUTH_SKIPS_NOTE] }, + ]; + expect(rows.map(cannotTrip)).toStrictEqual([false, false, false]); + expect(rows.map(rowRank)).toStrictEqual([0, 2, 3]); + expect(isBlockingRow(rows[0])).toBe(true); + }); + + it("still ranks a permanently inert row below a live one carrying the same note", () => { + expect(rowRank({ ...HEALTHY, notes: [CUSTOM_AUTH_SKIPS_NOTE] })).toBeLessThan( + rowRank({ ...HEALTHY, notes: [CUSTOM_AUTH_SKIPS_NOTE, USER_ON_TEAM_KEY_NOTE] }), ); }); }); diff --git a/ui/litellm-dashboard/src/components/templates/KeyBudgetsTableColumns.tsx b/ui/litellm-dashboard/src/components/templates/KeyBudgetsTableColumns.tsx index f42dd75a571..67debc4594d 100644 --- a/ui/litellm-dashboard/src/components/templates/KeyBudgetsTableColumns.tsx +++ b/ui/litellm-dashboard/src/components/templates/KeyBudgetsTableColumns.tsx @@ -38,9 +38,10 @@ const isAlertOnly = (entry: KeyBudgetEntry): boolean => entry.enforcement === "s /** * Whether a note means the row is dead: it cannot reject a request no matter what the numbers say. - * Only a permanent property counts. `model_budget_fails_open` is deliberately not dead, because a - * cold counter is transient: the budget is live and blocks as soon as one request warms it, so - * calling it dead tells someone to ignore a cap that stops them a minute later. + * Only a permanent property of this row counts. `custom_auth_skips_read_time_checks` is deliberately + * not dead even though it says budgets go unchecked, because the reservation layer still enforces + * the scopes it covers, so which rows survive depends on scope rather than on the code. Encoding + * that split here would duplicate the resolver's coverage set and drift from it. * * Deadness is a property of the code alone. Severity does not imply it in either direction, since * dead codes appear under both values, so an unclassified code is assumed live: calling a row dead @@ -51,8 +52,8 @@ const isAlertOnly = (entry: KeyBudgetEntry): boolean => entry.enforcement === "s const CODE_KILLS_ROW: Readonly> = { alert_only: false, custom_auth_may_override_end_user_cap: false, + custom_auth_skips_read_time_checks: false, end_user_route_only: false, - model_budget_fails_open: false, per_model_counters: false, project_spend_not_tracked: true, request_tags_add_budgets: false, diff --git a/ui/litellm-dashboard/src/components/templates/key_info_view.budgets_tab.test.tsx b/ui/litellm-dashboard/src/components/templates/key_info_view.budgets_tab.test.tsx index d9f846f6c58..90b048c0000 100644 --- a/ui/litellm-dashboard/src/components/templates/key_info_view.budgets_tab.test.tsx +++ b/ui/litellm-dashboard/src/components/templates/key_info_view.budgets_tab.test.tsx @@ -233,7 +233,9 @@ const assertServerCouldEmit = (budgets: readonly KeyBudgetEntry[]): void => { // Scoped to the states the resolver defines today; a fixture simulating a newer server is // deliberately outside them and only has to satisfy the invariants that are not state-specific. if (["live", "no_counter", "unavailable"].includes(budget.spend_state)) { - expect(budget.spend_state === "live").toBe(budget.spend !== null); + // A missing per-model counter is reported as a real 0.0, because zero is what the cap will be + // compared against. Only a failed read has no number at all. + expect(budget.spend_state === "unavailable").toBe(budget.spend === null); } expect(budget.status === "unlimited").toBe(budget.max_budget === null); if (budget.spend == null || budget.max_budget == null) { @@ -242,7 +244,7 @@ const assertServerCouldEmit = (budgets: readonly KeyBudgetEntry[]): void => { expect(budget.remaining).toBeCloseTo(budget.max_budget - budget.spend, 6); } if (budget.spend_state === "no_counter") { - expect(budget.notes.map((note) => note.code)).toContain("model_budget_fails_open"); + expect(budget.notes.map((note) => note.code)).toContain("per_model_counters"); } } }; @@ -280,6 +282,14 @@ const rowFor = (panel: ReturnType, entityLabel: string): HTMLElem return row; }; +const cellUnder = (panel: ReturnType, row: HTMLElement, column: string): HTMLElement => { + const index = panel + .getAllByRole("columnheader") + .findIndex((header: HTMLElement) => header.textContent?.trim() === column); + if (index < 0) throw new Error(`no ${column} column rendered`); + return within(row).getAllByRole("cell")[index]; +}; + describe("KeyInfoView Budgets tab", () => { beforeEach(() => { apiMocks.useQuery.mockReset(); @@ -349,11 +359,12 @@ describe("KeyInfoView Budgets tab", () => { expect(row).not.toHaveTextContent("$0.00"); // The meter is the part that lies loudest: drawn at 0% it reads as untouched headroom. expect(within(row).queryByRole("meter")).not.toBeInTheDocument(); + expect(cellUnder(panel, row, "Remaining")).toHaveTextContent("-"); expect(row).toHaveTextContent("Blocks at ≥ $1,000.00"); }); - // The resolver nulls `remaining` whenever spend is null and always attaches the cold note to a - // `no_counter` reading, so a fixture without both is one the server cannot produce. + // The resolver reports a cold counter as the 0.0 it will be enforced as, and always attaches the + // per-model note to a `no_counter` reading, so a fixture without both is one the server cannot produce. const COLD_PER_MODEL: KeyBudgetEntry = { ...UNCONFIGURED_BUDGET, scope: "key_model", @@ -361,16 +372,13 @@ describe("KeyInfoView Budgets tab", () => { entity_id: "claude-opus-5", entity_label: "claude-opus-5", max_budget: 40, - spend: null, + spend: 0, spend_state: "no_counter", - remaining: null, + remaining: 40, comparison: ">", source: "key.model_max_budget[claude-opus-5]", status: "ok", - notes: [ - { code: "per_model_counters", severity: "warning", text: noteText("per_model_counters") }, - { code: "model_budget_fails_open", severity: "info", text: noteText("model_budget_fails_open") }, - ], + notes: [{ code: "per_model_counters", severity: "warning", text: noteText("per_model_counters") }], }; it("shows a genuine no-counter-yet zero as $0.00 with its meter, not as unknown", async () => { @@ -381,6 +389,9 @@ describe("KeyInfoView Budgets tab", () => { expect(row).toHaveTextContent("$0.00 of $40.00"); expect(row).not.toHaveTextContent("Unknown"); expect(within(row).getByRole("meter")).toBeInTheDocument(); + // A cold counter reports a real zero, so the whole cap is still available and the remaining + // cell says so rather than falling back to the dash it shows when spend cannot be read. + expect(cellUnder(panel, row, "Remaining")).toHaveTextContent("$40.00"); }); it("keeps a cold per-model budget live, since one request warms the counter and it starts blocking", async () => { diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index 59d5476c868..000280b2108 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -6833,16 +6833,21 @@ export interface paths { * `team_member`, `user`, `organization`, `project`, `tag`, `end_user` or `end_user_model` * - entity_type: Litellm_EntityType - The entity a `BudgetExceededError` from this scope * names, so a denial message maps back to a row here - * - entity_id / entity_label: str | None - Which entity is limited, and its human-facing alias + * - entity_id / entity_label: str | None - Which entity is limited, and its human-facing alias. + * On the per-model scopes this is one row per counter rather than per request model, so + * `entity_id` is the model whose counter was read and `entity_label` is the configured cap + * it is compared against; several request models can share one counter, and they are not + * listed separately because their spend is not separate * - enforcement: str - `hard` blocks the request, `soft` only raises an alert, `throttled` * scales the key's rate limits down instead of denying anything. Only the key's own * `max_budget` can be `throttled`; every other scope on the same key still blocks * - max_budget: float | None - The limit in effect. `null` means this scope applies to the key * but places no limit on it * - spend: float | None - Spend as the enforcing check reads it, from the same cross-pod - * counter, not the periodically-synced database column - * - spend_state: str - Whether `spend` is `live`, missing because no counter exists yet - * (`no_counter`), or missing because the read failed (`unavailable`) + * counter, not the periodically-synced database column. `null` only when the read failed + * - spend_state: str - Whether `spend` came from a counter (`live`), is the zero that will be + * enforced because no counter has been created yet (`no_counter`), or is missing because the + * read failed (`unavailable`) * - remaining: float | None - `max_budget - spend`, when both are known * - comparison: str - The operator the enforcing check uses, which differs per scope * - budget_duration / budget_reset_at / window_start: When spend next resets to zero @@ -7533,16 +7538,21 @@ export interface paths { * `team_member`, `user`, `organization`, `project`, `tag`, `end_user` or `end_user_model` * - entity_type: Litellm_EntityType - The entity a `BudgetExceededError` from this scope * names, so a denial message maps back to a row here - * - entity_id / entity_label: str | None - Which entity is limited, and its human-facing alias + * - entity_id / entity_label: str | None - Which entity is limited, and its human-facing alias. + * On the per-model scopes this is one row per counter rather than per request model, so + * `entity_id` is the model whose counter was read and `entity_label` is the configured cap + * it is compared against; several request models can share one counter, and they are not + * listed separately because their spend is not separate * - enforcement: str - `hard` blocks the request, `soft` only raises an alert, `throttled` * scales the key's rate limits down instead of denying anything. Only the key's own * `max_budget` can be `throttled`; every other scope on the same key still blocks * - max_budget: float | None - The limit in effect. `null` means this scope applies to the key * but places no limit on it * - spend: float | None - Spend as the enforcing check reads it, from the same cross-pod - * counter, not the periodically-synced database column - * - spend_state: str - Whether `spend` is `live`, missing because no counter exists yet - * (`no_counter`), or missing because the read failed (`unavailable`) + * counter, not the periodically-synced database column. `null` only when the read failed + * - spend_state: str - Whether `spend` came from a counter (`live`), is the zero that will be + * enforced because no counter has been created yet (`no_counter`), or is missing because the + * read failed (`unavailable`) * - remaining: float | None - `max_budget - spend`, when both are known * - comparison: str - The operator the enforcing check uses, which differs per scope * - budget_duration / budget_reset_at / window_start: When spend next resets to zero @@ -26418,7 +26428,7 @@ export interface components { * Code * @enum {string} */ - code: "alert_only" | "custom_auth_may_override_end_user_cap" | "end_user_route_only" | "model_budget_fails_open" | "per_model_counters" | "project_spend_not_tracked" | "request_tags_add_budgets" | "reservation_blocks_at_limit" | "rolling_window" | "throttled_instead_of_blocked" | "user_budget_not_applied_to_team_key"; + code: "alert_only" | "custom_auth_may_override_end_user_cap" | "custom_auth_skips_read_time_checks" | "end_user_route_only" | "per_model_counters" | "project_spend_not_tracked" | "request_tags_add_budgets" | "reservation_blocks_at_limit" | "rolling_window" | "throttled_instead_of_blocked" | "user_budget_not_applied_to_team_key"; /** * Severity * @enum {string}