refactor(budgets): report only the budgets that can block, and lead with a verdict

Six of the thirteen scopes could not answer "what stopped my request". The three
soft budgets only ever raise an alert, and the two per-model scopes fan out over
every model on the router rather than the key's own, so one wildcard cap on a
proxy with 500 deployments planned ~500 rows and fired ~500 cache reads for one
page. Neither has a counter the other scopes share, and the per-model ones have
no database column to fall back on either.

Dropping them takes the resolver from 1457 to 1172 lines and removes the
`_KeyModelSpend`, `_EndUserModelSpend` and `no_counter` machinery that existed
only for them, along with the router enumeration and the shared-counter collapse.
`BudgetEnforcement` is now `hard | throttled` and `BudgetSpendState` is now
`live | unavailable`.

The tab now opens with the answer rather than a grid to read: a verdict line that
names the blocking scope, or the one closest to its limit, or the scopes nobody
could read, over bullet graphs sorted by headroom. The table stays underneath as
the evidence, with every row two lines tall and caveats collapsed into their own
column, so its height no longer comes from whichever note happened to be longest.
This commit is contained in:
ryan-crabbe-berri 2026-08-20 10:27:29 -07:00
parent 7520005db6
commit 65fad58f70
12 changed files with 876 additions and 2044 deletions

View file

@ -48,7 +48,6 @@ from litellm.proxy.auth.auth_checks import (
user_budget_applies_to_key,
)
from litellm.proxy.common_utils.user_api_key_cache import UserApiKeyCache
from litellm.proxy.hooks.model_max_budget_limiter import ModelSpendHit
from litellm.proxy.spend_tracking.budget_reservation import get_budget_window_start
from litellm.proxy.spend_tracking.spend_counter_keys import (
end_user_spend_counter,
@ -73,14 +72,12 @@ from litellm.types.proxy.management_endpoints.key_management_endpoints import (
KeyBudgetEntry,
KeyBudgetNote,
)
from litellm.types.utils import BudgetConfig
_ENTITY_TYPE_BY_SCOPE: Final[Mapping[BudgetScope, Litellm_EntityType]] = MappingProxyType(
{
"proxy": Litellm_EntityType.PROXY,
"key": Litellm_EntityType.KEY,
"key_window": Litellm_EntityType.KEY,
"key_model": Litellm_EntityType.KEY,
"team": Litellm_EntityType.TEAM,
"team_window": Litellm_EntityType.TEAM,
"team_member": Litellm_EntityType.TEAM_MEMBER,
@ -89,35 +86,18 @@ _ENTITY_TYPE_BY_SCOPE: Final[Mapping[BudgetScope, Litellm_EntityType]] = Mapping
"project": Litellm_EntityType.PROJECT,
"tag": Litellm_EntityType.TAG,
"end_user": Litellm_EntityType.END_USER,
"end_user_model": Litellm_EntityType.END_USER,
}
)
_MODEL_SCOPES: Final[frozenset[BudgetScope]] = frozenset({"key_model", "end_user_model"})
_RESERVATION_COVERED_SCOPES: Final[frozenset[BudgetScope]] = frozenset(
{"key", "key_window", "team", "team_window", "team_member", "user", "organization", "tag", "end_user"}
)
_ALERT_ONLY_NOTE: Final = KeyBudgetNote(
code="alert_only",
severity="info",
text="alert only, never blocks; compared against recorded spend rather than the live counter",
)
_ROLLING_WINDOW_NOTE: Final = KeyBudgetNote(
code="rolling_window",
severity="info",
text="rolling window; the start moves with reset_at so consecutive windows can overlap",
)
_MODEL_BUDGET_NOTE: Final = KeyBudgetNote(
code="per_model_counters",
severity="warning",
text=(
"one row per counter behind this cap, because each counter is compared against it alone; these "
"counters are cache-only and fail open while one is missing, and a model this proxy cannot "
"enumerate has a counter that is not reported here"
),
)
_RESERVATION_NOTE: Final = KeyBudgetNote(
code="reservation_blocks_at_limit",
severity="info",
@ -195,14 +175,6 @@ class SpendReader(Protocol):
) -> float: ...
class ModelSpendReader(Protocol):
async def __call__(self, *, entity_id: str, model: str, budget_config: BudgetConfig) -> ModelSpendHit | None: ...
class ModelBudgetKeyMatcher(Protocol):
def __call__(self, *, model: str, configured: Mapping[str, BudgetConfig]) -> str | None: ...
async def _read_counter_spend(
*,
counter_key: str,
@ -226,54 +198,6 @@ async def _read_counter_spend(
)
async def _read_key_model_spend(*, entity_id: str, model: str, budget_config: BudgetConfig) -> ModelSpendHit | None:
from litellm.proxy.proxy_server import model_max_budget_limiter
return await model_max_budget_limiter.resolve_model_spend(
scope="virtual_key", entity_id=entity_id, model=model, budget_config=budget_config
)
async def _read_end_user_model_spend(
*, entity_id: str, model: str, budget_config: BudgetConfig
) -> ModelSpendHit | None:
from litellm.proxy.proxy_server import model_max_budget_limiter
return await model_max_budget_limiter.resolve_model_spend(
scope="end_user", entity_id=entity_id, model=model, budget_config=budget_config
)
def _match_model_budget_key(*, model: str, configured: Mapping[str, BudgetConfig]) -> str | None:
from litellm.proxy.proxy_server import model_max_budget_limiter
return model_max_budget_limiter.get_request_model_budget_key(model=model, internal_model_max_budget=configured)
def _deployment_models(model_list: Sequence[object]) -> tuple[str, ...]:
"""
``Router.deployment_names`` is appended to and never pruned, so it names deployments that were
deleted. ``model_list`` is the live registry, so read the same names back out of it instead.
"""
parsed: Final = (_validated(_DEPLOYMENT_ENTRY, entry, "model_list entry") for entry in model_list)
return tuple(entry.litellm_params.model for entry in parsed if entry is not None)
def _request_models(allowed_models: tuple[str, ...]) -> tuple[str, ...]:
"""
Model names a request could carry, since per-model counters are keyed by the request model.
Deployment names are in here because routing straight at one bypasses the model group, and a
wildcard or non-router deployment can still produce a counter no enumeration can predict.
"""
from litellm.proxy.proxy_server import llm_router
if llm_router is None:
return allowed_models
routable: Final = (*llm_router.get_model_names(), *_deployment_models(llm_router.model_list or ()))
return tuple(dict.fromkeys((*allowed_models, *routable)))
@dataclass(frozen=True, slots=True)
class KeyBudgetResolverDeps:
prisma_client: PrismaClient
@ -282,9 +206,6 @@ class KeyBudgetResolverDeps:
general_settings: Mapping[str, object]
custom_auth_enabled: bool = False
read_spend: SpendReader = field(default=_read_counter_spend)
read_key_model_spend: ModelSpendReader = field(default=_read_key_model_spend)
read_end_user_model_spend: ModelSpendReader = field(default=_read_end_user_model_spend)
match_model_budget_key: ModelBudgetKeyMatcher = field(default=_match_model_budget_key)
@dataclass(frozen=True, slots=True)
@ -315,21 +236,7 @@ class _RecordedSpend:
value: float
@dataclass(frozen=True, slots=True)
class _KeyModelSpend:
key_hash: str
model: str
budget_config: BudgetConfig
@dataclass(frozen=True, slots=True)
class _EndUserModelSpend:
end_user_id: str
model: str
budget_config: BudgetConfig
_SpendSource = _CounterSpend | _RecordedSpend | _KeyModelSpend | _EndUserModelSpend | _UnknownSpend
_SpendSource = _CounterSpend | _RecordedSpend | _UnknownSpend
@dataclass(frozen=True, slots=True)
@ -338,7 +245,6 @@ class _SpendReading:
value: float | None
state: BudgetSpendState
counter_model: str | None = None
@dataclass(frozen=True, slots=True)
@ -372,12 +278,6 @@ class _MetadataFields(BaseModel):
metadata: _MetadataTags = _MetadataTags()
class _KeyModelsFields(BaseModel):
model_config = ConfigDict(extra="ignore")
models: tuple[str, ...] = ()
class _WindowFields(BaseModel):
"""Entries stay unparsed here so that one malformed window cannot discard the rest."""
@ -386,33 +286,11 @@ class _WindowFields(BaseModel):
budget_limits: tuple[object, ...] = ()
class _ModelBudgetFields(BaseModel):
model_config = ConfigDict(extra="ignore", protected_namespaces=())
model_max_budget: Mapping[str, object] = MappingProxyType({})
class _DeploymentParamsFields(BaseModel):
model_config = ConfigDict(extra="ignore", protected_namespaces=())
model: str
class _DeploymentEntryFields(BaseModel):
model_config = ConfigDict(extra="ignore")
litellm_params: _DeploymentParamsFields
_T = TypeVar("_T")
_METADATA_FIELDS: Final = TypeAdapter(_MetadataFields)
_WINDOW_FIELDS: Final = TypeAdapter(_WindowFields)
_MODEL_BUDGET_FIELDS: Final = TypeAdapter(_ModelBudgetFields)
_DEPLOYMENT_ENTRY: Final = TypeAdapter(_DeploymentEntryFields)
_BUDGET_LIMIT_ENTRY: Final = TypeAdapter(BudgetLimitEntry)
_BUDGET_CONFIG: Final = TypeAdapter(BudgetConfig)
_KEY_MODELS: Final = TypeAdapter(_KeyModelsFields)
def _validated(adapter: TypeAdapter[_T], value: object, field_name: str) -> _T | None:
@ -429,19 +307,11 @@ class _TokenBudgetInputs:
tags: tuple[str, ...]
budget_limits: tuple[BudgetLimitEntry, ...]
model_max_budget: Mapping[str, BudgetConfig]
models: tuple[str, ...]
def _token_budget_inputs(valid_token: UserAPIKeyAuth) -> _TokenBudgetInputs:
dumped: Final = valid_token.model_dump()
container: Final = _validated(_KEY_MODELS, dumped, "models")
return _TokenBudgetInputs(
tags=_key_tags(dumped),
budget_limits=_budget_windows(dumped),
model_max_budget=_model_budgets(dumped),
models=container.models if container is not None else (),
)
return _TokenBudgetInputs(tags=_key_tags(dumped), budget_limits=_budget_windows(dumped))
def _key_tags(dumped: Mapping[str, object]) -> tuple[str, ...]:
@ -463,20 +333,6 @@ def _budget_windows(dumped: Mapping[str, object]) -> tuple[BudgetLimitEntry, ...
return tuple(entry for entry in parsed if entry is not None)
def _model_budgets(dumped: Mapping[str, object] | None) -> Mapping[str, BudgetConfig]:
"""Same as ``_budget_windows``: drop only the per-model caps that fail to parse."""
if dumped is None:
return MappingProxyType({})
container: Final = _validated(_MODEL_BUDGET_FIELDS, dumped, "model_max_budget")
if container is None:
return MappingProxyType({})
parsed: Final = (
(model, _validated(_BUDGET_CONFIG, config, "model_max_budget"))
for model, config in container.model_max_budget.items()
)
return MappingProxyType({model: config for model, config in parsed if config is not None})
@dataclass(frozen=True, slots=True)
class _ProxyBudget:
spend: float
@ -498,8 +354,6 @@ class _KeyBudgetContext:
end_user_id: str | None
custom_auth_enabled: bool
custom_auth_skips_checks: bool
request_models: tuple[str, ...]
match_model_budget_key: ModelBudgetKeyMatcher
general_settings: Mapping[str, object]
proxy: _ProxyBudget | _Unavailable | None
team: LiteLLM_TeamTable | None
@ -532,45 +386,10 @@ async def resolve_key_budgets(
reservation_enabled=reservation_enabled,
proxy_notes=proxy_notes,
)
for plan, reading in _collapse_shared_counters(tuple(zip(plans, readings, strict=True)))
for plan, reading in zip(plans, readings, strict=True)
)
def _counter_identity(index: int, plan: _PlannedBudget, reading: _SpendReading) -> object:
"""Two rows share an identity only when enforcement compares both against the same counter."""
if plan.scope not in _MODEL_SCOPES:
return index
return (plan.source, reading.counter_model)
def _collapse_shared_counters(pairs: tuple[_ReadPlan, ...]) -> tuple[_ReadPlan, ...]:
"""
One counter, one row.
A per-model cap is planned once per request model routed to it, but the reader falls back to the
provider-stripped name, so several of those models read one counter. Reporting each of them would
show a single balance several times over as though it were several balances.
"""
identities: Final = tuple(
_counter_identity(index=index, plan=plan, reading=reading) for index, (plan, reading) in enumerate(pairs)
)
first_index: Final = {identity: index for index, identity in reversed(tuple(enumerate(identities)))}
return tuple(pair for index, pair in enumerate(pairs) if first_index[identities[index]] == index)
def _model_reading(hit: ModelSpendHit | None) -> _SpendReading:
"""
A missing per-model counter is reported as no spend, because that is what will be enforced.
The check reading it treats the absence as untouched headroom rather than as an error, so zero is
the number the cap will actually be compared against. ``spend_state`` still says the counter does
not exist, which is the part a reader needs to tell this apart from a counter that says zero.
"""
if hit is None:
return _SpendReading(value=0.0, state="no_counter")
return _SpendReading(value=hit.spend, state="live", counter_model=hit.model)
async def _read_spend(plan: _PlannedBudget, deps: KeyBudgetResolverDeps) -> _SpendReading:
source: Final = plan.spend_source
try:
@ -592,18 +411,6 @@ async def _read_spend(plan: _PlannedBudget, deps: KeyBudgetResolverDeps) -> _Spe
),
state="live",
)
case _KeyModelSpend():
return _model_reading(
await deps.read_key_model_spend(
entity_id=source.key_hash, model=source.model, budget_config=source.budget_config
)
)
case _EndUserModelSpend():
return _model_reading(
await deps.read_end_user_model_spend(
entity_id=source.end_user_id, model=source.model, budget_config=source.budget_config
)
)
case _:
assert_never(source)
except Exception: # noqa: BLE001 # one unreadable counter must not blank the whole report
@ -668,7 +475,7 @@ def _to_entry(
return KeyBudgetEntry(
scope=plan.scope,
entity_type=_ENTITY_TYPE_BY_SCOPE[plan.scope],
entity_id=reading.counter_model or plan.entity_id,
entity_id=plan.entity_id,
entity_label=plan.entity_label,
enforcement=plan.enforcement,
max_budget=plan.max_budget,
@ -726,8 +533,6 @@ async def _load_context(
custom_auth_skips_checks=(
deps.custom_auth_enabled and deps.general_settings.get("custom_auth_run_common_checks") is not True
),
request_models=_request_models(token_inputs.models),
match_model_budget_key=deps.match_model_budget_key,
general_settings=deps.general_settings,
proxy=loaded_proxy,
team=team,
@ -1006,7 +811,6 @@ def _plan_budgets(context: _KeyBudgetContext) -> tuple[_PlannedBudget, ...]:
*_plan_proxy(context),
*_plan_key(context),
*_plan_key_windows(context),
*_plan_key_models(context),
*_plan_team(context),
*_plan_team_windows(context),
*_plan_team_member(context),
@ -1121,20 +925,7 @@ def _plan_key(context: _KeyBudgetContext) -> tuple[_PlannedBudget, ...]:
budget_reset_at=token.budget_reset_at,
notes=(_THROTTLE_NOTE,) if throttled else (),
)
soft: Final = _PlannedBudget(
scope="key",
entity_id=token.key_alias,
entity_label=token.key_alias,
enforcement="soft",
max_budget=token.soft_budget,
comparison=">=",
source=f"budget_table:{token.budget_id}.soft_budget" if token.budget_id else "key.budget_id.soft_budget",
spend_source=_RecordedSpend(token.spend or 0.0),
budget_duration=token.budget_duration,
budget_reset_at=token.budget_reset_at,
notes=(_ALERT_ONLY_NOTE,),
)
return (hard, soft)
return (hard,)
def _plan_key_windows(context: _KeyBudgetContext) -> tuple[_PlannedBudget, ...]:
@ -1164,44 +955,6 @@ def _plan_key_windows(context: _KeyBudgetContext) -> tuple[_PlannedBudget, ...]:
)
def _model_counter_candidates(
config_key: str, configured: Mapping[str, BudgetConfig], context: _KeyBudgetContext
) -> tuple[str, ...]:
"""Counters are keyed by the request model, so a config key has to be probed under each model routed to it."""
routed: Final = (
model
for model in context.request_models
if context.match_model_budget_key(model=model, configured=configured) == config_key
)
return tuple(dict.fromkeys((config_key, *routed)))
def _plan_key_models(context: _KeyBudgetContext) -> tuple[_PlannedBudget, ...]:
token: Final = context.valid_token
configured: Final = context.token_inputs.model_max_budget
return tuple(
_PlannedBudget(
scope="key_model",
entity_id=request_model,
entity_label=cap_key,
enforcement="hard",
max_budget=_positive_or_none(config.max_budget),
comparison=">",
source=f"key.model_max_budget[{cap_key}]",
spend_source=_KeyModelSpend(key_hash=token.token or "", model=request_model, budget_config=config),
budget_duration=config.budget_duration,
notes=(_MODEL_BUDGET_NOTE,),
)
for cap_key, config in configured.items()
for request_model in _model_counter_candidates(config_key=cap_key, configured=configured, context=context)
)
def _positive_or_none(max_budget: float | None) -> float | None:
"""Several checks treat a non-positive cap as 'unset' rather than as an immediate block."""
return max_budget if max_budget is not None and max_budget > 0 else None
def _plan_team(context: _KeyBudgetContext) -> tuple[_PlannedBudget, ...]:
team: Final = context.team
if team is None:
@ -1218,20 +971,7 @@ def _plan_team(context: _KeyBudgetContext) -> tuple[_PlannedBudget, ...]:
budget_duration=team.budget_duration,
budget_reset_at=team.budget_reset_at,
)
soft: Final = _PlannedBudget(
scope="team",
entity_id=team.team_id,
entity_label=team.team_alias,
enforcement="soft",
max_budget=team.soft_budget,
comparison=">=",
source="team.soft_budget",
spend_source=_RecordedSpend(team.spend or 0.0),
budget_duration=team.budget_duration,
budget_reset_at=team.budget_reset_at,
notes=(_ALERT_ONLY_NOTE,),
)
return (hard, soft)
return (hard,)
def _plan_team_windows(context: _KeyBudgetContext) -> tuple[_PlannedBudget, ...]:
@ -1316,6 +1056,11 @@ def _plan_user(context: _KeyBudgetContext) -> tuple[_PlannedBudget, ...]:
)
def _positive_or_none(max_budget: float | None) -> float | None:
"""Several checks treat a non-positive cap as 'unset' rather than as an immediate block."""
return max_budget if max_budget is not None and max_budget > 0 else None
def _plan_organization(context: _KeyBudgetContext) -> tuple[_PlannedBudget, ...]:
org: Final = context.organization
if org is None or org.organization_id is None:
@ -1359,20 +1104,7 @@ def _plan_project(context: _KeyBudgetContext) -> tuple[_PlannedBudget, ...]:
budget_reset_at=meta.budget_reset_at,
notes=(_PROJECT_SPEND_NOTE,),
)
soft: Final = _PlannedBudget(
scope="project",
entity_id=project.project_id,
entity_label=project.project_alias,
enforcement="soft",
max_budget=linked.soft_budget if linked is not None else None,
comparison=">=",
source=f"budget_table:{project.budget_id}.soft_budget" if project.budget_id else "project.budget_id",
spend_source=_RecordedSpend(project.spend or 0.0),
budget_duration=meta.budget_duration,
budget_reset_at=meta.budget_reset_at,
notes=(_ALERT_ONLY_NOTE,),
)
return (hard, soft)
return (hard,)
def _plan_tags(context: _KeyBudgetContext) -> tuple[_PlannedBudget, ...]:
@ -1437,21 +1169,4 @@ def _plan_end_user(context: _KeyBudgetContext) -> tuple[_PlannedBudget, ...]:
else (_END_USER_ROUTE_NOTE,)
),
)
configured: Final = _model_budgets(budget.model_dump() if budget is not None else None)
per_model: Final = tuple(
_PlannedBudget(
scope="end_user_model",
entity_id=request_model,
entity_label=cap_key,
enforcement="hard",
max_budget=_positive_or_none(config.max_budget),
comparison=">",
source=f"{row_source}.model_max_budget[{cap_key}]",
spend_source=_EndUserModelSpend(end_user_id=end_user_id, model=request_model, budget_config=config),
budget_duration=config.budget_duration,
notes=(_MODEL_BUDGET_NOTE,),
)
for cap_key, config in configured.items()
for request_model in _model_counter_candidates(config_key=cap_key, configured=configured, context=context)
)
return (primary, *per_model)
return (primary,)

View file

@ -3817,26 +3817,21 @@ async def key_budgets_fn(
`team_member`, `user`, `organization`, `project`, `tag`, `end_user` or `end_user_model`
- entity_type: Litellm_EntityType - The entity a `BudgetExceededError` from this scope
names, so a denial message maps back to a row here
- entity_id / entity_label: str | None - Which entity is limited, and its human-facing alias.
On the per-model scopes this is one row per counter rather than per request model, so
`entity_id` is the model whose counter was read and `entity_label` is the configured cap
it is compared against; several request models can share one counter, and they are not
listed separately because their spend is not separate
- enforcement: str - `hard` blocks the request, `soft` only raises an alert, `throttled`
scales the key's rate limits down instead of denying anything. Only the key's own
`max_budget` can be `throttled`; every other scope on the same key still blocks
- entity_id / entity_label: str | None - Which entity is limited, and its human-facing alias
- enforcement: str - `hard` blocks the request, `throttled` scales the key's rate limits down
instead of denying anything. Only the key's own `max_budget` can be `throttled`; every
other scope on the same key still blocks
- max_budget: float | None - The limit in effect. `null` means this scope applies to the key
but places no limit on it
- spend: float | None - Spend as the enforcing check reads it, from the same cross-pod
counter, not the periodically-synced database column. `null` only when the read failed
- spend_state: str - Whether `spend` came from a counter (`live`), is the zero that will be
enforced because no counter has been created yet (`no_counter`), or is missing because the
read failed (`unavailable`)
- spend_state: str - Whether `spend` was read (`live`) or is missing because the entity or its
counter could not be read (`unavailable`)
- remaining: float | None - `max_budget - spend`, when both are known
- comparison: str - The operator the enforcing check uses, which differs per scope
- budget_duration / budget_reset_at / window_start: When spend next resets to zero
- source: str - Where the limit is configured, e.g. `key.max_budget`, `budget_table:<id>`
- status: str - `unlimited`, `ok` or `exceeded`
- status: str - `unlimited`, `ok`, `exceeded`, or `unknown` when the row could not be evaluated
- notes: list - Caveats worth knowing before trusting the row, each with a stable `code`
to branch on and human-facing `text` that is free to be reworded. `severity` is for a
`code` a client does not know yet: `info` only explains a field the row already carries,

View file

@ -114,7 +114,6 @@ BudgetScope = Literal[
"proxy",
"key",
"key_window",
"key_model",
"team",
"team_window",
"team_member",
@ -123,22 +122,19 @@ BudgetScope = Literal[
"project",
"tag",
"end_user",
"end_user_model",
]
BudgetEnforcement = Literal["hard", "soft", "throttled"]
BudgetEnforcement = Literal["hard", "throttled"]
BudgetComparison = Literal[">=", ">"]
BudgetStatus = Literal["unlimited", "ok", "exceeded", "unknown"]
BudgetNoteCode = Literal[
"alert_only",
"custom_auth_may_override_end_user_cap",
"custom_auth_skips_read_time_checks",
"end_user_route_only",
"entity_unavailable",
"per_model_counters",
"project_spend_not_tracked",
"request_tags_add_budgets",
"reservation_blocks_at_limit",
@ -149,7 +145,7 @@ BudgetNoteCode = Literal[
BudgetNoteSeverity = Literal["info", "warning"]
BudgetSpendState = Literal["live", "no_counter", "unavailable"]
BudgetSpendState = Literal["live", "unavailable"]
class KeyBudgetNote(BaseModel):

View file

@ -0,0 +1,141 @@
import { render, screen, within } from "@testing-library/react";
import { describe, expect, it } from "vitest";
import type { KeyBudgetEntry } from "@/app/(dashboard)/hooks/keys/useKeyBudgets";
import { KeyBudgetsBulletChart, plottable, utilization } from "./KeyBudgetsBulletChart";
// The chart answers "what stopped my request" before the table lists the evidence, so every test
// here is about a wrong answer being worse than no answer: a scope ruled out that should not be.
const BASE = {
scope: "key",
entity_type: "key",
entity_id: null,
entity_label: null,
enforcement: "hard",
max_budget: null,
spend: 0,
remaining: null,
comparison: ">=",
budget_duration: null,
budget_reset_at: null,
window_start: null,
source: "key.max_budget",
status: "unlimited",
spend_state: "live",
notes: [],
} as KeyBudgetEntry;
const OK: KeyBudgetEntry = { ...BASE, status: "ok" };
const TEAM_HALF: KeyBudgetEntry = { ...OK, scope: "team", max_budget: 100, spend: 50, remaining: 50 };
const ORG_NEARLY: KeyBudgetEntry = { ...OK, scope: "organization", max_budget: 100, spend: 88, remaining: 12 };
const USER_TENTH: KeyBudgetEntry = { ...OK, scope: "user", max_budget: 100, spend: 10, remaining: 90 };
const MEMBER_OVER: KeyBudgetEntry = {
...OK,
scope: "team_member",
max_budget: 100,
spend: 140,
remaining: -40,
status: "exceeded",
};
const UNLIMITED_KEY: KeyBudgetEntry = { ...OK, spend: 900, status: "unlimited" };
const ZERO_CAP: KeyBudgetEntry = { ...OK, max_budget: 0, spend: 5, remaining: -5 };
const TINY_SHARE: KeyBudgetEntry = { ...OK, scope: "tag", max_budget: 1000, spend: 0.4, remaining: 999.6 };
const UNREADABLE_TEAM: KeyBudgetEntry = {
...OK,
scope: "team",
spend: null,
spend_state: "unavailable",
status: "unknown",
notes: [{ code: "entity_unavailable", severity: "warning", text: "could not read" }],
};
const DEAD_PROJECT: KeyBudgetEntry = {
...OK,
scope: "project",
max_budget: 100,
spend: 99,
remaining: 1,
notes: [{ code: "project_spend_not_tracked", severity: "warning", text: "never increments" }],
};
const verdict = () => screen.getByTestId("key-budgets-verdict");
describe("utilization", () => {
it("is the fraction of the limit spent", () => {
expect(utilization(ORG_NEARLY)).toBeCloseTo(0.88, 6);
});
it("refuses to invent a fraction for a row with no limit or no reading", () => {
expect(utilization(UNLIMITED_KEY)).toBeNull();
expect(utilization(UNREADABLE_TEAM)).toBeNull();
// A zero cap is "unset" to every enforcing check, so dividing by it would plot a phantom bar.
expect(utilization(ZERO_CAP)).toBeNull();
});
});
describe("plottable", () => {
it("orders by how close each budget is to its limit, worst first", () => {
const ordered = plottable([USER_TENTH, ORG_NEARLY, TEAM_HALF]).map((entry) => entry.scope);
expect(ordered).toStrictEqual(["organization", "team", "user"]);
});
it("leaves out a budget that structurally cannot trip, however close to its cap it looks", () => {
expect(utilization(DEAD_PROJECT)).toBeCloseTo(0.99, 6);
expect(plottable([DEAD_PROJECT, TEAM_HALF]).map((entry) => entry.scope)).toStrictEqual(["team"]);
});
it("leaves out rows with nothing to draw rather than drawing them at zero", () => {
expect(plottable([UNLIMITED_KEY, UNREADABLE_TEAM])).toHaveLength(0);
});
});
describe("KeyBudgetsBulletChart", () => {
it("names the budget that is blocking, not merely that something is", () => {
render(<KeyBudgetsBulletChart budgets={[TEAM_HALF, MEMBER_OVER, UNLIMITED_KEY]} />);
expect(verdict()).toHaveTextContent("Blocked by Team member");
expect(verdict()).not.toHaveTextContent("Nothing is blocking");
expect(screen.getByTestId("key-budget-bullet-blocking")).toBeInTheDocument();
});
it("names the budget closest to its limit when nothing is blocking", () => {
render(<KeyBudgetsBulletChart budgets={[USER_TENTH, ORG_NEARLY, TEAM_HALF]} />);
expect(verdict()).toHaveTextContent("Nothing is blocking this key.");
expect(verdict()).toHaveTextContent("Closest to its limit: Organization, 88% used.");
expect(screen.queryByTestId("key-budget-bullet-blocking")).not.toBeInTheDocument();
});
it("names a scope nobody could read, since a verdict that ignores it is a verdict ruling it out", () => {
render(<KeyBudgetsBulletChart budgets={[TEAM_HALF, UNREADABLE_TEAM]} />);
expect(verdict()).toHaveTextContent("1 scope could not be read");
expect(verdict()).toHaveTextContent("Team");
});
it("draws each bar in proportion to its budget, and never past the end of its track", () => {
render(<KeyBudgetsBulletChart budgets={[ORG_NEARLY, MEMBER_OVER]} />);
const [over, nearly] = [screen.getByTestId("key-budget-bullet-blocking"), screen.getByTestId("key-budget-bullet")];
expect(nearly).toHaveStyle({ width: "88%" });
// 140% of the cap would otherwise render as a bar overflowing its own track.
expect(over).toHaveStyle({ width: "100%" });
});
it("keeps a sub-percent balance visible instead of rounding it away to 0%", () => {
render(<KeyBudgetsBulletChart budgets={[TINY_SHARE]} />);
expect(screen.getByTestId("key-budgets-chart")).toHaveTextContent("<1%");
});
it("plots nothing at all rather than an empty grid when no budget carries a limit", () => {
render(<KeyBudgetsBulletChart budgets={[UNLIMITED_KEY]} />);
const chart = screen.getByTestId("key-budgets-chart");
expect(within(chart).queryByTestId("key-budget-bullet")).not.toBeInTheDocument();
expect(verdict()).toHaveTextContent("Nothing is blocking this key.");
expect(verdict()).not.toHaveTextContent("Closest to its limit");
});
});

View file

@ -0,0 +1,138 @@
"use client";
import type { KeyBudgetEntry } from "@/app/(dashboard)/hooks/keys/useKeyBudgets";
import { CellTooltip } from "@/components/shared/table_cells";
import { cn } from "@/lib/cva.config";
import { formatNumberWithCommas } from "@/utils/dataUtils";
import { budgetThresholdRule, cannotTrip, isBlockingRow, scopeLabel } from "./KeyBudgetsTableColumns";
/**
* Fraction of its limit a row has spent, or null when the row has no limit or no reading.
*
* A row without both numbers cannot be drawn at all, and drawing it at zero would read as untouched
* headroom on a budget that may be exhausted.
*/
export const utilization = (entry: KeyBudgetEntry): number | null => {
if (entry.max_budget == null || entry.max_budget <= 0 || entry.spend == null) return null;
return entry.spend / entry.max_budget;
};
/** Rows worth plotting: a real limit, a real reading, and the ability to reject a request. */
export const plottable = (entries: readonly KeyBudgetEntry[]): readonly KeyBudgetEntry[] =>
[...entries]
.filter((entry) => utilization(entry) != null && !cannotTrip(entry))
.sort((a, b) => (utilization(b) ?? 0) - (utilization(a) ?? 0));
const BAND_BOUNDS = { comfortable: 0.7, tight: 0.9 } as const;
const measureTone = (fraction: number, blocking: boolean): string => {
if (blocking) return "bg-red-500";
if (fraction >= BAND_BOUNDS.tight) return "bg-amber-500";
return "bg-sky-500";
};
const percentLabel = (fraction: number): string => {
const percent = fraction * 100;
if (percent > 0 && percent < 1) return "<1%";
return `${formatNumberWithCommas(percent, percent >= 10 ? 0 : 1)}%`;
};
/**
* One bullet graph row: qualitative bands behind a measure bar, with the limit as the track's end.
*
* The bands are what carry "how close is close", which is the question a spend column next to a
* limit column makes the reader answer by subtraction.
*/
function BulletRow({ entry }: { entry: KeyBudgetEntry }) {
const fraction = utilization(entry) ?? 0;
const blocking = isBlockingRow(entry);
const rule = budgetThresholdRule(entry);
return (
<div className="grid grid-cols-[10rem_1fr_9rem_3.5rem] items-center gap-3 text-xs">
<span className="truncate font-medium text-foreground" title={scopeLabel(entry)}>
{scopeLabel(entry)}
</span>
<CellTooltip
content={rule ?? undefined}
trigger={
<div className="relative h-3 w-full overflow-hidden rounded-sm bg-muted">
<div
className="absolute inset-y-0 left-0 bg-amber-100"
style={{ left: `${BAND_BOUNDS.comfortable * 100}%`, right: `${(1 - BAND_BOUNDS.tight) * 100}%` }}
/>
<div
className="absolute inset-y-0 right-0 bg-red-100"
style={{ width: `${(1 - BAND_BOUNDS.tight) * 100}%` }}
/>
<div
className={cn("absolute inset-y-0.5 left-0 rounded-sm", measureTone(fraction, blocking))}
style={{ width: `${Math.min(fraction, 1) * 100}%` }}
data-testid={blocking ? "key-budget-bullet-blocking" : "key-budget-bullet"}
/>
<div className="absolute inset-y-0 right-0 w-0.5 bg-foreground/70" />
</div>
}
/>
<span className="truncate tabular-nums text-muted-foreground">
${formatNumberWithCommas(entry.spend ?? 0, 2)} of ${formatNumberWithCommas(entry.max_budget ?? 0, 2)}
</span>
<span className={cn("text-right tabular-nums", blocking ? "font-medium text-red-600" : "text-muted-foreground")}>
{percentLabel(fraction)}
</span>
</div>
);
}
/**
* The one-line answer to "what stopped my request", stated before the evidence under it.
*
* An unreadable scope is named here rather than left to the table, because a verdict that ignores
* the rows nobody could read is a verdict that rules them out.
*/
function Verdict({ budgets }: { budgets: readonly KeyBudgetEntry[] }) {
const blocking = budgets.filter(isBlockingRow);
const unknown = budgets.filter((entry) => entry.status === "unknown");
const closest = plottable(budgets).find((entry) => !isBlockingRow(entry));
const closestFraction = closest ? utilization(closest) : null;
if (blocking.length > 0) {
return (
<p className="text-sm font-medium text-red-600" data-testid="key-budgets-verdict">
Blocked by {blocking.map(scopeLabel).join(", ")}
</p>
);
}
return (
<div className="flex flex-col gap-0.5" data-testid="key-budgets-verdict">
<p className="text-sm font-medium text-foreground">Nothing is blocking this key.</p>
{closest && closestFraction != null && (
<p className="text-xs text-muted-foreground">
Closest to its limit: {scopeLabel(closest)}, {percentLabel(closestFraction)} used.
</p>
)}
{unknown.length > 0 && (
<p className="text-xs text-amber-600">
{unknown.length} {unknown.length === 1 ? "scope" : "scopes"} could not be read, so nothing on{" "}
{unknown.length === 1 ? "it" : "them"} can be ruled out: {unknown.map(scopeLabel).join(", ")}.
</p>
)}
</div>
);
}
export function KeyBudgetsBulletChart({ budgets }: { budgets: readonly KeyBudgetEntry[] }) {
const rows = plottable(budgets);
return (
<div className="flex flex-col gap-3 rounded-md border border-border bg-card p-4" data-testid="key-budgets-chart">
<Verdict budgets={budgets} />
{rows.length > 0 && (
<div className="flex flex-col gap-2">
{rows.map((entry) => (
<BulletRow key={`${entry.scope}:${entry.entity_id ?? ""}`} entry={entry} />
))}
</div>
)}
</div>
);
}

View file

@ -5,8 +5,10 @@ import { useMemo } from "react";
import { useKeyBudgets, type KeyBudgetEntry } from "@/app/(dashboard)/hooks/keys/useKeyBudgets";
import { Alert, AlertDescription, AlertTitle } from "@/components/shared/Alert";
import { DataTable } from "@/components/shared/DataTable";
import { cn } from "@/lib/cva.config";
import { parseErrorMessage } from "../shared/errorUtils";
import { KeyBudgetsBulletChart } from "./KeyBudgetsBulletChart";
import { getKeyBudgetsTableColumns, isBlockingRow, rowRank } from "./KeyBudgetsTableColumns";
function BudgetRows({ budgets, isLoading }: { budgets: readonly KeyBudgetEntry[]; isLoading: boolean }) {
@ -14,16 +16,21 @@ function BudgetRows({ budgets, isLoading }: { budgets: readonly KeyBudgetEntry[]
const rows = useMemo(() => [...budgets].sort((a, b) => rowRank(a) - rowRank(b)), [budgets]);
return (
<DataTable
data={rows}
columns={columns}
getRowId={(entry, index) => `${entry.scope}:${entry.entity_id ?? ""}:${index}`}
isLoading={isLoading}
loadingMessage="Loading budgets…"
noDataMessage="No budgets apply to this key."
rowClassName={(row) => (isBlockingRow(row.original) ? "bg-red-50 hover:bg-red-50" : "")}
size="compact"
/>
<>
{!isLoading && budgets.length > 0 && <KeyBudgetsBulletChart budgets={budgets} />}
<DataTable
data={rows}
columns={columns}
getRowId={(entry, index) => `${entry.scope}:${entry.entity_id ?? ""}:${index}`}
isLoading={isLoading}
loadingMessage="Loading budgets…"
noDataMessage="No budgets apply to this key."
// Every row is two lines tall whatever it carries, so the table reads as a grid rather than
// taking its height from whichever caveat happened to be longest.
rowClassName={(row) => cn("h-14", isBlockingRow(row.original) ? "bg-red-50 hover:bg-red-50" : "")}
size="compact"
/>
</>
);
}

View file

@ -31,12 +31,6 @@ const PROJECT_DEAD_NOTE = {
text: noteText("project_spend_not_tracked"),
} as const;
const ALERT_ONLY_NOTE = {
code: "alert_only",
severity: "info",
text: noteText("alert_only"),
} as const;
const ROLLING_NOTE = {
code: "rolling_window",
severity: "info",
@ -105,24 +99,14 @@ const RESERVED_TEAM_AT_300: KeyBudgetEntry = {
const UNRESERVED_TEAM_AT_300: KeyBudgetEntry = { ...RESERVED_TEAM_AT_300, comparison: ">", status: "ok" };
const SOFT_OVER: KeyBudgetEntry = {
...UNCONFIGURED_BUDGET,
enforcement: "soft",
comparison: ">=",
max_budget: 500,
spend: 900,
remaining: -400,
status: "exceeded",
};
const SUB_DOLLAR_LIMIT: KeyBudgetEntry = { ...UNCONFIGURED_BUDGET, comparison: ">", max_budget: 0.1 };
const BLOCKING: KeyBudgetEntry = { ...UNCONFIGURED_BUDGET, status: "exceeded", enforcement: "hard" };
const ALERT_ONLY: KeyBudgetEntry = {
const THROTTLING: KeyBudgetEntry = {
...UNCONFIGURED_BUDGET,
status: "exceeded",
enforcement: "soft",
notes: [ALERT_ONLY_NOTE],
enforcement: "throttled",
notes: [THROTTLE_NOTE],
};
const HEALTHY: KeyBudgetEntry = { ...UNCONFIGURED_BUDGET, status: "ok", enforcement: "hard" };
const INERT: KeyBudgetEntry = {
@ -150,8 +134,9 @@ describe("budgetThresholdRule", () => {
expect(budgetThresholdRule(UNRESERVED_TEAM_AT_300)).toBe("Blocks at > $300.00");
});
it("never promises a soft budget will block", () => {
expect(budgetThresholdRule(SOFT_OVER)).toBe("Alerts at ≥ $500.00");
it("never promises a throttling budget will block", () => {
const throttlingOver: KeyBudgetEntry = { ...THROTTLING, comparison: ">=", max_budget: 500 };
expect(budgetThresholdRule(throttlingOver)).toBe("Throttles at ≥ $500.00");
});
it("states no threshold for a scope with nothing configured", () => {
@ -168,8 +153,8 @@ describe("cannotTrip", () => {
expect(cannotTrip(INERT)).toBe(true);
});
it("does not call a soft budget dead, since alert_only only restates the enforcement column", () => {
expect(cannotTrip(ALERT_ONLY)).toBe(false);
it("does not call a throttling budget dead, since the note only restates the enforcement column", () => {
expect(cannotTrip(THROTTLING)).toBe(false);
});
it("leaves a warning note trippable, because it qualifies the number rather than killing it", () => {
@ -259,7 +244,7 @@ describe("isThrottled", () => {
describe("isBlockingRow", () => {
it("counts only a hard budget that is over as blocking", () => {
expect(isBlockingRow(BLOCKING)).toBe(true);
expect(isBlockingRow(ALERT_ONLY)).toBe(false);
expect(isBlockingRow(THROTTLING)).toBe(false);
expect(isBlockingRow(HEALTHY)).toBe(false);
expect(isBlockingRow(UNCONFIGURED_BUDGET)).toBe(false);
});
@ -309,15 +294,15 @@ describe("a scope the server could not resolve", () => {
});
describe("rowRank", () => {
it("ranks a blocking budget above an alert-only one, and both above healthy and unlimited", () => {
expect(rowRank(BLOCKING)).toBeLessThan(rowRank(ALERT_ONLY));
expect(rowRank(ALERT_ONLY)).toBeLessThan(rowRank(HEALTHY));
it("ranks a blocking budget above a throttling one, and both above healthy and unlimited", () => {
expect(rowRank(BLOCKING)).toBeLessThan(rowRank(THROTTLING));
expect(rowRank(THROTTLING)).toBeLessThan(rowRank(HEALTHY));
expect(rowRank(HEALTHY)).toBeLessThan(rowRank(UNCONFIGURED_BUDGET));
});
it("sinks a row that cannot trip below every row that can, since it never stopped anything", () => {
expect(rowRank(INERT)).toBeGreaterThan(rowRank(BLOCKING));
expect(rowRank(INERT)).toBeGreaterThan(rowRank(ALERT_ONLY));
expect(rowRank(INERT)).toBeGreaterThan(rowRank(THROTTLING));
expect(rowRank(INERT)).toBeGreaterThan(rowRank(HEALTHY));
expect(rowRank(INERT)).toBeGreaterThan(rowRank(UNCONFIGURED_BUDGET));
});

View file

@ -16,13 +16,13 @@ import {
StatusBadge,
type StatusTone,
} from "@/components/shared/table_cells";
import { cn } from "@/lib/cva.config";
import { formatNumberWithCommas } from "@/utils/dataUtils";
const SCOPE_LABELS: Record<string, string> = {
proxy: "Proxy",
key: "Key",
key_window: "Key window",
key_model: "Key per-model",
team: "Team",
team_window: "Team window",
team_member: "Team member",
@ -31,10 +31,9 @@ const SCOPE_LABELS: Record<string, string> = {
project: "Project",
tag: "Tag",
end_user: "End user",
end_user_model: "End user per-model",
};
const isAlertOnly = (entry: KeyBudgetEntry): boolean => entry.enforcement === "soft";
export const scopeLabel = (entry: KeyBudgetEntry): string => SCOPE_LABELS[entry.scope] ?? entry.scope;
/**
* Whether a note means the row is dead: it cannot reject a request no matter what the numbers say.
@ -50,12 +49,10 @@ const isAlertOnly = (entry: KeyBudgetEntry): boolean => entry.enforcement === "s
* fails this build until someone classifies it.
*/
const CODE_KILLS_ROW: Readonly<Record<KeyBudgetNoteCode, boolean>> = {
alert_only: false,
custom_auth_may_override_end_user_cap: false,
custom_auth_skips_read_time_checks: false,
end_user_route_only: false,
entity_unavailable: false,
per_model_counters: false,
project_spend_not_tracked: true,
request_tags_add_budgets: false,
reservation_blocks_at_limit: false,
@ -75,7 +72,7 @@ export const cannotTrip = (entry: KeyBudgetEntry): boolean => entry.notes.some(n
export const isThrottled = (entry: KeyBudgetEntry): boolean => entry.enforcement === "throttled";
/** Whether going over this budget rejects a request, as opposed to alerting, throttling or nothing. */
const canDeny = (entry: KeyBudgetEntry): boolean => !isAlertOnly(entry) && !cannotTrip(entry) && !isThrottled(entry);
const canDeny = (entry: KeyBudgetEntry): boolean => !cannotTrip(entry) && !isThrottled(entry);
export const isBlockingRow = (entry: KeyBudgetEntry): boolean => entry.status === "exceeded" && canDeny(entry);
@ -95,12 +92,11 @@ export const rowRank = (entry: KeyBudgetEntry): number => {
};
/**
* Only `live` and `no_counter` carry a number worth drawing. Any other state, including one the
* server adds after this ships, is rendered as unknown rather than as a confident zero: overstating
* a spend is the failure that matters here, and a new state is by definition not the normal one.
* Only `live` carries a number worth drawing. Any other state, including one the server adds after
* this ships, is rendered as unknown rather than as a confident zero: overstating a spend is the
* failure that matters here, and a new state is by definition not the normal one.
*/
const spendIsReadable = (entry: KeyBudgetEntry): boolean =>
entry.spend_state === "live" || entry.spend_state === "no_counter";
const spendIsReadable = (entry: KeyBudgetEntry): boolean => entry.spend_state === "live";
const COMPARISON_GLYPH: Record<string, string> = { ">=": "≥", ">": ">" };
@ -110,10 +106,7 @@ const COMPARISON_GLYPH: Record<string, string> = { ">=": "≥", ">": ">" };
* to `>`. So this reads `comparison` off each row rather than assuming a constant per scope. Two
* rows can show identical numbers and opposite statuses, so state the threshold each one enforces.
*/
const thresholdVerb = (entry: KeyBudgetEntry): string => {
if (isAlertOnly(entry)) return "Alerts";
return isThrottled(entry) ? "Throttles" : "Blocks";
};
const thresholdVerb = (entry: KeyBudgetEntry): string => (isThrottled(entry) ? "Throttles" : "Blocks");
export const budgetThresholdRule = (entry: KeyBudgetEntry): string | null => {
if (entry.max_budget == null) return null;
@ -127,7 +120,6 @@ const statusPresentation = (entry: KeyBudgetEntry): { tone: StatusTone; label: s
if (entry.status === "unknown") return { tone: "warning", label: "Unknown" };
if (cannotTrip(entry)) return { tone: "neutral", label: "Cannot trip" };
if (entry.status !== "exceeded") return { tone: "success", label: "Within budget" };
if (isAlertOnly(entry)) return { tone: "warning", label: "Exceeded (alert only)" };
return isThrottled(entry)
? { tone: "warning", label: "Exceeded (throttling)" }
: { tone: "error", label: "Exceeded" };
@ -135,52 +127,64 @@ const statusPresentation = (entry: KeyBudgetEntry): { tone: StatusTone; label: s
function ScopeCell({ entry }: { entry: KeyBudgetEntry }) {
const entity = entry.entity_label || entry.entity_id;
// Per-model rows split one cap across every request model that routes onto it, so `entity_id` is
// what tells two rows apart while `entity_label` repeats the cap. Showing only the label would
// render them as duplicates.
const measured = entry.entity_label && entry.entity_id !== entry.entity_label ? entry.entity_id : null;
return (
<div className="flex min-w-0 flex-col gap-0.5">
<CellTooltip
content={`Limit source: ${entry.source}`}
trigger={
<span className="w-fit truncate text-sm font-medium text-foreground">
{SCOPE_LABELS[entry.scope] ?? entry.scope}
</span>
}
trigger={<span className="w-fit truncate text-sm font-medium text-foreground">{scopeLabel(entry)}</span>}
/>
{entity && (
<span className="truncate font-mono text-xs text-muted-foreground" title={entity}>
{entity}
</span>
)}
{measured && (
<span className="truncate font-mono text-xs text-muted-foreground/70" title={measured}>
{measured}
</span>
)}
</div>
);
}
/**
* Caveats collapse to one marker per row rather than printing as prose in the scope cell.
*
* Inline they set each row's height from the length of its longest note, which is what made the
* table read as ragged. Every note still ships in the DOM, one element each and in the server's
* order, so none is dropped, truncated, or reachable only by hovering.
*/
function NotesCell({ entry }: { entry: KeyBudgetEntry }) {
if (entry.notes.length === 0) return <span className="text-xs text-muted-foreground">-</span>;
const warnings = entry.notes.filter((note) => note.severity === "warning").length;
return (
<span className="flex flex-col">
<CellTooltip
content={
<span className="flex max-w-md flex-col gap-2">
{entry.notes.map((note) => (
<span key={note.code}>{note.text}</span>
))}
</span>
}
trigger={
<span
className={cn(
"w-fit cursor-help text-xs italic",
warnings > 0 ? "text-amber-600" : "text-muted-foreground",
)}
>
{entry.notes.length === 1 ? "1 caveat" : `${entry.notes.length} caveats`}
</span>
}
/>
{entry.notes.map((note) => (
<span
key={note.code}
className={
note.severity === "warning" ? "text-xs italic text-amber-600" : "text-xs italic text-muted-foreground"
}
>
<span key={note.code} className="sr-only">
{note.text}
</span>
))}
</div>
</span>
);
}
/** Exhaustive, so a fourth enforcement mode fails this build rather than defaulting to a claim. */
const ENFORCEMENT_BADGE: Readonly<Record<KeyBudgetEnforcement, { tone: StatusTone; label: string; tooltip: string }>> =
{
soft: {
tone: "neutral",
label: "Alert only",
tooltip: "Soft budget. Going over raises an alert and never rejects a request.",
},
throttled: {
tone: "warning",
label: "Throttles requests",
@ -282,6 +286,14 @@ export const getKeyBudgetsTableColumns = (): ColumnDef<KeyBudgetEntry>[] => [
);
},
},
{
id: "notes",
meta: { title: "Caveats" },
header: "Caveats",
size: 110,
enableSorting: false,
cell: ({ row }) => <NotesCell entry={row.original} />,
},
{
id: "resets",
meta: { title: "Resets" },

View file

@ -135,12 +135,6 @@ const UNCONFIGURED_BUDGET = {
// text below is synthetic. Copying the resolver's prose only buys tests that go stale silently.
const noteText = (code: string): string => `synthetic ${code} caveat`;
const ALERT_ONLY_NOTE = {
code: "alert_only",
severity: "info",
text: noteText("alert_only"),
} as const;
const PROJECT_DEAD_NOTE = {
code: "project_spend_not_tracked",
severity: "warning",
@ -167,19 +161,18 @@ const USER_WITHIN_BUDGET: KeyBudgetEntry = {
status: "ok",
};
const TEAM_SOFT_OVER: KeyBudgetEntry = {
const TEAM_WITHIN_BUDGET: KeyBudgetEntry = {
...UNCONFIGURED_BUDGET,
scope: "team",
entity_type: "team",
entity_id: "team-123",
entity_label: "Platform",
enforcement: "soft",
max_budget: 500,
spend: 900,
remaining: -400,
source: "budget_table:b-soft",
status: "exceeded",
notes: [ALERT_ONLY_NOTE],
spend: 400,
remaining: 100,
comparison: ">",
source: "budget_table:b-team",
status: "ok",
};
const TEAM_MEMBER_BLOCKING: KeyBudgetEntry = {
@ -223,7 +216,7 @@ const WORST_CASE_NOTES = [
// guard. Only its length and the absence of clipping matter, never its wording.
const OVERLONG_NOTE_TEXT = `${"a caveat clause that keeps going ".repeat(16)}end`;
const ALL_BUDGETS = [KEY_UNLIMITED, USER_WITHIN_BUDGET, TEAM_SOFT_OVER, TEAM_MEMBER_BLOCKING, ORG_UNCONFIGURED];
const ALL_BUDGETS = [KEY_UNLIMITED, USER_WITHIN_BUDGET, TEAM_WITHIN_BUDGET, TEAM_MEMBER_BLOCKING, ORG_UNCONFIGURED];
// A fixture the resolver cannot produce is how a rendering bug hides: a cold per-model row built
// with no notes and a computed `remaining` looked healthy while the real one rendered as dead.
@ -232,9 +225,7 @@ const assertServerCouldEmit = (budgets: readonly KeyBudgetEntry[]): void => {
for (const budget of budgets) {
// Scoped to the states the resolver defines today; a fixture simulating a newer server is
// deliberately outside them and only has to satisfy the invariants that are not state-specific.
if (["live", "no_counter", "unavailable"].includes(budget.spend_state)) {
// A missing per-model counter is reported as a real 0.0, because zero is what the cap will be
// compared against. Only a failed read has no number at all.
if (["live", "unavailable"].includes(budget.spend_state)) {
expect(budget.spend_state === "unavailable").toBe(budget.spend === null);
}
// A row nobody could resolve has no cap to call unlimited and no spend to compare, so it is the
@ -249,9 +240,6 @@ const assertServerCouldEmit = (budgets: readonly KeyBudgetEntry[]): void => {
} else {
expect(budget.remaining).toBeCloseTo(budget.max_budget - budget.spend, 6);
}
if (budget.spend_state === "no_counter") {
expect(budget.notes.map((note) => note.code)).toContain("per_model_counters");
}
}
};
@ -332,17 +320,6 @@ describe("KeyInfoView Budgets tab", () => {
expect(blockedRow).toHaveTextContent("$1,000.2000 of $1,000.00");
});
it("shows an over-budget soft limit as alert-only, never as a blocker", async () => {
const panel = await renderAndOpenBudgetsTab();
const softRow = rowFor(panel, "Platform");
expect(within(softRow).getByText("Alert only")).toBeInTheDocument();
expect(within(softRow).getByText("Exceeded (alert only)")).toBeInTheDocument();
expect(within(softRow).queryByTestId("key-budget-blocking")).not.toBeInTheDocument();
expect(within(softRow).queryByText("Blocks requests")).not.toBeInTheDocument();
expect(softRow).toHaveTextContent(ALERT_ONLY_NOTE.text);
});
it("does not render a spend the server could not read as a healthy $0.00", async () => {
const unreadable: KeyBudgetEntry = {
...UNCONFIGURED_BUDGET,
@ -398,7 +375,7 @@ describe("KeyInfoView Budgets tab", () => {
expect(within(row).queryByText("Within budget")).not.toBeInTheDocument();
expect(within(row).queryByText("Cannot trip")).not.toBeInTheDocument();
expect(cellUnder(panel, row, "Remaining")).toHaveTextContent("-");
expect(row).toHaveTextContent(noteText("entity_unavailable"));
expect(cellUnder(panel, row, "Caveats")).toHaveTextContent(noteText("entity_unavailable"));
});
it("sorts a budget it could not read above the ones it could, since only that row needs chasing", async () => {
@ -409,52 +386,6 @@ describe("KeyInfoView Budgets tab", () => {
expect(dataRows[0]).toHaveTextContent("team-123");
});
// The resolver reports a cold counter as the 0.0 it will be enforced as, and always attaches the
// per-model note to a `no_counter` reading, so a fixture without both is one the server cannot produce.
const COLD_PER_MODEL: KeyBudgetEntry = {
...UNCONFIGURED_BUDGET,
scope: "key_model",
entity_type: "key",
entity_id: "claude-opus-5",
entity_label: "claude-opus-5",
max_budget: 40,
spend: 0,
spend_state: "no_counter",
remaining: 40,
comparison: ">",
source: "key.model_max_budget[claude-opus-5]",
status: "ok",
notes: [{ code: "per_model_counters", severity: "warning", text: noteText("per_model_counters") }],
};
it("shows a genuine no-counter-yet zero as $0.00 with its meter, not as unknown", async () => {
mockBudgets([COLD_PER_MODEL]);
const panel = await renderAndOpenBudgetsTab();
const row = rowFor(panel, "claude-opus-5");
expect(row).toHaveTextContent("$0.00 of $40.00");
expect(row).not.toHaveTextContent("Unknown");
expect(within(row).getByRole("meter")).toBeInTheDocument();
// A cold counter reports a real zero, so the whole cap is still available and the remaining
// cell says so rather than falling back to the dash it shows when spend cannot be read.
expect(cellUnder(panel, row, "Remaining")).toHaveTextContent("$40.00");
});
it("keeps a cold per-model budget live, since one request warms the counter and it starts blocking", async () => {
mockBudgets([COLD_PER_MODEL, KEY_UNLIMITED]);
const panel = await renderAndOpenBudgetsTab();
const row = rowFor(panel, "claude-opus-5");
expect(within(row).getByText("Within budget")).toBeInTheDocument();
expect(within(row).queryByText("Cannot trip")).not.toBeInTheDocument();
expect(within(row).getByText("Blocks requests")).toBeInTheDocument();
expect(row).toHaveTextContent("Blocks at > $40.00");
// A live budget must outrank a scope with nothing configured, never sink below it.
const [, ...dataRows] = panel.getAllByRole("row");
expect(dataRows[0]).toHaveTextContent("claude-opus-5");
});
it("treats a spend state this build predates as unreadable rather than as a confident number", async () => {
const future: KeyBudgetEntry = {
...UNCONFIGURED_BUDGET,
@ -493,8 +424,9 @@ describe("KeyInfoView Budgets tab", () => {
mockBudgets([endUser]);
const panel = await renderAndOpenBudgetsTab();
const rendered = WORST_CASE_NOTES.map((note) => panel.getByText(note.text));
expect(rendered).toHaveLength(3);
const caveats = cellUnder(panel, rowFor(panel, "customer-42"), "Caveats");
expect(caveats).toHaveTextContent("3 caveats");
const rendered = WORST_CASE_NOTES.map((note) => within(caveats).getByText(note.text));
// Separate elements, not one joined blob, so each caveat can carry its own severity.
expect(new Set(rendered).size).toBe(3);
});
@ -516,86 +448,16 @@ describe("KeyInfoView Budgets tab", () => {
const panel = await renderAndOpenBudgetsTab();
expect(OVERLONG_NOTE_TEXT.length).toBeGreaterThan(500);
const note = panel.getByText(OVERLONG_NOTE_TEXT);
const caveats = cellUnder(panel, rowFor(panel, "Platform"), "Caveats");
const note = within(caveats).getByText(OVERLONG_NOTE_TEXT);
// jsdom has no layout, so nothing here can prove the text is visually unclipped. Asserting the
// absence of the clipping utilities is the only mechanical guard against re-truncating a note.
// The cell shows a count so every row is the same height, but the text itself must still ship
// whole: jsdom has no layout, so the mechanical guard is that nothing clips or truncates it.
expect(note).not.toHaveClass("truncate");
expect(note.className).not.toMatch(/line-clamp|overflow-hidden|whitespace-nowrap/);
expect(note).not.toHaveAttribute("title");
});
it("keeps two per-model rows on one cap apart by the request model each measures", async () => {
const direct: KeyBudgetEntry = {
...UNCONFIGURED_BUDGET,
scope: "key_model",
entity_type: "key",
entity_id: "claude-opus-5",
entity_label: "claude-opus-5",
max_budget: 40,
spend: 5,
remaining: 35,
comparison: ">",
source: "key.model_max_budget[claude-opus-5]",
status: "ok",
};
const routed: KeyBudgetEntry = {
...direct,
entity_id: "bedrock/claude-opus-5",
spend: 38,
remaining: 2,
};
mockBudgets([direct, routed]);
const panel = await renderAndOpenBudgetsTab();
const routedRow = rowFor(panel, "bedrock/claude-opus-5");
expect(routedRow).toHaveTextContent("$38.0000 of $40.00");
// The cap is repeated on both rows, so it cannot be what tells them apart.
expect(panel.getAllByText("claude-opus-5")).toHaveLength(2);
expect(routedRow).toHaveTextContent("claude-opus-5");
const [, ...dataRows] = panel.getAllByRole("row");
expect(dataRows).toHaveLength(2);
expect(dataRows.filter((row) => row.textContent?.includes("bedrock/claude-opus-5"))).toHaveLength(1);
});
it("floats the one exceeded per-model row above its healthy siblings on the same cap", async () => {
const cap = {
...UNCONFIGURED_BUDGET,
scope: "key_model",
entity_type: "key",
entity_label: "claude-opus-5",
max_budget: 40,
comparison: ">",
source: "key.model_max_budget[claude-opus-5]",
} as const;
const healthy: KeyBudgetEntry = { ...cap, entity_id: "claude-opus-5", spend: 1, remaining: 39, status: "ok" };
const alsoHealthy: KeyBudgetEntry = {
...cap,
entity_id: "vertex_ai/claude-opus-5",
spend: 2,
remaining: 38,
status: "ok",
};
const over: KeyBudgetEntry = {
...cap,
entity_id: "bedrock/claude-opus-5",
spend: 41,
remaining: -1,
status: "exceeded",
};
mockBudgets([healthy, alsoHealthy, over]);
const panel = await renderAndOpenBudgetsTab();
const [, ...dataRows] = panel.getAllByRole("row");
expect(dataRows[0]).toHaveTextContent("bedrock/claude-opus-5");
expect(within(dataRows[0]).getByTestId("key-budget-blocking")).toBeInTheDocument();
// Siblings keep the server's order behind it rather than being reshuffled among themselves.
expect(dataRows[1]).toHaveTextContent("claude-opus-5");
expect(dataRows[2]).toHaveTextContent("vertex_ai/claude-opus-5");
expect(panel.getAllByTestId("key-budget-blocking")).toHaveLength(1);
});
it("renders notes in the order the server sent them, most specific to these numbers first", async () => {
const endUser: KeyBudgetEntry = {
...UNCONFIGURED_BUDGET,
@ -613,8 +475,8 @@ describe("KeyInfoView Budgets tab", () => {
mockBudgets([endUser]);
const panel = await renderAndOpenBudgetsTab();
const texts = WORST_CASE_NOTES.map((note) => note.text);
const rendered = texts.map((text) => panel.getByText(text));
const caveats = cellUnder(panel, rowFor(panel, "customer-42"), "Caveats");
const rendered = WORST_CASE_NOTES.map((note) => within(caveats).getByText(note.text));
const positions = rendered.map((node) => Array.from(node.parentElement?.children ?? []).indexOf(node));
expect(positions).toStrictEqual([...positions].sort((a, b) => a - b));
// Server order is meaningful: the reservation note explains the comparison the row renders.
@ -709,7 +571,7 @@ describe("KeyInfoView Budgets tab", () => {
const deadRow = rowFor(panel, "checkout");
expect(within(deadRow).getByText("Cannot trip")).toBeInTheDocument();
expect(within(deadRow).queryByText("Within budget")).not.toBeInTheDocument();
expect(deadRow).toHaveTextContent(PROJECT_DEAD_NOTE.text);
expect(cellUnder(panel, deadRow, "Caveats")).toHaveTextContent(PROJECT_DEAD_NOTE.text);
const [, ...dataRows] = panel.getAllByRole("row");
expect(dataRows[dataRows.length - 1]).toHaveTextContent("checkout");
@ -771,12 +633,12 @@ describe("KeyInfoView Budgets tab", () => {
expect(within(orgRow).queryByRole("meter")).not.toBeInTheDocument();
});
it("puts the blocking budget above the alert-only, healthy and unlimited ones", async () => {
it("puts the blocking budget above the healthy and unlimited ones", async () => {
const panel = await renderAndOpenBudgetsTab();
const [, ...dataRows] = panel.getAllByRole("row");
expect(dataRows[0]).toHaveTextContent("alice @ Platform");
expect(dataRows[1]).toHaveTextContent("Exceeded (alert only)");
expect(dataRows[1]).toHaveTextContent("Within budget");
expect(dataRows[2]).toHaveTextContent("Within budget");
expect(dataRows.slice(3).every((row) => row.textContent?.includes("Unlimited"))).toBe(true);
});

View file

@ -6833,26 +6833,21 @@ export interface paths {
* `team_member`, `user`, `organization`, `project`, `tag`, `end_user` or `end_user_model`
* - entity_type: Litellm_EntityType - The entity a `BudgetExceededError` from this scope
* names, so a denial message maps back to a row here
* - entity_id / entity_label: str | None - Which entity is limited, and its human-facing alias.
* On the per-model scopes this is one row per counter rather than per request model, so
* `entity_id` is the model whose counter was read and `entity_label` is the configured cap
* it is compared against; several request models can share one counter, and they are not
* listed separately because their spend is not separate
* - enforcement: str - `hard` blocks the request, `soft` only raises an alert, `throttled`
* scales the key's rate limits down instead of denying anything. Only the key's own
* `max_budget` can be `throttled`; every other scope on the same key still blocks
* - entity_id / entity_label: str | None - Which entity is limited, and its human-facing alias
* - enforcement: str - `hard` blocks the request, `throttled` scales the key's rate limits down
* instead of denying anything. Only the key's own `max_budget` can be `throttled`; every
* other scope on the same key still blocks
* - max_budget: float | None - The limit in effect. `null` means this scope applies to the key
* but places no limit on it
* - spend: float | None - Spend as the enforcing check reads it, from the same cross-pod
* counter, not the periodically-synced database column. `null` only when the read failed
* - spend_state: str - Whether `spend` came from a counter (`live`), is the zero that will be
* enforced because no counter has been created yet (`no_counter`), or is missing because the
* read failed (`unavailable`)
* - spend_state: str - Whether `spend` was read (`live`) or is missing because the entity or its
* counter could not be read (`unavailable`)
* - remaining: float | None - `max_budget - spend`, when both are known
* - comparison: str - The operator the enforcing check uses, which differs per scope
* - budget_duration / budget_reset_at / window_start: When spend next resets to zero
* - source: str - Where the limit is configured, e.g. `key.max_budget`, `budget_table:<id>`
* - status: str - `unlimited`, `ok` or `exceeded`
* - status: str - `unlimited`, `ok`, `exceeded`, or `unknown` when the row could not be evaluated
* - notes: list - Caveats worth knowing before trusting the row, each with a stable `code`
* to branch on and human-facing `text` that is free to be reworded. `severity` is for a
* `code` a client does not know yet: `info` only explains a field the row already carries,
@ -7538,26 +7533,21 @@ export interface paths {
* `team_member`, `user`, `organization`, `project`, `tag`, `end_user` or `end_user_model`
* - entity_type: Litellm_EntityType - The entity a `BudgetExceededError` from this scope
* names, so a denial message maps back to a row here
* - entity_id / entity_label: str | None - Which entity is limited, and its human-facing alias.
* On the per-model scopes this is one row per counter rather than per request model, so
* `entity_id` is the model whose counter was read and `entity_label` is the configured cap
* it is compared against; several request models can share one counter, and they are not
* listed separately because their spend is not separate
* - enforcement: str - `hard` blocks the request, `soft` only raises an alert, `throttled`
* scales the key's rate limits down instead of denying anything. Only the key's own
* `max_budget` can be `throttled`; every other scope on the same key still blocks
* - entity_id / entity_label: str | None - Which entity is limited, and its human-facing alias
* - enforcement: str - `hard` blocks the request, `throttled` scales the key's rate limits down
* instead of denying anything. Only the key's own `max_budget` can be `throttled`; every
* other scope on the same key still blocks
* - max_budget: float | None - The limit in effect. `null` means this scope applies to the key
* but places no limit on it
* - spend: float | None - Spend as the enforcing check reads it, from the same cross-pod
* counter, not the periodically-synced database column. `null` only when the read failed
* - spend_state: str - Whether `spend` came from a counter (`live`), is the zero that will be
* enforced because no counter has been created yet (`no_counter`), or is missing because the
* read failed (`unavailable`)
* - spend_state: str - Whether `spend` was read (`live`) or is missing because the entity or its
* counter could not be read (`unavailable`)
* - remaining: float | None - `max_budget - spend`, when both are known
* - comparison: str - The operator the enforcing check uses, which differs per scope
* - budget_duration / budget_reset_at / window_start: When spend next resets to zero
* - source: str - Where the limit is configured, e.g. `key.max_budget`, `budget_table:<id>`
* - status: str - `unlimited`, `ok` or `exceeded`
* - status: str - `unlimited`, `ok`, `exceeded`, or `unknown` when the row could not be evaluated
* - notes: list - Caveats worth knowing before trusting the row, each with a stable `code`
* to branch on and human-facing `text` that is free to be reworded. `severity` is for a
* `code` a client does not know yet: `info` only explains a field the row already carries,
@ -26379,7 +26369,7 @@ export interface components {
* Enforcement
* @enum {string}
*/
enforcement: "hard" | "soft" | "throttled";
enforcement: "hard" | "throttled";
/** Entity Id */
entity_id?: string | null;
/** Entity Label */
@ -26398,7 +26388,7 @@ export interface components {
* Scope
* @enum {string}
*/
scope: "proxy" | "key" | "key_window" | "key_model" | "team" | "team_window" | "team_member" | "user" | "organization" | "project" | "tag" | "end_user" | "end_user_model";
scope: "proxy" | "key" | "key_window" | "team" | "team_window" | "team_member" | "user" | "organization" | "project" | "tag" | "end_user";
/** Source */
source: string;
/** Spend */
@ -26407,7 +26397,7 @@ export interface components {
* Spend State
* @enum {string}
*/
spend_state: "live" | "no_counter" | "unavailable";
spend_state: "live" | "unavailable";
/**
* Status
* @enum {string}
@ -26432,7 +26422,7 @@ export interface components {
* Code
* @enum {string}
*/
code: "alert_only" | "custom_auth_may_override_end_user_cap" | "custom_auth_skips_read_time_checks" | "end_user_route_only" | "entity_unavailable" | "per_model_counters" | "project_spend_not_tracked" | "request_tags_add_budgets" | "reservation_blocks_at_limit" | "rolling_window" | "throttled_instead_of_blocked" | "user_budget_not_applied_to_team_key";
code: "custom_auth_may_override_end_user_cap" | "custom_auth_skips_read_time_checks" | "end_user_route_only" | "entity_unavailable" | "project_spend_not_tracked" | "request_tags_add_budgets" | "reservation_blocks_at_limit" | "rolling_window" | "throttled_instead_of_blocked" | "user_budget_not_applied_to_team_key";
/**
* Severity
* @enum {string}

File diff suppressed because one or more lines are too long