refactor(e2e): bucket rate limits, budgets, and spend tracking under quota_management

This commit is contained in:
mateo-berri 2026-07-11 11:43:54 -07:00
parent 0710cf2990
commit 320a55f01f
19 changed files with 123 additions and 15 deletions

View file

@ -16,7 +16,7 @@ Each subdirectory under `tests/e2e/` is one suite, scoped to an endpoint family
- `management/` - key/team/user/organization management routes: create/update/delete persistence via the info routes, team membership, and llm-only-key route denials; also the dashboard UI behavior on top of them, driven through the proxy-served UI at /ui with playwright (optional dep behind importorskip)
- `logging/` - logging-integration delivery (datadog and friends)
- `security/` - secret handling and log-leak protection
- `router/` - routing and reliability behavior (rate limits, fallbacks, cooldowns)
- `router/` - routing and reliability behavior (fallbacks, cooldowns)
- `gateway/` - proxy configuration only (`litellm-config.yml`); no tests
## Lay the pattern down in a class
@ -63,7 +63,7 @@ The harness is fully typed with no error budget: `make lint-e2e-basedpyright` mu
The set of tests we want is a registry checked into this repo, one row per behavior; that file is the definition of done and the denominator. Each e2e test declares what it covers with `@pytest.mark.covers("...")`, and a small collector diffs the registry against the tests and ships coverage to the existing Grafana. No Allure, no new dependencies
Coverage is organized as module > feature > test. Dashboard modules are `Core LLMs`, `Non-Core LLMs`, `MCPs`, `Management/UI`, `Reliability & Performance`, `Logging & Guardrails`, and `Other`. The Loki stdout formatter maps those display modules to log-safe labels (`core_llms`, `non_core_llms`, `mcp`, `management_ui`, `reliability_performance`, `logging_guardrails`, and `other`) without changing JSON or Prometheus labels. A feature is either an endpoint (`/chat/completions`) or a behavior (fallbacks, rate limits; config-driven, with no route of its own). A cell reads like `llm.chat_completions.bedrock_converse.tool_use.stream.works`
Coverage is organized as module > feature > test. Dashboard modules are `Core LLMs`, `Non-Core LLMs`, `MCPs`, `Management/UI`, `Reliability & Performance`, `Quota Management`, `Logging & Guardrails`, and `Other`. The Loki stdout formatter maps those display modules to log-safe labels (`core_llms`, `non_core_llms`, `mcp`, `management_ui`, `reliability_performance`, `quota_management`, `logging_guardrails`, and `other`) without changing JSON or Prometheus labels. A feature is either an endpoint (`/chat/completions`) or a behavior (fallbacks, rate limits; config-driven, with no route of its own). A cell reads like `llm.chat_completions.bedrock_converse.tool_use.stream.works`
The metric is coverage: the share of registry rows that have a passing covering test, reported to Grafana per module so a gap surfaces as an uncovered row rather than a silent absence
@ -115,14 +115,35 @@ Reliability & Performance - behavior features (no route; endpoint is exercised_o
```
reliability.<behavior>.<variant>.<assertion>
behavior : fallback | retry | cooldown | timeout | ratelimit | routing | cache | circuit_breaker | perf
behavior : fallback | retry | cooldown | timeout | routing | cache | circuit_breaker | perf
variant : <trigger> 5xx | context_window | content_policy | 429 | timeout
<strategy> simple_shuffle | usage_based | latency_based | cost_based | least_busy
<dimension> latency | throughput (perf only; SLO/threshold assertion, not binary)
assertion : routes_to_fallback | succeeds_within_retries | picks_under_tpm | returns_cached
| trips_then_recovers | under_slo
e.g. reliability.fallback.context_window.routes_to_fallback exercised_on=[chat_completions]
reliability.ratelimit.rpm.blocks_over_limit exercised_on=[chat_completions, messages]
reliability.cooldown.429.trips_then_recovers exercised_on=[chat_completions, messages]
```
Quota Management - behavior features (entity- or config-driven caps and their accounting; endpoint is exercised_on)
```
quota_management.<behavior>.<variant>.<assertion>
behavior : ratelimit | budget | spend_tracking
variant : <ratelimit> rpm | tpm | priority_generous | priority_strict
<budget> key | internal_user | end_user | organization | team_member | tag
| model_max | soft | key_multi_window | team_multi_window
| fallback | spend_counter
<spend_tracking> chat_completions | stream | embeddings | cache_hit | key_rollup
| concurrent_burst | tags | end_user | per_model | failure
| spend_calculate | pagination
assertion : blocks_over_limit | resets_after_window | headers_report_remaining | picks_under_tpm
| blocks_then_resets | resets_windows_independently | alerts_without_blocking
| isolates_per_model | routes_to_fallback | reseed_matches_db | logs_cost | zero_cost
| matches_sum_of_logs | loses_no_spend | attributes_spend | writes_own_rows
| writes_failure_row | returns_cost | keeps_total
e.g. quota_management.ratelimit.rpm.blocks_over_limit exercised_on=[chat_completions, messages]
quota_management.budget.key.blocks_over_limit exercised_on=[chat_completions]
```
Logging & Guardrails - behavior features (config-driven; endpoint is exercised_on)

View file

@ -15,6 +15,7 @@ from lifecycle import ResourceManager
pytestmark = pytest.mark.e2e
@pytest.mark.covers("mgmt.budget.new.persists")
def test_budget_crud_roundtrip(client: BudgetClient, resources: ResourceManager) -> None:
budget_id = client.create_budget(max_budget=12.5, soft_budget=10.0, budget_duration="30d")
resources.defer(lambda: client.delete_budget(budget_id))
@ -36,6 +37,7 @@ def test_budget_crud_roundtrip(client: BudgetClient, resources: ResourceManager)
)
@pytest.mark.covers("mgmt.budget.delete.persists")
def test_budget_delete_removes_it(client: BudgetClient, resources: ResourceManager) -> None:
budget_id = client.create_budget(max_budget=1.0)
resources.defer(lambda: client.delete_budget(budget_id))

View file

@ -134,11 +134,26 @@ def _case_id(case_cls: Type[_BudgetCase]) -> str:
@pytest.mark.parametrize(
"case_cls",
[
KeyBudgetCase,
InternalUserBudgetCase,
EndUserBudgetCase,
OrganizationBudgetCase,
TeamMemberBudgetCase,
pytest.param(
KeyBudgetCase,
marks=pytest.mark.covers("quota_management.budget.key.blocks_over_limit"),
),
pytest.param(
InternalUserBudgetCase,
marks=pytest.mark.covers("quota_management.budget.internal_user.blocks_over_limit"),
),
pytest.param(
EndUserBudgetCase,
marks=pytest.mark.covers("quota_management.budget.end_user.blocks_over_limit"),
),
pytest.param(
OrganizationBudgetCase,
marks=pytest.mark.covers("quota_management.budget.organization.blocks_over_limit"),
),
pytest.param(
TeamMemberBudgetCase,
marks=pytest.mark.covers("quota_management.budget.team_member.blocks_over_limit"),
),
],
ids=_case_id,
)

View file

@ -19,6 +19,7 @@ PRIMARY_MODEL = "claude-haiku-4-5"
FALLBACK_MODEL = "gpt-5.5"
@pytest.mark.covers("quota_management.budget.fallback.routes_to_fallback")
def test_budget_fallback_reroutes_anthropic_messages_to_openai(
client: BudgetClient, resources: ResourceManager
) -> None:

View file

@ -70,6 +70,7 @@ def test_key_with_budget_duration_schedules_reset_at_creation(
# ---- Rung 2: enforcement trips at the cap ------------------------------------
@pytest.mark.covers("quota_management.budget.key.blocks_over_limit")
def test_key_spend_blocks_at_cap(client: BudgetClient, resources: ResourceManager) -> None:
"""Sanity that the tiny cap is enforced before we test that it resets: spend
accrues across calls and eventually returns budget_exceeded, never a 5xx."""
@ -86,6 +87,7 @@ def test_key_spend_blocks_at_cap(client: BudgetClient, resources: ResourceManage
# ---- Rung 3: the core regression - reset_at strictly advances + spend zeroes --
@pytest.mark.covers("quota_management.budget.key.resets_after_window")
def test_key_budget_reset_at_advances_after_window(
client: BudgetClient, resources: ResourceManager
) -> None:
@ -123,6 +125,7 @@ def test_key_budget_reset_at_advances_after_window(
# ---- Rung 4: multi-window - tight window resets, roomy window keeps spend -----
@pytest.mark.covers("quota_management.budget.key_multi_window.resets_windows_independently")
def test_multi_window_key_resets_each_window_independently(
client: BudgetClient, resources: ResourceManager
) -> None:
@ -164,6 +167,7 @@ def test_multi_window_key_resets_each_window_independently(
# ---- Rung 5: team-member window advances (JSON-backed per-team budget) --------
@pytest.mark.covers("quota_management.budget.team_member.resets_after_window")
def test_team_member_budget_reset_at_advances(
client: BudgetClient, resources: ResourceManager
) -> None:

View file

@ -26,6 +26,7 @@ def _call(client: BudgetClient, key: str):
)
@pytest.mark.covers("quota_management.budget.key.resets_after_window")
def test_key_budget_resets_after_duration(
client: BudgetClient, resources: ResourceManager
) -> None:

View file

@ -28,6 +28,7 @@ def _call(client: BudgetClient, key: str, model: str):
return result
@pytest.mark.covers("quota_management.budget.model_max.isolates_per_model")
def test_model_max_budget_isolates_per_model(
client: BudgetClient, resources: ResourceManager
) -> None:

View file

@ -29,6 +29,7 @@ def _call(client: BudgetClient, key: str):
)
@pytest.mark.covers("quota_management.budget.key_multi_window.blocks_then_resets")
def test_short_window_blocks_then_resets(
client: BudgetClient, resources: ResourceManager
) -> None:

View file

@ -17,6 +17,7 @@ from lifecycle import ResourceManager
pytestmark = pytest.mark.e2e
@pytest.mark.covers("quota_management.budget.soft.alerts_without_blocking")
def test_soft_budget_does_not_block(
client: BudgetClient, resources: ResourceManager
) -> None:

View file

@ -121,6 +121,7 @@ def _accumulate(client: BudgetClient, key: str, count: int) -> None:
list(pool.map(one, range(count)))
@pytest.mark.covers("quota_management.budget.spend_counter.reseed_matches_db")
def test_cold_counter_reseed_keeps_counter_equal_to_db_spend(
client: BudgetClient, resources: ResourceManager
) -> None:

View file

@ -33,6 +33,7 @@ def _tagged_call(client: BudgetClient, key: str, tag: str):
return result
@pytest.mark.covers("quota_management.budget.tag.blocks_over_limit")
def test_tag_budget_blocks_tagged_requests(
client: BudgetClient, scoped_key: str, resources: ResourceManager
) -> None:

View file

@ -97,6 +97,7 @@ class TestTeamMemberBudget:
f"call {row.request_id} logged under user {row.user}, not member {member.user_id}"
)
@pytest.mark.covers("quota_management.budget.team_member.blocks_over_limit")
def test_member_spend_over_budget_is_blocked(self, client: BudgetClient, member: _Member) -> None:
for _ in range(40):
result = client.chat(member.key, MODEL, f"spend {unique_marker()}", max_tokens=16)

View file

@ -16,6 +16,7 @@ def _as_datetime(value: str) -> datetime:
return datetime.fromisoformat(value.replace("Z", "+00:00"))
@pytest.mark.covers("quota_management.budget.team_member.resets_after_window")
def test_team_member_budget_reset_keeps_advancing(client: BudgetClient, resources: ResourceManager) -> None:
team_id = client.create_team(alias=f"e2e-member-reset-{unique_marker()}", max_budget=100.0)
resources.defer(lambda: client.delete_team(team_id))

View file

@ -32,6 +32,7 @@ def _call(client: BudgetClient, key: str):
return client.chat(key, "claude-haiku-4-5", f"team-window {unique_marker()}", max_tokens=16)
@pytest.mark.covers("quota_management.budget.team_multi_window.blocks_then_resets")
def test_team_short_window_blocks_then_resets(client: BudgetClient, resources: ResourceManager) -> None:
team_id = client.create_team(
alias=f"e2e-team-window-{unique_marker()}",

View file

@ -14,7 +14,8 @@ grouped `module > feature > test`, with LLM cells split into `Core LLMs` and
`fail_before_fix` flag.
The rows live in per-prefix YAML files (`llm_*.yaml`, `mgmt.yaml`, `mcp.yaml`,
`reliability.yaml`, `logging.yaml`, `guardrail.yaml`, `other.yaml`) and validate against
`reliability.yaml`, `quota_management.yaml`, `logging.yaml`, `guardrail.yaml`,
`other.yaml`) and validate against
the discriminated union in `schema.py`, so an LLM row cannot carry a guardrail field and
vice versa. `llm` rows with `subject_endpoint` of `chat_completions`, `messages`, or
`responses` roll up to `Core LLMs`; all other LLM endpoints roll up to `Non-Core LLMs`.

View file

@ -0,0 +1,35 @@
# Quota Management (behavior features): rate limits, budgets, spend tracking. Grounded in
# litellm/proxy/hooks/ + litellm/proxy/auth/auth_checks.py + litellm/proxy/spend_tracking/.
- {id: quota_management.ratelimit.rpm.blocks_over_limit, module: quota_management, tier: P0, behavior: ratelimit, variant: rpm, assertions: [blocks_over_limit], exercised_on: [chat_completions, messages], source: "parallel_request_limiter_v3.py", rationale: "v3 limiter enforces RPM per key/team/model; 429 on breach"}
- {id: quota_management.ratelimit.tpm.blocks_over_limit, module: quota_management, tier: P0, behavior: ratelimit, variant: tpm, assertions: [blocks_over_limit], exercised_on: [chat_completions, messages], source: "parallel_request_limiter_v3.py", rationale: "v3 limiter enforces TPM per key/team/model; 429 on breach"}
- {id: quota_management.ratelimit.rpm.resets_after_window, module: quota_management, tier: P1, behavior: ratelimit, variant: rpm, assertions: [resets_after_window], exercised_on: [chat_completions], source: "parallel_request_limiter_v3.py", rationale: "Rate-limit window (LITELLM_RATE_LIMIT_WINDOW_SIZE, 60s default) expires; a blocked key serves again in the next window"}
- {id: quota_management.ratelimit.rpm.headers_report_remaining, module: quota_management, tier: P1, behavior: ratelimit, variant: rpm, assertions: [headers_report_remaining], exercised_on: [chat_completions], source: "parallel_request_limiter_v3.py async_post_call_success_hook", rationale: "Successful responses carry x-ratelimit-api_key-{limit,remaining}-{requests,tokens} so clients can pace"}
- {id: quota_management.ratelimit.priority_generous.picks_under_tpm, module: quota_management, tier: P1, behavior: ratelimit, variant: priority_generous, assertions: [picks_under_tpm], exercised_on: [chat_completions, messages], source: "dynamic_rate_limiter_v3.py:36-52", rationale: "Generous mode (<80% sat) allows priority borrowing"}
- {id: quota_management.ratelimit.priority_strict.picks_under_tpm, module: quota_management, tier: P1, behavior: ratelimit, variant: priority_strict, assertions: [picks_under_tpm], exercised_on: [chat_completions, messages], source: "dynamic_rate_limiter_v3.py:53-71", rationale: "Strict mode (>=80% sat) enforces priority fairness"}
- {id: quota_management.budget.key.blocks_over_limit, module: quota_management, tier: P0, behavior: budget, variant: key, assertions: [blocks_over_limit], exercised_on: [chat_completions], source: "proxy/auth/auth_checks.py", rationale: "A key's max_budget blocks further paid calls once spend crosses it"}
- {id: quota_management.budget.internal_user.blocks_over_limit, module: quota_management, tier: P1, behavior: budget, variant: internal_user, assertions: [blocks_over_limit], exercised_on: [chat_completions], source: "proxy/auth/auth_checks.py", rationale: "An internal user's max_budget governs personal keys"}
- {id: quota_management.budget.end_user.blocks_over_limit, module: quota_management, tier: P1, behavior: budget, variant: end_user, assertions: [blocks_over_limit], exercised_on: [chat_completions], source: "proxy/auth/auth_checks.py", rationale: "A customer (end-user) max_budget blocks calls attributed via user="}
- {id: quota_management.budget.organization.blocks_over_limit, module: quota_management, tier: P1, behavior: budget, variant: organization, assertions: [blocks_over_limit], exercised_on: [chat_completions], source: "proxy/auth/auth_checks.py", rationale: "An organization's max_budget blocks keys under its teams"}
- {id: quota_management.budget.team_member.blocks_over_limit, module: quota_management, tier: P1, behavior: budget, variant: team_member, assertions: [blocks_over_limit], exercised_on: [chat_completions], source: "proxy/auth/auth_checks.py", rationale: "A member's per-team budget blocks independently of the team budget"}
- {id: quota_management.budget.tag.blocks_over_limit, module: quota_management, tier: P1, behavior: budget, variant: tag, assertions: [blocks_over_limit], exercised_on: [chat_completions], source: "router_strategy/budget_limiter.py", rationale: "Proxy-level tag budgets block tagged requests at the cap"}
- {id: quota_management.budget.model_max.isolates_per_model, module: quota_management, tier: P1, behavior: budget, variant: model_max, assertions: [isolates_per_model], exercised_on: [chat_completions], source: "proxy/hooks/model_max_budget_limiter.py", rationale: "model_max_budget caps one model without touching a sibling's budget"}
- {id: quota_management.budget.soft.alerts_without_blocking, module: quota_management, tier: P1, behavior: budget, variant: soft, assertions: [alerts_without_blocking], exercised_on: [chat_completions], source: "proxy/auth/auth_checks.py", rationale: "soft_budget alerts but never blocks traffic"}
- {id: quota_management.budget.key.resets_after_window, module: quota_management, tier: P1, behavior: budget, variant: key, assertions: [resets_after_window], exercised_on: [chat_completions], source: "proxy/common_utils/reset_budget_job.py", rationale: "budget_duration zeroes key spend after the window; a blocked key serves again"}
- {id: quota_management.budget.team_member.resets_after_window, module: quota_management, tier: P1, behavior: budget, variant: team_member, assertions: [resets_after_window], exercised_on: [chat_completions], source: "proxy/common_utils/reset_budget_job.py", rationale: "Member per-team budget reset keeps advancing window after window"}
- {id: quota_management.budget.key_multi_window.blocks_then_resets, module: quota_management, tier: P1, behavior: budget, variant: key_multi_window, assertions: [blocks_then_resets], exercised_on: [chat_completions], source: "proxy/common_utils/reset_budget_job.py", rationale: "budget_limits enforce within a short window and serve again in the next"}
- {id: quota_management.budget.key_multi_window.resets_windows_independently, module: quota_management, tier: P2, behavior: budget, variant: key_multi_window, assertions: [resets_windows_independently], exercised_on: [chat_completions], source: "proxy/common_utils/reset_budget_job.py", rationale: "Each window of a multi-window budget resets on its own schedule"}
- {id: quota_management.budget.team_multi_window.blocks_then_resets, module: quota_management, tier: P1, behavior: budget, variant: team_multi_window, assertions: [blocks_then_resets], exercised_on: [chat_completions], source: "proxy/common_utils/reset_budget_job.py", rationale: "Team budget_limits enforce and reset per window"}
- {id: quota_management.budget.fallback.routes_to_fallback, module: quota_management, tier: P1, behavior: budget, variant: fallback, assertions: [routes_to_fallback], exercised_on: [messages], source: "proxy/hooks/model_max_budget_limiter.py", rationale: "budget_fallbacks reroute to the fallback model once the primary's budget is exhausted"}
- {id: quota_management.budget.spend_counter.reseed_matches_db, module: quota_management, tier: P2, behavior: budget, variant: spend_counter, assertions: [reseed_matches_db], exercised_on: [chat_completions], source: "proxy/spend_tracking/budget_reservation.py", rationale: "Concurrent cold-counter reseeds keep the enforcement counter equal to DB spend (#26829)"}
- {id: quota_management.spend_tracking.chat_completions.logs_cost, module: quota_management, tier: P0, behavior: spend_tracking, variant: chat_completions, assertions: [logs_cost], exercised_on: [chat_completions], source: "proxy/spend_tracking/spend_tracking_utils.py", rationale: "A paid chat call writes a nonzero spend row"}
- {id: quota_management.spend_tracking.stream.logs_cost, module: quota_management, tier: P1, behavior: spend_tracking, variant: stream, assertions: [logs_cost], exercised_on: [chat_completions], source: "proxy/spend_tracking/spend_tracking_utils.py", rationale: "Streaming responses aggregate token counts into a spend row"}
- {id: quota_management.spend_tracking.embeddings.logs_cost, module: quota_management, tier: P1, behavior: spend_tracking, variant: embeddings, assertions: [logs_cost], exercised_on: [embeddings], source: "proxy/spend_tracking/spend_tracking_utils.py", rationale: "Embedding calls write nonzero spend rows"}
- {id: quota_management.spend_tracking.cache_hit.zero_cost, module: quota_management, tier: P1, behavior: spend_tracking, variant: cache_hit, assertions: [zero_cost], exercised_on: [chat_completions], source: "proxy/spend_tracking/spend_tracking_utils.py", rationale: "A response-cache hit logs at zero cost with the cache-hit marker"}
- {id: quota_management.spend_tracking.key_rollup.matches_sum_of_logs, module: quota_management, tier: P1, behavior: spend_tracking, variant: key_rollup, assertions: [matches_sum_of_logs], exercised_on: [chat_completions], source: "proxy/db/db_spend_update_writer.py", rationale: "A key's rolled-up spend equals the sum of its log rows"}
- {id: quota_management.spend_tracking.concurrent_burst.loses_no_spend, module: quota_management, tier: P1, behavior: spend_tracking, variant: concurrent_burst, assertions: [loses_no_spend], exercised_on: [chat_completions], source: "proxy/db/db_spend_update_writer.py", rationale: "Concurrent calls all land as spend; no row lost to write contention"}
- {id: quota_management.spend_tracking.tags.attributes_spend, module: quota_management, tier: P1, behavior: spend_tracking, variant: tags, assertions: [attributes_spend], exercised_on: [chat_completions], source: "proxy/spend_tracking/spend_tracking_utils.py", rationale: "Request tags round-trip to spend rows and tag rollups match tagged logs"}
- {id: quota_management.spend_tracking.end_user.attributes_spend, module: quota_management, tier: P1, behavior: spend_tracking, variant: end_user, assertions: [attributes_spend], exercised_on: [chat_completions], source: "proxy/spend_tracking/spend_tracking_utils.py", rationale: "user= attribution lands the end-user id on the spend row"}
- {id: quota_management.spend_tracking.per_model.writes_own_rows, module: quota_management, tier: P2, behavior: spend_tracking, variant: per_model, assertions: [writes_own_rows], exercised_on: [chat_completions], source: "proxy/spend_tracking/spend_tracking_utils.py", rationale: "Each model on a shared key gets its own spend row"}
- {id: quota_management.spend_tracking.failure.writes_failure_row, module: quota_management, tier: P1, behavior: spend_tracking, variant: failure, assertions: [writes_failure_row], exercised_on: [chat_completions], source: "proxy/spend_tracking/spend_log_error_logger.py", rationale: "A failed call writes a failure-status spend row"}
- {id: quota_management.spend_tracking.spend_calculate.returns_cost, module: quota_management, tier: P2, behavior: spend_tracking, variant: spend_calculate, assertions: [returns_cost], exercised_on: [chat_completions], source: "proxy/spend_tracking/spend_management_endpoints.py", rationale: "/spend/calculate prices a hypothetical request at nonzero cost"}
- {id: quota_management.spend_tracking.pagination.keeps_total, module: quota_management, tier: P2, behavior: spend_tracking, variant: pagination, assertions: [keeps_total], exercised_on: [chat_completions], source: "proxy/spend_tracking/spend_management_endpoints.py", rationale: "Spend-logs v2 pagination caps page size without losing the total"}

View file

@ -12,10 +12,6 @@
- {id: reliability.cooldown.429.trips_then_recovers, module: reliability, tier: P0, behavior: cooldown, variant: "429", assertions: [trips_then_recovers], exercised_on: [chat_completions, messages], source: "cooldown_handlers.py:69", rationale: "Cools on 429, avoids hammering exhausted provider"}
- {id: reliability.cooldown.auth.trips_then_recovers, module: reliability, tier: P1, behavior: cooldown, variant: auth, assertions: [trips_then_recovers], exercised_on: [chat_completions, messages], source: "cooldown_handlers.py:74", rationale: "Cools on 401 auth error"}
- {id: reliability.cooldown.timeout.trips_then_recovers, module: reliability, tier: P1, behavior: cooldown, variant: timeout, assertions: [trips_then_recovers], exercised_on: [chat_completions, messages], source: "cooldown_handlers.py:77", rationale: "Cools on 408 timeout"}
- {id: reliability.ratelimit.rpm.blocks_over_limit, module: reliability, tier: P0, behavior: ratelimit, variant: rpm, assertions: [blocks_over_limit], exercised_on: [chat_completions, messages], source: "dynamic_rate_limiter_v3.py", rationale: "v3 limiter enforces RPM per key/team/model; 429 on breach"}
- {id: reliability.ratelimit.tpm.blocks_over_limit, module: reliability, tier: P0, behavior: ratelimit, variant: tpm, assertions: [blocks_over_limit], exercised_on: [chat_completions, messages], source: "dynamic_rate_limiter_v3.py", rationale: "v3 limiter enforces TPM per key/team/model; 429 on breach"}
- {id: reliability.ratelimit.priority_generous.picks_under_tpm, module: reliability, tier: P1, behavior: ratelimit, variant: priority_generous, assertions: [picks_under_tpm], exercised_on: [chat_completions, messages], source: "dynamic_rate_limiter_v3.py:36-52", rationale: "Generous mode (<80% sat) allows priority borrowing"}
- {id: reliability.ratelimit.priority_strict.picks_under_tpm, module: reliability, tier: P1, behavior: ratelimit, variant: priority_strict, assertions: [picks_under_tpm], exercised_on: [chat_completions, messages], source: "dynamic_rate_limiter_v3.py:53-71", rationale: "Strict mode (>=80% sat) enforces priority fairness"}
- {id: reliability.routing.simple_shuffle.picks_healthy_deployment, module: reliability, tier: P1, behavior: routing, variant: simple_shuffle, assertions: [picks_healthy_deployment], exercised_on: [chat_completions, messages], source: "router_strategy/simple_shuffle.py", rationale: "Baseline weighted/uniform pick"}
- {id: reliability.routing.latency_based.picks_lowest_latency, module: reliability, tier: P1, behavior: routing, variant: latency_based, assertions: [picks_lowest_latency], exercised_on: [chat_completions, messages], source: "router_strategy/lowest_latency.py", rationale: "Routes to lowest-latency deployment"}
- {id: reliability.routing.cost_based.picks_lowest_cost, module: reliability, tier: P1, behavior: routing, variant: cost_based, assertions: [picks_lowest_cost], exercised_on: [chat_completions, messages], source: "router_strategy/lowest_cost.py", rationale: "Spend-aware routing"}

View file

@ -1,7 +1,7 @@
"""Registry row schema: the contract every denominator cell validates against.
A cell is one customer-noticeable behavior a single e2e test can assert pass/fail
on. `module` is the id's segment-1 prefix (seven of them); dashboard rollups can
on. `module` is the id's segment-1 prefix (eight of them); dashboard rollups can
split or merge those prefixes. The union is discriminated on `module`, so an LLM
row cannot carry a guardrail field and vice versa.
"""
@ -100,6 +100,13 @@ class ReliabilityCell(_Base):
exercised_on: tuple[str, ...]
class QuotaCell(_Base):
module: Literal["quota_management"]
behavior: Literal["ratelimit", "budget", "spend_tracking"]
variant: str
exercised_on: tuple[str, ...]
class LoggingCell(_Base):
module: Literal["logging"]
event: str
@ -122,6 +129,7 @@ Cell = Annotated[
| MgmtCell
| McpCell
| ReliabilityCell
| QuotaCell
| LoggingCell
| GuardrailCell
| OtherCell,
@ -142,6 +150,7 @@ PREFIX_ROLLUP: dict[str, str] = {
"mcp": "MCPs",
"mgmt": "Management/UI",
"reliability": "Reliability & Performance",
"quota_management": "Quota Management",
"logging": "Logging & Guardrails",
"guardrail": "Logging & Guardrails",
"other": "Other",
@ -153,6 +162,7 @@ MODULE_ORDER: tuple[str, ...] = (
"MCPs",
"Management/UI",
"Reliability & Performance",
"Quota Management",
"Logging & Guardrails",
"Other",
)
@ -163,6 +173,7 @@ LOKI_MODULE_LABELS: dict[str, str] = {
"MCPs": "mcp",
"Management/UI": "management_ui",
"Reliability & Performance": "reliability_performance",
"Quota Management": "quota_management",
"Logging & Guardrails": "logging_guardrails",
"Other": "other",
}

View file

@ -59,6 +59,7 @@ def _require_row(
return matches[0]
@pytest.mark.covers("quota_management.spend_tracking.chat_completions.logs_cost")
def test_chat_completion_writes_nonzero_spend_row(
client: SpendClient, scoped_key: str
) -> None:
@ -93,6 +94,7 @@ def test_chat_completion_writes_nonzero_spend_row(
), f"row request_id != client response.id ({chat.id})"
@pytest.mark.covers("quota_management.spend_tracking.stream.logs_cost")
def test_streaming_chat_completion_tracks_spend(
client: SpendClient, scoped_key: str
) -> None:
@ -120,6 +122,7 @@ def test_streaming_chat_completion_tracks_spend(
assert (row.total_tokens or 0) == prompt + completion
@pytest.mark.covers("quota_management.spend_tracking.embeddings.logs_cost")
def test_embedding_writes_nonzero_spend_row(
client: SpendClient, scoped_key: str
) -> None:
@ -142,6 +145,7 @@ def test_embedding_writes_nonzero_spend_row(
assert "text-embedding-3-small" in (row.model or "")
@pytest.mark.covers("quota_management.spend_tracking.cache_hit.zero_cost")
def test_cache_hit_is_zero_cost_and_suffixed(
client: SpendClient, scoped_key: str
) -> None:
@ -177,6 +181,7 @@ def test_cache_hit_is_zero_cost_and_suffixed(
), f"the non-cached call should still be charged: {_summarize(rows)}"
@pytest.mark.covers("quota_management.spend_tracking.key_rollup.matches_sum_of_logs")
def test_key_spend_equals_sum_of_logs(client: SpendClient, scoped_key: str) -> None:
for _ in range(2):
_ = unwrap(
@ -203,6 +208,7 @@ def test_key_spend_equals_sum_of_logs(client: SpendClient, scoped_key: str) -> N
), f"key aggregate {key_spend} != sum of logs {logs_total}; rows: {_summarize(rows)}"
@pytest.mark.covers("quota_management.spend_tracking.concurrent_burst.loses_no_spend")
def test_burst_of_concurrent_calls_loses_no_spend(
client: SpendClient, scoped_key: str
) -> None:
@ -249,6 +255,7 @@ def test_burst_of_concurrent_calls_loses_no_spend(
)
@pytest.mark.covers("quota_management.spend_tracking.pagination.keeps_total")
def test_spend_logs_v2_pagination_caps_pages_and_keeps_total(
client: SpendClient, scoped_key: str
) -> None:
@ -301,6 +308,7 @@ def test_spend_logs_v2_pagination_caps_pages_and_keeps_total(
)
@pytest.mark.covers("quota_management.spend_tracking.tags.attributes_spend")
def test_request_tags_round_trip(client: SpendClient, scoped_key: str) -> None:
tag = f"e2e-spend-{unique_marker()}"
_ = unwrap(
@ -317,6 +325,7 @@ def test_request_tags_round_trip(client: SpendClient, scoped_key: str) -> None:
)
@pytest.mark.covers("quota_management.spend_tracking.tags.attributes_spend")
def test_tag_spend_matches_sum_of_tagged_logs(
client: SpendClient, scoped_key: str
) -> None:
@ -354,6 +363,7 @@ def test_tag_spend_matches_sum_of_tagged_logs(
)
@pytest.mark.covers("quota_management.spend_tracking.end_user.attributes_spend")
def test_end_user_spend_attributed_on_row(
client: SpendClient, scoped_key: str, resources: ResourceManager
) -> None:
@ -371,6 +381,7 @@ def test_end_user_spend_attributed_on_row(
assert (row.spend or 0) > 0, f"end-user row should cost > 0: {_summarize(rows)}"
@pytest.mark.covers("quota_management.spend_tracking.per_model.writes_own_rows")
def test_each_model_on_a_shared_key_gets_its_own_row(
client: SpendClient, scoped_key: str
) -> None:
@ -419,6 +430,7 @@ def test_each_model_on_a_shared_key_gets_its_own_row(
), f"claude row request_id {claude_row.request_id} != response id {claude.id}"
@pytest.mark.covers("quota_management.spend_tracking.failure.writes_failure_row")
def test_failure_call_writes_failure_status_row(
client: SpendClient, scoped_key: str
) -> None:
@ -438,6 +450,7 @@ def test_failure_call_writes_failure_status_row(
assert (failure_rows[0].spend or 0) == 0.0, "failed call must not be charged"
@pytest.mark.covers("quota_management.spend_tracking.spend_calculate.returns_cost")
def test_spend_calculate_returns_nonzero_cost(client: SpendClient) -> None:
cost = client.calculate_spend(
"gemini-2.5-flash", "estimate the cost of this request"