mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-10 03:28:53 +00:00
Merge pull request #32928 from BerriAI/litellm_quota_management_registry
refactor(e2e): bucket rate limits, budgets, and spend tracking under quota_management
This commit is contained in:
commit
5e1928c753
29 changed files with 127 additions and 20 deletions
|
|
@ -11,12 +11,11 @@ Each subdirectory under `tests/e2e/` is one suite, scoped to an endpoint family
|
|||
- `embeddings/` - the `/embeddings` endpoint across providers
|
||||
- `batches/` - the `/batches` endpoint (placeholder until the first test lands)
|
||||
- `realtime/` - realtime websocket sessions, including the pipecat audio path
|
||||
- `budgets/` - budget definition, enforcement, and reset windows (key, team, tag, soft, multi-window)
|
||||
- `spend_tracking/` - spend logging and cost attribution on `/spend/*`
|
||||
- `quota_management/` - quota enforcement and accounting, one subfolder per behavior: `budgets/` (budget definition, enforcement, and reset windows: key, team, tag, soft, multi-window) and `spend_tracking/` (spend logging and cost attribution on `/spend/*`)
|
||||
- `management/` - key/team/user/organization management routes: create/update/delete persistence via the info routes, team membership, and llm-only-key route denials; also the dashboard UI behavior on top of them, driven through the proxy-served UI at /ui with playwright (optional dep behind importorskip)
|
||||
- `logging/` - logging-integration delivery (datadog and friends)
|
||||
- `security/` - secret handling and log-leak protection
|
||||
- `router/` - routing and reliability behavior (rate limits, fallbacks, cooldowns)
|
||||
- `router/` - routing and reliability behavior (fallbacks, cooldowns)
|
||||
- `gateway/` - proxy configuration only (`litellm-config.yml`); no tests
|
||||
|
||||
## Lay the pattern down in a class
|
||||
|
|
@ -63,7 +62,7 @@ The harness is fully typed with no error budget: `make lint-e2e-basedpyright` mu
|
|||
|
||||
The set of tests we want is a registry checked into this repo, one row per behavior; that file is the definition of done and the denominator. Each e2e test declares what it covers with `@pytest.mark.covers("...")`, and a small collector diffs the registry against the tests and ships coverage to the existing Grafana. No Allure, no new dependencies
|
||||
|
||||
Coverage is organized as module > feature > test. Dashboard modules are `Core LLMs`, `Non-Core LLMs`, `MCPs`, `Management/UI`, `Reliability & Performance`, `Logging & Guardrails`, and `Other`. The Loki stdout formatter maps those display modules to log-safe labels (`core_llms`, `non_core_llms`, `mcp`, `management_ui`, `reliability_performance`, `logging_guardrails`, and `other`) without changing JSON or Prometheus labels. A feature is either an endpoint (`/chat/completions`) or a behavior (fallbacks, rate limits; config-driven, with no route of its own). A cell reads like `llm.chat_completions.bedrock_converse.tool_use.stream.works`
|
||||
Coverage is organized as module > feature > test. Dashboard modules are `Core LLMs`, `Non-Core LLMs`, `MCPs`, `Management/UI`, `Reliability & Performance`, `Quota Management`, `Logging & Guardrails`, and `Other`. The Loki stdout formatter maps those display modules to log-safe labels (`core_llms`, `non_core_llms`, `mcp`, `management_ui`, `reliability_performance`, `quota_management`, `logging_guardrails`, and `other`) without changing JSON or Prometheus labels. A feature is either an endpoint (`/chat/completions`) or a behavior (fallbacks, rate limits; config-driven, with no route of its own). A cell reads like `llm.chat_completions.bedrock_converse.tool_use.stream.works`
|
||||
|
||||
The metric is coverage: the share of registry rows that have a passing covering test, reported to Grafana per module so a gap surfaces as an uncovered row rather than a silent absence
|
||||
|
||||
|
|
@ -115,14 +114,35 @@ Reliability & Performance - behavior features (no route; endpoint is exercised_o
|
|||
|
||||
```
|
||||
reliability.<behavior>.<variant>.<assertion>
|
||||
behavior : fallback | retry | cooldown | timeout | ratelimit | routing | cache | circuit_breaker | perf
|
||||
behavior : fallback | retry | cooldown | timeout | routing | cache | circuit_breaker | perf
|
||||
variant : <trigger> 5xx | context_window | content_policy | 429 | timeout
|
||||
<strategy> simple_shuffle | usage_based | latency_based | cost_based | least_busy
|
||||
<dimension> latency | throughput (perf only; SLO/threshold assertion, not binary)
|
||||
assertion : routes_to_fallback | succeeds_within_retries | picks_under_tpm | returns_cached
|
||||
| trips_then_recovers | under_slo
|
||||
e.g. reliability.fallback.context_window.routes_to_fallback exercised_on=[chat_completions]
|
||||
reliability.ratelimit.rpm.blocks_over_limit exercised_on=[chat_completions, messages]
|
||||
reliability.cooldown.429.trips_then_recovers exercised_on=[chat_completions, messages]
|
||||
```
|
||||
|
||||
Quota Management - behavior features (entity- or config-driven caps and their accounting; endpoint is exercised_on)
|
||||
|
||||
```
|
||||
quota_management.<behavior>.<variant>.<assertion>
|
||||
behavior : ratelimit | budget | spend_tracking
|
||||
variant : <ratelimit> rpm | tpm | priority_generous | priority_strict
|
||||
<budget> key | internal_user | end_user | organization | team_member | tag
|
||||
| model_max | soft | key_multi_window | team_multi_window
|
||||
| fallback | spend_counter
|
||||
<spend_tracking> chat_completions | stream | embeddings | cache_hit | key_rollup
|
||||
| concurrent_burst | tags | end_user | per_model | failure
|
||||
| spend_calculate | pagination
|
||||
assertion : blocks_over_limit | resets_after_window | headers_report_remaining | picks_under_tpm
|
||||
| blocks_then_resets | resets_windows_independently | alerts_without_blocking
|
||||
| isolates_per_model | routes_to_fallback | reseed_matches_db | logs_cost | zero_cost
|
||||
| matches_sum_of_logs | loses_no_spend | attributes_spend | writes_own_rows
|
||||
| writes_failure_row | returns_cost | keeps_total
|
||||
e.g. quota_management.ratelimit.rpm.blocks_over_limit exercised_on=[chat_completions, messages]
|
||||
quota_management.budget.key.blocks_over_limit exercised_on=[chat_completions]
|
||||
```
|
||||
|
||||
Logging & Guardrails - behavior features (config-driven; endpoint is exercised_on)
|
||||
|
|
|
|||
|
|
@ -95,7 +95,7 @@ def pytest_sessionfinish(session: pytest.Session, exitstatus: int) -> None:
|
|||
path."""
|
||||
if not session.stash.get(_E2E_TEST_RAN, False):
|
||||
return
|
||||
spend_dir = str(Path(__file__).parent / "spend_tracking")
|
||||
spend_dir = str(Path(__file__).parent / "quota_management" / "spend_tracking")
|
||||
sys.path.insert(0, spend_dir)
|
||||
try:
|
||||
from spend_e2e_client import reset_spend_logs # pyright: ignore
|
||||
|
|
|
|||
|
|
@ -14,7 +14,8 @@ grouped `module > feature > test`, with LLM cells split into `Core LLMs` and
|
|||
`fail_before_fix` flag.
|
||||
|
||||
The rows live in per-prefix YAML files (`llm_*.yaml`, `mgmt.yaml`, `mcp.yaml`,
|
||||
`reliability.yaml`, `logging.yaml`, `guardrail.yaml`, `other.yaml`) and validate against
|
||||
`reliability.yaml`, `quota_management.yaml`, `logging.yaml`, `guardrail.yaml`,
|
||||
`other.yaml`) and validate against
|
||||
the discriminated union in `schema.py`, so an LLM row cannot carry a guardrail field and
|
||||
vice versa. `llm` rows with `subject_endpoint` of `chat_completions`, `messages`, or
|
||||
`responses` roll up to `Core LLMs`; all other LLM endpoints roll up to `Non-Core LLMs`.
|
||||
|
|
|
|||
35
tests/e2e/coverage_registry/quota_management.yaml
Normal file
35
tests/e2e/coverage_registry/quota_management.yaml
Normal file
|
|
@ -0,0 +1,35 @@
|
|||
# Quota Management (behavior features): rate limits, budgets, spend tracking. Grounded in
|
||||
# litellm/proxy/hooks/ + litellm/proxy/auth/auth_checks.py + litellm/proxy/spend_tracking/.
|
||||
- {id: quota_management.ratelimit.rpm.blocks_over_limit, module: quota_management, tier: P0, behavior: ratelimit, variant: rpm, assertions: [blocks_over_limit], exercised_on: [chat_completions, messages], source: "parallel_request_limiter_v3.py", rationale: "v3 limiter enforces RPM per key/team/model; 429 on breach"}
|
||||
- {id: quota_management.ratelimit.tpm.blocks_over_limit, module: quota_management, tier: P0, behavior: ratelimit, variant: tpm, assertions: [blocks_over_limit], exercised_on: [chat_completions, messages], source: "parallel_request_limiter_v3.py", rationale: "v3 limiter enforces TPM per key/team/model; 429 on breach"}
|
||||
- {id: quota_management.ratelimit.rpm.resets_after_window, module: quota_management, tier: P1, behavior: ratelimit, variant: rpm, assertions: [resets_after_window], exercised_on: [chat_completions], source: "parallel_request_limiter_v3.py", rationale: "Rate-limit window (LITELLM_RATE_LIMIT_WINDOW_SIZE, 60s default) expires; a blocked key serves again in the next window"}
|
||||
- {id: quota_management.ratelimit.rpm.headers_report_remaining, module: quota_management, tier: P1, behavior: ratelimit, variant: rpm, assertions: [headers_report_remaining], exercised_on: [chat_completions], source: "parallel_request_limiter_v3.py async_post_call_success_hook", rationale: "Successful responses carry x-ratelimit-api_key-{limit,remaining}-{requests,tokens} so clients can pace"}
|
||||
- {id: quota_management.ratelimit.priority_generous.picks_under_tpm, module: quota_management, tier: P1, behavior: ratelimit, variant: priority_generous, assertions: [picks_under_tpm], exercised_on: [chat_completions, messages], source: "dynamic_rate_limiter_v3.py:36-52", rationale: "Generous mode (<80% sat) allows priority borrowing"}
|
||||
- {id: quota_management.ratelimit.priority_strict.picks_under_tpm, module: quota_management, tier: P1, behavior: ratelimit, variant: priority_strict, assertions: [picks_under_tpm], exercised_on: [chat_completions, messages], source: "dynamic_rate_limiter_v3.py:53-71", rationale: "Strict mode (>=80% sat) enforces priority fairness"}
|
||||
- {id: quota_management.budget.key.blocks_over_limit, module: quota_management, tier: P0, behavior: budget, variant: key, assertions: [blocks_over_limit], exercised_on: [chat_completions], source: "proxy/auth/auth_checks.py", rationale: "A key's max_budget blocks further paid calls once spend crosses it"}
|
||||
- {id: quota_management.budget.internal_user.blocks_over_limit, module: quota_management, tier: P1, behavior: budget, variant: internal_user, assertions: [blocks_over_limit], exercised_on: [chat_completions], source: "proxy/auth/auth_checks.py", rationale: "An internal user's max_budget governs personal keys"}
|
||||
- {id: quota_management.budget.end_user.blocks_over_limit, module: quota_management, tier: P1, behavior: budget, variant: end_user, assertions: [blocks_over_limit], exercised_on: [chat_completions], source: "proxy/auth/auth_checks.py", rationale: "A customer (end-user) max_budget blocks calls attributed via user="}
|
||||
- {id: quota_management.budget.organization.blocks_over_limit, module: quota_management, tier: P1, behavior: budget, variant: organization, assertions: [blocks_over_limit], exercised_on: [chat_completions], source: "proxy/auth/auth_checks.py", rationale: "An organization's max_budget blocks keys under its teams"}
|
||||
- {id: quota_management.budget.team_member.blocks_over_limit, module: quota_management, tier: P1, behavior: budget, variant: team_member, assertions: [blocks_over_limit], exercised_on: [chat_completions], source: "proxy/auth/auth_checks.py", rationale: "A member's per-team budget blocks independently of the team budget"}
|
||||
- {id: quota_management.budget.tag.blocks_over_limit, module: quota_management, tier: P1, behavior: budget, variant: tag, assertions: [blocks_over_limit], exercised_on: [chat_completions], source: "router_strategy/budget_limiter.py", rationale: "Proxy-level tag budgets block tagged requests at the cap"}
|
||||
- {id: quota_management.budget.model_max.isolates_per_model, module: quota_management, tier: P1, behavior: budget, variant: model_max, assertions: [isolates_per_model], exercised_on: [chat_completions], source: "proxy/hooks/model_max_budget_limiter.py", rationale: "model_max_budget caps one model without touching a sibling's budget"}
|
||||
- {id: quota_management.budget.soft.alerts_without_blocking, module: quota_management, tier: P1, behavior: budget, variant: soft, assertions: [alerts_without_blocking], exercised_on: [chat_completions], source: "proxy/auth/auth_checks.py", rationale: "soft_budget alerts but never blocks traffic"}
|
||||
- {id: quota_management.budget.key.resets_after_window, module: quota_management, tier: P1, behavior: budget, variant: key, assertions: [resets_after_window], exercised_on: [chat_completions], source: "proxy/common_utils/reset_budget_job.py", rationale: "budget_duration zeroes key spend after the window; a blocked key serves again"}
|
||||
- {id: quota_management.budget.team_member.resets_after_window, module: quota_management, tier: P1, behavior: budget, variant: team_member, assertions: [resets_after_window], exercised_on: [chat_completions], source: "proxy/common_utils/reset_budget_job.py", rationale: "Member per-team budget reset keeps advancing window after window"}
|
||||
- {id: quota_management.budget.key_multi_window.blocks_then_resets, module: quota_management, tier: P1, behavior: budget, variant: key_multi_window, assertions: [blocks_then_resets], exercised_on: [chat_completions], source: "proxy/common_utils/reset_budget_job.py", rationale: "budget_limits enforce within a short window and serve again in the next"}
|
||||
- {id: quota_management.budget.key_multi_window.resets_windows_independently, module: quota_management, tier: P2, behavior: budget, variant: key_multi_window, assertions: [resets_windows_independently], exercised_on: [chat_completions], source: "proxy/common_utils/reset_budget_job.py", rationale: "Each window of a multi-window budget resets on its own schedule"}
|
||||
- {id: quota_management.budget.team_multi_window.blocks_then_resets, module: quota_management, tier: P1, behavior: budget, variant: team_multi_window, assertions: [blocks_then_resets], exercised_on: [chat_completions], source: "proxy/common_utils/reset_budget_job.py", rationale: "Team budget_limits enforce and reset per window"}
|
||||
- {id: quota_management.budget.fallback.routes_to_fallback, module: quota_management, tier: P1, behavior: budget, variant: fallback, assertions: [routes_to_fallback], exercised_on: [messages], source: "proxy/hooks/model_max_budget_limiter.py", rationale: "budget_fallbacks reroute to the fallback model once the primary's budget is exhausted"}
|
||||
- {id: quota_management.budget.spend_counter.reseed_matches_db, module: quota_management, tier: P2, behavior: budget, variant: spend_counter, assertions: [reseed_matches_db], exercised_on: [chat_completions], source: "proxy/spend_tracking/budget_reservation.py", rationale: "Concurrent cold-counter reseeds keep the enforcement counter equal to DB spend (#26829)"}
|
||||
- {id: quota_management.spend_tracking.chat_completions.logs_cost, module: quota_management, tier: P0, behavior: spend_tracking, variant: chat_completions, assertions: [logs_cost], exercised_on: [chat_completions], source: "proxy/spend_tracking/spend_tracking_utils.py", rationale: "A paid chat call writes a nonzero spend row"}
|
||||
- {id: quota_management.spend_tracking.stream.logs_cost, module: quota_management, tier: P1, behavior: spend_tracking, variant: stream, assertions: [logs_cost], exercised_on: [chat_completions], source: "proxy/spend_tracking/spend_tracking_utils.py", rationale: "Streaming responses aggregate token counts into a spend row"}
|
||||
- {id: quota_management.spend_tracking.embeddings.logs_cost, module: quota_management, tier: P1, behavior: spend_tracking, variant: embeddings, assertions: [logs_cost], exercised_on: [embeddings], source: "proxy/spend_tracking/spend_tracking_utils.py", rationale: "Embedding calls write nonzero spend rows"}
|
||||
- {id: quota_management.spend_tracking.cache_hit.zero_cost, module: quota_management, tier: P1, behavior: spend_tracking, variant: cache_hit, assertions: [zero_cost], exercised_on: [chat_completions], source: "proxy/spend_tracking/spend_tracking_utils.py", rationale: "A response-cache hit logs at zero cost with the cache-hit marker"}
|
||||
- {id: quota_management.spend_tracking.key_rollup.matches_sum_of_logs, module: quota_management, tier: P1, behavior: spend_tracking, variant: key_rollup, assertions: [matches_sum_of_logs], exercised_on: [chat_completions], source: "proxy/db/db_spend_update_writer.py", rationale: "A key's rolled-up spend equals the sum of its log rows"}
|
||||
- {id: quota_management.spend_tracking.concurrent_burst.loses_no_spend, module: quota_management, tier: P1, behavior: spend_tracking, variant: concurrent_burst, assertions: [loses_no_spend], exercised_on: [chat_completions], source: "proxy/db/db_spend_update_writer.py", rationale: "Concurrent calls all land as spend; no row lost to write contention"}
|
||||
- {id: quota_management.spend_tracking.tags.attributes_spend, module: quota_management, tier: P1, behavior: spend_tracking, variant: tags, assertions: [attributes_spend], exercised_on: [chat_completions], source: "proxy/spend_tracking/spend_tracking_utils.py", rationale: "Request tags round-trip to spend rows and tag rollups match tagged logs"}
|
||||
- {id: quota_management.spend_tracking.end_user.attributes_spend, module: quota_management, tier: P1, behavior: spend_tracking, variant: end_user, assertions: [attributes_spend], exercised_on: [chat_completions], source: "proxy/spend_tracking/spend_tracking_utils.py", rationale: "user= attribution lands the end-user id on the spend row"}
|
||||
- {id: quota_management.spend_tracking.per_model.writes_own_rows, module: quota_management, tier: P2, behavior: spend_tracking, variant: per_model, assertions: [writes_own_rows], exercised_on: [chat_completions], source: "proxy/spend_tracking/spend_tracking_utils.py", rationale: "Each model on a shared key gets its own spend row"}
|
||||
- {id: quota_management.spend_tracking.failure.writes_failure_row, module: quota_management, tier: P1, behavior: spend_tracking, variant: failure, assertions: [writes_failure_row], exercised_on: [chat_completions], source: "proxy/spend_tracking/spend_log_error_logger.py", rationale: "A failed call writes a failure-status spend row"}
|
||||
- {id: quota_management.spend_tracking.spend_calculate.returns_cost, module: quota_management, tier: P2, behavior: spend_tracking, variant: spend_calculate, assertions: [returns_cost], exercised_on: [spend_calculate], source: "proxy/spend_tracking/spend_management_endpoints.py", rationale: "/spend/calculate prices a hypothetical request at nonzero cost"}
|
||||
- {id: quota_management.spend_tracking.pagination.keeps_total, module: quota_management, tier: P2, behavior: spend_tracking, variant: pagination, assertions: [keeps_total], exercised_on: [chat_completions], source: "proxy/spend_tracking/spend_management_endpoints.py", rationale: "Spend-logs v2 pagination caps page size without losing the total"}
|
||||
|
|
@ -12,10 +12,6 @@
|
|||
- {id: reliability.cooldown.429.trips_then_recovers, module: reliability, tier: P0, behavior: cooldown, variant: "429", assertions: [trips_then_recovers], exercised_on: [chat_completions, messages], source: "cooldown_handlers.py:69", rationale: "Cools on 429, avoids hammering exhausted provider"}
|
||||
- {id: reliability.cooldown.auth.trips_then_recovers, module: reliability, tier: P1, behavior: cooldown, variant: auth, assertions: [trips_then_recovers], exercised_on: [chat_completions, messages], source: "cooldown_handlers.py:74", rationale: "Cools on 401 auth error"}
|
||||
- {id: reliability.cooldown.timeout.trips_then_recovers, module: reliability, tier: P1, behavior: cooldown, variant: timeout, assertions: [trips_then_recovers], exercised_on: [chat_completions, messages], source: "cooldown_handlers.py:77", rationale: "Cools on 408 timeout"}
|
||||
- {id: reliability.ratelimit.rpm.blocks_over_limit, module: reliability, tier: P0, behavior: ratelimit, variant: rpm, assertions: [blocks_over_limit], exercised_on: [chat_completions, messages], source: "dynamic_rate_limiter_v3.py", rationale: "v3 limiter enforces RPM per key/team/model; 429 on breach"}
|
||||
- {id: reliability.ratelimit.tpm.blocks_over_limit, module: reliability, tier: P0, behavior: ratelimit, variant: tpm, assertions: [blocks_over_limit], exercised_on: [chat_completions, messages], source: "dynamic_rate_limiter_v3.py", rationale: "v3 limiter enforces TPM per key/team/model; 429 on breach"}
|
||||
- {id: reliability.ratelimit.priority_generous.picks_under_tpm, module: reliability, tier: P1, behavior: ratelimit, variant: priority_generous, assertions: [picks_under_tpm], exercised_on: [chat_completions, messages], source: "dynamic_rate_limiter_v3.py:36-52", rationale: "Generous mode (<80% sat) allows priority borrowing"}
|
||||
- {id: reliability.ratelimit.priority_strict.picks_under_tpm, module: reliability, tier: P1, behavior: ratelimit, variant: priority_strict, assertions: [picks_under_tpm], exercised_on: [chat_completions, messages], source: "dynamic_rate_limiter_v3.py:53-71", rationale: "Strict mode (>=80% sat) enforces priority fairness"}
|
||||
- {id: reliability.routing.simple_shuffle.picks_healthy_deployment, module: reliability, tier: P1, behavior: routing, variant: simple_shuffle, assertions: [picks_healthy_deployment], exercised_on: [chat_completions, messages], source: "router_strategy/simple_shuffle.py", rationale: "Baseline weighted/uniform pick"}
|
||||
- {id: reliability.routing.latency_based.picks_lowest_latency, module: reliability, tier: P1, behavior: routing, variant: latency_based, assertions: [picks_lowest_latency], exercised_on: [chat_completions, messages], source: "router_strategy/lowest_latency.py", rationale: "Routes to lowest-latency deployment"}
|
||||
- {id: reliability.routing.cost_based.picks_lowest_cost, module: reliability, tier: P1, behavior: routing, variant: cost_based, assertions: [picks_lowest_cost], exercised_on: [chat_completions, messages], source: "router_strategy/lowest_cost.py", rationale: "Spend-aware routing"}
|
||||
|
|
|
|||
|
|
@ -1,7 +1,7 @@
|
|||
"""Registry row schema: the contract every denominator cell validates against.
|
||||
|
||||
A cell is one customer-noticeable behavior a single e2e test can assert pass/fail
|
||||
on. `module` is the id's segment-1 prefix (seven of them); dashboard rollups can
|
||||
on. `module` is the id's segment-1 prefix (eight of them); dashboard rollups can
|
||||
split or merge those prefixes. The union is discriminated on `module`, so an LLM
|
||||
row cannot carry a guardrail field and vice versa.
|
||||
"""
|
||||
|
|
@ -100,6 +100,13 @@ class ReliabilityCell(_Base):
|
|||
exercised_on: tuple[str, ...]
|
||||
|
||||
|
||||
class QuotaCell(_Base):
|
||||
module: Literal["quota_management"]
|
||||
behavior: Literal["ratelimit", "budget", "spend_tracking"]
|
||||
variant: str
|
||||
exercised_on: tuple[str, ...]
|
||||
|
||||
|
||||
class LoggingCell(_Base):
|
||||
module: Literal["logging"]
|
||||
event: str
|
||||
|
|
@ -122,6 +129,7 @@ Cell = Annotated[
|
|||
| MgmtCell
|
||||
| McpCell
|
||||
| ReliabilityCell
|
||||
| QuotaCell
|
||||
| LoggingCell
|
||||
| GuardrailCell
|
||||
| OtherCell,
|
||||
|
|
@ -142,6 +150,7 @@ PREFIX_ROLLUP: dict[str, str] = {
|
|||
"mcp": "MCPs",
|
||||
"mgmt": "Management/UI",
|
||||
"reliability": "Reliability & Performance",
|
||||
"quota_management": "Quota Management",
|
||||
"logging": "Logging & Guardrails",
|
||||
"guardrail": "Logging & Guardrails",
|
||||
"other": "Other",
|
||||
|
|
@ -153,6 +162,7 @@ MODULE_ORDER: tuple[str, ...] = (
|
|||
"MCPs",
|
||||
"Management/UI",
|
||||
"Reliability & Performance",
|
||||
"Quota Management",
|
||||
"Logging & Guardrails",
|
||||
"Other",
|
||||
)
|
||||
|
|
@ -163,6 +173,7 @@ LOKI_MODULE_LABELS: dict[str, str] = {
|
|||
"MCPs": "mcp",
|
||||
"Management/UI": "management_ui",
|
||||
"Reliability & Performance": "reliability_performance",
|
||||
"Quota Management": "quota_management",
|
||||
"Logging & Guardrails": "logging_guardrails",
|
||||
"Other": "other",
|
||||
}
|
||||
|
|
|
|||
|
|
@ -1,6 +1,6 @@
|
|||
[pytest]
|
||||
# Config when any e2e suite under tests/e2e/ is run directly, e.g.
|
||||
# uv run pytest tests/e2e/spend_tracking/ -v
|
||||
# uv run pytest tests/e2e/quota_management/spend_tracking/ -v
|
||||
# The e2e marker is also registered in conftest.py for runs rooted elsewhere.
|
||||
addopts = --strict-markers --strict-config
|
||||
markers =
|
||||
|
|
|
|||
|
|
@ -12,7 +12,7 @@ Pre-existing live coverage outside this suite:
|
|||
- `tests/local_testing/test_router_budget_limiter.py` - provider / tag / deployment
|
||||
budgets at the router.
|
||||
|
||||
This suite (`tests/e2e/budgets/`) adds the missing live coverage and runs
|
||||
This suite (`tests/e2e/quota_management/budgets/`) adds the missing live coverage and runs
|
||||
on the shared lifecycle (every entity it creates is deleted on teardown).
|
||||
|
||||
---
|
||||
|
|
@ -15,6 +15,7 @@ from lifecycle import ResourceManager
|
|||
pytestmark = pytest.mark.e2e
|
||||
|
||||
|
||||
@pytest.mark.covers("mgmt.budget.new.persists")
|
||||
def test_budget_crud_roundtrip(client: BudgetClient, resources: ResourceManager) -> None:
|
||||
budget_id = client.create_budget(max_budget=12.5, soft_budget=10.0, budget_duration="30d")
|
||||
resources.defer(lambda: client.delete_budget(budget_id))
|
||||
|
|
@ -36,6 +37,7 @@ def test_budget_crud_roundtrip(client: BudgetClient, resources: ResourceManager)
|
|||
)
|
||||
|
||||
|
||||
@pytest.mark.covers("mgmt.budget.delete.persists")
|
||||
def test_budget_delete_removes_it(client: BudgetClient, resources: ResourceManager) -> None:
|
||||
budget_id = client.create_budget(max_budget=1.0)
|
||||
resources.defer(lambda: client.delete_budget(budget_id))
|
||||
|
|
@ -134,11 +134,26 @@ def _case_id(case_cls: Type[_BudgetCase]) -> str:
|
|||
@pytest.mark.parametrize(
|
||||
"case_cls",
|
||||
[
|
||||
KeyBudgetCase,
|
||||
InternalUserBudgetCase,
|
||||
EndUserBudgetCase,
|
||||
OrganizationBudgetCase,
|
||||
TeamMemberBudgetCase,
|
||||
pytest.param(
|
||||
KeyBudgetCase,
|
||||
marks=pytest.mark.covers("quota_management.budget.key.blocks_over_limit"),
|
||||
),
|
||||
pytest.param(
|
||||
InternalUserBudgetCase,
|
||||
marks=pytest.mark.covers("quota_management.budget.internal_user.blocks_over_limit"),
|
||||
),
|
||||
pytest.param(
|
||||
EndUserBudgetCase,
|
||||
marks=pytest.mark.covers("quota_management.budget.end_user.blocks_over_limit"),
|
||||
),
|
||||
pytest.param(
|
||||
OrganizationBudgetCase,
|
||||
marks=pytest.mark.covers("quota_management.budget.organization.blocks_over_limit"),
|
||||
),
|
||||
pytest.param(
|
||||
TeamMemberBudgetCase,
|
||||
marks=pytest.mark.covers("quota_management.budget.team_member.blocks_over_limit"),
|
||||
),
|
||||
],
|
||||
ids=_case_id,
|
||||
)
|
||||
|
|
@ -19,6 +19,7 @@ PRIMARY_MODEL = "claude-haiku-4-5"
|
|||
FALLBACK_MODEL = "gpt-5.5"
|
||||
|
||||
|
||||
@pytest.mark.covers("quota_management.budget.fallback.routes_to_fallback")
|
||||
def test_budget_fallback_reroutes_anthropic_messages_to_openai(
|
||||
client: BudgetClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -70,6 +70,7 @@ def test_key_with_budget_duration_schedules_reset_at_creation(
|
|||
# ---- Rung 2: enforcement trips at the cap ------------------------------------
|
||||
|
||||
|
||||
@pytest.mark.covers("quota_management.budget.key.blocks_over_limit")
|
||||
def test_key_spend_blocks_at_cap(client: BudgetClient, resources: ResourceManager) -> None:
|
||||
"""Sanity that the tiny cap is enforced before we test that it resets: spend
|
||||
accrues across calls and eventually returns budget_exceeded, never a 5xx."""
|
||||
|
|
@ -86,6 +87,7 @@ def test_key_spend_blocks_at_cap(client: BudgetClient, resources: ResourceManage
|
|||
# ---- Rung 3: the core regression - reset_at strictly advances + spend zeroes --
|
||||
|
||||
|
||||
@pytest.mark.covers("quota_management.budget.key.resets_after_window")
|
||||
def test_key_budget_reset_at_advances_after_window(
|
||||
client: BudgetClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -123,6 +125,7 @@ def test_key_budget_reset_at_advances_after_window(
|
|||
# ---- Rung 4: multi-window - tight window resets, roomy window keeps spend -----
|
||||
|
||||
|
||||
@pytest.mark.covers("quota_management.budget.key_multi_window.resets_windows_independently")
|
||||
def test_multi_window_key_resets_each_window_independently(
|
||||
client: BudgetClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -164,6 +167,7 @@ def test_multi_window_key_resets_each_window_independently(
|
|||
# ---- Rung 5: team-member window advances (JSON-backed per-team budget) --------
|
||||
|
||||
|
||||
@pytest.mark.covers("quota_management.budget.team_member.resets_after_window")
|
||||
def test_team_member_budget_reset_at_advances(
|
||||
client: BudgetClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -26,6 +26,7 @@ def _call(client: BudgetClient, key: str):
|
|||
)
|
||||
|
||||
|
||||
@pytest.mark.covers("quota_management.budget.key.resets_after_window")
|
||||
def test_key_budget_resets_after_duration(
|
||||
client: BudgetClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -28,6 +28,7 @@ def _call(client: BudgetClient, key: str, model: str):
|
|||
return result
|
||||
|
||||
|
||||
@pytest.mark.covers("quota_management.budget.model_max.isolates_per_model")
|
||||
def test_model_max_budget_isolates_per_model(
|
||||
client: BudgetClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -29,6 +29,7 @@ def _call(client: BudgetClient, key: str):
|
|||
)
|
||||
|
||||
|
||||
@pytest.mark.covers("quota_management.budget.key_multi_window.blocks_then_resets")
|
||||
def test_short_window_blocks_then_resets(
|
||||
client: BudgetClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -17,6 +17,7 @@ from lifecycle import ResourceManager
|
|||
pytestmark = pytest.mark.e2e
|
||||
|
||||
|
||||
@pytest.mark.covers("quota_management.budget.soft.alerts_without_blocking")
|
||||
def test_soft_budget_does_not_block(
|
||||
client: BudgetClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -121,6 +121,7 @@ def _accumulate(client: BudgetClient, key: str, count: int) -> None:
|
|||
list(pool.map(one, range(count)))
|
||||
|
||||
|
||||
@pytest.mark.covers("quota_management.budget.spend_counter.reseed_matches_db")
|
||||
def test_cold_counter_reseed_keeps_counter_equal_to_db_spend(
|
||||
client: BudgetClient, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -33,6 +33,7 @@ def _tagged_call(client: BudgetClient, key: str, tag: str):
|
|||
return result
|
||||
|
||||
|
||||
@pytest.mark.covers("quota_management.budget.tag.blocks_over_limit")
|
||||
def test_tag_budget_blocks_tagged_requests(
|
||||
client: BudgetClient, scoped_key: str, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -97,6 +97,7 @@ class TestTeamMemberBudget:
|
|||
f"call {row.request_id} logged under user {row.user}, not member {member.user_id}"
|
||||
)
|
||||
|
||||
@pytest.mark.covers("quota_management.budget.team_member.blocks_over_limit")
|
||||
def test_member_spend_over_budget_is_blocked(self, client: BudgetClient, member: _Member) -> None:
|
||||
for _ in range(40):
|
||||
result = client.chat(member.key, MODEL, f"spend {unique_marker()}", max_tokens=16)
|
||||
|
|
@ -16,6 +16,7 @@ def _as_datetime(value: str) -> datetime:
|
|||
return datetime.fromisoformat(value.replace("Z", "+00:00"))
|
||||
|
||||
|
||||
@pytest.mark.covers("quota_management.budget.team_member.resets_after_window")
|
||||
def test_team_member_budget_reset_keeps_advancing(client: BudgetClient, resources: ResourceManager) -> None:
|
||||
team_id = client.create_team(alias=f"e2e-member-reset-{unique_marker()}", max_budget=100.0)
|
||||
resources.defer(lambda: client.delete_team(team_id))
|
||||
|
|
@ -32,6 +32,7 @@ def _call(client: BudgetClient, key: str):
|
|||
return client.chat(key, "claude-haiku-4-5", f"team-window {unique_marker()}", max_tokens=16)
|
||||
|
||||
|
||||
@pytest.mark.covers("quota_management.budget.team_multi_window.blocks_then_resets")
|
||||
def test_team_short_window_blocks_then_resets(client: BudgetClient, resources: ResourceManager) -> None:
|
||||
team_id = client.create_team(
|
||||
alias=f"e2e-team-window-{unique_marker()}",
|
||||
|
|
@ -59,6 +59,7 @@ def _require_row(
|
|||
return matches[0]
|
||||
|
||||
|
||||
@pytest.mark.covers("quota_management.spend_tracking.chat_completions.logs_cost")
|
||||
def test_chat_completion_writes_nonzero_spend_row(
|
||||
client: SpendClient, scoped_key: str
|
||||
) -> None:
|
||||
|
|
@ -93,6 +94,7 @@ def test_chat_completion_writes_nonzero_spend_row(
|
|||
), f"row request_id != client response.id ({chat.id})"
|
||||
|
||||
|
||||
@pytest.mark.covers("quota_management.spend_tracking.stream.logs_cost")
|
||||
def test_streaming_chat_completion_tracks_spend(
|
||||
client: SpendClient, scoped_key: str
|
||||
) -> None:
|
||||
|
|
@ -120,6 +122,7 @@ def test_streaming_chat_completion_tracks_spend(
|
|||
assert (row.total_tokens or 0) == prompt + completion
|
||||
|
||||
|
||||
@pytest.mark.covers("quota_management.spend_tracking.embeddings.logs_cost")
|
||||
def test_embedding_writes_nonzero_spend_row(
|
||||
client: SpendClient, scoped_key: str
|
||||
) -> None:
|
||||
|
|
@ -142,6 +145,7 @@ def test_embedding_writes_nonzero_spend_row(
|
|||
assert "text-embedding-3-small" in (row.model or "")
|
||||
|
||||
|
||||
@pytest.mark.covers("quota_management.spend_tracking.cache_hit.zero_cost")
|
||||
def test_cache_hit_is_zero_cost_and_suffixed(
|
||||
client: SpendClient, scoped_key: str
|
||||
) -> None:
|
||||
|
|
@ -177,6 +181,7 @@ def test_cache_hit_is_zero_cost_and_suffixed(
|
|||
), f"the non-cached call should still be charged: {_summarize(rows)}"
|
||||
|
||||
|
||||
@pytest.mark.covers("quota_management.spend_tracking.key_rollup.matches_sum_of_logs")
|
||||
def test_key_spend_equals_sum_of_logs(client: SpendClient, scoped_key: str) -> None:
|
||||
for _ in range(2):
|
||||
_ = unwrap(
|
||||
|
|
@ -203,6 +208,7 @@ def test_key_spend_equals_sum_of_logs(client: SpendClient, scoped_key: str) -> N
|
|||
), f"key aggregate {key_spend} != sum of logs {logs_total}; rows: {_summarize(rows)}"
|
||||
|
||||
|
||||
@pytest.mark.covers("quota_management.spend_tracking.concurrent_burst.loses_no_spend")
|
||||
def test_burst_of_concurrent_calls_loses_no_spend(
|
||||
client: SpendClient, scoped_key: str
|
||||
) -> None:
|
||||
|
|
@ -249,6 +255,7 @@ def test_burst_of_concurrent_calls_loses_no_spend(
|
|||
)
|
||||
|
||||
|
||||
@pytest.mark.covers("quota_management.spend_tracking.pagination.keeps_total")
|
||||
def test_spend_logs_v2_pagination_caps_pages_and_keeps_total(
|
||||
client: SpendClient, scoped_key: str
|
||||
) -> None:
|
||||
|
|
@ -301,6 +308,7 @@ def test_spend_logs_v2_pagination_caps_pages_and_keeps_total(
|
|||
)
|
||||
|
||||
|
||||
@pytest.mark.covers("quota_management.spend_tracking.tags.attributes_spend")
|
||||
def test_request_tags_round_trip(client: SpendClient, scoped_key: str) -> None:
|
||||
tag = f"e2e-spend-{unique_marker()}"
|
||||
_ = unwrap(
|
||||
|
|
@ -317,6 +325,7 @@ def test_request_tags_round_trip(client: SpendClient, scoped_key: str) -> None:
|
|||
)
|
||||
|
||||
|
||||
@pytest.mark.covers("quota_management.spend_tracking.tags.attributes_spend")
|
||||
def test_tag_spend_matches_sum_of_tagged_logs(
|
||||
client: SpendClient, scoped_key: str
|
||||
) -> None:
|
||||
|
|
@ -354,6 +363,7 @@ def test_tag_spend_matches_sum_of_tagged_logs(
|
|||
)
|
||||
|
||||
|
||||
@pytest.mark.covers("quota_management.spend_tracking.end_user.attributes_spend")
|
||||
def test_end_user_spend_attributed_on_row(
|
||||
client: SpendClient, scoped_key: str, resources: ResourceManager
|
||||
) -> None:
|
||||
|
|
@ -371,6 +381,7 @@ def test_end_user_spend_attributed_on_row(
|
|||
assert (row.spend or 0) > 0, f"end-user row should cost > 0: {_summarize(rows)}"
|
||||
|
||||
|
||||
@pytest.mark.covers("quota_management.spend_tracking.per_model.writes_own_rows")
|
||||
def test_each_model_on_a_shared_key_gets_its_own_row(
|
||||
client: SpendClient, scoped_key: str
|
||||
) -> None:
|
||||
|
|
@ -419,6 +430,7 @@ def test_each_model_on_a_shared_key_gets_its_own_row(
|
|||
), f"claude row request_id {claude_row.request_id} != response id {claude.id}"
|
||||
|
||||
|
||||
@pytest.mark.covers("quota_management.spend_tracking.failure.writes_failure_row")
|
||||
def test_failure_call_writes_failure_status_row(
|
||||
client: SpendClient, scoped_key: str
|
||||
) -> None:
|
||||
|
|
@ -438,6 +450,7 @@ def test_failure_call_writes_failure_status_row(
|
|||
assert (failure_rows[0].spend or 0) == 0.0, "failed call must not be charged"
|
||||
|
||||
|
||||
@pytest.mark.covers("quota_management.spend_tracking.spend_calculate.returns_cost")
|
||||
def test_spend_calculate_returns_nonzero_cost(client: SpendClient) -> None:
|
||||
cost = client.calculate_spend(
|
||||
"gemini-2.5-flash", "estimate the cost of this request"
|
||||
Loading…
Add table
Reference in a new issue