This commit is contained in:
Yassin Kortam 2026-08-28 01:15:19 +08:00 • committed by GitHub
commit 217f67f3b7
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
2 changed files with 767 additions and 0 deletions

View file

@ -0,0 +1,387 @@
"""Live e2e: the proxy's configuration lifecycle - what the running process took
from its config file at boot, and what it accepts at runtime without a restart.
Boot is only observable through the state the process now exposes, so each test
keys off something that exists *only* because the startup config loader ran:
- the config file's `model_list` is live in the routing catalog. /model/info marks
a deployment the process read from its file with `db_model: false` (a DB-stored
deployment is `true`), so a `db_model: false` entry that serves a real
completion is the file's model_list loaded, credentials and all. The file's
`general_settings` are proven live the same way: /model/new only persists a
deployment when `store_model_in_db` came off that file, and the deployment comes
back from the catalog flagged `db_model: true`
- the `os.environ/` references in that file were resolved into real values. A
credential reference is a poor witness: the router resolves `os.environ/` refs
again at call time, so a completion would succeed even if boot-time resolution
had never happened. The cache block is not re-resolved anywhere - the response
cache is built once at startup from `cache_params.host` / `.port` - so a
redis-typed cache that actually serves a repeated request can only exist if
those two references resolved at boot; an unresolved literal would leave the
cache dialling the hostname "os.environ/REDIS_HOST" and nothing would ever hit
- /config/update reaches the running process, not just the DB. The test adds a
uniquely named model-group alias, reads it back off the live router through
/get/config/callbacks (process state, not a DB row), drives a completion through
the alias, then removes it and watches the alias stop resolving. The alias is
namespaced per run, and every write goes through `_write_alias`, which compares
the aliases the test does not own against a baseline taken at the start, names
and targets both, before writing and again after. Anything else appearing,
vanishing or being repointed stops the test with a message naming the
difference, and nothing is written, so a concurrent writer's change is never
reverted with a stale view. A second alias of the test's own stands across the
whole exchange and is asserted to survive. /config/update rewrites the alias map
wholesale and offers neither a per-key write nor a version to compare against,
so this is the honest ceiling: the exposure becomes a loud, diagnosable failure
rather than silent data loss
These tests assume the proxy under test was booted from a config file that wires
the example models and a redis cache through `os.environ/` references, which is
the setup tests/e2e/CONTRIBUTING.md prescribes.
"""
from __future__ import annotations
import time
from collections.abc import Callable
from dataclasses import dataclass
import pytest
from pydantic import BaseModel, ConfigDict
from e2e_config import CHEAP_OPENAI_MODEL, unique_marker
from e2e_http import NoBody, StreamingResponse, unwrap
from lifecycle import ResourceManager
from models import ChatBody, ChatMessage, LiteLLMParamsBody
from proxy_client import ProxyClient
pytestmark = pytest.mark.e2e
CACHE_HIT_HEADER = "x-litellm-cache-key"
CACHE_HIT_DEADLINE_SECONDS = 30.0
class CatalogModelInfo(BaseModel):
"""The /model/info `model_info` block, narrowed to the flag that says where the
deployment came from: `db_model` is false for a deployment read off the config
file at startup and true for one stored in the DB."""
db_model: bool = False
class CatalogEntry(BaseModel):
model_config = ConfigDict(protected_namespaces=())
model_name: str
model_info: CatalogModelInfo = CatalogModelInfo()
class CatalogResponse(BaseModel):
data: list[CatalogEntry] = []
class CacheReadiness(BaseModel):
"""GET /health/readiness/details, narrowed to the cache the process built at
startup (`litellm.cache.type`)."""
status: str
cache: str | None = None
class LiveRouterSettings(BaseModel):
model_config = ConfigDict(protected_namespaces=())
model_group_alias: dict[str, str] = {}
class ConfigCallbacksResponse(BaseModel):
"""GET /get/config/callbacks. `router_settings` is read straight off the live
Router object, so it reports what the running process is using rather than
what any config row holds."""
router_settings: LiveRouterSettings
class RouterSettingsUpdate(BaseModel):
model_config = ConfigDict(protected_namespaces=())
model_group_alias: dict[str, str]
class ConfigUpdateBody(BaseModel):
router_settings: RouterSettingsUpdate
class ConfigUpdateResult(BaseModel):
message: str
class CompletionId(BaseModel):
id: str
@dataclass(frozen=True, slots=True)
class ConfigClient:
"""The catalog, health and config routes these tests read the proxy's own
configuration state back through."""
proxy: ProxyClient
def catalog(self) -> list[CatalogEntry]:
return unwrap(
self.proxy.transport.get(
"/model/info",
headers=self.proxy.transport.master,
params=NoBody(),
response_type=CatalogResponse,
)
).data
def cache_readiness(self) -> CacheReadiness:
return unwrap(
self.proxy.transport.get(
"/health/readiness/details",
headers=self.proxy.transport.master,
params=NoBody(),
response_type=CacheReadiness,
)
)
def live_router_settings(self) -> LiveRouterSettings:
return unwrap(
self.proxy.transport.get(
"/get/config/callbacks",
headers=self.proxy.transport.master,
params=NoBody(),
response_type=ConfigCallbacksResponse,
)
).router_settings
def set_model_group_alias(self, aliases: dict[str, str]) -> None:
_ = unwrap(
self.proxy.transport.post(
"/config/update",
headers=self.proxy.transport.master,
json=ConfigUpdateBody(router_settings=RouterSettingsUpdate(model_group_alias=aliases)),
response_type=ConfigUpdateResult,
)
)
def chat_status(self, key: str, model: str, content: str) -> StreamingResponse:
return self.proxy.transport.send(
"/chat/completions",
headers=self.proxy.transport.bearer(key),
json=ChatBody(
model=model,
messages=[ChatMessage(role="user", content=content)],
max_tokens=8,
),
)
@pytest.fixture
def config(proxy: ProxyClient) -> ConfigClient:
return ConfigClient(proxy=proxy)
def _poll[T](attempt: Callable[[], T | None], *, deadline_seconds: float, interval: float, failure: str) -> T:
deadline = time.monotonic() + deadline_seconds
while time.monotonic() < deadline:
found = attempt()
if found is not None:
return found
time.sleep(interval)
pytest.fail(failure)
def _entry(catalog: list[CatalogEntry], model_name: str) -> CatalogEntry | None:
return next((entry for entry in catalog if entry.model_name == model_name), None)
def _foreign_aliases(live: dict[str, str], owned: tuple[str, ...]) -> dict[str, str]:
"""Every alias the test does not own, names and targets both."""
return {name: target for name, target in live.items() if name not in owned}
def _write_alias(
config: ConfigClient,
*,
alias: str,
target: str | None,
owned: tuple[str, ...],
baseline: dict[str, str],
) -> None:
"""Set or remove one owned alias, refusing to write at all if anything else in
the map has moved.
/config/update is the only route that writes model_group_alias, it takes the
whole map, and it offers neither a per-key write nor a version to compare
against, so every writer is a read-modify-write that can revert a concurrent
change. Rather than paper over that, the aliases the test does not own are
compared against `baseline` as full name/target pairs before the write and
again after it: a name that appeared or vanished, or an existing alias
repointed at a different model group, stops the test with a message naming the
difference instead of being quietly overwritten with a stale value. Nothing is
written once the map has moved, so the other writer's state stands.
"""
live = config.live_router_settings().model_group_alias
_assert_unmoved(_foreign_aliases(live, owned), baseline, when=f"before writing {alias!r}")
keep = _foreign_aliases(live, owned) | {name: live[name] for name in owned if name in live and name != alias}
config.set_model_group_alias(keep if target is None else {**keep, alias: target})
settled = config.live_router_settings().model_group_alias
_assert_unmoved(_foreign_aliases(settled, owned), baseline, when=f"after writing {alias!r}")
def _release_alias(config: ConfigClient, alias: str) -> None:
"""Teardown safety net: drop one alias from the map as it stands right now.
Deliberately assertion-free and built from a fresh read, so it removes the
test's own entry without restoring anything else to an older value; teardown
swallows failures, so a stop-the-test check here would be invisible anyway.
"""
live = config.live_router_settings().model_group_alias
config.set_model_group_alias({name: target for name, target in live.items() if name != alias})
def _assert_unmoved(current: dict[str, str], baseline: dict[str, str], *, when: str) -> None:
added = sorted(name for name in current if name not in baseline)
removed = sorted(name for name in baseline if name not in current)
repointed = sorted(name for name, target in baseline.items() if name in current and current[name] != target)
assert current == baseline, (
f"the model_group_alias map moved under the test {when}: added {added}, removed {removed}, "
f"repointed {repointed}. /config/update rewrites the whole map, so continuing would write a stale "
"view back over another writer's change; nothing was written"
)
class TestStartupConfigLoad:
@pytest.mark.covers("other.lifecycle.startup.config_loads")
def test_config_file_model_list_and_general_settings_are_live(
self, config: ConfigClient, resources: ResourceManager, scoped_key: str
) -> None:
from_file = _entry(config.catalog(), CHEAP_OPENAI_MODEL)
assert from_file is not None, (
f"the proxy serves no {CHEAP_OPENAI_MODEL!r} deployment, so its config file's model_list never loaded"
)
assert not from_file.model_info.db_model, (
f"{CHEAP_OPENAI_MODEL!r} is flagged db_model=true, so it came from the DB rather than the config "
"file the process booted with; the file's model_list is not what is being served"
)
served = config.chat_status(scoped_key, CHEAP_OPENAI_MODEL, f"reply with one word {unique_marker()}")
assert served.status_code == 200, (
f"the deployment loaded from the config file must serve a real completion, got "
f"{served.status_code}: {served.body[:300]}"
)
stored_name = f"e2e-boot-cfg-{unique_marker()}"
model_id = config.proxy.create_model(
stored_name,
LiteLLMParamsBody(model="openai/gpt-4o-mini", api_key="e2e-dummy-key"),
)
resources.defer(lambda: config.proxy.delete_model(model_id))
stored = _entry(config.catalog(), stored_name)
assert stored is not None, f"{stored_name!r} is absent from /model/info right after /model/new"
assert stored.model_info.db_model, (
f"{stored_name!r} came back flagged db_model=false; the config file's general_settings "
"(store_model_in_db) are not in effect on the running process"
)
@pytest.mark.covers("other.lifecycle.startup.env_vars_resolved")
def test_env_referenced_cache_block_resolved_at_startup(self, config: ConfigClient, scoped_key: str) -> None:
readiness = config.cache_readiness()
assert readiness.cache == "redis", (
f"the process reports cache {readiness.cache!r}; the config file's redis cache_params block, "
"whose host and port are os.environ/ references, is not the cache the proxy built at startup"
)
prompt = f"reply with one word {unique_marker()}"
first = config.chat_status(scoped_key, CHEAP_OPENAI_MODEL, prompt)
assert first.status_code == 200, (
f"the call being cached must succeed first, got {first.status_code}: {first.body[:300]}"
)
assert CACHE_HIT_HEADER not in first.headers, (
f"a first-of-its-kind prompt came back as a cache hit ({first.headers.get(CACHE_HIT_HEADER)}), "
"so the repeat below would prove nothing"
)
repeated = _poll(
lambda: (lambda outcome: outcome if CACHE_HIT_HEADER in outcome.headers else None)(
config.chat_status(scoped_key, CHEAP_OPENAI_MODEL, prompt)
),
deadline_seconds=CACHE_HIT_DEADLINE_SECONDS,
interval=5.0,
failure=(
"an identical repeat call was never served from the redis response cache, so the "
"os.environ/ host and port that cache was configured with never resolved into a "
"reachable redis at startup"
),
)
assert CompletionId.model_validate_json(repeated.body).id == CompletionId.model_validate_json(first.body).id, (
"the repeat call carried a cache-key header but returned a different completion, so it was "
"not the stored response coming back"
)
class TestRuntimeConfigUpdate:
@pytest.mark.covers("other.config.runtime_update.applies_at_runtime")
def test_config_update_reaches_the_running_router_and_can_be_taken_back(
self, config: ConfigClient, resources: ResourceManager, scoped_key: str
) -> None:
alias = f"e2e-cfg-alias-{unique_marker()}"
bystander = f"e2e-cfg-bystander-{unique_marker()}"
owned = (alias, bystander)
baseline = _foreign_aliases(config.live_router_settings().model_group_alias, owned)
assert alias not in baseline and bystander not in baseline, (
f"the run's aliases {owned} are somehow already configured"
)
resources.defer(lambda: _release_alias(config, bystander))
_write_alias(config, alias=bystander, target=CHEAP_OPENAI_MODEL, owned=owned, baseline=baseline)
unknown = config.chat_status(scoped_key, alias, "should not route yet")
assert unknown.status_code == 400, (
f"an unconfigured model group must be rejected 400 before the update, got {unknown.status_code}: "
f"{unknown.body[:300]}"
)
resources.defer(lambda: _release_alias(config, alias))
_write_alias(config, alias=alias, target=CHEAP_OPENAI_MODEL, owned=owned, baseline=baseline)
applied = _poll(
lambda: (lambda live: live if live.model_group_alias.get(alias) == CHEAP_OPENAI_MODEL else None)(
config.live_router_settings()
),
deadline_seconds=config.proxy.poll_timeout,
interval=5.0,
failure=(
f"the live router never picked up model_group_alias {alias!r} after /config/update, so the "
"update only reached the DB and would need a restart to take effect"
),
)
assert applied.model_group_alias[alias] == CHEAP_OPENAI_MODEL
routed = config.chat_status(scoped_key, alias, f"reply with one word {unique_marker()}")
assert routed.status_code == 200, (
f"the alias added at runtime must route to {CHEAP_OPENAI_MODEL!r}, got {routed.status_code}: "
f"{routed.body[:300]}"
)
_write_alias(config, alias=alias, target=None, owned=owned, baseline=baseline)
_poll(
lambda: (
True
if config.chat_status(scoped_key, alias, "should not route any more").status_code == 400
else None
),
deadline_seconds=config.proxy.poll_timeout,
interval=5.0,
failure=f"the alias {alias!r} still routed after being removed at runtime",
)
settled = config.live_router_settings().model_group_alias
assert alias not in settled, f"the live router still carries {alias!r} after the removing /config/update"
assert settled.get(bystander) == CHEAP_OPENAI_MODEL, (
f"removing {alias!r} also took {bystander!r} with it; /config/update replaces the alias map "
"wholesale, so a caller that writes back anything other than the map as it currently stands "
"wipes aliases it never touched"
)

View file

@ -0,0 +1,380 @@
"""Live e2e: what an already-issued virtual key can still do to itself.
Three post-issue behaviors of a virtual key, each driven through the admin route
that owns it and judged on live traffic:
- rotation with a grace period: `/key/{key}/regenerate` with `grace_period` keeps
the old secret authenticating until the window closes, then stops accepting it.
A sibling key rotated in the same test *without* a grace period is the control:
it is refused immediately, so the graced key still being served cannot be an
auth-cache artifact
- spend reset: `/key/{key}/reset_spend` puts the key's accumulated spend back to
the requested value, and a key that its own max_budget had blocked serves again.
Spend arrives on a buffered batch write, so the test settles on a value that has
stopped moving before it resets, and afterwards asserts only relationships a
late flush cannot violate: the reset reports a previous value no smaller than
the settled one (spend never shrinks on its own), reports zero, and the DB is
polled until it agrees. Exact equality against a single sample would go red on
correct behavior the moment a flush landed between two reads
- the allow side of `allowed_routes`: a key granted the `llm_api_routes` group
reaches the LLM endpoints, while a management route stays refused, so the grant
is a real whitelist rather than an absent check (the deny side lives in
tests/e2e/access_control/)
Every key is minted here and deleted on teardown; auth is judged by status code
(401 vs not-401), never by requiring a completion, so a provider hiccup cannot be
mistaken for a revocation.
"""
from __future__ import annotations
import time
from collections.abc import Callable
from dataclasses import dataclass
import pytest
from pydantic import BaseModel
from e2e_config import CHEAP_OPENAI_MODEL, unique_marker
from e2e_http import NoBody, Result, StreamingResponse, is_ok, unwrap
from lifecycle import ResourceManager
from models import (
ChatBody,
ChatMessage,
EmbedBody,
KeyGenerateBody,
KeyInfoParams,
LiteLLMParamsBody,
ModelInfoBody,
ModelNewBody,
)
from proxy_client import ProxyClient
pytestmark = pytest.mark.e2e
EMBEDDING_MODEL = "openai-text-embedding-3-small"
ROUTE_NOT_ALLOWED_MARKER = "not allowed to call this route"
BUDGET_BLOCK_MARKER = "budget_exceeded"
GRACE_PERIOD = "60s"
REVOCATION_DEADLINE_SECONDS = 240.0
TINY_BUDGET = 3e-6
# Spend lands on a buffered batch write (proxy_batch_write_at), so a freshly read
# value can still be superseded. Re-read across a window wider than that flush
# before treating a value as final.
SPEND_SETTLE_SECONDS = 25.0
SPEND_SAMPLE_INTERVAL = 5.0
class RegeneratedKey(BaseModel):
key: str
class KeyRegenerateBody(BaseModel):
"""POST /key/{key}/regenerate. `grace_period` is a duration string ("60s",
"24h"); omitted means the old secret is revoked the moment the new one is
issued."""
grace_period: str | None = None
class ResetSpendBody(BaseModel):
reset_to: float
class ResetSpendResult(BaseModel):
"""POST /key/{key}/reset_spend. Reports the spend the key carried before the
reset alongside the value it now holds."""
spend: float
previous_spend: float
class ScopedKeyInfo(BaseModel):
allowed_routes: list[str] | None = None
models: list[str] = []
spend: float | None = None
max_budget: float | None = None
class ScopedKeyInfoResponse(BaseModel):
info: ScopedKeyInfo
@dataclass(frozen=True, slots=True)
class KeyLifecycleClient:
"""The admin routes that act on an existing key, plus the LLM and management
calls used to judge what that key may still do."""
proxy: ProxyClient
def key_info(self, key: str) -> ScopedKeyInfo:
return unwrap(
self.proxy.transport.get(
"/key/info",
headers=self.proxy.transport.master,
params=KeyInfoParams(key=key),
response_type=ScopedKeyInfoResponse,
)
).info
def regenerate(self, key: str, *, grace_period: str | None = None) -> str:
return unwrap(
self.proxy.transport.post(
f"/key/{key}/regenerate",
headers=self.proxy.transport.master,
json=KeyRegenerateBody(grace_period=grace_period),
response_type=RegeneratedKey,
)
).key
def reset_spend(self, key: str, *, reset_to: float) -> ResetSpendResult:
return unwrap(
self.proxy.transport.post(
f"/key/{key}/reset_spend",
headers=self.proxy.transport.master,
json=ResetSpendBody(reset_to=reset_to),
response_type=ResetSpendResult,
)
)
def chat_status(self, key: str, model: str = CHEAP_OPENAI_MODEL) -> StreamingResponse:
return self.proxy.transport.send(
"/chat/completions",
headers=self.proxy.transport.bearer(key),
json=ChatBody(
model=model,
messages=[ChatMessage(role="user", content=f"reply with one word {unique_marker()}")],
max_tokens=8,
),
)
def embed_status(self, key: str, model: str = EMBEDDING_MODEL) -> StreamingResponse:
return self.proxy.transport.send(
"/embeddings",
headers=self.proxy.transport.bearer(key),
json=EmbedBody(model=model, input=f"route group probe {unique_marker()}"),
)
def create_model_status(self, key: str, model_name: str) -> StreamingResponse:
return self.proxy.transport.send(
"/model/new",
headers=self.proxy.transport.bearer(key),
json=ModelNewBody(
model_name=model_name,
litellm_params=LiteLLMParamsBody(model="openai/gpt-4o-mini"),
model_info=ModelInfoBody(id=model_name),
),
)
def model_catalog_status(self, key: str) -> Result[NoBody]:
return self.proxy.transport.get(
"/v1/models",
headers=self.proxy.transport.bearer(key),
params=NoBody(),
response_type=NoBody,
)
@pytest.fixture
def keys(proxy: ProxyClient) -> KeyLifecycleClient:
return KeyLifecycleClient(proxy=proxy)
def _poll[T](attempt: Callable[[], T | None], *, deadline_seconds: float, interval: float, failure: str) -> T:
deadline = time.monotonic() + deadline_seconds
while time.monotonic() < deadline:
found = attempt()
if found is not None:
return found
time.sleep(interval)
pytest.fail(failure)
def _mint(keys: KeyLifecycleClient, resources: ResourceManager, body: KeyGenerateBody) -> str:
key = keys.proxy.generate_key(body)
resources.defer(lambda: keys.proxy.delete_key(key))
return key
def _assert_authenticates(keys: KeyLifecycleClient, key: str, context: str) -> None:
outcome = keys.chat_status(key)
assert outcome.status_code != 401, f"{context}: the key was refused at auth ({outcome.body[:300]})"
def _settled_spend(keys: KeyLifecycleClient, key: str) -> float:
"""The key's recorded spend once it has stopped moving.
Spend reaches the DB on a buffered batch write, so a single sample can be
taken mid-flight and a value read now can be superseded a moment later. The
key this runs against has made exactly one successful call (every later one
is refused over budget and costs nothing), so once a non-zero value appears
there is no second increment left to land. That is asserted rather than
assumed: the value is re-read across a settle window wider than the batch
write, and a change means either a flush was still outstanding or the one
call got counted twice.
"""
first = _poll(
lambda: (lambda spend: spend if spend is not None and spend > 0 else None)(keys.key_info(key).spend),
deadline_seconds=keys.proxy.poll_timeout,
interval=5.0,
failure="/key/info never recorded any spend for the key, so the reset would have nothing to clear",
)
deadline = time.monotonic() + SPEND_SETTLE_SECONDS
while time.monotonic() < deadline:
time.sleep(SPEND_SAMPLE_INTERVAL)
again = keys.key_info(key).spend
assert again == first, (
f"the key's recorded spend moved from {first} to {again} while settling, after a single "
"successful call; a second increment means a write was still outstanding or the call was "
"billed twice"
)
return first
def _await_rejected(keys: KeyLifecycleClient, key: str, *, deadline_seconds: float, failure: str) -> None:
_poll(
lambda: True if keys.chat_status(key).status_code == 401 else None,
deadline_seconds=deadline_seconds,
interval=5.0,
failure=failure,
)
class TestKeyRegenerationGracePeriod:
@pytest.mark.covers("other.key_mgmt.regenerate.grace_period_honored")
def test_old_key_serves_through_the_grace_period_then_is_revoked(
self, keys: KeyLifecycleClient, resources: ResourceManager
) -> None:
graced = _mint(keys, resources, KeyGenerateBody(models=[CHEAP_OPENAI_MODEL]))
control = _mint(keys, resources, KeyGenerateBody(models=[CHEAP_OPENAI_MODEL]))
_assert_authenticates(keys, graced, "before rotation the key to be graced")
_assert_authenticates(keys, control, "before rotation the control key")
rotated_graced = keys.regenerate(graced, grace_period=GRACE_PERIOD)
resources.defer(lambda: keys.proxy.delete_key(rotated_graced))
rotated_control = keys.regenerate(control)
resources.defer(lambda: keys.proxy.delete_key(rotated_control))
assert rotated_graced != graced, "regenerate handed back the same secret, so nothing rotated"
_assert_authenticates(keys, graced, f"inside the {GRACE_PERIOD} grace period the rotated-out key")
_assert_authenticates(keys, rotated_graced, "the replacement key")
_await_rejected(
keys,
control,
deadline_seconds=keys.proxy.poll_timeout,
failure=(
"a key rotated with no grace period was still accepted at auth; the graced key's "
"survival cannot be attributed to the grace period"
),
)
_await_rejected(
keys,
graced,
deadline_seconds=REVOCATION_DEADLINE_SECONDS,
failure=(
f"the rotated-out key was still accepted at auth well past its {GRACE_PERIOD} grace "
"period, so the window never closes"
),
)
_assert_authenticates(keys, rotated_graced, "after the grace period closed the replacement key")
class TestKeySpendReset:
@pytest.mark.covers("other.key_mgmt.spend_reset.resets_to_value")
def test_reset_spend_clears_recorded_spend_and_unblocks_the_key(
self, keys: KeyLifecycleClient, resources: ResourceManager
) -> None:
key = _mint(keys, resources, KeyGenerateBody(models=[CHEAP_OPENAI_MODEL], max_budget=TINY_BUDGET))
first = keys.chat_status(key)
assert first.status_code == 200, (
f"the key must serve its first call before its budget is spent, got {first.status_code}: "
f"{first.body[:300]}"
)
blocked = _poll(
lambda: (lambda outcome: outcome if BUDGET_BLOCK_MARKER in outcome.body else None)(keys.chat_status(key)),
deadline_seconds=keys.proxy.poll_timeout,
interval=3.0,
failure=f"the key never got blocked over its {TINY_BUDGET} budget, so there is no block to reset away",
)
assert blocked.status_code == 429, (
f"an over-budget call must be refused 429, got {blocked.status_code}: {blocked.body[:300]}"
)
spent = _settled_spend(keys, key)
assert spent > TINY_BUDGET, (
f"/key/info recorded {spent}, at or under the {TINY_BUDGET} cap the key was refused for; the "
"reset would then be clearing something other than the over-budget spend"
)
reset = keys.reset_spend(key, reset_to=0.0)
assert reset.previous_spend >= spent, (
f"reset_spend reports previous_spend {reset.previous_spend}, less than the {spent} the key had "
"settled on. Spend accounting only ever grows as buffered writes land, so a late flush can raise "
"this value; nothing may lower it except the reset itself"
)
assert reset.spend == 0.0, f"reset_spend to 0.0 reports the key still holding {reset.spend}"
cleared = _poll(
lambda: (lambda spend: spend if spend == 0.0 else None)(keys.key_info(key).spend),
deadline_seconds=keys.proxy.poll_timeout,
interval=5.0,
failure=(
f"/key/info never reported the key's spend back at 0.0 after the reset; it had settled on "
f"{spent} beforehand"
),
)
assert cleared == 0.0
budget = keys.key_info(key).max_budget
assert budget == TINY_BUDGET, (
f"the reset must not disturb the key's budget; /key/info reports max_budget {budget}"
)
_poll(
lambda: (lambda outcome: True if BUDGET_BLOCK_MARKER not in outcome.body else None)(
keys.chat_status(key)
),
deadline_seconds=keys.proxy.poll_timeout,
interval=5.0,
failure="the key was still refused over budget after its spend was reset to 0.0",
)
class TestVirtualKeyRouteGroupGrant:
@pytest.mark.covers("other.auth.virtual_key.route_group_allowed")
def test_llm_route_group_grants_the_llm_endpoints_and_nothing_else(
self, keys: KeyLifecycleClient, resources: ResourceManager
) -> None:
key = _mint(keys, resources, KeyGenerateBody(models=[], allowed_routes=["llm_api_routes"]))
granted = keys.key_info(key).allowed_routes
assert granted == ["llm_api_routes"], (
f"/key/info reports allowed_routes {granted}, configured ['llm_api_routes']"
)
chat = keys.chat_status(key)
assert chat.status_code == 200, (
f"a key granted the llm_api_routes group must reach /chat/completions, got {chat.status_code}: "
f"{chat.body[:300]}"
)
embeddings = keys.embed_status(key)
assert embeddings.status_code == 200, (
f"the same grant must cover /embeddings, got {embeddings.status_code}: {embeddings.body[:300]}"
)
catalog = keys.model_catalog_status(key)
assert is_ok(catalog), f"the grant must cover the /v1/models catalog LLM clients read, got {catalog}"
denied = keys.create_model_status(key, f"e2e-route-group-{unique_marker()}")
assert denied.status_code == 403, (
f"the grant is a whitelist: a management route must stay refused 403, got {denied.status_code}: "
f"{denied.body[:300]}"
)
assert ROUTE_NOT_ALLOWED_MARKER in denied.body, (
f"the 403 must be a route-permission denial, got: {denied.body[:300]}"
)