mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
Merge acb54a03e2 into 982d3a5476
This commit is contained in:
commit
217f67f3b7
2 changed files with 767 additions and 0 deletions
387
tests/e2e/other/test_config_lifecycle_e2e.py
Normal file
387
tests/e2e/other/test_config_lifecycle_e2e.py
Normal file
|
|
@ -0,0 +1,387 @@
|
|||
"""Live e2e: the proxy's configuration lifecycle - what the running process took
|
||||
from its config file at boot, and what it accepts at runtime without a restart.
|
||||
|
||||
Boot is only observable through the state the process now exposes, so each test
|
||||
keys off something that exists *only* because the startup config loader ran:
|
||||
|
||||
- the config file's `model_list` is live in the routing catalog. /model/info marks
|
||||
a deployment the process read from its file with `db_model: false` (a DB-stored
|
||||
deployment is `true`), so a `db_model: false` entry that serves a real
|
||||
completion is the file's model_list loaded, credentials and all. The file's
|
||||
`general_settings` are proven live the same way: /model/new only persists a
|
||||
deployment when `store_model_in_db` came off that file, and the deployment comes
|
||||
back from the catalog flagged `db_model: true`
|
||||
- the `os.environ/` references in that file were resolved into real values. A
|
||||
credential reference is a poor witness: the router resolves `os.environ/` refs
|
||||
again at call time, so a completion would succeed even if boot-time resolution
|
||||
had never happened. The cache block is not re-resolved anywhere - the response
|
||||
cache is built once at startup from `cache_params.host` / `.port` - so a
|
||||
redis-typed cache that actually serves a repeated request can only exist if
|
||||
those two references resolved at boot; an unresolved literal would leave the
|
||||
cache dialling the hostname "os.environ/REDIS_HOST" and nothing would ever hit
|
||||
- /config/update reaches the running process, not just the DB. The test adds a
|
||||
uniquely named model-group alias, reads it back off the live router through
|
||||
/get/config/callbacks (process state, not a DB row), drives a completion through
|
||||
the alias, then removes it and watches the alias stop resolving. The alias is
|
||||
namespaced per run, and every write goes through `_write_alias`, which compares
|
||||
the aliases the test does not own against a baseline taken at the start, names
|
||||
and targets both, before writing and again after. Anything else appearing,
|
||||
vanishing or being repointed stops the test with a message naming the
|
||||
difference, and nothing is written, so a concurrent writer's change is never
|
||||
reverted with a stale view. A second alias of the test's own stands across the
|
||||
whole exchange and is asserted to survive. /config/update rewrites the alias map
|
||||
wholesale and offers neither a per-key write nor a version to compare against,
|
||||
so this is the honest ceiling: the exposure becomes a loud, diagnosable failure
|
||||
rather than silent data loss
|
||||
|
||||
These tests assume the proxy under test was booted from a config file that wires
|
||||
the example models and a redis cache through `os.environ/` references, which is
|
||||
the setup tests/e2e/CONTRIBUTING.md prescribes.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import time
|
||||
from collections.abc import Callable
|
||||
from dataclasses import dataclass
|
||||
|
||||
import pytest
|
||||
from pydantic import BaseModel, ConfigDict
|
||||
|
||||
from e2e_config import CHEAP_OPENAI_MODEL, unique_marker
|
||||
from e2e_http import NoBody, StreamingResponse, unwrap
|
||||
from lifecycle import ResourceManager
|
||||
from models import ChatBody, ChatMessage, LiteLLMParamsBody
|
||||
from proxy_client import ProxyClient
|
||||
|
||||
pytestmark = pytest.mark.e2e
|
||||
|
||||
CACHE_HIT_HEADER = "x-litellm-cache-key"
|
||||
CACHE_HIT_DEADLINE_SECONDS = 30.0
|
||||
|
||||
|
||||
class CatalogModelInfo(BaseModel):
|
||||
"""The /model/info `model_info` block, narrowed to the flag that says where the
|
||||
deployment came from: `db_model` is false for a deployment read off the config
|
||||
file at startup and true for one stored in the DB."""
|
||||
|
||||
db_model: bool = False
|
||||
|
||||
|
||||
class CatalogEntry(BaseModel):
|
||||
model_config = ConfigDict(protected_namespaces=())
|
||||
model_name: str
|
||||
model_info: CatalogModelInfo = CatalogModelInfo()
|
||||
|
||||
|
||||
class CatalogResponse(BaseModel):
|
||||
data: list[CatalogEntry] = []
|
||||
|
||||
|
||||
class CacheReadiness(BaseModel):
|
||||
"""GET /health/readiness/details, narrowed to the cache the process built at
|
||||
startup (`litellm.cache.type`)."""
|
||||
|
||||
status: str
|
||||
cache: str | None = None
|
||||
|
||||
|
||||
class LiveRouterSettings(BaseModel):
|
||||
model_config = ConfigDict(protected_namespaces=())
|
||||
model_group_alias: dict[str, str] = {}
|
||||
|
||||
|
||||
class ConfigCallbacksResponse(BaseModel):
|
||||
"""GET /get/config/callbacks. `router_settings` is read straight off the live
|
||||
Router object, so it reports what the running process is using rather than
|
||||
what any config row holds."""
|
||||
|
||||
router_settings: LiveRouterSettings
|
||||
|
||||
|
||||
class RouterSettingsUpdate(BaseModel):
|
||||
model_config = ConfigDict(protected_namespaces=())
|
||||
model_group_alias: dict[str, str]
|
||||
|
||||
|
||||
class ConfigUpdateBody(BaseModel):
|
||||
router_settings: RouterSettingsUpdate
|
||||
|
||||
|
||||
class ConfigUpdateResult(BaseModel):
|
||||
message: str
|
||||
|
||||
|
||||
class CompletionId(BaseModel):
|
||||
id: str
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class ConfigClient:
|
||||
"""The catalog, health and config routes these tests read the proxy's own
|
||||
configuration state back through."""
|
||||
|
||||
proxy: ProxyClient
|
||||
|
||||
def catalog(self) -> list[CatalogEntry]:
|
||||
return unwrap(
|
||||
self.proxy.transport.get(
|
||||
"/model/info",
|
||||
headers=self.proxy.transport.master,
|
||||
params=NoBody(),
|
||||
response_type=CatalogResponse,
|
||||
)
|
||||
).data
|
||||
|
||||
def cache_readiness(self) -> CacheReadiness:
|
||||
return unwrap(
|
||||
self.proxy.transport.get(
|
||||
"/health/readiness/details",
|
||||
headers=self.proxy.transport.master,
|
||||
params=NoBody(),
|
||||
response_type=CacheReadiness,
|
||||
)
|
||||
)
|
||||
|
||||
def live_router_settings(self) -> LiveRouterSettings:
|
||||
return unwrap(
|
||||
self.proxy.transport.get(
|
||||
"/get/config/callbacks",
|
||||
headers=self.proxy.transport.master,
|
||||
params=NoBody(),
|
||||
response_type=ConfigCallbacksResponse,
|
||||
)
|
||||
).router_settings
|
||||
|
||||
def set_model_group_alias(self, aliases: dict[str, str]) -> None:
|
||||
_ = unwrap(
|
||||
self.proxy.transport.post(
|
||||
"/config/update",
|
||||
headers=self.proxy.transport.master,
|
||||
json=ConfigUpdateBody(router_settings=RouterSettingsUpdate(model_group_alias=aliases)),
|
||||
response_type=ConfigUpdateResult,
|
||||
)
|
||||
)
|
||||
|
||||
def chat_status(self, key: str, model: str, content: str) -> StreamingResponse:
|
||||
return self.proxy.transport.send(
|
||||
"/chat/completions",
|
||||
headers=self.proxy.transport.bearer(key),
|
||||
json=ChatBody(
|
||||
model=model,
|
||||
messages=[ChatMessage(role="user", content=content)],
|
||||
max_tokens=8,
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def config(proxy: ProxyClient) -> ConfigClient:
|
||||
return ConfigClient(proxy=proxy)
|
||||
|
||||
|
||||
def _poll[T](attempt: Callable[[], T | None], *, deadline_seconds: float, interval: float, failure: str) -> T:
|
||||
deadline = time.monotonic() + deadline_seconds
|
||||
while time.monotonic() < deadline:
|
||||
found = attempt()
|
||||
if found is not None:
|
||||
return found
|
||||
time.sleep(interval)
|
||||
pytest.fail(failure)
|
||||
|
||||
|
||||
def _entry(catalog: list[CatalogEntry], model_name: str) -> CatalogEntry | None:
|
||||
return next((entry for entry in catalog if entry.model_name == model_name), None)
|
||||
|
||||
|
||||
def _foreign_aliases(live: dict[str, str], owned: tuple[str, ...]) -> dict[str, str]:
|
||||
"""Every alias the test does not own, names and targets both."""
|
||||
return {name: target for name, target in live.items() if name not in owned}
|
||||
|
||||
|
||||
def _write_alias(
|
||||
config: ConfigClient,
|
||||
*,
|
||||
alias: str,
|
||||
target: str | None,
|
||||
owned: tuple[str, ...],
|
||||
baseline: dict[str, str],
|
||||
) -> None:
|
||||
"""Set or remove one owned alias, refusing to write at all if anything else in
|
||||
the map has moved.
|
||||
|
||||
/config/update is the only route that writes model_group_alias, it takes the
|
||||
whole map, and it offers neither a per-key write nor a version to compare
|
||||
against, so every writer is a read-modify-write that can revert a concurrent
|
||||
change. Rather than paper over that, the aliases the test does not own are
|
||||
compared against `baseline` as full name/target pairs before the write and
|
||||
again after it: a name that appeared or vanished, or an existing alias
|
||||
repointed at a different model group, stops the test with a message naming the
|
||||
difference instead of being quietly overwritten with a stale value. Nothing is
|
||||
written once the map has moved, so the other writer's state stands.
|
||||
"""
|
||||
live = config.live_router_settings().model_group_alias
|
||||
_assert_unmoved(_foreign_aliases(live, owned), baseline, when=f"before writing {alias!r}")
|
||||
|
||||
keep = _foreign_aliases(live, owned) | {name: live[name] for name in owned if name in live and name != alias}
|
||||
config.set_model_group_alias(keep if target is None else {**keep, alias: target})
|
||||
|
||||
settled = config.live_router_settings().model_group_alias
|
||||
_assert_unmoved(_foreign_aliases(settled, owned), baseline, when=f"after writing {alias!r}")
|
||||
|
||||
|
||||
def _release_alias(config: ConfigClient, alias: str) -> None:
|
||||
"""Teardown safety net: drop one alias from the map as it stands right now.
|
||||
|
||||
Deliberately assertion-free and built from a fresh read, so it removes the
|
||||
test's own entry without restoring anything else to an older value; teardown
|
||||
swallows failures, so a stop-the-test check here would be invisible anyway.
|
||||
"""
|
||||
live = config.live_router_settings().model_group_alias
|
||||
config.set_model_group_alias({name: target for name, target in live.items() if name != alias})
|
||||
|
||||
|
||||
def _assert_unmoved(current: dict[str, str], baseline: dict[str, str], *, when: str) -> None:
|
||||
added = sorted(name for name in current if name not in baseline)
|
||||
removed = sorted(name for name in baseline if name not in current)
|
||||
repointed = sorted(name for name, target in baseline.items() if name in current and current[name] != target)
|
||||
assert current == baseline, (
|
||||
f"the model_group_alias map moved under the test {when}: added {added}, removed {removed}, "
|
||||
f"repointed {repointed}. /config/update rewrites the whole map, so continuing would write a stale "
|
||||
"view back over another writer's change; nothing was written"
|
||||
)
|
||||
|
||||
|
||||
class TestStartupConfigLoad:
|
||||
@pytest.mark.covers("other.lifecycle.startup.config_loads")
|
||||
def test_config_file_model_list_and_general_settings_are_live(
|
||||
self, config: ConfigClient, resources: ResourceManager, scoped_key: str
|
||||
) -> None:
|
||||
from_file = _entry(config.catalog(), CHEAP_OPENAI_MODEL)
|
||||
assert from_file is not None, (
|
||||
f"the proxy serves no {CHEAP_OPENAI_MODEL!r} deployment, so its config file's model_list never loaded"
|
||||
)
|
||||
assert not from_file.model_info.db_model, (
|
||||
f"{CHEAP_OPENAI_MODEL!r} is flagged db_model=true, so it came from the DB rather than the config "
|
||||
"file the process booted with; the file's model_list is not what is being served"
|
||||
)
|
||||
|
||||
served = config.chat_status(scoped_key, CHEAP_OPENAI_MODEL, f"reply with one word {unique_marker()}")
|
||||
assert served.status_code == 200, (
|
||||
f"the deployment loaded from the config file must serve a real completion, got "
|
||||
f"{served.status_code}: {served.body[:300]}"
|
||||
)
|
||||
|
||||
stored_name = f"e2e-boot-cfg-{unique_marker()}"
|
||||
model_id = config.proxy.create_model(
|
||||
stored_name,
|
||||
LiteLLMParamsBody(model="openai/gpt-4o-mini", api_key="e2e-dummy-key"),
|
||||
)
|
||||
resources.defer(lambda: config.proxy.delete_model(model_id))
|
||||
|
||||
stored = _entry(config.catalog(), stored_name)
|
||||
assert stored is not None, f"{stored_name!r} is absent from /model/info right after /model/new"
|
||||
assert stored.model_info.db_model, (
|
||||
f"{stored_name!r} came back flagged db_model=false; the config file's general_settings "
|
||||
"(store_model_in_db) are not in effect on the running process"
|
||||
)
|
||||
|
||||
@pytest.mark.covers("other.lifecycle.startup.env_vars_resolved")
|
||||
def test_env_referenced_cache_block_resolved_at_startup(self, config: ConfigClient, scoped_key: str) -> None:
|
||||
readiness = config.cache_readiness()
|
||||
assert readiness.cache == "redis", (
|
||||
f"the process reports cache {readiness.cache!r}; the config file's redis cache_params block, "
|
||||
"whose host and port are os.environ/ references, is not the cache the proxy built at startup"
|
||||
)
|
||||
|
||||
prompt = f"reply with one word {unique_marker()}"
|
||||
first = config.chat_status(scoped_key, CHEAP_OPENAI_MODEL, prompt)
|
||||
assert first.status_code == 200, (
|
||||
f"the call being cached must succeed first, got {first.status_code}: {first.body[:300]}"
|
||||
)
|
||||
assert CACHE_HIT_HEADER not in first.headers, (
|
||||
f"a first-of-its-kind prompt came back as a cache hit ({first.headers.get(CACHE_HIT_HEADER)}), "
|
||||
"so the repeat below would prove nothing"
|
||||
)
|
||||
|
||||
repeated = _poll(
|
||||
lambda: (lambda outcome: outcome if CACHE_HIT_HEADER in outcome.headers else None)(
|
||||
config.chat_status(scoped_key, CHEAP_OPENAI_MODEL, prompt)
|
||||
),
|
||||
deadline_seconds=CACHE_HIT_DEADLINE_SECONDS,
|
||||
interval=5.0,
|
||||
failure=(
|
||||
"an identical repeat call was never served from the redis response cache, so the "
|
||||
"os.environ/ host and port that cache was configured with never resolved into a "
|
||||
"reachable redis at startup"
|
||||
),
|
||||
)
|
||||
assert CompletionId.model_validate_json(repeated.body).id == CompletionId.model_validate_json(first.body).id, (
|
||||
"the repeat call carried a cache-key header but returned a different completion, so it was "
|
||||
"not the stored response coming back"
|
||||
)
|
||||
|
||||
|
||||
class TestRuntimeConfigUpdate:
|
||||
@pytest.mark.covers("other.config.runtime_update.applies_at_runtime")
|
||||
def test_config_update_reaches_the_running_router_and_can_be_taken_back(
|
||||
self, config: ConfigClient, resources: ResourceManager, scoped_key: str
|
||||
) -> None:
|
||||
alias = f"e2e-cfg-alias-{unique_marker()}"
|
||||
bystander = f"e2e-cfg-bystander-{unique_marker()}"
|
||||
owned = (alias, bystander)
|
||||
baseline = _foreign_aliases(config.live_router_settings().model_group_alias, owned)
|
||||
assert alias not in baseline and bystander not in baseline, (
|
||||
f"the run's aliases {owned} are somehow already configured"
|
||||
)
|
||||
|
||||
resources.defer(lambda: _release_alias(config, bystander))
|
||||
_write_alias(config, alias=bystander, target=CHEAP_OPENAI_MODEL, owned=owned, baseline=baseline)
|
||||
|
||||
unknown = config.chat_status(scoped_key, alias, "should not route yet")
|
||||
assert unknown.status_code == 400, (
|
||||
f"an unconfigured model group must be rejected 400 before the update, got {unknown.status_code}: "
|
||||
f"{unknown.body[:300]}"
|
||||
)
|
||||
|
||||
resources.defer(lambda: _release_alias(config, alias))
|
||||
_write_alias(config, alias=alias, target=CHEAP_OPENAI_MODEL, owned=owned, baseline=baseline)
|
||||
|
||||
applied = _poll(
|
||||
lambda: (lambda live: live if live.model_group_alias.get(alias) == CHEAP_OPENAI_MODEL else None)(
|
||||
config.live_router_settings()
|
||||
),
|
||||
deadline_seconds=config.proxy.poll_timeout,
|
||||
interval=5.0,
|
||||
failure=(
|
||||
f"the live router never picked up model_group_alias {alias!r} after /config/update, so the "
|
||||
"update only reached the DB and would need a restart to take effect"
|
||||
),
|
||||
)
|
||||
assert applied.model_group_alias[alias] == CHEAP_OPENAI_MODEL
|
||||
|
||||
routed = config.chat_status(scoped_key, alias, f"reply with one word {unique_marker()}")
|
||||
assert routed.status_code == 200, (
|
||||
f"the alias added at runtime must route to {CHEAP_OPENAI_MODEL!r}, got {routed.status_code}: "
|
||||
f"{routed.body[:300]}"
|
||||
)
|
||||
|
||||
_write_alias(config, alias=alias, target=None, owned=owned, baseline=baseline)
|
||||
|
||||
_poll(
|
||||
lambda: (
|
||||
True
|
||||
if config.chat_status(scoped_key, alias, "should not route any more").status_code == 400
|
||||
else None
|
||||
),
|
||||
deadline_seconds=config.proxy.poll_timeout,
|
||||
interval=5.0,
|
||||
failure=f"the alias {alias!r} still routed after being removed at runtime",
|
||||
)
|
||||
settled = config.live_router_settings().model_group_alias
|
||||
assert alias not in settled, f"the live router still carries {alias!r} after the removing /config/update"
|
||||
assert settled.get(bystander) == CHEAP_OPENAI_MODEL, (
|
||||
f"removing {alias!r} also took {bystander!r} with it; /config/update replaces the alias map "
|
||||
"wholesale, so a caller that writes back anything other than the map as it currently stands "
|
||||
"wipes aliases it never touched"
|
||||
)
|
||||
380
tests/e2e/other/test_key_lifecycle_e2e.py
Normal file
380
tests/e2e/other/test_key_lifecycle_e2e.py
Normal file
|
|
@ -0,0 +1,380 @@
|
|||
"""Live e2e: what an already-issued virtual key can still do to itself.
|
||||
|
||||
Three post-issue behaviors of a virtual key, each driven through the admin route
|
||||
that owns it and judged on live traffic:
|
||||
|
||||
- rotation with a grace period: `/key/{key}/regenerate` with `grace_period` keeps
|
||||
the old secret authenticating until the window closes, then stops accepting it.
|
||||
A sibling key rotated in the same test *without* a grace period is the control:
|
||||
it is refused immediately, so the graced key still being served cannot be an
|
||||
auth-cache artifact
|
||||
- spend reset: `/key/{key}/reset_spend` puts the key's accumulated spend back to
|
||||
the requested value, and a key that its own max_budget had blocked serves again.
|
||||
Spend arrives on a buffered batch write, so the test settles on a value that has
|
||||
stopped moving before it resets, and afterwards asserts only relationships a
|
||||
late flush cannot violate: the reset reports a previous value no smaller than
|
||||
the settled one (spend never shrinks on its own), reports zero, and the DB is
|
||||
polled until it agrees. Exact equality against a single sample would go red on
|
||||
correct behavior the moment a flush landed between two reads
|
||||
- the allow side of `allowed_routes`: a key granted the `llm_api_routes` group
|
||||
reaches the LLM endpoints, while a management route stays refused, so the grant
|
||||
is a real whitelist rather than an absent check (the deny side lives in
|
||||
tests/e2e/access_control/)
|
||||
|
||||
Every key is minted here and deleted on teardown; auth is judged by status code
|
||||
(401 vs not-401), never by requiring a completion, so a provider hiccup cannot be
|
||||
mistaken for a revocation.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import time
|
||||
from collections.abc import Callable
|
||||
from dataclasses import dataclass
|
||||
|
||||
import pytest
|
||||
from pydantic import BaseModel
|
||||
|
||||
from e2e_config import CHEAP_OPENAI_MODEL, unique_marker
|
||||
from e2e_http import NoBody, Result, StreamingResponse, is_ok, unwrap
|
||||
from lifecycle import ResourceManager
|
||||
from models import (
|
||||
ChatBody,
|
||||
ChatMessage,
|
||||
EmbedBody,
|
||||
KeyGenerateBody,
|
||||
KeyInfoParams,
|
||||
LiteLLMParamsBody,
|
||||
ModelInfoBody,
|
||||
ModelNewBody,
|
||||
)
|
||||
from proxy_client import ProxyClient
|
||||
|
||||
pytestmark = pytest.mark.e2e
|
||||
|
||||
EMBEDDING_MODEL = "openai-text-embedding-3-small"
|
||||
ROUTE_NOT_ALLOWED_MARKER = "not allowed to call this route"
|
||||
BUDGET_BLOCK_MARKER = "budget_exceeded"
|
||||
|
||||
GRACE_PERIOD = "60s"
|
||||
REVOCATION_DEADLINE_SECONDS = 240.0
|
||||
TINY_BUDGET = 3e-6
|
||||
# Spend lands on a buffered batch write (proxy_batch_write_at), so a freshly read
|
||||
# value can still be superseded. Re-read across a window wider than that flush
|
||||
# before treating a value as final.
|
||||
SPEND_SETTLE_SECONDS = 25.0
|
||||
SPEND_SAMPLE_INTERVAL = 5.0
|
||||
|
||||
|
||||
class RegeneratedKey(BaseModel):
|
||||
key: str
|
||||
|
||||
|
||||
class KeyRegenerateBody(BaseModel):
|
||||
"""POST /key/{key}/regenerate. `grace_period` is a duration string ("60s",
|
||||
"24h"); omitted means the old secret is revoked the moment the new one is
|
||||
issued."""
|
||||
|
||||
grace_period: str | None = None
|
||||
|
||||
|
||||
class ResetSpendBody(BaseModel):
|
||||
reset_to: float
|
||||
|
||||
|
||||
class ResetSpendResult(BaseModel):
|
||||
"""POST /key/{key}/reset_spend. Reports the spend the key carried before the
|
||||
reset alongside the value it now holds."""
|
||||
|
||||
spend: float
|
||||
previous_spend: float
|
||||
|
||||
|
||||
class ScopedKeyInfo(BaseModel):
|
||||
allowed_routes: list[str] | None = None
|
||||
models: list[str] = []
|
||||
spend: float | None = None
|
||||
max_budget: float | None = None
|
||||
|
||||
|
||||
class ScopedKeyInfoResponse(BaseModel):
|
||||
info: ScopedKeyInfo
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class KeyLifecycleClient:
|
||||
"""The admin routes that act on an existing key, plus the LLM and management
|
||||
calls used to judge what that key may still do."""
|
||||
|
||||
proxy: ProxyClient
|
||||
|
||||
def key_info(self, key: str) -> ScopedKeyInfo:
|
||||
return unwrap(
|
||||
self.proxy.transport.get(
|
||||
"/key/info",
|
||||
headers=self.proxy.transport.master,
|
||||
params=KeyInfoParams(key=key),
|
||||
response_type=ScopedKeyInfoResponse,
|
||||
)
|
||||
).info
|
||||
|
||||
def regenerate(self, key: str, *, grace_period: str | None = None) -> str:
|
||||
return unwrap(
|
||||
self.proxy.transport.post(
|
||||
f"/key/{key}/regenerate",
|
||||
headers=self.proxy.transport.master,
|
||||
json=KeyRegenerateBody(grace_period=grace_period),
|
||||
response_type=RegeneratedKey,
|
||||
)
|
||||
).key
|
||||
|
||||
def reset_spend(self, key: str, *, reset_to: float) -> ResetSpendResult:
|
||||
return unwrap(
|
||||
self.proxy.transport.post(
|
||||
f"/key/{key}/reset_spend",
|
||||
headers=self.proxy.transport.master,
|
||||
json=ResetSpendBody(reset_to=reset_to),
|
||||
response_type=ResetSpendResult,
|
||||
)
|
||||
)
|
||||
|
||||
def chat_status(self, key: str, model: str = CHEAP_OPENAI_MODEL) -> StreamingResponse:
|
||||
return self.proxy.transport.send(
|
||||
"/chat/completions",
|
||||
headers=self.proxy.transport.bearer(key),
|
||||
json=ChatBody(
|
||||
model=model,
|
||||
messages=[ChatMessage(role="user", content=f"reply with one word {unique_marker()}")],
|
||||
max_tokens=8,
|
||||
),
|
||||
)
|
||||
|
||||
def embed_status(self, key: str, model: str = EMBEDDING_MODEL) -> StreamingResponse:
|
||||
return self.proxy.transport.send(
|
||||
"/embeddings",
|
||||
headers=self.proxy.transport.bearer(key),
|
||||
json=EmbedBody(model=model, input=f"route group probe {unique_marker()}"),
|
||||
)
|
||||
|
||||
def create_model_status(self, key: str, model_name: str) -> StreamingResponse:
|
||||
return self.proxy.transport.send(
|
||||
"/model/new",
|
||||
headers=self.proxy.transport.bearer(key),
|
||||
json=ModelNewBody(
|
||||
model_name=model_name,
|
||||
litellm_params=LiteLLMParamsBody(model="openai/gpt-4o-mini"),
|
||||
model_info=ModelInfoBody(id=model_name),
|
||||
),
|
||||
)
|
||||
|
||||
def model_catalog_status(self, key: str) -> Result[NoBody]:
|
||||
return self.proxy.transport.get(
|
||||
"/v1/models",
|
||||
headers=self.proxy.transport.bearer(key),
|
||||
params=NoBody(),
|
||||
response_type=NoBody,
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def keys(proxy: ProxyClient) -> KeyLifecycleClient:
|
||||
return KeyLifecycleClient(proxy=proxy)
|
||||
|
||||
|
||||
def _poll[T](attempt: Callable[[], T | None], *, deadline_seconds: float, interval: float, failure: str) -> T:
|
||||
deadline = time.monotonic() + deadline_seconds
|
||||
while time.monotonic() < deadline:
|
||||
found = attempt()
|
||||
if found is not None:
|
||||
return found
|
||||
time.sleep(interval)
|
||||
pytest.fail(failure)
|
||||
|
||||
|
||||
def _mint(keys: KeyLifecycleClient, resources: ResourceManager, body: KeyGenerateBody) -> str:
|
||||
key = keys.proxy.generate_key(body)
|
||||
resources.defer(lambda: keys.proxy.delete_key(key))
|
||||
return key
|
||||
|
||||
|
||||
def _assert_authenticates(keys: KeyLifecycleClient, key: str, context: str) -> None:
|
||||
outcome = keys.chat_status(key)
|
||||
assert outcome.status_code != 401, f"{context}: the key was refused at auth ({outcome.body[:300]})"
|
||||
|
||||
|
||||
def _settled_spend(keys: KeyLifecycleClient, key: str) -> float:
|
||||
"""The key's recorded spend once it has stopped moving.
|
||||
|
||||
Spend reaches the DB on a buffered batch write, so a single sample can be
|
||||
taken mid-flight and a value read now can be superseded a moment later. The
|
||||
key this runs against has made exactly one successful call (every later one
|
||||
is refused over budget and costs nothing), so once a non-zero value appears
|
||||
there is no second increment left to land. That is asserted rather than
|
||||
assumed: the value is re-read across a settle window wider than the batch
|
||||
write, and a change means either a flush was still outstanding or the one
|
||||
call got counted twice.
|
||||
"""
|
||||
first = _poll(
|
||||
lambda: (lambda spend: spend if spend is not None and spend > 0 else None)(keys.key_info(key).spend),
|
||||
deadline_seconds=keys.proxy.poll_timeout,
|
||||
interval=5.0,
|
||||
failure="/key/info never recorded any spend for the key, so the reset would have nothing to clear",
|
||||
)
|
||||
deadline = time.monotonic() + SPEND_SETTLE_SECONDS
|
||||
while time.monotonic() < deadline:
|
||||
time.sleep(SPEND_SAMPLE_INTERVAL)
|
||||
again = keys.key_info(key).spend
|
||||
assert again == first, (
|
||||
f"the key's recorded spend moved from {first} to {again} while settling, after a single "
|
||||
"successful call; a second increment means a write was still outstanding or the call was "
|
||||
"billed twice"
|
||||
)
|
||||
return first
|
||||
|
||||
|
||||
def _await_rejected(keys: KeyLifecycleClient, key: str, *, deadline_seconds: float, failure: str) -> None:
|
||||
_poll(
|
||||
lambda: True if keys.chat_status(key).status_code == 401 else None,
|
||||
deadline_seconds=deadline_seconds,
|
||||
interval=5.0,
|
||||
failure=failure,
|
||||
)
|
||||
|
||||
|
||||
class TestKeyRegenerationGracePeriod:
|
||||
@pytest.mark.covers("other.key_mgmt.regenerate.grace_period_honored")
|
||||
def test_old_key_serves_through_the_grace_period_then_is_revoked(
|
||||
self, keys: KeyLifecycleClient, resources: ResourceManager
|
||||
) -> None:
|
||||
graced = _mint(keys, resources, KeyGenerateBody(models=[CHEAP_OPENAI_MODEL]))
|
||||
control = _mint(keys, resources, KeyGenerateBody(models=[CHEAP_OPENAI_MODEL]))
|
||||
_assert_authenticates(keys, graced, "before rotation the key to be graced")
|
||||
_assert_authenticates(keys, control, "before rotation the control key")
|
||||
|
||||
rotated_graced = keys.regenerate(graced, grace_period=GRACE_PERIOD)
|
||||
resources.defer(lambda: keys.proxy.delete_key(rotated_graced))
|
||||
rotated_control = keys.regenerate(control)
|
||||
resources.defer(lambda: keys.proxy.delete_key(rotated_control))
|
||||
assert rotated_graced != graced, "regenerate handed back the same secret, so nothing rotated"
|
||||
|
||||
_assert_authenticates(keys, graced, f"inside the {GRACE_PERIOD} grace period the rotated-out key")
|
||||
_assert_authenticates(keys, rotated_graced, "the replacement key")
|
||||
|
||||
_await_rejected(
|
||||
keys,
|
||||
control,
|
||||
deadline_seconds=keys.proxy.poll_timeout,
|
||||
failure=(
|
||||
"a key rotated with no grace period was still accepted at auth; the graced key's "
|
||||
"survival cannot be attributed to the grace period"
|
||||
),
|
||||
)
|
||||
|
||||
_await_rejected(
|
||||
keys,
|
||||
graced,
|
||||
deadline_seconds=REVOCATION_DEADLINE_SECONDS,
|
||||
failure=(
|
||||
f"the rotated-out key was still accepted at auth well past its {GRACE_PERIOD} grace "
|
||||
"period, so the window never closes"
|
||||
),
|
||||
)
|
||||
_assert_authenticates(keys, rotated_graced, "after the grace period closed the replacement key")
|
||||
|
||||
|
||||
class TestKeySpendReset:
|
||||
@pytest.mark.covers("other.key_mgmt.spend_reset.resets_to_value")
|
||||
def test_reset_spend_clears_recorded_spend_and_unblocks_the_key(
|
||||
self, keys: KeyLifecycleClient, resources: ResourceManager
|
||||
) -> None:
|
||||
key = _mint(keys, resources, KeyGenerateBody(models=[CHEAP_OPENAI_MODEL], max_budget=TINY_BUDGET))
|
||||
|
||||
first = keys.chat_status(key)
|
||||
assert first.status_code == 200, (
|
||||
f"the key must serve its first call before its budget is spent, got {first.status_code}: "
|
||||
f"{first.body[:300]}"
|
||||
)
|
||||
|
||||
blocked = _poll(
|
||||
lambda: (lambda outcome: outcome if BUDGET_BLOCK_MARKER in outcome.body else None)(keys.chat_status(key)),
|
||||
deadline_seconds=keys.proxy.poll_timeout,
|
||||
interval=3.0,
|
||||
failure=f"the key never got blocked over its {TINY_BUDGET} budget, so there is no block to reset away",
|
||||
)
|
||||
assert blocked.status_code == 429, (
|
||||
f"an over-budget call must be refused 429, got {blocked.status_code}: {blocked.body[:300]}"
|
||||
)
|
||||
|
||||
spent = _settled_spend(keys, key)
|
||||
assert spent > TINY_BUDGET, (
|
||||
f"/key/info recorded {spent}, at or under the {TINY_BUDGET} cap the key was refused for; the "
|
||||
"reset would then be clearing something other than the over-budget spend"
|
||||
)
|
||||
|
||||
reset = keys.reset_spend(key, reset_to=0.0)
|
||||
assert reset.previous_spend >= spent, (
|
||||
f"reset_spend reports previous_spend {reset.previous_spend}, less than the {spent} the key had "
|
||||
"settled on. Spend accounting only ever grows as buffered writes land, so a late flush can raise "
|
||||
"this value; nothing may lower it except the reset itself"
|
||||
)
|
||||
assert reset.spend == 0.0, f"reset_spend to 0.0 reports the key still holding {reset.spend}"
|
||||
|
||||
cleared = _poll(
|
||||
lambda: (lambda spend: spend if spend == 0.0 else None)(keys.key_info(key).spend),
|
||||
deadline_seconds=keys.proxy.poll_timeout,
|
||||
interval=5.0,
|
||||
failure=(
|
||||
f"/key/info never reported the key's spend back at 0.0 after the reset; it had settled on "
|
||||
f"{spent} beforehand"
|
||||
),
|
||||
)
|
||||
assert cleared == 0.0
|
||||
budget = keys.key_info(key).max_budget
|
||||
assert budget == TINY_BUDGET, (
|
||||
f"the reset must not disturb the key's budget; /key/info reports max_budget {budget}"
|
||||
)
|
||||
|
||||
_poll(
|
||||
lambda: (lambda outcome: True if BUDGET_BLOCK_MARKER not in outcome.body else None)(
|
||||
keys.chat_status(key)
|
||||
),
|
||||
deadline_seconds=keys.proxy.poll_timeout,
|
||||
interval=5.0,
|
||||
failure="the key was still refused over budget after its spend was reset to 0.0",
|
||||
)
|
||||
|
||||
|
||||
class TestVirtualKeyRouteGroupGrant:
|
||||
@pytest.mark.covers("other.auth.virtual_key.route_group_allowed")
|
||||
def test_llm_route_group_grants_the_llm_endpoints_and_nothing_else(
|
||||
self, keys: KeyLifecycleClient, resources: ResourceManager
|
||||
) -> None:
|
||||
key = _mint(keys, resources, KeyGenerateBody(models=[], allowed_routes=["llm_api_routes"]))
|
||||
|
||||
granted = keys.key_info(key).allowed_routes
|
||||
assert granted == ["llm_api_routes"], (
|
||||
f"/key/info reports allowed_routes {granted}, configured ['llm_api_routes']"
|
||||
)
|
||||
|
||||
chat = keys.chat_status(key)
|
||||
assert chat.status_code == 200, (
|
||||
f"a key granted the llm_api_routes group must reach /chat/completions, got {chat.status_code}: "
|
||||
f"{chat.body[:300]}"
|
||||
)
|
||||
|
||||
embeddings = keys.embed_status(key)
|
||||
assert embeddings.status_code == 200, (
|
||||
f"the same grant must cover /embeddings, got {embeddings.status_code}: {embeddings.body[:300]}"
|
||||
)
|
||||
|
||||
catalog = keys.model_catalog_status(key)
|
||||
assert is_ok(catalog), f"the grant must cover the /v1/models catalog LLM clients read, got {catalog}"
|
||||
|
||||
denied = keys.create_model_status(key, f"e2e-route-group-{unique_marker()}")
|
||||
assert denied.status_code == 403, (
|
||||
f"the grant is a whitelist: a management route must stay refused 403, got {denied.status_code}: "
|
||||
f"{denied.body[:300]}"
|
||||
)
|
||||
assert ROUTE_NOT_ALLOWED_MARKER in denied.body, (
|
||||
f"the 403 must be a route-permission denial, got: {denied.body[:300]}"
|
||||
)
|
||||
Loading…
Add table
Reference in a new issue