feat(e2e): record each e2e test's steps, starting with ProxyClient

@step on a harness method records a plain-English line for every call, in
order, as repeated JUnit step properties. Labels are templates filled from the
call's parameters, like "Generate a virtual key with models: claude-haiku-4-5
and rpm limit: 3", and secret request fields are marked Field(repr=False) so
they never print. ProxyClient and the rate-limit QuotaClient carry steps first;
the other harnesses follow one area at a time. The recorder and JUnit tests run
in the Code Quality workflow's test_e2e_metadata step.
This commit is contained in:
ryan-crabbe-berri 2026-09-28 11:57:31 -07:00
parent cbe69723b1
commit fde8aa453e
10 changed files with 1063 additions and 18 deletions

View file

@ -80,6 +80,11 @@ jobs:
- name: test_e2e_changed_gate
run: uv run --no-sync pytest -q --noconftest -p no:cacheprovider -c /dev/null tests/code_coverage_tests/test_e2e_changed_gate.py tests/code_coverage_tests/test_e2e_idp_stack.py
- name: test_e2e_metadata
env:
PYTHONPATH: tests/e2e
run: uv run --no-sync pytest -q --noconftest -p no:cacheprovider -c /dev/null tests/code_coverage_tests/test_e2e_metadata.py tests/code_coverage_tests/test_e2e_junit_report.py
- name: Check merge smoke harness
run: uv run --no-sync pytest -q --noconftest -p no:cacheprovider -c /dev/null tests/code_coverage_tests/test_merge_smoke.py

View file

@ -0,0 +1,288 @@
"""The JUnit report itself, written by a real pytest run.
No proxy. test_e2e_metadata.py pins the recorder's edge cases;
this pins what reaches the XML once pytest, its junitxml plugin,
pytest-rerunfailures and xdist are all in the loop. Each case writes a throwaway
suite into a tmp dir and runs it in a child interpreter with tests/e2e's
conftest.py loaded as a plugin, so the hooks under test are the ones the live
suite runs and the recorder is the real one, never a copy of either.
The timing that makes the recorded half work is pytest's, which is why it is
pinned here against the real thing: junitxml writes a testcase's properties from
its TEARDOWN report, and pytest builds that report from ``item.user_properties``
after the setup and call phases have both attached the steps. The suite runs
distributed, so every assertion is made in-process and again under ``-n 2``.
"""
from __future__ import annotations
import os
import shlex
import subprocess
import sys
from collections.abc import Mapping
from importlib.util import find_spec
from pathlib import Path
from types import MappingProxyType
from typing import Final
from xml.etree import ElementTree
import pytest
SUITE_DIR: Final = Path(__file__).resolve().parents[1] / "e2e"
CHILD_TIMEOUT_SECONDS: Final = 180
STORY_SUITE: Final = """
from collections.abc import Iterator
from pathlib import Path
import pytest
from e2e_metadata import step
FIRST_ATTEMPT_MADE = Path(__file__).with_name("first-attempt-made")
@step("generate virtual key")
def generate_key() -> None:
return None
@step("create team")
def create_team() -> None:
raise RuntimeError("/team/new answered 500")
@step("POST /chat/completions")
def chat(*, ok: bool) -> None:
if not ok:
raise AssertionError("status_code=502 from upstream")
@step("poll /spend/logs")
def poll_spend_logs() -> None:
return None
@step("delete virtual key")
def delete_key() -> None:
return None
@pytest.fixture
def key() -> Iterator[None]:
generate_key()
yield
delete_key()
@pytest.fixture
def team(key: None) -> None:
create_team()
def test_passes(key: None) -> None:
chat(ok=True)
poll_spend_logs()
def test_fails(key: None) -> None:
chat(ok=False)
poll_spend_logs()
def test_errors_in_setup(team: None) -> None:
poll_spend_logs()
def test_passes_on_the_rerun(key: None) -> None:
first_attempt = not FIRST_ATTEMPT_MADE.exists()
FIRST_ATTEMPT_MADE.touch()
chat(ok=not first_attempt)
poll_spend_logs()
"""
WIDE_FINALIZER_SUITE: Final = """
from collections.abc import Iterator
import pytest
from e2e_metadata import step
@step("generate virtual key")
def generate_key() -> None:
return None
@step("delete shared team")
def delete_shared_team() -> None:
return None
@pytest.fixture(scope="module")
def shared_team() -> Iterator[None]:
yield
delete_shared_team()
def test_uses_the_shared_team(shared_team: None) -> None:
generate_key()
"""
WIDE_SETUP_ERROR_SUITE: Final = """
import pytest
from e2e_metadata import step
@step("log in to the identity provider")
def log_in() -> None:
raise RuntimeError("identity provider is down")
@pytest.fixture(scope="module")
def identity() -> None:
log_in()
def test_dies_in_a_module_scoped_fixture(identity: None) -> None:
assert identity is None
"""
Properties = tuple[tuple[str, str], ...]
def write_suite(directory: Path, modules: Mapping[str, str]) -> None:
"""Lay a child suite out in ``directory``, with an ini file of its own.
The ini pins the child's rootdir to the tmp dir wherever that lives, and its
``pythonpath`` is what makes tests/e2e's conftest.py, and the harness
modules the child suite imports, importable under ``-I``.
"""
_ = (directory / "pytest.ini").write_text(f"[pytest]\npythonpath = {shlex.quote(str(SUITE_DIR))}\n")
for name, source in modules.items():
_ = (directory / name).write_text(source)
def run_child_pytest(suite: Path, *args: str) -> subprocess.CompletedProcess[str]:
"""Run pytest over ``suite`` in a fresh interpreter, hooked up like the live suite.
``-p conftest`` registers tests/e2e's conftest.py as a plugin, since a
tmp dir outside tests/e2e would never pick it up by location. The parent's
fixture-mode and addopts settings are dropped so a replay lane cannot leak
into the child.
"""
inherited: Final = {
name: value
for name, value in os.environ.items()
if name != "PYTEST_ADDOPTS" and not name.startswith("E2E_FIXTURE_")
}
return subprocess.run(
[sys.executable, "-I", "-m", "pytest", "-p", "conftest", "-p", "no:cacheprovider", *args, str(suite)],
cwd=suite,
env=inherited,
capture_output=True,
text=True,
timeout=CHILD_TIMEOUT_SECONDS,
check=False,
)
def properties_by_test(testsuite: ElementTree.Element) -> Mapping[str, Properties]:
"""Every testcase's <property> pairs, in document order, keyed by test name."""
return MappingProxyType(
{
testcase.get("name", ""): tuple(
(prop.get("name", ""), prop.get("value", "")) for prop in testcase.iter("property")
)
for testcase in testsuite.iter("testcase")
}
)
def values(properties: Properties, name: str) -> tuple[str, ...]:
return tuple(value for prop, value in properties if prop == name)
@pytest.fixture(
scope="module",
params=[
pytest.param((), id="in-process"),
pytest.param(
("-n", "2"),
id="xdist",
marks=pytest.mark.skipif(find_spec("xdist") is None, reason="pytest-xdist is not installed"),
),
],
)
def report(request: pytest.FixtureRequest, tmp_path_factory: pytest.TempPathFactory) -> Mapping[str, Properties]:
"""One child run per distribution mode, shared by every assertion below.
``--reruns 1`` and the ``--only-rerun`` pattern are the live suite's own
addopts. The two wide-scope modules sort ahead of the story, and next to each
other, so in-process the second one's setup runs right after the first one's
module-scoped finalizer.
"""
distribution: Final[tuple[str, ...]] = request.param # pyright: ignore[reportAny] # pytest types request.param as Any
suite: Final = tmp_path_factory.mktemp("suite")
write_suite(
suite,
{
"test_scope_a_finalizer.py": WIDE_FINALIZER_SUITE,
"test_scope_b_setup_error.py": WIDE_SETUP_ERROR_SUITE,
"test_story.py": STORY_SUITE,
},
)
xml: Final = suite / "report.xml"
child: Final = run_child_pytest(
suite, f"--junitxml={xml}", "--reruns", "1", "--only-rerun", "status_code=5[0-9][0-9]", *distribution
)
assert xml.exists(), f"the child run wrote no JUnit report:\n{child.stdout}\n{child.stderr}"
testsuite: Final = next(ElementTree.parse(xml).getroot().iter("testsuite"))
outcomes: Final = {name: testsuite.get(name) for name in ("tests", "failures", "errors", "skipped")}
assert outcomes == {"tests": "6", "failures": "1", "errors": "2", "skipped": "0"}, child.stdout
return properties_by_test(testsuite)
class TestStepsReachTheReport:
def test_a_passing_test_tells_its_story_in_call_order(self, report: Mapping[str, Properties]) -> None:
"""Fixture setup first, then the body. The finalizer's "delete virtual key"
is cleanup and is deliberately not part of the story."""
assert values(report["test_passes"], "step") == (
"generate virtual key",
"POST /chat/completions",
"poll /spend/logs",
)
def test_a_failing_test_s_last_step_is_where_it_died(self, report: Mapping[str, Properties]) -> None:
"""The reason the field exists. Nothing the test never reached is listed,
and no teardown step is appended behind the one it died on."""
assert values(report["test_fails"], "step") == ("generate virtual key", "POST /chat/completions")
def test_a_setup_error_keeps_the_steps_recorded_before_the_crash(self, report: Mapping[str, Properties]) -> None:
"""A fixture that raises never reaches the call phase, and setup is where
an e2e test most often dies (proxy not ready, key creation failing), so
the steps have to be attached after setup too."""
assert values(report["test_errors_in_setup"], "step") == ("generate virtual key", "create team")
def test_a_rerun_reports_only_the_attempt_junit_records(self, report: Mapping[str, Properties]) -> None:
"""The first attempt died on the chat call and the rerun got through. Steps
are attached twice per attempt, and none of that may show up as a doubled
or a stale story."""
assert values(report["test_passes_on_the_rerun"], "step") == (
"generate virtual key",
"POST /chat/completions",
"poll /spend/logs",
)
def test_a_setup_error_does_not_inherit_a_wider_finalizer_s_steps(self, report: Mapping[str, Properties]) -> None:
"""A module-scoped finalizer runs after the last test of its module, and
a module-scoped fixture is set up before any function-scoped one. The log
is emptied ahead of both, so the next test's setup error reports its own
steps and not "delete shared team"."""
assert values(report["test_uses_the_shared_team"], "step") == ("generate virtual key",)
assert values(report["test_dies_in_a_module_scoped_fixture"], "step") == ("log in to the identity provider",)
def test_steps_ride_behind_the_fixed_prefix(self, report: Mapping[str, Properties]) -> None:
"""`package`/`covers`/`source` are what Loki, Grafana and the status page
already read, on every outcome including a setup error."""
for name in ("test_passes", "test_fails", "test_errors_in_setup"):
assert tuple(prop for prop, _ in report[name])[:4] == ("package", "covers", "source", "step"), name

View file

@ -0,0 +1,388 @@
"""The e2e step recorder's edge cases: label templates, dedupe, the cap, nesting, context managers.
Harness logic, so it lives here rather than under tests/e2e, which holds only
tests that drive a live proxy. The harness modules are imported off
``PYTHONPATH=tests/e2e``, the way the Code Quality workflow's
test_e2e_metadata step runs this file. Call order, the failing test's last step,
the per-test reset and the JUnit attach are pinned end to end in
test_e2e_junit_report.py.
"""
from __future__ import annotations
import ast
import inspect
import string
import threading
import warnings
from collections.abc import Callable, Generator, Iterator, Mapping
from contextlib import contextmanager
from pathlib import Path
from types import UnionType
from typing import Final, cast, get_args, get_type_hints
import pytest
from e2e_metadata import MAX_STEPS, STEP_FRAMES, STEPS, step
from proxy_client import ProxyClient
from pydantic import BaseModel, Field
@pytest.fixture(autouse=True)
def empty_step_log() -> Generator[None]:
"""Each test starts from an empty log and leaves none behind, as conftest's
`pytest_runtest_setup` hook arranges for every live test."""
STEPS.reset()
yield
STEPS.reset()
class TestStepRecording:
"""`@step`-decorated harness helpers append to the running test's story as
they execute.
Each test here starts from an empty log because `empty_step_log` resets the
recorder first, the same reset conftest's `pytest_runtest_setup` gives every
live test.
"""
def test_a_decorated_helper_still_returns_exactly_what_it_did(self) -> None:
"""`@step` records, it does not intercept: arguments, return value and
`__name__` all survive it, so decorating a live harness method cannot
change what the test observes."""
@step("POST /chat/completions")
def chat(key: str, *, model: str) -> str:
return f"{key}:{model}"
assert chat("sk-x", model="gpt-5.5") == "sk-x:gpt-5.5"
assert chat.__name__ == "chat"
def test_a_poll_loop_is_one_step_in_the_story_not_fifty(self) -> None:
@step("poll /spend/logs for the request id")
def poll() -> None:
return None
for _ in range(20):
poll()
assert STEPS.taken() == ("poll /spend/logs for the request id",)
def test_the_same_label_recorded_again_later_is_a_new_step(self) -> None:
"""Only CONSECUTIVE duplicates collapse; a helper called again after
something else happened is a genuine second beat of the story."""
STEPS.record("POST /chat/completions")
STEPS.record("poll /spend/logs")
STEPS.record("POST /chat/completions")
assert STEPS.taken() == ("POST /chat/completions", "poll /spend/logs", "POST /chat/completions")
def test_a_full_log_keeps_the_latest_steps_so_the_last_is_where_the_test_died(self) -> None:
"""A load test cannot bury the story in thousands of entries, and the cap
drops from the front: the step a test died on is the newest, so it is the
one that has to survive. The leading line says the story is partial."""
for index in range(MAX_STEPS + 10):
STEPS.record(f"call {index}")
assert STEPS.taken() == (
"(10 earlier steps not recorded)",
*(f"call {index}" for index in range(10, MAX_STEPS + 10)),
)
def test_reset_forgets_what_a_full_log_dropped(self) -> None:
for index in range(MAX_STEPS + 1):
STEPS.record(f"call {index}")
STEPS.reset()
STEPS.record("register deployment")
assert STEPS.taken() == ("register deployment",)
def test_whitespace_is_normalized_and_an_empty_label_records_nothing(self) -> None:
STEPS.record(" POST /chat/completions\n ")
STEPS.record(" ")
assert STEPS.taken() == ("POST /chat/completions",)
def test_a_decorated_helper_warns_at_its_caller_with_step_frames(self) -> None:
"""`stacklevel` counts frames, and the wrapper is one of them: a cleanup
helper that warns about its caller would otherwise report every warning at
e2e_metadata.py. Pins `STEP_FRAMES` to the frames the wrapper really adds."""
@step("delete team")
def delete_team() -> None:
warnings.warn("delete_team('t') failed", stacklevel=2 + STEP_FRAMES)
with warnings.catch_warnings(record=True) as caught:
warnings.simplefilter("always")
delete_team()
assert [Path(warning.filename).name for warning in caught] == [Path(__file__).name]
class _KeyBody(BaseModel):
models: list[str] = []
rpm_limit: int | None = None
tpm_limit: int | None = None
team_id: str | None = None
api_key: str | None = Field(default=None, repr=False)
class _Params(BaseModel):
model: str
api_key: str | None = Field(default=None, repr=False)
class _DeploymentBody(BaseModel):
model_name: str
params: _Params
def _field_type(annotation: object) -> object:
"""`X | None` is `X`: a placeholder reads the field when it is set."""
present: Final = tuple(arg for arg in get_args(annotation) if arg is not type(None))
return present[0] if isinstance(annotation, UnionType) and len(present) == 1 else annotation
def _dotted_placeholders(owner: type) -> Iterator[tuple[str, str]]:
tree: Final = ast.parse(inspect.getsource(owner))
for node in ast.walk(tree):
if not isinstance(node, ast.FunctionDef):
continue
for decorator in node.decorator_list:
match decorator:
case ast.Call(func=ast.Name(id="step"), args=[ast.Constant(value=str(label))]):
for _, field, _, _ in string.Formatter().parse(label):
if field is not None and "." in field:
yield node.name, field
case _:
pass
def _resolves(owner: type, method: str, field: str) -> bool:
root, *attributes = field.split(".")
wrapped: Final = cast("Callable[..., object]", getattr(owner, method))
hints: Final[Mapping[str, object]] = get_type_hints(inspect.unwrap(wrapped))
current: object = _field_type(hints[root]) # rebind-ok: walks one type per attribute
for attribute in attributes:
if not (isinstance(current, type) and issubclass(current, BaseModel) and attribute in current.model_fields):
return False
current = _field_type(current.model_fields[attribute].annotation) # rebind-ok: walks one type per attribute
return True
class TestLabelTemplates:
"""A label's `{placeholders}` are filled from the call's own arguments, so the
story says what the test asked for in words, and nothing the label doesn't name
ever reaches the report."""
def test_placeholders_take_the_call_arguments_and_defaults(self) -> None:
@step('Send a request to {model} with the prompt "{content}" capped at {max_tokens} tokens')
def chat(key: str, model: str, content: str, *, max_tokens: int = 16) -> None:
return None
chat("sk-live", "claude-haiku-4-5", content="hi")
assert STEPS.taken() == ('Send a request to claude-haiku-4-5 with the prompt "hi" capped at 16 tokens',)
def test_a_request_model_reads_as_only_the_fields_the_test_set(self) -> None:
@step("Generate a virtual key with {body}")
def generate_key(body: _KeyBody) -> None:
return None
generate_key(_KeyBody(models=["a", "b"], rpm_limit=3, tpm_limit=None, api_key="sk-live"))
generate_key(_KeyBody())
assert STEPS.taken() == (
"Generate a virtual key with models: a, b and rpm limit: 3",
"Generate a virtual key with default settings",
)
def test_calls_differing_only_in_arguments_are_separate_steps(self) -> None:
@step('Send "{content}"')
def chat(content: str) -> None:
return None
for content in ("one", "one", "two"):
chat(content)
assert STEPS.taken() == ('Send "one"', 'Send "two"')
def test_a_placeholder_the_helper_does_not_take_fails_at_import(self) -> None:
def chat(model: str) -> None:
return None
with pytest.raises(TypeError, match="modle"):
_ = step("Send a request to {modle}")(chat)
def test_a_dotted_placeholder_reads_one_field_of_a_request_model(self) -> None:
@step("Add a deployment named {body.model_name} that calls {body.params.model}")
def register_model(body: _DeploymentBody) -> None:
return None
register_model(_DeploymentBody(model_name="gpt", params=_Params(model="openai/gpt-5.5")))
assert STEPS.taken() == ("Add a deployment named gpt that calls openai/gpt-5.5",)
def test_a_placeholder_that_indexes_or_calls_is_refused(self) -> None:
def chat(body: _DeploymentBody) -> None:
return None
with pytest.raises(TypeError, match=r"body\.messages\[0\]"):
_ = step("Send {body.messages[0]}")(chat)
@pytest.mark.parametrize("owner", [ProxyClient], ids=["ProxyClient"])
def test_every_dotted_placeholder_in_the_harness_names_a_real_field(self, owner: type) -> None:
"""A dotted placeholder is read on every live call, so one naming a field the
request model doesn't have would fail the test calling it, not the label."""
placeholders: Final = tuple(_dotted_placeholders(owner))
assert placeholders
assert [f"{method}: {field}" for method, field in placeholders if not _resolves(owner, method, field)] == []
def test_escaped_braces_stay_literal(self) -> None:
@step("GET /v1/batches/{{id}}")
def retrieve_batch(batch_id: str) -> None:
return None
retrieve_batch("batch_123")
assert STEPS.taken() == ("GET /v1/batches/{id}",)
class TestNestedSteps:
"""Harness layers call each other, so a step's helper routinely calls other
decorated helpers. Only the outermost records."""
def test_a_step_called_inside_a_step_is_not_recorded(self) -> None:
"""`ProxyClient.create_model` wraps `register_model`: one action, one
beat of the story, at the level the test called in at."""
@step("POST /key/generate")
def generate_key() -> str:
return "sk-x"
@step("generate virtual key")
def key() -> str:
return generate_key()
assert key() == "sk-x"
assert STEPS.taken() == ("generate virtual key",)
def test_the_inner_step_records_again_once_the_outer_one_returns(self) -> None:
@step("POST /key/generate")
def generate_key() -> str:
return "sk-x"
@step("generate virtual key")
def key() -> str:
return generate_key()
_ = key()
_ = generate_key()
assert STEPS.taken() == ("generate virtual key", "POST /key/generate")
def test_an_inner_step_that_raises_leaves_the_outer_label_last_and_unwinds(self) -> None:
"""The helper the test called is where it died, and the nesting flag is
released on the way out, so the next top-level call still records."""
@step("POST /team/new")
def post_team() -> None:
raise RuntimeError("/team/new answered 500")
@step("create team with a budget")
def create_team() -> None:
post_team()
@step("POST /chat/completions")
def chat() -> None:
return None
with pytest.raises(RuntimeError, match="answered 500"):
create_team()
chat()
assert STEPS.taken() == ("create team with a budget", "POST /chat/completions")
def test_a_worker_thread_a_step_fans_out_to_records_its_own_steps(self) -> None:
"""Nesting is per thread: a load helper that fans chats out to workers is
not inside a step on those workers, so their calls are still recorded."""
@step("POST /chat/completions")
def chat() -> None:
return None
@step("fire concurrent chats")
def fan_out() -> None:
worker = threading.Thread(target=chat)
worker.start()
worker.join()
fan_out()
assert STEPS.taken() == ("fire concurrent chats", "POST /chat/completions")
class TestContextManagerSteps:
"""A `@contextmanager` helper's setup and cleanup run at `__enter__` and
`__exit__`, after the decorated call has returned. Both still count as part
of its step; the `with` body is the test's own code and records as usual."""
def test_setup_and_cleanup_stay_inside_the_step_and_the_body_records(self) -> None:
@step("run a SQL statement")
def execute() -> None:
return None
@step("create a read-only database role")
@contextmanager
def restricted_user() -> Generator[str]:
execute()
try:
yield "reader"
finally:
execute()
@step("POST /chat/completions")
def chat() -> None:
return None
with restricted_user() as user:
assert user == "reader"
chat()
assert STEPS.taken() == ("create a read-only database role", "POST /chat/completions")
def test_a_test_that_dies_in_the_with_body_keeps_its_last_step_last(self) -> None:
"""The guarantee the field makes: the cleanup that runs on the way out of
the `with` must not append a step behind the one the test died on."""
@step("drop the role")
def drop_role() -> None:
return None
@step("create a read-only database role")
@contextmanager
def restricted_user() -> Generator[None]:
try:
yield
finally:
drop_role()
@step("POST /chat/completions")
def chat() -> None:
raise RuntimeError("502 from upstream")
with pytest.raises(RuntimeError, match="502 from upstream"), restricted_user():
chat()
assert STEPS.taken() == ("create a read-only database role", "POST /chat/completions")
def test_the_wrapped_context_keeps_its_exception_handling(self) -> None:
"""`__exit__` is forwarded, return value included, so a context that
suppresses an exception still does."""
@step("hold an advisory lock")
@contextmanager
def swallowing() -> Generator[None]:
try:
yield
except KeyError:
pass
with swallowing():
raise KeyError("suppressed by the context")
assert STEPS.taken() == ("hold an advisory lock",)
def test_a_bare_generator_is_refused_where_the_decorator_runs(self) -> None:
"""Its body runs only as the caller iterates, interleaved with the caller's
own steps, so no single point in the story is where it happened. Refused at
decoration, which for a harness module is import, so it lands as a
collection error rather than a story that quietly reads out of order."""
def rows() -> Generator[int]:
yield 1
with pytest.raises(TypeError, match="cannot wrap the generator function"):
_ = step("poll /spend/logs")(rows)

View file

@ -131,6 +131,14 @@ Current limits: Bedrock cannot be mounted in record or replay (SigV4 signs the H
The harness is fully typed with no error budget: `make lint-e2e-basedpyright` must report zero basedpyright errors, and CI enforces that on any PR touching `tests/e2e/**/*.py`. When a response field is untyped, model it in `models.py` (just the fields you read) and let pydantic validate it, rather than threading a `dict` or `Any` through the test
## Recorded test steps
`@step("POST /chat/completions")` from `e2e_metadata.py` goes on HARNESS helpers - client methods and poll loops - never on a test, and appends its label to the running test's `user_properties` in call order. The list IS the test's user story, and because the label is recorded BEFORE the wrapped call, a failing test's LAST step is where it died. Nothing about steps is hand-written: the call sequence cannot drift from what the test actually did. Steps are rolling out one harness at a time: `ProxyClient` and the rate-limit suite's `QuotaClient` carry them today. In a harness that has them, a new public method that performs an action (an HTTP call, a poll, a login, a CLI run) gets a `@step`; pure builders, parsers and `_private` helpers do not. Labels are one beat of the story in plain English, written for a reader who never opens the code. A label may name the helper's own parameters as `{placeholders}`, filled from each call: `@step('Send a /chat/completions request to {model} with the prompt "{content}"')`. A request model reads as the fields the test set ("models: claude-haiku-4-5 and rpm limit: 3"), and a dotted placeholder reads one field of it (`{body.litellm_params.model}`). A field marked `Field(repr=False)` is never shown, so mark any secret that way. Only what a label names reaches the report, so never name a key, token or credential. A placeholder the helper does not take fails at import, and a literal brace is written `{{id}}`
Only the outermost step records. Harness layers call each other - `ProxyClient.create_model` goes through `register_model`, a domain client wraps the shared `ProxyClient` - so every layer carries its own label and the story still reads at the level the test called in at, one beat per action. On a `@contextmanager` helper `@step` goes ABOVE `@contextmanager`: the setup and cleanup around its `yield` count as part of the step, while the `with` body - the test's own code - records as usual, so cleanup never lands behind the step a test died on. A bare generator function is refused at import, since its body interleaves with the caller's. A decorated helper that warns about its caller uses `stacklevel=2 + STEP_FRAMES`, because the wrapper is a frame too. Nesting is tracked per thread, so a helper that fans work out to worker threads still records their steps. Consecutive duplicates collapse, so a poll loop is one beat rather than fifty, and the log keeps the latest 50 steps behind a line counting the ones it dropped: the cap drops from the front because the last step is where a failing test died. The log is emptied first thing in every test's setup phase and attached after setup and again after call, so a test that errors in a fixture keeps the steps recorded before the crash. Teardown steps are left out on purpose: they are cleanup, and listing them would put a finalizer's step after the one a failing test died on
Steps ride out as repeated JUnit `<property name="step">` entries (`junit_properties.py`), one per step rather than one delimiter-joined value, since a free-text label has no separator that can be reserved. The results JSON downstream regroups them into a `steps` array. The harness tests for it sit outside the suite, in `tests/code_coverage_tests/test_e2e_metadata.py` and `test_e2e_junit_report.py`; the latter runs real pytest with `--junitxml` through this conftest, in-process and under `-n 2`, and pins what reaches the XML
## Coverage registry
The set of tests we want is a registry checked into this repo, one row per behavior; that file is the definition of done and the denominator. Each e2e test declares what it covers with `@pytest.mark.covers("...")`, and a small collector diffs the registry against the tests and ships coverage to the existing Grafana. No Allure, no new dependencies

View file

@ -42,10 +42,11 @@ from e2e_config import (
)
from e2e_db import RESET_OPT_IN_ENV, reset_spend_logs, run_spend_log_cleanup
from e2e_http import unwrap
from e2e_metadata import STEPS
from fixture_mode import fixture_mode_collection_error, fixture_report_lines
from fixture_mode import pytest_fixture_setup as pytest_fixture_setup
from idp import Identity, Keycloak, keycloak_from_env
from junit_properties import attach_result_properties
from junit_properties import attach_result_properties, attach_step_properties
from lifecycle import ProxyClientProvider, ResourceManager
from memory_readings import RssCapture, read_rss_everywhere
from models import TeamNewBody, UserNewBody, UserNewResponse
@ -289,7 +290,14 @@ def pytest_runtest_setup(item: pytest.Item) -> None:
"""Hard-fail `e2e`-marked tests unless a proxy answers its liveness probe.
Unmarked tests (unit coverage of the harness) don't touch the proxy, so they
run even when none is up. Never skip for a missing proxy. Replay mode needs
the proxy too: only provider-bound traffic replays from the bundle."""
the proxy too: only provider-bound traffic replays from the bundle.
Also empties the step log, so the story a test tells is its own. It happens
here, first in the setup phase, rather than in a fixture: a fixture only runs
once every wider-scoped fixture ahead of it has been set up, so a step a
module-scoped finalizer recorded after the previous test would still be in
the log when this test's setup dies early, and would be reported as its own."""
STEPS.reset()
LIVE_PROVIDER_REQUIRED.set(item.get_closest_marker("provider_live") is not None)
if _uses_idle_rss(item):
item.user_properties.extend(item.config.stash[_IDLE_RSS].junit_properties)
@ -318,7 +326,24 @@ def pytest_runtest_makereport(
item: pytest.Item, call: pytest.CallInfo[None]
) -> Generator[None, pytest.TestReport, pytest.TestReport]:
"""Stash the call-phase outcome so teardown can tell a passed test from a
failed one without re-deriving it."""
failed one without re-deriving it, and attach the runtime-recorded steps.
The steps cannot ride along with the other properties in
`pytest_collection_modifyitems`: that hook runs before any test body has, so
the recorder is empty there. They are attached after setup and again after
call, on every outcome -- a failing test's last step is where it died, which
is the whole reason the field exists. Setup has to attach too because a test
whose fixture raises never reaches the call phase, and setup is where an e2e
test most often dies (proxy not ready, key creation failing). The second
attach replaces the first, so nothing is doubled. JUnit writes properties
from the teardown report, which pytest builds from `item.user_properties`
after both of these have run.
Teardown deliberately does not attach. Steps recorded by fixture finalizers
are cleanup, and appending them would put "delete virtual key" after the step
a failing test died on, which breaks the one guarantee the field makes. A
finalizer that raises is still reported by JUnit with its own traceback.
"""
report = yield
if item.get_closest_marker("mcp_oauth_live") is not None and call.excinfo is not None:
# Publish code locations only, never exception messages, source text or locals.
@ -329,6 +354,8 @@ def pytest_runtest_makereport(
report.user_properties = list(item.user_properties)
if report.when == "call":
item.stash[_CALL_PASSED] = report.passed
if report.when in ("setup", "call"):
attach_step_properties(item)
return report

262
tests/e2e/e2e_metadata.py Normal file
View file

@ -0,0 +1,262 @@
"""Per-test metadata for the e2e suite: the step log each test records as it runs.
`steps` is appended at runtime by `@step`-decorated harness helpers, in call
order, so the list IS the test's user story and its last element is where a
failing test died. Nothing about it is hand-written, so it cannot drift from
what the test actually did.
tests/e2e is a black-box HTTP suite that imports litellm in zero files and is
shipped to the runner image as tests/e2e alone, and every harness module imports
this one, so it imports only the stdlib and pydantic.
"""
from __future__ import annotations
import inspect
import re
import string
import threading
from collections import deque
from collections.abc import Callable, Generator, Iterable, Mapping
from contextlib import AbstractContextManager, contextmanager
from enum import Enum
from functools import reduce, wraps
from types import TracebackType
from typing import Final, ParamSpec, TypeVar, cast
from pydantic import BaseModel
_P = ParamSpec("_P")
_R = TypeVar("_R")
_Y = TypeVar("_Y")
MAX_STEPS: Final = 50
MAX_STEP_CHARS: Final = 200
STEP_FRAMES: Final = 1
"""Frames a `@step` wrapper puts between a helper and its caller. A decorated
helper that warns about its caller adds this to `stacklevel`
(`stacklevel=2 + STEP_FRAMES`), or the warning is reported at the wrapper."""
class _StepRecorder:
"""The ordered step log for the running test.
A plain lock-guarded list rather than a ContextVar: ContextVars do not
propagate into worker threads, and several e2e helpers call out from
threads. Under xdist each worker is its own process, so there is no
cross-test bleed beyond what the per-test reset already handles.
"""
def __init__(self) -> None:
self._lock = threading.Lock()
self._steps: deque[str] = deque(maxlen=MAX_STEPS)
self._dropped = 0
def reset(self) -> None:
"""Called first thing in every test's setup phase, so each test starts
empty."""
with self._lock:
self._steps.clear()
self._dropped = 0
def record(self, label: str) -> None:
"""Append `label`, unless it repeats the previous step.
A retrying helper (poll_cost_row) or a load test calling a decorated
helper in a loop would otherwise emit thousands of <property> entries per
testcase: a consecutive repeat collapses, so a poll loop is one step in
the story rather than fifty, and past MAX_STEPS the oldest step makes way.
It is the oldest that goes because the last step is the one that has to
survive: it is where a failing test died.
"""
cleaned = " ".join(label.split())[:MAX_STEP_CHARS]
if not cleaned:
return
with self._lock:
if self._steps and self._steps[-1] == cleaned:
return
if len(self._steps) == MAX_STEPS:
self._dropped += 1
self._steps.append(cleaned)
def taken(self) -> tuple[str, ...]:
"""The story so far, led by a line counting the steps a full log dropped,
so a story that starts mid-test says so rather than reading as complete."""
with self._lock:
dropped: Final = (f"({self._dropped} earlier steps not recorded)",) if self._dropped else ()
return dropped + tuple(self._steps)
STEPS: Final = _StepRecorder()
def _joined(phrases: tuple[str, ...]) -> str:
if len(phrases) <= 1:
return "".join(phrases)
return f"{', '.join(phrases[:-1])} and {phrases[-1]}"
def _model_phrase(model: BaseModel) -> str:
"""The fields the caller set, as "models: a, b and rpm limit: 3". A
`Field(repr=False)` field, pydantic's flag for a secret, is never shown."""
values: Final = (
(name, cast("object", getattr(model, name)))
for name, field in type(model).model_fields.items()
if name in model.model_fields_set and field.repr
)
phrases: Final = tuple(f"{name.replace('_', ' ')}: {_phrase(value)}" for name, value in values if _given(value))
return _joined(phrases) or "default settings"
def _given(value: object) -> bool:
return value is not None and value != [] and value != ()
def _phrase(value: object) -> str:
if isinstance(value, BaseModel):
return _model_phrase(value)
if isinstance(value, Enum):
return _phrase(cast("object", value.value))
if isinstance(value, Mapping):
entries: Final = cast("Mapping[object, object]", value)
return _joined(tuple(f"{str(key).replace('_', ' ')}: {_phrase(item)}" for key, item in entries.items()))
if isinstance(value, (list, tuple, set, frozenset)):
return ", ".join(map(_phrase, cast("Iterable[object]", value)))
return str(value)
_PLACEHOLDER: Final = re.compile(r"[A-Za-z_]\w*(\.[A-Za-z_]\w*)*")
def _placeholders(label: str) -> frozenset[str]:
return frozenset(field for _, field, _, _ in string.Formatter().parse(label) if field is not None)
def _resolved(field: str, arguments: Mapping[str, object]) -> object:
"""`body.litellm_params.model` is the `body` argument's `litellm_params.model`."""
root, *attributes = field.split(".")
return reduce(lambda value, attribute: cast("object", getattr(value, attribute)), attributes, arguments[root])
def _filled(label: str, bound: inspect.BoundArguments) -> str:
bound.apply_defaults()
arguments: Final = cast("Mapping[str, object]", bound.arguments)
return "".join(
literal + ("" if field is None else _phrase(_resolved(field, arguments)))
for literal, field, _, _ in string.Formatter().parse(label)
)
class _Nesting(threading.local):
"""Whether this thread is already inside a `@step` helper.
Per thread, like the helpers themselves: a worker thread a step fans out to
starts outside any step, so its own decorated calls still record."""
def __init__(self) -> None:
self.inside: bool = False
_NESTING: Final = _Nesting()
@contextmanager
def _inside_step() -> Generator[None]:
"""Hold the nesting guard for the duration, restoring whatever it was."""
outer: Final = _NESTING.inside
_NESTING.inside = True
try:
yield
finally:
_NESTING.inside = outer
class _StepContext(AbstractContextManager[_Y]):
"""A `@contextmanager` helper's context, entered and exited inside its step.
Calling a `@contextmanager` function runs none of its body: the setup runs at
`__enter__` and the cleanup at `__exit__`, both after the call has returned
and so both outside the guard the call held. Here each runs inside it, so the
helpers they call stay out of the story, while the `with` body in between --
the test's own code -- still records. Without this, a test that died inside
the `with` would have the cleanup's steps appended behind the one it died on.
"""
def __init__(self, inner: AbstractContextManager[_Y]) -> None:
self._inner: Final = inner
def __enter__(self) -> _Y:
with _inside_step():
return self._inner.__enter__()
def __exit__(
self,
exc_type: type[BaseException] | None,
exc: BaseException | None,
traceback: TracebackType | None,
) -> bool | None:
with _inside_step():
return self._inner.__exit__(exc_type, exc, traceback)
def step(label: str) -> Callable[[Callable[_P, _R]], Callable[_P, _R]]:
"""Record `label` on the running test whenever this helper is called.
Goes on HARNESS helpers (client methods, fixtures), never on tests. The
label is recorded BEFORE the wrapped call, so a helper that raises still
leaves its own label as the last element -- which is the whole point: the
last step is where the test died.
Only the outermost step records. Harness layers call each other --
`ProxyClient.create_model` goes through `register_model`, a domain
client wraps the shared `ProxyClient` -- so every layer can carry its own
label without one action showing up in the story once per layer. The story
reads at the level the test called in at, and the label of the helper the
test called is still the last one when anything beneath it raises.
On a `@contextmanager` helper `@step` goes ABOVE `@contextmanager`, and the
setup and cleanup around its `yield` count as part of the step (see
`_StepContext`). A bare generator function is refused where the decorator
runs: its body only runs as the caller iterates, interleaved with the
caller's own steps, so no single point in the story is where it happened.
"""
def decorate(fn: Callable[_P, _R]) -> Callable[_P, _R]:
signature: Final = inspect.signature(fn)
placeholders: Final = _placeholders(label)
malformed: Final = sorted(field for field in placeholders if not _PLACEHOLDER.fullmatch(field))
if malformed:
raise TypeError(f"@step({label!r}) has {malformed}: a placeholder is a parameter or its dotted attribute")
unknown: Final = {field.split(".")[0] for field in placeholders} - signature.parameters.keys()
if unknown:
raise TypeError(f"@step({label!r}) names {sorted(unknown)}, which {fn.__qualname__} doesn't take")
static_label: Final = None if placeholders else label.format()
if inspect.isgeneratorfunction(fn):
raise TypeError(
f"@step({label!r}) cannot wrap the generator function {fn!r}: put it on a helper that"
" returns, or above @contextmanager on one that yields a context"
)
underlying: Final[object] = inspect.unwrap(fn) # pyright: ignore[reportAny] # inspect.unwrap is typed as returning Any
opens_a_context: Final = inspect.isgeneratorfunction(underlying)
@wraps(fn)
def wrapper(*args: _P.args, **kwargs: _P.kwargs) -> _R:
if not _NESTING.inside:
STEPS.record(static_label or _filled(label, signature.bind(*args, **kwargs)))
with _inside_step():
result = fn(*args, **kwargs)
if opens_a_context and isinstance(result, AbstractContextManager):
context: Final = cast("AbstractContextManager[object]", result)
return cast("_R", _StepContext(context))
return result
return wrapper
return decorate
def step_properties() -> tuple[tuple[str, str], ...]:
"""The step log as repeated `step` properties. Appended after the setup and
call phases, never at collection."""
return tuple(("step", label) for label in STEPS.taken())

View file

@ -20,6 +20,7 @@ from collections.abc import Iterable
import pytest
from coverage_registry.management_cases import case_properties
from e2e_metadata import step_properties
# Hardcoded because the runner image copies tests/e2e/ to /app/e2e, so nothing
# at runtime names this suite's place in the repo. test_junit_properties.py
@ -105,3 +106,21 @@ def attach_result_properties(item: pytest.Item) -> None:
if any(name == "package" for name, _ in item.user_properties):
return
item.user_properties.extend(result_properties(item))
def attach_step_properties(item: pytest.Item) -> None:
"""Attach the runtime-recorded steps; called after setup and after call.
Separate from `attach_result_properties` because it cannot share its home:
that one runs in `pytest_collection_modifyitems`, before any test body has
executed, so the recorder is necessarily empty there.
Any `step` entries already on the item are dropped first, which is what makes
the second call of a test safe: the story attached after setup is replaced by
the longer one attached after call. It also covers `--reruns 1`, where a flaky
test's second attempt would otherwise append a second copy of the story behind
the first, and the report would read as one very long test that did everything
twice. Last attempt wins, which is the attempt whose outcome JUnit records.
"""
item.user_properties[:] = [entry for entry in item.user_properties if entry[0] != "step"]
item.user_properties.extend(step_properties())

View file

@ -979,7 +979,7 @@ class SpendLogMetadata(BaseModel):
class SpendLogRow(BaseModel):
request_id: str | None = None
api_key: str | None = None
api_key: str | None = Field(default=None, repr=False)
model: str | None = None
spend: float | None = None
status: str | None = None
@ -1007,7 +1007,7 @@ class SpendLogs(RootModel[list[SpendLogRow]]):
class SpendLogsParams(BaseModel):
request_id: str | None = None
api_key: str | None = None
api_key: str | None = Field(default=None, repr=False)
@model_validator(mode="after")
def require_filter(self) -> SpendLogsParams:
@ -1028,7 +1028,7 @@ class SpendLogsPageParams(BaseModel):
end_date: str
page: int
page_size: int
api_key: str | None = None
api_key: str | None = Field(default=None, repr=False)
class SessionSpendLogsParams(BaseModel):
@ -1229,25 +1229,25 @@ class LiteLLMParamsBody(BaseModel):
backend's canonical rate."""
model: str
api_key: str | None = None
api_key: str | None = Field(default=None, repr=False)
litellm_credential_name: str | None = None
api_base: str | None = None
api_version: str | None = None
realtime_protocol: str | None = None
allowed_openai_params: list[str] | None = None
aws_access_key_id: str | None = None
aws_secret_access_key: str | None = None
aws_access_key_id: str | None = Field(default=None, repr=False)
aws_secret_access_key: str | None = Field(default=None, repr=False)
aws_region_name: str | None = None
aws_bedrock_runtime_endpoint: str | None = None
vertex_project: str | None = None
vertex_location: str | None = None
vertex_credentials: str | None = None
vertex_credentials: str | None = Field(default=None, repr=False)
gcs_bucket_name: str | None = None
bucket_name: str | None = None
s3_bucket_name: str | None = None
s3_region_name: str | None = None
s3_access_key_id: str | None = None
s3_secret_access_key: str | None = None
s3_access_key_id: str | None = Field(default=None, repr=False)
s3_secret_access_key: str | None = Field(default=None, repr=False)
s3_encryption_key_id: str | None = None
aws_batch_role_arn: str | None = None
aws_role_name: str | None = None
@ -1368,7 +1368,7 @@ class ConnectionTestResponse(BaseModel):
class CredentialCreateBody(BaseModel):
credential_name: str
credential_values: dict[str, str]
credential_values: dict[str, str] = Field(repr=False)
credential_info: dict[str, str] = {}

View file

@ -43,6 +43,7 @@ from e2e_http import (
is_ok,
unwrap,
)
from e2e_metadata import STEP_FRAMES, step
from models import (
AnthropicMessagesBody,
AnthropicMessagesResponse,
@ -472,6 +473,7 @@ class ProxyClient:
# ---- keys / customers (satisfies lifecycle.ResourceClient) ----------
@step("Generate a virtual key with {body}")
def generate_key(self, body: KeyGenerateBody) -> str:
return unwrap(
self.transport.post(
@ -482,6 +484,7 @@ class ProxyClient:
)
).key
@step("Delete the virtual key")
def delete_key(self, key: str) -> None:
_ = self.transport.post(
"/key/delete",
@ -490,6 +493,7 @@ class ProxyClient:
response_type=NoBody,
)
@step("Delete the end users {user_ids}")
def delete_customers(self, user_ids: list[str]) -> None:
if not user_ids:
return
@ -500,6 +504,7 @@ class ProxyClient:
response_type=NoBody,
)
@step("Read the key's settings back from /key/info")
def key_info(self, key: str) -> KeyInfo:
return unwrap(
self.transport.get(
@ -510,6 +515,7 @@ class ProxyClient:
)
).info
@step("Read memory usage from /debug/memory/summary on every proxy replica")
def memory_summary_everywhere(
self, *, timeout: float | None = None
) -> Mapping[str, Result[MemorySummaryResponse]]:
@ -524,6 +530,7 @@ class ProxyClient:
for url, transport in self.replicas.items()
}
@step("Read {path} on every proxy replica until they all agree")
def read_back_everywhere[R: BaseModel](
self,
path: str,
@ -571,6 +578,7 @@ class ProxyClient:
path, headers=self.management_headers(transport=transport), params=params, response_type=response_type
)
@step("List the deployments from /model/info")
def model_info(self) -> list[ModelInfoEntry]:
"""Every configured deployment with the price the proxy resolved for it
(config override merged over cost-map defaults)."""
@ -583,6 +591,7 @@ class ProxyClient:
)
).data
@step("Read the router settings from /router/settings")
def router_settings(self) -> RouterCurrentValues:
"""The router knobs the proxy is running with, for a test whose behavior
needs one of them switched on in the proxy config."""
@ -595,6 +604,7 @@ class ProxyClient:
)
).current_values
@step("Read the model cost map")
def model_cost_map(self) -> dict[str, CostMapEntry]:
return unwrap(
self.transport.get(
@ -605,6 +615,7 @@ class ProxyClient:
)
).root
@step("List files from /v1/files")
def list_files(self, key: str) -> Result[FileListResponse]:
return self.transport.get(
"/v1/files",
@ -613,6 +624,7 @@ class ProxyClient:
response_type=FileListResponse,
)
@step("List {params.custom_llm_provider} fine-tuning jobs from /v1/fine_tuning/jobs")
def list_fine_tuning_jobs(self, key: str, params: FineTuningJobsParams) -> Result[FineTuningJobsResponse]:
return self.transport.get(
"/v1/fine_tuning/jobs",
@ -621,6 +633,7 @@ class ProxyClient:
response_type=FineTuningJobsResponse,
)
@step("Add a deployment named {model_name} that calls {litellm_params.model}")
def create_model(
self,
model_name: str,
@ -640,6 +653,7 @@ class ProxyClient:
provider_live=provider_live,
)
@step("Check whether the general setting {field_name} is on")
def general_setting_enabled(self, field_name: str) -> bool:
"""Whether the proxy is running with the named general_settings flag on, for
a test whose behavior only exists under a config flag the stack has to carry."""
@ -653,6 +667,7 @@ class ProxyClient:
).root
return any(entry.field_name == field_name and entry.field_value is True for entry in fields)
@step("Add a deployment named {body.model_name} that calls {body.litellm_params.model}")
def register_model(
self, body: ModelNewBody, listed_for: str | None = None, *, provider_live: bool = False
) -> str:
@ -735,6 +750,7 @@ class ProxyClient:
timeout=poll_timeout,
)
@step("Update a deployment's settings to {litellm_params}")
def update_model(self, model_id: str, litellm_params: LiteLLMParamsBody) -> None:
"""Merge `litellm_params` over the deployment `model_id`'s stored params via
POST /model/update. The proxy overlays only the non-null fields and clears
@ -752,6 +768,7 @@ class ProxyClient:
)
)
@step("Delete the deployment")
def delete_model(self, model_id: str) -> None:
result = self.transport.post(
"/model/delete",
@ -760,7 +777,7 @@ class ProxyClient:
response_type=NoBody,
)
if not is_ok(result):
warnings.warn(f"delete_model({model_id!r}) failed: {result}", stacklevel=2)
warnings.warn(f"delete_model({model_id!r}) failed: {result}", stacklevel=2 + STEP_FRAMES)
# ---- replica read-back ----------------------------------------------
@ -776,6 +793,7 @@ class ProxyClient:
assert replicas, f"no replica is configured to serve {path}, so a read-back there would prove nothing"
return replicas
@step("Read {path} on every proxy replica until it settles")
def read_body_back_everywhere[R: BaseModel](
self, path: str, response_type: type[R], *, settled: Callable[[R], bool]
) -> Mapping[str, R]:
@ -801,6 +819,7 @@ class ProxyClient:
f"last read: {last}"
)
@step("Check that {path} returns 404 on every proxy replica")
def gone_everywhere(self, path: str) -> Mapping[str, int]:
"""Poll GET `path` on every replica that serves it until each stops serving
it, and fail naming the first replica that still does at poll_timeout.
@ -833,6 +852,7 @@ class ProxyClient:
# ---- mcp toolsets ---------------------------------------------------
@step("Create an MCP toolset with the tools {body.tools}")
def create_toolset(self, body: ToolsetCreateBody) -> ToolsetRow:
return unwrap(
self.transport.post(
@ -843,6 +863,7 @@ class ProxyClient:
)
)
@step("Update an MCP toolset with {body}")
def update_toolset(self, body: ToolsetUpdateBody) -> ToolsetRow:
"""PUT /v1/mcp/toolset: a partial update where a field left unset keeps its
stored value and None clears it."""
@ -855,6 +876,7 @@ class ProxyClient:
)
)
@step("Delete the MCP toolset")
def delete_toolset(self, toolset_id: str) -> Result[NoBody]:
"""DELETE /v1/mcp/toolset/{toolset_id}. Returns the outcome so the act phase
can unwrap it while a deferred teardown can ignore an already-deleted row."""
@ -865,6 +887,7 @@ class ProxyClient:
response_type=NoBody,
)
@step("Create a search tool backed by {body.search_tool.litellm_params.search_provider}")
def create_search_tool(self, body: SearchToolCreateBody) -> str:
"""POST /search_tools: register a search tool on the running proxy and return its id
once every worker has had a config-reload window to pick it up from the DB."""
@ -879,6 +902,7 @@ class ProxyClient:
settle_propagation(time.monotonic())
return search_tool_id
@step("Delete the search tool")
def delete_search_tool(self, search_tool_id: str) -> None:
result = self.transport.delete(
f"/search_tools/{search_tool_id}",
@ -887,8 +911,9 @@ class ProxyClient:
response_type=NoBody,
)
if not is_ok(result):
warnings.warn(f"delete_search_tool({search_tool_id!r}) failed: {result}", stacklevel=2)
warnings.warn(f"delete_search_tool({search_tool_id!r}) failed: {result}", stacklevel=2 + STEP_FRAMES)
@step("Save a provider credential for {body.credential_info}")
def create_credential(self, body: CredentialCreateBody) -> None:
unwrap(
self.transport.post(
@ -899,6 +924,7 @@ class ProxyClient:
)
)
@step("Delete the provider credential")
def delete_credential(self, credential_name: str) -> None:
result = self.transport.delete(
f"/credentials/{credential_name}",
@ -907,8 +933,9 @@ class ProxyClient:
response_type=NoBody,
)
if not is_ok(result):
warnings.warn(f"delete_credential({credential_name!r}) failed: {result}", stacklevel=2)
warnings.warn(f"delete_credential({credential_name!r}) failed: {result}", stacklevel=2 + STEP_FRAMES)
@step("Create a team with {body}")
def create_team(self, body: TeamNewBody) -> str:
return unwrap(
self.transport.post(
@ -919,6 +946,7 @@ class ProxyClient:
)
).team_id
@step("Update a team with {body}")
def update_team(self, body: TeamUpdateBody) -> None:
unwrap(
self.transport.post(
@ -929,6 +957,7 @@ class ProxyClient:
)
)
@step("Delete the team")
def delete_team(self, team_id: str) -> None:
result = self.transport.post(
"/team/delete",
@ -937,8 +966,9 @@ class ProxyClient:
response_type=NoBody,
)
if not is_ok(result):
warnings.warn(f"delete_team({team_id!r}) failed: {result}", stacklevel=2)
warnings.warn(f"delete_team({team_id!r}) failed: {result}", stacklevel=2 + STEP_FRAMES)
@step("Delete the internal user")
def delete_user(self, user_id: str) -> None:
"""Best-effort teardown; a 404 is not a leak, since JWT tests defer this for
a user the proxy only upserts after a successful auth."""
@ -952,10 +982,11 @@ class ProxyClient:
case Success() | UnknownApiError(status_code=404):
return
case _:
warnings.warn(f"delete_user({user_id!r}) failed: {result}", stacklevel=2)
warnings.warn(f"delete_user({user_id!r}) failed: {result}", stacklevel=2 + STEP_FRAMES)
# ---- LLM calls ------------------------------------------------------
@step("Send a /chat/completions request to {body.model}")
def chat(self, key: str, body: ChatBody) -> Result[ChatResponse]:
return self.transport.post(
"/chat/completions",
@ -964,15 +995,19 @@ class ProxyClient:
response_type=ChatResponse,
)
@step("Send a streaming /chat/completions request to {body.model}")
def chat_stream(self, key: str, body: ChatBody) -> StreamingResponse:
return self.transport.stream("/chat/completions", headers=self.transport.bearer(key), json=body)
@step("Send a streaming /v1/messages request to {body.model}")
def messages_stream(self, key: str, body: AnthropicMessagesBody) -> StreamingResponse:
return self.transport.stream("/v1/messages", headers=self.transport.bearer(key), json=body)
@step("Send a streaming /v1/responses request to {body.model}")
def responses_stream(self, key: str, body: ResponsesStreamBody) -> StreamingResponse:
return self.transport.stream("/v1/responses", headers=self.transport.bearer(key), json=body)
@step('Send an /embeddings request to {body.model} for "{body.input}"')
def embed(self, key: str, body: EmbedBody) -> Result[EmbedResponse]:
return self.transport.post(
"/embeddings",
@ -981,6 +1016,7 @@ class ProxyClient:
response_type=EmbedResponse,
)
@step("Send a /v1/ocr request to {body.model}")
def ocr(self, key: str, body: OcrBody) -> Result[OcrResponse]:
return self.transport.post(
"/v1/ocr",
@ -990,6 +1026,7 @@ class ProxyClient:
timeout=SLOW_PROVIDER_TIMEOUT_SECONDS,
)
@step('Send a /v1/rerank request to {body.model} for "{body.query}"')
def rerank(self, key: str, body: RerankBody) -> Result[RerankResponse]:
"""POST /v1/rerank (Cohere-format). No official OpenAI/Anthropic SDK
covers this route, so it stays on the shared typed transport."""
@ -1000,6 +1037,7 @@ class ProxyClient:
response_type=RerankResponse,
)
@step("Count tokens with /v1/messages/count_tokens for {body.model}")
def count_tokens(self, key: str, body: CountTokensBody) -> Result[CountTokensResponse]:
"""POST /v1/messages/count_tokens (Anthropic-native). Sends the
anthropic-version header so the native path accepts it; harmless on the
@ -1011,6 +1049,7 @@ class ProxyClient:
response_type=CountTokensResponse,
)
@step("Send a /v1/messages request to {body.model}")
def messages(
self, key: str, body: AnthropicMessagesBody, *, session_id: str | None = None
) -> Result[AnthropicMessagesResponse]:
@ -1034,6 +1073,7 @@ class ProxyClient:
# ---- spend read-back ------------------------------------------------
@step("Read /spend/logs")
def spend_logs(self, params: SpendLogsParams) -> list[SpendLogRow]:
result = self.transport.get(
"/spend/logs",
@ -1047,6 +1087,7 @@ class ProxyClient:
case _:
return []
@step("Read /spend/logs between {start} and {end}")
def spend_logs_window(self, *, start: datetime, end: datetime) -> list[SpendLogRow]:
def fetch(page: int) -> SpendLogsPage:
return unwrap(
@ -1069,11 +1110,13 @@ class ProxyClient:
*(row for page in range(2, first.total_pages + 1) for row in fetch(page).data),
]
@step("Wait for at least {min_rows} of the key's spend logs in /spend/logs")
def poll_logs_for_key(
self, key: str, *, min_rows: int = 1, predicate: RowsPredicate | None = None
) -> list[SpendLogRow]:
return self._poll(lambda: self.spend_logs(SpendLogsParams(api_key=key)), min_rows, predicate)
@step("Read the session's spend logs from /spend/logs/session/ui")
def session_spend_logs(self, session_id: str) -> list[SpendLogRow]:
"""GET /spend/logs/session/ui, the per-session view the Admin UI logs page
opens when a session id is clicked."""
@ -1086,6 +1129,7 @@ class ProxyClient:
)
).data
@step("Wait for at least {min_rows} of the session's spend logs in /spend/logs")
def poll_logs_for_session(
self,
session_id: str,
@ -1095,6 +1139,7 @@ class ProxyClient:
) -> list[SpendLogRow]:
return self._poll(lambda: self.session_spend_logs(session_id), min_rows, predicate)
@step("Wait for the request's spend log in /spend/logs")
def poll_logs_for_request_id(
self,
request_id: str,
@ -1125,6 +1170,7 @@ class ProxyClient:
# ---- route probe ----------------------------------------------------
@step("Call the management route {path}")
def probe(self, path: str, *, params: NoBody) -> ProbeResult:
return self.transport.probe(path, params=params, headers=self.management_headers())

View file

@ -9,6 +9,7 @@ from dataclasses import dataclass
from proxy_client import ProxyClient
from e2e_http import StreamingResponse
from e2e_metadata import step
from models import ChatBody, ChatMessage
@ -16,6 +17,7 @@ from models import ChatBody, ChatMessage
class QuotaClient:
proxy: ProxyClient
@step('Send a /chat/completions request to {model} with the prompt "{content}"')
def chat(self, key: str, model: str, content: str, *, max_tokens: int = 16) -> StreamingResponse:
return self.proxy.transport.send(
"/chat/completions",