mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-02 02:11:58 +00:00
* feat(e2e): give e2e tests typed metadata for what they drive @meta(Subject(domain, route, providers, models, capabilities, mode)) declares what a test is about with closed enums, and each field lands in the JUnit report as a property. The quota_management suites are the first to declare it. * docs(e2e): say e2e_metadata avoids litellm, not that it is stdlib-only It already imports pydantic and pytest, both of which the suite needs to collect. The rule that matters is no litellm import * test(e2e): declare models through the constant each test drives 43 @meta declarations in quota_management typed the model name out again, so changing the call would leave the coverage report naming the old model. Each file now has one constant used by both, and a guard fails on any model written as a string literal in @meta * refactor(e2e): set route only when the endpoint is what the test checks A budget or rate-limit test whose chat call only triggers the block now leaves route unset, since its steps already name the call. Tests of an endpoint keep it: budget CRUD, key creation, spend reporting reads, and the per-endpoint spend tests for chat, messages, embeddings, batches and health. The two /spend/logs tests tagged chat_completions are now spend_reporting * refactor(e2e): build the declared properties without mutating a list subject_properties seeded a list and grew it with append and extend. It now flattens one tuple per field, and the plural-name table is a read-only mapping * fix(e2e): tag each spend-route probe with the endpoint it checks The breadth test gave all 33 probes spend_reporting, so /key/list, /user/list, /team/list, /organization/list and /customer/list counted as spend reporting. Each case now carries its own route, with organization and customer management added to Route
128 lines
5.7 KiB
Python
128 lines
5.7 KiB
Python
"""Custom per-test signals for the standard JUnit reporter.
|
|
|
|
The e2e suite ships results to Loki/Grafana from a standard pytest JUnit report
|
|
(`--junitxml=e2e-report.xml`), not a bespoke log line. JUnit already records
|
|
outcome, duration, and node id for every `<testcase>`; the signals it cannot
|
|
derive on its own are the normalized suite package, the coverage-registry cell
|
|
ids a test covers, and where the test's source lives. Those ride along as JUnit
|
|
`<property>` entries via each item's `user_properties`, attached in
|
|
`conftest.py::pytest_collection_modifyitems`.
|
|
|
|
`source` is a property rather than the `file=` / `line=` attributes pytest used
|
|
to write, because the `xunit2` family this suite runs on drops those, and
|
|
switching families would change the XML for every consumer of it -- the
|
|
Buildkite Test Engine upload and the Loki pipeline included.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
from collections.abc import Iterable
|
|
|
|
import pytest
|
|
from coverage_registry.management_cases import case_properties
|
|
from e2e_metadata import step_properties, subject_properties
|
|
|
|
# Hardcoded because the runner image copies tests/e2e/ to /app/e2e, so nothing
|
|
# at runtime names this suite's place in the repo. test_junit_properties.py
|
|
# fails from a checkout if it moves.
|
|
SUITE_ROOT = "tests/e2e"
|
|
|
|
|
|
def suite_parts(path_part: str) -> tuple[str, ...]:
|
|
"""Path components of a suite file relative to tests/e2e, however it ran.
|
|
|
|
Pytest paths are rootdir-relative, and rootdir moves with the invocation: a
|
|
repo-root run gives `tests/e2e/logging/test_x.py`, a suite-cwd run (the
|
|
runner image) gives `logging/test_x.py`. Both collapse to the same tuple.
|
|
"""
|
|
raw = tuple(p for p in path_part.replace("\\", "/").split("/") if p and p != ".")
|
|
return raw[2:] if len(raw) >= 3 and raw[0] == "tests" and raw[1] == "e2e" else raw
|
|
|
|
|
|
def package_from_nodeid(nodeid: str) -> str:
|
|
"""Top-level suite package under tests/e2e/, or 'root' for top-level files."""
|
|
parts = suite_parts(nodeid.split("::", 1)[0])
|
|
if len(parts) <= 1:
|
|
return "root"
|
|
return parts[0]
|
|
|
|
|
|
def source_from_location(path: str, lineno: int | None) -> str:
|
|
"""Repo-relative `path:line` for a test, or '' when nothing is linkable.
|
|
|
|
`pytest.Item.location` gives a rootdir-relative path and a ZERO-based line.
|
|
The path is re-rooted at SUITE_ROOT so consumers need not know how pytest was
|
|
started, and the line is emitted ONE-based to match editors, tracebacks and
|
|
code hosts. A decorated test anchors at its first decorator, which is where
|
|
pytest reports it.
|
|
|
|
Empty rather than a guess for anything unlinkable: no line, a path reaching
|
|
upward, or a path carrying a colon, which is both how an absolute Windows
|
|
path arrives and a character `path:line` has no way to represent.
|
|
"""
|
|
if lineno is None:
|
|
return ""
|
|
normalized = path.replace("\\", "/")
|
|
if normalized.startswith("/") or ":" in normalized or ".." in normalized.split("/"):
|
|
return ""
|
|
parts = suite_parts(normalized)
|
|
if not parts:
|
|
return ""
|
|
return f"{'/'.join((SUITE_ROOT, *parts))}:{lineno + 1}"
|
|
|
|
|
|
def source_from_item(item: pytest.Item) -> str:
|
|
"""Read the repo-relative `path:line` off a pytest Item's reported location."""
|
|
path, lineno, _ = item.location
|
|
return source_from_location(path, lineno)
|
|
|
|
|
|
def dedupe_covers(marker_args: Iterable[tuple[object, ...]]) -> tuple[str, ...]:
|
|
"""Flatten @pytest.mark.covers arg lists into unique, order-preserving cell
|
|
ids, dropping anything that is not a non-empty string."""
|
|
return tuple(dict.fromkeys(arg for args in marker_args for arg in args if isinstance(arg, str) and arg))
|
|
|
|
|
|
def covers_from_item(item: pytest.Item) -> tuple[str, ...]:
|
|
"""Read @pytest.mark.covers cell ids off a pytest Item, order-preserving."""
|
|
return dedupe_covers(marker.args for marker in item.iter_markers(name="covers"))
|
|
|
|
|
|
def result_properties(item: pytest.Item) -> tuple[tuple[str, str], ...]:
|
|
"""The custom signals a standard reporter cannot derive.
|
|
|
|
Loki, Grafana and tests/integration/conftest.py read the `package`/`covers`/`source` prefix, so it never moves
|
|
"""
|
|
fixed = (
|
|
("package", package_from_nodeid(item.nodeid)),
|
|
("covers", ",".join(covers_from_item(item))),
|
|
("source", source_from_item(item)),
|
|
)
|
|
return fixed + case_properties(item.nodeid) + subject_properties(item)
|
|
|
|
|
|
def attach_result_properties(item: pytest.Item) -> None:
|
|
"""Attach result_properties to an item's user_properties, idempotently: a
|
|
second call is a no-op, so a collection that runs the hook more than once
|
|
never emits duplicate <property> entries."""
|
|
if any(name == "package" for name, _ in item.user_properties):
|
|
return
|
|
item.user_properties.extend(result_properties(item))
|
|
|
|
|
|
def attach_step_properties(item: pytest.Item) -> None:
|
|
"""Attach the runtime-recorded steps; called after setup and after call.
|
|
|
|
Separate from `attach_result_properties` because it cannot share its home:
|
|
that one runs in `pytest_collection_modifyitems`, before any test body has
|
|
executed, so the recorder is necessarily empty there.
|
|
|
|
Any `step` entries already on the item are dropped first, which is what makes
|
|
the second call of a test safe: the story attached after setup is replaced by
|
|
the longer one attached after call. It also covers `--reruns 1`, where a flaky
|
|
test's second attempt would otherwise append a second copy of the story behind
|
|
the first, and the report would read as one very long test that did everything
|
|
twice. Last attempt wins, which is the attempt whose outcome JUnit records.
|
|
"""
|
|
item.user_properties[:] = [entry for entry in item.user_properties if entry[0] != "step"]
|
|
item.user_properties.extend(step_properties())
|