litellm/tests/e2e/e2e_result_reporter.py
mubashir1osmani 98765f65af
feat(e2e): emit structured E2E_RESULT lines for package status history (#33578)
* feat(e2e): emit structured E2E_RESULT lines for package status history

Pytest progress logs only expose file basenames and collapse multi-test files
into one status-history row. Emit one logfmt E2E_RESULT per finished node with
package, file, outcome, duration_ms, node_id, and covers so Grafana can roll up
by package (stable cardinality) and drill down by node_id in Explore

* test(e2e): drop unit tests from the live e2e tree

tests/e2e is for live proxy suites only. Remove harness, coverage-registry,
and claude_code unit trees so the e2e run does not collect them

* fix(e2e): type E2E_RESULT hook for basedpyright zero-error gate

Protocol-typed covers extraction and pluggy Result typing on the
makereport hook so tests/e2e stays under the e2e basedpyright ceiling

* fix(e2e): drop unused xfailed/xpassed from E2E_RESULT Outcome

We never emit those states; xfail collapses to skipped/passed via pytest
report flags. Keep the literal honest so the dashboard only sees real outcomes

* fix(e2e): strip tests/e2e prefix when deriving E2E_RESULT package

Repo-root pytest nodeids are tests/e2e/<suite>/...; without stripping,
every line would package=tests and status history would be useless

* fix(e2e): use pytest wrapper=True instead of deprecated hookwrapper

pytest 8.1+ deprecates hookwrapper; yield returns the TestReport directly
so we return it for the outer chain and drop pluggy.Result

* fix(e2e): import e2e_result_reporter at module load

Surface a missing module as a collection-time ImportError instead of a
per-test hook failure mid-run

* test(e2e): restore coverage_registry/test_collector.py

Needed to validate registry coverage math and the checked-in cell
denominator; not a live proxy suite
2026-07-16 15:01:01 -07:00

144 lines
4.2 KiB
Python

"""Structured e2e result lines for Loki / Grafana status history.
Pytest progress lines are a bad dashboard source: they only expose file basenames,
break under quiet modes, and force status-history rows to explode with suite growth.
Each finished test emits one logfmt line:
E2E_RESULT package=logging file=test_langfuse_e2e.py outcome=failed
duration_ms=1234 node_id=logging/test_langfuse_e2e.py::TestX::test_y
covers=logging.langfuse.team.success
Grafana package status-history queries max(fail) by package over E2E_RESULT lines.
Drill-down uses node_id / covers in Explore, not status-history cardinality.
"""
from __future__ import annotations
from collections.abc import Iterable, Sequence
from dataclasses import dataclass
from pathlib import Path
from typing import Literal, Protocol, runtime_checkable
Outcome = Literal["passed", "failed", "error", "skipped"]
@dataclass(frozen=True, slots=True)
class E2EResult:
package: str
file: str
outcome: Outcome
duration_ms: int
node_id: str
covers: tuple[str, ...]
@runtime_checkable
class _MarkerArgs(Protocol):
args: Sequence[object]
@runtime_checkable
class _ItemWithCovers(Protocol):
def iter_markers(self, name: str) -> Iterable[object]: ...
def package_from_nodeid(nodeid: str) -> str:
"""Top-level suite package under tests/e2e/, or 'root' for top-level files.
Pytest nodeids are relative to the invocation cwd. Repo-root runs look like
`tests/e2e/logging/...`; suite-cwd runs look like `logging/...`. Strip the
`tests/e2e` prefix so package is the suite dir either way.
"""
path_part = nodeid.split("::", 1)[0].replace("\\", "/")
parts = tuple(p for p in path_part.split("/") if p and p != ".")
if len(parts) >= 3 and parts[0] == "tests" and parts[1] == "e2e":
parts = parts[2:]
if len(parts) <= 1:
return "root"
return parts[0]
def file_from_nodeid(nodeid: str) -> str:
path_part = nodeid.split("::", 1)[0].replace("\\", "/")
return Path(path_part).name
def covers_from_item(item: object) -> tuple[str, ...]:
"""Read @pytest.mark.covers cell ids from a pytest Item."""
if not isinstance(item, _ItemWithCovers):
return ()
return tuple(
dict.fromkeys(
arg
for marker in item.iter_markers(name="covers")
if isinstance(marker, _MarkerArgs)
for arg in marker.args
if isinstance(arg, str) and arg
)
)
def outcome_from_report(when: str, failed: bool, skipped: bool, passed: bool) -> Outcome | None:
"""Map pytest TestReport fields to a terminal outcome. None if not final."""
if when == "setup" and skipped:
return "skipped"
if when == "setup" and failed:
return "error"
if when != "call":
return None
if skipped:
return "skipped"
if failed:
return "failed"
if passed:
return "passed"
return "failed"
def _logfmt_escape(value: str) -> str:
if value == "":
return '""'
needs_quote = any(ch.isspace() or ch in "\"=\\" for ch in value)
if not needs_quote:
return value
escaped = value.replace("\\", "\\\\").replace('"', '\\"')
return f'"{escaped}"'
def format_e2e_result_line(result: E2EResult) -> str:
covers = ",".join(result.covers)
fields = (
("package", result.package),
("file", result.file),
("outcome", result.outcome),
("duration_ms", str(result.duration_ms)),
("node_id", result.node_id),
("covers", covers),
)
body = " ".join(f"{key}={_logfmt_escape(value)}" for key, value in fields)
return f"E2E_RESULT {body}"
def result_from_pytest(
*,
nodeid: str,
when: str,
failed: bool,
skipped: bool,
passed: bool,
duration_seconds: float,
covers: tuple[str, ...] = (),
) -> E2EResult | None:
outcome = outcome_from_report(when=when, failed=failed, skipped=skipped, passed=passed)
if outcome is None:
return None
duration_ms = max(0, int(round(duration_seconds * 1000)))
return E2EResult(
package=package_from_nodeid(nodeid),
file=file_from_nodeid(nodeid),
outcome=outcome,
duration_ms=duration_ms,
node_id=nodeid,
covers=covers,
)