From 5523d74236fe2b4a4d85e100ffbe3b6eb3f3a0ae Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Sat, 16 May 2026 02:23:26 +0000 Subject: [PATCH] fix(claude_code/conftest): record fail on setup error or partial crash MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two related gaps in pytest_runtest_makereport let real failures show up as green (or absent) cells in the published compatibility matrix: 1. Setup-phase failures (broken fixtures / imports) only produce a report with when="setup"; the call phase never runs. The hook filtered on when=="call" and returned, leaving no row for the cell. The matrix builder then aggregated the empty cell to "not_tested" instead of "fail". 2. A test that called compat_result.add({"status": "pass"}) for some models and then raised before completing the rest produced a partial list of pass entries. The "if not collected" guard was bypassed because the list was non-empty, so no fail row was added. The cell aggregator's all-pass check then returned pass for a cell that was never fully exercised. Now the hook also handles when=="setup" on failure, and always appends a fail row when report.failed — preserving any partial pass entries from add() for diagnostics while ensuring the cell aggregator surfaces the crash. Co-authored-by: Yassin Kortam --- tests/claude_code/conftest.py | 52 ++++++++++++++++++++++------------- 1 file changed, 33 insertions(+), 19 deletions(-) diff --git a/tests/claude_code/conftest.py b/tests/claude_code/conftest.py index 0d2593b99e4..3d3b31de557 100644 --- a/tests/claude_code/conftest.py +++ b/tests/claude_code/conftest.py @@ -191,7 +191,18 @@ def pytest_runtest_makereport(item, call): """ outcome = yield report = outcome.get_result() - if report.when != "call": + # We record on two phases: + # - "call": the normal end-of-test path. + # - "setup" but only on failure: fixture/import errors that prevent the + # test body from running. Without recording these, a broken setup + # silently becomes "not_tested" in the published matrix instead of + # "fail". Teardown is ignored — by then "call" already recorded the + # outcome, and a teardown-only failure (e.g. fixture finalizer) is + # not a cell-level signal. + if report.when == "setup": + if not report.failed: + return + elif report.when != "call": return inferred = _infer_feature_and_provider(Path(str(item.path))) @@ -204,24 +215,27 @@ def pytest_runtest_makereport(item, call): fixture.collected() if isinstance(fixture, CompatResult) else [] ) - if not collected: - if report.passed: - collected = [ - { - "status": "fail", - "error": "test passed without reporting via compat_result; " - "every compat test must report a status.", - } - ] - else: - collected = [ - { - "status": "fail", - "error": ( - str(report.longrepr) if report.longrepr else "test failed" - ), - } - ] + if report.failed: + # The test body (or setup) raised. If the test had already recorded + # some per-model passes via `.add(...)` before crashing, those + # partial entries would otherwise aggregate to "pass" and hide the + # crash from the published matrix. Append an explicit "fail" row so + # the cell aggregator (which gives precedence to any fail) surfaces + # the breakage. + collected = collected + [ + { + "status": "fail", + "error": (str(report.longrepr) if report.longrepr else "test failed"), + } + ] + elif not collected: + collected = [ + { + "status": "fail", + "error": "test passed without reporting via compat_result; " + "every compat test must report a status.", + } + ] for reported in collected: _COLLECTOR.items.append(