feat(eval): log why a benchmark cell failed

The sweep printed error_kind=plan-evidence-invalid and nothing else, so
the reason a cell failed stayed in results.jsonl — an artifact uploaded
after the run, not something a watcher can read while it is still going.

Print the cell's error_detail next to its result line, redacted through
the same credential list as the artifact and bounded, since a
session-error detail carries stdout/stderr tails.

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
Gergo Magyar 2026-09-03 17:33:59 +00:00
parent 227a3502b8
commit 636d45364a
2 changed files with 67 additions and 0 deletions

View file

@ -184,3 +184,37 @@ def test_progress_sanitizes_a_hostile_tool_name() -> None:
assert "FAKE-LOG-LINE" not in stream.getvalue()
assert len(_drain_lines(stream)) == 1
def test_cell_failure_detail_line_explains_why_a_cell_failed() -> None:
from workflow_bench.runner import cell_failure_detail_line
assert cell_failure_detail_line("t", "workflow", 0, {"error_kind": None}) is None
assert cell_failure_detail_line("t", "workflow", 0, {"error_kind": "x"}) is None
line = cell_failure_detail_line(
"trivial-version-alias",
"candidate_workflow",
1,
{
"error_kind": "plan-evidence-invalid",
"error_detail": "unauthorized workspace path\ntoken=sk-secret-value",
},
("sk-secret-value",),
)
assert line is not None
assert line.startswith("[trivial-version-alias][candidate_workflow][run 1] detail: ")
assert "unauthorized workspace path" in line
assert "sk-secret-value" not in line
assert "\n" not in line
def test_cell_failure_detail_line_bounds_a_huge_detail() -> None:
from workflow_bench.runner import MAX_CELL_DETAIL_CHARS, cell_failure_detail_line
line = cell_failure_detail_line(
"t", "workflow", 0, {"error_kind": "session-error", "error_detail": {"stdout_tail": "y" * 50_000}}
)
assert line is not None
assert "truncated" in line
assert len(line) < MAX_CELL_DETAIL_CHARS + 200

View file

@ -1089,6 +1089,36 @@ def cell_progress_line(task_id: str, arm: str, run_idx: int, record: dict[str, A
)
# A failing cell's error_kind names the category; the detail names the cause.
# Bounded because a session-error detail carries stdout/stderr tails.
MAX_CELL_DETAIL_CHARS = 1200
def cell_failure_detail_line(
task_id: str,
arm: str,
run_idx: int,
record: Mapping[str, Any],
secrets: Sequence[str] = (),
) -> str | None:
"""The redacted reason a cell failed, or None when it succeeded.
Without this the log says only ``error_kind=plan-evidence-invalid`` and the
reason stays locked in results.jsonl, which is an uploaded artifact rather
than something a watcher can read while the sweep is still running.
"""
if not record.get("error_kind"):
return None
detail = record.get("error_detail")
if detail in (None, "", {}, []):
return None
rendered = detail if isinstance(detail, str) else json.dumps(detail, default=str, sort_keys=True)
rendered = redact_text(rendered, secrets).replace("\n", " ⏎ ")
if len(rendered) > MAX_CELL_DETAIL_CHARS:
rendered = f"{rendered[:MAX_CELL_DETAIL_CHARS]}…[truncated {len(rendered) - MAX_CELL_DETAIL_CHARS} chars]"
return f"[{task_id}][{arm}][run {run_idx}] detail: {rendered}"
def aggregate(records: list[dict[str, Any]]) -> dict[str, Any]:
"""Median metrics + resolve rate across repeated runs of one task+arm.
@ -1608,6 +1638,9 @@ def _run_sweep(
# sink was not).
fh.write(redact_text(json.dumps(record), credential_secrets(args)) + "\n")
print(cell_progress_line(task["id"], arm, run_idx, record))
failure = cell_failure_detail_line(task["id"], arm, run_idx, record, credential_secrets(args))
if failure:
print(failure)
outage_streak, outage_tripped = sweep_task_cells(
cells,