diff --git a/eval/tests/test_session_progress.py b/eval/tests/test_session_progress.py index 7f207b447..0a80290cb 100644 --- a/eval/tests/test_session_progress.py +++ b/eval/tests/test_session_progress.py @@ -184,3 +184,37 @@ def test_progress_sanitizes_a_hostile_tool_name() -> None: assert "FAKE-LOG-LINE" not in stream.getvalue() assert len(_drain_lines(stream)) == 1 + + +def test_cell_failure_detail_line_explains_why_a_cell_failed() -> None: + from workflow_bench.runner import cell_failure_detail_line + + assert cell_failure_detail_line("t", "workflow", 0, {"error_kind": None}) is None + assert cell_failure_detail_line("t", "workflow", 0, {"error_kind": "x"}) is None + + line = cell_failure_detail_line( + "trivial-version-alias", + "candidate_workflow", + 1, + { + "error_kind": "plan-evidence-invalid", + "error_detail": "unauthorized workspace path\ntoken=sk-secret-value", + }, + ("sk-secret-value",), + ) + assert line is not None + assert line.startswith("[trivial-version-alias][candidate_workflow][run 1] detail: ") + assert "unauthorized workspace path" in line + assert "sk-secret-value" not in line + assert "\n" not in line + + +def test_cell_failure_detail_line_bounds_a_huge_detail() -> None: + from workflow_bench.runner import MAX_CELL_DETAIL_CHARS, cell_failure_detail_line + + line = cell_failure_detail_line( + "t", "workflow", 0, {"error_kind": "session-error", "error_detail": {"stdout_tail": "y" * 50_000}} + ) + assert line is not None + assert "truncated" in line + assert len(line) < MAX_CELL_DETAIL_CHARS + 200 diff --git a/eval/workflow_bench/runner.py b/eval/workflow_bench/runner.py index aa76d3fee..7eb41a41e 100644 --- a/eval/workflow_bench/runner.py +++ b/eval/workflow_bench/runner.py @@ -1089,6 +1089,36 @@ def cell_progress_line(task_id: str, arm: str, run_idx: int, record: dict[str, A ) +# A failing cell's error_kind names the category; the detail names the cause. +# Bounded because a session-error detail carries stdout/stderr tails. +MAX_CELL_DETAIL_CHARS = 1200 + + +def cell_failure_detail_line( + task_id: str, + arm: str, + run_idx: int, + record: Mapping[str, Any], + secrets: Sequence[str] = (), +) -> str | None: + """The redacted reason a cell failed, or None when it succeeded. + + Without this the log says only ``error_kind=plan-evidence-invalid`` and the + reason stays locked in results.jsonl, which is an uploaded artifact rather + than something a watcher can read while the sweep is still running. + """ + if not record.get("error_kind"): + return None + detail = record.get("error_detail") + if detail in (None, "", {}, []): + return None + rendered = detail if isinstance(detail, str) else json.dumps(detail, default=str, sort_keys=True) + rendered = redact_text(rendered, secrets).replace("\n", " ⏎ ") + if len(rendered) > MAX_CELL_DETAIL_CHARS: + rendered = f"{rendered[:MAX_CELL_DETAIL_CHARS]}…[truncated {len(rendered) - MAX_CELL_DETAIL_CHARS} chars]" + return f"[{task_id}][{arm}][run {run_idx}] detail: {rendered}" + + def aggregate(records: list[dict[str, Any]]) -> dict[str, Any]: """Median metrics + resolve rate across repeated runs of one task+arm. @@ -1608,6 +1638,9 @@ def _run_sweep( # sink was not). fh.write(redact_text(json.dumps(record), credential_secrets(args)) + "\n") print(cell_progress_line(task["id"], arm, run_idx, record)) + failure = cell_failure_detail_line(task["id"], arm, run_idx, record, credential_secrets(args)) + if failure: + print(failure) outage_streak, outage_tripped = sweep_task_cells( cells,