refactor(eval): make a benchmark cell a callable instead of loop-body scope

The sweep's innermost body — clone, sandbox, sessions, verify, oracle,
teardown — was ~260 lines of `main()`'s scope, reachable only by running
the whole sweep. Nothing tested it, and it could not run anywhere except
that loop.

`run_cell(ctx, run_idx, arm)` now owns one (run, arm) cell and returns
its row; `TaskCellContext` is a frozen dataclass holding the per-task
inputs a cell reads, so a cell depends on named fields rather than on
whatever `main()` happens to have in scope. The try/except/finally moves
verbatim, exception whitelist unchanged: a harness bug still escapes
rather than being recorded as an ordinary infra-error and averaged into
the evidence.

The task asset snapshot is now prepared once per task, next to the graph
snapshot, instead of lazily inside whichever cell reached it first.
`TaskAssetCache` is a plain dict (task_assets.py:222-226,340), so the
lazy build was a read-then-write race waiting for a caller that is not
strictly serial. One behavior delta: when that preparation fails, every
cell of the task now reports the same wrapped RuntimeError, where before
the first cell reported the original OSError/SandboxError/ValueError.

The loop keeps all the bookkeeping — progress counter, prints, the
results.jsonl append, per-arm accumulation, the outage streak. Behavior
is otherwise unchanged; this is the seam, not the concurrency.

Six new tests cover it, the first coverage this body has ever had: the
row's task/arm/run/digest bindings, each of the five expected failure
kinds still removing the clone, an unexpected KeyError escaping while
cleanup still runs, cleanup failure overriding the primary outcome, and a
failed per-task snapshot failing every cell closed.
This commit is contained in:
Gergo Magyar 2026-08-01 18:38:45 +00:00
parent 48a03464de
commit b30f530698
3 changed files with 463 additions and 234 deletions

View file

@ -2,12 +2,15 @@
import hashlib
import json
from contextlib import nullcontext
from types import SimpleNamespace
import pytest
from workflow_bench import runner, runner_artifacts, runner_sessions
from workflow_bench.evolution import skill_fingerprint
from workflow_bench.process_control import ManagedProcessError, ManagedProcessResult
from workflow_bench.proposer_sandbox import SandboxError
def _report(**overrides) -> str:
@ -381,3 +384,170 @@ def test_phase_workspace_still_sees_writes_under_a_pre_existing_nested_claude_di
with pytest.raises(ValueError, match="unauthorized workspace path"):
runner_artifacts.enforce_phase_workspace(tmp_path, before, allowed_artifact=artifact)
def _cell_context(tmp_path, **overrides):
"""A TaskCellContext whose per-task inputs are all present and valid."""
snapshot = SimpleNamespace(
digest="asset-digest",
manifest_digest="asset-manifest",
dependency_content_digest="dep-content",
dependency_manifest_digest="dep-manifest",
)
graph = SimpleNamespace(
digest="graph-digest",
manifest_digest="graph-manifest",
materialize=lambda *_a, **_k: None,
)
oracle = SimpleNamespace(
digest="oracle-digest",
command_digest="oracle-command",
manifest_digest="oracle-manifest",
)
fields = {
"task": {"id": "task-a", "class": "demo", "prompt": "do the thing"},
"oracle_snapshot": oracle,
"repo": tmp_path / "repo",
"task_sha": "a" * 40,
"graph_snapshot": graph,
"graph_snapshot_error": None,
"asset_snapshot": snapshot,
"asset_snapshot_error": None,
"args": SimpleNamespace(
claude_bin="claude",
model="pinned-model",
proposer_model=None,
auth_token=None,
),
"out_dir": tmp_path / "out",
"oracle_mask": tmp_path / "mask",
"ce_plugin_snapshot": None,
"trees_dir": tmp_path / "trees",
"bwrap_bin": tmp_path / "bwrap",
"runtime_mounts": (),
"candidate_overlay": None,
"overlay_digest": None,
}
fields.update(overrides)
fields["out_dir"].mkdir(parents=True, exist_ok=True)
return runner.TaskCellContext(**fields)
def _stub_cell_dependencies(monkeypatch, tmp_path, removed):
"""Replace everything a cell shells out to, so only its own logic runs."""
worktree = tmp_path / "clone"
worktree.mkdir()
monkeypatch.setattr(runner, "make_worktree", lambda *_a, **_k: worktree)
monkeypatch.setattr(runner, "sanitize_clone_for_hidden_oracles", lambda *_a, **_k: "b" * 40)
monkeypatch.setattr(runner, "stage_task_assets", lambda *_a, **_k: [])
monkeypatch.setattr(runner, "isolated_gitnexus_registry_mount", lambda *_a, **_k: None)
monkeypatch.setattr(runner, "ce_plugin_mounts_for_arm", lambda *_a, **_k: [])
monkeypatch.setattr(runner, "ce_plugin_dir_for_arm", lambda *_a, **_k: None)
monkeypatch.setattr(runner, "prepare_sandbox", lambda **_k: nullcontext(SimpleNamespace(run=None)))
monkeypatch.setattr(runner, "skill_fingerprint", lambda *_a, **_k: "skill-digest")
monkeypatch.setattr(runner, "require_skill_fingerprint", lambda *_a, **_k: None)
monkeypatch.setattr(runner, "_sandbox_git", lambda *_a, **_k: "c" * 40)
monkeypatch.setattr(runner, "implementation_diff_digest", lambda *_a, **_k: "")
monkeypatch.setattr(runner, "_prepare_untracked_for_diff", lambda *_a, **_k: None)
monkeypatch.setattr(runner, "diff_churn", lambda *_a, **_k: {})
monkeypatch.setattr(runner, "enforce_work_evidence", lambda *_a, **_k: None)
monkeypatch.setattr(runner, "capture_patch", lambda *_a, **_k: b"diff")
monkeypatch.setattr(runner, "run_arm", lambda *_a, **_k: {"resolved": True, "ok": True, "error_kind": None})
monkeypatch.setattr(runner, "remove_clone", lambda path: removed.append(path))
return worktree
def test_run_cell_returns_a_row_bound_to_its_task_and_snapshots(monkeypatch, tmp_path):
removed: list = []
_stub_cell_dependencies(monkeypatch, tmp_path, removed)
record = runner.run_cell(_cell_context(tmp_path), 2, "workflow")
assert record["resolved"] is True
assert record["error_kind"] is None
# The row has to carry its own coordinates: once cells stop running in a
# predictable order, position in results.jsonl identifies nothing.
assert record["task"] == "task-a"
assert record["arm"] == "workflow"
assert record["run"] == 2
assert record["task_asset_snapshot_digest"] == "asset-digest"
assert record["sanitized_graph_snapshot_digest"] == "graph-digest"
assert record["oracle_digest"] == "oracle-digest"
assert removed == [tmp_path / "clone"]
@pytest.mark.parametrize(
"failure",
[
ManagedProcessError(
["setup"],
ManagedProcessResult(
state="timeout",
returncode=-15,
stdout_tail="",
stderr_tail="",
duration_s=1.0,
),
),
SandboxError("sandbox refused"),
OSError("disk went away"),
RuntimeError("overlay drifted"),
ValueError("bad binding"),
],
ids=["managed-process", "sandbox", "os", "runtime", "value"],
)
def test_run_cell_records_an_expected_failure_and_still_removes_its_clone(monkeypatch, tmp_path, failure):
removed: list = []
_stub_cell_dependencies(monkeypatch, tmp_path, removed)
def explode(*_args, **_kwargs):
raise failure
monkeypatch.setattr(runner, "run_arm", explode)
record = runner.run_cell(_cell_context(tmp_path), 0, "workflow")
assert record["error_kind"] == "infra-error"
assert record["resolved"] is False
# A cell owns its clone for its whole lifetime; the sweep has no other
# chance to reclaim it, so the finally must survive every expected failure.
assert removed == [tmp_path / "clone"]
def test_run_cell_lets_an_unexpected_failure_escape_rather_than_scoring_it(monkeypatch, tmp_path):
removed: list = []
_stub_cell_dependencies(monkeypatch, tmp_path, removed)
def explode(*_args, **_kwargs):
raise KeyError("harness bug")
monkeypatch.setattr(runner, "run_arm", explode)
# A harness bug recorded as an ordinary infra-error would be averaged into
# the evidence and counted toward the outage breaker. It must crash instead.
with pytest.raises(KeyError):
runner.run_cell(_cell_context(tmp_path), 0, "workflow")
assert removed == [tmp_path / "clone"]
def test_run_cell_reports_a_cleanup_failure_over_its_primary_outcome(monkeypatch, tmp_path):
_stub_cell_dependencies(monkeypatch, tmp_path, [])
def refuse(_path):
raise OSError("clone is busy")
monkeypatch.setattr(runner, "remove_clone", refuse)
record = runner.run_cell(_cell_context(tmp_path), 1, "workflow")
assert record["error_kind"] == "cleanup-failure"
assert record["resolved"] is False
assert "primary=None" in record["error_detail"]
assert "clone is busy" in record["error_detail"]
def test_run_cell_fails_closed_when_a_per_task_snapshot_never_materialized(tmp_path):
# The snapshots are prepared once per task, before any cell. If that failed,
# every cell of the task has to record it rather than run against nothing.
context = _cell_context(tmp_path, asset_snapshot=None, asset_snapshot_error=OSError("no assets"))
record = runner.run_cell(context, 0, "workflow")
assert record["error_kind"] == "infra-error"
assert "no assets" in str(record["error_detail"])

View file

@ -615,9 +615,7 @@ def evaluate_candidate(
# when it is fed back to the proposer (evolve.summarize_gate), and a
# growing set of unsolvable tasks must not crowd out the reason the
# candidate actually won or lost. The full list ships structurally.
reasons.append(
f"not gated on {len(ungated_tasks)} task(s) neither arm resolved: {', '.join(ungated_tasks)}"
)
reasons.append(f"not gated on {len(ungated_tasks)} task(s) neither arm resolved: {', '.join(ungated_tasks)}")
if not task_rows:
insufficient = True
reasons.append("no paired task results were found")

View file

@ -45,7 +45,7 @@ import statistics
import tempfile
import time
from collections.abc import Mapping
from dataclasses import replace
from dataclasses import dataclass, replace
from datetime import UTC, datetime, timedelta
from pathlib import Path
from typing import Any
@ -140,6 +140,7 @@ from .sanitized_graph import (
)
from .runtime_mounts import (
CE_ARMS,
CePluginSnapshot,
HARNESS_ROOT as HARNESS_ROOT,
PINNED_GITNEXUS_VERSION as PINNED_GITNEXUS_VERSION,
ce_plugin_dir_for_arm,
@ -638,6 +639,264 @@ def systemic_outage_streak(error_kind: str | None, prior_streak: int) -> int:
return prior_streak + 1 if error_kind in SYSTEMIC_ERROR_KINDS else 0
@dataclass(frozen=True)
class TaskCellContext:
"""Everything one benchmark cell needs from its task, prepared once.
A cell is one (run, arm) pair: a private clone, a sandboxed session set, and
the row it produces. Cells of the same task share this context read-only, so
it is what makes them independent of each other every per-cell mutable is
local to ``run_cell``. Holding the fields explicitly, rather than closing
over ``main``'s scope, is what lets a cell run off the main thread without
dragging the whole sweep's state along with it.
``args`` is treated as immutable: ``main`` finishes mutating it during
setup, well before any cell starts. ``argparse.Namespace`` cannot enforce
that, so it is stated here.
"""
task: dict[str, Any]
oracle_snapshot: TaskOracleSnapshot
repo: Path
task_sha: str
graph_snapshot: SanitizedGraphSnapshot | None
graph_snapshot_error: BaseException | None
asset_snapshot: TaskAssetSnapshot | None
asset_snapshot_error: BaseException | None
args: argparse.Namespace
out_dir: Path
oracle_mask: Path
ce_plugin_snapshot: CePluginSnapshot | None
trees_dir: Path
bwrap_bin: Path
runtime_mounts: tuple[ReadOnlyMount, ...]
candidate_overlay: Path | None
overlay_digest: str | None
def run_cell(ctx: TaskCellContext, run_idx: int, arm: str) -> dict[str, Any]:
"""Run one (run, arm) cell end to end and return its result row.
Owns its clone for the whole call, including teardown: the ``finally``
removes the worktree whatever happens, and an exception outside the five
expected kinds is deliberately left to propagate a harness bug must not be
recorded as an ordinary infra-error and averaged into the evidence.
"""
args = ctx.args
task = ctx.task
worktree: Path | None = None
record: dict[str, Any] | None = None
cleanup_error: OSError | None = None
try:
if ctx.asset_snapshot_error is not None:
raise RuntimeError(f"task asset snapshot preparation failed: {ctx.asset_snapshot_error}")
if ctx.graph_snapshot_error is not None:
raise RuntimeError(f"sanitized graph snapshot preparation failed: {ctx.graph_snapshot_error}")
if ctx.graph_snapshot is None:
raise RuntimeError("sanitized graph snapshot is unavailable")
if ctx.asset_snapshot is None:
raise RuntimeError("task asset snapshot is unavailable")
worktree = make_worktree(ctx.repo, ctx.task_sha, ctx.trees_dir)
sanitized_head = sanitize_clone_for_hidden_oracles(worktree)
ctx.graph_snapshot.materialize(worktree, sanitized_head=sanitized_head)
dependency_mounts = stage_task_assets(
task,
repo=ctx.repo,
clone=worktree,
snapshot=ctx.asset_snapshot,
)
registry_mount = isolated_gitnexus_registry_mount(worktree, ctx.trees_dir)
hidden_harness = worktree / "eval" / "workflow_bench"
oracle_visibility_mounts: list[ReadOnlyMount] = []
if hidden_harness.exists() or hidden_harness.is_symlink():
hidden_metadata = hidden_harness.lstat()
if stat.S_ISLNK(hidden_metadata.st_mode) or not stat.S_ISDIR(hidden_metadata.st_mode):
raise SandboxError("benchmark harness path must be a real directory before it can be hidden")
oracle_visibility_mounts.append(
ReadOnlyMount(
source=ctx.oracle_mask,
target=f"{SANDBOX_WORKSPACE}/eval/workflow_bench",
)
)
execution_arm = CANDIDATE_ARMS.get(arm, arm)
ce_mounts = ce_plugin_mounts_for_arm(execution_arm, ctx.ce_plugin_snapshot)
with prepare_sandbox(
clone=worktree,
claude_bin=args.claude_bin,
bwrap_bin=ctx.bwrap_bin,
read_only_mounts=[
*dependency_mounts,
*ctx.runtime_mounts,
registry_mount,
*ce_mounts,
*oracle_visibility_mounts,
],
preflight=False,
) as sandbox:
# Capture the BASE (pre-overlay) skill digest — identical
# for the incumbent and candidate arms — then run the
# task's untrusted setup against those base skills. The
# candidate overlay is applied only afterwards, so setup
# can never observe candidate prose and both arms share
# byte-identical pre-overlay state.
base_skill_digest = skill_fingerprint(worktree, execution_arm)
if task.get("setup"):
setup_command = ["/bin/sh", "-lc", str(task["setup"])]
setup = sandbox.run(
setup_command,
timeout=600,
env=build_sandbox_environment(),
)
if not setup.ok:
raise ManagedProcessError(setup_command, setup)
# Tamper-evidence: setup must not have rewritten the base
# skills, verified before any candidate overlay lands.
require_skill_fingerprint(
worktree,
execution_arm,
base_skill_digest,
phase="task setup",
)
if arm in CANDIDATE_ARMS:
assert ctx.candidate_overlay is not None
applied_digest = apply_candidate_overlay(
ctx.candidate_overlay,
worktree,
sandbox=sandbox,
)
if applied_digest != ctx.overlay_digest:
raise RuntimeError("candidate overlay changed during the benchmark run")
# The digest the model must preserve during its run is the
# post-overlay skill surface (candidate skills for
# candidate arms; unchanged base skills otherwise).
expected_skill_digest = skill_fingerprint(worktree, execution_arm)
orig_sha = _sandbox_git(sandbox, ["rev-parse", "HEAD"]).strip()
if not re.fullmatch(r"[0-9a-fA-F]{40,64}", orig_sha):
raise RuntimeError("sandboxed candidate setup did not produce an immutable commit")
before_work_digest = (
implementation_diff_digest(sandbox, orig_sha) if execution_arm in IMPLEMENTATION_ARMS else ""
)
record = run_arm(
execution_arm,
task,
worktree,
args,
sandbox=sandbox,
transcript_output_dir=ctx.out_dir,
transcript_output_prefix=f"{task['id']}-{arm}-run{run_idx}",
expected_skill_digest=expected_skill_digest,
enforce_phase_boundary=True,
ce_plugin_dir=ce_plugin_dir_for_arm(execution_arm, ctx.ce_plugin_snapshot),
oracle_snapshot=ctx.oracle_snapshot,
)
_prepare_untracked_for_diff(sandbox)
after_work_digest = (
implementation_diff_digest(
sandbox,
orig_sha,
prepare_untracked=False,
)
if execution_arm in IMPLEMENTATION_ARMS
else ""
)
record.update(
diff_churn(
sandbox,
orig_sha,
prepare_untracked=False,
)
)
enforce_work_evidence(
record,
arm=execution_arm,
before_digest=before_work_digest,
after_digest=after_work_digest,
)
patch_bytes = capture_patch(sandbox, worktree, orig_sha)
record["arm"] = arm
record.update(
{
"model": args.model,
"benchmark_model": args.model,
"proposer_model": args.proposer_model,
"task_ref": task.get("ref", "HEAD"),
"task_base_sha": ctx.task_sha,
"sanitized_task_sha": sanitized_head,
"variant_head_sha": orig_sha,
"task_prompt_digest": hashlib.sha256(task["prompt"].encode()).hexdigest(),
"skill_digest": expected_skill_digest,
"candidate_overlay_digest": (ctx.overlay_digest if arm in CANDIDATE_ARMS else None),
"recorded_at": datetime.now(UTC).isoformat(),
}
)
# Final working-tree patch — the clone is destroyed, so
# this is the only artifact for diagnosing verify fails.
patch_path = ctx.out_dir / f"{task['id']}-{arm}-run{run_idx}.patch"
patch_path.write_bytes(patch_bytes)
except (
ManagedProcessError,
SandboxError,
OSError,
RuntimeError,
ValueError,
) as exc:
# One hung session or failed setup must not abort the
# sweep — record the run as infra-error and move on so
# report.md/promotion.json still get written.
record = infra_error_record(exc)
record["arm"] = arm
print(f"[{task['id']}][{arm}][run {run_idx}] infra-error: {exc}")
finally:
if worktree is not None and worktree.exists():
try:
remove_clone(worktree)
except OSError as exc:
cleanup_error = exc
assert record is not None
if cleanup_error is not None:
primary_kind = record.get("error_kind")
primary_detail = record.get("error_detail")
record["resolved"] = False
record["ok"] = False
record["error_kind"] = "cleanup-failure"
record["error_detail"] = (
f"primary={primary_kind}: {primary_detail}; cleanup: {type(cleanup_error).__name__}: {cleanup_error}"
)[:2000]
record.update(
{
"task": task["id"],
"class": task.get("class", ""),
"run": run_idx,
"task_asset_snapshot_digest": (ctx.asset_snapshot.digest if ctx.asset_snapshot is not None else None),
"task_asset_manifest_digest": (
ctx.asset_snapshot.manifest_digest if ctx.asset_snapshot is not None else None
),
"sandbox_dependency_content_digest": (
ctx.asset_snapshot.dependency_content_digest if ctx.asset_snapshot is not None else None
),
"sandbox_dependency_manifest_digest": (
ctx.asset_snapshot.dependency_manifest_digest if ctx.asset_snapshot is not None else None
),
"sanitized_graph_snapshot_digest": (ctx.graph_snapshot.digest if ctx.graph_snapshot is not None else None),
"sanitized_graph_manifest_digest": (
ctx.graph_snapshot.manifest_digest if ctx.graph_snapshot is not None else None
),
"oracle_digest": ctx.oracle_snapshot.digest,
"oracle_command_digest": ctx.oracle_snapshot.command_digest,
"oracle_manifest_digest": ctx.oracle_snapshot.manifest_digest,
"ce_plugin_version": (
ctx.ce_plugin_snapshot.version if arm in CE_ARMS and ctx.ce_plugin_snapshot is not None else None
),
"ce_plugin_manifest_digest": (
ctx.ce_plugin_snapshot.manifest_digest
if arm in CE_ARMS and ctx.ce_plugin_snapshot is not None
else None
),
}
)
return record
def infra_error_record(exc: BaseException) -> dict[str, Any]:
"""Row for a run the harness itself killed (timeout, setup failure)."""
if isinstance(exc, ManagedProcessError):
@ -1059,6 +1318,37 @@ def main() -> None:
except (ManagedProcessError, OSError, SandboxError, RuntimeError, ValueError) as exc:
graph_snapshot_error = exc
graph_snapshot_errors[graph_key] = exc
try:
# Prepared here, once, rather than lazily inside the first cell:
# TaskAssetCache is a plain dict, so a lazy build would be a
# read-then-write race the moment cells stop running serially.
asset_snapshot = task_asset_cache.prepare(
task,
repo=repo,
resolved_sha=task_sha,
expected_dependency_binding=task_binding,
)
except (OSError, SandboxError, ValueError) as exc:
asset_snapshot_error = exc
cell_context = TaskCellContext(
task=task,
oracle_snapshot=oracle_snapshot,
repo=repo,
task_sha=task_sha,
graph_snapshot=graph_snapshot,
graph_snapshot_error=graph_snapshot_error,
asset_snapshot=asset_snapshot,
asset_snapshot_error=asset_snapshot_error,
args=args,
out_dir=out_dir,
oracle_mask=oracle_mask,
ce_plugin_snapshot=ce_plugin_snapshot,
trees_dir=Path(trees),
bwrap_bin=bwrap_bin,
runtime_mounts=runtime_mounts,
candidate_overlay=candidate_overlay,
overlay_digest=overlay_digest,
)
per_arm: dict[str, list[dict[str, Any]]] = {a: [] for a in args.arms}
for run_idx in range(args.runs):
if outage_tripped:
@ -1071,236 +1361,7 @@ def main() -> None:
f"[{task['id']}][{arm}][run {run_idx}] starting "
f"({started_cells}/{total_cells}, {(time.monotonic() - sweep_started) / 60:.0f}m elapsed)"
)
worktree: Path | None = None
record: dict[str, Any] | None = None
cleanup_error: OSError | None = None
try:
if asset_snapshot_error is not None:
raise RuntimeError(f"task asset snapshot preparation failed: {asset_snapshot_error}")
if graph_snapshot_error is not None:
raise RuntimeError(f"sanitized graph snapshot preparation failed: {graph_snapshot_error}")
if graph_snapshot is None:
raise RuntimeError("sanitized graph snapshot is unavailable")
if asset_snapshot is None:
try:
asset_snapshot = task_asset_cache.prepare(
task,
repo=repo,
resolved_sha=task_sha,
expected_dependency_binding=task_binding,
)
except (OSError, SandboxError, ValueError) as exc:
asset_snapshot_error = exc
raise
worktree = make_worktree(repo, task_sha, Path(trees))
sanitized_head = sanitize_clone_for_hidden_oracles(worktree)
graph_snapshot.materialize(worktree, sanitized_head=sanitized_head)
dependency_mounts = stage_task_assets(
task,
repo=repo,
clone=worktree,
snapshot=asset_snapshot,
)
registry_mount = isolated_gitnexus_registry_mount(worktree, Path(trees))
hidden_harness = worktree / "eval" / "workflow_bench"
oracle_visibility_mounts: list[ReadOnlyMount] = []
if hidden_harness.exists() or hidden_harness.is_symlink():
hidden_metadata = hidden_harness.lstat()
if stat.S_ISLNK(hidden_metadata.st_mode) or not stat.S_ISDIR(hidden_metadata.st_mode):
raise SandboxError(
"benchmark harness path must be a real directory before it can be hidden"
)
oracle_visibility_mounts.append(
ReadOnlyMount(
source=oracle_mask,
target=f"{SANDBOX_WORKSPACE}/eval/workflow_bench",
)
)
execution_arm = CANDIDATE_ARMS.get(arm, arm)
ce_mounts = ce_plugin_mounts_for_arm(execution_arm, ce_plugin_snapshot)
with prepare_sandbox(
clone=worktree,
claude_bin=args.claude_bin,
bwrap_bin=bwrap_bin,
read_only_mounts=[
*dependency_mounts,
*runtime_mounts,
registry_mount,
*ce_mounts,
*oracle_visibility_mounts,
],
preflight=False,
) as sandbox:
# Capture the BASE (pre-overlay) skill digest — identical
# for the incumbent and candidate arms — then run the
# task's untrusted setup against those base skills. The
# candidate overlay is applied only afterwards, so setup
# can never observe candidate prose and both arms share
# byte-identical pre-overlay state.
base_skill_digest = skill_fingerprint(worktree, execution_arm)
if task.get("setup"):
setup_command = ["/bin/sh", "-lc", str(task["setup"])]
setup = sandbox.run(
setup_command,
timeout=600,
env=build_sandbox_environment(),
)
if not setup.ok:
raise ManagedProcessError(setup_command, setup)
# Tamper-evidence: setup must not have rewritten the base
# skills, verified before any candidate overlay lands.
require_skill_fingerprint(
worktree,
execution_arm,
base_skill_digest,
phase="task setup",
)
if arm in CANDIDATE_ARMS:
assert candidate_overlay is not None
applied_digest = apply_candidate_overlay(
candidate_overlay,
worktree,
sandbox=sandbox,
)
if applied_digest != overlay_digest:
raise RuntimeError("candidate overlay changed during the benchmark run")
# The digest the model must preserve during its run is the
# post-overlay skill surface (candidate skills for
# candidate arms; unchanged base skills otherwise).
expected_skill_digest = skill_fingerprint(worktree, execution_arm)
orig_sha = _sandbox_git(sandbox, ["rev-parse", "HEAD"]).strip()
if not re.fullmatch(r"[0-9a-fA-F]{40,64}", orig_sha):
raise RuntimeError("sandboxed candidate setup did not produce an immutable commit")
before_work_digest = (
implementation_diff_digest(sandbox, orig_sha)
if execution_arm in IMPLEMENTATION_ARMS
else ""
)
record = run_arm(
execution_arm,
task,
worktree,
args,
sandbox=sandbox,
transcript_output_dir=out_dir,
transcript_output_prefix=f"{task['id']}-{arm}-run{run_idx}",
expected_skill_digest=expected_skill_digest,
enforce_phase_boundary=True,
ce_plugin_dir=ce_plugin_dir_for_arm(execution_arm, ce_plugin_snapshot),
oracle_snapshot=oracle_snapshot,
)
_prepare_untracked_for_diff(sandbox)
after_work_digest = (
implementation_diff_digest(
sandbox,
orig_sha,
prepare_untracked=False,
)
if execution_arm in IMPLEMENTATION_ARMS
else ""
)
record.update(
diff_churn(
sandbox,
orig_sha,
prepare_untracked=False,
)
)
enforce_work_evidence(
record,
arm=execution_arm,
before_digest=before_work_digest,
after_digest=after_work_digest,
)
patch_bytes = capture_patch(sandbox, worktree, orig_sha)
record["arm"] = arm
record.update(
{
"model": args.model,
"benchmark_model": args.model,
"proposer_model": args.proposer_model,
"task_ref": task.get("ref", "HEAD"),
"task_base_sha": task_sha,
"sanitized_task_sha": sanitized_head,
"variant_head_sha": orig_sha,
"task_prompt_digest": hashlib.sha256(task["prompt"].encode()).hexdigest(),
"skill_digest": expected_skill_digest,
"candidate_overlay_digest": (overlay_digest if arm in CANDIDATE_ARMS else None),
"recorded_at": datetime.now(UTC).isoformat(),
}
)
# Final working-tree patch — the clone is destroyed, so
# this is the only artifact for diagnosing verify fails.
patch_path = out_dir / f"{task['id']}-{arm}-run{run_idx}.patch"
patch_path.write_bytes(patch_bytes)
except (
ManagedProcessError,
SandboxError,
OSError,
RuntimeError,
ValueError,
) as exc:
# One hung session or failed setup must not abort the
# sweep — record the run as infra-error and move on so
# report.md/promotion.json still get written.
record = infra_error_record(exc)
record["arm"] = arm
print(f"[{task['id']}][{arm}][run {run_idx}] infra-error: {exc}")
finally:
if worktree is not None and worktree.exists():
try:
remove_clone(worktree)
except OSError as exc:
cleanup_error = exc
assert record is not None
if cleanup_error is not None:
primary_kind = record.get("error_kind")
primary_detail = record.get("error_detail")
record["resolved"] = False
record["ok"] = False
record["error_kind"] = "cleanup-failure"
record["error_detail"] = (
f"primary={primary_kind}: {primary_detail}; cleanup: "
f"{type(cleanup_error).__name__}: {cleanup_error}"
)[:2000]
record.update(
{
"task": task["id"],
"class": task.get("class", ""),
"run": run_idx,
"task_asset_snapshot_digest": (
asset_snapshot.digest if asset_snapshot is not None else None
),
"task_asset_manifest_digest": (
asset_snapshot.manifest_digest if asset_snapshot is not None else None
),
"sandbox_dependency_content_digest": (
asset_snapshot.dependency_content_digest if asset_snapshot is not None else None
),
"sandbox_dependency_manifest_digest": (
asset_snapshot.dependency_manifest_digest if asset_snapshot is not None else None
),
"sanitized_graph_snapshot_digest": (
graph_snapshot.digest if graph_snapshot is not None else None
),
"sanitized_graph_manifest_digest": (
graph_snapshot.manifest_digest if graph_snapshot is not None else None
),
"oracle_digest": oracle_snapshot.digest,
"oracle_command_digest": oracle_snapshot.command_digest,
"oracle_manifest_digest": oracle_snapshot.manifest_digest,
"ce_plugin_version": (
ce_plugin_snapshot.version
if arm in CE_ARMS and ce_plugin_snapshot is not None
else None
),
"ce_plugin_manifest_digest": (
ce_plugin_snapshot.manifest_digest
if arm in CE_ARMS and ce_plugin_snapshot is not None
else None
),
}
)
record = run_cell(cell_context, run_idx, arm)
per_arm[arm].append(record)
with results_path.open("a") as fh:
# Redact any API token a session-error stderr_tail