diff --git a/.github/workflows/release-evaluation.yml b/.github/workflows/release-evaluation.yml index 1fd987be3..f5571c7eb 100644 --- a/.github/workflows/release-evaluation.yml +++ b/.github/workflows/release-evaluation.yml @@ -295,6 +295,17 @@ jobs: GITNEXUS_REQUIRE_CLAUDE_CANARY: '1' CLAUDE_CANARY_BIN: ${{ runner.temp }}/claude-canary/node_modules/@anthropic-ai/claude-code-linux-x64/claude run: uv run --locked --extra dev python -m pytest tests/test_proposer_sandbox.py -q + - name: Materialize both release graphs before paid sessions + timeout-minutes: 120 + working-directory: eval + run: | + set -euo pipefail + for runtime in stable candidate; do + uv run --locked --extra dev python -m workflow_bench.release_preflight \ + --tasks "$RUNNER_TEMP/release-evaluation/$runtime/tasks.yaml" \ + --gitnexus-root "$GITHUB_WORKSPACE/$runtime" \ + --claude-bin "$RUNNER_TEMP/claude-canary/node_modules/@anthropic-ai/claude-code-linux-x64/claude" + done - name: Run repeated paired agent sessions timeout-minutes: 1140 working-directory: eval diff --git a/eval/tests/test_ec2_workflows.py b/eval/tests/test_ec2_workflows.py index ad80640f3..3066f2c6d 100644 --- a/eval/tests/test_ec2_workflows.py +++ b/eval/tests/test_ec2_workflows.py @@ -119,6 +119,20 @@ def test_both_runtimes_grade_tasks_against_one_pinned_task_dependency_checkout() assert names.count("Resolve pinned task revision") == 1 +def test_both_release_graphs_are_materialized_before_any_paid_sessions(): + steps = workflow("release-evaluation.yml")["jobs"]["evaluate"]["steps"] + preflight = next(step for step in steps if "workflow_bench.release_preflight" in step.get("run", "")) + paid = next(step for step in steps if "workflow_bench.runner" in step.get("run", "")) + assert steps.index(preflight) < steps.index(paid) + assert "for runtime in stable candidate; do" in preflight["run"] + assert "set -euo pipefail" in preflight["run"] + assert '--tasks "$RUNNER_TEMP/release-evaluation/$runtime/tasks.yaml"' in preflight["run"] + assert '--gitnexus-root "$GITHUB_WORKSPACE/$runtime"' in preflight["run"] + assert preflight.get("continue-on-error", False) is False + assert "secrets." not in json.dumps(preflight) + assert "if" not in paid # A failed preflight must skip the paid step. + + @pytest.mark.parametrize( "name,paid", [("gitnexus-skill-evolution.yml", "evolve"), ("release-evaluation.yml", "evaluate")] ) diff --git a/eval/tests/test_release_evaluation_smoke.py b/eval/tests/test_release_evaluation_smoke.py index 943737414..3ec232fe9 100644 --- a/eval/tests/test_release_evaluation_smoke.py +++ b/eval/tests/test_release_evaluation_smoke.py @@ -10,7 +10,7 @@ from pathlib import Path import pytest import yaml -from workflow_bench import oracle_assets, release_report, runner +from workflow_bench import oracle_assets, release_preflight, release_report, runner, runner_tasks from workflow_bench.mock_provider import MockProvider, Reply pytestmark = pytest.mark.skipif( @@ -63,6 +63,7 @@ def test_native_paired_evaluator_produces_valid_report_and_keeps_negative_result capture = partial(oracle_assets.capture_task_oracles, root=oracles) monkeypatch.setattr(release_report, "capture_task_oracles", capture) monkeypatch.setattr(runner, "capture_task_oracles", capture) + monkeypatch.setattr(runner_tasks, "capture_task_oracles", capture) monkeypatch.setattr(release_report, "suite_binding", partial(release_report.suite_binding, suite)) prepared = tmp_path / "prepared" model = "claude-canary-20260718" @@ -90,6 +91,15 @@ def test_native_paired_evaluator_produces_valid_report_and_keeps_negative_result meta = json.loads((prepared / "metadata.json").read_text()) assert meta["runtime_sha"] == sha and meta["harness_sha"] == release_report._git_sha(root) + # Exercise actual graph construction/copying before even starting a model + # provider. This must leave no measurements that could count as evidence. + release_preflight.preflight_release( + yaml.safe_load((prepared / "tasks.yaml").read_text())["tasks"], + gitnexus_root=root, + claude_bin=Path(os.environ["CLAUDE_CANARY_BIN"]), + ) + assert not (prepared / "raw").exists() + initial = [] counts = {"baseline_nomcp": 0, "baseline": 0} diff --git a/eval/tests/test_release_preflight.py b/eval/tests/test_release_preflight.py new file mode 100644 index 000000000..ec41ac259 --- /dev/null +++ b/eval/tests/test_release_preflight.py @@ -0,0 +1,103 @@ +"""Release preflight must exercise actual materialization without paid sessions.""" + +import json +import subprocess +import sys +from functools import partial +from pathlib import Path + +import pytest +import yaml + +from workflow_bench import oracle_assets, release_preflight, runner_tasks, task_assets +from workflow_bench.proposer_sandbox import SandboxError +from workflow_bench.sanitized_graph import SanitizedGraphSnapshot + + +@pytest.fixture +def prepared(monkeypatch, tmp_path): + repo = tmp_path / "repo" + repo.mkdir() + (repo / "source").write_text("source") + for args in ( + ["init", "-q"], ["add", "."], + ["-c", "user.name=Test", "-c", "user.email=test@example.test", "commit", "-qm", "fixture"], + ): + subprocess.run(["git", "-C", str(repo), *args], check=True, capture_output=True) + sha = subprocess.check_output(["git", "-C", str(repo), "rev-parse", "HEAD"], text=True).strip() + task = dict(id="release-probe", repo=str(repo), ref=sha, prompt="test", verify="true", **{"class": "test"}) + hidden = tmp_path / "hidden" + hidden.mkdir() + (hidden / "check.py").write_text("assert True\n") + task["oracle"] = { + "command": 'python "$GITNEXUS_BENCH_ORACLE_ROOT/check.py"', + "files": [{"source": "check.py", "target": "check.py"}], + } + monkeypatch.setattr(runner_tasks, "capture_task_oracles", partial(oracle_assets.capture_task_oracles, root=hidden)) + builds = [] + probes = [] + monkeypatch.setattr(release_preflight, "preflight_bubblewrap", lambda: Path("/fake/bwrap")) + monkeypatch.setattr(release_preflight, "require_claude_sandbox_helpers", lambda: None) + monkeypatch.setattr(release_preflight, "trusted_gitnexus_runtime_mounts", lambda **_: ()) + monkeypatch.setattr(task_assets, "_try_reflink", lambda *_: False) + + # Only graph construction/containment are substituted. Binding resolution, + # capture, graph materialization, staging, and cleanup are production code. + def build(task, *, repo, resolved_sha, parent, cache, **kwargs): + builds.append(resolved_sha) + seed = parent / f"graph-{len(builds)}" + seed.mkdir() + (seed / "index").write_bytes(b"graph") + assets = cache.prepare({"sandbox_copy": ["index"]}, repo=seed, resolved_sha=resolved_sha) + return SanitizedGraphSnapshot(assets=assets, sanitized_head=resolved_sha) + + stage = release_preflight.stage_task_assets + + def checked_stage(task, *, repo, clone, snapshot): + assert (clone / "index").read_bytes() == b"graph" + probes.append(clone) + return stage(task, repo=repo, clone=clone, snapshot=snapshot) + + monkeypatch.setattr(release_preflight, "prepare_sanitized_graph", build) + monkeypatch.setattr(release_preflight, "stage_task_assets", checked_stage) + return task, builds, probes + + +def test_preflight_materializes_every_task_but_builds_shared_graph_once(prepared, capsys, tmp_path): + task, builds, probes = prepared + release_preflight.preflight_release( + [task, {**task, "id": "second-task"}], gitnexus_root=tmp_path, claude_bin=Path("claude"), + ) + assert builds == [task["ref"]] + assert len(probes) == 2 + assert all(not probe.parent.exists() for probe in probes) + assert "graph snapshot 5 bytes" in capsys.readouterr().out + + +def test_copy_failure_aborts_preflight_and_cli_without_evidence(prepared, monkeypatch, tmp_path): + task, builds, probes = prepared + monkeypatch.setattr(task_assets, "MAX_BUFFERED_FALLBACK_BYTES", 4) + with pytest.raises(SandboxError, match="cannot reflink"): + release_preflight.preflight_release([task], gitnexus_root=tmp_path, claude_bin=Path("claude")) + assert probes == [] # Failure must happen in real graph materialization. + tasks_file = tmp_path / "tasks.yaml" + tasks_file.write_text(yaml.safe_dump({"tasks": [task]})) + monkeypatch.setattr(sys, "argv", [ + "preflight", "--tasks", str(tasks_file), "--gitnexus-root", str(tmp_path), "--claude-bin", "claude", + ]) + with pytest.raises(SystemExit) as error: + release_preflight.main() + assert error.value.code == 2 + assert builds == [task["ref"], task["ref"]] + assert not list(tmp_path.rglob("results.jsonl")) + + +def test_preflight_rejects_task_assets_that_replace_the_graph(prepared, tmp_path): + task, _, _ = prepared + index = Path(task["repo"]) / ".gitnexus" + index.mkdir() + (index / "meta.json").write_text(json.dumps({"fake": True})) + with pytest.raises(SandboxError, match="prebuilt graph"): + release_preflight.preflight_release( + [{**task, "sandbox_copy": [".gitnexus/meta.json"]}], gitnexus_root=tmp_path, claude_bin=Path("claude"), + ) diff --git a/eval/tests/test_task_assets.py b/eval/tests/test_task_assets.py index 34affd5f0..d0c636495 100644 --- a/eval/tests/test_task_assets.py +++ b/eval/tests/test_task_assets.py @@ -3,6 +3,7 @@ from __future__ import annotations import os +import hashlib import stat import subprocess from pathlib import Path, PurePosixPath @@ -117,12 +118,13 @@ def test_default_buffered_fallback_budget_covers_a_realistic_large_asset( monkeypatch, tmp_path: Path, ) -> None: - # 20 MiB exceeds the old 16 MiB default but must fit comfortably under - # the current default, proving the real (non-monkeypatched) budget - # constant is sized for a realistic large sandbox_copy asset such as the - # harness's own pre-built graph index, not just tiny fixtures. - payload = os.urandom(20 * 1024 * 1024) - repo, task = _repo_and_task(tmp_path, {"large": payload}) + # Exercise a valid snapshot above 512 MiB. Use a sparse source to avoid a + # huge Python allocation; capture and materialization still copy every byte. + repo, task = _repo_and_task(tmp_path, {"large": b"graph-start"}) + source = repo / "large" + with source.open("r+b") as stream: + stream.seek(513 * 1024 * 1024) + stream.write(b"graph-end") clone = tmp_path / "clone" clone.mkdir() monkeypatch.setattr(task_assets, "_try_reflink", lambda *_args: False) @@ -131,7 +133,9 @@ def test_default_buffered_fallback_budget_covers_a_realistic_large_asset( snapshot = cache.prepare(task, repo=repo, resolved_sha=SHA) snapshot.materialize(clone) - assert (clone / "large").read_bytes() == payload + with source.open("rb") as original, (clone / "large").open("rb") as copied: + assert hashlib.file_digest(copied, "sha256").digest() == hashlib.file_digest(original, "sha256").digest() + assert (clone / "large").stat().st_ino != source.stat().st_ino def test_large_asset_without_reflink_fails_before_publish_and_cleans_staging( diff --git a/eval/workflow_bench/README.md b/eval/workflow_bench/README.md index 2d23e43e5..2659f9f10 100644 --- a/eval/workflow_bench/README.md +++ b/eval/workflow_bench/README.md @@ -124,6 +124,13 @@ Candidate installs run in Bubblewrap with only that checkout (minus `.git`) writable, system tools read-only, and a cleared environment. Locked downloads run without package scripts; every lifecycle script then runs with no network. The trusted harness and host command files are not mounted into the candidate build. +Before any paid sessions, `workflow_bench.release_preflight` builds and +materializes the real sanitized graphs for both runtimes on the runner's temp +filesystem. A copy or containment failure stops the workflow at that point. +Buffered copies use the same 2 GiB hard ceiling as snapshot capture, including +on filesystems without reflinks. This preflight adds an offline graph build per +runtime; the measured runner rebuilds its own graphs to retain its existing +provenance and cache lifecycle. Preflight output is not release evidence. The report records runtime/harness SHAs, task and oracle digests, model/effort, every repetition, solve counts, cost and agent wall time. Failed solutions stay in the denominator. Infrastructure/session failures, missing diff --git a/eval/workflow_bench/release_preflight.py b/eval/workflow_bench/release_preflight.py new file mode 100644 index 000000000..cf228dbe2 --- /dev/null +++ b/eval/workflow_bench/release_preflight.py @@ -0,0 +1,73 @@ +"""Build and materialize release graphs before any paid agent sessions.""" + +from __future__ import annotations + +import argparse +import tempfile +from pathlib import Path +from typing import Any + +import yaml + +from .process_control import ManagedProcessError, cancellation_scope +from .proposer_sandbox import SandboxError, preflight_bubblewrap, require_claude_sandbox_helpers +from .runner_tasks import resolve_task_bindings, select_tasks +from .runtime_mounts import trusted_gitnexus_runtime_mounts +from .sanitized_graph import prepare_sanitized_graph, validate_no_prebuilt_graph_assets +from .task_assets import TaskAssetCache, stage_task_assets + + +def preflight_release(tasks: list[dict[str, Any]], *, gitnexus_root: Path, claude_bin: Path) -> None: + """Exercise the real snapshot/copy path on the same temp filesystem as the runner. + + No provider or session code is involved. Graphs are rebuilt by the paid + runner afterwards, preserving its existing provenance and cache lifecycle. + """ + bwrap_bin = preflight_bubblewrap() + require_claude_sandbox_helpers() + mounts = trusted_gitnexus_runtime_mounts(root=gitnexus_root) + with ( + tempfile.TemporaryDirectory(prefix="wfbench-preflight-") as directory, + TaskAssetCache(Path(directory) / ".task-assets") as cache, + ): + bindings = resolve_task_bindings(tasks, task_asset_cache=cache) + graphs = {} + for task, binding in zip(tasks, bindings, strict=True): + validate_no_prebuilt_graph_assets(task) + repo = Path(binding["repo_identity"]) + sha = binding["resolved_sha"] + key = (str(repo), sha) + if key not in graphs: + graphs[key] = prepare_sanitized_graph( + task, repo=repo, resolved_sha=sha, parent=Path(directory), cache=cache, + claude_bin=claude_bin, bwrap_bin=bwrap_bin, runtime_mounts=mounts, + ) + graph = graphs[key] + assets = cache.prepare(task, repo=repo, resolved_sha=sha, expected_dependency_binding=binding) + print(f"Preflight {task['id']}: graph snapshot {graph.assets.total_bytes} bytes", flush=True) + with tempfile.TemporaryDirectory(prefix="probe-", dir=directory) as probe: + clone = Path(probe) + graph.materialize(clone, sanitized_head=graph.sanitized_head) + stage_task_assets(task, repo=repo, clone=clone, snapshot=assets) + print(f"Preflight {task['id']}: materialization passed", flush=True) + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--tasks", required=True, type=Path) + parser.add_argument("--gitnexus-root", required=True, type=Path) + parser.add_argument("--claude-bin", required=True, type=Path) + args = parser.parse_args() + try: + document = yaml.safe_load(args.tasks.read_text()) + if not isinstance(document, dict) or not isinstance(document.get("tasks"), list): + raise ValueError("task file must contain a tasks list") + tasks, _ = select_tasks(document["tasks"], include_expensive=True) + with cancellation_scope(handle_signals=True): + preflight_release(tasks, gitnexus_root=args.gitnexus_root, claude_bin=args.claude_bin) + except (ManagedProcessError, OSError, SandboxError, RuntimeError, ValueError, yaml.YAMLError) as exc: + parser.error(str(exc)) + + +if __name__ == "__main__": + main() diff --git a/eval/workflow_bench/task_assets.py b/eval/workflow_bench/task_assets.py index 995c63e9a..14caa435c 100644 --- a/eval/workflow_bench/task_assets.py +++ b/eval/workflow_bench/task_assets.py @@ -35,21 +35,17 @@ from .proposer_sandbox import ( real_directory, ) -# The shipped index is roughly 428 MiB. These are containment limits rather -# than expected-size assertions: they admit normal growth while preventing a -# task declaration from turning snapshot preparation into an unbounded walk. +# These are containment limits rather than expected-size assertions: they +# prevent a task declaration from turning snapshot preparation into an +# unbounded walk. MAX_TASK_ASSET_ENTRIES = 100_000 MAX_TASK_ASSET_PATH_BYTES = 4_096 MAX_TASK_ASSET_BYTES = 2 * 1024 * 1024 * 1024 -# The largest known real sandbox_copy asset in this harness is the shipped -# index above (~428 MiB estimated, ~290 MiB measured); budget comfortably -# above that so it can still materialize via buffered copy on a filesystem -# that cannot reflink (ext4 CI runners, 9p-backed dev mounts), while staying -# well below MAX_TASK_ASSET_BYTES so a genuinely oversized or malformed -# declaration still fails closed instead of silently paying for a slow full -# copy. -MAX_BUFFERED_FALLBACK_BYTES = 512 * 1024 * 1024 +# Every accepted snapshot must also be materializable on filesystems without +# reflinks. Use the existing hard capture ceiling, so buffered copies stay +# bounded without imposing a filesystem-dependent acceptance limit. +MAX_BUFFERED_FALLBACK_BYTES = MAX_TASK_ASSET_BYTES COPY_CHUNK_BYTES = 1024 * 1024 # linux/fs.h: #define FICLONE _IOW(0x94, 9, int)