From 22aeeebaf3237f4a05ef28ef73796953f3fe6a65 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Gerg=C5=91=20Magyar?= Date: Sat, 10 Oct 2026 18:24:09 +0300 Subject: [PATCH] fix(eval): materialize release graphs before paid sessions (#3556) Use the existing snapshot capture ceiling for buffered materialization so valid graphs remain usable on filesystems without reflinks. Preflight both release runtimes before paid sessions and exercise the real copy boundary with a 513 MiB regression and the native release smoke. --- .github/workflows/release-evaluation.yml | 11 +++ eval/tests/test_ec2_workflows.py | 14 +++ eval/tests/test_release_evaluation_smoke.py | 12 ++- eval/tests/test_release_preflight.py | 103 ++++++++++++++++++++ eval/tests/test_task_assets.py | 18 ++-- eval/workflow_bench/README.md | 7 ++ eval/workflow_bench/release_preflight.py | 73 ++++++++++++++ eval/workflow_bench/task_assets.py | 18 ++-- 8 files changed, 237 insertions(+), 19 deletions(-) create mode 100644 eval/tests/test_release_preflight.py create mode 100644 eval/workflow_bench/release_preflight.py diff --git a/.github/workflows/release-evaluation.yml b/.github/workflows/release-evaluation.yml index 1fd987be3..f5571c7eb 100644 --- a/.github/workflows/release-evaluation.yml +++ b/.github/workflows/release-evaluation.yml @@ -295,6 +295,17 @@ jobs: GITNEXUS_REQUIRE_CLAUDE_CANARY: '1' CLAUDE_CANARY_BIN: ${{ runner.temp }}/claude-canary/node_modules/@anthropic-ai/claude-code-linux-x64/claude run: uv run --locked --extra dev python -m pytest tests/test_proposer_sandbox.py -q + - name: Materialize both release graphs before paid sessions + timeout-minutes: 120 + working-directory: eval + run: | + set -euo pipefail + for runtime in stable candidate; do + uv run --locked --extra dev python -m workflow_bench.release_preflight \ + --tasks "$RUNNER_TEMP/release-evaluation/$runtime/tasks.yaml" \ + --gitnexus-root "$GITHUB_WORKSPACE/$runtime" \ + --claude-bin "$RUNNER_TEMP/claude-canary/node_modules/@anthropic-ai/claude-code-linux-x64/claude" + done - name: Run repeated paired agent sessions timeout-minutes: 1140 working-directory: eval diff --git a/eval/tests/test_ec2_workflows.py b/eval/tests/test_ec2_workflows.py index ad80640f3..3066f2c6d 100644 --- a/eval/tests/test_ec2_workflows.py +++ b/eval/tests/test_ec2_workflows.py @@ -119,6 +119,20 @@ def test_both_runtimes_grade_tasks_against_one_pinned_task_dependency_checkout() assert names.count("Resolve pinned task revision") == 1 +def test_both_release_graphs_are_materialized_before_any_paid_sessions(): + steps = workflow("release-evaluation.yml")["jobs"]["evaluate"]["steps"] + preflight = next(step for step in steps if "workflow_bench.release_preflight" in step.get("run", "")) + paid = next(step for step in steps if "workflow_bench.runner" in step.get("run", "")) + assert steps.index(preflight) < steps.index(paid) + assert "for runtime in stable candidate; do" in preflight["run"] + assert "set -euo pipefail" in preflight["run"] + assert '--tasks "$RUNNER_TEMP/release-evaluation/$runtime/tasks.yaml"' in preflight["run"] + assert '--gitnexus-root "$GITHUB_WORKSPACE/$runtime"' in preflight["run"] + assert preflight.get("continue-on-error", False) is False + assert "secrets." not in json.dumps(preflight) + assert "if" not in paid # A failed preflight must skip the paid step. + + @pytest.mark.parametrize( "name,paid", [("gitnexus-skill-evolution.yml", "evolve"), ("release-evaluation.yml", "evaluate")] ) diff --git a/eval/tests/test_release_evaluation_smoke.py b/eval/tests/test_release_evaluation_smoke.py index 943737414..3ec232fe9 100644 --- a/eval/tests/test_release_evaluation_smoke.py +++ b/eval/tests/test_release_evaluation_smoke.py @@ -10,7 +10,7 @@ from pathlib import Path import pytest import yaml -from workflow_bench import oracle_assets, release_report, runner +from workflow_bench import oracle_assets, release_preflight, release_report, runner, runner_tasks from workflow_bench.mock_provider import MockProvider, Reply pytestmark = pytest.mark.skipif( @@ -63,6 +63,7 @@ def test_native_paired_evaluator_produces_valid_report_and_keeps_negative_result capture = partial(oracle_assets.capture_task_oracles, root=oracles) monkeypatch.setattr(release_report, "capture_task_oracles", capture) monkeypatch.setattr(runner, "capture_task_oracles", capture) + monkeypatch.setattr(runner_tasks, "capture_task_oracles", capture) monkeypatch.setattr(release_report, "suite_binding", partial(release_report.suite_binding, suite)) prepared = tmp_path / "prepared" model = "claude-canary-20260718" @@ -90,6 +91,15 @@ def test_native_paired_evaluator_produces_valid_report_and_keeps_negative_result meta = json.loads((prepared / "metadata.json").read_text()) assert meta["runtime_sha"] == sha and meta["harness_sha"] == release_report._git_sha(root) + # Exercise actual graph construction/copying before even starting a model + # provider. This must leave no measurements that could count as evidence. + release_preflight.preflight_release( + yaml.safe_load((prepared / "tasks.yaml").read_text())["tasks"], + gitnexus_root=root, + claude_bin=Path(os.environ["CLAUDE_CANARY_BIN"]), + ) + assert not (prepared / "raw").exists() + initial = [] counts = {"baseline_nomcp": 0, "baseline": 0} diff --git a/eval/tests/test_release_preflight.py b/eval/tests/test_release_preflight.py new file mode 100644 index 000000000..ec41ac259 --- /dev/null +++ b/eval/tests/test_release_preflight.py @@ -0,0 +1,103 @@ +"""Release preflight must exercise actual materialization without paid sessions.""" + +import json +import subprocess +import sys +from functools import partial +from pathlib import Path + +import pytest +import yaml + +from workflow_bench import oracle_assets, release_preflight, runner_tasks, task_assets +from workflow_bench.proposer_sandbox import SandboxError +from workflow_bench.sanitized_graph import SanitizedGraphSnapshot + + +@pytest.fixture +def prepared(monkeypatch, tmp_path): + repo = tmp_path / "repo" + repo.mkdir() + (repo / "source").write_text("source") + for args in ( + ["init", "-q"], ["add", "."], + ["-c", "user.name=Test", "-c", "user.email=test@example.test", "commit", "-qm", "fixture"], + ): + subprocess.run(["git", "-C", str(repo), *args], check=True, capture_output=True) + sha = subprocess.check_output(["git", "-C", str(repo), "rev-parse", "HEAD"], text=True).strip() + task = dict(id="release-probe", repo=str(repo), ref=sha, prompt="test", verify="true", **{"class": "test"}) + hidden = tmp_path / "hidden" + hidden.mkdir() + (hidden / "check.py").write_text("assert True\n") + task["oracle"] = { + "command": 'python "$GITNEXUS_BENCH_ORACLE_ROOT/check.py"', + "files": [{"source": "check.py", "target": "check.py"}], + } + monkeypatch.setattr(runner_tasks, "capture_task_oracles", partial(oracle_assets.capture_task_oracles, root=hidden)) + builds = [] + probes = [] + monkeypatch.setattr(release_preflight, "preflight_bubblewrap", lambda: Path("/fake/bwrap")) + monkeypatch.setattr(release_preflight, "require_claude_sandbox_helpers", lambda: None) + monkeypatch.setattr(release_preflight, "trusted_gitnexus_runtime_mounts", lambda **_: ()) + monkeypatch.setattr(task_assets, "_try_reflink", lambda *_: False) + + # Only graph construction/containment are substituted. Binding resolution, + # capture, graph materialization, staging, and cleanup are production code. + def build(task, *, repo, resolved_sha, parent, cache, **kwargs): + builds.append(resolved_sha) + seed = parent / f"graph-{len(builds)}" + seed.mkdir() + (seed / "index").write_bytes(b"graph") + assets = cache.prepare({"sandbox_copy": ["index"]}, repo=seed, resolved_sha=resolved_sha) + return SanitizedGraphSnapshot(assets=assets, sanitized_head=resolved_sha) + + stage = release_preflight.stage_task_assets + + def checked_stage(task, *, repo, clone, snapshot): + assert (clone / "index").read_bytes() == b"graph" + probes.append(clone) + return stage(task, repo=repo, clone=clone, snapshot=snapshot) + + monkeypatch.setattr(release_preflight, "prepare_sanitized_graph", build) + monkeypatch.setattr(release_preflight, "stage_task_assets", checked_stage) + return task, builds, probes + + +def test_preflight_materializes_every_task_but_builds_shared_graph_once(prepared, capsys, tmp_path): + task, builds, probes = prepared + release_preflight.preflight_release( + [task, {**task, "id": "second-task"}], gitnexus_root=tmp_path, claude_bin=Path("claude"), + ) + assert builds == [task["ref"]] + assert len(probes) == 2 + assert all(not probe.parent.exists() for probe in probes) + assert "graph snapshot 5 bytes" in capsys.readouterr().out + + +def test_copy_failure_aborts_preflight_and_cli_without_evidence(prepared, monkeypatch, tmp_path): + task, builds, probes = prepared + monkeypatch.setattr(task_assets, "MAX_BUFFERED_FALLBACK_BYTES", 4) + with pytest.raises(SandboxError, match="cannot reflink"): + release_preflight.preflight_release([task], gitnexus_root=tmp_path, claude_bin=Path("claude")) + assert probes == [] # Failure must happen in real graph materialization. + tasks_file = tmp_path / "tasks.yaml" + tasks_file.write_text(yaml.safe_dump({"tasks": [task]})) + monkeypatch.setattr(sys, "argv", [ + "preflight", "--tasks", str(tasks_file), "--gitnexus-root", str(tmp_path), "--claude-bin", "claude", + ]) + with pytest.raises(SystemExit) as error: + release_preflight.main() + assert error.value.code == 2 + assert builds == [task["ref"], task["ref"]] + assert not list(tmp_path.rglob("results.jsonl")) + + +def test_preflight_rejects_task_assets_that_replace_the_graph(prepared, tmp_path): + task, _, _ = prepared + index = Path(task["repo"]) / ".gitnexus" + index.mkdir() + (index / "meta.json").write_text(json.dumps({"fake": True})) + with pytest.raises(SandboxError, match="prebuilt graph"): + release_preflight.preflight_release( + [{**task, "sandbox_copy": [".gitnexus/meta.json"]}], gitnexus_root=tmp_path, claude_bin=Path("claude"), + ) diff --git a/eval/tests/test_task_assets.py b/eval/tests/test_task_assets.py index 34affd5f0..d0c636495 100644 --- a/eval/tests/test_task_assets.py +++ b/eval/tests/test_task_assets.py @@ -3,6 +3,7 @@ from __future__ import annotations import os +import hashlib import stat import subprocess from pathlib import Path, PurePosixPath @@ -117,12 +118,13 @@ def test_default_buffered_fallback_budget_covers_a_realistic_large_asset( monkeypatch, tmp_path: Path, ) -> None: - # 20 MiB exceeds the old 16 MiB default but must fit comfortably under - # the current default, proving the real (non-monkeypatched) budget - # constant is sized for a realistic large sandbox_copy asset such as the - # harness's own pre-built graph index, not just tiny fixtures. - payload = os.urandom(20 * 1024 * 1024) - repo, task = _repo_and_task(tmp_path, {"large": payload}) + # Exercise a valid snapshot above 512 MiB. Use a sparse source to avoid a + # huge Python allocation; capture and materialization still copy every byte. + repo, task = _repo_and_task(tmp_path, {"large": b"graph-start"}) + source = repo / "large" + with source.open("r+b") as stream: + stream.seek(513 * 1024 * 1024) + stream.write(b"graph-end") clone = tmp_path / "clone" clone.mkdir() monkeypatch.setattr(task_assets, "_try_reflink", lambda *_args: False) @@ -131,7 +133,9 @@ def test_default_buffered_fallback_budget_covers_a_realistic_large_asset( snapshot = cache.prepare(task, repo=repo, resolved_sha=SHA) snapshot.materialize(clone) - assert (clone / "large").read_bytes() == payload + with source.open("rb") as original, (clone / "large").open("rb") as copied: + assert hashlib.file_digest(copied, "sha256").digest() == hashlib.file_digest(original, "sha256").digest() + assert (clone / "large").stat().st_ino != source.stat().st_ino def test_large_asset_without_reflink_fails_before_publish_and_cleans_staging( diff --git a/eval/workflow_bench/README.md b/eval/workflow_bench/README.md index 2d23e43e5..2659f9f10 100644 --- a/eval/workflow_bench/README.md +++ b/eval/workflow_bench/README.md @@ -124,6 +124,13 @@ Candidate installs run in Bubblewrap with only that checkout (minus `.git`) writable, system tools read-only, and a cleared environment. Locked downloads run without package scripts; every lifecycle script then runs with no network. The trusted harness and host command files are not mounted into the candidate build. +Before any paid sessions, `workflow_bench.release_preflight` builds and +materializes the real sanitized graphs for both runtimes on the runner's temp +filesystem. A copy or containment failure stops the workflow at that point. +Buffered copies use the same 2 GiB hard ceiling as snapshot capture, including +on filesystems without reflinks. This preflight adds an offline graph build per +runtime; the measured runner rebuilds its own graphs to retain its existing +provenance and cache lifecycle. Preflight output is not release evidence. The report records runtime/harness SHAs, task and oracle digests, model/effort, every repetition, solve counts, cost and agent wall time. Failed solutions stay in the denominator. Infrastructure/session failures, missing diff --git a/eval/workflow_bench/release_preflight.py b/eval/workflow_bench/release_preflight.py new file mode 100644 index 000000000..cf228dbe2 --- /dev/null +++ b/eval/workflow_bench/release_preflight.py @@ -0,0 +1,73 @@ +"""Build and materialize release graphs before any paid agent sessions.""" + +from __future__ import annotations + +import argparse +import tempfile +from pathlib import Path +from typing import Any + +import yaml + +from .process_control import ManagedProcessError, cancellation_scope +from .proposer_sandbox import SandboxError, preflight_bubblewrap, require_claude_sandbox_helpers +from .runner_tasks import resolve_task_bindings, select_tasks +from .runtime_mounts import trusted_gitnexus_runtime_mounts +from .sanitized_graph import prepare_sanitized_graph, validate_no_prebuilt_graph_assets +from .task_assets import TaskAssetCache, stage_task_assets + + +def preflight_release(tasks: list[dict[str, Any]], *, gitnexus_root: Path, claude_bin: Path) -> None: + """Exercise the real snapshot/copy path on the same temp filesystem as the runner. + + No provider or session code is involved. Graphs are rebuilt by the paid + runner afterwards, preserving its existing provenance and cache lifecycle. + """ + bwrap_bin = preflight_bubblewrap() + require_claude_sandbox_helpers() + mounts = trusted_gitnexus_runtime_mounts(root=gitnexus_root) + with ( + tempfile.TemporaryDirectory(prefix="wfbench-preflight-") as directory, + TaskAssetCache(Path(directory) / ".task-assets") as cache, + ): + bindings = resolve_task_bindings(tasks, task_asset_cache=cache) + graphs = {} + for task, binding in zip(tasks, bindings, strict=True): + validate_no_prebuilt_graph_assets(task) + repo = Path(binding["repo_identity"]) + sha = binding["resolved_sha"] + key = (str(repo), sha) + if key not in graphs: + graphs[key] = prepare_sanitized_graph( + task, repo=repo, resolved_sha=sha, parent=Path(directory), cache=cache, + claude_bin=claude_bin, bwrap_bin=bwrap_bin, runtime_mounts=mounts, + ) + graph = graphs[key] + assets = cache.prepare(task, repo=repo, resolved_sha=sha, expected_dependency_binding=binding) + print(f"Preflight {task['id']}: graph snapshot {graph.assets.total_bytes} bytes", flush=True) + with tempfile.TemporaryDirectory(prefix="probe-", dir=directory) as probe: + clone = Path(probe) + graph.materialize(clone, sanitized_head=graph.sanitized_head) + stage_task_assets(task, repo=repo, clone=clone, snapshot=assets) + print(f"Preflight {task['id']}: materialization passed", flush=True) + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--tasks", required=True, type=Path) + parser.add_argument("--gitnexus-root", required=True, type=Path) + parser.add_argument("--claude-bin", required=True, type=Path) + args = parser.parse_args() + try: + document = yaml.safe_load(args.tasks.read_text()) + if not isinstance(document, dict) or not isinstance(document.get("tasks"), list): + raise ValueError("task file must contain a tasks list") + tasks, _ = select_tasks(document["tasks"], include_expensive=True) + with cancellation_scope(handle_signals=True): + preflight_release(tasks, gitnexus_root=args.gitnexus_root, claude_bin=args.claude_bin) + except (ManagedProcessError, OSError, SandboxError, RuntimeError, ValueError, yaml.YAMLError) as exc: + parser.error(str(exc)) + + +if __name__ == "__main__": + main() diff --git a/eval/workflow_bench/task_assets.py b/eval/workflow_bench/task_assets.py index 995c63e9a..14caa435c 100644 --- a/eval/workflow_bench/task_assets.py +++ b/eval/workflow_bench/task_assets.py @@ -35,21 +35,17 @@ from .proposer_sandbox import ( real_directory, ) -# The shipped index is roughly 428 MiB. These are containment limits rather -# than expected-size assertions: they admit normal growth while preventing a -# task declaration from turning snapshot preparation into an unbounded walk. +# These are containment limits rather than expected-size assertions: they +# prevent a task declaration from turning snapshot preparation into an +# unbounded walk. MAX_TASK_ASSET_ENTRIES = 100_000 MAX_TASK_ASSET_PATH_BYTES = 4_096 MAX_TASK_ASSET_BYTES = 2 * 1024 * 1024 * 1024 -# The largest known real sandbox_copy asset in this harness is the shipped -# index above (~428 MiB estimated, ~290 MiB measured); budget comfortably -# above that so it can still materialize via buffered copy on a filesystem -# that cannot reflink (ext4 CI runners, 9p-backed dev mounts), while staying -# well below MAX_TASK_ASSET_BYTES so a genuinely oversized or malformed -# declaration still fails closed instead of silently paying for a slow full -# copy. -MAX_BUFFERED_FALLBACK_BYTES = 512 * 1024 * 1024 +# Every accepted snapshot must also be materializable on filesystems without +# reflinks. Use the existing hard capture ceiling, so buffered copies stay +# bounded without imposing a filesystem-dependent acceptance limit. +MAX_BUFFERED_FALLBACK_BYTES = MAX_TASK_ASSET_BYTES COPY_CHUNK_BYTES = 1024 * 1024 # linux/fs.h: #define FICLONE _IOW(0x94, 9, int)