fix(eval): materialize release graphs before paid sessions (#3556)

Use the existing snapshot capture ceiling for buffered materialization so
valid graphs remain usable on filesystems without reflinks. Preflight both
release runtimes before paid sessions and exercise the real copy boundary
with a 513 MiB regression and the native release smoke.
This commit is contained in:
Gergő Magyar 2026-10-10 18:24:09 +03:00 • committed by GitHub
parent a532e08ef9
commit 22aeeebaf3
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
8 changed files with 237 additions and 19 deletions

View file

@ -295,6 +295,17 @@ jobs:
GITNEXUS_REQUIRE_CLAUDE_CANARY: '1'
CLAUDE_CANARY_BIN: ${{ runner.temp }}/claude-canary/node_modules/@anthropic-ai/claude-code-linux-x64/claude
run: uv run --locked --extra dev python -m pytest tests/test_proposer_sandbox.py -q
- name: Materialize both release graphs before paid sessions
timeout-minutes: 120
working-directory: eval
run: |
set -euo pipefail
for runtime in stable candidate; do
uv run --locked --extra dev python -m workflow_bench.release_preflight \
--tasks "$RUNNER_TEMP/release-evaluation/$runtime/tasks.yaml" \
--gitnexus-root "$GITHUB_WORKSPACE/$runtime" \
--claude-bin "$RUNNER_TEMP/claude-canary/node_modules/@anthropic-ai/claude-code-linux-x64/claude"
done
- name: Run repeated paired agent sessions
timeout-minutes: 1140
working-directory: eval

View file

@ -119,6 +119,20 @@ def test_both_runtimes_grade_tasks_against_one_pinned_task_dependency_checkout()
assert names.count("Resolve pinned task revision") == 1
def test_both_release_graphs_are_materialized_before_any_paid_sessions():
steps = workflow("release-evaluation.yml")["jobs"]["evaluate"]["steps"]
preflight = next(step for step in steps if "workflow_bench.release_preflight" in step.get("run", ""))
paid = next(step for step in steps if "workflow_bench.runner" in step.get("run", ""))
assert steps.index(preflight) < steps.index(paid)
assert "for runtime in stable candidate; do" in preflight["run"]
assert "set -euo pipefail" in preflight["run"]
assert '--tasks "$RUNNER_TEMP/release-evaluation/$runtime/tasks.yaml"' in preflight["run"]
assert '--gitnexus-root "$GITHUB_WORKSPACE/$runtime"' in preflight["run"]
assert preflight.get("continue-on-error", False) is False
assert "secrets." not in json.dumps(preflight)
assert "if" not in paid # A failed preflight must skip the paid step.
@pytest.mark.parametrize(
"name,paid", [("gitnexus-skill-evolution.yml", "evolve"), ("release-evaluation.yml", "evaluate")]
)

View file

@ -10,7 +10,7 @@ from pathlib import Path
import pytest
import yaml
from workflow_bench import oracle_assets, release_report, runner
from workflow_bench import oracle_assets, release_preflight, release_report, runner, runner_tasks
from workflow_bench.mock_provider import MockProvider, Reply
pytestmark = pytest.mark.skipif(
@ -63,6 +63,7 @@ def test_native_paired_evaluator_produces_valid_report_and_keeps_negative_result
capture = partial(oracle_assets.capture_task_oracles, root=oracles)
monkeypatch.setattr(release_report, "capture_task_oracles", capture)
monkeypatch.setattr(runner, "capture_task_oracles", capture)
monkeypatch.setattr(runner_tasks, "capture_task_oracles", capture)
monkeypatch.setattr(release_report, "suite_binding", partial(release_report.suite_binding, suite))
prepared = tmp_path / "prepared"
model = "claude-canary-20260718"
@ -90,6 +91,15 @@ def test_native_paired_evaluator_produces_valid_report_and_keeps_negative_result
meta = json.loads((prepared / "metadata.json").read_text())
assert meta["runtime_sha"] == sha and meta["harness_sha"] == release_report._git_sha(root)
# Exercise actual graph construction/copying before even starting a model
# provider. This must leave no measurements that could count as evidence.
release_preflight.preflight_release(
yaml.safe_load((prepared / "tasks.yaml").read_text())["tasks"],
gitnexus_root=root,
claude_bin=Path(os.environ["CLAUDE_CANARY_BIN"]),
)
assert not (prepared / "raw").exists()
initial = []
counts = {"baseline_nomcp": 0, "baseline": 0}

View file

@ -0,0 +1,103 @@
"""Release preflight must exercise actual materialization without paid sessions."""
import json
import subprocess
import sys
from functools import partial
from pathlib import Path
import pytest
import yaml
from workflow_bench import oracle_assets, release_preflight, runner_tasks, task_assets
from workflow_bench.proposer_sandbox import SandboxError
from workflow_bench.sanitized_graph import SanitizedGraphSnapshot
@pytest.fixture
def prepared(monkeypatch, tmp_path):
repo = tmp_path / "repo"
repo.mkdir()
(repo / "source").write_text("source")
for args in (
["init", "-q"], ["add", "."],
["-c", "user.name=Test", "-c", "user.email=test@example.test", "commit", "-qm", "fixture"],
):
subprocess.run(["git", "-C", str(repo), *args], check=True, capture_output=True)
sha = subprocess.check_output(["git", "-C", str(repo), "rev-parse", "HEAD"], text=True).strip()
task = dict(id="release-probe", repo=str(repo), ref=sha, prompt="test", verify="true", **{"class": "test"})
hidden = tmp_path / "hidden"
hidden.mkdir()
(hidden / "check.py").write_text("assert True\n")
task["oracle"] = {
"command": 'python "$GITNEXUS_BENCH_ORACLE_ROOT/check.py"',
"files": [{"source": "check.py", "target": "check.py"}],
}
monkeypatch.setattr(runner_tasks, "capture_task_oracles", partial(oracle_assets.capture_task_oracles, root=hidden))
builds = []
probes = []
monkeypatch.setattr(release_preflight, "preflight_bubblewrap", lambda: Path("/fake/bwrap"))
monkeypatch.setattr(release_preflight, "require_claude_sandbox_helpers", lambda: None)
monkeypatch.setattr(release_preflight, "trusted_gitnexus_runtime_mounts", lambda **_: ())
monkeypatch.setattr(task_assets, "_try_reflink", lambda *_: False)
# Only graph construction/containment are substituted. Binding resolution,
# capture, graph materialization, staging, and cleanup are production code.
def build(task, *, repo, resolved_sha, parent, cache, **kwargs):
builds.append(resolved_sha)
seed = parent / f"graph-{len(builds)}"
seed.mkdir()
(seed / "index").write_bytes(b"graph")
assets = cache.prepare({"sandbox_copy": ["index"]}, repo=seed, resolved_sha=resolved_sha)
return SanitizedGraphSnapshot(assets=assets, sanitized_head=resolved_sha)
stage = release_preflight.stage_task_assets
def checked_stage(task, *, repo, clone, snapshot):
assert (clone / "index").read_bytes() == b"graph"
probes.append(clone)
return stage(task, repo=repo, clone=clone, snapshot=snapshot)
monkeypatch.setattr(release_preflight, "prepare_sanitized_graph", build)
monkeypatch.setattr(release_preflight, "stage_task_assets", checked_stage)
return task, builds, probes
def test_preflight_materializes_every_task_but_builds_shared_graph_once(prepared, capsys, tmp_path):
task, builds, probes = prepared
release_preflight.preflight_release(
[task, {**task, "id": "second-task"}], gitnexus_root=tmp_path, claude_bin=Path("claude"),
)
assert builds == [task["ref"]]
assert len(probes) == 2
assert all(not probe.parent.exists() for probe in probes)
assert "graph snapshot 5 bytes" in capsys.readouterr().out
def test_copy_failure_aborts_preflight_and_cli_without_evidence(prepared, monkeypatch, tmp_path):
task, builds, probes = prepared
monkeypatch.setattr(task_assets, "MAX_BUFFERED_FALLBACK_BYTES", 4)
with pytest.raises(SandboxError, match="cannot reflink"):
release_preflight.preflight_release([task], gitnexus_root=tmp_path, claude_bin=Path("claude"))
assert probes == [] # Failure must happen in real graph materialization.
tasks_file = tmp_path / "tasks.yaml"
tasks_file.write_text(yaml.safe_dump({"tasks": [task]}))
monkeypatch.setattr(sys, "argv", [
"preflight", "--tasks", str(tasks_file), "--gitnexus-root", str(tmp_path), "--claude-bin", "claude",
])
with pytest.raises(SystemExit) as error:
release_preflight.main()
assert error.value.code == 2
assert builds == [task["ref"], task["ref"]]
assert not list(tmp_path.rglob("results.jsonl"))
def test_preflight_rejects_task_assets_that_replace_the_graph(prepared, tmp_path):
task, _, _ = prepared
index = Path(task["repo"]) / ".gitnexus"
index.mkdir()
(index / "meta.json").write_text(json.dumps({"fake": True}))
with pytest.raises(SandboxError, match="prebuilt graph"):
release_preflight.preflight_release(
[{**task, "sandbox_copy": [".gitnexus/meta.json"]}], gitnexus_root=tmp_path, claude_bin=Path("claude"),
)

View file

@ -3,6 +3,7 @@
from __future__ import annotations
import os
import hashlib
import stat
import subprocess
from pathlib import Path, PurePosixPath
@ -117,12 +118,13 @@ def test_default_buffered_fallback_budget_covers_a_realistic_large_asset(
monkeypatch,
tmp_path: Path,
) -> None:
# 20 MiB exceeds the old 16 MiB default but must fit comfortably under
# the current default, proving the real (non-monkeypatched) budget
# constant is sized for a realistic large sandbox_copy asset such as the
# harness's own pre-built graph index, not just tiny fixtures.
payload = os.urandom(20 * 1024 * 1024)
repo, task = _repo_and_task(tmp_path, {"large": payload})
# Exercise a valid snapshot above 512 MiB. Use a sparse source to avoid a
# huge Python allocation; capture and materialization still copy every byte.
repo, task = _repo_and_task(tmp_path, {"large": b"graph-start"})
source = repo / "large"
with source.open("r+b") as stream:
stream.seek(513 * 1024 * 1024)
stream.write(b"graph-end")
clone = tmp_path / "clone"
clone.mkdir()
monkeypatch.setattr(task_assets, "_try_reflink", lambda *_args: False)
@ -131,7 +133,9 @@ def test_default_buffered_fallback_budget_covers_a_realistic_large_asset(
snapshot = cache.prepare(task, repo=repo, resolved_sha=SHA)
snapshot.materialize(clone)
assert (clone / "large").read_bytes() == payload
with source.open("rb") as original, (clone / "large").open("rb") as copied:
assert hashlib.file_digest(copied, "sha256").digest() == hashlib.file_digest(original, "sha256").digest()
assert (clone / "large").stat().st_ino != source.stat().st_ino
def test_large_asset_without_reflink_fails_before_publish_and_cleans_staging(

View file

@ -124,6 +124,13 @@ Candidate installs run in Bubblewrap with only that checkout (minus `.git`)
writable, system tools read-only, and a cleared environment. Locked downloads
run without package scripts; every lifecycle script then runs with no network. The trusted harness
and host command files are not mounted into the candidate build.
Before any paid sessions, `workflow_bench.release_preflight` builds and
materializes the real sanitized graphs for both runtimes on the runner's temp
filesystem. A copy or containment failure stops the workflow at that point.
Buffered copies use the same 2 GiB hard ceiling as snapshot capture, including
on filesystems without reflinks. This preflight adds an offline graph build per
runtime; the measured runner rebuilds its own graphs to retain its existing
provenance and cache lifecycle. Preflight output is not release evidence.
The report records runtime/harness SHAs, task and oracle digests, model/effort,
every repetition, solve counts, cost and agent wall time. Failed
solutions stay in the denominator. Infrastructure/session failures, missing

View file

@ -0,0 +1,73 @@
"""Build and materialize release graphs before any paid agent sessions."""
from __future__ import annotations
import argparse
import tempfile
from pathlib import Path
from typing import Any
import yaml
from .process_control import ManagedProcessError, cancellation_scope
from .proposer_sandbox import SandboxError, preflight_bubblewrap, require_claude_sandbox_helpers
from .runner_tasks import resolve_task_bindings, select_tasks
from .runtime_mounts import trusted_gitnexus_runtime_mounts
from .sanitized_graph import prepare_sanitized_graph, validate_no_prebuilt_graph_assets
from .task_assets import TaskAssetCache, stage_task_assets
def preflight_release(tasks: list[dict[str, Any]], *, gitnexus_root: Path, claude_bin: Path) -> None:
"""Exercise the real snapshot/copy path on the same temp filesystem as the runner.
No provider or session code is involved. Graphs are rebuilt by the paid
runner afterwards, preserving its existing provenance and cache lifecycle.
"""
bwrap_bin = preflight_bubblewrap()
require_claude_sandbox_helpers()
mounts = trusted_gitnexus_runtime_mounts(root=gitnexus_root)
with (
tempfile.TemporaryDirectory(prefix="wfbench-preflight-") as directory,
TaskAssetCache(Path(directory) / ".task-assets") as cache,
):
bindings = resolve_task_bindings(tasks, task_asset_cache=cache)
graphs = {}
for task, binding in zip(tasks, bindings, strict=True):
validate_no_prebuilt_graph_assets(task)
repo = Path(binding["repo_identity"])
sha = binding["resolved_sha"]
key = (str(repo), sha)
if key not in graphs:
graphs[key] = prepare_sanitized_graph(
task, repo=repo, resolved_sha=sha, parent=Path(directory), cache=cache,
claude_bin=claude_bin, bwrap_bin=bwrap_bin, runtime_mounts=mounts,
)
graph = graphs[key]
assets = cache.prepare(task, repo=repo, resolved_sha=sha, expected_dependency_binding=binding)
print(f"Preflight {task['id']}: graph snapshot {graph.assets.total_bytes} bytes", flush=True)
with tempfile.TemporaryDirectory(prefix="probe-", dir=directory) as probe:
clone = Path(probe)
graph.materialize(clone, sanitized_head=graph.sanitized_head)
stage_task_assets(task, repo=repo, clone=clone, snapshot=assets)
print(f"Preflight {task['id']}: materialization passed", flush=True)
def main() -> None:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--tasks", required=True, type=Path)
parser.add_argument("--gitnexus-root", required=True, type=Path)
parser.add_argument("--claude-bin", required=True, type=Path)
args = parser.parse_args()
try:
document = yaml.safe_load(args.tasks.read_text())
if not isinstance(document, dict) or not isinstance(document.get("tasks"), list):
raise ValueError("task file must contain a tasks list")
tasks, _ = select_tasks(document["tasks"], include_expensive=True)
with cancellation_scope(handle_signals=True):
preflight_release(tasks, gitnexus_root=args.gitnexus_root, claude_bin=args.claude_bin)
except (ManagedProcessError, OSError, SandboxError, RuntimeError, ValueError, yaml.YAMLError) as exc:
parser.error(str(exc))
if __name__ == "__main__":
main()

View file

@ -35,21 +35,17 @@ from .proposer_sandbox import (
real_directory,
)
# The shipped index is roughly 428 MiB. These are containment limits rather
# than expected-size assertions: they admit normal growth while preventing a
# task declaration from turning snapshot preparation into an unbounded walk.
# These are containment limits rather than expected-size assertions: they
# prevent a task declaration from turning snapshot preparation into an
# unbounded walk.
MAX_TASK_ASSET_ENTRIES = 100_000
MAX_TASK_ASSET_PATH_BYTES = 4_096
MAX_TASK_ASSET_BYTES = 2 * 1024 * 1024 * 1024
# The largest known real sandbox_copy asset in this harness is the shipped
# index above (~428 MiB estimated, ~290 MiB measured); budget comfortably
# above that so it can still materialize via buffered copy on a filesystem
# that cannot reflink (ext4 CI runners, 9p-backed dev mounts), while staying
# well below MAX_TASK_ASSET_BYTES so a genuinely oversized or malformed
# declaration still fails closed instead of silently paying for a slow full
# copy.
MAX_BUFFERED_FALLBACK_BYTES = 512 * 1024 * 1024
# Every accepted snapshot must also be materializable on filesystems without
# reflinks. Use the existing hard capture ceiling, so buffered copies stay
# bounded without imposing a filesystem-dependent acceptance limit.
MAX_BUFFERED_FALLBACK_BYTES = MAX_TASK_ASSET_BYTES
COPY_CHUNK_BYTES = 1024 * 1024
# linux/fs.h: #define FICLONE _IOW(0x94, 9, int)