mirror of
https://github.com/abhigyanpatwari/GitNexus.git
synced 2026-10-11 03:38:07 +00:00
fix(eval): materialize release graphs before paid sessions (#3556)
Use the existing snapshot capture ceiling for buffered materialization so valid graphs remain usable on filesystems without reflinks. Preflight both release runtimes before paid sessions and exercise the real copy boundary with a 513 MiB regression and the native release smoke.
This commit is contained in:
parent
a532e08ef9
commit
22aeeebaf3
8 changed files with 237 additions and 19 deletions
11
.github/workflows/release-evaluation.yml
vendored
11
.github/workflows/release-evaluation.yml
vendored
|
|
@ -295,6 +295,17 @@ jobs:
|
|||
GITNEXUS_REQUIRE_CLAUDE_CANARY: '1'
|
||||
CLAUDE_CANARY_BIN: ${{ runner.temp }}/claude-canary/node_modules/@anthropic-ai/claude-code-linux-x64/claude
|
||||
run: uv run --locked --extra dev python -m pytest tests/test_proposer_sandbox.py -q
|
||||
- name: Materialize both release graphs before paid sessions
|
||||
timeout-minutes: 120
|
||||
working-directory: eval
|
||||
run: |
|
||||
set -euo pipefail
|
||||
for runtime in stable candidate; do
|
||||
uv run --locked --extra dev python -m workflow_bench.release_preflight \
|
||||
--tasks "$RUNNER_TEMP/release-evaluation/$runtime/tasks.yaml" \
|
||||
--gitnexus-root "$GITHUB_WORKSPACE/$runtime" \
|
||||
--claude-bin "$RUNNER_TEMP/claude-canary/node_modules/@anthropic-ai/claude-code-linux-x64/claude"
|
||||
done
|
||||
- name: Run repeated paired agent sessions
|
||||
timeout-minutes: 1140
|
||||
working-directory: eval
|
||||
|
|
|
|||
|
|
@ -119,6 +119,20 @@ def test_both_runtimes_grade_tasks_against_one_pinned_task_dependency_checkout()
|
|||
assert names.count("Resolve pinned task revision") == 1
|
||||
|
||||
|
||||
def test_both_release_graphs_are_materialized_before_any_paid_sessions():
|
||||
steps = workflow("release-evaluation.yml")["jobs"]["evaluate"]["steps"]
|
||||
preflight = next(step for step in steps if "workflow_bench.release_preflight" in step.get("run", ""))
|
||||
paid = next(step for step in steps if "workflow_bench.runner" in step.get("run", ""))
|
||||
assert steps.index(preflight) < steps.index(paid)
|
||||
assert "for runtime in stable candidate; do" in preflight["run"]
|
||||
assert "set -euo pipefail" in preflight["run"]
|
||||
assert '--tasks "$RUNNER_TEMP/release-evaluation/$runtime/tasks.yaml"' in preflight["run"]
|
||||
assert '--gitnexus-root "$GITHUB_WORKSPACE/$runtime"' in preflight["run"]
|
||||
assert preflight.get("continue-on-error", False) is False
|
||||
assert "secrets." not in json.dumps(preflight)
|
||||
assert "if" not in paid # A failed preflight must skip the paid step.
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"name,paid", [("gitnexus-skill-evolution.yml", "evolve"), ("release-evaluation.yml", "evaluate")]
|
||||
)
|
||||
|
|
|
|||
|
|
@ -10,7 +10,7 @@ from pathlib import Path
|
|||
import pytest
|
||||
import yaml
|
||||
|
||||
from workflow_bench import oracle_assets, release_report, runner
|
||||
from workflow_bench import oracle_assets, release_preflight, release_report, runner, runner_tasks
|
||||
from workflow_bench.mock_provider import MockProvider, Reply
|
||||
|
||||
pytestmark = pytest.mark.skipif(
|
||||
|
|
@ -63,6 +63,7 @@ def test_native_paired_evaluator_produces_valid_report_and_keeps_negative_result
|
|||
capture = partial(oracle_assets.capture_task_oracles, root=oracles)
|
||||
monkeypatch.setattr(release_report, "capture_task_oracles", capture)
|
||||
monkeypatch.setattr(runner, "capture_task_oracles", capture)
|
||||
monkeypatch.setattr(runner_tasks, "capture_task_oracles", capture)
|
||||
monkeypatch.setattr(release_report, "suite_binding", partial(release_report.suite_binding, suite))
|
||||
prepared = tmp_path / "prepared"
|
||||
model = "claude-canary-20260718"
|
||||
|
|
@ -90,6 +91,15 @@ def test_native_paired_evaluator_produces_valid_report_and_keeps_negative_result
|
|||
meta = json.loads((prepared / "metadata.json").read_text())
|
||||
assert meta["runtime_sha"] == sha and meta["harness_sha"] == release_report._git_sha(root)
|
||||
|
||||
# Exercise actual graph construction/copying before even starting a model
|
||||
# provider. This must leave no measurements that could count as evidence.
|
||||
release_preflight.preflight_release(
|
||||
yaml.safe_load((prepared / "tasks.yaml").read_text())["tasks"],
|
||||
gitnexus_root=root,
|
||||
claude_bin=Path(os.environ["CLAUDE_CANARY_BIN"]),
|
||||
)
|
||||
assert not (prepared / "raw").exists()
|
||||
|
||||
initial = []
|
||||
counts = {"baseline_nomcp": 0, "baseline": 0}
|
||||
|
||||
|
|
|
|||
103
eval/tests/test_release_preflight.py
Normal file
103
eval/tests/test_release_preflight.py
Normal file
|
|
@ -0,0 +1,103 @@
|
|||
"""Release preflight must exercise actual materialization without paid sessions."""
|
||||
|
||||
import json
|
||||
import subprocess
|
||||
import sys
|
||||
from functools import partial
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
import yaml
|
||||
|
||||
from workflow_bench import oracle_assets, release_preflight, runner_tasks, task_assets
|
||||
from workflow_bench.proposer_sandbox import SandboxError
|
||||
from workflow_bench.sanitized_graph import SanitizedGraphSnapshot
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def prepared(monkeypatch, tmp_path):
|
||||
repo = tmp_path / "repo"
|
||||
repo.mkdir()
|
||||
(repo / "source").write_text("source")
|
||||
for args in (
|
||||
["init", "-q"], ["add", "."],
|
||||
["-c", "user.name=Test", "-c", "user.email=test@example.test", "commit", "-qm", "fixture"],
|
||||
):
|
||||
subprocess.run(["git", "-C", str(repo), *args], check=True, capture_output=True)
|
||||
sha = subprocess.check_output(["git", "-C", str(repo), "rev-parse", "HEAD"], text=True).strip()
|
||||
task = dict(id="release-probe", repo=str(repo), ref=sha, prompt="test", verify="true", **{"class": "test"})
|
||||
hidden = tmp_path / "hidden"
|
||||
hidden.mkdir()
|
||||
(hidden / "check.py").write_text("assert True\n")
|
||||
task["oracle"] = {
|
||||
"command": 'python "$GITNEXUS_BENCH_ORACLE_ROOT/check.py"',
|
||||
"files": [{"source": "check.py", "target": "check.py"}],
|
||||
}
|
||||
monkeypatch.setattr(runner_tasks, "capture_task_oracles", partial(oracle_assets.capture_task_oracles, root=hidden))
|
||||
builds = []
|
||||
probes = []
|
||||
monkeypatch.setattr(release_preflight, "preflight_bubblewrap", lambda: Path("/fake/bwrap"))
|
||||
monkeypatch.setattr(release_preflight, "require_claude_sandbox_helpers", lambda: None)
|
||||
monkeypatch.setattr(release_preflight, "trusted_gitnexus_runtime_mounts", lambda **_: ())
|
||||
monkeypatch.setattr(task_assets, "_try_reflink", lambda *_: False)
|
||||
|
||||
# Only graph construction/containment are substituted. Binding resolution,
|
||||
# capture, graph materialization, staging, and cleanup are production code.
|
||||
def build(task, *, repo, resolved_sha, parent, cache, **kwargs):
|
||||
builds.append(resolved_sha)
|
||||
seed = parent / f"graph-{len(builds)}"
|
||||
seed.mkdir()
|
||||
(seed / "index").write_bytes(b"graph")
|
||||
assets = cache.prepare({"sandbox_copy": ["index"]}, repo=seed, resolved_sha=resolved_sha)
|
||||
return SanitizedGraphSnapshot(assets=assets, sanitized_head=resolved_sha)
|
||||
|
||||
stage = release_preflight.stage_task_assets
|
||||
|
||||
def checked_stage(task, *, repo, clone, snapshot):
|
||||
assert (clone / "index").read_bytes() == b"graph"
|
||||
probes.append(clone)
|
||||
return stage(task, repo=repo, clone=clone, snapshot=snapshot)
|
||||
|
||||
monkeypatch.setattr(release_preflight, "prepare_sanitized_graph", build)
|
||||
monkeypatch.setattr(release_preflight, "stage_task_assets", checked_stage)
|
||||
return task, builds, probes
|
||||
|
||||
|
||||
def test_preflight_materializes_every_task_but_builds_shared_graph_once(prepared, capsys, tmp_path):
|
||||
task, builds, probes = prepared
|
||||
release_preflight.preflight_release(
|
||||
[task, {**task, "id": "second-task"}], gitnexus_root=tmp_path, claude_bin=Path("claude"),
|
||||
)
|
||||
assert builds == [task["ref"]]
|
||||
assert len(probes) == 2
|
||||
assert all(not probe.parent.exists() for probe in probes)
|
||||
assert "graph snapshot 5 bytes" in capsys.readouterr().out
|
||||
|
||||
|
||||
def test_copy_failure_aborts_preflight_and_cli_without_evidence(prepared, monkeypatch, tmp_path):
|
||||
task, builds, probes = prepared
|
||||
monkeypatch.setattr(task_assets, "MAX_BUFFERED_FALLBACK_BYTES", 4)
|
||||
with pytest.raises(SandboxError, match="cannot reflink"):
|
||||
release_preflight.preflight_release([task], gitnexus_root=tmp_path, claude_bin=Path("claude"))
|
||||
assert probes == [] # Failure must happen in real graph materialization.
|
||||
tasks_file = tmp_path / "tasks.yaml"
|
||||
tasks_file.write_text(yaml.safe_dump({"tasks": [task]}))
|
||||
monkeypatch.setattr(sys, "argv", [
|
||||
"preflight", "--tasks", str(tasks_file), "--gitnexus-root", str(tmp_path), "--claude-bin", "claude",
|
||||
])
|
||||
with pytest.raises(SystemExit) as error:
|
||||
release_preflight.main()
|
||||
assert error.value.code == 2
|
||||
assert builds == [task["ref"], task["ref"]]
|
||||
assert not list(tmp_path.rglob("results.jsonl"))
|
||||
|
||||
|
||||
def test_preflight_rejects_task_assets_that_replace_the_graph(prepared, tmp_path):
|
||||
task, _, _ = prepared
|
||||
index = Path(task["repo"]) / ".gitnexus"
|
||||
index.mkdir()
|
||||
(index / "meta.json").write_text(json.dumps({"fake": True}))
|
||||
with pytest.raises(SandboxError, match="prebuilt graph"):
|
||||
release_preflight.preflight_release(
|
||||
[{**task, "sandbox_copy": [".gitnexus/meta.json"]}], gitnexus_root=tmp_path, claude_bin=Path("claude"),
|
||||
)
|
||||
|
|
@ -3,6 +3,7 @@
|
|||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import hashlib
|
||||
import stat
|
||||
import subprocess
|
||||
from pathlib import Path, PurePosixPath
|
||||
|
|
@ -117,12 +118,13 @@ def test_default_buffered_fallback_budget_covers_a_realistic_large_asset(
|
|||
monkeypatch,
|
||||
tmp_path: Path,
|
||||
) -> None:
|
||||
# 20 MiB exceeds the old 16 MiB default but must fit comfortably under
|
||||
# the current default, proving the real (non-monkeypatched) budget
|
||||
# constant is sized for a realistic large sandbox_copy asset such as the
|
||||
# harness's own pre-built graph index, not just tiny fixtures.
|
||||
payload = os.urandom(20 * 1024 * 1024)
|
||||
repo, task = _repo_and_task(tmp_path, {"large": payload})
|
||||
# Exercise a valid snapshot above 512 MiB. Use a sparse source to avoid a
|
||||
# huge Python allocation; capture and materialization still copy every byte.
|
||||
repo, task = _repo_and_task(tmp_path, {"large": b"graph-start"})
|
||||
source = repo / "large"
|
||||
with source.open("r+b") as stream:
|
||||
stream.seek(513 * 1024 * 1024)
|
||||
stream.write(b"graph-end")
|
||||
clone = tmp_path / "clone"
|
||||
clone.mkdir()
|
||||
monkeypatch.setattr(task_assets, "_try_reflink", lambda *_args: False)
|
||||
|
|
@ -131,7 +133,9 @@ def test_default_buffered_fallback_budget_covers_a_realistic_large_asset(
|
|||
snapshot = cache.prepare(task, repo=repo, resolved_sha=SHA)
|
||||
snapshot.materialize(clone)
|
||||
|
||||
assert (clone / "large").read_bytes() == payload
|
||||
with source.open("rb") as original, (clone / "large").open("rb") as copied:
|
||||
assert hashlib.file_digest(copied, "sha256").digest() == hashlib.file_digest(original, "sha256").digest()
|
||||
assert (clone / "large").stat().st_ino != source.stat().st_ino
|
||||
|
||||
|
||||
def test_large_asset_without_reflink_fails_before_publish_and_cleans_staging(
|
||||
|
|
|
|||
|
|
@ -124,6 +124,13 @@ Candidate installs run in Bubblewrap with only that checkout (minus `.git`)
|
|||
writable, system tools read-only, and a cleared environment. Locked downloads
|
||||
run without package scripts; every lifecycle script then runs with no network. The trusted harness
|
||||
and host command files are not mounted into the candidate build.
|
||||
Before any paid sessions, `workflow_bench.release_preflight` builds and
|
||||
materializes the real sanitized graphs for both runtimes on the runner's temp
|
||||
filesystem. A copy or containment failure stops the workflow at that point.
|
||||
Buffered copies use the same 2 GiB hard ceiling as snapshot capture, including
|
||||
on filesystems without reflinks. This preflight adds an offline graph build per
|
||||
runtime; the measured runner rebuilds its own graphs to retain its existing
|
||||
provenance and cache lifecycle. Preflight output is not release evidence.
|
||||
The report records runtime/harness SHAs, task and oracle digests, model/effort,
|
||||
every repetition, solve counts, cost and agent wall time. Failed
|
||||
solutions stay in the denominator. Infrastructure/session failures, missing
|
||||
|
|
|
|||
73
eval/workflow_bench/release_preflight.py
Normal file
73
eval/workflow_bench/release_preflight.py
Normal file
|
|
@ -0,0 +1,73 @@
|
|||
"""Build and materialize release graphs before any paid agent sessions."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import tempfile
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import yaml
|
||||
|
||||
from .process_control import ManagedProcessError, cancellation_scope
|
||||
from .proposer_sandbox import SandboxError, preflight_bubblewrap, require_claude_sandbox_helpers
|
||||
from .runner_tasks import resolve_task_bindings, select_tasks
|
||||
from .runtime_mounts import trusted_gitnexus_runtime_mounts
|
||||
from .sanitized_graph import prepare_sanitized_graph, validate_no_prebuilt_graph_assets
|
||||
from .task_assets import TaskAssetCache, stage_task_assets
|
||||
|
||||
|
||||
def preflight_release(tasks: list[dict[str, Any]], *, gitnexus_root: Path, claude_bin: Path) -> None:
|
||||
"""Exercise the real snapshot/copy path on the same temp filesystem as the runner.
|
||||
|
||||
No provider or session code is involved. Graphs are rebuilt by the paid
|
||||
runner afterwards, preserving its existing provenance and cache lifecycle.
|
||||
"""
|
||||
bwrap_bin = preflight_bubblewrap()
|
||||
require_claude_sandbox_helpers()
|
||||
mounts = trusted_gitnexus_runtime_mounts(root=gitnexus_root)
|
||||
with (
|
||||
tempfile.TemporaryDirectory(prefix="wfbench-preflight-") as directory,
|
||||
TaskAssetCache(Path(directory) / ".task-assets") as cache,
|
||||
):
|
||||
bindings = resolve_task_bindings(tasks, task_asset_cache=cache)
|
||||
graphs = {}
|
||||
for task, binding in zip(tasks, bindings, strict=True):
|
||||
validate_no_prebuilt_graph_assets(task)
|
||||
repo = Path(binding["repo_identity"])
|
||||
sha = binding["resolved_sha"]
|
||||
key = (str(repo), sha)
|
||||
if key not in graphs:
|
||||
graphs[key] = prepare_sanitized_graph(
|
||||
task, repo=repo, resolved_sha=sha, parent=Path(directory), cache=cache,
|
||||
claude_bin=claude_bin, bwrap_bin=bwrap_bin, runtime_mounts=mounts,
|
||||
)
|
||||
graph = graphs[key]
|
||||
assets = cache.prepare(task, repo=repo, resolved_sha=sha, expected_dependency_binding=binding)
|
||||
print(f"Preflight {task['id']}: graph snapshot {graph.assets.total_bytes} bytes", flush=True)
|
||||
with tempfile.TemporaryDirectory(prefix="probe-", dir=directory) as probe:
|
||||
clone = Path(probe)
|
||||
graph.materialize(clone, sanitized_head=graph.sanitized_head)
|
||||
stage_task_assets(task, repo=repo, clone=clone, snapshot=assets)
|
||||
print(f"Preflight {task['id']}: materialization passed", flush=True)
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--tasks", required=True, type=Path)
|
||||
parser.add_argument("--gitnexus-root", required=True, type=Path)
|
||||
parser.add_argument("--claude-bin", required=True, type=Path)
|
||||
args = parser.parse_args()
|
||||
try:
|
||||
document = yaml.safe_load(args.tasks.read_text())
|
||||
if not isinstance(document, dict) or not isinstance(document.get("tasks"), list):
|
||||
raise ValueError("task file must contain a tasks list")
|
||||
tasks, _ = select_tasks(document["tasks"], include_expensive=True)
|
||||
with cancellation_scope(handle_signals=True):
|
||||
preflight_release(tasks, gitnexus_root=args.gitnexus_root, claude_bin=args.claude_bin)
|
||||
except (ManagedProcessError, OSError, SandboxError, RuntimeError, ValueError, yaml.YAMLError) as exc:
|
||||
parser.error(str(exc))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
|
@ -35,21 +35,17 @@ from .proposer_sandbox import (
|
|||
real_directory,
|
||||
)
|
||||
|
||||
# The shipped index is roughly 428 MiB. These are containment limits rather
|
||||
# than expected-size assertions: they admit normal growth while preventing a
|
||||
# task declaration from turning snapshot preparation into an unbounded walk.
|
||||
# These are containment limits rather than expected-size assertions: they
|
||||
# prevent a task declaration from turning snapshot preparation into an
|
||||
# unbounded walk.
|
||||
MAX_TASK_ASSET_ENTRIES = 100_000
|
||||
MAX_TASK_ASSET_PATH_BYTES = 4_096
|
||||
MAX_TASK_ASSET_BYTES = 2 * 1024 * 1024 * 1024
|
||||
|
||||
# The largest known real sandbox_copy asset in this harness is the shipped
|
||||
# index above (~428 MiB estimated, ~290 MiB measured); budget comfortably
|
||||
# above that so it can still materialize via buffered copy on a filesystem
|
||||
# that cannot reflink (ext4 CI runners, 9p-backed dev mounts), while staying
|
||||
# well below MAX_TASK_ASSET_BYTES so a genuinely oversized or malformed
|
||||
# declaration still fails closed instead of silently paying for a slow full
|
||||
# copy.
|
||||
MAX_BUFFERED_FALLBACK_BYTES = 512 * 1024 * 1024
|
||||
# Every accepted snapshot must also be materializable on filesystems without
|
||||
# reflinks. Use the existing hard capture ceiling, so buffered copies stay
|
||||
# bounded without imposing a filesystem-dependent acceptance limit.
|
||||
MAX_BUFFERED_FALLBACK_BYTES = MAX_TASK_ASSET_BYTES
|
||||
COPY_CHUNK_BYTES = 1024 * 1024
|
||||
|
||||
# linux/fs.h: #define FICLONE _IOW(0x94, 9, int)
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue