fix(eval): make historical review evolution score instead of aborting

Seed the current gitnexus-review skill into older PR checkouts, force-add
historically gitignored skill paths, accept plugin-qualified Skill ids,
and lock host-unsafe workspaces to review-output.json so a generation can
finish and score. Sandbox cleanup restores owner write bits before delete
because a session that copytrees the locked clone otherwise leaves 0555
trees that rmtree cannot remove.

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
Gergo Magyar 2026-09-04 18:37:39 +00:00
parent 9047bf00a5
commit 8491cf4203
22 changed files with 1549 additions and 141 deletions

View file

@ -17,6 +17,7 @@ from workflow_bench.runner_sessions import PARENT_EVENT_STREAM_SOURCE
from workflow_bench.evolve import (
build_parser,
build_proposer_prompt,
executed_benchmark_arms,
generation_timeout_seconds,
load_jsonl,
proposer_evidence_entries,
@ -871,6 +872,25 @@ def test_runner_argv_pairs_each_incumbent_with_its_candidate(tmp_path):
assert json.loads(argv[argv.index("--promotion-target-bases-json") + 1]) == target_bases
def test_runner_argv_inserts_ce_review_for_review_overlay(tmp_path):
args = build_parser().parse_args(
["--tasks", "t.yaml", "--model", "pinned", "--arms", "review"]
)
overlay = tmp_path / "overlay"
skill = overlay / ".claude" / "skills" / "gitnexus-review" / "SKILL.md"
skill.parent.mkdir(parents=True)
skill.write_text("candidate")
argv = runner_argv(
args,
tmp_path / "bench",
overlay,
task_bindings=[{"id": "task"}],
target_base_digests={},
)
arms = argv[argv.index("--arms") + 1 : argv.index("--promotion-metric")]
assert arms == ["ce_review", "review", "candidate_review"]
def test_runner_argv_omits_proposer_for_manual_overlay(tmp_path):
args = build_parser().parse_args(["--tasks", "t.yaml", "--model", "pinned"])
overlay = tmp_path / "overlay"
@ -890,6 +910,26 @@ def test_runner_argv_omits_proposer_for_manual_overlay(tmp_path):
assert "--proposer-model" not in argv
def test_runner_argv_forwards_explicit_unsafe_backend(tmp_path):
args = build_parser().parse_args(
["--tasks", "t.yaml", "--model", "pinned", "--unsafe-no-bwrap"]
)
overlay = tmp_path / "overlay"
skill = overlay / ".claude" / "skills" / "gitnexus-plan" / "SKILL.md"
skill.parent.mkdir(parents=True)
skill.write_text("candidate")
argv = runner_argv(
args,
tmp_path / "bench",
overlay,
task_bindings=[{"id": "task"}],
target_base_digests={},
)
assert "--unsafe-no-bwrap" in argv
def test_runner_argv_keeps_task_commit_pinned_when_ref_moves(tmp_path):
repo = tmp_path / "task-repo"
repo.mkdir()
@ -992,6 +1032,59 @@ def test_generation_timeout_budgets_three_task_workflow_pair():
assert timeout >= 3 * (1 + 3 * paired_arm_cells) * evolve.WORKTREE_PREPARATION_TIMEOUT_SECONDS
def test_executed_benchmark_arms_inserts_review_comparator() -> None:
assert executed_benchmark_arms(["workflow"]) == ["workflow", "candidate_workflow"]
assert executed_benchmark_arms(["review"]) == ["ce_review", "review", "candidate_review"]
def test_generation_timeout_budgets_review_pair_plus_ce_comparator() -> None:
timeout = generation_timeout_seconds(
task_count=6,
runs=1,
session_timeout=3600,
incumbent_arms=["review"],
)
per_task_preparation = (
evolve.TASK_BINDING_GIT_PHASES * evolve.GIT_COMMAND_TIMEOUT_SECONDS
+ 2 * evolve.TASK_SNAPSHOT_TIMEOUT_SECONDS
+ evolve.WORKTREE_PREPARATION_TIMEOUT_SECONDS
+ evolve.GRAPH_SOURCE_PREPARATION_TIMEOUT_SECONDS
+ evolve.GRAPH_BUILD_TIMEOUT_SECONDS
+ 2 * evolve.GRAPH_QUERY_TIMEOUT_SECONDS
+ evolve.CLEANUP_TIMEOUT_SECONDS
)
paired_arm_cells = 3
session_slots = 3
workspace_snapshot_slots = 3
per_task_run = session_slots * (3600 + evolve.SESSION_FINALIZATION_TIMEOUT_SECONDS) + paired_arm_cells * (
evolve.WORKTREE_PREPARATION_TIMEOUT_SECONDS
+ evolve.ARM_ASSET_MATERIALIZATION_PHASES * evolve.TASK_SNAPSHOT_TIMEOUT_SECONDS
+ evolve.SETUP_TIMEOUT_SECONDS
+ 2 * 3600
+ evolve.ARM_EVIDENCE_GIT_PHASES * evolve.GIT_COMMAND_TIMEOUT_SECONDS
+ evolve.CLEANUP_TIMEOUT_SECONDS
)
per_task_run += workspace_snapshot_slots * evolve.TASK_SNAPSHOT_TIMEOUT_SECONDS
per_task_run += evolve.CANDIDATE_OVERLAY_GIT_PHASES * evolve.GIT_COMMAND_TIMEOUT_SECONDS
assert timeout == (
evolve.PROMOTION_BASE_TIMEOUT_SECONDS
+ 6 * (per_task_preparation + per_task_run)
+ evolve.DRIVER_OVERHEAD_SECONDS
)
def test_generation_timeout_rejects_unknown_arm() -> None:
with pytest.raises(ValueError, match="unsupported evolution arm: mystery"):
generation_timeout_seconds(
task_count=1,
runs=1,
session_timeout=60,
incumbent_arms=["mystery"],
)
@pytest.mark.skipif(sys.platform != "linux", reason="Bubblewrap PID namespaces require Linux")
def test_outer_runner_pid_namespace_kills_setsid_descendant(tmp_path):
try:

View file

@ -10,8 +10,11 @@ import pytest
import yaml
from workflow_bench.model_gateway import (
DEFAULT_GATEWAY_READY_TIMEOUT_S,
GATEWAY_READY_TIMEOUT_ENV,
GATEWAY_REQUEST_TIMEOUT_S,
OpenAIGateway,
gateway_ready_timeout_s,
anthropic_api_key_from_environ,
claude_gateway_model_env,
credential_secrets,
@ -141,6 +144,49 @@ def test_openai_gateway_never_leaves_proxy_output_on_an_undrained_pipe(tmp_path:
assert gateway.log_path.stat().st_mode & 0o777 == 0o600
def test_gateway_startup_budget_outlives_a_cold_litellm_import(monkeypatch, tmp_path: Path) -> None:
# Importing LiteLLM takes ~17s on a cold container filesystem and the proxy
# binds its port only afterwards, so a sub-20s budget fails as "connection
# refused" on a proxy that was merely still starting.
monkeypatch.delenv(GATEWAY_READY_TIMEOUT_ENV, raising=False)
assert DEFAULT_GATEWAY_READY_TIMEOUT_S >= 60
assert gateway_ready_timeout_s() == DEFAULT_GATEWAY_READY_TIMEOUT_S
assert (
OpenAIGateway(
openai_api_key="sk-openai-secret",
model_names=["gpt-4.1"],
work_dir=tmp_path / "gw",
).ready_timeout_s
== DEFAULT_GATEWAY_READY_TIMEOUT_S
)
monkeypatch.setenv(GATEWAY_READY_TIMEOUT_ENV, "42.5")
assert gateway_ready_timeout_s() == 42.5
for bad in ("0", "-1", "soon"):
monkeypatch.setenv(GATEWAY_READY_TIMEOUT_ENV, bad)
with pytest.raises(ValueError, match=GATEWAY_READY_TIMEOUT_ENV):
gateway_ready_timeout_s()
def test_gateway_readiness_timeout_reports_the_proxy_log_and_the_override(tmp_path: Path) -> None:
gateway = OpenAIGateway(
openai_api_key="sk-openai-secret",
model_names=["gpt-4.1"],
work_dir=tmp_path / "gw",
ready_timeout_s=0.1,
)
gateway.work_dir.mkdir(parents=True)
gateway.log_path.write_text("ImportError: litellm proxy extras missing")
with pytest.raises(RuntimeError) as excinfo:
gateway._wait_until_ready()
message = str(excinfo.value)
assert "ImportError: litellm proxy extras missing" in message
assert GATEWAY_READY_TIMEOUT_ENV in message
def test_runner_environment_does_not_forward_the_openai_key() -> None:
args = build_parser().parse_args(
[

View file

@ -5,6 +5,7 @@ from __future__ import annotations
import argparse
import hashlib
import os
import shutil
import subprocess
from pathlib import Path
from types import SimpleNamespace
@ -13,7 +14,12 @@ import pytest
from workflow_bench import oracle_assets, runner
from workflow_bench.evolution import evaluate_candidate
from workflow_bench.oracle_assets import capture_task_oracle, staged_task_oracle
from workflow_bench.oracle_assets import (
capture_task_oracle,
review_case_setup_command,
staged_task_oracle,
with_hidden_harness_apply_exclude,
)
def oracle_task(*, command: str = "true", source: str = "oracle.test.ts") -> dict[str, object]:
@ -149,6 +155,60 @@ def test_clone_sanitization_prunes_harness_checkout_and_recoverable_history(tmp_
assert git(clone, "status", "--porcelain=v1", "--untracked-files=all").stdout == ""
def test_hidden_harness_apply_exclude_is_idempotent() -> None:
raw = "git apply eval/workflow_bench/review_cases/pr.patch && rm -rf eval/workflow_bench"
once = with_hidden_harness_apply_exclude(raw)
assert once == review_case_setup_command("pr.patch")
assert with_hidden_harness_apply_exclude(once) == once
assert with_hidden_harness_apply_exclude("true") == "true"
def test_review_setup_skips_sanitized_harness_hunks(tmp_path: Path) -> None:
repo = tmp_path / "repo"
repo.mkdir()
def git(*args: str, check: bool = True) -> subprocess.CompletedProcess[str]:
result = subprocess.run(
["git", "-C", str(repo), *args],
check=False,
capture_output=True,
text=True,
)
if check and result.returncode != 0:
pytest.fail(f"git {' '.join(args)} failed: {result.stderr}")
return result
git("init", "--quiet", "--initial-branch=main")
git("config", "user.name", "Review Setup")
git("config", "user.email", "review-setup.invalid")
(repo / "visible.py").write_text("old\n")
hidden = repo / "eval" / "workflow_bench"
hidden.mkdir(parents=True)
(hidden / "learnings.jsonl").write_text("{}\n")
git("add", "--all")
git("commit", "--quiet", "-m", "base with harness file")
(repo / "visible.py").write_text("new\n")
(hidden / "learnings.jsonl").write_text("{}\nextra\n")
patch = git("diff").stdout
git("checkout", "--", ".")
shutil.rmtree(hidden)
patch_path = hidden / "review_cases" / "case.patch"
patch_path.parent.mkdir(parents=True)
patch_path.write_text(patch)
rejected = git("apply", "--check", str(patch_path.relative_to(repo)), check=False)
assert rejected.returncode != 0
assert "learnings.jsonl" in rejected.stderr
setup = with_hidden_harness_apply_exclude(
"git apply eval/workflow_bench/review_cases/case.patch && rm -rf eval/workflow_bench"
)
applied = subprocess.run(["/bin/sh", "-lc", setup], cwd=repo, check=False, capture_output=True, text=True)
assert applied.returncode == 0, applied.stderr
assert (repo / "visible.py").read_text() == "new\n"
assert not hidden.exists()
def test_clone_sanitization_prunes_remote_history_when_head_never_had_harness(tmp_path: Path) -> None:
source = tmp_path / "source"
source.mkdir()

View file

@ -11,6 +11,7 @@ import sys
import threading
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
from pathlib import Path
from types import SimpleNamespace
import pytest
@ -20,6 +21,7 @@ from workflow_bench.process_control import ManagedProcessResult, run_managed
from workflow_bench.proposer_sandbox import (
MAX_BUNDLE_BYTES,
MAX_EVIDENCE_FILE_BYTES,
SANDBOX_CLAUDE,
SANDBOX_NODE,
SANDBOX_NODE_PREFIX,
SANDBOX_EVIDENCE,
@ -35,8 +37,11 @@ from workflow_bench.proposer_sandbox import (
_runtime_mount_args,
build_claude_settings,
build_sandbox_environment,
_force_rmtree,
host_workspace_write_boundary,
prepare_sandbox,
preflight_bubblewrap,
sandbox_workspace_write_boundary,
stage_evidence_bundle,
stage_task_assets,
)
@ -80,6 +85,128 @@ def test_environment_is_allowlisted_and_shell_children_are_credential_free(monke
assert "defaultMode" not in settings["permissions"]
def test_unsafe_host_session_translates_virtual_paths_and_disables_containment(tmp_path) -> None:
clone = tmp_path / "clone"
evidence = tmp_path / "evidence"
for directory in (clone, evidence):
directory.mkdir()
with prepare_sandbox(
clone=clone,
backend="host-unsafe",
read_only_mounts=(ReadOnlyMount(evidence, "/evidence"),),
) as sandbox:
assert sandbox.command_prefix == []
assert sandbox.require_pid_namespace is False
assert sandbox.host_path("/workspace/review-output.json") == str(clone / "review-output.json")
assert sandbox.host_path("/evidence/selected-rows.json") == str(evidence / "selected-rows.json")
assert sandbox.host_text("read /evidence and write /workspace/out") == (
f"read {evidence} and write {clone}/out"
)
# Sessions spawn the binary directly, so it must be the host executable
# rather than the sandbox-only mount target.
assert sandbox.claude_bin != SANDBOX_CLAUDE
assert Path(sandbox.claude_bin).exists()
assert sandbox.environment()["HOME"] == str(sandbox.home)
assert "CLAUDE_CODE_SUBPROCESS_ENV_SCRUB" not in sandbox.environment()
assert all(
Path(entry).is_dir() for entry in sandbox.environment()["PATH"].split(":")
)
unsafe_settings = json.loads(sandbox.settings_json)
assert unsafe_settings["sandbox"]["enabled"] is False
assert unsafe_settings["sandbox"]["failIfUnavailable"] is False
assert "disableBypassPermissionsMode" not in unsafe_settings["permissions"]
def test_host_workspace_write_boundary_keeps_only_the_review_artifact_writable(tmp_path) -> None:
clone = tmp_path / "clone"
nested = clone / "src"
nested.mkdir(parents=True)
source = nested / "source.ts"
source.write_text("trusted\n")
output = clone / "review-output.json"
output.write_text("")
original_source_mode = stat.S_IMODE(source.stat().st_mode)
original_output_mode = stat.S_IMODE(output.stat().st_mode)
with host_workspace_write_boundary(clone, writable=(output,)):
with pytest.raises(OSError):
source.write_text("tampered\n")
with pytest.raises(OSError):
(clone / "extra.py").write_text("nope\n")
output.write_text('{"schema_version":1}\n')
assert source.read_text() == "trusted\n"
assert output.read_text() == '{"schema_version":1}\n'
assert not (clone / "extra.py").exists()
assert stat.S_IMODE(source.stat().st_mode) == original_source_mode
assert stat.S_IMODE(output.stat().st_mode) == original_output_mode
def test_sandbox_workspace_write_boundary_is_noop_unless_host_unsafe(tmp_path) -> None:
clone = tmp_path / "clone"
clone.mkdir()
source = clone / "source.ts"
source.write_text("trusted\n")
bwrap_sandbox = SimpleNamespace(backend="bwrap", clone=clone)
with sandbox_workspace_write_boundary(
bwrap_sandbox,
read_only_workspace=True,
writable=(),
):
source.write_text("still writable under bwrap no-op\n")
assert source.read_text() == "still writable under bwrap no-op\n"
output = clone / "review-output.json"
output.write_text("")
source.write_text("trusted\n")
unsafe = SimpleNamespace(backend="host-unsafe", clone=clone)
with sandbox_workspace_write_boundary(
unsafe,
read_only_workspace=True,
writable=(output,),
):
with pytest.raises(OSError):
source.write_text("tampered\n")
output.write_text("ok\n")
assert source.read_text() == "trusted\n"
assert output.read_text() == "ok\n"
def test_force_rmtree_deletes_nonempty_directories_copied_from_a_locked_workspace(tmp_path) -> None:
locked = tmp_path / "locked"
nested = locked / "gitnexus-shared" / "src"
nested.mkdir(parents=True)
(nested / "index.ts").write_text("export {}\n")
os.chmod(nested, 0o555)
os.chmod(locked / "gitnexus-shared", 0o555)
os.chmod(locked, 0o555)
copied = tmp_path / "sandbox-tmp" / "tmp.XXXX" / "gitnexus-shared"
copied.parent.mkdir(parents=True)
shutil.copytree(locked / "gitnexus-shared", copied)
assert stat.S_IMODE(copied.stat().st_mode) & 0o222 == 0
_force_rmtree(copied.parent)
assert not copied.parent.exists()
def test_host_unsafe_sandbox_cleanup_survives_readonly_tmpdir_copies(tmp_path) -> None:
clone = tmp_path / "clone"
clone.mkdir()
leftover = None
with prepare_sandbox(clone=clone, backend="host-unsafe") as sandbox:
leftover = sandbox.private_root
copied = sandbox.temp / "tmp.XXXX" / "gitnexus-shared" / "src"
copied.mkdir(parents=True)
(copied / "index.ts").write_text("export {}\n")
os.chmod(copied, 0o555)
os.chmod(copied.parent, 0o555)
os.chmod(copied.parent.parent, 0o555)
assert leftover is not None
assert not leftover.exists()
@pytest.mark.parametrize(
"bad_url",
["https://user:secret@example.test", "https://example.test/path?token=x", "file:///tmp/model"],

View file

@ -4,6 +4,8 @@ from pathlib import Path
import yaml
from workflow_bench.oracle_assets import review_case_setup_command
BENCH_ROOT = Path(__file__).parents[1] / "workflow_bench"
@ -26,6 +28,7 @@ def test_review_corpus_is_immutable_and_task_bound():
task = by_id[case["id"]]
assert task["ref"] == case["base_sha"]
assert task["sandbox_copy"] == [f"eval/workflow_bench/review_cases/{patch.name}"]
assert task["setup"] == review_case_setup_command(patch.name)
def test_hidden_labels_are_not_recoverable_from_visible_task_input():

View file

@ -27,6 +27,32 @@ def test_prebuilt_graph_and_harness_assets_are_rejected(task):
sanitized_graph.validate_no_prebuilt_graph_assets(task)
def test_review_case_patches_are_allowed_sandbox_copy():
sanitized_graph.validate_no_prebuilt_graph_assets(
{"sandbox_copy": ["eval/workflow_bench/review_cases/pr-2718-defect.patch"]}
)
@pytest.mark.parametrize(
"task",
[
{"sandbox_copy": ["eval/workflow_bench"]},
{"sandbox_copy": ["eval/workflow_bench/evolve.py"]},
{
"sandbox_dependencies": [
{
"source": "eval/workflow_bench/review_cases/pr-2718-defect.patch",
"target": "patch",
}
]
},
],
)
def test_non_corpus_harness_paths_stay_rejected(task):
with pytest.raises(SandboxError, match="prebuilt graph or harness"):
sanitized_graph.validate_no_prebuilt_graph_assets(task)
def test_graph_environment_is_offline_deterministic_and_ignores_target_gitignore():
env = sanitized_graph._graph_environment()
@ -34,6 +60,10 @@ def test_graph_environment_is_offline_deterministic_and_ignores_target_gitignore
assert env["GITNEXUS_NO_GITIGNORE"] == "1"
assert env["GITNEXUS_WORKER_POOL_SIZE"] == "1"
assert env["GITNEXUS_PARSE_CHUNK_CONCURRENCY"] == "1"
assert env["GITNEXUS_WORKER_READY_TIMEOUT_MS"] == str(
sanitized_graph.GRAPH_WORKER_READY_TIMEOUT_MS
)
assert int(env["GITNEXUS_WORKER_READY_TIMEOUT_MS"]) >= 60_000
assert "ANTHROPIC_API_KEY" not in env

View file

@ -5,15 +5,15 @@ from __future__ import annotations
import os
import stat
import subprocess
from pathlib import Path
from pathlib import Path, PurePosixPath
import pytest
from workflow_bench.proposer_sandbox import VITE_TEMP_DIR, SandboxError
from workflow_bench.oracle_assets import TaskOracleSnapshot
from workflow_bench.runner_tasks import resolve_task_bindings
from workflow_bench.task_assets import TaskAssetCache, stage_task_assets
from workflow_bench import task_assets
from workflow_bench.task_assets import TaskAssetCache, _is_harness_sandbox_copy, stage_task_assets
from workflow_bench import runtime_mounts, task_assets
SHA = "a" * 40
@ -443,3 +443,53 @@ def test_non_node_modules_dependency_snapshot_has_no_vite_temp(tmp_path: Path) -
snapshot = cache.prepare(task, repo=repo, resolved_sha=SHA)
captured = {entry.path.as_posix() for entry in snapshot.dependencies[0].entries}
assert not any(path.endswith(VITE_TEMP_DIR) for path in captured)
def test_review_case_sandbox_copy_is_read_from_the_harness_not_the_task_repo(
monkeypatch, tmp_path: Path
) -> None:
repo = tmp_path / "task-repo"
repo.mkdir()
(repo / "eval" / "workflow_bench").mkdir(parents=True)
harness = tmp_path / "harness"
patch = harness / "eval" / "workflow_bench" / "review_cases" / "pr-2718-defect.patch"
patch.parent.mkdir(parents=True)
patch.write_bytes(b"diff --git a/a b/a\n")
monkeypatch.setattr(runtime_mounts, "HARNESS_ROOT", harness)
task = {
"sandbox_copy": ["eval/workflow_bench/review_cases/pr-2718-defect.patch"],
"sandbox_dependencies": [],
}
with TaskAssetCache(tmp_path / "cache") as cache:
snapshot = cache.prepare(task, repo=repo, resolved_sha=SHA)
copied = snapshot.root / "sandbox-copy" / "eval" / "workflow_bench" / "review_cases" / "pr-2718-defect.patch"
assert copied.read_bytes() == b"diff --git a/a b/a\n"
def test_review_case_sandbox_copy_does_not_fall_back_to_the_task_repo(
monkeypatch, tmp_path: Path
) -> None:
repo = tmp_path / "task-repo"
planted = repo / "eval" / "workflow_bench" / "review_cases" / "pr-2718-defect.patch"
planted.parent.mkdir(parents=True)
planted.write_bytes(b"from-task-repo")
harness = tmp_path / "harness"
harness.mkdir()
monkeypatch.setattr(runtime_mounts, "HARNESS_ROOT", harness)
task = {
"sandbox_copy": ["eval/workflow_bench/review_cases/pr-2718-defect.patch"],
"sandbox_dependencies": [],
}
with TaskAssetCache(tmp_path / "cache") as cache:
with pytest.raises(SandboxError, match="unavailable"):
cache.prepare(task, repo=repo, resolved_sha=SHA)
def test_harness_sandbox_copy_does_not_treat_parent_escapes_as_corpus() -> None:
assert _is_harness_sandbox_copy(PurePosixPath("eval/workflow_bench/review_cases/pr.patch"))
assert not _is_harness_sandbox_copy(
PurePosixPath("eval/workflow_bench/review_cases/../oracles/hidden.json")
)
assert not _is_harness_sandbox_copy(PurePosixPath("eval/workflow_bench/oracles"))

View file

@ -15,6 +15,7 @@ from workflow_bench.evolution import (
evaluate_candidate,
evaluate_review_candidate,
required_candidate_arms,
seed_evaluated_skills,
skill_fingerprint,
unexercised_overlay_skills,
)
@ -387,6 +388,12 @@ def test_apply_candidate_overlay_creates_a_clean_ephemeral_commit(tmp_path):
) == candidate_overlay_digest(overlay)
assert incumbent.read_text() == "candidate\n"
git_commands = [command for command in sandbox.commands if command[0] == "/usr/bin/git"]
assert git_commands[0][-4:] == [
"add",
"-f",
"--",
".claude/skills/gitnexus-work/SKILL.md",
]
assert [command[-1] for command in git_commands[:2]] == [
".claude/skills/gitnexus-work/SKILL.md",
"--",
@ -422,6 +429,181 @@ def test_candidate_overlay_rejects_linked_destination_parents(tmp_path):
apply_candidate_overlay(overlay, repo, sandbox=sandbox)
@pytest.mark.skipif(os.name == "nt", reason="candidate overlays require the Linux outer sandbox")
def test_apply_candidate_overlay_force_adds_historically_ignored_skill(tmp_path):
repo = tmp_path / "repo"
repo.mkdir()
subprocess.run(["git", "init", "--quiet", str(repo)], check=True)
(repo / ".gitignore").write_text(".claude/skills/*\n")
(repo / "README").write_text("subject\n")
subprocess.run(["git", "-C", str(repo), "add", "."], check=True)
subprocess.run(
[
"git",
"-C",
str(repo),
"-c",
"user.name=test",
"-c",
"user.email=test@invalid",
"commit",
"--quiet",
"-m",
"historical checkout that ignores skills",
],
check=True,
)
overlay = tmp_path / "candidate"
write_overlay_skill(overlay, "gitnexus-review")
class LocalSandbox:
def __init__(self):
self.clone = repo
def run(self, command, **kwargs):
if command[0] == "/bin/mkdir":
return ManagedProcessResult(
state="exited",
returncode=0,
stdout_tail="",
stderr_tail="",
duration_s=0.0,
)
translated = [str(repo) if item == "/workspace" else item for item in command]
completed = subprocess.run(
translated,
cwd=repo,
env=dict(kwargs["env"]),
capture_output=True,
text=True,
check=False,
)
return ManagedProcessResult(
state="exited",
returncode=completed.returncode,
stdout_tail=completed.stdout,
stderr_tail=completed.stderr,
duration_s=0.0,
)
apply_candidate_overlay(overlay, repo, sandbox=LocalSandbox())
assert (repo / ".claude" / "skills" / "gitnexus-review" / "SKILL.md").read_text() == (
"gitnexus-review candidate\n"
)
status = subprocess.run(
["git", "-C", str(repo), "status", "--porcelain"],
check=True,
capture_output=True,
text=True,
)
assert status.stdout == ""
@pytest.mark.skipif(os.name == "nt", reason="skill seeds require the Linux outer sandbox")
def test_seed_evaluated_skills_installs_missing_review_skill_and_is_idempotent(tmp_path):
repo = tmp_path / "clone"
repo.mkdir()
subprocess.run(["git", "init", "--quiet", str(repo)], check=True)
# Historical review SHAs ignore the whole skill tree and lack today's
# `!.claude/skills/gitnexus-review/` allowlist. Seeding must still commit.
(repo / ".gitignore").write_text(".claude/skills/*\n")
(repo / "README").write_text("subject\n")
subprocess.run(["git", "-C", str(repo), "add", "."], check=True)
subprocess.run(
[
"git",
"-C",
str(repo),
"-c",
"user.name=test",
"-c",
"user.email=test@invalid",
"commit",
"--quiet",
"-m",
"historical checkout without review skill",
],
check=True,
)
before = subprocess.run(
["git", "-C", str(repo), "rev-parse", "HEAD"],
check=True,
capture_output=True,
text=True,
).stdout.strip()
source = tmp_path / "harness"
skill = source / ".claude" / "skills" / "gitnexus-review" / "SKILL.md"
persona = source / ".claude" / "skills" / "gitnexus-review" / "ci-personas" / "lens.md"
persona.parent.mkdir(parents=True)
skill.write_text("current incumbent review skill\n")
persona.write_text("persona\n")
class LocalSandbox:
def __init__(self):
self.clone = repo
def run(self, command, **kwargs):
if command[0] == "/bin/mkdir":
return ManagedProcessResult(
state="exited",
returncode=0,
stdout_tail="",
stderr_tail="",
duration_s=0.0,
)
translated = [str(repo) if item == "/workspace" else item for item in command]
completed = subprocess.run(
translated,
cwd=repo,
env=dict(kwargs["env"]),
capture_output=True,
text=True,
check=False,
)
return ManagedProcessResult(
state="exited",
returncode=completed.returncode,
stdout_tail=completed.stdout,
stderr_tail=completed.stderr,
duration_s=0.0,
)
sandbox = LocalSandbox()
seed_evaluated_skills(source, repo, sandbox=sandbox, arm="review")
assert skill_fingerprint(repo, "review") is not None
assert (repo / ".claude" / "skills" / "gitnexus-review" / "SKILL.md").read_text() == (
"current incumbent review skill\n"
)
assert (
repo / ".claude" / "skills" / "gitnexus-review" / "ci-personas" / "lens.md"
).read_text() == "persona\n"
status = subprocess.run(
["git", "-C", str(repo), "status", "--porcelain"],
check=True,
capture_output=True,
text=True,
)
assert status.stdout == ""
after = subprocess.run(
["git", "-C", str(repo), "rev-parse", "HEAD"],
check=True,
capture_output=True,
text=True,
).stdout.strip()
assert after != before
seed_evaluated_skills(source, repo, sandbox=sandbox, arm="review")
again = subprocess.run(
["git", "-C", str(repo), "rev-parse", "HEAD"],
check=True,
capture_output=True,
text=True,
).stdout.strip()
assert again == after
@pytest.mark.skipif(os.name == "nt", reason="skill links are rejected by the Linux sandbox harness")
def test_skill_fingerprint_rejects_linked_skill_roots(tmp_path):
outside = tmp_path / "outside"

View file

@ -4,6 +4,7 @@ import argparse
import hashlib
import json
import os
import shutil
import subprocess
import sys
from pathlib import Path
@ -14,6 +15,7 @@ import pytest
from workflow_bench import evolve, runner, runner_sessions, runtime_mounts
from workflow_bench.evolution import skill_fingerprint
from workflow_bench.process_control import ManagedProcessResult
from workflow_bench.proposer_sandbox import SandboxError
from workflow_bench.runner import snapshot_plan_docs
@ -315,10 +317,18 @@ def test_run_arm_keeps_session_error_kind_over_verify(monkeypatch, tmp_path):
def test_agent_tool_grants_are_exact_and_nomcp_has_no_graph_tools(monkeypatch, tmp_path):
read_only = runner.allowed_agent_tools(implementation=False)
review_tools = runner.allowed_agent_tools(implementation=False, allow_edit=False)
implementation = runner.allowed_agent_tools(implementation=True)
no_mcp = runner.allowed_agent_tools(implementation=True, include_mcp=False)
assert read_only == [*runner.BUILTIN_AGENT_TOOLS, *runner.GITNEXUS_READ_ONLY_TOOLS]
assert review_tools == [
tool
for tool in [*runner.BUILTIN_AGENT_TOOLS, *runner.GITNEXUS_READ_ONLY_TOOLS]
if tool != "Edit"
]
assert "Write" in review_tools
assert "Edit" not in review_tools
assert implementation == [
*runner.BUILTIN_AGENT_TOOLS,
*runner.GITNEXUS_READ_ONLY_TOOLS,
@ -346,7 +356,7 @@ def test_agent_tool_grants_are_exact_and_nomcp_has_no_graph_tools(monkeypatch, t
)
assert captured[0]["allowed_tools"] == read_only # planning
assert captured[1]["allowed_tools"] == read_only # review
assert captured[1]["allowed_tools"] == review_tools # review
assert captured[2]["allowed_tools"] == implementation
assert captured[3]["allowed_tools"] == list(runner.BUILTIN_AGENT_TOOLS)
assert captured[3]["mcp_config_json"] == '{"mcpServers":{}}'
@ -416,6 +426,93 @@ def test_mcp_config_uses_only_the_minimal_pinned_harness_runtime(monkeypatch, tm
assert f"{runner.SANDBOX_GITNEXUS}/hooks" not in mounted_targets
def _install_pinned_runtime(root: Path) -> None:
runtime = root / "gitnexus"
shared = root / "gitnexus-shared"
for directory in (
runtime / "dist" / "cli",
runtime / "node_modules",
runtime / "vendor",
runtime / "hooks" / "claude",
shared / "dist",
):
directory.mkdir(parents=True)
(runtime / "dist" / "cli" / "index.js").write_text("")
(runtime / "hooks" / "claude" / "resolve-analyze-cmd.cjs").write_text("")
(runtime / "package.json").write_text(json.dumps({"version": "9.9.9-test"}))
(runtime / "node_modules" / "gitnexus-shared").symlink_to(shared, target_is_directory=True)
(shared / "package.json").write_text(json.dumps({"name": "gitnexus-shared"}))
def test_runtime_mounts_reuse_primary_checkout_node_modules_from_a_worktree(
monkeypatch, tmp_path
) -> None:
primary = tmp_path / "primary"
worktree = tmp_path / "worktree"
_install_pinned_runtime(primary)
(primary / ".git" / "worktrees" / "wt").mkdir(parents=True)
_install_pinned_runtime(worktree)
shutil.rmtree(worktree / "gitnexus" / "node_modules")
(worktree / "gitnexus" / "node_modules").symlink_to(
primary / "gitnexus" / "node_modules",
target_is_directory=True,
)
(worktree / ".git").write_text(f"gitdir: {primary / '.git' / 'worktrees' / 'wt'}\n")
monkeypatch.setattr(runtime_mounts, "HARNESS_ROOT", worktree)
mounts = runner.trusted_gitnexus_runtime_mounts()
by_target = {mount.target: mount.source for mount in mounts}
assert by_target[f"{runner.SANDBOX_GITNEXUS}/node_modules"] == (
primary / "gitnexus" / "node_modules"
)
assert by_target[f"{runner.SANDBOX_GITNEXUS_SHARED}/package.json"] == (
worktree / "gitnexus-shared" / "package.json"
)
assert by_target[f"{runner.SANDBOX_GITNEXUS}/dist"] == worktree / "gitnexus" / "dist"
def test_runtime_mounts_reject_a_node_modules_symlink_outside_the_primary_checkout(
monkeypatch, tmp_path
) -> None:
primary = tmp_path / "primary"
worktree = tmp_path / "worktree"
outsider = tmp_path / "outsider"
_install_pinned_runtime(primary)
_install_pinned_runtime(outsider)
(primary / ".git" / "worktrees" / "wt").mkdir(parents=True)
_install_pinned_runtime(worktree)
shutil.rmtree(worktree / "gitnexus" / "node_modules")
(worktree / "gitnexus" / "node_modules").symlink_to(
outsider / "gitnexus" / "node_modules",
target_is_directory=True,
)
(worktree / ".git").write_text(f"gitdir: {primary / '.git' / 'worktrees' / 'wt'}\n")
monkeypatch.setattr(runtime_mounts, "HARNESS_ROOT", worktree)
with pytest.raises(SandboxError, match="primary checkout"):
runner.trusted_gitnexus_runtime_mounts()
def test_runtime_mounts_reject_a_node_modules_symlink_in_a_regular_checkout(
monkeypatch, tmp_path
) -> None:
checkout = tmp_path / "checkout"
other = tmp_path / "other"
_install_pinned_runtime(checkout)
_install_pinned_runtime(other)
(checkout / ".git").mkdir()
shutil.rmtree(checkout / "gitnexus" / "node_modules")
(checkout / "gitnexus" / "node_modules").symlink_to(
other / "gitnexus" / "node_modules",
target_is_directory=True,
)
monkeypatch.setattr(runtime_mounts, "HARNESS_ROOT", checkout)
with pytest.raises(SandboxError, match="must be a real directory"):
runner.trusted_gitnexus_runtime_mounts()
@pytest.mark.skipif(
os.environ.get("GITNEXUS_REQUIRE_BWRAP_CANARY") != "1",
reason="real Bubblewrap canary is mandatory in the named Ubuntu CI job",
@ -662,6 +759,23 @@ def test_skill_invocation_parses_supported_exact_identifier_fields(skill_input):
)
def test_skill_invocation_accepts_plugin_qualified_identifier():
assert (
runner_sessions.skill_was_invoked_events(
skill_events({"skill": "compound-engineering:ce-code-review"}),
"ce-code-review",
)
is True
)
assert (
runner_sessions.skill_was_invoked_events(
skill_events({"skill": "compound-engineering:ce-plan"}),
"ce-code-review",
)
is False
)
@pytest.mark.parametrize(
"skill_input",
[

View file

@ -139,6 +139,28 @@ Native benchmark execution is therefore Linux/WSL2-only. Evidence assembly
and hand-authored overlay preparation can happen elsewhere, but
`--initial-overlay` does not bypass containment.
For a local diagnostic inside a container that blocks user namespaces, an
operator may explicitly choose the non-containment host backend:
```bash
UNSAFE_NO_BWRAP=1 RUNS=1 ./workflow_bench/run-evolution.sh
```
This mode runs review sessions directly in disposable host worktrees and is
**not** a security boundary: it does not isolate the network or create a PID
namespace, and a session that can `chmod` can undo the workspace lock. The
harness still drops write bits on the clone except `review-output.json` so
accidental `npm install` / analyze writes cannot invalidate review evidence.
Sandbox cleanup restores owner write bits before deleting the private TMPDIR,
because a session that `copytree`s the locked clone would otherwise leave
non-empty 0555 directories that `rmtree` cannot remove. Historical review
SHAs that gitignore `.claude/skills/*` are force-added when the harness seeds
or overlays the evaluated `gitnexus-review` skill.
Treat model and verifier processes as able to access host files and
credentials available to the invoking user. It is restricted to the review
benchmark, forbidden with `--apply` and whenever `CI` is set;
promotion-capable and CI runs must use Bubblewrap.
## Prompt and skill evolution loop
Prompts age as models and tool harnesses change. Treat the current skills and

View file

@ -7,6 +7,7 @@ import os
import secrets
import stat
import statistics
from collections.abc import Sequence
from pathlib import Path, PurePosixPath
from typing import Any
@ -383,6 +384,69 @@ def candidate_overlay_digest(overlay: Path) -> str:
return digest
def _commit_sandbox_paths(
sandbox: SandboxSession,
relative_paths: Sequence[str],
*,
message: str,
require_change: bool,
) -> bool:
"""Stage and commit paths inside the outer sandbox.
Returns True when a commit was created. ``require_change`` keeps the
candidate-overlay contract: a no-op overlay is an error, while an
incumbent skill seed may already match the historical tree.
"""
mkdir_command = ["/bin/mkdir", "-p", f"{SANDBOX_TMP}/wfbench-empty-hooks"]
mkdir_result = sandbox.run(
mkdir_command,
timeout=60,
env=build_sandbox_environment(),
)
if not mkdir_result.ok:
raise ManagedProcessError(mkdir_command, mkdir_result)
if not relative_paths:
if require_change:
raise ValueError("candidate overlay is byte-identical to the incumbent skills")
return False
# Historical review SHAs gitignore `.claude/skills/*` and lack the current
# per-skill allowlist. Force-add so a seed or overlay of harness-owned
# skill bytes is not rejected as an ignored path.
command, added = _sandbox_overlay_git(sandbox, ["add", "-f", "--", *relative_paths])
if not added.ok:
raise ManagedProcessError(command, added)
command, changed = _sandbox_overlay_git(
sandbox,
["diff", "--cached", "--quiet", "--no-ext-diff", "--no-textconv", "--"],
)
if changed.returncode == 0:
if require_change:
raise ValueError("candidate overlay is byte-identical to the incumbent skills")
return False
if changed.returncode != 1:
raise ManagedProcessError(command, changed)
command, committed = _sandbox_overlay_git(
sandbox,
[
"commit",
"--quiet",
"--no-verify",
"-m",
message,
],
extra_config=(
"user.name=workflow-bench",
"user.email=workflow-bench@invalid",
),
)
if not committed.ok:
raise ManagedProcessError(command, committed)
return True
def apply_candidate_overlay(
overlay: Path,
worktree: Path,
@ -401,47 +465,96 @@ def apply_candidate_overlay(
for relative, content in payload:
_replace_regular_file(worktree, relative, content)
relative_paths.append(relative.as_posix())
mkdir_command = ["/bin/mkdir", "-p", f"{SANDBOX_TMP}/wfbench-empty-hooks"]
mkdir_result = sandbox.run(
mkdir_command,
timeout=60,
env=build_sandbox_environment(),
)
if not mkdir_result.ok:
raise ManagedProcessError(mkdir_command, mkdir_result)
command, added = _sandbox_overlay_git(sandbox, ["add", "--", *relative_paths])
if not added.ok:
raise ManagedProcessError(command, added)
command, changed = _sandbox_overlay_git(
_commit_sandbox_paths(
sandbox,
["diff", "--cached", "--quiet", "--no-ext-diff", "--no-textconv", "--"],
relative_paths,
message="benchmark candidate skill overlay",
require_change=True,
)
if changed.returncode == 0:
raise ValueError("candidate overlay is byte-identical to the incumbent skills")
if changed.returncode != 1:
raise ManagedProcessError(command, changed)
command, committed = _sandbox_overlay_git(
sandbox,
[
"commit",
"--quiet",
"--no-verify",
"-m",
"benchmark candidate skill overlay",
],
extra_config=(
"user.name=workflow-bench",
"user.email=workflow-bench@invalid",
),
)
if not committed.ok:
raise ManagedProcessError(command, committed)
return digest
def seed_evaluated_skills(
source_repo: Path,
worktree: Path,
*,
sandbox: SandboxSession,
arm: str,
) -> None:
"""Install the current evaluated skill tree into a historical clone.
Review evolution scores the current (or overlay) ``gitnexus-review`` skill
against a historical PR checkout. Older SHAs predate that skill, and
using whatever prose happened to exist at the reviewed commit would make
the incumbent arm a moving target. Copy the harness checkout's skill
bytes and commit them before setup so ``git status`` still shows only
the task patch.
"""
skill_names = EVALUATED_ARM_SKILLS.get(arm)
if not skill_names:
return
source_repo = source_repo.expanduser().absolute()
expected_clone = Path(os.path.abspath(worktree.expanduser()))
sandbox_clone = Path(os.path.abspath(sandbox.clone.expanduser()))
if sandbox_clone != expected_clone:
raise ValueError("skill seed sandbox does not bind the requested clone")
_require_real_directory(source_repo, label="incumbent skill repository")
if source_repo.resolve(strict=True) != source_repo:
raise ValueError(f"incumbent skill repository cannot traverse symlinks: {source_repo}")
relative_paths: list[str] = []
total = 0
for skill_name in skill_names:
_require_directory_chain(
source_repo,
Path(".claude") / "skills" / skill_name,
label="incumbent skill root",
)
skill_root = source_repo / ".claude" / "skills" / skill_name
pending = [skill_root]
while pending:
directory = pending.pop()
try:
children = list(os.scandir(directory))
except OSError as exc:
raise ValueError(f"incumbent skill directory is unreadable: {directory}: {exc}") from exc
for item in children:
path = Path(item.path)
if item.is_symlink():
raise ValueError(
"incumbent skill seed cannot contain symlinks: "
f"{path.relative_to(source_repo)}"
)
if item.is_dir(follow_symlinks=False):
pending.append(path)
continue
if not item.is_file(follow_symlinks=False):
raise ValueError(
"incumbent skill seed entries must be regular files: "
f"{path.relative_to(source_repo)}"
)
total += item.stat(follow_symlinks=False).st_size
if total > MAX_SKILL_FINGERPRINT_BYTES:
raise ValueError("incumbent skill seed exceeds the bounded evidence limit")
relative = Path(".claude") / "skills" / skill_name / path.relative_to(skill_root)
content = _bounded_regular_bytes(
path,
limit=MAX_SKILL_FINGERPRINT_BYTES,
label="incumbent skill file",
)
_replace_regular_file(worktree, relative, content)
relative_paths.append(PurePosixPath(relative.as_posix()).as_posix())
_commit_sandbox_paths(
sandbox,
relative_paths,
message="benchmark incumbent review skill",
require_change=False,
)
def unexercised_overlay_skills(overlay: Path, candidate_arms: list[str]) -> list[str]:
"""Overlay skills that no selected candidate arm would ever load.

View file

@ -82,6 +82,7 @@ from .proposer_sandbox import (
SandboxError,
build_sandbox_environment,
preflight_bubblewrap,
preflight_unsafe_host,
pid_namespace_command,
prepare_sandbox,
redact_text,
@ -127,8 +128,8 @@ WORKTREE_PREPARATION_TIMEOUT_SECONDS = (
# from that commit. Use the overlay boundary rather than the current candidate
# size so this helper remains conservative before the runner starts.
PROMOTION_BASE_TIMEOUT_SECONDS = (1 + 3 * MAX_CANDIDATE_FILES) * GIT_COMMAND_TIMEOUT_SECONDS
ARM_SESSION_COUNTS = {"workflow": 2, "workflow_direct": 1}
ARM_WORKSPACE_SNAPSHOT_COUNTS = {"workflow": 2, "workflow_direct": 0}
ARM_SESSION_COUNTS = {"workflow": 2, "workflow_direct": 1, "review": 1}
ARM_WORKSPACE_SNAPSHOT_COUNTS = {"workflow": 2, "workflow_direct": 0, "review": 1}
REPO_ROOT = Path(__file__).resolve().parents[2]
@ -692,6 +693,7 @@ def run_proposer(
proposal_path: Path,
evidence_bundle: Path,
bwrap_bin: Path,
sandbox_backend: str = "bwrap",
progress_label: str | None = None,
) -> dict[str, Any]:
"""Run one proposer in confinement and copy only validated outputs out."""
@ -722,9 +724,13 @@ def run_proposer(
bwrap_bin=bwrap_bin,
read_only_mounts=[evidence_mount],
preflight=False,
backend=sandbox_backend,
) as sandbox:
host_text = getattr(sandbox, "host_text", lambda value: value)
environment_builder = getattr(sandbox, "environment", build_sandbox_environment)
backend = getattr(sandbox, "backend", "bwrap")
record = runner.run_claude(
prompt,
host_text(prompt),
clone,
claude_bin=sandbox.claude_bin,
timeout=args.timeout,
@ -734,7 +740,7 @@ def run_proposer(
auth_token=args.auth_token,
base_url=args.base_url,
model=args.proposer_model,
build_sandbox_environment=build_sandbox_environment,
build_sandbox_environment=environment_builder,
),
# No permission_mode: CLAUDE_CODE_SUBPROCESS_ENV_SCRUB
# forces "default", so requesting dontAsk only warns. Tools
@ -743,7 +749,10 @@ def run_proposer(
# bare ignores --tools and imposes its own Bash/Edit/Read
# ceiling, which would cost the proposer Grep and Glob.
command_prefix=sandbox.command_prefix,
require_pid_namespace=True,
require_pid_namespace=getattr(sandbox, "require_pid_namespace", True),
permission_mode=(
"bypassPermissions" if backend == "host-unsafe" else None
),
settings_json=sandbox.settings_json,
strict_mcp_config=True,
mcp_config_json='{"mcpServers":{}}',
@ -796,6 +805,21 @@ def resolve_incumbent_arms(overlay: Path, explicit_arms: list[str] | None) -> li
return required
def executed_benchmark_arms(incumbent_arms: Sequence[str]) -> list[str]:
"""Incumbent/candidate pairs plus the review comparator when needed."""
paired = [arm for incumbent in incumbent_arms for arm in (incumbent, INCUMBENT_ARMS[incumbent])]
if "review" in incumbent_arms:
paired.insert(0, "ce_review")
return paired
def _timeout_arm_key(arm: str) -> str:
if arm == "ce_review":
return "review"
return CANDIDATE_ARMS.get(arm, arm)
def generation_timeout_seconds(
*,
task_count: int,
@ -808,11 +832,14 @@ def generation_timeout_seconds(
if task_count < 1 or runs < 1 or session_timeout < 1:
raise ValueError("task count, runs, and session timeout must be positive")
try:
session_slots = sum(2 * ARM_SESSION_COUNTS[arm] for arm in incumbent_arms)
executed = executed_benchmark_arms(incumbent_arms)
session_slots = sum(ARM_SESSION_COUNTS[_timeout_arm_key(arm)] for arm in executed)
workspace_snapshot_slots = sum(
ARM_WORKSPACE_SNAPSHOT_COUNTS[_timeout_arm_key(arm)] for arm in executed
)
except KeyError as exc:
raise ValueError(f"unsupported evolution arm: {exc.args[0]}") from exc
paired_arm_cells = 2 * len(incumbent_arms)
workspace_snapshot_slots = sum(2 * ARM_WORKSPACE_SNAPSHOT_COUNTS[arm] for arm in incumbent_arms)
paired_arm_cells = len(executed)
per_task_preparation = (
TASK_BINDING_GIT_PHASES * GIT_COMMAND_TIMEOUT_SECONDS
+ 2 * TASK_SNAPSHOT_TIMEOUT_SECONDS
@ -849,9 +876,7 @@ def runner_argv(
proposer_model: str | None = None,
) -> list[str]:
incumbent_arms = resolve_incumbent_arms(overlay_dir, args.arms)
paired_arms = [arm for incumbent in incumbent_arms for arm in (incumbent, INCUMBENT_ARMS[incumbent])]
if "review" in incumbent_arms:
paired_arms.insert(0, "ce_review")
paired_arms = executed_benchmark_arms(incumbent_arms)
argv = [
sys.executable,
"-m",
@ -897,6 +922,8 @@ def runner_argv(
argv.append("--include-expensive")
if args.ce_plugin_dir is not None:
argv += ["--ce-plugin-dir", str(args.ce_plugin_dir), "--ce-plugin-version", args.ce_plugin_version]
if args.unsafe_no_bwrap:
argv.append("--unsafe-no-bwrap")
return argv
@ -1186,6 +1213,12 @@ def build_parser() -> argparse.ArgumentParser:
)
parser.add_argument("--ce-plugin-dir", type=Path, default=None)
parser.add_argument("--ce-plugin-version", default=None)
parser.add_argument(
"--unsafe-no-bwrap",
action="store_true",
help="LOCAL DIAGNOSTICS ONLY: use PRoot path translation without filesystem, "
"network, or PID isolation; forbidden with --apply and in CI",
)
return parser
@ -1196,6 +1229,10 @@ def main() -> int:
parser.error("--generations must be positive")
if args.runs < 1 or args.timeout < 1:
parser.error("--runs and --timeout must be positive")
if args.unsafe_no_bwrap and args.apply:
parser.error("--unsafe-no-bwrap cannot be combined with --apply")
if args.unsafe_no_bwrap and os.environ.get("CI"):
parser.error("--unsafe-no-bwrap is forbidden when CI is set")
try:
args.model = runner.normalized_model_identifier(args.model)
args.proposer_model = runner.normalized_model_identifier(
@ -1213,6 +1250,8 @@ def main() -> int:
parser.error(str(exc))
raise AssertionError("ArgumentParser.error() returned unexpectedly")
requested_arms = args.arms or ["workflow", "workflow_direct"]
if args.unsafe_no_bwrap and requested_arms != ["review"]:
parser.error("--unsafe-no-bwrap is restricted to --arms review")
if "review" in requested_arms and (
args.ce_plugin_dir is None
or not args.ce_plugin_dir.expanduser().is_dir()
@ -1229,8 +1268,19 @@ def main() -> int:
parser.error(str(exc))
selected_tasks = runner.selected_task_bindings(selected_task_rows)
try:
bwrap_bin = preflight_bubblewrap()
require_claude_sandbox_helpers()
if args.unsafe_no_bwrap:
bwrap_bin = preflight_unsafe_host()
sandbox_backend = "host-unsafe"
print(
"WARNING: --unsafe-no-bwrap runs sessions directly on the host with no "
"containment; model and verifier processes can access the host filesystem, "
"network, and credentials.",
file=sys.stderr,
)
else:
bwrap_bin = preflight_bubblewrap()
sandbox_backend = "bwrap"
require_claude_sandbox_helpers()
except SandboxError as exc:
parser.error(str(exc))
raise AssertionError("ArgumentParser.error() returned unexpectedly")
@ -1250,6 +1300,7 @@ def main() -> int:
requested_arms=requested_arms,
initial_overlay=initial_overlay,
bwrap_bin=bwrap_bin,
sandbox_backend=sandbox_backend,
)
finally:
gateway.__exit__(None, None, None)
@ -1264,6 +1315,7 @@ def _run_generations(
requested_arms: list[str],
initial_overlay: Path | None,
bwrap_bin: Path,
sandbox_backend: str,
) -> int:
policy_binding = {
"metric": args.promotion_metric,
@ -1344,6 +1396,7 @@ def _run_generations(
proposal_path=gen_dir / "proposal.md",
evidence_bundle=bundle,
bwrap_bin=bwrap_bin,
sandbox_backend=sandbox_backend,
progress_label=f"gen {generation} proposer",
)
# Redact any API token echoed into the session record (e.g. an
@ -1383,18 +1436,21 @@ def _run_generations(
print(f"[gen {generation}] promotion targets contain uncommitted or drifted bytes")
return 1
print(f"[gen {generation}] benchmarking candidate…")
benchmark_argv = runner_argv(
args,
bench_dir,
frozen_overlay,
task_bindings=selected_tasks,
target_base_digests=target_base_digests,
proposer_model=generation_proposer_model,
)
benchmark_command = (
benchmark_argv
if sandbox_backend == "host-unsafe"
else pid_namespace_command(benchmark_argv, bwrap_bin=bwrap_bin)
)
bench = run_managed(
pid_namespace_command(
runner_argv(
args,
bench_dir,
frozen_overlay,
task_bindings=selected_tasks,
target_base_digests=target_base_digests,
proposer_model=generation_proposer_model,
),
bwrap_bin=bwrap_bin,
),
benchmark_command,
timeout=generation_timeout_seconds(
task_count=len(selected_task_rows),
runs=args.runs,
@ -1402,7 +1458,7 @@ def _run_generations(
incumbent_arms=incumbent_arms,
),
env=runner_environment(args),
require_pid_namespace=True,
require_pid_namespace=sandbox_backend == "bwrap",
# The sweep is the multi-hour phase; without this its per-run
# progress lines only reach the log as a bounded tail, and only
# when it fails.

View file

@ -42,6 +42,27 @@ _OPENAI_MODEL = re.compile(
# shorter than that, so both ends of the loopback hop get the same generous
# budget and the session fails on real errors instead of on the clock.
GATEWAY_REQUEST_TIMEOUT_S = 1800
# Importing LiteLLM alone costs ~17s on a cold container filesystem, and the
# proxy only binds its port after that. A budget tight enough to lose that race
# reads as "connection refused", which looks like a dead proxy rather than a
# slow import.
GATEWAY_READY_TIMEOUT_ENV = "GITNEXUS_BENCH_GATEWAY_READY_TIMEOUT_S"
DEFAULT_GATEWAY_READY_TIMEOUT_S = 180.0
def gateway_ready_timeout_s() -> float:
"""Startup budget for the loopback proxy, overridable for slow hosts."""
raw = (os.environ.get(GATEWAY_READY_TIMEOUT_ENV) or "").strip()
if not raw:
return DEFAULT_GATEWAY_READY_TIMEOUT_S
try:
value = float(raw)
except ValueError as exc:
raise ValueError(f"{GATEWAY_READY_TIMEOUT_ENV} must be a number of seconds, not {raw!r}") from exc
if value <= 0:
raise ValueError(f"{GATEWAY_READY_TIMEOUT_ENV} must be positive, not {raw!r}")
return value
def is_openai_model(model: str) -> bool:
@ -241,12 +262,12 @@ class OpenAIGateway(AbstractContextManager["OpenAIGateway"]):
openai_api_key: str,
model_names: Sequence[str],
work_dir: Path,
ready_timeout_s: float = 20.0,
ready_timeout_s: float | None = None,
) -> None:
self.openai_api_key = openai_api_key
self.model_names = tuple(model_names)
self.work_dir = work_dir
self.ready_timeout_s = ready_timeout_s
self.ready_timeout_s = gateway_ready_timeout_s() if ready_timeout_s is None else ready_timeout_s
self.auth_token = secrets.token_hex(16)
self.port = _free_loopback_port()
self.base_url = f"http://127.0.0.1:{self.port}"
@ -342,7 +363,13 @@ class OpenAIGateway(AbstractContextManager["OpenAIGateway"]):
except (urllib.error.URLError, TimeoutError, ConnectionError, OSError) as exc:
last_error = str(exc)
time.sleep(0.1)
raise RuntimeError(f"OpenAI LiteLLM gateway was not ready on {self.base_url}: {last_error}")
detail = self.log_tail()
raise RuntimeError(
f"OpenAI LiteLLM gateway was not ready on {self.base_url} after "
f"{self.ready_timeout_s:.0f}s: {last_error}"
+ (f"; proxy log: {detail}" if detail else "; proxy wrote no output yet")
+ f" (raise {GATEWAY_READY_TIMEOUT_ENV} if this host is simply slow to import LiteLLM)"
)
class attach_openai_gateway(AbstractContextManager[argparse.Namespace]):

View file

@ -24,6 +24,10 @@ MAX_ORACLE_PATH_BYTES = 240
MAX_ORACLE_COMMAND_BYTES = 8 * 1024
ORACLE_ENV_VAR = "GITNEXUS_BENCH_ORACLE_ROOT"
HIDDEN_HARNESS_PATH = PurePosixPath("eval/workflow_bench")
# Historical PR diffs can still edit this tree. Sanitization deletes it before
# setup, so `git apply` must skip those hunks or it fails with
# `error: eval/workflow_bench/<file>: No such file or directory`.
HIDDEN_HARNESS_APPLY_EXCLUDE = f"{HIDDEN_HARNESS_PATH.as_posix()}/*"
MAX_CLONE_REFS = 1024
MAX_CLONE_REF_BYTES = 2 * 1024 * 1024
@ -240,6 +244,36 @@ def _git_checked(
return result.stdout_tail.strip()
def with_hidden_harness_apply_exclude(setup: str) -> str:
"""Skip hunks for the harness tree sanitization already deleted.
Review cells copy a historical PR patch back under ``eval/workflow_bench``
and then ``git apply`` it. That patch may still mention harness files
(``learnings.jsonl`` on review-pr-2718-defect). Those files are gone from
the parentless snapshot, and setup deletes the directory again after apply,
so the hunks are never model-visible.
"""
if "git apply" not in setup:
return setup
flag = f"--exclude='{HIDDEN_HARNESS_APPLY_EXCLUDE}'"
if flag in setup or f'--exclude="{HIDDEN_HARNESS_APPLY_EXCLUDE}"' in setup:
return setup
return setup.replace("git apply ", f"git apply {flag} ", 1)
def review_case_setup_command(patch_name: str) -> str:
"""Setup that applies one review-case patch and then hides the harness."""
if not patch_name or "/" in patch_name or "\\" in patch_name or patch_name in {".", ".."}:
raise ValueError(f"review patch name must be a single path segment: {patch_name!r}")
patch = f"{HIDDEN_HARNESS_PATH.as_posix()}/review_cases/{patch_name}"
return (
f"git apply --exclude='{HIDDEN_HARNESS_APPLY_EXCLUDE}' {patch} "
f"&& rm -rf {HIDDEN_HARNESS_PATH.as_posix()}"
)
def sanitize_clone_for_hidden_oracles(clone: Path) -> str:
"""Remove the harness and its recoverable Git history from a disposable clone.

View file

@ -82,13 +82,58 @@ class SandboxSession:
home: Path
temp: Path
bwrap_bin: Path
backend: str
claude_host_bin: Path
command_prefix: list[str]
read_only_mounts: tuple[ReadOnlyMount, ...]
@property
def require_pid_namespace(self) -> bool:
return self.backend == "bwrap"
def host_path(self, raw: str | Path) -> str:
"""Translate a sandbox path to its real host path in unsafe mode."""
value = str(raw)
if self.backend == "bwrap":
return value
node = Path(shutil.which("node") or "/usr/bin/node").resolve()
mappings = [
*(self.read_only_mounts),
ReadOnlyMount(node.parent.parent, SANDBOX_NODE_PREFIX),
ReadOnlyMount(node, SANDBOX_NODE),
ReadOnlyMount(self.claude_host_bin, SANDBOX_CLAUDE),
ReadOnlyMount(self.clone, SANDBOX_WORKSPACE),
ReadOnlyMount(self.home, SANDBOX_HOME),
ReadOnlyMount(self.temp, SANDBOX_TMP),
]
for mount in sorted(mappings, key=lambda item: len(item.target), reverse=True):
target = mount.target.rstrip("/")
if value == target:
return str(mount.source)
if value.startswith(f"{target}/"):
return str(mount.source / value[len(target) + 1 :])
return value
def host_text(self, value: str) -> str:
if self.backend == "bwrap":
return value
targets = [
*(mount.target for mount in self.read_only_mounts),
SANDBOX_NODE_PREFIX,
SANDBOX_NODE,
SANDBOX_CLAUDE,
SANDBOX_WORKSPACE,
SANDBOX_HOME,
SANDBOX_TMP,
]
ordered = sorted(set(targets), key=len, reverse=True)
pattern = re.compile("|".join(re.escape(target) for target in ordered))
return pattern.sub(lambda match: self.host_path(match.group(0)), value)
@property
def claude_bin(self) -> str:
return SANDBOX_CLAUDE
return self.host_path(SANDBOX_CLAUDE)
@property
def transcript_projects(self) -> Path:
@ -96,7 +141,7 @@ class SandboxSession:
@property
def settings_json(self) -> str:
return build_claude_settings()
return build_claude_settings(sandbox_enabled=self.backend == "bwrap")
def environment(
self,
@ -104,7 +149,17 @@ class SandboxSession:
auth_token: str | None = None,
base_url: str | None = None,
) -> dict[str, str]:
return build_sandbox_environment(auth_token=auth_token, base_url=base_url)
env = build_sandbox_environment(auth_token=auth_token, base_url=base_url)
if self.backend != "bwrap":
env = {key: self.host_text(value) for key, value in env.items()}
env.pop("CLAUDE_CODE_SUBPROCESS_ENV_SCRUB", None)
env.pop("CLAUDE_CODE_SHELL_PREFIX", None)
# Sandbox-only bin dirs have no single host counterpart; drop the
# entries that survive translation as non-existent paths.
env["PATH"] = ":".join(
entry for entry in env["PATH"].split(":") if Path(entry).is_dir()
)
return env
def run(
self,
@ -114,11 +169,17 @@ class SandboxSession:
env: Mapping[str, str] | None = None,
stdin_data: bytes | None = None,
) -> ManagedProcessResult:
translated = [self.host_text(str(part)) for part in command]
return run_managed(
[*self.command_prefix, *command],
[*self.command_prefix, *translated],
cwd=None if self.command_prefix else self.clone,
timeout=timeout,
env=dict(env) if env is not None else build_sandbox_environment(),
require_pid_namespace=True,
env=(
{key: self.host_text(value) for key, value in env.items()}
if env is not None
else self.environment()
),
require_pid_namespace=self.require_pid_namespace,
stdin_data=stdin_data,
)
@ -139,6 +200,9 @@ class SandboxSession:
for harness-owned, post-session evidence such as hidden oracles.
"""
if self.backend != "bwrap":
return []
additional: list[ReadOnlyMount] = []
clone = _real_directory(self.clone, label="sandbox clone")
for raw_path in read_only_paths:
@ -347,7 +411,7 @@ def build_sandbox_environment(
return env
def build_claude_settings() -> str:
def build_claude_settings(*, sandbox_enabled: bool = True) -> str:
"""Inline settings that keep every Bash sandboxed and pre-approve the tools.
Deliberately hook-free: headless ``claude -p`` (2.1.247) never dispatches
@ -358,11 +422,16 @@ def build_claude_settings() -> str:
below, ``--tools``/``--allowedTools``, and the bwrap mounts.
"""
permissions = {
"allow": ["Read", "Grep", "Glob", "Bash"],
}
if sandbox_enabled:
permissions["disableBypassPermissionsMode"] = "disable"
settings = {
"sandbox": {
"enabled": True,
"failIfUnavailable": True,
"autoAllowBashIfSandboxed": True,
"enabled": sandbox_enabled,
"failIfUnavailable": sandbox_enabled,
"autoAllowBashIfSandboxed": sandbox_enabled,
"allowUnsandboxedCommands": False,
"enableWeakerNestedSandbox": True,
"network": {
@ -398,13 +467,16 @@ def build_claude_settings() -> str:
# allow rule, so pre-approve the proposer's exact tool surface. Bash
# is the only writable tool under --bare (it writes the candidate
# overlay) and stays sandbox-confined by the sandbox.* policy above.
"allow": ["Read", "Grep", "Glob", "Bash"],
"disableBypassPermissionsMode": "disable",
},
"env": {
"CLAUDE_CODE_SUBPROCESS_ENV_SCRUB": "1",
"CLAUDE_CODE_DONT_INHERIT_ENV": "1",
**permissions,
},
"env": (
{
"CLAUDE_CODE_SUBPROCESS_ENV_SCRUB": "1",
"CLAUDE_CODE_DONT_INHERIT_ENV": "1",
}
if sandbox_enabled
else {"CLAUDE_CODE_DONT_INHERIT_ENV": "1"}
),
}
return json.dumps(settings, sort_keys=True, separators=(",", ":"))
@ -570,6 +642,12 @@ def preflight_bubblewrap(bwrap_bin: Path | str | None = None) -> Path:
return bwrap
def preflight_unsafe_host() -> Path:
"""Return a harmless sentinel for the explicit non-containment backend."""
return _resolve_executable(None, "env")
def pid_namespace_command(
command: Sequence[str],
*,
@ -778,6 +856,145 @@ def _sandbox_command_prefix(
return args
def _drop_host_workspace_write_bits(
root: Path,
*,
writable: Sequence[Path],
) -> list[tuple[Path, int]]:
"""Clear write bits on a host-unsafe clone except explicit artifact paths."""
root = root.expanduser().absolute()
try:
metadata = root.lstat()
resolved = root.resolve(strict=True)
except OSError as exc:
raise SandboxError(f"host workspace lock root is unavailable: {root}: {exc}") from exc
if stat.S_ISLNK(metadata.st_mode) or not stat.S_ISDIR(metadata.st_mode) or resolved != root:
raise SandboxError(f"host workspace lock root must be a real directory: {root}")
allowed: set[Path] = set()
for raw in writable:
path = raw.expanduser().absolute()
try:
path.relative_to(root)
except ValueError as exc:
raise SandboxError(f"writable host path escapes the workspace: {raw}") from exc
allowed.add(path)
records: list[tuple[Path, int]] = []
pending = [root]
while pending:
current = pending.pop()
try:
metadata = current.lstat()
except OSError as exc:
raise SandboxError(f"host workspace lock path is unreadable: {current}: {exc}") from exc
records.append((current, stat.S_IMODE(metadata.st_mode)))
if stat.S_ISLNK(metadata.st_mode):
continue
if stat.S_ISDIR(metadata.st_mode):
try:
children = [Path(entry.path) for entry in os.scandir(current)]
except OSError as exc:
raise SandboxError(f"host workspace lock directory is unreadable: {current}: {exc}") from exc
pending.extend(children)
if current not in allowed:
os.chmod(current, 0o555)
continue
if current in allowed:
os.chmod(current, stat.S_IMODE(metadata.st_mode) | 0o222)
continue
if stat.S_ISREG(metadata.st_mode):
os.chmod(current, stat.S_IMODE(metadata.st_mode) & ~0o222)
return records
def _force_rmtree(path: Path) -> None:
"""Delete a tree even when leftover copies inherited 0555 directory modes.
Host-unsafe review sessions drop write bits on the clone. An agent that
``copytree``s those directories into the private TMPDIR leaves a tree
``shutil.rmtree`` cannot remove: a non-empty 0555 directory raises
PermissionError. Restore owner write bits, then delete.
"""
root = Path(path)
if not root.exists():
return
for dirpath, _dirnames, filenames in os.walk(root, topdown=False, followlinks=False):
try:
os.chmod(
dirpath,
stat.S_IMODE(os.lstat(dirpath).st_mode) | 0o700,
follow_symlinks=False,
)
except OSError:
pass
for name in filenames:
child = os.path.join(dirpath, name)
try:
metadata = os.lstat(child)
except OSError:
continue
if stat.S_ISLNK(metadata.st_mode):
continue
try:
os.chmod(child, stat.S_IMODE(metadata.st_mode) | 0o200, follow_symlinks=False)
except OSError:
pass
shutil.rmtree(root)
def _restore_host_workspace_modes(records: Sequence[tuple[Path, int]]) -> None:
errors: list[str] = []
for path, mode in reversed(records):
try:
os.chmod(path, mode)
except FileNotFoundError:
continue
except OSError as exc:
errors.append(f"{path}: {exc}")
if errors:
raise SandboxError("failed to restore host workspace modes: " + "; ".join(errors[:8]))
@contextmanager
def host_workspace_write_boundary(
root: Path,
*,
writable: Sequence[Path] = (),
) -> Iterator[None]:
"""Best-effort host analog of bwrap ``--ro-bind`` plus one writable artifact.
This is not a security boundary: a session that can ``chmod`` can undo it.
It is enough to stop accidental ``npm install`` / analyze writes from
invalidating review evidence on a host-unsafe diagnostic run.
"""
records = _drop_host_workspace_write_bits(root, writable=writable)
try:
yield
finally:
_restore_host_workspace_modes(records)
@contextmanager
def sandbox_workspace_write_boundary(
sandbox: Any,
*,
read_only_workspace: bool,
writable: Sequence[Path] = (),
) -> Iterator[None]:
"""Apply the host-unsafe write lock only when bwrap is not the backend."""
if getattr(sandbox, "backend", "bwrap") != "host-unsafe" or not read_only_workspace:
yield
return
clone = Path(sandbox.clone)
with host_workspace_write_boundary(clone, writable=writable):
yield
@contextmanager
def prepare_sandbox(
*,
@ -786,17 +1003,21 @@ def prepare_sandbox(
bwrap_bin: Path | str | None = None,
read_only_mounts: Sequence[ReadOnlyMount] = (),
preflight: bool = True,
backend: str = "bwrap",
) -> Iterator[SandboxSession]:
"""Create private host backing dirs and one immutable Bubblewrap command."""
"""Create private host backing dirs and one virtualized command."""
# Validate the lexical path before resolving it. Resolving first would
# erase the evidence that the caller supplied a symlinked clone root.
clone = _real_directory(clone, label="sandbox clone")
if backend not in ("bwrap", "host-unsafe"):
raise SandboxError(f"unknown sandbox backend: {backend}")
if preflight:
bwrap = preflight_bubblewrap(bwrap_bin)
require_claude_sandbox_helpers()
bwrap = preflight_bubblewrap(bwrap_bin) if backend == "bwrap" else preflight_unsafe_host()
if backend == "bwrap":
require_claude_sandbox_helpers()
else:
bwrap = _resolve_executable(bwrap_bin, "bwrap")
bwrap = _resolve_executable(bwrap_bin, "bwrap") if backend == "bwrap" else preflight_unsafe_host()
claude = _resolve_executable(claude_bin, "claude")
private_root = Path(tempfile.mkdtemp(prefix="wfbench-sandbox-"))
private_root.chmod(0o700)
@ -825,13 +1046,17 @@ def prepare_sandbox(
)
primary: BaseException | None = None
try:
command_prefix = _sandbox_command_prefix(
bwrap=bwrap,
clone=clone,
home=home,
temp=temp,
claude_bin=claude,
mounts=protected_mounts,
command_prefix = (
_sandbox_command_prefix(
bwrap=bwrap,
clone=clone,
home=home,
temp=temp,
claude_bin=claude,
mounts=protected_mounts,
)
if backend == "bwrap"
else []
)
yield SandboxSession(
private_root=private_root,
@ -839,6 +1064,7 @@ def prepare_sandbox(
home=home,
temp=temp,
bwrap_bin=bwrap,
backend=backend,
claude_host_bin=claude,
command_prefix=command_prefix,
read_only_mounts=protected_mounts,
@ -848,7 +1074,7 @@ def prepare_sandbox(
raise
finally:
try:
shutil.rmtree(private_root)
_force_rmtree(private_root)
except OSError as cleanup:
if primary is None:
raise

View file

@ -15,6 +15,7 @@
# MODEL PROPOSER_MODEL EFFORT GENERATIONS RUNS WORKERS PROVIDER
# EVOLUTION_PROFILE CE_PLUGIN_DIR CE_PLUGIN_VERSION
# INCLUDE_EXPENSIVE SEED_RESULTS CLAUDE_BIN OUT_ROOT
# UNSAFE_NO_BWRAP=1 (local review diagnostics only)
# GITNEXUS_BENCH_ANTHROPIC_API_KEY (legacy GITNEXUS_BENCH_AUTH_TOKEN)
# GITNEXUS_BENCH_OPENAI_API_KEY
set -euo pipefail
@ -181,6 +182,13 @@ fi
if [[ -n "${SEED_RESULTS}" ]]; then
cmd+=(--seed-results "${SEED_RESULTS}")
fi
if [[ -n "${UNSAFE_NO_BWRAP:-}" && "${UNSAFE_NO_BWRAP}" != "0" && "${UNSAFE_NO_BWRAP}" != "false" ]]; then
if [[ -n "${CI:-}" ]]; then
echo "UNSAFE_NO_BWRAP is forbidden in CI." >&2
exit 1
fi
cmd+=(--unsafe-no-bwrap)
fi
if ((${#passthrough[@]})); then
cmd+=("${passthrough[@]}")
fi
@ -200,13 +208,15 @@ runtime_digest="$(
sha256sum "${eval_dir}/../gitnexus-shared/package-lock.json"
} | sha256sum | cut -d' ' -f1
)"
SOURCE_SHA="${source_sha}" RUNTIME_DIGEST="${runtime_digest}" \
unsafe_backend="$([[ -n "${UNSAFE_NO_BWRAP:-}" && "${UNSAFE_NO_BWRAP}" != "0" && "${UNSAFE_NO_BWRAP}" != "false" ]] && echo host-unsafe || echo bwrap)"
SOURCE_SHA="${source_sha}" RUNTIME_DIGEST="${runtime_digest}" SANDBOX_BACKEND="${unsafe_backend}" \
node -e 'require("fs").writeFileSync(process.argv[1], JSON.stringify({
schema_version: 1,
source_sha: process.env.SOURCE_SHA,
runtime_digest: process.env.RUNTIME_DIGEST,
profile: process.env.EVOLUTION_PROFILE,
ce_plugin_version: process.env.CE_PLUGIN_VERSION
ce_plugin_version: process.env.CE_PLUGIN_VERSION,
sandbox_backend: process.env.SANDBOX_BACKEND
}, null, 2) + "\n")' "${out_root}/runtime-provenance.json"
export PYTHONUNBUFFERED=1

View file

@ -43,6 +43,7 @@ import re
import secrets
import stat
import statistics
import sys
import tempfile
import time
from collections.abc import Callable, Mapping, Sequence
@ -68,6 +69,7 @@ from .evolution import (
evaluate_candidate,
evaluate_review_candidate,
required_candidate_arms,
seed_evaluated_skills,
skill_fingerprint,
)
from .model_gateway import (
@ -83,6 +85,7 @@ from .oracle_assets import (
capture_task_oracles,
sanitize_clone_for_hidden_oracles,
staged_task_oracle,
with_hidden_harness_apply_exclude,
)
from .process_control import ManagedProcessError
from .promotion_apply import committed_destination_base_digests
@ -98,9 +101,11 @@ from .proposer_sandbox import (
SandboxSession,
build_sandbox_environment,
preflight_bubblewrap,
preflight_unsafe_host,
prepare_sandbox,
redact_text,
require_claude_sandbox_helpers,
sandbox_workspace_write_boundary,
)
from .runner_artifacts import (
IMPLEMENTATION_ARMS,
@ -329,7 +334,7 @@ def _run_hidden_oracle(
extra_read_only_mounts=(ReadOnlyMount(source=stage_root, target=oracle_mount),),
),
env=oracle_env,
require_pid_namespace=True,
require_pid_namespace=getattr(sandbox, "require_pid_namespace", True),
)
)
# Candidate code executes in this process. Never persist its stdout
@ -423,11 +428,15 @@ def run_arm(
oracle_snapshot: TaskOracleSnapshot | None = None,
) -> dict[str, Any]:
sessions: list[dict[str, Any]] = []
environment_builder = getattr(sandbox, "environment", build_sandbox_environment)
host_text = getattr(sandbox, "host_text", lambda value: value)
host_path = getattr(sandbox, "host_path", lambda value: str(value))
backend = getattr(sandbox, "backend", "bwrap")
env = model_session_environment(
auth_token=args.auth_token,
base_url=args.base_url,
model=args.model,
build_sandbox_environment=build_sandbox_environment,
build_sandbox_environment=environment_builder,
)
# --bare hard-disables the Skill tool and every mcp__* tool — by Claude
# Code design, not a bug (--allowedTools can't restore what --bare
@ -444,15 +453,17 @@ def run_arm(
"model": args.model,
"effort": args.effort,
"env": env,
"permission_mode": "dontAsk",
"permission_mode": (
"bypassPermissions" if backend == "host-unsafe" else "dontAsk"
),
"command_prefix": sandbox.command_prefix_for(
read_only_paths=_evaluated_skill_roots(worktree, arm),
),
"require_pid_namespace": True,
"require_pid_namespace": getattr(sandbox, "require_pid_namespace", True),
"bare": bare,
"settings_json": sandbox.settings_json,
"strict_mcp_config": True,
"mcp_config_json": sandbox_mcp_config(),
"mcp_config_json": host_text(sandbox_mcp_config()),
"transcript_projects": sandbox.transcript_projects,
"transcript_cwd": Path(SANDBOX_WORKSPACE),
"transcript_wait_seconds": 5,
@ -461,7 +472,7 @@ def run_arm(
"transcript_secrets": tuple(credential_secrets(args)),
}
if ce_plugin_dir is not None:
common["plugin_dirs"] = (ce_plugin_dir,)
common["plugin_dirs"] = (host_path(ce_plugin_dir),)
expected_skills = ARM_EXPECTED_SKILLS.get(arm, ())
plan_doc: Path | None = None
if arm in ("workflow", "ce_workflow"):
@ -538,7 +549,7 @@ def run_arm(
phase_before = workspace_snapshot(worktree) if enforce_phase_boundary else None
review_common = {
**common,
"allowed_tools": allowed_agent_tools(implementation=False),
"allowed_tools": allowed_agent_tools(implementation=False, allow_edit=False),
"command_prefix": sandbox.command_prefix_for(
read_only_workspace=True,
read_only_paths=_evaluated_skill_roots(worktree, arm),
@ -550,12 +561,17 @@ def run_arm(
),
),
}
review_session = run_claude(
review_prompt.format(task=task["prompt"]),
worktree,
expected_skill=expected_skills[0],
**review_common,
)
with sandbox_workspace_write_boundary(
sandbox,
read_only_workspace=True,
writable=(review_output,),
):
review_session = run_claude(
host_text(review_prompt.format(task=task["prompt"])),
worktree,
expected_skill=expected_skills[0],
**review_common,
)
sessions.append(review_session)
if review_session["ok"] and phase_before is not None:
try:
@ -627,8 +643,8 @@ def run_arm(
read_only_workspace=True,
unshare_network=True,
),
env=build_sandbox_environment(),
require_pid_namespace=True,
env=environment_builder(),
require_pid_namespace=getattr(sandbox, "require_pid_namespace", True),
)
)
review_score: dict[str, Any] | None = None
@ -849,6 +865,7 @@ class TaskCellContext:
runtime_mounts: tuple[ReadOnlyMount, ...]
candidate_overlay: Path | None
overlay_digest: str | None
sandbox_backend: str = "bwrap"
def run_cell(ctx: TaskCellContext, run_idx: int, arm: str) -> dict[str, Any]:
@ -909,6 +926,7 @@ def run_cell(ctx: TaskCellContext, run_idx: int, arm: str) -> dict[str, Any]:
*oracle_visibility_mounts,
],
preflight=False,
backend=ctx.sandbox_backend,
) as sandbox:
# Capture the BASE (pre-overlay) skill digest — identical
# for the incumbent and candidate arms — then run the
@ -916,9 +934,22 @@ def run_cell(ctx: TaskCellContext, run_idx: int, arm: str) -> dict[str, Any]:
# candidate overlay is applied only afterwards, so setup
# can never observe candidate prose and both arms share
# byte-identical pre-overlay state.
# Historical review SHAs may predate gitnexus-review. Seed
# the current evaluated skill first so fingerprinting and
# the model see the same incumbent prose on every case.
if execution_arm == "review":
seed_evaluated_skills(
HARNESS_ROOT,
worktree,
sandbox=sandbox,
arm=execution_arm,
)
base_skill_digest = skill_fingerprint(worktree, execution_arm)
if task.get("setup"):
setup_command = ["/bin/sh", "-lc", str(task["setup"])]
# Sanitization already removed eval/workflow_bench. A historical
# PR patch that still edits that tree (gitignored learnings.jsonl)
# must skip those hunks or `git apply` fails closed.
setup_command = ["/bin/sh", "-lc", with_hidden_harness_apply_exclude(str(task["setup"]))]
setup = sandbox.run(
setup_command,
timeout=600,
@ -1058,6 +1089,7 @@ def run_cell(ctx: TaskCellContext, run_idx: int, arm: str) -> dict[str, Any]:
"task": task["id"],
"class": task.get("class", ""),
"run": run_idx,
"sandbox_backend": ctx.sandbox_backend,
"task_asset_snapshot_digest": (ctx.asset_snapshot.digest if ctx.asset_snapshot is not None else None),
"task_asset_manifest_digest": (
ctx.asset_snapshot.manifest_digest if ctx.asset_snapshot is not None else None
@ -1498,6 +1530,7 @@ def build_parser() -> argparse.ArgumentParser:
parser.add_argument("--promotion-max-task-regression", type=float, default=20.0)
parser.add_argument("--task-bindings-json", default=None, help=argparse.SUPPRESS)
parser.add_argument("--promotion-target-bases-json", default=None, help=argparse.SUPPRESS)
parser.add_argument("--unsafe-no-bwrap", action="store_true", help=argparse.SUPPRESS)
return parser
@ -1574,9 +1607,24 @@ def main() -> None:
parser.error("--promotion-target-bases-json requires --candidate-overlay")
promotion_target_bases = {}
if args.unsafe_no_bwrap and os.environ.get("CI"):
parser.error("--unsafe-no-bwrap is forbidden when CI is set")
if args.unsafe_no_bwrap and args.arms != ["ce_review", "review", "candidate_review"]:
parser.error("--unsafe-no-bwrap is restricted to the paired review arms")
try:
bwrap_bin = preflight_bubblewrap()
require_claude_sandbox_helpers()
if args.unsafe_no_bwrap:
bwrap_bin = preflight_unsafe_host()
sandbox_backend = "host-unsafe"
print(
"WARNING: --unsafe-no-bwrap runs sessions directly on the host with no "
"containment; model and verifier processes can access the host filesystem, "
"network, and credentials.",
file=sys.stderr,
)
else:
bwrap_bin = preflight_bubblewrap()
sandbox_backend = "bwrap"
require_claude_sandbox_helpers()
runtime_mounts = trusted_gitnexus_runtime_mounts()
except SandboxError as exc:
parser.error(str(exc))
@ -1597,6 +1645,7 @@ def main() -> None:
expected_task_bindings=expected_task_bindings,
ce_plugin_config=ce_plugin_config,
bwrap_bin=bwrap_bin,
sandbox_backend=sandbox_backend,
runtime_mounts=runtime_mounts,
candidate_arms=candidate_arms,
candidate_overlay=candidate_overlay,
@ -1617,6 +1666,7 @@ def _run_sweep(
expected_task_bindings: Any,
ce_plugin_config: Any,
bwrap_bin: Any,
sandbox_backend: str,
runtime_mounts: Any,
candidate_arms: list[str],
candidate_overlay: Path | None,
@ -1691,6 +1741,7 @@ def _run_sweep(
cache=task_asset_cache,
claude_bin=args.claude_bin,
bwrap_bin=bwrap_bin,
sandbox_backend=sandbox_backend,
runtime_mounts=runtime_mounts,
)
graph_snapshots[graph_key] = graph_snapshot
@ -1724,6 +1775,7 @@ def _run_sweep(
ce_plugin_snapshot=ce_plugin_snapshot,
trees_dir=Path(trees),
bwrap_bin=bwrap_bin,
sandbox_backend=sandbox_backend,
runtime_mounts=runtime_mounts,
candidate_overlay=candidate_overlay,
overlay_digest=overlay_digest,

View file

@ -361,8 +361,13 @@ def sandbox_mcp_config() -> str:
return json.dumps(config, sort_keys=True, separators=(",", ":"))
def allowed_agent_tools(*, implementation: bool, include_mcp: bool = True) -> list[str]:
tools = [*BUILTIN_AGENT_TOOLS]
def allowed_agent_tools(
*,
implementation: bool,
include_mcp: bool = True,
allow_edit: bool = True,
) -> list[str]:
tools = [tool for tool in BUILTIN_AGENT_TOOLS if allow_edit or tool != "Edit"]
if include_mcp:
tools.extend(GITNEXUS_READ_ONLY_TOOLS)
if include_mcp and implementation:
@ -461,7 +466,13 @@ def _persist_parent_event_stream(
def _normalized_skill_identifier(value: Any) -> str | None:
"""Return the exact identifier token accepted by the Skill tool."""
"""Return the exact identifier token accepted by the Skill tool.
Plugin skills are requested as ``plugin:skill`` (observed in review
evolution: ``compound-engineering:ce-code-review``). Compare on the
skill token after the last colon so a successful plugin-qualified
invocation still counts as the expected skill.
"""
if not isinstance(value, str):
return None
@ -471,6 +482,8 @@ def _normalized_skill_identifier(value: Any) -> str | None:
token = stripped.split(maxsplit=1)[0]
if token.startswith("/"):
token = token[1:]
if ":" in token:
token = token.rsplit(":", 1)[-1]
return token or None

View file

@ -146,24 +146,91 @@ def _validated_runtime_root(path: Path, *, label: str) -> Path:
return root
def _primary_checkout_root(harness_root: Path) -> Path | None:
"""Return the main worktree when *harness_root* is a linked git worktree.
Linked worktrees store a regular ``.git`` file pointing at
``<primary>/.git/worktrees/<name>``. A symlink or oversized file is
ignored so this helper cannot be used to follow an unexpected path.
"""
git_file = harness_root / ".git"
try:
metadata = git_file.lstat()
except OSError:
return None
if stat.S_ISLNK(metadata.st_mode) or not stat.S_ISREG(metadata.st_mode):
return None
if metadata.st_size > 4096:
return None
try:
text = git_file.read_text(encoding="utf-8").strip()
except (OSError, UnicodeError):
return None
if not text.startswith("gitdir:"):
return None
raw = text.split(":", 1)[1].strip()
if not raw:
return None
gitdir = Path(raw)
if not gitdir.is_absolute():
gitdir = harness_root / gitdir
if gitdir.parent.name != "worktrees" or gitdir.parent.parent.name != ".git":
return None
primary = gitdir.parent.parent.parent
try:
primary_git = (primary / ".git").lstat()
except OSError:
return None
if stat.S_ISLNK(primary_git.st_mode) or not stat.S_ISDIR(primary_git.st_mode):
return None
resolved_primary = primary.resolve()
if resolved_primary == harness_root.resolve():
return None
return resolved_primary
def _validated_runtime_component(
root: Path,
relative: str,
target: str,
*,
directory: bool,
allow_primary_worktree_symlink: bool = False,
) -> ReadOnlyMount:
"""Validate one direct runtime component before exposing only that path."""
source = root / relative
kind = "directory" if directory else "file"
try:
mode = source.lstat().st_mode
resolved = source.resolve(strict=True)
except OSError as exc:
raise SandboxError(f"pinned GitNexus runtime component is unavailable: {source}: {exc}") from exc
if stat.S_ISLNK(mode) and allow_primary_worktree_symlink and directory:
primary = _primary_checkout_root(HARNESS_ROOT)
if primary is None:
raise SandboxError(f"pinned GitNexus runtime component must be a real {kind}: {source}")
expected = primary / "gitnexus" / relative
try:
expected_mode = expected.lstat().st_mode
expected_resolved = expected.resolve(strict=True)
except OSError as exc:
raise SandboxError(
f"pinned GitNexus runtime component symlink must point at the primary checkout: {source}"
) from exc
if (
stat.S_ISLNK(expected_mode)
or not stat.S_ISDIR(expected_mode)
or expected_resolved != expected
or resolved != expected_resolved
):
raise SandboxError(
f"pinned GitNexus runtime component symlink must point at the primary checkout: {source}"
)
return ReadOnlyMount(source=expected_resolved, target=target)
expected_type = stat.S_ISDIR(mode) if directory else stat.S_ISREG(mode)
if stat.S_ISLNK(mode) or not expected_type or resolved != source:
kind = "directory" if directory else "file"
raise SandboxError(f"pinned GitNexus runtime component must be a real {kind}: {source}")
return ReadOnlyMount(source=source, target=target)
@ -197,6 +264,7 @@ def trusted_gitnexus_runtime_mounts() -> tuple[ReadOnlyMount, ...]:
"node_modules",
f"{SANDBOX_GITNEXUS}/node_modules",
directory=True,
allow_primary_worktree_symlink=True,
),
_validated_runtime_component(
runtime,
@ -236,7 +304,19 @@ def trusted_gitnexus_runtime_mounts() -> tuple[ReadOnlyMount, ...]:
raise SandboxError("pinned GitNexus runtime package.json has no version")
linked_shared = mounts[2].source / "gitnexus-shared"
if not linked_shared.is_symlink() or linked_shared.resolve(strict=True) != shared:
allowed_shared = {shared}
primary = _primary_checkout_root(HARNESS_ROOT)
if primary is not None:
try:
allowed_shared.add(
_validated_runtime_root(
primary / "gitnexus-shared",
label="primary GitNexus shared runtime",
)
)
except SandboxError:
pass
if not linked_shared.is_symlink() or linked_shared.resolve(strict=True) not in allowed_shared:
raise SandboxError("pinned GitNexus runtime has an unexpected gitnexus-shared dependency")
try:
shared_package = json.loads(mounts[5].source.read_text())

View file

@ -20,11 +20,12 @@ from .proposer_sandbox import (
SANDBOX_WORKSPACE,
ReadOnlyMount,
SandboxError,
SandboxSession,
build_sandbox_environment,
prepare_sandbox,
)
from .runner_artifacts import make_worktree, remove_clone
from .task_assets import TaskAssetCache, TaskAssetSnapshot
from .task_assets import TaskAssetCache, TaskAssetSnapshot, _is_harness_sandbox_copy
GRAPH_ASSET_PATHS = (
".gitnexus/gitnexus.json",
@ -41,6 +42,12 @@ GRAPH_MARKERS = (
)
GRAPH_BUILD_TIMEOUT_SECONDS = 3600
GRAPH_QUERY_TIMEOUT_SECONDS = 300
# CLI default is 5s. Parse-worker top-of-script init loads every required
# tree-sitter binding before it can post `{type:'ready'}`; on a loaded WSL
# host that handshake is >15s, and a GitNexus-sized --pdg analyze can slow it
# further. Integration tests already stub 60s; graph prep uses 120s so a
# replacement worker is not classified as deterministic-startup (#2649).
GRAPH_WORKER_READY_TIMEOUT_MS = 120_000
MAX_GRAPH_SCRUB_ENTRIES = 250_000
MAX_GRAPH_SCRUB_FILE_BYTES = 512 * 1024
MAX_GRAPH_SCRUB_TOTAL_BYTES = 2 * 1024 * 1024 * 1024
@ -88,7 +95,13 @@ def validate_no_prebuilt_graph_assets(task: Mapping[str, Any]) -> None:
if not isinstance(sandbox_copy, list):
raise SandboxError("sandbox_copy must be a list")
for value in sandbox_copy:
if isinstance(value, str) and _is_restricted_path(value):
if not isinstance(value, str):
continue
# Review corpus patches are harness-owned and applied in setup, then
# deleted with eval/workflow_bench. They are not a prebuilt graph.
if _is_harness_sandbox_copy(PurePosixPath(value)):
continue
if _is_restricted_path(value):
raise SandboxError(f"sandbox_copy cannot import prebuilt graph or harness data: {value}")
dependencies = task.get("sandbox_dependencies", [])
@ -235,6 +248,7 @@ def _graph_environment() -> dict[str, str]:
"GITNEXUS_NO_GITIGNORE": "1",
"GITNEXUS_WORKER_POOL_SIZE": "1",
"GITNEXUS_PARSE_CHUNK_CONCURRENCY": "1",
"GITNEXUS_WORKER_READY_TIMEOUT_MS": str(GRAPH_WORKER_READY_TIMEOUT_MS),
}
)
return env
@ -244,20 +258,23 @@ def _run_graph_cli(
prefix: Sequence[str],
arguments: Sequence[str],
*,
sandbox: SandboxSession | None = None,
timeout: int,
capture_stdout: bool = False,
) -> bytes | None:
host_path = sandbox.host_path if sandbox is not None else str
host_text = sandbox.host_text if sandbox is not None else str
command = [
*prefix,
SANDBOX_NODE,
SANDBOX_GITNEXUS_ENTRYPOINT,
*arguments,
host_path(SANDBOX_NODE),
host_path(SANDBOX_GITNEXUS_ENTRYPOINT),
*(host_text(argument) for argument in arguments),
]
result = run_managed(
command,
timeout=timeout,
env=_graph_environment(),
require_pid_namespace=True,
env={key: host_text(value) for key, value in _graph_environment().items()},
require_pid_namespace=(sandbox.require_pid_namespace if sandbox is not None else True),
capture_stdout_bytes=(2 * 1024 * 1024 if capture_stdout else None),
)
if not result.ok:
@ -286,12 +303,14 @@ def _parse_empty_query(raw: bytes, *, label: str) -> None:
raise SandboxError(f"{label} found recoverable benchmark harness references")
def _scrub_and_verify_graph(prefix: Sequence[str]) -> None:
def _scrub_and_verify_graph(prefix: Sequence[str], sandbox: SandboxSession | None = None) -> None:
node_predicate = _marker_predicate("n")
relation_predicate = _marker_predicate("r")
sandbox_kwargs = {"sandbox": sandbox} if sandbox is not None else {}
node_result = _run_graph_cli(
prefix,
("cypher", f"MATCH (n) WHERE {node_predicate} RETURN n LIMIT 1", "-r", "benchmark-target", "--limit", "1"),
**sandbox_kwargs,
timeout=GRAPH_QUERY_TIMEOUT_SECONDS,
capture_stdout=True,
)
@ -305,6 +324,7 @@ def _scrub_and_verify_graph(prefix: Sequence[str]) -> None:
"--limit",
"1",
),
**sandbox_kwargs,
timeout=GRAPH_QUERY_TIMEOUT_SECONDS,
capture_stdout=True,
)
@ -339,6 +359,7 @@ def prepare_sanitized_graph(
claude_bin: Path | str,
bwrap_bin: Path | str,
runtime_mounts: Sequence[ReadOnlyMount],
sandbox_backend: str = "bwrap",
) -> SanitizedGraphSnapshot:
"""Sanitize, index offline once, scrub, and freeze graph assets for all arms."""
@ -355,8 +376,10 @@ def prepare_sanitized_graph(
bwrap_bin=bwrap_bin,
read_only_mounts=runtime_mounts,
preflight=False,
backend=sandbox_backend,
) as sandbox:
prefix = sandbox.command_prefix_for(unshare_network=True)
unsafe_sandbox = sandbox if getattr(sandbox, "backend", "bwrap") == "host-unsafe" else None
_run_graph_cli(
prefix,
(
@ -375,9 +398,10 @@ def prepare_sanitized_graph(
"--workers",
"1",
),
**({"sandbox": unsafe_sandbox} if unsafe_sandbox is not None else {}),
timeout=GRAPH_BUILD_TIMEOUT_SECONDS,
)
_scrub_and_verify_graph(prefix)
_scrub_and_verify_graph(prefix, unsafe_sandbox)
_validate_graph_metadata(seed, sanitized_head)
assets = cache.prepare(
{"sandbox_copy": list(GRAPH_ASSET_PATHS)},

View file

@ -24,6 +24,7 @@ from dataclasses import dataclass
from pathlib import Path, PurePosixPath
from typing import Any
from . import runtime_mounts
from .proposer_sandbox import (
DEPENDENCY_MOUNT_BASENAME,
SANDBOX_WORKSPACE,
@ -63,6 +64,10 @@ _REFLINK_UNAVAILABLE = {
DEPENDENCY_CONTENT_BINDING_FIELD = "sandbox_dependency_content_digest"
DEPENDENCY_MANIFEST_BINDING_FIELD = "sandbox_dependency_manifest_digest"
# Review corpus patches are harness-owned. The task repo (`~/GitNexus`) is the
# subject checkout and often a different worktree or branch, so these paths
# must be read from the package that defined the benchmark.
HARNESS_SANDBOX_COPY_PREFIXES = (PurePosixPath("eval/workflow_bench/review_cases"),)
@dataclass(frozen=True)
@ -191,7 +196,7 @@ class TaskAssetCache:
self.root = root.expanduser().absolute()
self.root.mkdir(mode=0o700, parents=True, exist_ok=False)
self._by_definition: dict[
tuple[str, str, tuple[str, ...], tuple[tuple[str, str], ...]],
tuple[str, str, tuple[str, ...], tuple[tuple[str, str], ...], str],
TaskAssetSnapshot,
] = {}
self._closed = False
@ -218,7 +223,14 @@ class TaskAssetCache:
declarations, relative_paths = _sandbox_copy_declarations(task)
dependency_declarations = _sandbox_dependency_declarations(task)
dependency_identity = tuple((declaration.source, declaration.target) for declaration in dependency_declarations)
definition = (str(repo_identity), resolved_sha, declarations, dependency_identity)
harness_identity = _harness_sandbox_copy_identity(relative_paths)
definition = (
str(repo_identity),
resolved_sha,
declarations,
dependency_identity,
harness_identity,
)
existing = self._by_definition.get(definition)
if existing is not None:
if expected_dependency_binding is not None:
@ -238,9 +250,22 @@ class TaskAssetCache:
repo_identity,
os.O_RDONLY | os.O_DIRECTORY | getattr(os, "O_CLOEXEC", 0) | getattr(os, "O_NOFOLLOW", 0),
)
harness_fd: int | None = None
try:
for relative in relative_paths:
descriptor = _open_relative(repo_fd, relative)
if _is_harness_sandbox_copy(relative):
if harness_fd is None:
harness_root = _harness_sandbox_copy_root()
harness_fd = os.open(
harness_root,
os.O_RDONLY
| os.O_DIRECTORY
| getattr(os, "O_CLOEXEC", 0)
| getattr(os, "O_NOFOLLOW", 0),
)
descriptor = _open_relative(harness_fd, relative)
else:
descriptor = _open_relative(repo_fd, relative)
try:
builder.copy_descriptor(descriptor, relative)
finally:
@ -299,6 +324,8 @@ class TaskAssetCache:
)
finally:
os.close(repo_fd)
if harness_fd is not None:
os.close(harness_fd)
entries = builder.finished_entries()
manifest_digest = _manifest_digest(entries)
@ -317,6 +344,7 @@ class TaskAssetCache:
manifest_digest=manifest_digest,
dependency_content_digest=dependency_content_digest,
dependency_manifest_digest=dependency_manifest_digest,
harness_identity=harness_identity,
)
destination = self.root / digest
if destination.exists():
@ -537,6 +565,22 @@ class _SnapshotBuilder:
return tuple(sorted(self.entries.values(), key=lambda entry: entry.path.as_posix()))
def _is_harness_sandbox_copy(relative: PurePosixPath) -> bool:
if relative.is_absolute() or not relative.parts or ".." in relative.parts:
return False
return any(relative == prefix or prefix in relative.parents for prefix in HARNESS_SANDBOX_COPY_PREFIXES)
def _harness_sandbox_copy_root() -> Path:
return _real_directory(runtime_mounts.HARNESS_ROOT, label="harness sandbox_copy root")
def _harness_sandbox_copy_identity(relative_paths: tuple[PurePosixPath, ...]) -> str:
if not any(_is_harness_sandbox_copy(relative) for relative in relative_paths):
return ""
return str(_harness_sandbox_copy_root())
def _sandbox_copy_declarations(
task: Mapping[str, Any],
) -> tuple[tuple[str, ...], tuple[PurePosixPath, ...]]:
@ -969,15 +1013,17 @@ def _snapshot_digest(
manifest_digest: str,
dependency_content_digest: str,
dependency_manifest_digest: str,
harness_identity: str = "",
) -> str:
payload = {
"declarations": declarations,
"dependency_content_digest": dependency_content_digest,
"dependency_manifest_digest": dependency_manifest_digest,
"harness_identity": harness_identity,
"manifest_digest": manifest_digest,
"repo_identity": str(repo_identity),
"resolved_sha": resolved_sha,
"schema_version": 2,
"schema_version": 3,
}
return hashlib.sha256(json.dumps(payload, sort_keys=True, separators=(",", ":")).encode()).hexdigest()

View file

@ -7,7 +7,7 @@ tasks:
repo: ~/GitNexus
ref: ff86ccf1e79cd7e4175da437ae8aeaf67b64aaa1
sandbox_copy: [eval/workflow_bench/review_cases/pr-2718-defect.patch]
setup: git apply eval/workflow_bench/review_cases/pr-2718-defect.patch && rm -rf eval/workflow_bench
setup: git apply --exclude='eval/workflow_bench/*' eval/workflow_bench/review_cases/pr-2718-defect.patch && rm -rf eval/workflow_bench
prompt: Review the historical snapshot of https://github.com/abhigyanpatwari/GitNexus/pull/2718. Report only actionable defects introduced by the local diff.
verify: test -s review-output.json
oracle:
@ -22,7 +22,7 @@ tasks:
id: review-pr-2794-defect
ref: 911151e2304f298a995fcc69c738ad2c6db9393a
sandbox_copy: [eval/workflow_bench/review_cases/pr-2794-defect.patch]
setup: git apply eval/workflow_bench/review_cases/pr-2794-defect.patch && rm -rf eval/workflow_bench
setup: git apply --exclude='eval/workflow_bench/*' eval/workflow_bench/review_cases/pr-2794-defect.patch && rm -rf eval/workflow_bench
prompt: Review the historical snapshot of https://github.com/abhigyanpatwari/GitNexus/pull/2794. Report only actionable defects introduced by the local diff.
oracle:
command: test -s review-output.json
@ -33,7 +33,7 @@ tasks:
id: review-pr-2108-defect
ref: 3a4247ec36b5ad86b1123d3bbce8183a643f7434
sandbox_copy: [eval/workflow_bench/review_cases/pr-2108-defect.patch]
setup: git apply eval/workflow_bench/review_cases/pr-2108-defect.patch && rm -rf eval/workflow_bench
setup: git apply --exclude='eval/workflow_bench/*' eval/workflow_bench/review_cases/pr-2108-defect.patch && rm -rf eval/workflow_bench
prompt: Review the historical snapshot of https://github.com/abhigyanpatwari/GitNexus/pull/2108. Report only actionable defects introduced by the local diff.
oracle:
command: test -s review-output.json
@ -44,7 +44,7 @@ tasks:
id: review-pr-2258-defect
ref: 78b4077d8acc86f1b0c32e41012174d484e81f12
sandbox_copy: [eval/workflow_bench/review_cases/pr-2258-defect.patch]
setup: git apply eval/workflow_bench/review_cases/pr-2258-defect.patch && rm -rf eval/workflow_bench
setup: git apply --exclude='eval/workflow_bench/*' eval/workflow_bench/review_cases/pr-2258-defect.patch && rm -rf eval/workflow_bench
prompt: Review the historical snapshot of https://github.com/abhigyanpatwari/GitNexus/pull/2258. Report only actionable defects introduced by the local diff.
oracle:
command: test -s review-output.json
@ -56,7 +56,7 @@ tasks:
class: review-clean
ref: 78b4077d8acc86f1b0c32e41012174d484e81f12
sandbox_copy: [eval/workflow_bench/review_cases/pr-2258-clean.patch]
setup: git apply eval/workflow_bench/review_cases/pr-2258-clean.patch && rm -rf eval/workflow_bench
setup: git apply --exclude='eval/workflow_bench/*' eval/workflow_bench/review_cases/pr-2258-clean.patch && rm -rf eval/workflow_bench
prompt: Review this historical snapshot of https://github.com/abhigyanpatwari/GitNexus/pull/2258. Report only actionable defects introduced by the local diff.
oracle:
command: test -s review-output.json
@ -68,7 +68,7 @@ tasks:
class: review-clean
ref: 84f584449de02376a8ffc096dceac2e8f732cab5
sandbox_copy: [eval/workflow_bench/review_cases/pr-2773-clean.patch]
setup: git apply eval/workflow_bench/review_cases/pr-2773-clean.patch && rm -rf eval/workflow_bench
setup: git apply --exclude='eval/workflow_bench/*' eval/workflow_bench/review_cases/pr-2773-clean.patch && rm -rf eval/workflow_bench
prompt: Review this historical snapshot of https://github.com/abhigyanpatwari/GitNexus/pull/2773. Report only actionable defects introduced by the local diff.
oracle:
command: test -s review-output.json