diff --git a/eval/tests/test_evolve.py b/eval/tests/test_evolve.py index fbb368a47..1096f32e1 100644 --- a/eval/tests/test_evolve.py +++ b/eval/tests/test_evolve.py @@ -17,6 +17,7 @@ from workflow_bench.runner_sessions import PARENT_EVENT_STREAM_SOURCE from workflow_bench.evolve import ( build_parser, build_proposer_prompt, + executed_benchmark_arms, generation_timeout_seconds, load_jsonl, proposer_evidence_entries, @@ -871,6 +872,25 @@ def test_runner_argv_pairs_each_incumbent_with_its_candidate(tmp_path): assert json.loads(argv[argv.index("--promotion-target-bases-json") + 1]) == target_bases +def test_runner_argv_inserts_ce_review_for_review_overlay(tmp_path): + args = build_parser().parse_args( + ["--tasks", "t.yaml", "--model", "pinned", "--arms", "review"] + ) + overlay = tmp_path / "overlay" + skill = overlay / ".claude" / "skills" / "gitnexus-review" / "SKILL.md" + skill.parent.mkdir(parents=True) + skill.write_text("candidate") + argv = runner_argv( + args, + tmp_path / "bench", + overlay, + task_bindings=[{"id": "task"}], + target_base_digests={}, + ) + arms = argv[argv.index("--arms") + 1 : argv.index("--promotion-metric")] + assert arms == ["ce_review", "review", "candidate_review"] + + def test_runner_argv_omits_proposer_for_manual_overlay(tmp_path): args = build_parser().parse_args(["--tasks", "t.yaml", "--model", "pinned"]) overlay = tmp_path / "overlay" @@ -890,6 +910,26 @@ def test_runner_argv_omits_proposer_for_manual_overlay(tmp_path): assert "--proposer-model" not in argv +def test_runner_argv_forwards_explicit_unsafe_backend(tmp_path): + args = build_parser().parse_args( + ["--tasks", "t.yaml", "--model", "pinned", "--unsafe-no-bwrap"] + ) + overlay = tmp_path / "overlay" + skill = overlay / ".claude" / "skills" / "gitnexus-plan" / "SKILL.md" + skill.parent.mkdir(parents=True) + skill.write_text("candidate") + + argv = runner_argv( + args, + tmp_path / "bench", + overlay, + task_bindings=[{"id": "task"}], + target_base_digests={}, + ) + + assert "--unsafe-no-bwrap" in argv + + def test_runner_argv_keeps_task_commit_pinned_when_ref_moves(tmp_path): repo = tmp_path / "task-repo" repo.mkdir() @@ -992,6 +1032,59 @@ def test_generation_timeout_budgets_three_task_workflow_pair(): assert timeout >= 3 * (1 + 3 * paired_arm_cells) * evolve.WORKTREE_PREPARATION_TIMEOUT_SECONDS +def test_executed_benchmark_arms_inserts_review_comparator() -> None: + assert executed_benchmark_arms(["workflow"]) == ["workflow", "candidate_workflow"] + assert executed_benchmark_arms(["review"]) == ["ce_review", "review", "candidate_review"] + + +def test_generation_timeout_budgets_review_pair_plus_ce_comparator() -> None: + timeout = generation_timeout_seconds( + task_count=6, + runs=1, + session_timeout=3600, + incumbent_arms=["review"], + ) + + per_task_preparation = ( + evolve.TASK_BINDING_GIT_PHASES * evolve.GIT_COMMAND_TIMEOUT_SECONDS + + 2 * evolve.TASK_SNAPSHOT_TIMEOUT_SECONDS + + evolve.WORKTREE_PREPARATION_TIMEOUT_SECONDS + + evolve.GRAPH_SOURCE_PREPARATION_TIMEOUT_SECONDS + + evolve.GRAPH_BUILD_TIMEOUT_SECONDS + + 2 * evolve.GRAPH_QUERY_TIMEOUT_SECONDS + + evolve.CLEANUP_TIMEOUT_SECONDS + ) + paired_arm_cells = 3 + session_slots = 3 + workspace_snapshot_slots = 3 + per_task_run = session_slots * (3600 + evolve.SESSION_FINALIZATION_TIMEOUT_SECONDS) + paired_arm_cells * ( + evolve.WORKTREE_PREPARATION_TIMEOUT_SECONDS + + evolve.ARM_ASSET_MATERIALIZATION_PHASES * evolve.TASK_SNAPSHOT_TIMEOUT_SECONDS + + evolve.SETUP_TIMEOUT_SECONDS + + 2 * 3600 + + evolve.ARM_EVIDENCE_GIT_PHASES * evolve.GIT_COMMAND_TIMEOUT_SECONDS + + evolve.CLEANUP_TIMEOUT_SECONDS + ) + per_task_run += workspace_snapshot_slots * evolve.TASK_SNAPSHOT_TIMEOUT_SECONDS + per_task_run += evolve.CANDIDATE_OVERLAY_GIT_PHASES * evolve.GIT_COMMAND_TIMEOUT_SECONDS + + assert timeout == ( + evolve.PROMOTION_BASE_TIMEOUT_SECONDS + + 6 * (per_task_preparation + per_task_run) + + evolve.DRIVER_OVERHEAD_SECONDS + ) + + +def test_generation_timeout_rejects_unknown_arm() -> None: + with pytest.raises(ValueError, match="unsupported evolution arm: mystery"): + generation_timeout_seconds( + task_count=1, + runs=1, + session_timeout=60, + incumbent_arms=["mystery"], + ) + + @pytest.mark.skipif(sys.platform != "linux", reason="Bubblewrap PID namespaces require Linux") def test_outer_runner_pid_namespace_kills_setsid_descendant(tmp_path): try: diff --git a/eval/tests/test_model_gateway.py b/eval/tests/test_model_gateway.py index 924aa6c51..f35cb74a6 100644 --- a/eval/tests/test_model_gateway.py +++ b/eval/tests/test_model_gateway.py @@ -10,8 +10,11 @@ import pytest import yaml from workflow_bench.model_gateway import ( + DEFAULT_GATEWAY_READY_TIMEOUT_S, + GATEWAY_READY_TIMEOUT_ENV, GATEWAY_REQUEST_TIMEOUT_S, OpenAIGateway, + gateway_ready_timeout_s, anthropic_api_key_from_environ, claude_gateway_model_env, credential_secrets, @@ -141,6 +144,49 @@ def test_openai_gateway_never_leaves_proxy_output_on_an_undrained_pipe(tmp_path: assert gateway.log_path.stat().st_mode & 0o777 == 0o600 +def test_gateway_startup_budget_outlives_a_cold_litellm_import(monkeypatch, tmp_path: Path) -> None: + # Importing LiteLLM takes ~17s on a cold container filesystem and the proxy + # binds its port only afterwards, so a sub-20s budget fails as "connection + # refused" on a proxy that was merely still starting. + monkeypatch.delenv(GATEWAY_READY_TIMEOUT_ENV, raising=False) + assert DEFAULT_GATEWAY_READY_TIMEOUT_S >= 60 + assert gateway_ready_timeout_s() == DEFAULT_GATEWAY_READY_TIMEOUT_S + assert ( + OpenAIGateway( + openai_api_key="sk-openai-secret", + model_names=["gpt-4.1"], + work_dir=tmp_path / "gw", + ).ready_timeout_s + == DEFAULT_GATEWAY_READY_TIMEOUT_S + ) + + monkeypatch.setenv(GATEWAY_READY_TIMEOUT_ENV, "42.5") + assert gateway_ready_timeout_s() == 42.5 + + for bad in ("0", "-1", "soon"): + monkeypatch.setenv(GATEWAY_READY_TIMEOUT_ENV, bad) + with pytest.raises(ValueError, match=GATEWAY_READY_TIMEOUT_ENV): + gateway_ready_timeout_s() + + +def test_gateway_readiness_timeout_reports_the_proxy_log_and_the_override(tmp_path: Path) -> None: + gateway = OpenAIGateway( + openai_api_key="sk-openai-secret", + model_names=["gpt-4.1"], + work_dir=tmp_path / "gw", + ready_timeout_s=0.1, + ) + gateway.work_dir.mkdir(parents=True) + gateway.log_path.write_text("ImportError: litellm proxy extras missing") + + with pytest.raises(RuntimeError) as excinfo: + gateway._wait_until_ready() + + message = str(excinfo.value) + assert "ImportError: litellm proxy extras missing" in message + assert GATEWAY_READY_TIMEOUT_ENV in message + + def test_runner_environment_does_not_forward_the_openai_key() -> None: args = build_parser().parse_args( [ diff --git a/eval/tests/test_oracle_assets.py b/eval/tests/test_oracle_assets.py index 580d96bfc..c8bf2ccad 100644 --- a/eval/tests/test_oracle_assets.py +++ b/eval/tests/test_oracle_assets.py @@ -5,6 +5,7 @@ from __future__ import annotations import argparse import hashlib import os +import shutil import subprocess from pathlib import Path from types import SimpleNamespace @@ -13,7 +14,12 @@ import pytest from workflow_bench import oracle_assets, runner from workflow_bench.evolution import evaluate_candidate -from workflow_bench.oracle_assets import capture_task_oracle, staged_task_oracle +from workflow_bench.oracle_assets import ( + capture_task_oracle, + review_case_setup_command, + staged_task_oracle, + with_hidden_harness_apply_exclude, +) def oracle_task(*, command: str = "true", source: str = "oracle.test.ts") -> dict[str, object]: @@ -149,6 +155,60 @@ def test_clone_sanitization_prunes_harness_checkout_and_recoverable_history(tmp_ assert git(clone, "status", "--porcelain=v1", "--untracked-files=all").stdout == "" +def test_hidden_harness_apply_exclude_is_idempotent() -> None: + raw = "git apply eval/workflow_bench/review_cases/pr.patch && rm -rf eval/workflow_bench" + once = with_hidden_harness_apply_exclude(raw) + assert once == review_case_setup_command("pr.patch") + assert with_hidden_harness_apply_exclude(once) == once + assert with_hidden_harness_apply_exclude("true") == "true" + + +def test_review_setup_skips_sanitized_harness_hunks(tmp_path: Path) -> None: + repo = tmp_path / "repo" + repo.mkdir() + + def git(*args: str, check: bool = True) -> subprocess.CompletedProcess[str]: + result = subprocess.run( + ["git", "-C", str(repo), *args], + check=False, + capture_output=True, + text=True, + ) + if check and result.returncode != 0: + pytest.fail(f"git {' '.join(args)} failed: {result.stderr}") + return result + + git("init", "--quiet", "--initial-branch=main") + git("config", "user.name", "Review Setup") + git("config", "user.email", "review-setup.invalid") + (repo / "visible.py").write_text("old\n") + hidden = repo / "eval" / "workflow_bench" + hidden.mkdir(parents=True) + (hidden / "learnings.jsonl").write_text("{}\n") + git("add", "--all") + git("commit", "--quiet", "-m", "base with harness file") + (repo / "visible.py").write_text("new\n") + (hidden / "learnings.jsonl").write_text("{}\nextra\n") + patch = git("diff").stdout + git("checkout", "--", ".") + shutil.rmtree(hidden) + patch_path = hidden / "review_cases" / "case.patch" + patch_path.parent.mkdir(parents=True) + patch_path.write_text(patch) + + rejected = git("apply", "--check", str(patch_path.relative_to(repo)), check=False) + assert rejected.returncode != 0 + assert "learnings.jsonl" in rejected.stderr + + setup = with_hidden_harness_apply_exclude( + "git apply eval/workflow_bench/review_cases/case.patch && rm -rf eval/workflow_bench" + ) + applied = subprocess.run(["/bin/sh", "-lc", setup], cwd=repo, check=False, capture_output=True, text=True) + assert applied.returncode == 0, applied.stderr + assert (repo / "visible.py").read_text() == "new\n" + assert not hidden.exists() + + def test_clone_sanitization_prunes_remote_history_when_head_never_had_harness(tmp_path: Path) -> None: source = tmp_path / "source" source.mkdir() diff --git a/eval/tests/test_proposer_sandbox.py b/eval/tests/test_proposer_sandbox.py index d7e015e0a..f8e3c4087 100644 --- a/eval/tests/test_proposer_sandbox.py +++ b/eval/tests/test_proposer_sandbox.py @@ -11,6 +11,7 @@ import sys import threading from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer from pathlib import Path +from types import SimpleNamespace import pytest @@ -20,6 +21,7 @@ from workflow_bench.process_control import ManagedProcessResult, run_managed from workflow_bench.proposer_sandbox import ( MAX_BUNDLE_BYTES, MAX_EVIDENCE_FILE_BYTES, + SANDBOX_CLAUDE, SANDBOX_NODE, SANDBOX_NODE_PREFIX, SANDBOX_EVIDENCE, @@ -35,8 +37,11 @@ from workflow_bench.proposer_sandbox import ( _runtime_mount_args, build_claude_settings, build_sandbox_environment, + _force_rmtree, + host_workspace_write_boundary, prepare_sandbox, preflight_bubblewrap, + sandbox_workspace_write_boundary, stage_evidence_bundle, stage_task_assets, ) @@ -80,6 +85,128 @@ def test_environment_is_allowlisted_and_shell_children_are_credential_free(monke assert "defaultMode" not in settings["permissions"] +def test_unsafe_host_session_translates_virtual_paths_and_disables_containment(tmp_path) -> None: + clone = tmp_path / "clone" + evidence = tmp_path / "evidence" + for directory in (clone, evidence): + directory.mkdir() + + with prepare_sandbox( + clone=clone, + backend="host-unsafe", + read_only_mounts=(ReadOnlyMount(evidence, "/evidence"),), + ) as sandbox: + assert sandbox.command_prefix == [] + assert sandbox.require_pid_namespace is False + assert sandbox.host_path("/workspace/review-output.json") == str(clone / "review-output.json") + assert sandbox.host_path("/evidence/selected-rows.json") == str(evidence / "selected-rows.json") + assert sandbox.host_text("read /evidence and write /workspace/out") == ( + f"read {evidence} and write {clone}/out" + ) + # Sessions spawn the binary directly, so it must be the host executable + # rather than the sandbox-only mount target. + assert sandbox.claude_bin != SANDBOX_CLAUDE + assert Path(sandbox.claude_bin).exists() + assert sandbox.environment()["HOME"] == str(sandbox.home) + assert "CLAUDE_CODE_SUBPROCESS_ENV_SCRUB" not in sandbox.environment() + assert all( + Path(entry).is_dir() for entry in sandbox.environment()["PATH"].split(":") + ) + unsafe_settings = json.loads(sandbox.settings_json) + assert unsafe_settings["sandbox"]["enabled"] is False + assert unsafe_settings["sandbox"]["failIfUnavailable"] is False + assert "disableBypassPermissionsMode" not in unsafe_settings["permissions"] + + +def test_host_workspace_write_boundary_keeps_only_the_review_artifact_writable(tmp_path) -> None: + clone = tmp_path / "clone" + nested = clone / "src" + nested.mkdir(parents=True) + source = nested / "source.ts" + source.write_text("trusted\n") + output = clone / "review-output.json" + output.write_text("") + original_source_mode = stat.S_IMODE(source.stat().st_mode) + original_output_mode = stat.S_IMODE(output.stat().st_mode) + + with host_workspace_write_boundary(clone, writable=(output,)): + with pytest.raises(OSError): + source.write_text("tampered\n") + with pytest.raises(OSError): + (clone / "extra.py").write_text("nope\n") + output.write_text('{"schema_version":1}\n') + + assert source.read_text() == "trusted\n" + assert output.read_text() == '{"schema_version":1}\n' + assert not (clone / "extra.py").exists() + assert stat.S_IMODE(source.stat().st_mode) == original_source_mode + assert stat.S_IMODE(output.stat().st_mode) == original_output_mode + + +def test_sandbox_workspace_write_boundary_is_noop_unless_host_unsafe(tmp_path) -> None: + clone = tmp_path / "clone" + clone.mkdir() + source = clone / "source.ts" + source.write_text("trusted\n") + bwrap_sandbox = SimpleNamespace(backend="bwrap", clone=clone) + with sandbox_workspace_write_boundary( + bwrap_sandbox, + read_only_workspace=True, + writable=(), + ): + source.write_text("still writable under bwrap no-op\n") + assert source.read_text() == "still writable under bwrap no-op\n" + + output = clone / "review-output.json" + output.write_text("") + source.write_text("trusted\n") + unsafe = SimpleNamespace(backend="host-unsafe", clone=clone) + with sandbox_workspace_write_boundary( + unsafe, + read_only_workspace=True, + writable=(output,), + ): + with pytest.raises(OSError): + source.write_text("tampered\n") + output.write_text("ok\n") + assert source.read_text() == "trusted\n" + assert output.read_text() == "ok\n" + + +def test_force_rmtree_deletes_nonempty_directories_copied_from_a_locked_workspace(tmp_path) -> None: + locked = tmp_path / "locked" + nested = locked / "gitnexus-shared" / "src" + nested.mkdir(parents=True) + (nested / "index.ts").write_text("export {}\n") + os.chmod(nested, 0o555) + os.chmod(locked / "gitnexus-shared", 0o555) + os.chmod(locked, 0o555) + + copied = tmp_path / "sandbox-tmp" / "tmp.XXXX" / "gitnexus-shared" + copied.parent.mkdir(parents=True) + shutil.copytree(locked / "gitnexus-shared", copied) + assert stat.S_IMODE(copied.stat().st_mode) & 0o222 == 0 + + _force_rmtree(copied.parent) + assert not copied.parent.exists() + + +def test_host_unsafe_sandbox_cleanup_survives_readonly_tmpdir_copies(tmp_path) -> None: + clone = tmp_path / "clone" + clone.mkdir() + leftover = None + with prepare_sandbox(clone=clone, backend="host-unsafe") as sandbox: + leftover = sandbox.private_root + copied = sandbox.temp / "tmp.XXXX" / "gitnexus-shared" / "src" + copied.mkdir(parents=True) + (copied / "index.ts").write_text("export {}\n") + os.chmod(copied, 0o555) + os.chmod(copied.parent, 0o555) + os.chmod(copied.parent.parent, 0o555) + assert leftover is not None + assert not leftover.exists() + + @pytest.mark.parametrize( "bad_url", ["https://user:secret@example.test", "https://example.test/path?token=x", "file:///tmp/model"], diff --git a/eval/tests/test_review_corpus.py b/eval/tests/test_review_corpus.py index a4db6b0b4..7c63327b4 100644 --- a/eval/tests/test_review_corpus.py +++ b/eval/tests/test_review_corpus.py @@ -4,6 +4,8 @@ from pathlib import Path import yaml +from workflow_bench.oracle_assets import review_case_setup_command + BENCH_ROOT = Path(__file__).parents[1] / "workflow_bench" @@ -26,6 +28,7 @@ def test_review_corpus_is_immutable_and_task_bound(): task = by_id[case["id"]] assert task["ref"] == case["base_sha"] assert task["sandbox_copy"] == [f"eval/workflow_bench/review_cases/{patch.name}"] + assert task["setup"] == review_case_setup_command(patch.name) def test_hidden_labels_are_not_recoverable_from_visible_task_input(): diff --git a/eval/tests/test_sanitized_graph.py b/eval/tests/test_sanitized_graph.py index 4f2c2cc0a..431f5059d 100644 --- a/eval/tests/test_sanitized_graph.py +++ b/eval/tests/test_sanitized_graph.py @@ -27,6 +27,32 @@ def test_prebuilt_graph_and_harness_assets_are_rejected(task): sanitized_graph.validate_no_prebuilt_graph_assets(task) +def test_review_case_patches_are_allowed_sandbox_copy(): + sanitized_graph.validate_no_prebuilt_graph_assets( + {"sandbox_copy": ["eval/workflow_bench/review_cases/pr-2718-defect.patch"]} + ) + + +@pytest.mark.parametrize( + "task", + [ + {"sandbox_copy": ["eval/workflow_bench"]}, + {"sandbox_copy": ["eval/workflow_bench/evolve.py"]}, + { + "sandbox_dependencies": [ + { + "source": "eval/workflow_bench/review_cases/pr-2718-defect.patch", + "target": "patch", + } + ] + }, + ], +) +def test_non_corpus_harness_paths_stay_rejected(task): + with pytest.raises(SandboxError, match="prebuilt graph or harness"): + sanitized_graph.validate_no_prebuilt_graph_assets(task) + + def test_graph_environment_is_offline_deterministic_and_ignores_target_gitignore(): env = sanitized_graph._graph_environment() @@ -34,6 +60,10 @@ def test_graph_environment_is_offline_deterministic_and_ignores_target_gitignore assert env["GITNEXUS_NO_GITIGNORE"] == "1" assert env["GITNEXUS_WORKER_POOL_SIZE"] == "1" assert env["GITNEXUS_PARSE_CHUNK_CONCURRENCY"] == "1" + assert env["GITNEXUS_WORKER_READY_TIMEOUT_MS"] == str( + sanitized_graph.GRAPH_WORKER_READY_TIMEOUT_MS + ) + assert int(env["GITNEXUS_WORKER_READY_TIMEOUT_MS"]) >= 60_000 assert "ANTHROPIC_API_KEY" not in env diff --git a/eval/tests/test_task_assets.py b/eval/tests/test_task_assets.py index dc5d63ac7..f4205a861 100644 --- a/eval/tests/test_task_assets.py +++ b/eval/tests/test_task_assets.py @@ -5,15 +5,15 @@ from __future__ import annotations import os import stat import subprocess -from pathlib import Path +from pathlib import Path, PurePosixPath import pytest from workflow_bench.proposer_sandbox import VITE_TEMP_DIR, SandboxError from workflow_bench.oracle_assets import TaskOracleSnapshot from workflow_bench.runner_tasks import resolve_task_bindings -from workflow_bench.task_assets import TaskAssetCache, stage_task_assets -from workflow_bench import task_assets +from workflow_bench.task_assets import TaskAssetCache, _is_harness_sandbox_copy, stage_task_assets +from workflow_bench import runtime_mounts, task_assets SHA = "a" * 40 @@ -443,3 +443,53 @@ def test_non_node_modules_dependency_snapshot_has_no_vite_temp(tmp_path: Path) - snapshot = cache.prepare(task, repo=repo, resolved_sha=SHA) captured = {entry.path.as_posix() for entry in snapshot.dependencies[0].entries} assert not any(path.endswith(VITE_TEMP_DIR) for path in captured) + + +def test_review_case_sandbox_copy_is_read_from_the_harness_not_the_task_repo( + monkeypatch, tmp_path: Path +) -> None: + repo = tmp_path / "task-repo" + repo.mkdir() + (repo / "eval" / "workflow_bench").mkdir(parents=True) + harness = tmp_path / "harness" + patch = harness / "eval" / "workflow_bench" / "review_cases" / "pr-2718-defect.patch" + patch.parent.mkdir(parents=True) + patch.write_bytes(b"diff --git a/a b/a\n") + monkeypatch.setattr(runtime_mounts, "HARNESS_ROOT", harness) + + task = { + "sandbox_copy": ["eval/workflow_bench/review_cases/pr-2718-defect.patch"], + "sandbox_dependencies": [], + } + with TaskAssetCache(tmp_path / "cache") as cache: + snapshot = cache.prepare(task, repo=repo, resolved_sha=SHA) + copied = snapshot.root / "sandbox-copy" / "eval" / "workflow_bench" / "review_cases" / "pr-2718-defect.patch" + assert copied.read_bytes() == b"diff --git a/a b/a\n" + + +def test_review_case_sandbox_copy_does_not_fall_back_to_the_task_repo( + monkeypatch, tmp_path: Path +) -> None: + repo = tmp_path / "task-repo" + planted = repo / "eval" / "workflow_bench" / "review_cases" / "pr-2718-defect.patch" + planted.parent.mkdir(parents=True) + planted.write_bytes(b"from-task-repo") + harness = tmp_path / "harness" + harness.mkdir() + monkeypatch.setattr(runtime_mounts, "HARNESS_ROOT", harness) + + task = { + "sandbox_copy": ["eval/workflow_bench/review_cases/pr-2718-defect.patch"], + "sandbox_dependencies": [], + } + with TaskAssetCache(tmp_path / "cache") as cache: + with pytest.raises(SandboxError, match="unavailable"): + cache.prepare(task, repo=repo, resolved_sha=SHA) + + +def test_harness_sandbox_copy_does_not_treat_parent_escapes_as_corpus() -> None: + assert _is_harness_sandbox_copy(PurePosixPath("eval/workflow_bench/review_cases/pr.patch")) + assert not _is_harness_sandbox_copy( + PurePosixPath("eval/workflow_bench/review_cases/../oracles/hidden.json") + ) + assert not _is_harness_sandbox_copy(PurePosixPath("eval/workflow_bench/oracles")) diff --git a/eval/tests/test_workflow_bench_evolution.py b/eval/tests/test_workflow_bench_evolution.py index c04489388..1aca6d019 100644 --- a/eval/tests/test_workflow_bench_evolution.py +++ b/eval/tests/test_workflow_bench_evolution.py @@ -15,6 +15,7 @@ from workflow_bench.evolution import ( evaluate_candidate, evaluate_review_candidate, required_candidate_arms, + seed_evaluated_skills, skill_fingerprint, unexercised_overlay_skills, ) @@ -387,6 +388,12 @@ def test_apply_candidate_overlay_creates_a_clean_ephemeral_commit(tmp_path): ) == candidate_overlay_digest(overlay) assert incumbent.read_text() == "candidate\n" git_commands = [command for command in sandbox.commands if command[0] == "/usr/bin/git"] + assert git_commands[0][-4:] == [ + "add", + "-f", + "--", + ".claude/skills/gitnexus-work/SKILL.md", + ] assert [command[-1] for command in git_commands[:2]] == [ ".claude/skills/gitnexus-work/SKILL.md", "--", @@ -422,6 +429,181 @@ def test_candidate_overlay_rejects_linked_destination_parents(tmp_path): apply_candidate_overlay(overlay, repo, sandbox=sandbox) +@pytest.mark.skipif(os.name == "nt", reason="candidate overlays require the Linux outer sandbox") +def test_apply_candidate_overlay_force_adds_historically_ignored_skill(tmp_path): + repo = tmp_path / "repo" + repo.mkdir() + subprocess.run(["git", "init", "--quiet", str(repo)], check=True) + (repo / ".gitignore").write_text(".claude/skills/*\n") + (repo / "README").write_text("subject\n") + subprocess.run(["git", "-C", str(repo), "add", "."], check=True) + subprocess.run( + [ + "git", + "-C", + str(repo), + "-c", + "user.name=test", + "-c", + "user.email=test@invalid", + "commit", + "--quiet", + "-m", + "historical checkout that ignores skills", + ], + check=True, + ) + + overlay = tmp_path / "candidate" + write_overlay_skill(overlay, "gitnexus-review") + + class LocalSandbox: + def __init__(self): + self.clone = repo + + def run(self, command, **kwargs): + if command[0] == "/bin/mkdir": + return ManagedProcessResult( + state="exited", + returncode=0, + stdout_tail="", + stderr_tail="", + duration_s=0.0, + ) + translated = [str(repo) if item == "/workspace" else item for item in command] + completed = subprocess.run( + translated, + cwd=repo, + env=dict(kwargs["env"]), + capture_output=True, + text=True, + check=False, + ) + return ManagedProcessResult( + state="exited", + returncode=completed.returncode, + stdout_tail=completed.stdout, + stderr_tail=completed.stderr, + duration_s=0.0, + ) + + apply_candidate_overlay(overlay, repo, sandbox=LocalSandbox()) + assert (repo / ".claude" / "skills" / "gitnexus-review" / "SKILL.md").read_text() == ( + "gitnexus-review candidate\n" + ) + status = subprocess.run( + ["git", "-C", str(repo), "status", "--porcelain"], + check=True, + capture_output=True, + text=True, + ) + assert status.stdout == "" + + +@pytest.mark.skipif(os.name == "nt", reason="skill seeds require the Linux outer sandbox") +def test_seed_evaluated_skills_installs_missing_review_skill_and_is_idempotent(tmp_path): + repo = tmp_path / "clone" + repo.mkdir() + subprocess.run(["git", "init", "--quiet", str(repo)], check=True) + # Historical review SHAs ignore the whole skill tree and lack today's + # `!.claude/skills/gitnexus-review/` allowlist. Seeding must still commit. + (repo / ".gitignore").write_text(".claude/skills/*\n") + (repo / "README").write_text("subject\n") + subprocess.run(["git", "-C", str(repo), "add", "."], check=True) + subprocess.run( + [ + "git", + "-C", + str(repo), + "-c", + "user.name=test", + "-c", + "user.email=test@invalid", + "commit", + "--quiet", + "-m", + "historical checkout without review skill", + ], + check=True, + ) + before = subprocess.run( + ["git", "-C", str(repo), "rev-parse", "HEAD"], + check=True, + capture_output=True, + text=True, + ).stdout.strip() + + source = tmp_path / "harness" + skill = source / ".claude" / "skills" / "gitnexus-review" / "SKILL.md" + persona = source / ".claude" / "skills" / "gitnexus-review" / "ci-personas" / "lens.md" + persona.parent.mkdir(parents=True) + skill.write_text("current incumbent review skill\n") + persona.write_text("persona\n") + + class LocalSandbox: + def __init__(self): + self.clone = repo + + def run(self, command, **kwargs): + if command[0] == "/bin/mkdir": + return ManagedProcessResult( + state="exited", + returncode=0, + stdout_tail="", + stderr_tail="", + duration_s=0.0, + ) + translated = [str(repo) if item == "/workspace" else item for item in command] + completed = subprocess.run( + translated, + cwd=repo, + env=dict(kwargs["env"]), + capture_output=True, + text=True, + check=False, + ) + return ManagedProcessResult( + state="exited", + returncode=completed.returncode, + stdout_tail=completed.stdout, + stderr_tail=completed.stderr, + duration_s=0.0, + ) + + sandbox = LocalSandbox() + seed_evaluated_skills(source, repo, sandbox=sandbox, arm="review") + assert skill_fingerprint(repo, "review") is not None + assert (repo / ".claude" / "skills" / "gitnexus-review" / "SKILL.md").read_text() == ( + "current incumbent review skill\n" + ) + assert ( + repo / ".claude" / "skills" / "gitnexus-review" / "ci-personas" / "lens.md" + ).read_text() == "persona\n" + status = subprocess.run( + ["git", "-C", str(repo), "status", "--porcelain"], + check=True, + capture_output=True, + text=True, + ) + assert status.stdout == "" + after = subprocess.run( + ["git", "-C", str(repo), "rev-parse", "HEAD"], + check=True, + capture_output=True, + text=True, + ).stdout.strip() + assert after != before + + seed_evaluated_skills(source, repo, sandbox=sandbox, arm="review") + again = subprocess.run( + ["git", "-C", str(repo), "rev-parse", "HEAD"], + check=True, + capture_output=True, + text=True, + ).stdout.strip() + assert again == after + + @pytest.mark.skipif(os.name == "nt", reason="skill links are rejected by the Linux sandbox harness") def test_skill_fingerprint_rejects_linked_skill_roots(tmp_path): outside = tmp_path / "outside" diff --git a/eval/tests/test_workflow_bench_sessions.py b/eval/tests/test_workflow_bench_sessions.py index 554443c1a..9723e2e4b 100644 --- a/eval/tests/test_workflow_bench_sessions.py +++ b/eval/tests/test_workflow_bench_sessions.py @@ -4,6 +4,7 @@ import argparse import hashlib import json import os +import shutil import subprocess import sys from pathlib import Path @@ -14,6 +15,7 @@ import pytest from workflow_bench import evolve, runner, runner_sessions, runtime_mounts from workflow_bench.evolution import skill_fingerprint from workflow_bench.process_control import ManagedProcessResult +from workflow_bench.proposer_sandbox import SandboxError from workflow_bench.runner import snapshot_plan_docs @@ -315,10 +317,18 @@ def test_run_arm_keeps_session_error_kind_over_verify(monkeypatch, tmp_path): def test_agent_tool_grants_are_exact_and_nomcp_has_no_graph_tools(monkeypatch, tmp_path): read_only = runner.allowed_agent_tools(implementation=False) + review_tools = runner.allowed_agent_tools(implementation=False, allow_edit=False) implementation = runner.allowed_agent_tools(implementation=True) no_mcp = runner.allowed_agent_tools(implementation=True, include_mcp=False) assert read_only == [*runner.BUILTIN_AGENT_TOOLS, *runner.GITNEXUS_READ_ONLY_TOOLS] + assert review_tools == [ + tool + for tool in [*runner.BUILTIN_AGENT_TOOLS, *runner.GITNEXUS_READ_ONLY_TOOLS] + if tool != "Edit" + ] + assert "Write" in review_tools + assert "Edit" not in review_tools assert implementation == [ *runner.BUILTIN_AGENT_TOOLS, *runner.GITNEXUS_READ_ONLY_TOOLS, @@ -346,7 +356,7 @@ def test_agent_tool_grants_are_exact_and_nomcp_has_no_graph_tools(monkeypatch, t ) assert captured[0]["allowed_tools"] == read_only # planning - assert captured[1]["allowed_tools"] == read_only # review + assert captured[1]["allowed_tools"] == review_tools # review assert captured[2]["allowed_tools"] == implementation assert captured[3]["allowed_tools"] == list(runner.BUILTIN_AGENT_TOOLS) assert captured[3]["mcp_config_json"] == '{"mcpServers":{}}' @@ -416,6 +426,93 @@ def test_mcp_config_uses_only_the_minimal_pinned_harness_runtime(monkeypatch, tm assert f"{runner.SANDBOX_GITNEXUS}/hooks" not in mounted_targets +def _install_pinned_runtime(root: Path) -> None: + runtime = root / "gitnexus" + shared = root / "gitnexus-shared" + for directory in ( + runtime / "dist" / "cli", + runtime / "node_modules", + runtime / "vendor", + runtime / "hooks" / "claude", + shared / "dist", + ): + directory.mkdir(parents=True) + (runtime / "dist" / "cli" / "index.js").write_text("") + (runtime / "hooks" / "claude" / "resolve-analyze-cmd.cjs").write_text("") + (runtime / "package.json").write_text(json.dumps({"version": "9.9.9-test"})) + (runtime / "node_modules" / "gitnexus-shared").symlink_to(shared, target_is_directory=True) + (shared / "package.json").write_text(json.dumps({"name": "gitnexus-shared"})) + + +def test_runtime_mounts_reuse_primary_checkout_node_modules_from_a_worktree( + monkeypatch, tmp_path +) -> None: + primary = tmp_path / "primary" + worktree = tmp_path / "worktree" + _install_pinned_runtime(primary) + (primary / ".git" / "worktrees" / "wt").mkdir(parents=True) + _install_pinned_runtime(worktree) + shutil.rmtree(worktree / "gitnexus" / "node_modules") + (worktree / "gitnexus" / "node_modules").symlink_to( + primary / "gitnexus" / "node_modules", + target_is_directory=True, + ) + (worktree / ".git").write_text(f"gitdir: {primary / '.git' / 'worktrees' / 'wt'}\n") + monkeypatch.setattr(runtime_mounts, "HARNESS_ROOT", worktree) + + mounts = runner.trusted_gitnexus_runtime_mounts() + by_target = {mount.target: mount.source for mount in mounts} + + assert by_target[f"{runner.SANDBOX_GITNEXUS}/node_modules"] == ( + primary / "gitnexus" / "node_modules" + ) + assert by_target[f"{runner.SANDBOX_GITNEXUS_SHARED}/package.json"] == ( + worktree / "gitnexus-shared" / "package.json" + ) + assert by_target[f"{runner.SANDBOX_GITNEXUS}/dist"] == worktree / "gitnexus" / "dist" + + +def test_runtime_mounts_reject_a_node_modules_symlink_outside_the_primary_checkout( + monkeypatch, tmp_path +) -> None: + primary = tmp_path / "primary" + worktree = tmp_path / "worktree" + outsider = tmp_path / "outsider" + _install_pinned_runtime(primary) + _install_pinned_runtime(outsider) + (primary / ".git" / "worktrees" / "wt").mkdir(parents=True) + _install_pinned_runtime(worktree) + shutil.rmtree(worktree / "gitnexus" / "node_modules") + (worktree / "gitnexus" / "node_modules").symlink_to( + outsider / "gitnexus" / "node_modules", + target_is_directory=True, + ) + (worktree / ".git").write_text(f"gitdir: {primary / '.git' / 'worktrees' / 'wt'}\n") + monkeypatch.setattr(runtime_mounts, "HARNESS_ROOT", worktree) + + with pytest.raises(SandboxError, match="primary checkout"): + runner.trusted_gitnexus_runtime_mounts() + + +def test_runtime_mounts_reject_a_node_modules_symlink_in_a_regular_checkout( + monkeypatch, tmp_path +) -> None: + checkout = tmp_path / "checkout" + other = tmp_path / "other" + _install_pinned_runtime(checkout) + _install_pinned_runtime(other) + (checkout / ".git").mkdir() + shutil.rmtree(checkout / "gitnexus" / "node_modules") + (checkout / "gitnexus" / "node_modules").symlink_to( + other / "gitnexus" / "node_modules", + target_is_directory=True, + ) + monkeypatch.setattr(runtime_mounts, "HARNESS_ROOT", checkout) + + with pytest.raises(SandboxError, match="must be a real directory"): + runner.trusted_gitnexus_runtime_mounts() + + @pytest.mark.skipif( os.environ.get("GITNEXUS_REQUIRE_BWRAP_CANARY") != "1", reason="real Bubblewrap canary is mandatory in the named Ubuntu CI job", @@ -662,6 +759,23 @@ def test_skill_invocation_parses_supported_exact_identifier_fields(skill_input): ) +def test_skill_invocation_accepts_plugin_qualified_identifier(): + assert ( + runner_sessions.skill_was_invoked_events( + skill_events({"skill": "compound-engineering:ce-code-review"}), + "ce-code-review", + ) + is True + ) + assert ( + runner_sessions.skill_was_invoked_events( + skill_events({"skill": "compound-engineering:ce-plan"}), + "ce-code-review", + ) + is False + ) + + @pytest.mark.parametrize( "skill_input", [ diff --git a/eval/workflow_bench/README.md b/eval/workflow_bench/README.md index e4f4b8623..d5642efdc 100644 --- a/eval/workflow_bench/README.md +++ b/eval/workflow_bench/README.md @@ -139,6 +139,28 @@ Native benchmark execution is therefore Linux/WSL2-only. Evidence assembly and hand-authored overlay preparation can happen elsewhere, but `--initial-overlay` does not bypass containment. +For a local diagnostic inside a container that blocks user namespaces, an +operator may explicitly choose the non-containment host backend: + +```bash +UNSAFE_NO_BWRAP=1 RUNS=1 ./workflow_bench/run-evolution.sh +``` + +This mode runs review sessions directly in disposable host worktrees and is +**not** a security boundary: it does not isolate the network or create a PID +namespace, and a session that can `chmod` can undo the workspace lock. The +harness still drops write bits on the clone except `review-output.json` so +accidental `npm install` / analyze writes cannot invalidate review evidence. +Sandbox cleanup restores owner write bits before deleting the private TMPDIR, +because a session that `copytree`s the locked clone would otherwise leave +non-empty 0555 directories that `rmtree` cannot remove. Historical review +SHAs that gitignore `.claude/skills/*` are force-added when the harness seeds +or overlays the evaluated `gitnexus-review` skill. +Treat model and verifier processes as able to access host files and +credentials available to the invoking user. It is restricted to the review +benchmark, forbidden with `--apply` and whenever `CI` is set; +promotion-capable and CI runs must use Bubblewrap. + ## Prompt and skill evolution loop Prompts age as models and tool harnesses change. Treat the current skills and diff --git a/eval/workflow_bench/evolution.py b/eval/workflow_bench/evolution.py index d5e6bcd3c..6a1699009 100644 --- a/eval/workflow_bench/evolution.py +++ b/eval/workflow_bench/evolution.py @@ -7,6 +7,7 @@ import os import secrets import stat import statistics +from collections.abc import Sequence from pathlib import Path, PurePosixPath from typing import Any @@ -383,6 +384,69 @@ def candidate_overlay_digest(overlay: Path) -> str: return digest +def _commit_sandbox_paths( + sandbox: SandboxSession, + relative_paths: Sequence[str], + *, + message: str, + require_change: bool, +) -> bool: + """Stage and commit paths inside the outer sandbox. + + Returns True when a commit was created. ``require_change`` keeps the + candidate-overlay contract: a no-op overlay is an error, while an + incumbent skill seed may already match the historical tree. + """ + + mkdir_command = ["/bin/mkdir", "-p", f"{SANDBOX_TMP}/wfbench-empty-hooks"] + mkdir_result = sandbox.run( + mkdir_command, + timeout=60, + env=build_sandbox_environment(), + ) + if not mkdir_result.ok: + raise ManagedProcessError(mkdir_command, mkdir_result) + if not relative_paths: + if require_change: + raise ValueError("candidate overlay is byte-identical to the incumbent skills") + return False + + # Historical review SHAs gitignore `.claude/skills/*` and lack the current + # per-skill allowlist. Force-add so a seed or overlay of harness-owned + # skill bytes is not rejected as an ignored path. + command, added = _sandbox_overlay_git(sandbox, ["add", "-f", "--", *relative_paths]) + if not added.ok: + raise ManagedProcessError(command, added) + command, changed = _sandbox_overlay_git( + sandbox, + ["diff", "--cached", "--quiet", "--no-ext-diff", "--no-textconv", "--"], + ) + if changed.returncode == 0: + if require_change: + raise ValueError("candidate overlay is byte-identical to the incumbent skills") + return False + if changed.returncode != 1: + raise ManagedProcessError(command, changed) + + command, committed = _sandbox_overlay_git( + sandbox, + [ + "commit", + "--quiet", + "--no-verify", + "-m", + message, + ], + extra_config=( + "user.name=workflow-bench", + "user.email=workflow-bench@invalid", + ), + ) + if not committed.ok: + raise ManagedProcessError(command, committed) + return True + + def apply_candidate_overlay( overlay: Path, worktree: Path, @@ -401,47 +465,96 @@ def apply_candidate_overlay( for relative, content in payload: _replace_regular_file(worktree, relative, content) relative_paths.append(relative.as_posix()) - - mkdir_command = ["/bin/mkdir", "-p", f"{SANDBOX_TMP}/wfbench-empty-hooks"] - mkdir_result = sandbox.run( - mkdir_command, - timeout=60, - env=build_sandbox_environment(), - ) - if not mkdir_result.ok: - raise ManagedProcessError(mkdir_command, mkdir_result) - - command, added = _sandbox_overlay_git(sandbox, ["add", "--", *relative_paths]) - if not added.ok: - raise ManagedProcessError(command, added) - command, changed = _sandbox_overlay_git( + _commit_sandbox_paths( sandbox, - ["diff", "--cached", "--quiet", "--no-ext-diff", "--no-textconv", "--"], + relative_paths, + message="benchmark candidate skill overlay", + require_change=True, ) - if changed.returncode == 0: - raise ValueError("candidate overlay is byte-identical to the incumbent skills") - if changed.returncode != 1: - raise ManagedProcessError(command, changed) - - command, committed = _sandbox_overlay_git( - sandbox, - [ - "commit", - "--quiet", - "--no-verify", - "-m", - "benchmark candidate skill overlay", - ], - extra_config=( - "user.name=workflow-bench", - "user.email=workflow-bench@invalid", - ), - ) - if not committed.ok: - raise ManagedProcessError(command, committed) return digest +def seed_evaluated_skills( + source_repo: Path, + worktree: Path, + *, + sandbox: SandboxSession, + arm: str, +) -> None: + """Install the current evaluated skill tree into a historical clone. + + Review evolution scores the current (or overlay) ``gitnexus-review`` skill + against a historical PR checkout. Older SHAs predate that skill, and + using whatever prose happened to exist at the reviewed commit would make + the incumbent arm a moving target. Copy the harness checkout's skill + bytes and commit them before setup so ``git status`` still shows only + the task patch. + """ + + skill_names = EVALUATED_ARM_SKILLS.get(arm) + if not skill_names: + return + + source_repo = source_repo.expanduser().absolute() + expected_clone = Path(os.path.abspath(worktree.expanduser())) + sandbox_clone = Path(os.path.abspath(sandbox.clone.expanduser())) + if sandbox_clone != expected_clone: + raise ValueError("skill seed sandbox does not bind the requested clone") + _require_real_directory(source_repo, label="incumbent skill repository") + if source_repo.resolve(strict=True) != source_repo: + raise ValueError(f"incumbent skill repository cannot traverse symlinks: {source_repo}") + + relative_paths: list[str] = [] + total = 0 + for skill_name in skill_names: + _require_directory_chain( + source_repo, + Path(".claude") / "skills" / skill_name, + label="incumbent skill root", + ) + skill_root = source_repo / ".claude" / "skills" / skill_name + pending = [skill_root] + while pending: + directory = pending.pop() + try: + children = list(os.scandir(directory)) + except OSError as exc: + raise ValueError(f"incumbent skill directory is unreadable: {directory}: {exc}") from exc + for item in children: + path = Path(item.path) + if item.is_symlink(): + raise ValueError( + "incumbent skill seed cannot contain symlinks: " + f"{path.relative_to(source_repo)}" + ) + if item.is_dir(follow_symlinks=False): + pending.append(path) + continue + if not item.is_file(follow_symlinks=False): + raise ValueError( + "incumbent skill seed entries must be regular files: " + f"{path.relative_to(source_repo)}" + ) + total += item.stat(follow_symlinks=False).st_size + if total > MAX_SKILL_FINGERPRINT_BYTES: + raise ValueError("incumbent skill seed exceeds the bounded evidence limit") + relative = Path(".claude") / "skills" / skill_name / path.relative_to(skill_root) + content = _bounded_regular_bytes( + path, + limit=MAX_SKILL_FINGERPRINT_BYTES, + label="incumbent skill file", + ) + _replace_regular_file(worktree, relative, content) + relative_paths.append(PurePosixPath(relative.as_posix()).as_posix()) + + _commit_sandbox_paths( + sandbox, + relative_paths, + message="benchmark incumbent review skill", + require_change=False, + ) + + def unexercised_overlay_skills(overlay: Path, candidate_arms: list[str]) -> list[str]: """Overlay skills that no selected candidate arm would ever load. diff --git a/eval/workflow_bench/evolve.py b/eval/workflow_bench/evolve.py index c1b95e133..24ae64387 100644 --- a/eval/workflow_bench/evolve.py +++ b/eval/workflow_bench/evolve.py @@ -82,6 +82,7 @@ from .proposer_sandbox import ( SandboxError, build_sandbox_environment, preflight_bubblewrap, + preflight_unsafe_host, pid_namespace_command, prepare_sandbox, redact_text, @@ -127,8 +128,8 @@ WORKTREE_PREPARATION_TIMEOUT_SECONDS = ( # from that commit. Use the overlay boundary rather than the current candidate # size so this helper remains conservative before the runner starts. PROMOTION_BASE_TIMEOUT_SECONDS = (1 + 3 * MAX_CANDIDATE_FILES) * GIT_COMMAND_TIMEOUT_SECONDS -ARM_SESSION_COUNTS = {"workflow": 2, "workflow_direct": 1} -ARM_WORKSPACE_SNAPSHOT_COUNTS = {"workflow": 2, "workflow_direct": 0} +ARM_SESSION_COUNTS = {"workflow": 2, "workflow_direct": 1, "review": 1} +ARM_WORKSPACE_SNAPSHOT_COUNTS = {"workflow": 2, "workflow_direct": 0, "review": 1} REPO_ROOT = Path(__file__).resolve().parents[2] @@ -692,6 +693,7 @@ def run_proposer( proposal_path: Path, evidence_bundle: Path, bwrap_bin: Path, + sandbox_backend: str = "bwrap", progress_label: str | None = None, ) -> dict[str, Any]: """Run one proposer in confinement and copy only validated outputs out.""" @@ -722,9 +724,13 @@ def run_proposer( bwrap_bin=bwrap_bin, read_only_mounts=[evidence_mount], preflight=False, + backend=sandbox_backend, ) as sandbox: + host_text = getattr(sandbox, "host_text", lambda value: value) + environment_builder = getattr(sandbox, "environment", build_sandbox_environment) + backend = getattr(sandbox, "backend", "bwrap") record = runner.run_claude( - prompt, + host_text(prompt), clone, claude_bin=sandbox.claude_bin, timeout=args.timeout, @@ -734,7 +740,7 @@ def run_proposer( auth_token=args.auth_token, base_url=args.base_url, model=args.proposer_model, - build_sandbox_environment=build_sandbox_environment, + build_sandbox_environment=environment_builder, ), # No permission_mode: CLAUDE_CODE_SUBPROCESS_ENV_SCRUB # forces "default", so requesting dontAsk only warns. Tools @@ -743,7 +749,10 @@ def run_proposer( # bare ignores --tools and imposes its own Bash/Edit/Read # ceiling, which would cost the proposer Grep and Glob. command_prefix=sandbox.command_prefix, - require_pid_namespace=True, + require_pid_namespace=getattr(sandbox, "require_pid_namespace", True), + permission_mode=( + "bypassPermissions" if backend == "host-unsafe" else None + ), settings_json=sandbox.settings_json, strict_mcp_config=True, mcp_config_json='{"mcpServers":{}}', @@ -796,6 +805,21 @@ def resolve_incumbent_arms(overlay: Path, explicit_arms: list[str] | None) -> li return required +def executed_benchmark_arms(incumbent_arms: Sequence[str]) -> list[str]: + """Incumbent/candidate pairs plus the review comparator when needed.""" + + paired = [arm for incumbent in incumbent_arms for arm in (incumbent, INCUMBENT_ARMS[incumbent])] + if "review" in incumbent_arms: + paired.insert(0, "ce_review") + return paired + + +def _timeout_arm_key(arm: str) -> str: + if arm == "ce_review": + return "review" + return CANDIDATE_ARMS.get(arm, arm) + + def generation_timeout_seconds( *, task_count: int, @@ -808,11 +832,14 @@ def generation_timeout_seconds( if task_count < 1 or runs < 1 or session_timeout < 1: raise ValueError("task count, runs, and session timeout must be positive") try: - session_slots = sum(2 * ARM_SESSION_COUNTS[arm] for arm in incumbent_arms) + executed = executed_benchmark_arms(incumbent_arms) + session_slots = sum(ARM_SESSION_COUNTS[_timeout_arm_key(arm)] for arm in executed) + workspace_snapshot_slots = sum( + ARM_WORKSPACE_SNAPSHOT_COUNTS[_timeout_arm_key(arm)] for arm in executed + ) except KeyError as exc: raise ValueError(f"unsupported evolution arm: {exc.args[0]}") from exc - paired_arm_cells = 2 * len(incumbent_arms) - workspace_snapshot_slots = sum(2 * ARM_WORKSPACE_SNAPSHOT_COUNTS[arm] for arm in incumbent_arms) + paired_arm_cells = len(executed) per_task_preparation = ( TASK_BINDING_GIT_PHASES * GIT_COMMAND_TIMEOUT_SECONDS + 2 * TASK_SNAPSHOT_TIMEOUT_SECONDS @@ -849,9 +876,7 @@ def runner_argv( proposer_model: str | None = None, ) -> list[str]: incumbent_arms = resolve_incumbent_arms(overlay_dir, args.arms) - paired_arms = [arm for incumbent in incumbent_arms for arm in (incumbent, INCUMBENT_ARMS[incumbent])] - if "review" in incumbent_arms: - paired_arms.insert(0, "ce_review") + paired_arms = executed_benchmark_arms(incumbent_arms) argv = [ sys.executable, "-m", @@ -897,6 +922,8 @@ def runner_argv( argv.append("--include-expensive") if args.ce_plugin_dir is not None: argv += ["--ce-plugin-dir", str(args.ce_plugin_dir), "--ce-plugin-version", args.ce_plugin_version] + if args.unsafe_no_bwrap: + argv.append("--unsafe-no-bwrap") return argv @@ -1186,6 +1213,12 @@ def build_parser() -> argparse.ArgumentParser: ) parser.add_argument("--ce-plugin-dir", type=Path, default=None) parser.add_argument("--ce-plugin-version", default=None) + parser.add_argument( + "--unsafe-no-bwrap", + action="store_true", + help="LOCAL DIAGNOSTICS ONLY: use PRoot path translation without filesystem, " + "network, or PID isolation; forbidden with --apply and in CI", + ) return parser @@ -1196,6 +1229,10 @@ def main() -> int: parser.error("--generations must be positive") if args.runs < 1 or args.timeout < 1: parser.error("--runs and --timeout must be positive") + if args.unsafe_no_bwrap and args.apply: + parser.error("--unsafe-no-bwrap cannot be combined with --apply") + if args.unsafe_no_bwrap and os.environ.get("CI"): + parser.error("--unsafe-no-bwrap is forbidden when CI is set") try: args.model = runner.normalized_model_identifier(args.model) args.proposer_model = runner.normalized_model_identifier( @@ -1213,6 +1250,8 @@ def main() -> int: parser.error(str(exc)) raise AssertionError("ArgumentParser.error() returned unexpectedly") requested_arms = args.arms or ["workflow", "workflow_direct"] + if args.unsafe_no_bwrap and requested_arms != ["review"]: + parser.error("--unsafe-no-bwrap is restricted to --arms review") if "review" in requested_arms and ( args.ce_plugin_dir is None or not args.ce_plugin_dir.expanduser().is_dir() @@ -1229,8 +1268,19 @@ def main() -> int: parser.error(str(exc)) selected_tasks = runner.selected_task_bindings(selected_task_rows) try: - bwrap_bin = preflight_bubblewrap() - require_claude_sandbox_helpers() + if args.unsafe_no_bwrap: + bwrap_bin = preflight_unsafe_host() + sandbox_backend = "host-unsafe" + print( + "WARNING: --unsafe-no-bwrap runs sessions directly on the host with no " + "containment; model and verifier processes can access the host filesystem, " + "network, and credentials.", + file=sys.stderr, + ) + else: + bwrap_bin = preflight_bubblewrap() + sandbox_backend = "bwrap" + require_claude_sandbox_helpers() except SandboxError as exc: parser.error(str(exc)) raise AssertionError("ArgumentParser.error() returned unexpectedly") @@ -1250,6 +1300,7 @@ def main() -> int: requested_arms=requested_arms, initial_overlay=initial_overlay, bwrap_bin=bwrap_bin, + sandbox_backend=sandbox_backend, ) finally: gateway.__exit__(None, None, None) @@ -1264,6 +1315,7 @@ def _run_generations( requested_arms: list[str], initial_overlay: Path | None, bwrap_bin: Path, + sandbox_backend: str, ) -> int: policy_binding = { "metric": args.promotion_metric, @@ -1344,6 +1396,7 @@ def _run_generations( proposal_path=gen_dir / "proposal.md", evidence_bundle=bundle, bwrap_bin=bwrap_bin, + sandbox_backend=sandbox_backend, progress_label=f"gen {generation} proposer", ) # Redact any API token echoed into the session record (e.g. an @@ -1383,18 +1436,21 @@ def _run_generations( print(f"[gen {generation}] promotion targets contain uncommitted or drifted bytes") return 1 print(f"[gen {generation}] benchmarking candidate…") + benchmark_argv = runner_argv( + args, + bench_dir, + frozen_overlay, + task_bindings=selected_tasks, + target_base_digests=target_base_digests, + proposer_model=generation_proposer_model, + ) + benchmark_command = ( + benchmark_argv + if sandbox_backend == "host-unsafe" + else pid_namespace_command(benchmark_argv, bwrap_bin=bwrap_bin) + ) bench = run_managed( - pid_namespace_command( - runner_argv( - args, - bench_dir, - frozen_overlay, - task_bindings=selected_tasks, - target_base_digests=target_base_digests, - proposer_model=generation_proposer_model, - ), - bwrap_bin=bwrap_bin, - ), + benchmark_command, timeout=generation_timeout_seconds( task_count=len(selected_task_rows), runs=args.runs, @@ -1402,7 +1458,7 @@ def _run_generations( incumbent_arms=incumbent_arms, ), env=runner_environment(args), - require_pid_namespace=True, + require_pid_namespace=sandbox_backend == "bwrap", # The sweep is the multi-hour phase; without this its per-run # progress lines only reach the log as a bounded tail, and only # when it fails. diff --git a/eval/workflow_bench/model_gateway.py b/eval/workflow_bench/model_gateway.py index 3899fda7b..553802609 100644 --- a/eval/workflow_bench/model_gateway.py +++ b/eval/workflow_bench/model_gateway.py @@ -42,6 +42,27 @@ _OPENAI_MODEL = re.compile( # shorter than that, so both ends of the loopback hop get the same generous # budget and the session fails on real errors instead of on the clock. GATEWAY_REQUEST_TIMEOUT_S = 1800 +# Importing LiteLLM alone costs ~17s on a cold container filesystem, and the +# proxy only binds its port after that. A budget tight enough to lose that race +# reads as "connection refused", which looks like a dead proxy rather than a +# slow import. +GATEWAY_READY_TIMEOUT_ENV = "GITNEXUS_BENCH_GATEWAY_READY_TIMEOUT_S" +DEFAULT_GATEWAY_READY_TIMEOUT_S = 180.0 + + +def gateway_ready_timeout_s() -> float: + """Startup budget for the loopback proxy, overridable for slow hosts.""" + + raw = (os.environ.get(GATEWAY_READY_TIMEOUT_ENV) or "").strip() + if not raw: + return DEFAULT_GATEWAY_READY_TIMEOUT_S + try: + value = float(raw) + except ValueError as exc: + raise ValueError(f"{GATEWAY_READY_TIMEOUT_ENV} must be a number of seconds, not {raw!r}") from exc + if value <= 0: + raise ValueError(f"{GATEWAY_READY_TIMEOUT_ENV} must be positive, not {raw!r}") + return value def is_openai_model(model: str) -> bool: @@ -241,12 +262,12 @@ class OpenAIGateway(AbstractContextManager["OpenAIGateway"]): openai_api_key: str, model_names: Sequence[str], work_dir: Path, - ready_timeout_s: float = 20.0, + ready_timeout_s: float | None = None, ) -> None: self.openai_api_key = openai_api_key self.model_names = tuple(model_names) self.work_dir = work_dir - self.ready_timeout_s = ready_timeout_s + self.ready_timeout_s = gateway_ready_timeout_s() if ready_timeout_s is None else ready_timeout_s self.auth_token = secrets.token_hex(16) self.port = _free_loopback_port() self.base_url = f"http://127.0.0.1:{self.port}" @@ -342,7 +363,13 @@ class OpenAIGateway(AbstractContextManager["OpenAIGateway"]): except (urllib.error.URLError, TimeoutError, ConnectionError, OSError) as exc: last_error = str(exc) time.sleep(0.1) - raise RuntimeError(f"OpenAI LiteLLM gateway was not ready on {self.base_url}: {last_error}") + detail = self.log_tail() + raise RuntimeError( + f"OpenAI LiteLLM gateway was not ready on {self.base_url} after " + f"{self.ready_timeout_s:.0f}s: {last_error}" + + (f"; proxy log: {detail}" if detail else "; proxy wrote no output yet") + + f" (raise {GATEWAY_READY_TIMEOUT_ENV} if this host is simply slow to import LiteLLM)" + ) class attach_openai_gateway(AbstractContextManager[argparse.Namespace]): diff --git a/eval/workflow_bench/oracle_assets.py b/eval/workflow_bench/oracle_assets.py index 1d1df7e99..f7c686506 100644 --- a/eval/workflow_bench/oracle_assets.py +++ b/eval/workflow_bench/oracle_assets.py @@ -24,6 +24,10 @@ MAX_ORACLE_PATH_BYTES = 240 MAX_ORACLE_COMMAND_BYTES = 8 * 1024 ORACLE_ENV_VAR = "GITNEXUS_BENCH_ORACLE_ROOT" HIDDEN_HARNESS_PATH = PurePosixPath("eval/workflow_bench") +# Historical PR diffs can still edit this tree. Sanitization deletes it before +# setup, so `git apply` must skip those hunks or it fails with +# `error: eval/workflow_bench/: No such file or directory`. +HIDDEN_HARNESS_APPLY_EXCLUDE = f"{HIDDEN_HARNESS_PATH.as_posix()}/*" MAX_CLONE_REFS = 1024 MAX_CLONE_REF_BYTES = 2 * 1024 * 1024 @@ -240,6 +244,36 @@ def _git_checked( return result.stdout_tail.strip() +def with_hidden_harness_apply_exclude(setup: str) -> str: + """Skip hunks for the harness tree sanitization already deleted. + + Review cells copy a historical PR patch back under ``eval/workflow_bench`` + and then ``git apply`` it. That patch may still mention harness files + (``learnings.jsonl`` on review-pr-2718-defect). Those files are gone from + the parentless snapshot, and setup deletes the directory again after apply, + so the hunks are never model-visible. + """ + + if "git apply" not in setup: + return setup + flag = f"--exclude='{HIDDEN_HARNESS_APPLY_EXCLUDE}'" + if flag in setup or f'--exclude="{HIDDEN_HARNESS_APPLY_EXCLUDE}"' in setup: + return setup + return setup.replace("git apply ", f"git apply {flag} ", 1) + + +def review_case_setup_command(patch_name: str) -> str: + """Setup that applies one review-case patch and then hides the harness.""" + + if not patch_name or "/" in patch_name or "\\" in patch_name or patch_name in {".", ".."}: + raise ValueError(f"review patch name must be a single path segment: {patch_name!r}") + patch = f"{HIDDEN_HARNESS_PATH.as_posix()}/review_cases/{patch_name}" + return ( + f"git apply --exclude='{HIDDEN_HARNESS_APPLY_EXCLUDE}' {patch} " + f"&& rm -rf {HIDDEN_HARNESS_PATH.as_posix()}" + ) + + def sanitize_clone_for_hidden_oracles(clone: Path) -> str: """Remove the harness and its recoverable Git history from a disposable clone. diff --git a/eval/workflow_bench/proposer_sandbox.py b/eval/workflow_bench/proposer_sandbox.py index a1bafbd62..cc6f714a4 100644 --- a/eval/workflow_bench/proposer_sandbox.py +++ b/eval/workflow_bench/proposer_sandbox.py @@ -82,13 +82,58 @@ class SandboxSession: home: Path temp: Path bwrap_bin: Path + backend: str claude_host_bin: Path command_prefix: list[str] read_only_mounts: tuple[ReadOnlyMount, ...] + @property + def require_pid_namespace(self) -> bool: + return self.backend == "bwrap" + + def host_path(self, raw: str | Path) -> str: + """Translate a sandbox path to its real host path in unsafe mode.""" + + value = str(raw) + if self.backend == "bwrap": + return value + node = Path(shutil.which("node") or "/usr/bin/node").resolve() + mappings = [ + *(self.read_only_mounts), + ReadOnlyMount(node.parent.parent, SANDBOX_NODE_PREFIX), + ReadOnlyMount(node, SANDBOX_NODE), + ReadOnlyMount(self.claude_host_bin, SANDBOX_CLAUDE), + ReadOnlyMount(self.clone, SANDBOX_WORKSPACE), + ReadOnlyMount(self.home, SANDBOX_HOME), + ReadOnlyMount(self.temp, SANDBOX_TMP), + ] + for mount in sorted(mappings, key=lambda item: len(item.target), reverse=True): + target = mount.target.rstrip("/") + if value == target: + return str(mount.source) + if value.startswith(f"{target}/"): + return str(mount.source / value[len(target) + 1 :]) + return value + + def host_text(self, value: str) -> str: + if self.backend == "bwrap": + return value + targets = [ + *(mount.target for mount in self.read_only_mounts), + SANDBOX_NODE_PREFIX, + SANDBOX_NODE, + SANDBOX_CLAUDE, + SANDBOX_WORKSPACE, + SANDBOX_HOME, + SANDBOX_TMP, + ] + ordered = sorted(set(targets), key=len, reverse=True) + pattern = re.compile("|".join(re.escape(target) for target in ordered)) + return pattern.sub(lambda match: self.host_path(match.group(0)), value) + @property def claude_bin(self) -> str: - return SANDBOX_CLAUDE + return self.host_path(SANDBOX_CLAUDE) @property def transcript_projects(self) -> Path: @@ -96,7 +141,7 @@ class SandboxSession: @property def settings_json(self) -> str: - return build_claude_settings() + return build_claude_settings(sandbox_enabled=self.backend == "bwrap") def environment( self, @@ -104,7 +149,17 @@ class SandboxSession: auth_token: str | None = None, base_url: str | None = None, ) -> dict[str, str]: - return build_sandbox_environment(auth_token=auth_token, base_url=base_url) + env = build_sandbox_environment(auth_token=auth_token, base_url=base_url) + if self.backend != "bwrap": + env = {key: self.host_text(value) for key, value in env.items()} + env.pop("CLAUDE_CODE_SUBPROCESS_ENV_SCRUB", None) + env.pop("CLAUDE_CODE_SHELL_PREFIX", None) + # Sandbox-only bin dirs have no single host counterpart; drop the + # entries that survive translation as non-existent paths. + env["PATH"] = ":".join( + entry for entry in env["PATH"].split(":") if Path(entry).is_dir() + ) + return env def run( self, @@ -114,11 +169,17 @@ class SandboxSession: env: Mapping[str, str] | None = None, stdin_data: bytes | None = None, ) -> ManagedProcessResult: + translated = [self.host_text(str(part)) for part in command] return run_managed( - [*self.command_prefix, *command], + [*self.command_prefix, *translated], + cwd=None if self.command_prefix else self.clone, timeout=timeout, - env=dict(env) if env is not None else build_sandbox_environment(), - require_pid_namespace=True, + env=( + {key: self.host_text(value) for key, value in env.items()} + if env is not None + else self.environment() + ), + require_pid_namespace=self.require_pid_namespace, stdin_data=stdin_data, ) @@ -139,6 +200,9 @@ class SandboxSession: for harness-owned, post-session evidence such as hidden oracles. """ + if self.backend != "bwrap": + return [] + additional: list[ReadOnlyMount] = [] clone = _real_directory(self.clone, label="sandbox clone") for raw_path in read_only_paths: @@ -347,7 +411,7 @@ def build_sandbox_environment( return env -def build_claude_settings() -> str: +def build_claude_settings(*, sandbox_enabled: bool = True) -> str: """Inline settings that keep every Bash sandboxed and pre-approve the tools. Deliberately hook-free: headless ``claude -p`` (2.1.247) never dispatches @@ -358,11 +422,16 @@ def build_claude_settings() -> str: below, ``--tools``/``--allowedTools``, and the bwrap mounts. """ + permissions = { + "allow": ["Read", "Grep", "Glob", "Bash"], + } + if sandbox_enabled: + permissions["disableBypassPermissionsMode"] = "disable" settings = { "sandbox": { - "enabled": True, - "failIfUnavailable": True, - "autoAllowBashIfSandboxed": True, + "enabled": sandbox_enabled, + "failIfUnavailable": sandbox_enabled, + "autoAllowBashIfSandboxed": sandbox_enabled, "allowUnsandboxedCommands": False, "enableWeakerNestedSandbox": True, "network": { @@ -398,13 +467,16 @@ def build_claude_settings() -> str: # allow rule, so pre-approve the proposer's exact tool surface. Bash # is the only writable tool under --bare (it writes the candidate # overlay) and stays sandbox-confined by the sandbox.* policy above. - "allow": ["Read", "Grep", "Glob", "Bash"], - "disableBypassPermissionsMode": "disable", - }, - "env": { - "CLAUDE_CODE_SUBPROCESS_ENV_SCRUB": "1", - "CLAUDE_CODE_DONT_INHERIT_ENV": "1", + **permissions, }, + "env": ( + { + "CLAUDE_CODE_SUBPROCESS_ENV_SCRUB": "1", + "CLAUDE_CODE_DONT_INHERIT_ENV": "1", + } + if sandbox_enabled + else {"CLAUDE_CODE_DONT_INHERIT_ENV": "1"} + ), } return json.dumps(settings, sort_keys=True, separators=(",", ":")) @@ -570,6 +642,12 @@ def preflight_bubblewrap(bwrap_bin: Path | str | None = None) -> Path: return bwrap +def preflight_unsafe_host() -> Path: + """Return a harmless sentinel for the explicit non-containment backend.""" + + return _resolve_executable(None, "env") + + def pid_namespace_command( command: Sequence[str], *, @@ -778,6 +856,145 @@ def _sandbox_command_prefix( return args +def _drop_host_workspace_write_bits( + root: Path, + *, + writable: Sequence[Path], +) -> list[tuple[Path, int]]: + """Clear write bits on a host-unsafe clone except explicit artifact paths.""" + + root = root.expanduser().absolute() + try: + metadata = root.lstat() + resolved = root.resolve(strict=True) + except OSError as exc: + raise SandboxError(f"host workspace lock root is unavailable: {root}: {exc}") from exc + if stat.S_ISLNK(metadata.st_mode) or not stat.S_ISDIR(metadata.st_mode) or resolved != root: + raise SandboxError(f"host workspace lock root must be a real directory: {root}") + + allowed: set[Path] = set() + for raw in writable: + path = raw.expanduser().absolute() + try: + path.relative_to(root) + except ValueError as exc: + raise SandboxError(f"writable host path escapes the workspace: {raw}") from exc + allowed.add(path) + + records: list[tuple[Path, int]] = [] + pending = [root] + while pending: + current = pending.pop() + try: + metadata = current.lstat() + except OSError as exc: + raise SandboxError(f"host workspace lock path is unreadable: {current}: {exc}") from exc + records.append((current, stat.S_IMODE(metadata.st_mode))) + if stat.S_ISLNK(metadata.st_mode): + continue + if stat.S_ISDIR(metadata.st_mode): + try: + children = [Path(entry.path) for entry in os.scandir(current)] + except OSError as exc: + raise SandboxError(f"host workspace lock directory is unreadable: {current}: {exc}") from exc + pending.extend(children) + if current not in allowed: + os.chmod(current, 0o555) + continue + if current in allowed: + os.chmod(current, stat.S_IMODE(metadata.st_mode) | 0o222) + continue + if stat.S_ISREG(metadata.st_mode): + os.chmod(current, stat.S_IMODE(metadata.st_mode) & ~0o222) + return records + + +def _force_rmtree(path: Path) -> None: + """Delete a tree even when leftover copies inherited 0555 directory modes. + + Host-unsafe review sessions drop write bits on the clone. An agent that + ``copytree``s those directories into the private TMPDIR leaves a tree + ``shutil.rmtree`` cannot remove: a non-empty 0555 directory raises + PermissionError. Restore owner write bits, then delete. + """ + + root = Path(path) + if not root.exists(): + return + for dirpath, _dirnames, filenames in os.walk(root, topdown=False, followlinks=False): + try: + os.chmod( + dirpath, + stat.S_IMODE(os.lstat(dirpath).st_mode) | 0o700, + follow_symlinks=False, + ) + except OSError: + pass + for name in filenames: + child = os.path.join(dirpath, name) + try: + metadata = os.lstat(child) + except OSError: + continue + if stat.S_ISLNK(metadata.st_mode): + continue + try: + os.chmod(child, stat.S_IMODE(metadata.st_mode) | 0o200, follow_symlinks=False) + except OSError: + pass + shutil.rmtree(root) + + +def _restore_host_workspace_modes(records: Sequence[tuple[Path, int]]) -> None: + errors: list[str] = [] + for path, mode in reversed(records): + try: + os.chmod(path, mode) + except FileNotFoundError: + continue + except OSError as exc: + errors.append(f"{path}: {exc}") + if errors: + raise SandboxError("failed to restore host workspace modes: " + "; ".join(errors[:8])) + + +@contextmanager +def host_workspace_write_boundary( + root: Path, + *, + writable: Sequence[Path] = (), +) -> Iterator[None]: + """Best-effort host analog of bwrap ``--ro-bind`` plus one writable artifact. + + This is not a security boundary: a session that can ``chmod`` can undo it. + It is enough to stop accidental ``npm install`` / analyze writes from + invalidating review evidence on a host-unsafe diagnostic run. + """ + + records = _drop_host_workspace_write_bits(root, writable=writable) + try: + yield + finally: + _restore_host_workspace_modes(records) + + +@contextmanager +def sandbox_workspace_write_boundary( + sandbox: Any, + *, + read_only_workspace: bool, + writable: Sequence[Path] = (), +) -> Iterator[None]: + """Apply the host-unsafe write lock only when bwrap is not the backend.""" + + if getattr(sandbox, "backend", "bwrap") != "host-unsafe" or not read_only_workspace: + yield + return + clone = Path(sandbox.clone) + with host_workspace_write_boundary(clone, writable=writable): + yield + + @contextmanager def prepare_sandbox( *, @@ -786,17 +1003,21 @@ def prepare_sandbox( bwrap_bin: Path | str | None = None, read_only_mounts: Sequence[ReadOnlyMount] = (), preflight: bool = True, + backend: str = "bwrap", ) -> Iterator[SandboxSession]: - """Create private host backing dirs and one immutable Bubblewrap command.""" + """Create private host backing dirs and one virtualized command.""" # Validate the lexical path before resolving it. Resolving first would # erase the evidence that the caller supplied a symlinked clone root. clone = _real_directory(clone, label="sandbox clone") + if backend not in ("bwrap", "host-unsafe"): + raise SandboxError(f"unknown sandbox backend: {backend}") if preflight: - bwrap = preflight_bubblewrap(bwrap_bin) - require_claude_sandbox_helpers() + bwrap = preflight_bubblewrap(bwrap_bin) if backend == "bwrap" else preflight_unsafe_host() + if backend == "bwrap": + require_claude_sandbox_helpers() else: - bwrap = _resolve_executable(bwrap_bin, "bwrap") + bwrap = _resolve_executable(bwrap_bin, "bwrap") if backend == "bwrap" else preflight_unsafe_host() claude = _resolve_executable(claude_bin, "claude") private_root = Path(tempfile.mkdtemp(prefix="wfbench-sandbox-")) private_root.chmod(0o700) @@ -825,13 +1046,17 @@ def prepare_sandbox( ) primary: BaseException | None = None try: - command_prefix = _sandbox_command_prefix( - bwrap=bwrap, - clone=clone, - home=home, - temp=temp, - claude_bin=claude, - mounts=protected_mounts, + command_prefix = ( + _sandbox_command_prefix( + bwrap=bwrap, + clone=clone, + home=home, + temp=temp, + claude_bin=claude, + mounts=protected_mounts, + ) + if backend == "bwrap" + else [] ) yield SandboxSession( private_root=private_root, @@ -839,6 +1064,7 @@ def prepare_sandbox( home=home, temp=temp, bwrap_bin=bwrap, + backend=backend, claude_host_bin=claude, command_prefix=command_prefix, read_only_mounts=protected_mounts, @@ -848,7 +1074,7 @@ def prepare_sandbox( raise finally: try: - shutil.rmtree(private_root) + _force_rmtree(private_root) except OSError as cleanup: if primary is None: raise diff --git a/eval/workflow_bench/run-evolution.sh b/eval/workflow_bench/run-evolution.sh index 23ad82248..c9d2a77a1 100755 --- a/eval/workflow_bench/run-evolution.sh +++ b/eval/workflow_bench/run-evolution.sh @@ -15,6 +15,7 @@ # MODEL PROPOSER_MODEL EFFORT GENERATIONS RUNS WORKERS PROVIDER # EVOLUTION_PROFILE CE_PLUGIN_DIR CE_PLUGIN_VERSION # INCLUDE_EXPENSIVE SEED_RESULTS CLAUDE_BIN OUT_ROOT +# UNSAFE_NO_BWRAP=1 (local review diagnostics only) # GITNEXUS_BENCH_ANTHROPIC_API_KEY (legacy GITNEXUS_BENCH_AUTH_TOKEN) # GITNEXUS_BENCH_OPENAI_API_KEY set -euo pipefail @@ -181,6 +182,13 @@ fi if [[ -n "${SEED_RESULTS}" ]]; then cmd+=(--seed-results "${SEED_RESULTS}") fi +if [[ -n "${UNSAFE_NO_BWRAP:-}" && "${UNSAFE_NO_BWRAP}" != "0" && "${UNSAFE_NO_BWRAP}" != "false" ]]; then + if [[ -n "${CI:-}" ]]; then + echo "UNSAFE_NO_BWRAP is forbidden in CI." >&2 + exit 1 + fi + cmd+=(--unsafe-no-bwrap) +fi if ((${#passthrough[@]})); then cmd+=("${passthrough[@]}") fi @@ -200,13 +208,15 @@ runtime_digest="$( sha256sum "${eval_dir}/../gitnexus-shared/package-lock.json" } | sha256sum | cut -d' ' -f1 )" -SOURCE_SHA="${source_sha}" RUNTIME_DIGEST="${runtime_digest}" \ +unsafe_backend="$([[ -n "${UNSAFE_NO_BWRAP:-}" && "${UNSAFE_NO_BWRAP}" != "0" && "${UNSAFE_NO_BWRAP}" != "false" ]] && echo host-unsafe || echo bwrap)" +SOURCE_SHA="${source_sha}" RUNTIME_DIGEST="${runtime_digest}" SANDBOX_BACKEND="${unsafe_backend}" \ node -e 'require("fs").writeFileSync(process.argv[1], JSON.stringify({ schema_version: 1, source_sha: process.env.SOURCE_SHA, runtime_digest: process.env.RUNTIME_DIGEST, profile: process.env.EVOLUTION_PROFILE, - ce_plugin_version: process.env.CE_PLUGIN_VERSION + ce_plugin_version: process.env.CE_PLUGIN_VERSION, + sandbox_backend: process.env.SANDBOX_BACKEND }, null, 2) + "\n")' "${out_root}/runtime-provenance.json" export PYTHONUNBUFFERED=1 diff --git a/eval/workflow_bench/runner.py b/eval/workflow_bench/runner.py index e000f5622..e0d972e3f 100644 --- a/eval/workflow_bench/runner.py +++ b/eval/workflow_bench/runner.py @@ -43,6 +43,7 @@ import re import secrets import stat import statistics +import sys import tempfile import time from collections.abc import Callable, Mapping, Sequence @@ -68,6 +69,7 @@ from .evolution import ( evaluate_candidate, evaluate_review_candidate, required_candidate_arms, + seed_evaluated_skills, skill_fingerprint, ) from .model_gateway import ( @@ -83,6 +85,7 @@ from .oracle_assets import ( capture_task_oracles, sanitize_clone_for_hidden_oracles, staged_task_oracle, + with_hidden_harness_apply_exclude, ) from .process_control import ManagedProcessError from .promotion_apply import committed_destination_base_digests @@ -98,9 +101,11 @@ from .proposer_sandbox import ( SandboxSession, build_sandbox_environment, preflight_bubblewrap, + preflight_unsafe_host, prepare_sandbox, redact_text, require_claude_sandbox_helpers, + sandbox_workspace_write_boundary, ) from .runner_artifacts import ( IMPLEMENTATION_ARMS, @@ -329,7 +334,7 @@ def _run_hidden_oracle( extra_read_only_mounts=(ReadOnlyMount(source=stage_root, target=oracle_mount),), ), env=oracle_env, - require_pid_namespace=True, + require_pid_namespace=getattr(sandbox, "require_pid_namespace", True), ) ) # Candidate code executes in this process. Never persist its stdout @@ -423,11 +428,15 @@ def run_arm( oracle_snapshot: TaskOracleSnapshot | None = None, ) -> dict[str, Any]: sessions: list[dict[str, Any]] = [] + environment_builder = getattr(sandbox, "environment", build_sandbox_environment) + host_text = getattr(sandbox, "host_text", lambda value: value) + host_path = getattr(sandbox, "host_path", lambda value: str(value)) + backend = getattr(sandbox, "backend", "bwrap") env = model_session_environment( auth_token=args.auth_token, base_url=args.base_url, model=args.model, - build_sandbox_environment=build_sandbox_environment, + build_sandbox_environment=environment_builder, ) # --bare hard-disables the Skill tool and every mcp__* tool — by Claude # Code design, not a bug (--allowedTools can't restore what --bare @@ -444,15 +453,17 @@ def run_arm( "model": args.model, "effort": args.effort, "env": env, - "permission_mode": "dontAsk", + "permission_mode": ( + "bypassPermissions" if backend == "host-unsafe" else "dontAsk" + ), "command_prefix": sandbox.command_prefix_for( read_only_paths=_evaluated_skill_roots(worktree, arm), ), - "require_pid_namespace": True, + "require_pid_namespace": getattr(sandbox, "require_pid_namespace", True), "bare": bare, "settings_json": sandbox.settings_json, "strict_mcp_config": True, - "mcp_config_json": sandbox_mcp_config(), + "mcp_config_json": host_text(sandbox_mcp_config()), "transcript_projects": sandbox.transcript_projects, "transcript_cwd": Path(SANDBOX_WORKSPACE), "transcript_wait_seconds": 5, @@ -461,7 +472,7 @@ def run_arm( "transcript_secrets": tuple(credential_secrets(args)), } if ce_plugin_dir is not None: - common["plugin_dirs"] = (ce_plugin_dir,) + common["plugin_dirs"] = (host_path(ce_plugin_dir),) expected_skills = ARM_EXPECTED_SKILLS.get(arm, ()) plan_doc: Path | None = None if arm in ("workflow", "ce_workflow"): @@ -538,7 +549,7 @@ def run_arm( phase_before = workspace_snapshot(worktree) if enforce_phase_boundary else None review_common = { **common, - "allowed_tools": allowed_agent_tools(implementation=False), + "allowed_tools": allowed_agent_tools(implementation=False, allow_edit=False), "command_prefix": sandbox.command_prefix_for( read_only_workspace=True, read_only_paths=_evaluated_skill_roots(worktree, arm), @@ -550,12 +561,17 @@ def run_arm( ), ), } - review_session = run_claude( - review_prompt.format(task=task["prompt"]), - worktree, - expected_skill=expected_skills[0], - **review_common, - ) + with sandbox_workspace_write_boundary( + sandbox, + read_only_workspace=True, + writable=(review_output,), + ): + review_session = run_claude( + host_text(review_prompt.format(task=task["prompt"])), + worktree, + expected_skill=expected_skills[0], + **review_common, + ) sessions.append(review_session) if review_session["ok"] and phase_before is not None: try: @@ -627,8 +643,8 @@ def run_arm( read_only_workspace=True, unshare_network=True, ), - env=build_sandbox_environment(), - require_pid_namespace=True, + env=environment_builder(), + require_pid_namespace=getattr(sandbox, "require_pid_namespace", True), ) ) review_score: dict[str, Any] | None = None @@ -849,6 +865,7 @@ class TaskCellContext: runtime_mounts: tuple[ReadOnlyMount, ...] candidate_overlay: Path | None overlay_digest: str | None + sandbox_backend: str = "bwrap" def run_cell(ctx: TaskCellContext, run_idx: int, arm: str) -> dict[str, Any]: @@ -909,6 +926,7 @@ def run_cell(ctx: TaskCellContext, run_idx: int, arm: str) -> dict[str, Any]: *oracle_visibility_mounts, ], preflight=False, + backend=ctx.sandbox_backend, ) as sandbox: # Capture the BASE (pre-overlay) skill digest — identical # for the incumbent and candidate arms — then run the @@ -916,9 +934,22 @@ def run_cell(ctx: TaskCellContext, run_idx: int, arm: str) -> dict[str, Any]: # candidate overlay is applied only afterwards, so setup # can never observe candidate prose and both arms share # byte-identical pre-overlay state. + # Historical review SHAs may predate gitnexus-review. Seed + # the current evaluated skill first so fingerprinting and + # the model see the same incumbent prose on every case. + if execution_arm == "review": + seed_evaluated_skills( + HARNESS_ROOT, + worktree, + sandbox=sandbox, + arm=execution_arm, + ) base_skill_digest = skill_fingerprint(worktree, execution_arm) if task.get("setup"): - setup_command = ["/bin/sh", "-lc", str(task["setup"])] + # Sanitization already removed eval/workflow_bench. A historical + # PR patch that still edits that tree (gitignored learnings.jsonl) + # must skip those hunks or `git apply` fails closed. + setup_command = ["/bin/sh", "-lc", with_hidden_harness_apply_exclude(str(task["setup"]))] setup = sandbox.run( setup_command, timeout=600, @@ -1058,6 +1089,7 @@ def run_cell(ctx: TaskCellContext, run_idx: int, arm: str) -> dict[str, Any]: "task": task["id"], "class": task.get("class", ""), "run": run_idx, + "sandbox_backend": ctx.sandbox_backend, "task_asset_snapshot_digest": (ctx.asset_snapshot.digest if ctx.asset_snapshot is not None else None), "task_asset_manifest_digest": ( ctx.asset_snapshot.manifest_digest if ctx.asset_snapshot is not None else None @@ -1498,6 +1530,7 @@ def build_parser() -> argparse.ArgumentParser: parser.add_argument("--promotion-max-task-regression", type=float, default=20.0) parser.add_argument("--task-bindings-json", default=None, help=argparse.SUPPRESS) parser.add_argument("--promotion-target-bases-json", default=None, help=argparse.SUPPRESS) + parser.add_argument("--unsafe-no-bwrap", action="store_true", help=argparse.SUPPRESS) return parser @@ -1574,9 +1607,24 @@ def main() -> None: parser.error("--promotion-target-bases-json requires --candidate-overlay") promotion_target_bases = {} + if args.unsafe_no_bwrap and os.environ.get("CI"): + parser.error("--unsafe-no-bwrap is forbidden when CI is set") + if args.unsafe_no_bwrap and args.arms != ["ce_review", "review", "candidate_review"]: + parser.error("--unsafe-no-bwrap is restricted to the paired review arms") try: - bwrap_bin = preflight_bubblewrap() - require_claude_sandbox_helpers() + if args.unsafe_no_bwrap: + bwrap_bin = preflight_unsafe_host() + sandbox_backend = "host-unsafe" + print( + "WARNING: --unsafe-no-bwrap runs sessions directly on the host with no " + "containment; model and verifier processes can access the host filesystem, " + "network, and credentials.", + file=sys.stderr, + ) + else: + bwrap_bin = preflight_bubblewrap() + sandbox_backend = "bwrap" + require_claude_sandbox_helpers() runtime_mounts = trusted_gitnexus_runtime_mounts() except SandboxError as exc: parser.error(str(exc)) @@ -1597,6 +1645,7 @@ def main() -> None: expected_task_bindings=expected_task_bindings, ce_plugin_config=ce_plugin_config, bwrap_bin=bwrap_bin, + sandbox_backend=sandbox_backend, runtime_mounts=runtime_mounts, candidate_arms=candidate_arms, candidate_overlay=candidate_overlay, @@ -1617,6 +1666,7 @@ def _run_sweep( expected_task_bindings: Any, ce_plugin_config: Any, bwrap_bin: Any, + sandbox_backend: str, runtime_mounts: Any, candidate_arms: list[str], candidate_overlay: Path | None, @@ -1691,6 +1741,7 @@ def _run_sweep( cache=task_asset_cache, claude_bin=args.claude_bin, bwrap_bin=bwrap_bin, + sandbox_backend=sandbox_backend, runtime_mounts=runtime_mounts, ) graph_snapshots[graph_key] = graph_snapshot @@ -1724,6 +1775,7 @@ def _run_sweep( ce_plugin_snapshot=ce_plugin_snapshot, trees_dir=Path(trees), bwrap_bin=bwrap_bin, + sandbox_backend=sandbox_backend, runtime_mounts=runtime_mounts, candidate_overlay=candidate_overlay, overlay_digest=overlay_digest, diff --git a/eval/workflow_bench/runner_sessions.py b/eval/workflow_bench/runner_sessions.py index dc2641320..4279facdc 100644 --- a/eval/workflow_bench/runner_sessions.py +++ b/eval/workflow_bench/runner_sessions.py @@ -361,8 +361,13 @@ def sandbox_mcp_config() -> str: return json.dumps(config, sort_keys=True, separators=(",", ":")) -def allowed_agent_tools(*, implementation: bool, include_mcp: bool = True) -> list[str]: - tools = [*BUILTIN_AGENT_TOOLS] +def allowed_agent_tools( + *, + implementation: bool, + include_mcp: bool = True, + allow_edit: bool = True, +) -> list[str]: + tools = [tool for tool in BUILTIN_AGENT_TOOLS if allow_edit or tool != "Edit"] if include_mcp: tools.extend(GITNEXUS_READ_ONLY_TOOLS) if include_mcp and implementation: @@ -461,7 +466,13 @@ def _persist_parent_event_stream( def _normalized_skill_identifier(value: Any) -> str | None: - """Return the exact identifier token accepted by the Skill tool.""" + """Return the exact identifier token accepted by the Skill tool. + + Plugin skills are requested as ``plugin:skill`` (observed in review + evolution: ``compound-engineering:ce-code-review``). Compare on the + skill token after the last colon so a successful plugin-qualified + invocation still counts as the expected skill. + """ if not isinstance(value, str): return None @@ -471,6 +482,8 @@ def _normalized_skill_identifier(value: Any) -> str | None: token = stripped.split(maxsplit=1)[0] if token.startswith("/"): token = token[1:] + if ":" in token: + token = token.rsplit(":", 1)[-1] return token or None diff --git a/eval/workflow_bench/runtime_mounts.py b/eval/workflow_bench/runtime_mounts.py index 04bacdf55..66bb21bcf 100644 --- a/eval/workflow_bench/runtime_mounts.py +++ b/eval/workflow_bench/runtime_mounts.py @@ -146,24 +146,91 @@ def _validated_runtime_root(path: Path, *, label: str) -> Path: return root +def _primary_checkout_root(harness_root: Path) -> Path | None: + """Return the main worktree when *harness_root* is a linked git worktree. + + Linked worktrees store a regular ``.git`` file pointing at + ``/.git/worktrees/``. A symlink or oversized file is + ignored so this helper cannot be used to follow an unexpected path. + """ + + git_file = harness_root / ".git" + try: + metadata = git_file.lstat() + except OSError: + return None + if stat.S_ISLNK(metadata.st_mode) or not stat.S_ISREG(metadata.st_mode): + return None + if metadata.st_size > 4096: + return None + try: + text = git_file.read_text(encoding="utf-8").strip() + except (OSError, UnicodeError): + return None + if not text.startswith("gitdir:"): + return None + raw = text.split(":", 1)[1].strip() + if not raw: + return None + gitdir = Path(raw) + if not gitdir.is_absolute(): + gitdir = harness_root / gitdir + if gitdir.parent.name != "worktrees" or gitdir.parent.parent.name != ".git": + return None + primary = gitdir.parent.parent.parent + try: + primary_git = (primary / ".git").lstat() + except OSError: + return None + if stat.S_ISLNK(primary_git.st_mode) or not stat.S_ISDIR(primary_git.st_mode): + return None + resolved_primary = primary.resolve() + if resolved_primary == harness_root.resolve(): + return None + return resolved_primary + + def _validated_runtime_component( root: Path, relative: str, target: str, *, directory: bool, + allow_primary_worktree_symlink: bool = False, ) -> ReadOnlyMount: """Validate one direct runtime component before exposing only that path.""" source = root / relative + kind = "directory" if directory else "file" try: mode = source.lstat().st_mode resolved = source.resolve(strict=True) except OSError as exc: raise SandboxError(f"pinned GitNexus runtime component is unavailable: {source}: {exc}") from exc + if stat.S_ISLNK(mode) and allow_primary_worktree_symlink and directory: + primary = _primary_checkout_root(HARNESS_ROOT) + if primary is None: + raise SandboxError(f"pinned GitNexus runtime component must be a real {kind}: {source}") + expected = primary / "gitnexus" / relative + try: + expected_mode = expected.lstat().st_mode + expected_resolved = expected.resolve(strict=True) + except OSError as exc: + raise SandboxError( + f"pinned GitNexus runtime component symlink must point at the primary checkout: {source}" + ) from exc + if ( + stat.S_ISLNK(expected_mode) + or not stat.S_ISDIR(expected_mode) + or expected_resolved != expected + or resolved != expected_resolved + ): + raise SandboxError( + f"pinned GitNexus runtime component symlink must point at the primary checkout: {source}" + ) + return ReadOnlyMount(source=expected_resolved, target=target) expected_type = stat.S_ISDIR(mode) if directory else stat.S_ISREG(mode) if stat.S_ISLNK(mode) or not expected_type or resolved != source: - kind = "directory" if directory else "file" raise SandboxError(f"pinned GitNexus runtime component must be a real {kind}: {source}") return ReadOnlyMount(source=source, target=target) @@ -197,6 +264,7 @@ def trusted_gitnexus_runtime_mounts() -> tuple[ReadOnlyMount, ...]: "node_modules", f"{SANDBOX_GITNEXUS}/node_modules", directory=True, + allow_primary_worktree_symlink=True, ), _validated_runtime_component( runtime, @@ -236,7 +304,19 @@ def trusted_gitnexus_runtime_mounts() -> tuple[ReadOnlyMount, ...]: raise SandboxError("pinned GitNexus runtime package.json has no version") linked_shared = mounts[2].source / "gitnexus-shared" - if not linked_shared.is_symlink() or linked_shared.resolve(strict=True) != shared: + allowed_shared = {shared} + primary = _primary_checkout_root(HARNESS_ROOT) + if primary is not None: + try: + allowed_shared.add( + _validated_runtime_root( + primary / "gitnexus-shared", + label="primary GitNexus shared runtime", + ) + ) + except SandboxError: + pass + if not linked_shared.is_symlink() or linked_shared.resolve(strict=True) not in allowed_shared: raise SandboxError("pinned GitNexus runtime has an unexpected gitnexus-shared dependency") try: shared_package = json.loads(mounts[5].source.read_text()) diff --git a/eval/workflow_bench/sanitized_graph.py b/eval/workflow_bench/sanitized_graph.py index 23339b1c7..3aea9e8c1 100644 --- a/eval/workflow_bench/sanitized_graph.py +++ b/eval/workflow_bench/sanitized_graph.py @@ -20,11 +20,12 @@ from .proposer_sandbox import ( SANDBOX_WORKSPACE, ReadOnlyMount, SandboxError, + SandboxSession, build_sandbox_environment, prepare_sandbox, ) from .runner_artifacts import make_worktree, remove_clone -from .task_assets import TaskAssetCache, TaskAssetSnapshot +from .task_assets import TaskAssetCache, TaskAssetSnapshot, _is_harness_sandbox_copy GRAPH_ASSET_PATHS = ( ".gitnexus/gitnexus.json", @@ -41,6 +42,12 @@ GRAPH_MARKERS = ( ) GRAPH_BUILD_TIMEOUT_SECONDS = 3600 GRAPH_QUERY_TIMEOUT_SECONDS = 300 +# CLI default is 5s. Parse-worker top-of-script init loads every required +# tree-sitter binding before it can post `{type:'ready'}`; on a loaded WSL +# host that handshake is >15s, and a GitNexus-sized --pdg analyze can slow it +# further. Integration tests already stub 60s; graph prep uses 120s so a +# replacement worker is not classified as deterministic-startup (#2649). +GRAPH_WORKER_READY_TIMEOUT_MS = 120_000 MAX_GRAPH_SCRUB_ENTRIES = 250_000 MAX_GRAPH_SCRUB_FILE_BYTES = 512 * 1024 MAX_GRAPH_SCRUB_TOTAL_BYTES = 2 * 1024 * 1024 * 1024 @@ -88,7 +95,13 @@ def validate_no_prebuilt_graph_assets(task: Mapping[str, Any]) -> None: if not isinstance(sandbox_copy, list): raise SandboxError("sandbox_copy must be a list") for value in sandbox_copy: - if isinstance(value, str) and _is_restricted_path(value): + if not isinstance(value, str): + continue + # Review corpus patches are harness-owned and applied in setup, then + # deleted with eval/workflow_bench. They are not a prebuilt graph. + if _is_harness_sandbox_copy(PurePosixPath(value)): + continue + if _is_restricted_path(value): raise SandboxError(f"sandbox_copy cannot import prebuilt graph or harness data: {value}") dependencies = task.get("sandbox_dependencies", []) @@ -235,6 +248,7 @@ def _graph_environment() -> dict[str, str]: "GITNEXUS_NO_GITIGNORE": "1", "GITNEXUS_WORKER_POOL_SIZE": "1", "GITNEXUS_PARSE_CHUNK_CONCURRENCY": "1", + "GITNEXUS_WORKER_READY_TIMEOUT_MS": str(GRAPH_WORKER_READY_TIMEOUT_MS), } ) return env @@ -244,20 +258,23 @@ def _run_graph_cli( prefix: Sequence[str], arguments: Sequence[str], *, + sandbox: SandboxSession | None = None, timeout: int, capture_stdout: bool = False, ) -> bytes | None: + host_path = sandbox.host_path if sandbox is not None else str + host_text = sandbox.host_text if sandbox is not None else str command = [ *prefix, - SANDBOX_NODE, - SANDBOX_GITNEXUS_ENTRYPOINT, - *arguments, + host_path(SANDBOX_NODE), + host_path(SANDBOX_GITNEXUS_ENTRYPOINT), + *(host_text(argument) for argument in arguments), ] result = run_managed( command, timeout=timeout, - env=_graph_environment(), - require_pid_namespace=True, + env={key: host_text(value) for key, value in _graph_environment().items()}, + require_pid_namespace=(sandbox.require_pid_namespace if sandbox is not None else True), capture_stdout_bytes=(2 * 1024 * 1024 if capture_stdout else None), ) if not result.ok: @@ -286,12 +303,14 @@ def _parse_empty_query(raw: bytes, *, label: str) -> None: raise SandboxError(f"{label} found recoverable benchmark harness references") -def _scrub_and_verify_graph(prefix: Sequence[str]) -> None: +def _scrub_and_verify_graph(prefix: Sequence[str], sandbox: SandboxSession | None = None) -> None: node_predicate = _marker_predicate("n") relation_predicate = _marker_predicate("r") + sandbox_kwargs = {"sandbox": sandbox} if sandbox is not None else {} node_result = _run_graph_cli( prefix, ("cypher", f"MATCH (n) WHERE {node_predicate} RETURN n LIMIT 1", "-r", "benchmark-target", "--limit", "1"), + **sandbox_kwargs, timeout=GRAPH_QUERY_TIMEOUT_SECONDS, capture_stdout=True, ) @@ -305,6 +324,7 @@ def _scrub_and_verify_graph(prefix: Sequence[str]) -> None: "--limit", "1", ), + **sandbox_kwargs, timeout=GRAPH_QUERY_TIMEOUT_SECONDS, capture_stdout=True, ) @@ -339,6 +359,7 @@ def prepare_sanitized_graph( claude_bin: Path | str, bwrap_bin: Path | str, runtime_mounts: Sequence[ReadOnlyMount], + sandbox_backend: str = "bwrap", ) -> SanitizedGraphSnapshot: """Sanitize, index offline once, scrub, and freeze graph assets for all arms.""" @@ -355,8 +376,10 @@ def prepare_sanitized_graph( bwrap_bin=bwrap_bin, read_only_mounts=runtime_mounts, preflight=False, + backend=sandbox_backend, ) as sandbox: prefix = sandbox.command_prefix_for(unshare_network=True) + unsafe_sandbox = sandbox if getattr(sandbox, "backend", "bwrap") == "host-unsafe" else None _run_graph_cli( prefix, ( @@ -375,9 +398,10 @@ def prepare_sanitized_graph( "--workers", "1", ), + **({"sandbox": unsafe_sandbox} if unsafe_sandbox is not None else {}), timeout=GRAPH_BUILD_TIMEOUT_SECONDS, ) - _scrub_and_verify_graph(prefix) + _scrub_and_verify_graph(prefix, unsafe_sandbox) _validate_graph_metadata(seed, sanitized_head) assets = cache.prepare( {"sandbox_copy": list(GRAPH_ASSET_PATHS)}, diff --git a/eval/workflow_bench/task_assets.py b/eval/workflow_bench/task_assets.py index 6cd96d84d..e97d6dd1c 100644 --- a/eval/workflow_bench/task_assets.py +++ b/eval/workflow_bench/task_assets.py @@ -24,6 +24,7 @@ from dataclasses import dataclass from pathlib import Path, PurePosixPath from typing import Any +from . import runtime_mounts from .proposer_sandbox import ( DEPENDENCY_MOUNT_BASENAME, SANDBOX_WORKSPACE, @@ -63,6 +64,10 @@ _REFLINK_UNAVAILABLE = { DEPENDENCY_CONTENT_BINDING_FIELD = "sandbox_dependency_content_digest" DEPENDENCY_MANIFEST_BINDING_FIELD = "sandbox_dependency_manifest_digest" +# Review corpus patches are harness-owned. The task repo (`~/GitNexus`) is the +# subject checkout and often a different worktree or branch, so these paths +# must be read from the package that defined the benchmark. +HARNESS_SANDBOX_COPY_PREFIXES = (PurePosixPath("eval/workflow_bench/review_cases"),) @dataclass(frozen=True) @@ -191,7 +196,7 @@ class TaskAssetCache: self.root = root.expanduser().absolute() self.root.mkdir(mode=0o700, parents=True, exist_ok=False) self._by_definition: dict[ - tuple[str, str, tuple[str, ...], tuple[tuple[str, str], ...]], + tuple[str, str, tuple[str, ...], tuple[tuple[str, str], ...], str], TaskAssetSnapshot, ] = {} self._closed = False @@ -218,7 +223,14 @@ class TaskAssetCache: declarations, relative_paths = _sandbox_copy_declarations(task) dependency_declarations = _sandbox_dependency_declarations(task) dependency_identity = tuple((declaration.source, declaration.target) for declaration in dependency_declarations) - definition = (str(repo_identity), resolved_sha, declarations, dependency_identity) + harness_identity = _harness_sandbox_copy_identity(relative_paths) + definition = ( + str(repo_identity), + resolved_sha, + declarations, + dependency_identity, + harness_identity, + ) existing = self._by_definition.get(definition) if existing is not None: if expected_dependency_binding is not None: @@ -238,9 +250,22 @@ class TaskAssetCache: repo_identity, os.O_RDONLY | os.O_DIRECTORY | getattr(os, "O_CLOEXEC", 0) | getattr(os, "O_NOFOLLOW", 0), ) + harness_fd: int | None = None try: for relative in relative_paths: - descriptor = _open_relative(repo_fd, relative) + if _is_harness_sandbox_copy(relative): + if harness_fd is None: + harness_root = _harness_sandbox_copy_root() + harness_fd = os.open( + harness_root, + os.O_RDONLY + | os.O_DIRECTORY + | getattr(os, "O_CLOEXEC", 0) + | getattr(os, "O_NOFOLLOW", 0), + ) + descriptor = _open_relative(harness_fd, relative) + else: + descriptor = _open_relative(repo_fd, relative) try: builder.copy_descriptor(descriptor, relative) finally: @@ -299,6 +324,8 @@ class TaskAssetCache: ) finally: os.close(repo_fd) + if harness_fd is not None: + os.close(harness_fd) entries = builder.finished_entries() manifest_digest = _manifest_digest(entries) @@ -317,6 +344,7 @@ class TaskAssetCache: manifest_digest=manifest_digest, dependency_content_digest=dependency_content_digest, dependency_manifest_digest=dependency_manifest_digest, + harness_identity=harness_identity, ) destination = self.root / digest if destination.exists(): @@ -537,6 +565,22 @@ class _SnapshotBuilder: return tuple(sorted(self.entries.values(), key=lambda entry: entry.path.as_posix())) +def _is_harness_sandbox_copy(relative: PurePosixPath) -> bool: + if relative.is_absolute() or not relative.parts or ".." in relative.parts: + return False + return any(relative == prefix or prefix in relative.parents for prefix in HARNESS_SANDBOX_COPY_PREFIXES) + + +def _harness_sandbox_copy_root() -> Path: + return _real_directory(runtime_mounts.HARNESS_ROOT, label="harness sandbox_copy root") + + +def _harness_sandbox_copy_identity(relative_paths: tuple[PurePosixPath, ...]) -> str: + if not any(_is_harness_sandbox_copy(relative) for relative in relative_paths): + return "" + return str(_harness_sandbox_copy_root()) + + def _sandbox_copy_declarations( task: Mapping[str, Any], ) -> tuple[tuple[str, ...], tuple[PurePosixPath, ...]]: @@ -969,15 +1013,17 @@ def _snapshot_digest( manifest_digest: str, dependency_content_digest: str, dependency_manifest_digest: str, + harness_identity: str = "", ) -> str: payload = { "declarations": declarations, "dependency_content_digest": dependency_content_digest, "dependency_manifest_digest": dependency_manifest_digest, + "harness_identity": harness_identity, "manifest_digest": manifest_digest, "repo_identity": str(repo_identity), "resolved_sha": resolved_sha, - "schema_version": 2, + "schema_version": 3, } return hashlib.sha256(json.dumps(payload, sort_keys=True, separators=(",", ":")).encode()).hexdigest() diff --git a/eval/workflow_bench/tasks.review.scenarios.yaml b/eval/workflow_bench/tasks.review.scenarios.yaml index 5c0de3cda..b508d9a72 100644 --- a/eval/workflow_bench/tasks.review.scenarios.yaml +++ b/eval/workflow_bench/tasks.review.scenarios.yaml @@ -7,7 +7,7 @@ tasks: repo: ~/GitNexus ref: ff86ccf1e79cd7e4175da437ae8aeaf67b64aaa1 sandbox_copy: [eval/workflow_bench/review_cases/pr-2718-defect.patch] - setup: git apply eval/workflow_bench/review_cases/pr-2718-defect.patch && rm -rf eval/workflow_bench + setup: git apply --exclude='eval/workflow_bench/*' eval/workflow_bench/review_cases/pr-2718-defect.patch && rm -rf eval/workflow_bench prompt: Review the historical snapshot of https://github.com/abhigyanpatwari/GitNexus/pull/2718. Report only actionable defects introduced by the local diff. verify: test -s review-output.json oracle: @@ -22,7 +22,7 @@ tasks: id: review-pr-2794-defect ref: 911151e2304f298a995fcc69c738ad2c6db9393a sandbox_copy: [eval/workflow_bench/review_cases/pr-2794-defect.patch] - setup: git apply eval/workflow_bench/review_cases/pr-2794-defect.patch && rm -rf eval/workflow_bench + setup: git apply --exclude='eval/workflow_bench/*' eval/workflow_bench/review_cases/pr-2794-defect.patch && rm -rf eval/workflow_bench prompt: Review the historical snapshot of https://github.com/abhigyanpatwari/GitNexus/pull/2794. Report only actionable defects introduced by the local diff. oracle: command: test -s review-output.json @@ -33,7 +33,7 @@ tasks: id: review-pr-2108-defect ref: 3a4247ec36b5ad86b1123d3bbce8183a643f7434 sandbox_copy: [eval/workflow_bench/review_cases/pr-2108-defect.patch] - setup: git apply eval/workflow_bench/review_cases/pr-2108-defect.patch && rm -rf eval/workflow_bench + setup: git apply --exclude='eval/workflow_bench/*' eval/workflow_bench/review_cases/pr-2108-defect.patch && rm -rf eval/workflow_bench prompt: Review the historical snapshot of https://github.com/abhigyanpatwari/GitNexus/pull/2108. Report only actionable defects introduced by the local diff. oracle: command: test -s review-output.json @@ -44,7 +44,7 @@ tasks: id: review-pr-2258-defect ref: 78b4077d8acc86f1b0c32e41012174d484e81f12 sandbox_copy: [eval/workflow_bench/review_cases/pr-2258-defect.patch] - setup: git apply eval/workflow_bench/review_cases/pr-2258-defect.patch && rm -rf eval/workflow_bench + setup: git apply --exclude='eval/workflow_bench/*' eval/workflow_bench/review_cases/pr-2258-defect.patch && rm -rf eval/workflow_bench prompt: Review the historical snapshot of https://github.com/abhigyanpatwari/GitNexus/pull/2258. Report only actionable defects introduced by the local diff. oracle: command: test -s review-output.json @@ -56,7 +56,7 @@ tasks: class: review-clean ref: 78b4077d8acc86f1b0c32e41012174d484e81f12 sandbox_copy: [eval/workflow_bench/review_cases/pr-2258-clean.patch] - setup: git apply eval/workflow_bench/review_cases/pr-2258-clean.patch && rm -rf eval/workflow_bench + setup: git apply --exclude='eval/workflow_bench/*' eval/workflow_bench/review_cases/pr-2258-clean.patch && rm -rf eval/workflow_bench prompt: Review this historical snapshot of https://github.com/abhigyanpatwari/GitNexus/pull/2258. Report only actionable defects introduced by the local diff. oracle: command: test -s review-output.json @@ -68,7 +68,7 @@ tasks: class: review-clean ref: 84f584449de02376a8ffc096dceac2e8f732cab5 sandbox_copy: [eval/workflow_bench/review_cases/pr-2773-clean.patch] - setup: git apply eval/workflow_bench/review_cases/pr-2773-clean.patch && rm -rf eval/workflow_bench + setup: git apply --exclude='eval/workflow_bench/*' eval/workflow_bench/review_cases/pr-2773-clean.patch && rm -rf eval/workflow_bench prompt: Review this historical snapshot of https://github.com/abhigyanpatwari/GitNexus/pull/2773. Report only actionable defects introduced by the local diff. oracle: command: test -s review-output.json