mirror of
https://github.com/abhigyanpatwari/GitNexus.git
synced 2026-10-06 02:49:56 +00:00
fix(eval): make historical review evolution score instead of aborting
Seed the current gitnexus-review skill into older PR checkouts, force-add historically gitignored skill paths, accept plugin-qualified Skill ids, and lock host-unsafe workspaces to review-output.json so a generation can finish and score. Sandbox cleanup restores owner write bits before delete because a session that copytrees the locked clone otherwise leaves 0555 trees that rmtree cannot remove. Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
parent
9047bf00a5
commit
8491cf4203
22 changed files with 1549 additions and 141 deletions
|
|
@ -17,6 +17,7 @@ from workflow_bench.runner_sessions import PARENT_EVENT_STREAM_SOURCE
|
|||
from workflow_bench.evolve import (
|
||||
build_parser,
|
||||
build_proposer_prompt,
|
||||
executed_benchmark_arms,
|
||||
generation_timeout_seconds,
|
||||
load_jsonl,
|
||||
proposer_evidence_entries,
|
||||
|
|
@ -871,6 +872,25 @@ def test_runner_argv_pairs_each_incumbent_with_its_candidate(tmp_path):
|
|||
assert json.loads(argv[argv.index("--promotion-target-bases-json") + 1]) == target_bases
|
||||
|
||||
|
||||
def test_runner_argv_inserts_ce_review_for_review_overlay(tmp_path):
|
||||
args = build_parser().parse_args(
|
||||
["--tasks", "t.yaml", "--model", "pinned", "--arms", "review"]
|
||||
)
|
||||
overlay = tmp_path / "overlay"
|
||||
skill = overlay / ".claude" / "skills" / "gitnexus-review" / "SKILL.md"
|
||||
skill.parent.mkdir(parents=True)
|
||||
skill.write_text("candidate")
|
||||
argv = runner_argv(
|
||||
args,
|
||||
tmp_path / "bench",
|
||||
overlay,
|
||||
task_bindings=[{"id": "task"}],
|
||||
target_base_digests={},
|
||||
)
|
||||
arms = argv[argv.index("--arms") + 1 : argv.index("--promotion-metric")]
|
||||
assert arms == ["ce_review", "review", "candidate_review"]
|
||||
|
||||
|
||||
def test_runner_argv_omits_proposer_for_manual_overlay(tmp_path):
|
||||
args = build_parser().parse_args(["--tasks", "t.yaml", "--model", "pinned"])
|
||||
overlay = tmp_path / "overlay"
|
||||
|
|
@ -890,6 +910,26 @@ def test_runner_argv_omits_proposer_for_manual_overlay(tmp_path):
|
|||
assert "--proposer-model" not in argv
|
||||
|
||||
|
||||
def test_runner_argv_forwards_explicit_unsafe_backend(tmp_path):
|
||||
args = build_parser().parse_args(
|
||||
["--tasks", "t.yaml", "--model", "pinned", "--unsafe-no-bwrap"]
|
||||
)
|
||||
overlay = tmp_path / "overlay"
|
||||
skill = overlay / ".claude" / "skills" / "gitnexus-plan" / "SKILL.md"
|
||||
skill.parent.mkdir(parents=True)
|
||||
skill.write_text("candidate")
|
||||
|
||||
argv = runner_argv(
|
||||
args,
|
||||
tmp_path / "bench",
|
||||
overlay,
|
||||
task_bindings=[{"id": "task"}],
|
||||
target_base_digests={},
|
||||
)
|
||||
|
||||
assert "--unsafe-no-bwrap" in argv
|
||||
|
||||
|
||||
def test_runner_argv_keeps_task_commit_pinned_when_ref_moves(tmp_path):
|
||||
repo = tmp_path / "task-repo"
|
||||
repo.mkdir()
|
||||
|
|
@ -992,6 +1032,59 @@ def test_generation_timeout_budgets_three_task_workflow_pair():
|
|||
assert timeout >= 3 * (1 + 3 * paired_arm_cells) * evolve.WORKTREE_PREPARATION_TIMEOUT_SECONDS
|
||||
|
||||
|
||||
def test_executed_benchmark_arms_inserts_review_comparator() -> None:
|
||||
assert executed_benchmark_arms(["workflow"]) == ["workflow", "candidate_workflow"]
|
||||
assert executed_benchmark_arms(["review"]) == ["ce_review", "review", "candidate_review"]
|
||||
|
||||
|
||||
def test_generation_timeout_budgets_review_pair_plus_ce_comparator() -> None:
|
||||
timeout = generation_timeout_seconds(
|
||||
task_count=6,
|
||||
runs=1,
|
||||
session_timeout=3600,
|
||||
incumbent_arms=["review"],
|
||||
)
|
||||
|
||||
per_task_preparation = (
|
||||
evolve.TASK_BINDING_GIT_PHASES * evolve.GIT_COMMAND_TIMEOUT_SECONDS
|
||||
+ 2 * evolve.TASK_SNAPSHOT_TIMEOUT_SECONDS
|
||||
+ evolve.WORKTREE_PREPARATION_TIMEOUT_SECONDS
|
||||
+ evolve.GRAPH_SOURCE_PREPARATION_TIMEOUT_SECONDS
|
||||
+ evolve.GRAPH_BUILD_TIMEOUT_SECONDS
|
||||
+ 2 * evolve.GRAPH_QUERY_TIMEOUT_SECONDS
|
||||
+ evolve.CLEANUP_TIMEOUT_SECONDS
|
||||
)
|
||||
paired_arm_cells = 3
|
||||
session_slots = 3
|
||||
workspace_snapshot_slots = 3
|
||||
per_task_run = session_slots * (3600 + evolve.SESSION_FINALIZATION_TIMEOUT_SECONDS) + paired_arm_cells * (
|
||||
evolve.WORKTREE_PREPARATION_TIMEOUT_SECONDS
|
||||
+ evolve.ARM_ASSET_MATERIALIZATION_PHASES * evolve.TASK_SNAPSHOT_TIMEOUT_SECONDS
|
||||
+ evolve.SETUP_TIMEOUT_SECONDS
|
||||
+ 2 * 3600
|
||||
+ evolve.ARM_EVIDENCE_GIT_PHASES * evolve.GIT_COMMAND_TIMEOUT_SECONDS
|
||||
+ evolve.CLEANUP_TIMEOUT_SECONDS
|
||||
)
|
||||
per_task_run += workspace_snapshot_slots * evolve.TASK_SNAPSHOT_TIMEOUT_SECONDS
|
||||
per_task_run += evolve.CANDIDATE_OVERLAY_GIT_PHASES * evolve.GIT_COMMAND_TIMEOUT_SECONDS
|
||||
|
||||
assert timeout == (
|
||||
evolve.PROMOTION_BASE_TIMEOUT_SECONDS
|
||||
+ 6 * (per_task_preparation + per_task_run)
|
||||
+ evolve.DRIVER_OVERHEAD_SECONDS
|
||||
)
|
||||
|
||||
|
||||
def test_generation_timeout_rejects_unknown_arm() -> None:
|
||||
with pytest.raises(ValueError, match="unsupported evolution arm: mystery"):
|
||||
generation_timeout_seconds(
|
||||
task_count=1,
|
||||
runs=1,
|
||||
session_timeout=60,
|
||||
incumbent_arms=["mystery"],
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.skipif(sys.platform != "linux", reason="Bubblewrap PID namespaces require Linux")
|
||||
def test_outer_runner_pid_namespace_kills_setsid_descendant(tmp_path):
|
||||
try:
|
||||
|
|
|
|||
|
|
@ -10,8 +10,11 @@ import pytest
|
|||
import yaml
|
||||
|
||||
from workflow_bench.model_gateway import (
|
||||
DEFAULT_GATEWAY_READY_TIMEOUT_S,
|
||||
GATEWAY_READY_TIMEOUT_ENV,
|
||||
GATEWAY_REQUEST_TIMEOUT_S,
|
||||
OpenAIGateway,
|
||||
gateway_ready_timeout_s,
|
||||
anthropic_api_key_from_environ,
|
||||
claude_gateway_model_env,
|
||||
credential_secrets,
|
||||
|
|
@ -141,6 +144,49 @@ def test_openai_gateway_never_leaves_proxy_output_on_an_undrained_pipe(tmp_path:
|
|||
assert gateway.log_path.stat().st_mode & 0o777 == 0o600
|
||||
|
||||
|
||||
def test_gateway_startup_budget_outlives_a_cold_litellm_import(monkeypatch, tmp_path: Path) -> None:
|
||||
# Importing LiteLLM takes ~17s on a cold container filesystem and the proxy
|
||||
# binds its port only afterwards, so a sub-20s budget fails as "connection
|
||||
# refused" on a proxy that was merely still starting.
|
||||
monkeypatch.delenv(GATEWAY_READY_TIMEOUT_ENV, raising=False)
|
||||
assert DEFAULT_GATEWAY_READY_TIMEOUT_S >= 60
|
||||
assert gateway_ready_timeout_s() == DEFAULT_GATEWAY_READY_TIMEOUT_S
|
||||
assert (
|
||||
OpenAIGateway(
|
||||
openai_api_key="sk-openai-secret",
|
||||
model_names=["gpt-4.1"],
|
||||
work_dir=tmp_path / "gw",
|
||||
).ready_timeout_s
|
||||
== DEFAULT_GATEWAY_READY_TIMEOUT_S
|
||||
)
|
||||
|
||||
monkeypatch.setenv(GATEWAY_READY_TIMEOUT_ENV, "42.5")
|
||||
assert gateway_ready_timeout_s() == 42.5
|
||||
|
||||
for bad in ("0", "-1", "soon"):
|
||||
monkeypatch.setenv(GATEWAY_READY_TIMEOUT_ENV, bad)
|
||||
with pytest.raises(ValueError, match=GATEWAY_READY_TIMEOUT_ENV):
|
||||
gateway_ready_timeout_s()
|
||||
|
||||
|
||||
def test_gateway_readiness_timeout_reports_the_proxy_log_and_the_override(tmp_path: Path) -> None:
|
||||
gateway = OpenAIGateway(
|
||||
openai_api_key="sk-openai-secret",
|
||||
model_names=["gpt-4.1"],
|
||||
work_dir=tmp_path / "gw",
|
||||
ready_timeout_s=0.1,
|
||||
)
|
||||
gateway.work_dir.mkdir(parents=True)
|
||||
gateway.log_path.write_text("ImportError: litellm proxy extras missing")
|
||||
|
||||
with pytest.raises(RuntimeError) as excinfo:
|
||||
gateway._wait_until_ready()
|
||||
|
||||
message = str(excinfo.value)
|
||||
assert "ImportError: litellm proxy extras missing" in message
|
||||
assert GATEWAY_READY_TIMEOUT_ENV in message
|
||||
|
||||
|
||||
def test_runner_environment_does_not_forward_the_openai_key() -> None:
|
||||
args = build_parser().parse_args(
|
||||
[
|
||||
|
|
|
|||
|
|
@ -5,6 +5,7 @@ from __future__ import annotations
|
|||
import argparse
|
||||
import hashlib
|
||||
import os
|
||||
import shutil
|
||||
import subprocess
|
||||
from pathlib import Path
|
||||
from types import SimpleNamespace
|
||||
|
|
@ -13,7 +14,12 @@ import pytest
|
|||
|
||||
from workflow_bench import oracle_assets, runner
|
||||
from workflow_bench.evolution import evaluate_candidate
|
||||
from workflow_bench.oracle_assets import capture_task_oracle, staged_task_oracle
|
||||
from workflow_bench.oracle_assets import (
|
||||
capture_task_oracle,
|
||||
review_case_setup_command,
|
||||
staged_task_oracle,
|
||||
with_hidden_harness_apply_exclude,
|
||||
)
|
||||
|
||||
|
||||
def oracle_task(*, command: str = "true", source: str = "oracle.test.ts") -> dict[str, object]:
|
||||
|
|
@ -149,6 +155,60 @@ def test_clone_sanitization_prunes_harness_checkout_and_recoverable_history(tmp_
|
|||
assert git(clone, "status", "--porcelain=v1", "--untracked-files=all").stdout == ""
|
||||
|
||||
|
||||
def test_hidden_harness_apply_exclude_is_idempotent() -> None:
|
||||
raw = "git apply eval/workflow_bench/review_cases/pr.patch && rm -rf eval/workflow_bench"
|
||||
once = with_hidden_harness_apply_exclude(raw)
|
||||
assert once == review_case_setup_command("pr.patch")
|
||||
assert with_hidden_harness_apply_exclude(once) == once
|
||||
assert with_hidden_harness_apply_exclude("true") == "true"
|
||||
|
||||
|
||||
def test_review_setup_skips_sanitized_harness_hunks(tmp_path: Path) -> None:
|
||||
repo = tmp_path / "repo"
|
||||
repo.mkdir()
|
||||
|
||||
def git(*args: str, check: bool = True) -> subprocess.CompletedProcess[str]:
|
||||
result = subprocess.run(
|
||||
["git", "-C", str(repo), *args],
|
||||
check=False,
|
||||
capture_output=True,
|
||||
text=True,
|
||||
)
|
||||
if check and result.returncode != 0:
|
||||
pytest.fail(f"git {' '.join(args)} failed: {result.stderr}")
|
||||
return result
|
||||
|
||||
git("init", "--quiet", "--initial-branch=main")
|
||||
git("config", "user.name", "Review Setup")
|
||||
git("config", "user.email", "review-setup.invalid")
|
||||
(repo / "visible.py").write_text("old\n")
|
||||
hidden = repo / "eval" / "workflow_bench"
|
||||
hidden.mkdir(parents=True)
|
||||
(hidden / "learnings.jsonl").write_text("{}\n")
|
||||
git("add", "--all")
|
||||
git("commit", "--quiet", "-m", "base with harness file")
|
||||
(repo / "visible.py").write_text("new\n")
|
||||
(hidden / "learnings.jsonl").write_text("{}\nextra\n")
|
||||
patch = git("diff").stdout
|
||||
git("checkout", "--", ".")
|
||||
shutil.rmtree(hidden)
|
||||
patch_path = hidden / "review_cases" / "case.patch"
|
||||
patch_path.parent.mkdir(parents=True)
|
||||
patch_path.write_text(patch)
|
||||
|
||||
rejected = git("apply", "--check", str(patch_path.relative_to(repo)), check=False)
|
||||
assert rejected.returncode != 0
|
||||
assert "learnings.jsonl" in rejected.stderr
|
||||
|
||||
setup = with_hidden_harness_apply_exclude(
|
||||
"git apply eval/workflow_bench/review_cases/case.patch && rm -rf eval/workflow_bench"
|
||||
)
|
||||
applied = subprocess.run(["/bin/sh", "-lc", setup], cwd=repo, check=False, capture_output=True, text=True)
|
||||
assert applied.returncode == 0, applied.stderr
|
||||
assert (repo / "visible.py").read_text() == "new\n"
|
||||
assert not hidden.exists()
|
||||
|
||||
|
||||
def test_clone_sanitization_prunes_remote_history_when_head_never_had_harness(tmp_path: Path) -> None:
|
||||
source = tmp_path / "source"
|
||||
source.mkdir()
|
||||
|
|
|
|||
|
|
@ -11,6 +11,7 @@ import sys
|
|||
import threading
|
||||
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
||||
from pathlib import Path
|
||||
from types import SimpleNamespace
|
||||
|
||||
import pytest
|
||||
|
||||
|
|
@ -20,6 +21,7 @@ from workflow_bench.process_control import ManagedProcessResult, run_managed
|
|||
from workflow_bench.proposer_sandbox import (
|
||||
MAX_BUNDLE_BYTES,
|
||||
MAX_EVIDENCE_FILE_BYTES,
|
||||
SANDBOX_CLAUDE,
|
||||
SANDBOX_NODE,
|
||||
SANDBOX_NODE_PREFIX,
|
||||
SANDBOX_EVIDENCE,
|
||||
|
|
@ -35,8 +37,11 @@ from workflow_bench.proposer_sandbox import (
|
|||
_runtime_mount_args,
|
||||
build_claude_settings,
|
||||
build_sandbox_environment,
|
||||
_force_rmtree,
|
||||
host_workspace_write_boundary,
|
||||
prepare_sandbox,
|
||||
preflight_bubblewrap,
|
||||
sandbox_workspace_write_boundary,
|
||||
stage_evidence_bundle,
|
||||
stage_task_assets,
|
||||
)
|
||||
|
|
@ -80,6 +85,128 @@ def test_environment_is_allowlisted_and_shell_children_are_credential_free(monke
|
|||
assert "defaultMode" not in settings["permissions"]
|
||||
|
||||
|
||||
def test_unsafe_host_session_translates_virtual_paths_and_disables_containment(tmp_path) -> None:
|
||||
clone = tmp_path / "clone"
|
||||
evidence = tmp_path / "evidence"
|
||||
for directory in (clone, evidence):
|
||||
directory.mkdir()
|
||||
|
||||
with prepare_sandbox(
|
||||
clone=clone,
|
||||
backend="host-unsafe",
|
||||
read_only_mounts=(ReadOnlyMount(evidence, "/evidence"),),
|
||||
) as sandbox:
|
||||
assert sandbox.command_prefix == []
|
||||
assert sandbox.require_pid_namespace is False
|
||||
assert sandbox.host_path("/workspace/review-output.json") == str(clone / "review-output.json")
|
||||
assert sandbox.host_path("/evidence/selected-rows.json") == str(evidence / "selected-rows.json")
|
||||
assert sandbox.host_text("read /evidence and write /workspace/out") == (
|
||||
f"read {evidence} and write {clone}/out"
|
||||
)
|
||||
# Sessions spawn the binary directly, so it must be the host executable
|
||||
# rather than the sandbox-only mount target.
|
||||
assert sandbox.claude_bin != SANDBOX_CLAUDE
|
||||
assert Path(sandbox.claude_bin).exists()
|
||||
assert sandbox.environment()["HOME"] == str(sandbox.home)
|
||||
assert "CLAUDE_CODE_SUBPROCESS_ENV_SCRUB" not in sandbox.environment()
|
||||
assert all(
|
||||
Path(entry).is_dir() for entry in sandbox.environment()["PATH"].split(":")
|
||||
)
|
||||
unsafe_settings = json.loads(sandbox.settings_json)
|
||||
assert unsafe_settings["sandbox"]["enabled"] is False
|
||||
assert unsafe_settings["sandbox"]["failIfUnavailable"] is False
|
||||
assert "disableBypassPermissionsMode" not in unsafe_settings["permissions"]
|
||||
|
||||
|
||||
def test_host_workspace_write_boundary_keeps_only_the_review_artifact_writable(tmp_path) -> None:
|
||||
clone = tmp_path / "clone"
|
||||
nested = clone / "src"
|
||||
nested.mkdir(parents=True)
|
||||
source = nested / "source.ts"
|
||||
source.write_text("trusted\n")
|
||||
output = clone / "review-output.json"
|
||||
output.write_text("")
|
||||
original_source_mode = stat.S_IMODE(source.stat().st_mode)
|
||||
original_output_mode = stat.S_IMODE(output.stat().st_mode)
|
||||
|
||||
with host_workspace_write_boundary(clone, writable=(output,)):
|
||||
with pytest.raises(OSError):
|
||||
source.write_text("tampered\n")
|
||||
with pytest.raises(OSError):
|
||||
(clone / "extra.py").write_text("nope\n")
|
||||
output.write_text('{"schema_version":1}\n')
|
||||
|
||||
assert source.read_text() == "trusted\n"
|
||||
assert output.read_text() == '{"schema_version":1}\n'
|
||||
assert not (clone / "extra.py").exists()
|
||||
assert stat.S_IMODE(source.stat().st_mode) == original_source_mode
|
||||
assert stat.S_IMODE(output.stat().st_mode) == original_output_mode
|
||||
|
||||
|
||||
def test_sandbox_workspace_write_boundary_is_noop_unless_host_unsafe(tmp_path) -> None:
|
||||
clone = tmp_path / "clone"
|
||||
clone.mkdir()
|
||||
source = clone / "source.ts"
|
||||
source.write_text("trusted\n")
|
||||
bwrap_sandbox = SimpleNamespace(backend="bwrap", clone=clone)
|
||||
with sandbox_workspace_write_boundary(
|
||||
bwrap_sandbox,
|
||||
read_only_workspace=True,
|
||||
writable=(),
|
||||
):
|
||||
source.write_text("still writable under bwrap no-op\n")
|
||||
assert source.read_text() == "still writable under bwrap no-op\n"
|
||||
|
||||
output = clone / "review-output.json"
|
||||
output.write_text("")
|
||||
source.write_text("trusted\n")
|
||||
unsafe = SimpleNamespace(backend="host-unsafe", clone=clone)
|
||||
with sandbox_workspace_write_boundary(
|
||||
unsafe,
|
||||
read_only_workspace=True,
|
||||
writable=(output,),
|
||||
):
|
||||
with pytest.raises(OSError):
|
||||
source.write_text("tampered\n")
|
||||
output.write_text("ok\n")
|
||||
assert source.read_text() == "trusted\n"
|
||||
assert output.read_text() == "ok\n"
|
||||
|
||||
|
||||
def test_force_rmtree_deletes_nonempty_directories_copied_from_a_locked_workspace(tmp_path) -> None:
|
||||
locked = tmp_path / "locked"
|
||||
nested = locked / "gitnexus-shared" / "src"
|
||||
nested.mkdir(parents=True)
|
||||
(nested / "index.ts").write_text("export {}\n")
|
||||
os.chmod(nested, 0o555)
|
||||
os.chmod(locked / "gitnexus-shared", 0o555)
|
||||
os.chmod(locked, 0o555)
|
||||
|
||||
copied = tmp_path / "sandbox-tmp" / "tmp.XXXX" / "gitnexus-shared"
|
||||
copied.parent.mkdir(parents=True)
|
||||
shutil.copytree(locked / "gitnexus-shared", copied)
|
||||
assert stat.S_IMODE(copied.stat().st_mode) & 0o222 == 0
|
||||
|
||||
_force_rmtree(copied.parent)
|
||||
assert not copied.parent.exists()
|
||||
|
||||
|
||||
def test_host_unsafe_sandbox_cleanup_survives_readonly_tmpdir_copies(tmp_path) -> None:
|
||||
clone = tmp_path / "clone"
|
||||
clone.mkdir()
|
||||
leftover = None
|
||||
with prepare_sandbox(clone=clone, backend="host-unsafe") as sandbox:
|
||||
leftover = sandbox.private_root
|
||||
copied = sandbox.temp / "tmp.XXXX" / "gitnexus-shared" / "src"
|
||||
copied.mkdir(parents=True)
|
||||
(copied / "index.ts").write_text("export {}\n")
|
||||
os.chmod(copied, 0o555)
|
||||
os.chmod(copied.parent, 0o555)
|
||||
os.chmod(copied.parent.parent, 0o555)
|
||||
assert leftover is not None
|
||||
assert not leftover.exists()
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"bad_url",
|
||||
["https://user:secret@example.test", "https://example.test/path?token=x", "file:///tmp/model"],
|
||||
|
|
|
|||
|
|
@ -4,6 +4,8 @@ from pathlib import Path
|
|||
|
||||
import yaml
|
||||
|
||||
from workflow_bench.oracle_assets import review_case_setup_command
|
||||
|
||||
|
||||
BENCH_ROOT = Path(__file__).parents[1] / "workflow_bench"
|
||||
|
||||
|
|
@ -26,6 +28,7 @@ def test_review_corpus_is_immutable_and_task_bound():
|
|||
task = by_id[case["id"]]
|
||||
assert task["ref"] == case["base_sha"]
|
||||
assert task["sandbox_copy"] == [f"eval/workflow_bench/review_cases/{patch.name}"]
|
||||
assert task["setup"] == review_case_setup_command(patch.name)
|
||||
|
||||
|
||||
def test_hidden_labels_are_not_recoverable_from_visible_task_input():
|
||||
|
|
|
|||
|
|
@ -27,6 +27,32 @@ def test_prebuilt_graph_and_harness_assets_are_rejected(task):
|
|||
sanitized_graph.validate_no_prebuilt_graph_assets(task)
|
||||
|
||||
|
||||
def test_review_case_patches_are_allowed_sandbox_copy():
|
||||
sanitized_graph.validate_no_prebuilt_graph_assets(
|
||||
{"sandbox_copy": ["eval/workflow_bench/review_cases/pr-2718-defect.patch"]}
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"task",
|
||||
[
|
||||
{"sandbox_copy": ["eval/workflow_bench"]},
|
||||
{"sandbox_copy": ["eval/workflow_bench/evolve.py"]},
|
||||
{
|
||||
"sandbox_dependencies": [
|
||||
{
|
||||
"source": "eval/workflow_bench/review_cases/pr-2718-defect.patch",
|
||||
"target": "patch",
|
||||
}
|
||||
]
|
||||
},
|
||||
],
|
||||
)
|
||||
def test_non_corpus_harness_paths_stay_rejected(task):
|
||||
with pytest.raises(SandboxError, match="prebuilt graph or harness"):
|
||||
sanitized_graph.validate_no_prebuilt_graph_assets(task)
|
||||
|
||||
|
||||
def test_graph_environment_is_offline_deterministic_and_ignores_target_gitignore():
|
||||
env = sanitized_graph._graph_environment()
|
||||
|
||||
|
|
@ -34,6 +60,10 @@ def test_graph_environment_is_offline_deterministic_and_ignores_target_gitignore
|
|||
assert env["GITNEXUS_NO_GITIGNORE"] == "1"
|
||||
assert env["GITNEXUS_WORKER_POOL_SIZE"] == "1"
|
||||
assert env["GITNEXUS_PARSE_CHUNK_CONCURRENCY"] == "1"
|
||||
assert env["GITNEXUS_WORKER_READY_TIMEOUT_MS"] == str(
|
||||
sanitized_graph.GRAPH_WORKER_READY_TIMEOUT_MS
|
||||
)
|
||||
assert int(env["GITNEXUS_WORKER_READY_TIMEOUT_MS"]) >= 60_000
|
||||
assert "ANTHROPIC_API_KEY" not in env
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -5,15 +5,15 @@ from __future__ import annotations
|
|||
import os
|
||||
import stat
|
||||
import subprocess
|
||||
from pathlib import Path
|
||||
from pathlib import Path, PurePosixPath
|
||||
|
||||
import pytest
|
||||
|
||||
from workflow_bench.proposer_sandbox import VITE_TEMP_DIR, SandboxError
|
||||
from workflow_bench.oracle_assets import TaskOracleSnapshot
|
||||
from workflow_bench.runner_tasks import resolve_task_bindings
|
||||
from workflow_bench.task_assets import TaskAssetCache, stage_task_assets
|
||||
from workflow_bench import task_assets
|
||||
from workflow_bench.task_assets import TaskAssetCache, _is_harness_sandbox_copy, stage_task_assets
|
||||
from workflow_bench import runtime_mounts, task_assets
|
||||
|
||||
|
||||
SHA = "a" * 40
|
||||
|
|
@ -443,3 +443,53 @@ def test_non_node_modules_dependency_snapshot_has_no_vite_temp(tmp_path: Path) -
|
|||
snapshot = cache.prepare(task, repo=repo, resolved_sha=SHA)
|
||||
captured = {entry.path.as_posix() for entry in snapshot.dependencies[0].entries}
|
||||
assert not any(path.endswith(VITE_TEMP_DIR) for path in captured)
|
||||
|
||||
|
||||
def test_review_case_sandbox_copy_is_read_from_the_harness_not_the_task_repo(
|
||||
monkeypatch, tmp_path: Path
|
||||
) -> None:
|
||||
repo = tmp_path / "task-repo"
|
||||
repo.mkdir()
|
||||
(repo / "eval" / "workflow_bench").mkdir(parents=True)
|
||||
harness = tmp_path / "harness"
|
||||
patch = harness / "eval" / "workflow_bench" / "review_cases" / "pr-2718-defect.patch"
|
||||
patch.parent.mkdir(parents=True)
|
||||
patch.write_bytes(b"diff --git a/a b/a\n")
|
||||
monkeypatch.setattr(runtime_mounts, "HARNESS_ROOT", harness)
|
||||
|
||||
task = {
|
||||
"sandbox_copy": ["eval/workflow_bench/review_cases/pr-2718-defect.patch"],
|
||||
"sandbox_dependencies": [],
|
||||
}
|
||||
with TaskAssetCache(tmp_path / "cache") as cache:
|
||||
snapshot = cache.prepare(task, repo=repo, resolved_sha=SHA)
|
||||
copied = snapshot.root / "sandbox-copy" / "eval" / "workflow_bench" / "review_cases" / "pr-2718-defect.patch"
|
||||
assert copied.read_bytes() == b"diff --git a/a b/a\n"
|
||||
|
||||
|
||||
def test_review_case_sandbox_copy_does_not_fall_back_to_the_task_repo(
|
||||
monkeypatch, tmp_path: Path
|
||||
) -> None:
|
||||
repo = tmp_path / "task-repo"
|
||||
planted = repo / "eval" / "workflow_bench" / "review_cases" / "pr-2718-defect.patch"
|
||||
planted.parent.mkdir(parents=True)
|
||||
planted.write_bytes(b"from-task-repo")
|
||||
harness = tmp_path / "harness"
|
||||
harness.mkdir()
|
||||
monkeypatch.setattr(runtime_mounts, "HARNESS_ROOT", harness)
|
||||
|
||||
task = {
|
||||
"sandbox_copy": ["eval/workflow_bench/review_cases/pr-2718-defect.patch"],
|
||||
"sandbox_dependencies": [],
|
||||
}
|
||||
with TaskAssetCache(tmp_path / "cache") as cache:
|
||||
with pytest.raises(SandboxError, match="unavailable"):
|
||||
cache.prepare(task, repo=repo, resolved_sha=SHA)
|
||||
|
||||
|
||||
def test_harness_sandbox_copy_does_not_treat_parent_escapes_as_corpus() -> None:
|
||||
assert _is_harness_sandbox_copy(PurePosixPath("eval/workflow_bench/review_cases/pr.patch"))
|
||||
assert not _is_harness_sandbox_copy(
|
||||
PurePosixPath("eval/workflow_bench/review_cases/../oracles/hidden.json")
|
||||
)
|
||||
assert not _is_harness_sandbox_copy(PurePosixPath("eval/workflow_bench/oracles"))
|
||||
|
|
|
|||
|
|
@ -15,6 +15,7 @@ from workflow_bench.evolution import (
|
|||
evaluate_candidate,
|
||||
evaluate_review_candidate,
|
||||
required_candidate_arms,
|
||||
seed_evaluated_skills,
|
||||
skill_fingerprint,
|
||||
unexercised_overlay_skills,
|
||||
)
|
||||
|
|
@ -387,6 +388,12 @@ def test_apply_candidate_overlay_creates_a_clean_ephemeral_commit(tmp_path):
|
|||
) == candidate_overlay_digest(overlay)
|
||||
assert incumbent.read_text() == "candidate\n"
|
||||
git_commands = [command for command in sandbox.commands if command[0] == "/usr/bin/git"]
|
||||
assert git_commands[0][-4:] == [
|
||||
"add",
|
||||
"-f",
|
||||
"--",
|
||||
".claude/skills/gitnexus-work/SKILL.md",
|
||||
]
|
||||
assert [command[-1] for command in git_commands[:2]] == [
|
||||
".claude/skills/gitnexus-work/SKILL.md",
|
||||
"--",
|
||||
|
|
@ -422,6 +429,181 @@ def test_candidate_overlay_rejects_linked_destination_parents(tmp_path):
|
|||
apply_candidate_overlay(overlay, repo, sandbox=sandbox)
|
||||
|
||||
|
||||
@pytest.mark.skipif(os.name == "nt", reason="candidate overlays require the Linux outer sandbox")
|
||||
def test_apply_candidate_overlay_force_adds_historically_ignored_skill(tmp_path):
|
||||
repo = tmp_path / "repo"
|
||||
repo.mkdir()
|
||||
subprocess.run(["git", "init", "--quiet", str(repo)], check=True)
|
||||
(repo / ".gitignore").write_text(".claude/skills/*\n")
|
||||
(repo / "README").write_text("subject\n")
|
||||
subprocess.run(["git", "-C", str(repo), "add", "."], check=True)
|
||||
subprocess.run(
|
||||
[
|
||||
"git",
|
||||
"-C",
|
||||
str(repo),
|
||||
"-c",
|
||||
"user.name=test",
|
||||
"-c",
|
||||
"user.email=test@invalid",
|
||||
"commit",
|
||||
"--quiet",
|
||||
"-m",
|
||||
"historical checkout that ignores skills",
|
||||
],
|
||||
check=True,
|
||||
)
|
||||
|
||||
overlay = tmp_path / "candidate"
|
||||
write_overlay_skill(overlay, "gitnexus-review")
|
||||
|
||||
class LocalSandbox:
|
||||
def __init__(self):
|
||||
self.clone = repo
|
||||
|
||||
def run(self, command, **kwargs):
|
||||
if command[0] == "/bin/mkdir":
|
||||
return ManagedProcessResult(
|
||||
state="exited",
|
||||
returncode=0,
|
||||
stdout_tail="",
|
||||
stderr_tail="",
|
||||
duration_s=0.0,
|
||||
)
|
||||
translated = [str(repo) if item == "/workspace" else item for item in command]
|
||||
completed = subprocess.run(
|
||||
translated,
|
||||
cwd=repo,
|
||||
env=dict(kwargs["env"]),
|
||||
capture_output=True,
|
||||
text=True,
|
||||
check=False,
|
||||
)
|
||||
return ManagedProcessResult(
|
||||
state="exited",
|
||||
returncode=completed.returncode,
|
||||
stdout_tail=completed.stdout,
|
||||
stderr_tail=completed.stderr,
|
||||
duration_s=0.0,
|
||||
)
|
||||
|
||||
apply_candidate_overlay(overlay, repo, sandbox=LocalSandbox())
|
||||
assert (repo / ".claude" / "skills" / "gitnexus-review" / "SKILL.md").read_text() == (
|
||||
"gitnexus-review candidate\n"
|
||||
)
|
||||
status = subprocess.run(
|
||||
["git", "-C", str(repo), "status", "--porcelain"],
|
||||
check=True,
|
||||
capture_output=True,
|
||||
text=True,
|
||||
)
|
||||
assert status.stdout == ""
|
||||
|
||||
|
||||
@pytest.mark.skipif(os.name == "nt", reason="skill seeds require the Linux outer sandbox")
|
||||
def test_seed_evaluated_skills_installs_missing_review_skill_and_is_idempotent(tmp_path):
|
||||
repo = tmp_path / "clone"
|
||||
repo.mkdir()
|
||||
subprocess.run(["git", "init", "--quiet", str(repo)], check=True)
|
||||
# Historical review SHAs ignore the whole skill tree and lack today's
|
||||
# `!.claude/skills/gitnexus-review/` allowlist. Seeding must still commit.
|
||||
(repo / ".gitignore").write_text(".claude/skills/*\n")
|
||||
(repo / "README").write_text("subject\n")
|
||||
subprocess.run(["git", "-C", str(repo), "add", "."], check=True)
|
||||
subprocess.run(
|
||||
[
|
||||
"git",
|
||||
"-C",
|
||||
str(repo),
|
||||
"-c",
|
||||
"user.name=test",
|
||||
"-c",
|
||||
"user.email=test@invalid",
|
||||
"commit",
|
||||
"--quiet",
|
||||
"-m",
|
||||
"historical checkout without review skill",
|
||||
],
|
||||
check=True,
|
||||
)
|
||||
before = subprocess.run(
|
||||
["git", "-C", str(repo), "rev-parse", "HEAD"],
|
||||
check=True,
|
||||
capture_output=True,
|
||||
text=True,
|
||||
).stdout.strip()
|
||||
|
||||
source = tmp_path / "harness"
|
||||
skill = source / ".claude" / "skills" / "gitnexus-review" / "SKILL.md"
|
||||
persona = source / ".claude" / "skills" / "gitnexus-review" / "ci-personas" / "lens.md"
|
||||
persona.parent.mkdir(parents=True)
|
||||
skill.write_text("current incumbent review skill\n")
|
||||
persona.write_text("persona\n")
|
||||
|
||||
class LocalSandbox:
|
||||
def __init__(self):
|
||||
self.clone = repo
|
||||
|
||||
def run(self, command, **kwargs):
|
||||
if command[0] == "/bin/mkdir":
|
||||
return ManagedProcessResult(
|
||||
state="exited",
|
||||
returncode=0,
|
||||
stdout_tail="",
|
||||
stderr_tail="",
|
||||
duration_s=0.0,
|
||||
)
|
||||
translated = [str(repo) if item == "/workspace" else item for item in command]
|
||||
completed = subprocess.run(
|
||||
translated,
|
||||
cwd=repo,
|
||||
env=dict(kwargs["env"]),
|
||||
capture_output=True,
|
||||
text=True,
|
||||
check=False,
|
||||
)
|
||||
return ManagedProcessResult(
|
||||
state="exited",
|
||||
returncode=completed.returncode,
|
||||
stdout_tail=completed.stdout,
|
||||
stderr_tail=completed.stderr,
|
||||
duration_s=0.0,
|
||||
)
|
||||
|
||||
sandbox = LocalSandbox()
|
||||
seed_evaluated_skills(source, repo, sandbox=sandbox, arm="review")
|
||||
assert skill_fingerprint(repo, "review") is not None
|
||||
assert (repo / ".claude" / "skills" / "gitnexus-review" / "SKILL.md").read_text() == (
|
||||
"current incumbent review skill\n"
|
||||
)
|
||||
assert (
|
||||
repo / ".claude" / "skills" / "gitnexus-review" / "ci-personas" / "lens.md"
|
||||
).read_text() == "persona\n"
|
||||
status = subprocess.run(
|
||||
["git", "-C", str(repo), "status", "--porcelain"],
|
||||
check=True,
|
||||
capture_output=True,
|
||||
text=True,
|
||||
)
|
||||
assert status.stdout == ""
|
||||
after = subprocess.run(
|
||||
["git", "-C", str(repo), "rev-parse", "HEAD"],
|
||||
check=True,
|
||||
capture_output=True,
|
||||
text=True,
|
||||
).stdout.strip()
|
||||
assert after != before
|
||||
|
||||
seed_evaluated_skills(source, repo, sandbox=sandbox, arm="review")
|
||||
again = subprocess.run(
|
||||
["git", "-C", str(repo), "rev-parse", "HEAD"],
|
||||
check=True,
|
||||
capture_output=True,
|
||||
text=True,
|
||||
).stdout.strip()
|
||||
assert again == after
|
||||
|
||||
|
||||
@pytest.mark.skipif(os.name == "nt", reason="skill links are rejected by the Linux sandbox harness")
|
||||
def test_skill_fingerprint_rejects_linked_skill_roots(tmp_path):
|
||||
outside = tmp_path / "outside"
|
||||
|
|
|
|||
|
|
@ -4,6 +4,7 @@ import argparse
|
|||
import hashlib
|
||||
import json
|
||||
import os
|
||||
import shutil
|
||||
import subprocess
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
|
@ -14,6 +15,7 @@ import pytest
|
|||
from workflow_bench import evolve, runner, runner_sessions, runtime_mounts
|
||||
from workflow_bench.evolution import skill_fingerprint
|
||||
from workflow_bench.process_control import ManagedProcessResult
|
||||
from workflow_bench.proposer_sandbox import SandboxError
|
||||
from workflow_bench.runner import snapshot_plan_docs
|
||||
|
||||
|
||||
|
|
@ -315,10 +317,18 @@ def test_run_arm_keeps_session_error_kind_over_verify(monkeypatch, tmp_path):
|
|||
|
||||
def test_agent_tool_grants_are_exact_and_nomcp_has_no_graph_tools(monkeypatch, tmp_path):
|
||||
read_only = runner.allowed_agent_tools(implementation=False)
|
||||
review_tools = runner.allowed_agent_tools(implementation=False, allow_edit=False)
|
||||
implementation = runner.allowed_agent_tools(implementation=True)
|
||||
no_mcp = runner.allowed_agent_tools(implementation=True, include_mcp=False)
|
||||
|
||||
assert read_only == [*runner.BUILTIN_AGENT_TOOLS, *runner.GITNEXUS_READ_ONLY_TOOLS]
|
||||
assert review_tools == [
|
||||
tool
|
||||
for tool in [*runner.BUILTIN_AGENT_TOOLS, *runner.GITNEXUS_READ_ONLY_TOOLS]
|
||||
if tool != "Edit"
|
||||
]
|
||||
assert "Write" in review_tools
|
||||
assert "Edit" not in review_tools
|
||||
assert implementation == [
|
||||
*runner.BUILTIN_AGENT_TOOLS,
|
||||
*runner.GITNEXUS_READ_ONLY_TOOLS,
|
||||
|
|
@ -346,7 +356,7 @@ def test_agent_tool_grants_are_exact_and_nomcp_has_no_graph_tools(monkeypatch, t
|
|||
)
|
||||
|
||||
assert captured[0]["allowed_tools"] == read_only # planning
|
||||
assert captured[1]["allowed_tools"] == read_only # review
|
||||
assert captured[1]["allowed_tools"] == review_tools # review
|
||||
assert captured[2]["allowed_tools"] == implementation
|
||||
assert captured[3]["allowed_tools"] == list(runner.BUILTIN_AGENT_TOOLS)
|
||||
assert captured[3]["mcp_config_json"] == '{"mcpServers":{}}'
|
||||
|
|
@ -416,6 +426,93 @@ def test_mcp_config_uses_only_the_minimal_pinned_harness_runtime(monkeypatch, tm
|
|||
assert f"{runner.SANDBOX_GITNEXUS}/hooks" not in mounted_targets
|
||||
|
||||
|
||||
def _install_pinned_runtime(root: Path) -> None:
|
||||
runtime = root / "gitnexus"
|
||||
shared = root / "gitnexus-shared"
|
||||
for directory in (
|
||||
runtime / "dist" / "cli",
|
||||
runtime / "node_modules",
|
||||
runtime / "vendor",
|
||||
runtime / "hooks" / "claude",
|
||||
shared / "dist",
|
||||
):
|
||||
directory.mkdir(parents=True)
|
||||
(runtime / "dist" / "cli" / "index.js").write_text("")
|
||||
(runtime / "hooks" / "claude" / "resolve-analyze-cmd.cjs").write_text("")
|
||||
(runtime / "package.json").write_text(json.dumps({"version": "9.9.9-test"}))
|
||||
(runtime / "node_modules" / "gitnexus-shared").symlink_to(shared, target_is_directory=True)
|
||||
(shared / "package.json").write_text(json.dumps({"name": "gitnexus-shared"}))
|
||||
|
||||
|
||||
def test_runtime_mounts_reuse_primary_checkout_node_modules_from_a_worktree(
|
||||
monkeypatch, tmp_path
|
||||
) -> None:
|
||||
primary = tmp_path / "primary"
|
||||
worktree = tmp_path / "worktree"
|
||||
_install_pinned_runtime(primary)
|
||||
(primary / ".git" / "worktrees" / "wt").mkdir(parents=True)
|
||||
_install_pinned_runtime(worktree)
|
||||
shutil.rmtree(worktree / "gitnexus" / "node_modules")
|
||||
(worktree / "gitnexus" / "node_modules").symlink_to(
|
||||
primary / "gitnexus" / "node_modules",
|
||||
target_is_directory=True,
|
||||
)
|
||||
(worktree / ".git").write_text(f"gitdir: {primary / '.git' / 'worktrees' / 'wt'}\n")
|
||||
monkeypatch.setattr(runtime_mounts, "HARNESS_ROOT", worktree)
|
||||
|
||||
mounts = runner.trusted_gitnexus_runtime_mounts()
|
||||
by_target = {mount.target: mount.source for mount in mounts}
|
||||
|
||||
assert by_target[f"{runner.SANDBOX_GITNEXUS}/node_modules"] == (
|
||||
primary / "gitnexus" / "node_modules"
|
||||
)
|
||||
assert by_target[f"{runner.SANDBOX_GITNEXUS_SHARED}/package.json"] == (
|
||||
worktree / "gitnexus-shared" / "package.json"
|
||||
)
|
||||
assert by_target[f"{runner.SANDBOX_GITNEXUS}/dist"] == worktree / "gitnexus" / "dist"
|
||||
|
||||
|
||||
def test_runtime_mounts_reject_a_node_modules_symlink_outside_the_primary_checkout(
|
||||
monkeypatch, tmp_path
|
||||
) -> None:
|
||||
primary = tmp_path / "primary"
|
||||
worktree = tmp_path / "worktree"
|
||||
outsider = tmp_path / "outsider"
|
||||
_install_pinned_runtime(primary)
|
||||
_install_pinned_runtime(outsider)
|
||||
(primary / ".git" / "worktrees" / "wt").mkdir(parents=True)
|
||||
_install_pinned_runtime(worktree)
|
||||
shutil.rmtree(worktree / "gitnexus" / "node_modules")
|
||||
(worktree / "gitnexus" / "node_modules").symlink_to(
|
||||
outsider / "gitnexus" / "node_modules",
|
||||
target_is_directory=True,
|
||||
)
|
||||
(worktree / ".git").write_text(f"gitdir: {primary / '.git' / 'worktrees' / 'wt'}\n")
|
||||
monkeypatch.setattr(runtime_mounts, "HARNESS_ROOT", worktree)
|
||||
|
||||
with pytest.raises(SandboxError, match="primary checkout"):
|
||||
runner.trusted_gitnexus_runtime_mounts()
|
||||
|
||||
|
||||
def test_runtime_mounts_reject_a_node_modules_symlink_in_a_regular_checkout(
|
||||
monkeypatch, tmp_path
|
||||
) -> None:
|
||||
checkout = tmp_path / "checkout"
|
||||
other = tmp_path / "other"
|
||||
_install_pinned_runtime(checkout)
|
||||
_install_pinned_runtime(other)
|
||||
(checkout / ".git").mkdir()
|
||||
shutil.rmtree(checkout / "gitnexus" / "node_modules")
|
||||
(checkout / "gitnexus" / "node_modules").symlink_to(
|
||||
other / "gitnexus" / "node_modules",
|
||||
target_is_directory=True,
|
||||
)
|
||||
monkeypatch.setattr(runtime_mounts, "HARNESS_ROOT", checkout)
|
||||
|
||||
with pytest.raises(SandboxError, match="must be a real directory"):
|
||||
runner.trusted_gitnexus_runtime_mounts()
|
||||
|
||||
|
||||
@pytest.mark.skipif(
|
||||
os.environ.get("GITNEXUS_REQUIRE_BWRAP_CANARY") != "1",
|
||||
reason="real Bubblewrap canary is mandatory in the named Ubuntu CI job",
|
||||
|
|
@ -662,6 +759,23 @@ def test_skill_invocation_parses_supported_exact_identifier_fields(skill_input):
|
|||
)
|
||||
|
||||
|
||||
def test_skill_invocation_accepts_plugin_qualified_identifier():
|
||||
assert (
|
||||
runner_sessions.skill_was_invoked_events(
|
||||
skill_events({"skill": "compound-engineering:ce-code-review"}),
|
||||
"ce-code-review",
|
||||
)
|
||||
is True
|
||||
)
|
||||
assert (
|
||||
runner_sessions.skill_was_invoked_events(
|
||||
skill_events({"skill": "compound-engineering:ce-plan"}),
|
||||
"ce-code-review",
|
||||
)
|
||||
is False
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"skill_input",
|
||||
[
|
||||
|
|
|
|||
|
|
@ -139,6 +139,28 @@ Native benchmark execution is therefore Linux/WSL2-only. Evidence assembly
|
|||
and hand-authored overlay preparation can happen elsewhere, but
|
||||
`--initial-overlay` does not bypass containment.
|
||||
|
||||
For a local diagnostic inside a container that blocks user namespaces, an
|
||||
operator may explicitly choose the non-containment host backend:
|
||||
|
||||
```bash
|
||||
UNSAFE_NO_BWRAP=1 RUNS=1 ./workflow_bench/run-evolution.sh
|
||||
```
|
||||
|
||||
This mode runs review sessions directly in disposable host worktrees and is
|
||||
**not** a security boundary: it does not isolate the network or create a PID
|
||||
namespace, and a session that can `chmod` can undo the workspace lock. The
|
||||
harness still drops write bits on the clone except `review-output.json` so
|
||||
accidental `npm install` / analyze writes cannot invalidate review evidence.
|
||||
Sandbox cleanup restores owner write bits before deleting the private TMPDIR,
|
||||
because a session that `copytree`s the locked clone would otherwise leave
|
||||
non-empty 0555 directories that `rmtree` cannot remove. Historical review
|
||||
SHAs that gitignore `.claude/skills/*` are force-added when the harness seeds
|
||||
or overlays the evaluated `gitnexus-review` skill.
|
||||
Treat model and verifier processes as able to access host files and
|
||||
credentials available to the invoking user. It is restricted to the review
|
||||
benchmark, forbidden with `--apply` and whenever `CI` is set;
|
||||
promotion-capable and CI runs must use Bubblewrap.
|
||||
|
||||
## Prompt and skill evolution loop
|
||||
|
||||
Prompts age as models and tool harnesses change. Treat the current skills and
|
||||
|
|
|
|||
|
|
@ -7,6 +7,7 @@ import os
|
|||
import secrets
|
||||
import stat
|
||||
import statistics
|
||||
from collections.abc import Sequence
|
||||
from pathlib import Path, PurePosixPath
|
||||
from typing import Any
|
||||
|
||||
|
|
@ -383,6 +384,69 @@ def candidate_overlay_digest(overlay: Path) -> str:
|
|||
return digest
|
||||
|
||||
|
||||
def _commit_sandbox_paths(
|
||||
sandbox: SandboxSession,
|
||||
relative_paths: Sequence[str],
|
||||
*,
|
||||
message: str,
|
||||
require_change: bool,
|
||||
) -> bool:
|
||||
"""Stage and commit paths inside the outer sandbox.
|
||||
|
||||
Returns True when a commit was created. ``require_change`` keeps the
|
||||
candidate-overlay contract: a no-op overlay is an error, while an
|
||||
incumbent skill seed may already match the historical tree.
|
||||
"""
|
||||
|
||||
mkdir_command = ["/bin/mkdir", "-p", f"{SANDBOX_TMP}/wfbench-empty-hooks"]
|
||||
mkdir_result = sandbox.run(
|
||||
mkdir_command,
|
||||
timeout=60,
|
||||
env=build_sandbox_environment(),
|
||||
)
|
||||
if not mkdir_result.ok:
|
||||
raise ManagedProcessError(mkdir_command, mkdir_result)
|
||||
if not relative_paths:
|
||||
if require_change:
|
||||
raise ValueError("candidate overlay is byte-identical to the incumbent skills")
|
||||
return False
|
||||
|
||||
# Historical review SHAs gitignore `.claude/skills/*` and lack the current
|
||||
# per-skill allowlist. Force-add so a seed or overlay of harness-owned
|
||||
# skill bytes is not rejected as an ignored path.
|
||||
command, added = _sandbox_overlay_git(sandbox, ["add", "-f", "--", *relative_paths])
|
||||
if not added.ok:
|
||||
raise ManagedProcessError(command, added)
|
||||
command, changed = _sandbox_overlay_git(
|
||||
sandbox,
|
||||
["diff", "--cached", "--quiet", "--no-ext-diff", "--no-textconv", "--"],
|
||||
)
|
||||
if changed.returncode == 0:
|
||||
if require_change:
|
||||
raise ValueError("candidate overlay is byte-identical to the incumbent skills")
|
||||
return False
|
||||
if changed.returncode != 1:
|
||||
raise ManagedProcessError(command, changed)
|
||||
|
||||
command, committed = _sandbox_overlay_git(
|
||||
sandbox,
|
||||
[
|
||||
"commit",
|
||||
"--quiet",
|
||||
"--no-verify",
|
||||
"-m",
|
||||
message,
|
||||
],
|
||||
extra_config=(
|
||||
"user.name=workflow-bench",
|
||||
"user.email=workflow-bench@invalid",
|
||||
),
|
||||
)
|
||||
if not committed.ok:
|
||||
raise ManagedProcessError(command, committed)
|
||||
return True
|
||||
|
||||
|
||||
def apply_candidate_overlay(
|
||||
overlay: Path,
|
||||
worktree: Path,
|
||||
|
|
@ -401,47 +465,96 @@ def apply_candidate_overlay(
|
|||
for relative, content in payload:
|
||||
_replace_regular_file(worktree, relative, content)
|
||||
relative_paths.append(relative.as_posix())
|
||||
|
||||
mkdir_command = ["/bin/mkdir", "-p", f"{SANDBOX_TMP}/wfbench-empty-hooks"]
|
||||
mkdir_result = sandbox.run(
|
||||
mkdir_command,
|
||||
timeout=60,
|
||||
env=build_sandbox_environment(),
|
||||
)
|
||||
if not mkdir_result.ok:
|
||||
raise ManagedProcessError(mkdir_command, mkdir_result)
|
||||
|
||||
command, added = _sandbox_overlay_git(sandbox, ["add", "--", *relative_paths])
|
||||
if not added.ok:
|
||||
raise ManagedProcessError(command, added)
|
||||
command, changed = _sandbox_overlay_git(
|
||||
_commit_sandbox_paths(
|
||||
sandbox,
|
||||
["diff", "--cached", "--quiet", "--no-ext-diff", "--no-textconv", "--"],
|
||||
relative_paths,
|
||||
message="benchmark candidate skill overlay",
|
||||
require_change=True,
|
||||
)
|
||||
if changed.returncode == 0:
|
||||
raise ValueError("candidate overlay is byte-identical to the incumbent skills")
|
||||
if changed.returncode != 1:
|
||||
raise ManagedProcessError(command, changed)
|
||||
|
||||
command, committed = _sandbox_overlay_git(
|
||||
sandbox,
|
||||
[
|
||||
"commit",
|
||||
"--quiet",
|
||||
"--no-verify",
|
||||
"-m",
|
||||
"benchmark candidate skill overlay",
|
||||
],
|
||||
extra_config=(
|
||||
"user.name=workflow-bench",
|
||||
"user.email=workflow-bench@invalid",
|
||||
),
|
||||
)
|
||||
if not committed.ok:
|
||||
raise ManagedProcessError(command, committed)
|
||||
return digest
|
||||
|
||||
|
||||
def seed_evaluated_skills(
|
||||
source_repo: Path,
|
||||
worktree: Path,
|
||||
*,
|
||||
sandbox: SandboxSession,
|
||||
arm: str,
|
||||
) -> None:
|
||||
"""Install the current evaluated skill tree into a historical clone.
|
||||
|
||||
Review evolution scores the current (or overlay) ``gitnexus-review`` skill
|
||||
against a historical PR checkout. Older SHAs predate that skill, and
|
||||
using whatever prose happened to exist at the reviewed commit would make
|
||||
the incumbent arm a moving target. Copy the harness checkout's skill
|
||||
bytes and commit them before setup so ``git status`` still shows only
|
||||
the task patch.
|
||||
"""
|
||||
|
||||
skill_names = EVALUATED_ARM_SKILLS.get(arm)
|
||||
if not skill_names:
|
||||
return
|
||||
|
||||
source_repo = source_repo.expanduser().absolute()
|
||||
expected_clone = Path(os.path.abspath(worktree.expanduser()))
|
||||
sandbox_clone = Path(os.path.abspath(sandbox.clone.expanduser()))
|
||||
if sandbox_clone != expected_clone:
|
||||
raise ValueError("skill seed sandbox does not bind the requested clone")
|
||||
_require_real_directory(source_repo, label="incumbent skill repository")
|
||||
if source_repo.resolve(strict=True) != source_repo:
|
||||
raise ValueError(f"incumbent skill repository cannot traverse symlinks: {source_repo}")
|
||||
|
||||
relative_paths: list[str] = []
|
||||
total = 0
|
||||
for skill_name in skill_names:
|
||||
_require_directory_chain(
|
||||
source_repo,
|
||||
Path(".claude") / "skills" / skill_name,
|
||||
label="incumbent skill root",
|
||||
)
|
||||
skill_root = source_repo / ".claude" / "skills" / skill_name
|
||||
pending = [skill_root]
|
||||
while pending:
|
||||
directory = pending.pop()
|
||||
try:
|
||||
children = list(os.scandir(directory))
|
||||
except OSError as exc:
|
||||
raise ValueError(f"incumbent skill directory is unreadable: {directory}: {exc}") from exc
|
||||
for item in children:
|
||||
path = Path(item.path)
|
||||
if item.is_symlink():
|
||||
raise ValueError(
|
||||
"incumbent skill seed cannot contain symlinks: "
|
||||
f"{path.relative_to(source_repo)}"
|
||||
)
|
||||
if item.is_dir(follow_symlinks=False):
|
||||
pending.append(path)
|
||||
continue
|
||||
if not item.is_file(follow_symlinks=False):
|
||||
raise ValueError(
|
||||
"incumbent skill seed entries must be regular files: "
|
||||
f"{path.relative_to(source_repo)}"
|
||||
)
|
||||
total += item.stat(follow_symlinks=False).st_size
|
||||
if total > MAX_SKILL_FINGERPRINT_BYTES:
|
||||
raise ValueError("incumbent skill seed exceeds the bounded evidence limit")
|
||||
relative = Path(".claude") / "skills" / skill_name / path.relative_to(skill_root)
|
||||
content = _bounded_regular_bytes(
|
||||
path,
|
||||
limit=MAX_SKILL_FINGERPRINT_BYTES,
|
||||
label="incumbent skill file",
|
||||
)
|
||||
_replace_regular_file(worktree, relative, content)
|
||||
relative_paths.append(PurePosixPath(relative.as_posix()).as_posix())
|
||||
|
||||
_commit_sandbox_paths(
|
||||
sandbox,
|
||||
relative_paths,
|
||||
message="benchmark incumbent review skill",
|
||||
require_change=False,
|
||||
)
|
||||
|
||||
|
||||
def unexercised_overlay_skills(overlay: Path, candidate_arms: list[str]) -> list[str]:
|
||||
"""Overlay skills that no selected candidate arm would ever load.
|
||||
|
||||
|
|
|
|||
|
|
@ -82,6 +82,7 @@ from .proposer_sandbox import (
|
|||
SandboxError,
|
||||
build_sandbox_environment,
|
||||
preflight_bubblewrap,
|
||||
preflight_unsafe_host,
|
||||
pid_namespace_command,
|
||||
prepare_sandbox,
|
||||
redact_text,
|
||||
|
|
@ -127,8 +128,8 @@ WORKTREE_PREPARATION_TIMEOUT_SECONDS = (
|
|||
# from that commit. Use the overlay boundary rather than the current candidate
|
||||
# size so this helper remains conservative before the runner starts.
|
||||
PROMOTION_BASE_TIMEOUT_SECONDS = (1 + 3 * MAX_CANDIDATE_FILES) * GIT_COMMAND_TIMEOUT_SECONDS
|
||||
ARM_SESSION_COUNTS = {"workflow": 2, "workflow_direct": 1}
|
||||
ARM_WORKSPACE_SNAPSHOT_COUNTS = {"workflow": 2, "workflow_direct": 0}
|
||||
ARM_SESSION_COUNTS = {"workflow": 2, "workflow_direct": 1, "review": 1}
|
||||
ARM_WORKSPACE_SNAPSHOT_COUNTS = {"workflow": 2, "workflow_direct": 0, "review": 1}
|
||||
REPO_ROOT = Path(__file__).resolve().parents[2]
|
||||
|
||||
|
||||
|
|
@ -692,6 +693,7 @@ def run_proposer(
|
|||
proposal_path: Path,
|
||||
evidence_bundle: Path,
|
||||
bwrap_bin: Path,
|
||||
sandbox_backend: str = "bwrap",
|
||||
progress_label: str | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Run one proposer in confinement and copy only validated outputs out."""
|
||||
|
|
@ -722,9 +724,13 @@ def run_proposer(
|
|||
bwrap_bin=bwrap_bin,
|
||||
read_only_mounts=[evidence_mount],
|
||||
preflight=False,
|
||||
backend=sandbox_backend,
|
||||
) as sandbox:
|
||||
host_text = getattr(sandbox, "host_text", lambda value: value)
|
||||
environment_builder = getattr(sandbox, "environment", build_sandbox_environment)
|
||||
backend = getattr(sandbox, "backend", "bwrap")
|
||||
record = runner.run_claude(
|
||||
prompt,
|
||||
host_text(prompt),
|
||||
clone,
|
||||
claude_bin=sandbox.claude_bin,
|
||||
timeout=args.timeout,
|
||||
|
|
@ -734,7 +740,7 @@ def run_proposer(
|
|||
auth_token=args.auth_token,
|
||||
base_url=args.base_url,
|
||||
model=args.proposer_model,
|
||||
build_sandbox_environment=build_sandbox_environment,
|
||||
build_sandbox_environment=environment_builder,
|
||||
),
|
||||
# No permission_mode: CLAUDE_CODE_SUBPROCESS_ENV_SCRUB
|
||||
# forces "default", so requesting dontAsk only warns. Tools
|
||||
|
|
@ -743,7 +749,10 @@ def run_proposer(
|
|||
# bare ignores --tools and imposes its own Bash/Edit/Read
|
||||
# ceiling, which would cost the proposer Grep and Glob.
|
||||
command_prefix=sandbox.command_prefix,
|
||||
require_pid_namespace=True,
|
||||
require_pid_namespace=getattr(sandbox, "require_pid_namespace", True),
|
||||
permission_mode=(
|
||||
"bypassPermissions" if backend == "host-unsafe" else None
|
||||
),
|
||||
settings_json=sandbox.settings_json,
|
||||
strict_mcp_config=True,
|
||||
mcp_config_json='{"mcpServers":{}}',
|
||||
|
|
@ -796,6 +805,21 @@ def resolve_incumbent_arms(overlay: Path, explicit_arms: list[str] | None) -> li
|
|||
return required
|
||||
|
||||
|
||||
def executed_benchmark_arms(incumbent_arms: Sequence[str]) -> list[str]:
|
||||
"""Incumbent/candidate pairs plus the review comparator when needed."""
|
||||
|
||||
paired = [arm for incumbent in incumbent_arms for arm in (incumbent, INCUMBENT_ARMS[incumbent])]
|
||||
if "review" in incumbent_arms:
|
||||
paired.insert(0, "ce_review")
|
||||
return paired
|
||||
|
||||
|
||||
def _timeout_arm_key(arm: str) -> str:
|
||||
if arm == "ce_review":
|
||||
return "review"
|
||||
return CANDIDATE_ARMS.get(arm, arm)
|
||||
|
||||
|
||||
def generation_timeout_seconds(
|
||||
*,
|
||||
task_count: int,
|
||||
|
|
@ -808,11 +832,14 @@ def generation_timeout_seconds(
|
|||
if task_count < 1 or runs < 1 or session_timeout < 1:
|
||||
raise ValueError("task count, runs, and session timeout must be positive")
|
||||
try:
|
||||
session_slots = sum(2 * ARM_SESSION_COUNTS[arm] for arm in incumbent_arms)
|
||||
executed = executed_benchmark_arms(incumbent_arms)
|
||||
session_slots = sum(ARM_SESSION_COUNTS[_timeout_arm_key(arm)] for arm in executed)
|
||||
workspace_snapshot_slots = sum(
|
||||
ARM_WORKSPACE_SNAPSHOT_COUNTS[_timeout_arm_key(arm)] for arm in executed
|
||||
)
|
||||
except KeyError as exc:
|
||||
raise ValueError(f"unsupported evolution arm: {exc.args[0]}") from exc
|
||||
paired_arm_cells = 2 * len(incumbent_arms)
|
||||
workspace_snapshot_slots = sum(2 * ARM_WORKSPACE_SNAPSHOT_COUNTS[arm] for arm in incumbent_arms)
|
||||
paired_arm_cells = len(executed)
|
||||
per_task_preparation = (
|
||||
TASK_BINDING_GIT_PHASES * GIT_COMMAND_TIMEOUT_SECONDS
|
||||
+ 2 * TASK_SNAPSHOT_TIMEOUT_SECONDS
|
||||
|
|
@ -849,9 +876,7 @@ def runner_argv(
|
|||
proposer_model: str | None = None,
|
||||
) -> list[str]:
|
||||
incumbent_arms = resolve_incumbent_arms(overlay_dir, args.arms)
|
||||
paired_arms = [arm for incumbent in incumbent_arms for arm in (incumbent, INCUMBENT_ARMS[incumbent])]
|
||||
if "review" in incumbent_arms:
|
||||
paired_arms.insert(0, "ce_review")
|
||||
paired_arms = executed_benchmark_arms(incumbent_arms)
|
||||
argv = [
|
||||
sys.executable,
|
||||
"-m",
|
||||
|
|
@ -897,6 +922,8 @@ def runner_argv(
|
|||
argv.append("--include-expensive")
|
||||
if args.ce_plugin_dir is not None:
|
||||
argv += ["--ce-plugin-dir", str(args.ce_plugin_dir), "--ce-plugin-version", args.ce_plugin_version]
|
||||
if args.unsafe_no_bwrap:
|
||||
argv.append("--unsafe-no-bwrap")
|
||||
return argv
|
||||
|
||||
|
||||
|
|
@ -1186,6 +1213,12 @@ def build_parser() -> argparse.ArgumentParser:
|
|||
)
|
||||
parser.add_argument("--ce-plugin-dir", type=Path, default=None)
|
||||
parser.add_argument("--ce-plugin-version", default=None)
|
||||
parser.add_argument(
|
||||
"--unsafe-no-bwrap",
|
||||
action="store_true",
|
||||
help="LOCAL DIAGNOSTICS ONLY: use PRoot path translation without filesystem, "
|
||||
"network, or PID isolation; forbidden with --apply and in CI",
|
||||
)
|
||||
return parser
|
||||
|
||||
|
||||
|
|
@ -1196,6 +1229,10 @@ def main() -> int:
|
|||
parser.error("--generations must be positive")
|
||||
if args.runs < 1 or args.timeout < 1:
|
||||
parser.error("--runs and --timeout must be positive")
|
||||
if args.unsafe_no_bwrap and args.apply:
|
||||
parser.error("--unsafe-no-bwrap cannot be combined with --apply")
|
||||
if args.unsafe_no_bwrap and os.environ.get("CI"):
|
||||
parser.error("--unsafe-no-bwrap is forbidden when CI is set")
|
||||
try:
|
||||
args.model = runner.normalized_model_identifier(args.model)
|
||||
args.proposer_model = runner.normalized_model_identifier(
|
||||
|
|
@ -1213,6 +1250,8 @@ def main() -> int:
|
|||
parser.error(str(exc))
|
||||
raise AssertionError("ArgumentParser.error() returned unexpectedly")
|
||||
requested_arms = args.arms or ["workflow", "workflow_direct"]
|
||||
if args.unsafe_no_bwrap and requested_arms != ["review"]:
|
||||
parser.error("--unsafe-no-bwrap is restricted to --arms review")
|
||||
if "review" in requested_arms and (
|
||||
args.ce_plugin_dir is None
|
||||
or not args.ce_plugin_dir.expanduser().is_dir()
|
||||
|
|
@ -1229,8 +1268,19 @@ def main() -> int:
|
|||
parser.error(str(exc))
|
||||
selected_tasks = runner.selected_task_bindings(selected_task_rows)
|
||||
try:
|
||||
bwrap_bin = preflight_bubblewrap()
|
||||
require_claude_sandbox_helpers()
|
||||
if args.unsafe_no_bwrap:
|
||||
bwrap_bin = preflight_unsafe_host()
|
||||
sandbox_backend = "host-unsafe"
|
||||
print(
|
||||
"WARNING: --unsafe-no-bwrap runs sessions directly on the host with no "
|
||||
"containment; model and verifier processes can access the host filesystem, "
|
||||
"network, and credentials.",
|
||||
file=sys.stderr,
|
||||
)
|
||||
else:
|
||||
bwrap_bin = preflight_bubblewrap()
|
||||
sandbox_backend = "bwrap"
|
||||
require_claude_sandbox_helpers()
|
||||
except SandboxError as exc:
|
||||
parser.error(str(exc))
|
||||
raise AssertionError("ArgumentParser.error() returned unexpectedly")
|
||||
|
|
@ -1250,6 +1300,7 @@ def main() -> int:
|
|||
requested_arms=requested_arms,
|
||||
initial_overlay=initial_overlay,
|
||||
bwrap_bin=bwrap_bin,
|
||||
sandbox_backend=sandbox_backend,
|
||||
)
|
||||
finally:
|
||||
gateway.__exit__(None, None, None)
|
||||
|
|
@ -1264,6 +1315,7 @@ def _run_generations(
|
|||
requested_arms: list[str],
|
||||
initial_overlay: Path | None,
|
||||
bwrap_bin: Path,
|
||||
sandbox_backend: str,
|
||||
) -> int:
|
||||
policy_binding = {
|
||||
"metric": args.promotion_metric,
|
||||
|
|
@ -1344,6 +1396,7 @@ def _run_generations(
|
|||
proposal_path=gen_dir / "proposal.md",
|
||||
evidence_bundle=bundle,
|
||||
bwrap_bin=bwrap_bin,
|
||||
sandbox_backend=sandbox_backend,
|
||||
progress_label=f"gen {generation} proposer",
|
||||
)
|
||||
# Redact any API token echoed into the session record (e.g. an
|
||||
|
|
@ -1383,18 +1436,21 @@ def _run_generations(
|
|||
print(f"[gen {generation}] promotion targets contain uncommitted or drifted bytes")
|
||||
return 1
|
||||
print(f"[gen {generation}] benchmarking candidate…")
|
||||
benchmark_argv = runner_argv(
|
||||
args,
|
||||
bench_dir,
|
||||
frozen_overlay,
|
||||
task_bindings=selected_tasks,
|
||||
target_base_digests=target_base_digests,
|
||||
proposer_model=generation_proposer_model,
|
||||
)
|
||||
benchmark_command = (
|
||||
benchmark_argv
|
||||
if sandbox_backend == "host-unsafe"
|
||||
else pid_namespace_command(benchmark_argv, bwrap_bin=bwrap_bin)
|
||||
)
|
||||
bench = run_managed(
|
||||
pid_namespace_command(
|
||||
runner_argv(
|
||||
args,
|
||||
bench_dir,
|
||||
frozen_overlay,
|
||||
task_bindings=selected_tasks,
|
||||
target_base_digests=target_base_digests,
|
||||
proposer_model=generation_proposer_model,
|
||||
),
|
||||
bwrap_bin=bwrap_bin,
|
||||
),
|
||||
benchmark_command,
|
||||
timeout=generation_timeout_seconds(
|
||||
task_count=len(selected_task_rows),
|
||||
runs=args.runs,
|
||||
|
|
@ -1402,7 +1458,7 @@ def _run_generations(
|
|||
incumbent_arms=incumbent_arms,
|
||||
),
|
||||
env=runner_environment(args),
|
||||
require_pid_namespace=True,
|
||||
require_pid_namespace=sandbox_backend == "bwrap",
|
||||
# The sweep is the multi-hour phase; without this its per-run
|
||||
# progress lines only reach the log as a bounded tail, and only
|
||||
# when it fails.
|
||||
|
|
|
|||
|
|
@ -42,6 +42,27 @@ _OPENAI_MODEL = re.compile(
|
|||
# shorter than that, so both ends of the loopback hop get the same generous
|
||||
# budget and the session fails on real errors instead of on the clock.
|
||||
GATEWAY_REQUEST_TIMEOUT_S = 1800
|
||||
# Importing LiteLLM alone costs ~17s on a cold container filesystem, and the
|
||||
# proxy only binds its port after that. A budget tight enough to lose that race
|
||||
# reads as "connection refused", which looks like a dead proxy rather than a
|
||||
# slow import.
|
||||
GATEWAY_READY_TIMEOUT_ENV = "GITNEXUS_BENCH_GATEWAY_READY_TIMEOUT_S"
|
||||
DEFAULT_GATEWAY_READY_TIMEOUT_S = 180.0
|
||||
|
||||
|
||||
def gateway_ready_timeout_s() -> float:
|
||||
"""Startup budget for the loopback proxy, overridable for slow hosts."""
|
||||
|
||||
raw = (os.environ.get(GATEWAY_READY_TIMEOUT_ENV) or "").strip()
|
||||
if not raw:
|
||||
return DEFAULT_GATEWAY_READY_TIMEOUT_S
|
||||
try:
|
||||
value = float(raw)
|
||||
except ValueError as exc:
|
||||
raise ValueError(f"{GATEWAY_READY_TIMEOUT_ENV} must be a number of seconds, not {raw!r}") from exc
|
||||
if value <= 0:
|
||||
raise ValueError(f"{GATEWAY_READY_TIMEOUT_ENV} must be positive, not {raw!r}")
|
||||
return value
|
||||
|
||||
|
||||
def is_openai_model(model: str) -> bool:
|
||||
|
|
@ -241,12 +262,12 @@ class OpenAIGateway(AbstractContextManager["OpenAIGateway"]):
|
|||
openai_api_key: str,
|
||||
model_names: Sequence[str],
|
||||
work_dir: Path,
|
||||
ready_timeout_s: float = 20.0,
|
||||
ready_timeout_s: float | None = None,
|
||||
) -> None:
|
||||
self.openai_api_key = openai_api_key
|
||||
self.model_names = tuple(model_names)
|
||||
self.work_dir = work_dir
|
||||
self.ready_timeout_s = ready_timeout_s
|
||||
self.ready_timeout_s = gateway_ready_timeout_s() if ready_timeout_s is None else ready_timeout_s
|
||||
self.auth_token = secrets.token_hex(16)
|
||||
self.port = _free_loopback_port()
|
||||
self.base_url = f"http://127.0.0.1:{self.port}"
|
||||
|
|
@ -342,7 +363,13 @@ class OpenAIGateway(AbstractContextManager["OpenAIGateway"]):
|
|||
except (urllib.error.URLError, TimeoutError, ConnectionError, OSError) as exc:
|
||||
last_error = str(exc)
|
||||
time.sleep(0.1)
|
||||
raise RuntimeError(f"OpenAI LiteLLM gateway was not ready on {self.base_url}: {last_error}")
|
||||
detail = self.log_tail()
|
||||
raise RuntimeError(
|
||||
f"OpenAI LiteLLM gateway was not ready on {self.base_url} after "
|
||||
f"{self.ready_timeout_s:.0f}s: {last_error}"
|
||||
+ (f"; proxy log: {detail}" if detail else "; proxy wrote no output yet")
|
||||
+ f" (raise {GATEWAY_READY_TIMEOUT_ENV} if this host is simply slow to import LiteLLM)"
|
||||
)
|
||||
|
||||
|
||||
class attach_openai_gateway(AbstractContextManager[argparse.Namespace]):
|
||||
|
|
|
|||
|
|
@ -24,6 +24,10 @@ MAX_ORACLE_PATH_BYTES = 240
|
|||
MAX_ORACLE_COMMAND_BYTES = 8 * 1024
|
||||
ORACLE_ENV_VAR = "GITNEXUS_BENCH_ORACLE_ROOT"
|
||||
HIDDEN_HARNESS_PATH = PurePosixPath("eval/workflow_bench")
|
||||
# Historical PR diffs can still edit this tree. Sanitization deletes it before
|
||||
# setup, so `git apply` must skip those hunks or it fails with
|
||||
# `error: eval/workflow_bench/<file>: No such file or directory`.
|
||||
HIDDEN_HARNESS_APPLY_EXCLUDE = f"{HIDDEN_HARNESS_PATH.as_posix()}/*"
|
||||
MAX_CLONE_REFS = 1024
|
||||
MAX_CLONE_REF_BYTES = 2 * 1024 * 1024
|
||||
|
||||
|
|
@ -240,6 +244,36 @@ def _git_checked(
|
|||
return result.stdout_tail.strip()
|
||||
|
||||
|
||||
def with_hidden_harness_apply_exclude(setup: str) -> str:
|
||||
"""Skip hunks for the harness tree sanitization already deleted.
|
||||
|
||||
Review cells copy a historical PR patch back under ``eval/workflow_bench``
|
||||
and then ``git apply`` it. That patch may still mention harness files
|
||||
(``learnings.jsonl`` on review-pr-2718-defect). Those files are gone from
|
||||
the parentless snapshot, and setup deletes the directory again after apply,
|
||||
so the hunks are never model-visible.
|
||||
"""
|
||||
|
||||
if "git apply" not in setup:
|
||||
return setup
|
||||
flag = f"--exclude='{HIDDEN_HARNESS_APPLY_EXCLUDE}'"
|
||||
if flag in setup or f'--exclude="{HIDDEN_HARNESS_APPLY_EXCLUDE}"' in setup:
|
||||
return setup
|
||||
return setup.replace("git apply ", f"git apply {flag} ", 1)
|
||||
|
||||
|
||||
def review_case_setup_command(patch_name: str) -> str:
|
||||
"""Setup that applies one review-case patch and then hides the harness."""
|
||||
|
||||
if not patch_name or "/" in patch_name or "\\" in patch_name or patch_name in {".", ".."}:
|
||||
raise ValueError(f"review patch name must be a single path segment: {patch_name!r}")
|
||||
patch = f"{HIDDEN_HARNESS_PATH.as_posix()}/review_cases/{patch_name}"
|
||||
return (
|
||||
f"git apply --exclude='{HIDDEN_HARNESS_APPLY_EXCLUDE}' {patch} "
|
||||
f"&& rm -rf {HIDDEN_HARNESS_PATH.as_posix()}"
|
||||
)
|
||||
|
||||
|
||||
def sanitize_clone_for_hidden_oracles(clone: Path) -> str:
|
||||
"""Remove the harness and its recoverable Git history from a disposable clone.
|
||||
|
||||
|
|
|
|||
|
|
@ -82,13 +82,58 @@ class SandboxSession:
|
|||
home: Path
|
||||
temp: Path
|
||||
bwrap_bin: Path
|
||||
backend: str
|
||||
claude_host_bin: Path
|
||||
command_prefix: list[str]
|
||||
read_only_mounts: tuple[ReadOnlyMount, ...]
|
||||
|
||||
@property
|
||||
def require_pid_namespace(self) -> bool:
|
||||
return self.backend == "bwrap"
|
||||
|
||||
def host_path(self, raw: str | Path) -> str:
|
||||
"""Translate a sandbox path to its real host path in unsafe mode."""
|
||||
|
||||
value = str(raw)
|
||||
if self.backend == "bwrap":
|
||||
return value
|
||||
node = Path(shutil.which("node") or "/usr/bin/node").resolve()
|
||||
mappings = [
|
||||
*(self.read_only_mounts),
|
||||
ReadOnlyMount(node.parent.parent, SANDBOX_NODE_PREFIX),
|
||||
ReadOnlyMount(node, SANDBOX_NODE),
|
||||
ReadOnlyMount(self.claude_host_bin, SANDBOX_CLAUDE),
|
||||
ReadOnlyMount(self.clone, SANDBOX_WORKSPACE),
|
||||
ReadOnlyMount(self.home, SANDBOX_HOME),
|
||||
ReadOnlyMount(self.temp, SANDBOX_TMP),
|
||||
]
|
||||
for mount in sorted(mappings, key=lambda item: len(item.target), reverse=True):
|
||||
target = mount.target.rstrip("/")
|
||||
if value == target:
|
||||
return str(mount.source)
|
||||
if value.startswith(f"{target}/"):
|
||||
return str(mount.source / value[len(target) + 1 :])
|
||||
return value
|
||||
|
||||
def host_text(self, value: str) -> str:
|
||||
if self.backend == "bwrap":
|
||||
return value
|
||||
targets = [
|
||||
*(mount.target for mount in self.read_only_mounts),
|
||||
SANDBOX_NODE_PREFIX,
|
||||
SANDBOX_NODE,
|
||||
SANDBOX_CLAUDE,
|
||||
SANDBOX_WORKSPACE,
|
||||
SANDBOX_HOME,
|
||||
SANDBOX_TMP,
|
||||
]
|
||||
ordered = sorted(set(targets), key=len, reverse=True)
|
||||
pattern = re.compile("|".join(re.escape(target) for target in ordered))
|
||||
return pattern.sub(lambda match: self.host_path(match.group(0)), value)
|
||||
|
||||
@property
|
||||
def claude_bin(self) -> str:
|
||||
return SANDBOX_CLAUDE
|
||||
return self.host_path(SANDBOX_CLAUDE)
|
||||
|
||||
@property
|
||||
def transcript_projects(self) -> Path:
|
||||
|
|
@ -96,7 +141,7 @@ class SandboxSession:
|
|||
|
||||
@property
|
||||
def settings_json(self) -> str:
|
||||
return build_claude_settings()
|
||||
return build_claude_settings(sandbox_enabled=self.backend == "bwrap")
|
||||
|
||||
def environment(
|
||||
self,
|
||||
|
|
@ -104,7 +149,17 @@ class SandboxSession:
|
|||
auth_token: str | None = None,
|
||||
base_url: str | None = None,
|
||||
) -> dict[str, str]:
|
||||
return build_sandbox_environment(auth_token=auth_token, base_url=base_url)
|
||||
env = build_sandbox_environment(auth_token=auth_token, base_url=base_url)
|
||||
if self.backend != "bwrap":
|
||||
env = {key: self.host_text(value) for key, value in env.items()}
|
||||
env.pop("CLAUDE_CODE_SUBPROCESS_ENV_SCRUB", None)
|
||||
env.pop("CLAUDE_CODE_SHELL_PREFIX", None)
|
||||
# Sandbox-only bin dirs have no single host counterpart; drop the
|
||||
# entries that survive translation as non-existent paths.
|
||||
env["PATH"] = ":".join(
|
||||
entry for entry in env["PATH"].split(":") if Path(entry).is_dir()
|
||||
)
|
||||
return env
|
||||
|
||||
def run(
|
||||
self,
|
||||
|
|
@ -114,11 +169,17 @@ class SandboxSession:
|
|||
env: Mapping[str, str] | None = None,
|
||||
stdin_data: bytes | None = None,
|
||||
) -> ManagedProcessResult:
|
||||
translated = [self.host_text(str(part)) for part in command]
|
||||
return run_managed(
|
||||
[*self.command_prefix, *command],
|
||||
[*self.command_prefix, *translated],
|
||||
cwd=None if self.command_prefix else self.clone,
|
||||
timeout=timeout,
|
||||
env=dict(env) if env is not None else build_sandbox_environment(),
|
||||
require_pid_namespace=True,
|
||||
env=(
|
||||
{key: self.host_text(value) for key, value in env.items()}
|
||||
if env is not None
|
||||
else self.environment()
|
||||
),
|
||||
require_pid_namespace=self.require_pid_namespace,
|
||||
stdin_data=stdin_data,
|
||||
)
|
||||
|
||||
|
|
@ -139,6 +200,9 @@ class SandboxSession:
|
|||
for harness-owned, post-session evidence such as hidden oracles.
|
||||
"""
|
||||
|
||||
if self.backend != "bwrap":
|
||||
return []
|
||||
|
||||
additional: list[ReadOnlyMount] = []
|
||||
clone = _real_directory(self.clone, label="sandbox clone")
|
||||
for raw_path in read_only_paths:
|
||||
|
|
@ -347,7 +411,7 @@ def build_sandbox_environment(
|
|||
return env
|
||||
|
||||
|
||||
def build_claude_settings() -> str:
|
||||
def build_claude_settings(*, sandbox_enabled: bool = True) -> str:
|
||||
"""Inline settings that keep every Bash sandboxed and pre-approve the tools.
|
||||
|
||||
Deliberately hook-free: headless ``claude -p`` (2.1.247) never dispatches
|
||||
|
|
@ -358,11 +422,16 @@ def build_claude_settings() -> str:
|
|||
below, ``--tools``/``--allowedTools``, and the bwrap mounts.
|
||||
"""
|
||||
|
||||
permissions = {
|
||||
"allow": ["Read", "Grep", "Glob", "Bash"],
|
||||
}
|
||||
if sandbox_enabled:
|
||||
permissions["disableBypassPermissionsMode"] = "disable"
|
||||
settings = {
|
||||
"sandbox": {
|
||||
"enabled": True,
|
||||
"failIfUnavailable": True,
|
||||
"autoAllowBashIfSandboxed": True,
|
||||
"enabled": sandbox_enabled,
|
||||
"failIfUnavailable": sandbox_enabled,
|
||||
"autoAllowBashIfSandboxed": sandbox_enabled,
|
||||
"allowUnsandboxedCommands": False,
|
||||
"enableWeakerNestedSandbox": True,
|
||||
"network": {
|
||||
|
|
@ -398,13 +467,16 @@ def build_claude_settings() -> str:
|
|||
# allow rule, so pre-approve the proposer's exact tool surface. Bash
|
||||
# is the only writable tool under --bare (it writes the candidate
|
||||
# overlay) and stays sandbox-confined by the sandbox.* policy above.
|
||||
"allow": ["Read", "Grep", "Glob", "Bash"],
|
||||
"disableBypassPermissionsMode": "disable",
|
||||
},
|
||||
"env": {
|
||||
"CLAUDE_CODE_SUBPROCESS_ENV_SCRUB": "1",
|
||||
"CLAUDE_CODE_DONT_INHERIT_ENV": "1",
|
||||
**permissions,
|
||||
},
|
||||
"env": (
|
||||
{
|
||||
"CLAUDE_CODE_SUBPROCESS_ENV_SCRUB": "1",
|
||||
"CLAUDE_CODE_DONT_INHERIT_ENV": "1",
|
||||
}
|
||||
if sandbox_enabled
|
||||
else {"CLAUDE_CODE_DONT_INHERIT_ENV": "1"}
|
||||
),
|
||||
}
|
||||
return json.dumps(settings, sort_keys=True, separators=(",", ":"))
|
||||
|
||||
|
|
@ -570,6 +642,12 @@ def preflight_bubblewrap(bwrap_bin: Path | str | None = None) -> Path:
|
|||
return bwrap
|
||||
|
||||
|
||||
def preflight_unsafe_host() -> Path:
|
||||
"""Return a harmless sentinel for the explicit non-containment backend."""
|
||||
|
||||
return _resolve_executable(None, "env")
|
||||
|
||||
|
||||
def pid_namespace_command(
|
||||
command: Sequence[str],
|
||||
*,
|
||||
|
|
@ -778,6 +856,145 @@ def _sandbox_command_prefix(
|
|||
return args
|
||||
|
||||
|
||||
def _drop_host_workspace_write_bits(
|
||||
root: Path,
|
||||
*,
|
||||
writable: Sequence[Path],
|
||||
) -> list[tuple[Path, int]]:
|
||||
"""Clear write bits on a host-unsafe clone except explicit artifact paths."""
|
||||
|
||||
root = root.expanduser().absolute()
|
||||
try:
|
||||
metadata = root.lstat()
|
||||
resolved = root.resolve(strict=True)
|
||||
except OSError as exc:
|
||||
raise SandboxError(f"host workspace lock root is unavailable: {root}: {exc}") from exc
|
||||
if stat.S_ISLNK(metadata.st_mode) or not stat.S_ISDIR(metadata.st_mode) or resolved != root:
|
||||
raise SandboxError(f"host workspace lock root must be a real directory: {root}")
|
||||
|
||||
allowed: set[Path] = set()
|
||||
for raw in writable:
|
||||
path = raw.expanduser().absolute()
|
||||
try:
|
||||
path.relative_to(root)
|
||||
except ValueError as exc:
|
||||
raise SandboxError(f"writable host path escapes the workspace: {raw}") from exc
|
||||
allowed.add(path)
|
||||
|
||||
records: list[tuple[Path, int]] = []
|
||||
pending = [root]
|
||||
while pending:
|
||||
current = pending.pop()
|
||||
try:
|
||||
metadata = current.lstat()
|
||||
except OSError as exc:
|
||||
raise SandboxError(f"host workspace lock path is unreadable: {current}: {exc}") from exc
|
||||
records.append((current, stat.S_IMODE(metadata.st_mode)))
|
||||
if stat.S_ISLNK(metadata.st_mode):
|
||||
continue
|
||||
if stat.S_ISDIR(metadata.st_mode):
|
||||
try:
|
||||
children = [Path(entry.path) for entry in os.scandir(current)]
|
||||
except OSError as exc:
|
||||
raise SandboxError(f"host workspace lock directory is unreadable: {current}: {exc}") from exc
|
||||
pending.extend(children)
|
||||
if current not in allowed:
|
||||
os.chmod(current, 0o555)
|
||||
continue
|
||||
if current in allowed:
|
||||
os.chmod(current, stat.S_IMODE(metadata.st_mode) | 0o222)
|
||||
continue
|
||||
if stat.S_ISREG(metadata.st_mode):
|
||||
os.chmod(current, stat.S_IMODE(metadata.st_mode) & ~0o222)
|
||||
return records
|
||||
|
||||
|
||||
def _force_rmtree(path: Path) -> None:
|
||||
"""Delete a tree even when leftover copies inherited 0555 directory modes.
|
||||
|
||||
Host-unsafe review sessions drop write bits on the clone. An agent that
|
||||
``copytree``s those directories into the private TMPDIR leaves a tree
|
||||
``shutil.rmtree`` cannot remove: a non-empty 0555 directory raises
|
||||
PermissionError. Restore owner write bits, then delete.
|
||||
"""
|
||||
|
||||
root = Path(path)
|
||||
if not root.exists():
|
||||
return
|
||||
for dirpath, _dirnames, filenames in os.walk(root, topdown=False, followlinks=False):
|
||||
try:
|
||||
os.chmod(
|
||||
dirpath,
|
||||
stat.S_IMODE(os.lstat(dirpath).st_mode) | 0o700,
|
||||
follow_symlinks=False,
|
||||
)
|
||||
except OSError:
|
||||
pass
|
||||
for name in filenames:
|
||||
child = os.path.join(dirpath, name)
|
||||
try:
|
||||
metadata = os.lstat(child)
|
||||
except OSError:
|
||||
continue
|
||||
if stat.S_ISLNK(metadata.st_mode):
|
||||
continue
|
||||
try:
|
||||
os.chmod(child, stat.S_IMODE(metadata.st_mode) | 0o200, follow_symlinks=False)
|
||||
except OSError:
|
||||
pass
|
||||
shutil.rmtree(root)
|
||||
|
||||
|
||||
def _restore_host_workspace_modes(records: Sequence[tuple[Path, int]]) -> None:
|
||||
errors: list[str] = []
|
||||
for path, mode in reversed(records):
|
||||
try:
|
||||
os.chmod(path, mode)
|
||||
except FileNotFoundError:
|
||||
continue
|
||||
except OSError as exc:
|
||||
errors.append(f"{path}: {exc}")
|
||||
if errors:
|
||||
raise SandboxError("failed to restore host workspace modes: " + "; ".join(errors[:8]))
|
||||
|
||||
|
||||
@contextmanager
|
||||
def host_workspace_write_boundary(
|
||||
root: Path,
|
||||
*,
|
||||
writable: Sequence[Path] = (),
|
||||
) -> Iterator[None]:
|
||||
"""Best-effort host analog of bwrap ``--ro-bind`` plus one writable artifact.
|
||||
|
||||
This is not a security boundary: a session that can ``chmod`` can undo it.
|
||||
It is enough to stop accidental ``npm install`` / analyze writes from
|
||||
invalidating review evidence on a host-unsafe diagnostic run.
|
||||
"""
|
||||
|
||||
records = _drop_host_workspace_write_bits(root, writable=writable)
|
||||
try:
|
||||
yield
|
||||
finally:
|
||||
_restore_host_workspace_modes(records)
|
||||
|
||||
|
||||
@contextmanager
|
||||
def sandbox_workspace_write_boundary(
|
||||
sandbox: Any,
|
||||
*,
|
||||
read_only_workspace: bool,
|
||||
writable: Sequence[Path] = (),
|
||||
) -> Iterator[None]:
|
||||
"""Apply the host-unsafe write lock only when bwrap is not the backend."""
|
||||
|
||||
if getattr(sandbox, "backend", "bwrap") != "host-unsafe" or not read_only_workspace:
|
||||
yield
|
||||
return
|
||||
clone = Path(sandbox.clone)
|
||||
with host_workspace_write_boundary(clone, writable=writable):
|
||||
yield
|
||||
|
||||
|
||||
@contextmanager
|
||||
def prepare_sandbox(
|
||||
*,
|
||||
|
|
@ -786,17 +1003,21 @@ def prepare_sandbox(
|
|||
bwrap_bin: Path | str | None = None,
|
||||
read_only_mounts: Sequence[ReadOnlyMount] = (),
|
||||
preflight: bool = True,
|
||||
backend: str = "bwrap",
|
||||
) -> Iterator[SandboxSession]:
|
||||
"""Create private host backing dirs and one immutable Bubblewrap command."""
|
||||
"""Create private host backing dirs and one virtualized command."""
|
||||
|
||||
# Validate the lexical path before resolving it. Resolving first would
|
||||
# erase the evidence that the caller supplied a symlinked clone root.
|
||||
clone = _real_directory(clone, label="sandbox clone")
|
||||
if backend not in ("bwrap", "host-unsafe"):
|
||||
raise SandboxError(f"unknown sandbox backend: {backend}")
|
||||
if preflight:
|
||||
bwrap = preflight_bubblewrap(bwrap_bin)
|
||||
require_claude_sandbox_helpers()
|
||||
bwrap = preflight_bubblewrap(bwrap_bin) if backend == "bwrap" else preflight_unsafe_host()
|
||||
if backend == "bwrap":
|
||||
require_claude_sandbox_helpers()
|
||||
else:
|
||||
bwrap = _resolve_executable(bwrap_bin, "bwrap")
|
||||
bwrap = _resolve_executable(bwrap_bin, "bwrap") if backend == "bwrap" else preflight_unsafe_host()
|
||||
claude = _resolve_executable(claude_bin, "claude")
|
||||
private_root = Path(tempfile.mkdtemp(prefix="wfbench-sandbox-"))
|
||||
private_root.chmod(0o700)
|
||||
|
|
@ -825,13 +1046,17 @@ def prepare_sandbox(
|
|||
)
|
||||
primary: BaseException | None = None
|
||||
try:
|
||||
command_prefix = _sandbox_command_prefix(
|
||||
bwrap=bwrap,
|
||||
clone=clone,
|
||||
home=home,
|
||||
temp=temp,
|
||||
claude_bin=claude,
|
||||
mounts=protected_mounts,
|
||||
command_prefix = (
|
||||
_sandbox_command_prefix(
|
||||
bwrap=bwrap,
|
||||
clone=clone,
|
||||
home=home,
|
||||
temp=temp,
|
||||
claude_bin=claude,
|
||||
mounts=protected_mounts,
|
||||
)
|
||||
if backend == "bwrap"
|
||||
else []
|
||||
)
|
||||
yield SandboxSession(
|
||||
private_root=private_root,
|
||||
|
|
@ -839,6 +1064,7 @@ def prepare_sandbox(
|
|||
home=home,
|
||||
temp=temp,
|
||||
bwrap_bin=bwrap,
|
||||
backend=backend,
|
||||
claude_host_bin=claude,
|
||||
command_prefix=command_prefix,
|
||||
read_only_mounts=protected_mounts,
|
||||
|
|
@ -848,7 +1074,7 @@ def prepare_sandbox(
|
|||
raise
|
||||
finally:
|
||||
try:
|
||||
shutil.rmtree(private_root)
|
||||
_force_rmtree(private_root)
|
||||
except OSError as cleanup:
|
||||
if primary is None:
|
||||
raise
|
||||
|
|
|
|||
|
|
@ -15,6 +15,7 @@
|
|||
# MODEL PROPOSER_MODEL EFFORT GENERATIONS RUNS WORKERS PROVIDER
|
||||
# EVOLUTION_PROFILE CE_PLUGIN_DIR CE_PLUGIN_VERSION
|
||||
# INCLUDE_EXPENSIVE SEED_RESULTS CLAUDE_BIN OUT_ROOT
|
||||
# UNSAFE_NO_BWRAP=1 (local review diagnostics only)
|
||||
# GITNEXUS_BENCH_ANTHROPIC_API_KEY (legacy GITNEXUS_BENCH_AUTH_TOKEN)
|
||||
# GITNEXUS_BENCH_OPENAI_API_KEY
|
||||
set -euo pipefail
|
||||
|
|
@ -181,6 +182,13 @@ fi
|
|||
if [[ -n "${SEED_RESULTS}" ]]; then
|
||||
cmd+=(--seed-results "${SEED_RESULTS}")
|
||||
fi
|
||||
if [[ -n "${UNSAFE_NO_BWRAP:-}" && "${UNSAFE_NO_BWRAP}" != "0" && "${UNSAFE_NO_BWRAP}" != "false" ]]; then
|
||||
if [[ -n "${CI:-}" ]]; then
|
||||
echo "UNSAFE_NO_BWRAP is forbidden in CI." >&2
|
||||
exit 1
|
||||
fi
|
||||
cmd+=(--unsafe-no-bwrap)
|
||||
fi
|
||||
if ((${#passthrough[@]})); then
|
||||
cmd+=("${passthrough[@]}")
|
||||
fi
|
||||
|
|
@ -200,13 +208,15 @@ runtime_digest="$(
|
|||
sha256sum "${eval_dir}/../gitnexus-shared/package-lock.json"
|
||||
} | sha256sum | cut -d' ' -f1
|
||||
)"
|
||||
SOURCE_SHA="${source_sha}" RUNTIME_DIGEST="${runtime_digest}" \
|
||||
unsafe_backend="$([[ -n "${UNSAFE_NO_BWRAP:-}" && "${UNSAFE_NO_BWRAP}" != "0" && "${UNSAFE_NO_BWRAP}" != "false" ]] && echo host-unsafe || echo bwrap)"
|
||||
SOURCE_SHA="${source_sha}" RUNTIME_DIGEST="${runtime_digest}" SANDBOX_BACKEND="${unsafe_backend}" \
|
||||
node -e 'require("fs").writeFileSync(process.argv[1], JSON.stringify({
|
||||
schema_version: 1,
|
||||
source_sha: process.env.SOURCE_SHA,
|
||||
runtime_digest: process.env.RUNTIME_DIGEST,
|
||||
profile: process.env.EVOLUTION_PROFILE,
|
||||
ce_plugin_version: process.env.CE_PLUGIN_VERSION
|
||||
ce_plugin_version: process.env.CE_PLUGIN_VERSION,
|
||||
sandbox_backend: process.env.SANDBOX_BACKEND
|
||||
}, null, 2) + "\n")' "${out_root}/runtime-provenance.json"
|
||||
|
||||
export PYTHONUNBUFFERED=1
|
||||
|
|
|
|||
|
|
@ -43,6 +43,7 @@ import re
|
|||
import secrets
|
||||
import stat
|
||||
import statistics
|
||||
import sys
|
||||
import tempfile
|
||||
import time
|
||||
from collections.abc import Callable, Mapping, Sequence
|
||||
|
|
@ -68,6 +69,7 @@ from .evolution import (
|
|||
evaluate_candidate,
|
||||
evaluate_review_candidate,
|
||||
required_candidate_arms,
|
||||
seed_evaluated_skills,
|
||||
skill_fingerprint,
|
||||
)
|
||||
from .model_gateway import (
|
||||
|
|
@ -83,6 +85,7 @@ from .oracle_assets import (
|
|||
capture_task_oracles,
|
||||
sanitize_clone_for_hidden_oracles,
|
||||
staged_task_oracle,
|
||||
with_hidden_harness_apply_exclude,
|
||||
)
|
||||
from .process_control import ManagedProcessError
|
||||
from .promotion_apply import committed_destination_base_digests
|
||||
|
|
@ -98,9 +101,11 @@ from .proposer_sandbox import (
|
|||
SandboxSession,
|
||||
build_sandbox_environment,
|
||||
preflight_bubblewrap,
|
||||
preflight_unsafe_host,
|
||||
prepare_sandbox,
|
||||
redact_text,
|
||||
require_claude_sandbox_helpers,
|
||||
sandbox_workspace_write_boundary,
|
||||
)
|
||||
from .runner_artifacts import (
|
||||
IMPLEMENTATION_ARMS,
|
||||
|
|
@ -329,7 +334,7 @@ def _run_hidden_oracle(
|
|||
extra_read_only_mounts=(ReadOnlyMount(source=stage_root, target=oracle_mount),),
|
||||
),
|
||||
env=oracle_env,
|
||||
require_pid_namespace=True,
|
||||
require_pid_namespace=getattr(sandbox, "require_pid_namespace", True),
|
||||
)
|
||||
)
|
||||
# Candidate code executes in this process. Never persist its stdout
|
||||
|
|
@ -423,11 +428,15 @@ def run_arm(
|
|||
oracle_snapshot: TaskOracleSnapshot | None = None,
|
||||
) -> dict[str, Any]:
|
||||
sessions: list[dict[str, Any]] = []
|
||||
environment_builder = getattr(sandbox, "environment", build_sandbox_environment)
|
||||
host_text = getattr(sandbox, "host_text", lambda value: value)
|
||||
host_path = getattr(sandbox, "host_path", lambda value: str(value))
|
||||
backend = getattr(sandbox, "backend", "bwrap")
|
||||
env = model_session_environment(
|
||||
auth_token=args.auth_token,
|
||||
base_url=args.base_url,
|
||||
model=args.model,
|
||||
build_sandbox_environment=build_sandbox_environment,
|
||||
build_sandbox_environment=environment_builder,
|
||||
)
|
||||
# --bare hard-disables the Skill tool and every mcp__* tool — by Claude
|
||||
# Code design, not a bug (--allowedTools can't restore what --bare
|
||||
|
|
@ -444,15 +453,17 @@ def run_arm(
|
|||
"model": args.model,
|
||||
"effort": args.effort,
|
||||
"env": env,
|
||||
"permission_mode": "dontAsk",
|
||||
"permission_mode": (
|
||||
"bypassPermissions" if backend == "host-unsafe" else "dontAsk"
|
||||
),
|
||||
"command_prefix": sandbox.command_prefix_for(
|
||||
read_only_paths=_evaluated_skill_roots(worktree, arm),
|
||||
),
|
||||
"require_pid_namespace": True,
|
||||
"require_pid_namespace": getattr(sandbox, "require_pid_namespace", True),
|
||||
"bare": bare,
|
||||
"settings_json": sandbox.settings_json,
|
||||
"strict_mcp_config": True,
|
||||
"mcp_config_json": sandbox_mcp_config(),
|
||||
"mcp_config_json": host_text(sandbox_mcp_config()),
|
||||
"transcript_projects": sandbox.transcript_projects,
|
||||
"transcript_cwd": Path(SANDBOX_WORKSPACE),
|
||||
"transcript_wait_seconds": 5,
|
||||
|
|
@ -461,7 +472,7 @@ def run_arm(
|
|||
"transcript_secrets": tuple(credential_secrets(args)),
|
||||
}
|
||||
if ce_plugin_dir is not None:
|
||||
common["plugin_dirs"] = (ce_plugin_dir,)
|
||||
common["plugin_dirs"] = (host_path(ce_plugin_dir),)
|
||||
expected_skills = ARM_EXPECTED_SKILLS.get(arm, ())
|
||||
plan_doc: Path | None = None
|
||||
if arm in ("workflow", "ce_workflow"):
|
||||
|
|
@ -538,7 +549,7 @@ def run_arm(
|
|||
phase_before = workspace_snapshot(worktree) if enforce_phase_boundary else None
|
||||
review_common = {
|
||||
**common,
|
||||
"allowed_tools": allowed_agent_tools(implementation=False),
|
||||
"allowed_tools": allowed_agent_tools(implementation=False, allow_edit=False),
|
||||
"command_prefix": sandbox.command_prefix_for(
|
||||
read_only_workspace=True,
|
||||
read_only_paths=_evaluated_skill_roots(worktree, arm),
|
||||
|
|
@ -550,12 +561,17 @@ def run_arm(
|
|||
),
|
||||
),
|
||||
}
|
||||
review_session = run_claude(
|
||||
review_prompt.format(task=task["prompt"]),
|
||||
worktree,
|
||||
expected_skill=expected_skills[0],
|
||||
**review_common,
|
||||
)
|
||||
with sandbox_workspace_write_boundary(
|
||||
sandbox,
|
||||
read_only_workspace=True,
|
||||
writable=(review_output,),
|
||||
):
|
||||
review_session = run_claude(
|
||||
host_text(review_prompt.format(task=task["prompt"])),
|
||||
worktree,
|
||||
expected_skill=expected_skills[0],
|
||||
**review_common,
|
||||
)
|
||||
sessions.append(review_session)
|
||||
if review_session["ok"] and phase_before is not None:
|
||||
try:
|
||||
|
|
@ -627,8 +643,8 @@ def run_arm(
|
|||
read_only_workspace=True,
|
||||
unshare_network=True,
|
||||
),
|
||||
env=build_sandbox_environment(),
|
||||
require_pid_namespace=True,
|
||||
env=environment_builder(),
|
||||
require_pid_namespace=getattr(sandbox, "require_pid_namespace", True),
|
||||
)
|
||||
)
|
||||
review_score: dict[str, Any] | None = None
|
||||
|
|
@ -849,6 +865,7 @@ class TaskCellContext:
|
|||
runtime_mounts: tuple[ReadOnlyMount, ...]
|
||||
candidate_overlay: Path | None
|
||||
overlay_digest: str | None
|
||||
sandbox_backend: str = "bwrap"
|
||||
|
||||
|
||||
def run_cell(ctx: TaskCellContext, run_idx: int, arm: str) -> dict[str, Any]:
|
||||
|
|
@ -909,6 +926,7 @@ def run_cell(ctx: TaskCellContext, run_idx: int, arm: str) -> dict[str, Any]:
|
|||
*oracle_visibility_mounts,
|
||||
],
|
||||
preflight=False,
|
||||
backend=ctx.sandbox_backend,
|
||||
) as sandbox:
|
||||
# Capture the BASE (pre-overlay) skill digest — identical
|
||||
# for the incumbent and candidate arms — then run the
|
||||
|
|
@ -916,9 +934,22 @@ def run_cell(ctx: TaskCellContext, run_idx: int, arm: str) -> dict[str, Any]:
|
|||
# candidate overlay is applied only afterwards, so setup
|
||||
# can never observe candidate prose and both arms share
|
||||
# byte-identical pre-overlay state.
|
||||
# Historical review SHAs may predate gitnexus-review. Seed
|
||||
# the current evaluated skill first so fingerprinting and
|
||||
# the model see the same incumbent prose on every case.
|
||||
if execution_arm == "review":
|
||||
seed_evaluated_skills(
|
||||
HARNESS_ROOT,
|
||||
worktree,
|
||||
sandbox=sandbox,
|
||||
arm=execution_arm,
|
||||
)
|
||||
base_skill_digest = skill_fingerprint(worktree, execution_arm)
|
||||
if task.get("setup"):
|
||||
setup_command = ["/bin/sh", "-lc", str(task["setup"])]
|
||||
# Sanitization already removed eval/workflow_bench. A historical
|
||||
# PR patch that still edits that tree (gitignored learnings.jsonl)
|
||||
# must skip those hunks or `git apply` fails closed.
|
||||
setup_command = ["/bin/sh", "-lc", with_hidden_harness_apply_exclude(str(task["setup"]))]
|
||||
setup = sandbox.run(
|
||||
setup_command,
|
||||
timeout=600,
|
||||
|
|
@ -1058,6 +1089,7 @@ def run_cell(ctx: TaskCellContext, run_idx: int, arm: str) -> dict[str, Any]:
|
|||
"task": task["id"],
|
||||
"class": task.get("class", ""),
|
||||
"run": run_idx,
|
||||
"sandbox_backend": ctx.sandbox_backend,
|
||||
"task_asset_snapshot_digest": (ctx.asset_snapshot.digest if ctx.asset_snapshot is not None else None),
|
||||
"task_asset_manifest_digest": (
|
||||
ctx.asset_snapshot.manifest_digest if ctx.asset_snapshot is not None else None
|
||||
|
|
@ -1498,6 +1530,7 @@ def build_parser() -> argparse.ArgumentParser:
|
|||
parser.add_argument("--promotion-max-task-regression", type=float, default=20.0)
|
||||
parser.add_argument("--task-bindings-json", default=None, help=argparse.SUPPRESS)
|
||||
parser.add_argument("--promotion-target-bases-json", default=None, help=argparse.SUPPRESS)
|
||||
parser.add_argument("--unsafe-no-bwrap", action="store_true", help=argparse.SUPPRESS)
|
||||
return parser
|
||||
|
||||
|
||||
|
|
@ -1574,9 +1607,24 @@ def main() -> None:
|
|||
parser.error("--promotion-target-bases-json requires --candidate-overlay")
|
||||
promotion_target_bases = {}
|
||||
|
||||
if args.unsafe_no_bwrap and os.environ.get("CI"):
|
||||
parser.error("--unsafe-no-bwrap is forbidden when CI is set")
|
||||
if args.unsafe_no_bwrap and args.arms != ["ce_review", "review", "candidate_review"]:
|
||||
parser.error("--unsafe-no-bwrap is restricted to the paired review arms")
|
||||
try:
|
||||
bwrap_bin = preflight_bubblewrap()
|
||||
require_claude_sandbox_helpers()
|
||||
if args.unsafe_no_bwrap:
|
||||
bwrap_bin = preflight_unsafe_host()
|
||||
sandbox_backend = "host-unsafe"
|
||||
print(
|
||||
"WARNING: --unsafe-no-bwrap runs sessions directly on the host with no "
|
||||
"containment; model and verifier processes can access the host filesystem, "
|
||||
"network, and credentials.",
|
||||
file=sys.stderr,
|
||||
)
|
||||
else:
|
||||
bwrap_bin = preflight_bubblewrap()
|
||||
sandbox_backend = "bwrap"
|
||||
require_claude_sandbox_helpers()
|
||||
runtime_mounts = trusted_gitnexus_runtime_mounts()
|
||||
except SandboxError as exc:
|
||||
parser.error(str(exc))
|
||||
|
|
@ -1597,6 +1645,7 @@ def main() -> None:
|
|||
expected_task_bindings=expected_task_bindings,
|
||||
ce_plugin_config=ce_plugin_config,
|
||||
bwrap_bin=bwrap_bin,
|
||||
sandbox_backend=sandbox_backend,
|
||||
runtime_mounts=runtime_mounts,
|
||||
candidate_arms=candidate_arms,
|
||||
candidate_overlay=candidate_overlay,
|
||||
|
|
@ -1617,6 +1666,7 @@ def _run_sweep(
|
|||
expected_task_bindings: Any,
|
||||
ce_plugin_config: Any,
|
||||
bwrap_bin: Any,
|
||||
sandbox_backend: str,
|
||||
runtime_mounts: Any,
|
||||
candidate_arms: list[str],
|
||||
candidate_overlay: Path | None,
|
||||
|
|
@ -1691,6 +1741,7 @@ def _run_sweep(
|
|||
cache=task_asset_cache,
|
||||
claude_bin=args.claude_bin,
|
||||
bwrap_bin=bwrap_bin,
|
||||
sandbox_backend=sandbox_backend,
|
||||
runtime_mounts=runtime_mounts,
|
||||
)
|
||||
graph_snapshots[graph_key] = graph_snapshot
|
||||
|
|
@ -1724,6 +1775,7 @@ def _run_sweep(
|
|||
ce_plugin_snapshot=ce_plugin_snapshot,
|
||||
trees_dir=Path(trees),
|
||||
bwrap_bin=bwrap_bin,
|
||||
sandbox_backend=sandbox_backend,
|
||||
runtime_mounts=runtime_mounts,
|
||||
candidate_overlay=candidate_overlay,
|
||||
overlay_digest=overlay_digest,
|
||||
|
|
|
|||
|
|
@ -361,8 +361,13 @@ def sandbox_mcp_config() -> str:
|
|||
return json.dumps(config, sort_keys=True, separators=(",", ":"))
|
||||
|
||||
|
||||
def allowed_agent_tools(*, implementation: bool, include_mcp: bool = True) -> list[str]:
|
||||
tools = [*BUILTIN_AGENT_TOOLS]
|
||||
def allowed_agent_tools(
|
||||
*,
|
||||
implementation: bool,
|
||||
include_mcp: bool = True,
|
||||
allow_edit: bool = True,
|
||||
) -> list[str]:
|
||||
tools = [tool for tool in BUILTIN_AGENT_TOOLS if allow_edit or tool != "Edit"]
|
||||
if include_mcp:
|
||||
tools.extend(GITNEXUS_READ_ONLY_TOOLS)
|
||||
if include_mcp and implementation:
|
||||
|
|
@ -461,7 +466,13 @@ def _persist_parent_event_stream(
|
|||
|
||||
|
||||
def _normalized_skill_identifier(value: Any) -> str | None:
|
||||
"""Return the exact identifier token accepted by the Skill tool."""
|
||||
"""Return the exact identifier token accepted by the Skill tool.
|
||||
|
||||
Plugin skills are requested as ``plugin:skill`` (observed in review
|
||||
evolution: ``compound-engineering:ce-code-review``). Compare on the
|
||||
skill token after the last colon so a successful plugin-qualified
|
||||
invocation still counts as the expected skill.
|
||||
"""
|
||||
|
||||
if not isinstance(value, str):
|
||||
return None
|
||||
|
|
@ -471,6 +482,8 @@ def _normalized_skill_identifier(value: Any) -> str | None:
|
|||
token = stripped.split(maxsplit=1)[0]
|
||||
if token.startswith("/"):
|
||||
token = token[1:]
|
||||
if ":" in token:
|
||||
token = token.rsplit(":", 1)[-1]
|
||||
return token or None
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -146,24 +146,91 @@ def _validated_runtime_root(path: Path, *, label: str) -> Path:
|
|||
return root
|
||||
|
||||
|
||||
def _primary_checkout_root(harness_root: Path) -> Path | None:
|
||||
"""Return the main worktree when *harness_root* is a linked git worktree.
|
||||
|
||||
Linked worktrees store a regular ``.git`` file pointing at
|
||||
``<primary>/.git/worktrees/<name>``. A symlink or oversized file is
|
||||
ignored so this helper cannot be used to follow an unexpected path.
|
||||
"""
|
||||
|
||||
git_file = harness_root / ".git"
|
||||
try:
|
||||
metadata = git_file.lstat()
|
||||
except OSError:
|
||||
return None
|
||||
if stat.S_ISLNK(metadata.st_mode) or not stat.S_ISREG(metadata.st_mode):
|
||||
return None
|
||||
if metadata.st_size > 4096:
|
||||
return None
|
||||
try:
|
||||
text = git_file.read_text(encoding="utf-8").strip()
|
||||
except (OSError, UnicodeError):
|
||||
return None
|
||||
if not text.startswith("gitdir:"):
|
||||
return None
|
||||
raw = text.split(":", 1)[1].strip()
|
||||
if not raw:
|
||||
return None
|
||||
gitdir = Path(raw)
|
||||
if not gitdir.is_absolute():
|
||||
gitdir = harness_root / gitdir
|
||||
if gitdir.parent.name != "worktrees" or gitdir.parent.parent.name != ".git":
|
||||
return None
|
||||
primary = gitdir.parent.parent.parent
|
||||
try:
|
||||
primary_git = (primary / ".git").lstat()
|
||||
except OSError:
|
||||
return None
|
||||
if stat.S_ISLNK(primary_git.st_mode) or not stat.S_ISDIR(primary_git.st_mode):
|
||||
return None
|
||||
resolved_primary = primary.resolve()
|
||||
if resolved_primary == harness_root.resolve():
|
||||
return None
|
||||
return resolved_primary
|
||||
|
||||
|
||||
def _validated_runtime_component(
|
||||
root: Path,
|
||||
relative: str,
|
||||
target: str,
|
||||
*,
|
||||
directory: bool,
|
||||
allow_primary_worktree_symlink: bool = False,
|
||||
) -> ReadOnlyMount:
|
||||
"""Validate one direct runtime component before exposing only that path."""
|
||||
|
||||
source = root / relative
|
||||
kind = "directory" if directory else "file"
|
||||
try:
|
||||
mode = source.lstat().st_mode
|
||||
resolved = source.resolve(strict=True)
|
||||
except OSError as exc:
|
||||
raise SandboxError(f"pinned GitNexus runtime component is unavailable: {source}: {exc}") from exc
|
||||
if stat.S_ISLNK(mode) and allow_primary_worktree_symlink and directory:
|
||||
primary = _primary_checkout_root(HARNESS_ROOT)
|
||||
if primary is None:
|
||||
raise SandboxError(f"pinned GitNexus runtime component must be a real {kind}: {source}")
|
||||
expected = primary / "gitnexus" / relative
|
||||
try:
|
||||
expected_mode = expected.lstat().st_mode
|
||||
expected_resolved = expected.resolve(strict=True)
|
||||
except OSError as exc:
|
||||
raise SandboxError(
|
||||
f"pinned GitNexus runtime component symlink must point at the primary checkout: {source}"
|
||||
) from exc
|
||||
if (
|
||||
stat.S_ISLNK(expected_mode)
|
||||
or not stat.S_ISDIR(expected_mode)
|
||||
or expected_resolved != expected
|
||||
or resolved != expected_resolved
|
||||
):
|
||||
raise SandboxError(
|
||||
f"pinned GitNexus runtime component symlink must point at the primary checkout: {source}"
|
||||
)
|
||||
return ReadOnlyMount(source=expected_resolved, target=target)
|
||||
expected_type = stat.S_ISDIR(mode) if directory else stat.S_ISREG(mode)
|
||||
if stat.S_ISLNK(mode) or not expected_type or resolved != source:
|
||||
kind = "directory" if directory else "file"
|
||||
raise SandboxError(f"pinned GitNexus runtime component must be a real {kind}: {source}")
|
||||
return ReadOnlyMount(source=source, target=target)
|
||||
|
||||
|
|
@ -197,6 +264,7 @@ def trusted_gitnexus_runtime_mounts() -> tuple[ReadOnlyMount, ...]:
|
|||
"node_modules",
|
||||
f"{SANDBOX_GITNEXUS}/node_modules",
|
||||
directory=True,
|
||||
allow_primary_worktree_symlink=True,
|
||||
),
|
||||
_validated_runtime_component(
|
||||
runtime,
|
||||
|
|
@ -236,7 +304,19 @@ def trusted_gitnexus_runtime_mounts() -> tuple[ReadOnlyMount, ...]:
|
|||
raise SandboxError("pinned GitNexus runtime package.json has no version")
|
||||
|
||||
linked_shared = mounts[2].source / "gitnexus-shared"
|
||||
if not linked_shared.is_symlink() or linked_shared.resolve(strict=True) != shared:
|
||||
allowed_shared = {shared}
|
||||
primary = _primary_checkout_root(HARNESS_ROOT)
|
||||
if primary is not None:
|
||||
try:
|
||||
allowed_shared.add(
|
||||
_validated_runtime_root(
|
||||
primary / "gitnexus-shared",
|
||||
label="primary GitNexus shared runtime",
|
||||
)
|
||||
)
|
||||
except SandboxError:
|
||||
pass
|
||||
if not linked_shared.is_symlink() or linked_shared.resolve(strict=True) not in allowed_shared:
|
||||
raise SandboxError("pinned GitNexus runtime has an unexpected gitnexus-shared dependency")
|
||||
try:
|
||||
shared_package = json.loads(mounts[5].source.read_text())
|
||||
|
|
|
|||
|
|
@ -20,11 +20,12 @@ from .proposer_sandbox import (
|
|||
SANDBOX_WORKSPACE,
|
||||
ReadOnlyMount,
|
||||
SandboxError,
|
||||
SandboxSession,
|
||||
build_sandbox_environment,
|
||||
prepare_sandbox,
|
||||
)
|
||||
from .runner_artifacts import make_worktree, remove_clone
|
||||
from .task_assets import TaskAssetCache, TaskAssetSnapshot
|
||||
from .task_assets import TaskAssetCache, TaskAssetSnapshot, _is_harness_sandbox_copy
|
||||
|
||||
GRAPH_ASSET_PATHS = (
|
||||
".gitnexus/gitnexus.json",
|
||||
|
|
@ -41,6 +42,12 @@ GRAPH_MARKERS = (
|
|||
)
|
||||
GRAPH_BUILD_TIMEOUT_SECONDS = 3600
|
||||
GRAPH_QUERY_TIMEOUT_SECONDS = 300
|
||||
# CLI default is 5s. Parse-worker top-of-script init loads every required
|
||||
# tree-sitter binding before it can post `{type:'ready'}`; on a loaded WSL
|
||||
# host that handshake is >15s, and a GitNexus-sized --pdg analyze can slow it
|
||||
# further. Integration tests already stub 60s; graph prep uses 120s so a
|
||||
# replacement worker is not classified as deterministic-startup (#2649).
|
||||
GRAPH_WORKER_READY_TIMEOUT_MS = 120_000
|
||||
MAX_GRAPH_SCRUB_ENTRIES = 250_000
|
||||
MAX_GRAPH_SCRUB_FILE_BYTES = 512 * 1024
|
||||
MAX_GRAPH_SCRUB_TOTAL_BYTES = 2 * 1024 * 1024 * 1024
|
||||
|
|
@ -88,7 +95,13 @@ def validate_no_prebuilt_graph_assets(task: Mapping[str, Any]) -> None:
|
|||
if not isinstance(sandbox_copy, list):
|
||||
raise SandboxError("sandbox_copy must be a list")
|
||||
for value in sandbox_copy:
|
||||
if isinstance(value, str) and _is_restricted_path(value):
|
||||
if not isinstance(value, str):
|
||||
continue
|
||||
# Review corpus patches are harness-owned and applied in setup, then
|
||||
# deleted with eval/workflow_bench. They are not a prebuilt graph.
|
||||
if _is_harness_sandbox_copy(PurePosixPath(value)):
|
||||
continue
|
||||
if _is_restricted_path(value):
|
||||
raise SandboxError(f"sandbox_copy cannot import prebuilt graph or harness data: {value}")
|
||||
|
||||
dependencies = task.get("sandbox_dependencies", [])
|
||||
|
|
@ -235,6 +248,7 @@ def _graph_environment() -> dict[str, str]:
|
|||
"GITNEXUS_NO_GITIGNORE": "1",
|
||||
"GITNEXUS_WORKER_POOL_SIZE": "1",
|
||||
"GITNEXUS_PARSE_CHUNK_CONCURRENCY": "1",
|
||||
"GITNEXUS_WORKER_READY_TIMEOUT_MS": str(GRAPH_WORKER_READY_TIMEOUT_MS),
|
||||
}
|
||||
)
|
||||
return env
|
||||
|
|
@ -244,20 +258,23 @@ def _run_graph_cli(
|
|||
prefix: Sequence[str],
|
||||
arguments: Sequence[str],
|
||||
*,
|
||||
sandbox: SandboxSession | None = None,
|
||||
timeout: int,
|
||||
capture_stdout: bool = False,
|
||||
) -> bytes | None:
|
||||
host_path = sandbox.host_path if sandbox is not None else str
|
||||
host_text = sandbox.host_text if sandbox is not None else str
|
||||
command = [
|
||||
*prefix,
|
||||
SANDBOX_NODE,
|
||||
SANDBOX_GITNEXUS_ENTRYPOINT,
|
||||
*arguments,
|
||||
host_path(SANDBOX_NODE),
|
||||
host_path(SANDBOX_GITNEXUS_ENTRYPOINT),
|
||||
*(host_text(argument) for argument in arguments),
|
||||
]
|
||||
result = run_managed(
|
||||
command,
|
||||
timeout=timeout,
|
||||
env=_graph_environment(),
|
||||
require_pid_namespace=True,
|
||||
env={key: host_text(value) for key, value in _graph_environment().items()},
|
||||
require_pid_namespace=(sandbox.require_pid_namespace if sandbox is not None else True),
|
||||
capture_stdout_bytes=(2 * 1024 * 1024 if capture_stdout else None),
|
||||
)
|
||||
if not result.ok:
|
||||
|
|
@ -286,12 +303,14 @@ def _parse_empty_query(raw: bytes, *, label: str) -> None:
|
|||
raise SandboxError(f"{label} found recoverable benchmark harness references")
|
||||
|
||||
|
||||
def _scrub_and_verify_graph(prefix: Sequence[str]) -> None:
|
||||
def _scrub_and_verify_graph(prefix: Sequence[str], sandbox: SandboxSession | None = None) -> None:
|
||||
node_predicate = _marker_predicate("n")
|
||||
relation_predicate = _marker_predicate("r")
|
||||
sandbox_kwargs = {"sandbox": sandbox} if sandbox is not None else {}
|
||||
node_result = _run_graph_cli(
|
||||
prefix,
|
||||
("cypher", f"MATCH (n) WHERE {node_predicate} RETURN n LIMIT 1", "-r", "benchmark-target", "--limit", "1"),
|
||||
**sandbox_kwargs,
|
||||
timeout=GRAPH_QUERY_TIMEOUT_SECONDS,
|
||||
capture_stdout=True,
|
||||
)
|
||||
|
|
@ -305,6 +324,7 @@ def _scrub_and_verify_graph(prefix: Sequence[str]) -> None:
|
|||
"--limit",
|
||||
"1",
|
||||
),
|
||||
**sandbox_kwargs,
|
||||
timeout=GRAPH_QUERY_TIMEOUT_SECONDS,
|
||||
capture_stdout=True,
|
||||
)
|
||||
|
|
@ -339,6 +359,7 @@ def prepare_sanitized_graph(
|
|||
claude_bin: Path | str,
|
||||
bwrap_bin: Path | str,
|
||||
runtime_mounts: Sequence[ReadOnlyMount],
|
||||
sandbox_backend: str = "bwrap",
|
||||
) -> SanitizedGraphSnapshot:
|
||||
"""Sanitize, index offline once, scrub, and freeze graph assets for all arms."""
|
||||
|
||||
|
|
@ -355,8 +376,10 @@ def prepare_sanitized_graph(
|
|||
bwrap_bin=bwrap_bin,
|
||||
read_only_mounts=runtime_mounts,
|
||||
preflight=False,
|
||||
backend=sandbox_backend,
|
||||
) as sandbox:
|
||||
prefix = sandbox.command_prefix_for(unshare_network=True)
|
||||
unsafe_sandbox = sandbox if getattr(sandbox, "backend", "bwrap") == "host-unsafe" else None
|
||||
_run_graph_cli(
|
||||
prefix,
|
||||
(
|
||||
|
|
@ -375,9 +398,10 @@ def prepare_sanitized_graph(
|
|||
"--workers",
|
||||
"1",
|
||||
),
|
||||
**({"sandbox": unsafe_sandbox} if unsafe_sandbox is not None else {}),
|
||||
timeout=GRAPH_BUILD_TIMEOUT_SECONDS,
|
||||
)
|
||||
_scrub_and_verify_graph(prefix)
|
||||
_scrub_and_verify_graph(prefix, unsafe_sandbox)
|
||||
_validate_graph_metadata(seed, sanitized_head)
|
||||
assets = cache.prepare(
|
||||
{"sandbox_copy": list(GRAPH_ASSET_PATHS)},
|
||||
|
|
|
|||
|
|
@ -24,6 +24,7 @@ from dataclasses import dataclass
|
|||
from pathlib import Path, PurePosixPath
|
||||
from typing import Any
|
||||
|
||||
from . import runtime_mounts
|
||||
from .proposer_sandbox import (
|
||||
DEPENDENCY_MOUNT_BASENAME,
|
||||
SANDBOX_WORKSPACE,
|
||||
|
|
@ -63,6 +64,10 @@ _REFLINK_UNAVAILABLE = {
|
|||
|
||||
DEPENDENCY_CONTENT_BINDING_FIELD = "sandbox_dependency_content_digest"
|
||||
DEPENDENCY_MANIFEST_BINDING_FIELD = "sandbox_dependency_manifest_digest"
|
||||
# Review corpus patches are harness-owned. The task repo (`~/GitNexus`) is the
|
||||
# subject checkout and often a different worktree or branch, so these paths
|
||||
# must be read from the package that defined the benchmark.
|
||||
HARNESS_SANDBOX_COPY_PREFIXES = (PurePosixPath("eval/workflow_bench/review_cases"),)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
|
|
@ -191,7 +196,7 @@ class TaskAssetCache:
|
|||
self.root = root.expanduser().absolute()
|
||||
self.root.mkdir(mode=0o700, parents=True, exist_ok=False)
|
||||
self._by_definition: dict[
|
||||
tuple[str, str, tuple[str, ...], tuple[tuple[str, str], ...]],
|
||||
tuple[str, str, tuple[str, ...], tuple[tuple[str, str], ...], str],
|
||||
TaskAssetSnapshot,
|
||||
] = {}
|
||||
self._closed = False
|
||||
|
|
@ -218,7 +223,14 @@ class TaskAssetCache:
|
|||
declarations, relative_paths = _sandbox_copy_declarations(task)
|
||||
dependency_declarations = _sandbox_dependency_declarations(task)
|
||||
dependency_identity = tuple((declaration.source, declaration.target) for declaration in dependency_declarations)
|
||||
definition = (str(repo_identity), resolved_sha, declarations, dependency_identity)
|
||||
harness_identity = _harness_sandbox_copy_identity(relative_paths)
|
||||
definition = (
|
||||
str(repo_identity),
|
||||
resolved_sha,
|
||||
declarations,
|
||||
dependency_identity,
|
||||
harness_identity,
|
||||
)
|
||||
existing = self._by_definition.get(definition)
|
||||
if existing is not None:
|
||||
if expected_dependency_binding is not None:
|
||||
|
|
@ -238,9 +250,22 @@ class TaskAssetCache:
|
|||
repo_identity,
|
||||
os.O_RDONLY | os.O_DIRECTORY | getattr(os, "O_CLOEXEC", 0) | getattr(os, "O_NOFOLLOW", 0),
|
||||
)
|
||||
harness_fd: int | None = None
|
||||
try:
|
||||
for relative in relative_paths:
|
||||
descriptor = _open_relative(repo_fd, relative)
|
||||
if _is_harness_sandbox_copy(relative):
|
||||
if harness_fd is None:
|
||||
harness_root = _harness_sandbox_copy_root()
|
||||
harness_fd = os.open(
|
||||
harness_root,
|
||||
os.O_RDONLY
|
||||
| os.O_DIRECTORY
|
||||
| getattr(os, "O_CLOEXEC", 0)
|
||||
| getattr(os, "O_NOFOLLOW", 0),
|
||||
)
|
||||
descriptor = _open_relative(harness_fd, relative)
|
||||
else:
|
||||
descriptor = _open_relative(repo_fd, relative)
|
||||
try:
|
||||
builder.copy_descriptor(descriptor, relative)
|
||||
finally:
|
||||
|
|
@ -299,6 +324,8 @@ class TaskAssetCache:
|
|||
)
|
||||
finally:
|
||||
os.close(repo_fd)
|
||||
if harness_fd is not None:
|
||||
os.close(harness_fd)
|
||||
|
||||
entries = builder.finished_entries()
|
||||
manifest_digest = _manifest_digest(entries)
|
||||
|
|
@ -317,6 +344,7 @@ class TaskAssetCache:
|
|||
manifest_digest=manifest_digest,
|
||||
dependency_content_digest=dependency_content_digest,
|
||||
dependency_manifest_digest=dependency_manifest_digest,
|
||||
harness_identity=harness_identity,
|
||||
)
|
||||
destination = self.root / digest
|
||||
if destination.exists():
|
||||
|
|
@ -537,6 +565,22 @@ class _SnapshotBuilder:
|
|||
return tuple(sorted(self.entries.values(), key=lambda entry: entry.path.as_posix()))
|
||||
|
||||
|
||||
def _is_harness_sandbox_copy(relative: PurePosixPath) -> bool:
|
||||
if relative.is_absolute() or not relative.parts or ".." in relative.parts:
|
||||
return False
|
||||
return any(relative == prefix or prefix in relative.parents for prefix in HARNESS_SANDBOX_COPY_PREFIXES)
|
||||
|
||||
|
||||
def _harness_sandbox_copy_root() -> Path:
|
||||
return _real_directory(runtime_mounts.HARNESS_ROOT, label="harness sandbox_copy root")
|
||||
|
||||
|
||||
def _harness_sandbox_copy_identity(relative_paths: tuple[PurePosixPath, ...]) -> str:
|
||||
if not any(_is_harness_sandbox_copy(relative) for relative in relative_paths):
|
||||
return ""
|
||||
return str(_harness_sandbox_copy_root())
|
||||
|
||||
|
||||
def _sandbox_copy_declarations(
|
||||
task: Mapping[str, Any],
|
||||
) -> tuple[tuple[str, ...], tuple[PurePosixPath, ...]]:
|
||||
|
|
@ -969,15 +1013,17 @@ def _snapshot_digest(
|
|||
manifest_digest: str,
|
||||
dependency_content_digest: str,
|
||||
dependency_manifest_digest: str,
|
||||
harness_identity: str = "",
|
||||
) -> str:
|
||||
payload = {
|
||||
"declarations": declarations,
|
||||
"dependency_content_digest": dependency_content_digest,
|
||||
"dependency_manifest_digest": dependency_manifest_digest,
|
||||
"harness_identity": harness_identity,
|
||||
"manifest_digest": manifest_digest,
|
||||
"repo_identity": str(repo_identity),
|
||||
"resolved_sha": resolved_sha,
|
||||
"schema_version": 2,
|
||||
"schema_version": 3,
|
||||
}
|
||||
return hashlib.sha256(json.dumps(payload, sort_keys=True, separators=(",", ":")).encode()).hexdigest()
|
||||
|
||||
|
|
|
|||
|
|
@ -7,7 +7,7 @@ tasks:
|
|||
repo: ~/GitNexus
|
||||
ref: ff86ccf1e79cd7e4175da437ae8aeaf67b64aaa1
|
||||
sandbox_copy: [eval/workflow_bench/review_cases/pr-2718-defect.patch]
|
||||
setup: git apply eval/workflow_bench/review_cases/pr-2718-defect.patch && rm -rf eval/workflow_bench
|
||||
setup: git apply --exclude='eval/workflow_bench/*' eval/workflow_bench/review_cases/pr-2718-defect.patch && rm -rf eval/workflow_bench
|
||||
prompt: Review the historical snapshot of https://github.com/abhigyanpatwari/GitNexus/pull/2718. Report only actionable defects introduced by the local diff.
|
||||
verify: test -s review-output.json
|
||||
oracle:
|
||||
|
|
@ -22,7 +22,7 @@ tasks:
|
|||
id: review-pr-2794-defect
|
||||
ref: 911151e2304f298a995fcc69c738ad2c6db9393a
|
||||
sandbox_copy: [eval/workflow_bench/review_cases/pr-2794-defect.patch]
|
||||
setup: git apply eval/workflow_bench/review_cases/pr-2794-defect.patch && rm -rf eval/workflow_bench
|
||||
setup: git apply --exclude='eval/workflow_bench/*' eval/workflow_bench/review_cases/pr-2794-defect.patch && rm -rf eval/workflow_bench
|
||||
prompt: Review the historical snapshot of https://github.com/abhigyanpatwari/GitNexus/pull/2794. Report only actionable defects introduced by the local diff.
|
||||
oracle:
|
||||
command: test -s review-output.json
|
||||
|
|
@ -33,7 +33,7 @@ tasks:
|
|||
id: review-pr-2108-defect
|
||||
ref: 3a4247ec36b5ad86b1123d3bbce8183a643f7434
|
||||
sandbox_copy: [eval/workflow_bench/review_cases/pr-2108-defect.patch]
|
||||
setup: git apply eval/workflow_bench/review_cases/pr-2108-defect.patch && rm -rf eval/workflow_bench
|
||||
setup: git apply --exclude='eval/workflow_bench/*' eval/workflow_bench/review_cases/pr-2108-defect.patch && rm -rf eval/workflow_bench
|
||||
prompt: Review the historical snapshot of https://github.com/abhigyanpatwari/GitNexus/pull/2108. Report only actionable defects introduced by the local diff.
|
||||
oracle:
|
||||
command: test -s review-output.json
|
||||
|
|
@ -44,7 +44,7 @@ tasks:
|
|||
id: review-pr-2258-defect
|
||||
ref: 78b4077d8acc86f1b0c32e41012174d484e81f12
|
||||
sandbox_copy: [eval/workflow_bench/review_cases/pr-2258-defect.patch]
|
||||
setup: git apply eval/workflow_bench/review_cases/pr-2258-defect.patch && rm -rf eval/workflow_bench
|
||||
setup: git apply --exclude='eval/workflow_bench/*' eval/workflow_bench/review_cases/pr-2258-defect.patch && rm -rf eval/workflow_bench
|
||||
prompt: Review the historical snapshot of https://github.com/abhigyanpatwari/GitNexus/pull/2258. Report only actionable defects introduced by the local diff.
|
||||
oracle:
|
||||
command: test -s review-output.json
|
||||
|
|
@ -56,7 +56,7 @@ tasks:
|
|||
class: review-clean
|
||||
ref: 78b4077d8acc86f1b0c32e41012174d484e81f12
|
||||
sandbox_copy: [eval/workflow_bench/review_cases/pr-2258-clean.patch]
|
||||
setup: git apply eval/workflow_bench/review_cases/pr-2258-clean.patch && rm -rf eval/workflow_bench
|
||||
setup: git apply --exclude='eval/workflow_bench/*' eval/workflow_bench/review_cases/pr-2258-clean.patch && rm -rf eval/workflow_bench
|
||||
prompt: Review this historical snapshot of https://github.com/abhigyanpatwari/GitNexus/pull/2258. Report only actionable defects introduced by the local diff.
|
||||
oracle:
|
||||
command: test -s review-output.json
|
||||
|
|
@ -68,7 +68,7 @@ tasks:
|
|||
class: review-clean
|
||||
ref: 84f584449de02376a8ffc096dceac2e8f732cab5
|
||||
sandbox_copy: [eval/workflow_bench/review_cases/pr-2773-clean.patch]
|
||||
setup: git apply eval/workflow_bench/review_cases/pr-2773-clean.patch && rm -rf eval/workflow_bench
|
||||
setup: git apply --exclude='eval/workflow_bench/*' eval/workflow_bench/review_cases/pr-2773-clean.patch && rm -rf eval/workflow_bench
|
||||
prompt: Review this historical snapshot of https://github.com/abhigyanpatwari/GitNexus/pull/2773. Report only actionable defects introduced by the local diff.
|
||||
oracle:
|
||||
command: test -s review-output.json
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue