mirror of
https://github.com/abhigyanpatwari/GitNexus.git
synced 2026-09-20 00:11:37 +00:00
Code review of this branch found the relocated review artifact was fixed in the bwrap mount and nowhere else. Four independent layers decide whether the agent can write it, and three still named the old location. Claude Code applies its own filesystem policy to its own tools, and build_claude_settings listed only /workspace, /tmp and /home/agent under allowWrite with denyRead ["/"]. The artifact used to live under /workspace, so this list was correct until it moved. SANDBOX_REVIEW_OUTPUT is now in allowWrite and allowRead; without it the bwrap bind grants a write the CLI then refuses. The task corpus still ran `test -s review-output.json` from the workspace, in a separate sandbox invocation that never sees the artifact mount. Every review cell would have been stamped verify-failed with resolved=False no matter how good the review was, which also made those rows permanently unreusable and so silently disabled this branch's own comparator reuse for review arms. The verify and hidden-oracle commands now read the location from GITNEXUS_BENCH_REVIEW_OUTPUT and get the directory bound read-only, mirroring the mount-plus-env-var shape _run_hidden_oracle already used. host_text and host_path did not translate the new path, so the host-unsafe backend told the agent to write somewhere that exists on neither backend. Adding the mapping exposed a second defect: host_text substituted every occurrence of a target, and "/review-output" appears twice in "/review-output/review-output.json" - once as the directory and once inside the filename. Matching is now anchored to a path boundary. Comparator reuse had three ways to accept evidence it should have rejected. A row with no runtime_digest passed the drift lock because the guard only compared when both sides were bound, and the branch's own test asserted that as correct; absence is now a mismatch and the test states the rule. materialize_reused_row overwrote recorded_at with the copy time while the age check read that field, so a row copied forward each generation refreshed its own clock and never aged out; the first measurement time is now preserved and aged against. A future-dated stamp passed a one-sided bound and is now rejected as corrupt. RUNTIME_DIGEST never reached the runner at all: runner_environment builds a fixed dict and process_control replaces the child environment wholesale, so the digest the workflow exports was dropped and the lock it feeds was inert. The instance-window deadline was also checked only after run_proposer returned, buying a proposal the generation had no room to benchmark. The graph prefetch thread was started without copy_context, so it never saw the cancellation ContextVar the rest of the sweep shares, and the outage breaker returned without setting cancel_event - together, a tripped breaker would block on joining a prefetch that was never told to stop. Both fixed, with outage checked before cancellation at the two exits so an outage keeps exit 1 instead of becoming a Ctrl-C's 130. Both bwrap canaries that actually execute a write still bound the pre-fix shape against a file this branch no longer creates, so they would have errored rather than caught anything. They now bind the directory and write atomically - temp file beside the target, then rename - which is the exact operation that failed with EROFS. A source-text assertion over inspect.getsource(run_arm) was replaced with one that inspects the real mount, and a wall-clock assertion was pinned to a fixed monotonic clock. Not applied, and why: binding task-asset and dependency digests into comparator reuse needs asset snapshots prepared before the reuse decision rather than inside the per-task loop, and shipping the comparison without that would add a guard that silently never fires. Forcing a paid canary cell per incumbent arm and folding reused rows into the outage streak are behaviour decisions, not fixes. Clone-template reuse still has no test. The cost model's per-cell residual still shrinks with arm count, overstating weekly savings by at most the 2541s residual; the docstring now says so rather than inventing a split. 585 eval tests pass, ruff clean, 29 workflow contract tests pass. The two test_model_gateway.py failures are pre-existing and fail on main.
1494 lines
52 KiB
Python
1494 lines
52 KiB
Python
"""Unit tests for the pure evidence/apply/driver helpers of workflow_bench.evolve."""
|
|
|
|
import hashlib
|
|
import json
|
|
import os
|
|
import subprocess
|
|
import sys
|
|
import time
|
|
from contextlib import contextmanager
|
|
from datetime import UTC, datetime, timedelta
|
|
from pathlib import Path
|
|
|
|
import pytest
|
|
|
|
from workflow_bench import evolve, evolution
|
|
from workflow_bench.runner_sessions import PARENT_EVENT_STREAM_SOURCE
|
|
from workflow_bench.evolve import (
|
|
MIN_INSTANCE_SWEEP_SECONDS,
|
|
build_parser,
|
|
build_proposer_prompt,
|
|
capped_timeout_seconds,
|
|
executed_benchmark_arms,
|
|
generation_timeout_seconds,
|
|
instance_window_budget_from_proc,
|
|
instance_window_budget_seconds,
|
|
load_jsonl,
|
|
proposer_evidence_entries,
|
|
read_learnings,
|
|
remaining_runtime_seconds,
|
|
resolve_incumbent_arms,
|
|
runner_argv,
|
|
select_evidence,
|
|
summarize_gate,
|
|
validate_promotion_for_apply,
|
|
)
|
|
from workflow_bench.process_control import ManagedProcessResult, run_managed
|
|
from workflow_bench.proposer_sandbox import pid_namespace_command, preflight_bubblewrap
|
|
|
|
|
|
def test_runner_environment_does_not_forward_the_openai_key() -> None:
|
|
from workflow_bench.model_gateway import credential_secrets
|
|
|
|
args = build_parser().parse_args(
|
|
[
|
|
"--tasks",
|
|
"t.yaml",
|
|
"--model",
|
|
"gpt-4.1",
|
|
"--anthropic-api-key",
|
|
"loopback-master",
|
|
"--openai-api-key",
|
|
"sk-openai-secret",
|
|
]
|
|
)
|
|
env = evolve.runner_environment(args)
|
|
assert env["GITNEXUS_BENCH_ANTHROPIC_API_KEY"] == "loopback-master"
|
|
assert "GITNEXUS_BENCH_AUTH_TOKEN" not in env
|
|
assert "OPENAI_API_KEY" not in env
|
|
assert "GITNEXUS_BENCH_OPENAI_API_KEY" not in env
|
|
assert "sk-openai-secret" not in env.values()
|
|
assert credential_secrets(args) == ["loopback-master", "sk-openai-secret"]
|
|
|
|
|
|
def test_parser_keeps_the_legacy_auth_token_alias() -> None:
|
|
args = build_parser().parse_args(["--tasks", "t.yaml", "--model", "pinned", "--auth-token", "alias-secret"])
|
|
assert args.auth_token == "alias-secret"
|
|
|
|
|
|
def row(**overrides):
|
|
base = {
|
|
"task": "demo-task",
|
|
"class": "trivial",
|
|
"arm": "workflow",
|
|
"run": 0,
|
|
"resolved": True,
|
|
"error_kind": None,
|
|
"cost_usd": 1.0,
|
|
"num_turns": 10,
|
|
"output_tokens": 400,
|
|
"session_ids": ["sess-1"],
|
|
"verify_output": "ok",
|
|
}
|
|
base.update(overrides)
|
|
return base
|
|
|
|
|
|
def test_select_evidence_puts_unresolved_before_expensive_resolved():
|
|
rows = [
|
|
row(task="cheap", resolved=True, cost_usd=0.5),
|
|
row(task="fail", resolved=False, error_kind="verify-failed"),
|
|
row(task="pricey", resolved=True, cost_usd=9.0),
|
|
]
|
|
picked = select_evidence(rows)
|
|
assert [r["task"] for r in picked] == ["fail", "pricey", "cheap"]
|
|
|
|
|
|
def test_select_evidence_excludes_infra_error_rows_and_caps():
|
|
rows = [
|
|
row(task="harness-died", resolved=False, error_kind="infra-error"),
|
|
row(task="session-died", resolved=False, error_kind="session-error"),
|
|
row(
|
|
task="missing-transcript",
|
|
resolved=False,
|
|
error_kind="evidence-unverified",
|
|
),
|
|
]
|
|
rows += [row(task=f"t{i}", cost_usd=float(i)) for i in range(20)]
|
|
picked = select_evidence(rows, max_rows=5)
|
|
assert len(picked) == 5
|
|
assert all(r["error_kind"] != "infra-error" for r in picked)
|
|
assert [r["task"] for r in picked] == ["t19", "t18", "t17", "t16", "t15"]
|
|
|
|
|
|
def test_select_evidence_tolerates_an_explicit_null_cost():
|
|
# A foreign --seed-results row (e.g. hand-edited or from another tool)
|
|
# can carry an explicit JSON null rather than omitting the key; .get's
|
|
# default only covers the missing-key case, so this must not raise.
|
|
rows = [row(task="no-cost", resolved=True, cost_usd=None), row(task="priced", cost_usd=5.0)]
|
|
picked = select_evidence(rows)
|
|
assert [r["task"] for r in picked] == ["priced", "no-cost"]
|
|
|
|
|
|
def test_load_jsonl_skips_blank_and_malformed_lines(tmp_path):
|
|
path = tmp_path / "learnings.jsonl"
|
|
path.write_text('{"skill": "gitnexus-plan"}\n\nnot json\n[1, 2]\n{"skill": "gitnexus-work"}\n')
|
|
assert load_jsonl(path) == [{"skill": "gitnexus-plan"}, {"skill": "gitnexus-work"}]
|
|
|
|
|
|
def test_load_jsonl_missing_file_is_empty(tmp_path):
|
|
assert load_jsonl(tmp_path / "absent.jsonl") == []
|
|
|
|
|
|
def test_read_learnings_keeps_the_most_recent_entries(tmp_path):
|
|
path = tmp_path / "learnings.jsonl"
|
|
rows = [{"skill": "gitnexus-work", "n": i} for i in range(10)] + [
|
|
{"skill": "gitnexus-review", "n": 10},
|
|
{"skill": "gitnexus-lfg", "n": 11},
|
|
]
|
|
path.write_text("\n".join(json.dumps(row) for row in rows) + "\n")
|
|
assert read_learnings(path, cap=3) == [
|
|
{"skill": "gitnexus-work", "n": 8},
|
|
{"skill": "gitnexus-work", "n": 9},
|
|
{"skill": "gitnexus-review", "n": 10},
|
|
]
|
|
|
|
|
|
def test_summarize_gate_one_line_per_decision():
|
|
promotion = {
|
|
"decisions": [
|
|
{
|
|
"candidate_arm": "candidate_workflow",
|
|
"decision": "keep_incumbent",
|
|
"reasons": ["a", "b", "c", "d"],
|
|
}
|
|
]
|
|
}
|
|
lines = summarize_gate(promotion)
|
|
assert lines == ["candidate_workflow: keep_incumbent — a; b; c"]
|
|
|
|
|
|
def test_build_proposer_prompt_carries_evidence_constraints_and_paths(tmp_path):
|
|
prompt = build_proposer_prompt(
|
|
results_dir=tmp_path / "bench",
|
|
evidence=[row(task="fail", resolved=False, error_kind="verify-failed")],
|
|
learnings=[{"skill": "gitnexus-work", "friction": "budget blown on reruns"}],
|
|
gate_summary=["candidate_workflow: keep_incumbent — quality regressed"],
|
|
overlay_dir=tmp_path / "overlay",
|
|
proposal_path=tmp_path / "proposal.md",
|
|
incumbent_arms=["workflow"],
|
|
)
|
|
assert str(tmp_path / "overlay") in prompt
|
|
assert str(tmp_path / "proposal.md") in prompt
|
|
assert "gitnexus-plan, gitnexus-work" in prompt
|
|
assert "node .gitnexus/run.cjs analyze" in prompt
|
|
assert "1 row(s) in /evidence/learnings.json" in prompt
|
|
assert "1 selected row(s) in /evidence/selected-rows.json" in prompt
|
|
assert "exact staged" in prompt
|
|
assert "no full results.jsonl" in prompt
|
|
assert "1 decision(s) in /evidence/gate-summary.json" in prompt
|
|
assert "budget blown on reruns" not in prompt
|
|
assert "verify-failed" not in prompt
|
|
assert "~/.claude/projects" not in prompt
|
|
|
|
|
|
def test_build_proposer_prompt_points_at_the_rejected_prior_proposal(tmp_path):
|
|
common = {
|
|
"results_dir": tmp_path / "bench",
|
|
"evidence": [],
|
|
"learnings": [],
|
|
"gate_summary": ["candidate_workflow: keep_incumbent — cost regressed"],
|
|
"overlay_dir": tmp_path / "overlay",
|
|
"proposal_path": tmp_path / "proposal.md",
|
|
"incumbent_arms": ["workflow"],
|
|
}
|
|
# The gate summary alone says a candidate lost, never what it proposed —
|
|
# so without this line the proposer can re-propose the same prose forever.
|
|
assert "/evidence/prior-proposal.md" in build_proposer_prompt(**common, prior_proposal=True)
|
|
assert "/evidence/prior-proposal.md" not in build_proposer_prompt(**common)
|
|
|
|
|
|
def test_build_proposer_prompt_first_generation_has_no_results_dir(tmp_path):
|
|
prompt = build_proposer_prompt(
|
|
results_dir=None,
|
|
evidence=[],
|
|
learnings=[],
|
|
gate_summary=[],
|
|
overlay_dir=tmp_path / "overlay",
|
|
proposal_path=tmp_path / "proposal.md",
|
|
incumbent_arms=["workflow_direct"],
|
|
)
|
|
assert "none (first generation)" in prompt
|
|
assert "none yet — use the incumbent skills and staged learning queue" in prompt
|
|
|
|
|
|
def test_proposer_reads_only_digest_bound_transcripts_below_results(tmp_path, monkeypatch):
|
|
results = tmp_path / "results"
|
|
transcripts = results / "transcripts"
|
|
transcripts.mkdir(parents=True, mode=0o700)
|
|
transcripts.chmod(0o700)
|
|
payload = b'{"message":{"content":[{"type":"text","text":"bound transcript"}]}}\n'
|
|
artifact = transcripts / "task-workflow-run0-session.jsonl"
|
|
artifact.write_bytes(payload)
|
|
artifact.chmod(0o600)
|
|
patch = results / "demo-task-workflow-run0.patch"
|
|
patch.write_text("diff --git a/a b/a\n")
|
|
metadata = {
|
|
"path": "transcripts/task-workflow-run0-session.jsonl",
|
|
"sha256": hashlib.sha256(payload).hexdigest(),
|
|
"bytes": len(payload),
|
|
"source": PARENT_EVENT_STREAM_SOURCE,
|
|
}
|
|
foreign_home = tmp_path / "foreign-home"
|
|
foreign = foreign_home / ".claude" / "projects" / "other" / "private.jsonl"
|
|
foreign.parent.mkdir(parents=True)
|
|
foreign.write_text("foreign host transcript")
|
|
monkeypatch.setenv("HOME", str(foreign_home))
|
|
|
|
entries = proposer_evidence_entries(
|
|
results_dir=results,
|
|
evidence=[row(session_ids=["**/*"], transcript_artifacts=[metadata])],
|
|
learnings=[],
|
|
gate_summary=[],
|
|
)
|
|
|
|
assert [json.loads(line) for line in entries["transcript-0-0.jsonl"].splitlines()] == [json.loads(payload)]
|
|
assert entries["patch-0.diff"] == patch.read_text()
|
|
staged_rows = entries["selected-rows.json"]
|
|
assert staged_rows[0]["patch_file"] == "patch-0.diff"
|
|
assert staged_rows[0]["transcript_files"] == ["transcript-0-0.jsonl"]
|
|
assert "foreign host transcript" not in json.dumps(entries)
|
|
|
|
bad_digest = {**metadata, "sha256": "0" * 64}
|
|
with pytest.raises(evolve.SandboxError, match="digest does not match"):
|
|
proposer_evidence_entries(
|
|
results_dir=results,
|
|
evidence=[row(transcript_artifacts=[bad_digest])],
|
|
learnings=[],
|
|
gate_summary=[],
|
|
)
|
|
|
|
|
|
def test_proposer_compacts_transcripts_as_complete_json_events(tmp_path):
|
|
results = tmp_path / "results"
|
|
transcripts = results / "transcripts"
|
|
transcripts.mkdir(parents=True, mode=0o700)
|
|
events = [
|
|
{
|
|
"type": "assistant",
|
|
"message": {
|
|
"content": [
|
|
{
|
|
"type": "thinking",
|
|
"thinking": "analysis-" + ("x" * 100_000),
|
|
"signature": "opaque-base64-signature",
|
|
}
|
|
]
|
|
},
|
|
},
|
|
{
|
|
"type": "result",
|
|
"session_id": "session-1",
|
|
"usage": {"input_tokens": 1, "output_tokens": 2},
|
|
},
|
|
]
|
|
payload = "".join(json.dumps(event) + "\n" for event in events).encode()
|
|
artifact = transcripts / "session.jsonl"
|
|
artifact.write_bytes(payload)
|
|
artifact.chmod(0o600)
|
|
|
|
entries = proposer_evidence_entries(
|
|
results_dir=results,
|
|
evidence=[
|
|
row(
|
|
transcript_artifacts=[
|
|
{
|
|
"path": "transcripts/session.jsonl",
|
|
"sha256": hashlib.sha256(payload).hexdigest(),
|
|
"bytes": len(payload),
|
|
"source": PARENT_EVENT_STREAM_SOURCE,
|
|
}
|
|
]
|
|
)
|
|
],
|
|
learnings=[],
|
|
gate_summary=[],
|
|
artifact_limit=8192,
|
|
)
|
|
|
|
staged = entries["transcript-0-0.jsonl"]
|
|
parsed = [json.loads(line) for line in staged.splitlines()]
|
|
assert len(staged.encode()) <= 8192
|
|
assert parsed[-1]["type"] == "result"
|
|
assert parsed[0]["message"]["content"][0]["signature"] == "[OMITTED]"
|
|
assert "opaque-base64-signature" not in staged
|
|
|
|
|
|
@pytest.mark.skipif(os.name == "nt", reason="transcript symlink containment is POSIX-only")
|
|
def test_proposer_rejects_symlink_and_foreign_transcript_artifacts(tmp_path):
|
|
results = tmp_path / "results"
|
|
transcripts = results / "transcripts"
|
|
transcripts.mkdir(parents=True, mode=0o700)
|
|
transcripts.chmod(0o700)
|
|
outside = tmp_path / "outside.jsonl"
|
|
outside.write_text("outside")
|
|
linked = transcripts / "linked.jsonl"
|
|
linked.symlink_to(outside)
|
|
link_metadata = {
|
|
"path": "transcripts/linked.jsonl",
|
|
"sha256": hashlib.sha256(outside.read_bytes()).hexdigest(),
|
|
"bytes": outside.stat().st_size,
|
|
"source": PARENT_EVENT_STREAM_SOURCE,
|
|
}
|
|
|
|
with pytest.raises(evolve.SandboxError, match="regular non-symlink"):
|
|
proposer_evidence_entries(
|
|
results_dir=results,
|
|
evidence=[row(transcript_artifacts=[link_metadata])],
|
|
learnings=[],
|
|
gate_summary=[],
|
|
)
|
|
with pytest.raises(evolve.SandboxError, match="unsafe results artifact path"):
|
|
proposer_evidence_entries(
|
|
results_dir=results,
|
|
evidence=[
|
|
row(
|
|
transcript_artifacts=[
|
|
{
|
|
"path": "../outside.jsonl",
|
|
"sha256": "0" * 64,
|
|
"bytes": 0,
|
|
"source": PARENT_EVENT_STREAM_SOURCE,
|
|
}
|
|
]
|
|
)
|
|
],
|
|
learnings=[],
|
|
gate_summary=[],
|
|
)
|
|
|
|
|
|
def test_proposer_rejects_duplicate_transcript_metadata_before_materializing():
|
|
metadata = {
|
|
"path": "transcripts/repeated.jsonl",
|
|
"sha256": "0" * 64,
|
|
"bytes": 0,
|
|
"source": PARENT_EVENT_STREAM_SOURCE,
|
|
}
|
|
|
|
with pytest.raises(evolve.SandboxError, match="duplicate transcript artifact path"):
|
|
proposer_evidence_entries(
|
|
results_dir=None,
|
|
evidence=[
|
|
row(
|
|
transcript_artifacts=[
|
|
metadata,
|
|
{**metadata, "path": "transcripts//repeated.jsonl"},
|
|
]
|
|
)
|
|
],
|
|
learnings=[],
|
|
gate_summary=[],
|
|
)
|
|
|
|
|
|
def test_proposer_bounds_transcript_metadata_per_row_and_globally_before_materializing():
|
|
def metadata(index):
|
|
return {
|
|
"path": f"transcripts/session-{index}.jsonl",
|
|
"sha256": "0" * 64,
|
|
"bytes": 0,
|
|
"source": PARENT_EVENT_STREAM_SOURCE,
|
|
}
|
|
|
|
with pytest.raises(evolve.SandboxError, match="per-row session limit"):
|
|
proposer_evidence_entries(
|
|
results_dir=None,
|
|
evidence=[row(transcript_artifacts=[metadata(index) for index in range(3)])],
|
|
learnings=[],
|
|
gate_summary=[],
|
|
)
|
|
|
|
rows = [
|
|
row(
|
|
run=index,
|
|
transcript_artifacts=[metadata(2 * index), metadata(2 * index + 1)],
|
|
)
|
|
for index in range(evolve.MAX_EVIDENCE_ROWS + 1)
|
|
]
|
|
with pytest.raises(evolve.SandboxError, match="global evidence limit"):
|
|
proposer_evidence_entries(
|
|
results_dir=None,
|
|
evidence=rows,
|
|
learnings=[],
|
|
gate_summary=[],
|
|
)
|
|
|
|
|
|
def test_proposer_refuses_a_selected_row_with_no_transcript_reference():
|
|
# Every selectable row comes from sum_sessions(), which always emits the
|
|
# key, and select_evidence() drops the kinds a failed transcript
|
|
# persistence produces (session-error, infra-error, evidence-unverified,
|
|
# cleanup-failure). A selected row without a transcript is therefore
|
|
# evidence lost between producer and proposer, not a row that had none.
|
|
with pytest.raises(evolve.SandboxError, match="missing transcript_artifacts"):
|
|
proposer_evidence_entries(
|
|
results_dir=None,
|
|
evidence=[row()],
|
|
learnings=[],
|
|
gate_summary=[],
|
|
)
|
|
with pytest.raises(evolve.SandboxError, match="carries no transcript artifact"):
|
|
proposer_evidence_entries(
|
|
results_dir=None,
|
|
evidence=[row(transcript_artifacts=[])],
|
|
learnings=[],
|
|
gate_summary=[],
|
|
)
|
|
|
|
|
|
def test_proposer_stages_the_bounded_prior_proposal(tmp_path):
|
|
proposal = tmp_path / "proposal.md"
|
|
proposal.write_text("# rejected candidate\n\nTightened the plan budget.\n")
|
|
proposal.chmod(0o600)
|
|
|
|
entries = proposer_evidence_entries(
|
|
results_dir=None,
|
|
evidence=[],
|
|
learnings=[],
|
|
gate_summary=[],
|
|
prior_proposal=proposal,
|
|
)
|
|
|
|
assert entries["prior-proposal.md"] == proposal.read_text()
|
|
|
|
oversized = tmp_path / "oversized.md"
|
|
oversized.write_bytes(b"x" * (evolve.MAX_EVIDENCE_FILE_BYTES + 4096))
|
|
oversized.chmod(0o600)
|
|
bounded = proposer_evidence_entries(
|
|
results_dir=None,
|
|
evidence=[],
|
|
learnings=[],
|
|
gate_summary=[],
|
|
prior_proposal=oversized,
|
|
)
|
|
assert len(bounded["prior-proposal.md"]) == evolve.MAX_EVIDENCE_FILE_BYTES
|
|
|
|
|
|
def test_stage_proposer_evidence_bundle_drops_prior_proposal_to_fit_budget(tmp_path, monkeypatch, capsys):
|
|
# Per-file caps alone can still exceed the aggregate budget; the helper must
|
|
# drop the prior proposal instead of aborting the generation.
|
|
monkeypatch.setattr(evolve, "MAX_BUNDLE_BYTES", 2048)
|
|
monkeypatch.setattr("workflow_bench.proposer_sandbox.MAX_BUNDLE_BYTES", 2048)
|
|
|
|
prior = tmp_path / "proposal.md"
|
|
prior.write_text("x" * 2500)
|
|
prior.chmod(0o600)
|
|
|
|
from workflow_bench.proposer_sandbox import SandboxError, stage_evidence_bundle
|
|
|
|
oversized = evolve.proposer_evidence_entries(
|
|
results_dir=None,
|
|
evidence=[],
|
|
learnings=[],
|
|
gate_summary=[],
|
|
prior_proposal=prior,
|
|
)
|
|
with pytest.raises(SandboxError, match="total byte limit"):
|
|
stage_evidence_bundle(tmp_path / "raw", oversized)
|
|
|
|
bundle = evolve.stage_proposer_evidence_bundle(
|
|
tmp_path / "bundle",
|
|
results_dir=None,
|
|
evidence=[],
|
|
learnings=[],
|
|
gate_summary=[],
|
|
prior_proposal=prior,
|
|
)
|
|
names = {path.name for path in bundle.iterdir()}
|
|
assert "selected-rows.json" in names
|
|
assert "prior-proposal.md" not in names
|
|
logged = capsys.readouterr().out
|
|
assert "trimmed proposer evidence" in logged
|
|
assert "omitted prior proposal" in logged
|
|
|
|
|
|
def test_stage_proposer_evidence_bundle_compacts_artifacts_before_dropping_rows(tmp_path, monkeypatch, capsys):
|
|
monkeypatch.setattr(evolve, "MAX_BUNDLE_BYTES", 150_000)
|
|
monkeypatch.setattr("workflow_bench.proposer_sandbox.MAX_BUNDLE_BYTES", 150_000)
|
|
results = tmp_path / "results"
|
|
transcripts = results / "transcripts"
|
|
transcripts.mkdir(parents=True, mode=0o700)
|
|
rows = []
|
|
for index in range(2):
|
|
payload = (
|
|
json.dumps(
|
|
{
|
|
"type": "assistant",
|
|
"message": {"content": [{"type": "text", "text": "x" * 100_000}]},
|
|
}
|
|
)
|
|
+ "\n"
|
|
+ json.dumps({"type": "result", "session_id": f"session-{index}"})
|
|
+ "\n"
|
|
).encode()
|
|
transcript = transcripts / f"session-{index}.jsonl"
|
|
transcript.write_bytes(payload)
|
|
transcript.chmod(0o600)
|
|
(results / f"task-{index}-workflow-run0.patch").write_bytes(b"p" * 100_000)
|
|
rows.append(
|
|
row(
|
|
task=f"task-{index}",
|
|
transcript_artifacts=[
|
|
{
|
|
"path": f"transcripts/session-{index}.jsonl",
|
|
"sha256": hashlib.sha256(payload).hexdigest(),
|
|
"bytes": len(payload),
|
|
"source": PARENT_EVENT_STREAM_SOURCE,
|
|
}
|
|
],
|
|
)
|
|
)
|
|
|
|
bundle = evolve.stage_proposer_evidence_bundle(
|
|
tmp_path / "bundle",
|
|
results_dir=results,
|
|
evidence=rows,
|
|
learnings=[],
|
|
gate_summary=[],
|
|
)
|
|
|
|
staged_rows = json.loads((bundle / "selected-rows.json").read_text())
|
|
assert len(staged_rows) == 2
|
|
assert all((bundle / staged["patch_file"]).is_file() for staged in staged_rows)
|
|
logged = capsys.readouterr().out
|
|
assert "artifact cap" in logged
|
|
assert "dropped 0 row(s)" in logged
|
|
|
|
|
|
@pytest.mark.skipif(os.name == "nt", reason="proposal containment checks are POSIX-only")
|
|
def test_proposer_refuses_a_prior_proposal_that_lost_its_trust_boundary(tmp_path):
|
|
outside = tmp_path / "outside.md"
|
|
outside.write_text("attacker-controlled prose")
|
|
linked = tmp_path / "linked-proposal.md"
|
|
linked.symlink_to(outside)
|
|
|
|
with pytest.raises(evolve.SandboxError, match="regular non-symlink"):
|
|
proposer_evidence_entries(
|
|
results_dir=None,
|
|
evidence=[],
|
|
learnings=[],
|
|
gate_summary=[],
|
|
prior_proposal=linked,
|
|
)
|
|
|
|
# run_proposer copies the proposal out 0600; anything looser means the
|
|
# bytes are no longer only the ones this driver wrote.
|
|
shared = tmp_path / "shared-proposal.md"
|
|
shared.write_text("proposal")
|
|
shared.chmod(0o644)
|
|
with pytest.raises(evolve.SandboxError, match="owner-only"):
|
|
proposer_evidence_entries(
|
|
results_dir=None,
|
|
evidence=[],
|
|
learnings=[],
|
|
gate_summary=[],
|
|
prior_proposal=shared,
|
|
)
|
|
|
|
with pytest.raises(evolve.SandboxError, match="unavailable"):
|
|
proposer_evidence_entries(
|
|
results_dir=None,
|
|
evidence=[],
|
|
learnings=[],
|
|
gate_summary=[],
|
|
prior_proposal=tmp_path / "absent.md",
|
|
)
|
|
|
|
|
|
def test_run_proposer_hides_the_hidden_harness_and_keeps_the_full_tool_surface(monkeypatch, tmp_path):
|
|
"""The proposer writes the artifact the arms are scored with.
|
|
|
|
So its clone must be sanitized before the session starts — a proposer that
|
|
can read eval/workflow_bench reads the task prompts and hidden oracles it
|
|
is about to be graded against, and can encode the answers into the skill.
|
|
"""
|
|
|
|
evidence = tmp_path / "evidence"
|
|
evidence.mkdir()
|
|
transcript_projects = tmp_path / "transcript-projects"
|
|
transcript_projects.mkdir()
|
|
captured: dict[str, object] = {}
|
|
sanitized: list[Path] = []
|
|
events: list[str] = []
|
|
|
|
def fake_make_worktree(_repo, _ref, destination):
|
|
clone = destination / "clone"
|
|
clone.mkdir()
|
|
return clone
|
|
|
|
def fake_sanitize(clone):
|
|
events.append("sanitize")
|
|
sanitized.append(clone)
|
|
return "0" * 40
|
|
|
|
class FakeSandbox:
|
|
claude_bin = "claude"
|
|
command_prefix: list[str] = []
|
|
settings_json = '{"permissions":{"allow":["Read"]}}'
|
|
|
|
@property
|
|
def transcript_projects(self):
|
|
return transcript_projects
|
|
|
|
@contextmanager
|
|
def fake_prepare_sandbox(**_kwargs):
|
|
events.append("prepare")
|
|
yield FakeSandbox()
|
|
|
|
def fake_run_claude(*_args, **kwargs):
|
|
captured.update(kwargs)
|
|
return {"ok": False, "error_kind": "session-error"}
|
|
|
|
monkeypatch.setattr(evolve.runner, "make_worktree", fake_make_worktree)
|
|
monkeypatch.setattr(evolve.runner, "remove_clone", lambda _clone: None)
|
|
monkeypatch.setattr(evolve, "sanitize_clone_for_hidden_oracles", fake_sanitize)
|
|
monkeypatch.setattr(evolve, "prepare_sandbox", fake_prepare_sandbox)
|
|
monkeypatch.setattr(evolve.runner, "run_claude", fake_run_claude)
|
|
args = build_parser().parse_args(["--tasks", "tasks.yaml", "--model", "model"])
|
|
|
|
record = evolve.run_proposer(
|
|
"prompt",
|
|
args,
|
|
overlay_dir=tmp_path / "overlay",
|
|
proposal_path=tmp_path / "proposal.md",
|
|
evidence_bundle=evidence,
|
|
bwrap_bin=tmp_path / "bwrap",
|
|
)
|
|
|
|
assert record["ok"] is False
|
|
# Sanitization has to happen on the clone the session actually runs in,
|
|
# and before the sandbox is prepared around it.
|
|
assert events[:2] == ["sanitize", "prepare"]
|
|
assert [clone.name for clone in sanitized] == ["clone"]
|
|
# Not --bare: bare ignores --tools and would cost the proposer Grep/Glob.
|
|
assert captured.get("bare", False) is False
|
|
assert captured["allowed_tools"] == evolve.PROPOSER_ALLOWED_TOOLS
|
|
assert captured["settings_json"] == FakeSandbox.settings_json
|
|
|
|
|
|
def test_parser_defaults_match_the_gate_minimums():
|
|
args = build_parser().parse_args(["--tasks", "t.yaml", "--model", "pinned"])
|
|
assert args.runs == 3
|
|
assert args.generations == 1
|
|
assert args.arms is None
|
|
assert args.apply is False
|
|
assert args.learnings.name == "learnings.jsonl"
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
"arguments",
|
|
[
|
|
["--model", "Auto"],
|
|
["--model", "provider/latest"],
|
|
["--model", "pinned-model", "--proposer-model", "vendor@LATEST"],
|
|
],
|
|
)
|
|
def test_evolve_rejects_mutable_model_aliases(monkeypatch, tmp_path, capsys, arguments):
|
|
monkeypatch.setattr(
|
|
sys,
|
|
"argv",
|
|
["workflow_bench.evolve", "--tasks", str(tmp_path / "missing.yaml"), *arguments],
|
|
)
|
|
with pytest.raises(SystemExit):
|
|
evolve.main()
|
|
assert "mutable auto/latest" in capsys.readouterr().err
|
|
|
|
|
|
def test_evolve_proposer_failure_returns_nonzero(monkeypatch, tmp_path):
|
|
tasks = tmp_path / "tasks.yaml"
|
|
tasks.write_text(
|
|
"""tasks:
|
|
- id: demo
|
|
class: test
|
|
repo: .
|
|
prompt: implement
|
|
verify: "true"
|
|
oracle:
|
|
command: "true"
|
|
files:
|
|
- source: hidden.test.ts
|
|
target: hidden.test.ts
|
|
"""
|
|
)
|
|
monkeypatch.setattr(
|
|
sys,
|
|
"argv",
|
|
[
|
|
"workflow_bench.evolve",
|
|
"--tasks",
|
|
str(tasks),
|
|
"--model",
|
|
"pinned-model",
|
|
"--out-root",
|
|
str(tmp_path / "out"),
|
|
],
|
|
)
|
|
monkeypatch.setattr(evolve.runner, "selected_task_bindings", lambda _tasks: [{"id": "demo"}])
|
|
monkeypatch.setattr(evolve, "preflight_bubblewrap", lambda: tmp_path / "bwrap")
|
|
monkeypatch.setattr(evolve, "require_claude_sandbox_helpers", lambda: None)
|
|
monkeypatch.setattr(
|
|
evolve,
|
|
"run_proposer",
|
|
lambda *args, **kwargs: {"ok": False, "error_detail": "proposer failed"},
|
|
)
|
|
|
|
assert evolve.main() == 1
|
|
|
|
|
|
def test_proposer_session_record_is_redacted_before_upload(monkeypatch, tmp_path, capsys):
|
|
tasks = tmp_path / "tasks.yaml"
|
|
tasks.write_text(
|
|
"""tasks:
|
|
- id: demo
|
|
class: test
|
|
repo: .
|
|
prompt: implement
|
|
verify: "true"
|
|
oracle:
|
|
command: "true"
|
|
files:
|
|
- source: hidden.test.ts
|
|
target: hidden.test.ts
|
|
"""
|
|
)
|
|
literal_token = "secret-LITERAL-XYZ"
|
|
pattern_token = "sk-ant-FAKEEXAMPLE0000"
|
|
monkeypatch.setattr(
|
|
sys,
|
|
"argv",
|
|
[
|
|
"workflow_bench.evolve",
|
|
"--tasks",
|
|
str(tasks),
|
|
"--model",
|
|
"pinned-model",
|
|
"--out-root",
|
|
str(tmp_path / "out"),
|
|
"--anthropic-api-key",
|
|
literal_token,
|
|
],
|
|
)
|
|
monkeypatch.setattr(evolve.runner, "selected_task_bindings", lambda _tasks: [{"id": "demo"}])
|
|
monkeypatch.setattr(evolve, "preflight_bubblewrap", lambda: tmp_path / "bwrap")
|
|
monkeypatch.setattr(evolve, "require_claude_sandbox_helpers", lambda: None)
|
|
# A session error whose stderr echoed both the literal API key and an
|
|
# sk-ant-shaped token into the record that gets written to the artifact.
|
|
monkeypatch.setattr(
|
|
evolve,
|
|
"run_proposer",
|
|
lambda *args, **kwargs: {
|
|
"ok": False,
|
|
"error_detail": {"stderr_tail": f"boom {literal_token} {pattern_token}"},
|
|
},
|
|
)
|
|
|
|
assert evolve.main() == 1
|
|
|
|
written = (tmp_path / "out" / "gen-0" / "proposer-session.json").read_text()
|
|
assert literal_token not in written
|
|
assert pattern_token not in written
|
|
assert "[REDACTED]" in written
|
|
|
|
# The same record is printed one line later, and the driver's stdout is a
|
|
# live CI log now that the sweep echoes it — same bar as the artifact.
|
|
printed = capsys.readouterr().out
|
|
assert "proposer session failed" in printed
|
|
assert literal_token not in printed
|
|
assert pattern_token not in printed
|
|
assert "[REDACTED]" in printed
|
|
|
|
|
|
def test_benchmark_failure_print_is_redacted(monkeypatch, tmp_path, capsys):
|
|
tasks = tmp_path / "tasks.yaml"
|
|
tasks.write_text(
|
|
"""tasks:
|
|
- id: demo
|
|
class: test
|
|
repo: .
|
|
prompt: implement
|
|
verify: "true"
|
|
oracle:
|
|
command: "true"
|
|
files:
|
|
- source: hidden.test.ts
|
|
target: hidden.test.ts
|
|
"""
|
|
)
|
|
overlay = tmp_path / "overlay"
|
|
skill = overlay / ".claude" / "skills" / "gitnexus-plan" / "SKILL.md"
|
|
skill.parent.mkdir(parents=True)
|
|
skill.write_text("candidate")
|
|
literal_token = "secret-LITERAL-XYZ"
|
|
pattern_token = "sk-ant-FAKEEXAMPLE0000"
|
|
monkeypatch.setattr(
|
|
sys,
|
|
"argv",
|
|
[
|
|
"workflow_bench.evolve",
|
|
"--tasks",
|
|
str(tasks),
|
|
"--model",
|
|
"pinned-model",
|
|
"--out-root",
|
|
str(tmp_path / "out"),
|
|
"--initial-overlay",
|
|
str(overlay),
|
|
"--anthropic-api-key",
|
|
literal_token,
|
|
],
|
|
)
|
|
monkeypatch.setattr(evolve.runner, "selected_task_bindings", lambda _tasks: [{"id": "demo"}])
|
|
monkeypatch.setattr(evolve, "preflight_bubblewrap", lambda: tmp_path / "bwrap")
|
|
monkeypatch.setattr(evolve, "require_claude_sandbox_helpers", lambda: None)
|
|
monkeypatch.setattr(evolve, "resolve_incumbent_arms", lambda *_args, **_kwargs: ["workflow"])
|
|
monkeypatch.setattr(evolve, "freeze_overlay", lambda _source, _destination: "d" * 64)
|
|
monkeypatch.setattr(evolve, "committed_destination_base_digests", lambda _overlay: {})
|
|
monkeypatch.setattr(evolve, "destination_base_digests", lambda _overlay: {})
|
|
# The sweep is launched with GITNEXUS_BENCH_ANTHROPIC_API_KEY in its environment,
|
|
# so its detail/stderr tail is as token-bearing as any session record.
|
|
monkeypatch.setattr(
|
|
evolve,
|
|
"run_managed",
|
|
lambda *_args, **_kwargs: ManagedProcessResult(
|
|
state="exited",
|
|
returncode=2,
|
|
stdout_tail="",
|
|
stderr_tail=f"ANTHROPIC_API_KEY={pattern_token}",
|
|
duration_s=1.0,
|
|
detail=f"sweep died with {literal_token}",
|
|
),
|
|
)
|
|
|
|
assert evolve.main() == 1
|
|
|
|
printed = capsys.readouterr().out
|
|
assert "benchmark run failed" in printed
|
|
assert literal_token not in printed
|
|
assert pattern_token not in printed
|
|
assert "[REDACTED]" in printed
|
|
|
|
|
|
def test_runner_argv_pairs_each_incumbent_with_its_candidate(tmp_path):
|
|
args = build_parser().parse_args(
|
|
[
|
|
"--tasks",
|
|
"t.yaml",
|
|
"--model",
|
|
"pinned",
|
|
"--arms",
|
|
"workflow",
|
|
"--workers",
|
|
"3",
|
|
"--include-expensive",
|
|
]
|
|
)
|
|
overlay = tmp_path / "overlay"
|
|
skill = overlay / ".claude" / "skills" / "gitnexus-plan" / "SKILL.md"
|
|
skill.parent.mkdir(parents=True)
|
|
skill.write_text("candidate")
|
|
task_bindings = [{"id": "task", "resolved_sha": "a" * 40}]
|
|
target_bases = {".claude/skills/gitnexus-plan/SKILL.md": "b" * 64}
|
|
argv = runner_argv(
|
|
args,
|
|
tmp_path / "bench",
|
|
overlay,
|
|
task_bindings=task_bindings,
|
|
target_base_digests=target_bases,
|
|
proposer_model="pinned",
|
|
)
|
|
arms = argv[argv.index("--arms") + 1 : argv.index("--promotion-metric")]
|
|
assert arms == ["workflow", "candidate_workflow"]
|
|
assert str(overlay) in argv
|
|
assert str(tmp_path / "bench") in argv
|
|
assert "pinned" in argv
|
|
assert argv[argv.index("--proposer-model") + 1] == "pinned"
|
|
assert argv[argv.index("--effort") + 1] == "xhigh"
|
|
assert argv[argv.index("--workers") + 1] == "3"
|
|
assert "--include-expensive" in argv
|
|
assert json.loads(argv[argv.index("--task-bindings-json") + 1]) == task_bindings
|
|
assert json.loads(argv[argv.index("--promotion-target-bases-json") + 1]) == target_bases
|
|
|
|
|
|
def test_runner_argv_inserts_ce_review_for_review_overlay(tmp_path):
|
|
args = build_parser().parse_args(
|
|
["--tasks", "t.yaml", "--model", "pinned", "--arms", "review"]
|
|
)
|
|
overlay = tmp_path / "overlay"
|
|
skill = overlay / ".claude" / "skills" / "gitnexus-review" / "SKILL.md"
|
|
skill.parent.mkdir(parents=True)
|
|
skill.write_text("candidate")
|
|
argv = runner_argv(
|
|
args,
|
|
tmp_path / "bench",
|
|
overlay,
|
|
task_bindings=[{"id": "task"}],
|
|
target_base_digests={},
|
|
)
|
|
arms = argv[argv.index("--arms") + 1 : argv.index("--promotion-metric")]
|
|
assert arms == ["ce_review", "review", "candidate_review"]
|
|
assert "--reuse-results" not in argv
|
|
|
|
|
|
def test_runner_argv_forwards_prior_results_for_comparator_reuse(tmp_path):
|
|
args = build_parser().parse_args(
|
|
["--tasks", "t.yaml", "--model", "pinned", "--arms", "review"]
|
|
)
|
|
overlay = tmp_path / "overlay"
|
|
skill = overlay / ".claude" / "skills" / "gitnexus-review" / "SKILL.md"
|
|
skill.parent.mkdir(parents=True)
|
|
skill.write_text("candidate")
|
|
prior = tmp_path / "prior-bench"
|
|
argv = runner_argv(
|
|
args,
|
|
tmp_path / "bench",
|
|
overlay,
|
|
task_bindings=[{"id": "task"}],
|
|
target_base_digests={},
|
|
reuse_results=prior,
|
|
)
|
|
assert argv[argv.index("--reuse-results") + 1] == str(prior)
|
|
|
|
|
|
def test_runner_argv_omits_proposer_for_manual_overlay(tmp_path):
|
|
args = build_parser().parse_args(["--tasks", "t.yaml", "--model", "pinned"])
|
|
overlay = tmp_path / "overlay"
|
|
skill = overlay / ".claude" / "skills" / "gitnexus-plan" / "SKILL.md"
|
|
skill.parent.mkdir(parents=True)
|
|
skill.write_text("candidate")
|
|
|
|
argv = runner_argv(
|
|
args,
|
|
tmp_path / "bench",
|
|
overlay,
|
|
task_bindings=[{"id": "task"}],
|
|
target_base_digests={},
|
|
proposer_model=None,
|
|
)
|
|
|
|
assert "--proposer-model" not in argv
|
|
|
|
|
|
def test_runner_argv_forwards_explicit_unsafe_backend(tmp_path):
|
|
args = build_parser().parse_args(
|
|
["--tasks", "t.yaml", "--model", "pinned", "--unsafe-no-bwrap"]
|
|
)
|
|
overlay = tmp_path / "overlay"
|
|
skill = overlay / ".claude" / "skills" / "gitnexus-plan" / "SKILL.md"
|
|
skill.parent.mkdir(parents=True)
|
|
skill.write_text("candidate")
|
|
|
|
argv = runner_argv(
|
|
args,
|
|
tmp_path / "bench",
|
|
overlay,
|
|
task_bindings=[{"id": "task"}],
|
|
target_base_digests={},
|
|
)
|
|
|
|
assert "--unsafe-no-bwrap" in argv
|
|
|
|
|
|
def test_runner_argv_keeps_task_commit_pinned_when_ref_moves(tmp_path):
|
|
repo = tmp_path / "task-repo"
|
|
repo.mkdir()
|
|
|
|
def git(*arguments):
|
|
return subprocess.run(
|
|
["git", "-C", str(repo), *arguments],
|
|
check=True,
|
|
capture_output=True,
|
|
text=True,
|
|
).stdout.strip()
|
|
|
|
git("init", "-b", "main")
|
|
git("config", "user.name", "Workflow Bench Test")
|
|
git("config", "user.email", "workflow-bench@example.invalid")
|
|
tracked = repo / "tracked.txt"
|
|
tracked.write_text("one")
|
|
git("add", "tracked.txt")
|
|
git("commit", "-m", "first")
|
|
first_sha = git("rev-parse", "HEAD")
|
|
task = {
|
|
"id": "moving-ref",
|
|
"class": "test",
|
|
"repo": str(repo),
|
|
"ref": "main",
|
|
"prompt": "test prompt",
|
|
"verify": "true",
|
|
"oracle": {
|
|
"command": "true",
|
|
"files": [
|
|
{
|
|
"source": "trivial-status-json-alias.oracle.test.ts",
|
|
"target": "oracle.test.ts",
|
|
}
|
|
],
|
|
},
|
|
}
|
|
bindings = evolve.runner.selected_task_bindings([task])
|
|
|
|
tracked.write_text("two")
|
|
git("commit", "-am", "second")
|
|
assert git("rev-parse", "main") != first_sha
|
|
|
|
args = build_parser().parse_args(["--tasks", "t.yaml", "--model", "pinned"])
|
|
overlay = tmp_path / "overlay"
|
|
skill = overlay / ".claude" / "skills" / "gitnexus-plan" / "SKILL.md"
|
|
skill.parent.mkdir(parents=True)
|
|
skill.write_text("candidate")
|
|
argv = runner_argv(
|
|
args,
|
|
tmp_path / "bench",
|
|
overlay,
|
|
task_bindings=bindings,
|
|
target_base_digests={},
|
|
)
|
|
forwarded = json.loads(argv[argv.index("--task-bindings-json") + 1])
|
|
|
|
assert forwarded[0]["resolved_sha"] == first_sha
|
|
assert evolve.runner.resolve_task_bindings([task], forwarded)[0]["resolved_sha"] == first_sha
|
|
|
|
|
|
def test_generation_timeout_budgets_three_task_workflow_pair():
|
|
timeout = generation_timeout_seconds(
|
|
task_count=3,
|
|
runs=3,
|
|
session_timeout=3600,
|
|
incumbent_arms=["workflow"],
|
|
)
|
|
|
|
per_task_preparation = (
|
|
evolve.TASK_BINDING_GIT_PHASES * evolve.GIT_COMMAND_TIMEOUT_SECONDS
|
|
+ 2 * evolve.TASK_SNAPSHOT_TIMEOUT_SECONDS
|
|
+ evolve.WORKTREE_PREPARATION_TIMEOUT_SECONDS
|
|
+ evolve.GRAPH_SOURCE_PREPARATION_TIMEOUT_SECONDS
|
|
+ evolve.GRAPH_BUILD_TIMEOUT_SECONDS
|
|
+ 2 * evolve.GRAPH_QUERY_TIMEOUT_SECONDS
|
|
+ evolve.CLEANUP_TIMEOUT_SECONDS
|
|
)
|
|
paired_arm_cells = 2
|
|
session_slots = 4
|
|
workspace_snapshot_slots = 4
|
|
per_task_run = session_slots * (3600 + evolve.SESSION_FINALIZATION_TIMEOUT_SECONDS) + paired_arm_cells * (
|
|
evolve.WORKTREE_PREPARATION_TIMEOUT_SECONDS
|
|
+ evolve.ARM_ASSET_MATERIALIZATION_PHASES * evolve.TASK_SNAPSHOT_TIMEOUT_SECONDS
|
|
+ evolve.SETUP_TIMEOUT_SECONDS
|
|
+ 2 * 3600
|
|
+ evolve.ARM_EVIDENCE_GIT_PHASES * evolve.GIT_COMMAND_TIMEOUT_SECONDS
|
|
+ evolve.CLEANUP_TIMEOUT_SECONDS
|
|
)
|
|
per_task_run += workspace_snapshot_slots * evolve.TASK_SNAPSHOT_TIMEOUT_SECONDS
|
|
per_task_run += evolve.CANDIDATE_OVERLAY_GIT_PHASES * evolve.GIT_COMMAND_TIMEOUT_SECONDS
|
|
|
|
assert timeout == (
|
|
evolve.PROMOTION_BASE_TIMEOUT_SECONDS
|
|
+ 3 * (per_task_preparation + 3 * per_task_run)
|
|
+ evolve.DRIVER_OVERHEAD_SECONDS
|
|
)
|
|
# The old deadline omitted clone sanitization entirely. Every graph seed
|
|
# and every paired arm cell must now receive the full bounded envelope.
|
|
assert timeout >= 3 * (1 + 3 * paired_arm_cells) * evolve.WORKTREE_PREPARATION_TIMEOUT_SECONDS
|
|
|
|
|
|
def test_executed_benchmark_arms_inserts_review_comparator() -> None:
|
|
assert executed_benchmark_arms(["workflow"]) == ["workflow", "candidate_workflow"]
|
|
assert executed_benchmark_arms(["review"]) == ["ce_review", "review", "candidate_review"]
|
|
|
|
|
|
def test_generation_timeout_budgets_review_pair_plus_ce_comparator() -> None:
|
|
timeout = generation_timeout_seconds(
|
|
task_count=6,
|
|
runs=1,
|
|
session_timeout=3600,
|
|
incumbent_arms=["review"],
|
|
)
|
|
|
|
per_task_preparation = (
|
|
evolve.TASK_BINDING_GIT_PHASES * evolve.GIT_COMMAND_TIMEOUT_SECONDS
|
|
+ 2 * evolve.TASK_SNAPSHOT_TIMEOUT_SECONDS
|
|
+ evolve.WORKTREE_PREPARATION_TIMEOUT_SECONDS
|
|
+ evolve.GRAPH_SOURCE_PREPARATION_TIMEOUT_SECONDS
|
|
+ evolve.GRAPH_BUILD_TIMEOUT_SECONDS
|
|
+ 2 * evolve.GRAPH_QUERY_TIMEOUT_SECONDS
|
|
+ evolve.CLEANUP_TIMEOUT_SECONDS
|
|
)
|
|
paired_arm_cells = 3
|
|
session_slots = 3
|
|
workspace_snapshot_slots = 3
|
|
per_task_run = session_slots * (3600 + evolve.SESSION_FINALIZATION_TIMEOUT_SECONDS) + paired_arm_cells * (
|
|
evolve.WORKTREE_PREPARATION_TIMEOUT_SECONDS
|
|
+ evolve.ARM_ASSET_MATERIALIZATION_PHASES * evolve.TASK_SNAPSHOT_TIMEOUT_SECONDS
|
|
+ evolve.SETUP_TIMEOUT_SECONDS
|
|
+ 2 * 3600
|
|
+ evolve.ARM_EVIDENCE_GIT_PHASES * evolve.GIT_COMMAND_TIMEOUT_SECONDS
|
|
+ evolve.CLEANUP_TIMEOUT_SECONDS
|
|
)
|
|
per_task_run += workspace_snapshot_slots * evolve.TASK_SNAPSHOT_TIMEOUT_SECONDS
|
|
per_task_run += evolve.CANDIDATE_OVERLAY_GIT_PHASES * evolve.GIT_COMMAND_TIMEOUT_SECONDS
|
|
|
|
assert timeout == (
|
|
evolve.PROMOTION_BASE_TIMEOUT_SECONDS
|
|
+ 6 * (per_task_preparation + per_task_run)
|
|
+ evolve.DRIVER_OVERHEAD_SECONDS
|
|
)
|
|
|
|
|
|
def test_generation_timeout_rejects_unknown_arm() -> None:
|
|
with pytest.raises(ValueError, match="unsupported evolution arm: mystery"):
|
|
generation_timeout_seconds(
|
|
task_count=1,
|
|
runs=1,
|
|
session_timeout=60,
|
|
incumbent_arms=["mystery"],
|
|
)
|
|
|
|
|
|
def test_instance_window_budget_leaves_upload_reserve() -> None:
|
|
# Friday 10:57 on a box that booted 02:45 Saturday-window: ~8.2h uptime.
|
|
leftover = instance_window_budget_seconds(8.2 * 3600)
|
|
assert leftover == int(86_400 - 8.2 * 3600 - 5_400)
|
|
assert leftover >= MIN_INSTANCE_SWEEP_SECONDS
|
|
with pytest.raises(ValueError, match="only .*s left"):
|
|
instance_window_budget_seconds(23.5 * 3600)
|
|
with pytest.raises(ValueError, match="uptime must be"):
|
|
instance_window_budget_seconds(float("nan"))
|
|
|
|
|
|
def test_instance_window_budget_from_proc_reads_uptime_and_env(
|
|
tmp_path: Path, monkeypatch: pytest.MonkeyPatch
|
|
) -> None:
|
|
uptime = tmp_path / "uptime"
|
|
uptime.write_text("3600.00 8000.00\n")
|
|
monkeypatch.setenv("EVENTBRIDGE_INSTANCE_WINDOW_SECONDS", "20000")
|
|
monkeypatch.setenv("EVENTBRIDGE_STOP_RESERVE_SECONDS", "1000")
|
|
assert instance_window_budget_from_proc(uptime) == 20000 - 3600 - 1000
|
|
with pytest.raises(ValueError, match="cannot read instance uptime"):
|
|
instance_window_budget_from_proc(tmp_path / "missing")
|
|
|
|
|
|
def test_capped_timeout_clamps_to_leftover_window(monkeypatch) -> None:
|
|
assert capped_timeout_seconds(10_000, None) == 10_000
|
|
assert capped_timeout_seconds(10_000, 90) == 90
|
|
with pytest.raises(ValueError, match="no time remains"):
|
|
capped_timeout_seconds(10_000, 0)
|
|
# Pinned clock: the helper is pure arithmetic, so a real elapsed-time window
|
|
# would assert on scheduling rather than on the behaviour under test.
|
|
monkeypatch.setattr(evolve.time, "monotonic", lambda: 1040.0)
|
|
started = 1000.0
|
|
assert remaining_runtime_seconds(max_runtime_seconds=None, started_monotonic=started) is None
|
|
leftover = remaining_runtime_seconds(max_runtime_seconds=100, started_monotonic=started)
|
|
assert leftover is not None
|
|
assert leftover == 60
|
|
|
|
|
|
def test_parser_rejects_non_positive_max_runtime() -> None:
|
|
with pytest.raises(SystemExit):
|
|
build_parser().parse_args(
|
|
["--tasks", "t.yaml", "--model", "pinned", "--max-runtime-seconds", "0"]
|
|
)
|
|
args = build_parser().parse_args(
|
|
["--tasks", "t.yaml", "--model", "pinned", "--max-runtime-seconds", "7200"]
|
|
)
|
|
assert args.max_runtime_seconds == 7200
|
|
|
|
|
|
@pytest.mark.skipif(sys.platform != "linux", reason="Bubblewrap PID namespaces require Linux")
|
|
def test_outer_runner_pid_namespace_kills_setsid_descendant(tmp_path):
|
|
try:
|
|
bwrap = preflight_bubblewrap()
|
|
except evolve.SandboxError as exc:
|
|
pytest.skip(str(exc))
|
|
raise AssertionError("pytest.skip() returned unexpectedly")
|
|
sentinel = tmp_path / "escaped"
|
|
child = (
|
|
"import os,subprocess,sys,time; "
|
|
f"subprocess.Popen([sys.executable,'-c',\"import time,pathlib;time.sleep(1);pathlib.Path({str(sentinel)!r}).touch()\"],preexec_fn=os.setsid); "
|
|
"time.sleep(10)"
|
|
)
|
|
result = run_managed(
|
|
pid_namespace_command([sys.executable, "-c", child], bwrap_bin=bwrap),
|
|
timeout=0.15,
|
|
terminate_grace=0.1,
|
|
require_pid_namespace=True,
|
|
)
|
|
time.sleep(1.1)
|
|
|
|
assert not result.ok
|
|
assert result.state in {"timeout", "forced-kill"}
|
|
assert not sentinel.exists()
|
|
|
|
|
|
def test_resolve_incumbent_arms_rejects_incomplete_and_extra_explicit_sets(tmp_path):
|
|
plan = tmp_path / "plan"
|
|
plan_skill = plan / ".claude" / "skills" / "gitnexus-plan" / "SKILL.md"
|
|
plan_skill.parent.mkdir(parents=True)
|
|
plan_skill.write_text("plan")
|
|
assert resolve_incumbent_arms(plan, None) == ["workflow"]
|
|
with pytest.raises(ValueError, match="exactly"):
|
|
resolve_incumbent_arms(plan, ["workflow", "workflow_direct"])
|
|
|
|
work = tmp_path / "work"
|
|
work_skill = work / ".claude" / "skills" / "gitnexus-work" / "SKILL.md"
|
|
work_skill.parent.mkdir(parents=True)
|
|
work_skill.write_text("work")
|
|
assert resolve_incumbent_arms(work, None) == ["workflow", "workflow_direct"]
|
|
with pytest.raises(ValueError, match="exactly"):
|
|
resolve_incumbent_arms(work, ["workflow"])
|
|
|
|
|
|
def bound_task_fixture(task_id="task-a"):
|
|
return {
|
|
"id": task_id,
|
|
"prompt_digest": "prompt",
|
|
"oracle_digest": "a" * 64,
|
|
"oracle_command_digest": "b" * 64,
|
|
"oracle_manifest_digest": "c" * 64,
|
|
"sandbox_dependency_content_digest": "e" * 64,
|
|
"sandbox_dependency_manifest_digest": "f" * 64,
|
|
"oracle_files": [{"target": "oracle.test.ts", "sha256": "d" * 64, "size": 10}],
|
|
}
|
|
|
|
|
|
def bound_task_fixtures():
|
|
return [bound_task_fixture("task-a"), bound_task_fixture("task-impossible")]
|
|
|
|
|
|
def promote_decision(**overrides):
|
|
"""A real producer decision, including its recomputable paired metrics."""
|
|
base = {"runs": 3, "valid_runs": 3, "excluded_runs": 0, "error_kinds": {}}
|
|
results = {
|
|
"task-a": {
|
|
"workflow": {**base, "resolved": 3, "cost_usd": 1.0},
|
|
"candidate_workflow": {**base, "resolved": 3, "cost_usd": 0.8},
|
|
},
|
|
"task-impossible": {
|
|
"workflow": {**base, "resolved": 0, "cost_usd": 1.0},
|
|
"candidate_workflow": {**base, "resolved": 0, "cost_usd": 1.0},
|
|
},
|
|
}
|
|
decision = evolution.promotion_evidence(
|
|
results,
|
|
policy=evolution.promotion_policy(["candidate_workflow"]),
|
|
model="bench-model",
|
|
complete=True,
|
|
)["decisions"][0]
|
|
decision.update(overrides)
|
|
return decision
|
|
|
|
|
|
def promotion_fixture(*, decisions=None, expires_delta=timedelta(days=1)):
|
|
now = datetime.now(UTC)
|
|
return {
|
|
"schema_version": 6,
|
|
"run_status": "complete",
|
|
"generated_at": now.isoformat(),
|
|
"evidence_expires_at": (now + expires_delta).isoformat(),
|
|
"benchmark_model": "bench-model",
|
|
"proposer_model": "proposer-model",
|
|
"effort": "xhigh",
|
|
"candidate_origin": "model-proposer",
|
|
"candidate_overlay_digest": "digest",
|
|
"target_base_digests": {"path": "base"},
|
|
"required_candidate_arms": ["candidate_workflow"],
|
|
"selected_tasks": bound_task_fixtures(),
|
|
"policy": evolution.promotion_policy(["candidate_workflow"]),
|
|
"decisions": decisions if decisions is not None else [promote_decision()],
|
|
}
|
|
|
|
|
|
def validate_fixture(promotion):
|
|
return validate_promotion_for_apply(
|
|
promotion,
|
|
overlay_digest="digest",
|
|
benchmark_model="bench-model",
|
|
proposer_model="proposer-model",
|
|
effort="xhigh",
|
|
selected_tasks=bound_task_fixtures(),
|
|
target_base_digests={"path": "base"},
|
|
required_candidate_arms=["candidate_workflow"],
|
|
policy=evolution.promotion_policy(["candidate_workflow"]),
|
|
)
|
|
|
|
|
|
def test_promotion_apply_requires_one_promote_for_every_bound_arm():
|
|
assert [d["candidate_arm"] for d in validate_fixture(promotion_fixture())] == ["candidate_workflow"]
|
|
|
|
for decisions in (
|
|
[],
|
|
[promote_decision(decision="keep_incumbent")],
|
|
[promote_decision(), promote_decision()],
|
|
[promote_decision(incumbent_arm="workflow_direct", candidate_arm="candidate_workflow_direct")],
|
|
):
|
|
with pytest.raises(ValueError):
|
|
validate_fixture(promotion_fixture(decisions=decisions))
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
("overrides", "match"),
|
|
[
|
|
# An older decision relabeled as schema 5: the verdict without the
|
|
# gated evidence base schema 5 promotes on.
|
|
({"tasks": None, "ungated_tasks": None}, "no per-task gate evidence"),
|
|
({"tasks": [{"task": "task-a"}]}, "malformed per-task gate evidence"),
|
|
({"tasks": [{"task": "task-a", "gated": "yes"}]}, "malformed per-task gate evidence"),
|
|
(
|
|
{"tasks": [{"task": "task-a", "gated": True}, {"task": "task-a", "gated": False}]},
|
|
"repeats a task",
|
|
),
|
|
(
|
|
{"ungated_tasks": [], "tasks": [{"task": "fabricated", "gated": True}]},
|
|
"does not match selected tasks",
|
|
),
|
|
({"ungated_tasks": None}, "missing its ungated task list"),
|
|
# The verdict claims a full gate; the per-task rows say a task sat
|
|
# outside it.
|
|
({"ungated_tasks": []}, "disagree with its per-task evidence"),
|
|
(
|
|
{
|
|
"ungated_tasks": ["task-a", "task-impossible"],
|
|
"tasks": [
|
|
{"task": "task-a", "gated": False},
|
|
{"task": "task-impossible", "gated": False},
|
|
],
|
|
},
|
|
"no gated task",
|
|
),
|
|
],
|
|
)
|
|
def test_promotion_apply_binds_the_schema_5_gate_evidence(overrides, match):
|
|
decision = promote_decision()
|
|
for field, value in overrides.items():
|
|
if value is None:
|
|
decision.pop(field)
|
|
else:
|
|
decision[field] = value
|
|
|
|
with pytest.raises(ValueError, match=match):
|
|
validate_fixture(promotion_fixture(decisions=[decision]))
|
|
|
|
|
|
def test_promotion_apply_rejects_a_gate_with_only_one_of_three_selected_tasks():
|
|
selected = [*bound_task_fixtures(), bound_task_fixture("task-impossible-2")]
|
|
decision = promote_decision(
|
|
ungated_tasks=["task-impossible", "task-impossible-2"],
|
|
tasks=[
|
|
{"task": "task-a", "gated": True},
|
|
{"task": "task-impossible", "gated": False},
|
|
{"task": "task-impossible-2", "gated": False},
|
|
],
|
|
)
|
|
promotion = promotion_fixture(decisions=[decision])
|
|
promotion["selected_tasks"] = selected
|
|
|
|
with pytest.raises(ValueError, match="too thin a gated evidence base"):
|
|
validate_promotion_for_apply(
|
|
promotion,
|
|
overlay_digest="digest",
|
|
benchmark_model="bench-model",
|
|
proposer_model="proposer-model",
|
|
effort="xhigh",
|
|
selected_tasks=selected,
|
|
target_base_digests={"path": "base"},
|
|
required_candidate_arms=["candidate_workflow"],
|
|
policy=promotion["policy"],
|
|
)
|
|
|
|
|
|
def test_manual_initial_overlay_has_no_fictitious_proposer_model():
|
|
promotion = promotion_fixture()
|
|
promotion["proposer_model"] = None
|
|
promotion["candidate_origin"] = "manual-initial-overlay"
|
|
|
|
decisions = validate_promotion_for_apply(
|
|
promotion,
|
|
overlay_digest="digest",
|
|
benchmark_model="bench-model",
|
|
proposer_model=None,
|
|
effort="xhigh",
|
|
selected_tasks=bound_task_fixtures(),
|
|
target_base_digests={"path": "base"},
|
|
required_candidate_arms=["candidate_workflow"],
|
|
policy=promotion["policy"],
|
|
)
|
|
|
|
assert decisions[0]["decision"] == "promote"
|
|
|
|
|
|
def test_promotion_apply_rejects_pre_oracle_schema_and_missing_oracle_bindings():
|
|
legacy = promotion_fixture()
|
|
legacy["schema_version"] = 3
|
|
with pytest.raises(ValueError, match="unsupported schema"):
|
|
validate_fixture(legacy)
|
|
|
|
weak_task = {"id": "task", "prompt_digest": "prompt"}
|
|
weak = promotion_fixture()
|
|
weak["selected_tasks"] = [weak_task]
|
|
with pytest.raises(ValueError, match="hidden-oracle or dependency digests"):
|
|
validate_promotion_for_apply(
|
|
weak,
|
|
overlay_digest="digest",
|
|
benchmark_model="bench-model",
|
|
proposer_model="proposer-model",
|
|
effort="xhigh",
|
|
selected_tasks=[weak_task],
|
|
target_base_digests={"path": "base"},
|
|
required_candidate_arms=["candidate_workflow"],
|
|
policy={
|
|
"metric": "cost_usd",
|
|
"min_runs": 3,
|
|
"min_improvement_pct": 5.0,
|
|
"max_task_regression_pct": 20.0,
|
|
},
|
|
)
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
("field", "value"),
|
|
[
|
|
("benchmark_model", "other"),
|
|
("proposer_model", "other"),
|
|
("candidate_overlay_digest", "other"),
|
|
("target_base_digests", {"path": "other"}),
|
|
("required_candidate_arms", ["candidate_workflow_direct"]),
|
|
("selected_tasks", [{"id": "other", "prompt_digest": "prompt"}]),
|
|
(
|
|
"policy",
|
|
{
|
|
"metric": "cost_usd",
|
|
"min_runs": 4,
|
|
"min_improvement_pct": 5.0,
|
|
"max_task_regression_pct": 20.0,
|
|
},
|
|
),
|
|
],
|
|
)
|
|
def test_promotion_apply_rejects_mismatched_evidence_bindings(field, value):
|
|
promotion = promotion_fixture()
|
|
promotion[field] = value
|
|
with pytest.raises(ValueError, match="binding"):
|
|
validate_fixture(promotion)
|
|
|
|
|
|
def test_promotion_apply_rejects_expired_evidence():
|
|
with pytest.raises(ValueError, match="expired"):
|
|
validate_fixture(promotion_fixture(expires_delta=timedelta(seconds=-1)))
|
|
|
|
|
|
def test_promotion_apply_rejects_extended_or_future_dated_evidence():
|
|
with pytest.raises(ValueError, match="expired"):
|
|
validate_fixture(promotion_fixture(expires_delta=timedelta(days=91)))
|
|
|
|
promotion = promotion_fixture()
|
|
future = datetime.now(UTC) + timedelta(days=1)
|
|
promotion["generated_at"] = future.isoformat()
|
|
promotion["evidence_expires_at"] = (future + timedelta(days=1)).isoformat()
|
|
with pytest.raises(ValueError, match="future"):
|
|
validate_fixture(promotion)
|
|
|
|
|
|
def test_promotion_apply_rejects_decision_metric_mismatch():
|
|
promotion = promotion_fixture()
|
|
promotion["decisions"][0]["metric"] = "output_tokens"
|
|
with pytest.raises(ValueError, match="metric mismatch"):
|
|
validate_fixture(promotion)
|