"""Unit tests for the pure evidence/apply/driver helpers of workflow_bench.evolve.""" import hashlib import json import os import subprocess import sys import time from contextlib import contextmanager from datetime import UTC, datetime, timedelta from pathlib import Path from types import SimpleNamespace import pytest from workflow_bench import evolve, evolution from workflow_bench.runner_sessions import PARENT_EVENT_STREAM_SOURCE from workflow_bench.evolve import ( MIN_INSTANCE_SWEEP_SECONDS, build_parser, build_proposer_prompt, capped_timeout_seconds, executed_benchmark_arms, generation_timeout_seconds, instance_window_budget_seconds, load_jsonl, proposer_evidence_entries, read_learnings, remaining_runtime_seconds, resolve_incumbent_arms, runner_argv, select_evidence, summarize_gate, validate_promotion_for_apply, ) from workflow_bench.process_control import ManagedProcessResult, run_managed from workflow_bench.proposer_sandbox import pid_namespace_command, preflight_bubblewrap def test_runner_environment_does_not_forward_the_openai_key() -> None: from workflow_bench.model_gateway import credential_secrets args = build_parser().parse_args( [ "--tasks", "t.yaml", "--model", "gpt-4.1", "--anthropic-api-key", "loopback-master", "--openai-api-key", "sk-openai-secret", ] ) env = evolve.runner_environment(args) assert env["GITNEXUS_BENCH_ANTHROPIC_API_KEY"] == "loopback-master" assert "GITNEXUS_BENCH_AUTH_TOKEN" not in env assert "OPENAI_API_KEY" not in env assert "GITNEXUS_BENCH_OPENAI_API_KEY" not in env assert "sk-openai-secret" not in env.values() assert credential_secrets(args) == ["loopback-master", "sk-openai-secret"] def test_parser_keeps_the_legacy_auth_token_alias() -> None: args = build_parser().parse_args(["--tasks", "t.yaml", "--model", "pinned", "--auth-token", "alias-secret"]) assert args.auth_token == "alias-secret" def row(**overrides): base = { "task": "demo-task", "class": "trivial", "arm": "workflow", "run": 0, "resolved": True, "error_kind": None, "cost_usd": 1.0, "num_turns": 10, "output_tokens": 400, "session_ids": ["sess-1"], "verify_output": "ok", } base.update(overrides) return base def test_select_evidence_puts_unresolved_before_expensive_resolved(): rows = [ row(task="cheap", resolved=True, cost_usd=0.5), row(task="fail", resolved=False, error_kind="verify-failed"), row(task="pricey", resolved=True, cost_usd=9.0), ] picked = select_evidence(rows) assert [r["task"] for r in picked] == ["fail", "pricey", "cheap"] def test_select_evidence_excludes_infra_error_rows_and_caps(): rows = [ row(task="harness-died", resolved=False, error_kind="infra-error"), row(task="session-died", resolved=False, error_kind="session-error"), row( task="missing-transcript", resolved=False, error_kind="evidence-unverified", ), ] rows += [row(task=f"t{i}", cost_usd=float(i)) for i in range(20)] picked = select_evidence(rows, max_rows=5) assert len(picked) == 5 assert all(r["error_kind"] != "infra-error" for r in picked) assert [r["task"] for r in picked] == ["t19", "t18", "t17", "t16", "t15"] def test_select_evidence_tolerates_an_explicit_null_cost(): # A foreign --seed-results row (e.g. hand-edited or from another tool) # can carry an explicit JSON null rather than omitting the key; .get's # default only covers the missing-key case, so this must not raise. rows = [row(task="no-cost", resolved=True, cost_usd=None), row(task="priced", cost_usd=5.0)] picked = select_evidence(rows) assert [r["task"] for r in picked] == ["priced", "no-cost"] def test_load_jsonl_skips_blank_and_malformed_lines(tmp_path): path = tmp_path / "learnings.jsonl" path.write_text('{"skill": "gitnexus-plan"}\n\nnot json\n[1, 2]\n{"skill": "gitnexus-work"}\n') assert load_jsonl(path) == [{"skill": "gitnexus-plan"}, {"skill": "gitnexus-work"}] def test_load_jsonl_missing_file_is_empty(tmp_path): assert load_jsonl(tmp_path / "absent.jsonl") == [] def test_read_learnings_keeps_the_most_recent_entries(tmp_path): path = tmp_path / "learnings.jsonl" rows = [{"skill": "gitnexus-work", "n": i} for i in range(10)] + [ {"skill": "gitnexus-review", "n": 10}, {"skill": "gitnexus-lfg", "n": 11}, ] path.write_text("\n".join(json.dumps(row) for row in rows) + "\n") assert read_learnings(path, cap=3) == [ {"skill": "gitnexus-work", "n": 8}, {"skill": "gitnexus-work", "n": 9}, {"skill": "gitnexus-review", "n": 10}, ] def test_summarize_gate_one_line_per_decision(): promotion = { "decisions": [ { "candidate_arm": "candidate_workflow", "decision": "keep_incumbent", "reasons": ["a", "b", "c", "d"], } ] } lines = summarize_gate(promotion) assert lines == ["candidate_workflow: keep_incumbent — a; b; c"] def test_build_proposer_prompt_carries_evidence_constraints_and_paths(tmp_path): prompt = build_proposer_prompt( results_dir=tmp_path / "bench", evidence=[row(task="fail", resolved=False, error_kind="verify-failed")], learnings=[{"skill": "gitnexus-work", "friction": "budget blown on reruns"}], gate_summary=["candidate_workflow: keep_incumbent — quality regressed"], overlay_dir=tmp_path / "overlay", proposal_path=tmp_path / "proposal.md", incumbent_arms=["workflow"], ) assert str(tmp_path / "overlay") in prompt assert str(tmp_path / "proposal.md") in prompt assert "gitnexus-plan, gitnexus-work" in prompt assert "node .gitnexus/run.cjs analyze" in prompt assert "1 row(s) in /evidence/learnings.json" in prompt assert "1 selected row(s) in /evidence/selected-rows.json" in prompt assert "exact staged" in prompt assert "no full results.jsonl" in prompt assert "1 decision(s) in /evidence/gate-summary.json" in prompt assert "budget blown on reruns" not in prompt assert "verify-failed" not in prompt assert "~/.claude/projects" not in prompt def test_build_proposer_prompt_points_at_the_rejected_prior_proposal(tmp_path): common = { "results_dir": tmp_path / "bench", "evidence": [], "learnings": [], "gate_summary": ["candidate_workflow: keep_incumbent — cost regressed"], "overlay_dir": tmp_path / "overlay", "proposal_path": tmp_path / "proposal.md", "incumbent_arms": ["workflow"], } # The gate summary alone says a candidate lost, never what it proposed — # so without this line the proposer can re-propose the same prose forever. assert "/evidence/prior-proposal.md" in build_proposer_prompt(**common, prior_proposal=True) assert "/evidence/prior-proposal.md" not in build_proposer_prompt(**common) def test_build_proposer_prompt_first_generation_has_no_results_dir(tmp_path): prompt = build_proposer_prompt( results_dir=None, evidence=[], learnings=[], gate_summary=[], overlay_dir=tmp_path / "overlay", proposal_path=tmp_path / "proposal.md", incumbent_arms=["workflow_direct"], ) assert "none (first generation)" in prompt assert "none yet — use the incumbent skills and staged learning queue" in prompt def test_proposer_reads_only_digest_bound_transcripts_below_results(tmp_path, monkeypatch): results = tmp_path / "results" transcripts = results / "transcripts" transcripts.mkdir(parents=True, mode=0o700) transcripts.chmod(0o700) payload = b'{"message":{"content":[{"type":"text","text":"bound transcript"}]}}\n' artifact = transcripts / "task-workflow-run0-session.jsonl" artifact.write_bytes(payload) artifact.chmod(0o600) patch = results / "demo-task-workflow-run0.patch" patch.write_text("diff --git a/a b/a\n") metadata = { "path": "transcripts/task-workflow-run0-session.jsonl", "sha256": hashlib.sha256(payload).hexdigest(), "bytes": len(payload), "source": PARENT_EVENT_STREAM_SOURCE, } foreign_home = tmp_path / "foreign-home" foreign = foreign_home / ".claude" / "projects" / "other" / "private.jsonl" foreign.parent.mkdir(parents=True) foreign.write_text("foreign host transcript") monkeypatch.setenv("HOME", str(foreign_home)) entries = proposer_evidence_entries( results_dir=results, evidence=[row(session_ids=["**/*"], transcript_artifacts=[metadata])], learnings=[], gate_summary=[], ) assert [json.loads(line) for line in entries["transcript-0-0.jsonl"].splitlines()] == [json.loads(payload)] assert entries["patch-0.diff"] == patch.read_text() staged_rows = entries["selected-rows.json"] assert staged_rows[0]["patch_file"] == "patch-0.diff" assert staged_rows[0]["transcript_files"] == ["transcript-0-0.jsonl"] assert "foreign host transcript" not in json.dumps(entries) bad_digest = {**metadata, "sha256": "0" * 64} with pytest.raises(evolve.SandboxError, match="digest does not match"): proposer_evidence_entries( results_dir=results, evidence=[row(transcript_artifacts=[bad_digest])], learnings=[], gate_summary=[], ) def test_proposer_compacts_transcripts_as_complete_json_events(tmp_path): results = tmp_path / "results" transcripts = results / "transcripts" transcripts.mkdir(parents=True, mode=0o700) events = [ { "type": "assistant", "message": { "content": [ { "type": "thinking", "thinking": "analysis-" + ("x" * 100_000), "signature": "opaque-base64-signature", } ] }, }, { "type": "result", "session_id": "session-1", "usage": {"input_tokens": 1, "output_tokens": 2}, }, ] payload = "".join(json.dumps(event) + "\n" for event in events).encode() artifact = transcripts / "session.jsonl" artifact.write_bytes(payload) artifact.chmod(0o600) entries = proposer_evidence_entries( results_dir=results, evidence=[ row( transcript_artifacts=[ { "path": "transcripts/session.jsonl", "sha256": hashlib.sha256(payload).hexdigest(), "bytes": len(payload), "source": PARENT_EVENT_STREAM_SOURCE, } ] ) ], learnings=[], gate_summary=[], artifact_limit=8192, ) staged = entries["transcript-0-0.jsonl"] parsed = [json.loads(line) for line in staged.splitlines()] assert len(staged.encode()) <= 8192 assert parsed[-1]["type"] == "result" assert parsed[0]["message"]["content"][0]["signature"] == "[OMITTED]" assert "opaque-base64-signature" not in staged @pytest.mark.skipif(os.name == "nt", reason="transcript symlink containment is POSIX-only") def test_proposer_rejects_symlink_and_foreign_transcript_artifacts(tmp_path): results = tmp_path / "results" transcripts = results / "transcripts" transcripts.mkdir(parents=True, mode=0o700) transcripts.chmod(0o700) outside = tmp_path / "outside.jsonl" outside.write_text("outside") linked = transcripts / "linked.jsonl" linked.symlink_to(outside) link_metadata = { "path": "transcripts/linked.jsonl", "sha256": hashlib.sha256(outside.read_bytes()).hexdigest(), "bytes": outside.stat().st_size, "source": PARENT_EVENT_STREAM_SOURCE, } with pytest.raises(evolve.SandboxError, match="regular non-symlink"): proposer_evidence_entries( results_dir=results, evidence=[row(transcript_artifacts=[link_metadata])], learnings=[], gate_summary=[], ) with pytest.raises(evolve.SandboxError, match="unsafe results artifact path"): proposer_evidence_entries( results_dir=results, evidence=[ row( transcript_artifacts=[ { "path": "../outside.jsonl", "sha256": "0" * 64, "bytes": 0, "source": PARENT_EVENT_STREAM_SOURCE, } ] ) ], learnings=[], gate_summary=[], ) def test_proposer_rejects_duplicate_transcript_metadata_before_materializing(): metadata = { "path": "transcripts/repeated.jsonl", "sha256": "0" * 64, "bytes": 0, "source": PARENT_EVENT_STREAM_SOURCE, } with pytest.raises(evolve.SandboxError, match="duplicate transcript artifact path"): proposer_evidence_entries( results_dir=None, evidence=[ row( transcript_artifacts=[ metadata, {**metadata, "path": "transcripts//repeated.jsonl"}, ] ) ], learnings=[], gate_summary=[], ) def test_proposer_bounds_transcript_metadata_per_row_and_globally_before_materializing(): def metadata(index): return { "path": f"transcripts/session-{index}.jsonl", "sha256": "0" * 64, "bytes": 0, "source": PARENT_EVENT_STREAM_SOURCE, } with pytest.raises(evolve.SandboxError, match="per-row session limit"): proposer_evidence_entries( results_dir=None, evidence=[row(transcript_artifacts=[metadata(index) for index in range(3)])], learnings=[], gate_summary=[], ) rows = [ row( run=index, transcript_artifacts=[metadata(2 * index), metadata(2 * index + 1)], ) for index in range(evolve.MAX_EVIDENCE_ROWS + 1) ] with pytest.raises(evolve.SandboxError, match="global evidence limit"): proposer_evidence_entries( results_dir=None, evidence=rows, learnings=[], gate_summary=[], ) def test_proposer_refuses_a_selected_row_with_no_transcript_reference(): # Every selectable row comes from sum_sessions(), which always emits the # key, and select_evidence() drops the kinds a failed transcript # persistence produces (session-error, infra-error, evidence-unverified, # cleanup-failure). A selected row without a transcript is therefore # evidence lost between producer and proposer, not a row that had none. with pytest.raises(evolve.SandboxError, match="missing transcript_artifacts"): proposer_evidence_entries( results_dir=None, evidence=[row()], learnings=[], gate_summary=[], ) with pytest.raises(evolve.SandboxError, match="carries no transcript artifact"): proposer_evidence_entries( results_dir=None, evidence=[row(transcript_artifacts=[])], learnings=[], gate_summary=[], ) def test_proposer_stages_the_bounded_prior_proposal(tmp_path): proposal = tmp_path / "proposal.md" proposal.write_text("# rejected candidate\n\nTightened the plan budget.\n") proposal.chmod(0o600) entries = proposer_evidence_entries( results_dir=None, evidence=[], learnings=[], gate_summary=[], prior_proposal=proposal, ) assert entries["prior-proposal.md"] == proposal.read_text() oversized = tmp_path / "oversized.md" oversized.write_bytes(b"x" * (evolve.MAX_EVIDENCE_FILE_BYTES + 4096)) oversized.chmod(0o600) bounded = proposer_evidence_entries( results_dir=None, evidence=[], learnings=[], gate_summary=[], prior_proposal=oversized, ) assert len(bounded["prior-proposal.md"]) == evolve.MAX_EVIDENCE_FILE_BYTES def test_stage_proposer_evidence_bundle_drops_prior_proposal_to_fit_budget(tmp_path, monkeypatch, capsys): # Per-file caps alone can still exceed the aggregate budget; the helper must # drop the prior proposal instead of aborting the generation. monkeypatch.setattr(evolve, "MAX_BUNDLE_BYTES", 2048) monkeypatch.setattr("workflow_bench.proposer_sandbox.MAX_BUNDLE_BYTES", 2048) prior = tmp_path / "proposal.md" prior.write_text("x" * 2500) prior.chmod(0o600) from workflow_bench.proposer_sandbox import SandboxError, stage_evidence_bundle oversized = evolve.proposer_evidence_entries( results_dir=None, evidence=[], learnings=[], gate_summary=[], prior_proposal=prior, ) with pytest.raises(SandboxError, match="total byte limit"): stage_evidence_bundle(tmp_path / "raw", oversized) bundle = evolve.stage_proposer_evidence_bundle( tmp_path / "bundle", results_dir=None, evidence=[], learnings=[], gate_summary=[], prior_proposal=prior, ) names = {path.name for path in bundle.iterdir()} assert "selected-rows.json" in names assert "prior-proposal.md" not in names logged = capsys.readouterr().out assert "trimmed proposer evidence" in logged assert "omitted prior proposal" in logged def test_stage_proposer_evidence_bundle_compacts_artifacts_before_dropping_rows(tmp_path, monkeypatch, capsys): monkeypatch.setattr(evolve, "MAX_BUNDLE_BYTES", 150_000) monkeypatch.setattr("workflow_bench.proposer_sandbox.MAX_BUNDLE_BYTES", 150_000) results = tmp_path / "results" transcripts = results / "transcripts" transcripts.mkdir(parents=True, mode=0o700) rows = [] for index in range(2): payload = ( json.dumps( { "type": "assistant", "message": {"content": [{"type": "text", "text": "x" * 100_000}]}, } ) + "\n" + json.dumps({"type": "result", "session_id": f"session-{index}"}) + "\n" ).encode() transcript = transcripts / f"session-{index}.jsonl" transcript.write_bytes(payload) transcript.chmod(0o600) (results / f"task-{index}-workflow-run0.patch").write_bytes(b"p" * 100_000) rows.append( row( task=f"task-{index}", transcript_artifacts=[ { "path": f"transcripts/session-{index}.jsonl", "sha256": hashlib.sha256(payload).hexdigest(), "bytes": len(payload), "source": PARENT_EVENT_STREAM_SOURCE, } ], ) ) bundle = evolve.stage_proposer_evidence_bundle( tmp_path / "bundle", results_dir=results, evidence=rows, learnings=[], gate_summary=[], ) staged_rows = json.loads((bundle / "selected-rows.json").read_text()) assert len(staged_rows) == 2 assert all((bundle / staged["patch_file"]).is_file() for staged in staged_rows) logged = capsys.readouterr().out assert "artifact cap" in logged assert "dropped 0 row(s)" in logged @pytest.mark.skipif(os.name == "nt", reason="proposal containment checks are POSIX-only") def test_proposer_refuses_a_prior_proposal_that_lost_its_trust_boundary(tmp_path): outside = tmp_path / "outside.md" outside.write_text("attacker-controlled prose") linked = tmp_path / "linked-proposal.md" linked.symlink_to(outside) with pytest.raises(evolve.SandboxError, match="regular non-symlink"): proposer_evidence_entries( results_dir=None, evidence=[], learnings=[], gate_summary=[], prior_proposal=linked, ) # run_proposer copies the proposal out 0600; anything looser means the # bytes are no longer only the ones this driver wrote. shared = tmp_path / "shared-proposal.md" shared.write_text("proposal") shared.chmod(0o644) with pytest.raises(evolve.SandboxError, match="owner-only"): proposer_evidence_entries( results_dir=None, evidence=[], learnings=[], gate_summary=[], prior_proposal=shared, ) with pytest.raises(evolve.SandboxError, match="unavailable"): proposer_evidence_entries( results_dir=None, evidence=[], learnings=[], gate_summary=[], prior_proposal=tmp_path / "absent.md", ) def test_run_proposer_hides_the_hidden_harness_and_keeps_the_full_tool_surface(monkeypatch, tmp_path): """The proposer writes the artifact the arms are scored with. So its clone must be sanitized before the session starts — a proposer that can read eval/workflow_bench reads the task prompts and hidden oracles it is about to be graded against, and can encode the answers into the skill. """ evidence = tmp_path / "evidence" evidence.mkdir() transcript_projects = tmp_path / "transcript-projects" transcript_projects.mkdir() captured: dict[str, object] = {} sanitized: list[Path] = [] events: list[str] = [] def fake_make_worktree(_repo, _ref, destination): clone = destination / "clone" clone.mkdir() return clone def fake_sanitize(clone): events.append("sanitize") sanitized.append(clone) return "0" * 40 class FakeSandbox: claude_bin = "claude" command_prefix: list[str] = [] settings_json = '{"permissions":{"allow":["Read"]}}' @property def transcript_projects(self): return transcript_projects @contextmanager def fake_prepare_sandbox(**_kwargs): events.append("prepare") yield FakeSandbox() def fake_run_claude(*_args, **kwargs): captured.update(kwargs) return {"ok": False, "error_kind": "session-error"} monkeypatch.setattr(evolve.runner, "make_worktree", fake_make_worktree) monkeypatch.setattr(evolve.runner, "remove_clone", lambda _clone: None) monkeypatch.setattr(evolve, "sanitize_clone_for_hidden_oracles", fake_sanitize) monkeypatch.setattr(evolve, "prepare_sandbox", fake_prepare_sandbox) monkeypatch.setattr(evolve.runner, "run_claude", fake_run_claude) args = build_parser().parse_args(["--tasks", "tasks.yaml", "--model", "model"]) record = evolve.run_proposer( "prompt", args, overlay_dir=tmp_path / "overlay", proposal_path=tmp_path / "proposal.md", evidence_bundle=evidence, bwrap_bin=tmp_path / "bwrap", ) assert record["ok"] is False # Sanitization has to happen on the clone the session actually runs in, # and before the sandbox is prepared around it. assert events[:2] == ["sanitize", "prepare"] assert [clone.name for clone in sanitized] == ["clone"] # Not --bare: bare ignores --tools and would cost the proposer Grep/Glob. assert captured.get("bare", False) is False assert captured["allowed_tools"] == evolve.PROPOSER_ALLOWED_TOOLS assert captured["settings_json"] == FakeSandbox.settings_json def test_proposer_session_cannot_outlive_the_remaining_instance_window(monkeypatch, tmp_path): """Clearing the sweep minimum is not a licence to run a full session. --timeout is sized for a whole generation, so a proposer started with the minimum left would run far past --max-runtime-seconds and the box would take the evidence with it. The budget is sampled after the clone, the sanitize pass and the sandbox setup, because a reading taken before them is already stale by the time the session it bounds actually starts. """ captured: dict[str, object] = {} # Pinned clock: real setup duration would make this assert on scheduling. clock = {"now": 1000.0} monkeypatch.setattr(evolve.time, "monotonic", lambda: clock["now"]) setup_seconds = 100.0 @contextmanager def fake_prepare_sandbox(**_kwargs): yield SimpleNamespace( claude_bin="claude", command_prefix=[], settings_json="{}", transcript_projects=tmp_path / "transcript-projects", ) def fake_run_claude(*_args, **kwargs): captured.update(kwargs) return {"ok": False, "error_kind": "session-error"} def slow_sanitize(_clone): # Stands in for the clone, the sanitize pass and the sandbox build — # all of which run between the caller's decision and the session. clock["now"] += setup_seconds return "0" * 40 monkeypatch.setattr(evolve.runner, "make_worktree", lambda _repo, _ref, destination: destination) monkeypatch.setattr(evolve.runner, "remove_clone", lambda _clone: None) monkeypatch.setattr(evolve, "sanitize_clone_for_hidden_oracles", slow_sanitize) monkeypatch.setattr(evolve, "prepare_sandbox", fake_prepare_sandbox) monkeypatch.setattr(evolve.runner, "run_claude", fake_run_claude) args = build_parser().parse_args(["--tasks", "tasks.yaml", "--model", "model"]) assert args.timeout > evolve.MIN_INSTANCE_SWEEP_SECONDS, "otherwise this test proves nothing" common = { "overlay_dir": tmp_path / "overlay", "proposal_path": tmp_path / "proposal.md", "evidence_bundle": tmp_path / "evidence", "bwrap_bin": tmp_path / "bwrap", } budget = evolve.MIN_INSTANCE_SWEEP_SECONDS + 1 args.max_runtime_seconds = budget started = clock["now"] evolve.run_proposer("prompt", args, **common, started_monotonic=started) # The setup time is charged, not handed back: a value sampled at `started` # would have allowed the whole budget. assert captured["timeout"] == budget - setup_seconds # No cap configured means no budget to overrun: the session keeps its own. args.max_runtime_seconds = None evolve.run_proposer("prompt", args, **common, started_monotonic=started) assert captured["timeout"] == args.timeout evolve.run_proposer("prompt", args, **common) assert captured["timeout"] == args.timeout def test_a_budget_spent_during_setup_stops_the_proposer_rather_than_buying_a_second( monkeypatch, tmp_path ): """An exhausted cap must end the generation, not start a one-second session. remaining_runtime_seconds floors at 0, and the call site wrapped it in max(1, ...) - so a cap fully consumed by the clone, the sanitize pass and the sandbox build produced a paid session with a one-second allowance instead of stopping before the upload reserve the cap exists to protect. """ captured: dict[str, object] = {} clock = {"now": 1000.0} monkeypatch.setattr(evolve.time, "monotonic", lambda: clock["now"]) @contextmanager def fake_prepare_sandbox(**_kwargs): yield SimpleNamespace( claude_bin="claude", command_prefix=[], settings_json="{}", transcript_projects=tmp_path / "transcript-projects", ) def fake_run_claude(*_args, **kwargs): captured.update(kwargs) return {"ok": True} def setup_that_spends_the_whole_budget(_clone): clock["now"] += budget return "0" * 40 monkeypatch.setattr(evolve.runner, "make_worktree", lambda _repo, _ref, destination: destination) monkeypatch.setattr(evolve.runner, "remove_clone", lambda _clone: None) monkeypatch.setattr(evolve, "sanitize_clone_for_hidden_oracles", setup_that_spends_the_whole_budget) monkeypatch.setattr(evolve, "prepare_sandbox", fake_prepare_sandbox) monkeypatch.setattr(evolve.runner, "run_claude", fake_run_claude) args = build_parser().parse_args(["--tasks", "tasks.yaml", "--model", "model"]) budget = evolve.MIN_INSTANCE_SWEEP_SECONDS + 1 args.max_runtime_seconds = budget record = evolve.run_proposer( "prompt", args, overlay_dir=tmp_path / "overlay", proposal_path=tmp_path / "proposal.md", evidence_bundle=tmp_path / "evidence", bwrap_bin=tmp_path / "bwrap", started_monotonic=clock["now"], ) assert not captured, "no session may start once the cap is exhausted" assert record["ok"] is False assert record["error_kind"] == "runtime-cap-exhausted" def test_parser_defaults_match_the_gate_minimums(): args = build_parser().parse_args(["--tasks", "t.yaml", "--model", "pinned"]) assert args.runs == 3 assert args.generations == 1 assert args.arms is None assert args.apply is False assert args.learnings.name == "learnings.jsonl" @pytest.mark.parametrize( "arguments", [ ["--model", "Auto"], ["--model", "provider/latest"], ["--model", "pinned-model", "--proposer-model", "vendor@LATEST"], ], ) def test_evolve_rejects_mutable_model_aliases(monkeypatch, tmp_path, capsys, arguments): monkeypatch.setattr( sys, "argv", ["workflow_bench.evolve", "--tasks", str(tmp_path / "missing.yaml"), *arguments], ) with pytest.raises(SystemExit): evolve.main() assert "mutable auto/latest" in capsys.readouterr().err def test_evolve_proposer_failure_returns_nonzero(monkeypatch, tmp_path): tasks = tmp_path / "tasks.yaml" tasks.write_text( """tasks: - id: demo class: test repo: . prompt: implement verify: "true" oracle: command: "true" files: - source: hidden.test.ts target: hidden.test.ts """ ) monkeypatch.setattr( sys, "argv", [ "workflow_bench.evolve", "--tasks", str(tasks), "--model", "pinned-model", "--out-root", str(tmp_path / "out"), ], ) monkeypatch.setattr(evolve.runner, "selected_task_bindings", lambda _tasks: [{"id": "demo"}]) monkeypatch.setattr(evolve, "preflight_bubblewrap", lambda: tmp_path / "bwrap") monkeypatch.setattr(evolve, "require_claude_sandbox_helpers", lambda: None) monkeypatch.setattr( evolve, "run_proposer", lambda *args, **kwargs: {"ok": False, "error_detail": "proposer failed"}, ) assert evolve.main() == 1 def test_proposer_session_record_is_redacted_before_upload(monkeypatch, tmp_path, capsys): tasks = tmp_path / "tasks.yaml" tasks.write_text( """tasks: - id: demo class: test repo: . prompt: implement verify: "true" oracle: command: "true" files: - source: hidden.test.ts target: hidden.test.ts """ ) literal_token = "secret-LITERAL-XYZ" pattern_token = "sk-ant-FAKEEXAMPLE0000" monkeypatch.setattr( sys, "argv", [ "workflow_bench.evolve", "--tasks", str(tasks), "--model", "pinned-model", "--out-root", str(tmp_path / "out"), "--anthropic-api-key", literal_token, ], ) monkeypatch.setattr(evolve.runner, "selected_task_bindings", lambda _tasks: [{"id": "demo"}]) monkeypatch.setattr(evolve, "preflight_bubblewrap", lambda: tmp_path / "bwrap") monkeypatch.setattr(evolve, "require_claude_sandbox_helpers", lambda: None) # A session error whose stderr echoed both the literal API key and an # sk-ant-shaped token into the record that gets written to the artifact. monkeypatch.setattr( evolve, "run_proposer", lambda *args, **kwargs: { "ok": False, "error_detail": {"stderr_tail": f"boom {literal_token} {pattern_token}"}, }, ) assert evolve.main() == 1 written = (tmp_path / "out" / "gen-0" / "proposer-session.json").read_text() assert literal_token not in written assert pattern_token not in written assert "[REDACTED]" in written # The same record is printed one line later, and the driver's stdout is a # live CI log now that the sweep echoes it — same bar as the artifact. printed = capsys.readouterr().out assert "proposer session failed" in printed assert literal_token not in printed assert pattern_token not in printed assert "[REDACTED]" in printed def test_benchmark_failure_print_is_redacted(monkeypatch, tmp_path, capsys): tasks = tmp_path / "tasks.yaml" tasks.write_text( """tasks: - id: demo class: test repo: . prompt: implement verify: "true" oracle: command: "true" files: - source: hidden.test.ts target: hidden.test.ts """ ) overlay = tmp_path / "overlay" skill = overlay / ".claude" / "skills" / "gitnexus-plan" / "SKILL.md" skill.parent.mkdir(parents=True) skill.write_text("candidate") literal_token = "secret-LITERAL-XYZ" pattern_token = "sk-ant-FAKEEXAMPLE0000" monkeypatch.setattr( sys, "argv", [ "workflow_bench.evolve", "--tasks", str(tasks), "--model", "pinned-model", "--out-root", str(tmp_path / "out"), "--initial-overlay", str(overlay), "--anthropic-api-key", literal_token, ], ) monkeypatch.setattr(evolve.runner, "selected_task_bindings", lambda _tasks: [{"id": "demo"}]) monkeypatch.setattr(evolve, "preflight_bubblewrap", lambda: tmp_path / "bwrap") monkeypatch.setattr(evolve, "require_claude_sandbox_helpers", lambda: None) monkeypatch.setattr(evolve, "resolve_incumbent_arms", lambda *_args, **_kwargs: ["workflow"]) monkeypatch.setattr(evolve, "freeze_overlay", lambda _source, _destination: "d" * 64) monkeypatch.setattr(evolve, "committed_destination_base_digests", lambda _overlay: {}) monkeypatch.setattr(evolve, "destination_base_digests", lambda _overlay: {}) # The sweep is launched with GITNEXUS_BENCH_ANTHROPIC_API_KEY in its environment, # so its detail/stderr tail is as token-bearing as any session record. monkeypatch.setattr( evolve, "run_managed", lambda *_args, **_kwargs: ManagedProcessResult( state="exited", returncode=2, stdout_tail="", stderr_tail=f"ANTHROPIC_API_KEY={pattern_token}", duration_s=1.0, detail=f"sweep died with {literal_token}", ), ) assert evolve.main() == 1 printed = capsys.readouterr().out assert "benchmark run failed" in printed assert literal_token not in printed assert pattern_token not in printed assert "[REDACTED]" in printed def test_runner_argv_pairs_each_incumbent_with_its_candidate(tmp_path): args = build_parser().parse_args( [ "--tasks", "t.yaml", "--model", "pinned", "--arms", "workflow", "--workers", "3", "--include-expensive", ] ) overlay = tmp_path / "overlay" skill = overlay / ".claude" / "skills" / "gitnexus-plan" / "SKILL.md" skill.parent.mkdir(parents=True) skill.write_text("candidate") task_bindings = [{"id": "task", "resolved_sha": "a" * 40}] target_bases = {".claude/skills/gitnexus-plan/SKILL.md": "b" * 64} argv = runner_argv( args, tmp_path / "bench", overlay, task_bindings=task_bindings, target_base_digests=target_bases, proposer_model="pinned", ) arms = argv[argv.index("--arms") + 1 : argv.index("--promotion-metric")] assert arms == ["workflow", "candidate_workflow"] assert str(overlay) in argv assert str(tmp_path / "bench") in argv assert "pinned" in argv assert argv[argv.index("--proposer-model") + 1] == "pinned" assert argv[argv.index("--effort") + 1] == "xhigh" assert argv[argv.index("--workers") + 1] == "3" assert "--include-expensive" in argv assert json.loads(argv[argv.index("--task-bindings-json") + 1]) == task_bindings assert json.loads(argv[argv.index("--promotion-target-bases-json") + 1]) == target_bases def test_runner_argv_inserts_ce_review_for_review_overlay(tmp_path): args = build_parser().parse_args( ["--tasks", "t.yaml", "--model", "pinned", "--arms", "review"] ) overlay = tmp_path / "overlay" skill = overlay / ".claude" / "skills" / "gitnexus-review" / "SKILL.md" skill.parent.mkdir(parents=True) skill.write_text("candidate") argv = runner_argv( args, tmp_path / "bench", overlay, task_bindings=[{"id": "task"}], target_base_digests={}, ) arms = argv[argv.index("--arms") + 1 : argv.index("--promotion-metric")] assert arms == ["ce_review", "review", "candidate_review"] assert "--reuse-results" not in argv def test_runner_argv_forwards_prior_results_for_comparator_reuse(tmp_path): args = build_parser().parse_args( ["--tasks", "t.yaml", "--model", "pinned", "--arms", "review"] ) overlay = tmp_path / "overlay" skill = overlay / ".claude" / "skills" / "gitnexus-review" / "SKILL.md" skill.parent.mkdir(parents=True) skill.write_text("candidate") prior = tmp_path / "prior-bench" argv = runner_argv( args, tmp_path / "bench", overlay, task_bindings=[{"id": "task"}], target_base_digests={}, reuse_results=prior, ) assert argv[argv.index("--reuse-results") + 1] == str(prior) def test_runner_argv_omits_proposer_for_manual_overlay(tmp_path): args = build_parser().parse_args(["--tasks", "t.yaml", "--model", "pinned"]) overlay = tmp_path / "overlay" skill = overlay / ".claude" / "skills" / "gitnexus-plan" / "SKILL.md" skill.parent.mkdir(parents=True) skill.write_text("candidate") argv = runner_argv( args, tmp_path / "bench", overlay, task_bindings=[{"id": "task"}], target_base_digests={}, proposer_model=None, ) assert "--proposer-model" not in argv def test_runner_argv_forwards_explicit_unsafe_backend(tmp_path): args = build_parser().parse_args( ["--tasks", "t.yaml", "--model", "pinned", "--unsafe-no-bwrap"] ) overlay = tmp_path / "overlay" skill = overlay / ".claude" / "skills" / "gitnexus-plan" / "SKILL.md" skill.parent.mkdir(parents=True) skill.write_text("candidate") argv = runner_argv( args, tmp_path / "bench", overlay, task_bindings=[{"id": "task"}], target_base_digests={}, ) assert "--unsafe-no-bwrap" in argv def test_runner_argv_keeps_task_commit_pinned_when_ref_moves(tmp_path): repo = tmp_path / "task-repo" repo.mkdir() def git(*arguments): return subprocess.run( ["git", "-C", str(repo), *arguments], check=True, capture_output=True, text=True, ).stdout.strip() git("init", "-b", "main") git("config", "user.name", "Workflow Bench Test") git("config", "user.email", "workflow-bench@example.invalid") tracked = repo / "tracked.txt" tracked.write_text("one") git("add", "tracked.txt") git("commit", "-m", "first") first_sha = git("rev-parse", "HEAD") task = { "id": "moving-ref", "class": "test", "repo": str(repo), "ref": "main", "prompt": "test prompt", "verify": "true", "oracle": { "command": "true", "files": [ { "source": "trivial-status-json-alias.oracle.test.ts", "target": "oracle.test.ts", } ], }, } bindings = evolve.runner.selected_task_bindings([task]) tracked.write_text("two") git("commit", "-am", "second") assert git("rev-parse", "main") != first_sha args = build_parser().parse_args(["--tasks", "t.yaml", "--model", "pinned"]) overlay = tmp_path / "overlay" skill = overlay / ".claude" / "skills" / "gitnexus-plan" / "SKILL.md" skill.parent.mkdir(parents=True) skill.write_text("candidate") argv = runner_argv( args, tmp_path / "bench", overlay, task_bindings=bindings, target_base_digests={}, ) forwarded = json.loads(argv[argv.index("--task-bindings-json") + 1]) assert forwarded[0]["resolved_sha"] == first_sha assert evolve.runner.resolve_task_bindings([task], forwarded)[0]["resolved_sha"] == first_sha def test_generation_timeout_budgets_three_task_workflow_pair(): timeout = generation_timeout_seconds( task_count=3, runs=3, session_timeout=3600, incumbent_arms=["workflow"], ) per_task_preparation = ( evolve.TASK_BINDING_GIT_PHASES * evolve.GIT_COMMAND_TIMEOUT_SECONDS + 2 * evolve.TASK_SNAPSHOT_TIMEOUT_SECONDS + evolve.WORKTREE_PREPARATION_TIMEOUT_SECONDS + evolve.GRAPH_SOURCE_PREPARATION_TIMEOUT_SECONDS + evolve.GRAPH_BUILD_TIMEOUT_SECONDS + 2 * evolve.GRAPH_QUERY_TIMEOUT_SECONDS + evolve.CLEANUP_TIMEOUT_SECONDS ) paired_arm_cells = 2 session_slots = 4 workspace_snapshot_slots = 4 per_task_run = session_slots * (3600 + evolve.SESSION_FINALIZATION_TIMEOUT_SECONDS) + paired_arm_cells * ( evolve.WORKTREE_PREPARATION_TIMEOUT_SECONDS + evolve.ARM_ASSET_MATERIALIZATION_PHASES * evolve.TASK_SNAPSHOT_TIMEOUT_SECONDS + evolve.SETUP_TIMEOUT_SECONDS + 2 * 3600 + evolve.ARM_EVIDENCE_GIT_PHASES * evolve.GIT_COMMAND_TIMEOUT_SECONDS + evolve.CLEANUP_TIMEOUT_SECONDS ) per_task_run += workspace_snapshot_slots * evolve.TASK_SNAPSHOT_TIMEOUT_SECONDS per_task_run += evolve.CANDIDATE_OVERLAY_GIT_PHASES * evolve.GIT_COMMAND_TIMEOUT_SECONDS assert timeout == ( evolve.PROMOTION_BASE_TIMEOUT_SECONDS + 3 * (per_task_preparation + 3 * per_task_run) + evolve.DRIVER_OVERHEAD_SECONDS ) # The old deadline omitted clone sanitization entirely. Every graph seed # and every paired arm cell must now receive the full bounded envelope. assert timeout >= 3 * (1 + 3 * paired_arm_cells) * evolve.WORKTREE_PREPARATION_TIMEOUT_SECONDS def test_executed_benchmark_arms_inserts_review_comparator() -> None: assert executed_benchmark_arms(["workflow"]) == ["workflow", "candidate_workflow"] assert executed_benchmark_arms(["review"]) == ["ce_review", "review", "candidate_review"] def test_generation_timeout_budgets_review_pair_plus_ce_comparator() -> None: timeout = generation_timeout_seconds( task_count=6, runs=1, session_timeout=3600, incumbent_arms=["review"], ) per_task_preparation = ( evolve.TASK_BINDING_GIT_PHASES * evolve.GIT_COMMAND_TIMEOUT_SECONDS + 2 * evolve.TASK_SNAPSHOT_TIMEOUT_SECONDS + evolve.WORKTREE_PREPARATION_TIMEOUT_SECONDS + evolve.GRAPH_SOURCE_PREPARATION_TIMEOUT_SECONDS + evolve.GRAPH_BUILD_TIMEOUT_SECONDS + 2 * evolve.GRAPH_QUERY_TIMEOUT_SECONDS + evolve.CLEANUP_TIMEOUT_SECONDS ) paired_arm_cells = 3 session_slots = 3 workspace_snapshot_slots = 3 per_task_run = session_slots * (3600 + evolve.SESSION_FINALIZATION_TIMEOUT_SECONDS) + paired_arm_cells * ( evolve.WORKTREE_PREPARATION_TIMEOUT_SECONDS + evolve.ARM_ASSET_MATERIALIZATION_PHASES * evolve.TASK_SNAPSHOT_TIMEOUT_SECONDS + evolve.SETUP_TIMEOUT_SECONDS + 2 * 3600 + evolve.ARM_EVIDENCE_GIT_PHASES * evolve.GIT_COMMAND_TIMEOUT_SECONDS + evolve.CLEANUP_TIMEOUT_SECONDS ) per_task_run += workspace_snapshot_slots * evolve.TASK_SNAPSHOT_TIMEOUT_SECONDS per_task_run += evolve.CANDIDATE_OVERLAY_GIT_PHASES * evolve.GIT_COMMAND_TIMEOUT_SECONDS assert timeout == ( evolve.PROMOTION_BASE_TIMEOUT_SECONDS + 6 * (per_task_preparation + per_task_run) + evolve.DRIVER_OVERHEAD_SECONDS ) def test_generation_timeout_rejects_unknown_arm() -> None: with pytest.raises(ValueError, match="unsupported evolution arm: mystery"): generation_timeout_seconds( task_count=1, runs=1, session_timeout=60, incumbent_arms=["mystery"], ) def test_instance_window_budget_leaves_upload_reserve() -> None: # Friday 10:57 on a box that booted 02:45 Saturday-window: ~8.2h uptime. leftover = instance_window_budget_seconds(8.2 * 3600) assert leftover == int(86_400 - 8.2 * 3600 - 5_400) assert leftover >= MIN_INSTANCE_SWEEP_SECONDS with pytest.raises(ValueError, match="only .*s left"): instance_window_budget_seconds(23.5 * 3600) with pytest.raises(ValueError, match="uptime must be"): instance_window_budget_seconds(float("nan")) def test_capped_timeout_clamps_to_leftover_window(monkeypatch) -> None: assert capped_timeout_seconds(10_000, None) == 10_000 assert capped_timeout_seconds(10_000, 90) == 90 with pytest.raises(ValueError, match="no time remains"): capped_timeout_seconds(10_000, 0) # Pinned clock: the helper is pure arithmetic, so a real elapsed-time window # would assert on scheduling rather than on the behaviour under test. monkeypatch.setattr(evolve.time, "monotonic", lambda: 1040.0) started = 1000.0 assert remaining_runtime_seconds(max_runtime_seconds=None, started_monotonic=started) is None leftover = remaining_runtime_seconds(max_runtime_seconds=100, started_monotonic=started) assert leftover is not None assert leftover == 60 def test_parser_rejects_non_positive_max_runtime() -> None: with pytest.raises(SystemExit): build_parser().parse_args( ["--tasks", "t.yaml", "--model", "pinned", "--max-runtime-seconds", "0"] ) args = build_parser().parse_args( ["--tasks", "t.yaml", "--model", "pinned", "--max-runtime-seconds", "7200"] ) assert args.max_runtime_seconds == 7200 assert args.max_runtime_from_instance_window is False def _task_file(tmp_path: Path) -> Path: tasks = tmp_path / "tasks.yaml" tasks.write_text( """tasks: - id: demo class: test repo: . prompt: implement verify: "true" oracle: command: "true" files: - source: hidden.test.ts target: hidden.test.ts """ ) return tasks def _stub_main_preflight(monkeypatch, tmp_path) -> None: """Everything main() shells out to before it reaches _run_generations.""" monkeypatch.setattr(evolve.runner, "selected_task_bindings", lambda _tasks: [{"id": "demo"}]) monkeypatch.setattr(evolve, "preflight_bubblewrap", lambda: tmp_path / "bwrap") monkeypatch.setattr(evolve, "require_claude_sandbox_helpers", lambda: None) def test_the_runtime_cap_is_derived_where_its_clock_starts(monkeypatch, tmp_path, capsys) -> None: """The budget and the clock it is measured against must be one instant. run-evolution.sh used to compute the budget in a separate `uv run python -c` and pass a number, so the script's remaining provenance work and this interpreter's startup were charged to the sweep — out of the upload reserve the cap exists to protect. main() reads /proc/uptime itself now, next to its own clock, so no interval exists to lose. """ monkeypatch.setenv("EVENTBRIDGE_INSTANCE_WINDOW_SECONDS", "20000") monkeypatch.setenv("EVENTBRIDGE_STOP_RESERVE_SECONDS", "1000") monkeypatch.setattr(evolve, "read_instance_uptime_seconds", lambda: 3600.0) captured: dict[str, object] = {} def record(args, **kwargs): captured["max_runtime_seconds"] = args.max_runtime_seconds captured["started_monotonic"] = kwargs["started_monotonic"] return 0 monkeypatch.setattr(evolve, "_run_generations", record) _stub_main_preflight(monkeypatch, tmp_path) monkeypatch.setattr( sys, "argv", [ "evolve", "--tasks", str(_task_file(tmp_path)), "--model", "pinned", "--out-root", str(tmp_path / "out"), "--max-runtime-from-instance-window", ], ) assert evolve.main() == 0 assert captured["max_runtime_seconds"] == 20000 - 3600 - 1000 # Derived here, not passed in: the clock handed to the sweep is the one # taken beside the uptime read. assert isinstance(captured["started_monotonic"], float) assert "capping the sweep to 15400s" in capsys.readouterr().out def test_the_runtime_cap_refuses_two_sources_of_truth(monkeypatch, tmp_path) -> None: monkeypatch.setattr(evolve, "read_instance_uptime_seconds", lambda: 3600.0) monkeypatch.setattr( sys, "argv", [ "evolve", "--tasks", str(_task_file(tmp_path)), "--model", "pinned", "--max-runtime-from-instance-window", "--max-runtime-seconds", "7200", ], ) with pytest.raises(SystemExit): evolve.main() def test_the_runtime_cap_fails_closed_without_a_readable_uptime(monkeypatch, tmp_path) -> None: def unreadable(): raise ValueError("cannot read instance uptime from /proc/uptime") monkeypatch.setattr(evolve, "read_instance_uptime_seconds", unreadable) monkeypatch.setattr( sys, "argv", [ "evolve", "--tasks", str(_task_file(tmp_path)), "--model", "pinned", "--max-runtime-from-instance-window", ], ) # Better to refuse than to run a box-stopped sweep believing it is uncapped. with pytest.raises(SystemExit): evolve.main() @pytest.mark.skipif(sys.platform != "linux", reason="Bubblewrap PID namespaces require Linux") def test_outer_runner_pid_namespace_kills_setsid_descendant(tmp_path): try: bwrap = preflight_bubblewrap() except evolve.SandboxError as exc: pytest.skip(str(exc)) raise AssertionError("pytest.skip() returned unexpectedly") sentinel = tmp_path / "escaped" child = ( "import os,subprocess,sys,time; " f"subprocess.Popen([sys.executable,'-c',\"import time,pathlib;time.sleep(1);pathlib.Path({str(sentinel)!r}).touch()\"],preexec_fn=os.setsid); " "time.sleep(10)" ) result = run_managed( pid_namespace_command([sys.executable, "-c", child], bwrap_bin=bwrap), timeout=0.15, terminate_grace=0.1, require_pid_namespace=True, ) time.sleep(1.1) assert not result.ok assert result.state in {"timeout", "forced-kill"} assert not sentinel.exists() def test_resolve_incumbent_arms_rejects_incomplete_and_extra_explicit_sets(tmp_path): plan = tmp_path / "plan" plan_skill = plan / ".claude" / "skills" / "gitnexus-plan" / "SKILL.md" plan_skill.parent.mkdir(parents=True) plan_skill.write_text("plan") assert resolve_incumbent_arms(plan, None) == ["workflow"] with pytest.raises(ValueError, match="exactly"): resolve_incumbent_arms(plan, ["workflow", "workflow_direct"]) work = tmp_path / "work" work_skill = work / ".claude" / "skills" / "gitnexus-work" / "SKILL.md" work_skill.parent.mkdir(parents=True) work_skill.write_text("work") assert resolve_incumbent_arms(work, None) == ["workflow", "workflow_direct"] with pytest.raises(ValueError, match="exactly"): resolve_incumbent_arms(work, ["workflow"]) def bound_task_fixture(task_id="task-a"): return { "id": task_id, "prompt_digest": "prompt", "oracle_digest": "a" * 64, "oracle_command_digest": "b" * 64, "oracle_manifest_digest": "c" * 64, "sandbox_dependency_content_digest": "e" * 64, "sandbox_dependency_manifest_digest": "f" * 64, "oracle_files": [{"target": "oracle.test.ts", "sha256": "d" * 64, "size": 10}], } def bound_task_fixtures(): return [bound_task_fixture("task-a"), bound_task_fixture("task-impossible")] def promote_decision(**overrides): """A real producer decision, including its recomputable paired metrics.""" base = {"runs": 3, "valid_runs": 3, "excluded_runs": 0, "error_kinds": {}} results = { "task-a": { "workflow": {**base, "resolved": 3, "cost_usd": 1.0}, "candidate_workflow": {**base, "resolved": 3, "cost_usd": 0.8}, }, "task-impossible": { "workflow": {**base, "resolved": 0, "cost_usd": 1.0}, "candidate_workflow": {**base, "resolved": 0, "cost_usd": 1.0}, }, } decision = evolution.promotion_evidence( results, policy=evolution.promotion_policy(["candidate_workflow"]), model="bench-model", complete=True, )["decisions"][0] decision.update(overrides) return decision def promotion_fixture(*, decisions=None, expires_delta=timedelta(days=1)): now = datetime.now(UTC) return { "schema_version": 6, "run_status": "complete", "generated_at": now.isoformat(), "evidence_expires_at": (now + expires_delta).isoformat(), "benchmark_model": "bench-model", "proposer_model": "proposer-model", "effort": "xhigh", "candidate_origin": "model-proposer", "candidate_overlay_digest": "digest", "target_base_digests": {"path": "base"}, "required_candidate_arms": ["candidate_workflow"], "selected_tasks": bound_task_fixtures(), "policy": evolution.promotion_policy(["candidate_workflow"]), "decisions": decisions if decisions is not None else [promote_decision()], } def validate_fixture(promotion): return validate_promotion_for_apply( promotion, overlay_digest="digest", benchmark_model="bench-model", proposer_model="proposer-model", effort="xhigh", selected_tasks=bound_task_fixtures(), target_base_digests={"path": "base"}, required_candidate_arms=["candidate_workflow"], policy=evolution.promotion_policy(["candidate_workflow"]), ) def test_promotion_apply_requires_one_promote_for_every_bound_arm(): assert [d["candidate_arm"] for d in validate_fixture(promotion_fixture())] == ["candidate_workflow"] for decisions in ( [], [promote_decision(decision="keep_incumbent")], [promote_decision(), promote_decision()], [promote_decision(incumbent_arm="workflow_direct", candidate_arm="candidate_workflow_direct")], ): with pytest.raises(ValueError): validate_fixture(promotion_fixture(decisions=decisions)) @pytest.mark.parametrize( ("overrides", "match"), [ # An older decision relabeled as schema 5: the verdict without the # gated evidence base schema 5 promotes on. ({"tasks": None, "ungated_tasks": None}, "no per-task gate evidence"), ({"tasks": [{"task": "task-a"}]}, "malformed per-task gate evidence"), ({"tasks": [{"task": "task-a", "gated": "yes"}]}, "malformed per-task gate evidence"), ( {"tasks": [{"task": "task-a", "gated": True}, {"task": "task-a", "gated": False}]}, "repeats a task", ), ( {"ungated_tasks": [], "tasks": [{"task": "fabricated", "gated": True}]}, "does not match selected tasks", ), ({"ungated_tasks": None}, "missing its ungated task list"), # The verdict claims a full gate; the per-task rows say a task sat # outside it. ({"ungated_tasks": []}, "disagree with its per-task evidence"), ( { "ungated_tasks": ["task-a", "task-impossible"], "tasks": [ {"task": "task-a", "gated": False}, {"task": "task-impossible", "gated": False}, ], }, "no gated task", ), ], ) def test_promotion_apply_binds_the_schema_5_gate_evidence(overrides, match): decision = promote_decision() for field, value in overrides.items(): if value is None: decision.pop(field) else: decision[field] = value with pytest.raises(ValueError, match=match): validate_fixture(promotion_fixture(decisions=[decision])) def test_promotion_apply_rejects_a_gate_with_only_one_of_three_selected_tasks(): selected = [*bound_task_fixtures(), bound_task_fixture("task-impossible-2")] decision = promote_decision( ungated_tasks=["task-impossible", "task-impossible-2"], tasks=[ {"task": "task-a", "gated": True}, {"task": "task-impossible", "gated": False}, {"task": "task-impossible-2", "gated": False}, ], ) promotion = promotion_fixture(decisions=[decision]) promotion["selected_tasks"] = selected with pytest.raises(ValueError, match="too thin a gated evidence base"): validate_promotion_for_apply( promotion, overlay_digest="digest", benchmark_model="bench-model", proposer_model="proposer-model", effort="xhigh", selected_tasks=selected, target_base_digests={"path": "base"}, required_candidate_arms=["candidate_workflow"], policy=promotion["policy"], ) def test_manual_initial_overlay_has_no_fictitious_proposer_model(): promotion = promotion_fixture() promotion["proposer_model"] = None promotion["candidate_origin"] = "manual-initial-overlay" decisions = validate_promotion_for_apply( promotion, overlay_digest="digest", benchmark_model="bench-model", proposer_model=None, effort="xhigh", selected_tasks=bound_task_fixtures(), target_base_digests={"path": "base"}, required_candidate_arms=["candidate_workflow"], policy=promotion["policy"], ) assert decisions[0]["decision"] == "promote" def test_promotion_apply_rejects_pre_oracle_schema_and_missing_oracle_bindings(): legacy = promotion_fixture() legacy["schema_version"] = 3 with pytest.raises(ValueError, match="unsupported schema"): validate_fixture(legacy) weak_task = {"id": "task", "prompt_digest": "prompt"} weak = promotion_fixture() weak["selected_tasks"] = [weak_task] with pytest.raises(ValueError, match="hidden-oracle or dependency digests"): validate_promotion_for_apply( weak, overlay_digest="digest", benchmark_model="bench-model", proposer_model="proposer-model", effort="xhigh", selected_tasks=[weak_task], target_base_digests={"path": "base"}, required_candidate_arms=["candidate_workflow"], policy={ "metric": "cost_usd", "min_runs": 3, "min_improvement_pct": 5.0, "max_task_regression_pct": 20.0, }, ) @pytest.mark.parametrize( ("field", "value"), [ ("benchmark_model", "other"), ("proposer_model", "other"), ("candidate_overlay_digest", "other"), ("target_base_digests", {"path": "other"}), ("required_candidate_arms", ["candidate_workflow_direct"]), ("selected_tasks", [{"id": "other", "prompt_digest": "prompt"}]), ( "policy", { "metric": "cost_usd", "min_runs": 4, "min_improvement_pct": 5.0, "max_task_regression_pct": 20.0, }, ), ], ) def test_promotion_apply_rejects_mismatched_evidence_bindings(field, value): promotion = promotion_fixture() promotion[field] = value with pytest.raises(ValueError, match="binding"): validate_fixture(promotion) def test_promotion_apply_rejects_expired_evidence(): with pytest.raises(ValueError, match="expired"): validate_fixture(promotion_fixture(expires_delta=timedelta(seconds=-1))) def test_promotion_apply_rejects_extended_or_future_dated_evidence(): with pytest.raises(ValueError, match="expired"): validate_fixture(promotion_fixture(expires_delta=timedelta(days=91))) promotion = promotion_fixture() future = datetime.now(UTC) + timedelta(days=1) promotion["generated_at"] = future.isoformat() promotion["evidence_expires_at"] = (future + timedelta(days=1)).isoformat() with pytest.raises(ValueError, match="future"): validate_fixture(promotion) def test_promotion_apply_rejects_decision_metric_mismatch(): promotion = promotion_fixture() promotion["decisions"][0]["metric"] = "output_tokens" with pytest.raises(ValueError, match="metric mismatch"): validate_fixture(promotion)