From c3eb5991c1108446a5f6a002a593d5cd4f2185aa Mon Sep 17 00:00:00 2001 From: Abhinav Pandey Date: Thu, 3 Sep 2026 05:18:46 +0530 Subject: [PATCH] fix(eval): preserve complete evolution evidence --- .../workflows/gitnexus-skill-evolution.yml | 25 ++++++++ eval/tests/test_runner_hardening.py | 8 +-- eval/workflow_bench/README.md | 19 ++++-- eval/workflow_bench/runner.py | 20 +++++-- .../unit/skill-evolution-workflow.test.ts | 60 +++++++++++++++++-- 5 files changed, 111 insertions(+), 21 deletions(-) diff --git a/.github/workflows/gitnexus-skill-evolution.yml b/.github/workflows/gitnexus-skill-evolution.yml index cc8d33e56..51ac148d1 100644 --- a/.github/workflows/gitnexus-skill-evolution.yml +++ b/.github/workflows/gitnexus-skill-evolution.yml @@ -307,6 +307,18 @@ jobs: echo "::warning::Evidence from run ${previous} could not be downloaded (expired or absent); trying an older run." continue fi + unsafe="$(find "${run_root}" ! -type d ! -type f -print -quit)" + if [[ -n "${unsafe}" ]]; then + echo "::warning::Run ${previous} contains a non-regular artifact entry; trying an older run." + continue + fi + # upload-artifact normalizes directories/files to 0755/0644, while + # the evidence reader deliberately requires transcript paths to be + # owner-only. Restore that trust-boundary invariant after download. + if ! chmod -R go-rwx "${run_root}"; then + echo "::warning::Evidence permissions from run ${previous} could not be restricted; trying an older run." + continue + fi # The artifact holds gen-N/bench/{results.jsonl,promotion.json,...}; # the highest generation is the one that actually reached the gate. latest="$(find "${run_root}" -type f -path '*/gen-*/bench/results.jsonl' | sort -V | tail -1)" @@ -314,6 +326,19 @@ jobs: echo "::warning::Run ${previous} uploaded no usable benchmark results; trying an older run." continue fi + # Existence is insufficient: an interrupted run may leave an empty, + # malformed, or session/infra-only JSONL. Reuse the same bounded + # selection and transcript/digest preflight the proposer will use, + # so an unusable newer run cannot hide an older useful one. + if uv run --project eval --locked --extra dev python -c \ + 'from pathlib import Path; import json, sys; from workflow_bench.evolve import load_jsonl, proposer_evidence_entries, select_evidence, summarize_gate; result = Path(sys.argv[1]); root = result.parent; rows = select_evidence(load_jsonl(result)); rows or sys.exit(10); promotion = root / "promotion.json"; gate = summarize_gate(json.loads(promotion.read_text())) if promotion.is_file() else []; proposer_evidence_entries(results_dir=root, evidence=rows, learnings=[], gate_summary=gate)' \ + "${latest}"; then + : + else + usability_status=$? + echo "::warning::Run ${previous} failed evidence preflight (exit ${usability_status}); trying an older run." + continue + fi seed="$(dirname "${latest}")" echo "Seeding the proposer from run ${previous}: ${seed}" break diff --git a/eval/tests/test_runner_hardening.py b/eval/tests/test_runner_hardening.py index 45f543f77..1b766a0a9 100644 --- a/eval/tests/test_runner_hardening.py +++ b/eval/tests/test_runner_hardening.py @@ -625,16 +625,16 @@ def test_sweep_keeps_rows_in_submission_order_whatever_order_they_finish(tmp_pat @pytest.mark.parametrize("workers", [1, 2, 3]) def test_sweep_trips_the_breaker_within_one_wave_of_the_serial_point(workers): # Serial stops after the 5th consecutive systemic failure. Cells already in - # flight when the breaker trips cannot be recalled, so the overrun is - # bounded by the wave — the point of waves is that it is never the whole - # task. Ten cells, so the bound is visible rather than hidden by the end. + # flight when the breaker trips cannot be recalled or erased from the + # evidence, so the overrun is bounded by the wave and every completed row + # is kept. Ten cells make the bound visible rather than hidden by the end. long_task = [(run_idx, arm) for run_idx in range(5) for arm in ("workflow", "candidate_workflow")] result = _sweep(long_task, workers=workers, run=lambda *_: _row("session-error")) assert result.tripped is True - assert len(result.kept) == 5 assert 5 <= len(result.started) <= 5 + workers - 1 + assert result.kept == result.started assert len(result.started) < len(long_task) diff --git a/eval/workflow_bench/README.md b/eval/workflow_bench/README.md index 3ba904607..3133a1e41 100644 --- a/eval/workflow_bench/README.md +++ b/eval/workflow_bench/README.md @@ -161,7 +161,7 @@ paid work. For a work overlay: cd eval uv run --locked --extra dev python -m workflow_bench.runner \ --tasks workflow_bench/tasks.scenarios.yaml \ - --runs 3 --model claude-sonnet-4-20250514 \ + --runs 3 --workers 1 --model claude-sonnet-4-20250514 \ --arms workflow candidate_workflow \ workflow_direct candidate_workflow_direct \ --candidate-overlay /tmp/gn-skill-candidate @@ -177,7 +177,7 @@ artifacts. Those artifacts are the trajectory evidence: cluster failures and expensive detours, propose one bounded prompt change, and feed it back as the next overlay. -When candidate arms are present the runner also writes schema-3 +When candidate arms are present the runner also writes schema-4 `promotion.json`. It binds the immutable overlay digest, benchmark model, truthful candidate origin (a named proposer model or `manual-initial-overlay`), selected @@ -188,7 +188,11 @@ deterministic gate is deliberately conservative: - at least 3 paired VALID runs per task, zero excluded runs in either arm (session/infra-error rows therefore block promotion), and a named model; -- the candidate must pass the hidden oracle on every valid run for every task; +- a fully measured task that neither arm ever resolves remains reported but is + ungated from quality and efficiency comparisons; if every task is ungated, + the generation is `insufficient_evidence`; +- the candidate must pass the hidden oracle on every valid run for every gated + task; - no per-task resolution-rate regression (quality is lexicographically first); - promotion by resolution needs a margin of at least 2 resolved runs — a 1-run difference is noise at this run count and falls through to the @@ -227,6 +231,7 @@ cd eval uv run --locked --extra dev python -m workflow_bench.evolve \ --tasks workflow_bench/tasks.scenarios.yaml \ --model claude-sonnet-4-20250514 --generations 2 \ + --workers 1 \ --seed-results results/wfbench- # optional gen-0 evidence ``` @@ -258,9 +263,11 @@ ground truth: a learning only reaches a shipped skill by surviving the same paired benchmark as any other candidate. Legacy review/LFG rows are ignored; those skills do not yet have honest candidate lanes or promotion gates. -Run the driver on the existing re-evaluation triggers (model/harness change, -90-day staleness), not on a tight schedule — every generation costs ≥3 paired -runs per task, and `--generations` is the only loop bound. +For ad-hoc use, run the driver on the existing re-evaluation triggers +(model/harness change or 90-day staleness). The repository workflow runs a +deliberate weekly drift check: scheduled concurrency stays serial unless +`GITNEXUS_EVOLUTION_WORKERS` is raised after a funded host-sized proof, and +`--generations` remains the only loop bound. ## Free-model setup (no paid tokens) diff --git a/eval/workflow_bench/runner.py b/eval/workflow_bench/runner.py index a0edbe399..ffd71edb0 100644 --- a/eval/workflow_bench/runner.py +++ b/eval/workflow_bench/runner.py @@ -685,16 +685,24 @@ def sweep_task_cells( # for it, so skipping this would turn a harness bug into a # silently missing run rather than a crash. records = [future.result() for future in futures] + wave_tripped = False for (run_idx, arm), record in zip(wave, records, strict=True): + # Every future in this wave has already completed and incurred its + # cost. Persist all of them in canonical submission order even if + # an earlier row trips the breaker; only later waves are skipped. on_record(run_idx, arm, record) + if wave_tripped: + continue outage_streak = systemic_outage_streak(record.get("error_kind"), outage_streak) if outage_limit and outage_streak >= outage_limit: - print( - f"[systemic-outage] {outage_streak} consecutive session/infra/cleanup " - "failures — aborting the remaining sweep; report and promotion are written " - "from partial evidence and the run exits non-zero." - ) - return outage_streak, True + wave_tripped = True + if wave_tripped: + print( + f"[systemic-outage] {outage_streak} consecutive session/infra/cleanup " + "failures — aborting the remaining sweep; report and promotion are written " + "from partial evidence and the run exits non-zero." + ) + return outage_streak, True return outage_streak, False diff --git a/gitnexus/test/unit/skill-evolution-workflow.test.ts b/gitnexus/test/unit/skill-evolution-workflow.test.ts index abf804b80..705cd0740 100644 --- a/gitnexus/test/unit/skill-evolution-workflow.test.ts +++ b/gitnexus/test/unit/skill-evolution-workflow.test.ts @@ -1,5 +1,14 @@ import { execFileSync } from 'node:child_process'; -import { chmodSync, mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs'; +import { + chmodSync, + existsSync, + mkdirSync, + mkdtempSync, + readFileSync, + rmSync, + statSync, + writeFileSync, +} from 'node:fs'; import os from 'node:os'; import path from 'node:path'; import { load } from 'js-yaml'; @@ -47,7 +56,12 @@ function stepRun(stepName: string): string { return typeof step?.run === 'string' ? step.run : ''; } -function runSeedStep(ghImplementation: string): { output: string; trace: string } { +function runSeedStep(ghImplementation: string): { + output: string; + trace: string; + transcriptDirectoryMode?: number; + transcriptMode?: number; +} { const root = mkdtempSync(path.join(os.tmpdir(), 'gitnexus-evolution-seed-')); try { const bin = path.join(root, 'bin'); @@ -60,6 +74,19 @@ function runSeedStep(ghImplementation: string): { output: string; trace: string const gh = path.join(bin, 'gh'); writeFileSync(gh, `#!/usr/bin/env bash\nset -euo pipefail\n${ghImplementation}\n`); chmodSync(gh, 0o700); + const uv = path.join(bin, 'uv'); + writeFileSync( + uv, + `#!/usr/bin/env bash +set -euo pipefail +results='' +for argument in "$@"; do results="$argument"; done +content="$(cat "$results")" +if [[ -z "$content" || "$content" == *'"error_kind":"session-error"'* ]]; then exit 10; fi +exit 0 +`, + ); + chmodSync(uv, 0o700); execFileSync( '/bin/bash', @@ -77,9 +104,21 @@ function runSeedStep(ghImplementation: string): { output: string; trace: string stdio: 'pipe', }, ); + const output = readFileSync(githubOutput, 'utf8'); + const seed = output.match(/^seed=(.+)$/m)?.[1]; + const transcriptDirectory = seed ? path.join(seed, 'transcripts') : undefined; + const transcript = transcriptDirectory + ? path.join(transcriptDirectory, 'session.jsonl') + : undefined; return { - output: readFileSync(githubOutput, 'utf8'), + output, trace: readFileSync(trace, 'utf8'), + transcriptDirectoryMode: + transcriptDirectory && existsSync(transcriptDirectory) + ? statSync(transcriptDirectory).mode & 0o777 + : undefined, + transcriptMode: + transcript && existsSync(transcript) ? statSync(transcript).mode & 0o777 : undefined, }; } finally { rmSync(root, { recursive: true, force: true }); @@ -117,6 +156,8 @@ describe('gitnexus skill-evolution workflow contract', () => { expect(seed).toContain('for previous in ${previous_runs}'); expect(seed).toContain('continue'); expect(seed).toContain('gen-*/bench/results.jsonl'); + expect(seed).toContain('chmod -R go-rwx'); + expect(seed).toContain('select_evidence(load_jsonl'); expect(seed).toContain('break'); }); @@ -136,9 +177,16 @@ if [[ "$1 $2" == 'run download' ]]; then if [[ "$1" == '--dir' ]]; then destination="$2"; shift 2; else shift; fi done printf '%s\\n' "\${run_id}" >> "\${TRACE}" - if [[ "\${run_id}" == '200' ]]; then + if [[ "\${run_id}" == '300' ]]; then + mkdir -p "\${destination}/artifact/gen-3/bench" + printf '%s\\n' '{"error_kind":"session-error","resolved":false}' > "\${destination}/artifact/gen-3/bench/results.jsonl" + elif [[ "\${run_id}" == '200' ]]; then mkdir -p "\${destination}/artifact/gen-2/bench" - printf '{}\\n' > "\${destination}/artifact/gen-2/bench/results.jsonl" + mkdir -p "\${destination}/artifact/gen-2/bench/transcripts" + printf '%s\\n' '{"task":"demo","arm":"workflow","run":0,"resolved":false,"error_kind":"oracle-failed","transcript_artifacts":[{"path":"transcripts/session.jsonl","sha256":"ca3d163bab055381827226140568f3bef7eaac187cebd76878e0b63e9e442356","bytes":3,"source":"parent-captured-stream-json"}]}' > "\${destination}/artifact/gen-2/bench/results.jsonl" + printf '{}\\n' > "\${destination}/artifact/gen-2/bench/transcripts/session.jsonl" + chmod 0755 "\${destination}/artifact/gen-2/bench/transcripts" + chmod 0644 "\${destination}/artifact/gen-2/bench/transcripts/session.jsonl" fi exit 0 fi @@ -146,6 +194,8 @@ exit 1`); expect(result.trace).toBe('300\n200\n'); expect(result.output).toMatch(/seed=.*\/200\/artifact\/gen-2\/bench\n/); + expect(result.transcriptDirectoryMode).toBe(0o700); + expect(result.transcriptMode).toBe(0o600); }, );