From 01be282667566c769d5c52162b9ed2200fa59e6c Mon Sep 17 00:00:00 2001 From: Gergo Magyar Date: Sat, 1 Aug 2026 17:42:48 +0000 Subject: [PATCH] fix(ci): make the evolution lane survive its own deadlines and remember prior runs MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit An end-to-end pass over the lane — instance start, job, artifacts, promotion — found three ways it loses work that has already been paid for. **Evidence died with the job.** Three budgets have to nest: EventBridge keeps the box up 24h from ~02:45, the job timeout was also 1440min, and the sweep had no budget of its own. A job-level timeout CANCELS the job, so the upload step never runs; and since the box stops 24h after it starts while a scheduled run can begin well after the cron (the 2026-08-01 run was queued 65min late), the box always won that race — the runner would simply vanish mid-step. The job now gets 21h, the sweep step 19h, so a wedged generation fails the step, keeps the job alive, and still uploads. The nesting is asserted in the contract test. **The upload could be skipped.** Its path came from an output the sweep step wrote — the same step whose death is the reason the upload matters. OUT_ROOT is now a job-level env constant known before anything runs, and the upload is unconditional: results.jsonl and transcripts are appended as the sweep goes, so a killed generation still holds the evidence that explains why it died. **The lane was memoryless.** `--seed-results` is how a run sees what already lost (summarize_gate feeds the prior promotion.json to the proposer), and with the default --generations 1 there is no earlier generation in-process to supply it — the workflow never passed it, so every Saturday proposed from a blank slate and could re-propose the same rejected candidate forever. The lane now seeds from the last successful run's artifact, best-effort: a first run, an expired artifact, a missing gh, or a failed download proceeds without it rather than costing a generation. Also guards the silent-promotion path: `.claude/skills/*` is gitignored with a hand-maintained per-skill allowlist, and `git status --porcelain` — how the workflow detects an applied promotion — is blind to ignored paths. A candidate skill missing from that allowlist would report "No promotion this run" after the gate said promote. A test now asserts every CANDIDATE_SKILLS entry is visible in all three shipped trees. --- .../workflows/gitnexus-skill-evolution.yml | 84 +++++++++++++++++-- eval/tests/test_workflow_bench_evolution.py | 28 +++++++ .../unit/skill-evolution-workflow.test.ts | 32 +++++++ 3 files changed, 136 insertions(+), 8 deletions(-) diff --git a/.github/workflows/gitnexus-skill-evolution.yml b/.github/workflows/gitnexus-skill-evolution.yml index 3b2d40373..329640e78 100644 --- a/.github/workflows/gitnexus-skill-evolution.yml +++ b/.github/workflows/gitnexus-skill-evolution.yml @@ -111,15 +111,30 @@ jobs: # it — server-side enforcement a dispatched non-main ref cannot bypass by # editing its own workflow copy. See the activation checklist above. environment: gitnexus-evolution - timeout-minutes: 1440 # self-hosted ceiling is 5 days (7200min); 24h is a generous margin over a single-generation serial run + # Three budgets have to nest, longest first, or the evidence is lost: + # EventBridge instance uptime (24h from ~02:45) + # > this job timeout (21h) + # > the benchmark step timeout (19h, set on the step below) + # A job-level timeout CANCELS the job, so the upload step never runs and a + # multi-hour generation's evidence dies with it; a step-level timeout only + # fails that step, and `if: always()` still uploads what the sweep wrote. + # The instance must outlive the job for the same reason — when the box + # stops the runner just disappears mid-step. Scheduled runs can start well + # after the cron (the 2026-08-01 run was queued 65min late), so the job + # budget has to absorb that delay and still land inside the uptime window. + timeout-minutes: 1260 permissions: contents: read # The promotion PR uses a short-lived App token minted below. + actions: read # Read the previous run's evidence artifact to seed the proposer. env: GENERATIONS: ${{ inputs.generations || '1' }} RUNS: ${{ inputs.runs || '3' }} MODEL: ${{ inputs.model || 'claude-sonnet-5' }} PROPOSER_MODEL: ${{ inputs.proposer_model || 'claude-opus-4-8' }} INCLUDE_EXPENSIVE: ${{ inputs.include_expensive && '1' || '' }} + # Fixed, known before the sweep starts: the upload below must not depend + # on a step that may have been killed having reported an output. + OUT_ROOT: ${{ runner.temp }}/wfevolve steps: - name: Require the benchmark auth secret env: @@ -214,22 +229,74 @@ jobs: # and mounts dependencies read-only, so the checkout is never mutated. ln -sfn "${GITHUB_WORKSPACE}" "${HOME}/GitNexus" + - name: Seed the proposer with the previous run's evidence + id: seed + env: + GH_TOKEN: ${{ github.token }} + run: | + set -euo pipefail + # Without this the weekly lane is memoryless: `--seed-results` is the + # only way a run sees what already lost (evolve.summarize_gate feeds + # the prior promotion.json to the proposer as "what already lost"), + # and with the default --generations 1 there is no earlier generation + # in-process to supply it. Every Saturday would otherwise propose + # from a blank slate and could re-propose the same rejected candidate + # forever. Best-effort by design: a first run, an expired artifact, + # or a download failure must not cost a whole generation. + if ! command -v gh >/dev/null; then + echo '::warning::gh is not installed on this runner — proposing without prior evidence. The promotion-PR step needs gh too.' + exit 0 + fi + previous="$(gh run list \ + --repo "${GITHUB_REPOSITORY}" \ + --workflow gitnexus-skill-evolution.yml \ + --branch main \ + --status success \ + --limit 5 \ + --json databaseId \ + --jq "map(.databaseId) | map(select(. != ${GITHUB_RUN_ID})) | first // empty")" + if [[ -z "${previous}" ]]; then + echo 'No prior successful run to seed from; the proposer starts from the learnings queue only.' + exit 0 + fi + seed_root="${RUNNER_TEMP}/wfseed" + rm -rf "${seed_root}" + mkdir -p "${seed_root}" + if ! gh run download "${previous}" --repo "${GITHUB_REPOSITORY}" --dir "${seed_root}"; then + echo "::warning::Evidence from run ${previous} could not be downloaded (expired past its retention?); proposing without it." + exit 0 + fi + # The artifact holds gen-N/bench/{results.jsonl,promotion.json,...}; + # the highest generation is the one that actually reached the gate. + latest="$(find "${seed_root}" -type f -path '*/gen-*/bench/results.jsonl' | sort -V | tail -1)" + if [[ -z "${latest}" ]]; then + echo "::warning::Run ${previous} uploaded no benchmark results; proposing without prior evidence." + exit 0 + fi + echo "Seeding the proposer from run ${previous}: $(dirname "${latest}")" + echo "seed=$(dirname "${latest}")" >> "${GITHUB_OUTPUT}" + - name: Run the propose → benchmark → gate loop id: loop + # Kill the sweep with time left in the job to upload what it produced. + # See the budget nesting on the job above. + timeout-minutes: 1140 env: GITNEXUS_BENCH_AUTH_TOKEN: ${{ secrets.GITNEXUS_BENCH_AUTH_TOKEN }} # The step's stdout is a pipe, so CPython block-buffers it and a # multi-hour generation would report nothing until it exits (run # 29907431284 emitted every line at the same timestamp, 14h45m in). PYTHONUNBUFFERED: '1' + SEED_RESULTS: ${{ steps.seed.outputs.seed }} run: | set -euo pipefail - out_root="${RUNNER_TEMP}/wfevolve" - echo "out_root=${out_root}" >> "${GITHUB_OUTPUT}" extra=() if [[ -n "${INCLUDE_EXPENSIVE}" ]]; then extra+=(--include-expensive) fi + if [[ -n "${SEED_RESULTS}" ]]; then + extra+=(--seed-results "${SEED_RESULTS}") + fi uv run --locked --extra dev python -m workflow_bench.evolve \ --tasks workflow_bench/tasks.scenarios.yaml \ --model "${MODEL}" \ @@ -237,24 +304,25 @@ jobs: --generations "${GENERATIONS}" \ --runs "${RUNS}" \ --claude-bin "${RUNNER_TEMP}/claude-canary/node_modules/@anthropic-ai/claude-code-linux-x64/claude" \ - --out-root "${out_root}" \ + --out-root "${OUT_ROOT}" \ --apply \ "${extra[@]}" working-directory: eval - name: Upload benchmark evidence - if: always() && steps.loop.outputs.out_root != '' + # Unconditional: the sweep writes results.jsonl and transcripts as it + # goes, so a killed or failed generation still has evidence worth + # keeping — and that is exactly the run whose evidence is needed. + if: always() uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 with: name: gitnexus-evolution-${{ github.run_id }}-${{ github.run_attempt }} - path: ${{ steps.loop.outputs.out_root }} + path: ${{ env.OUT_ROOT }} retention-days: 14 if-no-files-found: warn - name: Detect and bound the applied promotion id: promotion - env: - OUT_ROOT: ${{ steps.loop.outputs.out_root }} run: | set -euo pipefail changed="$(git status --porcelain)" diff --git a/eval/tests/test_workflow_bench_evolution.py b/eval/tests/test_workflow_bench_evolution.py index 4fae6ab28..7fc7e4b8b 100644 --- a/eval/tests/test_workflow_bench_evolution.py +++ b/eval/tests/test_workflow_bench_evolution.py @@ -8,6 +8,7 @@ from types import SimpleNamespace import pytest from workflow_bench.evolution import ( + CANDIDATE_SKILLS, MAX_CANDIDATE_ENTRIES, apply_candidate_overlay, candidate_overlay_digest, @@ -16,6 +17,7 @@ from workflow_bench.evolution import ( skill_fingerprint, unexercised_overlay_skills, ) +from workflow_bench.promotion_apply import MIRROR_SKILL_ROOTS from workflow_bench.process_control import ManagedProcessResult from workflow_bench.runner import aggregate, build_parser @@ -601,3 +603,29 @@ def test_overlay_skills_must_be_exercised_by_selected_candidate_arms(tmp_path): write_overlay_skill(plan_overlay, "gitnexus-plan") assert unexercised_overlay_skills(plan_overlay, ["candidate_workflow_direct"]) == ["gitnexus-plan"] assert unexercised_overlay_skills(plan_overlay, ["candidate_workflow"]) == [] + + +@pytest.mark.parametrize("skill", sorted(CANDIDATE_SKILLS)) +def test_a_promoted_skill_is_visible_to_git_status_in_every_shipped_tree(skill): + """A promotion the repository cannot see is a promotion that never happens. + + The workflow detects an applied promotion with `git status --porcelain`, + which is blind to ignored paths, and `.claude/skills/*` is ignored with a + hand-maintained per-skill allowlist. A candidate skill missing from that + allowlist would leave the run reporting "No promotion this run" after the + gate had already said promote — silently, and only after a full generation + of benchmark spend. + """ + repo_root = Path(__file__).resolve().parents[2] + targets = [f".claude/skills/{skill}"] + [f"{root}/{skill}" for root in MIRROR_SKILL_ROOTS] + ignored = [ + target + for target in targets + if subprocess.run( + ["git", "check-ignore", "-q", f"{target}/SKILL.md"], + cwd=repo_root, + check=False, + ).returncode + == 0 + ] + assert ignored == [] diff --git a/gitnexus/test/unit/skill-evolution-workflow.test.ts b/gitnexus/test/unit/skill-evolution-workflow.test.ts index c2c3240fe..ea484abf3 100644 --- a/gitnexus/test/unit/skill-evolution-workflow.test.ts +++ b/gitnexus/test/unit/skill-evolution-workflow.test.ts @@ -18,10 +18,13 @@ const workflowDocument = load(workflow) as { string, { environment?: unknown; + 'timeout-minutes'?: unknown; steps?: Array<{ name?: string; + if?: unknown; run?: unknown; uses?: string; + 'timeout-minutes'?: unknown; with?: Record; }>; } @@ -119,6 +122,35 @@ describe('gitnexus skill-evolution workflow contract', () => { } }); + it('kills the benchmark with job time left to upload its evidence', () => { + // A job-level timeout cancels the job outright, so the upload step never + // runs and a multi-hour generation's evidence is lost. The sweep therefore + // needs its own, strictly shorter budget: a step timeout only fails that + // step, and the always() upload below still ships what it wrote. + const jobBudget = evolveJob?.['timeout-minutes']; + const loopStep = evolveJob?.steps?.find( + ({ name }) => name === 'Run the propose → benchmark → gate loop', + ); + const stepBudget = loopStep?.['timeout-minutes']; + expect(typeof jobBudget).toBe('number'); + expect(typeof stepBudget).toBe('number'); + expect(stepBudget as number).toBeLessThan(jobBudget as number); + // The runner is an EC2 box an EventBridge schedule stops 24h after it + // starts; when the box goes the runner vanishes mid-step and nothing + // uploads. The job must finish inside that window even when the schedule + // fires late (the 2026-08-01 run was queued 65 minutes after the cron). + expect(jobBudget as number).toBeLessThanOrEqual(21 * 60); + }); + + it('uploads benchmark evidence unconditionally, on a path fixed before the sweep runs', () => { + const upload = evolveJob?.steps?.find(({ name }) => name === 'Upload benchmark evidence'); + // The sweep appends results.jsonl and transcripts as it goes, so a killed + // generation still holds the evidence explaining why — and a path taken + // from the killed step's outputs is exactly what would not be there. + expect(upload?.if).toBe('always()'); + expect(upload?.with?.path).toBe('${{ env.OUT_ROOT }}'); + }); + it('documents the App secrets and protected Environment on the activation checklist', () => { expect(workflow).toContain('RELEASE_APP_ID'); expect(workflow).toContain('RELEASE_APP_PRIVATE_KEY');