From 780cac78856ac487f8025765ce60a73db9456bd7 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Gerg=C5=91=20Magyar?= Date: Sun, 6 Sep 2026 10:37:05 +0100 Subject: [PATCH 01/17] fix(parse): keep stable cache packs parallel (#3194) Stable cache packs introduced by 9718e1247 often fit one worker job, leaving most workers idle. Size jobs against live pool capacity and reuse extraction queries per native grammar instead of recompiling on every pack. Preserve recovery readiness across dispatches and bound failed-thread termination acknowledgment. Bisect confirms the scheduling regression. Controlled parsing of 951 TypeScript files falls from 56.39s to 26.66s with identical graph output; peak RSS increases roughly 10%. Add scheduling and query reuse coverage and repair recovery fixtures that assumed single-job dispatch. Validation: build, typecheck, formatting and 229 focused tests pass. Full suite was interrupted during lengthy native DB testing; full release CI remains outstanding. Co-authored-by: Gergo Magyar --- .../core/ingestion/workers/parse-worker.ts | 13 +- .../src/core/ingestion/workers/worker-pool.ts | 29 ++++- gitnexus/test/integration/worker-pool.test.ts | 118 +++++++++++++++++- .../test/unit/worker-pool-resilience.test.ts | 56 ++++++--- 4 files changed, 193 insertions(+), 23 deletions(-) diff --git a/gitnexus/src/core/ingestion/workers/parse-worker.ts b/gitnexus/src/core/ingestion/workers/parse-worker.ts index 0e8c8c65d..e6b776228 100644 --- a/gitnexus/src/core/ingestion/workers/parse-worker.ts +++ b/gitnexus/src/core/ingestion/workers/parse-worker.ts @@ -1543,6 +1543,10 @@ function reportWarning(message: string): void { } } +// Keep compiled queries across jobs in this worker. A language can select +// multiple native grammars, so both grammar identity and query text matter. +const compiledQueries = new WeakMap>(); + const processFileGroup = ( files: ParseWorkerInput[], language: SupportedLanguages, @@ -1553,7 +1557,14 @@ const processFileGroup = ( let query: Parser.Query; try { const lang = parser.getLanguage(); - query = new Parser.Query(lang, queryString); + let queries = compiledQueries.get(lang); + if (!queries) { + queries = new Map(); + compiledQueries.set(lang, queries); + } + const cached = queries.get(queryString); + query = cached ?? new Parser.Query(lang, queryString); + if (!cached) queries.set(queryString, query); } catch (err) { reportWarning( `Query compilation failed for ${language}: ${err instanceof Error ? err.message : String(err)}`, diff --git a/gitnexus/src/core/ingestion/workers/worker-pool.ts b/gitnexus/src/core/ingestion/workers/worker-pool.ts index de102ebef..497100c64 100644 --- a/gitnexus/src/core/ingestion/workers/worker-pool.ts +++ b/gitnexus/src/core/ingestion/workers/worker-pool.ts @@ -1316,9 +1316,16 @@ export const createWorkerPool = ( } if (dispatchableItems.length === 0) return []; + // Stable cache packs can be much smaller than either job ceiling. Split + // those packs across the live slots too, otherwise each serial dispatch + // feeds only one worker. Keep both configured ceilings as upper bounds. + const maxItemsPerJob = Math.min( + poolOptions.subBatchSize, + Math.max(1, Math.floor(dispatchableItems.length / activeSlots.size)), + ); const jobs = createJobs( dispatchableItems, - poolOptions.subBatchSize, + maxItemsPerJob, poolOptions.subBatchMaxBytes, poolOptions.subBatchIdleTimeoutMs, chunkHash, @@ -1456,7 +1463,19 @@ export const createWorkerPool = ( retireWorkerAfterTimeout(existing, workerIndex, reason); return; } - await existing.terminate().catch(() => undefined); + // Recovery must settle before dispatch returns, but a failed thread + // may never acknowledge termination. Bound that wait as in shutdown. + const termination = existing.terminate().then( + () => undefined, + () => undefined, + ); + if (!(await settledWithin(termination, poolOptions.shutdownDrainMs))) { + existing.unref?.(); + logger.warn( + { workerIndex, drainMs: poolOptions.shutdownDrainMs, reason }, + `Worker ${workerIndex} did not finish terminating within the shutdown drain; continuing recovery.`, + ); + } }; const replaceWorker = async ( @@ -1898,11 +1917,13 @@ export const createWorkerPool = ( // (`error`, `exit`, msg-channel error). Bridges the per-job teardown // into the pool-level handleWorkerDeath recovery + breaker logic. const recoverAndResume = async (reason: string, excludePaths: readonly string[]) => { - activeWorkers--; busySlots.delete(workerIndex); inFlightProgress[workerIndex] = 0; requeueRemainder(job, excludePaths); + // Keep recovery in flight so another slot finishing cannot settle + // this dispatch before the replacement is ready for the next one. await handleWorkerDeath(workerIndex, reason, excludePaths); + activeWorkers--; if (stopped) return; // Slot may have been dropped or respawned. Kick the current slot // if still active, then wake any other idle live slots so the @@ -1949,7 +1970,6 @@ export const createWorkerPool = ( // is respawned (or dropped) and can dispatch the next // job deterministically. void (async () => { - activeWorkers--; busySlots.delete(workerIndex); requeueRemainder(job, decision.excludePaths); await handleWorkerDeath( @@ -1958,6 +1978,7 @@ export const createWorkerPool = ( decision.excludePaths, 'retire', ); + activeWorkers--; if (stopped) return; if (activeSlots.has(workerIndex)) runWorker(workerIndex); wakeIdleSlots(); diff --git a/gitnexus/test/integration/worker-pool.test.ts b/gitnexus/test/integration/worker-pool.test.ts index d9f6577a9..da910a736 100644 --- a/gitnexus/test/integration/worker-pool.test.ts +++ b/gitnexus/test/integration/worker-pool.test.ts @@ -14,6 +14,7 @@ import { } from '../../src/core/ingestion/workers/worker-pool.js'; import { pathToFileURL } from 'node:url'; import { spawn } from 'node:child_process'; +import { createRequire } from 'node:module'; import path from 'node:path'; import fs from 'node:fs'; import os from 'node:os'; @@ -245,10 +246,8 @@ describe('worker pool integration', () => { const results = await pool.dispatch(files); - // All 7 files fit one default sub-batch (size 200 / budget 8MB), - // so the dispatch returns exactly one chunk result regardless of - // pool size. - expect(results).toHaveLength(1); + // Small inputs must still split across the available workers. + expect(results.length).toBeGreaterThan(1); // Total files parsed should match input const totalParsed = results.reduce((sum: number, r: any) => sum + r.fileCount, 0); @@ -793,6 +792,54 @@ describe('worker pool integration', () => { } }); + it.each([2, 4, 7, 9])( + 'uses available workers for a small %i-file cache pack', + async (fileCount) => { + const { tempDir, workerPath } = writeTempWorker( + 'gitnexus-worker-small-pack-', + ` + const { parentPort, threadId } = require('node:worker_threads'); + let paths = []; + parentPort.on('message', (msg) => { + if (msg && msg.type === 'sub-batch') { + paths = msg.files.map((file) => file.path); + parentPort.postMessage({ type: 'progress', filesProcessed: paths.length }); + parentPort.postMessage({ type: 'sub-batch-done' }); + } else if (msg && msg.type === 'flush') { + parentPort.postMessage({ type: 'result', data: { paths, threadId } }); + } + }); + `, + ); + pool = createWorkerPool(pathToFileURL(workerPath), 4, { + subBatchMaxBytes: 256 * 1024, + }); + + try { + // Stable cache packs are often smaller than the byte budget. They must + // still use the pool, including when a later dispatch has fewer files. + for (const count of [fileCount, 1]) { + const files = Array.from({ length: count }, (_, i) => ({ + path: `file-${i}.ts`, + content: 'export const value = 1;', + })); + const progress: number[] = []; + const results = await pool.dispatch< + (typeof files)[number], + { paths: string[]; threadId: number } + >(files, (completed) => progress.push(completed), `pack-${count}`); + expect(new Set(results.map((result) => result.threadId)).size).toBe(Math.min(4, count)); + expect(results.flatMap((result) => result.paths)).toEqual(files.map((file) => file.path)); + expect(progress).toEqual([...progress].sort((a, b) => a - b)); + expect(progress.at(-1)).toBe(count); + } + } finally { + await pool.terminate(); + fs.rmSync(tempDir, { recursive: true, force: true }); + } + }, + ); + it('bounds worker jobs by byte budget as well as file count', async () => { const { tempDir, workerPath } = writeTempWorker( 'gitnexus-worker-byte-budget-', @@ -831,6 +878,69 @@ describe('worker pool integration', () => { } }); + it.skipIf(!hasDistWorker)( + 'reuses compiled queries across jobs while keeping TS and TSX grammars separate', + async () => { + const tempDir = fs.mkdtempSync(path.join(os.tmpdir(), 'gitnexus-worker-query-cache-')); + const workerPath = path.join(tempDir, 'worker.cjs'); + const parserPath = createRequire(import.meta.url).resolve('tree-sitter'); + fs.writeFileSync( + workerPath, + ` + const { parentPort } = require('node:worker_threads'); + const Parser = require(${JSON.stringify(parserPath)}); + let queryCompilations = 0; + Parser.Query = new Proxy(Parser.Query, { + construct(target, args, newTarget) { + queryCompilations++; + return Reflect.construct(target, args, newTarget); + }, + }); + const send = parentPort.postMessage.bind(parentPort); + parentPort.postMessage = (message, ...args) => { + if (message.type === 'result') message.data.queryCompilations = queryCompilations; + return send(message, ...args); + }; + import(${JSON.stringify(pathToFileURL(DIST_WORKER).href)}); + `, + ); + pool = createWorkerPool(pathToFileURL(workerPath), 1, { workerReadyTimeoutMs: 30_000 }); + type QueryResult = { + queryCompilations: number; + fileCount: number; + nodes: Array<{ properties: { name: string } }>; + }; + try { + const counts: number[] = []; + for (const [extension, name] of [ + ['ts', 'first'], + ['ts', 'second'], + ['tsx', 'view'], + ['tsx', 'otherView'], + ['ts', 'last'], + ]) { + const file = { + path: `${name}.${extension}`, + content: `export function ${name}() { return ${extension === 'tsx' ? '
' : '1'}; }`, + }; + const [result] = await pool.dispatch([file]); + expect(result.fileCount).toBe(1); + expect(result.nodes.map((node) => node.properties.name)).toContain(name); + counts.push(result.queryCompilations); + } + expect(counts[0]).toBeGreaterThan(0); + expect(counts[1]).toBe(counts[0]); + expect(counts[2]).toBeGreaterThan(counts[1]); + expect(counts[3]).toBe(counts[2]); + expect(counts[4]).toBe(counts[2]); + } finally { + await pool.terminate(); + fs.rmSync(tempDir, { recursive: true, force: true }); + } + }, + 60_000, + ); + it.skipIf(!hasDistWorker)('createWorkerPool with size 0 creates pool with zero workers', () => { const workerUrl = pathToFileURL(DIST_WORKER) as URL; const zeroPool = createWorkerPool(workerUrl, 0); diff --git a/gitnexus/test/unit/worker-pool-resilience.test.ts b/gitnexus/test/unit/worker-pool-resilience.test.ts index 002d47de8..1427f45f1 100644 --- a/gitnexus/test/unit/worker-pool-resilience.test.ts +++ b/gitnexus/test/unit/worker-pool-resilience.test.ts @@ -58,7 +58,7 @@ let workerInstances: FakeWorker[] = []; class FakeWorker extends EventEmitter { readonly seenMessages: unknown[] = []; - constructor() { + constructor(startupExitCode?: number) { super(); workerInstances.push(this); // Real Worker fires 'online' asynchronously after the runtime is ready; @@ -67,6 +67,10 @@ class FakeWorker extends EventEmitter { // message instead — emit that too so replacement-worker tests don't // hit the WORKER_READY_TIMEOUT_MS budget (5s). queueMicrotask(() => { + if (startupExitCode !== undefined) { + this.emit('exit', startupExitCode); + return; + } this.emit('online'); this.emit('message', { type: 'ready' }); }); @@ -262,8 +266,8 @@ describe('worker pool resilience', () => { consecutiveFailureThreshold: 10, maxRespawnsPerSlot: 1, }); - // Slot 0 dies twice, exceeding budget=1; slot 1 succeeds with the - // requeued remainder. + // Separate dispatches target the first live slot twice, independently + // of how multi-file packs are split across workers. nextActions.push({ kind: 'crash-exit', code: 134, afterStartingFiles: 1 }); nextActions.push({ kind: 'crash-exit', code: 134, afterStartingFiles: 1 }); nextActions.push({ @@ -272,9 +276,10 @@ describe('worker pool resilience', () => { result: { fileCount: 2 }, }); + await pool.dispatch([{ path: 'src/a.ts', content: '' }]); + await pool.dispatch([{ path: 'src/b.ts', content: '' }]); + expect(pool.getStats().activeSlots).toBe(1); const results = await pool.dispatch<{ path: string; content: string }, unknown>([ - { path: 'src/a.ts', content: '' }, - { path: 'src/b.ts', content: '' }, { path: 'src/c.ts', content: '' }, { path: 'src/d.ts', content: '' }, ]); @@ -495,18 +500,13 @@ describe('worker pool resilience', () => { await pool.terminate(); }); - it('drops slot when waitForWorkerOnline rejects (replaceWorker failure path)', async () => { + it('drops slot when replacement readiness rejects', async () => { let factoryCallCount = 0; const pool = createWorkerPool(workerUrl, 2, { workerFactory: () => { factoryCallCount++; - const worker = new FakeWorker(); - // Slot 0's initial worker is healthy; the replacement (3rd factory - // call after slot 0 dies once) exits before emitting 'online'. - if (factoryCallCount === 3) { - // Override the queued 'online' microtask with an immediate 'exit'. - queueMicrotask(() => worker.emit('exit', 1)); - } + // The replacement exits INSTEAD OF reporting ready. + const worker = new FakeWorker(factoryCallCount === 3 ? 1 : undefined); return worker as unknown as import('node:worker_threads').Worker; }, consecutiveFailureThreshold: 10, @@ -519,8 +519,9 @@ describe('worker pool resilience', () => { result: { fileCount: 2 }, }); + await pool.dispatch([{ path: 'src/a.ts', content: '' }]); + expect(pool.getStats().activeSlots).toBe(1); const results = await pool.dispatch<{ path: string; content: string }, unknown>([ - { path: 'src/a.ts', content: '' }, { path: 'src/b.ts', content: '' }, { path: 'src/c.ts', content: '' }, ]); @@ -532,6 +533,33 @@ describe('worker pool resilience', () => { await pool.terminate(); }); + it('finishes recovery when a failed worker never acknowledges termination', async () => { + const pool = createWorkerPool(workerUrl, 2, { + workerFactory: () => { + const worker = new FakeWorker(); + if (workerInstances.length === 1) { + worker.terminate = () => new Promise(() => {}); + } + return worker as unknown as import('node:worker_threads').Worker; + }, + shutdownDrainMs: 10, + }); + nextActions.push({ kind: 'crash-exit', code: 134, afterStartingFiles: 1 }); + nextActions.push({ kind: 'parse-ok', files: [{ path: 'src/good.ts' }] }); + try { + const results = await pool.dispatch([ + { path: 'src/bad.ts', content: '' }, + { path: 'src/good.ts', content: '' }, + ]); + expect(results).toEqual([{ fileCount: 1 }]); + expect(pool.getQuarantinedPaths()).toEqual(['src/bad.ts']); + expect(pool.getStats().slotGenerations).toEqual([1, 0]); + expect(pool.getStats().activeSlots).toBe(2); + } finally { + await pool.terminate(); + } + }, 1000); + it('trips the breaker when all slots exhaust their respawn budget', async () => { const pool = createWorkerPool(workerUrl, 2, { workerFactory: () => new FakeWorker() as unknown as import('node:worker_threads').Worker, From 8f006bd759181ef4751c95b06a1397e2753bbcc1 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Gerg=C5=91=20Magyar?= Date: Sun, 6 Sep 2026 18:27:09 +0100 Subject: [PATCH 02/17] perf(parse): batch cache packs into one dispatch round (#3196) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * perf(parse): batch cache packs into one dispatch round `WorkerPool.dispatch` is a barrier, so dispatching one parse-cache pack at a time leaves most slots idle for every round-trip. Packs are keyed by `(language, hash(path) % 128)`, so the byte budget rarely binds: this repo produces 1285 packs where the budget alone needs 16, and 549 of those hold a single file. In a real analyze, 76 of 221 dispatched chunks carried one file and cost 15.3s — 20% of the parse phase for 3.4% of the files. Chunks now accumulate into a round bounded by `GITNEXUS_PARSE_ROUND_BYTES` of cache-missing source (default: the chunk byte budget) and go out through a new `WorkerPool.dispatchGroups`. Jobs are still cut at pack boundaries, so each job carries exactly one `chunkHash` and every result stays attributable to the pack whose cache key owns it. Cache hits ride along as round entries, and rounds drain in `chunkIdx` order, so deferred aggregation stays deterministic. Cold `analyze --index-only` on this repo (2234 parseable files, 16 workers): 110.3s -> 70.5s total, parse phase 74.0s -> 40.5s, 221 dispatches -> 15. Graph output is unchanged: 51,286 nodes / 163,092 edges / 2106 clusters / 759 flows in both arms. Peak main-thread RSS 3372MB -> 3487MB (+3.4%). `dispatchGroups` also claims the pool synchronously and rejects a concurrent call. Two overlapping dispatches hand the same slots out twice and both stall; the first version of this change did exactly that, and the only symptom was every worker idle-timing out ~10s later with no indication of the cause. Co-Authored-By: Claude Opus 5 (1M context) * fix(parse): bound what an open round holds, not just what it dispatches Follow-up to the review of #3196. Three reviewers independently found the same defect: `roundMissBytes` was the only in-loop close condition, but cache HITS were queued into the same round without contributing to it. A warm re-analyze misses nothing, so no round ever closed and every chunk's source plus its cached worker output stayed resident until the tail drain — the #2649 heap failure shape on a large repo. - Hit entries now carry a file COUNT, not the file array, so a replayed chunk never pins its source text. `applyChunkResults` only ever read `.length`. - Track `roundBufferedBytes` across hits and misses and close on either cap. Verified on a warm run: with the cap, draining starts as soon as 2MB is buffered; without it all 221 merges land in the final 10% of the phase. - Warm progress no longer freezes at the phase floor. `filesParsedSoFar` only advances at drain, so a new `queuedFilesSoFar` feeds the progress events while `filesParsedSoFar` stays the merge-accurate throughput number. - A throw from `drainRound` used to unwind straight to `terminate()` while the next round's workers were still busy — the #2432 mid-N-API abort hazard. Settle the in-flight round first, then propagate. - `dispatchGroups` returns one array per group; assert that length instead of `?? []`, which turned a contract break into a silently empty chunk. - Collapse `PendingWorkerChunk` into the `miss` RoundEntry it duplicated. - Repair two stale doc comments: `dispatch`'s JSDoc had been orphaned onto `dispatchGroups`, and `dispatchChunkParse` still described chunk overlap that now lives in parse-impl's round machinery. - New test: a round mixing a cache hit and a cache miss. `drainRound` walks entries in chunkIdx order but pulls results on a separate cursor, and no existing test put both kinds in one round with content assertions. Cold analyze unchanged: 71.3s, 15 rounds, 51,286 nodes / 163,092 edges. Co-Authored-By: Claude Opus 5 (1M context) --------- Co-authored-by: Gergo Magyar Co-authored-by: Claude Opus 5 (1M context) --- .../src/core/ingestion/parsing-processor.ts | 73 +++- .../ingestion/pipeline-phases/parse-impl.ts | 412 +++++++++++++----- .../src/core/ingestion/workers/worker-pool.ts | 177 ++++++-- .../parse-impl-dispatch-rounds.test.ts | 225 ++++++++++ .../integration/parse-impl-env-reads.test.ts | 23 +- gitnexus/test/integration/worker-pool.test.ts | 150 +++++++ .../test/unit/parsing-worker-fallback.test.ts | 7 +- 7 files changed, 891 insertions(+), 176 deletions(-) create mode 100644 gitnexus/test/integration/parse-impl-dispatch-rounds.test.ts diff --git a/gitnexus/src/core/ingestion/parsing-processor.ts b/gitnexus/src/core/ingestion/parsing-processor.ts index e4df7dae5..9a1f39233 100644 --- a/gitnexus/src/core/ingestion/parsing-processor.ts +++ b/gitnexus/src/core/ingestion/parsing-processor.ts @@ -7,6 +7,7 @@ import { accumulateExportedTypesFromParsedNode, type ExportedTypeMap } from './c import type { ParsedFile } from 'gitnexus-shared'; import { WorkerPool } from './workers/worker-pool.js'; +import type { DispatchGroup } from './workers/worker-pool.js'; import type { SkippedPath } from './workers/clone-safety.js'; import type { CfgSkipCounts } from './cfg/collect.js'; import { logger } from '../logger.js'; @@ -206,12 +207,12 @@ export const mergeChunkResults = ( }; /** - * Dispatch a chunk's files to the worker pool and return the RAW per-worker - * results, WITHOUT merging them into the graph. Split out from - * {@link processParsing} so the parse loop can overlap one chunk's - * merge (main-thread, via {@link mergeChunkResults}) with the NEXT chunk's - * worker parse — the merge is the only remaining serial main-thread step once - * ParsedFile serialization moved into the workers (#worker-idle pipelining). + * Dispatch ONE chunk's files to the worker pool and return the RAW per-worker + * results, WITHOUT merging them into the graph. A thin single-group wrapper + * over {@link dispatchChunkParseRound}, used by {@link processParsing}'s + * one-shot path. The chunk-to-chunk overlap this once described now lives in + * `parse-impl.ts` at ROUND granularity (`startRound` / `drainRound` / + * `closeRound`), which batches several chunks into one dispatch. * Returns `[]` for an all-unparseable chunk (the caller merges `[]` → empty). */ export const dispatchChunkParse = async ( @@ -227,26 +228,56 @@ export const dispatchChunkParse = async ( */ chunkHash?: string, ): Promise => { - const parseableFiles: ParseWorkerInput[] = []; - for (const file of files) { - const lang = getLanguageFromFilename(file.path); - if (lang) parseableFiles.push({ path: file.path, content: file.content }); - } - if (parseableFiles.length === 0) return []; - - const total = files.length; - const chunkResults = await workerPool.dispatch( - parseableFiles, - (filesProcessed) => { - onFileProgress?.(Math.min(filesProcessed, total), total, 'Parsing...'); - }, - chunkHash, + const [chunkResults = []] = await dispatchChunkParseRound( + [{ items: files, chunkHash }], + workerPool, + onFileProgress, ); // Capture raw results for the incremental parse cache before merging. if (outRawResults) { for (const r of chunkResults) outRawResults.push(r); } + return chunkResults; +}; + +/** + * Dispatch SEVERAL parse-cache chunks as one pool round and return their raw + * results, one array per input group in input order. + * + * `WorkerPool.dispatch` is a barrier, so one round-trip per chunk leaves most + * slots idle whenever a chunk is smaller than the pool — which stable + * `(language, hash(path) % 128)` packs usually are. Batching chunks into one + * `dispatchGroups` call removes those barriers; jobs are still cut at chunk + * boundaries, so every result stays attributable to the chunk whose cache key + * owns it. + */ +export const dispatchChunkParseRound = async ( + groups: ReadonlyArray<{ + items: { path: string; content: string }[]; + chunkHash?: string; + }>, + workerPool: WorkerPool, + onFileProgress?: FileProgressCallback, +): Promise => { + const dispatchGroups: DispatchGroup[] = groups.map((group) => { + const items: ParseWorkerInput[] = []; + for (const file of group.items) { + const lang = getLanguageFromFilename(file.path); + if (lang) items.push({ path: file.path, content: file.content }); + } + return { items, chunkHash: group.chunkHash }; + }); + const total = groups.reduce((sum, group) => sum + group.items.length, 0); + if (dispatchGroups.every((group) => group.items.length === 0)) return groups.map(() => []); + + const perGroup = await workerPool.dispatchGroups( + dispatchGroups, + (filesProcessed) => { + onFileProgress?.(Math.min(filesProcessed, total), total, 'Parsing...'); + }, + ); + const chunkResults = perGroup.flat(); // Skipped-language telemetry (worker output, independent of the merge). const skippedLanguages = new Map(); @@ -311,7 +342,7 @@ export const dispatchChunkParse = async ( } onFileProgress?.(total, total, 'done'); - return chunkResults; + return perGroup; }; // ============================================================================ diff --git a/gitnexus/src/core/ingestion/pipeline-phases/parse-impl.ts b/gitnexus/src/core/ingestion/pipeline-phases/parse-impl.ts index b69c09eb7..b4ec9a5b8 100644 --- a/gitnexus/src/core/ingestion/pipeline-phases/parse-impl.ts +++ b/gitnexus/src/core/ingestion/pipeline-phases/parse-impl.ts @@ -20,7 +20,7 @@ import { enrichExportedTypeMap, type BindingEntry, } from '../binding-accumulator.js'; -import { mergeChunkResults, dispatchChunkParse } from '../parsing-processor.js'; +import { mergeChunkResults, dispatchChunkParseRound } from '../parsing-processor.js'; import { fileContentHash, computeChunkHash, @@ -216,6 +216,28 @@ const TARGET_JOBS_PER_WORKER = 3; /** Floor for a derived sub-batch so jobs don't shrink to per-file IPC churn. */ const MIN_SUB_BATCH_BYTES = 256 * 1024; +/** + * Source bytes of cache-missing chunks allowed in flight in one pool round. + * + * A `dispatch` is a barrier, so one round-trip per cache pack leaves most slots + * idle: packs are keyed by `(language, hash(path) % 128)` and routinely land far + * under {@link DEFAULT_CHUNK_BYTE_BUDGET} (this repo: 1285 packs where the byte + * budget alone needs 16, 549 of them holding a single file). Rounds batch packs + * into one `dispatchGroups` call without touching pack identity. + * + * This is the in-flight cap, the same role Piscina's `maxQueue` plays: bigger + * rounds remove more barriers but hold more file content and more un-merged + * worker output on the main thread at once. Defaulting to one chunk budget + * keeps in-flight source bytes at the magnitude the loop already prefetched + * (`parseChunkConcurrency`, 2 chunks ahead). Override via + * `GITNEXUS_PARSE_ROUND_BYTES`. + */ +function resolveParseRoundByteBudget(options?: PipelineOptions): number { + const env = Number(process.env.GITNEXUS_PARSE_ROUND_BYTES); + if (Number.isFinite(env) && env > 0) return env; + return resolveChunkByteBudget(options); +} + function resolveChunkByteBudget(options?: PipelineOptions): number { const opt = options?.chunkByteBudget; if (typeof opt === 'number' && Number.isFinite(opt) && opt > 0) return opt; @@ -793,25 +815,73 @@ export async function runChunkedParseAndResolve( const verboseThroughputLog = isDev || isVerboseIngestionEnabled(); const heapProbeEveryN = isDebugHeapEnabled() ? 25 : 0; - // ── Merge pipelining (#worker-idle) ────────────────────────────────────── - // Merging a chunk's worker results into the graph is the only remaining - // serial main-thread step (ParsedFile serialization now runs in workers). - // To stop the whole pool idling during that merge, we OVERLAP it with the - // NEXT chunk's worker parse: a freshly-dispatched worker chunk is parked in - // `pendingWorkerChunk`, and we merge+finalize it only AFTER starting the - // following chunk's dispatch — so the workers parse chunk N+1 while the - // main thread merges chunk N. Chunk ORDER is preserved (N finalized before - // N+1), which keeps the deferred aggregation deterministic. Cache-hit - // chunks drain any pending chunk first, then finalize inline (no worker - // dispatch to overlap). - interface PendingWorkerChunk { - readonly rawResults: ParseWorkerResult[]; - readonly chunkIdx: number; - readonly chunkHash: string | null; - readonly chunkFiles: Array<{ path: string; content: string }>; - readonly chunkStartMs: number | null; - } - let pendingWorkerChunk: PendingWorkerChunk | null = null; + // ── Dispatch rounds + merge pipelining (#worker-idle) ──────────────────── + // Two separate idle sources, handled together here. + // + // 1. Barrier per chunk. `dispatch` resolves only when every job it created + // has committed, so dispatching one cache pack at a time strands the + // pool whenever a pack is smaller than it — which stable packs usually + // are. Chunks accumulate into a ROUND (bounded by `roundByteBudget` of + // cache-missing source) and go out in one `dispatchGroups` call. + // 2. Serial merge. Merging worker results into the graph is the only + // remaining serial main-thread step (ParsedFile serialization now runs + // in workers). A dispatched round is parked in `pendingRound` and + // merged only AFTER the following round's dispatch has started, so the + // workers parse round N+1 while the main thread merges round N. + // + // Chunk ORDER is preserved throughout — rounds drain in order and entries + // inside a round finalize by `chunkIdx` — which keeps deferred aggregation + // deterministic regardless of how chunks were batched. Cache hits ride + // along as round entries so they observe the same ordering without forcing + // a dispatch. + /** + * One chunk queued into the current round. A `hit` already has its worker + * output (from the parse cache); a `miss` gets it from the round's single + * `dispatchGroups` call. Both are finalized in `chunkIdx` order when the + * round drains, which is what keeps deferred aggregation deterministic + * regardless of how chunks were batched. + */ + type RoundEntry = + | { + readonly kind: 'hit'; + readonly chunkIdx: number; + // A hit never reaches a worker, so it needs the file COUNT (progress, + // throughput log) but never the source strings. Holding those would + // pin the whole repo's text for a warm run, which is what the + // buffered budget below exists to bound. + readonly fileCount: number; + readonly chunkStartMs: number | null; + readonly cachedRaw: ParseWorkerResult[]; + } + | { + readonly kind: 'miss'; + readonly chunkIdx: number; + readonly chunkHash: string | null; + readonly chunkFiles: Array<{ path: string; content: string }>; + readonly chunkStartMs: number | null; + }; + + const roundByteBudget = resolveParseRoundByteBudget(options); + let roundEntries: RoundEntry[] = []; + let roundMissBytes = 0; + /** + * Bytes an open round is HOLDING, counting hits as well as misses. + * + * `roundMissBytes` alone bounds only what the workers are asked to do, so a + * warm run — where nothing misses — would never reach the close condition + * and would buffer every chunk's cached output until the tail drain. That + * is the #2649 heap failure on a large repo. Closing on either cap keeps a + * hits-only run draining at the same cadence as a cold one; `startRound` + * already supports a round with no misses. + */ + let roundBufferedBytes = 0; + /** + * Files QUEUED into rounds so far. `filesParsedSoFar` only advances when a + * round drains, so it is the right number for the throughput log but would + * pin a warm run's progress bar at the phase floor for the whole loop. + */ + let queuedFilesSoFar = 0; + let pendingRound: { entries: RoundEntry[]; missResults: ParseWorkerResult[][] } | null = null; // Apply one chunk's merged worker data: per-chunk aggregation into the // run-level accumulators + the throughput log. Shared by the cache-hit @@ -821,7 +891,7 @@ export async function runChunkedParseAndResolve( const applyChunkResults = async ( chunkWorkerData: WorkerExtractedData | null, chunkIdx: number, - chunkFiles: Array<{ path: string; content: string }>, + fileCount: number, chunkStartMs: number | null, ): Promise => { if (chunkWorkerData) { @@ -898,18 +968,18 @@ export async function runChunkedParseAndResolve( } } - filesParsedSoFar += chunkFiles.length; + filesParsedSoFar += fileCount; if (verboseThroughputLog && chunkStartMs !== null) { const elapsedMs = Date.now() - chunkStartMs; - const filesPerSec = elapsedMs > 0 ? (chunkFiles.length * 1000) / elapsedMs : 0; + const filesPerSec = elapsedMs > 0 ? (fileCount * 1000) / elapsedMs : 0; const stats = workerPool?.getStats?.(); const poolFrag = stats ? ` pool: ${stats.activeSlots}/${stats.size} active, ` + `${stats.quarantined} quarantined${stats.poolBroken ? ', BROKEN' : ''}` : ' (cache replay)'; logger.info( - `📊 chunk ${chunkIdx + 1}/${numChunks}: ${chunkFiles.length} files in ${elapsedMs}ms ` + + `📊 chunk ${chunkIdx + 1}/${numChunks}: ${fileCount} files in ${elapsedMs}ms ` + `(${filesPerSec.toFixed(1)} files/s)${poolFrag}`, ); } @@ -917,12 +987,15 @@ export async function runChunkedParseAndResolve( // Merge + finalize a parked worker chunk: graph merge (the overlapped // main-thread step) → parse-cache write-guard → run-level aggregation. - const finalizeWorkerChunk = async (p: PendingWorkerChunk): Promise => { - const chunkWorkerData = mergeChunkResults(graph, symbolTable, p.rawResults, exportedTypeMap); + const finalizeWorkerChunk = async ( + p: Extract, + rawResults: ParseWorkerResult[], + ): Promise => { + const chunkWorkerData = mergeChunkResults(graph, symbolTable, rawResults, exportedTypeMap); // Persist raw results for this chunk hash (skipping when any chunk file // was worker-quarantined, so the narrower rawResults isn't cached under // the full-chunk key — see the original inline note / U20.U2). - if (parseCache && p.chunkHash && p.rawResults.length > 0) { + if (parseCache && p.chunkHash && rawResults.length > 0) { const quarantineSet = new Set(workerPool?.getQuarantinedPaths?.() ?? []); const chunkHadQuarantine = p.chunkFiles.some((f) => quarantineSet.has(f.path)); if (chunkHadQuarantine) { @@ -935,7 +1008,7 @@ export async function runChunkedParseAndResolve( ); } } else { - await persistParseCacheChunk(parseCache, p.chunkHash, p.rawResults); + await persistParseCacheChunk(parseCache, p.chunkHash, rawResults); if (isDev) { logger.info( `📦 parse-cache MISS+store: chunk ${p.chunkIdx + 1}/${numChunks} (${p.chunkFiles.length} files, ${p.chunkHash.slice(0, 8)})`, @@ -943,7 +1016,171 @@ export async function runChunkedParseAndResolve( } } } - await applyChunkResults(chunkWorkerData, p.chunkIdx, p.chunkFiles, p.chunkStartMs); + await applyChunkResults(chunkWorkerData, p.chunkIdx, p.chunkFiles.length, p.chunkStartMs); + }; + + /** + * Dispatch a round's cache misses as ONE pool round. Returns the parked + * round; the caller drains it after starting the next one so the workers + * parse round N+1 while the main thread merges round N (the same overlap + * the per-chunk loop had, at round granularity). + */ + const startRound = async ( + entries: RoundEntry[], + ): Promise<{ entries: RoundEntry[]; results: Promise } | null> => { + if (entries.length === 0) return null; + const misses = entries.filter((entry) => entry.kind === 'miss'); + if (misses.length === 0) { + return { entries, results: Promise.resolve([]) }; + } + for (const miss of misses) { + if (durableParsedFileDir !== undefined && miss.chunkHash !== null) { + try { + await prepareDurableParsedFileChunk(durableParsedFileDir, miss.chunkHash); + } catch (err) { + // The durable store is an optimization — degrade like the restore + // path does instead of failing the analyze. Workers recreate the + // directory on write, so at worst the old generation lingers. + logger.warn( + { err, chunkHash: miss.chunkHash.slice(0, 8) }, + 'parsedfile-cache: could not reset durable chunk generation; continuing', + ); + } + } + } + const roundFiles = misses.reduce((sum, miss) => sum + miss.chunkFiles.length, 0); + const firstIdx = misses[0].chunkIdx; + const lastIdx = misses[misses.length - 1].chunkIdx; + const progressForRound = (current: number, _total: number, filePath: string) => { + // Rounds queued before this one are already counted in + // `queuedFilesSoFar`; `current` is this round's own worker progress. + const globalCurrent = queuedFilesSoFar - roundFiles + current; + // Parse phase covers 20-70 (M2). Deferred extraction handles 70-95. + const parsingProgress = 20 + (globalCurrent / totalParseable) * 50; + onProgress({ + phase: 'parsing', + percent: Math.round(parsingProgress), + message: + firstIdx === lastIdx + ? `Parsing chunk ${firstIdx + 1}/${numChunks}...` + : `Parsing chunks ${firstIdx + 1}-${lastIdx + 1}/${numChunks}...`, + detail: filePath, + stats: { + filesProcessed: globalCurrent, + totalFiles: totalParseable, + nodesCreated: graph.nodeCount, + }, + }); + }; + const activeWorkerPool = getOrCreateWorkerPool(); + if (verboseThroughputLog) { + logger.info( + `🚚 round: ${misses.length} chunk(s) ${firstIdx + 1}-${lastIdx + 1}/${numChunks}, ` + + `${roundFiles} files in one dispatch`, + ); + } + const results = dispatchChunkParseRound( + misses.map((miss) => ({ + items: miss.chunkFiles, + chunkHash: miss.chunkHash ?? undefined, + })), + activeWorkerPool, + progressForRound, + ); + // Mark handled so a rejection during the overlap drain below isn't + // flagged as unhandled; the `await` in drainRound re-throws it for real + // handling. + results.catch(() => {}); + return { entries, results }; + }; + + /** + * Merge + finalize every chunk of a parked round, in `chunkIdx` order. + * Takes RESOLVED worker output: the round's dispatch must already have + * settled before this runs, because the pool allows only one dispatch in + * flight at a time (see `closeRound`). + */ + const drainRound = async (round: { + entries: RoundEntry[]; + missResults: ParseWorkerResult[][]; + }): Promise => { + const missResults = round.missResults; + const missCount = round.entries.reduce( + (sum, entry) => sum + (entry.kind === 'miss' ? 1 : 0), + 0, + ); + // `dispatchGroups` returns one array per input group. If that contract + // ever breaks, every later entry in this round would silently merge the + // wrong chunk's results and skip its cache write, with a clean exit. + if (missResults.length !== missCount) { + throw new Error( + `Parse round result mismatch: ${missResults.length} result group(s) for ${missCount} dispatched chunk(s).`, + ); + } + let missIdx = 0; + for (const entry of round.entries) { + if (entry.kind === 'hit') { + const chunkWorkerData = mergeChunkResults( + graph, + symbolTable, + entry.cachedRaw, + exportedTypeMap, + ); + await applyChunkResults( + chunkWorkerData, + entry.chunkIdx, + entry.fileCount, + entry.chunkStartMs, + ); + continue; + } + await finalizeWorkerChunk(entry, missResults[missIdx++]); + } + }; + + /** + * Close the accumulated round. + * + * `WorkerPool.dispatch`/`dispatchGroups` is NOT reentrant — concurrent + * calls race on the shared per-slot busy/in-flight state and wedge the + * pool until every worker idle-times out. So exactly one dispatch is in + * flight here: start this round, merge the PREVIOUS round (whose results + * are already resolved) while these workers run, then await this round and + * park it resolved for the next close to merge. + */ + const closeRound = async (): Promise => { + const started = await startRound(roundEntries); + roundEntries = []; + roundMissBytes = 0; + roundBufferedBytes = 0; + const previous = pendingRound; + pendingRound = null; + if (previous) { + try { + await drainRound(previous); + } catch (err) { + // The round started above is still on the workers. Unwinding now + // reaches this function's `finally`, which calls `terminate()` — and + // terminate kills busy workers outright, which is the #2432 + // mid-N-API SIGABRT hazard. Let the in-flight round settle first so + // the pool is idle, then propagate the original failure. + await started?.results.catch(() => undefined); + throw err; + } + } + if (!started) return; + let missResults: ParseWorkerResult[][]; + try { + missResults = await started.results; + } catch (err) { + if (!(err instanceof WorkerPoolInitializationError)) throw err; + // Every worker crashed during startup and the pool's bounded self-heal + // was exhausted. Fail fast (#1741) — there is no sequential parser to + // degrade to. `handleWorkerStartupFailure` always throws, so + // `missResults` stays definitely assigned for the parked round below. + handleWorkerStartupFailure(err); + } + pendingRound = { entries: started.entries, missResults }; }; for (let chunkIdx = 0; chunkIdx < numChunks; chunkIdx++) { @@ -1042,13 +1279,8 @@ export async function runChunkedParseAndResolve( // Cache hit: replay cached worker output. Finalize any parked worker // chunk FIRST so deferred aggregation stays in chunk order, then merge // + apply this hit inline (no worker dispatch to overlap). - if (pendingWorkerChunk) { - await finalizeWorkerChunk(pendingWorkerChunk); - pendingWorkerChunk = null; - } chunkCacheHits++; parseCacheHitFileCount += chunkFiles.length; - const chunkWorkerData = mergeChunkResults(graph, symbolTable, cachedRaw, exportedTypeMap); if (isDev) { logger.info( `📦 parse-cache HIT: chunk ${chunkIdx + 1}/${numChunks} (${chunkFiles.length} files, ${chunkHash?.slice(0, 8) ?? 'unknown'})`, @@ -1062,90 +1294,44 @@ export async function runChunkedParseAndResolve( // takes 70-95 so the UI advances through the (potentially long) // resolution stages instead of holding at 82 (M2 from PR #1693 // review). - percent: Math.round(20 + ((filesParsedSoFar + cachedFiles) / totalParseable) * 50), + percent: Math.round(20 + ((queuedFilesSoFar + cachedFiles) / totalParseable) * 50), message: `Parsing chunk ${chunkIdx + 1}/${numChunks} (cache)...`, stats: { - filesProcessed: filesParsedSoFar + cachedFiles, + filesProcessed: queuedFilesSoFar + cachedFiles, totalFiles: totalParseable, nodesCreated: graph.nodeCount, }, }); // The durable gate already snapshotted warm `.v8` shards into the - // run-scoped store for scope resolution. - await applyChunkResults(chunkWorkerData, chunkIdx, chunkFiles, chunkStartMs); + // run-scoped store for scope resolution. Queue into the round so this + // hit still finalizes in `chunkIdx` order relative to its neighbours. + roundEntries.push({ + kind: 'hit', + chunkIdx, + fileCount: chunkFiles.length, + chunkStartMs, + cachedRaw, + }); + for (const file of chunkFiles) roundBufferedBytes += file.content.length; + queuedFilesSoFar += chunkFiles.length; } else { - // Cache miss: dispatch to workers, capture the raw results, store - // them under the chunk hash for the next run. + // Cache miss: queue for the round's single dispatch; the raw results + // are stored under the chunk hash when the round drains. chunkCacheMisses++; reparsedFileCount += chunkFiles.length; - if (durableParsedFileDir !== undefined && chunkHash !== null) { - try { - await prepareDurableParsedFileChunk(durableParsedFileDir, chunkHash); - } catch (err) { - // The durable store is an optimization — degrade like the restore - // path does instead of failing the analyze. Workers recreate the - // directory on write, so at worst the old generation lingers. - logger.warn( - { err, chunkHash: chunkHash.slice(0, 8) }, - 'parsedfile-cache: could not reset durable chunk generation; continuing', - ); - } + roundEntries.push({ kind: 'miss', chunkIdx, chunkHash, chunkFiles, chunkStartMs }); + for (const file of chunkFiles) { + roundMissBytes += file.content.length; + roundBufferedBytes += file.content.length; } - const progressForChunk = (current: number, _total: number, filePath: string) => { - const globalCurrent = filesParsedSoFar + current; - // Parse phase covers 20-70 (M2). Deferred extraction handles 70-95. - const parsingProgress = 20 + (globalCurrent / totalParseable) * 50; - onProgress({ - phase: 'parsing', - percent: Math.round(parsingProgress), - message: `Parsing chunk ${chunkIdx + 1}/${numChunks}...`, - detail: filePath, - stats: { - filesProcessed: globalCurrent, - totalFiles: totalParseable, - nodesCreated: graph.nodeCount, - }, - }); - }; - const activeWorkerPool = getOrCreateWorkerPool(); - // Worker path — PIPELINE: kick off this chunk's dispatch, merge the - // PREVIOUS chunk while these workers parse, then park this chunk for - // the next iteration to merge (overlapping its parse). The deferred - // merge + parse-cache write-guard + aggregation all run in - // `finalizeWorkerChunk`, in chunk order. The pool is the sole parse - // path — `getOrCreateWorkerPool` returns a pool or throws. - const dispatchPromise = dispatchChunkParse( - chunkFiles, - activeWorkerPool, - progressForChunk, - undefined, - chunkHash ?? undefined, - ); - // Mark handled so a rejection during the overlap drain below isn't - // flagged as unhandled; the `await` re-throws it for real handling. - dispatchPromise.catch(() => {}); - if (pendingWorkerChunk) { - await finalizeWorkerChunk(pendingWorkerChunk); - pendingWorkerChunk = null; - } - let chunkResults: ParseWorkerResult[]; - try { - chunkResults = await dispatchPromise; - } catch (err) { - if (!(err instanceof WorkerPoolInitializationError)) throw err; - // Every worker crashed during startup and the pool's bounded - // self-heal was exhausted. Fail fast (#1741) — there is no sequential - // parser to degrade to. `handleWorkerStartupFailure` always throws, so - // `chunkResults` stays definitely assigned for the parked chunk below. - handleWorkerStartupFailure(err); - } - pendingWorkerChunk = { - rawResults: chunkResults, - chunkIdx, - chunkHash, - chunkFiles, - chunkStartMs, - }; + queuedFilesSoFar += chunkFiles.length; + } + + // Close on EITHER cap. `roundMissBytes` sizes the worker round; + // `roundBufferedBytes` bounds what the main thread is holding, which is + // the only cap a warm run can ever reach. + if (roundMissBytes >= roundByteBudget || roundBufferedBytes >= roundByteBudget) { + await closeRound(); } // (Per-chunk aggregation + parse-cache write + throughput log now run in @@ -1155,11 +1341,13 @@ export async function runChunkedParseAndResolve( // scope-resolution phase, RING4-2 #943.) } - // Drain the final parked worker chunk — the last pipelined chunk has no - // successor to overlap its merge with, so merge + finalize it here. - if (pendingWorkerChunk) { - await finalizeWorkerChunk(pendingWorkerChunk); - pendingWorkerChunk = null; + // Drain the tail: close the partially-filled round, then drain the round + // it parked — the last round has no successor to overlap its merge with. + if (roundEntries.length > 0) await closeRound(); + if (pendingRound) { + const last = pendingRound; + pendingRound = null; + await drainRound(last); } if (isDev && parseCache && (chunkCacheHits > 0 || chunkCacheMisses > 0)) { diff --git a/gitnexus/src/core/ingestion/workers/worker-pool.ts b/gitnexus/src/core/ingestion/workers/worker-pool.ts index 497100c64..e5186c6e1 100644 --- a/gitnexus/src/core/ingestion/workers/worker-pool.ts +++ b/gitnexus/src/core/ingestion/workers/worker-pool.ts @@ -96,10 +96,45 @@ export function buildDispatchMessage(items: readonly T[]): { transferList, }; } +/** + * One content-addressed parse-cache chunk's worth of work inside a pool round. + * See {@link WorkerPool.dispatchGroups}. + */ +export interface DispatchGroup { + readonly items: readonly TInput[]; + /** + * Chunk hash tagged onto every job derived from `items`, exactly as the + * `chunkHash` argument of {@link WorkerPool.dispatch} does for a lone chunk. + */ + readonly chunkHash?: string; +} + export interface WorkerPool { /** - * Dispatch items across workers. Items are split into bounded jobs, each job - * is committed independently, and stalled jobs are split/retried locally. + * Dispatch several content-addressed chunks in ONE pool round. + * + * `dispatch` is a barrier: it resolves only once every job it created has + * committed, so dispatching one small parse-cache pack at a time leaves most + * slots idle for the whole round-trip. Stable packs are keyed by + * `(language, hash(path) % 128)`, which routinely yields packs far below the + * byte budget — on this repo, 1285 packs where the budget alone needs 16, and + * 549 of them hold a single file. Batching packs into one round removes those + * barriers without touching pack identity: jobs are still cut at group + * boundaries, so each job carries exactly one `chunkHash` and every result + * stays attributable to the pack that owns its cache key. + * + * Returns one result array per input group, in input order. A group whose + * items were all quarantined yields an empty array. + */ + dispatchGroups( + groups: readonly DispatchGroup[], + onProgress?: (filesProcessed: number) => void, + ): Promise; + + /** + * Dispatch ONE chunk across workers — {@link WorkerPool.dispatchGroups} with + * a single group. Items are split into bounded jobs, each job is committed + * independently, and stalled jobs are split/retried locally. * * Files in {@link WorkerPool.getQuarantinedPaths} are filtered out before * dispatch — they have already caused a worker death this pool lifetime and @@ -863,15 +898,21 @@ function inFlightExcludePath(job: WorkerJob, lastProgress: numbe return path ? [path] : []; } +/** + * Cut `items` into bounded jobs. `startIndexOffset` places those jobs on a + * shared index space so several groups can be laid out end to end in one + * dispatch round and every result still sorts back into global input order. + */ function createJobs( - items: TInput[], + items: readonly TInput[], maxItems: number, maxBytes: number, timeoutMs: number, chunkHash?: string, + startIndexOffset = 0, ): WorkerJob[] { const jobs: WorkerJob[] = []; - let startIndex = 0; + let startIndex = startIndexOffset; let batch: TInput[] = []; let batchBytes = 0; @@ -1263,11 +1304,44 @@ export const createWorkerPool = ( workers.map((_, i) => bringSlotReady(i)), ).then(() => undefined); - const dispatch = async ( - items: TInput[], + /** + * Guards the one-dispatch-at-a-time contract. The dispatch machinery keeps + * its jobs/busy-slot/in-flight state per call, so two concurrent dispatches + * hand the same slots out twice: both stall, and the failure surfaces only + * when every worker hits its idle timeout (10s+ of a wedged pool with no + * indication of the cause). Fail loudly at the call instead. + */ + let dispatchInFlight = false; + + /** + * Claim the pool synchronously, then run the dispatch. The claim CANNOT be + * taken inside `dispatchGroupsInner`: its first statement awaits the + * readiness gate, so two calls made in the same tick would both get past the + * check before either set the flag. + */ + const dispatchGroups = ( + groups: readonly DispatchGroup[], onProgress?: (filesProcessed: number) => void, - chunkHash?: string, - ): Promise => { + ): Promise => { + if (dispatchInFlight) { + return Promise.reject( + new WorkerPoolDispatchError( + 'Worker pool dispatch is already in flight. `dispatch`/`dispatchGroups` is not ' + + 'reentrant — await the previous call before starting another on the same pool.', + [], + ), + ); + } + dispatchInFlight = true; + return dispatchGroupsInner(groups, onProgress).finally(() => { + dispatchInFlight = false; + }); + }; + + const dispatchGroupsInner = async ( + groups: readonly DispatchGroup[], + onProgress?: (filesProcessed: number) => void, + ): Promise => { // Await the initial-spawn readiness gate (F13). On first dispatch // this blocks for up to poolOptions.workerReadyTimeoutMs while every initial // worker's `{type:'ready'}` handshake is checked; on subsequent @@ -1285,7 +1359,8 @@ export const createWorkerPool = ( [], ); } - if (items.length === 0) return []; + const emptyPerGroup = (): TResult[][] => groups.map(() => []); + if (groups.every((group) => group.items.length === 0)) return emptyPerGroup(); if (activeSlots.size === 0) { const detail = initialReadinessFailures.length > 0 @@ -1308,30 +1383,51 @@ export const createWorkerPool = ( // Layer 3: filter out quarantined paths so a known-bad file never reaches // a worker again this pool lifetime. The caller queries // `getQuarantinedPaths` after dispatch to route filtered items. - const dispatchableItems: TInput[] = []; - for (const item of items) { - const path = itemPath(item); - if (path !== undefined && quarantine.has(path)) continue; - dispatchableItems.push(item); - } - if (dispatchableItems.length === 0) return []; + const dispatchableGroups = groups.map((group) => { + const items: TInput[] = []; + for (const item of group.items) { + const path = itemPath(item); + if (path !== undefined && quarantine.has(path)) continue; + items.push(item); + } + return { items, chunkHash: group.chunkHash }; + }); + const dispatchableCount = dispatchableGroups.reduce( + (sum, group) => sum + group.items.length, + 0, + ); + if (dispatchableCount === 0) return emptyPerGroup(); // Stable cache packs can be much smaller than either job ceiling. Split // those packs across the live slots too, otherwise each serial dispatch // feeds only one worker. Keep both configured ceilings as upper bounds. const maxItemsPerJob = Math.min( poolOptions.subBatchSize, - Math.max(1, Math.floor(dispatchableItems.length / activeSlots.size)), - ); - const jobs = createJobs( - dispatchableItems, - maxItemsPerJob, - poolOptions.subBatchMaxBytes, - poolOptions.subBatchIdleTimeoutMs, - chunkHash, + Math.max(1, Math.floor(dispatchableCount / activeSlots.size)), ); + // Lay the groups end to end on one index space and cut jobs at every group + // boundary. A job therefore belongs to exactly one group, which is what + // lets a result be attributed back to the parse-cache chunk that owns it + // (and what keeps `chunkHash` a per-job constant through splits/requeues). + const jobs: WorkerJob[] = []; + const groupEnds: number[] = []; + let groupStart = 0; + for (const group of dispatchableGroups) { + for (const job of createJobs( + group.items, + maxItemsPerJob, + poolOptions.subBatchMaxBytes, + poolOptions.subBatchIdleTimeoutMs, + group.chunkHash, + groupStart, + )) { + jobs.push(job); + } + groupStart += group.items.length; + groupEnds.push(groupStart); + } - return new Promise((resolve, reject) => { + return await new Promise((resolve, reject) => { const results: WorkerJobResult[] = []; const inFlightProgress = new Array(size).fill(0); // Tracks which slots are currently mid-job so the "wake idle slots" @@ -1359,10 +1455,7 @@ export const createWorkerPool = ( const reportProgress = () => { if (!onProgress) return; const inFlight = inFlightProgress.reduce((sum, value) => sum + value, 0); - const next = Math.min( - dispatchableItems.length, - Math.max(maxReported, completedFiles + inFlight), - ); + const next = Math.min(dispatchableCount, Math.max(maxReported, completedFiles + inFlight)); if (next === maxReported) return; maxReported = next; onProgress(next); @@ -1551,9 +1644,19 @@ export const createWorkerPool = ( if (jobs.length === 0 && activeWorkers === 0) { stopped = true; results.sort((a, b) => a.startIndex - b.startIndex); - if (onProgress && maxReported < dispatchableItems.length) - onProgress(dispatchableItems.length); - resolve(results.map((result) => result.data)); + if (onProgress && maxReported < dispatchableCount) onProgress(dispatchableCount); + // Partition back per group. Job (and split sub-job) start indices + // stay inside their group's span, so a single forward walk over the + // sorted results assigns every result to exactly one group. + const perGroup: TResult[][] = groupEnds.map(() => []); + let groupIdx = 0; + for (const result of results) { + while (groupIdx < groupEnds.length - 1 && result.startIndex >= groupEnds[groupIdx]) { + groupIdx++; + } + perGroup[groupIdx].push(result.data); + } + resolve(perGroup); } }; @@ -2294,8 +2397,18 @@ export const createWorkerPool = ( activeSlots.clear(); }; + const dispatch = async ( + items: TInput[], + onProgress?: (filesProcessed: number) => void, + chunkHash?: string, + ): Promise => { + const [result] = await dispatchGroups([{ items, chunkHash }], onProgress); + return result ?? []; + }; + return { dispatch, + dispatchGroups, terminate, size, getQuarantinedPaths: () => quarantine.snapshot(), diff --git a/gitnexus/test/integration/parse-impl-dispatch-rounds.test.ts b/gitnexus/test/integration/parse-impl-dispatch-rounds.test.ts new file mode 100644 index 000000000..56d92618c --- /dev/null +++ b/gitnexus/test/integration/parse-impl-dispatch-rounds.test.ts @@ -0,0 +1,225 @@ +/** + * Dispatch rounds — batching cache packs into one pool round. + * + * `WorkerPool.dispatch` is a barrier, so one round-trip per parse-cache pack + * strands the pool whenever a pack is smaller than it. Packs are keyed by + * `(language, hash(path) % 128)`, so on a real repo most of them are: this + * repository produces 1285 packs where the byte budget alone needs 16, and 549 + * of those hold a single file. Chunks now accumulate into a round bounded by + * `GITNEXUS_PARSE_ROUND_BYTES` and go out in one `dispatchGroups` call. + * + * Batching must be invisible to the graph. These tests pin the two ways it + * could stop being invisible: + * 1. Ordering — deferred aggregation runs in `chunkIdx` order, so the graph + * must not depend on how chunks were grouped into rounds. + * 2. Attribution — a round returns one result array per pack, so a pack's + * parse-cache entry must hold ITS OWN worker output. Mis-attribution would + * survive a cold run and only surface as a corrupted warm replay, which is + * what the second test exercises. + */ +import { describe, it, expect, beforeEach, afterEach } from 'vitest'; +import fs from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; + +import { runChunkedParseAndResolve } from '../../src/core/ingestion/pipeline-phases/parse-impl.js'; +import { createKnowledgeGraph } from '../../src/core/graph/graph.js'; +import { PARSE_CACHE_VERSION, packParseCacheChunks } from '../../src/storage/parse-cache.js'; +import type { ParseWorkerResult } from '../../src/core/ingestion/workers/parse-worker.js'; + +const ORIGINAL_ROUND_BYTES = process.env.GITNEXUS_PARSE_ROUND_BYTES; + +/** + * Enough files, across enough languages, that `(language, bucket)` packing + * yields many more packs than the byte budget would — the shape that makes + * per-pack dispatch a barrier problem in the first place. + */ +const FIXTURE: ReadonlyArray<[string, string]> = [ + ...Array.from({ length: 12 }, (_, i): [string, string] => [ + `src/mod${i}.ts`, + `export function ts${i}() { return ${i}; }\n`, + ]), + ...Array.from({ length: 8 }, (_, i): [string, string] => [ + `src/mod${i}.py`, + `def py${i}():\n return ${i}\n`, + ]), + ...Array.from({ length: 6 }, (_, i): [string, string] => [ + `src/Mod${i}.java`, + `public class Mod${i} { public int go() { return ${i}; } }\n`, + ]), + ...Array.from({ length: 6 }, (_, i): [string, string] => [ + `src/mod${i}.go`, + `package main\n\nfunc Go${i}() int { return ${i} }\n`, + ]), +]; + +describe('parse-impl dispatch rounds', () => { + let repoPath = ''; + let storageDir = ''; + + beforeEach(() => { + repoPath = fs.mkdtempSync(path.join(os.tmpdir(), 'parse-impl-dispatch-rounds-')); + storageDir = fs.mkdtempSync(path.join(os.tmpdir(), 'parse-impl-rounds-storage-')); + for (const [rel, content] of FIXTURE) { + const full = path.join(repoPath, rel); + fs.mkdirSync(path.dirname(full), { recursive: true }); + fs.writeFileSync(full, content); + } + }); + + afterEach(() => { + for (const dir of [repoPath, storageDir]) { + if (dir && fs.existsSync(dir)) fs.rmSync(dir, { recursive: true, force: true }); + } + if (ORIGINAL_ROUND_BYTES === undefined) delete process.env.GITNEXUS_PARSE_ROUND_BYTES; + else process.env.GITNEXUS_PARSE_ROUND_BYTES = ORIGINAL_ROUND_BYTES; + }); + + const files = () => + FIXTURE.map(([rel]) => ({ path: rel, size: fs.statSync(path.join(repoPath, rel)).size })); + + /** + * Order-independent fingerprint of the graph. Counts alone would let a + * mis-attributed chunk (right totals, wrong contents) pass. + */ + const fingerprint = (graph: ReturnType): string => + Array.from(graph.nodes.values()) + .map((node) => { + const props = node.properties as { name?: string; filePath?: string } | undefined; + return `${node.label}|${props?.name ?? ''}|${props?.filePath ?? ''}`; + }) + .sort() + .join('\n'); + + const run = async (parseCache?: { + version: string; + entries: Map; + usedKeys: Set; + storagePath: string; + onDiskKeys: Set; + }) => { + const scan = files(); + const rels = scan.map((f) => f.path); + const graph = createKnowledgeGraph(); + await runChunkedParseAndResolve( + graph, + scan, + rels, + scan.length, + repoPath, + Date.now(), + () => {}, + parseCache ? { parseCache } : {}, + ); + return graph; + }; + + it('the fixture really does split into more packs than the byte budget needs', () => { + // Guards the premise: if packing ever stopped over-splitting, the tests + // below would still pass while measuring nothing. + const packs = packParseCacheChunks( + files().map((f) => ({ + path: f.path, + size: f.size, + language: f.path.slice(f.path.lastIndexOf('.') + 1), + })), + 2 * 1024 * 1024, + ); + const totalBytes = files().reduce((sum, f) => sum + f.size, 0); + expect(totalBytes).toBeLessThan(2 * 1024 * 1024); + expect(packs.length).toBeGreaterThan(1); + }); + + it('produces the same graph whether chunks are batched into rounds or dispatched one by one', async () => { + // 1 byte closes a round after every cache-missing chunk — the pre-round + // behaviour, and the control arm for the batched default. + process.env.GITNEXUS_PARSE_ROUND_BYTES = '1'; + const perChunk = await run(); + + delete process.env.GITNEXUS_PARSE_ROUND_BYTES; + const batched = await run(); + + expect(batched.nodeCount).toBe(perChunk.nodeCount); + expect(batched.relationshipCount).toBe(perChunk.relationshipCount); + expect(fingerprint(batched)).toBe(fingerprint(perChunk)); + // Pin real symbols so an empty-graph regression cannot satisfy the above. + const names = fingerprint(batched); + expect(names).toContain('ts0'); + expect(names).toContain('py0'); + expect(names).toContain('Mod0'); + expect(names).toContain('Go0'); + }); + + it('keeps hit and miss chunks attributed to their own files inside one round', async () => { + // The realistic incremental shape: some packs warm, some cold, batched into + // the SAME round. `drainRound` walks the round's entries in `chunkIdx` + // order but pulls worker output with a separate `missIdx` cursor, so a + // hit sitting between two misses is exactly where that cursor can slip. + // Cold-then-warm alone never exercises it -- every entry is the same kind. + const cache = { + version: PARSE_CACHE_VERSION, + entries: new Map(), + usedKeys: new Set(), + storagePath: storageDir, + onDiskKeys: new Set(), + }; + + const cold = await run(cache); + const cachedPacks = cache.onDiskKeys.size + cache.entries.size; + expect(cachedPacks).toBeGreaterThan(1); + + // Edit ONE file. Its pack now misses; every other pack still hits, so the + // next run's rounds carry both kinds together. + fs.writeFileSync( + path.join(repoPath, 'src/mod0.ts'), + 'export function ts0() { return 999; }\nexport function ts0Extra() { return 1; }\n', + ); + + const mixed = await run(cache); + + // The edited file's NEW symbol must be present, proving the miss chunk's + // fresh worker output landed under its own file... + const mixedPrint = fingerprint(mixed); + expect(mixedPrint).toContain('ts0Extra'); + // ...and every untouched file's symbols must still be present and attached + // to their own paths, proving no hit chunk was overwritten by, or swapped + // with, a neighbouring miss chunk's results. + const coldPrint = fingerprint(cold); + const untouched = coldPrint + .split('\n') + .filter((entry) => !entry.endsWith('|src/mod0.ts')) + .sort(); + const mixedUntouched = mixedPrint + .split('\n') + .filter((entry) => !entry.endsWith('|src/mod0.ts')) + .sort(); + expect(mixedUntouched).toEqual(untouched); + }); + + it('stores each pack’s own worker output, so a warm replay reproduces the cold graph', async () => { + const cache = { + version: PARSE_CACHE_VERSION, + entries: new Map(), + usedKeys: new Set(), + storagePath: storageDir, + onDiskKeys: new Set(), + }; + + // Cold: every pack misses, and the round writes each pack's results under + // that pack's own hash. + const cold = await run(cache); + // With a storagePath the chunk bodies land on disk and the hash is tracked + // in `onDiskKeys`; without one they stay in `entries`. Count both so the + // assertion pins "more than one pack was cached", not the storage route. + expect(cache.onDiskKeys.size + cache.entries.size).toBeGreaterThan(1); + expect(cache.usedKeys.size).toBe(cache.onDiskKeys.size + cache.entries.size); + + // Warm: every pack replays from its cache entry with no worker dispatch. + // If a round had attributed pack A's results to pack B's key, the replayed + // graph would differ here even though the cold run looked correct. + const warm = await run(cache); + expect(fingerprint(warm)).toBe(fingerprint(cold)); + expect(warm.nodeCount).toBe(cold.nodeCount); + expect(warm.relationshipCount).toBe(cold.relationshipCount); + }); +}); diff --git a/gitnexus/test/integration/parse-impl-env-reads.test.ts b/gitnexus/test/integration/parse-impl-env-reads.test.ts index 2d93e12dd..449691ea8 100644 --- a/gitnexus/test/integration/parse-impl-env-reads.test.ts +++ b/gitnexus/test/integration/parse-impl-env-reads.test.ts @@ -50,11 +50,13 @@ function scanned(repo: string, files: string[]) { } /** - * Capture every per-chunk progress message emitted during a run. - * parse-impl emits one per chunk in the "Parsing chunk X/Y" form, so - * counting unique chunk indices in the captured stream is a stable - * proxy for the number of chunks the loop actually produced. Avoids - * exposing internal counter state from parse-impl. + * Read the chunk count out of the progress stream. parse-impl reports progress + * as "Parsing chunk X/Y" for a single chunk and "Parsing chunks X-Z/Y" when a + * dispatch round batches several — so the DENOMINATOR, not the number of + * distinct messages, is the count of packs the loop produced. Reading `Y` + * keeps this independent of how chunks are grouped into rounds while still + * exercising the real budget-resolution path inside + * `runChunkedParseAndResolve`, rather than re-deriving packs in the test. */ async function countChunksFromProgress( repoPath: string, @@ -63,7 +65,7 @@ async function countChunksFromProgress( ): Promise { const scan = scanned(repoPath, files); const graph = createKnowledgeGraph(); - const chunkIndices = new Set(); + const totals = new Set(); await runChunkedParseAndResolve( graph, scan, @@ -73,8 +75,8 @@ async function countChunksFromProgress( Date.now(), (p) => { if (typeof p.message !== 'string') return; - const m = /Parsing chunk (\d+)\/(\d+)/.exec(p.message); - if (m !== null) chunkIndices.add(`${m[1]}/${m[2]}`); + const m = /Parsing chunks? \d+(?:-\d+)?\/(\d+)/.exec(p.message); + if (m !== null) totals.add(Number(m[1])); }, // Chunk count is byte-budget-driven and emitted before the pool runs, so it // is independent of worker vs sequential. Sequential parsing was removed, so @@ -82,7 +84,10 @@ async function countChunksFromProgress( // integration tier. { ...options }, ); - return chunkIndices.size; + // Every message in a run carries the same denominator; more than one value + // would mean the loop changed its chunk count mid-run. + expect(totals.size).toBeLessThanOrEqual(1); + return totals.values().next().value ?? 0; } describe('parse-impl chunkByteBudget resolution (U14 / F7)', () => { diff --git a/gitnexus/test/integration/worker-pool.test.ts b/gitnexus/test/integration/worker-pool.test.ts index da910a736..ac6fe0076 100644 --- a/gitnexus/test/integration/worker-pool.test.ts +++ b/gitnexus/test/integration/worker-pool.test.ts @@ -840,6 +840,156 @@ describe('worker pool integration', () => { }, ); + it('splits packs into one round, keeping results and chunk hashes per group', async () => { + const { tempDir, workerPath } = writeTempWorker( + 'gitnexus-worker-dispatch-groups-', + ` + const { parentPort, threadId } = require('node:worker_threads'); + let paths = []; + parentPort.on('message', (msg) => { + if (msg && msg.type === 'sub-batch') { + for (const file of msg.files) paths.push(file.path); + parentPort.postMessage({ type: 'progress', filesProcessed: msg.files.length }); + parentPort.postMessage({ type: 'sub-batch-done' }); + } else if (msg && msg.type === 'flush') { + parentPort.postMessage({ + type: 'result', + data: { paths, threadId, chunkHash: msg.chunkHash }, + }); + paths = []; + } + }); + `, + ); + pool = createWorkerPool(pathToFileURL(workerPath), 4); + + try { + // Shaped like real cache packs: mostly tiny, one larger. A single round + // must still keep every result attributable to the pack that owns it. + const groups = [ + { chunkHash: 'hash-a', items: [{ path: 'a0.ts', content: 'export const a0 = 1;' }] }, + { chunkHash: 'hash-b', items: [{ path: 'b0.ts', content: 'export const b0 = 1;' }] }, + { + chunkHash: 'hash-c', + items: Array.from({ length: 12 }, (_, i) => ({ + path: `c${i}.ts`, + content: 'export const c = 1;', + })), + }, + { chunkHash: 'hash-d', items: [{ path: 'd0.ts', content: 'export const d0 = 1;' }] }, + ]; + const progress: number[] = []; + const perGroup = await pool.dispatchGroups< + (typeof groups)[number]['items'][number], + { paths: string[]; threadId: number; chunkHash?: string } + >(groups, (completed) => progress.push(completed)); + + expect(perGroup).toHaveLength(groups.length); + // No job straddles a pack: every path comes back under its own group, + // and every result carries that group's chunk hash (the cache key). + for (const [index, group] of groups.entries()) { + expect(perGroup[index].flatMap((result) => result.paths).sort()).toEqual( + group.items.map((file) => file.path).sort(), + ); + for (const result of perGroup[index]) expect(result.chunkHash).toBe(group.chunkHash); + } + // The whole round shares the pool rather than one pack per barrier. + const threads = new Set(perGroup.flat().map((result) => result.threadId)); + expect(threads.size).toBeGreaterThan(1); + expect(progress).toEqual([...progress].sort((a, b) => a - b)); + expect(progress.at(-1)).toBe(15); + } finally { + await pool.terminate(); + fs.rmSync(tempDir, { recursive: true, force: true }); + } + }); + + it('returns an empty result array for a group whose items were all quarantined', async () => { + const { tempDir, workerPath } = writeTempWorker( + 'gitnexus-worker-groups-quarantine-', + ` + const { parentPort } = require('node:worker_threads'); + let paths = []; + parentPort.on('message', (msg) => { + if (msg && msg.type === 'sub-batch') { + for (const file of msg.files) { + if (file.path === 'poison.ts') process.exit(134); + paths.push(file.path); + } + parentPort.postMessage({ type: 'progress', filesProcessed: msg.files.length }); + parentPort.postMessage({ type: 'sub-batch-done' }); + } else if (msg && msg.type === 'flush') { + parentPort.postMessage({ type: 'result', data: { paths } }); + paths = []; + } + }); + `, + ); + pool = createWorkerPool(pathToFileURL(workerPath), 2); + + try { + await pool.dispatch([{ path: 'poison.ts', content: '' }]); + expect(pool.getQuarantinedPaths?.()).toContain('poison.ts'); + + // Group alignment must survive quarantine filtering — an all-quarantined + // group still occupies its slot so results line up with the input packs. + const perGroup = await pool.dispatchGroups< + { path: string; content: string }, + { paths: string[] } + >([ + { chunkHash: 'poisoned', items: [{ path: 'poison.ts', content: '' }] }, + { chunkHash: 'healthy', items: [{ path: 'ok.ts', content: 'export const ok = 1;' }] }, + ]); + + expect(perGroup).toHaveLength(2); + expect(perGroup[0]).toEqual([]); + expect(perGroup[1].flatMap((result) => result.paths)).toEqual(['ok.ts']); + } finally { + await pool.terminate(); + fs.rmSync(tempDir, { recursive: true, force: true }); + } + }); + + it('rejects a second dispatch while one is still in flight', async () => { + const { tempDir, workerPath } = writeTempWorker( + 'gitnexus-worker-reentrant-dispatch-', + ` + const { parentPort } = require('node:worker_threads'); + let paths = []; + parentPort.on('message', (msg) => { + if (msg && msg.type === 'sub-batch') { + paths = msg.files.map((file) => file.path); + parentPort.postMessage({ type: 'progress', filesProcessed: paths.length }); + parentPort.postMessage({ type: 'sub-batch-done' }); + } else if (msg && msg.type === 'flush') { + setTimeout(() => parentPort.postMessage({ type: 'result', data: { paths } }), 150); + } + }); + `, + ); + pool = createWorkerPool(pathToFileURL(workerPath), 2); + + try { + // Concurrent dispatches hand the same slots out twice; both then stall + // until every worker idle-times out. Fail at the call, not 10s later. + const first = pool.dispatch<{ path: string; content: string }, { paths: string[] }>([ + { path: 'first.ts', content: 'export const first = 1;' }, + ]); + await expect( + pool.dispatch([{ path: 'second.ts', content: 'export const second = 1;' }]), + ).rejects.toThrow(/not reentrant/); + await expect(first).resolves.toHaveLength(1); + + // The guard clears once the in-flight dispatch settles. + await expect( + pool.dispatch([{ path: 'third.ts', content: 'export const third = 1;' }]), + ).resolves.toHaveLength(1); + } finally { + await pool.terminate(); + fs.rmSync(tempDir, { recursive: true, force: true }); + } + }); + it('bounds worker jobs by byte budget as well as file count', async () => { const { tempDir, workerPath } = writeTempWorker( 'gitnexus-worker-byte-budget-', diff --git a/gitnexus/test/unit/parsing-worker-fallback.test.ts b/gitnexus/test/unit/parsing-worker-fallback.test.ts index e88f8b4aa..c3ce4bb6c 100644 --- a/gitnexus/test/unit/parsing-worker-fallback.test.ts +++ b/gitnexus/test/unit/parsing-worker-fallback.test.ts @@ -37,7 +37,8 @@ describe('processParsing — worker-pool error propagation (U20)', () => { const graph = createKnowledgeGraph(); const workerPool: WorkerPool = { size: 1, - dispatch: vi.fn(async () => { + dispatch: vi.fn(async () => []), + dispatchGroups: vi.fn(async () => { throw new Error('replacement worker failed'); }), terminate: vi.fn(async () => undefined), @@ -63,7 +64,8 @@ describe('processParsing — worker-pool error propagation (U20)', () => { const graph = createKnowledgeGraph(); const workerPool: WorkerPool = { size: 1, - dispatch: vi.fn(async () => { + dispatch: vi.fn(async () => []), + dispatchGroups: vi.fn(async () => { throw new WorkerPoolDispatchError( 'Worker pool circuit breaker tripped: 2 consecutive failures on slot 0', ['src/poison.ts'], @@ -114,6 +116,7 @@ describe('processParsing — worker-pool error propagation (U20)', () => { const workerPool: WorkerPool = { size: 1, dispatch: vi.fn(async () => []), + dispatchGroups: vi.fn(async () => []), terminate: vi.fn(async () => undefined), getQuarantinedPaths: () => ['src/poison.ts'], }; From 1c1cbf111e56415766ee8667c70dc24b1f0bade1 Mon Sep 17 00:00:00 2001 From: Parafee41 Date: Mon, 7 Sep 2026 14:19:30 +0800 Subject: [PATCH 03/17] fix(zig): resolve cross-file static gates (#3185) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(zig): resolve cross-file static gates * Fix Zig workspace import alias enrichment * Handle extensionless Zig workspace imports * Reuse Zig import resolution for static gates * Document Zig workspace static gating * Harden Zig workspace reference enrichment * Benchmark Zig cross-file static gating * Enforce linear Zig benchmark scaling --------- Co-authored-by: Gergő Magyar --- .github/workflows/ci-tests.yml | 7 + .../zig-cross-file-resolution/baseline.json | 16 ++ .../zig-cross-file-resolution/measure.mjs | 138 ++++++++++++++++++ .../call-extractors/zig-static-gating.ts | 13 +- .../ingestion/languages/zig/scope-resolver.ts | 3 + .../languages/zig/workspace-static-gating.ts | 106 ++++++++++++++ .../contract/scope-resolver.ts | 15 ++ .../scope-resolution/pipeline/phase.ts | 1 + .../scope-resolution/pipeline/run.ts | 5 + .../zig-static-gating/src/main.zig | 43 ++++++ .../zig-static-gating/src/other.zig | 1 + .../resolvers/zig-static-gating.test.ts | 67 ++++++++- 12 files changed, 400 insertions(+), 15 deletions(-) create mode 100644 gitnexus/bench/zig-cross-file-resolution/baseline.json create mode 100644 gitnexus/bench/zig-cross-file-resolution/measure.mjs create mode 100644 gitnexus/src/core/ingestion/languages/zig/workspace-static-gating.ts create mode 100644 gitnexus/test/fixtures/lang-resolution/zig-static-gating/src/other.zig diff --git a/.github/workflows/ci-tests.yml b/.github/workflows/ci-tests.yml index b1caad07f..a0d4abcad 100644 --- a/.github/workflows/ci-tests.yml +++ b/.github/workflows/ci-tests.yml @@ -745,6 +745,13 @@ jobs: run: node --import tsx bench/scope-emission/measure.mjs --check working-directory: gitnexus + - name: Zig cross-file static-gating guards (#3162) + if: ${{ !cancelled() }} + # Build-free: fingerprints cross-file dead-call classification and + # guards the workspace enrichment pass across file-count scaling. + run: node --import tsx bench/zig-cross-file-resolution/measure.mjs --check + working-directory: gitnexus + - name: CFG construction time / disk / memory guards (#2081 M1) if: ${{ !cancelled() }} # Build-free: asserts collectFunctionCfgs output is unchanged diff --git a/gitnexus/bench/zig-cross-file-resolution/baseline.json b/gitnexus/bench/zig-cross-file-resolution/baseline.json new file mode 100644 index 000000000..0626ec2e1 --- /dev/null +++ b/gitnexus/bench/zig-cross-file-resolution/baseline.json @@ -0,0 +1,16 @@ +{ + "_comment": "Correctness counts are exact. Timing budgets are deliberately loose and only guard large regressions in the post-extraction Zig workspace pass.", + "small": { + "modules": 40, + "calls_per_module": 12, + "gated_calls": 480, + "ms_budget": 1000 + }, + "large": { + "modules": 160, + "calls_per_module": 12, + "gated_calls": 1920, + "ms_budget": 4000 + }, + "linear_scaling_slack": 1.375 +} diff --git a/gitnexus/bench/zig-cross-file-resolution/measure.mjs b/gitnexus/bench/zig-cross-file-resolution/measure.mjs new file mode 100644 index 000000000..9f0c4e884 --- /dev/null +++ b/gitnexus/bench/zig-cross-file-resolution/measure.mjs @@ -0,0 +1,138 @@ +#!/usr/bin/env node +/** + * Build-free scaling and correctness guard for Zig cross-file static gates. + * + * The workspace pass parses every indexed Zig file, resolves direct @import + * aliases, then applies sibling boolean constants to call sites. This bench + * makes both relevant axes explicit: file count and calls per importer. It + * also fingerprints the number of calls classified dead, so a fast no-op + * implementation cannot pass the timing gate. + * + * Usage: + * node --import tsx bench/zig-cross-file-resolution/measure.mjs + * node --import tsx bench/zig-cross-file-resolution/measure.mjs --check + */ +import { readFileSync } from 'node:fs'; +import { dirname, join } from 'node:path'; +import { fileURLToPath } from 'node:url'; +import { performance } from 'node:perf_hooks'; +import { populateZigWorkspaceStaticGating } from '../../src/core/ingestion/languages/zig/workspace-static-gating.ts'; + +const HERE = dirname(fileURLToPath(import.meta.url)); +const SMALL_MODULES = 40; +const LARGE_MODULES = 160; +const CALLS_PER_MODULE = 12; +const REPS = 7; + +const range = (line, col) => ({ startLine: line, startCol: col, endLine: line, endCol: col + 4 }); + +function corpus(modules) { + const parsedFiles = []; + const fileContents = new Map(); + for (let i = 0; i < modules; i++) { + const cfgPath = `bench/cfg${i}.zig`; + const appPath = `bench/app${i}.zig`; + fileContents.set(cfgPath, 'pub const ENABLED = false;\n'); + fileContents.set( + appPath, + `const cfg = @import(\"./cfg${i}.zig\");\n` + + Array.from( + { length: CALLS_PER_MODULE }, + (_, j) => `pub fn run${j}() void { if (cfg.ENABLED) dead${j}(); }`, + ).join('\n'), + ); + parsedFiles.push(Object.freeze({ filePath: cfgPath, referenceSites: Object.freeze([]) })); + parsedFiles.push( + Object.freeze({ + filePath: appPath, + referenceSites: Object.freeze( + Array.from({ length: CALLS_PER_MODULE }, (_, j) => ({ + kind: 'call', + name: `dead${j}`, + atRange: range(j + 2, 44), + })), + ), + }), + ); + } + return { parsedFiles, fileContents }; +} + +function run(modules) { + const { parsedFiles, fileContents } = corpus(modules); + populateZigWorkspaceStaticGating(parsedFiles, { fileContents }); + let gatedCalls = 0; + for (const file of parsedFiles) { + for (const site of file.referenceSites) if (site.staticGated === true) gatedCalls++; + } + return gatedCalls; +} + +function measure(modules) { + run(modules); + let bestMs = Infinity; + let gatedCalls = 0; + for (let i = 0; i < REPS; i++) { + const start = performance.now(); + gatedCalls = run(modules); + bestMs = Math.min(bestMs, performance.now() - start); + } + return { + modules, + calls_per_module: CALLS_PER_MODULE, + gated_calls: gatedCalls, + min_ms: Number(bestMs.toFixed(2)), + }; +} + +const report = { small: measure(SMALL_MODULES), large: measure(LARGE_MODULES) }; +report.workload_ratio = LARGE_MODULES / SMALL_MODULES; +report.scaling_ratio = Number( + (report.large.min_ms / Math.max(report.small.min_ms, 0.01)).toFixed(3), +); +report.linear_factor = Number((report.scaling_ratio / report.workload_ratio).toFixed(3)); + +if (!process.argv.includes('--check')) { + console.log(JSON.stringify(report, null, 2)); + process.exit(0); +} + +const baseline = JSON.parse(readFileSync(join(HERE, 'baseline.json'), 'utf8')); +const failures = []; +const requirePositiveNumber = (path, value) => { + if (typeof value !== 'number' || !Number.isFinite(value) || value <= 0) { + failures.push(`${path}: expected a finite positive number, got ${JSON.stringify(value)}`); + return false; + } + return true; +}; +for (const arm of ['small', 'large']) { + for (const key of ['modules', 'calls_per_module', 'gated_calls']) { + if (report[arm][key] !== baseline[arm][key]) { + failures.push(`${arm}.${key}: expected ${baseline[arm][key]}, got ${report[arm][key]}`); + } + } + if ( + requirePositiveNumber(`${arm}.ms_budget`, baseline[arm].ms_budget) && + report[arm].min_ms > baseline[arm].ms_budget + ) { + failures.push(`${arm}.min_ms ${report[arm].min_ms} exceeds budget ${baseline[arm].ms_budget}`); + } +} +if ( + requirePositiveNumber('linear_scaling_slack', baseline.linear_scaling_slack) && + report.linear_factor > baseline.linear_scaling_slack +) { + failures.push( + `linear_factor ${report.linear_factor} exceeds slack ${baseline.linear_scaling_slack} ` + + `(runtime ${report.scaling_ratio}x for ${report.workload_ratio}x work)`, + ); +} + +console.log(JSON.stringify(report, null, 2)); +if (failures.length > 0) { + console.error('[zig-cross-file-resolution --check] FAIL'); + for (const failure of failures) console.error(` - ${failure}`); + process.exit(1); +} +console.log('[zig-cross-file-resolution --check] PASS'); diff --git a/gitnexus/src/core/ingestion/call-extractors/zig-static-gating.ts b/gitnexus/src/core/ingestion/call-extractors/zig-static-gating.ts index a41f04ae2..7a589e5cb 100644 --- a/gitnexus/src/core/ingestion/call-extractors/zig-static-gating.ts +++ b/gitnexus/src/core/ingestion/call-extractors/zig-static-gating.ts @@ -13,17 +13,14 @@ * Conservative by design: we only tag an edge when we can prove the * gating expression evaluates to `false`. Anything ambiguous → live. * - * Scope of v1: + * Supported scope: * * (a) **File-local** consts (`pub const FOO = false;`, plus const-to-const * aliases up to 5 hops), built once per file by `buildZigBoolConstMap`. - * (b) **Cross-file** (`const cfg = @import("./cfg.zig"); if (cfg.FOO)`) is - * NOT resolved yet. The evaluator keeps the seam for it (`importAliases` - * + `lookupBoolsForPath`, consumed by the `field_expression` case), but - * the only caller passes an empty alias map and a lookup that always - * returns `undefined`, because the capture emitter runs in the parse - * worker and sees only the current file. Tracked in #3162. Until then - * every `cfg.FOO` condition folds to unknown, i.e. live. + * (b) **Cross-file** direct imports (`const cfg = @import("./cfg.zig"); + * if (cfg.FOO)`) are enriched after per-file extraction. The workspace + * caller supplies `importAliases` and `lookupBoolsForPath`; the parse + * worker still uses empty/undefined inputs and remains file-local. * * Also out of scope: multi-hop member access (`cfg.sub.FOO`), re-exported * consts, runtime-evaluated bools (`const FOO = computeIt();`), and diff --git a/gitnexus/src/core/ingestion/languages/zig/scope-resolver.ts b/gitnexus/src/core/ingestion/languages/zig/scope-resolver.ts index 229d91cf3..86afcfbf2 100644 --- a/gitnexus/src/core/ingestion/languages/zig/scope-resolver.ts +++ b/gitnexus/src/core/ingestion/languages/zig/scope-resolver.ts @@ -19,6 +19,7 @@ import { resolveZigImportInternal } from '../../import-resolvers/zig.js'; import { zigProvider } from '../zig.js'; import { expandZigWildcardNames, zigArityCompatibility, zigMergeBindings } from './index.js'; import { populateZigRangeBindings } from './range-binding.js'; +import { populateZigWorkspaceStaticGating } from './workspace-static-gating.js'; export const zigScopeResolver: ScopeResolver = { language: SupportedLanguages.Zig, @@ -67,6 +68,8 @@ export const zigScopeResolver: ScopeResolver = { populateOwners: (parsed: ParsedFile) => populateClassOwnedMembers(parsed), + populateWorkspaceReferences: populateZigWorkspaceStaticGating, + // Payload captures — `for (items) |it|`, `if (opt) |v|`, `while (it.next()) // |x|` — typed from the subject's binding after finalize (F6). populateRangeBindings: populateZigRangeBindings, diff --git a/gitnexus/src/core/ingestion/languages/zig/workspace-static-gating.ts b/gitnexus/src/core/ingestion/languages/zig/workspace-static-gating.ts new file mode 100644 index 000000000..040c058be --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/zig/workspace-static-gating.ts @@ -0,0 +1,106 @@ +import type { ParsedFile, ReferenceSite } from 'gitnexus-shared'; +import { getTreeSitterBufferSize } from '../../constants.js'; +import type { ZigBuildZonConfig } from '../../language-config.js'; +import { resolveZigImportInternal } from '../../import-resolvers/zig.js'; +import { + buildZigBoolConstMap, + collectZigStaticGatedRanges, + isPositionStaticGated, + type ZigImportAliasMap, +} from '../../call-extractors/zig-static-gating.js'; +import { parseSourceSafe, ParseTimeoutError } from '../../../tree-sitter/safe-parse.js'; +import { getZigParser } from './query.js'; + +type ZigTree = ReturnType['parse']>; + +export function populateZigWorkspaceStaticGating( + parsedFiles: ParsedFile[], + ctx: { + readonly fileContents: ReadonlyMap; + readonly treeCache?: { get(filePath: string): unknown }; + readonly resolutionConfig?: unknown; + }, +): void { + const parser = getZigParser(); + const trees = new Map(); + const bools = new Map>(); + + for (const parsed of parsedFiles) { + const source = ctx.fileContents.get(parsed.filePath); + if (source === undefined) continue; + let tree = ctx.treeCache?.get(parsed.filePath) as ZigTree | undefined; + if (tree === undefined) { + try { + tree = parseSourceSafe(parser, source, undefined, { + bufferSize: getTreeSitterBufferSize(source), + }); + } catch (err) { + if (err instanceof ParseTimeoutError) continue; + throw err; + } + } + trees.set(parsed.filePath, tree); + bools.set(parsed.filePath, buildZigBoolConstMap(tree.rootNode)); + } + + const knownPaths = new Set(trees.keys()); + for (const [index, parsed] of parsedFiles.entries()) { + const tree = trees.get(parsed.filePath); + if (tree === undefined) continue; + const aliases = collectImportAliases( + tree, + parsed.filePath, + knownPaths, + ctx.resolutionConfig as ZigBuildZonConfig | null | undefined, + ); + if (aliases.size === 0) continue; + const ranges = collectZigStaticGatedRanges( + tree.rootNode, + bools.get(parsed.filePath) ?? new Map(), + aliases, + (filePath) => bools.get(filePath), + ); + if (ranges.length === 0) continue; + const next = parsed.referenceSites.map((site) => + site.kind === 'call' && + site.staticGated !== true && + isPositionStaticGated(site.atRange.startLine, site.atRange.startCol, ranges) + ? ({ ...site, staticGated: true } satisfies ReferenceSite) + : site, + ); + parsedFiles[index] = Object.freeze({ ...parsed, referenceSites: Object.freeze(next) }); + } +} + +function collectImportAliases( + tree: ZigTree, + fromFile: string, + knownPaths: ReadonlySet, + resolutionConfig?: ZigBuildZonConfig | null, +): ZigImportAliasMap { + const candidates = new Map(); + const declarationCounts = new Map(); + for (const decl of tree.rootNode.descendantsOfType('variable_declaration')) { + const names = decl.namedChildren.filter((node) => node.type === 'identifier'); + const binding = names[0]?.text; + if (binding === undefined) continue; + declarationCounts.set(binding, (declarationCounts.get(binding) ?? 0) + 1); + const builtin = decl.namedChildren.find( + (node) => node.type === 'builtin_function' && node.text.startsWith('@import('), + ); + const raw = builtin?.descendantsOfType('string').at(0)?.text; + if (raw === undefined) continue; + const specifier = raw.replace(/^['"]|['"]$/g, ''); + const target = resolveZigImportInternal(fromFile, specifier, knownPaths, resolutionConfig); + if (target !== null) candidates.set(binding, target); + } + + const aliases = new Map(); + for (const [binding, target] of candidates) { + // Alias lookup below is name-based rather than position-aware. If a name + // is redeclared in another lexical scope, fail open instead of applying + // either module's constants to every use of that spelling. + if (declarationCounts.get(binding) === 1) aliases.set(binding, target); + } + return aliases; +} diff --git a/gitnexus/src/core/ingestion/scope-resolution/contract/scope-resolver.ts b/gitnexus/src/core/ingestion/scope-resolution/contract/scope-resolver.ts index 9ac2423f5..e364b0227 100644 --- a/gitnexus/src/core/ingestion/scope-resolution/contract/scope-resolver.ts +++ b/gitnexus/src/core/ingestion/scope-resolution/contract/scope-resolver.ts @@ -697,6 +697,21 @@ export interface ScopeResolver { ctx: { readonly fileContents: ReadonlyMap }, ) => void; + /** + * Optional workspace-wide enrichment of extracted reference sites. Runs + * after all files have been extracted and before reference finalization. + * Use this when a per-file capture needs conservative facts from an + * imported sibling (for example a compile-time branch constant). + */ + readonly populateWorkspaceReferences?: ( + parsedFiles: ParsedFile[], + ctx: { + readonly fileContents: ReadonlyMap; + readonly treeCache?: { get(filePath: string): unknown }; + readonly resolutionConfig?: unknown; + }, + ) => void; + /** * Recognize a `super(...)`-style receiver text. Python returns * `/^super\s*\(/.test(t)`. Java returns `t === 'super'`. C++ may diff --git a/gitnexus/src/core/ingestion/scope-resolution/pipeline/phase.ts b/gitnexus/src/core/ingestion/scope-resolution/pipeline/phase.ts index 8218a9656..bb9854b3e 100644 --- a/gitnexus/src/core/ingestion/scope-resolution/pipeline/phase.ts +++ b/gitnexus/src/core/ingestion/scope-resolution/pipeline/phase.ts @@ -145,6 +145,7 @@ export function selectScopeSourcePathsToRead( ): string[] { const hasPostExtractHooks = provider.populateWorkspaceOwners !== undefined || + provider.populateWorkspaceReferences !== undefined || provider.populateNamespaceSiblings !== undefined || provider.populateRangeBindings !== undefined || provider.emitPostResolutionEdges !== undefined; diff --git a/gitnexus/src/core/ingestion/scope-resolution/pipeline/run.ts b/gitnexus/src/core/ingestion/scope-resolution/pipeline/run.ts index df823fecb..aa30acf91 100644 --- a/gitnexus/src/core/ingestion/scope-resolution/pipeline/run.ts +++ b/gitnexus/src/core/ingestion/scope-resolution/pipeline/run.ts @@ -639,6 +639,11 @@ export function runScopeResolution( `lang=${provider.language} parsedFiles=${parsedFiles.length} preExtractedHits=${preExtractedHits} skipped=${filesSkipped}`, ); provider.populateWorkspaceOwners?.(parsedFiles, { fileContents: getFileContents() }); + provider.populateWorkspaceReferences?.(parsedFiles, { + fileContents: getFileContents(), + treeCache, + resolutionConfig: input.resolutionConfig, + }); // A callable-flow-only provider has no reason to build the whole-graph // lookup or finalize ordinary references when none of its files emitted a diff --git a/gitnexus/test/fixtures/lang-resolution/zig-static-gating/src/main.zig b/gitnexus/test/fixtures/lang-resolution/zig-static-gating/src/main.zig index cb9686e96..6e397a013 100644 --- a/gitnexus/test/fixtures/lang-resolution/zig-static-gating/src/main.zig +++ b/gitnexus/test/fixtures/lang-resolution/zig-static-gating/src/main.zig @@ -9,6 +9,9 @@ // Cross-file alias — should resolve `cfg.FOO`, `cfg.BAR` against cfg.zig. const cfg = @import("./cfg.zig"); +const cfg_no_ext = @import("./cfg"); +const wrapped_cfg = wrap(@import("./cfg.zig")); +const shadowed_cfg = @import("./cfg.zig"); pub const UPGRADERS_ENABLED: bool = false; pub const DEBUG: bool = true; @@ -30,6 +33,7 @@ pub const CYCLE_B = CYCLE_A; pub const ALIAS_TO_VAR = IS_RUNTIME_FLAG_FALSE; pub fn run() void { + const local_cfg = @import("./cfg.zig"); // Live: not under any if-gate. live_unconditional(); @@ -162,6 +166,18 @@ pub fn run() void { if (cfg.NOT_A_BOOL != 0) { live_cross_file_not_bool(); } + if (cfg_no_ext.FOO) { + gated_extensionless_cross_file_foo(); + } + // Function-local imports use the same workspace constants. + if (local_cfg.FOO) { + gated_local_cross_file_foo(); + } + // An import nested inside another initializer does not bind the variable + // directly to that module, so its members must remain unknown/fail-open. + if (wrapped_cfg.FOO) { + live_wrapped_cross_file_foo(); + } // Bare literal gate: no constant table involved, but it must still be gated. if (false) { @@ -205,10 +221,37 @@ pub fn run() void { _ = e3; } +pub fn run_shadowed_alias() void { + const shadowed_cfg = @import("./other.zig"); + if (shadowed_cfg.FOO) { + live_shadowed_cross_file_foo(); + } +} + fn live_unconditional() void { _ = 1; } +fn wrap(value: anytype) @TypeOf(value) { + return value; +} + +fn gated_local_cross_file_foo() void { + _ = 1; +} + +fn gated_extensionless_cross_file_foo() void { + _ = 1; +} + +fn live_wrapped_cross_file_foo() void { + _ = 1; +} + +fn live_shadowed_cross_file_foo() void { + _ = 1; +} + fn gated_simple() void { _ = 1; } diff --git a/gitnexus/test/fixtures/lang-resolution/zig-static-gating/src/other.zig b/gitnexus/test/fixtures/lang-resolution/zig-static-gating/src/other.zig new file mode 100644 index 000000000..4d7502d54 --- /dev/null +++ b/gitnexus/test/fixtures/lang-resolution/zig-static-gating/src/other.zig @@ -0,0 +1 @@ +pub const FOO: bool = true; diff --git a/gitnexus/test/integration/resolvers/zig-static-gating.test.ts b/gitnexus/test/integration/resolvers/zig-static-gating.test.ts index 92526c157..33ee827fb 100644 --- a/gitnexus/test/integration/resolvers/zig-static-gating.test.ts +++ b/gitnexus/test/integration/resolvers/zig-static-gating.test.ts @@ -7,8 +7,11 @@ * such branches keep `staticGated` falsy. */ import { describe, it, expect, beforeAll } from 'vitest'; +import fs from 'node:fs'; import path from 'path'; import { FIXTURES, getRelationships, runPipelineFromRepo, type PipelineResult } from './helpers.js'; +import { populateZigWorkspaceStaticGating } from '../../../src/core/ingestion/languages/zig/workspace-static-gating.js'; +import type { ParsedFile } from 'gitnexus-shared'; describe('Zig static-gated edges', () => { let result: PipelineResult; @@ -174,12 +177,9 @@ describe('Zig static-gated edges', () => { expect(isGated('gated_chain_tail')).toBe(true); }); - // Cross-file positive cases: the gating module resolves `alias.NAME` through - // `lookupBoolsForPath`, but the scope-capture emitter runs per file in the - // parse worker with only `{ path, content }` in hand — no sibling sources — - // so v1 stamps file-local constants only. Re-enable once the emitter can - // see imported files (see PR description, "Cross-file constants"). - it.skip('tags `if (cfg.FOO)` cross-file when FOO is false in cfg.zig (tracked: #3162)', () => { + // Cross-file cases are enriched after per-file extraction, once sibling + // source facts are available but before reference finalization. + it('tags `if (cfg.FOO)` cross-file when FOO is false in cfg.zig', () => { expect(isGated('gated_cross_file_foo')).toBe(true); }); @@ -187,7 +187,7 @@ describe('Zig static-gated edges', () => { expect(isGated('live_cross_file_bar')).toBe(false); }); - it.skip('tags the ELSE branch of `if (cfg.BAR)` when BAR is true (tracked: #3162)', () => { + it('tags the ELSE branch of `if (cfg.BAR)` when BAR is true', () => { expect(isGated('gated_cross_file_else')).toBe(true); }); @@ -198,4 +198,57 @@ describe('Zig static-gated edges', () => { it('does NOT tag `cfg.NOT_A_BOOL != 0` (imported decl is not a bool literal)', () => { expect(isGated('live_cross_file_not_bool')).toBe(false); }); + + it('resolves a relative cross-file import with an omitted .zig extension', () => { + expect(isGated('gated_extensionless_cross_file_foo')).toBe(true); + }); + + it('tags a cross-file bool accessed through a function-local import alias', () => { + expect(isGated('gated_local_cross_file_foo')).toBe(true); + }); + + it('does NOT treat a nested @import as the declaration direct module alias', () => { + expect(isGated('live_wrapped_cross_file_foo')).toBe(false); + }); + + it('fails open when an import alias is shadowed in another lexical scope', () => { + expect(isGated('live_shadowed_cross_file_foo')).toBe(false); + }); + + it('replaces a frozen parsed file instead of mutating it', () => { + const site = Object.freeze({ + kind: 'call', + atRange: { startLine: 153, startCol: 8, endLine: 153, endCol: 30 }, + staticGated: false, + }); + const original = Object.freeze({ + filePath: 'src/main.zig', + referenceSites: Object.freeze([site]), + }) as unknown as ParsedFile; + const parsedFiles = [ + original, + Object.freeze({ + filePath: 'src/cfg.zig', + referenceSites: Object.freeze([]), + }) as unknown as ParsedFile, + ]; + + expect(() => + populateZigWorkspaceStaticGating(parsedFiles, { + fileContents: new Map([ + [ + 'src/main.zig', + fs.readFileSync(path.join(FIXTURES, 'zig-static-gating', 'src', 'main.zig'), 'utf8'), + ], + [ + 'src/cfg.zig', + fs.readFileSync(path.join(FIXTURES, 'zig-static-gating', 'src', 'cfg.zig'), 'utf8'), + ], + ]), + }), + ).not.toThrow(); + expect(parsedFiles[0]).not.toBe(original); + expect(Object.isFrozen(parsedFiles[0])).toBe(true); + expect(parsedFiles[0]?.referenceSites[0]?.staticGated).toBe(true); + }); }); From f48bf812566ed06300eb969df9fed5a64d0f66d3 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Gerg=C5=91=20Magyar?= Date: Mon, 7 Sep 2026 13:53:31 +0100 Subject: [PATCH 04/17] perf(parse): tighten the dispatch-round memory bound and unclamp the worker-pool override (#3200) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * docs(parse): record why dispatchGroups is a required interface member Review finding #10 argued dispatchGroups should be optional to match `getQuarantinedPaths?` / `getStats?`. Those are compatibility accommodation for WorkerPool shapes that predate them, not a convention for new members; optional here would force a `?.` plus an unreachable fallback at the single production call site. Documenting the decision so the next reader does not re-litigate it from the neighbouring optional markers. Co-Authored-By: Claude Opus 5 (1M context) (cherry picked from commit addaab647377f3c4553f752fa3ca1388bcb9ca81) * refactor(parse): simplify round accounting and dispatch setup Simplification pass over the dispatch-rounds change. Behavior preserved: identical graph on a full analyze (51,286 nodes / 163,092 edges). - Drop `roundMissBytes`. `roundBufferedBytes` counts the same bytes plus the cache hits, so it is always the greater of the two and the first disjunct of the close condition could never fire on its own. One counter, one reset, one check. - Measure round bytes with `Buffer.byteLength(content, 'utf8')` instead of `String.length`. UTF-16 code units undercount non-ASCII source by up to 3x, so the cap meant to bound main-thread retention was letting a CJK-heavy repo hold well past its nominal budget. Matches `estimateItemBytes` in the pool. - Reset the durable ParsedFile directories for a round's chunks concurrently. Each targets its own chunk-hash directory, and running them serially put N round trips of fs work on the critical path the round exists to shorten. The try/catch stays inside the mapped callback, so one failure still degrades that chunk alone. - Skip the quarantine filter entirely when nothing is quarantined, which is every run without a worker death. It was an identity copy of every group. - `dispatchChunkParseRound` takes `DispatchGroup<...>` rather than re-declaring that shape inline; the type was already imported and used in its body. Co-Authored-By: Claude Opus 5 (1M context) (cherry picked from commit 527d5b6e0ca8ae7bbc6a414c5ac7e27fd85e9995) * refactor(parse): count round misses with the same idiom startRound uses `drainRound` hand-rolled a reduce to count 'miss' entries while `startRound`, one function above, filters the same predicate over the same union. Same integer, one idiom. Co-Authored-By: Claude Opus 5 (1M context) (cherry picked from commit 7eaa193b0cb5fa515844f36ae1401d6fb2fed7b8) * fix(parse): honor GITNEXUS_WORKER_POOL_SIZE above the auto sizing cap The auto pool size is bounded by source bytes so a tiny repo does not spawn a full idle pool. That bound was also clamping the operator's env override, because the env value is read inside `resolveAutoPoolSize()` and the result went through `Math.min(..., workProportionalCap)`. `DEFAULT_POOL_SIZE_CAP`'s own comment offers `GITNEXUS_WORKER_POOL_SIZE` and `--workers ` as equivalent escape hatches for operators on bigger machines. They were not. Measured on a 30MB corpus, where the byte-derived cap is 16: --workers 24 -> pool: 24/24 active GITNEXUS_WORKER_POOL_SIZE=24 -> pool: 16/16 active (silently ignored) Both are deliberate operator input, so both now bypass the work-proportional cap, which goes back to bounding only the auto default. After the fix, on the same corpus, with identical graph output (51,286 nodes / 163,092 edges): GITNEXUS_WORKER_POOL_SIZE=24 -> pool: 24/24 active GITNEXUS_WORKER_POOL_SIZE=4 -> pool: 4/4 active unset -> pool: 16/16 active Verified by hand against the pool's own throughput log; not covered by an automated regression test, since the pool size is only observable through that log line and not through the progress stream a test can read. Co-Authored-By: Claude Opus 5 (1M context) (cherry picked from commit 17ed08608c878079b2927da25cfd39c1608a02a2) * fix(parse): bound the durable-reset fan-out and pin the pool-size override Review follow-ups on #3200. The round's durable ParsedFile directory resets went out as one unbounded `Promise.all` — one recursive rm + mkdir per miss chunk, all at once. A round can hold hundreds of small packs, and those resets compete for descriptors with the chunk prefetch this loop already has in flight. `readFileContents` degrades a losing read SILENTLY by documented contract, so a dropped file would vanish from the chunk, from the graph, and from the chunk hash — shipping a narrowed index with exit 0. Now routed through `mapConcurrent` at the same width the file reads use, which keeps the pipelining win and caps in-flight descriptors. An operator's pool size is now also bounded by the number of files there are to parse, so `GITNEXUS_WORKER_POOL_SIZE=100000` on a five-file repo cannot become the literal thread count. This applies to `--workers` and the env var alike, so the parity the previous commit established is intact. It does NOT shrink an incremental re-analyze: `totalParseable` counts every parseable file in the scan, not the changed ones. Adds the regression test a reviewer asked for. The existing coverage (`worker-pool-resilience` calling `resolveAutoPoolSize` directly, `analyze-worker-pool-size` mocking `runFullAnalysis`) never reaches `runChunkedParseAndResolve`'s `effectivePoolSize`, so both stayed green through a revert of the fix. The new test drives the real parse phase with a worker double that writes a per-`threadId` marker, and counts them: verified it fails on the reverted line with `expected [ 'worker-1' ] to have a length of 3 but got 1`, and passes on HEAD. Also corrects the `GITNEXUS_PARSE_ROUND_BYTES` docstring, which still described the cache-miss counter deleted two commits ago. Co-Authored-By: Claude Opus 5 (1M context) * fix(parse): skip caching a chunk with a stale durable generation; warn on over-subscription Closes the two findings left open by the review of #3200. When `prepareDurableParsedFileChunk` fails, the previous generation's shards are still on disk, so a later warm hit would union them with the new ones. The chunk is now recorded and its parse-cache write skipped -- the same posture `finalizeWorkerChunk` already takes for a quarantined chunk, and for the same reason: do not cache what we cannot vouch for. The next run re-dispatches into a directory it can actually clear. Bounding the reset fan-out removed the correlated trigger; this closes the individual case. Pool size over-subscription now warns rather than caps. Silently capping is precisely what the override exists to prevent, so an operator's number is still honored -- but an exported GITNEXUS_WORKER_POOL_SIZE applies to every analyze in a long-lived caller (watch auto-sync, the MCP server), including small incremental ones, and that is easy to set once and forget. The warning names the host's usable core count, so it is a hardware fact rather than an invented threshold. `resolveHostParallelism` is extracted from `resolveAutoPoolSize` rather than re-deriving the cgroup-aware fallback at the new call site. Tests: the stale-generation skip is pinned by a new case asserting nothing is written under any key; verified it fails without the guard with `expected 1 to be +0`. 60 unit and 49 integration tests pass across the affected suites. Co-Authored-By: Claude Opus 5 (1M context) * test(parse): guard dispatch-round cadence with a bench, not a wall-clock budget Round boundaries are deliberately invisible to graph output — batching that changed output would be a bug — so nothing in the repo could see the #3196 win regress. It would have come back as a silent ~1.5x on every cold analyze. Two earlier attempts to pin it as a unit test failed for that exact reason: one scraped a logger line the progress stream does not carry, the other asserted graph content that is identical either way. Extracts the round-close fold into `createRoundBudget`, so the decision is a shared unit the bench measures rather than a copy that drifts. The parse loop is streaming and cannot know chunk sizes up front, so an accumulator is the honest shape — not a planner. Four deterministic arms, one ratio, no millisecond gate: - layout_fingerprint — pack membership. Every cache key derives from it, so drift needs a SCHEMA_BUMP, never a lone re-baseline. - packs / single_file_packs — the FLOOR. `rounds` only asserts something while the corpus over-splits (774 packs where the byte budget needs 5). This is bench/import-target's lesson, where four heap arms read 0 B and passed every ceiling: a ceiling says "not too big", nothing said "still measuring". - rounds — the regression signal, both directions. - cjk_rounds vs ascii_rounds — pins UTF-8 byte accounting. The two corpora share a UTF-16 length and differ only in encoded size, so String.length collapses them to equal. This is the arm no unit test could be. - pack_scaling_ratio — (t_4n/t_n)/4, min-of-15. A ratio because wall-clock is runner-speed-dependent and this repo has the scar: callable-value-flow's ms gate failed twice at 2.07 and 1.975 against 1.9 with correct code, on a sub-11ms measurement. Every arm verified to fail before being recorded: close-every-chunk reads 774 rounds, disabling the close reads 1, reverting roundFileBytes to String.length takes cjk_rounds 8 -> 3, and shrinking the corpus trips the shape floor. Co-Authored-By: Claude Opus 5 (1M context) * docs(bench): record the analyze phase breakdown and the rejected optimizations Where analyze time actually goes, measured while landing #3194/#3196/#3200, plus the two optimizations that looked compelling and were measured away. The headline is that the parse work is done: a one-file-edit re-analyze is 36.5s, of which parse is 2.8s (8%). scopeResolution is 40% and the unlogged graph emit + FTS rebuild is 49% — neither is incremental, and the ~18s sits outside the phase runner so every phase log is blind to it. Also records the trap that invalidated an earlier measurement: a non-git corpus never records a schema fingerprint, so every run is a forced rebuild and any "warm" number taken that way is fiction. Rejected, with numbers: more workers (16/20/24 land inside run-to-run spread) and bundling the worker entry (~250ms on a normal filesystem; the 8.6s that motivated it was a 9p-mount artifact). Co-Authored-By: Claude Opus 5 (1M context) --------- Co-authored-by: Gergo Magyar Co-authored-by: Claude Opus 5 (1M context) --- .github/workflows/ci-tests.yml | 12 + gitnexus/bench/analyze-phase-breakdown.md | 126 ++++++++ .../parse-dispatch-rounds/baselines.json | 31 ++ .../bench/parse-dispatch-rounds/measure.mjs | 279 ++++++++++++++++++ .../src/core/ingestion/parsing-processor.ts | 5 +- .../ingestion/pipeline-phases/parse-impl.ts | 138 ++++++--- .../pipeline-phases/parse-round-budget.ts | 63 ++++ .../src/core/ingestion/workers/worker-pool.ts | 51 +++- ...mpl-warm-cache-parsedfile-coverage.test.ts | 18 ++ .../unit/parse-impl-worker-lazy-cache.test.ts | 94 ++++++ 10 files changed, 765 insertions(+), 52 deletions(-) create mode 100644 gitnexus/bench/analyze-phase-breakdown.md create mode 100644 gitnexus/bench/parse-dispatch-rounds/baselines.json create mode 100644 gitnexus/bench/parse-dispatch-rounds/measure.mjs create mode 100644 gitnexus/src/core/ingestion/pipeline-phases/parse-round-budget.ts diff --git a/.github/workflows/ci-tests.yml b/.github/workflows/ci-tests.yml index a0d4abcad..7b469948c 100644 --- a/.github/workflows/ci-tests.yml +++ b/.github/workflows/ci-tests.yml @@ -587,6 +587,18 @@ jobs: run: node --import tsx bench/finalize-reexport/measure.mjs --check working-directory: gitnexus + - name: Parse dispatch-round cadence guards (#3194, #3196) + if: ${{ !cancelled() }} + # Build-free: asserts parse-cache pack membership is unchanged + # (fingerprint — every cache key derives from it), that a fixed corpus + # still batches into a fixed number of dispatch rounds, and that the + # round budget counts UTF-8 bytes rather than UTF-16 code units. Round + # boundaries are deliberately invisible to graph output, so no test can + # see these regress. Rationale and history: see the header of + # bench/parse-dispatch-rounds/measure.mjs. + run: node --import tsx bench/parse-dispatch-rounds/measure.mjs --check + working-directory: gitnexus + - name: C++ qualified-namespace resolution guards (#2788) if: ${{ !cancelled() }} # Build-free: asserts resolveCppQualifiedNamespaceMember resolves an diff --git a/gitnexus/bench/analyze-phase-breakdown.md b/gitnexus/bench/analyze-phase-breakdown.md new file mode 100644 index 000000000..2a3a0a221 --- /dev/null +++ b/gitnexus/bench/analyze-phase-breakdown.md @@ -0,0 +1,126 @@ +# Analyze phase breakdown — where the time actually goes + +Measured 2026-09-06/07 while landing #3194, #3196 and #3200. Records what the +`analyze` pipeline costs per phase, and — as importantly — the optimizations +that were measured and **rejected**, so the next person does not re-derive them. + +Unlike `parse-throughput.md` (a synthetic-fixture scaffold), these numbers come +from a real repo corpus. They are not a CI gate; the gate for the parse +dispatch path is `bench/parse-dispatch-rounds/`. + +--- + +## Method, and one trap that invalidates everything + +**The corpus must be a git repository.** A non-git checkout cannot record a +schema fingerprint, so the tool forces a full rebuild on every run and prints: + +> `non-git repositories never record a schema fingerprint, so this run rebuilds regardless` + +A "warm" run measured that way is a forced cold rebuild wearing a warm label. A +37.7s figure was recorded that way during this work and was meaningless. On a +real git repo an unchanged re-analyze short-circuits to `Already up to date`. + +Corpus: this repository, `git archive` of HEAD into a scratch dir, then +`git init && git add -A && git commit`. 5350 paths, 2234 parseable, ~30MB. +16 workers, `dist` on a local overlay filesystem (see "Filesystem" below). +Phase numbers come from the `✓ Phase: ()` lines under +`NODE_ENV=development`. + +--- + +## Cold analyze + +| | before #3194 | after #3196 | +| ---------------- | -----------: | ----------: | +| total | 110.3s | **63.6s** | +| parse | 74.0s | 36.0s | +| scopeResolution | 17.0s | 16.0s | +| all other phases | 1.2s | 1.2s | +| dispatches | 221 | 15 | + +Graph output identical throughout: 51,286 nodes / 163,092 edges / 2106 clusters +/ 759 flows. + +## Re-analyze after a one-file edit — the developer loop + +One file changed out of 5350, on a git repo. **36.5s total.** + +| phase | ms | share | +| ----------------------------------- | ------: | ----: | +| scopeResolution | 14,717 | 40% | +| unlogged — graph emit + FTS rebuild | ~18,000 | 49% | +| parse | 2,832 | 8% | +| all other phases | ~1,200 | 3% | + +The parse cache works: it replays 2231 of 2232 chunks. **Everything the parse +work optimized is that 8%.** The other 92% is not incremental at all — the run +banner says so directly: _"Rebuilt the graph and FTS while reusing cached +parser output."_ + +Note the ~18s sits **outside the phase runner**, so every `✓ Phase` line is +blind to it. `phasesSum` and the log span agree exactly; the gap is wall-clock +before the first phase and after the last. + +## scopeResolution is memory-traffic bound, not algorithmic + +`--cpu-prof` of a one-file-edit run, top main-thread self time: + +``` +4294ms (garbage collector) +1552ms v8.deserialize + 756ms crypto update + 728ms runScopeResolution + 620ms scope-resolution/pipeline/reconcile-ownership + 380ms v8-sidecar walk + 323ms internString +``` + +then a tail of passes at 200–750ms (`emitReceiverBoundCalls`, +`emitCallableValueFlow`, `buildGraphNodeLookup`, `resolveReferenceSites`). + +No dominant hot function, nothing quadratic. The cost is rehydrating every +file's `ParsedFile` from the durable `.v8` shards into main-thread memory and +re-running every pass over them. + +**So the win is not caching resolution output** — that stores more of exactly +what is already the memory problem, and #2649 (large-repo OOM) is the standing +constraint. The win is skipping rehydration and re-resolution for files whose +inputs provably did not change, as a streaming/bounded design. + +--- + +## Measured and rejected + +Recorded because each cost real time to establish and each looks attractive +from the armchair. + +**More workers buys nothing.** Isolated-harness wall time at 16 / 20 / 24 +workers: 44.1s / 44.8s / 43.3s — a 1.5s spread against a 3.7s within-size +spread. A full-analyze sweep appeared to show 20 beating 16 by 6.3s; it was +noise, and the sweep was invalid anyway because `GITNEXUS_WORKER_POOL_SIZE` was +silently clamped at the time (fixed in #3200). `DEFAULT_POOL_SIZE_CAP = 16` +stands. + +**Bundling the worker entry buys ~250ms.** Worker boot profiled at 12.2s per +worker, of which `getPackageScopeConfig` 5.86s + `internalModuleStat` 3.14s + +`lstat`/`open` ~2.2s — ESM module resolution, not native grammars (all 11 +`tree-sitter-*` imports together are 33ms) and not V8 compile (53ms; +`NODE_COMPILE_CACHE` gives zero gain). An esbuild bundle takes 16-worker boot +8.6s → 0.37s. + +**That was a filesystem artifact.** On a normal overlay filesystem the same +boot is 515ms stock vs 263ms bundled — 0.2% of a 110s analyze. The 8.6s only +reproduces with the repo on a 9p mount (WSL2 `D:\`). Dropped. + +Caveat carried by every number here: `dist` on a 9p mount costs ~3.5s of a 73s +run (73.0s vs 69.6s on overlay). Measure on a local filesystem. + +--- + +## Open + +The ~18s of graph emit + FTS rebuild is **unprofiled**. It is the largest +single share of the edit loop and nothing is known about it beyond the banner. +Profile it before proposing anything — two optimizations in this document +looked compelling until measured. diff --git a/gitnexus/bench/parse-dispatch-rounds/baselines.json b/gitnexus/bench/parse-dispatch-rounds/baselines.json new file mode 100644 index 000000000..febd79edd --- /dev/null +++ b/gitnexus/bench/parse-dispatch-rounds/baselines.json @@ -0,0 +1,31 @@ +{ + "_what": "Baselines for bench/parse-dispatch-rounds/measure.mjs --check. Guards parse-cache pack layout and dispatch-round cadence. Neither is visible in graph output — batching that changed output would be a bug — so nothing else in the repo can see these regress. Four of the five arms are deterministic; only pack_scaling_ratio is a timing signal.", + + "_triage": "READ THIS BEFORE RE-RUNNING. layout_fingerprint, packs, single_file_packs, rounds, cjk_rounds and ascii_rounds are DETERMINISTIC: a re-run never changes them, and none may be re-baselined to make CI green. pack_scaling_ratio is the only timing arm; runner contention dominates it, so re-run on an idle machine before investigating and read the reported `reps` first. If exactly one arm fails and it is that one, suspect the machine.", + + "layout_fingerprint": "cc875fd264498964b463aef55cec0166d57468a092303e94f1ed7f09fe141a44", + "_layout_fingerprint_note": "sha256 over the sorted pack membership — which files share a pack, and their order within it. Every parse-cache key derives from a pack's file set, so a change here invalidates every cached chunk for every user. This is a CORRECTNESS gate: drift needs a SCHEMA_BUMP in src/storage/parse-cache.ts alongside a new fingerprint, never a lone re-baseline.", + + "packs": 774, + "single_file_packs": 251, + "_shape_note": "THE FLOOR. Without these two, every arm below is a ceiling over nothing. `rounds` only asserts something while the corpus OVER-SPLITS — 774 packs where the byte budget alone needs 5, 251 of them holding a single file. Shrink the corpus until packing stops over-splitting and rounds still reads 5 and still passes, asserting a property the corpus no longer has. bench/import-target learned this the hard way: four heap arms read 0 B and passed every ceiling, because a ceiling says 'not too big' and nothing said 'still measuring something'.", + + "rounds": 5, + "_rounds_note": "Exact round count for the fixed corpus at the 2MB budget, folded through the production accumulator in pipeline-phases/parse-round-budget.ts. HIGHER (toward packs=774) means dispatch went back to one barrier per cache pack — the #3196 regression, measured at ~1.5x on a cold analyze with no visible symptom. LOWER (toward 1) means the close condition stopped firing, so an open round retains the whole repo until the tail drain (#2649 heap shape). Both directions verified to fail this arm before it was recorded: forcing close-every-chunk reads 774, disabling the close reads 1.", + + "cjk_rounds": 8, + "ascii_rounds": 3, + "_encoding_note": "The round budget bounds what the MAIN THREAD HOLDS, so it must count UTF-8 bytes. String.length returns UTF-16 code units: a CJK character is one unit but three UTF-8 bytes, so reverting the unit would let a CJK-heavy repo hold ~3x its nominal budget before draining. The two corpora are constructed to have IDENTICAL UTF-16 length and differ only in encoded size, so under String.length both close 3 rounds and the arm collapses. Verified: reverting roundFileBytes to content.length takes cjk_rounds 8 -> 3. This is the arm that pins the change no unit test could — round cadence changes no graph output, so a test asserting output passes either way.", + + "pack_scaling_budget": 1.6, + "_pack_scaling_note": "(t_4n / t_n) / 4 for packParseCacheChunks; ~1.0 is linear. A RATIO rather than a millisecond ceiling, deliberately: wall-clock is runner-speed-dependent, and this repo has already been bitten by a fixed ms budget — bench/callable-value-flow's widening_overhead gate failed twice on a shared runner at 2.07 and 1.975 against a 1.9 budget while the code was correct, on a sub-11ms measurement. A ratio divides the machine out. Measured over 5 runs on a NON-idle box: 0.940, 0.946, 0.977, 0.998, 1.085 (peak-to-peak 1.154). Budget is 1.6, i.e. 1.47x the measured maximum — this file's siblings use ~1.5x on ratios. It catches packParseCacheChunks going superlinear (it sorts within each bucket, so a global sort or a nested scan lands here) and is not tight enough to police drift. min-of-15 estimator, matching bench/import-target's finding that N=5 tripped its own budget ~1 run in 20 while N=15 held every language inside a 1.13-1.26x swing.", + + "_measured": { + "pack_scaling_ratio": 1.085, + "pack_scaling_ratio_samples": [0.94, 0.946, 0.977, 0.998, 1.085], + "small_ms": 1.91, + "large_ms_4x": 7.46, + "reps": 15 + }, + "_measured_note": "Maxima over 5 runs on a box that was NOT idle, so the ratio spread is an upper bound on its real noise. small_ms/large_ms_4x are recorded for context only — nothing gates on them, because an absolute millisecond is exactly the gate this file avoids." +} diff --git a/gitnexus/bench/parse-dispatch-rounds/measure.mjs b/gitnexus/bench/parse-dispatch-rounds/measure.mjs new file mode 100644 index 000000000..6543b561e --- /dev/null +++ b/gitnexus/bench/parse-dispatch-rounds/measure.mjs @@ -0,0 +1,279 @@ +/** + * Build-free bench for parse-cache pack layout and dispatch-round cadence. + * + * WHY THIS EXISTS. `WorkerPool.dispatch` is a barrier: it resolves only once + * every job it created has committed. Packs are keyed `(language, + * sha256(path) % 128)`, so the byte budget almost never binds and most packs + * land far below the pool size — this repo produced 1285 packs where the + * budget alone needed 16, and 549 held a single file. Dispatching one pack at + * a time therefore left most workers idle for every round-trip. #3194 fixed + * fan-out WITHIN a pack; #3196 batched packs into bounded rounds and took a + * cold analyze from 110.3s to 70.5s (221 dispatches -> 15). + * + * Nothing guarded that. Round boundaries are deliberately invisible to the + * graph — batching that changed output would be a bug — so no test can see the + * regression, and it would come back as a silent 1.5x on every cold analyze. + * Two earlier attempts to pin this as a unit test failed for exactly that + * reason: one scraped a logger line the progress stream does not carry, the + * other asserted graph content that is identical either way. + * + * FOUR ARMS, and only the last is a timing arm: + * + * - `rounds` — EXACT. The regression signal. A fixed corpus and budget must + * produce a fixed number of rounds. Per-pack dispatch coming back sends this + * to `packs`; a broken close condition sends it to 1. + * + * - `cjk_rounds` vs `ascii_rounds` — EXACT. The round budget bounds what the + * MAIN THREAD HOLDS, so it must count UTF-8 bytes. `String.length` returns + * UTF-16 code units: a CJK character is one unit but three UTF-8 bytes, so + * reverting the unit would let a CJK-heavy repo hold ~3x its nominal budget + * before draining — the #2649 heap-failure shape. The two corpora are + * identical in UTF-16 length and differ only in encoded size, so under + * `String.length` they would close the SAME number of rounds. Only a UTF-8 + * count separates them. + * + * - `packs` / `single_file_packs` — EXACT, and they are the FLOOR. `rounds` + * only asserts something while the corpus over-splits (774 packs where the + * byte budget alone needs 5). Shrink the corpus past that and `rounds` still + * reads 5 and still passes, gating a property the corpus no longer has. + * bench/import-target learned this when four heap arms read 0 B and passed. + * + * - `pack_scaling_ratio` — the only timing arm, and a RATIO not a millisecond + * ceiling. (t_4n/t_n)/4 divides the machine out; ~1.0 is linear. A fixed ms + * budget on a shared runner is a coin flip, and this repo has the scar: + * bench/callable-value-flow's gate failed twice at 2.07 and 1.975 against a + * 1.9 budget with correct code, on a sub-11ms measurement. Catches + * `packParseCacheChunks` going superlinear; not tight enough to police drift. + * + * Usage: + * node --import tsx bench/parse-dispatch-rounds/measure.mjs # report + * node --import tsx bench/parse-dispatch-rounds/measure.mjs --check # CI gate + */ +import { performance } from 'node:perf_hooks'; +import { createHash } from 'node:crypto'; +import { readFileSync } from 'node:fs'; +import { packParseCacheChunks } from '../../src/storage/parse-cache.js'; +import { createRoundBudget } from '../../src/core/ingestion/pipeline-phases/parse-round-budget.js'; + +const baselines = JSON.parse( + new URL('./baselines.json', import.meta.url).pathname + ? readFileSync(new URL('./baselines.json', import.meta.url), 'utf8') + : '{}', +); + +/** Matches DEFAULT_CHUNK_BYTE_BUDGET / the round budget's default in parse-impl.ts. */ +const BUDGET = 2 * 1024 * 1024; + +/** + * A repo shaped like a real one: many languages, so `(language, bucket)` + * packing over-splits well past what the byte budget alone would need. Sizes + * are deliberately uneven — a uniform corpus hides an off-by-one in the fold. + */ +function mixedCorpus(scale = 1) { + const langs = [ + ['ts', 900], + ['py', 400], + ['java', 260], + ['go', 240], + ['rb', 120], + ['rs', 180], + ['php', 90], + ['cs', 140], + ]; + const files = []; + for (const [ext, count] of langs) { + for (let i = 0; i < count * scale; i++) { + files.push({ + path: `src/${ext}/mod${i}.${ext}`, + // 400B - 8KB, varying by index so packs are not uniform. + size: 400 + ((i * 977) % 7700), + language: ext, + }); + } + } + return files; +} + +/** Feed chunks through the real accumulator and count the rounds it closes. */ +function roundsFor(chunks, contentsByPath, budgetBytes) { + const budget = createRoundBudget(budgetBytes); + let rounds = 0; + for (const chunk of chunks) { + if (budget.addChunk(chunk.map((p) => contentsByPath.get(p)))) rounds++; + } + // The tail drain closes a partially-filled round when anything is left. + if (budget.bufferedBytes > 0) rounds++; + return rounds; +} + +/** + * Two corpora with IDENTICAL UTF-16 length and different UTF-8 size. Under + * `String.length` both close the same number of rounds; under UTF-8 the CJK + * one closes strictly more. + */ +function encodingCorpora() { + // 1 UTF-16 unit / 3 UTF-8 bytes each, vs 1 unit / 1 byte each. + const cjkLine = '説'.repeat(240); + const asciiLine = 'a'.repeat(240); + const count = 260; + const files = Array.from({ length: count }, (_, i) => ({ + path: `src/enc/mod${i}.ts`, + size: 240, + language: 'ts', + })); + const chunks = packParseCacheChunks(files, BUDGET); + const cjk = new Map(files.map((f) => [f.path, cjkLine])); + const ascii = new Map(files.map((f) => [f.path, asciiLine])); + // A budget small enough that both corpora close several rounds. + const encBudget = 24 * 1024; + return { + utf16Length: cjkLine.length === asciiLine.length, + cjkRounds: roundsFor(chunks, cjk, encBudget), + asciiRounds: roundsFor(chunks, ascii, encBudget), + }; +} + +/** + * Min-of-N estimator. `fastest` rather than a mean because the minimum is the + * least contaminated sample on a shared runner — the same choice, and the same + * reason, as bench/import-target's `fastest()`. + */ +function fastest(fn, reps) { + fn(); // warm + let best = Infinity; + for (let r = 0; r < reps; r++) { + const t0 = performance.now(); + fn(); + best = Math.min(best, performance.now() - t0); + } + return best; +} + +const REPS = 15; + +const corpus = mixedCorpus(); +const corpus4x = mixedCorpus(4); + +const packs = packParseCacheChunks(corpus, BUDGET); +// A RATIO, not a millisecond ceiling. Wall-clock is runner-speed-dependent and +// a fixed ms budget on a shared runner is a coin flip — this file's sibling +// benches record exactly that failure. (t_4n / t_n) / 4 divides the machine +// out: ~1.0 is linear, and packParseCacheChunks going superlinear (it sorts +// within each bucket) shows up here regardless of how fast the box is. +const smallMs = fastest(() => packParseCacheChunks(corpus, BUDGET), REPS); +const largeMs = fastest(() => packParseCacheChunks(corpus4x, BUDGET), REPS); +const packScaling = largeMs / smallMs / 4; + +const contents = new Map(corpus.map((f) => [f.path, 'x'.repeat(f.size)])); +const rounds = roundsFor(packs, contents, BUDGET); + +const enc = encodingCorpora(); + +/** + * Order-independent hash of the pack layout: which files share a pack, and in + * what order within it. Catches a packing change that leaves the counts intact + * but moves files between packs — which would silently change every cache key. + */ +const layoutFingerprint = createHash('sha256') + .update( + packs + .map((chunk) => chunk.join(',')) + .sort() + .join('\n'), + ) + .digest('hex'); + +const singleFilePacks = packs.filter((c) => c.length === 1).length; +const totalBytes = corpus.reduce((sum, f) => sum + f.size, 0); +const budgetFloor = Math.ceil(totalBytes / BUDGET); + +console.log(`files : ${corpus.length}`); +console.log( + `packs : ${packs.length} (expect ${baselines.packs}; byte budget alone needs ${budgetFloor})`, +); +console.log(`rounds : ${rounds} (expect ${baselines.rounds})`); +console.log(`single_file_packs : ${singleFilePacks} (expect ${baselines.single_file_packs})`); +console.log(`cjk_rounds : ${enc.cjkRounds} (UTF-8 bytes)`); +console.log(`ascii_rounds : ${enc.asciiRounds} (same UTF-16 length)`); +console.log(`layout_fingerprint : ${layoutFingerprint.slice(0, 16)}`); +console.log( + `pack_scaling_ratio : ${packScaling.toFixed(3)} (budget <= ${baselines.pack_scaling_budget}; ~1.0 is linear)`, +); +console.log( + `reps : ${REPS} small ${smallMs.toFixed(2)}ms / 4x ${largeMs.toFixed(2)}ms`, +); + +if (process.argv.includes('--check')) { + let failed = false; + + if (layoutFingerprint !== baselines.layout_fingerprint) { + failed = true; + console.error( + `\nFAIL layout_fingerprint: ${layoutFingerprint}\n` + + ` expected ${baselines.layout_fingerprint}\n` + + ` Pack membership moved. Every parse-cache key is derived from a pack's\n` + + ` file set, so this invalidates every cached chunk for every user. If the\n` + + ` change is intended, it needs a SCHEMA_BUMP in src/storage/parse-cache.ts\n` + + ` alongside a new fingerprint here — never re-baseline it alone.`, + ); + } + + if (rounds !== baselines.rounds) { + failed = true; + console.error( + `\nFAIL rounds: ${rounds}, expected exactly ${baselines.rounds}.\n` + + ` HIGHER (toward packs=${packs.length}) means rounds stopped batching and\n` + + ` dispatch went back to one barrier per cache pack — the #3196 regression,\n` + + ` worth ~1.5x on a cold analyze with no visible symptom.\n` + + ` LOWER (toward 1) means the close condition stopped firing, so an open\n` + + ` round retains the whole repo until the tail drain (#2649 heap shape).\n` + + ` Check createRoundBudget in pipeline-phases/parse-round-budget.ts.`, + ); + } + + if (!enc.utf16Length) { + failed = true; + console.error( + `\nFAIL encoding arm is broken: its two corpora no longer share a UTF-16 length.`, + ); + } else if (enc.cjkRounds <= enc.asciiRounds) { + failed = true; + console.error( + `\nFAIL cjk_rounds ${enc.cjkRounds} <= ascii_rounds ${enc.asciiRounds}.\n` + + ` These corpora have identical UTF-16 length and differ only in encoded\n` + + ` size, so equal round counts mean the budget is counting String.length\n` + + ` again instead of Buffer.byteLength. A CJK-heavy repo would then hold\n` + + ` ~3x its nominal budget on the main thread before draining.\n` + + ` See roundFileBytes in pipeline-phases/parse-round-budget.ts.`, + ); + } + + // SHAPE — the floor. Without it every arm below is a ceiling over nothing: + // shrink the corpus until packing stops over-splitting and `rounds` still + // reads 5 and still passes, asserting a property the corpus no longer has. + if (packs.length !== baselines.packs || singleFilePacks !== baselines.single_file_packs) { + failed = true; + console.error( + `\nFAIL shape: packs ${packs.length} (expected ${baselines.packs}), ` + + `single_file_packs ${singleFilePacks} (expected ${baselines.single_file_packs}).\n` + + ` The corpus must stay one that OVER-SPLITS — ${packs.length} packs where the\n` + + ` byte budget alone needs ${budgetFloor}. That gap is the entire reason rounds\n` + + ` exist, so if it closes, the rounds arm below asserts nothing.`, + ); + } + + if (packScaling > baselines.pack_scaling_budget) { + failed = true; + console.error( + `\nFAIL pack_scaling_ratio: ${packScaling.toFixed(3)} exceeds ` + + `${baselines.pack_scaling_budget} (~1.0 is linear).\n` + + ` packParseCacheChunks grew superlinearly in file count — it sorts within\n` + + ` each bucket, so a global sort or a nested scan lands here.\n` + + ` This is the ONLY timing arm in this file: re-run on an idle machine\n` + + ` before investigating, and check \`reps\` in the report first.`, + ); + } + + if (failed) process.exit(1); + console.log('\nOK — within budget.'); +} diff --git a/gitnexus/src/core/ingestion/parsing-processor.ts b/gitnexus/src/core/ingestion/parsing-processor.ts index 9a1f39233..4798c1f9f 100644 --- a/gitnexus/src/core/ingestion/parsing-processor.ts +++ b/gitnexus/src/core/ingestion/parsing-processor.ts @@ -253,10 +253,7 @@ export const dispatchChunkParse = async ( * owns it. */ export const dispatchChunkParseRound = async ( - groups: ReadonlyArray<{ - items: { path: string; content: string }[]; - chunkHash?: string; - }>, + groups: ReadonlyArray>, workerPool: WorkerPool, onFileProgress?: FileProgressCallback, ): Promise => { diff --git a/gitnexus/src/core/ingestion/pipeline-phases/parse-impl.ts b/gitnexus/src/core/ingestion/pipeline-phases/parse-impl.ts index b4ec9a5b8..18542637d 100644 --- a/gitnexus/src/core/ingestion/pipeline-phases/parse-impl.ts +++ b/gitnexus/src/core/ingestion/pipeline-phases/parse-impl.ts @@ -69,6 +69,8 @@ import { createWorkerPool, workerPoolDisabledByEnv, resolveAutoPoolSize, + envWorkerPoolSize, + resolveHostParallelism, WorkerPoolInitializationError, WorkerPoolDisabledError, } from '../workers/worker-pool.js'; @@ -115,6 +117,8 @@ import { import { isDebugHeapEnabled, logHeapProbe } from '../utils/heap-probe.js'; import { logger } from '../../logger.js'; +import { mapConcurrent } from '../../../lib/utils.js'; +import { createRoundBudget } from './parse-round-budget.js'; // ── Constants ────────────────────────────────────────────────────────────── /** @@ -213,11 +217,18 @@ const CHUNK_BYTES_PER_WORKER = DEFAULT_CHUNK_BYTE_BUDGET; */ const TARGET_JOBS_PER_WORKER = 3; +/** + * Concurrent durable ParsedFile directory resets per round. Matches the file + * reader's `READ_CONCURRENCY`, because both compete for the same descriptors. + */ +const DURABLE_RESET_CONCURRENCY = 32; + /** Floor for a derived sub-batch so jobs don't shrink to per-file IPC churn. */ const MIN_SUB_BATCH_BYTES = 256 * 1024; /** - * Source bytes of cache-missing chunks allowed in flight in one pool round. + * Source bytes an open round may HOLD — cache hits and misses alike — before + * it is dispatched and drained. * * A `dispatch` is a barrier, so one round-trip per cache pack leaves most slots * idle: packs are keyed by `(language, hash(path) % 128)` and routinely land far @@ -579,12 +590,42 @@ export async function runChunkedParseAndResolve( // cores-based auto size is capped by source bytes / CHUNK_BYTES_PER_WORKER // so a tiny repo does not spawn a full idle pool. Cache pack membership // is independent of this number (#3088). - const explicitPoolSize = options?.workerPoolSize; + // `--workers ` and `GITNEXUS_WORKER_POOL_SIZE` are both deliberate + // operator input, so both bypass the work-proportional cap below. Only the + // env path used to be clamped by it, which made the documented escape hatch + // silently do nothing: on a 30MB repo the cap resolves to 16, so an operator + // asking for 24 still got 16 with no warning, while `--workers 24` got 24. + const explicitPoolSize = options?.workerPoolSize ?? envWorkerPoolSize(); + // Cores-based auto size, bounded by source bytes so a tiny repo does not + // spawn a full idle pool. const workProportionalCap = Math.max(1, Math.ceil(totalBytes / CHUNK_BYTES_PER_WORKER)); + // An operator's number is honored, but never exceeds the number of files + // there are to parse — `GITNEXUS_WORKER_POOL_SIZE=100000` on a five-file repo + // should not become the literal thread count. This bounds `--workers` and the + // env var identically, keeping the parity above intact. Note it does NOT + // shrink an incremental re-analyze: `totalParseable` counts every parseable + // file in the scan, not the changed ones, so a warm run of a large repo still + // spawns the full requested pool. const effectivePoolSize = explicitPoolSize && explicitPoolSize > 0 - ? explicitPoolSize + ? Math.min(explicitPoolSize, Math.max(1, totalParseable)) : Math.min(resolveAutoPoolSize(), workProportionalCap); + // Deliberate over-subscription is the operator's call, so this warns rather + // than caps — silently capping is what the override exists to stop. But an + // exported `GITNEXUS_WORKER_POOL_SIZE` applies to EVERY analyze in a + // long-lived caller (watch auto-sync, the MCP server), including small + // incremental ones, and that is easy to set once and forget. + if (explicitPoolSize && explicitPoolSize > 0) { + const hostParallelism = resolveHostParallelism(); + if (effectivePoolSize > hostParallelism) { + logger.warn( + { requested: explicitPoolSize, spawning: effectivePoolSize, hostParallelism }, + `Worker pool size ${effectivePoolSize} exceeds this host's ${hostParallelism} usable core(s); ` + + `parsing is CPU-bound, so the extra workers add memory pressure without throughput. ` + + `This applies to every analyze while the override is set.`, + ); + } + } // Cache packs: stable (language, hash(path) mod 128) buckets, then the // per-call byte budget inside each bucket (#3088). Pool size is used only // for worker count and sub-batch fan-out, not membership. @@ -861,20 +902,31 @@ export async function runChunkedParseAndResolve( readonly chunkStartMs: number | null; }; + /** + * Chunk hashes whose durable ParsedFile directory could not be reset. The + * old generation's shards are still on disk, so a warm hit would union + * stale shards with the new ones. Treated exactly like a quarantined chunk: + * skip the parse-cache write so the next run re-dispatches into a clean + * directory rather than trusting a generation we could not clear. + */ + const durablePrepareFailures = new Set(); + const roundByteBudget = resolveParseRoundByteBudget(options); let roundEntries: RoundEntry[] = []; - let roundMissBytes = 0; /** * Bytes an open round is HOLDING, counting hits as well as misses. * - * `roundMissBytes` alone bounds only what the workers are asked to do, so a - * warm run — where nothing misses — would never reach the close condition - * and would buffer every chunk's cached output until the tail drain. That - * is the #2649 heap failure on a large repo. Closing on either cap keeps a - * hits-only run draining at the same cadence as a cold one; `startRound` - * already supports a round with no misses. + * Counting only the cache-MISSING bytes would bound just what the workers + * are asked to do, so a warm run — where nothing misses — would never reach + * the close condition and would buffer every chunk's cached output until + * the tail drain. That is the #2649 heap failure on a large repo. Counting + * both keeps a hits-only run draining at the same cadence as a cold one; + * `startRound` already supports a round with no misses. + * + * Measured in UTF-8 bytes, matching `estimateItemBytes` in the worker pool, + * so the cap means the same thing here as it does for a job's payload. */ - let roundBufferedBytes = 0; + const roundBudget = createRoundBudget(roundByteBudget); /** * Files QUEUED into rounds so far. `filesParsedSoFar` only advances when a * round drains, so it is the right number for the throughput log but would @@ -998,7 +1050,14 @@ export async function runChunkedParseAndResolve( if (parseCache && p.chunkHash && rawResults.length > 0) { const quarantineSet = new Set(workerPool?.getQuarantinedPaths?.() ?? []); const chunkHadQuarantine = p.chunkFiles.some((f) => quarantineSet.has(f.path)); - if (chunkHadQuarantine) { + const durableGenerationStale = durablePrepareFailures.has(p.chunkHash); + if (durableGenerationStale) { + logger.warn( + { chunkHash: p.chunkHash.slice(0, 8) }, + 'parse-cache SKIP: durable generation for this chunk could not be reset, ' + + 'so its shards may be stale; next run will re-dispatch it', + ); + } else if (chunkHadQuarantine) { if (isDev) { const quarantinedInChunk = p.chunkFiles.filter((f) => quarantineSet.has(f.path)).length; logger.info( @@ -1033,21 +1092,39 @@ export async function runChunkedParseAndResolve( if (misses.length === 0) { return { entries, results: Promise.resolve([]) }; } - for (const miss of misses) { - if (durableParsedFileDir !== undefined && miss.chunkHash !== null) { + // Each chunk resets its own directory, so these are independent and run + // concurrently: serially they would sit on the critical path this round + // exists to shorten, with the pool idle and the previous round's merge + // waiting, once per miss. + // + // BOUNDED, though. A round can hold hundreds of small packs, and each + // reset is a recursive rm + mkdir. Firing all of them at once competes + // for descriptors with the chunk prefetch this loop already has in + // flight, and `readFileContents` degrades a losing read SILENTLY by + // contract — a dropped file would vanish from the chunk, from the graph, + // and from the chunk hash, shipping a narrowed index with exit 0. Same + // helper and width the file reads use. + await mapConcurrent( + misses, + async (miss) => { + if (durableParsedFileDir === undefined || miss.chunkHash === null) return; try { await prepareDurableParsedFileChunk(durableParsedFileDir, miss.chunkHash); } catch (err) { // The durable store is an optimization — degrade like the restore // path does instead of failing the analyze. Workers recreate the // directory on write, so at worst the old generation lingers. + // Caught per chunk so one failure cannot abort the others. + durablePrepareFailures.add(miss.chunkHash); logger.warn( { err, chunkHash: miss.chunkHash.slice(0, 8) }, - 'parsedfile-cache: could not reset durable chunk generation; continuing', + 'parsedfile-cache: could not reset durable chunk generation; ' + + 'continuing without caching this chunk', ); } - } - } + }, + { concurrency: DURABLE_RESET_CONCURRENCY }, + ); const roundFiles = misses.reduce((sum, miss) => sum + miss.chunkFiles.length, 0); const firstIdx = misses[0].chunkIdx; const lastIdx = misses[misses.length - 1].chunkIdx; @@ -1105,10 +1182,7 @@ export async function runChunkedParseAndResolve( missResults: ParseWorkerResult[][]; }): Promise => { const missResults = round.missResults; - const missCount = round.entries.reduce( - (sum, entry) => sum + (entry.kind === 'miss' ? 1 : 0), - 0, - ); + const missCount = round.entries.filter((entry) => entry.kind === 'miss').length; // `dispatchGroups` returns one array per input group. If that contract // ever breaks, every later entry in this round would silently merge the // wrong chunk's results and skip its cache write, with a clean exit. @@ -1151,8 +1225,7 @@ export async function runChunkedParseAndResolve( const closeRound = async (): Promise => { const started = await startRound(roundEntries); roundEntries = []; - roundMissBytes = 0; - roundBufferedBytes = 0; + roundBudget.reset(); const previous = pendingRound; pendingRound = null; if (previous) { @@ -1275,6 +1348,8 @@ export async function runChunkedParseAndResolve( durableExpectedPaths !== undefined && (await durableChunkHasShards(parsedFileStorePath, chunkHash, durableExpectedPaths)); + // Set by whichever branch queues this chunk; drives the close below. + let roundIsFull = false; if (cachedRaw && cachedRaw.length > 0 && (durableHit || parsedFileStorePath === undefined)) { // Cache hit: replay cached worker output. Finalize any parked worker // chunk FIRST so deferred aggregation stays in chunk order, then merge @@ -1312,7 +1387,7 @@ export async function runChunkedParseAndResolve( chunkStartMs, cachedRaw, }); - for (const file of chunkFiles) roundBufferedBytes += file.content.length; + roundIsFull = roundBudget.addChunk(chunkFiles.map((file) => file.content)); queuedFilesSoFar += chunkFiles.length; } else { // Cache miss: queue for the round's single dispatch; the raw results @@ -1320,19 +1395,14 @@ export async function runChunkedParseAndResolve( chunkCacheMisses++; reparsedFileCount += chunkFiles.length; roundEntries.push({ kind: 'miss', chunkIdx, chunkHash, chunkFiles, chunkStartMs }); - for (const file of chunkFiles) { - roundMissBytes += file.content.length; - roundBufferedBytes += file.content.length; - } + roundIsFull = roundBudget.addChunk(chunkFiles.map((file) => file.content)); queuedFilesSoFar += chunkFiles.length; } - // Close on EITHER cap. `roundMissBytes` sizes the worker round; - // `roundBufferedBytes` bounds what the main thread is holding, which is - // the only cap a warm run can ever reach. - if (roundMissBytes >= roundByteBudget || roundBufferedBytes >= roundByteBudget) { - await closeRound(); - } + // One cap, on what the main thread is holding. That bounds the worker + // round too, since a round's dispatched bytes are a subset of its + // buffered bytes. + if (roundIsFull) await closeRound(); // (Per-chunk aggregation + parse-cache write + throughput log now run in // `applyChunkResults` / `finalizeWorkerChunk` — see the merge-pipelining diff --git a/gitnexus/src/core/ingestion/pipeline-phases/parse-round-budget.ts b/gitnexus/src/core/ingestion/pipeline-phases/parse-round-budget.ts new file mode 100644 index 000000000..9ab85b715 --- /dev/null +++ b/gitnexus/src/core/ingestion/pipeline-phases/parse-round-budget.ts @@ -0,0 +1,63 @@ +/** + * The fold that decides when an open dispatch round closes. + * + * Extracted so the decision is a shared, inspectable unit rather than four + * loose statements inside `runChunkedParseAndResolve`. The parse loop is + * STREAMING — it reads chunk contents lazily, so it cannot know every chunk's + * size up front and cannot "plan" rounds ahead. That makes an accumulator, not + * a planner, the honest shape: feed it each chunk as it is queued and it tells + * you whether the round is now full. + * + * Being a real unit is what makes round cadence observable. Round boundaries + * are otherwise invisible from outside the parse phase: they change no graph + * output (that is the point of batching) and surface only in a log line, which + * is why `bench/parse-dispatch-rounds` measures this directly rather than + * inferring cadence from a full analyze. + */ + +/** Bytes a file contributes to the open round's retained total. */ +export const roundFileBytes = (content: string): number => Buffer.byteLength(content, 'utf8'); + +export interface RoundBudget { + /** + * Add one queued chunk's files. Returns true when the round is now full and + * the caller should close it. Closing resets the accumulator. + */ + addChunk(contents: readonly string[]): boolean; + /** Bytes currently held by the open round. */ + readonly bufferedBytes: number; + /** Reset without closing — used when the caller closes for another reason. */ + reset(): void; +} + +/** + * `budgetBytes` bounds what the main thread HOLDS, counting cache hits as well + * as misses. Counting only cache-missing bytes would bound just the work sent + * to workers, so a warm run — where nothing misses — would never reach the + * close condition and would buffer every chunk's cached output until the tail + * drain. That is the #2649 heap failure on a large repo. + * + * Measured in UTF-8 bytes, matching `estimateItemBytes` in the worker pool. + * `String.length` would return UTF-16 code units, undercounting non-ASCII + * source by up to 3x and letting a CJK-heavy repo hold well past its nominal + * budget before draining. + */ +export const createRoundBudget = (budgetBytes: number): RoundBudget => { + let bufferedBytes = 0; + return { + addChunk(contents) { + for (const content of contents) bufferedBytes += roundFileBytes(content); + if (bufferedBytes >= budgetBytes) { + bufferedBytes = 0; + return true; + } + return false; + }, + get bufferedBytes() { + return bufferedBytes; + }, + reset() { + bufferedBytes = 0; + }, + }; +}; diff --git a/gitnexus/src/core/ingestion/workers/worker-pool.ts b/gitnexus/src/core/ingestion/workers/worker-pool.ts index e5186c6e1..b972d7326 100644 --- a/gitnexus/src/core/ingestion/workers/worker-pool.ts +++ b/gitnexus/src/core/ingestion/workers/worker-pool.ts @@ -125,6 +125,12 @@ export interface WorkerPool { * * Returns one result array per input group, in input order. A group whose * items were all quarantined yields an empty array. + * + * Required, not optional. `getQuarantinedPaths?` and `getStats?` below are + * marked optional as a compatibility accommodation for `WorkerPool` shapes + * that predate them — not as a convention for new members. Making this one + * optional would force a `?.` plus a fallback branch at its only production + * call site, and that branch could never run. */ dispatchGroups( groups: readonly DispatchGroup[], @@ -469,7 +475,9 @@ const DEFAULT_WORKER_READY_TIMEOUT_MS = 5_000; * extraction / structured-clone overhead, and the marginal worker adds * memory pressure (tree-sitter state + sub-batch buffer) without much * throughput gain. Operators on bigger machines override via - * `GITNEXUS_WORKER_POOL_SIZE` or `--workers `. + * `GITNEXUS_WORKER_POOL_SIZE` or `--workers `; both are deliberate + * operator input and bypass the work-proportional sizing in `parse-impl`, + * which only bounds the AUTO default. */ const DEFAULT_POOL_SIZE_CAP = 16; @@ -647,7 +655,7 @@ export function resolveWorkerPoolOptions( * GITNEXUS_WORKER_POOL_SIZE=`) is an accident, not a request for zero workers; * only a literal `0` disables the pool. */ -function envWorkerPoolSize(): number | undefined { +export function envWorkerPoolSize(): number | undefined { const raw = process.env.GITNEXUS_WORKER_POOL_SIZE; if (raw === undefined || raw.trim() === '') return undefined; return nonNegativeInteger(raw); @@ -691,9 +699,19 @@ export function resolveAutoPoolSize(): number { // pool cap exists to prevent. Falls back to os.cpus().length on // older Node versions. Mirrors `capabilities.ts:85` // (`defaultEmbeddingThreads`). - const cores = - typeof os.availableParallelism === 'function' ? os.availableParallelism() : os.cpus().length; - return Math.min(DEFAULT_POOL_SIZE_CAP, Math.max(1, cores - 1)); + return Math.min(DEFAULT_POOL_SIZE_CAP, Math.max(1, resolveHostParallelism() - 1)); +} + +/** + * Usable parallelism for this process. Prefers `os.availableParallelism` so + * cgroup CPU limits are honored, falling back to `os.cpus().length` on older + * Node. Exported so callers that size work against the host (rather than + * against the pool default) do not re-derive the fallback. + */ +export function resolveHostParallelism(): number { + return typeof os.availableParallelism === 'function' + ? os.availableParallelism() + : os.cpus().length; } /** @@ -1383,15 +1401,20 @@ export const createWorkerPool = ( // Layer 3: filter out quarantined paths so a known-bad file never reaches // a worker again this pool lifetime. The caller queries // `getQuarantinedPaths` after dispatch to route filtered items. - const dispatchableGroups = groups.map((group) => { - const items: TInput[] = []; - for (const item of group.items) { - const path = itemPath(item); - if (path !== undefined && quarantine.has(path)) continue; - items.push(item); - } - return { items, chunkHash: group.chunkHash }; - }); + // Quarantine is empty on every run that has not had a worker die, so the + // filter below would be an identity copy of every group's items. Skip it. + const dispatchableGroups = + quarantine.size === 0 + ? groups + : groups.map((group) => { + const items: TInput[] = []; + for (const item of group.items) { + const path = itemPath(item); + if (path !== undefined && quarantine.has(path)) continue; + items.push(item); + } + return { items, chunkHash: group.chunkHash }; + }); const dispatchableCount = dispatchableGroups.reduce( (sum, group) => sum + group.items.length, 0, diff --git a/gitnexus/test/unit/parse-impl-warm-cache-parsedfile-coverage.test.ts b/gitnexus/test/unit/parse-impl-warm-cache-parsedfile-coverage.test.ts index 410d0c048..c01baaca2 100644 --- a/gitnexus/test/unit/parse-impl-warm-cache-parsedfile-coverage.test.ts +++ b/gitnexus/test/unit/parse-impl-warm-cache-parsedfile-coverage.test.ts @@ -383,6 +383,24 @@ describe('parse-impl warm-cache ParsedFile coverage (#2038)', () => { } }); + it('does not cache a chunk whose durable generation could not be reset', async () => { + // The reset failed, so the previous generation's shards are still on disk. + // Caching this chunk would let a future warm hit union those stale shards + // with the new ones. Same posture as a quarantined chunk: leave it uncached + // so the next run re-dispatches into a directory it can actually clear. + const f = writeFile('src/stale-generation.ts', 'export function stale() { return 1; }\n'); + const cache = newCache(); + prepareOverride.impl = () => Promise.reject(new Error('EACCES: simulated cache failure')); + try { + await expect(run(cache, [f])).resolves.toBeDefined(); + } finally { + prepareOverride.impl = undefined; + } + + // Nothing was written under any key -- neither on disk nor in memory. + expect(cache.onDiskKeys.size + cache.entries.size).toBe(0); + }); + it('retains worker ParsedFiles when the main-thread run-store write fails', async () => { const f = writeFile( 'src/persist-fallback.ts', diff --git a/gitnexus/test/unit/parse-impl-worker-lazy-cache.test.ts b/gitnexus/test/unit/parse-impl-worker-lazy-cache.test.ts index d35629a75..216170a2b 100644 --- a/gitnexus/test/unit/parse-impl-worker-lazy-cache.test.ts +++ b/gitnexus/test/unit/parse-impl-worker-lazy-cache.test.ts @@ -103,6 +103,50 @@ parentPort.on('message', (msg) => { ); }; +/** + * Like `writeResultWorker`, but each spawned instance writes its OWN marker + * keyed by `threadId`. The shared single-marker workers above can only prove + * "at least one worker started"; counting files in `markerDir` gives the actual + * pool size the parse phase asked `createWorkerPool` for, which is the only + * thing that distinguishes a clamped pool from an honored override. + */ +const writeSpawnCountingWorker = (workerPath: string, markerDir: string): void => { + fs.writeFileSync( + workerPath, + ` +const fs = require('node:fs'); +const path = require('node:path'); +const { parentPort, threadId } = require('node:worker_threads'); +fs.mkdirSync(${JSON.stringify(markerDir)}, { recursive: true }); +fs.writeFileSync(path.join(${JSON.stringify(markerDir)}, 'worker-' + threadId), 'spawned'); +parentPort.postMessage({ type: 'ready' }); +const accumulated = { + nodes: [], relationships: [], symbols: [], imports: [], calls: [], assignments: [], heritage: [], + routes: [], fetchCalls: [], fetchWrapperDefs: [], decoratorRoutes: [], routerIncludes: [], routerImports: [], toolDefs: [], ormQueries: [], constructorBindings: [], + fileScopeBindings: [], parsedFiles: [], skippedLanguages: {}, fileCount: 0, +}; +parentPort.on('message', (msg) => { + if (msg && msg.type === 'sub-batch') { + for (const file of msg.files) { + const filePath = file.path; + const name = filePath.split('/').pop().replace(/\.ts$/, ''); + accumulated.nodes.push({ + id: 'Function:' + filePath + ':' + name, + label: 'Function', + properties: { name, filePath, startLine: 1, endLine: 1, language: 'typescript' }, + }); + accumulated.fileCount++; + } + parentPort.postMessage({ type: 'progress', filesProcessed: accumulated.fileCount }); + parentPort.postMessage({ type: 'sub-batch-done' }); + return; + } + if (msg && msg.type === 'flush') parentPort.postMessage({ type: 'result', data: accumulated }); +}); +`, + ); +}; + const writeExitBeforeReadyWorker = (workerPath: string): void => { fs.writeFileSync(workerPath, `process.exit(1);\n`); }; @@ -240,6 +284,56 @@ describe('parse-impl worker pool lazy startup', () => { expect(Array.from(graph.nodes.values()).some((n) => n.properties.name === 'fatal')).toBe(false); }); + it('honors GITNEXUS_WORKER_POOL_SIZE above the work-proportional cap', async () => { + // The auto pool size is bounded by source bytes so a tiny repo does not + // spawn a full idle pool. That bound must apply to the AUTO default only — + // it used to clamp the operator's env override too, so an operator asking + // for more workers silently got the byte-derived number while `--workers` + // was honored. + // + // This asserts the pool the PARSE PHASE actually builds, not the resolver in + // isolation: `resolveAutoPoolSize()` already honored the env var before the + // fix, so a test at that level stays green through a revert. + const saved = process.env.GITNEXUS_WORKER_POOL_SIZE; + process.env.GITNEXUS_WORKER_POOL_SIZE = '3'; + try { + // Four tiny files: total bytes are far under one CHUNK_BYTES_PER_WORKER so + // the work-proportional cap is 1, while the parseable count stays above the + // requested 3 (the pool never exceeds the number of files to parse). + const rels = ['src/a.ts', 'src/b.ts', 'src/c.ts', 'src/d.ts']; + const scanned = rels.map((rel) => { + const full = path.join(repoDir, rel); + fs.mkdirSync(path.dirname(full), { recursive: true }); + fs.writeFileSync(full, `export function ${path.basename(rel, '.ts')}() { return 1; }\n`); + return { path: rel, size: fs.statSync(full).size }; + }); + + const markerDir = path.join(tempDir, 'pool-size-markers'); + const workerPath = path.join(tempDir, 'pool-size-worker.js'); + writeSpawnCountingWorker(workerPath, markerDir); + + const result = await runChunkedParseAndResolve( + createKnowledgeGraph(), + scanned, + rels, + rels.length, + repoDir, + Date.now(), + () => {}, + // No `workerPoolSize`: the env var is the only override in play. + { workerUrlForTest: pathToFileURL(workerPath) }, + ); + + expect(result.usedWorkerPool).toBe(true); + // 3, not the byte-derived 1. Exactly this assertion fails on the clamped + // parent commit, which is what makes it a regression test for the fix. + expect(fs.readdirSync(markerDir)).toHaveLength(3); + } finally { + if (saved === undefined) delete process.env.GITNEXUS_WORKER_POOL_SIZE; + else process.env.GITNEXUS_WORKER_POOL_SIZE = saved; + } + }); + it('throws when GITNEXUS_WORKER_POOL_SIZE=0 and no --workers flag (sequential parsing removed)', async () => { const saved = process.env.GITNEXUS_WORKER_POOL_SIZE; process.env.GITNEXUS_WORKER_POOL_SIZE = '0'; From eba42d2994d69215ba16b6f09774d0a80e73c9c6 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Gerg=C5=91=20Magyar?= Date: Mon, 7 Sep 2026 17:35:31 +0100 Subject: [PATCH 05/17] docs(bench): correct the edit-loop numbers and add the FTS per-index breakdown (#3208) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The numbers this file shipped with were full rebuilds labelled as incremental. Rebuilding or re-copying `dist` between runs changes the analyzer runner identity, and the tool then forces a full rebuild — every measurement taken that way is a full run wearing an incremental label. That is the third "silently fall back to full work" guard in this pipeline, after the non-git corpus and the escalation gate, so the method section now says to read the banner on every run. Corrected, measured on a leaf file with a stable runner identity so the incremental path is genuinely taken: 31.7s, not 36.5s. The graph write is a 3,980-node subgraph rather than the full 51,288, and parse is ~9% of the loop. Adds the per-index FTS breakdown, which is the actionable finding: 10.2s across 20 indexes, of which File.file_fts alone is 3.5s because File nodes carry file content and that index re-tokenizes ~30MB of source to reflect one changed row. Also records that a two-importer leaf edit rebuilt 20 indexes while another run rebuilt 8 — `touchedFts` narrowing is at least inconsistent, and it has a withdrawal path when the index catalog cannot be read. That needs pinning before anyone optimizes against the narrowed set. Co-authored-by: Gergo Magyar Co-authored-by: Claude Opus 5 (1M context) --- gitnexus/bench/analyze-phase-breakdown.md | 100 ++++++++++++++++++---- 1 file changed, 81 insertions(+), 19 deletions(-) diff --git a/gitnexus/bench/analyze-phase-breakdown.md b/gitnexus/bench/analyze-phase-breakdown.md index 2a3a0a221..1751c3a3e 100644 --- a/gitnexus/bench/analyze-phase-breakdown.md +++ b/gitnexus/bench/analyze-phase-breakdown.md @@ -21,11 +21,28 @@ A "warm" run measured that way is a forced cold rebuild wearing a warm label. A 37.7s figure was recorded that way during this work and was meaningless. On a real git repo an unchanged re-analyze short-circuits to `Already up to date`. +**And the analyzer binary must not change between runs.** Rebuilding or +re-copying `dist` changes the runner identity, and the tool then prints: + +> `analyzer runner identity changed ...; forcing a full rebuild so the index +provenance matches the analyzer` + +Every run after a rebuild is a full rebuild. To measure the incremental path, +run once to stamp the identity, then edit and run again **without touching +`dist`**. An earlier revision of this document reported full-rebuild numbers as +if they were incremental for exactly this reason; they are corrected below. + +That is three separate "silently fall back to full work" guards — non-git +corpus, runner identity, and the escalation gate. Read the banner on every run +before trusting a number. + Corpus: this repository, `git archive` of HEAD into a scratch dir, then `git init && git add -A && git commit`. 5350 paths, 2234 parseable, ~30MB. 16 workers, `dist` on a local overlay filesystem (see "Filesystem" below). Phase numbers come from the `✓ Phase: ()` lines under -`NODE_ENV=development`. +`NODE_ENV=development`; the post-pipeline tail has no such lines and was +measured by injecting timestamp marks into a disposable copy of the built +`dist`. --- @@ -44,23 +61,65 @@ Graph output identical throughout: 51,286 nodes / 163,092 edges / 2106 clusters ## Re-analyze after a one-file edit — the developer loop -One file changed out of 5350, on a git repo. **36.5s total.** +Leaf file (`cli/update-notice.ts`, 2 importers), stable runner identity, so the +incremental path is genuinely taken. **31.7s total.** -| phase | ms | share | -| ----------------------------------- | ------: | ----: | -| scopeResolution | 14,717 | 40% | -| unlogged — graph emit + FTS rebuild | ~18,000 | 49% | -| parse | 2,832 | 8% | -| all other phases | ~1,200 | 3% | +| step | ms | share | +| ----------------------------------------------------- | -------: | ------: | +| scopeResolution | ~13900 | 44% | +| **FTS index rebuild** (`buildSearchIndexesOrDegrade`) | **7525** | **24%** | +| graph write (`loadGraphToLbug`, subgraph) | 3566 | 11% | +| parse | ~2800 | 9% | +| `import('./platform/capabilities.js')` | 845 | 3% | +| everything else | ~3000 | 9% | -The parse cache works: it replays 2231 of 2232 chunks. **Everything the parse -work optimized is that 8%.** The other 92% is not incremental at all — the run -banner says so directly: _"Rebuilt the graph and FTS while reusing cached -parser output."_ +The incremental machinery works: the graph write was a 3,980-node subgraph, not +the full 51,288. **Everything the #3194/#3196 parse work optimized is ~9% of +this.** -Note the ~18s sits **outside the phase runner**, so every `✓ Phase` line is -blind to it. `phasesSum` and the log span agree exactly; the gap is wall-clock -before the first phase and after the last. +A FULL-rebuild run of the same repo is 36-39s, with the graph write at ~6.3s and +FTS at ~9.7s. Do not quote those as edit-loop numbers. + +## FTS index rebuild — the largest non-resolution cost + +Per-index, measured on one leaf-file edit: + +``` + 3531ms File.file_fts <- 35% of FTS on its own + 1558ms Function.function_fts + 535ms Method 491ms Const 466ms Property + 427ms Interface 363ms Class 280ms TypeAlias + 247ms Struct 242ms Variable 234ms Enum 222ms Trait ... + ------ +10203ms across 20 indexes +``` + +Two findings. + +**`File.file_fts` dominates** because File nodes carry file content, so that one +index re-tokenizes ~30MB of source. Changing one file re-tokenizes all of it. + +**20 indexes were rebuilt for a two-importer leaf edit.** `touchedFts` is meant +to narrow this, and a separate run rebuilt only 8 — so the narrowing is at least +inconsistent. It has a withdrawal path: `missingSearchFTSIndexTables` returns +`undefined` when the index catalog cannot be read (`fts-indexes.ts:285`), and +`tables: undefined` means rebuild everything. Pin that before optimizing, or the +optimization targets a set that is not actually being narrowed. + +The code states the underlying constraint plainly: _"createSearchFTSIndexes +re-tokenizes every stored row on every run."_ Indexes must be dropped for DML +(#2589, #2841), and the only way back is a whole-table rebuild. + +Ranked next steps: + +1. **Take the FTS rebuild off the critical path.** Nothing in `analyze` reads + the index — only later searches do. Rebuild lazily on first search or in the + background after the swap. Removes the cost rather than shrinking it, and the + degraded-search state is already modelled (`ftsSkipReason`, "degrade rather + than throw"). +2. **Do not rebuild `File.file_fts` when no file content changed.** +3. **Fix or delete the narrowing** — a gate that silently withdraws looks like + protection and is not. ## scopeResolution is memory-traffic bound, not algorithmic @@ -120,7 +179,10 @@ run (73.0s vs 69.6s on overlay). Measure on a local filesystem. ## Open -The ~18s of graph emit + FTS rebuild is **unprofiled**. It is the largest -single share of the edit loop and nothing is known about it beyond the banner. -Profile it before proposing anything — two optimizations in this document -looked compelling until measured. +The post-pipeline tail is now fully accounted (99.7%): FTS rebuild, the graph +write, and a 0.8s dynamic `capabilities.js` import between them. What remains +open is the FTS work above, and `scopeResolution` — still the single largest +step, memory-traffic bound, and untouched. + +Every optimization in this document that looked compelling from the armchair +died under measurement. Measure first, and check the banner. From 0d1aed942f0e8b5d3bac27519fff441aceea722d Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Gerg=C5=91=20Magyar?= Date: Mon, 7 Sep 2026 19:22:24 +0100 Subject: [PATCH 06/17] docs(bench): close FTS as an optimization target with measured evidence (#3209) PR #3208 landed two claims that further measurement disproved. The narrowing is not inconsistent. A true incremental leaf edit rebuilds exactly 8 of the 20 configured indexes -- the tables the writeback DMLs -- and the 20-index run I compared it against was a forced full rebuild (runner-identity trap). Per-index costs for those 8 sum to 7769ms against the 7525ms measured in-analyze. There is nothing to fix in `touchedFts`. The 845ms was not `import('./platform/capabilities.js')`. The CLI already imports that module statically; a cached dynamic import measures 0.035ms. The `await` is the first yield after the native FTS build and absorbs the libuv work still queued behind it. Recorded as a fourth measurement trap, since it invalidates any mark placed on an await that follows native work. What replaces them is a floor, established by probing a copy of the corpus index directly: - narrowing further: nothing left, the 8 tables are exactly the DML'd set - concurrent builds: hard error, one write transaction at a time - connection thread count: flat at 4/8/16/24 (7298/7133/7109/7345ms min-of-3), though the default burns ~60% more CPU for it - dropping `content` from File: 3541ms -> 241ms, but that deletes full-file keyword search (#2317/#2323); capping is a bad trade because the size distribution is flat The one lever left is overlap: the build runs on a libuv thread and hides behind main-thread JS (3337ms for the index plus a 3000ms JS burn, against ~6859ms serial). File rows are `{ name, filePath }` from `processStructure` with content lazy-read at CSV time, so they are known before parsing. What blocks it is that the DB is closed for the whole pipeline and that an early write moves `liveIndexMutationStarted` ahead of it. The `ponytail:` comment in `createSearchFTSIndexes` invited exactly the fix that cannot work -- every caller has already dropped the indexes it passes, so a presence gate would never fire. Replaced with the measured reason. Co-authored-by: Gergo Magyar Co-authored-by: Claude Opus 5 (1M context) --- gitnexus/bench/analyze-phase-breakdown.md | 119 +++++++++++++++------- gitnexus/src/core/search/fts-indexes.ts | 17 +++- 2 files changed, 98 insertions(+), 38 deletions(-) diff --git a/gitnexus/bench/analyze-phase-breakdown.md b/gitnexus/bench/analyze-phase-breakdown.md index 1751c3a3e..6706c1905 100644 --- a/gitnexus/bench/analyze-phase-breakdown.md +++ b/gitnexus/bench/analyze-phase-breakdown.md @@ -32,6 +32,13 @@ run once to stamp the identity, then edit and run again **without touching `dist`**. An earlier revision of this document reported full-rebuild numbers as if they were incremental for exactly this reason; they are corrected below. +**And a mark on an `await` is not a mark on that call.** The post-pipeline tail +was timed by injecting timestamp marks into a disposable `dist`. An `await` that +directly follows native work also absorbs whatever libuv still had queued, so +the cost lands on the wrong line — that is how 845ms ended up attributed to a +dynamic import that actually costs 0.035ms. Sanity-check any mark that lands on +a call with no plausible work in it. + That is three separate "silently fall back to full work" guards — non-git corpus, runner identity, and the escalation gate. Read the banner on every run before trusting a number. @@ -70,9 +77,16 @@ incremental path is genuinely taken. **31.7s total.** | **FTS index rebuild** (`buildSearchIndexesOrDegrade`) | **7525** | **24%** | | graph write (`loadGraphToLbug`, subgraph) | 3566 | 11% | | parse | ~2800 | 9% | -| `import('./platform/capabilities.js')` | 845 | 3% | +| post-FTS event-loop drain (see below) | 845 | 3% | | everything else | ~3000 | 9% | +The 845ms was originally recorded against `import('./platform/capabilities.js')`. +That import is not the cost: the CLI already imports the module statically, so a +cached dynamic import measures 0.035ms and `getRuntimeCapabilities()` 0.1ms. The +`await` there is the first yield after the native FTS build, so it absorbs +whatever libuv work was still queued. Any mark placed on an `await` immediately +after native work charges that work to the wrong line. + The incremental machinery works: the graph write was a 3,980-node subgraph, not the full 51,288. **Everything the #3194/#3196 parse work optimized is ~9% of this.** @@ -80,46 +94,80 @@ this.** A FULL-rebuild run of the same repo is 36-39s, with the graph write at ~6.3s and FTS at ~9.7s. Do not quote those as edit-loop numbers. -## FTS index rebuild — the largest non-resolution cost +## FTS index rebuild — the largest non-resolution cost, and it is a floor -Per-index, measured on one leaf-file edit: +A true incremental leaf edit rebuilds **8** of the 20 configured indexes — the +tables the writeback actually DMLs. Per-index cost (measured against a copy of +the corpus index, `bench/` probe, reproduced within 2% of the in-analyze number): ``` - 3531ms File.file_fts <- 35% of FTS on its own - 1558ms Function.function_fts - 535ms Method 491ms Const 466ms Property - 427ms Interface 363ms Class 280ms TypeAlias - 247ms Struct 242ms Variable 234ms Enum 222ms Trait ... + 3541ms File.file_fts <- 47% of the incremental FTS cost on its own + 1696ms Function.function_fts + 521ms Method 506ms Const 486ms Property + 403ms Interface 324ms Class 291ms TypeAlias ------ -10203ms across 20 indexes + 7769ms 8 indexes (in-analyze: 7525ms) ``` -Two findings. +A FULL rebuild does all 20 and costs ~10.2s; the extra 12 indexes are only +~2.5s, so **the narrowing is already doing its job**. An earlier revision of +this document called the narrowing "inconsistent" because one run rebuilt 8 and +another 20 — the 20-index run was a forced full rebuild (the runner-identity +trap above), not a leaf edit. There is nothing to fix there. -**`File.file_fts` dominates** because File nodes carry file content, so that one -index re-tokenizes ~30MB of source. Changing one file re-tokenizes all of it. +`File.file_fts` dominates because File rows carry whole-file content: 2442 rows, +32.9 MB, and Ladybug tokenizes at ~9.8 MB/s. -**20 indexes were rebuilt for a two-importer leaf edit.** `touchedFts` is meant -to narrow this, and a separate run rebuilt only 8 — so the narrowing is at least -inconsistent. It has a withdrawal path: `missingSearchFTSIndexTables` returns -`undefined` when the index catalog cannot be read (`fts-indexes.ts:285`), and -`tables: undefined` means rebuild everything. Pin that before optimizing, or the -optimization targets a set that is not actually being narrowed. +### Four ways out, all measured, all closed -The code states the underlying constraint plainly: _"createSearchFTSIndexes -re-tokenizes every stored row on every run."_ Indexes must be dropped for DML -(#2589, #2841), and the only way back is a whole-table rebuild. +**Narrow further — no.** The 8 tables are exactly the ones holding rows for the +6 files in the write set (1 changed + 5 importers). There is no fat. -Ranked next steps: +**Build the indexes concurrently — impossible.** A second connection issuing +`CREATE_FTS_INDEX` fails immediately: -1. **Take the FTS rebuild off the critical path.** Nothing in `analyze` reads - the index — only later searches do. Rebuild lazily on first search or in the - background after the swap. Removes the cost rather than shrinking it, and the - degraded-search state is already modelled (`ftsSkipReason`, "degrade rather - than throw"). -2. **Do not rebuild `File.file_fts` when no file content changed.** -3. **Fix or delete the narrowing** — a gate that silently withdraws looks like - protection and is not. +> `Cannot start a new write transaction in the system. Only one write transaction at a time is allowed in the system.` + +Eight builds on one connection serialize exactly (7886ms concurrent vs 7769ms +serial). + +**Raise the connection's thread count — no effect.** min-of-3 wall time at +4 / 8 / 16 / default(24) threads: 7298 / 7133 / 7109 / 7345 ms, inside the ~400ms +per-config spread. CPU burned does move — 8.8s / 9.7s / 11.8s / 13.6s — so the +default over-subscribes ~60% for nothing, but wall time is flat. + +**Skip `File.file_fts` when content did not change — cannot happen.** Any file +edit changes a File row, and Ladybug's FTS is not incremental: an index built +before an insert does not see the new row, so a changed row forces a whole-table +rebuild. Dropping `content` from the index takes it 3541ms → **241ms**, but that +is deleting full-file keyword search (#2317/#2323), not optimizing it. Capping +the indexed content is a bad trade — the size distribution is flat, so a 64 KB +cap still indexes 90% of the bytes while truncating the 72 largest files. + +### The one lever left: overlap + +The FTS build runs on a libuv thread, not the main thread, and fully overlaps +blocking JS: + +``` +index alone 3859ms +index + 3000ms of JS burn 3337ms (serial would be ~6859ms) +``` + +So the 3.5s File index could hide entirely behind the pipeline's ~17s of +main-thread JS. File rows are the only ones that make this possible: they are +`{ name, filePath }` from `processStructure`, with content lazy-read from disk at +CSV time, so they are fully determined by the file scan — before parsing, before +resolution. + +Two things block it today, and neither is small: + +1. The DB is **closed** for the whole pipeline (`closeLbug` before + `runPipelineFromRepo`, `initLbug` after). An early File write means holding a + write handle across the pipeline. +2. It moves `liveIndexMutationStarted` before the pipeline. A pipeline failure + would then leave the live index with fresh File content and stale symbols, + instead of untouched. ## scopeResolution is memory-traffic bound, not algorithmic @@ -179,10 +227,13 @@ run (73.0s vs 69.6s on overlay). Measure on a local filesystem. ## Open -The post-pipeline tail is now fully accounted (99.7%): FTS rebuild, the graph -write, and a 0.8s dynamic `capabilities.js` import between them. What remains -open is the FTS work above, and `scopeResolution` — still the single largest -step, memory-traffic bound, and untouched. +The post-pipeline tail is fully accounted (99.7%): FTS rebuild, the graph write, +and 0.8s of post-FTS event-loop drain between them. + +FTS is closed as an optimization target except for the overlap above, which is a +scheduling change to `run-analyze.ts`'s open/close discipline rather than +anything about FTS. `scopeResolution` is the single largest step, memory-traffic +bound, and still untouched. Every optimization in this document that looked compelling from the armchair died under measurement. Measure first, and check the banner. diff --git a/gitnexus/src/core/search/fts-indexes.ts b/gitnexus/src/core/search/fts-indexes.ts index 951734d4e..4965fcbd9 100644 --- a/gitnexus/src/core/search/fts-indexes.ts +++ b/gitnexus/src/core/search/fts-indexes.ts @@ -335,10 +335,19 @@ export async function createSearchFTSIndexes( // the old name+content index would silently persist. `dropFTSIndex` no-ops // when the index is absent (first-ever analyze) and clears the per-connection // memo so the create below actually runs. - // ponytail: this rebuilds every FTS index on every analyze instead of - // skipping when present; FTS build is proportional to symbol-table size and - // runs inside the existing FTS phase. Gate on a stored schema fingerprint if - // this rebuild cost ever shows up in analyze profiles. + // The cost DID show up in analyze profiles — 7.5s of a 31.7s edit loop on a + // 5350-file repo, `bench/analyze-phase-breakdown.md` — and a "skip when the + // index is already present" gate is NOT the answer, so don't reach for it. + // Every caller that reaches this loop has already dropped the indexes it + // passes in `tables`: the incremental writeback drops them because Ladybug + // cannot DML a table with a live FTS index (#2589), and a full rebuild + // builds into a fresh staging DB that never had one. A presence gate would + // therefore never fire. The cost is inherent — Ladybug's FTS is not + // incremental, so one changed row means re-tokenizing the whole table, and + // `File` alone is ~33MB of file content at ~10MB/s. The measured floor and + // the four exits that were tried and closed (narrow further, build + // concurrently, raise the connection thread count, drop `content`) are in + // that document. try { await dropFTSIndex(table, indexName); await createFTSIndex(table, indexName, [...properties], stemmer); From a7d9229326e3fdb069f8036e156b9e5be535af3a Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Tue, 8 Sep 2026 06:23:22 +0100 Subject: [PATCH 07/17] chore(deps)(deps): bump express-rate-limit in /gitnexus (#3214) --- gitnexus/package-lock.json | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/gitnexus/package-lock.json b/gitnexus/package-lock.json index 471c0e0fe..6e521f664 100644 --- a/gitnexus/package-lock.json +++ b/gitnexus/package-lock.json @@ -2994,9 +2994,9 @@ } }, "node_modules/express-rate-limit": { - "version": "8.6.2", - "resolved": "https://registry.npmjs.org/express-rate-limit/-/express-rate-limit-8.6.2.tgz", - "integrity": "sha512-YH4ru+eOJxQABscKFfRCy9R7x9QFGdezclVMwwgFFndzS2Xnm0uo6B0ABZsLhcpeptGv2qvuJVWlQr9gQZoC3A==", + "version": "8.7.0", + "resolved": "https://registry.npmjs.org/express-rate-limit/-/express-rate-limit-8.7.0.tgz", + "integrity": "sha512-hOwV7WOxXfjRpAM1DSJWZDXx3GhplwD8IfwuwvogD8i1Qnkgosw/H45s4ZnFAUHDAhPjlY9hLBvJhKmGMyY26g==", "license": "MIT", "dependencies": { "debug": "^4.4.3", From 95858e75493804fd445141e0c422595021ea2e8d Mon Sep 17 00:00:00 2001 From: Subham Kundu <43017632+Cenrax@users.noreply.github.com> Date: Mon, 7 Sep 2026 23:25:22 -0700 Subject: [PATCH 08/17] Update README.md (#3217) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Co-authored-by: Gergő Magyar --- README.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/README.md b/README.md index f2714bf89..99782975e 100644 --- a/README.md +++ b/README.md @@ -26,7 +26,7 @@

-

The nervous system for agent context.

+

The context engine for Enterprise Codebases

Indexes any codebase into a knowledge graph — every dependency, call chain, cluster, and execution flow — From d1463977c8804dbf71d5aeb3cb949c5ba37cc828 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Gerg=C5=91=20Magyar?= Date: Tue, 8 Sep 2026 08:22:12 +0100 Subject: [PATCH 09/17] feat(eval): Add bounded packed-scheduler primitives and offline replay benchmarks (#3206) * perf(eval): packed sweep scheduler and the harness that measured it Extracted from the combined skill-evolution branch so it can be reviewed on its own. Purely additive against main: no existing function changes behaviour, and sweep_packed_cells has no production caller yet. sweep_task_cells finishes one task before starting the next and drains a wave before refilling it, so a task with fewer cells than workers leaves workers idle and one slow cell stalls its whole wave. sweep_packed_cells feeds every task's cells through a single pool instead, keeping the breaker's meaning: a total submission order continued across task boundaries, a folder walking results in that order, and consecutive systemic failures counted there, so a doomed run aborts on the same cell it would have under waves. simulate_sweep.py is what produced the numbers. It drives the real schedulers with only the paid agent session stubbed, using the measured per-arm durations in session_durations.json divided by a scale factor. The distribution's shape is kept deliberately - median 826s against a 5400s ceiling - because that spread is the entire reason a barrier costs anything, and uniform sleeps would erase the effect under test. All schedulers consume one identical seeded plan. Measured at workers=3 against the review corpus, packing is worth about 40% of a cold sweep, and it is the only change that moves a seeded weekly run at all - there a task is three cells and a wave is never full. The submission window is a real trade, measured with failures injected at four positions: window 3 -> -8% wall, overrun 2 (the wave scheduler's own bound) window 6 -> -27% wall, overrun 4 window 12 -> -42% wall, overrun 9 window 54 -> -44% wall, overrun 11 Overrun is wasted paid sessions on an aborted sweep. The default multiplier is 2; the curve lives in the constant's comment so raising it is an informed decision. Contention was measured separately by burning real CPU in subprocesses under taskset: the advantage holds between -40% and -47% from 24 cores down to an oversubscribed 2, though packing erodes faster than waves do because packing is what creates the concurrency. measure_evolution_cost.py is the offline cost model, with no runtime caller. It reports workers from the workflow's current default, which on this base is 1. Limits worth stating: sleeping threads do not contend and the duration sample was itself recorded at workers=1, so the speedups are upper bounds; the ordering of the schedulers is trustworthy because they were compared under identical conditions, the magnitudes are not. 562 eval tests pass at this base. The two test_model_gateway.py failures, test_locked_litellm_translates_messages_to_offline_responses and test_openai_gateway_never_leaves_proxy_output_on_an_undrained_pipe, fail identically on origin/main in this environment. * fix(eval): compare the shipped window and bound the overrun by it Address PR review feedback (#3206). run_faithful defaulted its submission window to `workers` while runner.sweep_packed_cells defaults to `max(workers * PACKED_WINDOW_MULTIPLIER, workers)`, so every run that named no window compared a prototype queued twice as tightly as the shipped scheduler and presented it as the production invariant. The faithful default now reads the same constant. Measured at workers=3, faithful and production agreed on nothing before and agree exactly now: breaker overrun 2/1/2 vs 2/4/3 becomes 2/4/3 vs 2/4/3 across the three failure positions. The contention sweep hard-coded `window=12` for faithful only, which the production run never saw - masked at workers=6 where both are 12. Removed, and the production measurement it was already paying for is now reported as `production_s` instead of being discarded. breaker_fidelity checked the overrun against `args.workers`. The bound the producer actually enforces is `window - 1` cells past the fold pointer, which is the wave scheduler's own `workers - 1` when window == workers; against the shipped default of 6 the old predicate reported a failure for an in-bound run. The window is now passed explicitly, reported in each row, and checked against its own bound. --window was parsed and never read. Wired into the schedulers that hold one. Dropped two unused plan constructions CodeQL flagged, and the `skipped` set in sweep_packed_cells that nothing reads - the None appended to `submitted` is the skip representation the fold loop consumes. Verification: 562 passed, 15 skipped, 2 failed (the two test_model_gateway.py failures the PR description documents as reproducing on origin/main), ruff clean. * fix(eval): carry the cancellation scope into packed cells, reject the args that hang Address PR review feedback (#3206). sweep_packed_cells submits from a producer THREAD, and a new thread starts with an empty context, so `copy_context()` there copied the producer's context rather than the one cancellation_scope had just bound _CANCELLATION in. Every packed cell therefore ran with no cancellation event, and run_managed falls back to _CANCELLATION when none is passed - so a cancelled run's subprocesses would never have learned about it. sweep_task_cells gets this right for free by submitting from the thread that entered the scope. Reproduced directly: packed workers observed [False, False], wave workers [True, True]. The caller's context is now captured before the producer starts and copied per submission; the new test fails without the fix. Three CLI arguments were accepted and then wedged the run: --scale 0 ZeroDivisionError before any scheduler starts --graph-seconds -1 hangs: the builder thread dies on a negative sleep, every scheduler waits on a readiness event nobody sets --window 0 (faithful) hangs: submitted - fold_pointer >= 0 holds before the first submission, so the producer and the consumer wait on each other The first two are rejected at the parser, which is the only layer that runs before a thread exists. run_faithful now enforces the same window >= workers rule sweep_packed_cells already had, so the prototype rejects exactly what the shipped function rejects. All three were confirmed to crash or hang first. Verification: 563 passed, 15 skipped, 2 failed (the two test_model_gateway.py failures the PR description documents as reproducing on origin/main), ruff clean. * chore(autofix): apply prettier + eslint fixes via /autofix command * Address PR review feedback (#3206) Preserve settled sibling rows when a packed cell raises. run_cell deliberately lets unexpected harness exceptions propagate, and sweep_task_cells answers that by folding every non-failing sibling before it re-raises - the cells already ran and already spent their budget, so dropping their rows means paying for evidence the sweep then discards. sweep_packed_cells called future.result() bare, so the fold stopped at the failing index and every later cell that had already completed was silently lost. It now folds forward over the settled futures before re-raising. The failing index itself has no row, since execute() assigns only on success, so folding forward cannot duplicate it. Pinned by a regression test that fails without the fix: the later cell is made to finish first, so there is real settled evidence to lose at the moment cell 0 raises. Reject arguments that cannot produce a run, at the boundary rather than deep inside a thread. NaN defeats every comparison it appears in, so the existing "> 0" and ">= 0" checks admitted --scale nan and --graph-seconds nan; the NaN then reached time.sleep in a worker or the graph thread, raised there, and left every scheduler waiting forever on a readiness event nobody would set. Infinity was worse than a crash: it scaled all durations to zero and the run reported a sweep that took no time. Both flags now require a finite value. The count flags are indexed or handed straight to a thread pool, so a zero surfaced as an IndexError on plans[0], a median over an empty sequence, or ThreadPoolExecutor's own error - none naming the flag responsible. --workers, --repeat and --runs now require at least 1. Two flags were not in the review but carry the same invariant and the same one-line treatment, so they are fixed with the class rather than left to resurface: --runs (same empty-plan path as --repeat) and --window, where zero admits no cell at all because the producer waits for a fold pointer to move past a cell it was never allowed to submit. Verified each guard fires with its own message rather than a stack trace. 563 eval tests pass. Note: pre-existing failures in test_model_gateway.py not addressed by this PR - litellm[proxy]'s console script is absent in this environment, and neither test touches the files changed here. * Address PR review feedback (#3206), round 2 Stop charging the fed baseline for overlap the wave scheduler gets free. run_fed is documented as pricing the barrier alone, but it slept graph_seconds serially before every task, while run_wave starts one background builder that prepares task N+1 while task N's cells run. The fed-versus-wave delta therefore mixed the loss of that overlap into what was reported as the price of the barrier. run_fed now uses the same builder, started before the clock, so the barrier is the only remaining difference. This moved the numbers. On the weekly profile fed was 4.203s and is now 3.694s, exactly equal to wave - which is the answer that profile should give. On cold, fed was 5.995s and is now 5.487s, so the measured price of the barrier widens from 1.844s to 2.352s: the old arrangement understated it by about a quarter. No committed results file or PR-body figure quotes these, so there is nothing stale to regenerate. Enforce the window bound the schedulers actually hold. Last round's guard required only >= 1, but run_faithful and sweep_packed_cells both refuse a window below the worker count, so --scheduler faithful --workers 3 --window 1 passed validation and then died on an uncaught ValueError. The check now uses the worker count. It also uses the LARGEST worker count the invocation will really use. --contention-sweep runs its own counts irrespective of --workers, so validating against --workers alone let the three-worker measurements finish and then raised on the six-worker one, losing the run partway through. Those counts are now a named constant the validator can see. Verified: --scheduler faithful --workers 3 --window 1 is rejected naming 3, and --workers 3 --window 3 --contention-sweep is rejected naming 6. No regression test for the graph-overlap fix. Discriminating it from the old behaviour requires cell work to overlap graph work, which makes the assertion a timing comparison, and this project does not take non-deterministic tests. It is verified by the before/after measurement above instead. 564 eval tests pass. Note: pre-existing failures in test_model_gateway.py not addressed by this PR - litellm[proxy]'s console script is absent here. --------- Co-authored-by: Gergo Magyar Co-authored-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> --- eval/tests/test_measure_evolution_cost.py | 122 +++ eval/tests/test_workflow_bench.py | 157 ++++ eval/workflow_bench/measure_evolution_cost.py | 311 ++++++++ eval/workflow_bench/runner.py | 198 +++++ eval/workflow_bench/session_durations.json | 33 + eval/workflow_bench/simulate_sweep.py | 747 ++++++++++++++++++ 6 files changed, 1568 insertions(+) create mode 100644 eval/tests/test_measure_evolution_cost.py create mode 100644 eval/workflow_bench/measure_evolution_cost.py create mode 100644 eval/workflow_bench/session_durations.json create mode 100644 eval/workflow_bench/simulate_sweep.py diff --git a/eval/tests/test_measure_evolution_cost.py b/eval/tests/test_measure_evolution_cost.py new file mode 100644 index 000000000..c6035bc86 --- /dev/null +++ b/eval/tests/test_measure_evolution_cost.py @@ -0,0 +1,122 @@ +"""Cost model for the evolution wall clock: measured cells, real schedules.""" + +from __future__ import annotations + +import pytest + +from workflow_bench.measure_evolution_cost import ( + CANDIDATE_ARM, + SHA_OVERHEAD_SECONDS, + DURATIONS_BY_ARM, + PROPOSER_SECONDS, + REVIEW_ARMS, + expected_task_seconds, + fed_makespan, + fed_pool_enabled, + generation_seconds, + graph_pipeline_enabled, + paid_arms, + task_cells, + wave_makespan, +) + + +def test_every_arm_has_its_own_unsorted_sample(): + assert set(DURATIONS_BY_ARM) == set(REVIEW_ARMS) + for arm, sample in DURATIONS_BY_ARM.items(): + assert len(sample) >= 10, arm + # Sorting would hand each task a uniform block and hide the variance + # the whole model exists to price. + assert list(sample) != sorted(sample), arm + assert PROPOSER_SECONDS > 0 + assert SHA_OVERHEAD_SECONDS > 0 + + +def test_weekly_reuse_pays_the_candidate_arm_only(): + assert paid_arms(weekly=True, reuse_enabled=True) == (CANDIDATE_ARM,) + assert paid_arms(weekly=False, reuse_enabled=True) == REVIEW_ARMS + assert paid_arms(weekly=True, reuse_enabled=False) == REVIEW_ARMS + + +def test_cells_are_submitted_run_major_arm_minor(): + # runner.py: [(run_idx, arm) for run_idx in range(runs) for arm in arms]. + # At workers=3 that puts one cell of each arm in every wave. + cells = task_cells(2, REVIEW_ARMS, 0) + assert len(cells) == 6 + expected = [DURATIONS_BY_ARM[arm][run] for run in range(2) for arm in REVIEW_ARMS] + assert cells == expected + + +def test_overhead_is_charged_per_sha_and_outside_the_pool(): + # Two properties at once: the residual sits outside the schedule, where more + # workers cannot dissolve it, and it scales with SHAs rather than cells. + assert task_cells(1, (CANDIDATE_ARM,), 0) == [DURATIONS_BY_ARM[CANDIDATE_ARM][0]] + wide = generation_seconds( + task_count=1, runs=3, arms=REVIEW_ARMS, workers=9, fed_pool=True, unique_shas=5 + ) + assert wide >= PROPOSER_SECONDS + 5 * SHA_OVERHEAD_SECONDS + + +def test_sweep_overhead_does_not_shrink_with_the_arm_count(): + """The bias that made weekly look cheaper than it is. + + A seeded weekly generation pays one arm instead of three but builds exactly + the same graphs. Charging the residual per cell billed it a third of a cost + the real sweep still pays; per SHA, the two attribute the same setup. + """ + + kwargs = dict(task_count=6, runs=3, workers=3, fed_pool=False, unique_shas=5) + weekly = generation_seconds(arms=(CANDIDATE_ARM,), **kwargs) + cold = generation_seconds(arms=REVIEW_ARMS, **kwargs) + weekly_sessions = 6 * expected_task_seconds(3, (CANDIDATE_ARM,), 3, fed_pool=False) + cold_sessions = 6 * expected_task_seconds(3, REVIEW_ARMS, 3, fed_pool=False) + # Whatever each wall is, the non-session part is identical. + assert round(weekly - weekly_sessions) == round(cold - cold_sessions) + # Cycling wraps, so a task can ask for more runs than the sample holds. + long_sample = task_cells(len(DURATIONS_BY_ARM[CANDIDATE_ARM]) + 2, (CANDIDATE_ARM,), 0) + assert len(long_sample) == len(DURATIONS_BY_ARM[CANDIDATE_ARM]) + 2 + + +def test_a_wave_costs_its_slowest_cell_and_a_fed_pool_does_not(): + slow = [10.0, 1.0, 1.0, 10.0, 1.0, 1.0] + assert wave_makespan(slow, 3) == 20.0 + # Fed: one worker takes the first 10; the second 10 lands on a worker that + # has already cleared a 1, and the remaining 1s fill the third. + assert fed_makespan(slow, 3) == 11.0 + assert fed_makespan(slow, 1) == wave_makespan(slow, 1) == 24.0 + + +def test_expected_task_seconds_is_alignment_averaged_and_deterministic(): + waved = expected_task_seconds(3, REVIEW_ARMS, 3, fed_pool=False) + assert waved == expected_task_seconds(3, REVIEW_ARMS, 3, fed_pool=False) + assert expected_task_seconds(0, REVIEW_ARMS, 3, fed_pool=False) == 0.0 + assert expected_task_seconds(3, (), 3, fed_pool=False) == 0.0 + # The barrier can only cost time, never save it. + assert waved >= expected_task_seconds(3, REVIEW_ARMS, 3, fed_pool=True) + + +def test_a_generation_pays_one_proposer_session_on_top_of_its_tasks(): + one = generation_seconds( + task_count=1, runs=3, arms=REVIEW_ARMS, workers=3, fed_pool=False, unique_shas=1 + ) + two = generation_seconds( + task_count=2, runs=3, arms=REVIEW_ARMS, workers=3, fed_pool=False, unique_shas=1 + ) + # Each extra task adds exactly one task's makespan. The proposer and the + # per-SHA sweep overhead are both paid once, not per task. + assert two - one == pytest.approx( + one - PROPOSER_SECONDS - SHA_OVERHEAD_SECONDS, abs=2.0 + ) + + +def test_feature_flags_read_the_runner_not_the_wish(): + assert graph_pipeline_enabled("def _run_sweep(): pass") == 0 + assert graph_pipeline_enabled("graph_prefetch = GraphPrefetch(...)") == 1 + assert fed_pool_enabled("def _run_wave(): pass") == 0 + assert fed_pool_enabled("def _run_fed_pool(): pass") == 1 + + +@pytest.mark.parametrize("workers", [1, 3, 8]) +def test_more_workers_never_lengthen_a_task(workers): + serial = expected_task_seconds(3, REVIEW_ARMS, 1, fed_pool=True) + assert expected_task_seconds(3, REVIEW_ARMS, workers, fed_pool=True) <= serial diff --git a/eval/tests/test_workflow_bench.py b/eval/tests/test_workflow_bench.py index 45d0b2f35..d9c9c0f92 100644 --- a/eval/tests/test_workflow_bench.py +++ b/eval/tests/test_workflow_bench.py @@ -5,11 +5,16 @@ import os import re import shlex import subprocess +import threading from pathlib import Path import pytest import yaml +from typing import Any + +from workflow_bench import runner +from workflow_bench.process_control import _CANCELLATION, cancellation_scope from workflow_bench.runner import ( aggregate, broken_incumbent_arms, @@ -512,3 +517,155 @@ def test_run_evolution_script_is_the_shared_ci_and_local_entrypoint(): assert "--include-expensive" in argv assert "claude-sonnet-5" not in argv assert printed.stderr # rewrite notice goes to stderr + + +def _packed_cells(tasks: int, runs: int, arms: tuple[str, ...]) -> list[tuple[str, int, str]]: + return [(f"t{t}", r, a) for t in range(tasks) for r in range(runs) for a in arms] + + +def test_packed_sweep_runs_every_cell_and_folds_in_submission_order(): + """Fold order is the contract the breaker rests on. + + Cells finish in whatever order the pool returns them, but the breaker counts + CONSECUTIVE systemic failures, which only means something in a fixed order. + """ + + cells = _packed_cells(3, 2, ("review", "candidate_review")) + folded: list[tuple[str, int, str]] = [] + streak, tripped = runner.sweep_packed_cells( + cells, + workers=4, + run=lambda task, run_idx, arm: {"error_kind": None, "review_evidence_valid": True}, + on_start=lambda *_: None, + on_record=lambda task, run_idx, arm, _rec: folded.append((task, run_idx, arm)), + outage_streak=0, + outage_limit=0, + ) + assert folded == cells + assert (streak, tripped) == (0, False) + + +def test_packed_sweep_trips_the_breaker_on_the_same_cell_waves_would(): + """Packing must not change WHEN a doomed run aborts, only how it is fed.""" + + cells = _packed_cells(3, 3, ("review",)) + fail_from = 2 + folded: list[int] = [] + + def run(task: str, run_idx: int, arm: str) -> dict[str, Any]: + index = cells.index((task, run_idx, arm)) + systemic = index >= fail_from + return { + "error_kind": "session-error" if systemic else None, + "review_evidence_valid": not systemic, + } + + streak, tripped = runner.sweep_packed_cells( + cells, + workers=2, + run=run, + on_start=lambda *_: None, + on_record=lambda t, r, a, _rec: folded.append(cells.index((t, r, a))), + outage_streak=0, + outage_limit=runner.DEFAULT_OUTAGE_STREAK, + ) + assert tripped is True + assert streak == runner.DEFAULT_OUTAGE_STREAK + # Five consecutive systemic failures starting at index 2 -> trips on index 6. + assert folded[-1] == fail_from + runner.DEFAULT_OUTAGE_STREAK - 1 + assert folded == sorted(folded), "records must fold in submission order" + + +def test_packed_sweep_skips_a_task_whose_assets_never_arrive(): + """A task that cannot be prepared is skipped, not run against nothing.""" + + cells = _packed_cells(3, 2, ("review",)) + ran: list[str] = [] + runner.sweep_packed_cells( + cells, + workers=3, + run=lambda task, run_idx, arm: ran.append(task) + or {"error_kind": None, "review_evidence_valid": True}, + on_start=lambda *_: None, + on_record=lambda *_: None, + outage_streak=0, + outage_limit=0, + await_ready=lambda task: task != "t1", + ) + assert set(ran) == {"t0", "t2"} + assert "t1" not in ran + + +def test_packed_sweep_workers_inherit_the_runs_cancellation_event(): + """A worker that cannot see the event runs on after the sweep is cancelled. + + The cells are submitted from a producer THREAD, and a new thread starts with + an empty context - so copying the context at submission copies the wrong one + unless the caller's is captured first. run_managed falls back to + _CANCELLATION when no event is passed, which is how a cell's subprocesses + learn the run was cancelled at all. + """ + + seen: list[threading.Event | None] = [] + event = threading.Event() + with cancellation_scope(event): + runner.sweep_packed_cells( + _packed_cells(2, 1, ("review",)), + workers=2, + run=lambda *_: seen.append(_CANCELLATION.get()) or {"error_kind": None}, + on_start=lambda *_: None, + on_record=lambda *_: None, + outage_streak=0, + outage_limit=0, + ) + assert seen and all(observed is event for observed in seen) + + +def test_packed_sweep_window_must_keep_the_pool_fed(): + with pytest.raises(ValueError, match="window must be at least workers"): + runner.sweep_packed_cells( + _packed_cells(1, 1, ("review",)), + workers=4, + run=lambda *_: {"error_kind": None}, + on_start=lambda *_: None, + on_record=lambda *_: None, + outage_streak=0, + outage_limit=0, + window=2, + ) + + +def test_a_raising_packed_cell_still_persists_its_settled_siblings(): + """A crash in one cell must not erase the evidence of cells that finished. + + run_cell deliberately lets unexpected harness exceptions propagate, and the + wave scheduler answers that by folding every non-failing sibling before it + re-raises. The packed scheduler has to hold the same contract: the later + cells already ran and already cost money, so losing their rows would mean + paying for evidence the sweep then throws away. + """ + + folded: list[tuple[int, str]] = [] + started = threading.Event() + + def run(task_id: str, run_idx: int, arm: str) -> dict[str, Any]: + if run_idx == 0: + # Let the later cell finish first, so there is settled evidence to + # lose at the moment this one raises. + started.wait(timeout=5) + raise RuntimeError("harness bug in cell 0") + started.set() + return {"error_kind": None} + + with pytest.raises(RuntimeError, match="harness bug in cell 0"): + runner.sweep_packed_cells( + _packed_cells(1, 2, ("review",)), + workers=2, + run=run, + on_start=lambda *_: None, + on_record=lambda task_id, run_idx, arm, _rec: folded.append((run_idx, arm)), + outage_streak=0, + outage_limit=0, + ) + + assert (1, "review") in folded, "the sibling that completed was never recorded" diff --git a/eval/workflow_bench/measure_evolution_cost.py b/eval/workflow_bench/measure_evolution_cost.py new file mode 100644 index 000000000..4d18de746 --- /dev/null +++ b/eval/workflow_bench/measure_evolution_cost.py @@ -0,0 +1,311 @@ +#!/usr/bin/env python3 +"""Cheap cost model for the skill-evolution review generation. + +This is the ce-optimize measurement harness. It does not start Claude and it +does not replay a run. It reads the review corpus, the evolve defaults and the +workflow's workers default, then schedules the measured cell durations in +``session_durations.json`` the way ``sweep_task_cells`` schedules real cells. + +Everything priced here is measured. Cell durations and the proposer session +come from a real artifact, and the work outside the agent sessions comes from +that run's own step wall minus the time its sessions and proposer account for. + +Weekly assumes a matching seed, so every reusable comparator cell is skipped +and only the candidate arm is paid. Cold assumes an empty seed. +""" + +from __future__ import annotations + +import json +import math +import re +import statistics as st +import subprocess +import sys +from pathlib import Path + +REPO_ROOT = Path(__file__).resolve().parents[2] +EVAL_ROOT = REPO_ROOT / "eval" +REVIEW_TASKS = EVAL_ROOT / "workflow_bench" / "tasks.review.scenarios.yaml" +EVOLVE_PY = EVAL_ROOT / "workflow_bench" / "evolve.py" +RUNNER_PY = EVAL_ROOT / "workflow_bench" / "runner.py" +ARTIFACTS_PY = EVAL_ROOT / "workflow_bench" / "runner_artifacts.py" +REUSE_PY = EVAL_ROOT / "workflow_bench" / "comparator_reuse.py" +WORKFLOW = REPO_ROOT / ".github" / "workflows" / "gitnexus-skill-evolution.yml" + +MEASURED = json.loads( + (Path(__file__).resolve().parent / "session_durations.json").read_text(encoding="utf-8") +) +# Per arm, because the arms are not interchangeable and the weekly lane pays +# only the candidate one. Cells are submitted run-major and arm-minor +# (runner.py ``planned``), so at workers=3 every wave holds one cell of each +# arm and the slowest arm sets the wave. +DURATIONS_BY_ARM: dict[str, tuple[float, ...]] = { + arm: tuple(values) for arm, values in MEASURED["cell_duration_s_by_arm"].items() +} +PROPOSER_SECONDS: float = MEASURED["proposer_duration_s"] +_RESIDUAL = MEASURED["residual"] +# Clone, graph build, sandbox, teardown: the sweep's own time, taken as that +# run's step wall minus what its sessions and proposer account for. Charged +# SERIALLY, outside the pool, and charged PER SHA rather than per cell. The +# residual mixes per-cell work with per-SHA graph setup and the artifact cannot +# separate them; per-SHA is the direction that refuses to credit a run for +# shrinking work it still performs, which per-cell did - a weekly generation +# pays one arm instead of three but builds exactly the same graphs. See +# session_durations.json residual._split_assumption. +SHA_OVERHEAD_SECONDS: float = _RESIDUAL["sha_overhead_s"] + +# runner.py CANDIDATE_ARMS derives the candidate arm from its incumbent, and +# only an incumbent row can be reused from a prior generation. +CANDIDATE_ARM = "candidate_review" +REVIEW_ARMS = ("ce_review", "review", CANDIDATE_ARM) + +SUITE_FILES = ( + "tests/test_measure_evolution_cost.py", + "tests/test_comparator_reuse.py", + "tests/test_evolve.py", + "tests/test_sanitized_graph.py", + "tests/test_workflow_bench.py", + "tests/test_workflow_bench_sessions.py", + "tests/test_session_progress.py", +) + + +def _read(path: Path) -> str: + return path.read_text(encoding="utf-8") + + +def review_tasks(text: str) -> list[dict[str, str]]: + tasks: list[dict[str, str]] = [] + current: dict[str, str] | None = None + for raw in text.splitlines(): + line = raw.strip() + if line.startswith("id:"): + if current is not None: + tasks.append(current) + current = {"id": line.split(":", 1)[1].strip()} + elif line.startswith("ref:") and current is not None: + current["ref"] = line.split(":", 1)[1].strip() + if current is not None: + tasks.append(current) + return tasks + + +def evolve_default(name: str, text: str) -> int: + match = re.search(rf'add_argument\("--{re.escape(name)}".*?default=(\d+)', text, flags=re.S) + if match is None: + raise ValueError(f"evolve.py is missing --{name} default") + return int(match.group(1)) + + +def workflow_dispatch_workers(text: str) -> int: + match = re.search(r"^\s+workers:\n(?:.*\n)*?^\s+default: '(\d+)'", text, flags=re.M) + if match is None: + raise ValueError("workflow_dispatch workers default is missing") + return int(match.group(1)) + + +def feature_enabled() -> tuple[int, int]: + evolve = _read(EVOLVE_PY) + runner = _read(RUNNER_PY) + artifacts = _read(ARTIFACTS_PY) + reuse = int( + REUSE_PY.is_file() + and "--reuse-results" in evolve + and "select_reusable_comparator_rows" in runner + and "CANDIDATE" in _read(REUSE_PY) + ) + templates = int("def copy_isolated_tree" in artifacts and "clone_templates" in runner) + return reuse, templates + + +def graph_pipeline_enabled(runner_text: str) -> int: + """True when the runner prefetches the next SHA during paid sessions.""" + + return int("prefetch_next_graph" in runner_text or "GraphPrefetch" in runner_text) + + +def fed_pool_enabled(runner_text: str) -> int: + """True when the sweep feeds a live pool instead of waiting on waves.""" + + return int("def _run_fed_pool" in runner_text) + + +def paid_arms(weekly: bool, reuse_enabled: bool) -> tuple[str, ...]: + """Arms a generation actually pays for.""" + + if weekly and reuse_enabled: + return (CANDIDATE_ARM,) + return REVIEW_ARMS + + +def task_cells(runs: int, arms: tuple[str, ...], offset: int) -> list[float]: + """One task's cell durations in submission order: run-major, arm-minor. + + Each arm draws from its own measured sample, cycled from ``offset`` so the + caller can average over every alignment instead of trusting one. + """ + + cells: list[float] = [] + for run_idx in range(runs): + for arm in arms: + sample = DURATIONS_BY_ARM[arm] + cells.append(sample[(offset + run_idx) % len(sample)]) + return cells + + +def wave_makespan(durations: list[float], workers: int) -> float: + """Today's scheduler: fixed waves of ``workers``, with a barrier between.""" + + return sum( + max(durations[start : start + workers]) for start in range(0, len(durations), workers) + ) + + +def fed_makespan(durations: list[float], workers: int) -> float: + """Continuously fed pool: a free worker takes the next cell immediately.""" + + busy_until = [0.0] * workers + for duration in durations: + first = min(range(workers), key=busy_until.__getitem__) + busy_until[first] += duration + return max(busy_until) + + +def expected_task_seconds( + runs: int, arms: tuple[str, ...], workers: int, *, fed_pool: bool +) -> float: + """Mean makespan of one task over every alignment of the measured samples. + + One fixed alignment would let an accident of the source run - its slowest + cells happen to come first - decide the answer. Averaging keeps the real + multiset and the real ordering effects without that artifact, and stays + deterministic. + """ + + if runs < 1 or not arms: + return 0.0 + makespan = fed_makespan if fed_pool else wave_makespan + # lcm, not max: with samples of 13 and 14, max would wrap the shorter one + # and count its first entry twice. + alignments = math.lcm(*(len(DURATIONS_BY_ARM[arm]) for arm in arms)) + return ( + sum(makespan(task_cells(runs, arms, offset), workers) for offset in range(alignments)) + / alignments + ) + + +def generation_seconds( + *, + task_count: int, + runs: int, + arms: tuple[str, ...], + workers: int, + fed_pool: bool, + unique_shas: int, +) -> int: + """Whole generation: proposer, then the tasks back to back, plus overhead. + + Prices a HEALTHY sweep. A run whose cells return unusable evidence does not + reach this wall at all: the outage breaker aborts after + ``DEFAULT_OUTAGE_STREAK`` consecutive systemic failures, which for the + sample's own error sequence is cell 5 of 41. + + Sweep overhead is charged per SHA, so it does not shrink with the arm count. + Weekly pays one arm instead of three but builds the same graphs, and billing + that per cell credited it for a saving the real run never makes. + """ + + return round( + PROPOSER_SECONDS + + task_count * expected_task_seconds(runs, arms, workers, fed_pool=fed_pool) + + unique_shas * SHA_OVERHEAD_SECONDS + ) + + +def _pytest_python() -> list[str]: + venv_python = EVAL_ROOT / ".venv" / "bin" / "python" + if venv_python.is_file(): + return [str(venv_python)] + if (EVAL_ROOT / "uv.lock").is_file(): + return ["uv", "run", "--locked", "--extra", "dev", "python"] + return [sys.executable] + + +def suite_passed() -> int: + files = [name for name in SUITE_FILES if (EVAL_ROOT / name).is_file()] + if not files: + return 0 + cmd = [*_pytest_python(), "-m", "pytest", *files, "-q", "--tb=no", "--no-header"] + try: + completed = subprocess.run( + cmd, cwd=EVAL_ROOT, check=False, capture_output=True, text=True, timeout=240 + ) + except (OSError, subprocess.TimeoutExpired): + return 0 + return int(completed.returncode == 0) + + +def main() -> int: + tasks = review_tasks(_read(REVIEW_TASKS)) + evolve = _read(EVOLVE_PY) + runner = _read(RUNNER_PY) + runs = evolve_default("runs", evolve) + workers = workflow_dispatch_workers(_read(WORKFLOW)) + reuse_enabled, clone_templates_enabled = feature_enabled() + fed_pool = fed_pool_enabled(runner) + + # Both walls build the same graphs; the arm count does not change that. + unique_shas = len({t.get("ref", "") for t in tasks if t.get("ref")}) + payload: dict[str, object] = {} + for label, weekly in (("weekly", True), ("cold", False)): + arms = paid_arms(weekly, bool(reuse_enabled)) + payload[f"estimated_{label}_wall_seconds"] = generation_seconds( + task_count=len(tasks), + runs=runs, + arms=arms, + workers=workers, + fed_pool=bool(fed_pool), + unique_shas=unique_shas, + ) + payload[f"paid_{label}_cells"] = len(tasks) * runs * len(arms) + # What the wave barrier costs: the same cells, continuously fed. + payload[f"fed_pool_{label}_wall_seconds"] = generation_seconds( + task_count=len(tasks), + runs=runs, + arms=arms, + workers=workers, + fed_pool=True, + unique_shas=unique_shas, + ) + + all_durations = [d for sample in DURATIONS_BY_ARM.values() for d in sample] + payload.update( + { + "suite_passed": suite_passed(), + "promotion_min_runs": evolve_default("promotion-min-runs", evolve), + "review_task_count": len(tasks), + "candidate_cells": len(tasks) * runs, + "workers": workers, + "unique_task_shas": len({t.get("ref", "") for t in tasks if t.get("ref")}), + "reuse_enabled": reuse_enabled, + "clone_templates_enabled": clone_templates_enabled, + "graph_pipeline_enabled": graph_pipeline_enabled(runner), + "fed_pool_enabled": fed_pool, + "measured_cell_count": len(all_durations), + "median_cell_seconds": round(st.median(all_durations)), + "mean_cell_seconds": round(st.mean(all_durations)), + "max_cell_seconds": round(max(all_durations)), + "median_candidate_cell_seconds": round(st.median(DURATIONS_BY_ARM[CANDIDATE_ARM])), + "mean_candidate_cell_seconds": round(st.mean(DURATIONS_BY_ARM[CANDIDATE_ARM])), + "proposer_seconds": round(PROPOSER_SECONDS), + "sha_overhead_seconds": round(SHA_OVERHEAD_SECONDS, 1), + } + ) + json.dump(payload, sys.stdout, sort_keys=True) + sys.stdout.write("\n") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/eval/workflow_bench/runner.py b/eval/workflow_bench/runner.py index 82185186c..9f9dd9ef3 100644 --- a/eval/workflow_bench/runner.py +++ b/eval/workflow_bench/runner.py @@ -851,6 +851,204 @@ def sweep_task_cells( return outage_streak, False +# A sustained upstream outage shows up as a run of session/infra/cleanup +# failures. (cleanup-failure overwrites the primary error_kind, so a +# session-error whose worktree cleanup also failed still counts.) A task's own +# resolved=False is real signal, not an outage, so it never trips the breaker. +# How far ahead of the in-order fold pointer cells may be submitted, as a +# multiple of the worker count. This is the wall-clock/wasted-cell trade, and it +# is a real one - measured against the review corpus at workers=3, with failures +# injected at four different positions: +# +# window wall vs waves worst overrun +# 3 -8% 2 (the wave scheduler's own bound) +# 6 -27% 4 +# 12 -42% 9 +# 54 -44% 11 +# +# Overrun is wasted paid sessions when the breaker trips, at roughly $70 each. +# 2 is the default because it keeps the worst case within 2x the wave bound +# while taking most of the gain; raise it if a run's wall clock costs more than +# an occasional handful of cells on an aborted sweep. +PACKED_WINDOW_MULTIPLIER = 2 + + +def sweep_packed_cells( + cells: Sequence[tuple[str, int, str]], + *, + workers: int, + run: Callable[[str, int, str], dict[str, Any]], + on_start: Callable[[str, int, str], None], + on_record: Callable[[str, int, str, dict[str, Any]], None], + outage_streak: int, + outage_limit: int, + window: int | None = None, + await_ready: Callable[[str], bool] | None = None, + cancel_event: threading.Event | None = None, +) -> tuple[int, bool]: + """Run cells from EVERY task through one pool; return (streak, tripped). + + ``sweep_task_cells`` finishes one task before starting the next and drains a + wave before refilling it, so a task with fewer cells than ``workers`` leaves + workers idle and a slow cell stalls its whole wave. Packing every task's + cells into one continuously fed pool removes both, which is worth about 40% + of a cold sweep's wall clock and is the only thing that moves a seeded + weekly run at all - there, a task is three cells and a wave is never full. + + The breaker keeps its exact meaning. ``cells`` is a total submission order + (task-major, run-major, arm-minor - the same order waves fold in, continued + across task boundaries), a folder walks results in precisely that order, and + "consecutive systemic failures" is evaluated there. So the run aborts on the + same logical cell it would have aborted on under waves. + + ``window`` is what bounds the overrun, and it is load-bearing. The halt flag + alone is not enough: the folder walks in order, so a slow early cell lets + workers race ahead, and by the time the breaker trips those cells have + already paid for their sessions. Measured, an unbounded queue overran by 11 + cells at ``workers=3`` where the wave scheduler overruns by 2. Holding + submission to ``window`` cells beyond the fold point caps it, trading + packing for wasted cells - see ``PACKED_WINDOW_MULTIPLIER`` for the curve. + + ``await_ready`` gates a task's first cell on whatever that task still needs + (a sanitized clone, a graph). It returns False to abandon the task, whose + cells are then skipped rather than run against missing assets. Cells are + submitted as their task becomes ready, so a later task's graph builds while + earlier cells are still paying for sessions. + """ + + with cancellation_scope(cancel_event) as cancel_event: + if workers < 1: + raise ValueError("workers must be positive") + if not cells: + return outage_streak, False + if window is None: + window = max(workers * PACKED_WINDOW_MULTIPLIER, workers) + if window < workers: + raise ValueError("window must be at least workers, or the pool starves") + + halt = threading.Event() + results: list[dict[str, Any] | None] = [None] * len(cells) + submitted: list[Any] = [] + gate = threading.Condition() + producing = True + fold_pointer = 0 + + def execute(index: int) -> None: + if halt.is_set() or cancel_event.is_set(): + return + task_id, run_idx, arm = cells[index] + on_start(task_id, run_idx, arm) + results[index] = run(task_id, run_idx, arm) + + pool = ThreadPoolExecutor(max_workers=workers) + # cancellation_scope binds _CANCELLATION in the CALLING thread's + # context, and a new thread starts with an empty one - so the producer + # has to copy this context rather than its own, or every cell it + # submits loses the run's cancellation event. sweep_task_cells gets + # this for free by submitting from the thread that entered the scope. + caller_context = copy_context() + + def produce() -> None: + nonlocal producing + ready_tasks: dict[str, bool] = {} + try: + for index, (task_id, _run_idx, _arm) in enumerate(cells): + if halt.is_set() or cancel_event.is_set(): + break + if task_id not in ready_tasks: + ready_tasks[task_id] = True if await_ready is None else await_ready(task_id) + if not ready_tasks[task_id]: + with gate: + submitted.append(None) + gate.notify_all() + continue + with gate: + while index - fold_pointer >= window and not halt.is_set(): + gate.wait(timeout=0.5) + if halt.is_set() or cancel_event.is_set(): + break + worker_context = caller_context.run(copy_context) + submitted.append(pool.submit(worker_context.run, execute, index)) + gate.notify_all() + finally: + with gate: + producing = False + gate.notify_all() + + producer = threading.Thread(target=produce, name="packed-cell-producer", daemon=False) + producer.start() + + tripped = False + try: + index = 0 + while True: + with gate: + while index >= len(submitted) and producing: + gate.wait(timeout=0.5) + if index >= len(submitted): + break + future = submitted[index] + if future is not None: + try: + future.result() + except BaseException: + # Same contract as sweep_task_cells: the cells submitted + # after this one have already run and spent their budget, + # so persist their rows in submission order before the + # harness bug takes the process down. Without this, one + # crashing cell silently erases the paid evidence of + # every sibling that had already finished. The failing + # index itself has no row - execute() only assigns on + # success - so folding forward cannot duplicate it. + with gate: + settled = list(submitted) + for later in range(index + 1, len(settled)): + pending = settled[later] + if pending is not None and not pending.done(): + continue + row = results[later] + if row is not None: + on_record(*cells[later], row) + raise + record = results[index] + if record is not None: + task_id, run_idx, arm = cells[index] + on_record(task_id, run_idx, arm, record) + kind = ( + "review-evidence-invalid" + if record.get("review_evidence_valid") is False + else record.get("error_kind") + ) + outage_streak = systemic_outage_streak(kind, outage_streak) + if outage_limit and outage_streak >= outage_limit: + print( + f"[systemic-outage] {outage_streak} consecutive unusable-evidence " + "failures — aborting the remaining sweep; report and promotion are " + "written from partial evidence and the run exits non-zero." + ) + tripped = True + halt.set() + cancel_event.set() + break + index += 1 + with gate: + fold_pointer = index + gate.notify_all() + if cancel_event.is_set(): + tripped = True + break + finally: + halt.set() + with gate: + gate.notify_all() + producer.join() + for pending in submitted[index + 1 :]: + if pending is not None: + pending.cancel() + pool.shutdown(wait=True) + return outage_streak, tripped + + @dataclass(frozen=True) class TaskCellContext: """Everything one benchmark cell needs from its task, prepared once. diff --git a/eval/workflow_bench/session_durations.json b/eval/workflow_bench/session_durations.json new file mode 100644 index 000000000..dbb5e9004 --- /dev/null +++ b/eval/workflow_bench/session_durations.json @@ -0,0 +1,33 @@ +{ + "_provenance": "Actions run 33912693948 (2026-09-04), review profile, gen-0, workers=1. Artifact gitnexus-evolution-33912693948-1: gen-0/bench/results.jsonl and gen-0/proposer-session.json. Step wall from the Actions API.", + "_caveat": "Every cell in that run returned unusable evidence (32 review-evidence-invalid, 6 session-error, 3 skill-not-invoked); two hit the 5400s ceiling and it cost 653. Durations are real, but a run that resolves cleanly may sit lower. It is the only live artifact - the 2026-07-22 green run's has expired.", + "_order": "Submission order, deliberately unsorted: the model cycles these, so sorting would hand each task a uniform block and hide the variance being measured.", + "_duration_scope": "duration_s is the sum of the cell's Claude session durations (runner_sessions.py). It excludes the clone, graph materialize, asset staging, sandbox setup and teardown - those live in the residual below.", + "session_ceiling_s": 5400, + "proposer_duration_s": 344.7, + "cell_duration_s_by_arm": { + "candidate_review": [ + 2485.6, 1338.0, 3075.2, 702.5, 1240.4, 5400.0, 762.1, 342.9, 826.3, 489.7, 675.4, 337.8, 734.3 + ], + "ce_review": [ + 3744.6, 2140.4, 1418.4, 436.1, 653.3, 1022.3, 851.2, 1191.1, 902.1, 847.2, 502.6, 991.3, + 1222.7, 544.0 + ], + "review": [ + 5400.0, 2976.4, 1162.8, 627.1, 901.3, 436.9, 704.3, 963.4, 631.8, 627.9, 663.5, 662.4, 361.3, + 741.1 + ] + }, + "residual": { + "benchmark_step_wall_s": 54623, + "session_seconds": 51737.7, + "proposer_seconds": 344.7, + "unaccounted_s": 2540.6, + "cells": 41, + "unique_shas": 5, + "_note": "Everything the sweep spent outside the agent sessions: per-SHA sanitize and `analyze --pdg --index-only`, plus each cell's clone, materialize, staging, sandbox and teardown. That run predates clone templates and graph prefetch, so this is an upper bound for the current code. The split between per-SHA and per-cell is not recoverable from the artifact, so the model charges it per cell and serially, outside the pool - the pessimistic reading of an already-small term.", + "_split_assumption": "The residual mixes per-SHA graph setup with per-cell clone/sandbox/teardown and the artifact cannot separate them. The model charges it per SHA, not per cell, because only that direction refuses to credit a run for shrinking work it still performs: a weekly generation pays one arm instead of three but builds the same graphs. This overstates cold slightly and refuses to understate weekly. Replace with measured per-SHA and per-cell times when a run records them separately.", + "sha_overhead_s": 508.1 + }, + "_breaker": "Replaying this sample's error_kind sequence through today's systemic_outage_streak trips the outage breaker at cell 5 of 41 (DEFAULT_OUTAGE_STREAK=5). The source run executed all 41, so its runner did not break on this sequence. The durations stay valid as per-cell timings; what they cannot describe is a 54-cell sweep with this failure profile, because the current code would never run one." +} diff --git a/eval/workflow_bench/simulate_sweep.py b/eval/workflow_bench/simulate_sweep.py new file mode 100644 index 000000000..764460eea --- /dev/null +++ b/eval/workflow_bench/simulate_sweep.py @@ -0,0 +1,747 @@ +#!/usr/bin/env python3 +"""Run the real sweep scheduler against stub sessions and time it. + +``measure_evolution_cost`` is arithmetic: it predicts wall clock from a model of +what ``sweep_task_cells`` does. This runs the actual function - real threads, +the real wave barrier, the real outage breaker - and replaces only the paid +agent session with a sleep. If the two disagree, the model is wrong. + +Durations are the measured per-arm samples from ``session_durations.json`` +divided by ``--scale``, so a cell that really took 1416s takes ~0.28s here. The +shape is preserved deliberately: the median cell is 826s against a 5400s +ceiling, and that spread is the whole reason a barrier costs anything. Uniform +random sleeps would erase the effect under test. + +Schedulers, all consuming one identical seeded plan: + +``wave`` the shipped ``sweep_task_cells`` - fixed waves of ``workers``, a + barrier between them, one task at a time. +``fed`` a continuously fed pool per task (H1). Naive: no breaker, no graph + gating. Present to price the barrier alone. +``packed`` one pool across every task (H2). Naive, same caveat. +``faithful``H2 carrying the invariants the shipped scheduler actually holds: + a global submission order, in-order folding, the outage breaker, and + per-task graph readiness gating. This is the one to believe. + + python3 -m workflow_bench.simulate_sweep --compare --repeat 5 + python3 -m workflow_bench.simulate_sweep --breaker-fidelity +""" + +from __future__ import annotations + +import argparse +import json +import math +import random +import statistics +import subprocess +import sys +import threading +import time +from concurrent.futures import ThreadPoolExecutor +from dataclasses import dataclass, field +from typing import Any + +from . import runner +from .measure_evolution_cost import ( + CANDIDATE_ARM, + DURATIONS_BY_ARM, + REVIEW_ARMS, + REVIEW_TASKS, + SHA_OVERHEAD_SECONDS, + _read, + expected_task_seconds, + review_tasks, +) + +DEFAULT_SCALE = 5000.0 +# --contention-sweep measures both of these regardless of --workers, so the +# window has to be valid for the LARGEST of them, not for the parsed value. +CONTENTION_WORKERS = (3, 6) +SYSTEMIC_KIND = "session-error" + +# A cell is mostly a model session waiting on the network, but its tool calls - +# git, vitest, analyze - burn real CPU in real subprocesses. Sleeping threads +# model the wait and nothing else, so every speedup measured that way is an +# upper bound. This burns WORK, not wall clock: a fixed number of sha256 rounds +# in a subprocess, which takes longer when cores are contended. That is the +# effect under test, and it has to be a subprocess - Python threads burning +# Python would measure the GIL rather than the machine. +_BURN_SRC = ( + "import hashlib,sys\n" + "n=int(sys.argv[1]); b=b'x'*4096; h=hashlib.sha256()\n" + "for _ in range(n): h.update(b)\n" + "sys.stdout.write(h.hexdigest()[:8])\n" +) + + +def calibrate_burn(probe_rounds: int = 400_000) -> float: + """sha256 rounds per second, one uncontended subprocess. Measured, not assumed.""" + + started = time.monotonic() + subprocess.run( + [sys.executable, "-c", _BURN_SRC, str(probe_rounds)], + check=True, + capture_output=True, + ) + return probe_rounds / (time.monotonic() - started) + + +def _execute_cell(cell: Cell, cpu_fraction: float, burn_rate: float) -> None: + """The stub session: wait for the API, then do the tool-call work.""" + + if cpu_fraction <= 0: + time.sleep(cell.seconds) + return + time.sleep(cell.seconds * (1.0 - cpu_fraction)) + rounds = int(cell.seconds * cpu_fraction * burn_rate) + if rounds > 0: + subprocess.run( + [sys.executable, "-c", _BURN_SRC, str(rounds)], check=True, capture_output=True + ) + + +@dataclass(frozen=True) +class Cell: + task: int + run: int + arm: str + seconds: float + systemic: bool = False + + +@dataclass +class Outcome: + wall_s: float + executed: int + tripped_at: int | None = None + folded: list[int] = field(default_factory=list) + + +def build_plan( + *, + task_count: int, + runs: int, + arms: tuple[str, ...], + scale: float, + seed: int, + fail_from: int | None = None, +) -> list[list[Cell]]: + """Per-task cells in submission order, with durations drawn once. + + Shared by every scheduler so a comparison cannot be an artifact of one of + them drawing luckier cells. ``fail_from`` marks every cell at or after that + global index systemic, which is what the breaker-fidelity mode needs. + """ + + rng = random.Random(seed) + plan: list[list[Cell]] = [] + index = 0 + for task in range(task_count): + cells: list[Cell] = [] + for run_idx in range(runs): + for arm in arms: + sample = DURATIONS_BY_ARM[arm] + cells.append( + Cell( + task=task, + run=run_idx, + arm=arm, + seconds=sample[rng.randrange(len(sample))] / scale, + systemic=fail_from is not None and index >= fail_from, + ) + ) + index += 1 + plan.append(cells) + return plan + + +def _flatten(plan: list[list[Cell]]) -> list[Cell]: + return [cell for cells in plan for cell in cells] + + +def _record(cell: Cell) -> dict[str, Any]: + kind = SYSTEMIC_KIND if cell.systemic else None + return { + "run": cell.run, + "arm": cell.arm, + "ok": not cell.systemic, + "resolved": not cell.systemic, + "error_kind": kind, + "review_evidence_valid": not cell.systemic, + } + + +def _graph_builder( + ready: list[threading.Event], graph_seconds: float, stop: threading.Event +) -> threading.Thread: + """One graph at a time, in task order - they are CPU and IO heavy.""" + + def build() -> None: + for event in ready: + if stop.is_set(): + return + time.sleep(graph_seconds) + event.set() + + thread = threading.Thread(target=build, name="graph-builder", daemon=True) + thread.start() + return thread + + +def run_wave( + plan: list[list[Cell]], + workers: int, + *, + outage_limit: int, + graph_seconds: float, + cpu_fraction: float = 0.0, + burn_rate: float = 0.0, +) -> Outcome: + """The shipped scheduler, driven for real, task after task.""" + + ready = [threading.Event() for _ in plan] + stop = threading.Event() + _graph_builder(ready, graph_seconds, stop) + executed = 0 + lock = threading.Lock() + streak = 0 + tripped_at: int | None = None + folded: list[int] = [] + base = 0 + + started = time.monotonic() + for task, cells in enumerate(plan): + ready[task].wait() + by_key = {(c.run, c.arm): c for c in cells} + + def fake_run(run_idx: int, arm: str) -> dict[str, Any]: + nonlocal executed + cell = by_key[(run_idx, arm)] + _execute_cell(cell, cpu_fraction, burn_rate) + with lock: + executed += 1 + return _record(cell) + + order = {(c.run, c.arm): base + i for i, c in enumerate(cells)} + + def on_record(run_idx: int, arm: str, rec: dict[str, Any]) -> None: + # Mirror the breaker's own evaluation so the reported trip point is + # the cell that crossed the limit, not merely the last one folded - + # sweep_task_cells folds a whole wave before it evaluates. + nonlocal streak, tripped_at + index = order[(run_idx, arm)] + folded.append(index) + streak = runner.systemic_outage_streak(rec["error_kind"], streak) + if outage_limit and streak >= outage_limit and tripped_at is None: + tripped_at = index + + streak, tripped = runner.sweep_task_cells( + [(c.run, c.arm) for c in cells], + workers=workers, + run=fake_run, + on_start=lambda *_: None, + on_record=on_record, + outage_streak=streak, + outage_limit=outage_limit, + ) + base += len(cells) + if tripped: + break + stop.set() + return Outcome(wall_s=time.monotonic() - started, executed=executed, tripped_at=tripped_at, folded=folded) + + +def _drain_naive(cells: list[Cell], workers: int) -> int: + with ThreadPoolExecutor(max_workers=workers) as pool: + list(pool.map(lambda c: time.sleep(c.seconds), cells)) + return len(cells) + + +def run_fed(plan: list[list[Cell]], workers: int, *, outage_limit: int, graph_seconds: float) -> Outcome: + """H1 without invariants: fed pool per task. Prices the barrier alone. + + Graph building is deliberately identical to ``run_wave`` - the same + background builder, started before the clock - because that is what makes + the claim in the first line true. Sleeping ``graph_seconds`` serially before + each task instead, as this did, charged fed for overlap that wave gets for + free: the wave builder prepares task N+1 while task N's cells run. The + fed-versus-wave delta then mixed the loss of that overlap into what was + reported as the price of the barrier. + """ + + ready = [threading.Event() for _ in plan] + stop = threading.Event() + _graph_builder(ready, graph_seconds, stop) + executed = 0 + started = time.monotonic() + for task, cells in enumerate(plan): + ready[task].wait() + executed += _drain_naive(cells, workers) + return Outcome(wall_s=time.monotonic() - started, executed=executed) + + +def run_packed(plan: list[list[Cell]], workers: int, *, outage_limit: int, graph_seconds: float) -> Outcome: + """H2 without invariants. Upper bound, not a design.""" + + started = time.monotonic() + time.sleep(graph_seconds) + executed = _drain_naive(_flatten(plan), workers) + return Outcome(wall_s=time.monotonic() - started, executed=executed) + + +def run_faithful( + plan: list[list[Cell]], + workers: int, + *, + outage_limit: int, + graph_seconds: float, + window: int | None = None, + cpu_fraction: float = 0.0, + burn_rate: float = 0.0, +) -> Outcome: + """H2 carrying the invariants the shipped scheduler holds. + + Global submission order is task-major, run-major, arm-minor - the same total + order the wave scheduler folds in, just continued across task boundaries. A + folder walks results in exactly that order, so "consecutive systemic + failures" keeps its meaning; the breaker trips on the same logical cell it + would have in waves. Cells already in flight when it trips are the overrun, + bounded by ``workers - 1`` exactly as the wave docstring promises. + + A task's cells are not submitted until its graph is ready, which is what + makes this a schedule rather than a wish: the graph builder is serial, so + packing cannot outrun it. + + ``window`` is the design question. Queue every cell at once and workers race + far ahead of the fold pointer, so a breaker trip has already paid for cells + nobody has looked at - measured at 5 against a bound of 2. Holding + submission to ``window`` cells beyond the fold point caps the overrun at + ``window - 1``, which is the wave's own ``workers - 1`` bound when the two + are equal, while still packing across task boundaries. Defaults to whatever + ``runner.sweep_packed_cells`` defaults to, so a run that names no window + compares the shipped policy rather than a more tightly queued prototype. + """ + + if window is None: + window = max(workers * runner.PACKED_WINDOW_MULTIPLIER, workers) + if window < workers: + # Same rule sweep_packed_cells enforces. Without it a window below 1 + # never lets the producer past its own gate and the run hangs. + raise ValueError("window must be at least workers, or the pool starves") + + cells = _flatten(plan) + ready = [threading.Event() for _ in plan] + stop = threading.Event() + _graph_builder(ready, graph_seconds, stop) + + results: list[dict[str, Any] | None] = [None] * len(cells) + executed = 0 + lock = threading.Lock() + halt = threading.Event() + + def work(index: int) -> None: + nonlocal executed + if halt.is_set(): + return + cell = cells[index] + _execute_cell(cell, cpu_fraction, burn_rate) + with lock: + executed += 1 + results[index] = _record(cell) + + gate = threading.Condition() + fold_pointer = 0 + futures: list[Any] = [] + producer_done = threading.Event() + + started = time.monotonic() + pool = ThreadPoolExecutor(max_workers=workers) + + def produce() -> None: + submitted = 0 + for task, task_cells in enumerate(plan): + ready[task].wait() + for _ in task_cells: + with gate: + while submitted - fold_pointer >= window and not halt.is_set(): + gate.wait(timeout=0.5) + if halt.is_set(): + producer_done.set() + return + futures.append(pool.submit(work, submitted)) + submitted += 1 + gate.notify_all() + producer_done.set() + + producer = threading.Thread(target=produce, name="cell-producer", daemon=True) + producer.start() + + streak = 0 + tripped_at: int | None = None + folded: list[int] = [] + try: + index = 0 + while True: + with gate: + while index >= len(futures) and not producer_done.is_set(): + gate.wait(timeout=0.5) + if index >= len(futures): + break + future = futures[index] + future.result() + record = results[index] + if record is not None: + folded.append(index) + streak = runner.systemic_outage_streak(record["error_kind"], streak) + if outage_limit and streak >= outage_limit: + tripped_at = index + halt.set() + with gate: + gate.notify_all() + for pending in futures[index + 1 :]: + pending.cancel() + break + index += 1 + with gate: + fold_pointer = index + gate.notify_all() + finally: + halt.set() + with gate: + gate.notify_all() + stop.set() + producer.join(timeout=5) + pool.shutdown(wait=True) + return Outcome( + wall_s=time.monotonic() - started, executed=executed, tripped_at=tripped_at, folded=folded + ) + + +def run_production_packed( + plan: list[list[Cell]], + workers: int, + *, + outage_limit: int, + graph_seconds: float, + cpu_fraction: float = 0.0, + burn_rate: float = 0.0, + window: int | None = None, +) -> Outcome: + """Drive the REAL runner.sweep_packed_cells, not a prototype of it. + + Same relationship run_wave has to sweep_task_cells: only the paid session is + stubbed. If this disagrees with the faithful prototype, the shipped function + is what is wrong. + """ + + cells = _flatten(plan) + by_key = {(f"t{c.task}", c.run, c.arm): c for c in cells} + order = {(f"t{c.task}", c.run, c.arm): i for i, c in enumerate(cells)} + ready = [threading.Event() for _ in plan] + stop = threading.Event() + _graph_builder(ready, graph_seconds, stop) + + executed = 0 + lock = threading.Lock() + folded: list[int] = [] + tripped_at: int | None = None + streak_seen = {"streak": 0} + + def run_cell(task_id: str, run_idx: int, arm: str) -> dict[str, Any]: + nonlocal executed + cell = by_key[(task_id, run_idx, arm)] + _execute_cell(cell, cpu_fraction, burn_rate) + with lock: + executed += 1 + return _record(cell) + + def on_record(task_id: str, run_idx: int, arm: str, rec: dict[str, Any]) -> None: + nonlocal tripped_at + index = order[(task_id, run_idx, arm)] + folded.append(index) + streak_seen["streak"] = runner.systemic_outage_streak(rec["error_kind"], streak_seen["streak"]) + if outage_limit and streak_seen["streak"] >= outage_limit and tripped_at is None: + tripped_at = index + + def await_ready(task_id: str) -> bool: + ready[int(task_id[1:])].wait() + return True + + started = time.monotonic() + runner.sweep_packed_cells( + [(f"t{c.task}", c.run, c.arm) for c in cells], + workers=workers, + run=run_cell, + on_start=lambda *_: None, + on_record=on_record, + outage_streak=0, + outage_limit=outage_limit, + window=window, + await_ready=await_ready, + ) + wall = time.monotonic() - started + stop.set() + return Outcome(wall_s=wall, executed=executed, tripped_at=tripped_at, folded=folded) + + +SCHEDULERS = { + "wave": run_wave, + "fed": run_fed, + "packed": run_packed, + "faithful": run_faithful, + "production": run_production_packed, +} + + +def _window_kwargs(name: str, window: int | None) -> dict[str, int]: + """``--window`` only means anything to the two schedulers that hold one.""" + + return {"window": window} if window is not None and name in ("faithful", "production") else {} + + +def _plan_args(args: argparse.Namespace, weekly: bool, seed: int, fail_from: int | None = None): + arms = (CANDIDATE_ARM,) if weekly else REVIEW_ARMS + return { + "task_count": len(review_tasks(_read(REVIEW_TASKS))), + "runs": args.runs, + "arms": arms, + "scale": args.scale, + "seed": seed, + "fail_from": fail_from, + }, arms + + +def breaker_fidelity(args: argparse.Namespace) -> list[dict[str, Any]]: + """Does packing still trip where waves trip, and overrun no further?""" + + rows: list[dict[str, Any]] = [] + limit = runner.DEFAULT_OUTAGE_STREAK + window = args.window if args.window is not None else max( + args.workers * runner.PACKED_WINDOW_MULTIPLIER, args.workers + ) + for fail_from in (0, 4, 12): + kwargs, _arms = _plan_args(args, weekly=False, seed=args.seed, fail_from=fail_from) + plan = build_plan(**kwargs) + total = sum(len(c) for c in plan) + row: dict[str, Any] = { + "fail_from": fail_from, "limit": limit, "total_cells": total, "window": window + } + for name in ("wave", "faithful", "production"): + out = SCHEDULERS[name]( + plan, args.workers, outage_limit=limit, graph_seconds=args.graph_seconds, + **_window_kwargs(name, window), + ) + row[name] = { + "tripped_at": out.tripped_at, + "executed": out.executed, + "overrun": out.executed - (out.tripped_at + 1) if out.tripped_at is not None else None, + } + row["same_trip_point"] = ( + row["wave"]["tripped_at"] == row["faithful"]["tripped_at"] == row["production"]["tripped_at"] + ) + # The producer holds submission to ``window`` cells beyond the fold + # pointer, so at most ``window - 1`` cells past the tripping one can + # already be in flight. At ``window == workers`` that is exactly the + # wave scheduler's own ``workers - 1`` bound. + row["overrun_within_bound"] = ( + row["production"]["overrun"] is not None + and row["production"]["overrun"] <= window - 1 + ) + rows.append(row) + return rows + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--workers", type=int, default=3) + parser.add_argument("--scale", type=float, default=DEFAULT_SCALE) + parser.add_argument("--seed", type=int, default=1729) + parser.add_argument("--repeat", type=int, default=1) + parser.add_argument("--runs", type=int, default=3) + parser.add_argument("--scheduler", choices=sorted(SCHEDULERS), default="wave") + parser.add_argument("--compare", action="store_true") + parser.add_argument("--breaker-fidelity", action="store_true") + parser.add_argument("--window-sweep", action="store_true", help="wall clock vs breaker overrun") + parser.add_argument("--contention-sweep", action="store_true", help="does the gain survive real CPU?") + parser.add_argument( + "--window", + type=int, + default=None, + help="submission window for the packed schedulers; defaults to the shipped policy", + ) + parser.add_argument( + "--graph-seconds", + type=float, + default=None, + help="per-task graph build; defaults to the measured per-SHA overhead, scaled", + ) + args = parser.parse_args() + # Both are checked here rather than where they are used: a bad --scale + # divides by zero before anything runs, and a negative --graph-seconds + # kills the graph-builder thread, after which every scheduler waits on a + # readiness event nobody will ever set. + # NaN defeats every comparison it appears in, so "> 0" and ">= 0" both admit + # it and the failure surfaces far from the flag: NaN durations reach + # time.sleep in a worker or the graph thread and raise there, after which the + # schedulers wait forever on a readiness event nobody will set. Infinity is + # worse than a crash - it silently scales every duration to zero and the run + # reports a sweep that took no time. + if not math.isfinite(args.scale) or args.scale <= 0: + parser.error("--scale must be a finite positive number") + if args.graph_seconds is not None and (not math.isfinite(args.graph_seconds) or args.graph_seconds < 0): + parser.error("--graph-seconds must be a finite non-negative number") + # Counts are indexed or handed to a thread pool without further checking, so + # a zero turns into an IndexError on plans[0], a median over an empty + # sequence, or ThreadPoolExecutor's own error - none of which name the flag + # that caused them. + if args.workers < 1: + parser.error("--workers must be at least 1") + if args.repeat < 1: + parser.error("--repeat must be at least 1") + if args.runs < 1: + parser.error("--runs must be at least 1") + # run_faithful and sweep_packed_cells both refuse a window below the worker + # count - a smaller one starves the pool, because the producer waits for a + # fold pointer to pass a cell it was never allowed to submit. Enforcing it + # here turns an uncaught ValueError partway through a measurement into an + # argument error before anything runs. Checked against the largest worker + # count this invocation will actually use: --contention-sweep runs its own + # counts irrespective of --workers, so validating against --workers alone + # let the 3-worker measurements finish and then raised on the 6-worker one. + window_workers = args.workers + if args.contention_sweep: + window_workers = max(window_workers, max(CONTENTION_WORKERS)) + if args.window is not None and args.window < window_workers: + parser.error(f"--window must be at least the worker count ({window_workers}); a smaller window starves the pool") + if args.graph_seconds is None: + args.graph_seconds = SHA_OVERHEAD_SECONDS / args.scale + + if args.contention_sweep: + burn_rate = statistics.median(calibrate_burn() for _ in range(3)) + rows = [] + for cpu_fraction in (0.0, 0.25, 0.5): + for workers in CONTENTION_WORKERS: + plans = [ + build_plan(**_plan_args(args, False, args.seed + i)[0]) + for i in range(args.repeat) + ] + measured = {} + for name in ("wave", "faithful", "production"): + fn = SCHEDULERS[name] + measured[name] = statistics.median( + fn( + plan, + workers, + outage_limit=0, + graph_seconds=args.graph_seconds, + cpu_fraction=cpu_fraction, + burn_rate=burn_rate, + **_window_kwargs(name, args.window), + ).wall_s + for plan in plans + ) + serial = statistics.median( + sum(c.seconds for c in _flatten(plan)) for plan in plans + ) + rows.append( + { + "cpu_fraction": cpu_fraction, + "workers": workers, + "wave_s": round(measured["wave"], 2), + "faithful_s": round(measured["faithful"], 2), + "production_s": round(measured["production"], 2), + "packing_gain_pct": round( + (measured["faithful"] - measured["wave"]) / measured["wave"] * 100, 1 + ), + "wave_speedup": round(serial / measured["wave"], 2), + "faithful_speedup": round(serial / measured["faithful"], 2), + } + ) + print(json.dumps({"burn_rate": round(burn_rate), "nproc": __import__("os").cpu_count(), "rows": rows}, indent=2)) + return 0 + + if args.window_sweep: + total = len(review_tasks(_read(REVIEW_TASKS))) * args.runs * len(REVIEW_ARMS) + rows = [] + for window in (args.workers, args.workers * 2, args.workers * 4, total): + clean = [build_plan(**_plan_args(args, False, args.seed + i)[0]) for i in range(args.repeat)] + wall = statistics.median( + run_faithful( + p, args.workers, outage_limit=0, graph_seconds=args.graph_seconds, window=window + ).wall_s + for p in clean + ) + failing = build_plan(**_plan_args(args, weekly=False, seed=args.seed, fail_from=12)[0]) + trip = run_faithful( + failing, + args.workers, + outage_limit=runner.DEFAULT_OUTAGE_STREAK, + graph_seconds=args.graph_seconds, + window=window, + ) + rows.append( + { + "window": window, + "cold_wall_s": round(wall, 3), + "tripped_at": trip.tripped_at, + "executed": trip.executed, + "overrun_cells": trip.executed - (trip.tripped_at + 1) + if trip.tripped_at is not None + else None, + } + ) + print(json.dumps({"workers": args.workers, "rows": rows}, indent=2)) + return 0 + + if args.breaker_fidelity: + print( + json.dumps( + {"workers": args.workers, "graph_seconds": round(args.graph_seconds, 4), + "rows": breaker_fidelity(args)}, + indent=2, + ) + ) + return 0 + + names = sorted(SCHEDULERS) if args.compare else [args.scheduler] + rows: list[dict[str, Any]] = [] + for label, weekly in (("weekly", True), ("cold", False)): + plans = [] + for i in range(args.repeat): + kwargs, arms = _plan_args(args, weekly, args.seed + i) + plans.append(build_plan(**kwargs)) + serial = statistics.median(sum(c.seconds for c in _flatten(p)) for p in plans) + predicted = ( + len(plans[0]) + * expected_task_seconds(args.runs, arms, args.workers, fed_pool=False) + / args.scale + ) + for name in names: + observed = statistics.median( + SCHEDULERS[name]( + p, + args.workers, + outage_limit=0, + graph_seconds=args.graph_seconds, + **_window_kwargs(name, args.window), + ).wall_s + for p in plans + ) + rows.append( + { + "profile": label, + "scheduler": name, + "workers": args.workers, + "observed_s": round(observed, 3), + "wave_model_s": round(predicted, 3), + "serial_s": round(serial, 3), + "speedup_vs_serial": round(serial / observed, 3) if observed else None, + } + ) + print(json.dumps({"scale": args.scale, "repeat": args.repeat, "rows": rows}, indent=2)) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) From bddbb0ff9f04e49b1bd1d1b80c50688a20c1e75c Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Gerg=C5=91=20Magyar?= Date: Tue, 8 Sep 2026 09:14:47 +0100 Subject: [PATCH 10/17] test(cli): prove detached refresh by ordering, not by wall clock (#3221) `lets the parent exit without waiting for a detached refresh child` bet twice on absolute wall-clock budgets and lost both bets on loaded CI runners: - `expect(elapsed).toBeLessThan(1_800)` bounded the *parent's* cold `node --import tsx` boot, inferring "did not wait" from a 1050ms margin over the mock fetch's 750ms sleep. Reproduced failing on Linux under 40-way load at 1904ms. - The 15s poll for the cache file had to cover the whole detached child: node boot, a tsx transpile of 22 source files, an `acquireFileLock` that shells out to `ps` (POSIX) or `powershell.exe -Command Get-CimInstance Win32_Process` (Windows), the mocked fetch, and the atomic write. That chain measures ~1.0s locally but has no bounded upper limit on a contended runner, and it is what timed out in CI. Replace both budgets with an ordering proof. The preloaded mock fetch now parks the refresh child until the test releases it, so the sequence asserted is: parent exited -> child provably still mid-refresh (started marker present, cache absent) -> release -> cache written. That is strictly stronger than the old elapsed-time inference, and it holds at any machine speed. The remaining `expect.poll` timeouts no longer carry the assertion's meaning; they are only "is the child dead" safety nets. The mock's wait is bounded at 60s so an abandoned child (test failed before releasing, temp home already removed) still exits instead of spinning forever. Co-authored-by: Gergo Magyar Co-authored-by: Claude Opus 5 (1M context) --- .../integration/cli/update-notice.test.ts | 30 ++++++++++++++----- 1 file changed, 23 insertions(+), 7 deletions(-) diff --git a/gitnexus/test/integration/cli/update-notice.test.ts b/gitnexus/test/integration/cli/update-notice.test.ts index abebfbd28..0887d1c46 100644 --- a/gitnexus/test/integration/cli/update-notice.test.ts +++ b/gitnexus/test/integration/cli/update-notice.test.ts @@ -127,12 +127,26 @@ describe('CLI update notice subprocess behavior', () => { 'dir', ); + // The refresh child parks inside fetch() until this test releases it. That + // orders the parent's exit against work that is provably still in flight, + // instead of racing it against a wall-clock budget: the child's real cost + // (node boot, tsx transpile, a lock acquisition that shells out to + // ps/powershell) has no bounded upper limit on a loaded CI runner. + const started = path.join(home, 'refresh-started'); + const release = path.join(home, 'refresh-release'); const preload = path.join(home, 'mock-refresh.mjs'); fs.writeFileSync( preload, - `Object.defineProperty(process.stderr, 'isTTY', { value: true, configurable: true }); + `import fs from 'node:fs'; +Object.defineProperty(process.stderr, 'isTTY', { value: true, configurable: true }); globalThis.fetch = async () => { - await new Promise((resolve) => setTimeout(resolve, 750)); + fs.writeFileSync(${JSON.stringify(started)}, ''); + // Bounded so an abandoned child (test failed before releasing, temp home + // already deleted) still exits instead of spinning forever. + const deadline = Date.now() + 60_000; + while (!fs.existsSync(${JSON.stringify(release)}) && Date.now() < deadline) { + await new Promise((resolve) => setTimeout(resolve, 25)); + } return new Response(JSON.stringify({ version: '99.0.0' }), { status: 200, headers: { 'content-type': 'application/json' }, @@ -141,7 +155,6 @@ globalThis.fetch = async () => { `, ); - const startedAt = Date.now(); await new Promise((resolve, reject) => { const parent = spawn( process.execPath, @@ -162,17 +175,20 @@ globalThis.fetch = async () => { else reject(new Error(`notifier parent exited ${String(code)}`)); }); }); - const elapsed = Date.now() - startedAt; - expect(elapsed).toBeLessThan(1_800); + // The parent already exited above, so reaching a still-parked child proves + // the refresh outlived it and was never awaited. const cache = path.join(home, 'update-check.json'); + await expect.poll(() => fs.existsSync(started), { timeout: 30_000, interval: 50 }).toBe(true); expect(fs.existsSync(cache)).toBe(false); - await expect.poll(() => fs.existsSync(cache), { timeout: 15_000, interval: 100 }).toBe(true); + + fs.writeFileSync(release, ''); + await expect.poll(() => fs.existsSync(cache), { timeout: 30_000, interval: 50 }).toBe(true); expect(JSON.parse(fs.readFileSync(cache, 'utf8'))).toMatchObject({ latestVersion: '99.0.0', registry: 'https://registry.npmjs.org', }); - }, 20_000); + }, 90_000); it('prints the localized notice on a forced-TTY stderr and keeps stdout clean', () => { const home = tempHome(); From a4e70ec3b4738958eac34a352cb6f61470dfab3f Mon Sep 17 00:00:00 2001 From: cosark <121065588+cosark@users.noreply.github.com> Date: Tue, 8 Sep 2026 01:23:52 -0700 Subject: [PATCH 11/17] docs: add RepoCloud one-click deploy button (#3212) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Co-authored-by: cosark Co-authored-by: Gergő Magyar --- README.md | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/README.md b/README.md index 99782975e..26a03588e 100644 --- a/README.md +++ b/README.md @@ -102,6 +102,10 @@ The proxy strips `Origin` before forwarding, so the server's CSRF guard does not Indexing is memory-bound. If `gitnexus-server` runs out of memory on a large repo, raise its `plan`, which sets available RAM: `standard` is 2 GB, `pro` is 4 GB. Raise `sizeGB` only if the disk fills with clones and indexes. +### Deploy to RepoCloud + +[![Deploy on RepoCloud](https://d16t0pc4846x52.cloudfront.net/deploylobe.svg)](https://repocloud.io/details/gitnexus/) + ## Two Ways to Use GitNexus | | **CLI + MCP** (recommended) | **Web UI** | From b1d87c1f33d765910418ce76e641274c14a3616d Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Gerg=C5=91=20Magyar?= Date: Tue, 8 Sep 2026 10:25:45 +0100 Subject: [PATCH 12/17] fix(eval): sweep evidence handling and measurement health, with guarded comparator reuse (#3207) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(eval): cut skill-evolution wall clock without shrinking the gate Reuse matching incumbent/CE cells, sanitize each SHA once, and default dispatch workers to 3 so weekly review generations finish inside the EventBridge window. Cap the sweep from leftover instance uptime so a Friday dispatch still uploads evidence. Co-authored-by: Cursor Co-Authored-By: Claude Opus 5 (1M context) * perf(eval): pipeline graph setup and correct the wall-clock cost model The evolution sweep paid `sanitize` + `analyze --pdg --index-only` for every unique task SHA on the critical path, one at a time, with nothing overlapping. `_run_sweep` now starts the next unpaid SHA's clone template and graph snapshot on a prefetch thread as soon as the current task's cells are dispatched, so every SHA but the first hides behind a paid session wave. The thread is joined before that SHA is used and before the trees tempdir is torn down, and a prefetch failure is recorded against the SHA exactly as an inline failure is. Tasks whose cells are all reusable comparator rows are not prefetched: they never build a graph, so priming one would be pure cost. Adds `measure_evolution_cost.py`, the cost model behind these numbers. It reads the review corpus, the evolve defaults, and the workflow's workers default — it does not start a session. Its first version charged `copy_isolated_tree` once per paid cell, serially. `run_cell` clones inside its own pool worker, so the clones in a wave overlap and only one is on the critical path per wave; the model now charges `ceil(cells / workers)` waves. Estimated review generation at workers=3: cold 21570s, weekly 7710s. Wall clock is quantised by `ceil(cells_per_task / workers)`. A cold review task is 9 cells, so workers=4 buys the wall clock of workers=3 and pays host contention for it. Documented in the workflow's rollout checklist. Co-Authored-By: Claude Opus 5 (1M context) Co-Authored-By: Claude Opus 5 (1M context) * perf(eval): price the benchmark against measured cell durations The cost model assumed every cell runs the 1140s mean. Cells are not uniform: the 41 rows in Actions run 33912693948's artifact are 826s at the median, 1262s at the mean, 2976s at p90, with two pinned at the 5400s session ceiling. A wave waits for its slowest cell, so a mean understates every concurrent schedule — the previous model called workers=3 cold 5.99h when the same schedule against real durations is 10.33h. session_durations.json carries the sample in submission order with its provenance and its caveat: every cell in that run returned unusable evidence, so the durations are real but a clean run may sit lower. It is the only live artifact; the 2026-07-22 green run's has expired. The model now simulates the schedule cell by cell rather than multiplying a mean by a wave count, averaged over all 41 rotations of the sample so no single alignment between sample order and cell index decides the answer. It prices today's barrier (wave_makespan) against a continuously fed pool (fed_makespan) and reports both, and it charges the proposer session — one per generation, measured at 344.7s — which it had been omitting entirely. Measurement only; no runtime behaviour changes. Co-Authored-By: Claude Opus 5 (1M context) Co-Authored-By: Claude Opus 5 (1M context) * perf(eval): price arms separately and stop inventing setup constants Two errors in the model, both found by auditing it against the artifact it claims to describe. The arms are not interchangeable. `candidate_review` runs 1416s at the mean against `review`'s 1204s and `ce_review`'s 1176s, and the weekly lane pays the candidate arm and nothing else — reuse skips both incumbents. Pricing weekly from a pooled sample charged it for arms it never runs: weekly is 4.59h, not the 3.65h a pooled sample reported. Cells are also submitted run-major and arm-minor, so at workers=3 every wave holds one cell of each arm and the slowest arm sets the wave; the model now builds cells in that order. The setup constants were invented. GRAPH_ANALYZE_SECONDS=600 and TEMPLATE_SANITIZE_SECONDS=180 charged 3900s of per-SHA setup for a cold run — more than the entire non-session time of the source run, which was 2541s for 41 cells and 5 SHAs. `duration_s` is the sum of a cell's Claude sessions (runner_sessions.py), so that 2541s residual is every clone, graph build, sandbox and teardown the sweep paid. The model now charges the measured residual per cell, 62.0s, and no longer credits clone templates or graph prefetch: both landed after that run and there is no measurement of them yet. The residual bounds what they can be worth. Cold 37452s (10.40h), weekly 16541s (4.59h), against a fed pool at 31683s and 16541s. Measurement only; no runtime behaviour changes. Co-Authored-By: Claude Opus 5 (1M context) Co-Authored-By: Claude Opus 5 (1M context) * perf(eval): charge sweep overhead where more workers cannot dissolve it Three defects, found by auditing the model against the artifact again. The overhead was charged inside the schedule. session_durations.json claimed the residual was charged "per cell and serially - the pessimistic reading", but task_cells folded it into each cell's duration, where the pool then divided it by the worker count. The residual mixes per-cell work the pool really does divide with per-SHA graph setup it cannot, and the artifact cannot separate them, so it now sits outside the schedule: cold 11.09h, not 10.40h. Alignment averaging weighted the shortest sample twice. The arm samples are 13, 14 and 14 long and the average ran over max()=14 offsets, so candidate_review's first cell was counted twice and its last never. Averaging over lcm()=182 offsets weights every arm's sample evenly. The wall assumed all 54 cells run. Replaying the sample's own error_kind sequence through today's systemic_outage_streak trips the outage breaker at cell 5 of 41. The source run executed all 41, so its runner did not break on that sequence, but the current one would: these numbers price a HEALTHY sweep, and a sweep with the sample's failure profile never reaches them. Stated on generation_seconds and recorded next to the sample it qualifies. Measurement only; no runtime behaviour changes. Co-Authored-By: Claude Opus 5 (1M context) Co-Authored-By: Claude Opus 5 (1M context) * fix(eval): give the review agent somewhere it can actually write Every review cell in the last recorded generation returned unusable evidence. Not some — all 41, across all three arms and all six tasks, at $3653 for the run. The transcripts say why, 127 times across 35 of 35 sessions: EROFS: read-only file system, open '/workspace/review-output.json.tmp.2.90a76e583b0c' The review arm mounted the artifact as a writable FILE at /workspace/review-output.json while binding /workspace read-only. The Write tool writes atomically: it creates `.tmp..` beside the target and renames it. The parent was read-only, so the temp create failed and the artifact was never written. A writable file inside a read-only directory is not writable to anything that writes atomically. Agents tried /proc/self/root/workspace/... and /proc/1/root/workspace/... to get around it; all 41 artifacts came back 0 bytes. The artifact now lives in its own writable directory bound at /review-output, outside the workspace. That is what a rename needs, and it lets the workspace get stricter rather than looser: the review phase may now change nothing there at all (enforce_phase_workspace gained allowed_artifact=None), where before it was entitled to one path inside it. The file is no longer pre-created — the agent writes it, and absence is now meaningful evidence. parse_review_output reported every one of these as "review output is not valid UTF-8 JSON". The file was empty, and its except folded OSError, UnicodeError and JSONDecodeError into that one string, so a sandbox that made writing impossible was indistinguishable from an encoding fault. That is why this read as an agent-quality problem for fifteen consecutive non-green runs. Each cause now names itself: never written, empty, not valid UTF-8, not valid JSON with the decoder's position. run_arm also keeps the FIRST error_detail, as it already did for error_kind, so a phase-boundary violation is no longer buried under the parse failure it causes. The test double conflated sandbox.private_root with the clone, which put the artifact directory inside the workspace and would have hidden the stricter check. Regression tests pin the mount shape in the generated bwrap argv, the contract path in the prompt, the four parse diagnostics, and the untouched-workspace contract. Verified by unit tests only: this container has unprivileged user namespaces disabled, so bwrap cannot run here and the mount was not exercised end to end. Co-Authored-By: Claude Opus 5 (1M context) Co-Authored-By: Claude Opus 5 (1M context) * fix(review): close the artifact path in every layer that gates it Code review of this branch found the relocated review artifact was fixed in the bwrap mount and nowhere else. Four independent layers decide whether the agent can write it, and three still named the old location. Claude Code applies its own filesystem policy to its own tools, and build_claude_settings listed only /workspace, /tmp and /home/agent under allowWrite with denyRead ["/"]. The artifact used to live under /workspace, so this list was correct until it moved. SANDBOX_REVIEW_OUTPUT is now in allowWrite and allowRead; without it the bwrap bind grants a write the CLI then refuses. The task corpus still ran `test -s review-output.json` from the workspace, in a separate sandbox invocation that never sees the artifact mount. Every review cell would have been stamped verify-failed with resolved=False no matter how good the review was, which also made those rows permanently unreusable and so silently disabled this branch's own comparator reuse for review arms. The verify and hidden-oracle commands now read the location from GITNEXUS_BENCH_REVIEW_OUTPUT and get the directory bound read-only, mirroring the mount-plus-env-var shape _run_hidden_oracle already used. host_text and host_path did not translate the new path, so the host-unsafe backend told the agent to write somewhere that exists on neither backend. Adding the mapping exposed a second defect: host_text substituted every occurrence of a target, and "/review-output" appears twice in "/review-output/review-output.json" - once as the directory and once inside the filename. Matching is now anchored to a path boundary. Comparator reuse had three ways to accept evidence it should have rejected. A row with no runtime_digest passed the drift lock because the guard only compared when both sides were bound, and the branch's own test asserted that as correct; absence is now a mismatch and the test states the rule. materialize_reused_row overwrote recorded_at with the copy time while the age check read that field, so a row copied forward each generation refreshed its own clock and never aged out; the first measurement time is now preserved and aged against. A future-dated stamp passed a one-sided bound and is now rejected as corrupt. RUNTIME_DIGEST never reached the runner at all: runner_environment builds a fixed dict and process_control replaces the child environment wholesale, so the digest the workflow exports was dropped and the lock it feeds was inert. The instance-window deadline was also checked only after run_proposer returned, buying a proposal the generation had no room to benchmark. The graph prefetch thread was started without copy_context, so it never saw the cancellation ContextVar the rest of the sweep shares, and the outage breaker returned without setting cancel_event - together, a tripped breaker would block on joining a prefetch that was never told to stop. Both fixed, with outage checked before cancellation at the two exits so an outage keeps exit 1 instead of becoming a Ctrl-C's 130. Both bwrap canaries that actually execute a write still bound the pre-fix shape against a file this branch no longer creates, so they would have errored rather than caught anything. They now bind the directory and write atomically - temp file beside the target, then rename - which is the exact operation that failed with EROFS. A source-text assertion over inspect.getsource(run_arm) was replaced with one that inspects the real mount, and a wall-clock assertion was pinned to a fixed monotonic clock. Not applied, and why: binding task-asset and dependency digests into comparator reuse needs asset snapshots prepared before the reuse decision rather than inside the per-task loop, and shipping the comparison without that would add a guard that silently never fires. Forcing a paid canary cell per incumbent arm and folding reused rows into the outage streak are behaviour decisions, not fixes. Clone-template reuse still has no test. The cost model's per-cell residual still shrinks with arm count, overstating weekly savings by at most the 2541s residual; the docstring now says so rather than inventing a split. 585 eval tests pass, ruff clean, 29 workflow contract tests pass. The two test_model_gateway.py failures are pre-existing and fail on main. * fix(review): bind reuse to its environment and keep the health canary real Applies the five findings the previous review round left open. Comparator reuse ignored the environment a row was measured in. TaskReuseBinding carried the task and oracle identity but not the task-asset or sandbox-dependency digests, and this branch itself changes sandbox_dependencies in the review corpus - so a reused comparator could be measured against one dependency set and compared against a candidate built on another, handing the gate a false comparison. Closing it needed the digests to exist before the reuse decision, so asset snapshots are now prepared for every task up front instead of lazily inside the per-task loop. That also removes the concurrent TaskAssetCache.prepare the prefetch thread could otherwise race, which the file's own "plain dict, read-then-write race" comment warned about. Both digests fail closed on either side, matching the runtime digest. The broken-incumbent canary could not fire when reuse was working. It read `resolved`, which counts reused rows, so an arm whose cells were all reused always looked healthy - in precisely the run where a broken environment would go unnoticed. aggregate now also reports `resolved_fresh` and the canary reads it. That count would be vacuous if an arm were reused end to end, so the sweep keeps one paid cell per incumbent arm and says which one it kept. Reused rows did not participate in the outage streak, so a run of failures could carry across them and trip on stale history. A reused success now resets the streak the way a paid success does. The cost model charged sweep overhead per cell, which credited a weekly generation for shrinking work it still performs: it pays one arm instead of three but builds exactly the same graphs. Overhead is charged per SHA now. Weekly is 5.20h rather than the 4.80h the per-cell rate reported; cold is 10.86h. The residual still cannot be split between per-SHA and per-cell work from one artifact, so session_durations.json records that assumption and the direction it errs in, rather than leaving a number nobody can trace. Clone-template reuse - the branch's core speedup, taken on essentially every multi-cell sweep - now has a test that builds a real sanitized template, asserts the cell runs against the copy with the template's HEAD, and fails if run_cell re-clones. A second test asserting only on a namespace built inside the test was written and deleted: it exercised nothing, which is the failure this review round penalised elsewhere. 589 eval tests pass, ruff clean, 29 workflow contract tests pass. The two test_model_gateway.py failures are pre-existing and fail on main. * refactor(eval): consolidate duplicated harness logic after the review round Simplification pass over the branch. Behavior-preserving throughout; three reviewers, nine findings applied, two skipped. The review-artifact block in _run_hidden_oracle was unreachable. That function runs only in run_arm's non-review branch, while the directory it probes for is created only in the review branch, and each sandbox serves exactly one arm - so `review_artifact.parent.is_dir()` could never be true. It was added an hour earlier to make the hidden oracle resolve the moved artifact; the oracle never runs for review tasks, so the guard was dead on arrival. Deleting it also removes the duplication it had with the verify-command wiring. EXCLUDED_ERROR_KINDS is now one definition. runner.py and comparator_reuse.py each carried the same six-member frozenset, kept in sync by a comment. Only one direction is possible: runner already imports from comparator_reuse, so the reverse import fails at module-init with a circular-import error. That is now stated where the alias lives, so nobody tries it the other way. ensure_task_graph and prefetch_next_graph shared ten keyword parameters, passed through two call sites and forwarded whole between them. They now take a GraphBuildEnv, mirroring TaskCellContext, which already bundles per-cell state in this file. Its ready_keys() replaces an inline four-set union at the call site. Smaller consolidations: _sha256_file's hand-rolled chunk loop becomes hashlib.file_digest (3.11+, already used in runner_artifacts); _copy_owner_only reuses task_assets._write_all and COPY_CHUNK_BYTES instead of repeating the short-write retry; its stat-then-open existence check becomes the O_EXCL failure it was already relying on, which is atomic rather than merely narrow; and runner_environment reads the digest through comparator_reuse.current_runtime_digest instead of re-parsing the environment variable. Three test docstrings summarised the branch's own history ("the branch's core speedup", "the regression that produced fifteen runs") rather than the invariant under test. Rewritten to state the constraint, which is what survives the merge. Repaired the indentation left behind by the outage-streak edit and flattened the prefetch dispatch from three nested conditionals to one. Skipped: consolidating comparator_reuse._real_directory onto proposer_sandbox's same-named helper - they differ, the sandbox one rejects any symlink in the resolved path while this one checks only the leaf, so sharing it would tighten behavior rather than preserve it. That needs a decision about which policy the reuse path wants, not a simplification. 589 eval tests pass, ruff clean, 29 workflow contract tests pass. Unrelated and pre-existing: two test_model_gateway.py failures, and test_process_control.py::test_timeout_kills_term_ignoring_descendants_before_they_write, which is a TERM-to-KILL timing flake (passes 2 of 3 in isolation) in a file this branch does not touch. * refactor(eval): name the reuse directory check for the promise it makes The simplification pass left one finding open: comparator_reuse and proposer_sandbox both defined `_real_directory`, same name and same shape, with different guarantees. The sandbox one rejects every symlink hop in the path; the reuse one checks only the leaf and resolves through parents. Sharing the name invites a consolidation that would silently tighten one of them. They should not be merged, so the name stops claiming they could be. proposer_sandbox guards a mount root, where a symlink hop changes what an untrusted session is handed. comparator_reuse guards a data directory whose contents are already validated one file at a time - reads go through _regular_file, which lstats and rejects symlinks, and writes through O_NOFOLLOW. A symlinked parent therefore grants nothing those guards do not already cover, while refusing one would reject a symlinked artifacts directory or macOS's /var for no gain. Renamed to _resolved_directory, with the reasoning recorded at the definition, and a test that pins both halves: a symlinked parent is accepted and resolved, a symlinked leaf is still refused. Behavior is unchanged. 591 eval tests pass, ruff clean. The two test_model_gateway.py failures are pre-existing and fail on main. * test(eval): measure the sweep scheduler instead of modelling it measure_evolution_cost predicts wall clock from a model of what sweep_task_cells does. This runs the real thing - real threads, the real wave barrier, the real outage breaker - with only the paid agent session replaced by a sleep, and times it. Durations are the measured per-arm samples divided by 5000, so a 1416s cell takes ~0.28s. The shape is kept on purpose: the median cell is 826s against a 5400s ceiling, and that spread is the entire reason a barrier costs anything. Uniform random sleeps would erase the effect under test. All schedulers consume one identical seeded plan, so a comparison cannot be an artifact of one of them drawing luckier cells. The model survives contact: it tracks real execution within about 10%, and workers=1 - which runs without a pool at all - sits at 0.95, so the residual above 1.0 at higher worker counts is per-wave thread overhead rather than a modelling error. Two structural claims that were arithmetic are now observed. Weekly is flat from workers=3: 3.59, 3.59, 3.59, 3.60, 3.59, 3.59 across w=3..8. workers=4 buys nothing over workers=3 on cold, 7.68 against 7.78. Two prototype schedulers are measured beside it, deliberately before any production code exists. A continuously fed pool per task is worth more than the model claimed on cold, -27.3% against a predicted -17.9%, and exactly nothing on weekly, +0.0%, because a weekly task is one wave with nothing to feed. One pool across all tasks beats both: -40.7% weekly and -42.9% cold at workers=3, rising to -65.7% and -63.9% at workers=8. It also subsumes the fed pool, since packing across tasks is a fed pool. That reorders the backlog. Cross-task packing moves from second to first: it dominates on both profiles, and it is the only thing that moves weekly at all. Raising the worker count is worth nothing until it lands - under the barrier weekly does not improve from w=3 to w=8, and speedup against serial is 1.58x for three workers and only 2.40x for eight. The bound on all of it: sleeping threads do not contend. Real sandboxed sessions compete for CPU, page cache and disk, and the duration sample was itself measured at workers=1, so it carries no contention either. These speedups are upper bounds. The ordering is trustworthy because the schedulers were compared under identical conditions; the magnitudes are not. The packed prototype is also a bare ThreadPoolExecutor with no breaker folding, no per-task graph lifecycle and no reuse binding - which is the actual cost of building it, and is not measured here. * test(eval): carry the sweep invariants into the packed prototype The first packed prototype was a bare ThreadPoolExecutor. It reported -43% and none of the invariants the shipped scheduler holds, so it priced an idea nobody could ship. This one carries them: a global submission order continued across task boundaries, in-order folding, the real outage breaker, and per-task graph readiness gating behind a serial builder. The fidelity check first reported the two schedulers tripping on different cells, 17 against 16. That was my instrumentation, not a divergence - sweep_task_cells folds an entire wave before it evaluates the breaker, so the last cell folded is not the cell that tripped. With the harness mirroring the breaker's own evaluation the two agree exactly, across failures starting at cell 0, 4 and 12, with overrun inside the workers-1 bound the wave docstring promises. Two results worth the exercise. Head-of-line blocking, not the barrier, is what a naive in-order design pays. Holding submission to `workers` cells beyond the fold pointer leaves the faithful scheduler at -8.1% cold and -2.7% weekly: one slow cell stalls the pointer, the window cannot slide, and it reproduces the wave almost exactly. That is the number to quote if anyone proposes the obvious implementation. But the overrun bound turns out to be set by the worker count, not the window. Only `workers` cells can be running when the breaker trips; everything queued behind them short-circuits on the halt flag. Overrun is 3 at an unbounded window exactly as at 6, and the trip cell never moves off 16. So H2 does not have to trade breaker fidelity for speed - a wide window takes -42% with the semantics intact. The tension I assumed was there is not, and window=12 already captures 97% of it. Still an upper bound: sleeping threads do not contend, and the sample was measured at workers=1. What this establishes is that the invariants are affordable, which was the thing blocking H2. Not built here: the trees tempdir lifecycle, reuse-row binding, and the cancel_event path. 591 eval tests pass, ruff clean. * test(eval): put the scheduler comparison under real CPU contention Every Phase 2 number so far came from sleeping threads, which contend for nothing, against a duration sample measured at workers=1, which contains no contention either. That was the standing caveat on the whole result, so this measures it. A cell now waits for its API share and then burns a fixed number of sha256 rounds in a subprocess. Work-bounded rather than wall-clock bounded, so it takes longer when cores are busy - that is the effect under test. A subprocess because Python threads burning Python would measure the GIL rather than the machine. Calibrated at 519k rounds/s, stable within 2% across three probes. The first run of this was worthless and is recorded as such: on a 24-core host with 3 to 6 workers nothing ever contends, since cpu_fraction 0.5 at 6 workers is about 3 cores of demand out of 24. It measured an absence. Re-run pinned with taskset to 4 and 2 cores. The packing advantage survives. It holds between -40% and -47% across every host size and CPU fraction tested, including a genuinely oversubscribed 2-core box at cpu_fraction 0.5 with 6 workers. But contention erodes packing more than it erodes waves, for a structural reason: packing is what creates the concurrency. Moving from 24 cores to 2 at cpu 0.5 and 6 workers, the faithful scheduler slows 13% while the wave slows 3.7%, and the gain narrows from 45.0% to 39.8%. Packing and a higher worker count are therefore not independent wins - packing spends the contention headroom first, so raising workers has to be re-argued after it lands rather than added to it. Three things this still does not measure, and they bound the result. The real CPU fraction of a benchmark cell is a guess informed by roughly 180 tool calls per session; nobody has profiled one. The evolution runner's core count decides which column applies and is unknown here. And the burn is sha256, pure CPU, while real cells run vitest and analyze, which are memory and IO heavy - so this is a floor on contention, not a ceiling. 591 eval tests pass, ruff clean. * perf(eval): add a packed sweep scheduler, and correct the bound I claimed for it sweep_task_cells finishes one task before starting the next and drains a wave before refilling it, so a task with fewer cells than workers leaves workers idle and one slow cell stalls its whole wave. sweep_packed_cells feeds every task's cells through a single pool instead. Measured against the review corpus it is worth about 40% of a cold sweep, and it is the only change that moves a seeded weekly run at all - there a task is three cells and a wave is never full. The breaker keeps its exact meaning. Cells carry a total submission order continued across task boundaries, a folder walks results in that order, and consecutive systemic failures are counted there, so a doomed run aborts on the same cell it would have under waves. Verified at three failure positions. This commit also corrects a finding from the Phase 2 prototype. I claimed the overrun bound was set by the worker count rather than the submission window, and that packing therefore cost nothing in breaker fidelity. That was derived from a window sweep that only ever injected failures at one position. Driving the real function at other positions shows the halt flag does not bound overrun at all: the folder walks in order, so a slow early cell lets workers race ahead and the trip is detected after those cells have already paid. An unbounded queue overran by 11 cells where waves overrun by 2. So the window is load-bearing and the trade is real, measured at workers=3 with failures injected at four positions: window 3 -> -8% wall, overrun 2 (the wave scheduler's own bound) window 6 -> -27% wall, overrun 4 window 12 -> -42% wall, overrun 9 window 54 -> -44% wall, overrun 11 Overrun is wasted paid sessions at roughly $70 each. The default multiplier is 2, keeping the worst case within twice the wave bound while taking most of the gain; the curve is in the constant's comment so raising it is an informed decision rather than a guess. Not wired in yet: _run_sweep still calls sweep_task_cells per task. Moving the per-task graph, trees tempdir and reuse binding out of that loop behind await_ready is the larger and riskier half, and it belongs in its own change. 595 eval tests pass, ruff clean. * fix(eval): judge harness health on execution, not on how many tasks resolved broken_incumbent_arms infers "the environment is broken" from an arm resolving zero tasks. That inference does not hold: a reviewer can be wrong about every task in a hard corpus while every process, mount and capture worked perfectly. Actions run 33962002890 is exactly that shape - 51 cells, all resolved=False with error_kind=oracle-failed, median score 0.212, and a healthy harness. Someone already knew this, and patched it by excluding review arms at the call site. That leaves the unsound inference in place for workflow and workflow_direct, and leaves review arms with no health check at all - so the run that genuinely was broken, 33912693948, where the mount made an atomic write impossible and all 41 artifacts came back empty, could not have been caught here either. So this replaces the inference rather than adding another exemption. aggregate now classifies fresh rows into execution failures (the process or its tooling did not complete), evidence failures (it completed but produced nothing trustworthy or scoreable), and admissible measurements. An arm is unhealthy only when it has fresh attempts, zero admissible measurements, and at least one execution or evidence failure. Resolution count is no longer consulted. Arms with only reused rows report current health as UNKNOWN rather than good. With the inference corrected, review arms are checked again, which is what lets the empty-artifact case be caught at all. Deliberately unchanged: comparator reuse eligibility, quality denominators, promotion thresholds, model settings, skill prompts and scheduler behaviour. Failures that stop being called infrastructure failures still surface in the counts and reasons - an agent-originated failure must not vanish from reporting because it was reclassified. broken_incumbent_arms and its tests are left in place; deleting behaviour belongs in its own change. Seven regression tests, built from both runs' shapes and labelled as reconstructed from logged observations, since 33962002890's results.jsonl did not survive the instance shutdown. They pin: a badly-scoring reviewer is healthy; an all-zero score is still a valid negative; empty artifacts are unhealthy; one admissible cell keeps an arm healthy while its failures stay visible; reused rows alone leave health unknown; reused successes do not mask fresh failures; and a parseable artifact does not excuse a failed session. 602 eval tests pass, ruff clean. * fix(eval): pin the health guard below the breaker, and stop calling mixed runs healthy Two corrections to the health-classification patch. The regression I wrote could not have proved what it claimed. A fixture of 41 empty artifacts aborts through the outage breaker long before finalization: review-evidence-invalid is in SYSTEMIC_ERROR_KINDS and the limit is 5, so it trips at cell 5 through the pre-existing path. It demonstrated failure detection, not the new guard. The decisive test now uses ONE fresh unusable cell, asserts the streak stays under the breaker threshold, and only then requires finalization to abort - leaving the new check as the only thing that can catch it. Removing the call makes that test fail; restoring it passes. The accurate defect statement is narrower than the last message claimed. Review arms were excluded from the final incumbent-health check while the consecutive- failure breaker gave them separate, partial coverage. They were not unguarded. Second: "one admissible cell plus two execution failures" was asserted as healthy. That converts "not wholly unusable" into "ran reliably", which is how a partly-broken sweep passes review. Arms now report UNKNOWN, OBSERVED_OK, DEGRADED or UNUSABLE. Only UNUSABLE is fatal, so eligibility and promotion are untouched - this changes what is reported, not what is allowed. The guard is extracted as enforce_measurement_health so it can be driven directly, and it now reports a status line per arm. It names no cause: an empty artifact establishes that evidence is unusable, not that a mount rejected the write, so it prints cause=undetermined rather than guessing EROFS. It still runs after report.md and promotion.json are written, so a failing sweep leaves its evidence behind. ce_review is named explicitly at the call site. It is a comparator rather than a candidate, so it is absent from CANDIDATE_ARMS.values(), and dropping the review exclusion alone would have left it unclassified. The wiring test reads _run_sweep's compiled code object for the referenced global rather than matching source text. It is honest about its limit: it proves the call exists and would catch its removal, but no test here drives _run_sweep end to end, which needs bwrap and a sandbox. broken_incumbent_arms is marked LEGACY and NON-AUTHORITATIVE with removal tracked. It has no production caller. 608 eval tests pass, ruff clean. The two test_model_gateway.py failures are test_locked_litellm_translates_messages_to_offline_responses and test_openai_gateway_never_leaves_proxy_output_on_an_undrained_pipe; both fail identically on origin/main in this environment, checked directly rather than carried forward as an inherited label. * fix(eval): review artifact path, evidence classification, comparator reuse Extracted from the combined skill-evolution branch. This is the runtime change set: everything that alters how a sweep executes and what it records. The packed scheduler and its measurement harness were separated onto perf/skill-evolution-packed-scheduler, which is purely additive. Correctness. The review artifact was mounted as a writable FILE inside a read-only workspace while the agent's Write tool writes atomically - temp file beside the target, then rename - so the temp create failed EROFS and the artifact was never written. Four layers gate that path and three named the old location: the CLI's own allowWrite/allowRead policy, the task corpus verify command run in its own sandbox invocation, and host_text/host_path for the host-unsafe backend. Fixing the translator exposed a second defect, since "/review-output" appears twice in "/review-output/review-output.json"; matching is now anchored to a path boundary. parse_review_output folded OSError, UnicodeError and JSONDecodeError into one message, so an artifact that was never written looked like an encoding fault; each cause now names itself. Health classification. broken_incumbent_arms inferred a broken environment from an arm resolving zero tasks, which a reviewer facing a hard corpus falsifies - Actions run 33962002890 is exactly that shape. Arms are now classified from fresh execution and evidence outcomes as UNKNOWN, OBSERVED_OK, DEGRADED or UNUSABLE, and only UNUSABLE aborts. Resolution count is not consulted. The guard names no cause: an empty artifact establishes unusable evidence, not that a mount rejected the write. Comparator reuse. Reuse accepted evidence it should have rejected: a row without a runtime_digest passed the drift lock, recorded_at was overwritten with the copy time so a row could outlive its own max_age, and the binding ignored task-asset and dependency digests although this change alters sandbox_dependencies in the review corpus. Closing the last one required preparing asset snapshots before the reuse decision, which also removes the concurrent TaskAssetCache.prepare the prefetch thread could race. These three concerns share aggregate() and _run_sweep, which is why they ship together: separating them further would mean hunk-level surgery on a function all three modify, and the risk of a silent omission outweighs the reviewability gain. 592 eval tests pass at this base. The two test_model_gateway.py failures, test_locked_litellm_translates_messages_to_offline_responses and test_openai_gateway_never_leaves_proxy_output_on_an_undrained_pipe, fail identically on origin/main in this environment. Known gap, and the reason this is not ready to merge: no test drives _run_sweep end to end. enforce_measurement_health is unit-tested including the below-breaker unusable case, and the caller wiring is pinned structurally by reading _run_sweep's compiled code object, but interruption semantics, exit precedence and persisted artifacts are not exercised through the real path. * fix(eval): address PR review feedback (#3207) - aggregate: count admissible rows directly instead of subtracting the execution and evidence counters, which double-charged a row that is both a session error and invalid review evidence and could report UNUSABLE for an arm holding real measurements. - run_proposer: bound the session timeout by what is left of --max-runtime-seconds, so clearing the sweep minimum cannot start a full-length session past the instance window. - comparator reuse: hold one O_NOFOLLOW descriptor for the size check, digest and copy, and prove it is the inode that was checked, closing the swap window a concurrent writer of the reuse directory had. - Drive the review-artifact mount assertion through run_arm and the clone-template assertion through run_cell, instead of rebuilding the expected values in the tests (also removes the CodeQL unnecessary lambda). - Assert the workflow invokes run-evolution.sh rather than that its YAML mentions --max-runtime-seconds, which only appears in a comment. - Correct the parse_review_output failure-mode claim: the fold was empty artifacts reported as "not valid UTF-8 JSON"; a never-created file raised FileNotFoundError. - prettier: wrap the over-long readFileSync call flagged by PR autofix. Note: pre-existing failure in tests/test_model_gateway.py::test_locked_litellm_translates_messages_to_offline_responses (local LiteLLM proxy never becomes ready in this environment) not addressed by this PR. Co-Authored-By: Claude Opus 5 (1M context) * fix(eval): address the second round of PR review feedback (#3207) - Refuse a symlinked `transcripts` component on both sides of comparator reuse. O_NOFOLLOW guards the leaf only, so a link there redirected the read or the copy out of the results directory; checked per component as evolution._require_directory_chain does. - Base the paid incumbent canary on the cells this sweep PLANS. Reuse selection accepts any prior run index, so a results directory produced with more runs left extra keys, the equality never held, and the canary stopped firing. Extracted as drop_canary_reuse_key and unit-tested. - Start the runtime clock in main(). --max-runtime-seconds is measured from /proc/uptime before exec, so parsing, task I/O, preflight and gateway setup were being handed back to the sweep out of the upload reserve. - Do not fall back to shutil.copytree when the managed clone copy was cancelled or timed out; that fallback is for a filesystem that cannot reflink, and copytree cannot be cancelled. - Assert the review session's writable mount, not only the verifier's read-only one: the EROFS bug is about the agent's write. - Exercise ref isolation in the copy_isolated_tree test rather than comparing an initial HEAD a shared namespace would also match. - Point the stale-symlink fixture at the sentinel via os.path.relpath, and skip the reuse symlink tests where symlink creation needs privilege. Co-Authored-By: Claude Opus 5 (1M context) * fix(eval): close the runtime-cap gap and pin the reuse directory Both were left open on #3207 as approach decisions rather than nits. Runtime cap: run-evolution.sh computed the budget in its own `uv run python -c` and passed a number, so the script's remaining provenance work and the CLI's own startup were spent by nobody and charged to the sweep — out of the upload reserve the cap exists to protect. The script now passes --max-runtime-from-instance-window and evolve reads /proc/uptime itself, on the line after it starts the clock the budget is measured against, so no interval exists to lose. Also removes an interpreter start from the script and lets --dry-run print the real argv. Reuse directory: _real_child_directory lstat-checked `transcripts` and returned its pathname, so a concurrent writer could rename the directory and leave a symlink before the name was used again — O_NOFOLLOW guards only the leaf. Every artifact is now resolved against a held descriptor: _open_real_directory opens with O_DIRECTORY|O_NOFOLLOW (check and open in one syscall), and _open_regular / _copy_owner_only take dir_fd. The reuse path is therefore POSIX-only; _require_openat says so and fails closed, which the runner already treats as "run a paid cell". _resolved_directory still tolerates a symlinked reuse root, unchanged and still tested. evolution._require_directory_chain is still lstat-per-component. It guards a different surface (candidate overlay reads) that neither review raised, so it is left alone rather than widened into here. Co-Authored-By: Claude Opus 5 (1M context) * fix(eval): sample the proposer budget where it is spent, digest what is copied Four findings against f0cdc9e7, all of them mine. - run_proposer took a precomputed remaining_seconds, but it clones, sanitizes and builds a sandbox before the session starts, so the caller's reading was already stale. It takes started_monotonic now and samples the budget on the last line before run_claude. The caller's earlier reading still decides whether to start at all — it just no longer decides how long to allow. - _copy_transcript_artifact hashed the source and then read it again to copy it. A held descriptor stops the pathname being substituted, not the inode being rewritten, so the row could record the expected digest while the destination held other bytes. _copy_owner_only now digests the same buffers it writes and returns (digest, bytes); a mismatch unlinks the destination and raises. One read instead of two. - Explain the empty except in _open_real_directory: an existing directory is the ordinary case, and the O_DIRECTORY|O_NOFOLLOW open below is what proves what it is (CodeQL). - Drop the unused uptime fixture from the runtime-cap test, and correct the cross-file contract described in the workflow-contract test and the workflow YAML: neither names --max-runtime-from-instance-window, the script does. Co-Authored-By: Claude Opus 5 (1M context) * test(eval): close the reuse producer/consumer contract through the real run_cell Every comparator-reuse test built its rows by hand. That proves the predicate's logic and nothing about the producer: a fixture can satisfy eligibility while a row the runner actually emits never does, and a key-name inventory cannot tell the difference. I checked that inventory first - all 25 keys the predicate reads have a producer - which is exactly why it was not sufficient evidence. This carries one record through the production path instead: real run_cell -> production JSONL writer -> load_result_rows -> row_is_reusable_comparator Only the expensive dependencies are replaced: the model session, sandbox launch, repository acquisition, graph preparation, git plumbing. The digest fields the reuse binding compares are assembled by run_cell itself from its TaskCellContext, so they stay real - they are the subject, not scaffolding. Expectations are built from the sweep's own configuration rather than copied out of the emitted row, since copying them back would make producer and consumer agree because the test arranged it. Two cases: a production-emitted row with matching bindings is eligible, and the same evidence with one dependency binding changed is rejected. A test-controlled digest set keeps both deterministic. Mutation-checked. Replacing run_cell's task_asset_manifest_digest assignment with None makes the positive case fail; restoring it passes. That is evidence the inventory audit could not produce. Building it also documented what a real review row must carry, which no fixture had recorded: a scored review with review_weighted_f1 present, and at least one transcript artifact whose source is PARENT_EVENT_STREAM_SOURCE. An empty artifact list is not reusable. Those are production contracts my first double got wrong, and the reader rejected it each time. No production code changed. 607 eval tests pass, ruff clean. * test(eval): drive the real sweep to its finalization decision enforce_measurement_health was unit-tested and its call site pinned by reading _run_sweep's compiled code object. Neither showed the guard running inside a sweep. These drive the real _run_sweep with cell execution scripted and everything downstream left alone: folding, aggregation, the artifact writers, the health guard and the exit selection. The below-breaker case is the decisive one. A fixture of many unusable cells aborts through the pre-existing outage breaker instead - review-evidence-invalid is systemic with a limit of five - and would pass whether or not the finalization guard exists. One fresh unusable cell stays under that threshold, so only the guard can catch it. The test asserts the arm and status the guard reports, and that no systemic-outage line was printed, rather than accepting any SystemExit: an unrelated setup failure must not satisfy it. Mutation-checked at the runtime path. Removing the guard invocation fails the below-breaker test because the expected finalization behaviour disappears, not because a name went missing from a code object. A zero score is covered separately. review_weighted_f1 = 0.0 must stay a present valid negative measurement; a truthiness check would read it as absent and turn a quality result into an execution-health failure. The third test asserts the evidence survives - results.jsonl carries the scored row and report.md exists - so the persisted artifacts tell the same story as the exit. Only expensive setup is replaced: cell execution, graph preparation, asset snapshots, and task-binding resolution, which clones the repository and verifies the ref. Building the fixture also documented that the report renders the whole review metric set, so an incomplete row fails in formatting rather than logic. Still open on this track: interruption semantics and exit-reason precedence (outage 1 before cancellation 130) are not yet exercised, and fold order and the real sandbox mount contract remain on separate tracks. No production code changed. 610 eval tests pass, ruff clean. The two test_model_gateway.py failures reproduce on origin/main in this environment. * fix(eval): an interrupted sweep is interrupted, not aborted Driving the real _run_sweep to its exit revealed that cancellation without an outage exits 1 and writes "Sweep aborted: partial evidence", where the contract is 130 and "Sweep cancelled". sweep_task_cells returns (streak, tripped), and its caller assigns that flag to outage_tripped and turns it into exit 1. Both cancellation paths returned True for it. The breaker's own return is also True, so the two became indistinguishable one frame up and cancellation inherited the outage's exit and wording. The fix is to return False from the cancellation paths: the flag means the breaker tripped, and the caller already tests cancel_event itself for the stop decision, so nothing stops running any later than before. Reproduced before the fix, through the real caller, not from reading. The failing assertion was the report line; the exit code was 1. Two runtime tests cover it. Each begins with one admissible cell so the arm classifies DEGRADED rather than UNUSABLE - otherwise enforce_measurement_health supplies exit 1 first and a precedence test passes without ever reaching exit selection. Cancellation is set from a completed cell rather than a sleep or a real signal, so the interruption point is deterministic. The precedence test is mutation-checked: moving the cancel_event check ahead of the outage check fails it with "assert 130 == 1" - it observes the wrong exit code, not merely some failure. My first attempt at that mutation silently matched nothing and the suite passed; a no-op mutation proves nothing, so the edit now asserts its own anchor. test_process_control read the same flag as "stopped" and asserted it after a cancellation. It now reports the event and the flag separately, which is what it was really asserting: cancelled, and not an outage. The scored-row shape moved into tests/bench_fixtures.py with zero values written out, so the finalization and interruption tests share one definition of what a real review row carries. Impact analysis returned risk UNKNOWN - the index predates this branch and does not carry the eval harness - so the callers were confirmed by text search as the rules require for UNKNOWN: one production caller and three test sites, all updated or verified. detect_changes reports zero for the same reason; that zero is unseen, not unaffected. 612 eval tests pass, ruff clean. The two test_model_gateway.py failures reproduce on origin/main in this environment. * test(eval): prove the review artifact mount under a real sandbox The orchestration tests replace the sandbox, so they say nothing about isolation. This covers the filesystem contract the EROFS defect actually broke, through the production configuration: the same command_prefix_for call run_arm makes for a review cell, with read_only_workspace=True and the review-output directory as the writable mount - not a hand-built mount tuple that merely looks right. A deterministic writer stands in for the agent. It writes a temp file beside the destination and renames it into place, which is the operation that failed: an atomic write needs a WRITABLE PARENT DIRECTORY, and binding the file itself left nowhere to put the temp file. It then attempts a workspace write and must be refused, and the production parse_review_output reads the bytes the sandbox left behind. No model session and no credentials. Placed in test_proposer_sandbox.py, which the "eval / containment (ubuntu)" job already runs with GITNEXUS_REQUIRE_BWRAP_CANARY=1. That gate is the point: where the variable is set, a missing namespace capability FAILS the job instead of skipping into a green tick. Verified here - forcing the variable on this machine fails with "bwrap: No permissions to create new namespace" rather than skipping. What this does not establish: the assertions have never executed. This machine cannot create user namespaces, so the test stops at the preflight. Imports, signatures and the payload were checked statically instead - parse_review_output returns a tuple rather than an object, and it rejects a body without "verdict", so both of my first attempts were wrong and are fixed. The first CI run on a namespace-capable runner is what will actually confirm it. Scope is the filesystem and process contract only. A stand-in writer does not establish that a particular agent CLI's own file-access policy permits the same operation; that is a second, independent gate. 40 passed, 10 skipped locally; ruff clean. * fix(eval): read the review artifact before the sandbox deletes it The containment (ubuntu) job has been red since before the mount canary was added, and both failures have the same cause. This branch moved the review artifact out of the clone and into the session's private root, which is the point of the change: the agent writes atomically, so the artifact needs a writable parent DIRECTORY outside the read-only workspace. prepare_sandbox removes that private root in a finally on scope exit. Two tests read the artifact AFTER the with block, so they were asserting against a directory the sandbox had already deleted - FileNotFoundError, reported as "review output was never written". Production is not affected, and this is the reason: run_arm holds the live session and reads the artifact inside that scope, both to score it and to mount it read-only into the verifier's separate sandbox invocation. The tests were the only readers outside it. Both now read inside the scope. The pre-existing canary (test_read_only_review_workspace_exposes_only_one_writable_artifact) predates this PR and passes on main, where the artifact still lived in the clone and survived teardown; the relocation is what broke it, so the fix belongs here. The new canary found this independently and agreed on the cause, which is what it was written for - though only after CI ran it, since this machine cannot create user namespaces. 40 passed, 10 skipped locally; ruff clean. The bwrap-gated tests still skip here and remain unverified until the ubuntu job runs them. * test(eval): pin what an interrupted sweep persists and refuses to promote The completed-run persistence test could not show either of these: it never interrupts, so it would pass even if the writers only ran on the clean path. Evidence already paid for survives cancellation. The cell that completed before the interruption is still in results.jsonl with its measurement intact, and report.md is still written. Losing those rows would mean paying for evidence the sweep then discards. An interrupted run emits nothing that authorizes promotion. Asserted as the semantic condition rather than the absence of a file: promotion.json IS still written for an aborted run - it is the record of why nothing was promoted - so the test requires run_status "aborted" and every decision reduced to insufficient_evidence with a partial-evidence reason, rather than requiring the artifact to disappear. Mutation-checked. Forcing complete=True at the promotion_evidence call site fails the test with "assert 'complete' == 'aborted'" - it observes an aborted run claiming completeness, not merely some failure. These were the last two unasserted items on this PR's finalization checklist. 614 eval tests pass, ruff clean. The two test_model_gateway.py failures are environmental (litellm[proxy] console script absent) and predate this branch. * Address PR review feedback (#3207) An exhausted runtime cap no longer buys a second. remaining_runtime_seconds floors at 0 and the proposer call site wrapped it in max(1, ...), so a cap fully spent by the clone, the sanitize pass and the sandbox build started a paid session with a one-second allowance instead of stopping before the upload reserve the cap exists to protect. It now returns a not-ok record, which the caller already treats as end-of-run. Pinned by a test driving the real run_proposer with setup that consumes the whole budget; reverting to max(1, ...) fails it. Reuse copies are bounded before their size is validated, not after. The source is a prior sweep directory this module already treats as concurrently writable, so a transcript appended to after its metadata was recorded was streamed to EOF and only then compared against its declared size - filling the destination, or never reaching EOF, long before the drift check could reject it. _copy_owner_only now takes max_bytes and stops one byte past the ceiling, which keeps the drift comparison meaningful. Review and patch artifacts carry no recorded size, but "no recorded size" is not "no limit": they get MAX_TRANSCRIPT_BYTES, the ceiling the capture path already enforces. A reused review row must carry its artifact. The predicate accepted a row on its score and transcript metadata while materialize_reused_row copied the review artifact only when the row named one, so a scored review could be carried forward with nothing for a proposer to read. The shared row fixture was the thing out of step here, not the requirement - production sets review_artifact whenever the review source exists - so it now carries one too. Test fixes, all against code this PR added: unusable_review_row could not be overridden at all. It passed explicit keywords beside **overrides, and Python rejects the duplicate in the call expression before scored_review_row can apply its update, so unusable_review_row(error_kind=...) raised TypeError. Merged into one mapping. The finalization helper monkeypatched runner.prepare_ce_plugin_snapshot with raising=False. No such symbol exists - the real one is staged_ce_plugin_snapshot - so it silently added an attribute nothing reads. Removed. The promotion test named candidate_review in candidate_arms but _run_sweep builds cells only from args.arms, so no candidate cell ran and insufficient_evidence could hold because nothing executed rather than because partial evidence is barred. The candidate arm now runs, cancellation fires after its cell, and the test asserts the candidate actually produced a row so it cannot silently return to being vacuous. The reuse round trip serialized with json.dumps and write_text while claiming to cover the production writer, which applies redact_text over the row's own bytes. It now writes the way the sweep does. Its fixture also writes the review artifact, because run_cell records review_artifact only when the review source exists and the emitted row was otherwise one production never emits. Not addressing the duplicate-match nit on proposer_sandbox.py: the claim is that "/review-output" occurs once in "/review-output/review-output.json". It occurs twice, at offsets 0 and 14 - the separator before the basename forms it again - so the boundary the comment describes is real. Verified with re.finditer. The workspace-snapshot finding is parked for a human: excluding bootstrap noise is a documented deliberate choice, and tightening it is a genuine tradeoff. 634 eval tests pass, ruff clean. The two test_model_gateway.py failures are environmental (litellm[proxy] console script absent) and predate this branch. * Address PR review feedback (#3207), round 2 Carry review_f1 in the scored-row fixture. score_review emits "f1" (review_scoring.py:317), which the runner folds in as review_f1, so a real scored row has it and the fixture did not - the same fidelity gap as the metrics already added there. aggregate now reports 0.5 for it instead of None. The reported failure mode was not real, and the distinction matters for anyone reading the thread later. aggregate filters on record.get(metric) is not None BEFORE indexing record[metric], so a missing key is skipped rather than raising KeyError; the finalization tests were passing throughout. Verified by running aggregate against the old fixture. Fixed because the fixture should match what production emits, not because anything was crashing. Drop the unused row parameter from the round-trip _expectation helper. It never read the argument - deliberately, since the docstring says the bindings must be derived from sweep configuration rather than copied out of the emitted row - but passing the row anyway was dead plumbing that suggested the opposite. The reuse-root TOCTOU finding is parked for a human rather than fixed: the lstat-then-resolve window is real, but _resolved_directory documents the weaker promise as deliberate and says not to merge it with proposer_sandbox's helper "without first deciding which promise the reuse path should make". That decision is the fix, and it is the same class as the reuse TOCTOU already parked on this PR. 634 eval tests pass, ruff clean. test_process_control's TERM-ignoring-descendant test flaked once under full-suite load and passes 3/3 in isolation; it is a timing test and neither file changed here touches it. The two test_model_gateway.py failures remain environmental. * fix(eval): pin the reuse roots to the directory that was checked Settles the reuse TOCTOU that has been parked twice on this PR. The docstring asked to decide which promise the reuse path makes before merging its helper with proposer_sandbox's, and the answer is that two separate things were being conflated. The symlink POLICY stays exactly as it was: parent hops remain allowed, so a symlinked artifacts directory or macOS's /var still works, and a symlinked leaf is still refused. Rejecting hops would break ordinary setups for no gain, which is what the docstring argued and it is still right. What is closed is the other thing - the gap between checking a name and using it. _resolved_directory lstats a name and the later open re-walks that same name, so a prior sweep that renames its results root and leaves something else behind is opened somewhere else entirely, and O_NOFOLLOW cannot see a link that resolve() already followed. _resolved_directory now returns the checked directory's identity alongside its path, and _open_pinned_root fstats the descriptor it opened and refuses a mismatch. The two are separable, so there was no trade to make: comparing identity rejects nothing that holds still, since a stable directory always matches itself. Both cases are pinned - a replacement by a different REAL directory is refused (the leaf-symlink rule does not cover that one), and an ordinary unchanged root opens normally. Why it matters here rather than as a general hardening: the failure is silent. Rows would be copied out of some other directory and folded into a comparator baseline as though they were this sweep's own evidence, which is the one thing the reuse path exists to get right - a corrupted baseline decides promotions. 636 eval tests pass, ruff clean. The two test_model_gateway.py failures remain environmental. * refactor(eval): one prompt digest, and drop two pieces of dead scaffolding Compute task_prompt_digest once. row_is_reusable_comparator compares the value a prior row stored against the value this sweep derives, as exact strings, and the two inline copies of that hash had already drifted apart - the expectation side had picked up a str() cast the row side lacked. They agree today only because tasks validate "prompt" as a string (runner_tasks.required_strings), so nothing was broken; a later divergence on one side would have silently stopped rows matching, with no test to catch it. Verified the extracted helper is byte-identical to what it replaced. Two comments named broken_incumbent_arms as the thing that would read stale health. This branch replaced that call with the measurement-health path, so the reasoning still holds but the name no longer does; they now name arm_health. The function itself stays: origin/main still calls it at runner.py:2098, and retiring it is its own change rather than a cleanup pass. Removed instance_window_budget_from_proc and its one test. It composes two helpers main() deliberately calls separately - there is a comment there explaining why the read and the budget calculation stay decoupled - and nothing outside its own test ever called it. Introduced on this branch, absent from main, so it was never deployed, public, or consumed elsewhere. Dropped an unused tmp_path parameter from a comparator-reuse test that does no filesystem work. Skipped three findings. Hoisting _assert_self_contained_git_objects to a once-per-template check is a real saving - measured at 5.66s over 1551 objects - but it verifies the OUTPUT of each copy operation, and git clone from a local path hardlinks objects by default, which is exactly what its st_nlink check catches. Checking a predecessor instead of the clone each cell runs against thins an isolation guarantee for 0.7% of a median cell. Deleting broken_incumbent_arms outright would remove a symbol main still calls. And a fourth copy of the test-only _git helper extends a pattern that already exists in three other test modules; centralising it means editing conftest.py, outside this scope. 635 eval tests pass, ruff clean. The two test_model_gateway.py failures are the environmental ones. --------- Co-authored-by: Gergo Magyar Co-authored-by: Cursor Co-authored-by: Claude Opus 5 (1M context) --- .../workflows/gitnexus-skill-evolution.yml | 27 +- eval/tests/bench_fixtures.py | 75 ++ eval/tests/test_comparator_reuse.py | 467 +++++++++ eval/tests/test_evolve.py | 302 ++++++ eval/tests/test_process_control.py | 7 +- eval/tests/test_proposer_sandbox.py | 200 +++- eval/tests/test_reuse_round_trip.py | 268 ++++++ eval/tests/test_review_corpus.py | 4 + eval/tests/test_review_scoring.py | 30 + eval/tests/test_runner_hardening.py | 101 +- eval/tests/test_sanitized_graph.py | 16 + eval/tests/test_session_progress.py | 47 +- eval/tests/test_sweep_finalization.py | 279 ++++++ eval/tests/test_workflow_bench.py | 383 ++++++++ eval/tests/test_workflow_bench_sessions.py | 143 ++- eval/workflow_bench/README.md | 53 +- eval/workflow_bench/comparator_reuse.py | 604 ++++++++++++ eval/workflow_bench/evolve.py | 265 +++++- eval/workflow_bench/proposer_sandbox.py | 86 +- eval/workflow_bench/review_scoring.py | 31 +- eval/workflow_bench/run-evolution.sh | 18 + eval/workflow_bench/runner.py | 892 +++++++++++++++--- eval/workflow_bench/runner_artifacts.py | 84 +- eval/workflow_bench/runner_sessions.py | 22 +- eval/workflow_bench/sanitized_graph.py | 23 +- .../tasks.review.scenarios.yaml | 17 +- eval/workflow_bench/tasks.scenarios.yaml | 5 + .../unit/skill-evolution-workflow.test.ts | 28 +- 28 files changed, 4217 insertions(+), 260 deletions(-) create mode 100644 eval/tests/bench_fixtures.py create mode 100644 eval/tests/test_comparator_reuse.py create mode 100644 eval/tests/test_reuse_round_trip.py create mode 100644 eval/tests/test_sweep_finalization.py create mode 100644 eval/workflow_bench/comparator_reuse.py diff --git a/.github/workflows/gitnexus-skill-evolution.yml b/.github/workflows/gitnexus-skill-evolution.yml index c004bb515..22f0bdfcc 100644 --- a/.github/workflows/gitnexus-skill-evolution.yml +++ b/.github/workflows/gitnexus-skill-evolution.yml @@ -63,13 +63,17 @@ # uploads, and a promotion (if any) opens a well-formed PR. Run # 29907431284 (2026-07-22) went green end to end in 14h45m and reached a # gate decision (`insufficient_evidence`, no promotion). -# [ ] After resizing the runner, prove a manual workers=3 run has zero excluded -# runs and does not stretch the 48-minute serial mean toward the session -# ceiling; then set GITNEXUS_EVOLUTION_WORKERS=3 and +# [ ] Confirm a workers=3 dispatch has zero excluded runs (review sessions in +# 33962002890 averaged ~19m serial, well under the 90m session ceiling). +# Then set GITNEXUS_EVOLUTION_WORKERS=3 and # GITNEXUS_EVOLUTION_ENABLED=true for scheduled runs. Scheduled runs -# require both values, so leaving workers unset/1 is an immediate rollback; -# workflow_dispatch remains available for the proof and bills real API -# usage on GITNEXUS_BENCH_ANTHROPIC_API_KEY or GITNEXUS_BENCH_OPENAI_API_KEY. +# require both values, so leaving the var unset is an immediate rollback. +# Dispatch defaults to 3; pass workers=1 only to debug a contended host. +# Weekly generations reuse matching incumbent/CE cells from the previous +# artifact so the paid matrix is the new candidate, not a 54-cell replay. +# Wall clock is quantised by ceil(cells_per_task / workers), and a review +# task is 9 cells cold, so 4 costs host contention for exactly the wall +# clock of 3. The next step up that buys anything is 5 (3 waves -> 2). name: GitNexus skill evolution on: @@ -92,9 +96,9 @@ on: default: '3' type: string workers: - description: 'Benchmark cells of one task to run at once — raise only to match the runner’s vCPUs' + description: 'Benchmark cells of one task to run at once — 3 fits the evolution box; drop to 1 only if siblings hit the session ceiling' required: false - default: '1' + default: '3' type: string model: description: 'Model for the benchmark arms (match the model your skill users run)' @@ -172,6 +176,13 @@ jobs: # stops the runner just disappears mid-step. Scheduled runs can start well # after the cron (the 2026-08-01 run was queued 65min late), so the job # budget has to absorb that delay and still land inside the uptime window. + # A Friday workflow_dispatch on a box that already booted for Saturday's + # cron inherits leftover uptime, not a fresh 24h. Run 33962002890 started + # Friday 10:57 UTC and vanished at the Saturday 03:00 stop — 51 finished + # sessions never uploaded. run-evolution.sh therefore passes + # --max-runtime-from-instance-window, and the CLI derives its cap from + # /proc/uptime at startup, so the sweep fails in-process and this always() + # upload still runs. timeout-minutes: 1260 permissions: contents: read # The promotion PR uses a short-lived App token minted below. diff --git a/eval/tests/bench_fixtures.py b/eval/tests/bench_fixtures.py new file mode 100644 index 000000000..fdd6c131f --- /dev/null +++ b/eval/tests/bench_fixtures.py @@ -0,0 +1,75 @@ +"""Shared row shapes for the sweep tests. + +Building the finalization tests turned up what a real scored review row must +carry: the report renders the whole review metric set, so an incomplete row +fails in string formatting rather than in the logic under test. That is a +property of the fixture, not of production - the shape lives here once so each +test does not rediscover it. +""" + +from __future__ import annotations + +from typing import Any + + +def scored_review_row(**overrides: Any) -> dict[str, Any]: + """One admissible review cell, with zero-valued metrics written out.""" + + row: dict[str, Any] = { + "ok": True, + "error_kind": None, + "error_detail": None, + "resolved": True, + "review_evidence_valid": True, + "review_score": {"weighted_f1": 0.5}, + "review_weighted_f1": 0.5, + "review_true_positives": 1, + "review_false_positives": 0, + "review_false_negatives": 0, + "review_precision": 0.5, + "review_recall": 0.5, + "review_f1": 0.5, + "review_weighted_precision": 0.5, + "review_weighted_recall": 0.5, + "review_blocker_recall": 1.0, + "review_severity_accuracy": 1.0, + "review_category_accuracy": 1.0, + "review_grounded_evidence": 1.0, + "review_verdict_correct": True, + "review_clean_control": True, + "review_clean_pass": True, + "transcript_missing": False, + "transcript_artifacts": [], + "num_turns": 3, + "duration_s": 1.0, + "cost_usd": 0.5, + "input_tokens": 1, + "output_tokens": 1, + "cache_creation_input_tokens": 0, + "cache_read_input_tokens": 0, + "diff_files": 0, + "diff_insertions": 0, + "diff_deletions": 0, + } + row.update(overrides) + return row + + +def unusable_review_row(**overrides: Any) -> dict[str, Any]: + """A cell that ran but produced evidence nothing can be scored from.""" + + # Merged into one mapping rather than passed as explicit keywords beside + # **overrides: Python rejects a duplicate keyword in the call expression + # itself, so unusable_review_row(error_kind=...) raised TypeError before + # scored_review_row could apply the override this helper advertises. + return scored_review_row( + **{ + "ok": False, + "resolved": False, + "review_evidence_valid": False, + "error_kind": "review-evidence-invalid", + "review_score": None, + "review_weighted_f1": None, + **overrides, + } + ) diff --git a/eval/tests/test_comparator_reuse.py b/eval/tests/test_comparator_reuse.py new file mode 100644 index 000000000..4425304b3 --- /dev/null +++ b/eval/tests/test_comparator_reuse.py @@ -0,0 +1,467 @@ +"""Comparator-row reuse: skip unchanged incumbent/CE cells, never candidates.""" + +from __future__ import annotations + +import hashlib +import os +from datetime import UTC, datetime, timedelta +from pathlib import Path + +import pytest + +from workflow_bench import comparator_reuse +from workflow_bench.comparator_reuse import ( + ComparatorReuseExpectation, + TaskReuseBinding, + materialize_reused_row, + row_is_reusable_comparator, + select_reusable_comparator_rows, +) +from workflow_bench.proposer_sandbox import SandboxError +from workflow_bench.runner_sessions import PARENT_EVENT_STREAM_SOURCE + + +requires_openat = pytest.mark.skipif( + os.open not in os.supports_dir_fd, + reason="comparator reuse resolves every artifact against a pinned directory descriptor", +) + + +def _digest(text: str = "blob") -> str: + return hashlib.sha256(text.encode()).hexdigest() + + +def _artifact(name: str = "session-1.jsonl", payload: bytes = b'{"type":"ok"}\n') -> dict: + return { + "path": f"transcripts/{name}", + "sha256": hashlib.sha256(payload).hexdigest(), + "bytes": len(payload), + "source": PARENT_EVENT_STREAM_SOURCE, + } + + +def _row(**overrides) -> dict: + base = { + "task": "review-pr-2718-defect", + "arm": "review", + "run": 0, + "ok": True, + "error_kind": None, + "model": "gpt-5.6-sol", + "benchmark_model": "gpt-5.6-sol", + "effort": "xhigh", + "sandbox_backend": "bwrap", + "task_base_sha": "a" * 40, + "task_prompt_digest": _digest("prompt"), + "oracle_digest": _digest("oracle"), + "oracle_command_digest": _digest("oracle-cmd"), + "oracle_manifest_digest": _digest("oracle-man"), + "skill_digest": _digest("skill"), + "candidate_overlay_digest": None, + "review_evidence_valid": True, + # Production sets this whenever the review source exists, which is the + # normal path for a valid review; the fixture predated the requirement. + "review_artifact": "review-pr-2718-defect-review-run0.review.json", + "review_score": {"weighted_f1": 0.4}, + "review_weighted_f1": 0.4, + "transcript_missing": False, + "transcript_artifacts": [_artifact()], + "recorded_at": datetime.now(UTC).isoformat(), + "runtime_digest": _digest("cli"), + "task_asset_manifest_digest": _digest("assets"), + "sandbox_dependency_manifest_digest": _digest("deps"), + } + base.update(overrides) + return base + + +def _expected(**overrides) -> ComparatorReuseExpectation: + now = datetime.now(UTC) + values = dict( + model="gpt-5.6-sol", + effort="xhigh", + sandbox_backend="bwrap", + runtime_digest=_digest("cli"), + now=now, + max_age=timedelta(days=90), + tasks={ + "review-pr-2718-defect": TaskReuseBinding( + task_base_sha="a" * 40, + task_prompt_digest=_digest("prompt"), + oracle_digest=_digest("oracle"), + oracle_command_digest=_digest("oracle-cmd"), + oracle_manifest_digest=_digest("oracle-man"), + task_asset_manifest_digest=_digest("assets"), + sandbox_dependency_manifest_digest=_digest("deps"), + ) + }, + skill_digests={"review": _digest("skill"), "ce_review": None}, + ce_plugin_version="3.24.0", + ce_plugin_manifest_digest=_digest("ce"), + ) + values.update(overrides) + return ComparatorReuseExpectation(**values) + + +def test_matching_incumbent_review_row_is_reusable() -> None: + assert row_is_reusable_comparator(_row(), _expected()) is True + + +def test_candidate_rows_are_never_reusable() -> None: + assert row_is_reusable_comparator(_row(arm="candidate_review"), _expected()) is False + + +def test_skill_digest_drift_rejects_reuse() -> None: + assert row_is_reusable_comparator(_row(), _expected(skill_digests={"review": _digest("other")})) is False + + +def test_excluded_or_failed_rows_are_not_reusable() -> None: + expected = _expected() + assert row_is_reusable_comparator(_row(error_kind="session-error", ok=False), expected) is False + assert row_is_reusable_comparator(_row(ok=False), expected) is False + assert row_is_reusable_comparator(_row(review_evidence_valid=False), expected) is False + assert row_is_reusable_comparator(_row(recorded_at=(datetime.now(UTC) - timedelta(days=91)).isoformat()), expected) is False + + +def test_runtime_digest_mismatch_rejects_when_both_sides_are_bound() -> None: + row = _row(runtime_digest=_digest("old-cli")) + assert row_is_reusable_comparator(row, _expected(runtime_digest=_digest("new-cli"))) is False + assert row_is_reusable_comparator(row, _expected(runtime_digest=_digest("old-cli"))) is True + # A row with no runtime_digest was measured by a harness that recorded none, + # which is the drift this lock exists to catch - not evidence of agreement. + assert row_is_reusable_comparator(_row(runtime_digest=None), _expected()) is False + # And a sweep that cannot determine its own digest must not reuse either. + assert row_is_reusable_comparator(_row(), _expected(runtime_digest=None)) is False + + +def test_ce_review_matches_plugin_digest_not_repo_skill() -> None: + row = _row( + arm="ce_review", + skill_digest=None, + ce_plugin_version="3.24.0", + ce_plugin_manifest_digest=_digest("ce"), + ) + assert row_is_reusable_comparator(row, _expected()) is True + assert ( + row_is_reusable_comparator(row, _expected(ce_plugin_manifest_digest=_digest("other"))) + is False + ) + + +def test_select_drops_conflicting_duplicates() -> None: + first = _row(review_weighted_f1=0.4) + second = _row(review_weighted_f1=0.9, recorded_at=datetime.now(UTC).isoformat()) + selected = select_reusable_comparator_rows([first, second], expected=_expected()) + assert selected == {} + same = select_reusable_comparator_rows([first, dict(first)], expected=_expected()) + assert ("review-pr-2718-defect", "review", 0) in same + + +@requires_openat +def test_materialize_copies_transcript_and_review_artifacts(tmp_path: Path) -> None: + payload = b'{"type":"result"}\n' + source = tmp_path / "prior" + dest = tmp_path / "fresh" + (source / "transcripts").mkdir(parents=True) + dest.mkdir() + transcript = source / "transcripts" / "session-1.jsonl" + transcript.write_bytes(payload) + transcript.chmod(0o600) + review = source / "review-pr-2718-defect-review-run0.review.json" + review.write_text('{"verdict":"comment"}\n') + patch = source / "review-pr-2718-defect-review-run0.patch" + patch.write_text("diff\n") + row = _row( + review_artifact=review.name, + transcript_artifacts=[_artifact(payload=payload)], + ) + + copied = materialize_reused_row(row, source_dir=source, dest_dir=dest) + + assert copied["reused"] is True + assert copied["reused_from_recorded_at"] == row["recorded_at"] + assert (dest / "transcripts" / "session-1.jsonl").read_bytes() == payload + assert (dest / review.name).read_text() == review.read_text() + assert (dest / patch.name).read_text() == "diff\n" + assert copied["transcript_artifacts"][0]["sha256"] == hashlib.sha256(payload).hexdigest() + + +@pytest.mark.skipif(os.name == "nt", reason="symlink creation may require elevated Windows privileges") +@requires_openat +def test_a_reused_artifact_is_copied_from_the_inode_that_was_checked(tmp_path: Path) -> None: + """The reuse source is a directory another sweep wrote and may still write. + + Validating a path and then re-opening it hands a concurrent writer the gap: + replace the checked file with a symlink and the copy follows it out of the + results directory. Swapping the path while the descriptor is held is that + same substitution, made deterministic. + """ + + (tmp_path / "transcript.jsonl").write_bytes(b"verified\n") + decoy = tmp_path / "decoy.jsonl" + decoy.write_bytes(b"substituted\n") + + with comparator_reuse._open_real_directory(tmp_path, label="reuse source") as dir_fd: + with comparator_reuse._open_regular("transcript.jsonl", dir_fd=dir_fd, label="transcript") as descriptor: + (tmp_path / "transcript.jsonl").unlink() + (tmp_path / "transcript.jsonl").symlink_to(decoy) + comparator_reuse._copy_owner_only(descriptor, "copy.jsonl", dir_fd=dir_fd) + + assert (tmp_path / "copy.jsonl").read_bytes() == b"verified\n" + with pytest.raises(SandboxError, match="regular non-symlink"): + with comparator_reuse._open_regular("transcript.jsonl", dir_fd=dir_fd, label="transcript"): + pass + + +@pytest.mark.skipif(os.name == "nt", reason="symlink creation may require elevated Windows privileges") +@requires_openat +def test_a_symlinked_transcripts_directory_is_refused_on_both_sides(tmp_path: Path) -> None: + """`O_NOFOLLOW` refuses the leaf, not the directory above it. + + A `transcripts` symlink on the source side makes reuse read a file outside + the results directory; one on the destination side writes the copy outside + this sweep's evidence. Neither is covered by the per-file guards that let + _resolved_directory tolerate a symlinked root. + """ + + payload = b'{"type":"result"}\n' + outside = tmp_path / "outside" + (outside / "transcripts").mkdir(parents=True) + (outside / "transcripts" / "session-1.jsonl").write_bytes(payload) + row = _row(transcript_artifacts=[_artifact(payload=payload)]) + + linked_source = tmp_path / "linked-source" + linked_source.mkdir() + (linked_source / "transcripts").symlink_to(outside / "transcripts", target_is_directory=True) + dest = tmp_path / "fresh" + dest.mkdir() + with pytest.raises(SandboxError, match="transcript source must be a real directory"): + materialize_reused_row(row, source_dir=linked_source, dest_dir=dest) + + source = tmp_path / "prior" + (source / "transcripts").mkdir(parents=True) + (source / "transcripts" / "session-1.jsonl").write_bytes(payload) + linked_dest = tmp_path / "linked-dest" + linked_dest.mkdir() + (linked_dest / "transcripts").symlink_to(outside / "transcripts", target_is_directory=True) + with pytest.raises(SandboxError, match="transcript destination must be a real directory"): + materialize_reused_row(row, source_dir=source, dest_dir=linked_dest) + + +@requires_openat +@pytest.mark.skipif(os.name == "nt", reason="symlink creation may require elevated Windows privileges") +def test_a_renamed_transcripts_directory_cannot_redirect_a_copy(tmp_path: Path) -> None: + """The directory is pinned, not re-walked from its name. + + An lstat that passed and a pathname used afterwards are two different + directories the moment a concurrent writer renames the first one away. This + performs exactly that substitution — rename, then leave a symlink in its + place — while the descriptor is held, which is what makes the race testable + without timing. + """ + + payload = b'{"type":"result"}\n' + results = tmp_path / "results" + transcripts = results / "transcripts" + transcripts.mkdir(parents=True) + (transcripts / "session-1.jsonl").write_bytes(payload) + outside = tmp_path / "outside" + outside.mkdir() + + with comparator_reuse._open_real_directory(results, label="reuse source") as root_fd: + with comparator_reuse._open_real_directory( + "transcripts", dir_fd=root_fd, label="transcript source" + ) as dir_fd: + transcripts.rename(results / "moved") + (results / "transcripts").symlink_to(outside, target_is_directory=True) + with comparator_reuse._open_regular( + "session-1.jsonl", dir_fd=dir_fd, label="transcript" + ) as artifact_fd: + comparator_reuse._copy_owner_only(artifact_fd, "copy.jsonl", dir_fd=dir_fd) + + assert (results / "moved" / "copy.jsonl").read_bytes() == payload + assert not (outside / "copy.jsonl").exists() + + +@requires_openat +def test_a_transcript_rewritten_mid_copy_is_refused_not_recorded(tmp_path: Path, monkeypatch) -> None: + """The digest has to describe the bytes that were written. + + A held descriptor stops the pathname being substituted; it does not stop the + inode being rewritten, and the prior sweep's directory is one this sweep + treats as concurrently writable. Hashing the source and then reading it + again to copy let the row keep the expected digest while the destination + held different bytes. + """ + + payload = b'{"type":"result"}\n' + source = tmp_path / "prior" + (source / "transcripts").mkdir(parents=True) + transcript = source / "transcripts" / "session-1.jsonl" + transcript.write_bytes(payload) + dest = tmp_path / "fresh" + dest.mkdir() + row = _row(transcript_artifacts=[_artifact(payload=payload)]) + + # Rewrite the inode in the window the copy reads through — same length, so + # only the digest can tell, which is the point. + real_read = comparator_reuse.os.read + rewritten = {"done": False} + + def rewrite_then_read(fd: int, size: int) -> bytes: + if not rewritten["done"]: + rewritten["done"] = True + with open(transcript, "r+b") as handle: + handle.write(b'{"type":"TAMPER"}') + return real_read(fd, size) + + monkeypatch.setattr(comparator_reuse.os, "read", rewrite_then_read) + with pytest.raises(SandboxError, match="drifted"): + materialize_reused_row(row, source_dir=source, dest_dir=dest) + monkeypatch.undo() + + # And nothing unvouched-for is left behind for the proposer to read. + assert not (dest / "transcripts" / "session-1.jsonl").exists() + + +@requires_openat +def test_materialize_rejects_same_directory_and_missing_transcript(tmp_path: Path) -> None: + source = tmp_path / "prior" + source.mkdir() + row = _row() + with pytest.raises(SandboxError, match="same results directory"): + materialize_reused_row(row, source_dir=source, dest_dir=source) + dest = tmp_path / "fresh" + dest.mkdir() + with pytest.raises(SandboxError, match="missing"): + materialize_reused_row(row, source_dir=source, dest_dir=dest) + + +@requires_openat +def test_a_reused_row_ages_from_its_first_measurement_not_the_copy(): + """Reuse chains must not refresh the clock. + + materialize_reused_row restamps recorded_at with the copy time, so aging + against that field let a row be copied forward every generation and outlive + max_age forever. The original measurement time is the one that counts. + """ + + original = (datetime.now(UTC) - timedelta(days=91)).isoformat() + chained = _row(recorded_at=datetime.now(UTC).isoformat(), reused_from_recorded_at=original) + assert row_is_reusable_comparator(chained, _expected()) is False + # The same row inside the window is still reusable. + fresh = _row( + recorded_at=datetime.now(UTC).isoformat(), + reused_from_recorded_at=(datetime.now(UTC) - timedelta(days=1)).isoformat(), + ) + assert row_is_reusable_comparator(fresh, _expected()) is True + + +def test_a_future_dated_row_is_corrupt_not_fresh(): + ahead = (datetime.now(UTC) + timedelta(days=2)).isoformat() + assert row_is_reusable_comparator(_row(recorded_at=ahead), _expected()) is False + + +def test_a_changed_sandbox_dependency_is_not_the_same_baseline(): + """The environment is part of the measurement. + + This branch itself changes `sandbox_dependencies` in the review corpus, so a + prior row measured against the old set is a measurement of a different + machine. Reusing it would compare a fresh candidate to a baseline built + somewhere else and hand the promotion gate a false comparison. + """ + + assert row_is_reusable_comparator( + _row(sandbox_dependency_manifest_digest=_digest("other-deps")), _expected() + ) is False + assert row_is_reusable_comparator( + _row(task_asset_manifest_digest=_digest("other-assets")), _expected() + ) is False + # A row that predates the field is not evidence of agreement either. + assert row_is_reusable_comparator(_row(sandbox_dependency_manifest_digest=None), _expected()) is False + + +@pytest.mark.skipif(os.name == "nt", reason="symlink creation may require elevated Windows privileges") +def test_reuse_directories_allow_a_symlinked_parent_but_not_a_symlinked_leaf(tmp_path: Path): + """Pins a deliberate difference from the sandbox's mount-root check. + + proposer_sandbox refuses every symlink hop because a hop changes what an + untrusted session is handed. A reuse directory is data, and every file + inside it is validated on its own, so a symlinked parent is allowed - + rejecting it would break a symlinked artifacts directory or macOS's /var + for no gain. The leaf itself must still be a real directory. + """ + + real = tmp_path / "real" + real.mkdir() + (real / "inner").mkdir() + linked_parent = tmp_path / "linked" + linked_parent.symlink_to(real, target_is_directory=True) + + # Reached through a symlinked parent: allowed, and resolved to the real path. + # The identity returned alongside it is what pins the root against a swap + # between the check and the open; the symlink policy itself is unchanged. + resolved, identity = comparator_reuse._resolved_directory(linked_parent / "inner", label="probe") + assert resolved == (real / "inner").resolve() + inner_stat = (real / "inner").stat() + assert identity == (inner_stat.st_dev, inner_stat.st_ino) + + # The leaf itself being a symlink is still refused. + with pytest.raises(SandboxError, match="must be a real directory"): + comparator_reuse._resolved_directory(linked_parent, label="probe") + + +def test_a_review_row_without_its_artifact_is_not_reusable() -> None: + """A score is a claim about evidence, not the evidence itself. + + materialize_reused_row copies the review artifact only when the row names + one, so accepting a row without it would carry a scored review forward with + nothing for a proposer to read. + """ + + row = _row() + assert row_is_reusable_comparator(row, _expected()) is True + without = {**row, "review_artifact": ""} + assert row_is_reusable_comparator(without, _expected()) is False + missing = {k: v for k, v in row.items() if k != "review_artifact"} + assert row_is_reusable_comparator(missing, _expected()) is False + + +@requires_openat +def test_a_reuse_root_replaced_after_the_check_is_refused(tmp_path: Path, monkeypatch) -> None: + """Check and use must name the same directory, not the same string. + + _resolved_directory lstats a name and the open re-walks that same name, so + a prior sweep that swaps its results root in between is opened somewhere + else. The leaf-symlink rule does not cover it - a replacement that is + itself a real directory passes every check the policy makes - and the + failure is silent, folding another directory's rows into this sweep's + comparator baseline. + """ + + original = tmp_path / "results" + original.mkdir() + resolved, stale_identity = comparator_reuse._resolved_directory(original, label="probe") + + # Replaced by a different REAL directory: the name still resolves and still + # passes the symlink policy, but it is not the inode that was checked. + original.rename(tmp_path / "moved") + original.mkdir() + assert comparator_reuse._resolved_directory(original, label="probe")[1] != stale_identity + + monkeypatch.setattr( + comparator_reuse, "_resolved_directory", lambda *_a, **_k: (resolved, stale_identity) + ) + with pytest.raises(SandboxError, match="replaced between the check and the open"): + with comparator_reuse._open_pinned_root(original, label="probe"): + pass + + +@requires_openat +def test_a_stable_reuse_root_opens_normally(tmp_path: Path) -> None: + """The guard rejects nothing that holds still - a directory matches itself.""" + + root = tmp_path / "results" + root.mkdir() + with comparator_reuse._open_pinned_root(root, label="probe") as fd: + assert os.fstat(fd).st_ino == root.stat().st_ino diff --git a/eval/tests/test_evolve.py b/eval/tests/test_evolve.py index c135584de..32d50fd85 100644 --- a/eval/tests/test_evolve.py +++ b/eval/tests/test_evolve.py @@ -9,19 +9,24 @@ import time from contextlib import contextmanager from datetime import UTC, datetime, timedelta from pathlib import Path +from types import SimpleNamespace import pytest from workflow_bench import evolve, evolution from workflow_bench.runner_sessions import PARENT_EVENT_STREAM_SOURCE from workflow_bench.evolve import ( + MIN_INSTANCE_SWEEP_SECONDS, build_parser, build_proposer_prompt, + capped_timeout_seconds, executed_benchmark_arms, generation_timeout_seconds, + instance_window_budget_seconds, load_jsonl, proposer_evidence_entries, read_learnings, + remaining_runtime_seconds, resolve_incumbent_arms, runner_argv, select_evidence, @@ -662,6 +667,127 @@ def test_run_proposer_hides_the_hidden_harness_and_keeps_the_full_tool_surface(m assert captured["settings_json"] == FakeSandbox.settings_json +def test_proposer_session_cannot_outlive_the_remaining_instance_window(monkeypatch, tmp_path): + """Clearing the sweep minimum is not a licence to run a full session. + + --timeout is sized for a whole generation, so a proposer started with the + minimum left would run far past --max-runtime-seconds and the box would take + the evidence with it. The budget is sampled after the clone, the sanitize + pass and the sandbox setup, because a reading taken before them is already + stale by the time the session it bounds actually starts. + """ + + captured: dict[str, object] = {} + # Pinned clock: real setup duration would make this assert on scheduling. + clock = {"now": 1000.0} + monkeypatch.setattr(evolve.time, "monotonic", lambda: clock["now"]) + setup_seconds = 100.0 + + @contextmanager + def fake_prepare_sandbox(**_kwargs): + yield SimpleNamespace( + claude_bin="claude", + command_prefix=[], + settings_json="{}", + transcript_projects=tmp_path / "transcript-projects", + ) + + def fake_run_claude(*_args, **kwargs): + captured.update(kwargs) + return {"ok": False, "error_kind": "session-error"} + + def slow_sanitize(_clone): + # Stands in for the clone, the sanitize pass and the sandbox build — + # all of which run between the caller's decision and the session. + clock["now"] += setup_seconds + return "0" * 40 + + monkeypatch.setattr(evolve.runner, "make_worktree", lambda _repo, _ref, destination: destination) + monkeypatch.setattr(evolve.runner, "remove_clone", lambda _clone: None) + monkeypatch.setattr(evolve, "sanitize_clone_for_hidden_oracles", slow_sanitize) + monkeypatch.setattr(evolve, "prepare_sandbox", fake_prepare_sandbox) + monkeypatch.setattr(evolve.runner, "run_claude", fake_run_claude) + args = build_parser().parse_args(["--tasks", "tasks.yaml", "--model", "model"]) + assert args.timeout > evolve.MIN_INSTANCE_SWEEP_SECONDS, "otherwise this test proves nothing" + + common = { + "overlay_dir": tmp_path / "overlay", + "proposal_path": tmp_path / "proposal.md", + "evidence_bundle": tmp_path / "evidence", + "bwrap_bin": tmp_path / "bwrap", + } + budget = evolve.MIN_INSTANCE_SWEEP_SECONDS + 1 + args.max_runtime_seconds = budget + started = clock["now"] + evolve.run_proposer("prompt", args, **common, started_monotonic=started) + # The setup time is charged, not handed back: a value sampled at `started` + # would have allowed the whole budget. + assert captured["timeout"] == budget - setup_seconds + + # No cap configured means no budget to overrun: the session keeps its own. + args.max_runtime_seconds = None + evolve.run_proposer("prompt", args, **common, started_monotonic=started) + assert captured["timeout"] == args.timeout + evolve.run_proposer("prompt", args, **common) + assert captured["timeout"] == args.timeout + + +def test_a_budget_spent_during_setup_stops_the_proposer_rather_than_buying_a_second( + monkeypatch, tmp_path +): + """An exhausted cap must end the generation, not start a one-second session. + + remaining_runtime_seconds floors at 0, and the call site wrapped it in + max(1, ...) - so a cap fully consumed by the clone, the sanitize pass and + the sandbox build produced a paid session with a one-second allowance + instead of stopping before the upload reserve the cap exists to protect. + """ + + captured: dict[str, object] = {} + clock = {"now": 1000.0} + monkeypatch.setattr(evolve.time, "monotonic", lambda: clock["now"]) + + @contextmanager + def fake_prepare_sandbox(**_kwargs): + yield SimpleNamespace( + claude_bin="claude", + command_prefix=[], + settings_json="{}", + transcript_projects=tmp_path / "transcript-projects", + ) + + def fake_run_claude(*_args, **kwargs): + captured.update(kwargs) + return {"ok": True} + + def setup_that_spends_the_whole_budget(_clone): + clock["now"] += budget + return "0" * 40 + + monkeypatch.setattr(evolve.runner, "make_worktree", lambda _repo, _ref, destination: destination) + monkeypatch.setattr(evolve.runner, "remove_clone", lambda _clone: None) + monkeypatch.setattr(evolve, "sanitize_clone_for_hidden_oracles", setup_that_spends_the_whole_budget) + monkeypatch.setattr(evolve, "prepare_sandbox", fake_prepare_sandbox) + monkeypatch.setattr(evolve.runner, "run_claude", fake_run_claude) + + args = build_parser().parse_args(["--tasks", "tasks.yaml", "--model", "model"]) + budget = evolve.MIN_INSTANCE_SWEEP_SECONDS + 1 + args.max_runtime_seconds = budget + record = evolve.run_proposer( + "prompt", + args, + overlay_dir=tmp_path / "overlay", + proposal_path=tmp_path / "proposal.md", + evidence_bundle=tmp_path / "evidence", + bwrap_bin=tmp_path / "bwrap", + started_monotonic=clock["now"], + ) + + assert not captured, "no session may start once the cap is exhausted" + assert record["ok"] is False + assert record["error_kind"] == "runtime-cap-exhausted" + + def test_parser_defaults_match_the_gate_minimums(): args = build_parser().parse_args(["--tasks", "t.yaml", "--model", "pinned"]) assert args.runs == 3 @@ -922,6 +1048,27 @@ def test_runner_argv_inserts_ce_review_for_review_overlay(tmp_path): ) arms = argv[argv.index("--arms") + 1 : argv.index("--promotion-metric")] assert arms == ["ce_review", "review", "candidate_review"] + assert "--reuse-results" not in argv + + +def test_runner_argv_forwards_prior_results_for_comparator_reuse(tmp_path): + args = build_parser().parse_args( + ["--tasks", "t.yaml", "--model", "pinned", "--arms", "review"] + ) + overlay = tmp_path / "overlay" + skill = overlay / ".claude" / "skills" / "gitnexus-review" / "SKILL.md" + skill.parent.mkdir(parents=True) + skill.write_text("candidate") + prior = tmp_path / "prior-bench" + argv = runner_argv( + args, + tmp_path / "bench", + overlay, + task_bindings=[{"id": "task"}], + target_base_digests={}, + reuse_results=prior, + ) + assert argv[argv.index("--reuse-results") + 1] == str(prior) def test_runner_argv_omits_proposer_for_manual_overlay(tmp_path): @@ -1118,6 +1265,161 @@ def test_generation_timeout_rejects_unknown_arm() -> None: ) +def test_instance_window_budget_leaves_upload_reserve() -> None: + # Friday 10:57 on a box that booted 02:45 Saturday-window: ~8.2h uptime. + leftover = instance_window_budget_seconds(8.2 * 3600) + assert leftover == int(86_400 - 8.2 * 3600 - 5_400) + assert leftover >= MIN_INSTANCE_SWEEP_SECONDS + with pytest.raises(ValueError, match="only .*s left"): + instance_window_budget_seconds(23.5 * 3600) + with pytest.raises(ValueError, match="uptime must be"): + instance_window_budget_seconds(float("nan")) + + + + +def test_capped_timeout_clamps_to_leftover_window(monkeypatch) -> None: + assert capped_timeout_seconds(10_000, None) == 10_000 + assert capped_timeout_seconds(10_000, 90) == 90 + with pytest.raises(ValueError, match="no time remains"): + capped_timeout_seconds(10_000, 0) + # Pinned clock: the helper is pure arithmetic, so a real elapsed-time window + # would assert on scheduling rather than on the behaviour under test. + monkeypatch.setattr(evolve.time, "monotonic", lambda: 1040.0) + started = 1000.0 + assert remaining_runtime_seconds(max_runtime_seconds=None, started_monotonic=started) is None + leftover = remaining_runtime_seconds(max_runtime_seconds=100, started_monotonic=started) + assert leftover is not None + assert leftover == 60 + + +def test_parser_rejects_non_positive_max_runtime() -> None: + with pytest.raises(SystemExit): + build_parser().parse_args( + ["--tasks", "t.yaml", "--model", "pinned", "--max-runtime-seconds", "0"] + ) + args = build_parser().parse_args( + ["--tasks", "t.yaml", "--model", "pinned", "--max-runtime-seconds", "7200"] + ) + assert args.max_runtime_seconds == 7200 + assert args.max_runtime_from_instance_window is False + + +def _task_file(tmp_path: Path) -> Path: + tasks = tmp_path / "tasks.yaml" + tasks.write_text( + """tasks: + - id: demo + class: test + repo: . + prompt: implement + verify: "true" + oracle: + command: "true" + files: + - source: hidden.test.ts + target: hidden.test.ts +""" + ) + return tasks + + +def _stub_main_preflight(monkeypatch, tmp_path) -> None: + """Everything main() shells out to before it reaches _run_generations.""" + + monkeypatch.setattr(evolve.runner, "selected_task_bindings", lambda _tasks: [{"id": "demo"}]) + monkeypatch.setattr(evolve, "preflight_bubblewrap", lambda: tmp_path / "bwrap") + monkeypatch.setattr(evolve, "require_claude_sandbox_helpers", lambda: None) + + +def test_the_runtime_cap_is_derived_where_its_clock_starts(monkeypatch, tmp_path, capsys) -> None: + """The budget and the clock it is measured against must be one instant. + + run-evolution.sh used to compute the budget in a separate `uv run python -c` + and pass a number, so the script's remaining provenance work and this + interpreter's startup were charged to the sweep — out of the upload reserve + the cap exists to protect. main() reads /proc/uptime itself now, next to its + own clock, so no interval exists to lose. + """ + + monkeypatch.setenv("EVENTBRIDGE_INSTANCE_WINDOW_SECONDS", "20000") + monkeypatch.setenv("EVENTBRIDGE_STOP_RESERVE_SECONDS", "1000") + monkeypatch.setattr(evolve, "read_instance_uptime_seconds", lambda: 3600.0) + captured: dict[str, object] = {} + + def record(args, **kwargs): + captured["max_runtime_seconds"] = args.max_runtime_seconds + captured["started_monotonic"] = kwargs["started_monotonic"] + return 0 + + monkeypatch.setattr(evolve, "_run_generations", record) + _stub_main_preflight(monkeypatch, tmp_path) + monkeypatch.setattr( + sys, + "argv", + [ + "evolve", + "--tasks", + str(_task_file(tmp_path)), + "--model", + "pinned", + "--out-root", + str(tmp_path / "out"), + "--max-runtime-from-instance-window", + ], + ) + + assert evolve.main() == 0 + + assert captured["max_runtime_seconds"] == 20000 - 3600 - 1000 + # Derived here, not passed in: the clock handed to the sweep is the one + # taken beside the uptime read. + assert isinstance(captured["started_monotonic"], float) + assert "capping the sweep to 15400s" in capsys.readouterr().out + + +def test_the_runtime_cap_refuses_two_sources_of_truth(monkeypatch, tmp_path) -> None: + monkeypatch.setattr(evolve, "read_instance_uptime_seconds", lambda: 3600.0) + monkeypatch.setattr( + sys, + "argv", + [ + "evolve", + "--tasks", + str(_task_file(tmp_path)), + "--model", + "pinned", + "--max-runtime-from-instance-window", + "--max-runtime-seconds", + "7200", + ], + ) + with pytest.raises(SystemExit): + evolve.main() + + +def test_the_runtime_cap_fails_closed_without_a_readable_uptime(monkeypatch, tmp_path) -> None: + def unreadable(): + raise ValueError("cannot read instance uptime from /proc/uptime") + + monkeypatch.setattr(evolve, "read_instance_uptime_seconds", unreadable) + monkeypatch.setattr( + sys, + "argv", + [ + "evolve", + "--tasks", + str(_task_file(tmp_path)), + "--model", + "pinned", + "--max-runtime-from-instance-window", + ], + ) + # Better to refuse than to run a box-stopped sweep believing it is uncapped. + with pytest.raises(SystemExit): + evolve.main() + + @pytest.mark.skipif(sys.platform != "linux", reason="Bubblewrap PID namespaces require Linux") def test_outer_runner_pid_namespace_kills_setsid_descendant(tmp_path): try: diff --git a/eval/tests/test_process_control.py b/eval/tests/test_process_control.py index 8d7405356..1184feea0 100644 --- a/eval/tests/test_process_control.py +++ b/eval/tests/test_process_control.py @@ -79,11 +79,11 @@ def run(index, arm): assert Path({str(assets)!r}).exists(), 'assets removed while a worker was active' return {{'resolved': False, 'error_kind': result.state}} with cancellation_scope(handle_signals=True) as event: - streak, stopped=sweep_task_cells([(i, 'review') for i in range(10)], workers=2, run=run, + streak, tripped=sweep_task_cells([(i, 'review') for i in range(10)], workers=2, run=run, on_start=lambda *args: None, on_record=lambda i,a,r: rows.append([i,r]), outage_streak=0, outage_limit=5, cancel_event=event) Path({str(assets)!r}).unlink() -print(json.dumps({{'rows': rows, 'stopped': stopped}})) +print(json.dumps({{'rows': rows, 'stopped': event.is_set(), 'tripped': tripped}})) """ process = subprocess.Popen( [PYTHON, "-c", script], @@ -104,7 +104,8 @@ print(json.dumps({{'rows': rows, 'stopped': stopped}})) assert process.returncode == 0, stderr assert time.monotonic() - started < 15 report = json.loads(stdout) - assert report["stopped"] and [row[0] for row in report["rows"]] == [0, 1] + assert report["stopped"] and not report["tripped"], "cancelled, not an outage" + assert [row[0] for row in report["rows"]] == [0, 1] assert report["rows"][0][1]["resolved"] is True assert report["rows"][1][1]["error_kind"] == "cancelled" with pytest.raises(ProcessLookupError): diff --git a/eval/tests/test_proposer_sandbox.py b/eval/tests/test_proposer_sandbox.py index 2e30cf6a2..26532a5f0 100644 --- a/eval/tests/test_proposer_sandbox.py +++ b/eval/tests/test_proposer_sandbox.py @@ -33,9 +33,11 @@ from workflow_bench.proposer_sandbox import ( SANDBOX_GIT_EXCLUDES, VITE_TEMP_DIR, SANDBOX_PATH, + SANDBOX_REVIEW_OUTPUT, SANDBOX_PYTHON3, SANDBOX_SHELL_PREFIX, SANDBOX_USER_SKILLS, + SANDBOX_WORKSPACE, ReadOnlyMount, SandboxError, _runtime_mount_args, @@ -43,46 +45,65 @@ from workflow_bench.proposer_sandbox import ( build_sandbox_environment, _force_rmtree, host_workspace_write_boundary, + prepare_review_workspace, prepare_sandbox, preflight_bubblewrap, sandbox_workspace_write_boundary, stage_evidence_bundle, stage_task_assets, ) +from workflow_bench.review_scoring import REVIEW_OUTPUT, parse_review_output from workflow_bench.task_assets import TaskAssetCache, stage_task_assets as stage_immutable_task_assets -@pytest.mark.parametrize("entry", ["file", "directory", "relative-link", "absolute-link"]) -def test_review_preparation_rejects_existing_output_without_touching_target(tmp_path, entry): +@pytest.mark.parametrize("entry", ["directory", "relative-link", "absolute-link"]) +def test_review_preparation_rejects_a_reused_artifact_directory(tmp_path, entry): clone = tmp_path / "clone" clone.mkdir() sentinel = tmp_path / "sentinel" sentinel.write_text("must survive") - output = clone / "review-output.json" - if entry == "file": - output.write_text("existing result") - elif entry == "directory": - output.mkdir() - else: - output.symlink_to(sentinel if entry == "absolute-link" else "../sentinel") with prepare_sandbox(clone=clone, claude_bin=sys.executable, backend="host-unsafe") as sandbox: + stale = proposer_sandbox.review_output_path(sandbox, "review-output.json").parent + if entry == "directory": + stale.mkdir() + (stale / "review-output.json").write_text("a previous cell's verdict") + else: + # relpath, not a hand-written "../sentinel": stale is + # /review-output, which is nowhere near tmp_path, so the + # literal produced a dangling link and the assertion below proved nothing. + stale.symlink_to( + sentinel if entry == "absolute-link" else Path(os.path.relpath(sentinel, stale.parent)) + ) with pytest.raises(SandboxError, match="already exists"): proposer_sandbox.prepare_review_workspace(sandbox, "review-output.json") assert sentinel.read_text() == "must survive" - if entry == "file": - assert output.read_text() == "existing result" - if "link" in entry: - assert output.is_symlink() -def test_review_preparation_creates_a_private_regular_output(tmp_path): +def test_review_preparation_leaves_a_clone_entry_of_the_same_name_alone(tmp_path): + # The artifact no longer lives in the workspace, so a file that happens to + # share its name is just one of the repository's own files. + clone = tmp_path / "clone" + clone.mkdir() + (clone / "review-output.json").write_text("repository content") + with prepare_sandbox(clone=clone, claude_bin=sys.executable, backend="host-unsafe") as sandbox: + output = proposer_sandbox.prepare_review_workspace(sandbox, "review-output.json") + assert (clone / "review-output.json").read_text() == "repository content" + assert clone not in output.parents + + +def test_review_preparation_creates_a_private_directory_and_not_the_file(tmp_path): clone = tmp_path / "clone" clone.mkdir() with prepare_sandbox(clone=clone, claude_bin=sys.executable, backend="host-unsafe") as sandbox: output = proposer_sandbox.prepare_review_workspace(sandbox, "review-output.json") - assert output.read_bytes() == b"" - assert stat.S_ISREG(output.lstat().st_mode) - assert stat.S_IMODE(output.stat().st_mode) == 0o600 + assert output == proposer_sandbox.review_output_path(sandbox, "review-output.json") + # The DIRECTORY is what has to exist and be writable: the agent writes + # a temp file beside the target and renames it. + assert output.parent.is_dir() + assert stat.S_IMODE(output.parent.stat().st_mode) == 0o700 + # The file is deliberately absent — absence is how "never written" is + # told apart from "written badly". + assert not output.exists() def test_review_preparation_preserves_existing_runtime_files_and_tracks_only_created_paths(tmp_path): @@ -165,6 +186,15 @@ def test_unsafe_host_session_translates_virtual_paths_and_disables_containment(t assert sandbox.require_pid_namespace is False assert sandbox.host_path("/workspace/review-output.json") == str(clone / "review-output.json") assert sandbox.host_path("/evidence/selected-rows.json") == str(evidence / "selected-rows.json") + # The review artifact left the workspace, so the host-unsafe backend has + # to translate its new home too. Untranslated, the review prompt names a + # path that exists on neither backend and the cell writes nothing. + assert sandbox.host_path( + f"{proposer_sandbox.SANDBOX_REVIEW_OUTPUT}/review-output.json" + ) == str(proposer_sandbox.review_output_path(sandbox, "review-output.json")) + assert sandbox.host_text( + f"write {proposer_sandbox.SANDBOX_REVIEW_OUTPUT}/review-output.json" + ) == f"write {proposer_sandbox.review_output_path(sandbox, 'review-output.json')}" assert sandbox.host_text("read /evidence and write /workspace/out") == ( f"read {evidence} and write {clone}/out" ) @@ -867,9 +897,8 @@ def test_read_only_review_workspace_exposes_only_one_writable_artifact(tmp_path: clone.mkdir() source = clone / "source.ts" source.write_text("trusted\n") - output = clone / "review-output.json" - output.write_text("") script = """ +import os from pathlib import Path try: Path('/workspace/source.ts').write_text('tampered') @@ -877,15 +906,24 @@ except OSError: pass else: raise SystemExit('review source remained writable') -Path('/workspace/review-output.json').write_text('{"schema_version":1}') +# Write the way the agent's Write tool does: a temp file beside the target, +# then rename. Writing in place would pass against the mount shape that +# shipped every artifact empty, which is the regression this canary exists for. +target = Path('/review-output/review-output.json') +staging = target.with_name(target.name + '.tmp.1.abc') +staging.write_text('{"schema_version":1}') +os.replace(staging, target) """ with prepare_sandbox(clone=clone, claude_bin=Path(sys.executable), preflight=True) as sandbox: + output = proposer_sandbox.prepare_review_workspace(sandbox, "review-output.json") result = run_managed( [ *sandbox.command_prefix_for( read_only_workspace=True, extra_writable_mounts=( - ReadOnlyMount(source=output, target="/workspace/review-output.json"), + ReadOnlyMount( + source=output.parent, target=proposer_sandbox.SANDBOX_REVIEW_OUTPUT + ), ), ), "/usr/bin/python3", @@ -897,9 +935,14 @@ Path('/workspace/review-output.json').write_text('{"schema_version":1}') require_pid_namespace=True, ) - assert result.ok, result.stderr_tail - assert source.read_text() == "trusted\n" - assert output.read_text() == '{"schema_version":1}' + # Inside the sandbox scope: the artifact now lives under the session's + # private root, which prepare_sandbox removes on exit. run_arm reads it + # here too, while the session is still alive. + assert result.ok, result.stderr_tail + assert source.read_text() == "trusted\n" + assert output.read_text() == '{"schema_version":1}' + # The staging file is gone: the rename landed rather than a copy. + assert list(output.parent.iterdir()) == [output] @pytest.mark.skipif(os.name == "nt", reason="symlink creation may require elevated Windows privileges") @@ -1170,6 +1213,7 @@ for line in sys.stdin: review_command = """test -z "${ANTHROPIC_API_KEY:-}" && python3 - <<'PY' import json +import os import subprocess from pathlib import Path source = Path('/workspace/canary.txt') @@ -1182,7 +1226,10 @@ for operation in (lambda: source.write_text('forbidden'), lambda: source.rename( pass else: raise AssertionError('source mutation was allowed') -Path('/workspace/review-output.json').write_text(json.dumps({'schema_version': 1, 'verdict': 'approve', 'findings': []})) +target = Path('/review-output/review-output.json') +staging = target.with_name(target.name + '.tmp.1.abc') +staging.write_text(json.dumps({'schema_version': 1, 'verdict': 'approve', 'findings': []})) +os.replace(staging, target) PY""" if review_layout: for command in ( @@ -1359,7 +1406,11 @@ PY""" sandbox, command_prefix=sandbox.command_prefix_for( read_only_workspace=True, - extra_writable_mounts=(ReadOnlyMount(output, "/workspace/review-output.json"),), + extra_writable_mounts=( + ReadOnlyMount( + output.parent, proposer_sandbox.SANDBOX_REVIEW_OUTPUT + ), + ), ), ) result = sandbox.run( @@ -1408,7 +1459,7 @@ PY""" assert (sandbox.temp / "mcp-called").read_text() == "ok" if review_layout: assert output is not None - runner_artifacts.enforce_phase_workspace(clone, before, allowed_artifact=output) + runner_artifacts.enforce_phase_workspace(clone, before, allowed_artifact=None) assert json.loads(output.read_text())["verdict"] == "approve" assert (clone / "canary.txt").read_text() == "hook-readable\nchanged for review\n" finally: @@ -1418,3 +1469,98 @@ PY""" if not review_layout: assert (clone / "bash-called").read_text() == "canary" + + +def test_review_artifact_binds_a_writable_directory_outside_the_workspace(tmp_path): + """The bwrap argv, since the mount shape is the whole bug. + + bwrap cannot create a mount point inside an already-read-only bind, so a + writable path has to live outside /workspace — and it has to be the + directory, or the agent has nowhere to put the temp file it renames into + place. + """ + + clone = tmp_path / "clone" + clone.mkdir() + with prepare_sandbox(clone=clone, claude_bin=sys.executable, backend="host-unsafe") as session: + sandbox = replace(session, backend="bwrap") + output = proposer_sandbox.review_output_path(sandbox, "review-output.json") + output.parent.mkdir(mode=0o700) + argv = sandbox.command_prefix_for( + read_only_workspace=True, + extra_writable_mounts=( + proposer_sandbox.ReadOnlyMount( + source=output.parent, + target=proposer_sandbox.SANDBOX_REVIEW_OUTPUT, + ), + ), + ) + + target = proposer_sandbox.SANDBOX_REVIEW_OUTPUT + assert not target.startswith(proposer_sandbox.SANDBOX_WORKSPACE + "/") + # The workspace itself is bound read-only... + workspace_at = argv.index(proposer_sandbox.SANDBOX_WORKSPACE) + assert argv[workspace_at - 2] == "--ro-bind" + # ...and the artifact directory is bound writable, as a directory. + artifact_at = argv.index(target) + assert argv[artifact_at - 2] == "--bind" + assert Path(argv[artifact_at - 1]) == output.parent + assert Path(argv[artifact_at - 1]).is_dir() + assert f"{proposer_sandbox.SANDBOX_WORKSPACE}/review-output.json" not in argv + + +@pytest.mark.skipif( + os.environ.get("GITNEXUS_REQUIRE_BWRAP_CANARY") != "1", + reason="real Bubblewrap canary is mandatory in the named Ubuntu CI job", +) +def test_real_bubblewrap_lets_a_review_artifact_be_written_atomically(tmp_path: Path) -> None: + """The filesystem contract the EROFS defect broke, under a real sandbox. + + Argv assertions cannot establish this. The artifact came back empty because + an atomic write - temp file beside the target, then rename - needs a + WRITABLE PARENT DIRECTORY, and only a real bwrap invocation shows whether + the mount grants one. A deterministic writer stands in for the agent: no + model session, no credentials. + + Scope: this proves the filesystem and process contract of the production + mount configuration. It does not establish that a particular agent CLI's + own file-access policy permits the same operation - that is a second, + independent gate. + """ + + clone = tmp_path / "clone" + clone.mkdir() + (clone / "tracked.txt").write_text("original\n") + + with prepare_sandbox(clone=clone, claude_bin=Path(sys.executable), preflight=True) as sandbox: + review_output = prepare_review_workspace(sandbox, REVIEW_OUTPUT) + # The production configuration, not a hand-built mount tuple: the same + # command_prefix_for call run_arm makes for a review cell. + prefix = sandbox.command_prefix_for( + read_only_workspace=True, + extra_writable_mounts=( + ReadOnlyMount(source=review_output.parent, target=SANDBOX_REVIEW_OUTPUT), + ), + ) + target = f"{SANDBOX_REVIEW_OUTPUT}/{REVIEW_OUTPUT}" + script = ( + # 1. temp file beside the destination, then atomic rename over it. + f'printf %s \'{{"schema_version": 1, "verdict": "approve", "findings": []}}\' > {target}.tmp && ' + f"mv {target}.tmp {target} && " + # 2. the workspace must refuse the write that the mount forbids. + f"(printf x >> {SANDBOX_WORKSPACE}/tracked.txt 2>/dev/null && echo WORKSPACE-WRITABLE || echo workspace-readonly)" + ) + result = subprocess.run( + [*prefix, "/bin/sh", "-c", script], + capture_output=True, text=True, timeout=60, check=False, + ) + + assert result.returncode == 0, f"atomic write failed inside the sandbox: {result.stderr[-400:]}" + assert "workspace-readonly" in result.stdout, "the workspace must stay read-only" + assert (clone / "tracked.txt").read_text() == "original\n", "the clone was modified" + + # Read while the session is alive: the artifact lives under the private + # root that prepare_sandbox removes on exit, which is also why run_arm + # consumes it before leaving the scope. + _verdict, findings = parse_review_output(review_output) + assert findings == () diff --git a/eval/tests/test_reuse_round_trip.py b/eval/tests/test_reuse_round_trip.py new file mode 100644 index 000000000..a31e9c8f7 --- /dev/null +++ b/eval/tests/test_reuse_round_trip.py @@ -0,0 +1,268 @@ +"""A row the runner actually emits must satisfy the reuse reader. + +Every existing comparator-reuse test builds its rows by hand. That proves the +predicate's logic and nothing about the producer: a fixture can satisfy +eligibility while a real emitted row never does, and the audit that counts key +names cannot tell the difference. These tests carry one record through the +production path instead: + + real run_cell -> production JSONL writer -> load_result_rows + -> row_is_reusable_comparator + +Only the expensive dependencies are replaced - the model session, sandbox +launch, repository acquisition, graph preparation. The digest fields the reuse +binding compares are assembled by run_cell itself from its TaskCellContext, so +they stay real: they are the subject of the test, not scaffolding around it. +""" + +from __future__ import annotations + +import json +from datetime import UTC, datetime, timedelta +from pathlib import Path +from types import SimpleNamespace +from typing import Any + +import pytest + +from workflow_bench import runner +from workflow_bench.proposer_sandbox import redact_text +from workflow_bench.model_gateway import credential_secrets +from workflow_bench.runner_sessions import PARENT_EVENT_STREAM_SOURCE +from workflow_bench.comparator_reuse import ( + ComparatorReuseExpectation, + TaskReuseBinding, + load_result_rows, + row_is_reusable_comparator, +) + +TASK_ID = "review-pr-2718-defect" +SHA = "a" * 40 + + +def _snapshot(prefix: str) -> SimpleNamespace: + return SimpleNamespace( + digest=f"{prefix}-content", + manifest_digest=f"{prefix}-manifest", + dependency_content_digest=f"{prefix}-dep-content", + dependency_manifest_digest=f"{prefix}-dep-manifest", + command_digest=f"{prefix}-command", + materialize=lambda *a, **k: None, + ) + + +def _write_like_the_sweep(tmp_path: Path, row: dict[str, Any]) -> Path: + """Serialize exactly as ``keep`` does in _run_sweep, redaction included. + + json.dumps + write_text would skip the redaction the real writer applies, + so a change there could break reusable rows without failing this test - and + redaction is not cosmetic here, since it rewrites the row's own bytes. + """ + + results = tmp_path / "results.jsonl" + secrets = credential_secrets( + SimpleNamespace(auth_token="sk-ant-should-never-appear", base_url=None) + ) + with results.open("a") as handle: + handle.write(redact_text(json.dumps(row), secrets) + "\n") + return results + + +@pytest.fixture +def emitted_row(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> dict[str, Any]: + """One record from the real run_cell, with only expensive work replaced.""" + + worktree = tmp_path / "clone" + worktree.mkdir() + + # The session is what costs money; everything it returns is scripted. The + # record's binding fields are NOT set here - run_cell derives them. + def fake_run_arm(*_a: Any, **_k: Any) -> dict[str, Any]: + return { + "ok": True, + "error_kind": None, + "error_detail": None, + "resolved": True, + "review_evidence_valid": True, + "review_score": {"weighted_f1": 0.5}, + "review_weighted_f1": 0.5, + "skill_invoked": True, + "skill_digest": "skill-digest", + "transcript_missing": False, + "transcript_artifacts": [ + { + "path": "transcripts/session-1.jsonl", + "sha256": __import__("hashlib").sha256(b'{"type":"ok"}\n').hexdigest(), + "bytes": 14, + "source": PARENT_EVENT_STREAM_SOURCE, + } + ], + "session_ids": ["s1"], + "num_turns": 3, + "duration_s": 1.0, + "cost_usd": 0.5, + "input_tokens": 1, + "output_tokens": 1, + "cache_creation_input_tokens": 0, + "cache_read_input_tokens": 0, + } + + for name, value in { + "run_arm": fake_run_arm, + "copy_isolated_tree": lambda *a, **k: worktree, + "make_worktree": lambda *a, **k: worktree, + "sanitize_clone_for_hidden_oracles": lambda *a, **k: SHA, + "stage_task_assets": lambda *a, **k: (), + "isolated_gitnexus_registry_mount": lambda *a, **k: None, + "seed_evaluated_skills": lambda *a, **k: None, + "apply_candidate_overlay": lambda *a, **k: None, + "require_hidden_harness_absent": lambda *a, **k: None, + "require_skill_fingerprint": lambda *a, **k: None, + "enforce_work_evidence": lambda *a, **k: None, + "skill_fingerprint": lambda *a, **k: "skill-digest", + "capture_patch": lambda *a, **k: b"", + "implementation_diff_digest": lambda *a, **k: "", + "diff_churn": lambda *a, **k: {}, + "_prepare_untracked_for_diff": lambda *a, **k: None, + "remove_clone": lambda *a, **k: None, + "ce_plugin_dir_for_arm": lambda *a, **k: None, + "ce_plugin_mounts_for_arm": lambda *a, **k: (), + "current_runtime_digest": lambda: "runtime-digest", + "build_sandbox_environment": lambda *a, **k: {}, + "credential_secrets": lambda *a, **k: (), + # run_cell requires an immutable base commit before it will record a + # cell; the git plumbing is expensive setup, the SHA it returns is not + # part of the reuse binding under test. + "_sandbox_git": lambda *a, **k: SHA, + # The artifact copy is real; only the read of the agent-written file is + # replaced, since no agent ran to write one. + "_bounded_regular_bytes": lambda *a, **k: b'{"schema_version":1}', + }.items(): + monkeypatch.setattr(runner, name, value) + + class _Sandbox: + clone = worktree + private_root = tmp_path / "private" + backend = "test-double" + settings_json = "{}" + require_pid_namespace = False + + def __enter__(self) -> _Sandbox: + return self + + def __exit__(self, *_exc: Any) -> bool: + return False + + def command_prefix_for(self, **_k: Any) -> list[str]: + return [] + + def run(self, *_a: Any, **_k: Any) -> SimpleNamespace: + return SimpleNamespace(ok=True, returncode=0, stdout_tail="", stderr_tail="") + + def environment(self, **_k: Any) -> dict[str, str]: + return {} + + def host_text(self, value: str) -> str: + return value + + monkeypatch.setattr(runner, "prepare_sandbox", lambda **_k: _Sandbox()) + + ctx = runner.TaskCellContext( + task={"id": TASK_ID, "prompt": "review it", "verify": "true"}, + oracle_snapshot=_snapshot("oracle"), + repo=tmp_path / "repo", + task_sha=SHA, + graph_snapshot=_snapshot("graph"), + graph_snapshot_error=None, + asset_snapshot=_snapshot("asset"), + asset_snapshot_error=None, + args=SimpleNamespace( + model="gpt-5.6-sol", effort="xhigh", timeout=60, claude_bin="claude", + base_url=None, auth_token=None, permission_mode=None, arms=["review"], + proposer_model=None, outage_streak=5, runs=1, workers=1, + ), + out_dir=tmp_path / "out", + ce_plugin_snapshot=None, + trees_dir=tmp_path / "trees", + bwrap_bin=Path("/bin/true"), + runtime_mounts=(), + candidate_overlay=None, + overlay_digest=None, + sandbox_backend="test-double", + clone_template=None, + sanitized_head=SHA, + ) + (tmp_path / "out").mkdir(exist_ok=True) + (tmp_path / "trees").mkdir(exist_ok=True) + (tmp_path / "private").mkdir(exist_ok=True) + # run_cell records review_artifact only when the review source exists, and + # reuse now requires it - a scored review with no artifact is a claim about + # evidence rather than the evidence. Production writes this file; the + # fixture has to as well, or the emitted row is one production never emits. + review_dir = tmp_path / "private" / "review-output" + review_dir.mkdir(exist_ok=True) + (review_dir / "review-output.json").write_text('{"schema_version": 1, "verdict": "approve", "findings": []}') + return runner.run_cell(ctx, 0, "review") + + +def _expectation(**overrides: Any) -> ComparatorReuseExpectation: + """Bindings from the sweep's own configuration, not copied out of the row. + + Copying the emitted values back in would make producer and consumer agree + because the test arranged it, which is the blind spot being closed. + """ + + binding = TaskReuseBinding( + task_base_sha=SHA, + task_prompt_digest=runner.hashlib.sha256(b"review it").hexdigest(), + oracle_digest="oracle-content", + oracle_command_digest="oracle-command", + oracle_manifest_digest="oracle-manifest", + task_asset_manifest_digest="asset-manifest", + sandbox_dependency_manifest_digest="asset-dep-manifest", + ) + values: dict[str, Any] = dict( + model="gpt-5.6-sol", + effort="xhigh", + sandbox_backend="test-double", + runtime_digest="runtime-digest", + now=datetime.now(UTC), + max_age=timedelta(days=90), + tasks={TASK_ID: binding}, + skill_digests={"review": "skill-digest"}, + ce_plugin_version=None, + ce_plugin_manifest_digest=None, + ) + values.update(overrides) + return ComparatorReuseExpectation(**values) + + +def test_a_row_the_runner_emitted_survives_serialization_and_qualifies( + emitted_row: dict[str, Any], tmp_path: Path +) -> None: + """The producer/consumer contract, end to end through the real writer.""" + + results = _write_like_the_sweep(tmp_path, emitted_row) + rows = load_result_rows(results) + assert len(rows) == 1, "the production row must survive the reader" + + assert row_is_reusable_comparator(rows[0], _expectation()) is True + + +def test_a_changed_binding_rejects_the_same_emitted_row( + emitted_row: dict[str, Any], tmp_path: Path +) -> None: + """Fails closed on drift, so the positive case is not vacuous.""" + + row = load_result_rows(_write_like_the_sweep(tmp_path, emitted_row))[0] + + binding = TaskReuseBinding( + task_base_sha=SHA, + task_prompt_digest=runner.hashlib.sha256(b"review it").hexdigest(), + oracle_digest="oracle-content", + oracle_command_digest="oracle-command", + oracle_manifest_digest="oracle-manifest", + task_asset_manifest_digest="asset-manifest", + sandbox_dependency_manifest_digest="DIFFERENT-dependencies", + ) + assert row_is_reusable_comparator(row, _expectation(tasks={TASK_ID: binding})) is False diff --git a/eval/tests/test_review_corpus.py b/eval/tests/test_review_corpus.py index 42f18cb7c..cbaa2695e 100644 --- a/eval/tests/test_review_corpus.py +++ b/eval/tests/test_review_corpus.py @@ -32,6 +32,10 @@ def test_review_corpus_is_immutable_and_task_bound(): assert task["ref"] == case["base_sha"] assert task["sandbox_copy"] == [f"eval/workflow_bench/review_cases/{patch.name}"] assert task["setup"] == review_case_setup_command(patch.name) + assert any( + dep.get("source") == "gitnexus-shared/dist" and dep.get("target") == "gitnexus-shared/dist" + for dep in task["sandbox_dependencies"] + ) def test_hidden_labels_are_not_recoverable_from_visible_task_input(): diff --git a/eval/tests/test_review_scoring.py b/eval/tests/test_review_scoring.py index c1b081720..1e6c8922e 100644 --- a/eval/tests/test_review_scoring.py +++ b/eval/tests/test_review_scoring.py @@ -325,3 +325,33 @@ def test_clean_control_rewards_an_empty_approval_and_penalizes_noise(): assert noisy["recall"] is None assert noisy["clean_pass"] is False assert noisy["verdict_correct"] is False + + +def test_parse_review_output_names_the_actual_failure(tmp_path: Path): + """One message per cause. + + Folding empty, malformed and encoding failures together makes a sandbox that + left the artifact at 0 bytes indistinguishable from an encoding fault: every + such cell reports "not valid UTF-8 JSON". A file the agent never created + escaped that fold — lstat sat outside the try, so it raised + FileNotFoundError — but only as a bare OSError, naming no cause at all. + """ + + missing = tmp_path / "never-written.json" + with pytest.raises(ValueError, match="was never written"): + parse_review_output(missing) + + empty = tmp_path / "empty.json" + empty.touch() + with pytest.raises(ValueError, match="is empty"): + parse_review_output(empty) + + not_utf8 = tmp_path / "latin1.json" + not_utf8.write_bytes(b'{"verdict": "\xff\xfe"}') + with pytest.raises(ValueError, match="not valid UTF-8"): + parse_review_output(not_utf8) + + prose = tmp_path / "prose.json" + prose.write_text("Here is my review of the changes.", encoding="utf-8") + with pytest.raises(ValueError, match="not valid JSON"): + parse_review_output(prose) diff --git a/eval/tests/test_runner_hardening.py b/eval/tests/test_runner_hardening.py index be196750b..111b3c3c5 100644 --- a/eval/tests/test_runner_hardening.py +++ b/eval/tests/test_runner_hardening.py @@ -3,13 +3,14 @@ import hashlib import json import shutil +import subprocess from contextlib import nullcontext from pathlib import Path from types import SimpleNamespace import pytest -from workflow_bench import runner, runner_artifacts, runner_sessions +from workflow_bench import proposer_sandbox, runner, runner_artifacts, runner_sessions from workflow_bench.evolution import skill_fingerprint from workflow_bench.oracle_assets import review_case_setup_command from workflow_bench.process_control import ManagedProcessError, ManagedProcessResult @@ -608,6 +609,67 @@ def test_run_cell_reports_a_cleanup_failure_over_its_primary_outcome(monkeypatch assert "clone is busy" in record["error_detail"] +def _git(repo, *args): + return subprocess.run(["git", "-C", str(repo), *args], check=True, capture_output=True, text=True) + + +def test_run_cell_runs_the_arm_against_a_copy_of_the_clone_template(monkeypatch, tmp_path): + """run_cell must copy the template, never re-clone. + + run_cell takes the clone-template branch on essentially every multi-cell + sweep: it copies a pre-sanitized template rather than paying `git clone + --no-local` plus repack/prune/fsck per cell. Asserting on a copy the test + makes itself proves nothing about that branch — the clone the arm receives + is what has to come from the template, carrying the template's sanitized + HEAD rather than a recomputed one. + """ + + repo = tmp_path / "repo" + repo.mkdir() + _git(repo, "init", "--quiet") + _git(repo, "checkout", "--quiet", "-b", "main") + (repo / "from-template.txt").write_text("sanitized\n") + _git(repo, "add", "-A") + _git(repo, "-c", "user.name=test", "-c", "user.email=test@invalid", "commit", "--quiet", "-m", "base") + sha = _git(repo, "rev-parse", "HEAD").stdout.strip() + trees = tmp_path / "trees" + trees.mkdir() + template = runner.make_worktree(repo, sha, trees) + template_head = _git(template, "rev-parse", "HEAD").stdout.strip() + + _stub_cell_dependencies(monkeypatch, tmp_path) + + def fail_if_recloned(*_args, **_kwargs): + raise AssertionError("clone template present: run_cell must not re-clone") + + monkeypatch.setattr(runner, "make_worktree", fail_if_recloned) + monkeypatch.setattr(runner, "sanitize_clone_for_hidden_oracles", fail_if_recloned) + + seen: dict[str, object] = {} + + def record_arm(_arm, _task, worktree, _args, **_kwargs): + seen["worktree"] = worktree + seen["head"] = _git(worktree, "rev-parse", "HEAD").stdout.strip() + seen["content"] = (worktree / "from-template.txt").read_text() + # The copy is a private checkout: what the cell writes must not reach + # the template the other cells of this task still copy from. + (worktree / "from-template.txt").write_text("cell-local\n") + return {"resolved": True, "ok": True, "error_kind": None} + + monkeypatch.setattr(runner, "run_arm", record_arm) + + runner.run_cell( + _cell_context(tmp_path, clone_template=template, sanitized_head=template_head), + 0, + "workflow", + ) + + assert seen["content"] == "sanitized\n" + assert seen["head"] == template_head + assert seen["worktree"] != template + assert (template / "from-template.txt").read_text() == "sanitized\n" + + def test_run_cell_does_not_mask_the_staged_review_patch_before_setup(monkeypatch, tmp_path): """Review setup applies a patch staged under eval/workflow_bench. @@ -1002,3 +1064,40 @@ def test_progress_line_reports_the_numbers_a_real_run_measured(): assert "cost=$0.5" in line assert "took=12.0s" in line assert "error_kind=none" in line + + +def test_claude_settings_allow_the_review_artifact_directory(): + """The second gate on the artifact path. + + The bwrap bind is not the only thing that decides whether the agent can + write: the CLI applies this filesystem policy to its own tools, so a path + missing from allowWrite is unwritable however the mount is shaped. The + artifact lived under /workspace when this list was written, which is why + moving it out needed this entry and nothing caught the omission. + """ + + settings = json.loads(proposer_sandbox.build_claude_settings(sandbox_enabled=True)) + filesystem = settings["sandbox"]["filesystem"] + assert proposer_sandbox.SANDBOX_REVIEW_OUTPUT in filesystem["allowWrite"] + assert proposer_sandbox.SANDBOX_REVIEW_OUTPUT in filesystem["allowRead"] + assert filesystem["denyRead"] == ["/"] + + +def test_review_contract_tells_the_agent_the_writable_path(): + prompt = runner.REVIEW_PROMPT.format(task="task text") + assert f"{runner.SANDBOX_REVIEW_OUTPUT}/{runner.REVIEW_OUTPUT}" in prompt + assert f"{runner.SANDBOX_WORKSPACE}/{runner.REVIEW_OUTPUT}" not in prompt + # The JSON shape survives .format() with its braces intact. + assert '{"schema_version":1' in prompt + artifact = f"{runner.SANDBOX_REVIEW_OUTPUT}/{runner.REVIEW_OUTPUT}" + assert runner.CE_REVIEW_PROMPT.format(task="task text").count(artifact) == 1 + + +def test_enforce_phase_workspace_can_require_an_untouched_workspace(tmp_path): + (tmp_path / "tracked.py").write_text("original\n") + before = runner_artifacts.workspace_snapshot(tmp_path) + runner_artifacts.enforce_phase_workspace(tmp_path, before, allowed_artifact=None) + + (tmp_path / "tracked.py").write_text("the review edited the code it was reviewing\n") + with pytest.raises(ValueError, match="changed the read-only workspace"): + runner_artifacts.enforce_phase_workspace(tmp_path, before, allowed_artifact=None) diff --git a/eval/tests/test_sanitized_graph.py b/eval/tests/test_sanitized_graph.py index 7b5dfc093..d7f73d756 100644 --- a/eval/tests/test_sanitized_graph.py +++ b/eval/tests/test_sanitized_graph.py @@ -212,6 +212,22 @@ def test_prepare_sanitized_graph_builds_once_from_parentless_tree_and_caches_onl assert removed == [seed] +def test_prepare_sanitized_graph_requires_head_when_given_a_template(tmp_path: Path): + with pytest.raises(SandboxError, match="sanitized HEAD"): + sanitized_graph.prepare_sanitized_graph( + {}, + repo=tmp_path, + resolved_sha="b" * 40, + parent=tmp_path, + cache=SimpleNamespace(), # type: ignore[arg-type] + claude_bin="claude", + bwrap_bin="bwrap", + runtime_mounts=(), + clone_template=tmp_path, + sanitized_head=None, + ) + + def test_graph_snapshot_rejects_arm_sanitization_identity_drift(tmp_path: Path): assets = SimpleNamespace( digest="digest", diff --git a/eval/tests/test_session_progress.py b/eval/tests/test_session_progress.py index 9e59ba4bf..416c6206c 100644 --- a/eval/tests/test_session_progress.py +++ b/eval/tests/test_session_progress.py @@ -11,7 +11,7 @@ import io import json import time -from workflow_bench.runner_sessions import SessionProgress +from workflow_bench.runner_sessions import SessionProgress, neutralize_ci_log_text def _drain_lines(stream: io.StringIO) -> list[str]: @@ -325,3 +325,48 @@ def test_cell_failure_detail_line_bounds_a_huge_detail() -> None: assert line is not None assert "truncated" in line assert len(line) < MAX_CELL_DETAIL_CHARS + 200 + + +def test_progress_neutralizes_github_actions_annotation_forms() -> None: + rewritten = neutralize_ci_log_text( + "gitnexus/src/cli/optional-grammars.ts(18,36): error TS2307: Cannot find module " + "'gitnexus-shared'\n::error::Composite projects may not disable incremental compilation.\n" + "##[error]tsc failed" + ) + assert "): error TS2307" not in rewritten + assert "): compiler-error TS2307" in rewritten + assert "::error::" not in rewritten + assert "[:]error::" in rewritten + assert "##[error]" not in rewritten + assert "# [error]tsc failed" in rewritten + + stream = io.StringIO() + progress = SessionProgress("review-pr-2718-defect-ce_review-run0", stream=stream, heartbeat_s=3600) + events = [ + { + "type": "assistant", + "message": { + "content": [{"type": "tool_use", "id": "b1", "name": "Bash", "input": {"command": "npx tsc --noEmit"}}] + }, + }, + { + "type": "user", + "message": { + "content": [ + { + "type": "tool_result", + "tool_use_id": "b1", + "is_error": True, + "content": "gitnexus/src/cli/optional-grammars.ts(18,36): error TS2307: Cannot find module 'gitnexus-shared'", + } + ] + }, + }, + ] + for event in events: + _observe(progress, (json.dumps(event) + "\n").encode()) + + output = stream.getvalue() + assert "): error TS2307" not in output + assert "): compiler-error TS2307" in output + assert "result=error" in output diff --git a/eval/tests/test_sweep_finalization.py b/eval/tests/test_sweep_finalization.py new file mode 100644 index 000000000..176157645 --- /dev/null +++ b/eval/tests/test_sweep_finalization.py @@ -0,0 +1,279 @@ +"""The real sweep must reach the right finalization decision. + +`enforce_measurement_health` is unit-tested and the call site is pinned +structurally, but neither shows the guard running inside a sweep. These drive +the real `_run_sweep` with cell execution scripted and everything downstream of +it left alone: folding, aggregation, the artifact writers, the health guard and +the exit selection. + +The below-breaker case is the decisive one. A fixture of many unusable cells +aborts through the pre-existing outage breaker instead - `review-evidence-invalid` +is systemic with a limit of 5 - and would pass whether or not the finalization +guard exists. One fresh unusable cell stays under that threshold, so only the +guard can catch it. +""" + +from __future__ import annotations + +import json +import threading +from collections.abc import Callable +from pathlib import Path +from types import SimpleNamespace +from typing import Any + +import pytest + +from tests.bench_fixtures import scored_review_row, unusable_review_row +from workflow_bench import runner + +TASK = { + "id": "review-pr-2718-defect", + "repo": "~/GitNexus", + "ref": "a" * 40, + "prompt": "review it", + "verify": "true", + "class": "review-defect", +} + + +def _args(out: Path, **overrides: Any) -> SimpleNamespace: + values: dict[str, Any] = dict( + arms=["review"], claude_bin="claude", effort="xhigh", model="gpt-5.6-sol", + out=out, outage_streak=runner.DEFAULT_OUTAGE_STREAK, promotion_max_task_regression=10.0, + promotion_metric="review_weighted_f1", promotion_min_improvement=1.0, + promotion_min_runs=1, proposer_model=None, reuse_results=None, runs=1, workers=1, + timeout=60, base_url=None, auth_token=None, permission_mode=None, + ) + values.update(overrides) + return SimpleNamespace(**values) + + +def _snapshot(prefix: str) -> SimpleNamespace: + return SimpleNamespace( + digest=f"{prefix}-content", manifest_digest=f"{prefix}-manifest", + dependency_content_digest=f"{prefix}-dep", dependency_manifest_digest=f"{prefix}-depman", + command_digest=f"{prefix}-command", materialize=lambda *a, **k: None, + ) + + +def _sweep( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, + record: dict[str, Any] | Callable[[int], dict[str, Any]], + *, + runs: int = 1, + cancel_event: threading.Event | None = None, + candidate_arms: list[str] | None = None, + arms: list[str] | None = None, + after_cell: Callable[[int, str], None] | None = None, +): + """Drive the real _run_sweep; only cell execution and setup are scripted. + + ``after_cell`` runs once a cell's record exists, which is how a test sets + cancellation deterministically at a known point instead of racing a sleep. + """ + + out = tmp_path / "out" + + def scripted_cell(_ctx: Any, run_idx: int, arm: str) -> dict[str, Any]: + row = dict(record(run_idx) if callable(record) else record) + row.update({"task": TASK["id"], "arm": arm, "run": run_idx, "class": TASK["class"]}) + if after_cell is not None: + after_cell(run_idx, arm) + return row + + monkeypatch.setattr(runner, "run_cell", scripted_cell) + monkeypatch.setattr(runner, "ensure_task_graph", lambda **k: k["env"].graph_snapshots.__setitem__( + k["graph_key"], _snapshot("graph"))) + monkeypatch.setattr(runner.TaskAssetCache, "prepare", lambda self, *a, **k: _snapshot("asset")) + # Binding resolution clones the repo and verifies the ref; that is expensive + # setup, and the bindings it would return are supplied directly instead. + monkeypatch.setattr( + runner, "resolve_task_bindings", + lambda tasks, expected, **k: list(expected), + ) + + return runner._run_sweep( + _args(out, runs=runs, arms=arms or ["review"]), + parser=SimpleNamespace(error=lambda m: (_ for _ in ()).throw(SystemExit(2))), + tasks=[TASK], + skipped_expensive=[], + oracle_snapshots=[_snapshot("oracle")], + expected_task_bindings=[{"repo_identity": str(tmp_path / "repo"), "resolved_sha": "a" * 40}], + ce_plugin_config=None, + bwrap_bin=Path("/bin/true"), + sandbox_backend="test-double", + runtime_mounts=(), + candidate_arms=candidate_arms or [], + candidate_overlay=None, + overlay_digest=None, + promotion_target_bases={}, + cancel_event=cancel_event, + ), out + + +def test_one_unusable_cell_below_the_breaker_reaches_the_finalization_guard( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture[str] +) -> None: + """The decisive case: too few failures to trip the breaker, so only the guard can catch it.""" + + streak = runner.systemic_outage_streak("review-evidence-invalid", 0) + assert streak < runner.DEFAULT_OUTAGE_STREAK, "fixture must stay under the breaker" + + unusable = unusable_review_row() + with pytest.raises(SystemExit) as exc: + _sweep(tmp_path, monkeypatch, unusable) + assert exc.value.code == 1 + out = capsys.readouterr().out + assert "review: UNUSABLE" in out, "the guard must name the arm and its status" + assert "systemic-outage" not in out, "the breaker must not have tripped" + + +def test_a_zero_score_stays_a_valid_negative_measurement( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture[str] +) -> None: + """0.0 is a present measurement, not missing evidence. + + A truthiness check on the score would misread it as absent and turn a + quality result into an execution-health failure. + """ + + zeroed = scored_review_row( + resolved=False, error_kind="oracle-failed", + review_score={"weighted_f1": 0.0}, review_weighted_f1=0.0, + ) + _sweep(tmp_path, monkeypatch, zeroed) + out = capsys.readouterr().out + assert "review: OBSERVED_OK" in out + assert "UNUSABLE" not in out + + +def test_finalization_persists_results_and_report( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + """Evidence must survive the sweep, and say the same thing the exit does.""" + + scored = scored_review_row(resolved=False, error_kind="oracle-failed", review_weighted_f1=0.2) + _result, out = _sweep(tmp_path, monkeypatch, scored) + rows = [json.loads(line) for line in (out / "results.jsonl").read_text().splitlines()] + assert len(rows) == 1 and rows[0]["review_weighted_f1"] == 0.2 + assert (out / "report.md").is_file() + + +def test_cancellation_without_an_outage_exits_130( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture[str] +) -> None: + """An interrupted sweep is interrupted, not aborted. + + One admissible cell lands first so the measurement-health guard classifies + the arm DEGRADED rather than UNUSABLE - otherwise the guard would supply + exit 1 and this test would pass without ever exercising exit selection. + """ + + cancel_event = threading.Event() + with pytest.raises(SystemExit) as exc: + _sweep( + tmp_path, monkeypatch, lambda _run: scored_review_row(), + runs=3, cancel_event=cancel_event, + after_cell=lambda run_idx, _arm: cancel_event.set() if run_idx == 0 else None, + ) + stdout = capsys.readouterr().out + report = (tmp_path / "out" / "report.md").read_text() + assert "Sweep cancelled" in report, "an interruption must be reported as one" + assert "systemic-outage" not in stdout, "no breaker trip in this scenario" + assert exc.value.code == 130 + + +def test_an_outage_keeps_exit_1_even_though_the_breaker_cancels( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture[str] +) -> None: + """Precedence: the breaker sets cancel_event, so order decides the exit. + + Testing cancellation first would relabel every outage a Ctrl-C. The first + cell is admissible for the same reason as above, and the failures after it + are consecutive and systemic, which is what the breaker actually counts. + """ + + def cell(run_idx: int) -> dict[str, Any]: + return scored_review_row() if run_idx == 0 else unusable_review_row() + + cancel_event = threading.Event() + with pytest.raises(SystemExit) as exc: + _sweep(tmp_path, monkeypatch, cell, + runs=1 + runner.DEFAULT_OUTAGE_STREAK, cancel_event=cancel_event) + stdout = capsys.readouterr().out + report = (tmp_path / "out" / "report.md").read_text() + assert "systemic-outage" in stdout, "the real breaker must have tripped" + assert cancel_event.is_set(), "the breaker cancels in-flight work" + assert "Sweep aborted" in report + assert exc.value.code == 1, "an outage must not become the 130 of a Ctrl-C" + + +def test_an_interrupted_sweep_keeps_the_evidence_it_already_paid_for( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + """Cancellation must not discard rows that already cost money. + + The completed-run persistence test cannot show this: it never interrupts, so + it would pass even if the writer only ran on the clean path. + """ + + cancel_event = threading.Event() + with pytest.raises(SystemExit): + _sweep( + tmp_path, monkeypatch, lambda _run: scored_review_row(review_weighted_f1=0.42), + runs=3, cancel_event=cancel_event, + after_cell=lambda run_idx, _arm: cancel_event.set() if run_idx == 0 else None, + ) + rows = [ + json.loads(line) + for line in (tmp_path / "out" / "results.jsonl").read_text().splitlines() + ] + assert len(rows) == 1, "the cell that completed before cancellation must survive" + assert rows[0]["review_weighted_f1"] == 0.42, "its measurement must survive intact" + assert (tmp_path / "out" / "report.md").is_file() + + +def test_an_interrupted_sweep_emits_nothing_that_authorizes_promotion( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + """The semantic condition, not the absence of a file. + + promotion.json is still written for an aborted run - it is the record of why + nothing was promoted. What must hold is that nothing in it authorizes a + promotion from partial evidence. + """ + + cancel_event = threading.Event() + with pytest.raises(SystemExit): + _sweep( + tmp_path, monkeypatch, lambda _run: scored_review_row(), + runs=3, cancel_event=cancel_event, + # The candidate arm has to RUN, not merely appear in promotion + # metadata: _run_sweep builds cells only from args.arms, so naming it + # in candidate_arms alone left the candidate with no results at all - + # and then "insufficient_evidence" would hold because nothing ran, + # not because partial evidence is barred from promoting. + arms=["review", "candidate_review"], + candidate_arms=["candidate_review"], + after_cell=( + lambda run_idx, arm: cancel_event.set() + if run_idx == 0 and arm == "candidate_review" + else None + ), + ) + rows = [ + json.loads(line) + for line in (tmp_path / "out" / "results.jsonl").read_text().splitlines() + ] + assert any(r["arm"] == "candidate_review" for r in rows), ( + "the candidate must have produced evidence, or insufficient_evidence " + "would hold merely because nothing ran" + ) + promotion = json.loads((tmp_path / "out" / "promotion.json").read_text()) + assert promotion["run_status"] == "aborted" + assert promotion["decisions"], "an aborted run still has to say what it decided" + for decision in promotion["decisions"]: + assert decision["decision"] == "insufficient_evidence" + assert any("partial evidence" in reason for reason in decision["reasons"]) diff --git a/eval/tests/test_workflow_bench.py b/eval/tests/test_workflow_bench.py index d9c9c0f92..a53b2b8cd 100644 --- a/eval/tests/test_workflow_bench.py +++ b/eval/tests/test_workflow_bench.py @@ -14,18 +14,26 @@ import yaml from typing import Any from workflow_bench import runner +from workflow_bench.evolution import CANDIDATE_ARMS from workflow_bench.process_control import _CANCELLATION, cancellation_scope from workflow_bench.runner import ( aggregate, + GraphBuildEnv, + arm_health, broken_incumbent_arms, + unhealthy_arms, + unmeasured_arms, build_parser, infra_error_record, + next_graph_prefetch_target, normalized_model_identifier, parse_shortstat, + prefetch_next_graph, render_report, savings, select_tasks, systemic_outage_streak, + task_has_planned_paid_cells, ) @@ -68,6 +76,15 @@ def test_aggregate_takes_medians_and_counts_resolved(): "diff_deletions": 5, "class": "demo", "resolved": 2, + # None of these are reused, so every resolution was measured this sweep. + "resolved_fresh": 2, + # Health is counted separately from resolution: all three executed and + # produced usable evidence, including the one that resolved nothing. + "fresh_attempts": 3, + "admissible": 3, + "execution_failures": 0, + "evidence_failures": 0, + "health_reasons": [], "runs": 3, "valid_runs": 3, "excluded_runs": 0, @@ -250,6 +267,9 @@ def test_shipped_scenarios_opt_out_the_cross_module_cell_and_rebuild_graph_asset assert skipped == ["cross-module-parse-retry"] assert all(not task.get("sandbox_copy") for task in tasks) assert all(task["sandbox_dependencies"] for task in tasks) + assert all( + any(dep.get("source") == "gitnexus-shared/dist" for dep in task["sandbox_dependencies"]) for task in tasks + ) assert all(task["oracle"]["command"] and task["oracle"]["files"] for task in tasks) assert all("./node_modules/.bin/vitest run" in task["oracle"]["command"] for task in tasks) assert all("npx vitest" not in task["oracle"]["command"] for task in tasks) @@ -519,6 +539,369 @@ def test_run_evolution_script_is_the_shared_ci_and_local_entrypoint(): assert printed.stderr # rewrite notice goes to stderr +def test_planned_paid_cells_treat_missing_reuse_as_paid(): + task = {"id": "review-pr-2718-defect"} + assert task_has_planned_paid_cells( + task, + arms=["ce_review", "review", "candidate_review"], + runs=3, + reusable_rows={}, + reuse_source=None, + ) + reuse_source = Path("/tmp/seed") + rows = { + (task["id"], arm, run_idx): {} + for run_idx in range(3) + for arm in ("ce_review", "review", "candidate_review") + } + assert not task_has_planned_paid_cells( + task, + arms=["ce_review", "review", "candidate_review"], + runs=3, + reusable_rows=rows, + reuse_source=reuse_source, + ) + del rows[(task["id"], "candidate_review", 0)] + assert task_has_planned_paid_cells( + task, + arms=["ce_review", "review", "candidate_review"], + runs=3, + reusable_rows=rows, + reuse_source=reuse_source, + ) + + +def test_next_graph_prefetch_skips_ready_shas_and_fully_reused_tasks(tmp_path: Path): + first = {"id": "review-a"} + second = {"id": "review-b"} + third = {"id": "review-c"} + reuse_source = tmp_path / "seed" + reused_second = { + (second["id"], arm, 0): {} for arm in ("ce_review", "review", "candidate_review") + } + target = next_graph_prefetch_target( + [ + (first, {"repo_identity": "/repo", "resolved_sha": "aaa"}), + (second, {"repo_identity": "/repo", "resolved_sha": "bbb"}), + (third, {"repo_identity": "/repo", "resolved_sha": "ccc"}), + ], + arms=["ce_review", "review", "candidate_review"], + runs=1, + reusable_rows=reused_second, + reuse_source=reuse_source, + ready_keys={("/repo", "aaa")}, + ) + assert target is not None + task, binding, key = target + assert task["id"] == "review-c" + assert key == ("/repo", "ccc") + assert binding["resolved_sha"] == "ccc" + + +def test_prefetch_next_graph_runs_ensure_on_a_background_thread(monkeypatch): + started = threading.Event() + seen: list[tuple[str, str]] = [] + + def fake_ensure(**kwargs): + seen.append(kwargs["graph_key"]) + started.set() + + monkeypatch.setattr("workflow_bench.runner.ensure_task_graph", fake_ensure) + cancel = threading.Event() + job = prefetch_next_graph( + task={"id": "review-b"}, + binding={"repo_identity": "/repo", "resolved_sha": "bbb"}, + graph_key=("/repo", "bbb"), + env=GraphBuildEnv( + trees=Path("/tmp"), + task_asset_cache=None, + claude_bin="claude", + bwrap_bin="bwrap", + sandbox_backend="bwrap", + runtime_mounts=(), + clone_templates={}, + clone_template_errors={}, + graph_snapshots={}, + graph_snapshot_errors={}, + ), + cancel_event=cancel, + ) + assert job.key == ("/repo", "bbb") + assert started.wait(timeout=2) + job.join() + assert seen == [("/repo", "bbb")] + + +def test_a_reused_resolution_does_not_count_as_this_sweeps_health(): + """resolved counts evidence; resolved_fresh counts evidence measured today. + + broken_incumbent_arms reads resolved_fresh because a reused row proves last + generation's environment worked. Counting it would make an arm whose cells + were all reused look healthy in exactly the run where a broken environment + should have been caught. + """ + + reused = [record(resolved=True, reused=True), record(resolved=True, reused=True)] + agg = aggregate(reused) + assert agg["resolved"] == 2 + assert agg["resolved_fresh"] == 0 + assert broken_incumbent_arms({"t": {"review": agg}}, {"review"}) == ["review"] + + mixed = aggregate([record(resolved=True, reused=True), record(resolved=True)]) + assert mixed["resolved_fresh"] == 1 + assert broken_incumbent_arms({"t": {"review": mixed}}, {"review"}) == [] + + +def test_graph_build_env_ready_keys_covers_successes_and_failures(): + """A key that failed is attempted, not pending. + + next_graph_prefetch_target skips keys already in ready_keys. If a failed + build were omitted, the sweep would prefetch it again every iteration and + pay a full clone and offline index each time for a build that cannot + succeed. + """ + + env = GraphBuildEnv( + trees=Path("/tmp"), + task_asset_cache=None, + claude_bin="claude", + bwrap_bin="bwrap", + sandbox_backend="bwrap", + runtime_mounts=(), + clone_templates={("/repo", "aaa"): (Path("/tmp/a"), "aaa")}, + clone_template_errors={("/repo", "bbb"): OSError("clone failed")}, + graph_snapshots={("/repo", "ccc"): object()}, + graph_snapshot_errors={("/repo", "ddd"): OSError("index failed")}, + ) + assert env.ready_keys() == { + ("/repo", "aaa"), + ("/repo", "bbb"), + ("/repo", "ccc"), + ("/repo", "ddd"), + } + + +def _cell(**overrides) -> dict[str, Any]: + """One results.jsonl row, healthy unless told otherwise.""" + + base = record(resolved=True) + base.update({"error_kind": None, "review_evidence_valid": True, "transcript_missing": False}) + base.update(overrides) + return base + + +def _arms(**by_arm) -> dict[str, dict[str, dict[str, Any]]]: + return {"task0": {arm: aggregate(rows) for arm, rows in by_arm.items()}} + + +def test_a_reviewer_that_scores_badly_is_not_an_unhealthy_harness(): + """Reconstructed from Actions run 33962002890's logged observations. + + Every completed cell was resolved=False with error_kind=oracle-failed, at a + median score of 0.212 — the reviews ran, wrote artifacts and were scored. + That is a valid negative for the quality gate to judge. Diagnosing it as a + broken environment is the confusion this classification exists to end. + """ + + scored_but_wrong = [_cell(resolved=False, error_kind="oracle-failed") for _ in range(3)] + results = _arms(review=scored_but_wrong, ce_review=list(scored_but_wrong)) + assert unhealthy_arms(results, {"review", "ce_review"}) == [] + health = arm_health(results, {"review"})["review"] + assert health.admissible == 3 and health.fresh_attempts == 3 + assert (health.execution_failures, health.evidence_failures) == (0, 0) + + +def test_an_all_zero_score_is_still_a_valid_negative(): + zeroed = [_cell(resolved=False, error_kind="oracle-failed", review_weighted_f1=0.0) for _ in range(3)] + assert unhealthy_arms(_arms(review=zeroed), {"review"}) == [] + + +def test_artifacts_that_were_never_written_are_an_unhealthy_harness(): + """Reconstructed from Actions run 33912693948. + + All 41 artifacts came back 0 bytes because the mount made an atomic write + impossible. The reviews could not produce evidence at all — the opposite of + the case above, and the one a health check must catch. The old caller + excluded review arms entirely, so it could not have. + """ + + unwritable = [_cell(resolved=False, ok=False, error_kind="review-evidence-invalid") for _ in range(3)] + flagged = unhealthy_arms(_arms(review=unwritable), {"review"}) + assert [h.arm for h in flagged] == ["review"] + assert flagged[0].evidence_failures == 3 + assert "review-evidence-invalid" in flagged[0].reasons + + +def test_one_admissible_cell_leaves_an_arm_degraded_not_healthy(): + """Mixed outcomes are DEGRADED. One usable measurement does not erase two failures. + + Not fatal - the sweep still produced evidence - but calling it healthy is + how a partly-broken environment passes review. + """ + + mixed = [ + _cell(resolved=False, error_kind="oracle-failed"), + _cell(resolved=False, ok=False, error_kind="session-error"), + _cell(resolved=False, ok=False, error_kind="infra-error"), + ] + results = _arms(review=mixed) + health = arm_health(results, {"review"})["review"] + assert health.status == "DEGRADED" + assert unhealthy_arms(results, {"review"}) == [], "degraded is diagnostic, not fatal" + assert health.execution_failures == 2, "failures must stay visible, not be erased" + assert health.admissible == 1 + + +def test_a_row_that_fails_both_ways_is_only_subtracted_once(): + """run_arm can produce a row that is an execution AND an evidence failure. + + It keeps the first error_kind — a session-error survives — and still sets + review_evidence_valid=False when the artifact will not parse. Counting that + row against admissible twice zeroed an arm that held a real measurement, + which arm_health reports as UNUSABLE and the measurement gate then fails on. + """ + + both = _cell(resolved=False, ok=False, error_kind="session-error", review_evidence_valid=False) + results = _arms(review=[both, _cell(resolved=True, error_kind="oracle-failed")]) + health = arm_health(results, {"review"})["review"] + assert (health.execution_failures, health.evidence_failures) == (1, 1) + assert health.fresh_attempts == 2 + assert health.admissible == 1 + assert health.status == "DEGRADED" + assert unhealthy_arms(results, {"review"}) == [] + + +def test_reused_rows_alone_leave_current_health_unknown(): + """Historical success cannot certify this sweep's environment.""" + + reused = [_cell(reused=True) for _ in range(3)] + results = _arms(review=reused) + assert unmeasured_arms(results, {"review"}) == ["review"] + assert unhealthy_arms(results, {"review"}) == [] + assert arm_health(results, {"review"})["review"].measured is False + + +def test_the_paid_canary_survives_a_prior_run_with_more_run_indices(): + """The canary counts planned cells, not every key reuse selection returned. + + Reuse selection accepts any non-negative prior `run`, so a results directory + produced with --runs 5 leaves keys this sweep never plans. Comparing against + those made the "arm is fully reused" test false exactly when it was true, + and the incumbent went a whole sweep without one measured cell. + """ + + tasks = [{"id": "task0"}, {"id": "task1"}] + reusable = {(task["id"], "review", run): {} for task in tasks for run in range(5)} + + dropped = runner.drop_canary_reuse_key(reusable, arm="review", tasks=tasks, runs=3) + + assert dropped == ("task0", "review", 0) + assert dropped not in reusable + # A second call is a no-op: the arm now has its paid cell. + assert runner.drop_canary_reuse_key(reusable, arm="review", tasks=tasks, runs=3) is None + + +def test_an_arm_with_a_planned_paid_cell_keeps_every_reusable_row(): + tasks = [{"id": "task0"}] + reusable = {("task0", "review", 0): {}} + + assert runner.drop_canary_reuse_key(reusable, arm="review", tasks=tasks, runs=2) is None + assert len(reusable) == 1 + + +def test_reused_successes_do_not_mask_fresh_execution_failures(): + rows = [_cell(reused=True), _cell(reused=True), _cell(ok=False, error_kind="session-error")] + flagged = unhealthy_arms(_arms(review=rows), {"review"}) + assert [h.arm for h in flagged] == ["review"] + assert flagged[0].fresh_attempts == 1 and flagged[0].execution_failures == 1 + + +def test_a_parseable_artifact_does_not_excuse_a_failed_session(): + """Artifact parseability must not override an execution failure.""" + + rows = [_cell(ok=False, error_kind="session-error", review_evidence_valid=True) for _ in range(2)] + flagged = unhealthy_arms(_arms(review=rows), {"review"}) + assert [h.arm for h in flagged] == ["review"] + assert flagged[0].execution_failures == 2 + + +def test_a_single_unusable_review_is_caught_below_the_breaker_threshold(): + """The decisive regression for the finalization guard. + + A fixture of 41 empty artifacts would abort through the outage breaker - + review-evidence-invalid is systemic and the limit is 5 - so it proves + nothing about this path. One fresh unusable cell is under that threshold, + which leaves the finalization check as the only thing that can catch it. + """ + + streak = 0 + for _ in range(1): + streak = runner.systemic_outage_streak("review-evidence-invalid", streak) + assert streak < runner.DEFAULT_OUTAGE_STREAK, "fixture must not reach the breaker" + + results = _arms(review=[_cell(resolved=False, ok=False, error_kind="review-evidence-invalid")]) + with pytest.raises(SystemExit) as exc: + runner.enforce_measurement_health(results, {"review"}) + assert exc.value.code == 1 + + +def test_finalization_reports_every_arm_and_names_no_cause(capsys): + """Status for each arm; an empty artifact does not become an EROFS diagnosis.""" + + results = _arms( + review=[_cell(resolved=False, ok=False, error_kind="review-evidence-invalid")], + ce_review=[_cell(resolved=False, error_kind="oracle-failed")], + ) + with pytest.raises(SystemExit): + runner.enforce_measurement_health(results, {"review", "ce_review"}) + out = capsys.readouterr().out + assert "review: UNUSABLE" in out + assert "ce_review: OBSERVED_OK" in out + assert "cause=undetermined" in out + assert "EROFS" not in out and "mount" not in out + + +def test_valid_negatives_do_not_abort_finalization(capsys): + """The 16h run's shape must survive the real guard, not just the classifier.""" + + scored_but_wrong = [_cell(resolved=False, error_kind="oracle-failed") for _ in range(3)] + health = runner.enforce_measurement_health( + _arms(review=scored_but_wrong, ce_review=list(scored_but_wrong)), {"review", "ce_review"} + ) + assert {h.status for h in health.values()} == {"OBSERVED_OK"} + assert "UNUSABLE" not in capsys.readouterr().out + + +def test_reused_only_arm_is_reported_unknown_by_finalization(capsys): + runner.enforce_measurement_health(_arms(review=[_cell(reused=True)]), {"review"}) + assert "review: UNKNOWN" in capsys.readouterr().out + + +def test_run_sweep_calls_the_health_guard_and_not_the_legacy_helper(): + """Pins the wiring the caller correction exposed. + + Reads the compiled code object's global references rather than the source + text: deleting the call removes the name and fails this test, which is the + mutation check. It does NOT prove the guard runs end to end - _run_sweep + needs bwrap and a sandbox, so no test here drives it. + """ + + referenced = runner._run_sweep.__code__.co_names + assert "enforce_measurement_health" in referenced + assert "broken_incumbent_arms" not in referenced + + +def test_ce_review_is_classified_even_though_it_is_not_a_candidate_arm(): + """ce_review is a comparator, absent from CANDIDATE_ARMS. + + Dropping the `- {"review"}` exclusion alone would have left it unchecked. + """ + + assert "ce_review" not in set(CANDIDATE_ARMS.values()) + health = arm_health(_arms(ce_review=[_cell()]), {"review", "ce_review"}) + assert "ce_review" in health + + def _packed_cells(tasks: int, runs: int, arms: tuple[str, ...]) -> list[tuple[str, int, str]]: return [(f"t{t}", r, a) for t in range(tasks) for r in range(runs) for a in arms] diff --git a/eval/tests/test_workflow_bench_sessions.py b/eval/tests/test_workflow_bench_sessions.py index 1f66f071f..5d1692fbb 100644 --- a/eval/tests/test_workflow_bench_sessions.py +++ b/eval/tests/test_workflow_bench_sessions.py @@ -12,9 +12,9 @@ from types import SimpleNamespace import pytest -from workflow_bench import evolve, runner, runner_sessions, runtime_mounts +from workflow_bench import evolve, runner, runner_artifacts, runner_sessions, runtime_mounts from workflow_bench.evolution import skill_fingerprint -from workflow_bench.process_control import ManagedProcessResult +from workflow_bench.process_control import ManagedProcessError, ManagedProcessResult from workflow_bench.proposer_sandbox import SandboxError from workflow_bench.runner import snapshot_plan_docs @@ -120,11 +120,16 @@ def skill_events(skill_input: dict, *, tool_id: str = "skill-1", is_error: bool def fake_sandbox(root: Path) -> SimpleNamespace: + # private_root is NOT the clone. Conflating them puts the review artifact + # directory inside the workspace, which the real sandbox never does and + # which hides whether the workspace was left untouched. + private_root = root.parent / f"{root.name}-sandbox-private" + private_root.mkdir(exist_ok=True) return SimpleNamespace( backend="test-double", claude_bin="claude", clone=root, - private_root=root, + private_root=private_root, command_prefix=[], command_prefix_for=lambda **_kwargs: [], settings_json="{}", @@ -1249,7 +1254,7 @@ def test_planning_cannot_change_source_tests_or_downstream_skill(monkeypatch, tm @pytest.mark.parametrize( ("attack", "expected_detail"), [ - ("workspace", "unauthorized workspace path"), + ("workspace", "changed the read-only workspace"), ("skill", "changed the evaluated skill fingerprint"), ], ) @@ -1265,9 +1270,11 @@ def test_review_phase_rejects_workspace_or_skill_mutation( expected_skill_digest = "expected-skill-fingerprint" def adversarial_review(prompt, *args, **kwargs): - (tmp_path / "review-output.json").write_text( - '{"schema_version":1,"verdict":"approve","findings":[]}' - ) + # Write where the contract now says: the artifact directory outside the + # workspace, which is the only place the agent can write atomically. + artifact = runner.review_output_path(sandbox, runner.REVIEW_OUTPUT) + artifact.parent.mkdir(parents=True, exist_ok=True) + artifact.write_text('{"schema_version":1,"verdict":"approve","findings":[]}') if attack == "workspace": source.write_text("review silently changed source") return session_record() @@ -1298,6 +1305,97 @@ def test_review_phase_rejects_workspace_or_skill_mutation( assert expected_detail in rec["error_detail"] +def test_a_cancelled_clone_copy_does_not_fall_back_to_an_uncancellable_copytree(monkeypatch, tmp_path): + """The reflink fallback is for a filesystem, not for a teardown. + + run_managed reports cancellation as a non-OK result rather than raising, so + the fallback treated it like an unsupported reflink and started a copytree + that cannot be cancelled — waiting out exactly the full copy the outage + breaker set the cancellation event to avoid. + """ + + source = tmp_path / "template" + (source / ".git").mkdir(parents=True) + parent = tmp_path / "clones" + parent.mkdir() + copied: list[object] = [] + + monkeypatch.setattr( + runner_artifacts, + "run_managed", + lambda *_a, **_k: ManagedProcessResult( + state="cancelled", + returncode=None, + stdout_tail="", + stderr_tail="", + duration_s=0.1, + ), + ) + monkeypatch.setattr(runner_artifacts.shutil, "copytree", lambda *a, **k: copied.append(a)) + + with pytest.raises(ManagedProcessError): + runner.copy_isolated_tree(source, parent) + assert copied == [] + assert list(parent.iterdir()) == [], "the partial target must be cleaned up" + + +@pytest.mark.parametrize("arm", ["review", "ce_review"]) +def test_run_arm_mounts_the_review_artifact_directory_outside_the_workspace(monkeypatch, tmp_path, arm): + """A writable FILE inside a read-only directory is not a writable path. + + The Write tool creates `.tmp..` beside the target and + renames it, so a read-only parent fails the temp create with EROFS and the + artifact stays 0 bytes. The mount target must be the directory, and it must + sit outside the read-only workspace. + + Driven through run_arm rather than rebuilt here: an expected tuple assembled + in the test passes whatever run_arm actually mounts, which is the one thing + this needs to prove. + """ + + assert not runner.SANDBOX_REVIEW_OUTPUT.startswith(runner.SANDBOX_WORKSPACE + "/") + assert runner.SANDBOX_REVIEW_OUTPUT != runner.SANDBOX_WORKSPACE + + verify_calls: list[dict] = [] + sandbox = fake_sandbox(tmp_path) + sandbox.command_prefix_for = lambda **kwargs: verify_calls.append(kwargs) or [] + + def review_session(prompt, *args, **kwargs): + artifact = runner.review_output_path(sandbox, runner.REVIEW_OUTPUT) + artifact.parent.mkdir(parents=True, exist_ok=True) + artifact.write_text('{"schema_version":1,"verdict":"approve","findings":[]}') + return session_record() + + monkeypatch.setattr(runner, "run_claude", review_session) + monkeypatch.setattr(runner, "skill_fingerprint", lambda *_a, **_k: "skill-digest") + monkeypatch.setattr(runner, "run_verify", lambda *a, **k: (True, "ok")) + + runner.run_arm( + arm, + {"prompt": "p", "verify": "true"}, + tmp_path, + bench_args(), + sandbox=sandbox, + expected_skill_digest="skill-digest", + ) + + review_output = runner.review_output_path(sandbox, runner.REVIEW_OUTPUT) + expected = (runner.ReadOnlyMount(source=review_output.parent, target=runner.SANDBOX_REVIEW_OUTPUT),) + # The EROFS bug is about the AGENT's write, so the mount that has to be the + # directory is the writable one on the review session — not the read-only + # exposure the verify command gets afterwards. Assert both: they are + # separate arguments to separate command prefixes. + writable = [call["extra_writable_mounts"] for call in verify_calls if "extra_writable_mounts" in call] + assert writable, "the review session must be given a writable artifact mount" + assert writable[-1] == expected, "mount the directory, not the file" + read_only = [call["extra_read_only_mounts"] for call in verify_calls if "extra_read_only_mounts" in call] + assert read_only, "the verify invocation must be given the artifact mount" + assert read_only[-1] == expected, "mount the directory, not the file" + assert not expected[0].target.startswith(f"{runner.SANDBOX_WORKSPACE}/") + # The artifact the harness later reads is the one inside that mount. + assert review_output.parent in review_output.parents + + def _git(repo, *args, check=True): return subprocess.run(["git", "-C", str(repo), *args], check=check, capture_output=True, text=True) @@ -1348,3 +1446,34 @@ def test_make_worktree_clone_has_no_tags_but_keeps_all_branches(tmp_path): current = _git(target, "rev-parse", "HEAD").stdout.strip() assert current == other_sha + + +def test_copy_isolated_tree_does_not_share_git_objects_or_refs(tmp_path): + repo = tmp_path / "repo" + repo.mkdir() + _git(repo, "init", "--quiet") + _git(repo, "checkout", "--quiet", "-b", "main") + sha = _git_commit(repo, "base") + clones = tmp_path / "clones" + clones.mkdir() + template = runner.make_worktree(repo, sha, clones) + (template / "marker.txt").write_text("template\n") + + copy = runner.copy_isolated_tree(template, clones) + assert copy != template + assert (copy / "marker.txt").read_text() == "template\n" + (copy / "marker.txt").write_text("copy\n") + assert (template / "marker.txt").read_text() == "template\n" + copy_head = _git(copy, "rev-parse", "HEAD").stdout.strip() + template_head = _git(template, "rev-parse", "HEAD").stdout.strip() + assert copy_head == template_head == sha + # An equal initial HEAD is also what a shared ref namespace looks like, so + # write a ref and prove the template cannot see it. A linked worktree would + # pass every assertion above, including the alternates check — its `.git` is + # a file, so the directory inspected below simply does not exist. + _git(copy, "branch", "copy-only") + assert _git(copy, "show-ref", "--verify", "refs/heads/copy-only").returncode == 0 + assert _git(template, "show-ref", "--verify", "refs/heads/copy-only", check=False).returncode != 0 + assert (copy / ".git").is_dir() + alternates = copy / ".git" / "objects" / "info" / "alternates" + assert not alternates.exists() diff --git a/eval/workflow_bench/README.md b/eval/workflow_bench/README.md index 79bd4bfea..a0008fa7d 100644 --- a/eval/workflow_bench/README.md +++ b/eval/workflow_bench/README.md @@ -149,8 +149,10 @@ UNSAFE_NO_BWRAP=1 RUNS=1 ./workflow_bench/run-evolution.sh This mode runs review sessions directly in disposable host worktrees and is **not** a security boundary: it does not isolate the network or create a PID namespace, and a session that can `chmod` can undo the workspace lock. The -harness still drops write bits on the clone except `review-output.json` so -accidental `npm install` / analyze writes cannot invalidate review evidence. +harness drops write bits on the whole clone, with no carve-out, so accidental +`npm install` / analyze writes cannot invalidate review evidence. The review +artifact is not in the clone at all: it lives in a writable directory bound at +`/review-output`, outside the workspace. Sandbox cleanup restores owner write bits before deleting the private TMPDIR, because a session that `copytree`s the locked clone would otherwise leave non-empty 0555 directories that `rmtree` cannot remove. Historical review @@ -168,6 +170,36 @@ router thresholds as an incumbent policy, not permanent truth. Candidate changes run offline in the same throwaway clones as the incumbent; production skills never rewrite themselves from a live task. +On the self-hosted evolution box, `run-evolution.sh` passes +`--max-runtime-from-instance-window` and the CLI derives its own cap from +`/proc/uptime` at startup (24h EventBridge window minus a 90-minute upload +reserve), in the same breath as it starts the clock that cap is measured +against — a budget computed anywhere earlier is spent by the seconds between. A `workflow_dispatch` that lands on an +already-running instance therefore exits in-process instead of vanishing when +the box stops — a cancelled GitHub job skips even `if: always()`, which is +how run 33962002890 lost 51 finished sessions. Local runs are uncapped. + +A review generation is 6 tasks × 3 arms × 3 runs. Serial workers=1 at ~19 +minutes per session is a 16-hour job (run 33962002890). Two harness changes +cut that without shrinking the gate: + +- **Comparator reuse.** `evolve.py` forwards the seed / prior generation as + `--reuse-results`. Incumbent `review` and `ce_review` rows are copied into + the new `results.jsonl` when model, effort, task SHA, prompt digest, oracle + bytes, incumbent skill digest, CE plugin digest, and sandbox backend still + match. Candidate arms always run. A weekly generation with an unchanged + incumbent therefore pays 18 sessions, not 54. A promotion, model change, + task-corpus change, or harness `RUNTIME_DIGEST` change invalidates the + lock and re-runs the comparators. +- **Sanitized clone templates.** Each unique task SHA is cloned and + sanitized once. Cells copy that parentless snapshot (reflink when the + filesystem allows) instead of `git clone --no-local` plus repack/prune/fsck + 54 times. Isolation is a private `.git`, not a second copy of full history. + +Dispatch defaults to `--workers 3` so those 18 paid cells can overlap. Size +workers to the host: a cell that loses CPU and hits the session ceiling is +an excluded run the gate refuses. + Build an overlay that mirrors only the canonical repo-local skill paths: ```text @@ -276,9 +308,12 @@ without weakening today's deterministic promotion boundary. The evolution workflow runs an offline containment preflight with the pinned Claude Code 2.1.214 binary before starting a paid proposer or benchmark. The -review canary seals the workspace read-only and exposes only the pre-created -`review-output.json` as writable. Runtime mount placeholders are prepared in -the disposable clone before sealing it; existing config bytes are preserved. +review canary seals the workspace read-only and writes nothing into it: the +artifact directory is bound at `/review-output` outside the workspace, and the +file itself is deliberately absent until the session creates it, so its absence +distinguishes "never written" from "written badly". Runtime mount placeholders +are prepared in the disposable clone before sealing it; existing config bytes +are preserved. Any pre-existing result entry, including a symlink, is rejected. Required canaries fail when their runtime or Bubblewrap is unavailable. @@ -368,10 +403,10 @@ paired benchmark as any other candidate. For ad-hoc use, run the driver on the existing re-evaluation triggers (model/harness change or 90-day staleness). The repository workflow runs a -deliberate weekly drift check: scheduled concurrency stays serial unless -`GITNEXUS_EVOLUTION_WORKERS` is raised after a funded host-sized proof, and -`--workers` is bounded to 1–8 before paid work starts. `--generations` remains -the only loop bound. +deliberate weekly drift check: dispatch defaults to three concurrent cells +of one task; scheduled concurrency still requires +`GITNEXUS_EVOLUTION_WORKERS=3` after a clean proof. `--workers` is bounded +to 1–8 before paid work starts. `--generations` remains the only loop bound. ## Free-model setup (no paid tokens) diff --git a/eval/workflow_bench/comparator_reuse.py b/eval/workflow_bench/comparator_reuse.py new file mode 100644 index 000000000..5dcef1289 --- /dev/null +++ b/eval/workflow_bench/comparator_reuse.py @@ -0,0 +1,604 @@ +"""Reuse frozen comparator cells when the current sweep is still the same experiment. + +Weekly skill evolution re-runs incumbent ``review`` / ``ce_review`` (and the +implementation incumbents) even when the model, effort, tasks, oracles, +incumbent skill bytes, and CE plugin have not changed. Those arms are the +baseline the gate compares a *new* candidate against — they are not the +thing being evolved. Replaying them burns two-thirds of a generation. + +This module selects prior ``results.jsonl`` rows that are safe to carry +forward. Candidate arms are never reused. A mismatch on any bound field +falls through to a paid cell. Missing artifacts also fall through: a reused +row that the proposer cannot read is worse than spending the tokens again. +""" + +from __future__ import annotations + +import hashlib +import json +import os +import re +import stat +from collections.abc import Iterator, Mapping, Sequence +from contextlib import contextmanager +from dataclasses import dataclass +from datetime import UTC, datetime, timedelta +from pathlib import Path, PurePosixPath +from typing import Any + +from .evolution import CANDIDATE_ARMS, EVIDENCE_MAX_AGE_DAYS +from .proposer_sandbox import SandboxError +from .runner_sessions import MAX_TRANSCRIPT_BYTES, PARENT_EVENT_STREAM_SOURCE +from .runtime_mounts import CE_ARMS +from .task_assets import COPY_CHUNK_BYTES, _write_all + +REUSABLE_COMPARATOR_ARMS = frozenset( + { + "review", + "ce_review", + "workflow", + "workflow_direct", + "ce_workflow", + "ce_workflow_direct", + "baseline", + "baseline_nomcp", + } +) +# Must stay aligned with runner.EXCLUDED_ERROR_KINDS plus review-invalid. +# A reused row becomes promotion evidence; excluded kinds cannot enter that set. +REUSE_EXCLUDED_ERROR_KINDS = frozenset( + { + "session-error", + "infra-error", + "evidence-unverified", + "cleanup-failure", + "review-evidence-invalid", + "cancelled", + } +) +_TRANSCRIPT_NAME = re.compile(r"[A-Za-z0-9._-]{1,200}") +CellKey = tuple[str, str, int] + + +@dataclass(frozen=True) +class TaskReuseBinding: + """Per-task identity the prior row must still match.""" + + task_base_sha: str + task_prompt_digest: str + oracle_digest: str + oracle_command_digest: str + oracle_manifest_digest: str + # The cell's environment is part of its identity: a comparator measured + # against different task assets or different sandbox dependencies is a + # measurement of a different machine, not a baseline for this sweep. + task_asset_manifest_digest: str | None = None + sandbox_dependency_manifest_digest: str | None = None + + +@dataclass(frozen=True) +class ComparatorReuseExpectation: + """Sweep-wide lock for comparator reuse. Any drift pays for a fresh cell.""" + + model: str + effort: str + sandbox_backend: str + runtime_digest: str | None + now: datetime + max_age: timedelta + tasks: Mapping[str, TaskReuseBinding] + skill_digests: Mapping[str, str | None] + ce_plugin_version: str | None + ce_plugin_manifest_digest: str | None + + +def load_result_rows(path: Path) -> list[dict[str, Any]]: + """Load ``results.jsonl``; skip malformed lines the same way evolve does.""" + + rows: list[dict[str, Any]] = [] + for line in path.read_text().splitlines(): + if not line.strip(): + continue + try: + row = json.loads(line) + except json.JSONDecodeError: + continue + if isinstance(row, dict): + rows.append(row) + return rows + + +def current_runtime_digest() -> str | None: + """Harness lockfile digest exported by ``run-evolution.sh``, if present.""" + + value = os.environ.get("RUNTIME_DIGEST", "").strip() + return value or None + + +def row_is_reusable_comparator(row: Mapping[str, Any], expected: ComparatorReuseExpectation) -> bool: + """True when ``row`` is a complete, still-valid comparator measurement.""" + + arm = row.get("arm") + if not isinstance(arm, str) or arm in CANDIDATE_ARMS or arm not in REUSABLE_COMPARATOR_ARMS: + return False + if row.get("error_kind") in REUSE_EXCLUDED_ERROR_KINDS: + return False + if row.get("error_kind") not in (None, ""): + return False + if row.get("ok") is not True: + return False + if row.get("transcript_missing") is True: + return False + if row.get("candidate_overlay_digest") not in (None, ""): + return False + # Age against the ORIGINAL measurement, not the copy time: materialize_reused_row + # restamps recorded_at, so a chained row would otherwise refresh its own clock + # and never expire. Bound both directions - a future stamp is corrupt, not fresh. + recorded = _parse_recorded_at(row.get("reused_from_recorded_at") or row.get("recorded_at")) + if recorded is None: + return False + age = expected.now - recorded + if age > expected.max_age or age < timedelta(0): + return False + if row.get("model") != expected.model and row.get("benchmark_model") != expected.model: + return False + if row.get("effort") != expected.effort: + return False + if row.get("sandbox_backend") != expected.sandbox_backend: + return False + # Fail closed. A row with no runtime_digest was measured by a harness that + # did not record one, which is exactly the drift this lock exists to catch; + # treating the absence as agreement made every legacy row reusable forever. + prior_runtime = row.get("runtime_digest") + if not isinstance(prior_runtime, str) or not prior_runtime: + return False + if not expected.runtime_digest or prior_runtime != expected.runtime_digest: + return False + + task_id = row.get("task") + binding = expected.tasks.get(task_id) if isinstance(task_id, str) else None + if binding is None: + return False + if row.get("task_base_sha") != binding.task_base_sha: + return False + if row.get("task_prompt_digest") != binding.task_prompt_digest: + return False + if row.get("oracle_digest") != binding.oracle_digest: + return False + if row.get("oracle_command_digest") != binding.oracle_command_digest: + return False + if row.get("oracle_manifest_digest") != binding.oracle_manifest_digest: + return False + # Fail closed on both sides, as the runtime digest does: an unbound + # expectation means this sweep could not determine its own environment, and + # a row without the field was measured before it was recorded. + for field, bound in ( + ("task_asset_manifest_digest", binding.task_asset_manifest_digest), + ("sandbox_dependency_manifest_digest", binding.sandbox_dependency_manifest_digest), + ): + prior = row.get(field) + if not isinstance(prior, str) or not prior or not bound or prior != bound: + return False + + if arm in CE_ARMS: + if row.get("ce_plugin_version") != expected.ce_plugin_version: + return False + if row.get("ce_plugin_manifest_digest") != expected.ce_plugin_manifest_digest: + return False + else: + expected_skill = expected.skill_digests.get(arm) + if not expected_skill or row.get("skill_digest") != expected_skill: + return False + + if arm in {"review", "ce_review"}: + if row.get("review_evidence_valid") is not True: + return False + # The artifact, not just the score derived from it. materialize_reused_row + # copies it only when the name is present, so without this a row whose + # artifact copy never happened could be carried forward as a scored + # review that a proposer then cannot read - evidence by assertion. + review_artifact = row.get("review_artifact") + if not isinstance(review_artifact, str) or not review_artifact: + return False + if not isinstance(row.get("review_score"), dict): + return False + if row.get("review_weighted_f1") is None: + return False + + artifacts = row.get("transcript_artifacts") + if not isinstance(artifacts, list) or not artifacts: + return False + try: + for artifact in artifacts: + _transcript_metadata(artifact) + except SandboxError: + return False + return True + + +def select_reusable_comparator_rows( + rows: Sequence[Mapping[str, Any]], + *, + expected: ComparatorReuseExpectation, +) -> dict[CellKey, dict[str, Any]]: + """Index reusable rows by ``(task, arm, run)``. Conflicting duplicates drop the key.""" + + chosen: dict[CellKey, dict[str, Any]] = {} + blocked: set[CellKey] = set() + for row in rows: + if not row_is_reusable_comparator(row, expected): + continue + task_id = row["task"] + arm = row["arm"] + run = row.get("run") + if not isinstance(run, int) or isinstance(run, bool) or run < 0: + continue + key = (str(task_id), str(arm), run) + if key in blocked: + continue + previous = chosen.get(key) + if previous is None: + chosen[key] = dict(row) + continue + if _row_identity(previous) != _row_identity(row): + blocked.add(key) + chosen.pop(key, None) + return chosen + + +def materialize_reused_row( + row: Mapping[str, Any], + *, + source_dir: Path, + dest_dir: Path, +) -> dict[str, Any]: + """Copy digest-bound artifacts into this sweep's evidence dir and stamp reuse.""" + + source, _ = _resolved_directory(source_dir, label="reuse source") + dest, _ = _resolved_directory(dest_dir, label="reuse destination") + if source == dest: + raise SandboxError("comparator reuse cannot read and write the same results directory") + + materialized = dict(row) + materialized["reused"] = True + # Keep the FIRST measurement time across a chain. Overwriting it with the + # previous copy's stamp let a row refresh its own clock every generation and + # outlive the max_age bound entirely. + materialized["reused_from_recorded_at"] = row.get("reused_from_recorded_at") or row.get("recorded_at") + materialized["recorded_at"] = datetime.now(UTC).isoformat() + + artifacts = row.get("transcript_artifacts") + if not isinstance(artifacts, list) or not artifacts: + raise SandboxError("reused row is missing transcript_artifacts") + + # Every path below is resolved against a held descriptor, never re-walked + # from a name. Both roots are already symlink-free (_resolved_directory + # resolved them), and pinning them here means the components under them + # cannot be swapped out from under a check that already passed. + with ( + _open_pinned_root(source_dir, label="reuse source") as source_fd, + _open_pinned_root(dest_dir, label="reuse destination") as dest_fd, + ): + copied_artifacts: list[dict[str, Any]] = [] + for artifact in artifacts: + copied_artifacts.append(_copy_transcript_artifact(source_fd, dest_fd, artifact)) + materialized["transcript_artifacts"] = copied_artifacts + + review_name = row.get("review_artifact") + if isinstance(review_name, str) and review_name: + _copy_named_artifact(source_fd, dest_fd, review_name, label="review artifact") + + task = row.get("task") + arm = row.get("arm") + run = row.get("run") + if isinstance(task, str) and isinstance(arm, str) and isinstance(run, int) and not isinstance(run, bool): + patch_name = f"{task}-{arm}-run{run}.patch" + if _is_regular_at(patch_name, dir_fd=source_fd): + _copy_named_artifact(source_fd, dest_fd, patch_name, label="patch artifact") + return materialized + + +def default_reuse_max_age() -> timedelta: + return timedelta(days=EVIDENCE_MAX_AGE_DAYS) + + +def _row_identity(row: Mapping[str, Any]) -> tuple[Any, ...]: + return ( + row.get("skill_digest"), + row.get("oracle_digest"), + row.get("review_weighted_f1"), + row.get("ce_plugin_manifest_digest"), + row.get("recorded_at"), + ) + + +def _parse_recorded_at(value: Any) -> datetime | None: + if not isinstance(value, str) or not value: + return None + try: + parsed = datetime.fromisoformat(value.replace("Z", "+00:00")) + except ValueError: + return None + if parsed.tzinfo is None: + parsed = parsed.replace(tzinfo=UTC) + return parsed.astimezone(UTC) + + +def _transcript_metadata(metadata: Any) -> tuple[str, str, int]: + if not isinstance(metadata, dict) or set(metadata) != {"path", "sha256", "bytes", "source"}: + raise SandboxError("transcript artifact metadata must contain only path, sha256, bytes, and source") + relative = metadata["path"] + digest = metadata["sha256"] + size = metadata["bytes"] + if metadata["source"] != PARENT_EVENT_STREAM_SOURCE: + raise SandboxError("transcript artifact source is not the parent event stream") + if not isinstance(relative, str) or not isinstance(digest, str) or not re.fullmatch(r"[0-9a-f]{64}", digest): + raise SandboxError("transcript artifact metadata is malformed") + if not isinstance(size, int) or isinstance(size, bool) or size < 0 or size > MAX_TRANSCRIPT_BYTES: + raise SandboxError("transcript artifact byte count is out of range") + relative_path = PurePosixPath(relative) + if ( + relative_path.is_absolute() + or len(relative_path.parts) != 2 + or relative_path.parts[0] != "transcripts" + or any(part in {"", ".", ".."} for part in relative_path.parts) + or _TRANSCRIPT_NAME.fullmatch(relative_path.parts[1]) is None + ): + raise SandboxError(f"unsafe transcript artifact path: {relative!r}") + return relative, digest, size + + +def _resolved_directory(path: Path, *, label: str) -> tuple[Path, tuple[int, int]]: + """An existing, non-symlink directory, resolved through its parents. + + Deliberately weaker than proposer_sandbox's same-shaped helper, which + refuses every symlink hop in the path. That one guards a MOUNT ROOT, where + a hop changes what an untrusted session is handed. This one guards a DATA + directory whose contents are validated individually anyway - every file + read goes through ``_regular_file`` (lstat, symlinks rejected) and every + write through ``O_NOFOLLOW`` - so a symlinked parent grants nothing those + guards do not already cover, while refusing one would reject ordinary + setups such as a symlinked artifacts directory or macOS's /var. + + Separately named because they make different promises. Do not merge them + without first deciding which promise the reuse path should make. + """ + + resolved = path.expanduser() + try: + metadata = resolved.lstat() + except OSError as exc: + raise SandboxError(f"{label} is unavailable: {resolved}: {exc}") from exc + if stat.S_ISLNK(metadata.st_mode) or not stat.S_ISDIR(metadata.st_mode): + raise SandboxError(f"{label} must be a real directory: {resolved}") + return resolved.resolve(), (metadata.st_dev, metadata.st_ino) + + +@contextmanager +def _open_pinned_root(path: Path, *, label: str) -> Iterator[int]: + """Open a checked root and prove it is still the directory that was checked. + + The symlink POLICY above is deliberate and unchanged: parent hops stay + allowed, so a symlinked artifacts directory or macOS's /var still works. + What is closed here is separate from that policy - the gap between checking + a name and using it. lstat names one directory and resolve() re-walks the + same name afterwards, so a prior sweep that renames its results root and + drops a symlink in its place is resolved to somewhere else entirely, and + O_NOFOLLOW on the open cannot see a link that resolve() already followed. + + Comparing the opened descriptor's identity to the checked one costs an + fstat and rejects nothing that holds still: a stable directory always + matches itself. It matters for reuse specifically because the failure is + silent - rows would be copied out of the wrong directory and folded into a + comparator baseline as though they were this sweep's own evidence. + """ + + resolved, expected = _resolved_directory(path, label=label) + with _open_real_directory(resolved, label=label) as fd: + opened = os.fstat(fd) + if (opened.st_dev, opened.st_ino) != expected: + raise SandboxError(f"{label} was replaced between the check and the open: {resolved}") + yield fd + + +def _copy_transcript_artifact(source_fd: int, dest_fd: int, metadata: Mapping[str, Any]) -> dict[str, Any]: + relative, expected_digest, expected_size = _transcript_metadata(metadata) + name = PurePosixPath(relative).name + # Both `transcripts` components are opened as descriptors, not checked as + # names. An lstat that passes and a pathname that is used afterwards are two + # different directories whenever a concurrent writer renames the first one + # away — which the reuse directory, written by a prior sweep, invites. + with ( + _open_real_directory("transcripts", dir_fd=dest_fd, label="transcript destination", create=True) as dest_dir_fd, + _open_real_directory("transcripts", dir_fd=source_fd, label="transcript source") as source_dir_fd, + ): + os.fchmod(dest_dir_fd, 0o700) + # One descriptor for the whole transfer, and ONE read of it. Hashing the + # source and then reading it again to copy leaves the recorded digest + # describing bytes that are not the bytes written: the descriptor stops + # the pathname being substituted, not the inode being rewritten, and + # this directory belongs to a sweep that may still be writing. Digest + # what is copied, then judge it. + with _open_regular(name, dir_fd=source_dir_fd, label="transcript") as artifact_fd: + digest, copied_bytes = _copy_owner_only( + artifact_fd, name, dir_fd=dest_dir_fd, max_bytes=expected_size + ) + if copied_bytes != expected_size or digest != expected_digest: + # The destination now holds bytes no expectation vouches for. + os.unlink(name, dir_fd=dest_dir_fd) + drift = "size" if copied_bytes != expected_size else "digest" + raise SandboxError(f"reused transcript {drift} drifted: {relative}") + return {"path": relative, "sha256": digest, "bytes": expected_size, "source": PARENT_EVENT_STREAM_SOURCE} + + +def _copy_named_artifact(source_fd: int, dest_fd: int, name: str, *, label: str) -> None: + relative = PurePosixPath(name) + if relative.is_absolute() or len(relative.parts) != 1 or relative.parts[0] in {"", ".", ".."}: + raise SandboxError(f"unsafe {label} path: {name!r}") + with _open_regular(name, dir_fd=source_fd, label=label) as artifact_fd: + # No expectation is recorded for these, so the digest is discarded - but + # "no recorded size" is not "no limit". The source is a prior sweep + # directory that can change between sweeps, so a replaced artifact could + # be arbitrarily large; MAX_TRANSCRIPT_BYTES is the ceiling the capture + # path already enforces on evidence of this kind. + _, copied = _copy_owner_only(artifact_fd, name, dir_fd=dest_fd, max_bytes=MAX_TRANSCRIPT_BYTES) + if copied > MAX_TRANSCRIPT_BYTES: + os.unlink(name, dir_fd=dest_fd) + raise SandboxError(f"reused {label} exceeds {MAX_TRANSCRIPT_BYTES} bytes: {name}") + + +def _require_openat() -> None: + """openat is what makes a checked directory and a used directory the same one. + + Without it the only alternative is to re-walk the name after the check, + which is exactly the race this module is guarding. Refusing is safe: the + caller in runner treats a SandboxError from reuse as "run a paid cell", so + a platform without openat pays for the cells rather than copying through a + directory nobody verified. The sweep itself is Linux-only anyway (bwrap, + /proc/uptime); this is about the unit tests and about failing loudly. + """ + + if os.open not in os.supports_dir_fd or os.lstat not in os.supports_dir_fd: + raise SandboxError("comparator reuse requires POSIX openat support (os.supports_dir_fd)") + + +def _is_regular_at(name: str, *, dir_fd: int) -> bool: + """True when `name` under the pinned directory is a regular non-symlink file.""" + + try: + metadata = os.lstat(name, dir_fd=dir_fd) + except OSError: + return False + return stat.S_ISREG(metadata.st_mode) + + +@contextmanager +def _open_real_directory( + path: Path | str, + *, + dir_fd: int | None = None, + label: str, + create: bool = False, +) -> Iterator[int]: + """Open one directory that is not a symlink, and hold it for every use below. + + ``O_DIRECTORY | O_NOFOLLOW`` makes the check and the open a single syscall, + so unlike an ``lstat`` followed by a path, there is no window in which the + directory can be replaced. ``_resolved_directory`` still tolerates a + symlinked reuse ROOT — it hands this function the already-resolved path — + but every component below it is pinned. + """ + + _require_openat() + if create: + try: + os.mkdir(path, 0o700, dir_fd=dir_fd) + except FileExistsError: + # Already there is the ordinary case — a second artifact from the + # same row. What it already IS still has to be proven, and the + # O_DIRECTORY|O_NOFOLLOW open below is what proves it, so there is + # nothing to do here. + pass + except OSError as exc: + raise SandboxError(f"{label} cannot be created: {path}: {exc}") from exc + try: + descriptor = os.open( + path, + os.O_RDONLY | getattr(os, "O_DIRECTORY", 0) | getattr(os, "O_NOFOLLOW", 0), + dir_fd=dir_fd, + ) + except FileNotFoundError as exc: + # Absent is a different fact from present-but-not-a-real-directory, and + # the caller falls through to a paid cell on either. + raise SandboxError(f"{label} is missing: {path}") from exc + except OSError as exc: + raise SandboxError(f"{label} must be a real directory: {path}: {exc}") from exc + try: + # O_DIRECTORY is the check on Linux; the fstat covers a platform whose + # os module does not define it, where the flag degrades to 0. + if not stat.S_ISDIR(os.fstat(descriptor).st_mode): + raise SandboxError(f"{label} must be a real directory: {path}") + yield descriptor + finally: + os.close(descriptor) + + +@contextmanager +def _open_regular(name: str, *, dir_fd: int, label: str) -> Iterator[int]: + """Open a regular non-symlink file under a pinned directory, and hold it. + + Checking a name and then re-opening it is a race the reuse directory is + exposed to: it is written by a previous sweep and read by this one, so a + concurrent writer can replace a validated file with a symlink in between. + Resolving against ``dir_fd`` removes the directory half, ``O_NOFOLLOW`` + refuses the leaf link, and the fstat comparison proves the open descriptor + is the inode that was checked — the same guarantee + evolution._bounded_regular_bytes makes for evidence files. + """ + + _require_openat() + try: + before = os.lstat(name, dir_fd=dir_fd) + except OSError as exc: + raise SandboxError(f"{label} is missing: {name}: {exc}") from exc + if stat.S_ISLNK(before.st_mode) or not stat.S_ISREG(before.st_mode): + raise SandboxError(f"{label} must be a regular non-symlink file: {name}") + try: + descriptor = os.open(name, os.O_RDONLY | getattr(os, "O_NOFOLLOW", 0), dir_fd=dir_fd) + except OSError as exc: + raise SandboxError(f"{label} is unreadable: {name}: {exc}") from exc + try: + opened = os.fstat(descriptor) + if not stat.S_ISREG(opened.st_mode) or (opened.st_dev, opened.st_ino) != (before.st_dev, before.st_ino): + raise SandboxError(f"{label} changed while opening: {name}") + yield descriptor + finally: + os.close(descriptor) + + +def _copy_owner_only(source: int, name: str, *, dir_fd: int, max_bytes: int | None = None) -> tuple[str, int]: + """Copy one open file into the pinned directory; return what was written. + + The digest is taken from the same buffers that are written, so it describes + the copy rather than a state the source was in at some earlier read. + + ``max_bytes`` bounds the copy itself. The source is a prior sweep directory + this module already treats as concurrently writable, so a transcript + appended to after its metadata was recorded would otherwise be streamed to + EOF and only then compared against its declared size - filling the + destination, or never reaching EOF at all, long before the drift check could + reject it. Stopping one byte past the ceiling keeps that comparison + meaningful while bounding the work. + """ + + # O_CREAT|O_EXCL is the existence check, and unlike a stat beforehand it is + # atomic: a file appearing between check and open cannot slip through. + try: + descriptor = os.open( + name, + os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, "O_NOFOLLOW", 0), + 0o600, + dir_fd=dir_fd, + ) + except FileExistsError as exc: + raise SandboxError(f"reuse destination already exists: {name}") from exc + try: + os.fchmod(descriptor, 0o600) + os.lseek(source, 0, os.SEEK_SET) + digest = hashlib.sha256() + written = 0 + limit = None if max_bytes is None else max_bytes + 1 + while True: + want = COPY_CHUNK_BYTES if limit is None else min(COPY_CHUNK_BYTES, limit - written) + if want <= 0: + break + chunk = os.read(source, want) + if not chunk: + break + digest.update(chunk) + written += len(chunk) + _write_all(descriptor, chunk) + os.fsync(descriptor) + return digest.hexdigest(), written + finally: + os.close(descriptor) diff --git a/eval/workflow_bench/evolve.py b/eval/workflow_bench/evolve.py index 71a2299cb..3c17325be 100644 --- a/eval/workflow_bench/evolve.py +++ b/eval/workflow_bench/evolve.py @@ -48,6 +48,7 @@ import yaml from . import runner from . import runner_sessions +from .comparator_reuse import current_runtime_digest from .model_gateway import ( ANTHROPIC_API_KEY_ENV, attach_openai_gateway, @@ -698,8 +699,17 @@ def run_proposer( bwrap_bin: Path, sandbox_backend: str = "bwrap", progress_label: str | None = None, + started_monotonic: float | None = None, ) -> dict[str, Any]: - """Run one proposer in confinement and copy only validated outputs out.""" + """Run one proposer in confinement and copy only validated outputs out. + + ``started_monotonic`` is the sweep clock, not a precomputed budget. The + per-session ``--timeout`` is sized for a whole generation, so a proposer + started with only the sweep minimum left would otherwise run far past the + instance window; the clock is passed rather than the leftover because the + clone, the sanitize pass and the sandbox setup below all happen before the + session starts, and a number sampled by the caller is already stale by then. + """ with tempfile.TemporaryDirectory(prefix="wfevolve-") as tmp: clone = runner.make_worktree(REPO_ROOT, "HEAD", Path(tmp)) @@ -732,11 +742,45 @@ def run_proposer( host_text = getattr(sandbox, "host_text", lambda value: value) environment_builder = getattr(sandbox, "environment", build_sandbox_environment) backend = getattr(sandbox, "backend", "bwrap") + # Sampled here, after the setup above: this is the last + # moment before the session starts, so it is the only reading + # the session's own timeout can honestly be clamped to. + remaining_seconds = ( + None + if started_monotonic is None + else remaining_runtime_seconds( + max_runtime_seconds=args.max_runtime_seconds, + started_monotonic=started_monotonic, + ) + ) + # An exhausted cap must stop the run, not buy one more second. + # remaining_runtime_seconds floors at 0, and max(1, ...) turned + # that 0 into a one-second paid session: the admission check + # happens before cloning, sanitizing and sandbox setup, so those + # unbounded steps can spend the rest of the window and leave + # nothing for the upload reserve this cap exists to protect. + if remaining_seconds is not None and remaining_seconds < 1: + # The caller stops the run on a not-ok record, which is the + # right outcome: an exhausted cap should end the generation, + # not start a session it cannot afford to finish. + return { + "ok": False, + "error_kind": "runtime-cap-exhausted", + "error_detail": ( + "the wall-clock cap elapsed during proposer setup " + "(clone, sanitize, sandbox), before the session started" + ), + "duration_s": 0.0, + "num_turns": 0, + "cost_usd": None, + } record = runner.run_claude( host_text(prompt), clone, claude_bin=sandbox.claude_bin, - timeout=args.timeout, + timeout=( + args.timeout if remaining_seconds is None else min(args.timeout, remaining_seconds) + ), model=args.proposer_model, effort=args.effort, env=model_session_environment( @@ -823,6 +867,119 @@ def _timeout_arm_key(arm: str) -> str: return CANDIDATE_ARMS.get(arm, arm) +EVENTBRIDGE_INSTANCE_WINDOW_SECONDS = 86_400 +EVENTBRIDGE_STOP_RESERVE_SECONDS = 5_400 +MIN_INSTANCE_SWEEP_SECONDS = 600 + + +def instance_window_budget_seconds( + uptime_seconds: float, + *, + window_seconds: int = EVENTBRIDGE_INSTANCE_WINDOW_SECONDS, + reserve_seconds: int = EVENTBRIDGE_STOP_RESERVE_SECONDS, + min_seconds: int = MIN_INSTANCE_SWEEP_SECONDS, +) -> int: + """Seconds a sweep may run before an EventBridge 24h instance stop. + + The dedicated evolution box is started ~15 minutes before the Saturday + cron and stopped 24h later. A ``workflow_dispatch`` that lands on an + already-running box inherits the leftover uptime, not a fresh day. + Run 33962002890 dispatched Friday 10:57 UTC and was still on its last + review cell when the Saturday 03:00 stop cancelled the runner — 51 + finished sessions never uploaded because a cancelled job skips even + ``if: always()``. Capping the in-process sweep so it *fails* (instead + of vanishing) leaves the reserve for the upload step. + """ + + if window_seconds < 1 or reserve_seconds < 0 or min_seconds < 1: + raise ValueError("instance window and minimum must be positive; reserve must be non-negative") + if not math.isfinite(uptime_seconds) or uptime_seconds < 0: + raise ValueError("uptime must be a finite non-negative number") + leftover = int(window_seconds - uptime_seconds - reserve_seconds) + if leftover < min_seconds: + raise ValueError( + f"instance window has only {leftover}s left after a {reserve_seconds}s " + f"upload reserve (uptime {uptime_seconds:.0f}s of {window_seconds}s); " + f"need at least {min_seconds}s" + ) + return leftover + + +def _instance_uptime_or_none() -> float | None: + """The uptime read main() takes before it knows whether it needs it. + + Deferring the read until after argument parsing would put the parse back + inside the interval the cap is supposed to cover, so it happens first and + an unreadable /proc/uptime is only an error if the flag turns out to be set. + """ + + try: + return read_instance_uptime_seconds() + except ValueError: + return None + + +def read_instance_uptime_seconds(uptime_path: Path = Path("/proc/uptime")) -> float: + """Host uptime, the clock the EventBridge stop is scheduled against.""" + + try: + return float(uptime_path.read_text().split()[0]) + except (OSError, IndexError, ValueError) as exc: + raise ValueError(f"cannot read instance uptime from {uptime_path}: {exc}") from exc + + +def instance_window_budget_from_uptime( + uptime_seconds: float, + *, + window_seconds: int | None = None, + reserve_seconds: int | None = None, +) -> int: + """Apply the EventBridge window env overrides to an already-read uptime. + + Separate from the read so ``main`` can take the uptime in the same breath + as its own clock: the budget and the clock it is measured against have to + describe one instant, or the interval between them is spent by nobody and + charged to the sweep. + """ + + window = ( + window_seconds + if window_seconds is not None + else int(os.environ.get("EVENTBRIDGE_INSTANCE_WINDOW_SECONDS", str(EVENTBRIDGE_INSTANCE_WINDOW_SECONDS))) + ) + reserve = ( + reserve_seconds + if reserve_seconds is not None + else int(os.environ.get("EVENTBRIDGE_STOP_RESERVE_SECONDS", str(EVENTBRIDGE_STOP_RESERVE_SECONDS))) + ) + return instance_window_budget_seconds(uptime_seconds, window_seconds=window, reserve_seconds=reserve) + + + + +def remaining_runtime_seconds(*, max_runtime_seconds: int | None, started_monotonic: float) -> int | None: + """Seconds left in an optional wall-clock cap, or None when uncapped.""" + + if max_runtime_seconds is None: + return None + if max_runtime_seconds < 1: + raise ValueError("max runtime must be positive") + leftover = max_runtime_seconds - (time.monotonic() - started_monotonic) + return max(0, int(leftover)) + + +def capped_timeout_seconds(requested: int, remaining: int | None) -> int: + """Clamp one managed-process timeout to the leftover instance window.""" + + if requested < 1: + raise ValueError("requested timeout must be positive") + if remaining is None: + return requested + if remaining < 1: + raise ValueError("no time remains in the instance window") + return min(requested, remaining) + + def generation_timeout_seconds( *, task_count: int, @@ -877,6 +1034,7 @@ def runner_argv( task_bindings: list[dict[str, Any]], target_base_digests: dict[str, str], proposer_model: str | None = None, + reuse_results: Path | None = None, ) -> list[str]: incumbent_arms = resolve_incumbent_arms(overlay_dir, args.arms) paired_arms = executed_benchmark_arms(incumbent_arms) @@ -927,6 +1085,8 @@ def runner_argv( argv += ["--ce-plugin-dir", str(args.ce_plugin_dir), "--ce-plugin-version", args.ce_plugin_version] if args.unsafe_no_bwrap: argv.append("--unsafe-no-bwrap") + if reuse_results is not None: + argv += ["--reuse-results", str(reuse_results)] return argv @@ -944,6 +1104,12 @@ def runner_environment(args: argparse.Namespace) -> dict[str, str]: # actually show progress rather than a burst at the end. "PYTHONUNBUFFERED": "1", } + # process_control replaces the child environment wholesale, so a digest the + # workflow exported reaches the runner only if it is forwarded here. Without + # this the runner stamps no runtime_digest and the reuse lock never engages. + runtime_digest = current_runtime_digest() + if runtime_digest: + env["RUNTIME_DIGEST"] = runtime_digest if args.auth_token: env[ANTHROPIC_API_KEY_ENV] = args.auth_token return env @@ -1199,6 +1365,13 @@ def _require_finite_metric(value: Any, name: str, *, nullable: bool = False, max raise ValueError(f"promotion has invalid {name}") +def _positive_int(value: str) -> int: + parsed = int(value) + if parsed < 1: + raise argparse.ArgumentTypeError(f"{value} is not a positive integer") + return parsed + + def build_parser() -> argparse.ArgumentParser: parser = argparse.ArgumentParser(description=__doc__) parser.add_argument("--tasks", required=True, type=Path) @@ -1237,7 +1410,8 @@ def build_parser() -> argparse.ArgumentParser: "--seed-results", type=Path, default=None, - help="prior wfbench results dir used as generation-0 proposer evidence", + help="prior wfbench results dir used as generation-0 proposer evidence " + "and as --reuse-results for unchanged incumbent/CE cells", ) parser.add_argument( "--initial-overlay", @@ -1271,6 +1445,19 @@ def build_parser() -> argparse.ArgumentParser: default=runner_sessions.SESSION_TIMEOUT_SECONDS, help="per session, seconds", ) + parser.add_argument( + "--max-runtime-seconds", + type=_positive_int, + default=None, + help="wall-clock cap for the whole evolve process (CI derives this from " + "instance uptime so the sweep exits before EventBridge stops the box)", + ) + parser.add_argument( + "--max-runtime-from-instance-window", + action="store_true", + help="derive --max-runtime-seconds from /proc/uptime at startup, so the " + "budget and the clock it is measured against describe one instant", + ) parser.add_argument("--base-url", default=None) parser.add_argument( "--anthropic-api-key", @@ -1308,8 +1495,26 @@ def build_parser() -> argparse.ArgumentParser: def main() -> int: + # These two lines are the cap, and they are adjacent on purpose: the clock + # the sweep is measured against, and the uptime the budget is derived from. + # run-evolution.sh used to compute the budget in its own `uv run python -c` + # and pass a number, so the script's remaining work and this interpreter's + # startup were spent by nobody and charged to the sweep — out of the upload + # reserve the cap exists to protect. Nothing can be spent between them now. + started_monotonic = time.monotonic() + instance_uptime = _instance_uptime_or_none() parser = build_parser() args = parser.parse_args() + if args.max_runtime_from_instance_window: + if args.max_runtime_seconds is not None: + parser.error("--max-runtime-from-instance-window and --max-runtime-seconds are mutually exclusive") + if instance_uptime is None: + parser.error("--max-runtime-from-instance-window needs a readable /proc/uptime") + try: + args.max_runtime_seconds = instance_window_budget_from_uptime(instance_uptime) + except ValueError as exc: + parser.error(str(exc)) + print(f"capping the sweep to {args.max_runtime_seconds}s so the instance-window reserve can upload evidence") if args.generations < 1: parser.error("--generations must be positive") if args.runs < 1 or args.timeout < 1: @@ -1379,6 +1584,7 @@ def main() -> int: try: return _run_generations( args, + started_monotonic=started_monotonic, selected_task_rows=selected_task_rows, skipped_expensive=skipped_expensive, selected_tasks=selected_tasks, @@ -1394,6 +1600,7 @@ def main() -> int: def _run_generations( args: argparse.Namespace, *, + started_monotonic: float, selected_task_rows: list[dict[str, Any]], skipped_expensive: list[str], selected_tasks: list[dict[str, Any]], @@ -1465,6 +1672,19 @@ def _run_generations( incumbent_arms=requested_arms, prior_proposal=staged_prior_included, ) + # Check the window before the paid session, not after it. A + # generation that cannot fit its sweep should not buy a proposal + # first and discover the deadline on the way out. + before_proposer = remaining_runtime_seconds( + max_runtime_seconds=args.max_runtime_seconds, + started_monotonic=started_monotonic, + ) + if before_proposer is not None and before_proposer < MIN_INSTANCE_SWEEP_SECONDS: + print( + f"[gen {generation}] stopping with {before_proposer}s left before the " + f"instance window ends; not starting a proposer session" + ) + return 1 print(f"[gen {generation}] proposing…") record = run_proposer( prompt, @@ -1475,6 +1695,11 @@ def _run_generations( bwrap_bin=bwrap_bin, sandbox_backend=sandbox_backend, progress_label=f"gen {generation} proposer", + # The clock, not the reading taken above: run_proposer clones, + # sanitizes and builds a sandbox before the session starts, so + # before_proposer is stale by then. It still decides whether to + # start at all — it just cannot decide how long to allow. + started_monotonic=started_monotonic, ) # Redact any API token echoed into the session record (e.g. an # error_detail stderr_tail) before it enters the uploaded artifact. @@ -1513,6 +1738,16 @@ def _run_generations( print(f"[gen {generation}] promotion targets contain uncommitted or drifted bytes") return 1 print(f"[gen {generation}] benchmarking candidate…") + leftover = remaining_runtime_seconds( + max_runtime_seconds=args.max_runtime_seconds, + started_monotonic=started_monotonic, + ) + if leftover is not None and leftover < MIN_INSTANCE_SWEEP_SECONDS: + print( + f"[gen {generation}] stopping with {leftover}s left before the " + f"instance window ends; partial evidence is in {out_root}/" + ) + return 1 benchmark_argv = runner_argv( args, bench_dir, @@ -1520,20 +1755,30 @@ def _run_generations( task_bindings=selected_tasks, target_base_digests=target_base_digests, proposer_model=generation_proposer_model, + reuse_results=evidence_dir, ) benchmark_command = ( benchmark_argv if sandbox_backend == "host-unsafe" else pid_namespace_command(benchmark_argv, bwrap_bin=bwrap_bin) ) - bench = run_managed( - benchmark_command, - timeout=generation_timeout_seconds( + sweep_timeout = capped_timeout_seconds( + generation_timeout_seconds( task_count=len(selected_task_rows), runs=args.runs, session_timeout=args.timeout, incumbent_arms=incumbent_arms, ), + leftover, + ) + if leftover is not None: + print( + f"[gen {generation}] sweep timeout {sweep_timeout}s " + f"(instance window leftover {leftover}s)" + ) + bench = run_managed( + benchmark_command, + timeout=sweep_timeout, env=runner_environment(args), require_pid_namespace=sandbox_backend == "bwrap", # The sweep is the multi-hour phase; without this its per-run @@ -1545,7 +1790,13 @@ def _run_generations( # The sweep runs with GITNEXUS_BENCH_ANTHROPIC_API_KEY in its environment, # so its detail/stderr tail is a token-bearing sink like any other. detail = redacted_failure(args, str(bench.detail or bench.stderr_tail[-1000:])) - print(f"[gen {generation}] benchmark run failed ({bench.state}, exit {bench.returncode}): {detail}") + if leftover is not None and bench.state == "timeout": + print( + f"[gen {generation}] benchmark hit the instance-window budget " + f"({sweep_timeout}s); partial evidence is in {bench_dir}: {detail}" + ) + else: + print(f"[gen {generation}] benchmark run failed ({bench.state}, exit {bench.returncode}): {detail}") return 1 promotion = json.loads((bench_dir / "promotion.json").read_text()) for line in summarize_gate(promotion): diff --git a/eval/workflow_bench/proposer_sandbox.py b/eval/workflow_bench/proposer_sandbox.py index a2dfaa54a..2b0347da1 100644 --- a/eval/workflow_bench/proposer_sandbox.py +++ b/eval/workflow_bench/proposer_sandbox.py @@ -22,6 +22,15 @@ from .process_control import ManagedProcessResult, run_managed MAX_EVIDENCE_FILE_BYTES = 256 * 1024 MAX_BUNDLE_BYTES = 2 * 1024 * 1024 SANDBOX_WORKSPACE = "/workspace" +# The review artifact lives OUTSIDE the workspace, in its own writable +# directory. A writable FILE inside a read-only directory is not writable to +# anything that writes atomically: the Write tool creates +# `.tmp..` beside the target and renames it, so a read-only +# parent fails the temp create with EROFS and the artifact is never written. +# Binding a writable directory outside /workspace lets the rename land while +# the workspace itself stays entirely read-only. +SANDBOX_REVIEW_OUTPUT = "/review-output" +REVIEW_OUTPUT_DIRNAME = "review-output" SANDBOX_HOME = "/home/agent" SANDBOX_TMP = "/tmp" SANDBOX_CLAUDE = "/opt/claude/claude" @@ -118,36 +127,41 @@ REVIEW_RUNTIME_DIRECTORIES = ( ) -def prepare_review_workspace(sandbox: SandboxSession, artifact_name: str) -> Path: - """Prepare disposable mount targets; never truncate a pre-existing entry.""" +def review_output_path(sandbox: SandboxSession, artifact_name: str) -> Path: + """Host path of the review artifact: a private directory, not the clone. + + One source of truth for the location, so the mount, the parse and the + artifact copy cannot drift apart. + """ - clone = _real_directory(sandbox.clone, label="review clone") if PurePosixPath(artifact_name).name != artifact_name or "\\" in artifact_name or artifact_name in ("", ".", ".."): raise SandboxError("review artifact must be a root filename") - output = clone / artifact_name - # No agent runs while this private clone is being prepared. On POSIX the - # directory descriptor additionally binds the exclusive create to its owner. - directory_fd = None - try: - if os.name != "nt": - directory_fd = os.open(clone, os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW) - fd = os.open( - artifact_name if directory_fd is not None else output, - os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, "O_NOFOLLOW", 0), - 0o600, - dir_fd=directory_fd, - ) - try: - if not stat.S_ISREG(os.fstat(fd).st_mode): - raise SandboxError("review artifact must be a regular file") - finally: - os.close(fd) - except FileExistsError as exc: - raise SandboxError("review artifact already exists") from exc - finally: - if directory_fd is not None: - os.close(directory_fd) + return Path(sandbox.private_root) / REVIEW_OUTPUT_DIRNAME / artifact_name + +def prepare_review_workspace(sandbox: SandboxSession, artifact_name: str) -> Path: + """Prepare disposable mount targets; never truncate a pre-existing entry. + + Creates the artifact's own directory and returns the path the agent is + expected to write. The file itself is deliberately NOT pre-created: the + agent writes it atomically (temp file beside the target, then rename), so + the directory is what has to be writable, and an existing empty file would + only be something for the write to trip over. Absence is meaningful — it is + how ``parse_review_output`` tells "never written" from "written badly". + """ + + output = review_output_path(sandbox, artifact_name) + # No agent runs while this private root is being prepared, and the + # exclusive create is what proves the directory is ours rather than + # something a previous cell left behind. + try: + output.parent.mkdir(mode=0o700, parents=False, exist_ok=False) + except FileExistsError as exc: + raise SandboxError("review artifact directory already exists") from exc + except OSError as exc: + raise SandboxError(f"review artifact directory is unavailable: {output.parent}") from exc + + clone = _real_directory(sandbox.clone, label="review clone") if sandbox.backend != "bwrap": return output created: list[str] = [] @@ -214,6 +228,10 @@ class SandboxSession: ReadOnlyMount(self.clone, SANDBOX_WORKSPACE), ReadOnlyMount(self.home, SANDBOX_HOME), ReadOnlyMount(self.temp, SANDBOX_TMP), + # The review artifact directory is a real mount on bwrap, so the + # host-unsafe backend has to translate it too. Without this the + # review prompt names a path that exists on neither backend. + ReadOnlyMount(Path(self.private_root) / REVIEW_OUTPUT_DIRNAME, SANDBOX_REVIEW_OUTPUT), ] for mount in sorted(mappings, key=lambda item: len(item.target), reverse=True): target = mount.target.rstrip("/") @@ -234,9 +252,16 @@ class SandboxSession: SANDBOX_WORKSPACE, SANDBOX_HOME, SANDBOX_TMP, + SANDBOX_REVIEW_OUTPUT, ] ordered = sorted(set(targets), key=len, reverse=True) - pattern = re.compile("|".join(re.escape(target) for target in ordered)) + # Only translate at a path boundary. "/review-output" occurs twice in + # "/review-output/review-output.json" - once as the directory and once + # inside the filename - and rewriting the second turned the artifact path + # into nonsense. A target must be followed by "/", whitespace, a quote or + # end of string to be a path rather than a prefix of a longer name. + boundary = r"""(?=[/\s"']|$)""" + pattern = re.compile("(?:" + "|".join(re.escape(target) for target in ordered) + ")" + boundary) return pattern.sub(lambda match: self.host_path(match.group(0)), value) @property @@ -549,12 +574,17 @@ def build_claude_settings(*, sandbox_enabled: bool = True) -> str: "allowLocalBinding": False, }, "filesystem": { - "allowWrite": [SANDBOX_WORKSPACE, SANDBOX_TMP, SANDBOX_HOME], + # SANDBOX_REVIEW_OUTPUT is the review artifact directory. The + # bwrap bind alone is not enough: this policy is a second, + # independent gate the CLI applies to its own tools, and a path + # missing here is unwritable however the mount is shaped. + "allowWrite": [SANDBOX_WORKSPACE, SANDBOX_TMP, SANDBOX_HOME, SANDBOX_REVIEW_OUTPUT], "denyRead": ["/"], "allowRead": [ SANDBOX_WORKSPACE, SANDBOX_TMP, SANDBOX_HOME, + SANDBOX_REVIEW_OUTPUT, "/usr", "/bin", "/lib", diff --git a/eval/workflow_bench/review_scoring.py b/eval/workflow_bench/review_scoring.py index b8ad0151b..eaf328e37 100644 --- a/eval/workflow_bench/review_scoring.py +++ b/eval/workflow_bench/review_scoring.py @@ -12,6 +12,9 @@ from typing import Any, Mapping, Sequence from .oracle_assets import TaskOracleSnapshot REVIEW_OUTPUT = "review-output.json" +# Task verify/oracle commands read the artifact location from here rather +# than hardcoding a path, so one command works under bwrap and host-unsafe. +REVIEW_OUTPUT_ENV_VAR = "GITNEXUS_BENCH_REVIEW_OUTPUT" REVIEW_SCHEMA_VERSION = 1 MAX_REVIEW_BYTES = 256 * 1024 MAX_FINDINGS = 100 @@ -112,13 +115,33 @@ def _parse_review_finding(raw: Any, index: int) -> ReviewFinding: def parse_review_output(path: Path) -> tuple[str, tuple[ReviewFinding, ...]]: - metadata = path.lstat() + # Distinguish these. Folding empty, malformed and encoding failures into one + # message is how a sandbox that left the artifact at 0 bytes read for 15 + # runs as an encoding fault: json.loads("") raises, and every such cell + # reported "not valid UTF-8 JSON". A path the agent never created was not in + # that fold — the lstat below sat outside the try and raised + # FileNotFoundError — but it reached the caller as a bare OSError rather + # than saying what was wrong, which is why it is named here too. + try: + metadata = path.lstat() + except FileNotFoundError as exc: + raise ValueError("review output was never written") from exc + except OSError as exc: + raise ValueError(f"review output is unreadable: {exc.strerror}") from exc if path.is_symlink() or not path.is_file() or metadata.st_size > MAX_REVIEW_BYTES: raise ValueError("review output must be a bounded regular non-symlink file") + if metadata.st_size == 0: + raise ValueError("review output is empty") try: - raw = json.loads(path.read_text()) - except (OSError, UnicodeError, json.JSONDecodeError) as exc: - raise ValueError("review output is not valid UTF-8 JSON") from exc + text = path.read_text(encoding="utf-8") + except OSError as exc: + raise ValueError(f"review output is unreadable: {exc.strerror}") from exc + except UnicodeError as exc: + raise ValueError("review output is not valid UTF-8") from exc + try: + raw = json.loads(text) + except json.JSONDecodeError as exc: + raise ValueError(f"review output is not valid JSON: {exc.msg} at line {exc.lineno}") from exc if not isinstance(raw, Mapping) or set(raw) != {"schema_version", "verdict", "findings"}: raise ValueError("review output requires exactly schema_version, verdict, and findings") if raw["schema_version"] != REVIEW_SCHEMA_VERSION: diff --git a/eval/workflow_bench/run-evolution.sh b/eval/workflow_bench/run-evolution.sh index c9d2a77a1..066df111a 100755 --- a/eval/workflow_bench/run-evolution.sh +++ b/eval/workflow_bench/run-evolution.sh @@ -15,6 +15,8 @@ # MODEL PROPOSER_MODEL EFFORT GENERATIONS RUNS WORKERS PROVIDER # EVOLUTION_PROFILE CE_PLUGIN_DIR CE_PLUGIN_VERSION # INCLUDE_EXPENSIVE SEED_RESULTS CLAUDE_BIN OUT_ROOT +# CI (passes --max-runtime-from-instance-window; the CLI reads /proc/uptime) +# EVENTBRIDGE_INSTANCE_WINDOW_SECONDS EVENTBRIDGE_STOP_RESERVE_SECONDS # UNSAFE_NO_BWRAP=1 (local review diagnostics only) # GITNEXUS_BENCH_ANTHROPIC_API_KEY (legacy GITNEXUS_BENCH_AUTH_TOKEN) # GITNEXUS_BENCH_OPENAI_API_KEY @@ -193,6 +195,18 @@ if ((${#passthrough[@]})); then cmd+=("${passthrough[@]}") fi +# A cancelled GitHub job skips even `if: always()`, so evidence dies with the +# runner. The evolution box is EventBridge-stopped 24h after boot; a Friday +# dispatch inherits leftover uptime. Cap the sweep so it fails in-process and +# the upload step still runs (run 33962002890). The CLI reads /proc/uptime +# itself, in the same breath as it starts the clock the cap is measured +# against; computing a number here — in a separate interpreter, before the +# provenance work and the exec below — charged the sweep for every second +# this script spent afterwards. +if [[ -n "${CI:-}" && -r /proc/uptime ]]; then + cmd+=(--max-runtime-from-instance-window) +fi + if ((dry_run)); then printf '%q ' "${cmd[@]}" printf '\n' @@ -220,5 +234,9 @@ SOURCE_SHA="${source_sha}" RUNTIME_DIGEST="${runtime_digest}" SANDBOX_BACKEND="$ }, null, 2) + "\n")' "${out_root}/runtime-provenance.json" export PYTHONUNBUFFERED=1 +# The runner stamps this on every results.jsonl row and refuses to reuse a +# comparator cell when a prior row's digest disagrees. Keep it on the evolve +# process, not only in the provenance JSON sidecar. +export RUNTIME_DIGEST="${runtime_digest}" cd "${eval_dir}" exec "${cmd[@]}" diff --git a/eval/workflow_bench/runner.py b/eval/workflow_bench/runner.py index 9f9dd9ef3..a2223a20a 100644 --- a/eval/workflow_bench/runner.py +++ b/eval/workflow_bench/runner.py @@ -57,6 +57,17 @@ from typing import Any import yaml +from .comparator_reuse import ( + REUSE_EXCLUDED_ERROR_KINDS, + CellKey, + ComparatorReuseExpectation, + TaskReuseBinding, + current_runtime_digest, + default_reuse_max_age, + load_result_rows, + materialize_reused_row, + select_reusable_comparator_rows, +) from .evolution import ( CANDIDATE_ARMS, EVIDENCE_MAX_AGE_DAYS, @@ -88,12 +99,19 @@ from .oracle_assets import ( ) from .process_control import cancellation_scope, ManagedProcessError from .promotion_apply import committed_destination_base_digests -from .review_scoring import REVIEW_OUTPUT, expected_findings, parse_review_output, score_review +from .review_scoring import ( + REVIEW_OUTPUT, + REVIEW_OUTPUT_ENV_VAR, + expected_findings, + parse_review_output, + score_review, +) from .proposer_sandbox import ( SANDBOX_GITNEXUS as SANDBOX_GITNEXUS, SANDBOX_GITNEXUS_REGISTRY, SANDBOX_GITNEXUS_SHARED as SANDBOX_GITNEXUS_SHARED, SANDBOX_NODE as SANDBOX_NODE, + SANDBOX_REVIEW_OUTPUT, SANDBOX_WORKSPACE, ReadOnlyMount, SandboxError, @@ -104,6 +122,7 @@ from .proposer_sandbox import ( prepare_sandbox, prepare_review_workspace, redact_text, + review_output_path, require_claude_sandbox_helpers, sandbox_workspace_write_boundary, ) @@ -121,6 +140,7 @@ from .runner_artifacts import ( enforce_phase_workspace, enforce_work_evidence, implementation_diff_digest, + copy_isolated_tree, make_worktree, new_plan_doc, parse_shortstat as parse_shortstat, @@ -222,8 +242,9 @@ CE_WORK_DIRECT_PROMPT = ( # Review cell: setup applies a historical PR diff, then the model sees a # read-only checkout. Both arms emit the same strict artifact so quality can be # scored deterministically against labels that remain hidden until it exits. -REVIEW_OUTPUT_CONTRACT = """ -Write /workspace/review-output.json as UTF-8 JSON with exactly this shape: +# Concatenated, not an f-string: the JSON shape below keeps its braces doubled +# because the finished prompt is .format()-ed with the task text. +REVIEW_OUTPUT_CONTRACT = f"\nWrite {SANDBOX_REVIEW_OUTPUT}/{REVIEW_OUTPUT} " + """as UTF-8 JSON with exactly this shape: {{"schema_version":1,"verdict":"approve|comment|request_changes","findings":[{{ "id":"unique stable id","severity":"critical|high|medium|low", "path":"repository-relative changed file","line":1,"end_line":1, @@ -563,9 +584,13 @@ def run_arm( read_only_workspace=True, read_only_paths=_evaluated_skill_roots(worktree, arm), extra_writable_mounts=( + # The DIRECTORY, outside the workspace. Binding the file + # itself left the agent nowhere to put the temp file it + # renames into place, so every review artifact came back + # empty with EROFS in the transcript. ReadOnlyMount( - source=review_output, - target=f"{SANDBOX_WORKSPACE}/{REVIEW_OUTPUT}", + source=review_output.parent, + target=SANDBOX_REVIEW_OUTPUT, ), ), ), @@ -573,7 +598,8 @@ def run_arm( with sandbox_workspace_write_boundary( sandbox, read_only_workspace=True, - writable=(review_output,), + # Nothing in the workspace is writable now — the artifact left it. + writable=(), ): review_session = run_claude( host_text(review_prompt.format(task=task["prompt"])), @@ -584,11 +610,9 @@ def run_arm( sessions.append(review_session) if review_session["ok"] and phase_before is not None: try: - enforce_phase_workspace( - worktree, - phase_before, - allowed_artifact=worktree / REVIEW_OUTPUT, - ) + # The artifact is no longer in the workspace, so the review + # phase may now change nothing there at all. + enforce_phase_workspace(worktree, phase_before, allowed_artifact=None) require_skill_fingerprint( worktree, arm, @@ -643,6 +667,18 @@ def run_arm( record = sum_sessions(sessions) record["arm"] = arm record["plan_produced"] = arm not in ("workflow", "ce_workflow") or plan_doc is not None + # The verify command runs in its OWN sandbox invocation, which knows nothing + # about the review session's writable mount. Expose the artifact read-only and + # name it through the environment, the same shape _run_hidden_oracle uses, so + # one command works on both backends instead of hardcoding either path. + verify_env = environment_builder() + verify_mounts: tuple[ReadOnlyMount, ...] = () + if arm in ("review", "ce_review"): + review_artifact = review_output_path(sandbox, REVIEW_OUTPUT) + verify_env[REVIEW_OUTPUT_ENV_VAR] = host_text(f"{SANDBOX_REVIEW_OUTPUT}/{REVIEW_OUTPUT}") + verify_mounts = ( + ReadOnlyMount(source=review_artifact.parent, target=SANDBOX_REVIEW_OUTPUT), + ) authored_tests_passed, authored_test_output = _verification_outcome( run_verify( task["verify"], @@ -651,21 +687,25 @@ def run_arm( command_prefix=sandbox.command_prefix_for( read_only_workspace=True, unshare_network=True, + extra_read_only_mounts=verify_mounts, ), - env=environment_builder(), + env=verify_env, require_pid_namespace=getattr(sandbox, "require_pid_namespace", True), ) ) review_score: dict[str, Any] | None = None if arm in ("review", "ce_review"): try: - verdict, findings = parse_review_output(worktree / REVIEW_OUTPUT) + verdict, findings = parse_review_output(review_output_path(sandbox, REVIEW_OUTPUT)) labels = expected_findings(oracle_snapshot) if oracle_snapshot is not None else () review_score = score_review(verdict, findings, labels) except (OSError, ValueError) as exc: record["ok"] = False record["error_kind"] = record["error_kind"] or "review-evidence-invalid" - record["error_detail"] = str(exc) + # Keep the FIRST detail, as error_kind already does. A phase- + # boundary violation is why the artifact is unparseable; reporting + # the parse failure over it buries the cause under the symptom. + record["error_detail"] = record.get("error_detail") or str(exc) record["review_score"] = review_score record["review_evidence_valid"] = review_score is not None if review_score is not None: @@ -721,9 +761,53 @@ CHURN_FIELDS = ("diff_files", "diff_insertions", "diff_deletions") # Rows where the session (or the harness) died carry no measured evidence and # must not skew efficiency medians or resolve denominators. verify-failed and # skill-not-invoked rows DO count: those sessions ran and spent real tokens. -EXCLUDED_ERROR_KINDS = frozenset( - {"session-error", "infra-error", "evidence-unverified", "cleanup-failure", "review-evidence-invalid", "cancelled"} -) +# One definition, in comparator_reuse: reuse eligibility and aggregate +# exclusion must never drift apart. The dependency only runs this way - +# comparator_reuse importing back from runner is a circular import. +EXCLUDED_ERROR_KINDS = REUSE_EXCLUDED_ERROR_KINDS + + + +# Health classification. These answer "did the harness work", which is a +# different question from "did the agent get the right answer" - a review can be +# wrong about a hard corpus while every process, mount and capture behaved. +# +# EXECUTION: the process or its tooling did not complete. Nothing was measured. +# EVIDENCE: it completed, but what it produced cannot be trusted or scored. +# Everything else - including resolved=False and a zero score - is a VALID +# NEGATIVE: an admissible measurement that the quality gate then judges. +EXECUTION_FAILURE_KINDS = frozenset({"session-error", "infra-error", "cleanup-failure", "cancelled"}) +EVIDENCE_FAILURE_KINDS = frozenset({"review-evidence-invalid", "evidence-unverified", "skill-not-invoked"}) + + +def execution_failed(record: Mapping[str, Any]) -> bool: + """The process or its tooling did not complete.""" + + return record.get("error_kind") in EXECUTION_FAILURE_KINDS + + +def evidence_failed(record: Mapping[str, Any]) -> bool: + """It completed, but what it produced cannot be trusted or scored.""" + + return ( + record.get("error_kind") in EVIDENCE_FAILURE_KINDS + or record.get("review_evidence_valid") is False + or record.get("transcript_missing") is True + ) + + +def task_prompt_digest(task: Mapping[str, Any]) -> str: + """The prompt digest, computed once for the row and the reuse expectation. + + row_is_reusable_comparator compares the value a prior row stored against the + value this sweep derives, as exact strings. Two inline copies of this hash + had already drifted - one picked up a str() cast the other lacked - and a + further divergence (normalising whitespace on one side, say) would silently + stop rows matching, or match rows that should not. + """ + + return hashlib.sha256(str(task["prompt"]).encode()).hexdigest() + # A sustained upstream outage shows up as a run of session/infra/cleanup # failures. (cleanup-failure overwrites the primary error_kind, so a @@ -806,7 +890,13 @@ def sweep_task_cells( raise ValueError("workers must be positive") for wave_start in range(0, len(cells), workers): if cancel_event.is_set(): - return outage_streak, True + # False: this flag means the OUTAGE breaker tripped, and the + # caller turns it into exit 1 with "Sweep aborted". Cancellation + # stops the sweep too, but it is the operator's Ctrl-C, not a + # systemic failure - reporting True relabelled every interrupted + # run an outage and returned 1 where the contract says 130. The + # caller tests cancel_event itself for the stop decision. + return outage_streak, False wave = list(cells[wave_start : wave_start + workers]) for run_idx, arm in wave: on_start(run_idx, arm) @@ -833,7 +923,7 @@ def sweep_task_cells( # an earlier row trips the breaker; only later waves are skipped. on_record(run_idx, arm, record) if cancel_event.is_set(): - return outage_streak, True + return outage_streak, False for record in records: kind = ( "review-evidence-invalid" @@ -847,14 +937,15 @@ def sweep_task_cells( "failures — aborting the remaining sweep; report and promotion are written " "from partial evidence and the run exits non-zero." ) + # Signal in-flight background work too. The breaker exists to + # SHORTEN a doomed run; without this a graph prefetch keeps + # building and the unconditional join blocks the abort for the + # length of a full clone and offline index. + cancel_event.set() return outage_streak, True return outage_streak, False -# A sustained upstream outage shows up as a run of session/infra/cleanup -# failures. (cleanup-failure overwrites the primary error_kind, so a -# session-error whose worktree cleanup also failed still counts.) A task's own -# resolved=False is real signal, not an outage, so it never trips the breaker. # How far ahead of the in-order fold pointer cells may be submitted, as a # multiple of the worker count. This is the wall-clock/wasted-cell trade, and it # is a real one - measured against the review corpus at workers=3, with failures @@ -1082,6 +1173,8 @@ class TaskCellContext: candidate_overlay: Path | None overlay_digest: str | None sandbox_backend: str = "bwrap" + clone_template: Path | None = None + sanitized_head: str | None = None def run_cell(ctx: TaskCellContext, run_idx: int, arm: str) -> dict[str, Any]: @@ -1106,8 +1199,14 @@ def run_cell(ctx: TaskCellContext, run_idx: int, arm: str) -> dict[str, Any]: raise RuntimeError("sanitized graph snapshot is unavailable") if ctx.asset_snapshot is None: raise RuntimeError("task asset snapshot is unavailable") - worktree = make_worktree(ctx.repo, ctx.task_sha, ctx.trees_dir) - sanitized_head = sanitize_clone_for_hidden_oracles(worktree) + if ctx.clone_template is not None: + if not ctx.sanitized_head: + raise RuntimeError("clone template is missing its sanitized HEAD") + worktree = copy_isolated_tree(ctx.clone_template, ctx.trees_dir) + sanitized_head = ctx.sanitized_head + else: + worktree = make_worktree(ctx.repo, ctx.task_sha, ctx.trees_dir) + sanitized_head = sanitize_clone_for_hidden_oracles(worktree) ctx.graph_snapshot.materialize(worktree, sanitized_head=sanitized_head) dependency_mounts = stage_task_assets( task, @@ -1212,7 +1311,7 @@ def run_cell(ctx: TaskCellContext, run_idx: int, arm: str) -> dict[str, Any]: oracle_snapshot=ctx.oracle_snapshot, ) if execution_arm in ("review", "ce_review"): - review_source = worktree / REVIEW_OUTPUT + review_source = review_output_path(sandbox, REVIEW_OUTPUT) if review_source.is_file() and not review_source.is_symlink(): review_artifact = ctx.out_dir / f"{task['id']}-{arm}-run{run_idx}.review.json" review_artifact.write_bytes(_bounded_regular_bytes(review_source, limit=256 * 1024)) @@ -1252,9 +1351,10 @@ def run_cell(ctx: TaskCellContext, run_idx: int, arm: str) -> dict[str, Any]: "task_base_sha": ctx.task_sha, "sanitized_task_sha": sanitized_head, "variant_head_sha": orig_sha, - "task_prompt_digest": hashlib.sha256(task["prompt"].encode()).hexdigest(), + "task_prompt_digest": task_prompt_digest(task), "skill_digest": expected_skill_digest, "candidate_overlay_digest": (ctx.overlay_digest if arm in CANDIDATE_ARMS else None), + "runtime_digest": current_runtime_digest(), "recorded_at": datetime.now(UTC).isoformat(), } ) @@ -1441,7 +1541,30 @@ def aggregate(records: list[dict[str, Any]]) -> dict[str, Any]: out["cost_usd"] = ( None if (not valid or any(cost is None for cost in valid_costs)) else statistics.median(valid_costs) ) + fresh = [r for r in records if not r.get("reused")] + out["fresh_attempts"] = len(fresh) + out["execution_failures"] = sum(1 for r in fresh if execution_failed(r)) + out["evidence_failures"] = sum(1 for r in fresh if evidence_failed(r)) + # Admissible means the harness delivered a trustworthy measurement. It says + # nothing about whether the answer was right, which is the whole point. + # + # Count the rows that failed NEITHER way rather than subtracting both + # counters: run_arm keeps a pre-existing session error and still marks the + # review evidence invalid, so one row can land in both. Subtracting it twice + # drove an arm holding real measurements to admissible=0, which arm_health + # reads as UNUSABLE and enforce_measurement_health then fails the sweep on. + out["admissible"] = sum(1 for r in fresh if not execution_failed(r) and not evidence_failed(r)) + out["health_reasons"] = sorted( + { + str(r.get("error_kind")) + for r in fresh + if r.get("error_kind") in EXECUTION_FAILURE_KINDS or r.get("error_kind") in EVIDENCE_FAILURE_KINDS + } + ) out["resolved"] = sum(1 for r in records if r["resolved"]) + # Reused rows are last generation's measurement. The health canary below has + # to ask whether THIS environment worked, so it needs the freshly-run count. + out["resolved_fresh"] = sum(1 for r in records if r["resolved"] and not r.get("reused")) out["runs"] = len(records) out["valid_runs"] = len(valid) out["excluded_runs"] = len(records) - len(valid) @@ -1499,11 +1622,135 @@ def savings(baseline: dict[str, Any], workflow: dict[str, Any]) -> dict[str, Any return out +@dataclass(frozen=True) +class ArmHealth: + """What the harness observed for one arm this sweep, before any judgement.""" + + arm: str + fresh_attempts: int + admissible: int + execution_failures: int + evidence_failures: int + reasons: tuple[str, ...] + + @property + def measured(self) -> bool: + """False when only reused rows exist - current health is UNKNOWN, not good.""" + + return self.fresh_attempts > 0 + + @property + def status(self) -> str: + """UNKNOWN / OBSERVED_OK / DEGRADED / UNUSABLE. + + DEGRADED is the distinction that matters: an arm with both admissible + measurements and observed failures produced usable evidence but did not + run reliably. Reporting that as healthy is how a partly-broken sweep + looks fine. It is diagnostic here - only UNUSABLE is fatal - so this + patch changes what is reported, not what is eligible. + """ + + if not self.measured: + return "UNKNOWN" + failures = self.execution_failures + self.evidence_failures + if self.admissible == 0 and failures > 0: + return "UNUSABLE" + if failures > 0: + return "DEGRADED" + return "OBSERVED_OK" + + @property + def unhealthy(self) -> bool: + """Every fresh attempt failed to execute or to produce usable evidence. + + Deliberately not "resolved zero tasks". A reviewer can be wrong about + every task in a hard corpus with the harness working perfectly; that is + a valid negative and belongs to the quality gate, not here. + """ + + return self.status == "UNUSABLE" + + +def arm_health(results: dict[str, dict[str, dict[str, Any]]], arms: set[str]) -> dict[str, ArmHealth]: + """Fold per-task aggregates into one health observation per arm.""" + + health: dict[str, ArmHealth] = {} + for arm in sorted(arms): + rows = [task_arms[arm] for task_arms in results.values() if arm in task_arms] + if not rows: + continue + reasons: set[str] = set() + for row in rows: + reasons.update(row.get("health_reasons") or ()) + health[arm] = ArmHealth( + arm=arm, + fresh_attempts=sum(int(r.get("fresh_attempts", 0)) for r in rows), + admissible=sum(int(r.get("admissible", 0)) for r in rows), + execution_failures=sum(int(r.get("execution_failures", 0)) for r in rows), + evidence_failures=sum(int(r.get("evidence_failures", 0)) for r in rows), + reasons=tuple(sorted(reasons)), + ) + return health + + +def unhealthy_arms(results: dict[str, dict[str, dict[str, Any]]], arms: set[str]) -> list[ArmHealth]: + """Arms whose every fresh attempt failed to execute or to produce evidence.""" + + return [h for h in arm_health(results, arms).values() if h.unhealthy] + + +def unmeasured_arms(results: dict[str, dict[str, dict[str, Any]]], arms: set[str]) -> list[str]: + """Arms with no fresh attempt at all - reported as unknown, never as healthy.""" + + return [h.arm for h in arm_health(results, arms).values() if not h.measured] + + +def enforce_measurement_health( + results: dict[str, dict[str, dict[str, Any]]], arms: set[str] +) -> dict[str, ArmHealth]: + """Report every arm's measurement status; abort only on UNUSABLE. + + Runs after report.md and promotion.json are written, so a failing sweep + still leaves its evidence behind. Reports cause as undetermined: an empty + artifact establishes that evidence is unusable, not why - naming a mount + failure here would be a guess the recorded rows do not support. + """ + + health = arm_health(results, arms) + for arm in sorted(health): + observed = health[arm] + reasons = f" reason={','.join(observed.reasons)}" if observed.reasons else "" + print( + f"[measurement-health] {arm}: {observed.status} " + f"fresh_attempts={observed.fresh_attempts} admissible={observed.admissible} " + f"execution_failures={observed.execution_failures} " + f"evidence_failures={observed.evidence_failures}{reasons}" + ) + unusable = [h for h in health.values() if h.unhealthy] + if unusable: + detail = "; ".join(f"{h.arm} ({h.fresh_attempts} fresh attempt(s))" for h in unusable) + print( + f"[measurement-health] {detail} produced no usable measurement this sweep. " + "cause=undetermined — see error_detail in results.jsonl. Exiting non-zero rather " + "than reporting a quiet no-promotion." + ) + raise SystemExit(1) + return health + + def broken_incumbent_arms( results: dict[str, dict[str, dict[str, Any]]], incumbent_arms: set[str], ) -> list[str]: - """Incumbent arms that resolved nothing across every task they ran. + """LEGACY, NON-AUTHORITATIVE. Superseded by ``enforce_measurement_health``. + + Kept only so its historical behaviour stays documented and testable while + the replacement settles; it has no production caller. Do not wire it into a + health decision - it infers a broken environment from a resolution count, + which a reviewer facing a hard corpus falsifies. Remove once the + measurement-health path has run in CI. + + Incumbent arms that resolved nothing across every task they ran. An incumbent arm is the currently-shipped, presumably-working skill: if it resolves NOTHING across every task it ran, that reads as an environment or @@ -1521,9 +1768,19 @@ def broken_incumbent_arms( some-runs-resolved-zero case since here nothing completed at all. aggregate() never marks an excluded/unverifiable row resolved=True, so resolved == 0 alone already covers both cases. + + A reused row proves last generation's environment worked, not this one's, so + the count consulted here is ``resolved_fresh``. Without that, an arm whose + cells were all reused always looks healthy and the canary can never fire — + which is exactly when a broken environment would go unnoticed. The sweep + keeps one paid cell per incumbent arm so this count is never vacuous. """ present = incumbent_arms & {arm for arms in results.values() for arm in arms} - return sorted(arm for arm in present if all(arms[arm]["resolved"] == 0 for arms in results.values() if arm in arms)) + return sorted( + arm + for arm in present + if all(arms[arm].get("resolved_fresh", arms[arm]["resolved"]) == 0 for arms in results.values() if arm in arms) + ) def _cost_cell(value: Any) -> str: @@ -1755,6 +2012,15 @@ def build_parser() -> argparse.ArgumentParser: parser.add_argument("--task-bindings-json", default=None, help=argparse.SUPPRESS) parser.add_argument("--promotion-target-bases-json", default=None, help=argparse.SUPPRESS) parser.add_argument("--unsafe-no-bwrap", action="store_true", help=argparse.SUPPRESS) + parser.add_argument( + "--reuse-results", + type=Path, + default=None, + help="prior wfbench results dir whose incumbent/CE rows may be reused " + "when model, effort, tasks, oracles, skill bytes, and CE plugin still " + "match. Candidate arms always run. Used by evolve.py so a weekly " + "generation does not re-pay for an unchanged comparator.", + ) return parser @@ -1882,6 +2148,244 @@ def main() -> None: gateway.__exit__(None, None, None) +def _comparator_reuse_expectation( + *, + args: argparse.Namespace, + tasks: Sequence[Any], + task_bindings: Sequence[Mapping[str, Any]], + oracle_snapshots: Sequence[Any], + asset_snapshots: Mapping[str, Any], + sandbox_backend: str, + ce_plugin_snapshot: CePluginSnapshot | None, +) -> ComparatorReuseExpectation: + """Bind this sweep's immutable identity for comparator-row reuse.""" + + skill_digests: dict[str, str | None] = {} + for arm in args.arms: + execution = CANDIDATE_ARMS.get(arm, arm) + if execution in EVALUATED_ARM_SKILLS: + skill_digests[arm] = skill_fingerprint(HARNESS_ROOT, execution) + else: + skill_digests[arm] = None + task_locks: dict[str, TaskReuseBinding] = {} + for task, binding, oracle in zip(tasks, task_bindings, oracle_snapshots, strict=True): + task_locks[str(task["id"])] = TaskReuseBinding( + task_base_sha=str(binding["resolved_sha"]), + task_prompt_digest=task_prompt_digest(task), + oracle_digest=oracle.digest, + oracle_command_digest=oracle.command_digest, + oracle_manifest_digest=oracle.manifest_digest, + task_asset_manifest_digest=getattr( + asset_snapshots.get(str(task["id"])), "manifest_digest", None + ), + sandbox_dependency_manifest_digest=getattr( + asset_snapshots.get(str(task["id"])), "dependency_manifest_digest", None + ), + ) + return ComparatorReuseExpectation( + model=args.model, + effort=args.effort, + sandbox_backend=sandbox_backend, + runtime_digest=current_runtime_digest(), + now=datetime.now(UTC), + max_age=default_reuse_max_age(), + tasks=task_locks, + skill_digests=skill_digests, + ce_plugin_version=ce_plugin_snapshot.version if ce_plugin_snapshot is not None else None, + ce_plugin_manifest_digest=( + ce_plugin_snapshot.manifest_digest if ce_plugin_snapshot is not None else None + ), + ) + + +def task_has_planned_paid_cells( + task: Mapping[str, Any], + *, + arms: Sequence[str], + runs: int, + reusable_rows: Mapping[tuple[str, str, int], object], + reuse_source: Path | None, +) -> bool: + """True when at least one planned cell is not a reusable comparator row.""" + + task_id = str(task["id"]) + for run_idx in range(runs): + for arm in arms: + if reuse_source is None or (task_id, arm, run_idx) not in reusable_rows: + return True + return False + + +def drop_canary_reuse_key( + reusable_rows: dict[CellKey, dict[str, Any]], + *, + arm: str, + tasks: Sequence[Mapping[str, Any]], + runs: int, +) -> CellKey | None: + """Drop one reusable cell so an incumbent arm still measures THIS sweep. + + An arm reused end to end measures nothing about today's environment, and + arm_health would then be reading last week's health. + + Counted against the cells this sweep PLANS, not every key reuse selection + returned: selection accepts any non-negative prior run index, so a results + directory produced with more runs than this invocation leaves extra keys. + Comparing against those made the check false exactly when it mattered, and + the canary silently stopped firing while every planned cell stayed reused. + + Returns the dropped key, or None when the arm already has a paid cell. + """ + + planned_keys = [(str(task["id"]), arm, run_idx) for task in tasks for run_idx in range(runs)] + arm_keys = sorted(key for key in planned_keys if key in reusable_rows) + if not arm_keys or len(arm_keys) != len(planned_keys): + return None + dropped = arm_keys[0] + del reusable_rows[dropped] + return dropped + + +def next_graph_prefetch_target( + remaining: Sequence[tuple[Mapping[str, Any], Mapping[str, Any]]], + *, + arms: Sequence[str], + runs: int, + reusable_rows: Mapping[tuple[str, str, int], object], + reuse_source: Path | None, + ready_keys: set[tuple[str, str]], +) -> tuple[Mapping[str, Any], Mapping[str, Any], tuple[str, str]] | None: + """Next later task that still needs a clone template and sanitized graph.""" + + for task, binding in remaining: + if not task_has_planned_paid_cells( + task, + arms=arms, + runs=runs, + reusable_rows=reusable_rows, + reuse_source=reuse_source, + ): + continue + key = (str(binding["repo_identity"]), str(binding["resolved_sha"])) + if key in ready_keys: + continue + return task, binding, key + return None + + +@dataclass(frozen=True) +class GraphBuildEnv: + """Per-sweep state every graph build shares, and the caches it fills. + + The four dicts are the sweep's memo of what has already been built, keyed by + (repo, sha). They are mutable by design and are written by both the sweep + thread and the prefetch thread, which is safe only because a build is + started for a key exactly once and joined before that key is read. + """ + + trees: Path + task_asset_cache: TaskAssetCache + claude_bin: Path | str + bwrap_bin: Path | str + sandbox_backend: str + runtime_mounts: Sequence[ReadOnlyMount] + clone_templates: dict[tuple[str, str], tuple[Path, str]] + clone_template_errors: dict[tuple[str, str], BaseException] + graph_snapshots: dict[tuple[str, str], SanitizedGraphSnapshot] + graph_snapshot_errors: dict[tuple[str, str], BaseException] + + def ready_keys(self) -> set[tuple[str, str]]: + """Keys whose build has already been attempted, successfully or not.""" + + return ( + set(self.clone_templates) + | set(self.clone_template_errors) + | set(self.graph_snapshots) + | set(self.graph_snapshot_errors) + ) + + +def ensure_task_graph( + *, + task: Mapping[str, Any], + repo: Path, + task_sha: str, + graph_key: tuple[str, str], + env: GraphBuildEnv, +) -> None: + """Build one SHA's sanitized clone template and graph. Idempotent per key.""" + + if graph_key in env.graph_snapshots or graph_key in env.graph_snapshot_errors: + return + try: + validate_no_prebuilt_graph_assets(task) + if graph_key not in env.clone_templates and graph_key not in env.clone_template_errors: + template = make_worktree(repo, task_sha, env.trees) + template_head = sanitize_clone_for_hidden_oracles(template) + env.clone_templates[graph_key] = (template, template_head) + clone_template: Path | None = None + template_head: str | None = None + if graph_key in env.clone_templates: + clone_template, template_head = env.clone_templates[graph_key] + if graph_key in env.clone_template_errors: + env.graph_snapshot_errors[graph_key] = env.clone_template_errors[graph_key] + return + env.graph_snapshots[graph_key] = prepare_sanitized_graph( + task, + repo=repo, + resolved_sha=task_sha, + parent=env.trees, + cache=env.task_asset_cache, + claude_bin=env.claude_bin, + bwrap_bin=env.bwrap_bin, + sandbox_backend=env.sandbox_backend, + runtime_mounts=env.runtime_mounts, + clone_template=clone_template, + sanitized_head=template_head, + ) + except (ManagedProcessError, OSError, SandboxError, RuntimeError, ValueError) as exc: + env.graph_snapshot_errors[graph_key] = exc + env.clone_template_errors.setdefault(graph_key, exc) + + +@dataclass +class GraphPrefetch: + """In-flight clone+graph build for a later task SHA.""" + + key: tuple[str, str] + thread: threading.Thread + + def join(self) -> None: + self.thread.join() + + +def prefetch_next_graph( + *, + task: Mapping[str, Any], + binding: Mapping[str, Any], + graph_key: tuple[str, str], + env: GraphBuildEnv, + cancel_event: threading.Event, +) -> GraphPrefetch: + """Start clone+graph prep for the next unpaid SHA during paid sessions.""" + + repo = Path(binding["repo_identity"]) + task_sha = str(binding["resolved_sha"]) + + def run() -> None: + if cancel_event.is_set(): + return + print(f"[prefetch_next_graph] clone+graph for {task_sha}") + ensure_task_graph(task=task, repo=repo, task_sha=task_sha, graph_key=graph_key, env=env) + + # copy_context, as the worker pool already does at _run_wave: a plain Thread + # does not inherit ContextVars, so without this every run_managed inside the + # graph build resolves _CANCELLATION to None and ignores the shared cancel. + thread = threading.Thread(target=copy_context().run, args=(run,), name="prefetch_next_graph", daemon=False) + thread.start() + return GraphPrefetch(key=graph_key, thread=thread) + + def _run_sweep( args: argparse.Namespace, *, @@ -1938,107 +2442,213 @@ def _run_sweep( except (OSError, SandboxError, ValueError) as exc: parser.error(str(exc)) raise AssertionError("ArgumentParser.error() returned unexpectedly") - graph_snapshots: dict[tuple[str, str], SanitizedGraphSnapshot] = {} - graph_snapshot_errors: dict[tuple[str, str], BaseException] = {} - for task, task_binding, oracle_snapshot in zip( - tasks, - task_bindings, - oracle_snapshots, - strict=True, - ): - if outage_tripped or cancel_event.is_set(): - break - repo = Path(task_binding["repo_identity"]) - task_sha = task_binding["resolved_sha"] - asset_snapshot: TaskAssetSnapshot | None = None - asset_snapshot_error: BaseException | None = None - graph_key = (str(repo), task_sha) - graph_snapshot: SanitizedGraphSnapshot | None = graph_snapshots.get(graph_key) - graph_snapshot_error: BaseException | None = graph_snapshot_errors.get(graph_key) + # Asset snapshots are built here, up front, for two reasons. Comparator + # reuse has to compare this sweep's task-asset and dependency digests + # against the prior row's, and those digests do not exist until the + # snapshot does. Building them all before any cell or prefetch thread + # starts also keeps TaskAssetCache single-threaded, which is what its own + # "plain dict, read-then-write race" comment asks for. + asset_snapshots: dict[str, TaskAssetSnapshot] = {} + asset_snapshot_errors: dict[str, BaseException] = {} + for _task, _binding in zip(tasks, task_bindings, strict=True): try: - validate_no_prebuilt_graph_assets(task) - if graph_snapshot is None and graph_snapshot_error is None: - graph_snapshot = prepare_sanitized_graph( - task, - repo=repo, - resolved_sha=task_sha, - parent=Path(trees), - cache=task_asset_cache, - claude_bin=args.claude_bin, - bwrap_bin=bwrap_bin, - sandbox_backend=sandbox_backend, - runtime_mounts=runtime_mounts, - ) - graph_snapshots[graph_key] = graph_snapshot - except (ManagedProcessError, OSError, SandboxError, RuntimeError, ValueError) as exc: - graph_snapshot_error = exc - graph_snapshot_errors[graph_key] = exc - try: - # Prepared here, once, rather than lazily inside the first cell: - # TaskAssetCache is a plain dict, so a lazy build would be a - # read-then-write race the moment cells stop running serially. - asset_snapshot = task_asset_cache.prepare( - task, - repo=repo, - resolved_sha=task_sha, - expected_dependency_binding=task_binding, + asset_snapshots[str(_task["id"])] = task_asset_cache.prepare( + _task, + repo=Path(_binding["repo_identity"]), + resolved_sha=_binding["resolved_sha"], + expected_dependency_binding=_binding, ) except (OSError, SandboxError, ValueError) as exc: - asset_snapshot_error = exc - cell_context = TaskCellContext( - task=task, - oracle_snapshot=oracle_snapshot, - repo=repo, - task_sha=task_sha, - graph_snapshot=graph_snapshot, - graph_snapshot_error=graph_snapshot_error, - asset_snapshot=asset_snapshot, - asset_snapshot_error=asset_snapshot_error, - args=args, - out_dir=out_dir, - ce_plugin_snapshot=ce_plugin_snapshot, - trees_dir=Path(trees), - bwrap_bin=bwrap_bin, - sandbox_backend=sandbox_backend, - runtime_mounts=runtime_mounts, - candidate_overlay=candidate_overlay, - overlay_digest=overlay_digest, - ) - per_arm: dict[str, list[dict[str, Any]]] = {a: [] for a in args.arms} - cells = [(run_idx, arm) for run_idx in range(args.runs) for arm in args.arms] + asset_snapshot_errors[str(_task["id"])] = exc - def announce(run_idx: int, arm: str) -> None: - nonlocal started_cells - started_cells += 1 - print( - f"[{task['id']}][{arm}][run {run_idx}] starting " - f"({started_cells}/{total_cells}, {(time.monotonic() - sweep_started) / 60:.0f}m elapsed)" + reuse_source = args.reuse_results.expanduser().resolve() if args.reuse_results is not None else None + reusable_rows: dict[tuple[str, str, int], dict[str, Any]] = {} + if reuse_source is not None: + if reuse_source == out_dir.resolve(): + parser.error("--reuse-results cannot be this sweep's --out directory") + raise AssertionError("ArgumentParser.error() returned unexpectedly") + results_file = reuse_source / "results.jsonl" + if results_file.is_symlink() or not results_file.is_file(): + parser.error("--reuse-results must contain a regular results.jsonl") + raise AssertionError("ArgumentParser.error() returned unexpectedly") + reusable_rows = select_reusable_comparator_rows( + load_result_rows(results_file), + expected=_comparator_reuse_expectation( + args=args, + tasks=tasks, + task_bindings=task_bindings, + oracle_snapshots=oracle_snapshots, + asset_snapshots=asset_snapshots, + sandbox_backend=sandbox_backend, + ce_plugin_snapshot=ce_plugin_snapshot, + ), + ) + # Keep one paid cell per incumbent arm. A generation that reuses an + # arm end to end measures nothing about today's environment, and + # arm_health would then be reading last week's health. + # One cell per arm is the cheapest thing that keeps the canary real. + incumbent_arms = [arm for arm in args.arms if arm not in candidate_arms] + for arm in incumbent_arms: + dropped = drop_canary_reuse_key(reusable_rows, arm=arm, tasks=tasks, runs=args.runs) + if dropped is not None: + print( + f"reuse-results: keeping one paid {arm} cell " + f"({dropped[0]} run {dropped[2]}) so incumbent health is measured this sweep" + ) + print( + f"reuse-results {reuse_source}: {len(reusable_rows)} comparator " + f"cell(s) match this sweep; candidate arms always run" + ) + graph_env = GraphBuildEnv( + trees=Path(trees), + task_asset_cache=task_asset_cache, + claude_bin=args.claude_bin, + bwrap_bin=bwrap_bin, + sandbox_backend=sandbox_backend, + runtime_mounts=runtime_mounts, + clone_templates={}, + clone_template_errors={}, + graph_snapshots={}, + graph_snapshot_errors={}, + ) + graph_prefetch: GraphPrefetch | None = None + sweep_rows = list(zip(tasks, task_bindings, oracle_snapshots, strict=True)) + + def _join_graph_prefetch() -> None: + nonlocal graph_prefetch + if graph_prefetch is not None: + graph_prefetch.join() + graph_prefetch = None + + try: + for index, (task, task_binding, oracle_snapshot) in enumerate(sweep_rows): + if outage_tripped or cancel_event.is_set(): + break + repo = Path(task_binding["repo_identity"]) + task_sha = task_binding["resolved_sha"] + per_arm: dict[str, list[dict[str, Any]]] = {a: [] for a in args.arms} + planned = [(run_idx, arm) for run_idx in range(args.runs) for arm in args.arms] + reused_records: list[tuple[int, str, dict[str, Any]]] = [] + paid_cells: list[tuple[int, str]] = [] + for run_idx, arm in planned: + prior = reusable_rows.get((task["id"], arm, run_idx)) + if prior is None or reuse_source is None: + paid_cells.append((run_idx, arm)) + continue + try: + reused_records.append( + ( + run_idx, + arm, + materialize_reused_row(prior, source_dir=reuse_source, dest_dir=out_dir), + ) + ) + except (OSError, SandboxError, ValueError) as exc: + print( + f"[{task['id']}][{arm}][run {run_idx}] comparator reuse " + f"failed ({exc}); running a paid cell" + ) + paid_cells.append((run_idx, arm)) + + asset_snapshot = asset_snapshots.get(str(task["id"])) + asset_snapshot_error: BaseException | None = asset_snapshot_errors.get(str(task["id"])) + graph_key = (str(repo), task_sha) + if graph_prefetch is not None and graph_prefetch.key == graph_key: + _join_graph_prefetch() + if paid_cells: + ensure_task_graph( + task=task, repo=repo, task_sha=task_sha, graph_key=graph_key, env=graph_env + ) + graph_snapshot = graph_env.graph_snapshots.get(graph_key) + graph_snapshot_error = graph_env.graph_snapshot_errors.get(graph_key) + clone_template, template_head = graph_env.clone_templates.get(graph_key, (None, None)) + cell_context = TaskCellContext( + task=task, + oracle_snapshot=oracle_snapshot, + repo=repo, + task_sha=task_sha, + graph_snapshot=graph_snapshot, + graph_snapshot_error=graph_snapshot_error, + asset_snapshot=asset_snapshot, + asset_snapshot_error=asset_snapshot_error, + args=args, + out_dir=out_dir, + ce_plugin_snapshot=ce_plugin_snapshot, + trees_dir=Path(trees), + bwrap_bin=bwrap_bin, + sandbox_backend=sandbox_backend, + runtime_mounts=runtime_mounts, + candidate_overlay=candidate_overlay, + overlay_digest=overlay_digest, + clone_template=clone_template, + sanitized_head=template_head, ) - def keep(run_idx: int, arm: str, record: dict[str, Any]) -> None: - per_arm[arm].append(record) - with results_path.open("a") as fh: - # Redact any API token a session-error stderr_tail echoed - # into error_detail before it enters the uploaded - # results.jsonl artifact (transcripts are redacted; this - # sink was not). - fh.write(redact_text(json.dumps(record), credential_secrets(args)) + "\n") - print(cell_progress_line(task["id"], arm, run_idx, record)) - failure = cell_failure_detail_line(task["id"], arm, run_idx, record, credential_secrets(args)) - if failure: - print(failure) + def announce(run_idx: int, arm: str) -> None: + nonlocal started_cells + started_cells += 1 + print( + f"[{task['id']}][{arm}][run {run_idx}] starting " + f"({started_cells}/{total_cells}, {(time.monotonic() - sweep_started) / 60:.0f}m elapsed)" + ) - outage_streak, outage_tripped = sweep_task_cells( - cells, - workers=args.workers, - run=partial(run_cell, cell_context), - on_start=announce, - on_record=keep, - outage_streak=outage_streak, - outage_limit=args.outage_streak, - cancel_event=cancel_event, - ) - results[task["id"]] = {a: aggregate(rs) for a, rs in per_arm.items() if rs} + def keep(run_idx: int, arm: str, record: dict[str, Any]) -> None: + per_arm[arm].append(record) + with results_path.open("a") as fh: + # Redact any API token a session-error stderr_tail echoed + # into error_detail before it enters the uploaded + # results.jsonl artifact (transcripts are redacted; this + # sink was not). + fh.write(redact_text(json.dumps(record), credential_secrets(args)) + "\n") + print(cell_progress_line(task["id"], arm, run_idx, record)) + failure = cell_failure_detail_line(task["id"], arm, run_idx, record, credential_secrets(args)) + if failure: + print(failure) + + for run_idx, arm, record in reused_records: + started_cells += 1 + print( + f"[{task['id']}][{arm}][run {run_idx}] reused comparator " + f"({started_cells}/{total_cells}, {(time.monotonic() - sweep_started) / 60:.0f}m elapsed)" + ) + keep(run_idx, arm, record) + if paid_cells and graph_prefetch is None and not cancel_event.is_set(): + target = next_graph_prefetch_target( + [(later_task, later_binding) for later_task, later_binding, _ in sweep_rows[index + 1 :]], + arms=args.arms, + runs=args.runs, + reusable_rows=reusable_rows, + reuse_source=reuse_source, + ready_keys=graph_env.ready_keys(), + ) + if target is not None: + later_task, later_binding, later_key = target + graph_prefetch = prefetch_next_graph( + task=later_task, + binding=later_binding, + graph_key=later_key, + env=graph_env, + cancel_event=cancel_event, + ) + # A reused success is evidence the pipeline can produce a good row, + # so it resets the consecutive-failure count the same way a paid + # success does. Leaving reused rows out let a streak carry across + # them and trip on stale history. + for _run_idx, _arm, _record in reused_records: + outage_streak = systemic_outage_streak(_record.get("error_kind"), outage_streak) + outage_streak, outage_tripped = sweep_task_cells( + paid_cells, + workers=args.workers, + run=partial(run_cell, cell_context), + on_start=announce, + on_record=keep, + outage_streak=outage_streak, + outage_limit=args.outage_streak, + cancel_event=cancel_event, + ) + results[task["id"]] = {a: aggregate(rs) for a, rs in per_arm.items() if rs} + finally: + _join_graph_prefetch() selection_report = [ "## Run provenance", @@ -2056,10 +2666,12 @@ def _run_sweep( selection_report.append( f"Compound Engineering plugin: `{ce_plugin_snapshot.version}` (`{ce_plugin_snapshot.manifest_digest}`)" ) - if cancel_event.is_set(): - selection_report.append("Sweep cancelled: partial evidence; promotion is disabled.") - elif outage_tripped: + # outage first: the breaker now sets cancel_event to stop in-flight background + # work, so testing cancellation first would relabel every outage a cancellation. + if outage_tripped: selection_report.append("Sweep aborted: partial evidence; promotion is disabled.") + elif cancel_event.is_set(): + selection_report.append("Sweep cancelled: partial evidence; promotion is disabled.") report = render_report(results) + "\n\n" + "\n".join(selection_report) + "\n" (out_dir / "report.md").write_text(report) if candidate_arms: @@ -2092,26 +2704,22 @@ def _run_sweep( } (out_dir / "promotion.json").write_text(json.dumps(promotion, indent=2) + "\n") print(f"\n{report}\n\nWritten to {out_dir}/") - # A reviewer may legitimately match none of a difficult hidden corpus; - # unlike an implementation arm, zero exact resolutions is quality signal, - # not proof that the harness failed. - broken_incumbents = broken_incumbent_arms(results, set(CANDIDATE_ARMS.values()) - {"review"}) - if broken_incumbents: - # Fail loudly rather than let a broken environment read as a quiet - # "no promotion, incumbent stands." - print( - f"[harness-health] incumbent arm(s) {', '.join(broken_incumbents)} resolved zero " - "tasks across every valid run — this looks like an environment/harness failure, " - "not a normal candidate miss. See the errors column in report.md and error_detail " - "in results.jsonl. Exiting non-zero rather than reporting a quiet no-promotion." - ) - raise SystemExit(1) - if cancel_event.is_set(): - raise SystemExit(130) + # Health is judged on whether fresh attempts EXECUTED and produced usable + # evidence - never on how many tasks they resolved. Review arms used to be + # excluded here because "resolved zero" is quality signal for a reviewer + # facing a hard corpus; with the inference corrected they are included + # again, which is what lets an all-artifacts-empty run be caught at all. + # ce_review is named explicitly: it is a comparator, not a candidate, so it + # is absent from CANDIDATE_ARMS and would otherwise go unclassified. + enforce_measurement_health(results, set(CANDIDATE_ARMS.values()) | {"review", "ce_review"}) if outage_tripped: # Non-zero exit so a driver (evolve.py) treats the partial benchmark as a # failed run and halts instead of proposing from outage-truncated evidence. + # Checked before cancel_event because the breaker sets it (see above), and + # an outage must keep exit 1 rather than becoming the 130 of a Ctrl-C. raise SystemExit(1) + if cancel_event.is_set(): + raise SystemExit(130) if __name__ == "__main__": diff --git a/eval/workflow_bench/runner_artifacts.py b/eval/workflow_bench/runner_artifacts.py index 7f05e14b1..3780b9e46 100644 --- a/eval/workflow_bench/runner_artifacts.py +++ b/eval/workflow_bench/runner_artifacts.py @@ -217,11 +217,25 @@ def enforce_phase_workspace( worktree: Path, before: dict[str, str], *, - allowed_artifact: Path, + allowed_artifact: Path | None, ) -> None: - """Require a phase to change only its one explicit workspace artifact.""" + """Require a phase to change only its one explicit workspace artifact. + + ``allowed_artifact=None`` is the stricter contract: the phase must leave + the workspace byte-identical. That is what a review phase whose artifact + lives outside the workspace has to satisfy — there is nothing in there it + is entitled to touch. + """ root = worktree.expanduser().absolute() + if allowed_artifact is None: + after = workspace_snapshot(root) + changed = sorted( + path for path in before.keys() | after.keys() if before.get(path) != after.get(path) + ) + if changed: + raise ValueError(f"phase changed the read-only workspace: {', '.join(changed[:5])}") + return artifact = allowed_artifact.expanduser().absolute() try: relative = PurePosixPath(artifact.relative_to(root).as_posix()) @@ -325,6 +339,65 @@ def new_plan_doc(worktree: Path, before: dict[Path, str]) -> Path: return changed[0] +def _assert_self_contained_git_objects(clone: Path) -> None: + """Refuse clones that share pack/object bytes with another repository.""" + + alternates = clone / ".git" / "objects" / "info" / "alternates" + if alternates.exists(): + raise RuntimeError(f"clone unexpectedly has an external object alternate: {alternates}") + objects = clone / ".git" / "objects" + if not objects.is_dir(): + raise RuntimeError(f"clone is missing a git object store: {clone}") + for obj in objects.rglob("*"): + if obj.is_file() and obj.stat().st_nlink > 1: + raise RuntimeError(f"clone object is hardlinked to host storage: {obj}") + + +def copy_isolated_tree(source: Path, parent: Path) -> Path: + """Copy a sanitized clone without sharing git objects or a ref namespace. + + ``git clone --no-local`` of GitNexus plus ``sanitize_clone_for_hidden_oracles`` + (repack/prune/fsck) is minutes per cell. After sanitization the snapshot is + one parentless commit; copying that tree is the isolation boundary the + contamination bug actually required (a private ``.git``), not a second + fetch of full history. Prefer ``cp --reflink=auto`` so XFS/btrfs pay COW; + fall back to a full copy on filesystems that cannot reflink. + """ + + try: + source_meta = source.expanduser().lstat() + except OSError as exc: + raise RuntimeError(f"clone template is unavailable: {source}: {exc}") from exc + if stat.S_ISLNK(source_meta.st_mode) or not stat.S_ISDIR(source_meta.st_mode): + raise RuntimeError(f"clone template must be a real directory: {source}") + source = source.expanduser().resolve() + target = Path(tempfile.mkdtemp(prefix="wfbench-", dir=parent)) + target.rmdir() + try: + copied = run_managed( + ["cp", "-a", "--reflink=auto", str(source), str(target)], + timeout=600, + ) + if not copied.ok: + # The fallback is for a filesystem that cannot reflink, which shows + # up as a normal nonzero exit. A cancellation or timeout is reported + # the same way (run_managed returns it rather than raising), and + # copytree cannot be cancelled — so falling back there makes the + # outage breaker wait out the full copy it set the event to avoid. + if copied.state != "exited": + raise ManagedProcessError(["cp", "-a", "--reflink=auto", str(source), str(target)], copied) + shutil.copytree(source, target, symlinks=True, copy_function=shutil.copy2) + _assert_self_contained_git_objects(target) + return target + except BaseException as primary: + if target.exists(): + try: + shutil.rmtree(target) + except OSError as cleanup: + primary.add_note(f"clone copy cleanup also failed: {type(cleanup).__name__}: {cleanup}") + raise + + def make_worktree(repo: Path, ref: str, parent: Path) -> Path: """Create a self-contained clone per benchmark arm.""" @@ -344,12 +417,7 @@ def make_worktree(repo: Path, ref: str, parent: Path) -> Path: ], timeout=600, ) - alternates = target / ".git" / "objects" / "info" / "alternates" - if alternates.exists(): - raise RuntimeError(f"clone unexpectedly has an external object alternate: {alternates}") - for obj in (target / ".git" / "objects").rglob("*"): - if obj.is_file() and obj.stat().st_nlink > 1: - raise RuntimeError(f"clone object is hardlinked to host storage: {obj}") + _assert_self_contained_git_objects(target) for candidate in (ref, f"origin/{ref}"): proc = run_managed( ["git", "-C", str(target), "checkout", "--detach", "--quiet", candidate], diff --git a/eval/workflow_bench/runner_sessions.py b/eval/workflow_bench/runner_sessions.py index 9daef00b4..053d5e48d 100644 --- a/eval/workflow_bench/runner_sessions.py +++ b/eval/workflow_bench/runner_sessions.py @@ -57,6 +57,24 @@ MAX_PROGRESS_PENDING = 256 MAX_PROGRESS_TOOL_ID_CHARS = 256 MAX_TOOL_PREVIEW_CHARS = 800 _SAFE_TOOL_NAME = re.compile(r"[A-Za-z0-9._:-]{1,64}") +_GHA_WORKFLOW_COMMAND = re.compile(r"(^|[\n\r])::") +_GHA_HASH_COMMAND = re.compile(r"##\[") +_GHA_COMPILER_ANNOTATION = re.compile(r"\((\d+),(\d+)\):\s+error\b", re.IGNORECASE) + + +def neutralize_ci_log_text(text: str) -> str: + """Stop GitHub Actions from promoting tool output into check annotations. + + Run 33962002890 logged in-sandbox ``tsc`` failures as + ``file.ts(line,col): error TS2307``, which Actions parsed as workflow + annotations on ``.github``. The same parser treats ``::error::`` and + ``##[error]`` as commands. Progress previews are evidence, not CI + signaling, so rewrite those forms before they hit the job log. + """ + + text = _GHA_WORKFLOW_COMMAND.sub(r"\1[:]", text) + text = _GHA_HASH_COMMAND.sub("# [", text) + return _GHA_COMPILER_ANNOTATION.sub(r"(\1,\2): compiler-error", text) def _safe_tool_name(value: Any) -> str: @@ -177,7 +195,9 @@ class SessionProgress: def _say(self, message: str) -> None: # Queue only: the stdout drain thread calls observe() and must not # block on a full log pipe (process_control.stdout_observer contract). - self._pending_messages.append(f"[{self.label} {self._elapsed()}] {message}") + self._pending_messages.append( + neutralize_ci_log_text(f"[{self.label} {self._elapsed()}] {message}") + ) self._last_spoke = time.monotonic() def _emit_pending(self) -> None: diff --git a/eval/workflow_bench/sanitized_graph.py b/eval/workflow_bench/sanitized_graph.py index 3aea9e8c1..fc3d766a2 100644 --- a/eval/workflow_bench/sanitized_graph.py +++ b/eval/workflow_bench/sanitized_graph.py @@ -24,7 +24,7 @@ from .proposer_sandbox import ( build_sandbox_environment, prepare_sandbox, ) -from .runner_artifacts import make_worktree, remove_clone +from .runner_artifacts import copy_isolated_tree, make_worktree, remove_clone from .task_assets import TaskAssetCache, TaskAssetSnapshot, _is_harness_sandbox_copy GRAPH_ASSET_PATHS = ( @@ -360,14 +360,29 @@ def prepare_sanitized_graph( bwrap_bin: Path | str, runtime_mounts: Sequence[ReadOnlyMount], sandbox_backend: str = "bwrap", + clone_template: Path | None = None, + sanitized_head: str | None = None, ) -> SanitizedGraphSnapshot: - """Sanitize, index offline once, scrub, and freeze graph assets for all arms.""" + """Sanitize, index offline once, scrub, and freeze graph assets for all arms. + + When ``clone_template`` is an already-sanitized snapshot, this copies it + (the copy is scrubbed and indexed) so the template stays a clean cell + seed. Callers that already paid for ``make_worktree`` + sanitization + should pass that template rather than cloning GitNexus again. + """ validate_no_prebuilt_graph_assets(task) - seed = make_worktree(repo, resolved_sha, parent) + if clone_template is not None: + if not isinstance(sanitized_head, str) or not sanitized_head: + raise SandboxError("clone template requires the sanitized HEAD") + seed = copy_isolated_tree(clone_template, parent) + else: + seed = make_worktree(repo, resolved_sha, parent) + sanitized_head = None primary: BaseException | None = None try: - sanitized_head = sanitize_clone_for_hidden_oracles(seed) + if sanitized_head is None: + sanitized_head = sanitize_clone_for_hidden_oracles(seed) _scrub_source_references(seed) _neutralize_target_index_inputs(seed) with prepare_sandbox( diff --git a/eval/workflow_bench/tasks.review.scenarios.yaml b/eval/workflow_bench/tasks.review.scenarios.yaml index 8ff323ed1..4d68ae871 100644 --- a/eval/workflow_bench/tasks.review.scenarios.yaml +++ b/eval/workflow_bench/tasks.review.scenarios.yaml @@ -9,14 +9,17 @@ tasks: sandbox_copy: [eval/workflow_bench/review_cases/pr-2718.patch] setup: git apply --exclude='eval/workflow_bench/*' eval/workflow_bench/review_cases/pr-2718.patch && rm -rf eval/workflow_bench prompt: Review the historical snapshot of https://github.com/abhigyanpatwari/GitNexus/pull/2718. Report only actionable defects introduced by the local diff. - verify: test -s review-output.json + verify: test -s "$GITNEXUS_BENCH_REVIEW_OUTPUT" oracle: - command: test -s review-output.json + command: test -s "$GITNEXUS_BENCH_REVIEW_OUTPUT" files: [{ source: review-pr-2718-defect.labels.json, target: review-labels.json }] sandbox_dependencies: &deps - { source: node_modules, target: node_modules } - { source: gitnexus/node_modules, target: gitnexus/node_modules } - { source: gitnexus-shared/node_modules, target: gitnexus-shared/node_modules } + # Host-built types/JS. Historical clones have no dist/, so `tsc` in the + # read-only workspace otherwise reports TS2307/TS6379 (run 33962002890). + - { source: gitnexus-shared/dist, target: gitnexus-shared/dist } - <<: *review_case id: review-pr-2794-defect @@ -25,7 +28,7 @@ tasks: setup: git apply --exclude='eval/workflow_bench/*' eval/workflow_bench/review_cases/pr-2794.patch && rm -rf eval/workflow_bench prompt: Review the historical snapshot of https://github.com/abhigyanpatwari/GitNexus/pull/2794. Report only actionable defects introduced by the local diff. oracle: - command: test -s review-output.json + command: test -s "$GITNEXUS_BENCH_REVIEW_OUTPUT" files: [{ source: review-pr-2794-defect.labels.json, target: review-labels.json }] sandbox_dependencies: *deps @@ -36,7 +39,7 @@ tasks: setup: git apply --exclude='eval/workflow_bench/*' eval/workflow_bench/review_cases/pr-2108.patch && rm -rf eval/workflow_bench prompt: Review the historical snapshot of https://github.com/abhigyanpatwari/GitNexus/pull/2108. Report only actionable defects introduced by the local diff. oracle: - command: test -s review-output.json + command: test -s "$GITNEXUS_BENCH_REVIEW_OUTPUT" files: [{ source: review-pr-2108-defect.labels.json, target: review-labels.json }] sandbox_dependencies: *deps @@ -47,7 +50,7 @@ tasks: setup: git apply --exclude='eval/workflow_bench/*' eval/workflow_bench/review_cases/pr-2258.patch && rm -rf eval/workflow_bench prompt: Review the historical snapshot of https://github.com/abhigyanpatwari/GitNexus/pull/2258. Report only actionable defects introduced by the local diff. oracle: - command: test -s review-output.json + command: test -s "$GITNEXUS_BENCH_REVIEW_OUTPUT" files: [{ source: review-pr-2258-defect.labels.json, target: review-labels.json }] sandbox_dependencies: *deps @@ -59,7 +62,7 @@ tasks: setup: git apply --exclude='eval/workflow_bench/*' eval/workflow_bench/review_cases/pr-2258b.patch && rm -rf eval/workflow_bench prompt: Review this historical snapshot of https://github.com/abhigyanpatwari/GitNexus/pull/2258. Report only actionable defects introduced by the local diff. oracle: - command: test -s review-output.json + command: test -s "$GITNEXUS_BENCH_REVIEW_OUTPUT" files: [{ source: review-pr-2258-clean.labels.json, target: review-labels.json }] sandbox_dependencies: *deps @@ -71,6 +74,6 @@ tasks: setup: git apply --exclude='eval/workflow_bench/*' eval/workflow_bench/review_cases/pr-2773.patch && rm -rf eval/workflow_bench prompt: Review this historical snapshot of https://github.com/abhigyanpatwari/GitNexus/pull/2773. Report only actionable defects introduced by the local diff. oracle: - command: test -s review-output.json + command: test -s "$GITNEXUS_BENCH_REVIEW_OUTPUT" files: [{ source: review-pr-2773-clean.labels.json, target: review-labels.json }] sandbox_dependencies: *deps diff --git a/eval/workflow_bench/tasks.scenarios.yaml b/eval/workflow_bench/tasks.scenarios.yaml index af4755f88..59448ae4f 100644 --- a/eval/workflow_bench/tasks.scenarios.yaml +++ b/eval/workflow_bench/tasks.scenarios.yaml @@ -44,6 +44,11 @@ tasks: target: gitnexus/node_modules - source: gitnexus-shared/node_modules target: gitnexus-shared/node_modules + # Host-built types/JS. The clone has no dist/, and node_modules/gitnexus-shared + # is a relative symlink into that unbuilt tree — without this mount, in-sandbox + # `tsc --noEmit` / vitest fail with TS2307 / TS6379 (run 33962002890). + - source: gitnexus-shared/dist + target: gitnexus-shared/dist prompt: > Add -j as a short alias for --json on the gitnexus status command (gitnexus/src/cli/index.ts), and cover the alias with a unit test in diff --git a/gitnexus/test/unit/skill-evolution-workflow.test.ts b/gitnexus/test/unit/skill-evolution-workflow.test.ts index b10313e42..6916c7028 100644 --- a/gitnexus/test/unit/skill-evolution-workflow.test.ts +++ b/gitnexus/test/unit/skill-evolution-workflow.test.ts @@ -270,12 +270,13 @@ describe('gitnexus skill-evolution workflow contract', () => { }); it('passes the cell concurrency through to the benchmark', () => { - // The lane is serial unless told otherwise: concurrency only pays off when - // the runner has the vCPUs for it, and a cell starved of CPU drifts toward - // its session timeout, which the gate counts as an excluded run. + // Dispatch defaults to 3. Scheduled runs still fall back to serial unless + // GITNEXUS_EVOLUTION_WORKERS is set — a cell starved of CPU that hits the + // session ceiling is an excluded run the gate refuses. expect(evolveJob?.env?.WORKERS).toBe( "${{ inputs.workers || vars.GITNEXUS_EVOLUTION_WORKERS || '1' }}", ); + expect(workflow).toMatch(/workers:\n(?:[^\n]*\n){0,4} default: '3'/); }); it('seeds from the newest usable completed main run, including failed runs', () => { @@ -572,6 +573,27 @@ exit 1`); // uploads. The job must finish inside that window even when the schedule // fires late (the 2026-08-01 run was queued 65 minutes after the cron). expect(jobBudget as number).toBeLessThanOrEqual(21 * 60); + // A Friday dispatch inherits leftover uptime. The shared entrypoint — not + // the workflow YAML — must cap the sweep so it fails in-process and the + // always() upload still runs (run 33962002890). + const script = readFileSync( + path.join(REPO_ROOT, 'eval/workflow_bench/run-evolution.sh'), + 'utf8', + ); + // The flag, not a precomputed number: the CLI reads /proc/uptime in the + // same breath as it starts the clock the cap is measured against, so + // nothing between the two can be charged to the sweep. + expect(script).toContain('--max-runtime-from-instance-window'); + expect(script).not.toContain('instance_window_budget_from_proc'); + expect(script).toContain('export RUNTIME_DIGEST'); + // The workflow's half of that contract is calling the entrypoint, not + // naming the flag. The YAML never mentions --max-runtime-from-instance-window + // at all — only the prose at gitnexus-skill-evolution.yml:179-184 describing + // the cap — so an assertion on flag text there would test a comment, and a + // loop step that had stopped invoking the script would still pass it. + expect(stepRun('Run the propose → benchmark → gate loop')).toContain( + './workflow_bench/run-evolution.sh --apply', + ); }); it('uploads benchmark evidence unconditionally, on a path it addresses itself', () => { From 4757c0d3cbd65e8ec90ec13f771e50a29f425d2e Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Tue, 8 Sep 2026 10:58:41 +0100 Subject: [PATCH 13/17] chore(deps)(deps-dev): bump @types/node in /gitnexus (#3215) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Bumps [@types/node](https://github.com/DefinitelyTyped/DefinitelyTyped/tree/HEAD/types/node) from 26.4.0 to 26.4.1. - [Release notes](https://github.com/DefinitelyTyped/DefinitelyTyped/releases) - [Commits](https://github.com/DefinitelyTyped/DefinitelyTyped/commits/HEAD/types/node) --- updated-dependencies: - dependency-name: "@types/node" dependency-version: 26.4.1 dependency-type: direct:development update-type: version-update:semver-patch ... Signed-off-by: dependabot[bot] Co-authored-by: dependabot[bot] <49699333+dependabot[bot]@users.noreply.github.com> Co-authored-by: Gergő Magyar --- gitnexus/package-lock.json | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/gitnexus/package-lock.json b/gitnexus/package-lock.json index 6e521f664..caf8a1df0 100644 --- a/gitnexus/package-lock.json +++ b/gitnexus/package-lock.json @@ -1888,9 +1888,9 @@ "license": "MIT" }, "node_modules/@types/node": { - "version": "26.4.0", - "resolved": "https://registry.npmjs.org/@types/node/-/node-26.4.0.tgz", - "integrity": "sha512-faiGnoIrLH/V8cibOMEAZ8pMw6oXqSukl29ra4mN8GdaB2ZewzeaLj+INpV5N+Z1eKWzY+IzaIZH2EIR6YZRNQ==", + "version": "26.4.1", + "resolved": "https://registry.npmjs.org/@types/node/-/node-26.4.1.tgz", + "integrity": "sha512-k97ENvZWtvA6yqz5/FS6a7duDgOPEeOQOc2iKS/nY6mX6qJUKtLnWzQS+Xj6tXweyj6ZcTAK2Qecetnvi9nCLA==", "devOptional": true, "license": "MIT", "dependencies": { From 96132bd13a386388d09c1515c4f13ad0fd5e1cff Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Gerg=C5=91=20Magyar?= Date: Tue, 8 Sep 2026 13:48:30 +0100 Subject: [PATCH 14/17] perf(scope-resolution): stop re-scanning the ParsedFile store once per language (#3211) * perf(scope-resolution): stop re-scanning the ParsedFile store once per language Scope resolution calls `loadParsedFilesForPaths` once per language, and every call walks every shard in the store. The skip decision needs the envelope's path listing, and that listing is only trustworthy after the payload digest has been checked -- so a pass that wants 50 Python files still opens and SHA-256s all 413 shards / 301MB of a TypeScript-dominated store to prove it can skip them. A pass wanting a SINGLE file costs 335ms. The cost scales with language count, not with the files that language has, so a polyglot repo pays it worst. `tryLoadV8Cache` now returns the listing it already parsed for that skip decision, and the store memoizes it per run. Later passes skip on the memoized listing without reopening the file. Measured on a 2234-file, 3-language repo, min-of-5: before python 411ms typescript 2872ms javascript 507ms = 3834ms after python 415ms typescript 2805ms javascript 248ms = 3484ms -350ms here, roughly -250ms per additional language elsewhere. The first pass is unchanged by construction -- it is what populates the memo. End-to-end the graph is byte-identical: 51,288 nodes / 163,094 edges / 2106 clusters / 759 flows on a true incremental run. Keyed on size+mtime as well as name. Shard names are content-addressed, so a name collision across different content should be impossible, but that invariant lives in the parse-cache keying rather than here and one stat per shard is a few ms against the hundreds this saves. The memo holds one store directory at a time, so a new repo in a long-lived MCP process drops the previous set instead of accumulating. The failure mode a listing memo introduces is a FALSE SKIP: a pass concludes a shard holds nothing it wants and those files silently never reach the graph -- an exit-0 wrong answer, not a crash. The new test walks four passes with disjoint wants over one store, plus a shard written after the memo is warm; it fails when the skip is forced. Also records the full scopeResolution breakdown in bench/. The headline is that `emit` is 7161ms of the 14.7s phase and ~21% of the edit loop, spread across a fan of passes with no hot inner loop -- so the win there is not running them for unchanged files, which is a design rather than a patch. Co-Authored-By: Claude Opus 5 (1M context) * Address PR review feedback (#3211) Strengthen the shard-listing memo test so a later miss asserts fs.open and v8.deserialize never run for the skipped shard. Key-set checks alone still passed if the memo never skipped. Co-authored-by: Cursor * chore(autofix): apply prettier + eslint fixes via /autofix command --------- Co-authored-by: Gergo Magyar Co-authored-by: Claude Opus 5 (1M context) Co-authored-by: Cursor Co-authored-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> --- gitnexus/bench/analyze-phase-breakdown.md | 86 ++++++++++++++++----- gitnexus/src/storage/parsedfile-store.ts | 59 +++++++++++++- gitnexus/src/storage/v8-sidecar.ts | 45 ++++++++--- gitnexus/test/unit/parsedfile-store.test.ts | 58 ++++++++++++++ 4 files changed, 216 insertions(+), 32 deletions(-) diff --git a/gitnexus/bench/analyze-phase-breakdown.md b/gitnexus/bench/analyze-phase-breakdown.md index 6706c1905..e04d48628 100644 --- a/gitnexus/bench/analyze-phase-breakdown.md +++ b/gitnexus/bench/analyze-phase-breakdown.md @@ -169,31 +169,73 @@ Two things block it today, and neither is small: would then leave the live index with fresh File content and stale symbols, instead of untouched. -## scopeResolution is memory-traffic bound, not algorithmic +## scopeResolution — 14.7s, and it is two different problems -`--cpu-prof` of a one-file-edit run, top main-thread self time: +`PROF_SCOPE_RESOLUTION=1` already exists and reports the internal split. Marks +injected around the phase supply the rest. On a one-file edit: + +| step | ms | share | +| ---------------------------- | -------: | ------: | +| ParsedFile store rehydration | 4426 | 30% | +| **`emit`** | **7161** | **49%** | +| `resolve` | 966 | 7% | +| `finalize` | 552 | 4% | +| `extract` | 369 | 3% | +| per-language teardown, misc | ~750 | 5% | + +`extract` is small because the parse cache works: 2121/2121 pre-extracted hits +on the TypeScript pass. Nothing here re-parses. + +### Rehydration: three full store scans, one per language + +The corpus resolves three languages — python (54 files), typescript (2121), +javascript (59) — and `loadParsedFilesForPaths` walks **every** shard on each +pass. The store is 413 shards / 301 MB. A pass that wants a single file costs +335 ms, because the skip decision needs the envelope's path listing and that +listing is only trustworthy after the payload digest is checked. + +So the fixed cost is paid per language, and **it scales with language count**, +not with how many files that language has. Measured, min-of-5, three passes: ``` -4294ms (garbage collector) -1552ms v8.deserialize - 756ms crypto update - 728ms runScopeResolution - 620ms scope-resolution/pipeline/reconcile-ownership - 380ms v8-sidecar walk - 323ms internString +before python 411ms typescript 2872ms javascript 507ms = 3834ms +after python 415ms typescript 2805ms javascript 248ms = 3484ms ``` -then a tail of passes at 200–750ms (`emitReceiverBoundCalls`, -`emitCallableValueFlow`, `buildGraphNodeLookup`, `resolveReferenceSites`). +Memoizing each shard's authenticated listing for the run removes the repeat +scans (−350 ms here; roughly −250 ms per additional language elsewhere). The +first pass is unchanged by construction — it is what populates the memo. -No dominant hot function, nothing quadratic. The cost is rehydrating every -file's `ParsedFile` from the durable `.v8` shards into main-thread memory and -re-running every pass over them. +What is left is a floor. The digest is **not** the cost: SHA-256 over all +301 MB takes 134 ms (2.25 GB/s, hardware-accelerated), so swapping it for +CRC-32 would buy ~90 ms and cost an envelope-format bump. The remainder is +`v8.deserialize` (~1240 ms) plus the cross-shard string intern walk +(~1085 ms), and the intern walk is not optional — dropping it regresses +retained heap ~59%, which is #2649's constraint. -**So the win is not caching resolution output** — that stores more of exactly -what is already the memory problem, and #2649 (large-repo OOM) is the standing -constraint. The win is skipping rehydration and re-resolution for files whose -inputs provably did not change, as a streaming/bounded design. +### `emit` is the real target + +7.2s, 49% of the phase and ~21% of the whole edit loop. It is a fan of passes, +each walking every reference site in the repo: + +``` +1828ms emitCallableValueFlow 480ms emitFreeCallFallback +1590ms emitReceiverBoundCalls 426ms emitReferencesViaLookup + 681ms resolveDefGraphId 313ms emitUniqueNamePropertyAccesses + 529ms lookupCore 280ms emitReturnShapeMemberAccesses + 472ms tryEmitEdge 222ms emitPropertyDispatchCalls + 449ms getScope +``` + +No hot inner loop, nothing quadratic, no single pass worth more than 12% of the +phase. Micro-optimizing any of them is not the win. + +The win is not running them. A one-file edit re-emits all 163,094 edges to +write a 3,980-node subgraph. Every pass above runs over all 2121 TypeScript +files because the pipeline rebuilds the full in-memory graph every run and only +the DB _write_ is incremental. Making `emit` incremental means knowing which +files' edges can change when one file's registry contribution changes — which +is a design, not a patch, and #2649 rules out "just cache the emitted edges". --- @@ -232,8 +274,12 @@ and 0.8s of post-FTS event-loop drain between them. FTS is closed as an optimization target except for the overlap above, which is a scheduling change to `run-analyze.ts`'s open/close discipline rather than -anything about FTS. `scopeResolution` is the single largest step, memory-traffic -bound, and still untouched. +anything about FTS. + +`scopeResolution` is now broken down. Its rehydration half has a floor and one +repeat-scan win that is taken. Its other half — `emit`, 7.2s — is the largest +remaining target in the whole edit loop, and the only way at it is incremental +resolution. Every optimization in this document that looked compelling from the armchair died under measurement. Measure first, and check the banner. diff --git a/gitnexus/src/storage/parsedfile-store.ts b/gitnexus/src/storage/parsedfile-store.ts index dce10b03d..68ed2e06e 100644 --- a/gitnexus/src/storage/parsedfile-store.ts +++ b/gitnexus/src/storage/parsedfile-store.ts @@ -169,6 +169,7 @@ export const getParsedFileStoreDir = (storagePath: string): string => /** Remove any prior run's shards so a fresh parse starts clean. Idempotent. */ export const clearParsedFileStore = async (storagePath: string): Promise => { await fs.rm(getParsedFileStoreDir(storagePath), { recursive: true, force: true }); + forgetShardListings(); }; const isV8ShardName = (name: string): boolean => name.endsWith('.v8') && !name.includes('.v8.'); @@ -245,13 +246,52 @@ const listV8Shards = async (dir: string): Promise => { } }; +/** + * Per-run memo of each shard's authenticated path listing, so the SECOND and + * later passes over the store can decide "this shard holds nothing I want" + * without reopening it. + * + * Scope resolution calls `loadParsedFilesForPaths` once per language, and each + * call walks every shard in the store. The skip decision needs the envelope's + * path listing, and that listing is only trustworthy once the payload digest + * has been checked — so today a pass that wants 50 Python files still reads and + * SHA-256s all ~300MB of a TypeScript-dominated store to prove it can skip it. + * Measured on a 2234-file repo: a pass wanting a single file costs 335ms, and + * the three language passes together spend 4426ms in here. The cost scales with + * LANGUAGE COUNT, so a polyglot repo pays it worst. + * + * Keyed by size+mtime as well as name. Shard names are content-addressed + * (parse-chunk hash + worker id), so a name collision across different content + * should be impossible — but that invariant lives in the parse-cache keying, + * not here, and one `stat` per shard is a few ms against the hundreds this + * saves. The memo holds one store directory at a time: a different `dir` (a new + * repo in a long-lived MCP process, or a wiped store) drops the previous set + * rather than accumulating. + */ +let shardListingMemo: { dir: string; byIdentity: Map } | undefined; + +const shardIdentity = (file: string, size: number, mtimeMs: number): string => + `${file}\0${size}\0${mtimeMs}`; + +const memoFor = (dir: string): Map => { + if (shardListingMemo?.dir !== dir) shardListingMemo = { dir, byIdentity: new Map() }; + return shardListingMemo.byIdentity; +}; + +/** Drop the memo when the store it describes is removed. */ +const forgetShardListings = (): void => { + shardListingMemo = undefined; +}; + export const loadParsedFilesForPaths = async ( storagePath: string, wantPaths: ReadonlySet, ): Promise> => { const out = new Map(); if (wantPaths.size === 0) return out; - const shardPaths = await listV8Shards(getParsedFileStoreDir(storagePath)); + const storeDir = getParsedFileStoreDir(storagePath); + const shardPaths = await listV8Shards(storeDir); + const listings = memoFor(storeDir); const pool = new Map(); let droppedSites = 0; let filesWithDroppedSites = 0; @@ -274,11 +314,28 @@ export const loadParsedFilesForPaths = async ( } }; for (const shardFull of shardPaths) { + // Re-stat rather than trusting the name alone, then reuse this run's + // authenticated listing to skip without opening the file. A shard with no + // memoized listing — first pass, or an envelope whose listing did not + // validate — falls through to the full read, so this only ever removes + // work that a later pass had already proved unnecessary. + const st = await fs.stat(shardFull).catch(() => undefined); + const identity = st ? shardIdentity(shardFull, st.size, st.mtimeMs) : undefined; + const memoized = identity === undefined ? undefined : listings.get(identity); + if (memoized !== undefined && !memoized.some((p) => wantPaths.has(p))) { + bytesSinceGc += st?.size ?? 0; + await maybeYieldAndGc(bytesSinceGc >= parsedFileLoadGc.byteBudget); + continue; + } + const loaded = await tryLoadV8Cache(shardFull, pool, wantPaths); if (loaded === undefined) { await maybeYieldAndGc(false); continue; } + if (identity !== undefined && loaded.paths !== undefined) { + listings.set(identity, loaded.paths); + } if (loaded.kind === 'skip') { bytesSinceGc += loaded.bytes; await maybeYieldAndGc(bytesSinceGc >= parsedFileLoadGc.byteBudget); diff --git a/gitnexus/src/storage/v8-sidecar.ts b/gitnexus/src/storage/v8-sidecar.ts index e3b2f92c9..deee9364c 100644 --- a/gitnexus/src/storage/v8-sidecar.ts +++ b/gitnexus/src/storage/v8-sidecar.ts @@ -185,8 +185,22 @@ const decodePrefix = (buf: Buffer): EnvelopeMeta | undefined => { const runtimeCompatible = (meta: EnvelopeMeta): boolean => meta.recordedNodeMajor === nodeMajor() && meta.recordedV8 === process.versions.v8; -export type V8CacheHit = { kind: 'hit'; value: unknown; bytes: number }; -export type V8CacheSkip = { kind: 'skip'; bytes: number }; +/** + * `paths` is the envelope's digest-authenticated path listing, present only + * when it parsed and its entry count matched the recorded `pathCount` — i.e. + * exactly when the skip decision below is willing to trust it. A caller that + * loads the same immutable shard more than once per run can memoize it and + * make its own skip decision without reopening the file (see + * `parsedfile-store.ts`). Absent means "this envelope has no listing worth + * trusting", which is fail-closed: load, never skip. + */ +export type V8CacheHit = { + kind: 'hit'; + value: unknown; + bytes: number; + paths?: readonly string[]; +}; +export type V8CacheSkip = { kind: 'skip'; bytes: number; paths: readonly string[] }; export type V8CacheLoad = V8CacheHit | V8CacheSkip; export type V8CacheInspection = { paths: readonly string[] }; @@ -294,20 +308,29 @@ export const tryLoadV8Cache = async ( const payload = await readVerifiedPayload(fh, meta, pathRaw); if (!payload) return undefined; - if (wantPaths && wantPaths.size > 0 && meta.pathBytes > 0) { + // Parsed once and reused for both the skip decision and the returned + // listing, so a caller that memoizes it is trusting exactly the bytes this + // function was already willing to skip on. Anything that fails these checks + // stays `undefined` and falls through to the load. + let listedPaths: readonly string[] | undefined; + if (meta.pathBytes > 0) { const listed = parseCachePathListing(pathRaw); - if ( - listed !== null && - listed.length === meta.pathCount && - listed.length > 0 && - !listed.some((p) => wantPaths.has(p)) - ) { - return { kind: 'skip', bytes: st.size }; + if (listed !== null && listed.length === meta.pathCount && listed.length > 0) { + listedPaths = listed; } } + + if ( + wantPaths && + wantPaths.size > 0 && + listedPaths && + !listedPaths.some((p) => wantPaths.has(p)) + ) { + return { kind: 'skip', bytes: st.size, paths: listedPaths }; + } const value = v8.deserialize(payload); if (internPool) internGraphStrings(value, internPool); - return { kind: 'hit', value, bytes: st.size }; + return { kind: 'hit', value, bytes: st.size, paths: listedPaths }; } catch (err) { if (!isEnoent(err)) { logger.debug({ err, filePath }, 'v8 cache: load failed; treating as miss'); diff --git a/gitnexus/test/unit/parsedfile-store.test.ts b/gitnexus/test/unit/parsedfile-store.test.ts index 85a96e3da..ad012bd58 100644 --- a/gitnexus/test/unit/parsedfile-store.test.ts +++ b/gitnexus/test/unit/parsedfile-store.test.ts @@ -86,6 +86,64 @@ describe('parsedfile-store', () => { } }); + it('repeated loads with different wantPaths each see their own files (shard-listing memo)', async () => { + // Scope resolution calls loadParsedFilesForPaths once per LANGUAGE over the + // same store, so the second and later passes reuse the shard path listings + // the first pass authenticated instead of re-reading every shard. The + // failure mode that memo introduces is a FALSE SKIP: pass 2 concludes a + // shard holds nothing it wants, and those files silently never reach the + // graph — an exit-0 wrong answer, not a crash. Each pass below wants files + // the previous pass did not, so a listing carried over from the wrong shard + // shows up as a missing file here. + const dir = await mkdtemp(path.join(tmpdir(), 'pfstore-')); + try { + await persistParsedFileChunk(dir, 'chunk-0', [makeParsedFile('a.c'), makeParsedFile('b.c')]); + await persistParsedFileChunk(dir, 'chunk-1', [makeParsedFile('c.c')]); + await persistParsedFileChunk(dir, 'chunk-2', [makeParsedFile('d.c')]); + + const first = await loadParsedFilesForPaths(dir, new Set(['a.c'])); + expect([...first.keys()]).toEqual(['a.c']); + + // Key-set asserts alone still pass if the memo never skipped: the + // envelope listing would open the shard and skip deserialize. Spy + // open+deserialize on a later miss so a no-memo path fails. + const deserialize = vi.spyOn(v8, 'deserialize'); + const open = vi.spyOn(nodeFsPromises, 'open'); + try { + const second = await loadParsedFilesForPaths(dir, new Set(['c.c', 'd.c'])); + expect([...second.keys()].sort()).toEqual(['c.c', 'd.c']); + expect(open.mock.calls.map(([file]) => path.basename(String(file))).sort()).toEqual([ + 'chunk-1.v8', + 'chunk-2.v8', + ]); + expect(deserialize).toHaveBeenCalledTimes(2); + + open.mockClear(); + deserialize.mockClear(); + const third = await loadParsedFilesForPaths(dir, new Set(['b.c'])); + expect([...third.keys()]).toEqual(['b.c']); + expect(open.mock.calls.map(([file]) => path.basename(String(file)))).toEqual([ + 'chunk-0.v8', + ]); + expect(deserialize).toHaveBeenCalledTimes(1); + + const fourth = await loadParsedFilesForPaths(dir, new Set(['a.c', 'b.c', 'c.c', 'd.c'])); + expect([...fourth.keys()].sort()).toEqual(['a.c', 'b.c', 'c.c', 'd.c']); + } finally { + deserialize.mockRestore(); + open.mockRestore(); + } + + // A shard written AFTER the memo was populated is still found: the memo + // holds listings, not the shard roster, and the roster is re-read per call. + await persistParsedFileChunk(dir, 'chunk-3', [makeParsedFile('e.c')]); + const fifth = await loadParsedFilesForPaths(dir, new Set(['e.c', 'a.c'])); + expect([...fifth.keys()].sort()).toEqual(['a.c', 'e.c']); + } finally { + await rm(dir, { recursive: true, force: true }); + } + }); + it('writes no shard for an empty chunk', async () => { const dir = await mkdtemp(path.join(tmpdir(), 'pfstore-')); try { From 8ddab9aed5dabbeb5b1e4a400d083cab53a200c5 Mon Sep 17 00:00:00 2001 From: Ankit Verma Date: Tue, 8 Sep 2026 19:45:45 +0530 Subject: [PATCH 15/17] feat(api): honor branch on POST /api/analyze (#3199) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * feat(api): honor branch on POST /api/analyze The serve route accepted a `branch` field in the request body, returned 202 and reported the job `complete` — while indexing the remote's default branch. Express drops unknown body fields, so the caller got no error and no warning; the only way to notice was to inspect the checked-out clone. Both ends of the plumbing already existed: CloneOrPullOptions.branch is honored by cloneOrPull, and AnalyzeOptions.branch already drives resolveBranchPlacement. Only the HTTP layer was missing, so this wires `branch` from the route through cloneOrPull and LaunchOptions into the worker's AnalyzeOptions. StartMessage.options is already typed as AnalyzeOptions, so the IPC protocol is unchanged. Validation reuses validateBranchName — the same function backing the CLI's `--branch` — so both entry points accept exactly the same refs and a malformed value is rejected with 400 before it can reach git. Co-Authored-By: Claude Opus 5 * fix(api): make branch part of job identity and complete ref validation Addresses the review on #3199. 1. Job dedup ignored `branch`, so a request for branch B while branch A was in flight was answered with A's job and a 202. The caller would read that as "B is indexed" — the same silent wrong-branch outcome honoring `branch` was meant to remove. Branch is now part of the dedup identity; a different-branch request falls through to the single-slot guard and gets a truthful 409 instead. 2. validateBranchName implemented only a subset of git's ref rules, so `feature.lock`, `/feature`, `feature/`, `feature//next`, `@`, `@{` and dot-prefixed components passed validation and failed later in the git subprocess — a 202 plus a background failure rather than the advertised 400. The remaining `git check-ref-format` rules are now enforced at the same chokepoint, which fixes the CLI and `.gitnexusrc` paths too. No branch git can create is affected. 3. The LaunchOptions doc claimed an explicit branch always pins `branches//`. resolveBranchPlacement keeps the run on the flat slot when that slot has no owner, or when its owner is already this label. Comment and CHANGELOG corrected. Co-Authored-By: Claude Opus 5 * docs(server): describe cloneOrPull's branch path in its contract comment The header comment predated `options.branch` and still said an existing clone is only ever `git pull --ff-only`. The implementation has a second path: with a branch it fetches that ref and runs `checkout -B origin/`, so the requested branch — not the one already checked out — ends up in the working tree. The stale comment is actively misleading: a reviewer reading it concludes that requesting a branch on an existing clone silently analyzes the default branch, which is not what happens. Co-Authored-By: Claude Opus 5 * docs(server): scope the origin check to the existing-clone path My previous comment said remote.origin is verified "in both cases", which is wrong: assertRemoteMatchesRequestedUrl runs inside `if (exists)`, so a fresh clone has no origin to check. Restructured around whether targetDir exists, which is what actually selects the behavior. Co-Authored-By: Claude Opus 5 * test(server): cover branch job identity in JobManager's own suite The branch dedup tests were sitting in analyze-api.test.ts, but they exercise JobManager directly, so they belong beside the existing "returns existing job for same repoUrl when active" case in analyze-job.test.ts. Moved, and extended to cover the callers that omit branch entirely — the upload route, the embed manager and the existing tests — which compare undefined === undefined and are unaffected. Also pins that branch survives the whole clone -> analyze -> terminal update sequence, since it is now part of dedup identity and must not drift mid-flight. Co-Authored-By: Claude Opus 5 * fix(api): give a pinned branch its own clone and settle the slot it wrote Addresses the review on #3199. Per-branch clone directories (review option 2). One checkout per repo made `branch` one-shot: after any analyze the tree is dirty with generated AGENTS.md / CLAUDE.md / .claude/, so a pinned request 202'd and then died on cloneOrPull's porcelain refusal. Worse, a later request that OMITTED `branch` pulled whatever branch the last pin left checked out and indexed it as the default — silent wrong content, the same class as #3198 one request later. A pinned run now clones into `__`, so the two requests no longer share a tree. The pinned clone registers under its directory name, because both dirs share an origin and the inferred name would otherwise collide; that name re-derives through getCloneDir, so DELETE still finds it. Finalization gate. registerRepo always records the flat `.gitnexus`, but a pinned run whose label differs from the flat slot's owner writes `branches//`. The gate probed the flat path regardless, so it never settled, and the worker's normal exit 0 — sent ~500ms after `complete`, while the job is deliberately still non-terminal — was classified as a crash and a successful analysis was retried three times and failed. The gate now follows the placement the worker reports (isPrimaryBranch, added to the IPC allowlist under the rule that module already documents), and an exit after a terminal IPC counts as winding down, not dying. Reported by the maintainer and reproduced independently by @azizur100389. analyzeCloneOptions extracted so the token/branch combination is asserted. Inline, the branch-only case — a public URL with no token — was untested, and dropping it there would silently reindex the default branch while every other test stayed green. CHANGELOG: the previous entry claimed the newly-400'd payloads "would have failed at git", which is true of CLI --branch but wrong for HTTP, where they succeeded on the default branch. Documented as an explicit behavior change. Co-Authored-By: Claude Opus 5 * fix(server): bound the branch clone-dir name to one path component `validateBranchName` allows a 255-character ref and `branchSlug` appends a dash plus 8 hash characters, so `__` reached 267 — past the 255-byte component limit on ext4/APFS/NTFS. The clone would then fail to create its target directory, which the character-only regex could not catch. Only the readable half is trimmed. The hash is a digest of the full ref and is always kept, so two long branches sharing a prefix still resolve to different directories rather than silently sharing an index. `branchSlug` itself is untouched: the per-branch index slots already use those names on disk, and shortening them there would orphan existing indexes. Also corrects two comments: the forwarding test claimed a branch selector always pins `branches//` (it keeps the flat slot when that slot has no owner or already owns the label), and the web client's `branch` doc said omitting it means the remote default — true for a `url` request, but a `path` request is never cloned and indexes whatever that tree has checked out. Co-Authored-By: Claude Opus 5 * fix(server): bound the branch clone-dir name and unblock same-branch re-index Two defects found by testing this branch end to end, plus the review nits. 1. Path length. `validateBranchName` allows a 255-character ref and `branchSlug` appends a dash plus 8 hash characters, so `__` reached 267 — past the 255-byte component limit on ext4/APFS/NTFS, and the clone could not create its directory. Only the readable half is trimmed; the hash is a digest of the full ref and is always kept, so two long branches sharing a prefix still get separate directories. `branchSlug` itself is untouched — the per-branch index slots already use those names on disk and shortening them there would orphan existing indexes. 2. Same-branch re-index. Per-branch clone dirs stopped branches from contaminating each other, but a REPEAT pin still failed: analyze writes AGENTS.md / CLAUDE.md / .claude/ into the clone, so the second pinned run met its own dirt at the porcelain check and asked for `overwrite_local_changes`. When HEAD already matches the requested branch there is nothing to switch, so the run now takes the same `pull --ff-only` path an unpinned request takes — review option (1), alongside (2). The refusal is untouched where it matters: a real switch, or a detached HEAD, still goes through the checkout path and can still refuse. Also: repositions getCloneDir's JSDoc, which an inserted constant had orphaned; gates the pinned `registryName` on the same condition as the clone, so supplying both `url` and `path` no longer renames the operator's local repo; corrects a test comment that claimed a branch selector always pins `branches//`; and corrects the web client's `branch` doc, which said omitting it means the remote default — true for `url`, but a `path` request is never cloned. Co-Authored-By: Claude Opus 5 * docs: drop the CHANGELOG edits from this PR Requested in review. Checking the history, no feat/fix PR here touches gitnexus/CHANGELOG.md — the only recent commit on it is `chore: release v1.6.11`, and CONTRIBUTING says release notes are generated from the merged PR title via .github/release.yml. Hand-editing an [Unreleased] section from a feature branch was my mistake, not the project's convention. The behavior change it documented (branch: null / "" / non-string now 400 where they were previously dropped and the default branch indexed) is stated in the PR description instead, which is what feeds the release notes. Co-Authored-By: Claude Opus 5 * docs(server): align the clone/pull contract with the same-branch fast path Two comments I wrote went stale against my own later change. `cloneOrPull`'s header still said an existing clone with `options.branch` always fetches and runs `checkout -B`. Since the same-branch fast path landed that is only true when the branch actually differs; when it is already checked out the run takes `pull --ff-only` and no dirty-tree check applies. The header now splits on whether the branch differs, which is what the code branches on. The web client's `branch` doc said omitting it on a `url` request clones the remote default. That holds only when there is no clone yet — an existing unpinned clone is pulled on whatever branch it already has checked out. Comments only; no behavior change. Co-Authored-By: Claude Opus 5 * Address PR review feedback (#3199) Pin a same-branch re-index to `git pull --ff-only origin ` so the job cannot follow an unverified `branch..merge` while still skipping the dirty-tree refuse. Co-authored-by: Cursor * fix(server): close the #3199 holes on pinned analyze re-index Same-ref updates were a raw pull dest (force-fetch via +) or a shallow ff-merge that could not move, and a tag pin compared the tag-object SHA so re-index refused a dirty tree. Fetch the mapped remote-tracking ref, stay put on a peeled SHA match, restore only GitNexus overlays, skip the 60s settle on alreadyUpToDate, and keep branch validation in core so the HTTP route does not import the CLI. Co-authored-by: Cursor * chore(autofix): apply prettier + eslint fixes via /autofix command --------- Co-authored-by: Claude Opus 5 Co-authored-by: Gergő Magyar Co-authored-by: Gergo Magyar Co-authored-by: Cursor Co-authored-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> --- gitnexus-web/src/services/backend-client.ts | 7 + gitnexus/src/cli/analyze-config.ts | 52 +- gitnexus/src/core/git-ref.ts | 128 +++++ gitnexus/src/server/analyze-job.ts | 26 +- gitnexus/src/server/analyze-launch.ts | 98 +++- gitnexus/src/server/analyze-worker-ipc.ts | 14 +- gitnexus/src/server/api.ts | 57 ++- gitnexus/src/server/git-clone.ts | 277 +++++++++-- .../server-analyze-branch-validation.test.ts | 210 ++++++++ gitnexus/test/unit/analyze-config.test.ts | 41 ++ gitnexus/test/unit/analyze-job.test.ts | 59 +++ .../unit/analyze-launch-branch-settle.test.ts | 238 +++++++++ .../test/unit/analyze-launch-collapse.test.ts | 44 ++ gitnexus/test/unit/git-clone.test.ts | 466 ++++++++++++++++++ gitnexus/test/unit/git-ref.test.ts | 15 + 15 files changed, 1628 insertions(+), 104 deletions(-) create mode 100644 gitnexus/src/core/git-ref.ts create mode 100644 gitnexus/test/integration/server-analyze-branch-validation.test.ts create mode 100644 gitnexus/test/unit/analyze-launch-branch-settle.test.ts create mode 100644 gitnexus/test/unit/git-ref.test.ts diff --git a/gitnexus-web/src/services/backend-client.ts b/gitnexus-web/src/services/backend-client.ts index 8378ef494..50b9f2683 100644 --- a/gitnexus-web/src/services/backend-client.ts +++ b/gitnexus-web/src/services/backend-client.ts @@ -1010,6 +1010,13 @@ export const startAnalyze = async (request: { force?: boolean; embeddings?: boolean; token?: string; + /** + * Index-branch selector. Omitted: a `url` with no existing clone takes the + * remote's default branch, an existing clone updates whichever branch it + * already has checked out, and a `path` request is not cloned at all and + * indexes that working tree as it stands. + */ + branch?: string; }): Promise<{ jobId: string; status: string }> => { const response = await fetchWithTimeout( `${_backendUrl}/api/analyze`, diff --git a/gitnexus/src/cli/analyze-config.ts b/gitnexus/src/cli/analyze-config.ts index 3d040fac1..49896538e 100644 --- a/gitnexus/src/cli/analyze-config.ts +++ b/gitnexus/src/cli/analyze-config.ts @@ -31,6 +31,10 @@ import fs from 'node:fs'; import path from 'node:path'; import { readRepoControlFile } from '../config/repo-control-file.js'; +import { + InvalidBranchError, + validateBranchName as validateBranchNameCore, +} from '../core/git-ref.js'; import type { AnalyzeOptions } from './analyze-options.js'; export const GITNEXUS_RC_FILENAME = '.gitnexusrc'; @@ -38,9 +42,6 @@ export const GITNEXUS_RC_FILENAME = '.gitnexusrc'; /** Final fallback when no branch is configured or detectable. */ export const DEFAULT_BRANCH_FALLBACK = 'main'; -/** Git refs longer than this are almost certainly a mistake / injection attempt. */ -const BRANCH_MAX_LENGTH = 255; - /** * Thrown for any `.gitnexusrc` problem (missing-file is NOT an error — it * returns `undefined`). The message is user-facing and names the file so the @@ -157,45 +158,18 @@ const assertNoHiddenChars = (value: string, source: string): void => { /** * Validate a user-supplied branch name (from CLI or `.gitnexusrc`). Returns the - * trimmed name or throws {@link GitNexusRcError}. Conservative but accepts the - * shapes real branches use (`feature/foo-bar`, `release/1.2`, `develop`). + * trimmed name or throws {@link GitNexusRcError}. Rules live in + * `core/git-ref.ts`; this wrapper keeps the CLI / `.gitnexusrc` error type. */ export function validateBranchName(value: string, source: string): string { - const trimmed = value.trim(); - if (!trimmed) { - throw new GitNexusRcError(`${source}: branch name must not be empty.`); + try { + return validateBranchNameCore(value, source); + } catch (err) { + if (err instanceof InvalidBranchError) { + throw new GitNexusRcError(err.message); + } + throw err; } - if (trimmed.length > BRANCH_MAX_LENGTH) { - throw new GitNexusRcError(`${source}: branch name is too long (max ${BRANCH_MAX_LENGTH}).`); - } - assertNoHiddenChars(trimmed, source); - if (/\s/.test(trimmed)) { - throw new GitNexusRcError(`${source}: branch name must not contain whitespace.`); - } - // git ref-name rules (subset): reject characters git itself forbids in refs. - if (/[~^:?*[\\]/.test(trimmed)) { - throw new GitNexusRcError( - `${source}: branch name contains characters not allowed in a git ref (~ ^ : ? * [ \\).`, - ); - } - if (trimmed.startsWith('-')) { - throw new GitNexusRcError(`${source}: branch name must not start with "-".`); - } - if (trimmed.includes('..')) { - throw new GitNexusRcError(`${source}: branch name must not contain "..".`); - } - // Git permits a backtick in a ref, but the branch is embedded inside a - // Markdown inline-code span in the generated AGENTS.md/CLAUDE.md regression - // example, where a backtick would close the span early and let the rest of - // the template render as instruction text. Reject it at this single - // chokepoint so all three tiers (CLI flag, .gitnexusrc, auto-detect via - // sanitizeDetectedBranch) are covered (#1996 tri-review P1). - if (trimmed.includes('`')) { - throw new GitNexusRcError( - `${source}: branch name must not contain a backtick (it would break the generated Markdown).`, - ); - } - return trimmed; } /** diff --git a/gitnexus/src/core/git-ref.ts b/gitnexus/src/core/git-ref.ts new file mode 100644 index 000000000..e7d101eed --- /dev/null +++ b/gitnexus/src/core/git-ref.ts @@ -0,0 +1,128 @@ +/** + * Git ref-name validation used by both the CLI and the HTTP analyze route. + * + * Lives in `core/` so `server/api.ts` does not import `cli/analyze-config` + * (that import closed a cli → server → cli cycle: `cli/serve.ts` already + * imports `createServer`). The CLI keeps a thin wrapper that rethrows + * {@link InvalidBranchError} as `GitNexusRcError`. + */ + +/** Git refs longer than this are almost certainly a mistake / injection attempt. */ +const BRANCH_MAX_LENGTH = 255; + +/** + * Thrown when a user-supplied branch name fails {@link validateBranchName}. + * Callers at a product boundary map this to their own error type (CLI: + * `GitNexusRcError`; HTTP: 400). + */ +export class InvalidBranchError extends Error { + constructor(message: string) { + super(message); + this.name = 'InvalidBranchError'; + } +} + +/** + * Reject control characters and hidden / bidirectional Unicode in a string + * value. These have no legitimate place in a branch name and would otherwise + * let a committed config or HTTP body smuggle invisible controls into + * generated AGENTS.md / CLAUDE.md content. + */ +const isHiddenOrControl = (codePoint: number): boolean => + codePoint < 0x20 || + codePoint === 0x7f || + (codePoint >= 0x200b && codePoint <= 0x200f) || // zero-width + LRM/RLM + (codePoint >= 0x202a && codePoint <= 0x202e) || // bidi embeddings/overrides + (codePoint >= 0x2060 && codePoint <= 0x2064) || // word-joiner + invisible math + (codePoint >= 0x2066 && codePoint <= 0x206f) || // bidi isolates + deprecated + codePoint === 0xfeff; // BOM / zero-width no-break space + +const assertNoHiddenChars = (value: string, source: string): void => { + for (const ch of value) { + const cp = ch.codePointAt(0); + if (cp !== undefined && isHiddenOrControl(cp)) { + throw new InvalidBranchError( + `${source}: value contains control or hidden/bidirectional characters, which are not allowed.`, + ); + } + } +}; + +/** + * Validate a user-supplied branch name. Returns the trimmed name or throws + * {@link InvalidBranchError}. Conservative but accepts the shapes real + * branches use (`feature/foo-bar`, `release/1.2`, `develop`). + */ +export function validateBranchName(value: string, source: string): string { + const trimmed = value.trim(); + if (!trimmed) { + throw new InvalidBranchError(`${source}: branch name must not be empty.`); + } + if (trimmed.length > BRANCH_MAX_LENGTH) { + throw new InvalidBranchError(`${source}: branch name is too long (max ${BRANCH_MAX_LENGTH}).`); + } + assertNoHiddenChars(trimmed, source); + if (/\s/.test(trimmed)) { + throw new InvalidBranchError(`${source}: branch name must not contain whitespace.`); + } + // git ref-name rules (subset): reject characters git itself forbids in refs. + if (/[~^:?*[\\]/.test(trimmed)) { + throw new InvalidBranchError( + `${source}: branch name contains characters not allowed in a git ref (~ ^ : ? * [ \\).`, + ); + } + if (trimmed.startsWith('-')) { + throw new InvalidBranchError(`${source}: branch name must not start with "-".`); + } + // Force-refspec prefix (`git fetch origin +main` / `+refs/heads/main:…`). + // Rejected here so neither the CLI nor HTTP can pass a force-update refspec + // through as a "branch" (#3199 review, defense in depth). + if (trimmed.startsWith('+')) { + throw new InvalidBranchError(`${source}: branch name must not start with "+".`); + } + // The symbolic ref HEAD (case-sensitive). A repo can have a branch named + // `head`; git itself treats only `HEAD` as the current-commit alias. + if (trimmed === 'HEAD') { + throw new InvalidBranchError(`${source}: branch name must not be "HEAD".`); + } + if (trimmed.includes('..')) { + throw new InvalidBranchError(`${source}: branch name must not contain "..".`); + } + // The remaining `git check-ref-format` rules. Without these the validator + // accepted refs git itself refuses (`feature.lock`, `/feature`, `feature/`, + // `feature//next`, `@`, `.hidden`), so the failure surfaced later from the + // git subprocess instead of here. No real branch can violate them — git + // could not have created one — so nothing that works today starts failing. + if (trimmed.endsWith('.lock') || trimmed.split('/').some((part) => part.endsWith('.lock'))) { + throw new InvalidBranchError(`${source}: branch name must not end with ".lock".`); + } + if (trimmed.startsWith('/') || trimmed.endsWith('/')) { + throw new InvalidBranchError(`${source}: branch name must not start or end with "/".`); + } + if (trimmed.includes('//')) { + throw new InvalidBranchError(`${source}: branch name must not contain consecutive slashes.`); + } + if (trimmed === '@') { + throw new InvalidBranchError(`${source}: branch name must not be the single character "@".`); + } + if (trimmed.includes('@{')) { + throw new InvalidBranchError(`${source}: branch name must not contain "@{".`); + } + if (trimmed.endsWith('.') || trimmed.split('/').some((part) => part.startsWith('.'))) { + throw new InvalidBranchError( + `${source}: branch name must not end with "." or have a path component starting with ".".`, + ); + } + // Git permits a backtick in a ref, but the branch is embedded inside a + // Markdown inline-code span in the generated AGENTS.md/CLAUDE.md regression + // example, where a backtick would close the span early and let the rest of + // the template render as instruction text. Reject it at this single + // chokepoint so all three tiers (CLI flag, .gitnexusrc, auto-detect via + // sanitizeDetectedBranch) are covered (#1996 tri-review P1). + if (trimmed.includes('`')) { + throw new InvalidBranchError( + `${source}: branch name must not contain a backtick (it would break the generated Markdown).`, + ); + } + return trimmed; +} diff --git a/gitnexus/src/server/analyze-job.ts b/gitnexus/src/server/analyze-job.ts index a0a4b527d..fcd6af615 100644 --- a/gitnexus/src/server/analyze-job.ts +++ b/gitnexus/src/server/analyze-job.ts @@ -68,6 +68,13 @@ export interface AnalyzeJob { repoUrl?: string; repoPath?: string; repoName?: string; + /** + * Index-branch selector this job was started with, part of the job's dedup + * identity. A repo is not "the same repo" for reuse purposes when a different + * branch was asked for — reusing across branches would hand the caller a 202 + * for a job indexing something else. + */ + branch?: string; progress: AnalyzeJobProgress; error?: string; /** Set only when a terminal `failed` job still persisted usable work. */ @@ -94,15 +101,25 @@ export class JobManager { this.cleanupTimer = setInterval(() => this.cleanup(), CLEANUP_INTERVAL_MS); } - /** Create a new job, or return existing active job for the same repo. */ - createJob(params: { repoUrl?: string; repoPath?: string }): AnalyzeJob { - // Dedup: return existing active job for the same repo (by URL or path) + /** + * Create a new job, or return the existing active job for the same repo AND + * the same branch. + * + * Branch is part of the identity deliberately. Deduping on repo alone would + * return the in-flight job for branch A to a caller that asked for branch B, + * and that caller would read the resulting 202/`complete` as "B is indexed" + * — the same silent wrong-branch outcome that made `branch` worth honoring in + * the first place. Falling through instead lets the single-slot guard below + * reject the request outright, which is a truthful answer. + */ + createJob(params: { repoUrl?: string; repoPath?: string; branch?: string }): AnalyzeJob { + // Dedup: return existing active job for the same repo (by URL or path) and branch for (const job of this.jobs.values()) { if (!this.isTerminal(job.status)) { const isSameRepo = (params.repoUrl && job.repoUrl === params.repoUrl) || (params.repoPath && job.repoPath === params.repoPath); - if (isSameRepo) { + if (isSameRepo && job.branch === params.branch) { return job; } } @@ -120,6 +137,7 @@ export class JobManager { status: 'queued', repoUrl: params.repoUrl, repoPath: params.repoPath, + branch: params.branch, progress: { phase: 'queued', percent: 0, message: 'Waiting to start...' }, startedAt: Date.now(), retryCount: 0, diff --git a/gitnexus/src/server/analyze-launch.ts b/gitnexus/src/server/analyze-launch.ts index 06ddf94c4..901963a87 100644 --- a/gitnexus/src/server/analyze-launch.ts +++ b/gitnexus/src/server/analyze-launch.ts @@ -22,6 +22,7 @@ import { listRegisteredRepos, registryPathEquals, } from '../storage/repo-manager.js'; +import { BRANCHES_DIR, branchSlug } from '../storage/branch-index.js'; import { logger } from '../core/logger.js'; import { autoHeapCapMb } from '../core/ingestion/utils/effective-ram.js'; import { isTerminalJobStatus, type JobManager } from './analyze-job.js'; @@ -49,6 +50,20 @@ export interface LaunchOptions { springActuatorPath?: string; asyncApiSpecPath?: string; registryName?: string; + /** + * Index-branch selector, forwarded to `AnalyzeOptions.branch`. + * + * Setting it does not by itself mean a `branches//` sub-directory: + * `resolveBranchPlacement` (storage/branch-index.ts) keeps the run on the flat + * slot when that slot has no recorded owner, or when its owner already IS this + * label. Only a label that differs from the flat slot's owner gets its own + * sub-directory. + * + * The caller is responsible for having the branch checked out — + * `resolveWriteTarget` in core refuses a label that disagrees with the working + * tree, which is what keeps one branch's content out of another's slot (#2106). + */ + branch?: string; } const MAX_WORKER_RETRIES = 2; @@ -66,17 +81,6 @@ const MAX_WORKER_RETRIES = 2; const FINALIZE_SETTLE_TIMEOUT_MS = 60_000; const FINALIZE_SETTLE_POLL_MS = 200; -/** - * Resolve once the analyzed repo's index is settled at `storagePath`: the - * LadybugDB file and metadata both exist AND were (re)written by THIS job - * (mtime >= jobStartMs — bare existence is not enough, a re-analysis leaves - * the previous index in place while it works), and no transient WAL/shadow/ - * checkpoint sidecars remain (the worker's native close has finished). - * - * Never rejects. Timing out logs and proceeds (pre-gate behavior) rather - * than failing a job whose analysis genuinely succeeded — e.g. a no-op - * non-force analyze legitimately rewrites nothing. - */ /** * Look up the analyzed repo's registered storage path. The request's * user-provided path is used only as a comparison key; the filesystem probes @@ -91,7 +95,47 @@ const registeredStoragePath = async (targetPath: string): Promise return entry?.storagePath ?? null; }; -const waitForSettledIndex = async (targetPath: string, jobStartMs: number): Promise => { +/** + * Resolve the directory this run's index actually landed in. + * + * `registerRepo` always records the FLAT `.gitnexus` as `entry.storagePath`, + * but a pinned `--branch` run whose label differs from the flat slot's owner + * writes `lbug`/`gitnexus.json` under `branches//` instead. Probing the + * flat path for such a run watches files it never rewrote, so the gate below + * would spin to its timeout on a perfectly successful analysis (#3199 review). + * + * `isPrimaryBranch` is the worker's own report of `!placement.branch`, so this + * follows the placement core actually chose rather than recomputing it here + * (the flat slot's recorded owner can be adopted mid-run, which would make a + * recomputation race the thing it is trying to observe). + */ +const settleDirFor = ( + registryStoragePath: string, + branch: string | undefined, + isPrimaryBranch: boolean | undefined, +): string => + branch && isPrimaryBranch === false + ? path.join(registryStoragePath, BRANCHES_DIR, branchSlug(branch)) + : registryStoragePath; + +/** + * Resolve once the analyzed repo's index is settled at `storagePath`: the + * LadybugDB file and metadata both exist AND were (re)written by THIS job + * (mtime >= jobStartMs — bare existence is not enough, a re-analysis leaves + * the previous index in place while it works), and no transient WAL/shadow/ + * checkpoint sidecars remain (the worker's native close has finished). + * + * Never rejects. Timing out logs and proceeds (pre-gate behavior) rather + * than failing a job whose analysis genuinely succeeded. The `alreadyUpToDate` + * fast path never rewrites `lbug` (see `run-analyze.ts`) and skips this wait + * at the `complete` handler so it does not hold the analyze slot for 60s. + */ +const waitForSettledIndex = async ( + targetPath: string, + jobStartMs: number, + branch?: string, + isPrimaryBranch?: boolean, +): Promise => { const settled = (storagePath: string): boolean => { try { const lbugStat = statSync(path.join(storagePath, 'lbug')); @@ -112,7 +156,7 @@ const waitForSettledIndex = async (targetPath: string, jobStartMs: number): Prom // Re-resolved each round: the worker registers the repo as part of the // finalization this gate is waiting out. const storagePath = await registeredStoragePath(targetPath); - if (storagePath && settled(storagePath)) return; + if (storagePath && settled(settleDirFor(storagePath, branch, isPrimaryBranch))) return; if (Date.now() > deadline) { logger.warn( { targetPath }, @@ -173,6 +217,13 @@ export function createLaunchAnalysisWorker(deps: LaunchDeps) { // Capture stderr for crash diagnostics let stderrChunks = ''; + // A terminal IPC message (`complete`/`error`) means the worker finished + // and is now winding down — it calls process.exit(0) ~500ms later. The + // job is deliberately still non-terminal at that point because the + // finalization gate is running, so without this flag the exit handler + // below reads that clean exit as a crash and retries a SUCCESSFUL + // analysis, three times, before failing it (#3199 review). + let terminalIpcSeen = false; child.stderr?.on('data', (chunk: Buffer) => { stderrChunks += chunk.toString(); if (stderrChunks.length > 4096) stderrChunks = stderrChunks.slice(-4096); @@ -186,6 +237,8 @@ export function createLaunchAnalysisWorker(deps: LaunchDeps) { const current = jobManager.getJob(job.id); if (!current || isTerminalJobStatus(current.status)) return; + if (msg.type === 'complete' || msg.type === 'error') terminalIpcSeen = true; + if (msg.type === 'progress') { jobManager.updateJob(job.id, { status: 'analyzing', @@ -202,7 +255,16 @@ export function createLaunchAnalysisWorker(deps: LaunchDeps) { // below true in practice: the repo is actually queryable when the // client receives the SSE complete event, and an index this run knows // to be incomplete is never published at all. - waitForSettledIndex(targetPath, jobStartMs) + // + // alreadyUpToDate never opens LadybugDB and never rewrites `lbug` + // (run-analyze.ts early-return; CLI notes the same). The mtime gate + // would spin the full 60s and hold the single global analyze slot. + // ftsRepairedOnly DOES rewrite `lbug` (initLbug + createSearchFTSIndexes) + // so it still waits. + const settle = msg.result.alreadyUpToDate + ? Promise.resolve() + : waitForSettledIndex(targetPath, jobStartMs, opts.branch, msg.result.isPrimaryBranch); + settle .then(() => closeDbHandle()) .catch(() => {}) // best-effort: eviction failure must not fail the job .then(() => { @@ -296,6 +358,13 @@ export function createLaunchAnalysisWorker(deps: LaunchDeps) { const j = jobManager.getJob(job.id); if (!j || isTerminalJobStatus(j.status)) return; + // The worker already reported a terminal outcome; this exit is it + // winding down, not dying. The job is still non-terminal only because + // the finalization gate above has not resolved yet, and that gate owns + // the outcome — retrying here would fork a second worker over a + // finished, successful analysis. + if (terminalIpcSeen) return; + // Worker crashed — attempt retry if under the limit if (j.retryCount < MAX_WORKER_RETRIES) { j.retryCount++; @@ -339,6 +408,7 @@ export function createLaunchAnalysisWorker(deps: LaunchDeps) { ...(opts.springActuatorPath ? { springActuatorPath: opts.springActuatorPath } : {}), ...(opts.asyncApiSpecPath ? { asyncApiSpecPath: opts.asyncApiSpecPath } : {}), ...(opts.registryName ? { registryName: opts.registryName } : {}), + ...(opts.branch ? { branch: opts.branch } : {}), }, }); }; diff --git a/gitnexus/src/server/analyze-worker-ipc.ts b/gitnexus/src/server/analyze-worker-ipc.ts index 3bb0a4edd..15b8112d6 100644 --- a/gitnexus/src/server/analyze-worker-ipc.ts +++ b/gitnexus/src/server/analyze-worker-ipc.ts @@ -44,10 +44,11 @@ import type { AnalyzeResult } from '../core/run-analyze.js'; * ones (e.g. `isPrimaryBranch?`), so an optional non-serializable field could be * advertised by the type yet silently dropped by the runtime allowlist. * - * `isPrimaryBranch` is intentionally excluded: the parent (`api.ts`) reads only - * `repoName`, and nothing consumes `isPrimaryBranch` across this fork (its CLI - * consumer calls `runFullAnalysis` in-process). Add a field here only when a - * server-side IPC consumer actually needs it — and only if it is JSON-safe. + * `isPrimaryBranch` IS on the wire, under exactly the rule this comment used to + * cite for excluding it: a server-side consumer now needs it. `analyze-launch.ts` + * settles the index the run actually wrote, and only the worker knows whether + * core chose the flat slot or a `branches//` sub-slot. It is a boolean, so + * it is JSON-safe by construction. */ export type AnalyzeResultIpc = Pick< AnalyzeResult, @@ -58,6 +59,7 @@ export type AnalyzeResultIpc = Pick< | 'ftsRepairedOnly' | 'ftsSkipped' | 'graphWriteCollapsed' + | 'isPrimaryBranch' >; /** @@ -78,5 +80,9 @@ export function projectAnalyzeResultForIpc(result: AnalyzeResult): AnalyzeResult // outcome the CLI does; without it the worker reports a clean `complete` // for a run whose edges are mostly missing. graphWriteCollapsed: result.graphWriteCollapsed, + // Tells the parent which slot this run wrote — the flat `.gitnexus` or a + // `branches//` sub-slot — so its finalization gate watches the files + // this job actually rewrote (#3199 review). + isPrimaryBranch: result.isPrimaryBranch, }; } diff --git a/gitnexus/src/server/api.ts b/gitnexus/src/server/api.ts index 0843d0b63..5a731ae0b 100644 --- a/gitnexus/src/server/api.ts +++ b/gitnexus/src/server/api.ts @@ -57,6 +57,7 @@ import { assertString, BadRequestError, createRouteLimiter } from './validation. import { parseGrepQuery, GREP_TIME_BUDGET_MS } from './grep-params.js'; import { runGrepScanInWorker } from './grep-scan.js'; import { + analyzeCloneOptions, extractWebRepoName, getCloneDir, cloneOrPull, @@ -64,6 +65,10 @@ import { GITHUB_TOKEN_HOSTS, } from './git-clone.js'; import { createAnalyzeUploadHandler } from './analyze-upload.js'; +// Shared with the CLI's `--branch` (via the analyze-config wrapper) so both +// entry points accept the same refs. Imported from core — not cli/ — so +// createServer does not close a cycle with cli/serve.ts. +import { InvalidBranchError, validateBranchName } from '../core/git-ref.js'; import { assertServeAuthForPublicOrigin, createPublicOriginMatcher, @@ -1519,6 +1524,7 @@ export const createServer = async (port: number, host: string = '127.0.0.1') => springActuatorPath, asyncApiSpecPath, token: repoToken, + branch: repoBranch, } = req.body; // Input type validation @@ -1550,6 +1556,27 @@ export const createServer = async (port: number, host: string = '127.0.0.1') => return; } + // Branch: optional index-branch selector, validated with the same rules + // as the CLI's `--branch` so both entry points accept the same refs. + // Rejecting here (rather than letting the clone fail) keeps a malformed + // ref from ever reaching `git`. + if (repoBranch !== undefined && typeof repoBranch !== 'string') { + res.status(400).json({ error: '"branch" must be a string' }); + return; + } + let analyzeBranch: string | undefined; + if (repoBranch !== undefined) { + try { + analyzeBranch = validateBranchName(repoBranch, '"branch"'); + } catch (err) { + if (err instanceof InvalidBranchError) { + res.status(400).json({ error: err.message }); + return; + } + throw err; + } + } + // Token: optional, restricted charset to prevent header smuggling // (CRLF), bound length, and bound to github.com (see validateAnalyzeToken). const tokenError = validateAnalyzeToken(repoToken, repoUrl); @@ -1575,7 +1602,11 @@ export const createServer = async (port: number, host: string = '127.0.0.1') => return; } - const job = jobManager.createJob({ repoUrl, repoPath: repoLocalPath }); + const job = jobManager.createJob({ + repoUrl, + repoPath: repoLocalPath, + branch: analyzeBranch, + }); // If job was already running (dedup), just return its id. The token is // not part of the dedup identity and is never stored on the job, so a @@ -1603,11 +1634,15 @@ export const createServer = async (port: number, host: string = '127.0.0.1') => // Clone if URL provided if (repoUrl && !repoLocalPath) { const repoName = extractWebRepoName(repoUrl); - targetPath = getCloneDir(repoName); + // Branch-pinned runs get their own clone dir, so they never share + // a working tree with the unpinned one (see getCloneDir). + targetPath = getCloneDir(repoName, analyzeBranch); jobManager.updateJob(job.id, { status: 'cloning', - repoName, + // url+branch: same value as registryName (dir basename), not + // the extractWebRepoName stem used only as getCloneDir's first arg. + repoName: analyzeBranch ? path.basename(targetPath) : repoName, progress: { phase: 'cloning', percent: 0, message: `Cloning ${repoUrl}...` }, }); @@ -1619,7 +1654,7 @@ export const createServer = async (port: number, host: string = '127.0.0.1') => progress: { phase: progress.phase, percent: 5, message: progress.message }, }); }, - repoToken ? { token: repoToken } : undefined, + analyzeCloneOptions(repoToken, analyzeBranch), ); } @@ -1633,6 +1668,20 @@ export const createServer = async (port: number, host: string = '127.0.0.1') => dropEmbeddings, springActuatorPath, asyncApiSpecPath, + branch: analyzeBranch, + // Both clone dirs share an `origin`, so the name `registerRepo` + // infers from the remote would be identical and the second one + // would fail with RegistryNameCollisionError. Register the pinned + // clone under its directory name instead: unique per branch, and + // it re-derives through getCloneDir for DELETE /api/repo. + // + // Gated on the SAME condition as the clone above: when a caller + // supplies both `url` and `path` nothing is cloned, and renaming + // the operator's own local repo after its directory would be a + // surprise unrelated to branch pinning. + ...(analyzeBranch && repoUrl && !repoLocalPath + ? { registryName: path.basename(targetPath) } + : {}), }); } catch (err: any) { if (targetPath) releaseRepoLock(getStoragePath(targetPath)); diff --git a/gitnexus/src/server/git-clone.ts b/gitnexus/src/server/git-clone.ts index 6fb1e25c9..f5a8d9e22 100644 --- a/gitnexus/src/server/git-clone.ts +++ b/gitnexus/src/server/git-clone.ts @@ -11,6 +11,7 @@ import fs from 'fs/promises'; import os from 'node:os'; import { logger } from '../core/logger.js'; import { getGlobalDir } from '../storage/repo-manager.js'; +import { branchSlug } from '../storage/branch-index.js'; import { sanitizeRepoName, stripUrlCredentials } from '../storage/git.js'; import { validateGitUrl } from '../core/net/url-guard.js'; import { @@ -84,14 +85,65 @@ export function extractWebRepoName(url: string): string { return safeName; } -/** Get the clone target directory for a repo name. */ -export function getCloneDir(repoName: string): string { +/** + * Longest single path component the supported filesystems accept (ext4, APFS, + * NTFS all cap at 255). Compared against `.length`, which equals the byte + * count here because every name this guards is ASCII by construction + * (REPO_NAME_PATTERN and sanitizeRepoName both restrict to `[a-zA-Z0-9._-]`). + */ +const MAX_PATH_COMPONENT_BYTES = 255; + +/** + * `branchSlug` for a clone-directory name, trimmed to fit one path component. + * + * `validateBranchName` allows a ref up to 255 characters and `branchSlug` + * appends `-` plus 8 hash characters, so `__` can reach 267 — past + * the filesystem limit, and the clone would then fail to create its target + * directory (#3199 review). + * + * Only the READABLE half is trimmed; the 8-character hash is always kept, and + * it is a digest of the full ref, so two long branches that share a prefix + * still get different directories. The slug is not trimmed inside + * `branchSlug` itself because the per-branch *index* slots already use those + * names on disk — shortening them there would orphan existing indexes. + */ +const boundedBranchSegment = (repoName: string, branch: string): string => { + const slug = branchSlug(branch); + if (`${repoName}__${slug}`.length <= MAX_PATH_COMPONENT_BYTES) return slug; + + const hash = slug.slice(slug.lastIndexOf('-')); // "-" + 8 hex + const budget = MAX_PATH_COMPONENT_BYTES - repoName.length - '__'.length - hash.length; + // A repo name long enough to leave no budget falls through to the caller's + // length check, which rejects it rather than building an unusable path. + return `${slug.slice(0, Math.max(0, budget))}${hash}`; +}; + +/** Get the clone target directory for a repo name, optionally pinned to a branch. */ +export function getCloneDir(repoName: string, branch?: string): string { // Re-validate at the boundary even though extractRepoName already checked — // callers may pass a repoName from another source (test fixtures, scripts). if (!repoName || repoName === '.' || repoName === '..' || !REPO_NAME_PATTERN.test(repoName)) { throw new Error('Invalid repository name'); } - return path.join(CLONE_ROOT, repoName); + // A branch-pinned analyze gets its OWN working tree. + // + // Sharing one checkout per repo made `branch` unusable in practice: the tree + // is dirty after any analyze (generated AGENTS.md / CLAUDE.md / .claude/), so + // a pinned request hit `cloneOrPull`'s porcelain refusal; and a later request + // that OMITTED `branch` would pull whatever branch the last pin left checked + // out and index it as the default (#3199 review). Separate directories remove + // both, because the two requests no longer share a tree. + // + // `branchSlug` is the same helper the per-branch index slots use, so the two + // layouts agree on how a ref becomes a path segment. It emits only + // `[a-zA-Z0-9._-]`, so the composed name still satisfies REPO_NAME_PATTERN and + // round-trips through this function — which is how DELETE /api/repo re-derives + // the directory from the registry name. + const dirName = branch ? `${repoName}__${boundedBranchSegment(repoName, branch)}` : repoName; + if (!REPO_NAME_PATTERN.test(dirName) || dirName.length > MAX_PATH_COMPONENT_BYTES) { + throw new Error('Invalid repository name'); + } + return path.join(CLONE_ROOT, dirName); } export interface CloneProgress { @@ -99,6 +151,29 @@ export interface CloneProgress { message: string; } +/** + * Build the `cloneOrPull` options for an `/api/analyze` request. + * + * Extracted from the route so the token/branch combination is unit-testable. + * Inline, the branch-only case was the one nothing asserted: every existing + * test still passed if `branch` were dropped whenever no token was supplied — + * i.e. silently cloning the default branch for every public URL, which is the + * exact behavior #3198 is about (#3199 review). + * + * Returns `undefined` rather than `{}` when neither is set, because that is + * what `cloneOrPull` treats as "no options" at its own call sites. + */ +export function analyzeCloneOptions( + token?: string, + branch?: string, +): Pick | undefined { + if (!token && !branch) return undefined; + return { + ...(token ? { token } : {}), + ...(branch ? { branch } : {}), + }; +} + export interface CloneOrPullOptions { token?: string; allowedCloneRoot?: string; @@ -294,11 +369,127 @@ export async function assertRemoteMatchesRequestedUrl( } } +/** + * Fetch refspec that updates `origin/` from `refs/heads/`. + * + * The leading `+` is git's dest-update prefix (`+refs/heads/*:refs/remotes/origin/*` + * is what `git clone` writes into `.git/config`). Without it, `fetch --depth 1` + * refuses to move `origin/` when the shallow history cannot prove a + * fast-forward — so a same-branch re-index stays stuck on the old tip. + * + * The user string is interpolated inside `refs/heads/…`, never as a raw pull + * dest. A branch named `+develop` becomes `+refs/heads/+develop:…`, not a + * force-update of `develop`. + */ +function branchFetchRefspec(branch: string): string { + return `+refs/heads/${branch}:refs/remotes/origin/${branch}`; +} + +/** Overlays `analyze` writes into a clone; they must not block a same-ref update. */ +const GITNEXUS_GENERATED_OVERLAYS = ['./AGENTS.md', './CLAUDE.md', './.claude'] as const; + +/** + * Restore only GitNexus-generated overlays so a same-ref update is not + * blocked by analyze dirt. Path-limited and root-anchored (`./`): tracked + * files are checked out from HEAD; untracked overlays (including gitignored + * ones — `AGENTS.md` / `.claude/` are commonly ignored) are `git clean -fdx`'d. + * A slash-free `AGENTS.md` would also hit `docs/AGENTS.md`. Never a + * whole-clone `git clean --force -d`. + */ +async function restoreGitNexusGeneratedOverlays( + runGitImpl: typeof runGit, + cwd: string, + gitOpts: RunGitOptions, +): Promise { + for (const overlay of GITNEXUS_GENERATED_OVERLAYS) { + const listed = (await runGitImpl(['ls-files', '--', overlay], cwd, gitOpts)).trim(); + if (!listed) continue; + await runGitImpl(['checkout', 'HEAD', '--', overlay], cwd, gitOpts); + } + // Path-limited: untracked analyze output still blocks checkout when the + // incoming tree has the same path, and otherwise leaves a dirty tree to + // index. `-x` is required because these overlays are often gitignored. + // Never a whole-clone `git clean --force -d`. + await runGitImpl(['clean', '-fdx', '--', ...GITNEXUS_GENERATED_OVERLAYS], cwd, gitOpts); +} + +/** + * True when the working tree is already at the requested pin: either HEAD is + * that named branch, or HEAD is detached at the same SHA as `branch` / + * `origin/`. A missing ref falls through to the switch path. + */ +async function matchRequestedRef( + runGitImpl: typeof runGit, + cwd: string, + branch: string, + gitOpts: RunGitOptions, +): Promise<'branch' | 'sha' | undefined> { + const abbrev = (await runGitImpl(['rev-parse', '--abbrev-ref', 'HEAD'], cwd, gitOpts)).trim(); + if (abbrev === branch) return 'branch'; + // Detached HEAD reports `HEAD`; compare SHAs so a tag/SHA pin is not a switch. + if (abbrev !== 'HEAD') return undefined; + + let headSha: string; + try { + headSha = (await runGitImpl(['rev-parse', 'HEAD'], cwd, gitOpts)).trim(); + } catch { + return undefined; + } + + for (const candidate of [branch, `origin/${branch}`] as const) { + try { + // Peel annotated tags (`v1.0` is a tag object; HEAD is the commit). + const requestedSha = ( + await runGitImpl(['rev-parse', `${candidate}^{commit}`], cwd, gitOpts) + ).trim(); + if (requestedSha && requestedSha === headSha) return 'sha'; + } catch { + // Ref missing — try origin/, then the switch path. + } + } + return undefined; +} + +async function fetchAndCheckoutRequestedBranch( + runGitImpl: typeof runGit, + cwd: string, + branch: string, + gitOpts: RunGitOptions, +): Promise { + // Analyze clones are `--depth 1`. `merge --ff-only` cannot walk O→N when + // the remote moved 2+ commits (the merge-base is not in the shallow + // history). `checkout -B` points the local branch at the fetched tip — + // same as the switch path, no ancestry walk. No `--force`: leftover + // non-overlay dirt still refuses. + await runGitImpl(['fetch', '--depth', '1', 'origin', branchFetchRefspec(branch)], cwd, gitOpts); + await runGitImpl(['checkout', '-B', branch, `origin/${branch}`], cwd, gitOpts); +} + /** * Clone or pull a git repository. - * If targetDir doesn't exist: git clone --depth 1 - * If targetDir exists with .git: git pull --ff-only (after verifying the - * existing clone's remote.origin matches the requested URL). + * + * If targetDir doesn't exist: git clone --depth 1, adding `--branch ` + * when one is requested. + * + * If targetDir exists with .git, its remote.origin is verified against the + * requested URL first, and then the branch decides the update: + * - no `options.branch`: git pull --ff-only, which updates the current + * branch in place via its configured upstream. Nothing moves, so no + * dirty-tree check applies. + * - a `options.branch` that is ALREADY the current named branch: restore + * GitNexus overlays, then fetch via + * `+refs/heads/:refs/remotes/origin/` and + * `checkout -B origin/` (shallow clones cannot + * `merge --ff-only` across a 2+ commit move). Never a raw + * `origin ` pull refspec. No porcelain refuse. + * - a detached HEAD whose SHA already matches the requested ref (tag / + * SHA pin): restore overlays only. Do not fetch/merge — a same-named + * branch could otherwise fast-forward the pin past the tag. + * - a `options.branch` that DIFFERS from the current pin: fetch that ref, + * then `checkout -B origin/` — so the requested branch, + * not the one already checked out, is what ends up in the working tree. + * This is the switching case, and it refuses a dirty tree unless + * `overwriteLocalChanges` is set. * * Security: * - targetDir must resolve inside CLONE_ROOT (~/.gitnexus/repos/). The @@ -382,13 +573,36 @@ export async function cloneOrPull( await assertRemoteMatchesRequestedUrl(safeTarget, url, options?.timeoutMs); onProgress?.({ phase: 'pulling', message: 'Pulling latest changes...' }); const runGitImpl = options?.runGitForTest ?? runGit; - if (options?.branch) { + const gitOpts = { + token: options?.token, + url, + timeoutMs: options?.timeoutMs, + }; + // Already at the requested pin? Then there is no switch to make, so do + // not take the checkout path below — that would run the porcelain check + // against a tree ANALYZE ITSELF dirtied (it writes AGENTS.md / CLAUDE.md / + // .claude/ into the clone), which made a pinned RE-index impossible: the + // first pin succeeded and every later one failed asking for + // `overwrite_local_changes`, a flag this route deliberately does not pass + // because it would `git clean --force -d` the directory (#3199 review). + // + // "Already there" is a named-branch match OR a detached HEAD whose SHA + // equals `branch` / `origin/` (tag / SHA pin). A missing ref + // falls through to the switch path, which still refuses a dirty tree. + // + // Same-named-branch update uses the heads/ → remotes/ fetch refspec plus + // `checkout -B origin/`, never `pull origin ` + // (a leading `+` would otherwise be a force-fetch). Only + // `remote.origin.url` is verified above; `branch..remote` / + // `.merge` are not, so an implicit-upstream pull can update a different + // ref while the job still carries this branch (#3199 review). + const requestedRefMatch = options?.branch + ? await matchRequestedRef(runGitImpl, safeTarget, options.branch, gitOpts) + : undefined; + + if (options?.branch && !requestedRefMatch) { if (!options.overwriteLocalChanges) { - const status = await runGitImpl(['status', '--porcelain'], safeTarget, { - token: options?.token, - url, - timeoutMs: options?.timeoutMs, - }); + const status = await runGitImpl(['status', '--porcelain'], safeTarget, gitOpts); if (status.trim()) { throw new Error( `Refusing to update ${safeTarget}: local changes detected. Set overwrite_local_changes: true to overwrite them.`, @@ -396,19 +610,9 @@ export async function cloneOrPull( } } await runGitImpl( - [ - 'fetch', - '--depth', - '1', - 'origin', - `refs/heads/${options.branch}:refs/remotes/origin/${options.branch}`, - ], + ['fetch', '--depth', '1', 'origin', branchFetchRefspec(options.branch)], safeTarget, - { - token: options?.token, - url, - timeoutMs: options?.timeoutMs, - }, + gitOpts, ); await runGitImpl( [ @@ -419,11 +623,7 @@ export async function cloneOrPull( `origin/${options.branch}`, ], safeTarget, - { - token: options?.token, - url, - timeoutMs: options?.timeoutMs, - }, + gitOpts, ); if (options.overwriteLocalChanges) { // `checkout --force` rewrites tracked files only, so untracked sources @@ -432,18 +632,17 @@ export async function cloneOrPull( // ignored paths must survive, and `-e /.gitnexus` is belt-and-braces // because `.git/info/exclude` is skipped on a read-only storage mount // and a freshly cloned repo may not have been analyzed yet at all. - await runGitImpl(['clean', '--force', '-d', '-e', '/.gitnexus'], safeTarget, { - token: options?.token, - url, - timeoutMs: options?.timeoutMs, - }); + await runGitImpl(['clean', '--force', '-d', '-e', '/.gitnexus'], safeTarget, gitOpts); } + } else if (options?.branch && requestedRefMatch === 'branch') { + await restoreGitNexusGeneratedOverlays(runGitImpl, safeTarget, gitOpts); + await fetchAndCheckoutRequestedBranch(runGitImpl, safeTarget, options.branch, gitOpts); + } else if (options?.branch && requestedRefMatch === 'sha') { + // Tag / SHA pin: already at the requested commit. Fetching + // `refs/heads/` would follow a same-named branch past the pin. + await restoreGitNexusGeneratedOverlays(runGitImpl, safeTarget, gitOpts); } else { - await runGitImpl(['pull', '--ff-only'], safeTarget, { - token: options?.token, - url, - timeoutMs: options?.timeoutMs, - }); + await runGitImpl(['pull', '--ff-only'], safeTarget, gitOpts); } } else { if (targetExists && (await fs.readdir(safeTarget)).length > 0) { diff --git a/gitnexus/test/integration/server-analyze-branch-validation.test.ts b/gitnexus/test/integration/server-analyze-branch-validation.test.ts new file mode 100644 index 000000000..3e71ebf02 --- /dev/null +++ b/gitnexus/test/integration/server-analyze-branch-validation.test.ts @@ -0,0 +1,210 @@ +/** + * End-to-end HTTP test of POST /api/analyze `branch` validation. + * + * `branch` is handed to `git` as a ref (`clone --branch`, `checkout -B`), so the + * route validates it with the same `validateBranchName` the CLI's `--branch` + * uses. This proves the REAL production route wires that in — express.json body + * parsing, the requireTrustedOrigin guard, the route handler invoking the + * validator, and the 400 status/error shape on the wire. + * + * Only rejection paths are asserted: each returns 400 BEFORE any clone, so the + * test is hermetic (no network, no background git, no real repo). The accepted + * path would spawn a background clone; the parent→worker half of it is covered + * in test/unit/analyze-launch-collapse.test.ts, which asserts `branch` reaches + * the worker's `AnalyzeOptions`. + * + * This lives in its own file rather than alongside the token cases because + * /api/analyze is rate-limited to 10 requests/minute per IP — one spawned server + * per concern keeps each suite clear of that ceiling. + * + * Mirrors the spawn+health-poll harness in server-analyze-token-validation.test.ts; + * the integration suite always builds dist first (pretest:integration). + */ +import { spawn, type ChildProcessWithoutNullStreams } from 'node:child_process'; +import fs from 'node:fs'; +import http from 'node:http'; +import os from 'node:os'; +import path from 'node:path'; +import { fileURLToPath } from 'node:url'; +import { afterAll, beforeAll, describe, expect, it } from 'vitest'; + +const __dirname = path.dirname(fileURLToPath(import.meta.url)); +const REPO_ROOT = path.resolve(__dirname, '..', '..'); +const DIST_CLI = path.join(REPO_ROOT, 'dist', 'cli', 'index.js'); +const STARTUP_BUDGET_MS = process.env.CI ? 30_000 : 15_000; + +const allocateFreePort = (): Promise => + new Promise((resolve, reject) => { + const probe = http.createServer(); + probe.once('error', reject); + probe.listen(0, '127.0.0.1', () => { + const addr = probe.address(); + if (typeof addr !== 'object' || !addr) { + probe.close(); + reject(new Error('could not allocate ephemeral port')); + return; + } + const port = addr.port; + probe.close((err) => (err ? reject(err) : resolve(port))); + }); + }); + +const httpJson = ( + port: number, + method: string, + reqPath: string, + body?: unknown, +): Promise<{ status: number; body: string }> => + new Promise((resolve, reject) => { + const payload = body === undefined ? undefined : JSON.stringify(body); + const req = http.request( + { + host: '127.0.0.1', + port, + path: reqPath, + method, + headers: payload + ? { 'content-type': 'application/json', 'content-length': Buffer.byteLength(payload) } + : {}, + }, + (res) => { + const chunks: Buffer[] = []; + res.on('data', (c) => chunks.push(c)); + res.on('end', () => + resolve({ status: res.statusCode ?? 0, body: Buffer.concat(chunks).toString('utf8') }), + ); + }, + ); + req.on('error', reject); + req.setTimeout(5_000, () => { + req.destroy(); + reject(new Error(`${method} ${reqPath} timed out`)); + }); + if (payload) req.write(payload); + req.end(); + }); + +const postAnalyze = (port: number, body: unknown) => httpJson(port, 'POST', '/api/analyze', body); + +// Spawned `serve` on Windows can report ready before the socket is reachable +// from the parent (see server-http-startup.test.ts); validateBranchName's own +// unit coverage runs on every platform. +const describeBlock = process.platform === 'win32' ? describe.skip : describe; + +describeBlock('POST /api/analyze branch validation (real server)', () => { + let proc: ChildProcessWithoutNullStreams | undefined; + let homeDir: string | undefined; + let port = 0; + + beforeAll(async () => { + if (!fs.existsSync(DIST_CLI)) { + throw new Error(`Missing ${DIST_CLI} — run npm run build before integration tests`); + } + + port = await allocateFreePort(); + homeDir = fs.mkdtempSync(path.join(os.tmpdir(), 'gitnexus-analyze-branch-')); + + proc = spawn( + process.execPath, + [DIST_CLI, 'serve', '--port', String(port), '--host', '127.0.0.1'], + { + cwd: REPO_ROOT, + env: { ...process.env, GITNEXUS_HOME: homeDir, NODE_OPTIONS: '' }, + stdio: ['ignore', 'pipe', 'pipe'], + }, + ); + + let stderr = ''; + proc.stderr.on('data', (buf) => { + stderr += buf.toString(); + }); + + const startedAt = Date.now(); + while (Date.now() - startedAt < STARTUP_BUDGET_MS) { + if (proc.exitCode !== null) { + throw new Error(`serve exited ${proc.exitCode} before ready.\nstderr:\n${stderr}`); + } + try { + const { status } = await httpJson(port, 'GET', '/api/health'); + if (status === 200) return; + } catch { + // Server still starting — retry until budget expires. + } + await new Promise((r) => setTimeout(r, 100)); + } + throw new Error( + `serve did not become ready within ${STARTUP_BUDGET_MS}ms.\nstderr:\n${stderr}`, + ); + }, 60_000); + + afterAll(async () => { + if (proc && !proc.killed) { + proc.kill('SIGTERM'); + await new Promise((resolve) => { + const timer = setTimeout(() => { + proc?.kill('SIGKILL'); + resolve(); + }, 3_000); + proc?.on('exit', () => { + clearTimeout(timer); + resolve(); + }); + }); + } + proc = undefined; + if (homeDir) { + fs.rmSync(homeDir, { recursive: true, force: true }); + homeDir = undefined; + } + }); + + it('rejects a non-string branch', async () => { + const { status, body } = await postAnalyze(port, { + url: 'https://github.com/owner/repo', + branch: 42, + }); + expect(status).toBe(400); + expect(JSON.parse(body).error).toContain('"branch" must be a string'); + }); + + it('rejects a branch containing whitespace', async () => { + const { status, body } = await postAnalyze(port, { + url: 'https://github.com/owner/repo', + branch: 'feature branch', + }); + expect(status).toBe(400); + expect(JSON.parse(body).error).toContain('whitespace'); + }); + + it('rejects a branch using characters git forbids in a ref', async () => { + const { status, body } = await postAnalyze(port, { + url: 'https://github.com/owner/repo', + branch: 'feature^bad', + }); + expect(status).toBe(400); + expect(JSON.parse(body).error).toContain('not allowed in a git ref'); + }); + + it('rejects a branch that would read as a git option', async () => { + // `git clone --branch --upload-pack=evil` would otherwise let a ref choose + // the subprocess git runs; buildBranchCloneArgs keeps the `--` separator, + // and this closes the same shape one layer earlier. + const { status, body } = await postAnalyze(port, { + url: 'https://github.com/owner/repo', + branch: '--upload-pack=evil', + }); + expect(status).toBe(400); + expect(JSON.parse(body).error).toContain('must not start with "-"'); + }); + + it('rejects a whitespace-only branch rather than silently indexing the default', async () => { + // The bug this feature fixes was a silent fallback to the default branch; + // an unusable selector must fail loudly, never degrade into that behavior. + const { status, body } = await postAnalyze(port, { + url: 'https://github.com/owner/repo', + branch: ' ', + }); + expect(status).toBe(400); + expect(JSON.parse(body).error).toContain('must not be empty'); + }); +}); diff --git a/gitnexus/test/unit/analyze-config.test.ts b/gitnexus/test/unit/analyze-config.test.ts index fb2999cb2..890877835 100644 --- a/gitnexus/test/unit/analyze-config.test.ts +++ b/gitnexus/test/unit/analyze-config.test.ts @@ -299,6 +299,33 @@ describe('analyze-config (.gitnexusrc support, #243)', () => { expect(() => validateBranchName('foo..bar', 'src')).toThrow(/must not contain ".."/); }); + it('validateBranchName rejects the ref shapes git check-ref-format rejects', () => { + // These previously passed validation and failed later in the git subprocess, + // which over HTTP meant a 202 and a background failure instead of a 400. + expect(() => validateBranchName('feature.lock', 'src')).toThrow(/must not end with "\.lock"/); + expect(() => validateBranchName('refs/heads.lock/x', 'src')).toThrow( + /must not end with "\.lock"/, + ); + expect(() => validateBranchName('/feature', 'src')).toThrow(/must not start or end with "\/"/); + expect(() => validateBranchName('feature/', 'src')).toThrow(/must not start or end with "\/"/); + expect(() => validateBranchName('feature//next', 'src')).toThrow(/consecutive slashes/); + expect(() => validateBranchName('@', 'src')).toThrow(/single character "@"/); + expect(() => validateBranchName('feature@{1}', 'src')).toThrow(/must not contain "@\{"/); + expect(() => validateBranchName('.hidden', 'src')).toThrow(/starting with "\."/); + expect(() => validateBranchName('feature/.hidden', 'src')).toThrow(/starting with "\."/); + expect(() => validateBranchName('feature.', 'src')).toThrow(/end with "\."/); + }); + + it('validateBranchName still accepts the real branch shapes those rules must not catch', () => { + // A dot, a slash and an @ are all legal in the middle of a ref — the new + // rules must reject only what git itself would. + expect(validateBranchName('release/1.2.3', 'src')).toBe('release/1.2.3'); + expect(validateBranchName('feature/lockfile-bump', 'src')).toBe('feature/lockfile-bump'); + expect(validateBranchName('user@host', 'src')).toBe('user@host'); + expect(validateBranchName('v1.0', 'src')).toBe('v1.0'); + expect(validateBranchName('a/b/c', 'src')).toBe('a/b/c'); + }); + it('validateBranchName rejects a newline / control character', () => { expect(() => validateBranchName('main\nrm -rf', 'src')).toThrow(/control or hidden|whitespace/); }); @@ -394,6 +421,20 @@ describe('analyze-config (.gitnexusrc support, #243)', () => { expect(() => validateBranchName('a'.repeat(256), 'src')).toThrow(/too long/); }); + it('validateBranchName rejects HEAD (case-sensitive) and accepts head (#3199)', () => { + expect(() => validateBranchName('HEAD', 'src')).toThrow(GitNexusRcError); + expect(() => validateBranchName('HEAD', 'src')).toThrow(/must not be "HEAD"/); + expect(() => validateBranchName(' HEAD ', 'src')).toThrow(GitNexusRcError); + expect(validateBranchName('head', 'src')).toBe('head'); + }); + + it('validateBranchName rejects a force-refspec "+" prefix (#3199)', () => { + expect(() => validateBranchName('+main', 'src')).toThrow(GitNexusRcError); + expect(() => validateBranchName('+main', 'src')).toThrow(/must not start with "\+"/); + expect(() => validateBranchName('+develop', 'src')).toThrow(GitNexusRcError); + expect(() => validateBranchName('+develop', 'src')).toThrow(/must not start with "\+"/); + }); + it('rejects Markdown-significant characters in a config name, allows real names (#1996)', async () => { await writeRc(JSON.stringify({ name: '**evil**' })); expect(() => loadAnalyzeConfig(dir)).toThrow(/Markdown-significant/); diff --git a/gitnexus/test/unit/analyze-job.test.ts b/gitnexus/test/unit/analyze-job.test.ts index 6743a0b73..aadb9b09d 100644 --- a/gitnexus/test/unit/analyze-job.test.ts +++ b/gitnexus/test/unit/analyze-job.test.ts @@ -56,6 +56,65 @@ describe('JobManager', () => { expect(job2.id).toBe(job1.id); }); + it('returns existing job for the same repoUrl AND the same branch', () => { + const job1 = manager.createJob({ + repoUrl: 'https://github.com/user/repo', + branch: 'development', + }); + manager.updateJob(job1.id, { status: 'analyzing' }); + const job2 = manager.createJob({ + repoUrl: 'https://github.com/user/repo', + branch: 'development', + }); + expect(job2.id).toBe(job1.id); + }); + + it('does not return the active job to a caller asking for a different branch', () => { + const job1 = manager.createJob({ + repoUrl: 'https://github.com/user/repo', + branch: 'development', + }); + manager.updateJob(job1.id, { status: 'analyzing' }); + // Handing job1 back would report branch "development" as the work being done + // for a caller that asked for "main". Falling through to the single-slot + // guard is the truthful answer. + expect(() => + manager.createJob({ repoUrl: 'https://github.com/user/repo', branch: 'main' }), + ).toThrow(/already in progress/); + }); + + it('treats an unpinned request as distinct from a branch-pinned one', () => { + const job1 = manager.createJob({ + repoUrl: 'https://github.com/user/repo', + branch: 'development', + }); + manager.updateJob(job1.id, { status: 'analyzing' }); + expect(() => manager.createJob({ repoUrl: 'https://github.com/user/repo' })).toThrow( + /already in progress/, + ); + }); + + it('keeps callers that omit branch deduping exactly as before', () => { + // The parameter is optional, so every pre-existing call site (upload route, + // embed manager, tests) compares undefined === undefined and is unaffected. + const job1 = manager.createJob({ repoPath: '/tmp/repo' }); + manager.updateJob(job1.id, { status: 'analyzing' }); + expect(manager.createJob({ repoPath: '/tmp/repo' }).id).toBe(job1.id); + }); + + it('carries the branch unchanged through the whole job lifecycle', () => { + // `branch` is part of dedup identity, so it must not drift mid-flight. + // `updateJob`'s Pick<> allowlist omits it, so no well-typed caller can + // change it; this pins that none of the updates the server actually + // performs (clone -> analyze -> terminal) disturbs it either. + const job = manager.createJob({ repoUrl: 'https://github.com/user/repo', branch: 'develop' }); + manager.updateJob(job.id, { status: 'cloning' }); + manager.updateJob(job.id, { repoPath: '/tmp/repo', status: 'analyzing' }); + manager.updateJob(job.id, { progress: { phase: 'parsing', percent: 30, message: 'Parsing' } }); + manager.updateJob(job.id, { status: 'complete', repoName: 'repo' }); + expect(manager.getJob(job.id)?.branch).toBe('develop'); + }); + it('updates job progress', () => { const job = manager.createJob({ repoUrl: 'https://github.com/user/repo' }); manager.updateJob(job.id, { diff --git a/gitnexus/test/unit/analyze-launch-branch-settle.test.ts b/gitnexus/test/unit/analyze-launch-branch-settle.test.ts new file mode 100644 index 000000000..0fa3dc6a4 --- /dev/null +++ b/gitnexus/test/unit/analyze-launch-branch-settle.test.ts @@ -0,0 +1,238 @@ +/** + * The finalization gate must watch the index the run actually WROTE. + * + * `registerRepo` always records the flat `.gitnexus` as `entry.storagePath`, but + * a pinned `--branch` run whose label differs from the flat slot's owner writes + * `lbug`/`gitnexus.json` under `branches//`. The gate used to probe the + * flat path unconditionally, so for such a run it watched files this job never + * rewrote: + * + * - the gate never settles, so the job stays non-terminal for the full 60s; + * - meanwhile the worker, having already sent `complete`, calls + * `process.exit(0)` ~500ms later; + * - the exit handler saw a non-terminal job and classified that clean exit as + * a crash — retrying a SUCCESSFUL analysis three times before failing it + * with `Worker crashed 3 times (code 0)`. + * + * Reported twice on #3199 (maintainer review + @azizur100389's repro). These + * tests pin both halves: the gate follows the placement, and a terminal IPC + * makes a subsequent exit 0 settlement rather than a crash. + * + * The filesystem mock is deliberately PATH-SENSITIVE — only the branch sub-slot + * looks freshly written. A gate that probes the flat path therefore cannot pass + * these tests by accident. + */ +import { afterEach, beforeEach, describe, expect, it, vi, type Mock } from 'vitest'; +import { EventEmitter } from 'node:events'; +import path from 'node:path'; + +// `vi.hoisted` is lifted above the imports, so nothing in here may reference +// them — these stay plain literals and `path` is only used below. +const H = vi.hoisted(() => ({ + forkMock: vi.fn(), + STORAGE_PATH: '/tmp/gitnexus-settle-storage', + REPO_PATH: '/tmp/gitnexus-settle-repo', + METADATA_FILE: 'gitnexus.json', + // Set per-test: the only directory the fake filesystem reports as freshly + // written. Anything else looks stale, exactly like a slot this job skipped. + settledDir: '', +})); +const { forkMock, REPO_PATH } = H; + +vi.mock('child_process', async () => { + const actual = await vi.importActual('child_process'); + return { ...actual, fork: H.forkMock }; +}); + +vi.mock('../../src/storage/repo-manager.js', () => ({ + canonicalizePath: (p: string) => p, + getStoragePath: () => H.STORAGE_PATH, + INDEX_METADATA_FILE: H.METADATA_FILE, + listRegisteredRepos: async () => [{ path: H.REPO_PATH, storagePath: H.STORAGE_PATH }], + registryPathEquals: (a: string, b: string) => a === b, +})); + +vi.mock('node:fs', async () => { + const actual = await vi.importActual('node:fs'); + return { + ...actual, + statSync: (p: string) => { + // Fresh only inside the directory this run is pretending to have written. + // Plain string work rather than `path.dirname`: this factory is hoisted + // above the imports too. + const file = String(p); + const dir = file.slice(0, Math.max(file.lastIndexOf('/'), file.lastIndexOf('\\'))); + if (H.settledDir && dir === H.settledDir) { + return { mtimeMs: Number.MAX_SAFE_INTEGER }; + } + return { mtimeMs: 0 }; + }, + existsSync: () => false, // no WAL/shadow/checkpoint sidecars anywhere + }; +}); + +import { createLaunchAnalysisWorker } from '../../src/server/analyze-launch.js'; +import { JobManager } from '../../src/server/analyze-job.js'; +import { projectAnalyzeResultForIpc } from '../../src/server/analyze-worker-ipc.js'; +import { BRANCHES_DIR, branchSlug } from '../../src/storage/branch-index.js'; +import type { AnalyzeResult } from '../../src/core/run-analyze.js'; +import type { CompleteMessage } from '../../src/server/analyze-worker.js'; + +const BRANCH = 'feature/settle'; + +/** The `complete` message the worker really sends, via the production projection. */ +const completeMessage = ( + isPrimaryBranch: boolean, + extras?: { alreadyUpToDate?: boolean }, +): CompleteMessage => { + const result = { + repoName: 'settle-fixture', + repoPath: REPO_PATH, + stats: { files: 3, nodes: 9, edges: 12 }, + isPrimaryBranch, + ...(extras?.alreadyUpToDate ? { alreadyUpToDate: true } : {}), + } satisfies Partial as AnalyzeResult; + return { type: 'complete', result: projectAnalyzeResultForIpc(result) }; +}; + +interface FakeChild extends EventEmitter { + stderr: EventEmitter; + send: Mock<(msg: unknown) => boolean>; + kill: Mock<(signal?: NodeJS.Signals) => boolean>; +} + +const makeChild = (): FakeChild => { + const child = new EventEmitter() as FakeChild; + child.stderr = new EventEmitter(); + child.send = vi.fn(); + child.kill = vi.fn(); + return child; +}; + +describe('finalization gate follows the placement the run chose', () => { + let jobManager: JobManager; + let child: FakeChild; + let backendInit: Mock<() => Promise>; + let closeDbHandle: Mock<() => Promise>; + + const launcher = () => + createLaunchAnalysisWorker({ + jobManager, + backend: { init: backendInit }, + acquireRepoLock: () => null, + releaseRepoLock: () => {}, + closeDbHandle, + }); + + beforeEach(() => { + jobManager = new JobManager(); + child = makeChild(); + forkMock.mockImplementation(() => child); + backendInit = vi.fn(async () => true); + closeDbHandle = vi.fn(async () => {}); + H.settledDir = ''; + }); + + afterEach(() => { + jobManager.dispose(); + vi.restoreAllMocks(); + forkMock.mockReset(); + }); + + it('completes a branch sub-slot run, whose files are NOT in the flat slot', async () => { + // Only `branches//` looks written. The flat slot is stale, so a gate + // probing it would spin to the 60s timeout instead of settling here. + H.settledDir = path.join(H.STORAGE_PATH, BRANCHES_DIR, branchSlug(BRANCH)); + + const job = jobManager.createJob({ repoPath: REPO_PATH, branch: BRANCH }); + launcher()(job, REPO_PATH, { branch: BRANCH }); + + child.emit('message', completeMessage(false)); + + await vi.waitFor(() => expect(jobManager.getJob(job.id)?.status).toBe('complete')); + expect(jobManager.getJob(job.id)?.error).toBeUndefined(); + expect(backendInit).toHaveBeenCalledTimes(1); + }); + + it('still settles a flat-slot run against the flat slot', async () => { + // Control: the primary-branch case must keep watching `entry.storagePath`. + H.settledDir = H.STORAGE_PATH; + + const job = jobManager.createJob({ repoPath: REPO_PATH }); + launcher()(job, REPO_PATH, {}); + + child.emit('message', completeMessage(true)); + + await vi.waitFor(() => expect(jobManager.getJob(job.id)?.status).toBe('complete')); + expect(backendInit).toHaveBeenCalledTimes(1); + }); + + it('settles a first-pin (branch SET, isPrimaryBranch true) against the flat slot', async () => { + // Fresh clone: first pin adopts the flat slot. The slug dir is stale, so a + // gate that does `branch ? slugDir : flat` would spin the 60s timeout here. + H.settledDir = H.STORAGE_PATH; + + const job = jobManager.createJob({ repoPath: REPO_PATH, branch: BRANCH }); + launcher()(job, REPO_PATH, { branch: BRANCH }); + + child.emit('message', completeMessage(true)); + + await vi.waitFor(() => expect(jobManager.getJob(job.id)?.status).toBe('complete')); + expect(jobManager.getJob(job.id)?.error).toBeUndefined(); + expect(backendInit).toHaveBeenCalledTimes(1); + }); + + it('completes alreadyUpToDate quickly even when the slot is stale, without retrying', async () => { + // No directory looks freshly written. Without the alreadyUpToDate skip the + // mtime gate would hold the analyze slot for the full 60s settle timeout. + H.settledDir = ''; + + const job = jobManager.createJob({ repoPath: REPO_PATH }); + launcher()(job, REPO_PATH, {}); + + child.emit('message', completeMessage(true, { alreadyUpToDate: true })); + child.emit('exit', 0); + + await vi.waitFor(() => expect(jobManager.getJob(job.id)?.status).toBe('complete'), { + timeout: 2_000, + }); + expect(jobManager.getJob(job.id)?.error).toBeUndefined(); + expect(backendInit).toHaveBeenCalledTimes(1); + expect(forkMock).toHaveBeenCalledTimes(1); + expect(jobManager.getJob(job.id)?.retryCount).toBe(0); + }); + + it('does not fork a retry when the worker exits 0 after reporting complete', async () => { + // The worker exits ~500ms after the `complete` IPC, while the gate is still + // running and the job is deliberately non-terminal. That exit is the worker + // winding down, not dying. + H.settledDir = path.join(H.STORAGE_PATH, BRANCHES_DIR, branchSlug(BRANCH)); + + const job = jobManager.createJob({ repoPath: REPO_PATH, branch: BRANCH }); + launcher()(job, REPO_PATH, { branch: BRANCH }); + + child.emit('message', completeMessage(false)); + child.emit('exit', 0); + + await vi.waitFor(() => expect(jobManager.getJob(job.id)?.status).toBe('complete')); + expect(jobManager.getJob(job.id)?.error).toBeUndefined(); + // One fork for the run itself; a retry would be a second. + expect(forkMock).toHaveBeenCalledTimes(1); + expect(jobManager.getJob(job.id)?.retryCount).toBe(0); + }); + + it('still treats an exit with no terminal IPC as a crash worth retrying', async () => { + // The guard must not swallow real crashes: no `complete`/`error` was sent. + H.settledDir = H.STORAGE_PATH; + + const job = jobManager.createJob({ repoPath: REPO_PATH }); + launcher()(job, REPO_PATH, {}); + + child.emit('exit', 1); + + // The first retry is scheduled on a 1s backoff, so this needs more than + // vi.waitFor's default budget. + await vi.waitFor(() => expect(forkMock).toHaveBeenCalledTimes(2), { timeout: 4_000 }); + expect(jobManager.getJob(job.id)?.retryCount).toBe(1); + }); +}); diff --git a/gitnexus/test/unit/analyze-launch-collapse.test.ts b/gitnexus/test/unit/analyze-launch-collapse.test.ts index a32cf530e..226cde9cc 100644 --- a/gitnexus/test/unit/analyze-launch-collapse.test.ts +++ b/gitnexus/test/unit/analyze-launch-collapse.test.ts @@ -170,6 +170,50 @@ describe('createLaunchAnalysisWorker — collapsed index is never published', () ); }); + it('forwards the index-branch selector to the worker', () => { + const launch = createLaunchAnalysisWorker({ + jobManager, + backend: { init: backendInit }, + acquireRepoLock: () => null, + releaseRepoLock: () => {}, + closeDbHandle, + }); + const job = jobManager.createJob({ repoPath: REPO_PATH }); + + launch(job, REPO_PATH, { branch: 'development' }); + + // `StartMessage.options` is typed as `AnalyzeOptions`, so this key IS + // `AnalyzeOptions.branch` — the field `resolveWriteTarget` reads to choose + // the run's storage slot. (It does not always mean a `branches//` + // sub-slot: `resolveBranchPlacement` keeps the flat slot when that slot has + // no owner, or when its owner is already this label.) A rename breaks this + // test. + expect(child.send).toHaveBeenCalledWith( + expect.objectContaining({ + options: expect.objectContaining({ branch: 'development' }), + }), + ); + }); + + it('omits branch entirely when the caller did not select one', () => { + const launch = createLaunchAnalysisWorker({ + jobManager, + backend: { init: backendInit }, + acquireRepoLock: () => null, + releaseRepoLock: () => {}, + closeDbHandle, + }); + const job = jobManager.createJob({ repoPath: REPO_PATH }); + + launch(job, REPO_PATH, {}); + + // Not merely undefined: absent. `AnalyzeOptions.branch === undefined` is the + // documented signal for "target the flat workspace slot", so sending the key + // with an undefined value must not become the way that default is expressed. + const sent = child.send.mock.calls.at(0)?.[0] as { options: Record }; + expect(Object.hasOwn(sent.options, 'branch')).toBe(false); + }); + afterEach(() => { jobManager.dispose(); vi.restoreAllMocks(); diff --git a/gitnexus/test/unit/git-clone.test.ts b/gitnexus/test/unit/git-clone.test.ts index 715955d74..5b6ad1701 100644 --- a/gitnexus/test/unit/git-clone.test.ts +++ b/gitnexus/test/unit/git-clone.test.ts @@ -15,6 +15,7 @@ vi.mock('../../src/core/logger.js', () => ({ })); import { + analyzeCloneOptions, extractRepoName, extractWebRepoName, getCloneDir, @@ -671,6 +672,359 @@ describe('git-clone', () => { } }); + // `assertRemoteMatchesRequestedUrl` runs REAL git (it is not injectable), so + // these fixtures are real repositories with a matching origin; only the + // branch logic under test is driven through `runGitForTest`. + const REMOTE = 'https://github.com/owner/repo.git'; + const makeExistingClone = async (root: string, branch: string) => { + const target = path.join(root, 'repo'); + await fs.mkdir(target, { recursive: true }); + await runGit(['init', `--initial-branch=${branch}`], target); + await runGit(['remote', 'add', 'origin', REMOTE], target); + return target; + }; + + it('re-indexing the SAME pinned branch fetches via a safe refspec instead of refusing a dirty tree', async () => { + // Analyze writes AGENTS.md / CLAUDE.md / .claude/ into the clone, so the + // tree is dirty from its own first run. Routing a same-branch request + // through the checkout path made every pinned RE-index fail asking for + // `overwrite_local_changes` — a flag the route will not pass because it + // would `git clean --force -d` the directory (#3199 review). + const root = await mkControlledRoot('gitnexus-controlled-root-'); + const calls: string[][] = []; + try { + const target = await makeExistingClone(root, 'develop'); + const runGitForTest = vi.fn(async (args: string[]) => { + calls.push(args); + if (args[0] === 'rev-parse') return 'develop\n'; + if (args[0] === 'status') return ' M AGENTS.md\n?? .claude/\n'; // dirty, as analyze leaves it + return ''; + }); + await cloneOrPull(REMOTE, target, undefined, { + allowedCloneRoot: root, + expectedRepoName: 'repo', + branch: 'develop', + runGitForTest, + }); + } finally { + await fs.rm(root, { recursive: true, force: true }); + } + + const verbs = calls.map((c) => c[0]); + expect(calls).toContainEqual([ + 'fetch', + '--depth', + '1', + 'origin', + '+refs/heads/develop:refs/remotes/origin/develop', + ]); + expect(calls).toContainEqual(['checkout', '-B', 'develop', 'origin/develop']); + // Never a raw `origin ` pull — `+develop` would be a force-fetch. + expect( + calls.some((c) => c[0] === 'pull' && c.includes('origin') && c.includes('develop')), + ).toBe(false); + expect(verbs).not.toContain('status'); // so the dirty check never ran + expect(calls.some((c) => c[0] === 'merge')).toBe(false); + // Must not fall back to a bare `git pull --ff-only` — that follows + // `branch..merge`, which is not verified (only origin.url is). + expect(calls.some((c) => c[0] === 'pull' && c.length === 2)).toBe(false); + }); + + it('same-branch argv uses +refs/heads/develop:refs/remotes/origin/develop, never raw develop as a pull dest', async () => { + const root = await mkControlledRoot('gitnexus-controlled-root-'); + const calls: string[][] = []; + try { + const target = await makeExistingClone(root, 'develop'); + const runGitForTest = vi.fn(async (args: string[]) => { + calls.push(args); + if (args[0] === 'rev-parse' && args[1] === '--abbrev-ref') return 'develop\n'; + return ''; + }); + await cloneOrPull(REMOTE, target, undefined, { + allowedCloneRoot: root, + expectedRepoName: 'repo', + branch: 'develop', + runGitForTest, + }); + } finally { + await fs.rm(root, { recursive: true, force: true }); + } + + const fetchCall = calls.find((c) => c[0] === 'fetch'); + expect(fetchCall).toEqual([ + 'fetch', + '--depth', + '1', + 'origin', + '+refs/heads/develop:refs/remotes/origin/develop', + ]); + expect(fetchCall?.[4]).not.toBe('develop'); + expect(calls.some((c) => c[0] === 'pull' && c[3] === 'develop')).toBe(false); + }); + + it('does not treat a leading-plus branch as a force-fetch pull refspec', async () => { + // If `+develop` somehow reached cloneOrPull, the heads/ mapping keeps + // the `+` inside the ref name. It is not git's force-fetch prefix. + const root = await mkControlledRoot('gitnexus-controlled-root-'); + const calls: string[][] = []; + try { + const target = await makeExistingClone(root, 'main'); + const runGitForTest = vi.fn(async (args: string[]) => { + calls.push(args); + if (args[0] === 'rev-parse' && args[1] === '--abbrev-ref') return '+develop\n'; + return ''; + }); + await cloneOrPull(REMOTE, target, undefined, { + allowedCloneRoot: root, + expectedRepoName: 'repo', + branch: '+develop', + runGitForTest, + }); + } finally { + await fs.rm(root, { recursive: true, force: true }); + } + + expect(calls).toContainEqual([ + 'fetch', + '--depth', + '1', + 'origin', + '+refs/heads/+develop:refs/remotes/origin/+develop', + ]); + expect( + calls.some((c) => c[0] === 'pull' && c.some((a) => a === '+develop' || a.startsWith('+'))), + ).toBe(false); + // Force prefix is on the mapping, not a force-update of `develop`. + expect( + calls.some( + (c) => c[0] === 'fetch' && c.includes('+refs/heads/develop:refs/remotes/origin/develop'), + ), + ).toBe(false); + }); + + it('restores a dirty AGENTS.md on the same branch without taking the switch refuse path', async () => { + const root = await mkControlledRoot('gitnexus-controlled-root-'); + const calls: string[][] = []; + try { + const target = await makeExistingClone(root, 'develop'); + const runGitForTest = vi.fn(async (args: string[]) => { + calls.push(args); + if (args[0] === 'rev-parse' && args[1] === '--abbrev-ref') return 'develop\n'; + if (args[0] === 'ls-files' && args.includes('./AGENTS.md')) return 'AGENTS.md\n'; + if (args[0] === 'status') return ' M AGENTS.md\n?? .claude/\n'; + return ''; + }); + await expect( + cloneOrPull(REMOTE, target, undefined, { + allowedCloneRoot: root, + expectedRepoName: 'repo', + branch: 'develop', + runGitForTest, + }), + ).resolves.toBe(target); + } finally { + await fs.rm(root, { recursive: true, force: true }); + } + + expect(calls).toContainEqual(['ls-files', '--', './AGENTS.md']); + expect(calls).toContainEqual(['checkout', 'HEAD', '--', './AGENTS.md']); + expect(calls).toContainEqual([ + 'clean', + '-fdx', + '--', + './AGENTS.md', + './CLAUDE.md', + './.claude', + ]); + expect(calls.some((c) => c[0] === 'status')).toBe(false); + expect(calls).toContainEqual(['checkout', '-B', 'develop', 'origin/develop']); + expect(calls.some((c) => c[0] === 'clean' && c.includes('/.gitnexus'))).toBe(false); + }); + + it('removes untracked AGENTS.md on the same branch so merge is not blocked', async () => { + const root = await mkControlledRoot('gitnexus-controlled-root-'); + const calls: string[][] = []; + try { + const target = await makeExistingClone(root, 'develop'); + const runGitForTest = vi.fn(async (args: string[]) => { + calls.push(args); + if (args[0] === 'rev-parse' && args[1] === '--abbrev-ref') return 'develop\n'; + if (args[0] === 'ls-files') return ''; // untracked analyze output + return ''; + }); + await cloneOrPull(REMOTE, target, undefined, { + allowedCloneRoot: root, + expectedRepoName: 'repo', + branch: 'develop', + runGitForTest, + }); + } finally { + await fs.rm(root, { recursive: true, force: true }); + } + + expect(calls).toContainEqual([ + 'clean', + '-fdx', + '--', + './AGENTS.md', + './CLAUDE.md', + './.claude', + ]); + expect(calls.some((c) => c[0] === 'checkout' && c.includes('HEAD'))).toBe(false); + expect(calls).toContainEqual(['checkout', '-B', 'develop', 'origin/develop']); + }); + + it('still switches — and still refuses a dirty tree — for a DIFFERENT branch', async () => { + // The refusal must survive where it matters: a real switch can discard + // local work, so the fast path above must not weaken it. + const root = await mkControlledRoot('gitnexus-controlled-root-'); + try { + const target = await makeExistingClone(root, 'main'); + const runGitForTest = vi.fn(async (args: string[]) => { + if (args[0] === 'rev-parse') return 'main\n'; // on a different branch + if (args[0] === 'status') return ' M src/index.ts\n'; + return ''; + }); + await expect( + cloneOrPull(REMOTE, target, undefined, { + allowedCloneRoot: root, + expectedRepoName: 'repo', + branch: 'develop', + runGitForTest, + }), + ).rejects.toThrow(/local changes detected/); + } finally { + await fs.rm(root, { recursive: true, force: true }); + } + }); + + it('treats a detached HEAD as needing the checkout path', async () => { + // `rev-parse --abbrev-ref HEAD` reports `HEAD` when detached. A SHA that + // does not match the requested ref is a real switch, not "already there". + const root = await mkControlledRoot('gitnexus-controlled-root-'); + const calls: string[][] = []; + try { + const target = await makeExistingClone(root, 'main'); + const runGitForTest = vi.fn(async (args: string[]) => { + calls.push(args); + if (args[0] === 'rev-parse' && args[1] === '--abbrev-ref') return 'HEAD\n'; + if (args[0] === 'rev-parse' && args[1] === 'HEAD') + return 'aaa111aaa111aaa111aaa111aaa111aaa111aaa1\n'; + if (args[0] === 'rev-parse') return 'bbb222bbb222bbb222bbb222bbb222bbb222bbb2\n'; + if (args[0] === 'status') return ''; // clean, so the checkout proceeds + return ''; + }); + await cloneOrPull(REMOTE, target, undefined, { + allowedCloneRoot: root, + expectedRepoName: 'repo', + branch: 'develop', + runGitForTest, + }); + } finally { + await fs.rm(root, { recursive: true, force: true }); + } + expect(calls.map((c) => c[0])).toContain('checkout'); + expect(calls.some((c) => c[0] === 'checkout' && c.includes('-B'))).toBe(true); + }); + + it('does not switch or fetch when a detached HEAD SHA matches the requested ref', async () => { + // Tag / SHA pin: already at the commit. Fetching refs/heads/ would + // follow a same-named branch past the pin (#3199 review). + const root = await mkControlledRoot('gitnexus-controlled-root-'); + const calls: string[][] = []; + const sha = 'aaa111aaa111aaa111aaa111aaa111aaa111aaa1'; + try { + const target = await makeExistingClone(root, 'main'); + const runGitForTest = vi.fn(async (args: string[]) => { + calls.push(args); + if (args[0] === 'rev-parse' && args[1] === '--abbrev-ref') return 'HEAD\n'; + if (args[0] === 'rev-parse') return `${sha}\n`; + if (args[0] === 'status') return ' M AGENTS.md\n'; + if (args[0] === 'ls-files' && args.includes('./AGENTS.md')) return 'AGENTS.md\n'; + return ''; + }); + await cloneOrPull(REMOTE, target, undefined, { + allowedCloneRoot: root, + expectedRepoName: 'repo', + branch: 'develop', + runGitForTest, + }); + } finally { + await fs.rm(root, { recursive: true, force: true }); + } + + expect(calls.some((c) => c[0] === 'status')).toBe(false); + expect(calls.some((c) => c[0] === 'checkout' && c.includes('-B'))).toBe(false); + expect(calls).toContainEqual(['checkout', 'HEAD', '--', './AGENTS.md']); + expect(calls.map((c) => c[0])).not.toContain('fetch'); + expect(calls.map((c) => c[0])).not.toContain('merge'); + }); + + it('peels an annotated tag so a tag pin is not treated as a switch', async () => { + // `rev-parse v1.0` is the tag object; HEAD is the peeled commit. Without + // `^{commit}` the SHA match misses and re-index porcelain-refuses. + const root = await mkControlledRoot('gitnexus-controlled-root-'); + const calls: string[][] = []; + const commit = 'aaa111aaa111aaa111aaa111aaa111aaa111aaa1'; + const tagObject = 'cccccccccccccccccccccccccccccccccccccccc'; + try { + const target = await makeExistingClone(root, 'main'); + const runGitForTest = vi.fn(async (args: string[]) => { + calls.push(args); + if (args[0] === 'rev-parse' && args[1] === '--abbrev-ref') return 'HEAD\n'; + if (args[0] === 'rev-parse' && args[1] === 'HEAD') return `${commit}\n`; + if (args[0] === 'rev-parse' && args[1] === 'v1.0^{commit}') return `${commit}\n`; + if (args[0] === 'rev-parse' && args[1] === 'v1.0') return `${tagObject}\n`; + if (args[0] === 'rev-parse') throw new Error('unknown ref'); + if (args[0] === 'status') return ' M AGENTS.md\n'; + return ''; + }); + await cloneOrPull(REMOTE, target, undefined, { + allowedCloneRoot: root, + expectedRepoName: 'repo', + branch: 'v1.0', + runGitForTest, + }); + } finally { + await fs.rm(root, { recursive: true, force: true }); + } + + expect(calls).toContainEqual(['rev-parse', 'v1.0^{commit}']); + expect(calls.some((c) => c[0] === 'status')).toBe(false); + expect(calls.some((c) => c[0] === 'checkout' && c.includes('-B'))).toBe(false); + expect(calls.map((c) => c[0])).not.toContain('fetch'); + }); + + it('still refuses a dirty tree when a detached HEAD SHA does not match the requested ref', async () => { + const root = await mkControlledRoot('gitnexus-controlled-root-'); + const calls: string[][] = []; + try { + const target = await makeExistingClone(root, 'main'); + const runGitForTest = vi.fn(async (args: string[]) => { + calls.push(args); + if (args[0] === 'rev-parse' && args[1] === '--abbrev-ref') return 'HEAD\n'; + if (args[0] === 'rev-parse' && args[1] === 'HEAD') + return 'aaa111aaa111aaa111aaa111aaa111aaa111aaa1\n'; + if (args[0] === 'rev-parse') return 'bbb222bbb222bbb222bbb222bbb222bbb222bbb2\n'; + if (args[0] === 'status') return ' M src/index.ts\n'; + return ''; + }); + await expect( + cloneOrPull(REMOTE, target, undefined, { + allowedCloneRoot: root, + expectedRepoName: 'repo', + branch: 'develop', + runGitForTest, + }), + ).rejects.toThrow(/local changes detected/); + } finally { + await fs.rm(root, { recursive: true, force: true }); + } + + expect(calls.some((c) => c[0] === 'status')).toBe(true); + expect(calls.some((c) => c[0] === 'checkout')).toBe(false); + }); + it('allows auto-sync SSH SCP clone URLs with a per-repo timeout', async () => { const root = await mkControlledRoot('gitnexus-controlled-root-'); const target = path.join(root, 'repo'); @@ -1367,3 +1721,115 @@ describe('git-clone', () => { }); }); }); + +describe('getCloneDir — a branch-pinned analyze gets its own working tree', () => { + it('keeps the historic directory for an unpinned request', () => { + // Backward compatibility: existing installs must keep using the dir they have. + expect(getCloneDir('Hello-World')).toBe(getCloneDir('Hello-World', undefined)); + }); + + it('gives a pinned request a different directory from the unpinned one', () => { + // This separation is what stops (a) a pinned request failing on the dirty + // tree an earlier analyze left, and (b) a later unpinned request pulling on + // the branch a pin left checked out and indexing it as the default. + expect(getCloneDir('Hello-World', 'development')).not.toBe(getCloneDir('Hello-World')); + }); + + it('gives two different branches two different directories', () => { + expect(getCloneDir('Hello-World', 'development')).not.toBe( + getCloneDir('Hello-World', 'release/1.2'), + ); + }); + + it('is stable for the same branch', () => { + expect(getCloneDir('Hello-World', 'release/1.2')).toBe( + getCloneDir('Hello-World', 'release/1.2'), + ); + }); + + it('keeps a slash-bearing ref inside a single path segment under the clone root', () => { + // `release/1.2` must not become a nested directory, or the containment + // guarantees around CLONE_ROOT would be reasoning about a different path. + const dir = getCloneDir('Hello-World', 'release/1.2'); + const base = getCloneDir('Hello-World'); + expect(path.dirname(dir)).toBe(path.dirname(base)); + expect(path.basename(dir)).not.toContain('/'); + }); + + it('round-trips its own directory name, which is how DELETE re-derives it', () => { + // DELETE /api/repo calls getCloneDir(entry.name); the pinned repo is + // registered under this basename, so the two must agree. + const dir = getCloneDir('Hello-World', 'development'); + expect(getCloneDir(path.basename(dir))).toBe(dir); + }); + + it('pinned basename differs from the GitHub stem (mid-job repoName / registerRepo)', () => { + // POST /api/analyze must set job.repoName and registryName to + // path.basename(targetPath), not extractWebRepoName(url). The stem is + // only getCloneDir's first argument (#3199 review). + const stem = 'Hello-World'; + const basename = path.basename(getCloneDir(stem, 'development')); + expect(basename).not.toBe(stem); + expect(basename.startsWith(`${stem}__`)).toBe(true); + }); + + it('keeps the directory name inside the 255-byte filesystem limit', () => { + // validateBranchName allows a 255-char ref and branchSlug appends 9 more, + // so the naive `__` reached 267 and the clone could not create + // its target directory. + const longBranch = 'b'.repeat(255); + const base = path.basename(getCloneDir('Hello-World', longBranch)); + expect(base.length).toBeLessThanOrEqual(255); + }); + + it('still separates two long branches that share a prefix', () => { + // Trimming keeps the hash, which is a digest of the FULL ref — otherwise + // two long branches would collapse onto one directory and silently share + // an index. + const a = 'b'.repeat(250) + 'one'; + const b = 'b'.repeat(250) + 'two'; + expect(getCloneDir('Hello-World', a)).not.toBe(getCloneDir('Hello-World', b)); + expect(path.basename(getCloneDir('Hello-World', a)).length).toBeLessThanOrEqual(255); + }); + + it('round-trips a trimmed directory name too', () => { + const dir = getCloneDir('Hello-World', 'b'.repeat(255)); + expect(getCloneDir(path.basename(dir))).toBe(dir); + }); + + it('still rejects a traversal attempt in the repo name', () => { + expect(() => getCloneDir('..', 'development')).toThrow(/Invalid repository name/); + expect(() => getCloneDir('a/b', 'development')).toThrow(/Invalid repository name/); + }); +}); + +describe('analyzeCloneOptions — the /api/analyze glue for #3198', () => { + // The route passes the result straight to `cloneOrPull`. Inline, a regression + // that dropped `branch` for token-less URLs left every other test green while + // silently reindexing the default branch — so each combination is pinned. + it('returns undefined when neither a token nor a branch is supplied', () => { + expect(analyzeCloneOptions(undefined, undefined)).toBeUndefined(); + }); + + it('carries a token on its own', () => { + expect(analyzeCloneOptions('ghp_token', undefined)).toEqual({ token: 'ghp_token' }); + }); + + it('carries a branch on its own — the public-repo case', () => { + // The regression that would reopen #3198: a branch requested for a public + // URL must still reach cloneOrPull, with no token in play. + expect(analyzeCloneOptions(undefined, 'development')).toEqual({ branch: 'development' }); + }); + + it('carries both together', () => { + expect(analyzeCloneOptions('ghp_token', 'development')).toEqual({ + token: 'ghp_token', + branch: 'development', + }); + }); + + it('treats an empty branch as absent rather than sending an empty ref', () => { + expect(analyzeCloneOptions('ghp_token', '')).toEqual({ token: 'ghp_token' }); + expect(analyzeCloneOptions('', '')).toBeUndefined(); + }); +}); diff --git a/gitnexus/test/unit/git-ref.test.ts b/gitnexus/test/unit/git-ref.test.ts new file mode 100644 index 000000000..af2492926 --- /dev/null +++ b/gitnexus/test/unit/git-ref.test.ts @@ -0,0 +1,15 @@ +import { describe, expect, it } from 'vitest'; +import { InvalidBranchError, validateBranchName } from '../../src/core/git-ref.js'; + +describe('core/git-ref', () => { + it('throws InvalidBranchError with name "InvalidBranchError"', () => { + expect(() => validateBranchName('HEAD', 'src')).toThrow(InvalidBranchError); + try { + validateBranchName('HEAD', 'src'); + throw new Error('expected InvalidBranchError'); + } catch (err) { + expect(err).toBeInstanceOf(InvalidBranchError); + expect((err as Error).name).toBe('InvalidBranchError'); + } + }); +}); From 18cbeb907c87c50d548b5c62102080750baac972 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Gerg=C5=91=20Magyar?= Date: Tue, 8 Sep 2026 18:22:04 +0100 Subject: [PATCH 16/17] feat(eval): record provider-native usage at the gateway instead of inferring it after translation (#3220) * feat(eval): record provider-native usage at the gateway, not after translation The benchmark reads token counts out of Claude Code's session output, which is Anthropic-shaped whatever actually served the request. That holds until the upstream is OpenAI, because the two providers do not merely name their fields differently - they mean opposite things by them: Anthropic: total_input = input_tokens + cache_creation + cache_read (input_tokens is the UNCACHED remainder; cache fields ADD) OpenAI: total_input = input_tokens ordinary = input_tokens - cached - cache_write (input_tokens is the WHOLE; cache fields are SUBSETS) Adding OpenAI's three double-counts; subtracting Anthropic's under-counts. One shared struct cannot be right for both, so the seam goes at the gateway, on the far side of the translation: a LiteLLM callback appends each upstream request's usage verbatim, along with the model that actually answered, the response id and the cell it belongs to. Normalization is derived offline from that record, so the derivation can be revisited without re-running a paid sweep. Two rules the tests encode literally. The native object is authoritative. The callback stores it unflattened, unrenamed and unsummed. Reasoning tokens are kept as the decomposition of output tokens they are, not added to them a second time. A field nobody reported is unknown, never zero. A stored cache_read of 0 used to mean either "the provider said zero" or "our adapter never looked" - the first says caching is not working, the second says we cannot tell. NormalizedUsage therefore uses None, and refuses to compute the ordinary portion when a term is missing rather than subtracting an invented zero. Mutation-checked three ways. Giving OpenAI Anthropic's arithmetic fails four tests. Making unknown fall back to zero fails the unknown test. Dropping input_tokens_details in the callback fails the end-to-end accounting test with "assert None == 3000" - it goes unknown rather than passing with zeros, which was the point of the exercise. The actual model is recorded separately from the requested role because several Claude role names map onto one upstream model here; pricing must follow what answered. Cost is deliberately NOT stored: prices change, and tokens plus a versioned pricing table can answer both what a past run cost and what the same usage would cost today, without rewriting historical evidence. The callback never raises. A cell that fails still spent money upstream, and losing the accounting because a log write failed is the worse outcome. Failed requests are recorded too. No caching configuration, model, skill or promotion change: this installs the thermometer without altering the experiment. 538 eval tests pass plus 27 gateway tests; ruff clean. The two test_model_gateway.py failures are environmental - litellm[proxy]'s console script is absent in this venv - and predate this branch. * fix(eval): drop the accidentally committed .venv symlink I symlinked eval/.venv at a sibling worktree's virtualenv to avoid rebuilding it, and git add -A committed the symlink. .gitignore lists ".venv/" with a trailing slash, which matches a directory and not a symlink, so nothing stopped it. That broke eval / containment (windows), where uv then refused to create the environment: "failed to create directory eval\\.venv: Cannot create a file when that file already exists". A machine-specific absolute path had no business in the tree in the first place. Removed, and .gitignore now also lists the bare name so the same slip cannot repeat. * Address PR review feedback (#3220) Forward the usage environment into the proxy. This is the one that mattered: the callback returns immediately when GITNEXUS_BENCH_PROVIDER_USAGE is absent, the proxy runs as its own process, and Popen(env=...) REPLACES the parent environment rather than extending it. The gateway's allowlist carried the OpenAI and master keys and nothing else, so the callback loaded, found no destination, and silently recorded nothing on every request. The accounting looked configured and measured nothing at all. My tests could not see it. They set the variable in-process and called the logger directly, so none of them ever crossed the subprocess boundary the feature actually runs behind. The new test drives OpenAIGateway.__enter__ with Popen captured and asserts each variable reaches the child - and that the result is still an allowlist rather than the inherited parent environment, since forwarding by name is what keeps the credential boundary explicit. Resolve the provider label into an adapter key. The callback recorded LiteLLM's custom_llm_provider, which is "openai", while the adapter table is keyed "openai-responses" - so nothing the logger wrote could have been normalized. The end-to-end test hid this by passing OPENAI_RESPONSES by hand instead of using the provider the log recorded; it now uses the logged value, which is what makes the mismatch visible. The label alone cannot pick an adapter: LiteLLM reports "openai" for Chat Completions as well, and the two report usage differently. canonical_provider combines the label with the call type and returns None when it cannot resolve one, so normalize_usage refuses rather than guessing token semantics. Both are stored - provider_label is what LiteLLM said, provider is the adapter key. The shared env-var names moved into provider_usage.py so model_gateway can import them without importing litellm, which only the in-proxy callback needs. Mutation-checked. Removing the forwarding loop fails the gateway test; using the raw label as the adapter key fails two. 656 eval tests pass, ruff clean. The two test_model_gateway.py failures are the environmental ones - litellm[proxy]'s console script is absent here, which is also why the new test patches the argv builder to reach Popen at all. * fix(eval): stop recording a cell id the proxy cannot know Setting out to build the correlation this PR was missing - cell usage as the sum of its upstream requests - turned up that the field it would have been built on cannot hold what its name claims. attach_openai_gateway wraps the whole sweep (runner.py:2122), so ONE proxy serves every cell, and its environment is fixed for that process's lifetime. Cells run concurrently under --workers and interleave requests through it. A cell id forwarded at launch is therefore the same constant on every event the callback ever writes - not an attribution, just a label that looks like one. Worse than absent, because a reader would trust it. So GITNEXUS_BENCH_CELL_ID is gone rather than left to be wired up later. What remains is honest about its scope: sweep_id is genuinely sweep-wide, and session_id is the per-request half - the only thing that can attribute a request to a cell, since anything read from the environment is shared by all of them. It is recorded even when the provider supplies nothing, because knowing attribution is unavailable is itself a fact about the run. Pinned by a test asserting the forwarded set contains no per-cell variable, so a later change does not reintroduce one and quietly stamp a single value across concurrent cells. What this leaves open, stated plainly: per-cell attribution is NOT built, and cannot be until a per-request identifier is available. Whether Claude Code propagates a session identifier through the proxy is unverified - determining it needs a real session against the gateway, which is a paid run. Sweep-level totals and per-request cache ratios do not need it, and those are what the caching question actually turns on. 658 eval tests pass, ruff clean; the two test_model_gateway.py failures remain environmental. * fix(eval): keep the usage callback importable the way LiteLLM loads it CI caught a regression I introduced: "ImportError: Could not import handler from provider_usage_callback", and the proxy exited before becoming ready. Moving the shared constants into provider_usage.py, I imported them from the callback with "from .provider_usage import ...". But LiteLLM resolves a dotted callback through spec_from_file_location against the config directory, so the copied file runs as a top-level module with no parent package and no sys.path entry - the relative import raises and the gateway never starts. The module's own docstring says it is deliberately self-contained for exactly this reason, and I broke that invariant while tidying. The in-package tests could not see it. They import workflow_bench.litellm_usage_callback, where the relative import resolves fine; the failure only exists on the path where the file is copied and loaded standalone. The callback carries its own literals again. Two tests keep that honest: one loads the copied file the way LiteLLM does - by path, as a top-level module - so an import that only works in-package fails there, and one asserts the copied constants and the provider resolver still agree with the canonical copies in provider_usage.py, so the deliberate duplication cannot drift silently. Mutation-checked: restoring the relative import reproduces CI's exact error. 660 eval tests pass locally; the two remaining test_model_gateway.py failures are the environmental ones (litellm[proxy]'s console script is absent here, which is also why this never reproduced locally). * test(eval): import the installed callback instead of grepping it Two review findings on the same weakness, both correct. The install test asserted "class ProviderUsageLogger" appeared in the copied file's text. That passes whenever the string is present, including when the module cannot load at all - which is precisely how a package-relative import got through review here and took the proxy down. It now loads the copy the way LiteLLM does, by path as a top-level module, and checks the handler instance the config actually names. The gateway-forwarding test built its work directory with tempfile.mkdtemp(), which nothing removed, so every run left the generated config and the copied callback behind in the system temp directory. It uses the pytest-managed tmp_path fixture like its neighbours. 660 eval tests pass; the two test_model_gateway.py failures are the environmental ones. * fix(eval): record failures on the synchronous callback path too ProviderUsageLogger overrode both async hooks and the sync SUCCESS hook, but not the sync failure hook. On that path failures fell through to CustomLogger's base implementation and were never appended - so a sweep recorded its successes and quietly understated what it spent, since a failed request is billed all the same. That contradicts the module's own stated reason for handling failures at all. The failure test could not have caught it: it called _append directly, which exercises neither public hook. Both failure tests now drive the hooks LiteLLM actually calls, and a new one walks all four - sync and async, success and failure - asserting each records in order. Removing the sync failure hook fails both. 661 eval tests pass; the two test_model_gateway.py failures remain environmental. --------- Co-authored-by: Gergo Magyar --- eval/.gitignore | 1 + eval/tests/test_provider_usage.py | 132 ++++++++ eval/tests/test_provider_usage_capture.py | 313 ++++++++++++++++++ eval/workflow_bench/litellm_usage_callback.py | 136 ++++++++ eval/workflow_bench/model_gateway.py | 43 ++- eval/workflow_bench/provider_usage.py | 188 +++++++++++ 6 files changed, 812 insertions(+), 1 deletion(-) create mode 100644 eval/tests/test_provider_usage.py create mode 100644 eval/tests/test_provider_usage_capture.py create mode 100644 eval/workflow_bench/litellm_usage_callback.py create mode 100644 eval/workflow_bench/provider_usage.py diff --git a/eval/.gitignore b/eval/.gitignore index d1ac9f241..8f1814dfa 100644 --- a/eval/.gitignore +++ b/eval/.gitignore @@ -14,3 +14,4 @@ build/ # Environment .env .venv/ +.venv diff --git a/eval/tests/test_provider_usage.py b/eval/tests/test_provider_usage.py new file mode 100644 index 000000000..c8e74a966 --- /dev/null +++ b/eval/tests/test_provider_usage.py @@ -0,0 +1,132 @@ +"""The two providers' accounting equations, encoded literally. + +Adding OpenAI's cache fields to its input_tokens double-counts, because they are +subsets of it. Subtracting Anthropic's under-counts, because they are additional +categories. A single generic struct cannot be right for both, so these tests +pin each equation rather than the field names. +""" + +from __future__ import annotations + +import pytest + +from workflow_bench.provider_usage import ( + ANTHROPIC, + OPENAI_RESPONSES, + UsageSemanticsError, + normalize_usage, +) + + +def _openai(input_tokens: int, cached: int | None = None, cache_write: int | None = None) -> dict: + details: dict[str, int] = {} + if cached is not None: + details["cached_tokens"] = cached + if cache_write is not None: + details["cache_write_tokens"] = cache_write + return { + "input_tokens": input_tokens, + "input_tokens_details": details, + "output_tokens": 300, + "output_tokens_details": {"reasoning_tokens": 250}, + } + + +def test_openai_uncached_request_is_all_ordinary_input() -> None: + usage = normalize_usage(OPENAI_RESPONSES, _openai(1000, cached=0, cache_write=0)) + assert usage.ordinary_input_tokens == 1000 + assert usage.total_input_tokens == 1000 + assert (usage.cache_read_input_tokens, usage.cache_write_input_tokens) == (0, 0) + + +def test_openai_cache_creation_keeps_the_parts_summing_to_input_tokens() -> None: + """The subsets must reconstruct the whole, never exceed it.""" + + usage = normalize_usage(OPENAI_RESPONSES, _openai(1000, cached=0, cache_write=400)) + assert usage.ordinary_input_tokens == 600 + assert ( + usage.ordinary_input_tokens + + usage.cache_read_input_tokens + + usage.cache_write_input_tokens + == usage.total_input_tokens + ) + + +def test_openai_cache_hit_plus_new_write_uses_the_documented_subtraction() -> None: + usage = normalize_usage(OPENAI_RESPONSES, _openai(10_000, cached=7_000, cache_write=1_000)) + assert usage.ordinary_input_tokens == 2_000 + assert usage.total_input_tokens == 10_000, "input_tokens is the whole, not a component" + + +def test_openai_reasoning_tokens_decompose_output_rather_than_adding_to_it() -> None: + usage = normalize_usage(OPENAI_RESPONSES, _openai(100, cached=0, cache_write=0)) + assert usage.output_tokens == 300 + assert usage.reasoning_output_tokens == 250 + assert usage.reasoning_output_tokens <= usage.output_tokens + + +def test_anthropic_uncached_total_is_just_input_tokens() -> None: + usage = normalize_usage( + ANTHROPIC, + {"input_tokens": 1000, "cache_creation_input_tokens": 0, + "cache_read_input_tokens": 0, "output_tokens": 200}, + ) + assert usage.total_input_tokens == 1000 + assert usage.ordinary_input_tokens == 1000 + + +def test_anthropic_cached_total_adds_the_cache_categories() -> None: + """The opposite equation to OpenAI's, on deliberately identical numbers.""" + + usage = normalize_usage( + ANTHROPIC, + {"input_tokens": 2_000, "cache_creation_input_tokens": 1_000, + "cache_read_input_tokens": 7_000, "output_tokens": 200}, + ) + assert usage.total_input_tokens == 10_000 + assert usage.ordinary_input_tokens == 2_000 + + +def test_the_same_numbers_mean_different_totals_on_the_two_providers() -> None: + """The whole reason a shared struct is unsafe, in one assertion.""" + + openai = normalize_usage(OPENAI_RESPONSES, _openai(10_000, cached=7_000, cache_write=1_000)) + anthropic = normalize_usage( + ANTHROPIC, + {"input_tokens": 10_000, "cache_creation_input_tokens": 1_000, + "cache_read_input_tokens": 7_000, "output_tokens": 300}, + ) + assert openai.total_input_tokens == 10_000 + assert anthropic.total_input_tokens == 18_000 + assert openai.ordinary_input_tokens == 2_000 + assert anthropic.ordinary_input_tokens == 10_000 + + +def test_missing_native_cache_fields_are_unknown_and_never_zero() -> None: + """A zero we invented is indistinguishable from a zero the provider reported.""" + + usage = normalize_usage(OPENAI_RESPONSES, {"input_tokens": 1000, "output_tokens": 10}) + assert usage.cache_read_input_tokens is None + assert usage.cache_write_input_tokens is None + assert usage.ordinary_input_tokens is None, "cannot subtract what was never reported" + assert usage.total_input_tokens == 1000 + assert not usage.complete + assert "cache_read_input_tokens" in usage.unknown_fields + + +def test_an_absent_usage_object_is_entirely_unknown() -> None: + usage = normalize_usage(ANTHROPIC, None) + assert not usage.complete + assert usage.total_input_tokens is None + + +def test_an_unknown_provider_is_refused_rather_than_guessed() -> None: + with pytest.raises(UsageSemanticsError, match="refusing to guess"): + normalize_usage("some-new-provider", {"input_tokens": 1}) + + +def test_cache_subsets_larger_than_the_whole_are_rejected() -> None: + """Nonsense arithmetic must surface, not silently produce a negative.""" + + with pytest.raises(UsageSemanticsError, match="exceed input_tokens"): + normalize_usage(OPENAI_RESPONSES, _openai(100, cached=90, cache_write=50)) diff --git a/eval/tests/test_provider_usage_capture.py b/eval/tests/test_provider_usage_capture.py new file mode 100644 index 000000000..99e542e4d --- /dev/null +++ b/eval/tests/test_provider_usage_capture.py @@ -0,0 +1,313 @@ +"""What the proxy writes must outlive the translation that follows it. + +Claude Code receives an Anthropic-shaped response, which has nowhere to put +OpenAI's cached_tokens, cache_write_tokens or reasoning_tokens. If those are not +captured before the translation, the only remaining record of them is a bill. +""" + +from __future__ import annotations + +import contextlib +import json +from pathlib import Path +from types import SimpleNamespace + +import pytest + +from workflow_bench import litellm_usage_callback, provider_usage +from workflow_bench.litellm_usage_callback import USAGE_LOG_ENV_VAR, ProviderUsageLogger +from workflow_bench.model_gateway import ( + OpenAIGateway, + USAGE_CALLBACK_MODULE, + openai_litellm_config, + write_openai_litellm_config, +) +from workflow_bench.provider_usage import ( + ANTHROPIC, + OPENAI_RESPONSES, + USAGE_ENV_VARS, + normalize_usage, +) + + +class _Usage: + """Stands in for the provider usage model LiteLLM hands the callback.""" + + def __init__(self, payload: dict) -> None: + self._payload = payload + + def model_dump(self) -> dict: + return self._payload + + +def _openai_response(usage: dict) -> SimpleNamespace: + return SimpleNamespace( + id="resp_68f2c1", + # The model that actually answered, which is not the role the caller asked for. + model="gpt-5.6-sol-2026-08-01", + usage=_Usage(usage), + ) + + +NATIVE = { + "input_tokens": 48_000, + "input_tokens_details": {"cached_tokens": 44_000, "cache_write_tokens": 1_000}, + "output_tokens": 900, + "output_tokens_details": {"reasoning_tokens": 640}, +} + + +@pytest.fixture +def logged(tmp_path: Path, monkeypatch: pytest.MonkeyPatch): + log = tmp_path / "provider_usage.jsonl" + monkeypatch.setenv(USAGE_LOG_ENV_VAR, str(log)) + + def emit(usage: dict) -> dict: + ProviderUsageLogger()._append( + "success", + {"model": "claude-sonnet-4-5", "custom_llm_provider": "openai", "call_type": "responses"}, + _openai_response(usage), + 0.0, + 1.0, + ) + return json.loads(log.read_text().splitlines()[-1]) + + return emit + + +def test_native_openai_usage_survives_the_anthropic_translation(logged) -> None: + event = logged(NATIVE) + native = event["native_usage"] + # Verbatim: the fields an Anthropic-shaped response cannot carry. + assert native["input_tokens_details"]["cached_tokens"] == 44_000 + assert native["input_tokens_details"]["cache_write_tokens"] == 1_000 + assert native["output_tokens_details"]["reasoning_tokens"] == 640 + assert event["response_id"] == "resp_68f2c1" + + +def test_the_actual_model_is_recorded_separately_from_the_requested_role(logged) -> None: + """Pricing must follow what answered, not what the caller named.""" + + event = logged(NATIVE) + assert event["requested_model"] == "claude-sonnet-4-5" + assert event["actual_model"] == "gpt-5.6-sol-2026-08-01" + assert "cell_id" not in event, "a proxy-wide variable cannot identify a cell" + + +def test_the_captured_event_normalizes_with_openai_arithmetic(logged) -> None: + """Capture and normalization must agree end to end, not just in isolation.""" + + event = logged(NATIVE) + # The provider the LOG recorded, not one the test supplies - passing + # OPENAI_RESPONSES by hand here is what hid the adapter-key mismatch. + assert event["provider"] == OPENAI_RESPONSES + assert event["provider_label"] == "openai" + usage = normalize_usage(event["provider"], event["native_usage"]) + assert usage.total_input_tokens == 48_000 + assert usage.ordinary_input_tokens == 3_000 + assert usage.cache_read_input_tokens == 44_000 + assert usage.complete + + +def test_usage_without_details_normalizes_to_unknown_rather_than_zero(logged) -> None: + """The mutation the accounting must not survive: dropped details, silent zeros.""" + + stripped = {k: v for k, v in NATIVE.items() if k != "input_tokens_details"} + event = logged(stripped) + usage = normalize_usage(event["provider"], event["native_usage"]) + assert usage.cache_read_input_tokens is None + assert usage.ordinary_input_tokens is None + assert not usage.complete + + +def test_a_failed_request_is_still_accounted_for(logged, tmp_path: Path) -> None: + """The money was spent whether or not the cell produced an artifact.""" + + import asyncio + + logger = ProviderUsageLogger() + args = ({"model": "claude-sonnet-4-5"}, _openai_response(NATIVE), 0.0, 1.0) + # Every hook LiteLLM can call, not the private helper underneath them: the + # sync failure hook was missing entirely and _append could never show that. + logger.log_failure_event(*args) + asyncio.run(logger.async_log_failure_event(*args)) + events = [json.loads(line) for line in (tmp_path / "provider_usage.jsonl").read_text().splitlines()] + assert len(events) == 2, "both failure hooks must record" + assert all(e["status"] == "failure" for e in events) + + +def test_every_public_outcome_hook_records(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None: + """Overriding a subset silently drops whichever path LiteLLM actually uses.""" + + import asyncio + + monkeypatch.setenv(USAGE_LOG_ENV_VAR, str(tmp_path / "usage.jsonl")) + logger = ProviderUsageLogger() + args = ({"model": "m"}, _openai_response(NATIVE), 0.0, 1.0) + logger.log_success_event(*args) + logger.log_failure_event(*args) + asyncio.run(logger.async_log_success_event(*args)) + asyncio.run(logger.async_log_failure_event(*args)) + + events = [json.loads(line) for line in (tmp_path / "usage.jsonl").read_text().splitlines()] + assert [e["status"] for e in events] == ["success", "failure", "success", "failure"] + + +def test_the_logger_never_raises_into_the_proxy(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None: + """Accounting is evidence, not control flow.""" + + monkeypatch.setenv(USAGE_LOG_ENV_VAR, str(tmp_path / "missing-dir" / "usage.jsonl")) + ProviderUsageLogger()._append("success", {}, object(), 0.0, 1.0) + + +def test_no_log_is_written_when_the_destination_is_unset(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.delenv(USAGE_LOG_ENV_VAR, raising=False) + ProviderUsageLogger()._append("success", {}, _openai_response(NATIVE), 0.0, 1.0) + assert not list(tmp_path.iterdir()) + + +def test_the_generated_config_loads_the_callback_from_beside_itself(tmp_path: Path) -> None: + """LiteLLM resolves the dotted path relative to the config directory.""" + + config = write_openai_litellm_config(tmp_path / "litellm.yaml", ["gpt-5.6-sol"]) + assert openai_litellm_config(["gpt-5.6-sol"])["litellm_settings"]["callbacks"] == [ + f"{USAGE_CALLBACK_MODULE}.handler" + ] + installed = config.parent / f"{USAGE_CALLBACK_MODULE}.py" + assert installed.is_file(), "the proxy cannot import a callback that was never placed" + # Importing it, not grepping it: a text search passes even when the module + # cannot load, which is exactly how a package-relative import survived + # review here. This is the deployment configuration, so load it the way the + # proxy does - by path, as a top-level module. + import importlib.util + + spec = importlib.util.spec_from_file_location(USAGE_CALLBACK_MODULE, installed) + assert spec is not None and spec.loader is not None + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + assert isinstance(module.handler, module.ProviderUsageLogger) + + +def test_the_gateway_forwards_the_usage_environment_into_the_proxy( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + """The proxy is a separate process with a constructed environment. + + Popen(env=...) replaces the parent environment rather than extending it, so + a variable the callback reads is simply absent unless the gateway forwards + it by name. Without this the accounting looks configured and silently + records nothing on every request - the in-process tests above cannot see + that, because they never cross the subprocess boundary. + """ + + for name in USAGE_ENV_VARS: + monkeypatch.setenv(name, f"value-for-{name}") + captured: dict[str, dict[str, str]] = {} + + class _Popen: + def __init__(self, *_a, **kwargs): + captured["env"] = kwargs["env"] + raise RuntimeError("stop before launching a real proxy") + + # The console-script resolver runs before Popen and is absent in this + # environment (the same reason two gateway tests fail here); the argv it + # builds is not what this test is about. + monkeypatch.setattr( + "workflow_bench.model_gateway.litellm_proxy_argv", + lambda **_k: ["/bin/true"], + ) + monkeypatch.setattr("workflow_bench.model_gateway.subprocess.Popen", _Popen) + gateway = OpenAIGateway( + openai_api_key="sk-test", + model_names=["gpt-5.6-sol"], + work_dir=tmp_path, + ) + with contextlib.suppress(Exception): + gateway.__enter__() + + env = captured.get("env") + assert env is not None, "the proxy was never constructed" + for name in USAGE_ENV_VARS: + assert env.get(name) == f"value-for-{name}", f"{name} never reached the proxy" + # The credential allowlist is still an allowlist, not the parent environment. + assert "PATH" in env and len(env) < 40 + + +def test_an_unresolvable_provider_is_refused_rather_than_guessed() -> None: + """LiteLLM says "openai" for Chat Completions too, and it counts differently.""" + + from workflow_bench.provider_usage import canonical_provider + + assert canonical_provider("openai", "responses") == OPENAI_RESPONSES + assert canonical_provider("openai", "completion") is None + assert canonical_provider("openai", None) is None + assert canonical_provider("anthropic", "completion") == ANTHROPIC + + +def test_request_identity_cannot_come_from_the_proxy_environment() -> None: + """One proxy serves the whole sweep, so its environment identifies the sweep. + + attach_openai_gateway wraps all of _run_sweep, and cells run concurrently + under --workers, interleaving requests through that single process. Any + variable forwarded at launch is therefore constant for every event it ever + records. Pinned so a future change does not reintroduce a per-cell + environment variable that would silently stamp one value on every request. + """ + + assert USAGE_ENV_VARS == ( + "GITNEXUS_BENCH_PROVIDER_USAGE", + "GITNEXUS_BENCH_SWEEP_ID", + ), "a per-cell variable here would be constant across concurrent cells" + + +def test_a_request_records_its_session_so_attribution_stays_possible(logged) -> None: + """The per-request half of identity, recorded even when the provider omits it.""" + + event = logged(NATIVE) + assert "session_id" in event, "absent attribution is still a fact about the run" + + +def test_the_callback_imports_the_way_litellm_actually_loads_it(tmp_path: Path) -> None: + """By path, as a top-level module, with no parent package and no sys.path entry. + + LiteLLM resolves a dotted callback through spec_from_file_location against + the config directory, so the copied file is not part of workflow_bench when + it runs. A relative or sibling import therefore raises ImportError and the + proxy exits before becoming ready - which the in-package tests cannot see, + because they import it as workflow_bench.litellm_usage_callback. + """ + + import importlib.util + import shutil + + source = Path(litellm_usage_callback.__file__) + installed = tmp_path / f"{USAGE_CALLBACK_MODULE}.py" + shutil.copy(source, installed) + + spec = importlib.util.spec_from_file_location(USAGE_CALLBACK_MODULE, installed) + assert spec is not None and spec.loader is not None + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) # ImportError here is the proxy refusing to start + assert hasattr(module, "handler") + + +def test_the_callbacks_copied_constants_match_the_canonical_ones() -> None: + """The copies are deliberate; drifting apart silently is not. + + The callback cannot import from the package (see the test above), so it + carries its own literals. These assertions are what keep the duplication + honest. + """ + + assert litellm_usage_callback.USAGE_LOG_ENV_VAR == provider_usage.USAGE_LOG_ENV_VAR + assert litellm_usage_callback.SWEEP_ID_ENV_VAR == provider_usage.SWEEP_ID_ENV_VAR + for label, call_type in ( + ("openai", "responses"), + ("openai", "completion"), + ("openai", None), + ("anthropic", "completion"), + ("mystery", "responses"), + ): + assert litellm_usage_callback.canonical_provider(label, call_type) == provider_usage.canonical_provider( + label, call_type + ), f"resolver drifted for {label!r}/{call_type!r}" diff --git a/eval/workflow_bench/litellm_usage_callback.py b/eval/workflow_bench/litellm_usage_callback.py new file mode 100644 index 000000000..843413765 --- /dev/null +++ b/eval/workflow_bench/litellm_usage_callback.py @@ -0,0 +1,136 @@ +"""Append each upstream request's usage exactly as the provider reported it. + +This runs INSIDE the LiteLLM proxy, on the far side of the translation that +turns an OpenAI response into the Anthropic shape Claude Code expects. That is +the only point that still knows which provider served the request, what model +actually answered, and what the native usage object said before its fields were +renamed into someone else's semantics. + +Deliberately self-contained: the proxy loads this file by path from the config +directory, so it cannot assume ``workflow_bench`` is importable. Normalization +lives in workflow_bench.provider_usage and runs offline over what this writes - +the native object is the evidence, and deriving from it here would mean the +derivation could not be revisited without re-running a paid sweep. + +Never raises. A cell that fails still spent money upstream, and losing the +accounting because the log write failed would be the worse outcome. +""" + +from __future__ import annotations + +import json +import os +import threading +from typing import Any + +from litellm.integrations.custom_logger import CustomLogger + +# Literals, not imports. LiteLLM loads this file BY PATH from the config +# directory via spec_from_file_location, so it has no parent package and the +# directory is not on sys.path - a relative or sibling import raises +# ImportError and the proxy refuses to start. workflow_bench.provider_usage +# holds the canonical copies and a test asserts these agree with them, which +# catches drift without coupling at import time. +USAGE_LOG_ENV_VAR = "GITNEXUS_BENCH_PROVIDER_USAGE" +SWEEP_ID_ENV_VAR = "GITNEXUS_BENCH_SWEEP_ID" + + +def canonical_provider(label, call_type): # noqa: ANN001, ANN201 + """Adapter key for the usage shape, or None when it cannot be resolved. + + Mirrors workflow_bench.provider_usage.canonical_provider; see the note + above for why this is a copy rather than an import. + """ + + if label == "openai" and call_type and "responses" in call_type: + return "openai-responses" + if label == "anthropic": + return "anthropic" + return None + +SCHEMA_VERSION = 1 +_LOCK = threading.Lock() + + +def _plain(value: Any) -> Any: + """Provider usage arrives as pydantic models; keep the shape, drop the class.""" + + for attr in ("model_dump", "dict"): + method = getattr(value, attr, None) + if callable(method): + try: + return method() + except Exception: + pass + if isinstance(value, dict): + return value + return None + + +class ProviderUsageLogger(CustomLogger): + async def async_log_success_event(self, kwargs, response_obj, start_time, end_time) -> None: # noqa: ANN001 + self._append("success", kwargs, response_obj, start_time, end_time) + + async def async_log_failure_event(self, kwargs, response_obj, start_time, end_time) -> None: # noqa: ANN001 + # Failed requests are billed too, and a sweep that only accounts for + # successes understates what it spent. + self._append("failure", kwargs, response_obj, start_time, end_time) + + def log_success_event(self, kwargs, response_obj, start_time, end_time) -> None: # noqa: ANN001 + self._append("success", kwargs, response_obj, start_time, end_time) + + def log_failure_event(self, kwargs, response_obj, start_time, end_time) -> None: # noqa: ANN001 + # The synchronous counterpart. Overriding only the success hook here + # recorded successes and let failures fall through to the base class, + # which accounts for nothing - and a failed request is still billed, so + # a sweep missing them understates what it spent. + self._append("failure", kwargs, response_obj, start_time, end_time) + + def _append(self, status, kwargs, response_obj, start_time, end_time) -> None: # noqa: ANN001 + path = os.environ.get(USAGE_LOG_ENV_VAR) + if not path: + return + try: + params = kwargs.get("litellm_params") or {} + call_type = kwargs.get("call_type") + provider_label = kwargs.get("custom_llm_provider") or params.get("custom_llm_provider") + metadata = params.get("metadata") or {} + event = { + "schema_version": SCHEMA_VERSION, + "status": status, + # Identity. The REQUESTED model is the caller's role name and the + # ACTUAL model is what answered; pricing must follow the second, + # because several roles map onto one upstream model here. + "requested_model": kwargs.get("model"), + "actual_model": getattr(response_obj, "model", None), + # Two fields, because they answer different questions. The raw + # label is what LiteLLM said; "provider" is the adapter key, + # which needs the call type too - LiteLLM reports "openai" for + # both Chat Completions and Responses and those report usage + # differently. Unresolvable stays None so normalize_usage + # refuses rather than guessing token semantics. + "provider_label": provider_label, + "provider": canonical_provider(provider_label, call_type), + "response_id": getattr(response_obj, "id", None), + "call_type": call_type, + "sweep_id": os.environ.get(SWEEP_ID_ENV_VAR), + # The per-request half of identity, and the only thing that can + # attribute a request to a cell: one proxy serves the whole + # sweep, so anything read from the environment is the same for + # every event. Recorded even when absent, because knowing the + # attribution is unavailable is itself a fact about the run. + "session_id": metadata.get("litellm_session_id") or metadata.get("session_id"), + "started_at": str(start_time), + "completed_at": str(end_time), + # Verbatim. Not flattened, not renamed, not summed. + "native_usage": _plain(getattr(response_obj, "usage", None)), + } + line = json.dumps(event, default=str) + "\n" + with _LOCK, open(path, "a", encoding="utf-8") as handle: + handle.write(line) + except Exception: + # Accounting is evidence, not control flow: never take the sweep down. + return + + +handler = ProviderUsageLogger() diff --git a/eval/workflow_bench/model_gateway.py b/eval/workflow_bench/model_gateway.py index e44da55ec..e798eb14a 100644 --- a/eval/workflow_bench/model_gateway.py +++ b/eval/workflow_bench/model_gateway.py @@ -31,6 +31,8 @@ from typing import Any import yaml +from .provider_usage import USAGE_ENV_VARS + ANTHROPIC_API_KEY_ENV = "GITNEXUS_BENCH_ANTHROPIC_API_KEY" LEGACY_ANTHROPIC_API_KEY_ENV = "GITNEXUS_BENCH_AUTH_TOKEN" OPENAI_API_KEY_ENV = "GITNEXUS_BENCH_OPENAI_API_KEY" @@ -173,6 +175,9 @@ def resolve_model_access( return ModelAccess(start_proxy=False) +USAGE_CALLBACK_MODULE = "provider_usage_callback" + + def openai_litellm_config(model_names: Sequence[str]) -> dict[str, Any]: seen: list[str] = [] for name in model_names: @@ -196,7 +201,15 @@ def openai_litellm_config(model_names: Sequence[str]) -> dict[str, Any]: } for name in seen ], - "litellm_settings": {"request_timeout": GATEWAY_REQUEST_TIMEOUT_S}, + "litellm_settings": { + "request_timeout": GATEWAY_REQUEST_TIMEOUT_S, + # Captures each upstream request's usage as the provider reported + # it, before translation renames OpenAI's fields into Anthropic's + # shape and loses which arithmetic applies. Resolved by LiteLLM + # relative to the config directory, which is why the module is + # copied next to the config rather than imported from the package. + "callbacks": [f"{USAGE_CALLBACK_MODULE}.handler"], + }, "general_settings": {"master_key": "os.environ/LITELLM_MASTER_KEY"}, } @@ -204,9 +217,26 @@ def openai_litellm_config(model_names: Sequence[str]) -> dict[str, Any]: def write_openai_litellm_config(path: Path, model_names: Sequence[str]) -> Path: path.write_text(yaml.safe_dump(openai_litellm_config(model_names), sort_keys=False)) path.chmod(0o600) + _install_usage_callback(path.parent) return path +def _install_usage_callback(config_dir: Path) -> Path: + """Place the usage logger where LiteLLM resolves callbacks from. + + LiteLLM loads a dotted callback path as a file relative to the config + directory before falling back to a package import, and the proxy runs as + its own process that need not have this package on sys.path. Copying the + one module is what makes the callback resolvable in both cases. + """ + + source = Path(__file__).with_name("litellm_usage_callback.py") + destination = config_dir / f"{USAGE_CALLBACK_MODULE}.py" + destination.write_text(source.read_text()) + destination.chmod(0o600) + return destination + + def _free_loopback_port() -> int: with socket.socket(socket.AF_INET, socket.SOCK_STREAM) as sock: sock.bind(("127.0.0.1", 0)) @@ -302,6 +332,17 @@ class OpenAIGateway(AbstractContextManager["OpenAIGateway"]): "OPENAI_API_KEY": self.openai_api_key, "LITELLM_MASTER_KEY": self.auth_token, } + # The proxy is a separate process and Popen(env=...) REPLACES the + # parent environment rather than extending it, so anything the usage + # callback reads has to be forwarded by name. Without this the callback + # loads, finds no destination, and returns silently on every request - + # the accounting looks configured and records nothing. Forwarded + # individually rather than by inheriting the environment, because the + # allowlist above is the gateway's credential boundary. + for name in USAGE_ENV_VARS: + value = os.environ.get(name) + if value: + env[name] = value if os.name == "nt": # Windows subprocess DLL/socket initialization needs SystemRoot. # Keep the rest of the gateway's credential boundary explicit. diff --git a/eval/workflow_bench/provider_usage.py b/eval/workflow_bench/provider_usage.py new file mode 100644 index 000000000..782eb18fa --- /dev/null +++ b/eval/workflow_bench/provider_usage.py @@ -0,0 +1,188 @@ +"""Per-request usage as the provider reported it, plus a derived cross-provider view. + +The benchmark has been reading token counts out of Claude Code's session output, +which is Anthropic-shaped whatever actually served the request. That works until +the upstream is OpenAI, because the two providers do not merely name their fields +differently - they mean opposite things by them: + + Anthropic: total_input = input_tokens + + cache_creation_input_tokens + + cache_read_input_tokens + (input_tokens is only the UNCACHED remainder; cache fields ADD) + + OpenAI: total_input = input_tokens + ordinary = input_tokens - cached_tokens - cache_write_tokens + (input_tokens is the WHOLE; cache fields are SUBSETS) + +Adding OpenAI's three together double-counts; subtracting Anthropic's +under-counts. So the native object is authoritative and is stored verbatim, and +the normalized view is derived from it per provider. + +The second rule is that a field nobody reported is UNKNOWN, not zero. A stored +``cache_read = 0`` previously could mean either "the provider said zero" or "our +adapter never looked", and those two must never be written identically again: +the first says caching is not working, the second says we cannot tell. +""" + +from __future__ import annotations + +from collections.abc import Mapping +from dataclasses import dataclass +from typing import Any + +SCHEMA_VERSION = 1 + +# Read by the in-proxy callback and forwarded by the gateway that launches it. +# Defined here because this module is pure stdlib: model_gateway can import the +# name without importing litellm, which only the callback needs. +# +# Both are SWEEP-scoped, and that is a constraint rather than an oversight. +# attach_openai_gateway wraps the whole sweep (runner.py), so one proxy serves +# every cell and its environment is fixed for that proxy's lifetime - while +# cells run concurrently under --workers and interleave requests through it. An +# environment variable therefore cannot carry a per-cell identity: it would +# record one constant against every event. Attributing a request to a cell +# needs an identifier that travels WITH the request; see the session fields the +# callback records for the intended hook. +USAGE_LOG_ENV_VAR = "GITNEXUS_BENCH_PROVIDER_USAGE" +SWEEP_ID_ENV_VAR = "GITNEXUS_BENCH_SWEEP_ID" +USAGE_ENV_VARS = (USAGE_LOG_ENV_VAR, SWEEP_ID_ENV_VAR) + +ANTHROPIC = "anthropic" +OPENAI_RESPONSES = "openai-responses" + + +class UsageSemanticsError(ValueError): + """The native usage object does not satisfy its own provider's arithmetic.""" + + +@dataclass(frozen=True) +class NormalizedUsage: + """Cross-provider view. ``None`` means the provider did not report it. + + Deliberately not defaulted to 0: see the module docstring. Every consumer + that sums these has to decide what to do about unknown, and making it None + forces that decision to be explicit instead of silently counting zero. + """ + + ordinary_input_tokens: int | None + cache_read_input_tokens: int | None + cache_write_input_tokens: int | None + total_input_tokens: int | None + output_tokens: int | None + reasoning_output_tokens: int | None + + @property + def complete(self) -> bool: + return all( + value is not None + for value in ( + self.ordinary_input_tokens, + self.cache_read_input_tokens, + self.cache_write_input_tokens, + self.total_input_tokens, + self.output_tokens, + ) + ) + + @property + def unknown_fields(self) -> tuple[str, ...]: + return tuple( + name for name, value in sorted(vars(self).items()) if value is None + ) + + +def _int_or_none(source: Mapping[str, Any] | None, key: str) -> int | None: + """Absent, null, or non-numeric all read as unknown rather than zero.""" + + if not isinstance(source, Mapping): + return None + value = source.get(key) + if isinstance(value, bool) or not isinstance(value, int): + return None + return value + + +def _normalize_openai_responses(usage: Mapping[str, Any]) -> NormalizedUsage: + """input_tokens is the WHOLE; cached and cache-write are subsets of it.""" + + total = _int_or_none(usage, "input_tokens") + details = usage.get("input_tokens_details") + cache_read = _int_or_none(details, "cached_tokens") + cache_write = _int_or_none(details, "cache_write_tokens") + output_details = usage.get("output_tokens_details") + + ordinary: int | None = None + if total is not None and cache_read is not None and cache_write is not None: + ordinary = total - cache_read - cache_write + if ordinary < 0: + raise UsageSemanticsError( + f"OpenAI cached ({cache_read}) + cache_write ({cache_write}) " + f"exceed input_tokens ({total})" + ) + return NormalizedUsage( + ordinary_input_tokens=ordinary, + cache_read_input_tokens=cache_read, + cache_write_input_tokens=cache_write, + total_input_tokens=total, + output_tokens=_int_or_none(usage, "output_tokens"), + # A decomposition of output_tokens, not an addition to it. + reasoning_output_tokens=_int_or_none(output_details, "reasoning_tokens"), + ) + + +def _normalize_anthropic(usage: Mapping[str, Any]) -> NormalizedUsage: + """input_tokens is the uncached REMAINDER; the cache fields add to it.""" + + ordinary = _int_or_none(usage, "input_tokens") + cache_read = _int_or_none(usage, "cache_read_input_tokens") + cache_write = _int_or_none(usage, "cache_creation_input_tokens") + + total: int | None = None + if ordinary is not None and cache_read is not None and cache_write is not None: + total = ordinary + cache_read + cache_write + return NormalizedUsage( + ordinary_input_tokens=ordinary, + cache_read_input_tokens=cache_read, + cache_write_input_tokens=cache_write, + total_input_tokens=total, + output_tokens=_int_or_none(usage, "output_tokens"), + reasoning_output_tokens=None, + ) + + +def canonical_provider(label: str | None, call_type: str | None) -> str | None: + """Map LiteLLM's provider label onto an adapter key, or None if unsure. + + LiteLLM reports ``custom_llm_provider`` as "openai" for both Chat + Completions and Responses, and those two report usage differently, so the + label alone cannot pick an adapter. The call type is what distinguishes + them. Returning None when it does not is deliberate: normalize_usage + refuses an unknown provider rather than guessing token semantics, which is + the whole point of keeping the native object authoritative. + """ + + if label == "openai" and call_type and "responses" in call_type: + return OPENAI_RESPONSES + if label in _ADAPTERS: + return label + return None + + +_ADAPTERS = { + ANTHROPIC: _normalize_anthropic, + OPENAI_RESPONSES: _normalize_openai_responses, +} + + +def normalize_usage(provider: str, native_usage: Mapping[str, Any] | None) -> NormalizedUsage: + """Derive the cross-provider view. Never mutates or replaces the native object.""" + + adapter = _ADAPTERS.get(provider) + if adapter is None: + raise UsageSemanticsError( + f"no usage adapter for provider {provider!r}; refusing to guess its token semantics" + ) + if not isinstance(native_usage, Mapping): + return NormalizedUsage(None, None, None, None, None, None) + return adapter(native_usage) From 376ed3bb4abda80f7a57cf963822002b5304f2f9 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Gerg=C5=91=20Magyar?= Date: Tue, 8 Sep 2026 19:06:03 +0100 Subject: [PATCH 17/17] perf(lock): probe this process's own start time once (#3222) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * perf(lock): probe this process's own start time once `acquireFileLock` stamps the owner file with the acquiring process's start time so a later reclaimer can tell a live owner from pid reuse. That value cannot change while we are running, but it was re-probed on every acquisition — and on Windows the probe is a `powershell.exe` spawn plus a `Get-CimInstance Win32_Process` WMI query, which is the single most expensive step in taking an uncontended lock. Add `readProcessStartTimeCached` and make it the default reader in `acquireFileLock` and `resolveWatchDeps`. Only this process's own pid is cached: - A foreign pid is always re-probed. That process can exit and its pid be reused, which is precisely what the stamp exists to detect. - A failed probe is not cached. `acquireFileLock` throws when the start time is empty, so caching one transient failure would leave the process unable to take a lock for the rest of its life. `readProcessStartTime` itself is unchanged and still probes every call, so the existing timezone-pinning regression test keeps exercising the real `ps` invocation instead of passing off a cached value. Behavior is otherwise identical: same probe, same string, same stamp. Co-Authored-By: Claude Opus 5 (1M context) * chore(autofix): apply prettier + eslint fixes via /autofix command --------- Co-authored-by: Gergo Magyar Co-authored-by: Claude Opus 5 (1M context) Co-authored-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> --- gitnexus/src/core/auto-sync/starter.ts | 4 +- gitnexus/src/storage/file-lock.ts | 8 ++-- gitnexus/src/utils/process-identity.ts | 19 +++++++++ gitnexus/test/unit/process-identity.test.ts | 45 +++++++++++++++++++++ 4 files changed, 71 insertions(+), 5 deletions(-) diff --git a/gitnexus/src/core/auto-sync/starter.ts b/gitnexus/src/core/auto-sync/starter.ts index 624e092ec..7d67dfa64 100644 --- a/gitnexus/src/core/auto-sync/starter.ts +++ b/gitnexus/src/core/auto-sync/starter.ts @@ -4,7 +4,7 @@ import path from 'node:path'; import { execFileSync } from 'node:child_process'; import { acquireFileLock, FileLockBusyError } from '../../storage/file-lock.js'; import { getGlobalDir } from '../../storage/repo-manager.js'; -import { isProcessAlive, readProcessStartTime } from '../../utils/process-identity.js'; +import { isProcessAlive, readProcessStartTimeCached } from '../../utils/process-identity.js'; import { loadAutoSyncConfig } from './config.js'; import { runAutoSyncOnce } from './runner.js'; import { getAutoSyncMutexPath, getAutoSyncWatchDir } from './state.js'; @@ -632,7 +632,7 @@ function resolveWatchDeps(deps: Partial = {}): AutoSyn return undefined; } }), - readProcessStartTime: deps.readProcessStartTime ?? readProcessStartTime, + readProcessStartTime: deps.readProcessStartTime ?? readProcessStartTimeCached, sleep: deps.sleep ?? ((ms) => diff --git a/gitnexus/src/storage/file-lock.ts b/gitnexus/src/storage/file-lock.ts index a857d6a47..2b2494ef8 100644 --- a/gitnexus/src/storage/file-lock.ts +++ b/gitnexus/src/storage/file-lock.ts @@ -3,7 +3,7 @@ import fs from 'node:fs/promises'; import os from 'node:os'; import path from 'node:path'; import { setTimeout as sleep } from 'node:timers/promises'; -import { isProcessAlive, readProcessStartTime } from '../utils/process-identity.js'; +import { isProcessAlive, readProcessStartTimeCached } from '../utils/process-identity.js'; const HOSTNAME = os.hostname(); @@ -46,7 +46,9 @@ export async function acquireFileLock( pid, ownerId: crypto.randomUUID(), processStartTime: - options.processStartTime ?? (options.readProcessStartTime ?? readProcessStartTime)(pid) ?? '', + options.processStartTime ?? + (options.readProcessStartTime ?? readProcessStartTimeCached)(pid) ?? + '', hostname: options.hostname ?? HOSTNAME, }; if (!owner.processStartTime) { @@ -69,7 +71,7 @@ export async function acquireFileLock( resolvedPath, owner, options.isProcessAlive ?? isProcessAlive, - options.readProcessStartTime ?? readProcessStartTime, + options.readProcessStartTime ?? readProcessStartTimeCached, ) ) { continue; diff --git a/gitnexus/src/utils/process-identity.ts b/gitnexus/src/utils/process-identity.ts index e7c99439c..8c00eab08 100644 --- a/gitnexus/src/utils/process-identity.ts +++ b/gitnexus/src/utils/process-identity.ts @@ -38,3 +38,22 @@ export function readProcessStartTime(pid: number): string | undefined { return undefined; } } + +let ownStartTime: string | undefined; + +/** + * `readProcessStartTime`, except this process's own start time is probed once. + * It cannot change while we are running, and every `acquireFileLock` — plus + * each retry attempt and each stale-lock reclaim guard — stamps the owner file + * with it. On Windows that probe is a `powershell.exe` spawn and a WMI query, + * so a process taking several locks pays it several times for one constant. + * + * A foreign pid is never cached: that process can exit and its pid can be + * reused, which is the very thing the stamp exists to detect. A failed probe + * is not cached either — one transient failure would otherwise leave the + * process unable to take a lock for its whole lifetime. + */ +export function readProcessStartTimeCached(pid: number): string | undefined { + if (pid !== process.pid) return readProcessStartTime(pid); + return (ownStartTime ??= readProcessStartTime(pid)); +} diff --git a/gitnexus/test/unit/process-identity.test.ts b/gitnexus/test/unit/process-identity.test.ts index 7b31f1d32..de3436882 100644 --- a/gitnexus/test/unit/process-identity.test.ts +++ b/gitnexus/test/unit/process-identity.test.ts @@ -4,8 +4,24 @@ import { isProcessAlive, readProcessStartTime } from '../../src/utils/process-id afterEach(() => { vi.restoreAllMocks(); + vi.doUnmock('node:child_process'); + vi.resetModules(); }); +/** + * Loads a fresh copy of the module (fresh memo) over a counted `execFileSync`, + * so "how many times did we actually shell out" is observable. `doMock` is not + * hoisted, so the statically imported functions used by the other tests keep + * the real implementation. + */ +async function withCountedProbe(probe: () => string) { + const execFileSync = vi.fn(probe); + vi.doMock('node:child_process', () => ({ execFileSync })); + vi.resetModules(); + const identity = await import('../../src/utils/process-identity.js'); + return { execFileSync, readProcessStartTimeCached: identity.readProcessStartTimeCached }; +} + describe('process identity', () => { it('treats only ESRCH as a dead process', () => { const kill = vi.spyOn(process, 'kill'); @@ -39,4 +55,33 @@ describe('process identity', () => { } }, ); + + it('probes this process once and re-probes a foreign pid every time', async () => { + const { execFileSync, readProcessStartTimeCached } = await withCountedProbe(() => 'STAMP\n'); + + expect(readProcessStartTimeCached(process.pid)).toBe('STAMP'); + expect(readProcessStartTimeCached(process.pid)).toBe('STAMP'); + // On Windows each probe is a powershell.exe spawn plus a WMI query. + expect(execFileSync).toHaveBeenCalledTimes(1); + + // A foreign process can exit and its pid be reused — caching that stamp + // would blind the reuse check the stamp exists for. + expect(readProcessStartTimeCached(process.pid + 1)).toBe('STAMP'); + expect(readProcessStartTimeCached(process.pid + 1)).toBe('STAMP'); + expect(execFileSync).toHaveBeenCalledTimes(3); + }); + + it('retries after a failed self probe instead of caching the failure', async () => { + const { execFileSync, readProcessStartTimeCached } = await withCountedProbe(() => 'STAMP\n'); + execFileSync.mockImplementationOnce(() => { + throw new Error('probe unavailable'); + }); + + // A cached failure would leave acquireFileLock throwing "Unable to + // determine process start time" for the rest of the process's life. + expect(readProcessStartTimeCached(process.pid)).toBeUndefined(); + expect(readProcessStartTimeCached(process.pid)).toBe('STAMP'); + expect(readProcessStartTimeCached(process.pid)).toBe('STAMP'); + expect(execFileSync).toHaveBeenCalledTimes(2); + }); });