diff --git a/.agents/plugins/marketplace.json b/.agents/plugins/marketplace.json index 944455615..5a130e4c6 100644 --- a/.agents/plugins/marketplace.json +++ b/.agents/plugins/marketplace.json @@ -6,7 +6,7 @@ "plugins": [ { "name": "gitnexus", - "version": "1.6.11", + "version": "1.6.12", "source": { "source": "local", "path": "./gitnexus-claude-plugin" diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json index 57073ec59..8ced2c56f 100644 --- a/.claude-plugin/marketplace.json +++ b/.claude-plugin/marketplace.json @@ -11,7 +11,7 @@ "plugins": [ { "name": "gitnexus", - "version": "1.6.11", + "version": "1.6.12", "source": "./gitnexus-claude-plugin", "description": "Code intelligence powered by a knowledge graph. Provides execution flow tracing, blast radius analysis, and augmented search across your codebase." } diff --git a/.claude/skills/gitnexus-cli/SKILL.md b/.claude/skills/gitnexus-cli/SKILL.md index 09c7af0d2..3a00ab546 100644 --- a/.claude/skills/gitnexus-cli/SKILL.md +++ b/.claude/skills/gitnexus-cli/SKILL.md @@ -34,6 +34,18 @@ Run from the project root. This parses all source files, builds the knowledge gr For Spring runtime enrichment, pass a JSON bundle, one endpoint JSON file, or a directory containing endpoint files. Route evidence is authoritative only when `runtimeConfirmed === true`; `runtimeSource` records provenance and may also accompany `handler-conflict`. Env/configprops values are never persisted. +## Index storage and retention + +Default location is `/.gitnexus/`. Override with environment variables (also documented in README): + +| Env | Effect | +| --- | ------ | +| `GITNEXUS_STORAGE_PATH` | One complete external index directory. Wins if both storage vars are set. | +| `GITNEXUS_STORAGE_ROOT` | Absolute root; GitNexus creates an isolated `-<12-hex>/` slot per repository. | +| `GITNEXUS_CONTENT_RETENTION` | `full` (default) keeps file text; `symbol` keeps snippets; `none` keeps the graph only. | + +`list_repos`, `gitnexus://repo/{name}/context`, and HTTP `GET /api/repos` / `GET /api/repo` expose `storagePath`, `contentRetention`, and `sourceAvailable`. HTTP `/api/file` and `/api/grep` return 410 unless retention is `full`. MCP `include_content` may still return symbol spans when retention is `symbol`. + Use `node .gitnexus/run.cjs analyze --watch` for a long-lived local Git repository. It performs an initial analysis, queues scanner-admitted file changes, and retries intact failed batches with bounded backoff. Watch refreshes update only the graph: they skip AGENTS.md / CLAUDE.md injection and standard skill installation, so run a one-shot `analyze` when those generated files need updating. Watch rejects one-shot or context-output flags including `--force`, embedding flags, `--skills`, `--default-branch`, `--skip-agents-md`, `--skip-skills`, `--no-stats`, `--self-commit`, `--index-only`, and `--skip-git`. It never pulls remotes. Scheduled remote clone/pull is a different command: `gitnexus auto-sync`. Bare `gitnexus watch` is reserved and does not start either job. Running MCP and `serve` processes periodically check for a published replacement and reopen it without a restart. MCP checks are throttled to once every five seconds, so a tool call before the next check can briefly use the previous index. ### status — Check index freshness diff --git a/.claude/skills/gitnexus-guide/SKILL.md b/.claude/skills/gitnexus-guide/SKILL.md index e52560422..bf3c73948 100644 --- a/.claude/skills/gitnexus-guide/SKILL.md +++ b/.claude/skills/gitnexus-guide/SKILL.md @@ -83,15 +83,23 @@ Notes: `offset` ≥ `total` returns an empty page (with `total` still reported). ### Inline staleness signal (`query` / `context` / `impact` / `cypher`) -These four hot read tools attach a non-blocking `staleness` field to their response when the index is behind the checkout's current HEAD — the same `{ commitsBehind, hint }` shape `list_repos` already reports — so a direct tool call surfaces a behind-HEAD index without a separate `list_repos` call: +These four hot read tools attach a non-blocking `staleness` field to their response when the index is not at the checkout's current HEAD — the same `{ status, commitsBehind?, hint? }` shape `list_repos` already reports — so a direct tool call surfaces a stale index without a separate `list_repos` call: ```jsonc { /* …the tool's normal result… */ - "staleness": { "commitsBehind": 3, "hint": "⚠️ Index is 3 commits behind HEAD. Run analyze tool to update." } + "staleness": { "status": "behind", "commitsBehind": 3, "hint": "⚠️ Index is 3 commits behind HEAD. Run analyze tool to update." } } ``` -The field is **absent when the index is current** (or when the freshness check can't run), so its presence is the signal. It is only ever added to object results — raw-array `cypher` output and error envelopes are returned unchanged. `@group`-targeted calls do not carry it (multi-repo staleness is ill-defined). When you see it, the graph may be behind the working tree — re-run `analyze` before trusting blast-radius or dependence answers. +`commitsBehind` is present only when git counted the gap. When git could not count it but HEAD still resolves to a commit other than the indexed one — usually because the indexed commit is no longer in the clone's history — the index is provably not at HEAD with no countable gap, so no number is reported: + +```jsonc +{ /* …the tool's normal result… */ + "staleness": { "status": "diverged", "hint": "⚠️ Index is not at HEAD and the commit gap could not be counted — the recorded commit may no longer be in this clone's history. Run analyze tool to update." } +} +``` + +The field is **absent when the index is current**, and these four tools also omit it when the freshness check could not run at all — that case is `status: "unknown"`, which only the `list_repos` listing reports. So its presence means the status is not `current`: read `status` before using `commitsBehind`. It is only ever added to object results — raw-array `cypher` output and error envelopes are returned unchanged. `@group`-targeted calls do not carry it (multi-repo staleness is ill-defined). When you see it, the graph may be behind the working tree — re-run `analyze` before trusting blast-radius or dependence answers. ### Taint findings (`explain`) diff --git a/.gitattributes b/.gitattributes index eeb1a0976..15fad158a 100644 --- a/.gitattributes +++ b/.gitattributes @@ -15,6 +15,7 @@ *.so binary *.dll binary *.dylib binary +*.lbug_extension binary # TypeScript sources are always text for diff purposes. Git's binary # heuristic fires when EITHER blob in a pair carries a NUL, so a source diff --git a/.github/dependabot.yml b/.github/dependabot.yml index 4a25e682d..535204a38 100644 --- a/.github/dependabot.yml +++ b/.github/dependabot.yml @@ -89,6 +89,10 @@ updates: # tree-sitter-cli follows the runtime's version cadence. Bump when # regenerating vendor/tree-sitter-proto/src/parser.c, not on a schedule. - dependency-name: tree-sitter-cli + # Pin @ladybugdb/core so a daily bump cannot ship a skewed FTS artifact. + # The extension version is a separate upstream constant, not derivable + # from the core version (see vendor/lbug-fts/manifest.json). + - dependency-name: '@ladybugdb/core' # gitnexus-web (thin frontend client). - package-ecosystem: npm diff --git a/.github/scripts/check-tree-sitter-upgrade-readiness.py b/.github/scripts/check-tree-sitter-upgrade-readiness.py index c1ee3c92d..30201ab57 100644 --- a/.github/scripts/check-tree-sitter-upgrade-readiness.py +++ b/.github/scripts/check-tree-sitter-upgrade-readiness.py @@ -68,6 +68,7 @@ GRAMMARS: dict[str, tuple[str, str, str]] = { "tree-sitter-typescript": ("tree-sitter/tree-sitter-typescript", "master", "typescript/src/parser.c"), # Vendored parsers — kept here so the upstream coords for drift # detection are co-located with every other grammar's coords. + "tree-sitter-objc": ("tree-sitter-grammars/tree-sitter-objc", "master", "src/parser.c"), "tree-sitter-proto": ("coder3101/tree-sitter-proto", "main", "src/parser.c"), "tree-sitter-zig": ("tree-sitter-grammars/tree-sitter-zig", "master", "src/parser.c"), } diff --git a/.github/scripts/fetch-lbug-fts-artifacts.mjs b/.github/scripts/fetch-lbug-fts-artifacts.mjs new file mode 100644 index 000000000..8ad916ed3 --- /dev/null +++ b/.github/scripts/fetch-lbug-fts-artifacts.mjs @@ -0,0 +1,144 @@ +#!/usr/bin/env node +/** + * Fetch Ladybug FTS artifacts into gitnexus/vendor/lbug-fts/prebuilds/. + * + * Lives outside the published package (`files` includes `scripts` wholesale). + * Reads versions, filename, and tuple→upstream-platform mapping from + * vendor/lbug-fts/manifest.json so the gate and runtime cannot drift. + * + * Usage: node .github/scripts/fetch-lbug-fts-artifacts.mjs + */ +import { createHash } from 'node:crypto'; +import { existsSync, mkdirSync, readFileSync, writeFileSync } from 'node:fs'; +import path from 'node:path'; +import { fileURLToPath, pathToFileURL } from 'node:url'; + +const REPO_ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..', '..'); +const VENDOR = path.join(REPO_ROOT, 'gitnexus', 'vendor', 'lbug-fts'); +const PREBUILDS = path.join(VENDOR, 'prebuilds'); +const MANIFEST_PATH = path.join(VENDOR, 'manifest.json'); + +/** Only the Ladybug official extension host — never a manifest-supplied origin. */ +const OFFICIAL_REPO = 'https://extension.ladybugdb.com/'; +const EXACT_VERSION = /^\d+\.\d+\.\d+$/; +const SAFE_UPSTREAM = /^(linux_amd64|linux_arm64|osx_amd64|osx_arm64|win_amd64)$/; + +/** + * Build the official artifact URL from allowlisted fields only. + * `officialRepo` in the manifest must match {@link OFFICIAL_REPO}; the + * origin itself is a constant so an edited manifest cannot redirect the fetch. + */ +export function officialArtifactUrl(manifest, upstreamPlatform) { + const officialRepo = String(manifest?.officialRepo ?? ''); + if (officialRepo !== OFFICIAL_REPO) { + throw new Error(`refusing unofficial FTS repo: '${officialRepo}'`); + } + const version = String(manifest?.extensionVersion ?? ''); + if (!EXACT_VERSION.test(version)) { + throw new Error(`unsafe extensionVersion: '${version}'`); + } + if (!SAFE_UPSTREAM.test(String(upstreamPlatform ?? ''))) { + throw new Error(`unsafe upstream platform: '${upstreamPlatform}'`); + } + const filename = String(manifest?.filename ?? ''); + if (!SAFE_FILENAME.test(filename)) { + throw new Error(`unsafe FTS artifact filename: '${filename}'`); + } + return `${OFFICIAL_REPO}v${version}/${upstreamPlatform}/fts/${filename}`; +} + +const sha256 = (buf) => createHash('sha256').update(buf).digest('hex'); + +const readExistingHash = (filePath) => { + if (!existsSync(filePath)) return null; + return sha256(readFileSync(filePath)); +}; + +export const supportedTuples = (manifest) => manifest.tuples.map((entry) => entry.tuple); + +const SAFE_TUPLE = /^(darwin|linux|win32)-(x64|arm64)$/; +const SAFE_FILENAME = /^[\w.-]+\.lbug_extension$/; + +/** Relative-path containment — not a prefix match (rejects `prebuilds-evil`). */ +const isPathInsideRoot = (root, candidate) => { + const relative = path.relative(root, candidate); + if (path.isAbsolute(relative)) return false; + return relative !== '' && !relative.startsWith(`..${path.sep}`) && relative !== '..'; +}; + +export function assertSafeArtifactDest({ prebuildsDir, tuple, filename }) { + if (!SAFE_TUPLE.test(String(tuple ?? ''))) { + throw new Error( + `unsafe FTS artifact tuple: '${tuple}' (expected (darwin|linux|win32)-(x64|arm64))`, + ); + } + if (!SAFE_FILENAME.test(String(filename ?? ''))) { + throw new Error(`unsafe FTS artifact filename: '${filename}' (expected *.lbug_extension)`); + } + const dest = path.join(prebuildsDir, tuple, filename); + if (!isPathInsideRoot(prebuildsDir, dest)) { + throw new Error(`FTS artifact dest is not inside prebuildsDir: ${dest}`); + } + return dest; +} + +async function fetchBuffer(url) { + // codeql[js/request-forgery] — origin is OFFICIAL_REPO; path segments are allowlisted. + // lgtm[js/request-forgery] + // codeql[js/file-access-to-http] — versions/platforms are regex-pinned, not raw file bytes. + const res = await fetch(url, { signal: AbortSignal.timeout(120_000) }); + if (!res.ok) { + throw new Error(`GET ${url} → ${res.status} ${res.statusText}`); + } + return Buffer.from(await res.arrayBuffer()); +} + +const writeAllowlistedArtifact = (prebuildsDir, dest, buf) => { + if (!isPathInsideRoot(prebuildsDir, dest)) { + throw new Error(`FTS artifact dest is not inside prebuildsDir: ${dest}`); + } + // codeql[js/http-to-file-access] — dest is assertSafeArtifactDest + containment-checked. + writeFileSync(dest, buf); +}; + +export async function refreshArtifacts({ + manifest = JSON.parse(readFileSync(MANIFEST_PATH, 'utf8')), + prebuildsDir = PREBUILDS, + download = fetchBuffer, +} = {}) { + mkdirSync(prebuildsDir, { recursive: true }); + const lines = []; + for (const { tuple, upstreamPlatform } of manifest.tuples) { + const dest = assertSafeArtifactDest({ + prebuildsDir, + tuple, + filename: manifest.filename, + }); + mkdirSync(path.dirname(dest), { recursive: true }); + const url = officialArtifactUrl(manifest, upstreamPlatform); + const previousHash = readExistingHash(dest); + const previousSize = previousHash ? readFileSync(dest).byteLength : 0; + const buf = await download(url); + const nextHash = sha256(buf); + writeAllowlistedArtifact(prebuildsDir, dest, buf); + const changed = previousHash !== nextHash; + console.log( + changed + ? `[fts-fetch] ${tuple}: ${previousHash ?? '(new)'} (${previousSize} B) → ${nextHash} (${buf.byteLength} B)` + : `[fts-fetch] ${tuple}: unchanged ${nextHash} (${buf.byteLength} B)`, + ); + lines.push(`${nextHash} ./${tuple}/${manifest.filename}`); + } + lines.sort(); + writeFileSync(path.join(prebuildsDir, 'SHA256SUMS'), `${lines.join('\n')}\n`); + return lines; +} + +const invokedDirectly = + process.argv[1] && pathToFileURL(path.resolve(process.argv[1])).href === import.meta.url; +if (invokedDirectly) { + refreshArtifacts().catch((err) => { + console.error(`[fts-fetch] ${err instanceof Error ? err.message : err}`); + process.exit(1); + }); +} diff --git a/.github/scripts/test_check_tree_sitter_upgrade_readiness.py b/.github/scripts/test_check_tree_sitter_upgrade_readiness.py index 5013b3549..f1b4baf97 100644 --- a/.github/scripts/test_check_tree_sitter_upgrade_readiness.py +++ b/.github/scripts/test_check_tree_sitter_upgrade_readiness.py @@ -9,8 +9,8 @@ which is deliberately dependency-free so it runs on any vanilla runner. Run with (pytest also discovers ``unittest.TestCase`` classes, so a future pytest CI job picks these up unchanged.) -These tests lock in the #858 fix: the 6 vendored grammars -(c/swift/kotlin/dart/proto/zig) are classified from the shared manifest +These tests lock in the #858 fix: the 7 vendored grammars +(c/swift/kotlin/dart/objc/proto/zig) are classified from the shared manifest (.github/vendored-grammars.json), their ABI is read from gitnexus/vendor/, and the report never renders a bare ``?`` placeholder. All network is mocked. """ @@ -192,7 +192,7 @@ class AssertCurrent(TestCase): def test_assert_current_is_network_free_and_passes(self): report, code = self._run_assert_current() # raises if any urlopen fires self.assertEqual(code, 0) - # All 6 vendored grammars are introspected from the repo (ABI 14), not skipped. + # All 7 vendored grammars are introspected from the repo (ABI 14), not skipped. for name in readiness.VENDORED_NAMES: self.assertIn(f"{name}: vendored ABI", report) @@ -365,15 +365,16 @@ class ReportRendering(TestCase): # Counts are derived from _render_report()'s mock corpus (all npm peer # deps mocked permissive): of the 10 npm-installed grammars, 9 render # Ready and 1 — tree-sitter-cpp — is the intentional pin (#1242), so it is - # not counted ready. The 3 blockers are that same pinned tree-sitter-cpp - # plus two held vendored grammars: ABI-held tree-sitter-c (#1242/#858) and + # not counted ready. The 4 blockers are that same pinned tree-sitter-cpp + # plus three held vendored grammars: ABI-held tree-sitter-c (#1242/#858), # tree-sitter-kotlin (pinned to an unreleased fwcd main commit for `fun # interface` support — ABI 14 is in range, but a hold counts as a blocker - # until it is lifted). If a grammar is added/removed or a pin/hold changes, + # until it is lifted), and tree-sitter-objc. If a grammar is added/removed + # or a pin/hold changes, # update _render_report()'s mock AND these expected counts together; a # mismatch here means the report prose drifted, not the regex. self.assertEqual(ready.groups(), ("9", "10")) - self.assertEqual(blockers.group(1), "3") + self.assertEqual(blockers.group(1), "4") def _matrix_row(self, name: str) -> str: for line in self.report.splitlines(): diff --git a/.github/scripts/update-vendored-grammars.mjs b/.github/scripts/update-vendored-grammars.mjs index 769310bd7..0554d3253 100644 --- a/.github/scripts/update-vendored-grammars.mjs +++ b/.github/scripts/update-vendored-grammars.mjs @@ -21,7 +21,7 @@ * node update-vendored-grammars.mjs # detect only → JSON report on stdout * node update-vendored-grammars.mjs --apply X # re-vendor grammar X in place * - * tree-sitter-c is MONITORED but report-only (`hold`): it is ABI-pinned at 0.21.4 + * tree-sitter-c and tree-sitter-objc are MONITORED but report-only (`hold`): c is ABI-pinned at 0.21.4 * (#1242/#858) and must not auto-bump without a tree-sitter runtime upgrade, so an * available c update is detected + reported but never auto-applied — even if it is * ABI-13/14. A maintainer re-vendors it deliberately. diff --git a/.github/scripts/verify-workflow-run-pr-identity.cjs b/.github/scripts/verify-workflow-run-pr-identity.cjs new file mode 100644 index 000000000..c638f92a6 --- /dev/null +++ b/.github/scripts/verify-workflow-run-pr-identity.cjs @@ -0,0 +1,274 @@ +// Resolve the open PR for a trusted workflow_run consumer. +// +// Shared by commit-fork-prebuilds.yml and pr-autofix-publish.yml. +// workflow_run.pull_requests[] is empty on fork PRs, and +// GET /repos/{base}/commits/{sha}/pulls is also empty because the fork head +// commit is not in the base repo's commit graph. The authoritative lookup is +// GET /repos/{base}/pulls?head={owner}:{branch}&state=open using +// workflow_run.head_repository + workflow_run.head_branch (server-controlled). +// That same query works for same-repo PRs (owner is the base repo owner). +// +// The current PR tip may have moved past the SHA the producer built; that is +// not an identity failure — the caller decides whether to lease-push or just +// comment. Two open PRs from the same fork head (same owner:branch into this +// repo) are an identity failure: artifact pr_number is untrusted and must not +// pick among them. Set SCHEMA_PATTERN to the artifact schema allowlist +// (defaults to the tree-sitter prebuild schema). +'use strict'; + +const fs = require('node:fs'); +const { spawnSync } = require('node:child_process'); + +const SCHEMA_PATTERN = /^gitnexus\.ts-prebuild\/v[0-9]+$/; +const IDENTITY_PATTERNS = { + pr_number: /^[0-9]+$/, + head_sha: /^[0-9a-f]{40}$/, + head_ref: /^[A-Za-z0-9._/-]+$/, + repo: /^[A-Za-z0-9._-]+\/[A-Za-z0-9._-]+$/, +}; + +function allowlistField(key, value, pattern) { + const text = value == null ? '' : String(value); + if (!text || !pattern.test(text)) { + throw new Error(`metadata.${key} failed allowlist (got: ${JSON.stringify(text)})`); + } + return text; +} + +function forkHeadOwner(headRepo) { + const slash = headRepo.indexOf('/'); + if (slash <= 0 || slash === headRepo.length - 1) { + throw new Error(`head_repo must be owner/name (got: ${JSON.stringify(headRepo)})`); + } + return headRepo.slice(0, slash); +} + +function compileSchemaPattern(value) { + if (value instanceof RegExp) return value; + if (typeof value === 'string' && value.length > 0) { + try { + return new RegExp(value); + } catch { + throw new Error('SCHEMA_PATTERN is not a valid regular expression'); + } + } + return SCHEMA_PATTERN; +} + +function allowlistMetadata(raw, schemaPattern) { + const parsed = typeof raw === 'string' ? JSON.parse(raw) : raw; + if (!parsed || typeof parsed !== 'object' || Array.isArray(parsed)) { + throw new Error('metadata.json must be an object'); + } + return { + schema: allowlistField('schema', parsed.schema, compileSchemaPattern(schemaPattern)), + pr_number: allowlistField('pr_number', parsed.pr_number, IDENTITY_PATTERNS.pr_number), + head_sha: allowlistField('head_sha', parsed.head_sha, IDENTITY_PATTERNS.head_sha), + head_ref: allowlistField('head_ref', parsed.head_ref, IDENTITY_PATTERNS.head_ref), + head_repo: allowlistField('head_repo', parsed.head_repo, IDENTITY_PATTERNS.repo), + base_repo: allowlistField('base_repo', parsed.base_repo, IDENTITY_PATTERNS.repo), + }; +} + +function allowlistAuthority(authority) { + return { + head_sha: allowlistField('head_sha', authority.head_sha, IDENTITY_PATTERNS.head_sha), + head_repo: allowlistField('head_repo', authority.head_repo, IDENTITY_PATTERNS.repo), + head_branch: allowlistField('head_ref', authority.head_branch, IDENTITY_PATTERNS.head_ref), + base_repo: allowlistField('base_repo', authority.base_repo, IDENTITY_PATTERNS.repo), + }; +} + +function verifyArtifactAgainstWorkflowRun(meta, authority) { + if (meta.head_sha !== authority.head_sha) { + throw new Error( + `Artifact head_sha (${meta.head_sha}) != workflow_run.head_sha (${authority.head_sha}) — refusing.`, + ); + } + if (meta.head_repo !== authority.head_repo) { + throw new Error( + `Artifact head_repo (${meta.head_repo}) != workflow_run.head_repository (${authority.head_repo}) — refusing.`, + ); + } + if (meta.base_repo !== authority.base_repo) { + throw new Error('Artifact base_repo does not match $GITHUB_REPOSITORY — refusing.'); + } + if (meta.head_ref !== authority.head_branch) { + throw new Error( + `Artifact head_ref (${meta.head_ref}) != workflow_run.head_branch (${authority.head_branch}) — refusing.`, + ); + } +} + +function matchOpenPullsFromForkHead(pulls, { headRepo, headBranch, baseRepo }) { + if (!Array.isArray(pulls)) { + throw new Error('GitHub pulls?head= lookup returned a non-array'); + } + return pulls.filter((pr) => { + return ( + pr && + pr.state === 'open' && + Number.isInteger(pr.number) && + pr.head && + pr.head.repo && + pr.head.repo.full_name === headRepo && + pr.head.ref === headBranch && + pr.base && + pr.base.repo && + pr.base.repo.full_name === baseRepo + ); + }); +} + +function resolveVerifiedPullRequest({ meta, authority, pulls, schemaPattern }) { + const cleanMeta = allowlistMetadata(meta, schemaPattern); + const cleanAuthority = allowlistAuthority(authority); + verifyArtifactAgainstWorkflowRun(cleanMeta, cleanAuthority); + + const matched = matchOpenPullsFromForkHead(pulls, { + headRepo: cleanAuthority.head_repo, + headBranch: cleanAuthority.head_branch, + baseRepo: cleanAuthority.base_repo, + }); + + if (matched.length === 0) { + throw new Error( + `No open PR from ${cleanAuthority.head_repo}:${cleanAuthority.head_branch} targeting ${cleanAuthority.base_repo} — refusing.`, + ); + } + + // Artifact pr_number is untrusted. Do not use it to pick among several open + // PRs that share this fork head (same owner:branch into this repo, different + // base branches). Fail closed unless GitHub-controlled fields leave exactly one. + if (matched.length !== 1) { + throw new Error( + `Ambiguous open PRs from ${cleanAuthority.head_repo}:${cleanAuthority.head_branch} targeting ${cleanAuthority.base_repo} (${matched + .map((pr) => pr.number) + .join(',')}) — refusing.`, + ); + } + + const chosen = matched[0]; + const expected = Number(cleanMeta.pr_number); + if (chosen.number !== expected) { + throw new Error( + `Artifact pr_number (${cleanMeta.pr_number}) is not the open PR(s) from this fork head (${chosen.number}) — refusing.`, + ); + } + + const currentHeadSha = typeof chosen.head.sha === 'string' ? chosen.head.sha : ''; + return { + pr_number: String(chosen.number), + head_ref: cleanAuthority.head_branch, + head_sha: cleanAuthority.head_sha, + head_repo: cleanAuthority.head_repo, + current_head_sha: currentHeadSha, + branch_moved: Boolean(currentHeadSha && currentHeadSha !== cleanAuthority.head_sha), + }; +} + +function flattenGhListPages(parsed) { + if (!Array.isArray(parsed)) { + throw new Error('GitHub pulls?head= lookup returned a non-array'); + } + if (parsed.length === 0) return parsed; + if (parsed.every((page) => Array.isArray(page))) { + return parsed.flat(); + } + return parsed; +} + +function listOpenPullsByHead({ ghRepo, headOwner, headBranch, runGh }) { + const run = runGh || ((args) => spawnSync('gh', args, { encoding: 'utf8' })); + const result = run([ + 'api', + '--paginate', + '--slurp', + '-X', + 'GET', + `repos/${ghRepo}/pulls`, + '-f', + 'state=open', + '-f', + `head=${headOwner}:${headBranch}`, + ]); + if (result.status !== 0) { + const err = (result.stderr || result.stdout || '').trim(); + throw new Error(`GitHub pulls?head= lookup failed: ${err || `exit ${result.status}`}`); + } + const stdout = (result.stdout || '').trim(); + if (!stdout) { + throw new Error('GitHub pulls?head= lookup returned an empty body'); + } + let parsed; + try { + parsed = JSON.parse(stdout); + } catch { + throw new Error('GitHub pulls?head= lookup returned non-JSON'); + } + return flattenGhListPages(parsed); +} + +function main() { + const schemaPattern = compileSchemaPattern(process.env.SCHEMA_PATTERN); + const raw = fs.readFileSync(process.env.META_PATH, 'utf8'); + const meta = allowlistMetadata(raw, schemaPattern); + const authority = allowlistAuthority({ + head_sha: process.env.WF_HEAD_SHA, + head_repo: process.env.WF_HEAD_REPO, + head_branch: process.env.WF_HEAD_BRANCH, + base_repo: process.env.GH_REPO, + }); + const pulls = listOpenPullsByHead({ + ghRepo: authority.base_repo, + headOwner: forkHeadOwner(authority.head_repo), + headBranch: authority.head_branch, + }); + const verified = resolveVerifiedPullRequest({ meta, authority, pulls, schemaPattern }); + if (verified.branch_moved) { + console.log( + `PR head moved to ${verified.current_head_sha}; delivering against built SHA ${verified.head_sha} (lease will refuse if the branch moved).`, + ); + } + console.log( + `Verified identity: PR=${verified.pr_number} head_sha=${verified.head_sha} head_repo=${verified.head_repo} head_ref=${verified.head_ref}.`, + ); + const out = process.env.GITHUB_OUTPUT; + if (!out) { + throw new Error('GITHUB_OUTPUT is unset'); + } + fs.appendFileSync( + out, + [ + `pr_number=${verified.pr_number}`, + `head_ref=${verified.head_ref}`, + `head_sha=${verified.head_sha}`, + `head_repo=${verified.head_repo}`, + ].join('\n') + '\n', + ); +} + +if (require.main === module) { + try { + main(); + } catch (err) { + console.error(`::error::${err instanceof Error ? err.message : String(err)}`); + process.exit(1); + } +} + +module.exports = { + SCHEMA_PATTERN, + IDENTITY_PATTERNS, + allowlistField, + compileSchemaPattern, + allowlistMetadata, + allowlistAuthority, + forkHeadOwner, + verifyArtifactAgainstWorkflowRun, + matchOpenPullsFromForkHead, + flattenGhListPages, + resolveVerifiedPullRequest, + listOpenPullsByHead, + main, +}; diff --git a/.github/vendored-grammars.json b/.github/vendored-grammars.json index bb4b18820..d42314aa2 100644 --- a/.github/vendored-grammars.json +++ b/.github/vendored-grammars.json @@ -6,6 +6,11 @@ "upstream": { "npm": "tree-sitter-c" }, "hold": "ABI-pinned at 0.21.4 (#1242/#858) — needs a tree-sitter runtime upgrade before bumping" }, + "objc": { + "name": "tree-sitter-objc", + "upstream": { "npm": "tree-sitter-objc" }, + "hold": "Pinned at 3.0.2 for the Objective-C provider MVP; carries darwin/linux arm64+x64 prebuilds compatible with the current tree-sitter runtime (linux-arm64 built from vendored source because the upstream npm artifact is mislabeled)" + }, "swift": { "name": "tree-sitter-swift", "upstream": { "npm": "tree-sitter-swift" } diff --git a/.github/workflows/build-tree-sitter-prebuilds.yml b/.github/workflows/build-tree-sitter-prebuilds.yml index ef64a64b8..b592ebe16 100644 --- a/.github/workflows/build-tree-sitter-prebuilds.yml +++ b/.github/workflows/build-tree-sitter-prebuilds.yml @@ -7,7 +7,7 @@ name: Build tree-sitter prebuilds # # Grammars covered here (the at-risk set — everything else already ships 6 # upstream prebuilds AND stays dependency-review-tracked, so it is left alone). -# All five are vendored under gitnexus/vendor/; `kind` (below) only picks where +# All seven are vendored under gitnexus/vendor/; `kind` (below) only picks where # the build job fetches the C source to compile: # - tree-sitter-c (vendored prebuild-only; built from the published npm # package — closes upstream's 4/6 ARM gap #2116 for a @@ -17,6 +17,8 @@ name: Build tree-sitter prebuilds # - tree-sitter-kotlin (vendored source; built from gitnexus/vendor/ — pinned to # an unreleased main commit for `fun interface` support # (#169) that no npm release carries yet) +# - tree-sitter-objc (vendored source; built from gitnexus/vendor/ — pinned +# for the Objective-C provider MVP) # - tree-sitter-swift (vendored source; built from gitnexus/vendor/ — its # prebuilds were originally upstream-shipped, now # GitNexus-cross-built like the rest for uniformity) @@ -24,13 +26,13 @@ name: Build tree-sitter prebuilds # off npm optionalDependency so `npm i -g gitnexus` # no longer warns on peerOptional tree-sitter@^0.22.1. # Upstream linux-arm64 prebuild is a mispackaged -# x86-64 binary; this workflow rebuilds all six.) +# x86-64 binary; this workflow rebuilds all seven.) # # Output: gitnexus/vendor//prebuilds//.node for # all 6 targets ({linux,darwin,win32}-{x64,arm64}). tree-sitter grammars are # N-API, so one ABI-stable .node per platform-arch works across all Node majors. # -# COST DISCIPLINE — this is a HEAVY native matrix (up to 3 grammars x 6 runners, +# COST DISCIPLINE — this is a HEAVY native matrix (up to 7 grammars x 6 runners, # incl. macOS + arm64). It is DELIBERATELY NOT wired into normal PR/push CI. It # runs only: # 1. on manual dispatch (workflow_dispatch); or @@ -61,7 +63,7 @@ on: workflow_dispatch: inputs: grammars: - description: 'Comma-separated grammar shortnames to build (c,dart,proto,kotlin,swift,zig), or "all".' + description: 'Comma-separated grammar shortnames to build (c,dart,proto,kotlin,objc,swift,zig), or "all".' required: false type: string default: 'all' @@ -93,7 +95,7 @@ on: - '!gitnexus/vendor/tree-sitter-*/prebuilds/**' # Self-test: re-run the guard if a future grammar pin is reintroduced in # the main package.json (optionalDependencies fallback). No-op otherwise — - # all six grammars are now fully vendored (kotlin and zig included). + # all seven grammars are now fully vendored (including kotlin, objc, and zig). - 'gitnexus/package.json' # Self-test: re-run the guard (normally a no-op) when the recipe changes. - '.github/workflows/build-tree-sitter-prebuilds.yml' @@ -163,6 +165,9 @@ jobs: // unreleased main commit for `fun interface` support (#169) that no // npm release carries yet — so it must build from the vendored source. kotlin: { name: 'tree-sitter-kotlin', kind: 'vendored' }, + // Objective-C is vendored WITH its source and its native bindings + // must be recut together with the pinned grammar snapshot. + objc: { name: 'tree-sitter-objc', kind: 'vendored' }, // swift is vendored WITH its source (parser.c/scanner.c/binding.gyp), // so it builds from gitnexus/vendor/ like dart/proto. Its prebuilds // were originally upstream-shipped; rebuilding them here unifies it. @@ -505,6 +510,7 @@ jobs: dart: "void main() { print(\"hi\"); }", proto: "syntax = \"proto3\";\nmessage M { int32 id = 1; }", kotlin: "fun main() { println(\"hi\") }", + objc: "@interface GNValidationProbe : NSObject\n@end", swift: "func greet() { print(\"hi\") }", zig: "pub fn main() void {}", }; diff --git a/.github/workflows/ci-tests.yml b/.github/workflows/ci-tests.yml index 7b469948c..641df1821 100644 --- a/.github/workflows/ci-tests.yml +++ b/.github/workflows/ci-tests.yml @@ -599,6 +599,26 @@ jobs: run: node --import tsx bench/parse-dispatch-rounds/measure.mjs --check working-directory: gitnexus + - name: Python workspace import-scan guards (#3254) + if: ${{ !cancelled() }} + # Build-free: same baseline approach as parse-dispatch-rounds — + # exact link/lookalike floors plus a fingerprint, then ratio timing + # only (scan scaling and from-token prefilter advantage). See + # bench/python-workspace-import-scan/measure.mjs. + run: node --import tsx bench/python-workspace-import-scan/measure.mjs --check + working-directory: gitnexus + + - name: MCP tools/list countRepos vs listRepos guards (#3259, #3184) + if: ${{ !cancelled() }} + # Build-free: exact registry cardinality + tool-roster + schema-flag + # floors, then ratio timing only (countRepos/listRepos and + # listTools/listRepos). No millisecond ceiling — this repo has + # already been bitten by a fixed ms budget. Isolated GITNEXUS_HOME; + # fixture is N real git repos so listRepos pays rev-list. See + # bench/mcp-tools-list/measure.mjs. + run: node --import tsx bench/mcp-tools-list/measure.mjs --check + working-directory: gitnexus + - name: C++ qualified-namespace resolution guards (#2788) if: ${{ !cancelled() }} # Build-free: asserts resolveCppQualifiedNamespaceMember resolves an @@ -718,6 +738,13 @@ jobs: run: node --import tsx bench/kotlin-import-target/measure.mjs --check working-directory: gitnexus + - name: Ruby gem-boundary correctness + scaling guards (#3096) + if: ${{ !cancelled() }} + # Includes real manifest loading; checks scoped resolution and scaling + # as sibling projects or declared gem counts grow independently. + run: node --import tsx bench/ruby-gem-resolution/measure.mjs --check + working-directory: gitnexus + - name: Receiver-resolution drop guards if: ${{ !cancelled() }} # NOT build-free: this one runs the real pipeline, so it needs dist/ @@ -764,6 +791,28 @@ jobs: run: node --import tsx bench/zig-cross-file-resolution/measure.mjs --check working-directory: gitnexus + - name: Objective-C workspace resolution guards (#3179) + if: ${{ !cancelled() }} + # Build-free: fingerprints spread (typed self/super/sibling) and + # protocol-candidate evidence, and gates linear file-count scaling + # of emitPostResolutionEdges. Import lookup is the shared + # import-target `objc` arm; this is the C#/Zig analog for the + # workspace message-send pass. + run: node --import tsx bench/objective-c-resolution/measure.mjs --check + working-directory: gitnexus + + - name: Callable-value reference resolution guards (#3399) + if: ${{ !cancelled() }} + # Build-free: pins the resolved-target SET of `resolveValueRefTarget` + # (exact site/resolved/declined counts plus an order-independent + # fingerprint) and asserts its per-site cost stays independent of + # workspace size across a 4x file-count step. The pass resolves a + # qualified receiver through `scopes.qualifiedNames`, a workspace-wide + # index: keyed it is O(1) per site, scanned it is O(files) — a + # regression a fixture cannot see and a 257-file binding table can. + run: node --import tsx bench/value-ref-resolution/measure.mjs --check + working-directory: gitnexus + - name: CFG construction time / disk / memory guards (#2081 M1) if: ${{ !cancelled() }} # Build-free: asserts collectFunctionCfgs output is unchanged @@ -803,11 +852,13 @@ jobs: # guards they hold never run in the main coverage job. env: GITNEXUS_BENCH: '1' + GITNEXUS_WORKER_READY_TIMEOUT_MS: '60000' run: >- npx vitest run --no-file-parallelism test/integration/cobol-pipeline-benchmark.test.ts test/integration/csharp-pipeline-benchmark.test.ts test/integration/csharp-razor-view-components-benchmark.test.ts + test/integration/objective-c-pipeline-benchmark.test.ts test/integration/cpp-adl-benchmark.test.ts test/integration/data-route-table-benchmark.test.ts test/integration/instance-ownership-pipeline-benchmark.test.ts @@ -846,6 +897,11 @@ jobs: timeout-minutes: 20 env: GITNEXUS_REQUIRE_BWRAP_CANARY: '1' + # This job installs bubblewrap, the pinned runtime and a built GitNexus, + # so the offline sweep runs here with nothing provisioning-stubbed: real + # containment, real mounts, real graph. A missing piece fails the job + # rather than silently falling back to the stubbed path. + GITNEXUS_REQUIRE_FULL_SWEEP: '1' GITNEXUS_REQUIRE_CLAUDE_CANARY: '1' steps: - uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0 @@ -904,7 +960,9 @@ jobs: tests/test_process_control.py tests/test_proposer_sandbox.py tests/test_workflow_bench_sessions.py - tests/test_ce_plugin_runtime.py -q + tests/test_ce_plugin_runtime.py + tests/test_offline_sweep_integration.py + tests/test_mock_provider.py -q working-directory: eval # Native Windows Job Object canary. POSIX-only tests skip by platform, while diff --git a/.github/workflows/codeql.yml b/.github/workflows/codeql.yml index b3103030e..c7c85ba03 100644 --- a/.github/workflows/codeql.yml +++ b/.github/workflows/codeql.yml @@ -77,6 +77,14 @@ jobs: # GitHub PR CodeQL gate, so this file is excluded to avoid # re-filing js/regex-injection on every push of the same line. - 'gitnexus/src/server/grep-params.ts' + # Tests construct tmpdir fixtures and pass them into production + # read-only probes (openSync(..., 'r')). CodeQL models that as + # js/insecure-temporary-file even though nothing is created. + - '**/test/**' + # CI vendor fetch: origin is the official Ladybug repo; dest is + # regex-pinned and containment-checked. Inline suppressions do + # not clear the PR CodeQL gate (same as grep-params.ts). + - '.github/scripts/fetch-lbug-fts-artifacts.mjs' - name: Perform CodeQL Analysis uses: github/codeql-action/analyze@cdf488f595d80d6e07e03d4674febd5ab45fa938 # v4.37.9 diff --git a/.github/workflows/commit-fork-prebuilds.yml b/.github/workflows/commit-fork-prebuilds.yml index 950402e35..3b6ef78d3 100644 --- a/.github/workflows/commit-fork-prebuilds.yml +++ b/.github/workflows/commit-fork-prebuilds.yml @@ -135,68 +135,53 @@ jobs: # workflow_run event. The allowlist above only proves the fields are # well-formed — not that they refer to the PR/SHA that actually triggered # us. A fork-controlled build could mutate metadata.json to reference - # another PR/SHA and redirect our write-scoped push. Authority sources are - # all server-controlled: workflow_run.head_sha, head_repository.full_name, - # and pull_requests[].number (empty on forks -> commits/{sha}/pulls). + # another PR/SHA and redirect our write-scoped push. + # + # This job's `if:` already restricts to forks, so pull_requests[] is empty + # by design and GET /repos/{base}/commits/{sha}/pulls is also empty (the + # fork commit is not in the base graph). Authority is workflow_run.head_sha + # + head_repository.full_name + head_branch, resolved via + # pulls?head={owner}:{branch}. The script comes from THIS default-branch + # checkout (the same trust anchor as this workflow file). + - name: Checkout identity verifier + if: steps.meta.outputs.deliver == 'true' + uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0 + with: + persist-credentials: false + sparse-checkout: .github/scripts/verify-workflow-run-pr-identity.cjs + sparse-checkout-cone-mode: false + path: trusted + - name: Verify metadata against workflow_run authority + id: verify if: steps.meta.outputs.deliver == 'true' env: GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} GH_REPO: ${{ github.repository }} - META_PR_NUMBER: ${{ steps.meta.outputs.pr_number }} - META_HEAD_SHA: ${{ steps.meta.outputs.head_sha }} - META_HEAD_REPO: ${{ steps.meta.outputs.head_repo }} + META_PATH: meta-in/metadata.json + SCHEMA_PATTERN: '^gitnexus\.ts-prebuild/v[0-9]+$' WF_HEAD_SHA: ${{ github.event.workflow_run.head_sha }} WF_HEAD_REPO: ${{ github.event.workflow_run.head_repository.full_name }} - WF_PR_NUMBERS: ${{ toJSON(github.event.workflow_run.pull_requests.*.number) }} + WF_HEAD_BRANCH: ${{ github.event.workflow_run.head_branch }} shell: bash - run: | - set -euo pipefail - - # 1) head_sha must match exactly — the commit GitHub ran the producer against. - if [ "${META_HEAD_SHA}" != "${WF_HEAD_SHA}" ]; then - echo "::error::Artifact head_sha (${META_HEAD_SHA}) != workflow_run.head_sha (${WF_HEAD_SHA}) — refusing." - exit 1 - fi - # 2) head_repo must match exactly. - if [ "${META_HEAD_REPO}" != "${WF_HEAD_REPO}" ]; then - echo "::error::Artifact head_repo (${META_HEAD_REPO}) != workflow_run.head_repository (${WF_HEAD_REPO}) — refusing." - exit 1 - fi - # 3) pr_number must reference an open PR with this head SHA. Forks have - # an empty pull_requests[] by design — fall back to commits/{sha}/pulls. - allowed_numbers=$(jq -c '.' <<< "${WF_PR_NUMBERS}") - if [ "${allowed_numbers}" = "[]" ]; then - echo "workflow_run.pull_requests empty (fork) — using commits/{sha}/pulls." - allowed_numbers=$(gh api "repos/${GH_REPO}/commits/${WF_HEAD_SHA}/pulls" \ - --jq '[.[] | select(.state == "open") | .number]' 2>/dev/null || echo "[]") - if [ "${allowed_numbers}" = "[]" ]; then - echo "::error::No open PR for head ${WF_HEAD_SHA} — refusing." - exit 1 - fi - fi - if ! jq -e --argjson n "${META_PR_NUMBER}" 'index($n) != null' <<< "${allowed_numbers}" >/dev/null; then - echo "::error::Artifact pr_number (${META_PR_NUMBER}) not in authoritative list (${allowed_numbers}) — refusing." - exit 1 - fi - echo "Verified identity: PR=${META_PR_NUMBER} head_sha=${META_HEAD_SHA} head_repo=${META_HEAD_REPO}." + run: node trusted/.github/scripts/verify-workflow-run-pr-identity.cjs # Pinned to v6.0.3 (same SHA used by build-tree-sitter-prebuilds.yml). # persist-credentials: false — push auth is provided inline at push time, # never written to .git/config on disk. - name: Checkout fork PR head - if: steps.meta.outputs.deliver == 'true' + if: steps.meta.outputs.deliver == 'true' && steps.verify.outcome == 'success' uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0 with: - repository: ${{ steps.meta.outputs.head_repo }} - ref: ${{ steps.meta.outputs.head_sha }} + repository: ${{ steps.verify.outputs.head_repo }} + ref: ${{ steps.verify.outputs.head_sha }} token: ${{ secrets.GITHUB_TOKEN }} persist-credentials: false fetch-depth: 0 path: pr-checkout - name: Place prebuilds into the fork checkout - if: steps.meta.outputs.deliver == 'true' + if: steps.meta.outputs.deliver == 'true' && steps.verify.outcome == 'success' env: DL: prebuilds-in CHECKOUT: pr-checkout @@ -238,12 +223,12 @@ jobs: - name: Commit and push to the fork branch id: push - if: steps.meta.outputs.deliver == 'true' + if: steps.meta.outputs.deliver == 'true' && steps.verify.outcome == 'success' working-directory: pr-checkout env: - HEAD_REF: ${{ steps.meta.outputs.head_ref }} - HEAD_REPO: ${{ steps.meta.outputs.head_repo }} - HEAD_SHA: ${{ steps.meta.outputs.head_sha }} + HEAD_REF: ${{ steps.verify.outputs.head_ref }} + HEAD_REPO: ${{ steps.verify.outputs.head_repo }} + HEAD_SHA: ${{ steps.verify.outputs.head_sha }} # Push auth only — supplied via env, never interpolated into the command. GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }} shell: bash @@ -303,11 +288,11 @@ jobs: fi - name: Comment delivery outcome - if: always() && steps.meta.outputs.deliver == 'true' && steps.push.outcome != 'skipped' + if: always() && steps.meta.outputs.deliver == 'true' && steps.verify.outcome == 'success' && steps.push.outcome != 'skipped' env: GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} GH_REPO: ${{ github.repository }} - PR: ${{ steps.meta.outputs.pr_number }} + PR: ${{ steps.verify.outputs.pr_number }} RESULT: ${{ steps.push.outputs.result }} RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} shell: bash diff --git a/.github/workflows/docker.yml b/.github/workflows/docker.yml index dac6a7398..0334531c1 100644 --- a/.github/workflows/docker.yml +++ b/.github/workflows/docker.yml @@ -138,7 +138,7 @@ jobs: # Required for multi-platform (linux/arm64) emulation. - name: Set up QEMU - uses: docker/setup-qemu-action@96fe6ef7f33517b61c61be40b68a1882f3264fb8 # v4.2.0 + uses: docker/setup-qemu-action@1f40c72289eff860ee54a304f1438e3cff362e0a # v4.3.0 - name: Set up Docker Buildx uses: docker/setup-buildx-action@37fe631027851001ddb9b187196cc803df7f5f0e # v4.3.0 diff --git a/.github/workflows/gitnexus-skill-evolution.yml b/.github/workflows/gitnexus-skill-evolution.yml index c004bb515..22f0bdfcc 100644 --- a/.github/workflows/gitnexus-skill-evolution.yml +++ b/.github/workflows/gitnexus-skill-evolution.yml @@ -63,13 +63,17 @@ # uploads, and a promotion (if any) opens a well-formed PR. Run # 29907431284 (2026-07-22) went green end to end in 14h45m and reached a # gate decision (`insufficient_evidence`, no promotion). -# [ ] After resizing the runner, prove a manual workers=3 run has zero excluded -# runs and does not stretch the 48-minute serial mean toward the session -# ceiling; then set GITNEXUS_EVOLUTION_WORKERS=3 and +# [ ] Confirm a workers=3 dispatch has zero excluded runs (review sessions in +# 33962002890 averaged ~19m serial, well under the 90m session ceiling). +# Then set GITNEXUS_EVOLUTION_WORKERS=3 and # GITNEXUS_EVOLUTION_ENABLED=true for scheduled runs. Scheduled runs -# require both values, so leaving workers unset/1 is an immediate rollback; -# workflow_dispatch remains available for the proof and bills real API -# usage on GITNEXUS_BENCH_ANTHROPIC_API_KEY or GITNEXUS_BENCH_OPENAI_API_KEY. +# require both values, so leaving the var unset is an immediate rollback. +# Dispatch defaults to 3; pass workers=1 only to debug a contended host. +# Weekly generations reuse matching incumbent/CE cells from the previous +# artifact so the paid matrix is the new candidate, not a 54-cell replay. +# Wall clock is quantised by ceil(cells_per_task / workers), and a review +# task is 9 cells cold, so 4 costs host contention for exactly the wall +# clock of 3. The next step up that buys anything is 5 (3 waves -> 2). name: GitNexus skill evolution on: @@ -92,9 +96,9 @@ on: default: '3' type: string workers: - description: 'Benchmark cells of one task to run at once — raise only to match the runner’s vCPUs' + description: 'Benchmark cells of one task to run at once — 3 fits the evolution box; drop to 1 only if siblings hit the session ceiling' required: false - default: '1' + default: '3' type: string model: description: 'Model for the benchmark arms (match the model your skill users run)' @@ -172,6 +176,13 @@ jobs: # stops the runner just disappears mid-step. Scheduled runs can start well # after the cron (the 2026-08-01 run was queued 65min late), so the job # budget has to absorb that delay and still land inside the uptime window. + # A Friday workflow_dispatch on a box that already booted for Saturday's + # cron inherits leftover uptime, not a fresh 24h. Run 33962002890 started + # Friday 10:57 UTC and vanished at the Saturday 03:00 stop — 51 finished + # sessions never uploaded. run-evolution.sh therefore passes + # --max-runtime-from-instance-window, and the CLI derives its cap from + # /proc/uptime at startup, so the sweep fails in-process and this always() + # upload still runs. timeout-minutes: 1260 permissions: contents: read # The promotion PR uses a short-lived App token minted below. diff --git a/.github/workflows/pr-autofix-publish.yml b/.github/workflows/pr-autofix-publish.yml index 08ad1d60f..0fffad2ff 100644 --- a/.github/workflows/pr-autofix-publish.yml +++ b/.github/workflows/pr-autofix-publish.yml @@ -121,64 +121,35 @@ jobs: # metadata.json to reference another PR or SHA, redirecting our # write-scoped sticky/check-run onto an attacker-chosen target. # - # Authority sources are all server-controlled GitHub event fields: - # - workflow_run.head_sha - # - workflow_run.head_repository.full_name - # - workflow_run.pull_requests[].number (within-repo PRs only; - # empty array on fork PRs — fall back to commits/{sha}/pulls) + # Authority is workflow_run.head_sha + head_repository.full_name + + # head_branch, resolved via pulls?head={owner}:{branch}. That query + # works for same-repo PRs and forks; commits/{sha}/pulls is empty + # for fork SHAs. The script comes from THIS default-branch checkout + # (the same trust anchor as this workflow file). # + # Always verify — including the changed_lines=0 path — so the + # check-run SHA cannot be an unverified artifact field. # Mismatch => fail loud BEFORE any sticky/check-run side effect. + - name: Checkout identity verifier + uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0 + with: + persist-credentials: false + sparse-checkout: .github/scripts/verify-workflow-run-pr-identity.cjs + sparse-checkout-cone-mode: false + path: trusted + - name: Verify metadata against workflow_run authority id: verify - if: steps.meta.outputs.changed_lines != '0' env: GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} GH_REPO: ${{ github.repository }} - META_PR_NUMBER: ${{ steps.meta.outputs.pr_number }} - META_HEAD_SHA: ${{ steps.meta.outputs.head_sha }} - META_HEAD_REPO: ${{ steps.meta.outputs.head_repo }} + META_PATH: autofix-in/metadata.json + SCHEMA_PATTERN: '^gitnexus\.pr-autofix/v[0-9]+$' WF_HEAD_SHA: ${{ github.event.workflow_run.head_sha }} WF_HEAD_REPO: ${{ github.event.workflow_run.head_repository.full_name }} - WF_PR_NUMBERS: ${{ toJSON(github.event.workflow_run.pull_requests.*.number) }} + WF_HEAD_BRANCH: ${{ github.event.workflow_run.head_branch }} shell: bash - run: | - set -euo pipefail - - # 1) head_sha must match exactly. workflow_run.head_sha is the - # commit GitHub actually ran the producer against — definitive. - if [ "${META_HEAD_SHA}" != "${WF_HEAD_SHA}" ]; then - echo "::error::Artifact head_sha (${META_HEAD_SHA}) does not match workflow_run.head_sha (${WF_HEAD_SHA}) — refusing to publish." - exit 1 - fi - - # 2) head_repo must match exactly. Same authority anchor. - if [ "${META_HEAD_REPO}" != "${WF_HEAD_REPO}" ]; then - echo "::error::Artifact head_repo (${META_HEAD_REPO}) does not match workflow_run.head_repository (${WF_HEAD_REPO}) — refusing to publish." - exit 1 - fi - - # 3) pr_number must reference an open PR with this head SHA. - # Within-repo PRs: workflow_run.pull_requests[] is populated. - # Fork PRs: that array is empty by GitHub design — fall back - # to the REST commit-to-PRs lookup. Fail closed if the lookup - # finds no matching open PR (avoids attacker-forged PR ids). - allowed_numbers=$(jq -c '.' <<< "${WF_PR_NUMBERS}") - if [ "${allowed_numbers}" = "[]" ]; then - echo "workflow_run.pull_requests is empty (fork PR) — falling back to commits/{sha}/pulls." - allowed_numbers=$(gh api "repos/${GH_REPO}/commits/${WF_HEAD_SHA}/pulls" \ - --jq '[.[] | select(.state == "open") | .number]' 2>/dev/null || echo "[]") - if [ "${allowed_numbers}" = "[]" ]; then - echo "::error::No open PR found for head ${WF_HEAD_SHA} via commits/{sha}/pulls — refusing to publish." - exit 1 - fi - fi - - if ! jq -e --argjson n "${META_PR_NUMBER}" 'index($n) != null' <<< "${allowed_numbers}" >/dev/null; then - echo "::error::Artifact pr_number (${META_PR_NUMBER}) is not in the authoritative PR list (${allowed_numbers}) — refusing to publish." - exit 1 - fi - - echo "Verified: metadata identity matches workflow_run authority (PR=${META_PR_NUMBER}, head_sha=${META_HEAD_SHA}, head_repo=${META_HEAD_REPO})." + run: node trusted/.github/scripts/verify-workflow-run-pr-identity.cjs - name: Upsert sticky summary comment # Only post when ci-quality found something fixable (= the @@ -187,14 +158,14 @@ jobs: # so we skip it. if: >- always() - && steps.meta.outputs.pr_number != '' + && steps.verify.outcome == 'success' && steps.meta.outputs.changed_lines != '0' env: GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} GH_REPO: ${{ github.repository }} - PR: ${{ steps.meta.outputs.pr_number }} + PR: ${{ steps.verify.outputs.pr_number }} CHANGED: ${{ steps.meta.outputs.changed_lines }} - HEAD_SHA: ${{ steps.meta.outputs.head_sha }} + HEAD_SHA: ${{ steps.verify.outputs.head_sha }} RUN_ID: ${{ github.run_id }} shell: bash run: | @@ -285,11 +256,11 @@ jobs: # fixes-available → conclusion: neutral # `neutral` does not block branch-protection required-checks but # is visually distinct from a green pass. - if: always() && steps.meta.outputs.head_sha != '' + if: always() && steps.verify.outcome == 'success' env: GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} GH_REPO: ${{ github.repository }} - HEAD_SHA: ${{ steps.meta.outputs.head_sha }} + HEAD_SHA: ${{ steps.verify.outputs.head_sha }} CHANGED: ${{ steps.meta.outputs.changed_lines }} shell: bash run: | diff --git a/.github/workflows/publish.yml b/.github/workflows/publish.yml index c03a801c1..ca81460f1 100644 --- a/.github/workflows/publish.yml +++ b/.github/workflows/publish.yml @@ -831,7 +831,7 @@ jobs: fi - name: Create GitHub Release - uses: softprops/action-gh-release@3d0d9888cb7fd7b750713d6e236d1fcb99157228 # v2 + uses: softprops/action-gh-release@efb35369e0ad2afab669f228072c1b0d510eae64 # v2 with: tag_name: ${{ steps.vtag-gate.outputs.vtag }} name: >- diff --git a/.github/workflows/tree-sitter-upgrade-readiness.yml b/.github/workflows/tree-sitter-upgrade-readiness.yml index 13b79e093..88ee0bd42 100644 --- a/.github/workflows/tree-sitter-upgrade-readiness.yml +++ b/.github/workflows/tree-sitter-upgrade-readiness.yml @@ -4,7 +4,7 @@ name: Tree-sitter Upgrade Readiness # 1. Peer-dep compatibility — can each NPM-installed grammar install cleanly # with tree-sitter@0.25.0 without --legacy-peer-deps? # 2. Vendored grammars — each grammar in .github/vendored-grammars.json -# (c/swift/kotlin/dart/proto) is classified by its vendored ABI, read +# (c/swift/kotlin/dart/proto/objc) is classified by its vendored ABI, read # straight from gitnexus/vendor//src/parser.c (NOT node_modules, # which is never populated for vendored grammars — that mismatch is why # the report used to render bare "?" placeholders, #858). diff --git a/.github/zizmor.yml b/.github/zizmor.yml index b18a76336..7426e31ae 100644 --- a/.github/zizmor.yml +++ b/.github/zizmor.yml @@ -18,9 +18,11 @@ rules: # untrusted half (pr-autofix.yml) runs fork code with permissions:{} # and produces only a diff artifact (data, not executable code). The # publish job consumes the artifact, allowlist-validates every field - # of metadata.json before exporting to $GITHUB_OUTPUT, never checks - # out fork code, and never executes anything fork-controlled. Header - # comment in the file documents the split. + # of metadata.json, then cross-checks identity against + # workflow_run.head_sha / head_repository / head_branch via + # pulls?head=owner:branch (commits/{sha}/pulls is empty for fork SHAs). + # It never checks out fork code and never executes anything + # fork-controlled. Header comment in the file documents the split. - pr-autofix-publish.yml # workflow_run is the trusted half of the vendored-grammar prebuild @@ -29,7 +31,8 @@ rules: # validates the .node prebuilds and uploads them as artifacts. This # consumer downloads ONLY those artifacts + metadata.json, # allowlist-validates every metadata field, cross-checks identity against - # the workflow_run authority (head_sha / head_repo / pr_number), and + # workflow_run.head_sha / head_repository / head_branch via + # pulls?head=owner:branch (commits/{sha}/pulls is empty for fork SHAs), and # checks out the fork head pinned to that HEAD SHA solely to ADD prebuild # files (never executes fork code) before pushing. Header comment in the # file documents the split. diff --git a/.gitignore b/.gitignore index 3d125d475..1f350e059 100644 --- a/.gitignore +++ b/.gitignore @@ -72,6 +72,8 @@ eval/.hypothesis/ # Local docs — planning output (gitnexus-plan / gitnexus-work) stays local, not tracked docs/* +!docs/fork/ +!docs/fork/** gitnexus/test/fixtures/mini-repo/*.md gitnexus/test/fixtures/mini-repo/.claude diff --git a/AGENTS.md b/AGENTS.md index fffdd6242..d15b02274 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -1,7 +1,7 @@ - - + + -Last reviewed: 2026-07-16 +Last reviewed: 2026-09-07 **Project:** GitNexus · **Environment:** dev · **Maintainer:** repository maintainers (see GitHub) @@ -39,6 +39,7 @@ Commands and gotchas live under **Repo reference** below and in **[CONTRIBUTING. ## Reference docs - **[ARCHITECTURE.md](ARCHITECTURE.md)**, **[CONTRIBUTING.md](CONTRIBUTING.md)**, **[GUARDRAILS.md](GUARDRAILS.md)** +- **Objective-C provider work:** read **[docs/languages/objective-c-provider.md](docs/languages/objective-c-provider.md)** before changing Objective-C parsing or resolution. - **Call & inheritance resolution (RFC #909 Ring 3):** See ARCHITECTURE.md § Scope-Resolution Pipeline. All languages resolve calls and inheritance through the scope-resolution pipeline (`Registry.lookup`, `preEmitInheritanceEdges`, `emitHeritageEdges`, `buildMro` → `MethodDispatchIndex`). **Shared code in `gitnexus/src/core/ingestion/` must not name languages** — plug language behavior in via `LanguageProvider` / `ScopeResolver` hooks. A language plugs in by implementing `ScopeResolver` (`scope-resolution/contract/scope-resolver.ts`) and registering it in `SCOPE_RESOLVERS`. (The legacy call-resolution DAG + `@heritage` capture path were removed in RING4-1 #942.) - **Cursor:** `.cursor/index.mdc` (always-on); `.cursor/rules/*.mdc` (glob-scoped). Legacy `.cursorrules` deprecated. - **GitNexus:** standard skills in `.claude/skills/gitnexus-*/`; MCP rules in `gitnexus:start` block below. @@ -90,6 +91,7 @@ mirror. `gitnexus/test/unit/shipped-skills-sync.test.ts` guards the copies. Toke | Date | Version | Change | |------|---------|--------| +| 2026-09-07 | 1.15.0 | Added the Objective-C provider guide as the required reference before changing Objective-C parsing or resolution. | | 2026-07-20 | 1.14.0 | `gitnexus-review` gains a coordinated swarm: six `ci-personas/` lanes the CI review agent dispatches as subagents (via the `Agent` tool), with a bounded critic gate and sidechain-excluded evidence. | | 2026-07-16 | 1.13.0 | `gitnexus-plan` asks plan depth up front (quick/standard/deep) in interactive runs; `gitnexus-lfg` gate slimmed to proceed/stop (Deepen stays as the route-back mechanism). | | 2026-07-16 | 1.12.0 | Renamed `gitnexus-pr-review` to `gitnexus-review`; added PR URL/number, branch/range, and local-change targets plus install migration (setup warns on a legacy `gitnexus-pr-review` dir and leaves it in place; uninstall removes it). | @@ -195,3 +197,4 @@ npx gitnexus serve # HTTP API on port 4747 (from any ind - `npm install` in `gitnexus/` triggers `prepare` (builds via `tsc`) and `postinstall` (`build-tree-sitter-grammars.cjs` activates committed prebuilds in place under `vendor/`, and only source-builds when none matches). A C/C++ toolchain (`python3`, `make`, `g++`) is needed only for that source-build fallback. - The vendored grammars `tree-sitter-{c,dart,proto,swift,kotlin,zig}` are handled uniformly: c is required; dart/proto/swift/kotlin/zig are optional and skippable via `GITNEXUS_SKIP_OPTIONAL_GRAMMARS=1`. Install warnings appear only when no prebuild matches the platform-arch and no toolchain is present, and are non-fatal — only that language's parsing is unavailable. - ESLint configured via `eslint.config.mjs` (TS, React Hooks, unused-imports). No `npm run lint` script; use `npx eslint .`. Prettier runs via lint-staged. CI checks both in `ci-quality.yml`. +- Index storage defaults to `/.gitnexus/`. `GITNEXUS_STORAGE_PATH` selects one complete external index directory and wins over `GITNEXUS_STORAGE_ROOT`, which creates an isolated `-<12-hex>/` slot per repository. `GITNEXUS_CONTENT_RETENTION` is `full` (default), `symbol`, or `none`. MCP `list_repos`, `gitnexus://repo/{name}/context`, and HTTP `GET /api/repos` / `GET /api/repo` expose `storagePath`, `contentRetention`, and `sourceAvailable`. HTTP `/api/file` and `/api/grep` return 410 unless retention is `full`; MCP `include_content` may still return symbol spans at `symbol`. diff --git a/ARCHITECTURE.md b/ARCHITECTURE.md index 10c511697..17b88d42e 100644 --- a/ARCHITECTURE.md +++ b/ARCHITECTURE.md @@ -24,7 +24,7 @@ Monorepo: **CLI/MCP** (`gitnexus/`) + **browser UI** (`gitnexus-web/`). - **HTTP bridge:** `serve.ts` → Express (`api.ts`, `mcp-http.ts`) for web UI - **CLI direct:** `gitnexus query|context|impact|cypher` in `tool.ts` -4. **Staleness** — `staleness.ts` compares indexed `lastCommit` to `HEAD`, surfaces hints. +4. **Staleness** — `core/git-staleness.ts` compares indexed `lastCommit` to `HEAD` and classifies the result as `current`, `behind`, `diverged` (HEAD moved off the indexed commit, gap uncountable) or `unknown`; `core/staleness-status.ts` builds the one `staleness` payload that MCP `list_repos`, the read tools and the `serve` repo routes all emit. ## MCP tools @@ -379,7 +379,7 @@ CI auto-discovers the set via `tsx`. No workflow edit required. ## Language-agnostic graph feeding -16 languages → single unified graph. Four abstraction layers: +18 languages → single unified graph. Four abstraction layers: ``` Unified Graph Schema (44 node types, 21 relationship types) @@ -407,7 +407,7 @@ Each language implements `LanguageProvider` (`language-provider.ts`). Key fields | `descriptionExtractor` | Optional hook returning a symbol's doc-comment text as its `description`; feeds the embedding metadata header so doc-only terms are semantically searchable (issue #2270). Most languages register `createLeadingDocDescriptionExtractor` (shared, language-neutral; per-language comment/wrapper config passed at the call site) | | `definitionPropertiesExtractor` | Optional language-owned hook for structured, clone-safe definition metadata. Shared ingestion persists these properties opaquely; the owning provider supplies the extraction semantics. | -16 providers in `languages/index.ts` via `satisfies Record` — missing a language is a compile error. +18 providers in `languages/index.ts` via `satisfies Record` — missing a language is a compile error. ### Unified capture tags @@ -544,4 +544,5 @@ Node IDs use arity suffix (`#`): `Method:file:Class.method#1` vs `#2 - [RUNBOOK.md](RUNBOOK.md) — operational commands and recovery - [GUARDRAILS.md](GUARDRAILS.md) — safety boundaries for humans and agents - [TESTING.md](TESTING.md) — how to run tests +- [docs/languages/objective-c-provider.md](docs/languages/objective-c-provider.md) — Objective-C provider behavior and limits - `AGENTS.md` / `CLAUDE.md` — agent workflows and tool usage diff --git a/README.md b/README.md index 38bce3ec5..b247b8b6f 100644 --- a/README.md +++ b/README.md @@ -1,7 +1,5 @@ # GitNexus (Akon Labs) -**⚠️ Important Notice:** GitNexus has NO official cryptocurrency, token, or coin. Any token/coin using the GitNexus name on Pump.fun or any other platform is **not affiliated with, endorsed by, or created by** this project or its maintainers. Do not purchase any cryptocurrency claiming association with GitNexus. -
@@ -26,7 +24,7 @@

-

The nervous system for agent context.

+

The context engine for Enterprise Codebases

Indexes any codebase into a knowledge graph — every dependency, call chain, cluster, and execution flow — @@ -102,6 +100,10 @@ The proxy strips `Origin` before forwarding, so the server's CSRF guard does not Indexing is memory-bound. If `gitnexus-server` runs out of memory on a large repo, raise its `plan`, which sets available RAM: `standard` is 2 GB, `pro` is 4 GB. Raise `sizeGB` only if the disk fills with clones and indexes. +### Deploy to RepoCloud + +[![Deploy on RepoCloud](https://d16t0pc4846x52.cloudfront.net/deploylobe.svg)](https://repocloud.io/details/gitnexus/) + ## Two Ways to Use GitNexus | | **CLI + MCP** (recommended) | **Web UI** | @@ -405,7 +407,7 @@ backoff. Invalid `.gitnexusrc` or ignore-file reloads pause ordinary refreshes until the control file is fixed. Stop the watcher with Ctrl+C. Watch mode accepts `--debounce`, `--workers`, `--worker-timeout`, -`--max-file-size`, `--branch`, `--pdg`, `--name`, `--allow-duplicate-name`, and +`--max-file-size`, `--branch`, `--pdg`, `--skip-fts`, `--name`, `--allow-duplicate-name`, and `--verbose`. Explicit one-shot options such as `--force`, `--repair-fts`, embedding flags, `--skills`, `--self-commit`, `--index-only`, and `--skip-git` are rejected. Unsupported defaults from `.gitnexusrc` are ignored with a @@ -439,6 +441,7 @@ The token may be set in the shell, `.env.local`, or `.env` in the working direct gitnexus analyze --force # Full graph + FTS rebuild (reuses unchanged parser output) gitnexus analyze --no-parse-cache # Full rebuild that re-parses every source file gitnexus analyze --repair-fts # Fast path: rebuild/verify only FTS indexes on existing index data +gitnexus analyze --skip-fts # Index graph/embeddings without loading FTS or building keyword indexes gitnexus analyze --skills # Generate repo-specific skill files from detected communities gitnexus analyze --skip-embeddings # Skip embedding generation (faster) gitnexus analyze --embeddings [limit] # Enable embedding generation (slower, better search) @@ -456,6 +459,8 @@ gitnexus analyze --wal-checkpoint-threshold 67108864 # LadybugDB WAL auto-check # (default 67108864 = 64 MiB; -1 keeps Ladybug stock ~16 MiB) ``` +`--skip-fts` (or `GITNEXUS_SKIP_FTS=1`) disables FTS extension loading and keyword-index construction for this analysis. Graph queries, communities, processes, and existing embeddings remain available. Status and search report "FTS disabled for this index". Remove both the flag and environment setting and run `analyze` again to restore keyword search, even at the same commit. Only the exact environment value `1` enables the opt-out; the flag takes precedence. It cannot be combined with `--repair-fts`. Disabling an existing FTS index may require one graph-store rebuild to avoid unsafe writes through native indexes. + `--spring-actuator` is explicitly opt-in and accepts either a JSON bundle keyed by `mappings`, `beans`, `conditions`, `configprops`, and/or `env`, or a directory containing endpoint-named JSON files. It confirms matching static nodes and adds conservative runtime-only routes, beans, and property keys. The configured input is excluded from source scanning; only normalized repository-relative exclusions are retained for future scans, never absolute paths. Env/configprops values, origins, condition messages, and source names are never persisted or printed. Because snapshots are external runtime state, an enabled run always rebuilds; the first later run without the option rebuilds once to remove runtime evidence. The same path can be set as `springActuator` in `.gitnexusrc`. `--asyncapi-spec` is explicitly opt-in and accepts a directory of AsyncAPI documents or a single document; the path is resolved against the repository root, so a committed `docs/asyncapi` and an absolute cache written by something else both work. Each `operations[]` entry of an **AsyncAPI 3.x** document can contribute a `Destination` node keyed by broker and address, with `action: send` emitting `PUBLISHES_TO` and `action: receive` emitting `CONSUMES_FROM`, so a document and source code that name one address on one broker land on the same node. Edges start at the document, not at a callable — a document states that the service talks to an address, not which method does — and no address a document names is ever attached to an unresolved source site. @@ -581,6 +586,7 @@ Most `analyze` knobs are also CLI flags (`--workers`, `--worker-timeout`, `--max | `GITNEXUS_WORKER_POOL_SIZE` | `cores - 1`, capped at 16 | Parse worker pool size (must be ≥ 1). Equivalent to `--workers `. The worker pool is the sole parse path — there is no sequential parser, so `0` is rejected with an actionable error (the pool self-heals via quarantine + respawn). | Constrained containers (cgroup CPU limits) or CI runners with explicit quotas. To narrow down a worker crash set `1` for a single-worker pool — not `0`. | | `GITNEXUS_PARSE_CHUNK_CONCURRENCY` | `2` | Number of chunks whose file contents may be read into memory in parallel while the pool dispatches the current chunk. Worker dispatch itself stays serial. | Repos large enough to chunk (multi-MB total source) where disk I/O is a measurable fraction of analyze wall-clock. | | `GITNEXUS_VERBOSE` | unset | When `1`, enables verbose ingestion logs (skipped-file warnings, per-chunk throughput, parse-cache stats). Equivalent to `--verbose`. | Debugging an analyze that "completed" but seems to have missed files; tuning `--workers` / chunk concurrency against observable throughput. | +| `GITNEXUS_EMBEDDING_RETRY_TIMEOUTS` | unset | When truthy (`1`/`true`/`yes`), per-attempt HTTP embedding timeouts (`TimeoutError` on fetch or body read) go through the bounded `GITNEXUS_EMBEDDING_MAX_ATTEMPTS` retry loop instead of failing the job. Any other value leaves it off, so cloud/default timeouts remain terminal. | Local accelerators that drop a device lock when the client disconnects and succeed on the next request (observed with FastFlowLM on Ryzen AI). | | `GITNEXUS_ANALYZER_IDENTITY_IN_PROCESS_GUARDS` | unset | When truthy (`1`/`true`/`yes`), forces in-process cache-guard validation once a batch has ≥128 requests. In-process mode also auto-selects when `packageRoot`/`buildRoot` fail `W_OK` with `EACCES`/`EROFS`. Otherwise those large batches use a Node subprocess probe. Batches under 128 always stay in-process. | Trusted or read-only installs where two identity subprocess spawns per analyze dominate wall time; leave unset to keep the default isolation path on writable trees. | | `GITNEXUS_RESOLVE_DEF_GRAPH_ID_MEMO` | on (unset) | Memoizes `resolveDefGraphId` per `nodeLookup` instance (WeakMap). Enabled by default. Set to `0`/`false`/`off`/`no` to disable and recompute on every call (debug / bisect memo bugs). | Suspecting stale graph-id resolution after a lookup rebuild, or comparing memo vs uncached cost on a large index. | | `GITNEXUS_AUTH_TOKEN` | unset | Bearer token required when `eval-server` binds beyond loopback. May also be read from `.env.local` or `.env`; shell values take precedence. | Exposing the evaluation HTTP tools to a container, VM, or LAN. | @@ -592,6 +598,10 @@ Most `analyze` knobs are also CLI flags (`--workers`, `--worker-timeout`, `--max | `GITNEXUS_WORKER_SUB_BATCH_TIMEOUT_MS` | `30000` | Worker idle timeout in milliseconds before retry/fallback. Equivalent to `--worker-timeout ` × 1000. | Slow-parsing files (large minified JS, deeply-nested TS types) that legitimately need more than 30s. | | `GITNEXUS_WORKER_READY_TIMEOUT_MS` | `5000` | Startup budget in milliseconds for a parse worker to load its grammar bindings and report `{type:'ready'}`. Slots that miss it are treated as startup crashes. | Slow or heavily loaded hosts where a full pool cold-starting concurrently needs more than 5s, and analyze aborts with "did not report ready within 5000ms". | | `GITNEXUS_FTS_STEMMER` | `porter` | Stemmer used when rebuilding BM25/FTS indexes. Use `none` for CJK-heavy repositories, or a language stemmer such as `german`, `french`, or `spanish` for matching repository comments. Re-run `gitnexus analyze --repair-fts` after changing it. | Keyword search quality is poor for non-English comments or identifiers under English stemming. | +| `GITNEXUS_STORAGE_PATH` | unset (`/.gitnexus/`) | Complete external index directory. This preserves the existing configuration semantics and takes precedence over `GITNEXUS_STORAGE_ROOT` when both are set. | You already keep one repository index outside its checkout or need one explicit index location. | +| `GITNEXUS_STORAGE_ROOT` | unset | Absolute root directory for external indexes. GitNexus creates an isolated `-/` slot beneath it for each repository, then registers the resolved slot so `status`, MCP, and `serve` can reopen it later. | You want to manage multiple repository indexes centrally or keep generated data outside source checkouts. | +| `GITNEXUS_CONTENT_RETENTION` | `full` | Source-text retention profile: `full` keeps file and symbol text, `symbol` keeps symbol snippets without full file content, and `none` keeps the structural graph without source body text. | You need to reduce persisted source text while preserving graph structure. | +| `GITNEXUS_SKIP_FTS` | unset | When exactly `1`, skips FTS extension loading and keyword index creation during analyze. Equivalent to `--skip-fts`; a later analyze without either option restores FTS. | Graph-only consumers with their own retrieval, or short-lived indexes that do not need keyword search. | | `GITNEXUS_WAL_CHECKPOINT_THRESHOLD` | `67108864` (64 MiB) | LadybugDB WAL auto-checkpoint threshold in bytes. Equivalent to `--wal-checkpoint-threshold `. `-1` keeps LadybugDB's stock threshold (~16 MiB). Larger thresholds reduce checkpoint frequency but increase the WAL size at rotation time — choose a smaller value on disk-constrained environments. | You need a larger or smaller WAL auto-checkpoint threshold for your analyze workload. | | `GITNEXUS_LBUG_BUFFER_POOL_SIZE` | min(2 GiB, 80% RAM) | LadybugDB buffer-pool ceiling in bytes for every GitNexus database (analyze, MCP server, serve, group bridges). `0` restores LadybugDB's native unbounded default of 80% of system RAM; invalid values warn and fall back to the default (#2557). During `analyze` the pool is right-sized to the graph, scaled on non-4 KiB-page hosts by the page-size granule ratio up to min(2 GiB × pageSize/4 KiB, 80% RAM) (#2631); this env var overrides all of that as an absolute value. | A long-lived `gitnexus mcp` or a big incremental `analyze` uses too much memory, or a huge repo's working set genuinely needs a pool larger than 2 GiB. | | `GITNEXUS_LBUG_MAX_DB_SIZE` | `17179869184` (16 GiB) | Maximum size in bytes of a single LadybugDB database file — an mmap/disk-address-space ceiling, not a memory limit (it does not constrain the buffer pool). Invalid values silently fall back to the default. | Indexing a genuinely huge monorepo whose on-disk graph index approaches 16 GiB. | @@ -659,6 +669,7 @@ XAML files are indexed as documents. Literal `x:Name`, `x:Key`, and `x:Class` de | Swift | — | — | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | | C | — | — | ✓ | — | ✓ | ✓ | — | ✓ | ✓ | | C++ | — | — | ✓ | ✓ | ✓ | ✓ | — | ✓ | ✓ | +| Objective-C | ✓ | — | ✓ | ✓ | ✓ | — | — | — | — | | Dart | ✓ | — | ✓ | ✓ | ✓ | ✓ | — | ✓ | ✓ | | Zig | ✓ | — | ✓ | — | ✓ | ✓ | ✓ | — | ✓ | @@ -670,7 +681,7 @@ XAML files are indexed as documents. Literal `x:Name`, `x:Key`, and `x:Class` de GitNexus uses a **global registry** so one MCP server can serve multiple indexed repos. No per-project MCP config needed — set it up once and it works everywhere. -Each `gitnexus analyze` stores the index in `.gitnexus/` inside the repo (portable, gitignored) and registers a pointer in `~/.gitnexus/registry.json`. When an AI agent starts, the MCP server reads the registry and can serve any indexed repo. LadybugDB connections are opened lazily on first query and evicted after 5 minutes of inactivity (max 5 concurrent). Read-only tools can omit `repo` when only one repo is indexed, an MCP default is configured, or the GitNexus process cwd is inside a registered path without crossing into an unindexed nested Git checkout. Outside those paths—and for mutating tools with multiple indexed repos and no MCP default—pass `repo` explicitly. +Each `gitnexus analyze` stores the index in `.gitnexus/` inside the repo by default (portable, gitignored). `GITNEXUS_STORAGE_PATH` selects one complete external index directory and preserves the established configuration behavior. To manage multiple repositories under one external directory, set `GITNEXUS_STORAGE_ROOT`; GitNexus derives an isolated `-/` slot beneath it for each repository. If both variables are set, `GITNEXUS_STORAGE_PATH` takes precedence. GitNexus registers the resolved slot in `~/.gitnexus/registry.json`, allowing later `status`, MCP, and `serve` commands to reopen the index without repeating the environment variable. LadybugDB connections are opened lazily on first query and evicted after 5 minutes of inactivity (max 5 concurrent). Read-only tools can omit `repo` when only one repo is indexed, an MCP default is configured, or the GitNexus process cwd is inside a registered path without crossing into an unindexed nested Git checkout. Outside those paths—and for mutating tools with multiple indexed repos and no MCP default—pass `repo` explicitly.

Architecture diagram @@ -1089,7 +1100,7 @@ Built by the community — not officially maintained, but worth checking out. ## Security & Privacy -- **CLI**: everything runs locally on your machine. No network calls. Index stored in `.gitnexus/` (gitignored). Global registry at `~/.gitnexus/` stores only paths and metadata. +- **CLI**: everything runs locally on your machine. No network calls. Indexes are stored in `.gitnexus/` by default (gitignored), in the complete external directory selected by `GITNEXUS_STORAGE_PATH`, or in repository-specific slots beneath `GITNEXUS_STORAGE_ROOT`. Global registry at `~/.gitnexus/` stores only paths and metadata. - **Web**: everything runs in your browser. No code uploaded to any server. API keys stored in localStorage only. - Open source — audit the code yourself. diff --git a/RUNBOOK.md b/RUNBOOK.md index d16ccd52d..2f2c06f14 100644 --- a/RUNBOOK.md +++ b/RUNBOOK.md @@ -187,6 +187,58 @@ If the error text is `"Only one write transaction at a time is allowed in the sy --- +## File acquisition/reclaim guard recovery + +The portable file-lock backend uses `analyze.lock.guard` beside `analyze.lock`. +Every acquisition, including an empty slot, exclusively creates the guard before +inspecting, reclaiming, creating, and verifying the main lock. It removes the +guard before returning a workload handle or waiting on a live workload holder. +Linux abstract-socket and Windows named-pipe locking are unchanged. + +A stalled or crashed guard owner blocks file acquisition even when its PID is +dead, its metadata is incomplete, or no main lock exists. **The guard is never +automatically stolen.** Guard contention times out after at most 30 seconds, +capped by the remaining acquisition timeout. This separate ceiling applies even +when `GITNEXUS_INDEX_LOCK_TIMEOUT_MS` is zero or negative (unbounded workload wait). +A guard-cleanup failure rejects acquisition; it must not start unprotected work. + +Manual recovery is an outage procedure, not an age/PID-based cleanup: + +1. Identify the exact lock directory named in the error. This shared primitive + also protects group sync and registry operations, not just repo analysis. +2. Stop **all relevant writers** and prevent restart: editor/agent hooks, watch + processes, scheduled jobs, services, and any containers sharing the directory. + Account for paused processes and every host with access. If quiescence cannot + be established, do not remove the guard. PID metadata is diagnostic only. +3. While restart remains disabled, inspect and preserve the guard/main records + for diagnosis, then remove only that directory's orphan `analyze.lock.guard` + and, if present, its orphan `analyze.lock`. Do not remove databases or sidecars + as part of lock recovery. Do not use a recursive or wildcard cleanup. +4. Ensure all participating writers use the guarded version and the same locking + backend/domain, then restart in a controlled fashion. + +**Upgrade requires a coordinated stop/upgrade/restart.** Concurrent older +versions ignore the guard and can still displace live locks; mixed-version +mutual exclusion is not guaranteed. The file protocol assumes reliable atomic +local-filesystem `O_EXCL` creation and cooperating processes. Network/distributed +filesystems, external file replacement, and uncoordinated manual deletion are not +covered. A process crash while holding the short-lived guard trades automatic +recovery for fail-closed safety. Denied file creation returns a non-owning +`lockFree` handle only when neither workload lock nor acquisition guard exists; +unreadable paths fail closed. No staging sweep runs without ownership. Analysis, +registry transactions, group synchronization, and embeddings sync refuse +`lockFree` handles, including an otherwise up-to-date analysis on a file-backend +read-only mount. +The socket backend can still acquire ownership on a read-only index mount. +Heterogeneous permissions are not proof that another process cannot write. + +If guard cleanup fails after this attempt created its workload record, acquisition +is refused and token-exact workload cleanup is attempted before returning the +error. Failed or unverifiable cleanup must be diagnosed under the same quiesced +recovery procedure above; never delete a possibly active successor's record. + +--- + ## Where to dig deeper - Architecture overview: [ARCHITECTURE.md](ARCHITECTURE.md) diff --git a/docs/languages/objective-c-provider.md b/docs/languages/objective-c-provider.md new file mode 100644 index 000000000..7051832c5 --- /dev/null +++ b/docs/languages/objective-c-provider.md @@ -0,0 +1,109 @@ +# Objective-C Language Provider + +Status: implemented + +The deterministic provider is covered by focused unit and integration tests. The parser-loader ABI +smoke runs in the published multi-OS test matrix, and the native prebuild workflow owns +Objective-C together with all six vendored grammar targets. This status describes the implemented +MVP; it does not promise full Objective-C runtime dispatch. + +## Goal + +Add deterministic, symbol-level Objective-C analysis to GitNexus. The first release must support high-confidence code navigation and direct static dependency analysis for `.m`, `.mm`, and Objective-C `.h` files. It must not imply that Objective-C runtime dispatch is fully resolved. + +The provider belongs in the existing language-provider and scope-resolution extension points. Shared ingestion code must remain language-agnostic. + +## Compatibility contract + +- Existing language detection and parsing must remain unchanged. +- A `.h` file must be classified from its content or surrounding context; it cannot be unconditionally claimed by Objective-C because C and C++ also use that extension. +- If Objective-C grammar loading fails, the error must clearly name the missing provider/grammar and cannot corrupt a previously valid index. +- Provider and grammar versions must be stored in index metadata. A version change that can alter node identity or edges requires a full rebuild. +- No LLM participates in parsing, name resolution, or edge creation. Analysis is Tree-sitter plus deterministic static resolution. + +## MVP model + +The provider must extract and connect: + +- Classes, superclasses, protocols, categories, class extensions, properties, ivars, C functions, imports, declarations, and implementations. +- Instance and class methods, preserving their complete multi-part selector. +- Inheritance, protocol conformance, import, declaration/implementation, host-class/category, and statically resolved call relationships. + +Stable identity must include enough ownership to distinguish same-named methods. Recommended forms are: + +```text +objc:class: +objc:protocol: +objc:category:: +objc:method::-: +objc:method::+: +objc:function: +``` + +For example, `-loadData:completion:` and `+loadData:completion:` are different symbols. A category method remains linked to both its category and host class; querying the host class must expose distributed implementations. + +## Resolution policy + +Resolution must be conservative. A missing or dynamic target is evidence of uncertainty, not proof that no target exists. + +| Receiver case | Required result | +| --------------------------------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------- | +| Explicit class name, `self`, or `super` | Resolve when the owner is statically known. | +| Local, parameter, property, or ivar with known static type | Resolve to matching owner and selector. | +| Protocol-typed receiver | Link the protocol method and identify possible implementations as candidates. | +| Multiple host-class/category implementations of one selector | Record all static candidates as evidence; do not emit a certain call edge because runtime image-load order is unknown. | +| `id`, `Class`, macros, reflection, `performSelector:`, `NSInvocation`, runtime injection, or unknown type | Store selector/location with `resolution=unresolved`; do not emit a certain call edge. | + +The provider should first collect file-local declarations, imports, and types, then resolve across the repository. It must use structured Tree-sitter captures or AST traversal, not regular expressions over source text. Multi-part selectors, block arguments, nullability annotations, generics, macros, and multiline declarations make a regex-only extractor unsafe. + +## Imports and incremental correctness + +- Resolve quoted project imports against the current directory, configured include roots, and indexed headers. Model framework imports as external-module evidence without downloading SDK source. +- Merge `@interface`, `@implementation`, categories, and extensions across files. +- A changed header, protocol, class declaration, or category invalidates importing and affected implementation/call-resolution state. Incremental output after such a change must match a full rebuild. +- Index metadata must record provider version, grammar version, include/exclude configuration, and parsing options used for resolution. + +## Implementation sequence + +1. Add and package a pinned Objective-C Tree-sitter grammar; verify macOS arm64 and the production Linux runner can load it. +2. Add language detection for `.m`, `.mm`, and content-classified `.h` files. +3. Implement AST extraction and stable IDs for declarations and definitions. +4. Implement repository-level merge, imports, inheritance, protocol, and category relationships. +5. Add conservative message-send resolution and explicit unresolved evidence. +6. Integrate invalidation, metadata comparison, MCP/CLI output, and fixtures. + +## Fixtures and acceptance + +Create a minimal Objective-C fixture containing a class, protocol, category, extension, superclass, properties, ivars, C function, imports, multi-part selector, block parameter, `self`, `super`, protocol receiver, and `id` receiver. Use `symodulebridge` as a real integration fixture after the minimal suite is stable. + +The acceptance bar is: + +- `query "SYModuleCaller"` yields class/method semantic nodes, not only file nodes. +- `context "SYModuleCaller" --file ` yields declaration, implementation, imports, and known references. +- Known statically typed message sends create call edges; dynamic sends are marked unresolved. +- Same selector on multiple classes, a category override, and `+` versus `-` methods remain distinct. +- A `.m`, `.h`, protocol, or category edit produces results equivalent to a clean rebuild. +- Generated documentation, dependency directories, and build output are excluded through explicit indexing configuration. + +## Non-goals + +The MVP does not promise exact runtime type inference for `id` or `instancetype`, reflection, swizzling, arbitrary category replacement, dynamic selector construction, or complete impact analysis across every runtime dispatch path. Tool results must surface confidence and unresolved evidence rather than presenting guesses as certain graph facts. + +## Current implementation coverage + +Implemented capabilities: + +- Vendored `tree-sitter-objc` grammar, registered through the existing Tree-sitter loader. +- `.m` and `.mm` language mapping plus content-based `.h` classification so plain C/C++ headers are not unconditionally claimed. +- LanguageProvider extraction for classes, protocols, categories, extensions, methods, properties, ivars, C functions, imports, unresolved message evidence, stable Objective-C qualified names, and provider/grammar metadata. +- Length-preserving preprocessing of bare, file-scope all-caps macro markers before Tree-sitter parsing. This recovers declarations after wrappers such as `RCT_EXTERN_C_BEGIN` / `RCT_EXTERN_C_END` without expanding macros or adding framework-specific rules. +- ScopeResolver edges for imports, inheritance, protocol conformance, category host membership, implementation evidence, and conservative static message sends. +- Persisted query/context support for Objective-C class and method nodes, including implementation evidence via `DECLARES`. +- Regression tests for grammar loading, `.h` classification, stable identities, conservative calls, metadata feature mismatch, persisted query/context behavior, and incremental-vs-force parity for Objective-C fixture edits. + +Known limits of this MVP: + +- The first version does not perform full Objective-C runtime dispatch, swizzling, dynamic selector construction, macro expansion, or `id` flow inference. Bare file-scope marker macros are elided only to preserve parser recovery; their expansion semantics are not interpreted. +- Protocol receiver handling records the protocol method and candidate implementation evidence, but candidate implementations are not emitted as certain call edges. +- When a host class and one or more named categories define the same selector, the provider records candidate evidence rather than choosing a runtime winner or emitting multiple certain call edges. +- Objective-C++ `.mm` files are parsed with the Objective-C grammar path for this MVP; deep C++ semantic extraction inside Objective-C++ bodies remains outside this provider. diff --git a/eval/.gitignore b/eval/.gitignore index d1ac9f241..8f1814dfa 100644 --- a/eval/.gitignore +++ b/eval/.gitignore @@ -14,3 +14,4 @@ build/ # Environment .env .venv/ +.venv diff --git a/eval/tests/bench_fixtures.py b/eval/tests/bench_fixtures.py new file mode 100644 index 000000000..fdd6c131f --- /dev/null +++ b/eval/tests/bench_fixtures.py @@ -0,0 +1,75 @@ +"""Shared row shapes for the sweep tests. + +Building the finalization tests turned up what a real scored review row must +carry: the report renders the whole review metric set, so an incomplete row +fails in string formatting rather than in the logic under test. That is a +property of the fixture, not of production - the shape lives here once so each +test does not rediscover it. +""" + +from __future__ import annotations + +from typing import Any + + +def scored_review_row(**overrides: Any) -> dict[str, Any]: + """One admissible review cell, with zero-valued metrics written out.""" + + row: dict[str, Any] = { + "ok": True, + "error_kind": None, + "error_detail": None, + "resolved": True, + "review_evidence_valid": True, + "review_score": {"weighted_f1": 0.5}, + "review_weighted_f1": 0.5, + "review_true_positives": 1, + "review_false_positives": 0, + "review_false_negatives": 0, + "review_precision": 0.5, + "review_recall": 0.5, + "review_f1": 0.5, + "review_weighted_precision": 0.5, + "review_weighted_recall": 0.5, + "review_blocker_recall": 1.0, + "review_severity_accuracy": 1.0, + "review_category_accuracy": 1.0, + "review_grounded_evidence": 1.0, + "review_verdict_correct": True, + "review_clean_control": True, + "review_clean_pass": True, + "transcript_missing": False, + "transcript_artifacts": [], + "num_turns": 3, + "duration_s": 1.0, + "cost_usd": 0.5, + "input_tokens": 1, + "output_tokens": 1, + "cache_creation_input_tokens": 0, + "cache_read_input_tokens": 0, + "diff_files": 0, + "diff_insertions": 0, + "diff_deletions": 0, + } + row.update(overrides) + return row + + +def unusable_review_row(**overrides: Any) -> dict[str, Any]: + """A cell that ran but produced evidence nothing can be scored from.""" + + # Merged into one mapping rather than passed as explicit keywords beside + # **overrides: Python rejects a duplicate keyword in the call expression + # itself, so unusable_review_row(error_kind=...) raised TypeError before + # scored_review_row could apply the override this helper advertises. + return scored_review_row( + **{ + "ok": False, + "resolved": False, + "review_evidence_valid": False, + "error_kind": "review-evidence-invalid", + "review_score": None, + "review_weighted_f1": None, + **overrides, + } + ) diff --git a/eval/tests/fixtures/fake_claude.py b/eval/tests/fixtures/fake_claude.py new file mode 100755 index 000000000..fbcedcdbe --- /dev/null +++ b/eval/tests/fixtures/fake_claude.py @@ -0,0 +1,175 @@ +#!/usr/bin/env python3 +"""A stand-in for the Claude Code CLI: real HTTP, real tool execution, real stream-json. + +Not a mock of the harness's own code. It does what the CLI does at the two +boundaries the harness depends on - it calls ANTHROPIC_BASE_URL for a turn, it +EXECUTES the tool blocks that come back, and it prints the stream-json event +sequence the parent parses. Only Write really executes - it is what produces the +review artifact, so the artifact path has to be genuine end to end. Skill is +MODELLED: it validates the request and returns a synthetic result, because the +parent's evidence gate keys on the request/result pair rather than on a skill +having loaded, and a fixture cannot load a real one. Bash is stubbed outright: +arbitrary shell from a scripted reply buys no fidelity for the paths this +exercises and plenty of ways to damage the host. Everything between +those boundaries (the sandbox, +the artifact capture, the scoring, the row) stays real, which is the whole +point: those are the layers that shipped bugs no unit test could see. + +Reads the prompt from stdin, as the real CLI does under "-p --input-format text". +""" + +from __future__ import annotations + +import json +import os +import pathlib +import sys +import urllib.request + + +def _turn(base_url: str, prompt: str) -> dict: + request = urllib.request.Request( + base_url.rstrip("/") + "/v1/messages", + data=json.dumps({"model": os.environ.get("ANTHROPIC_MODEL", "mock"), "max_tokens": 1024, + "messages": [{"role": "user", "content": prompt}]}).encode(), + headers={"Content-Type": "application/json", + "x-api-key": os.environ.get("ANTHROPIC_API_KEY", ""), + "anthropic-version": "2023-06-01"}, + ) + with urllib.request.urlopen(request, timeout=30) as response: + return json.load(response) + + +def _run_tool(name: str, params: dict) -> str: + """Write executes for real - it is what produces the review artifact. + + Skill and Bash do not: see the module docstring for which is modelled and + which is stubbed, and why neither can be genuine here. + """ + + if name == "Write": + target = pathlib.Path(params["file_path"]) + target.parent.mkdir(parents=True, exist_ok=True) + # Atomic, exactly as the real Write tool does it: temp file beside the + # target, then rename. This is the operation the read-only workspace + # boundary has to permit for the artifact directory and refuse for the + # workspace, so a stand-in that wrote in place would prove nothing. + staging = target.with_name(target.name + ".tmp.fake") + staging.write_text(params.get("content", "")) + os.replace(staging, target) + return f"wrote {target}" + if name == "Skill": + # Modelled explicitly rather than falling through to a generic success. + # The parent's evidence gate keys on a Skill request with a non-error + # result, so leaving this unimplemented let an unexecuted skill satisfy + # the gate - the gate would have been measuring the fixture, not a skill. + skill = params.get("skill") or params.get("command") or params.get("name") + if not skill: + raise NotImplementedError("Skill request carried no skill name") + return f"loaded skill {skill}" + if name == "Bash": + return "(bash suppressed in the stand-in)" + # An unsupported tool is a FAILED tool run, not a quiet success. Returning a + # plain string here made the parent's evidence gate read an unexecuted Skill + # request as a successful invocation. + raise NotImplementedError(f"unsupported tool {name}") + + +def main() -> int: + # stdin, because that is where the real CLI takes it under + # "-p --input-format text": the parent pipes prompt bytes in. Scanning argv + # for a non-flag token picks up a flag's VALUE instead ("text"), which is + # exactly what the prompt-fidelity test caught. + prompt = sys.stdin.read() + base_url = os.environ.get("ANTHROPIC_BASE_URL") + if not base_url: + print(json.dumps({"type": "result", "subtype": "error", "is_error": True, + "session_id": "fake-session", "num_turns": 0}), flush=True) + return 1 + + emit = lambda event: print(json.dumps(event), flush=True) # noqa: E731 + emit({"type": "system", "subtype": "init", "session_id": "fake-session"}) + + try: + message = _turn(base_url, prompt) + except (OSError, ValueError) as exc: + # A provider failure is a failed SESSION, not a crashed process: dying + # here leaves no terminal result event, so the parent reports a generic + # stream error instead of the upstream failure it actually saw. + emit({"type": "result", "subtype": "error", "is_error": True, + "session_id": "fake-session", "num_turns": 0, + "error": f"provider request failed: {type(exc).__name__}: {exc}"}) + return 1 + blocks = message.get("content", []) + emit({"type": "assistant", "message": {"role": "assistant", "content": blocks}}) + + tool_results = [] + for block in blocks: + if block.get("type") == "tool_use": + # A refused write is a tool ERROR the session reports and carries + # on from, not a crash. Letting it kill the process would lose the + # result event and misreport a working boundary as a broken run. + failed = False + try: + output = _run_tool(block["name"], block.get("input", {})) + except (OSError, NotImplementedError) as exc: + output, failed = f"error: {type(exc).__name__}: {exc}", True + # is_error is load-bearing: the parent treats an ABSENT is_error as + # success, so a refused or unsupported tool would otherwise be + # scored as a completed one. + tool_results.append({ + "type": "tool_result", "tool_use_id": block["id"], + "content": output, "is_error": failed, + }) + if tool_results: + emit({"type": "user", "message": {"role": "user", "content": tool_results}}) + + # Unknown is not zero. A reply carrying no usage used to become four + # zero-valued fields plus a fabricated cost, which the harness then treats + # as a real measurement - the exact confusion the accounting this fixture + # feeds exists to prevent. + usage = message.get("usage") + # Every field that gets forwarded is validated, not just the required two. + # The parent's well_formed check tests only that the four keys are PRESENT, + # so an unvalidated cache value rides into a success result and is recorded + # as a real measurement. A field good enough to report is good enough to + # check. + countable = lambda v: isinstance(v, int) and not isinstance(v, bool) and v >= 0 # noqa: E731 + if not isinstance(usage, dict) or not all( + countable(usage.get(f)) for f in ("input_tokens", "output_tokens") + ) or not all( + countable(usage[f]) + for f in ("cache_read_input_tokens", "cache_creation_input_tokens") + if f in usage + ): + emit({"type": "result", "subtype": "error", "is_error": True, + "session_id": "fake-session", "num_turns": 1, + "error": "provider reply carried no usable usage; refusing to report a measured run"}) + return 1 + emit({ + "type": "result", + "subtype": "success", + "is_error": False, + "session_id": "fake-session", + "num_turns": 1, + "duration_ms": 1200, + # A measured zero is not the same as unmeasured; the parent rejects a + # collapsed cost, so report a real one. + "total_cost_usd": 0.42, + # Forward exactly the fields the provider reported. Defaulting the + # absent ones to 0 fabricated a complete measurement out of an + # incomplete reply - and worse, it made the parent's own completeness + # check (runner_sessions.USAGE_FIELDS / well_formed) unfirable from any + # offline test, because the stand-in always satisfied it. + "usage": { + field: usage[field] + for field in ("input_tokens", "output_tokens", + "cache_read_input_tokens", "cache_creation_input_tokens") + if field in usage + }, + }) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/eval/tests/test_comparator_reuse.py b/eval/tests/test_comparator_reuse.py new file mode 100644 index 000000000..4425304b3 --- /dev/null +++ b/eval/tests/test_comparator_reuse.py @@ -0,0 +1,467 @@ +"""Comparator-row reuse: skip unchanged incumbent/CE cells, never candidates.""" + +from __future__ import annotations + +import hashlib +import os +from datetime import UTC, datetime, timedelta +from pathlib import Path + +import pytest + +from workflow_bench import comparator_reuse +from workflow_bench.comparator_reuse import ( + ComparatorReuseExpectation, + TaskReuseBinding, + materialize_reused_row, + row_is_reusable_comparator, + select_reusable_comparator_rows, +) +from workflow_bench.proposer_sandbox import SandboxError +from workflow_bench.runner_sessions import PARENT_EVENT_STREAM_SOURCE + + +requires_openat = pytest.mark.skipif( + os.open not in os.supports_dir_fd, + reason="comparator reuse resolves every artifact against a pinned directory descriptor", +) + + +def _digest(text: str = "blob") -> str: + return hashlib.sha256(text.encode()).hexdigest() + + +def _artifact(name: str = "session-1.jsonl", payload: bytes = b'{"type":"ok"}\n') -> dict: + return { + "path": f"transcripts/{name}", + "sha256": hashlib.sha256(payload).hexdigest(), + "bytes": len(payload), + "source": PARENT_EVENT_STREAM_SOURCE, + } + + +def _row(**overrides) -> dict: + base = { + "task": "review-pr-2718-defect", + "arm": "review", + "run": 0, + "ok": True, + "error_kind": None, + "model": "gpt-5.6-sol", + "benchmark_model": "gpt-5.6-sol", + "effort": "xhigh", + "sandbox_backend": "bwrap", + "task_base_sha": "a" * 40, + "task_prompt_digest": _digest("prompt"), + "oracle_digest": _digest("oracle"), + "oracle_command_digest": _digest("oracle-cmd"), + "oracle_manifest_digest": _digest("oracle-man"), + "skill_digest": _digest("skill"), + "candidate_overlay_digest": None, + "review_evidence_valid": True, + # Production sets this whenever the review source exists, which is the + # normal path for a valid review; the fixture predated the requirement. + "review_artifact": "review-pr-2718-defect-review-run0.review.json", + "review_score": {"weighted_f1": 0.4}, + "review_weighted_f1": 0.4, + "transcript_missing": False, + "transcript_artifacts": [_artifact()], + "recorded_at": datetime.now(UTC).isoformat(), + "runtime_digest": _digest("cli"), + "task_asset_manifest_digest": _digest("assets"), + "sandbox_dependency_manifest_digest": _digest("deps"), + } + base.update(overrides) + return base + + +def _expected(**overrides) -> ComparatorReuseExpectation: + now = datetime.now(UTC) + values = dict( + model="gpt-5.6-sol", + effort="xhigh", + sandbox_backend="bwrap", + runtime_digest=_digest("cli"), + now=now, + max_age=timedelta(days=90), + tasks={ + "review-pr-2718-defect": TaskReuseBinding( + task_base_sha="a" * 40, + task_prompt_digest=_digest("prompt"), + oracle_digest=_digest("oracle"), + oracle_command_digest=_digest("oracle-cmd"), + oracle_manifest_digest=_digest("oracle-man"), + task_asset_manifest_digest=_digest("assets"), + sandbox_dependency_manifest_digest=_digest("deps"), + ) + }, + skill_digests={"review": _digest("skill"), "ce_review": None}, + ce_plugin_version="3.24.0", + ce_plugin_manifest_digest=_digest("ce"), + ) + values.update(overrides) + return ComparatorReuseExpectation(**values) + + +def test_matching_incumbent_review_row_is_reusable() -> None: + assert row_is_reusable_comparator(_row(), _expected()) is True + + +def test_candidate_rows_are_never_reusable() -> None: + assert row_is_reusable_comparator(_row(arm="candidate_review"), _expected()) is False + + +def test_skill_digest_drift_rejects_reuse() -> None: + assert row_is_reusable_comparator(_row(), _expected(skill_digests={"review": _digest("other")})) is False + + +def test_excluded_or_failed_rows_are_not_reusable() -> None: + expected = _expected() + assert row_is_reusable_comparator(_row(error_kind="session-error", ok=False), expected) is False + assert row_is_reusable_comparator(_row(ok=False), expected) is False + assert row_is_reusable_comparator(_row(review_evidence_valid=False), expected) is False + assert row_is_reusable_comparator(_row(recorded_at=(datetime.now(UTC) - timedelta(days=91)).isoformat()), expected) is False + + +def test_runtime_digest_mismatch_rejects_when_both_sides_are_bound() -> None: + row = _row(runtime_digest=_digest("old-cli")) + assert row_is_reusable_comparator(row, _expected(runtime_digest=_digest("new-cli"))) is False + assert row_is_reusable_comparator(row, _expected(runtime_digest=_digest("old-cli"))) is True + # A row with no runtime_digest was measured by a harness that recorded none, + # which is the drift this lock exists to catch - not evidence of agreement. + assert row_is_reusable_comparator(_row(runtime_digest=None), _expected()) is False + # And a sweep that cannot determine its own digest must not reuse either. + assert row_is_reusable_comparator(_row(), _expected(runtime_digest=None)) is False + + +def test_ce_review_matches_plugin_digest_not_repo_skill() -> None: + row = _row( + arm="ce_review", + skill_digest=None, + ce_plugin_version="3.24.0", + ce_plugin_manifest_digest=_digest("ce"), + ) + assert row_is_reusable_comparator(row, _expected()) is True + assert ( + row_is_reusable_comparator(row, _expected(ce_plugin_manifest_digest=_digest("other"))) + is False + ) + + +def test_select_drops_conflicting_duplicates() -> None: + first = _row(review_weighted_f1=0.4) + second = _row(review_weighted_f1=0.9, recorded_at=datetime.now(UTC).isoformat()) + selected = select_reusable_comparator_rows([first, second], expected=_expected()) + assert selected == {} + same = select_reusable_comparator_rows([first, dict(first)], expected=_expected()) + assert ("review-pr-2718-defect", "review", 0) in same + + +@requires_openat +def test_materialize_copies_transcript_and_review_artifacts(tmp_path: Path) -> None: + payload = b'{"type":"result"}\n' + source = tmp_path / "prior" + dest = tmp_path / "fresh" + (source / "transcripts").mkdir(parents=True) + dest.mkdir() + transcript = source / "transcripts" / "session-1.jsonl" + transcript.write_bytes(payload) + transcript.chmod(0o600) + review = source / "review-pr-2718-defect-review-run0.review.json" + review.write_text('{"verdict":"comment"}\n') + patch = source / "review-pr-2718-defect-review-run0.patch" + patch.write_text("diff\n") + row = _row( + review_artifact=review.name, + transcript_artifacts=[_artifact(payload=payload)], + ) + + copied = materialize_reused_row(row, source_dir=source, dest_dir=dest) + + assert copied["reused"] is True + assert copied["reused_from_recorded_at"] == row["recorded_at"] + assert (dest / "transcripts" / "session-1.jsonl").read_bytes() == payload + assert (dest / review.name).read_text() == review.read_text() + assert (dest / patch.name).read_text() == "diff\n" + assert copied["transcript_artifacts"][0]["sha256"] == hashlib.sha256(payload).hexdigest() + + +@pytest.mark.skipif(os.name == "nt", reason="symlink creation may require elevated Windows privileges") +@requires_openat +def test_a_reused_artifact_is_copied_from_the_inode_that_was_checked(tmp_path: Path) -> None: + """The reuse source is a directory another sweep wrote and may still write. + + Validating a path and then re-opening it hands a concurrent writer the gap: + replace the checked file with a symlink and the copy follows it out of the + results directory. Swapping the path while the descriptor is held is that + same substitution, made deterministic. + """ + + (tmp_path / "transcript.jsonl").write_bytes(b"verified\n") + decoy = tmp_path / "decoy.jsonl" + decoy.write_bytes(b"substituted\n") + + with comparator_reuse._open_real_directory(tmp_path, label="reuse source") as dir_fd: + with comparator_reuse._open_regular("transcript.jsonl", dir_fd=dir_fd, label="transcript") as descriptor: + (tmp_path / "transcript.jsonl").unlink() + (tmp_path / "transcript.jsonl").symlink_to(decoy) + comparator_reuse._copy_owner_only(descriptor, "copy.jsonl", dir_fd=dir_fd) + + assert (tmp_path / "copy.jsonl").read_bytes() == b"verified\n" + with pytest.raises(SandboxError, match="regular non-symlink"): + with comparator_reuse._open_regular("transcript.jsonl", dir_fd=dir_fd, label="transcript"): + pass + + +@pytest.mark.skipif(os.name == "nt", reason="symlink creation may require elevated Windows privileges") +@requires_openat +def test_a_symlinked_transcripts_directory_is_refused_on_both_sides(tmp_path: Path) -> None: + """`O_NOFOLLOW` refuses the leaf, not the directory above it. + + A `transcripts` symlink on the source side makes reuse read a file outside + the results directory; one on the destination side writes the copy outside + this sweep's evidence. Neither is covered by the per-file guards that let + _resolved_directory tolerate a symlinked root. + """ + + payload = b'{"type":"result"}\n' + outside = tmp_path / "outside" + (outside / "transcripts").mkdir(parents=True) + (outside / "transcripts" / "session-1.jsonl").write_bytes(payload) + row = _row(transcript_artifacts=[_artifact(payload=payload)]) + + linked_source = tmp_path / "linked-source" + linked_source.mkdir() + (linked_source / "transcripts").symlink_to(outside / "transcripts", target_is_directory=True) + dest = tmp_path / "fresh" + dest.mkdir() + with pytest.raises(SandboxError, match="transcript source must be a real directory"): + materialize_reused_row(row, source_dir=linked_source, dest_dir=dest) + + source = tmp_path / "prior" + (source / "transcripts").mkdir(parents=True) + (source / "transcripts" / "session-1.jsonl").write_bytes(payload) + linked_dest = tmp_path / "linked-dest" + linked_dest.mkdir() + (linked_dest / "transcripts").symlink_to(outside / "transcripts", target_is_directory=True) + with pytest.raises(SandboxError, match="transcript destination must be a real directory"): + materialize_reused_row(row, source_dir=source, dest_dir=linked_dest) + + +@requires_openat +@pytest.mark.skipif(os.name == "nt", reason="symlink creation may require elevated Windows privileges") +def test_a_renamed_transcripts_directory_cannot_redirect_a_copy(tmp_path: Path) -> None: + """The directory is pinned, not re-walked from its name. + + An lstat that passed and a pathname used afterwards are two different + directories the moment a concurrent writer renames the first one away. This + performs exactly that substitution — rename, then leave a symlink in its + place — while the descriptor is held, which is what makes the race testable + without timing. + """ + + payload = b'{"type":"result"}\n' + results = tmp_path / "results" + transcripts = results / "transcripts" + transcripts.mkdir(parents=True) + (transcripts / "session-1.jsonl").write_bytes(payload) + outside = tmp_path / "outside" + outside.mkdir() + + with comparator_reuse._open_real_directory(results, label="reuse source") as root_fd: + with comparator_reuse._open_real_directory( + "transcripts", dir_fd=root_fd, label="transcript source" + ) as dir_fd: + transcripts.rename(results / "moved") + (results / "transcripts").symlink_to(outside, target_is_directory=True) + with comparator_reuse._open_regular( + "session-1.jsonl", dir_fd=dir_fd, label="transcript" + ) as artifact_fd: + comparator_reuse._copy_owner_only(artifact_fd, "copy.jsonl", dir_fd=dir_fd) + + assert (results / "moved" / "copy.jsonl").read_bytes() == payload + assert not (outside / "copy.jsonl").exists() + + +@requires_openat +def test_a_transcript_rewritten_mid_copy_is_refused_not_recorded(tmp_path: Path, monkeypatch) -> None: + """The digest has to describe the bytes that were written. + + A held descriptor stops the pathname being substituted; it does not stop the + inode being rewritten, and the prior sweep's directory is one this sweep + treats as concurrently writable. Hashing the source and then reading it + again to copy let the row keep the expected digest while the destination + held different bytes. + """ + + payload = b'{"type":"result"}\n' + source = tmp_path / "prior" + (source / "transcripts").mkdir(parents=True) + transcript = source / "transcripts" / "session-1.jsonl" + transcript.write_bytes(payload) + dest = tmp_path / "fresh" + dest.mkdir() + row = _row(transcript_artifacts=[_artifact(payload=payload)]) + + # Rewrite the inode in the window the copy reads through — same length, so + # only the digest can tell, which is the point. + real_read = comparator_reuse.os.read + rewritten = {"done": False} + + def rewrite_then_read(fd: int, size: int) -> bytes: + if not rewritten["done"]: + rewritten["done"] = True + with open(transcript, "r+b") as handle: + handle.write(b'{"type":"TAMPER"}') + return real_read(fd, size) + + monkeypatch.setattr(comparator_reuse.os, "read", rewrite_then_read) + with pytest.raises(SandboxError, match="drifted"): + materialize_reused_row(row, source_dir=source, dest_dir=dest) + monkeypatch.undo() + + # And nothing unvouched-for is left behind for the proposer to read. + assert not (dest / "transcripts" / "session-1.jsonl").exists() + + +@requires_openat +def test_materialize_rejects_same_directory_and_missing_transcript(tmp_path: Path) -> None: + source = tmp_path / "prior" + source.mkdir() + row = _row() + with pytest.raises(SandboxError, match="same results directory"): + materialize_reused_row(row, source_dir=source, dest_dir=source) + dest = tmp_path / "fresh" + dest.mkdir() + with pytest.raises(SandboxError, match="missing"): + materialize_reused_row(row, source_dir=source, dest_dir=dest) + + +@requires_openat +def test_a_reused_row_ages_from_its_first_measurement_not_the_copy(): + """Reuse chains must not refresh the clock. + + materialize_reused_row restamps recorded_at with the copy time, so aging + against that field let a row be copied forward every generation and outlive + max_age forever. The original measurement time is the one that counts. + """ + + original = (datetime.now(UTC) - timedelta(days=91)).isoformat() + chained = _row(recorded_at=datetime.now(UTC).isoformat(), reused_from_recorded_at=original) + assert row_is_reusable_comparator(chained, _expected()) is False + # The same row inside the window is still reusable. + fresh = _row( + recorded_at=datetime.now(UTC).isoformat(), + reused_from_recorded_at=(datetime.now(UTC) - timedelta(days=1)).isoformat(), + ) + assert row_is_reusable_comparator(fresh, _expected()) is True + + +def test_a_future_dated_row_is_corrupt_not_fresh(): + ahead = (datetime.now(UTC) + timedelta(days=2)).isoformat() + assert row_is_reusable_comparator(_row(recorded_at=ahead), _expected()) is False + + +def test_a_changed_sandbox_dependency_is_not_the_same_baseline(): + """The environment is part of the measurement. + + This branch itself changes `sandbox_dependencies` in the review corpus, so a + prior row measured against the old set is a measurement of a different + machine. Reusing it would compare a fresh candidate to a baseline built + somewhere else and hand the promotion gate a false comparison. + """ + + assert row_is_reusable_comparator( + _row(sandbox_dependency_manifest_digest=_digest("other-deps")), _expected() + ) is False + assert row_is_reusable_comparator( + _row(task_asset_manifest_digest=_digest("other-assets")), _expected() + ) is False + # A row that predates the field is not evidence of agreement either. + assert row_is_reusable_comparator(_row(sandbox_dependency_manifest_digest=None), _expected()) is False + + +@pytest.mark.skipif(os.name == "nt", reason="symlink creation may require elevated Windows privileges") +def test_reuse_directories_allow_a_symlinked_parent_but_not_a_symlinked_leaf(tmp_path: Path): + """Pins a deliberate difference from the sandbox's mount-root check. + + proposer_sandbox refuses every symlink hop because a hop changes what an + untrusted session is handed. A reuse directory is data, and every file + inside it is validated on its own, so a symlinked parent is allowed - + rejecting it would break a symlinked artifacts directory or macOS's /var + for no gain. The leaf itself must still be a real directory. + """ + + real = tmp_path / "real" + real.mkdir() + (real / "inner").mkdir() + linked_parent = tmp_path / "linked" + linked_parent.symlink_to(real, target_is_directory=True) + + # Reached through a symlinked parent: allowed, and resolved to the real path. + # The identity returned alongside it is what pins the root against a swap + # between the check and the open; the symlink policy itself is unchanged. + resolved, identity = comparator_reuse._resolved_directory(linked_parent / "inner", label="probe") + assert resolved == (real / "inner").resolve() + inner_stat = (real / "inner").stat() + assert identity == (inner_stat.st_dev, inner_stat.st_ino) + + # The leaf itself being a symlink is still refused. + with pytest.raises(SandboxError, match="must be a real directory"): + comparator_reuse._resolved_directory(linked_parent, label="probe") + + +def test_a_review_row_without_its_artifact_is_not_reusable() -> None: + """A score is a claim about evidence, not the evidence itself. + + materialize_reused_row copies the review artifact only when the row names + one, so accepting a row without it would carry a scored review forward with + nothing for a proposer to read. + """ + + row = _row() + assert row_is_reusable_comparator(row, _expected()) is True + without = {**row, "review_artifact": ""} + assert row_is_reusable_comparator(without, _expected()) is False + missing = {k: v for k, v in row.items() if k != "review_artifact"} + assert row_is_reusable_comparator(missing, _expected()) is False + + +@requires_openat +def test_a_reuse_root_replaced_after_the_check_is_refused(tmp_path: Path, monkeypatch) -> None: + """Check and use must name the same directory, not the same string. + + _resolved_directory lstats a name and the open re-walks that same name, so + a prior sweep that swaps its results root in between is opened somewhere + else. The leaf-symlink rule does not cover it - a replacement that is + itself a real directory passes every check the policy makes - and the + failure is silent, folding another directory's rows into this sweep's + comparator baseline. + """ + + original = tmp_path / "results" + original.mkdir() + resolved, stale_identity = comparator_reuse._resolved_directory(original, label="probe") + + # Replaced by a different REAL directory: the name still resolves and still + # passes the symlink policy, but it is not the inode that was checked. + original.rename(tmp_path / "moved") + original.mkdir() + assert comparator_reuse._resolved_directory(original, label="probe")[1] != stale_identity + + monkeypatch.setattr( + comparator_reuse, "_resolved_directory", lambda *_a, **_k: (resolved, stale_identity) + ) + with pytest.raises(SandboxError, match="replaced between the check and the open"): + with comparator_reuse._open_pinned_root(original, label="probe"): + pass + + +@requires_openat +def test_a_stable_reuse_root_opens_normally(tmp_path: Path) -> None: + """The guard rejects nothing that holds still - a directory matches itself.""" + + root = tmp_path / "results" + root.mkdir() + with comparator_reuse._open_pinned_root(root, label="probe") as fd: + assert os.fstat(fd).st_ino == root.stat().st_ino diff --git a/eval/tests/test_evolve.py b/eval/tests/test_evolve.py index c135584de..32d50fd85 100644 --- a/eval/tests/test_evolve.py +++ b/eval/tests/test_evolve.py @@ -9,19 +9,24 @@ import time from contextlib import contextmanager from datetime import UTC, datetime, timedelta from pathlib import Path +from types import SimpleNamespace import pytest from workflow_bench import evolve, evolution from workflow_bench.runner_sessions import PARENT_EVENT_STREAM_SOURCE from workflow_bench.evolve import ( + MIN_INSTANCE_SWEEP_SECONDS, build_parser, build_proposer_prompt, + capped_timeout_seconds, executed_benchmark_arms, generation_timeout_seconds, + instance_window_budget_seconds, load_jsonl, proposer_evidence_entries, read_learnings, + remaining_runtime_seconds, resolve_incumbent_arms, runner_argv, select_evidence, @@ -662,6 +667,127 @@ def test_run_proposer_hides_the_hidden_harness_and_keeps_the_full_tool_surface(m assert captured["settings_json"] == FakeSandbox.settings_json +def test_proposer_session_cannot_outlive_the_remaining_instance_window(monkeypatch, tmp_path): + """Clearing the sweep minimum is not a licence to run a full session. + + --timeout is sized for a whole generation, so a proposer started with the + minimum left would run far past --max-runtime-seconds and the box would take + the evidence with it. The budget is sampled after the clone, the sanitize + pass and the sandbox setup, because a reading taken before them is already + stale by the time the session it bounds actually starts. + """ + + captured: dict[str, object] = {} + # Pinned clock: real setup duration would make this assert on scheduling. + clock = {"now": 1000.0} + monkeypatch.setattr(evolve.time, "monotonic", lambda: clock["now"]) + setup_seconds = 100.0 + + @contextmanager + def fake_prepare_sandbox(**_kwargs): + yield SimpleNamespace( + claude_bin="claude", + command_prefix=[], + settings_json="{}", + transcript_projects=tmp_path / "transcript-projects", + ) + + def fake_run_claude(*_args, **kwargs): + captured.update(kwargs) + return {"ok": False, "error_kind": "session-error"} + + def slow_sanitize(_clone): + # Stands in for the clone, the sanitize pass and the sandbox build — + # all of which run between the caller's decision and the session. + clock["now"] += setup_seconds + return "0" * 40 + + monkeypatch.setattr(evolve.runner, "make_worktree", lambda _repo, _ref, destination: destination) + monkeypatch.setattr(evolve.runner, "remove_clone", lambda _clone: None) + monkeypatch.setattr(evolve, "sanitize_clone_for_hidden_oracles", slow_sanitize) + monkeypatch.setattr(evolve, "prepare_sandbox", fake_prepare_sandbox) + monkeypatch.setattr(evolve.runner, "run_claude", fake_run_claude) + args = build_parser().parse_args(["--tasks", "tasks.yaml", "--model", "model"]) + assert args.timeout > evolve.MIN_INSTANCE_SWEEP_SECONDS, "otherwise this test proves nothing" + + common = { + "overlay_dir": tmp_path / "overlay", + "proposal_path": tmp_path / "proposal.md", + "evidence_bundle": tmp_path / "evidence", + "bwrap_bin": tmp_path / "bwrap", + } + budget = evolve.MIN_INSTANCE_SWEEP_SECONDS + 1 + args.max_runtime_seconds = budget + started = clock["now"] + evolve.run_proposer("prompt", args, **common, started_monotonic=started) + # The setup time is charged, not handed back: a value sampled at `started` + # would have allowed the whole budget. + assert captured["timeout"] == budget - setup_seconds + + # No cap configured means no budget to overrun: the session keeps its own. + args.max_runtime_seconds = None + evolve.run_proposer("prompt", args, **common, started_monotonic=started) + assert captured["timeout"] == args.timeout + evolve.run_proposer("prompt", args, **common) + assert captured["timeout"] == args.timeout + + +def test_a_budget_spent_during_setup_stops_the_proposer_rather_than_buying_a_second( + monkeypatch, tmp_path +): + """An exhausted cap must end the generation, not start a one-second session. + + remaining_runtime_seconds floors at 0, and the call site wrapped it in + max(1, ...) - so a cap fully consumed by the clone, the sanitize pass and + the sandbox build produced a paid session with a one-second allowance + instead of stopping before the upload reserve the cap exists to protect. + """ + + captured: dict[str, object] = {} + clock = {"now": 1000.0} + monkeypatch.setattr(evolve.time, "monotonic", lambda: clock["now"]) + + @contextmanager + def fake_prepare_sandbox(**_kwargs): + yield SimpleNamespace( + claude_bin="claude", + command_prefix=[], + settings_json="{}", + transcript_projects=tmp_path / "transcript-projects", + ) + + def fake_run_claude(*_args, **kwargs): + captured.update(kwargs) + return {"ok": True} + + def setup_that_spends_the_whole_budget(_clone): + clock["now"] += budget + return "0" * 40 + + monkeypatch.setattr(evolve.runner, "make_worktree", lambda _repo, _ref, destination: destination) + monkeypatch.setattr(evolve.runner, "remove_clone", lambda _clone: None) + monkeypatch.setattr(evolve, "sanitize_clone_for_hidden_oracles", setup_that_spends_the_whole_budget) + monkeypatch.setattr(evolve, "prepare_sandbox", fake_prepare_sandbox) + monkeypatch.setattr(evolve.runner, "run_claude", fake_run_claude) + + args = build_parser().parse_args(["--tasks", "tasks.yaml", "--model", "model"]) + budget = evolve.MIN_INSTANCE_SWEEP_SECONDS + 1 + args.max_runtime_seconds = budget + record = evolve.run_proposer( + "prompt", + args, + overlay_dir=tmp_path / "overlay", + proposal_path=tmp_path / "proposal.md", + evidence_bundle=tmp_path / "evidence", + bwrap_bin=tmp_path / "bwrap", + started_monotonic=clock["now"], + ) + + assert not captured, "no session may start once the cap is exhausted" + assert record["ok"] is False + assert record["error_kind"] == "runtime-cap-exhausted" + + def test_parser_defaults_match_the_gate_minimums(): args = build_parser().parse_args(["--tasks", "t.yaml", "--model", "pinned"]) assert args.runs == 3 @@ -922,6 +1048,27 @@ def test_runner_argv_inserts_ce_review_for_review_overlay(tmp_path): ) arms = argv[argv.index("--arms") + 1 : argv.index("--promotion-metric")] assert arms == ["ce_review", "review", "candidate_review"] + assert "--reuse-results" not in argv + + +def test_runner_argv_forwards_prior_results_for_comparator_reuse(tmp_path): + args = build_parser().parse_args( + ["--tasks", "t.yaml", "--model", "pinned", "--arms", "review"] + ) + overlay = tmp_path / "overlay" + skill = overlay / ".claude" / "skills" / "gitnexus-review" / "SKILL.md" + skill.parent.mkdir(parents=True) + skill.write_text("candidate") + prior = tmp_path / "prior-bench" + argv = runner_argv( + args, + tmp_path / "bench", + overlay, + task_bindings=[{"id": "task"}], + target_base_digests={}, + reuse_results=prior, + ) + assert argv[argv.index("--reuse-results") + 1] == str(prior) def test_runner_argv_omits_proposer_for_manual_overlay(tmp_path): @@ -1118,6 +1265,161 @@ def test_generation_timeout_rejects_unknown_arm() -> None: ) +def test_instance_window_budget_leaves_upload_reserve() -> None: + # Friday 10:57 on a box that booted 02:45 Saturday-window: ~8.2h uptime. + leftover = instance_window_budget_seconds(8.2 * 3600) + assert leftover == int(86_400 - 8.2 * 3600 - 5_400) + assert leftover >= MIN_INSTANCE_SWEEP_SECONDS + with pytest.raises(ValueError, match="only .*s left"): + instance_window_budget_seconds(23.5 * 3600) + with pytest.raises(ValueError, match="uptime must be"): + instance_window_budget_seconds(float("nan")) + + + + +def test_capped_timeout_clamps_to_leftover_window(monkeypatch) -> None: + assert capped_timeout_seconds(10_000, None) == 10_000 + assert capped_timeout_seconds(10_000, 90) == 90 + with pytest.raises(ValueError, match="no time remains"): + capped_timeout_seconds(10_000, 0) + # Pinned clock: the helper is pure arithmetic, so a real elapsed-time window + # would assert on scheduling rather than on the behaviour under test. + monkeypatch.setattr(evolve.time, "monotonic", lambda: 1040.0) + started = 1000.0 + assert remaining_runtime_seconds(max_runtime_seconds=None, started_monotonic=started) is None + leftover = remaining_runtime_seconds(max_runtime_seconds=100, started_monotonic=started) + assert leftover is not None + assert leftover == 60 + + +def test_parser_rejects_non_positive_max_runtime() -> None: + with pytest.raises(SystemExit): + build_parser().parse_args( + ["--tasks", "t.yaml", "--model", "pinned", "--max-runtime-seconds", "0"] + ) + args = build_parser().parse_args( + ["--tasks", "t.yaml", "--model", "pinned", "--max-runtime-seconds", "7200"] + ) + assert args.max_runtime_seconds == 7200 + assert args.max_runtime_from_instance_window is False + + +def _task_file(tmp_path: Path) -> Path: + tasks = tmp_path / "tasks.yaml" + tasks.write_text( + """tasks: + - id: demo + class: test + repo: . + prompt: implement + verify: "true" + oracle: + command: "true" + files: + - source: hidden.test.ts + target: hidden.test.ts +""" + ) + return tasks + + +def _stub_main_preflight(monkeypatch, tmp_path) -> None: + """Everything main() shells out to before it reaches _run_generations.""" + + monkeypatch.setattr(evolve.runner, "selected_task_bindings", lambda _tasks: [{"id": "demo"}]) + monkeypatch.setattr(evolve, "preflight_bubblewrap", lambda: tmp_path / "bwrap") + monkeypatch.setattr(evolve, "require_claude_sandbox_helpers", lambda: None) + + +def test_the_runtime_cap_is_derived_where_its_clock_starts(monkeypatch, tmp_path, capsys) -> None: + """The budget and the clock it is measured against must be one instant. + + run-evolution.sh used to compute the budget in a separate `uv run python -c` + and pass a number, so the script's remaining provenance work and this + interpreter's startup were charged to the sweep — out of the upload reserve + the cap exists to protect. main() reads /proc/uptime itself now, next to its + own clock, so no interval exists to lose. + """ + + monkeypatch.setenv("EVENTBRIDGE_INSTANCE_WINDOW_SECONDS", "20000") + monkeypatch.setenv("EVENTBRIDGE_STOP_RESERVE_SECONDS", "1000") + monkeypatch.setattr(evolve, "read_instance_uptime_seconds", lambda: 3600.0) + captured: dict[str, object] = {} + + def record(args, **kwargs): + captured["max_runtime_seconds"] = args.max_runtime_seconds + captured["started_monotonic"] = kwargs["started_monotonic"] + return 0 + + monkeypatch.setattr(evolve, "_run_generations", record) + _stub_main_preflight(monkeypatch, tmp_path) + monkeypatch.setattr( + sys, + "argv", + [ + "evolve", + "--tasks", + str(_task_file(tmp_path)), + "--model", + "pinned", + "--out-root", + str(tmp_path / "out"), + "--max-runtime-from-instance-window", + ], + ) + + assert evolve.main() == 0 + + assert captured["max_runtime_seconds"] == 20000 - 3600 - 1000 + # Derived here, not passed in: the clock handed to the sweep is the one + # taken beside the uptime read. + assert isinstance(captured["started_monotonic"], float) + assert "capping the sweep to 15400s" in capsys.readouterr().out + + +def test_the_runtime_cap_refuses_two_sources_of_truth(monkeypatch, tmp_path) -> None: + monkeypatch.setattr(evolve, "read_instance_uptime_seconds", lambda: 3600.0) + monkeypatch.setattr( + sys, + "argv", + [ + "evolve", + "--tasks", + str(_task_file(tmp_path)), + "--model", + "pinned", + "--max-runtime-from-instance-window", + "--max-runtime-seconds", + "7200", + ], + ) + with pytest.raises(SystemExit): + evolve.main() + + +def test_the_runtime_cap_fails_closed_without_a_readable_uptime(monkeypatch, tmp_path) -> None: + def unreadable(): + raise ValueError("cannot read instance uptime from /proc/uptime") + + monkeypatch.setattr(evolve, "read_instance_uptime_seconds", unreadable) + monkeypatch.setattr( + sys, + "argv", + [ + "evolve", + "--tasks", + str(_task_file(tmp_path)), + "--model", + "pinned", + "--max-runtime-from-instance-window", + ], + ) + # Better to refuse than to run a box-stopped sweep believing it is uncapped. + with pytest.raises(SystemExit): + evolve.main() + + @pytest.mark.skipif(sys.platform != "linux", reason="Bubblewrap PID namespaces require Linux") def test_outer_runner_pid_namespace_kills_setsid_descendant(tmp_path): try: diff --git a/eval/tests/test_measure_evolution_cost.py b/eval/tests/test_measure_evolution_cost.py new file mode 100644 index 000000000..c6035bc86 --- /dev/null +++ b/eval/tests/test_measure_evolution_cost.py @@ -0,0 +1,122 @@ +"""Cost model for the evolution wall clock: measured cells, real schedules.""" + +from __future__ import annotations + +import pytest + +from workflow_bench.measure_evolution_cost import ( + CANDIDATE_ARM, + SHA_OVERHEAD_SECONDS, + DURATIONS_BY_ARM, + PROPOSER_SECONDS, + REVIEW_ARMS, + expected_task_seconds, + fed_makespan, + fed_pool_enabled, + generation_seconds, + graph_pipeline_enabled, + paid_arms, + task_cells, + wave_makespan, +) + + +def test_every_arm_has_its_own_unsorted_sample(): + assert set(DURATIONS_BY_ARM) == set(REVIEW_ARMS) + for arm, sample in DURATIONS_BY_ARM.items(): + assert len(sample) >= 10, arm + # Sorting would hand each task a uniform block and hide the variance + # the whole model exists to price. + assert list(sample) != sorted(sample), arm + assert PROPOSER_SECONDS > 0 + assert SHA_OVERHEAD_SECONDS > 0 + + +def test_weekly_reuse_pays_the_candidate_arm_only(): + assert paid_arms(weekly=True, reuse_enabled=True) == (CANDIDATE_ARM,) + assert paid_arms(weekly=False, reuse_enabled=True) == REVIEW_ARMS + assert paid_arms(weekly=True, reuse_enabled=False) == REVIEW_ARMS + + +def test_cells_are_submitted_run_major_arm_minor(): + # runner.py: [(run_idx, arm) for run_idx in range(runs) for arm in arms]. + # At workers=3 that puts one cell of each arm in every wave. + cells = task_cells(2, REVIEW_ARMS, 0) + assert len(cells) == 6 + expected = [DURATIONS_BY_ARM[arm][run] for run in range(2) for arm in REVIEW_ARMS] + assert cells == expected + + +def test_overhead_is_charged_per_sha_and_outside_the_pool(): + # Two properties at once: the residual sits outside the schedule, where more + # workers cannot dissolve it, and it scales with SHAs rather than cells. + assert task_cells(1, (CANDIDATE_ARM,), 0) == [DURATIONS_BY_ARM[CANDIDATE_ARM][0]] + wide = generation_seconds( + task_count=1, runs=3, arms=REVIEW_ARMS, workers=9, fed_pool=True, unique_shas=5 + ) + assert wide >= PROPOSER_SECONDS + 5 * SHA_OVERHEAD_SECONDS + + +def test_sweep_overhead_does_not_shrink_with_the_arm_count(): + """The bias that made weekly look cheaper than it is. + + A seeded weekly generation pays one arm instead of three but builds exactly + the same graphs. Charging the residual per cell billed it a third of a cost + the real sweep still pays; per SHA, the two attribute the same setup. + """ + + kwargs = dict(task_count=6, runs=3, workers=3, fed_pool=False, unique_shas=5) + weekly = generation_seconds(arms=(CANDIDATE_ARM,), **kwargs) + cold = generation_seconds(arms=REVIEW_ARMS, **kwargs) + weekly_sessions = 6 * expected_task_seconds(3, (CANDIDATE_ARM,), 3, fed_pool=False) + cold_sessions = 6 * expected_task_seconds(3, REVIEW_ARMS, 3, fed_pool=False) + # Whatever each wall is, the non-session part is identical. + assert round(weekly - weekly_sessions) == round(cold - cold_sessions) + # Cycling wraps, so a task can ask for more runs than the sample holds. + long_sample = task_cells(len(DURATIONS_BY_ARM[CANDIDATE_ARM]) + 2, (CANDIDATE_ARM,), 0) + assert len(long_sample) == len(DURATIONS_BY_ARM[CANDIDATE_ARM]) + 2 + + +def test_a_wave_costs_its_slowest_cell_and_a_fed_pool_does_not(): + slow = [10.0, 1.0, 1.0, 10.0, 1.0, 1.0] + assert wave_makespan(slow, 3) == 20.0 + # Fed: one worker takes the first 10; the second 10 lands on a worker that + # has already cleared a 1, and the remaining 1s fill the third. + assert fed_makespan(slow, 3) == 11.0 + assert fed_makespan(slow, 1) == wave_makespan(slow, 1) == 24.0 + + +def test_expected_task_seconds_is_alignment_averaged_and_deterministic(): + waved = expected_task_seconds(3, REVIEW_ARMS, 3, fed_pool=False) + assert waved == expected_task_seconds(3, REVIEW_ARMS, 3, fed_pool=False) + assert expected_task_seconds(0, REVIEW_ARMS, 3, fed_pool=False) == 0.0 + assert expected_task_seconds(3, (), 3, fed_pool=False) == 0.0 + # The barrier can only cost time, never save it. + assert waved >= expected_task_seconds(3, REVIEW_ARMS, 3, fed_pool=True) + + +def test_a_generation_pays_one_proposer_session_on_top_of_its_tasks(): + one = generation_seconds( + task_count=1, runs=3, arms=REVIEW_ARMS, workers=3, fed_pool=False, unique_shas=1 + ) + two = generation_seconds( + task_count=2, runs=3, arms=REVIEW_ARMS, workers=3, fed_pool=False, unique_shas=1 + ) + # Each extra task adds exactly one task's makespan. The proposer and the + # per-SHA sweep overhead are both paid once, not per task. + assert two - one == pytest.approx( + one - PROPOSER_SECONDS - SHA_OVERHEAD_SECONDS, abs=2.0 + ) + + +def test_feature_flags_read_the_runner_not_the_wish(): + assert graph_pipeline_enabled("def _run_sweep(): pass") == 0 + assert graph_pipeline_enabled("graph_prefetch = GraphPrefetch(...)") == 1 + assert fed_pool_enabled("def _run_wave(): pass") == 0 + assert fed_pool_enabled("def _run_fed_pool(): pass") == 1 + + +@pytest.mark.parametrize("workers", [1, 3, 8]) +def test_more_workers_never_lengthen_a_task(workers): + serial = expected_task_seconds(3, REVIEW_ARMS, 1, fed_pool=True) + assert expected_task_seconds(3, REVIEW_ARMS, workers, fed_pool=True) <= serial diff --git a/eval/tests/test_mock_provider.py b/eval/tests/test_mock_provider.py new file mode 100644 index 000000000..201845fbf --- /dev/null +++ b/eval/tests/test_mock_provider.py @@ -0,0 +1,280 @@ +"""The mock has to be right about the wire, or every test built on it lies.""" + +from __future__ import annotations + +import json +import os +import shutil +import subprocess +import urllib.request +from pathlib import Path + +import pytest + +from workflow_bench.mock_provider import MockProvider, Reply +from workflow_bench.provider_usage import ( + ANTHROPIC, + LITELLM_NORMALIZED, + OPENAI_RESPONSES, + normalize_usage, +) + + +def _post(url: str, payload: dict) -> tuple[int, bytes]: + request = urllib.request.Request( + url, data=json.dumps(payload).encode(), headers={"Content-Type": "application/json"} + ) + with urllib.request.urlopen(request, timeout=10) as response: + return response.status, response.read() + + +def test_anthropic_messages_returns_a_usable_message() -> None: + with MockProvider([Reply(text="reviewed")]) as provider: + _status, raw = _post(provider.base_url + "/v1/messages", {"model": "m", "messages": []}) + body = json.loads(raw) + assert body["role"] == "assistant" + assert body["content"][0]["text"] == "reviewed" + assert body["stop_reason"] == "end_turn" + + +def test_a_scripted_tool_call_is_carried_as_a_tool_use_block() -> None: + """Tool blocks are how a mocked run produces real artifacts. + + The CLI executes what it is asked to run, so a Write block makes it write + that file for real inside the sandbox - which is how an artifact-producing + cell can be exercised with no model involved. + """ + + write = {"name": "Write", "input": {"file_path": "/review-output/review-output.json", "content": "{}"}} + with MockProvider([Reply(text="writing", tools=[write])]) as provider: + _status, raw = _post(provider.base_url + "/v1/messages", {"model": "m", "messages": []}) + body = json.loads(raw) + block = body["content"][1] + assert block["type"] == "tool_use" and block["name"] == "Write" + assert block["input"]["file_path"] == "/review-output/review-output.json" + assert body["stop_reason"] == "tool_use", "a turn ending in a tool call must say so" + + +def test_streaming_emits_the_event_sequence_a_consumer_expects() -> None: + with MockProvider([Reply(text="hi")]) as provider: + request = urllib.request.Request( + provider.base_url + "/v1/messages", + data=json.dumps({"model": "m", "messages": [], "stream": True}).encode(), + headers={"Content-Type": "application/json"}, + ) + with urllib.request.urlopen(request, timeout=10) as response: + assert response.headers["Content-Type"] == "text/event-stream" + body = response.read().decode() + + events = [line[len("event: ") :] for line in body.splitlines() if line.startswith("event: ")] + assert events[0] == "message_start" + assert events[-1] == "message_stop" + assert "content_block_delta" in events + # message_delta carries the final usage, which is where output tokens land. + assert events[-2] == "message_delta" + + +def test_each_protocol_reports_usage_in_its_own_arithmetic() -> None: + """The whole point: the two providers count the same numbers differently. + + Anthropic's cache fields ADD to input_tokens; OpenAI's are SUBSETS of it. + Scripting one Reply and serving it both ways is what makes that asymmetry + testable without a paid request. + """ + + reply = Reply(input_tokens=2_000, output_tokens=300, cache_read_input_tokens=7_000, cache_creation_input_tokens=1_000) + + with MockProvider([reply, reply]) as provider: + _s, anthropic_raw = _post(provider.base_url + "/v1/messages", {"model": "m", "messages": []}) + _s, openai_raw = _post(provider.base_url + "/v1/responses", {"model": "m", "input": []}) + + anthropic = normalize_usage(ANTHROPIC, json.loads(anthropic_raw)["usage"]) + openai = normalize_usage(OPENAI_RESPONSES, json.loads(openai_raw)["usage"]) + + assert anthropic.total_input_tokens == 10_000 + assert openai.total_input_tokens == 10_000, "same billed work, stated as the whole" + assert anthropic.ordinary_input_tokens == 2_000 + assert openai.ordinary_input_tokens == 2_000, "recovered by subtraction, not addition" + assert openai.cache_read_input_tokens == 7_000 + + +def test_a_scripted_failure_is_returned_as_one() -> None: + """Billed failures are part of what the accounting must survive.""" + + with MockProvider([Reply(status_code=529, error_body={"error": {"type": "overloaded_error"}})]) as provider: + try: + _post(provider.base_url + "/v1/messages", {"model": "m", "messages": []}) + raise AssertionError("the scripted failure was not returned") + except urllib.error.HTTPError as exc: + assert exc.code == 529 + + +def test_requests_are_recorded_for_assertions() -> None: + with MockProvider() as provider: + _post(provider.base_url + "/v1/messages", {"model": "claude-sonnet-4-5", "messages": [{"role": "user"}]}) + assert len(provider.requests) == 1 + assert provider.requests[0].body["model"] == "claude-sonnet-4-5" + assert provider.requests[0].path.endswith("/v1/messages") + + +def test_an_unscripted_turn_gets_the_default_rather_than_stalling() -> None: + """A real run makes more calls than a test wants to enumerate.""" + + with MockProvider([Reply(text="first")], default=Reply(text="fallback")) as provider: + _s, one = _post(provider.base_url + "/v1/messages", {"model": "m", "messages": []}) + _s, two = _post(provider.base_url + "/v1/messages", {"model": "m", "messages": []}) + assert json.loads(one)["content"][0]["text"] == "first" + assert json.loads(two)["content"][0]["text"] == "fallback" + + +def test_a_request_through_the_real_gateway_records_native_usage(tmp_path, monkeypatch) -> None: + """The whole stack minus the model: proxy, translation, callback, log. + + This is the path that shipped three separate defects invisible to unit + tests - the usage variable never reaching the proxy subprocess, the + callback failing to import when loaded by path, and failures never + recorded. All three live between the gateway and the provider, which is + exactly the span this exercises. + """ + + + import yaml + + from workflow_bench import model_gateway + from workflow_bench.model_gateway import OpenAIGateway + from workflow_bench.provider_usage import USAGE_LOG_ENV_VAR + + if shutil.which("litellm") is None: + import pytest + + pytest.skip("litellm console script absent; the proxy cannot start here") + + usage_log = tmp_path / "provider_usage.jsonl" + monkeypatch.setenv(USAGE_LOG_ENV_VAR, str(usage_log)) + + reply = Reply(input_tokens=2_000, output_tokens=300, cache_read_input_tokens=7_000, cache_creation_input_tokens=1_000) + with MockProvider(default=reply) as provider: + original = model_gateway.write_openai_litellm_config + + def config(path, names): + original(path, names) + document = yaml.safe_load(path.read_text()) + for entry in document["model_list"]: + entry["litellm_params"]["api_base"] = f"{provider.base_url}/v1" + path.write_text(yaml.safe_dump(document)) + return path + + monkeypatch.setattr(model_gateway, "write_openai_litellm_config", config) + with OpenAIGateway( + openai_api_key="mock-key", model_names=["gpt-4.1"], work_dir=tmp_path / "gw", ready_timeout_s=60 + ) as gateway: + request = urllib.request.Request( + gateway.base_url + "/v1/messages", + data=json.dumps({"model": "gpt-4.1", "max_tokens": 32, "messages": [{"role": "user", "content": "ping"}]}).encode(), + headers={"Content-Type": "application/json", "x-api-key": gateway.auth_token, "anthropic-version": "2023-06-01"}, + ) + with urllib.request.urlopen(request, timeout=60): + pass + + assert usage_log.exists(), "the callback never wrote - the env did not reach the proxy" + events = [json.loads(line) for line in usage_log.read_text().splitlines()] + assert events, "the proxy started but recorded nothing" + event = events[-1] + native = event["native_usage"] + # LiteLLM hands a callback its OWN normalised object, not the upstream body: + # an OpenAI Responses reply arrives as prompt_tokens / prompt_tokens_details. + # Asserting the wire shape here is what proved the shipped adapter read keys + # that are never present. + assert native["prompt_tokens_details"]["cached_tokens"] == 7_000 + assert native["prompt_tokens_details"]["cache_write_tokens"] == 1_000 + assert event["provider"] == LITELLM_NORMALIZED + assert event["call_type"] == "anthropic_messages", "the observed call type, not a Responses one" + + usage = normalize_usage(event["provider"], native) + assert usage.total_input_tokens == 10_000 + assert usage.cache_read_input_tokens == 7_000 + assert usage.cache_write_input_tokens == 1_000 + assert usage.ordinary_input_tokens == 2_000 + assert usage.complete, "a run that cannot interpret its own usage measured nothing" + + +def test_probe_what_identity_the_real_cli_actually_sends(tmp_path: Path) -> None: + """An experiment, not an assertion: which fields could correlate a request to a cell? + + Per-cell usage attribution is unbuilt because one proxy serves the whole + sweep, so anything read from the proxy environment is identical for every + request. Attribution needs something that travels WITH the request, and + what the Claude Code CLI actually sends is not documented anywhere I can + check - guessing it is how the last three accounting bugs happened. + + So this drives the REAL pinned CLI against the mock and prints the + identity-bearing fields that arrive. It asserts only that a request was + made; the value is the recorded evidence, which the job log preserves. + """ + + claude = os.environ.get("CLAUDE_CANARY_BIN") + if not claude or not Path(claude).exists(): + pytest.skip("no pinned Claude CLI here; the containment job supplies CLAUDE_CANARY_BIN") + + with MockProvider(default=Reply(text="ok")) as provider: + subprocess.run( + [claude, "-p", "--input-format", "text", "--output-format", "stream-json", "--verbose"], + input=b"say ok", + capture_output=True, + timeout=120, + env={ + **os.environ, + "ANTHROPIC_BASE_URL": provider.base_url, + "ANTHROPIC_API_KEY": "offline-probe", + "HOME": str(tmp_path), + }, + ) + + assert provider.requests, "the real CLI never reached the mock provider" + request = provider.requests[0] + interesting = { + "header:" + name: value + for name, value in request.headers.items() + if any(k in name.lower() for k in ("session", "user", "trace", "request-id", "conversation", "metadata")) + } + interesting.update( + {f"body:{key}": request.body[key] for key in ("metadata", "user", "session_id") if key in request.body} + ) + print("\nIDENTITY FIELDS THE REAL CLI SENDS:") + print(" body keys:", sorted(request.body)) + print(" candidate correlators:", interesting or "NONE — per-cell attribution needs another mechanism") + + +def test_scripted_tools_survive_the_responses_protocol_too() -> None: + """The gateway uses Responses BECAUSE it carries tool use. + + Emitting only output_text there meant a scripted Write or Skill crossed the + gateway with the tool dropped, so a mock claiming to serve both protocols + was wrong about the one the gateway actually runs. + """ + + write = {"name": "Write", "input": {"file_path": "/review-output/review-output.json", "content": "{}"}} + with MockProvider([Reply(text="writing", tools=[write])]) as provider: + _status, raw = _post(provider.base_url + "/v1/responses", {"model": "m", "input": []}) + + output = json.loads(raw)["output"] + calls = [item for item in output if item["type"] == "function_call"] + assert len(calls) == 1, "the scripted tool must cross the Responses path" + assert calls[0]["name"] == "Write" + assert json.loads(calls[0]["arguments"])["file_path"] == "/review-output/review-output.json" + + +def test_an_omitted_cache_field_stays_omitted_on_the_responses_wire_too() -> None: + """Absence must survive both protocols, not just the Anthropic one. + + `_int_or_none` reads an absent detail key as unknown and a present 0 as a + measured zero, so serializing 0 for a scripted None would claim a + measurement the reply never made. + """ + + with MockProvider([Reply(input_tokens=2_000, cache_read_input_tokens=None)]) as provider: + _status, raw = _post(provider.base_url + "/v1/responses", {"model": "m", "input": []}) + + details = json.loads(raw)["usage"]["input_tokens_details"] + assert "cached_tokens" not in details, "an omitted field must not serialize as a measured zero" + assert details["cache_write_tokens"] == 0, "a scripted 0 is still a real measurement" diff --git a/eval/tests/test_offline_session_integration.py b/eval/tests/test_offline_session_integration.py new file mode 100644 index 000000000..cb5d1fe8f --- /dev/null +++ b/eval/tests/test_offline_session_integration.py @@ -0,0 +1,195 @@ +"""A session end to end with only the model faked. + +The layers between the CLI and the row are where this harness has actually +shipped bugs - the artifact that could not be written, the usage that was never +recorded, the evidence that was scored from the wrong directory. Every one of +them sat below the level its tests exercised. These run the real session path +against a scripted provider, so the only thing not real is what the model says. +""" + +from __future__ import annotations + +import sys +from pathlib import Path + +import pytest + +from workflow_bench.mock_provider import MockProvider, Reply +from workflow_bench.proposer_sandbox import ( + host_workspace_write_boundary, + prepare_review_workspace, + prepare_sandbox, +) +from workflow_bench.review_scoring import REVIEW_OUTPUT, parse_review_output +from workflow_bench.runner_sessions import run_claude + +FAKE_CLI = Path(__file__).parent / "fixtures" / "fake_claude.py" +REVIEW_JSON = '{"schema_version": 1, "verdict": "approve", "findings": []}' + + +def _session(clone: Path, provider: MockProvider, **overrides): + return run_claude( + "review the change", + clone, + claude_bin=str(FAKE_CLI), + timeout=60, + env={ + "ANTHROPIC_BASE_URL": provider.base_url, + "ANTHROPIC_API_KEY": "offline", + "PATH": "/usr/bin:/bin", + }, + **overrides, + ) + + +@pytest.fixture +def clone(tmp_path: Path) -> Path: + workspace = tmp_path / "clone" + workspace.mkdir() + (workspace / "source.ts").write_text("export const answer = 42;\n") + return workspace + + +def test_a_session_records_the_usage_the_provider_reported(clone: Path) -> None: + """Token counts must survive the CLI boundary, not be invented after it.""" + + reply = Reply(input_tokens=2_000, output_tokens=300, cache_read_input_tokens=7_000, cache_creation_input_tokens=1_000) + with MockProvider(default=reply) as provider: + record = _session(clone, provider) + + assert record["ok"] is True, record.get("error_detail") + assert record["input_tokens"] == 2_000 + assert record["cache_read_input_tokens"] == 7_000 + assert record["cache_creation_input_tokens"] == 1_000 + assert record["output_tokens"] == 300 + # A measured zero would be indistinguishable from an unmeasured one. + assert record["cost_usd"] == 0.42 + assert record["num_turns"] == 1 + + +def test_a_scripted_write_produces_a_review_artifact_the_scorer_accepts(clone: Path) -> None: + """The full artifact path: model asks, CLI writes atomically, scorer reads. + + This is the operation that shipped empty for a whole run. Nothing here + fakes the write, the directory, or the parse - only the decision to write. + """ + + with prepare_sandbox( + clone=clone, claude_bin=Path(sys.executable), backend="host-unsafe", preflight=False + ) as sandbox: + artifact = prepare_review_workspace(sandbox, REVIEW_OUTPUT) + write = {"name": "Write", "input": {"file_path": str(artifact), "content": REVIEW_JSON}} + with MockProvider(default=Reply(text="reviewing", tools=[write])) as provider: + # Take the command configuration from the sandbox the way run_arm + # does, rather than calling run_claude bare. On host-unsafe the + # prefix is [] by construction, so this pins the WIRING, not the + # isolation - a bwrap run would carry a real prefix through here. + record = _session( + clone, + provider, + command_prefix=sandbox.command_prefix_for(), + require_pid_namespace=sandbox.require_pid_namespace, + ) + assert record["ok"] is True, record.get("error_detail") + # Read inside the scope: prepare_sandbox removes the private root on exit. + verdict, findings = parse_review_output(artifact) + + assert verdict == "approve" + assert findings == () + + +def test_the_provider_saw_the_prompt_the_harness_meant_to_send(clone: Path) -> None: + """A run that measures the wrong prompt measures nothing.""" + + with MockProvider() as provider: + _session(clone, provider) + + assert provider.requests, "the session never reached the provider" + sent = provider.requests[0].body["messages"][0]["content"] + assert "review the change" in sent + + +def test_a_provider_failure_surfaces_as_a_failed_session_not_a_silent_pass(clone: Path) -> None: + """An upstream 529 must not be recorded as a usable measurement.""" + + failing = Reply(status_code=529, error_body={"error": {"type": "overloaded_error"}}) + with MockProvider(default=failing) as provider: + record = _session(clone, provider) + + assert record["ok"] is False + assert record["error_kind"] is not None + + +def test_the_write_boundary_refuses_the_workspace_and_permits_the_artifact(clone: Path, tmp_path: Path) -> None: + """The contract the empty-artifact run violated, on the backend available here. + + A review must not change the workspace, and must still be able to write its + artifact ATOMICALLY - temp file beside the target, then rename - which is + what needs a writable parent DIRECTORY rather than a writable file. Both + halves are asserted through the real session, with the real boundary + applied, and the model scripted to attempt each one. + + Scope: this is the host-unsafe boundary, which its own docstring calls + best-effort because a session that can chmod can undo it. The kernel-enforced + version is bubblewrap's --ro-bind, which needs namespaces this machine cannot + create; that half stays with the real-sandbox canary in CI. + """ + + artifacts = tmp_path / "artifacts" + artifacts.mkdir() + target = artifacts / REVIEW_OUTPUT + protected = clone / "source.ts" + before = protected.read_text() + + write_artifact = {"name": "Write", "input": {"file_path": str(target), "content": REVIEW_JSON}} + tamper = {"name": "Write", "input": {"file_path": str(protected), "content": "tampered"}} + + # No writable= entry: the boundary only governs paths INSIDE the workspace + # (it refuses one that escapes), and the artifact directory deliberately + # lives outside it - that relocation is the fix for the empty-artifact run. + with host_workspace_write_boundary(clone): + with MockProvider(default=Reply(text="writing", tools=[write_artifact, tamper])) as provider: + record = _session(clone, provider) + + assert record["ok"] is True, record.get("error_detail") + # The artifact landed, written the way the agent's Write tool does it. + verdict, _findings = parse_review_output(target) + assert verdict == "approve" + assert not list(artifacts.glob("*.tmp.*")), "the rename landed rather than a copy" + # The workspace did not move. + assert protected.read_text() == before, "the read-only workspace was modified" + + +def test_a_reply_missing_cache_usage_is_refused_not_zero_filled(clone: Path) -> None: + """An omitted cache field must not arrive as a measured zero. + + The parent already demands all four USAGE_FIELDS before it calls a session + measured (runner_sessions.well_formed). The stand-in used to default the + absent ones to 0, which both fabricated a complete measurement AND made + that parent guard unfirable from any offline test - it was always + satisfied. Scripting the absence is what proves the guard still fires. + """ + + partial = Reply(input_tokens=2_000, output_tokens=300, cache_read_input_tokens=None) + with MockProvider(default=partial) as provider: + record = _session(clone, provider) + + assert record["ok"] is False, "an incomplete usage report is not a usable measurement" + assert record["error_kind"] == "session-error" + + +@pytest.mark.parametrize("bad", [-5, True, "1200"], ids=["negative", "boolean", "string"]) +def test_a_nonsense_cache_value_is_refused_rather_than_forwarded(clone: Path, bad: object) -> None: + """A field good enough to report is good enough to validate. + + The parent's well_formed check tests only that the four keys are PRESENT, + so an unvalidated cache value would ride into a success result and be + recorded as a real measurement. + """ + + reply = Reply(input_tokens=2_000, output_tokens=300) + object.__setattr__(reply, "cache_read_input_tokens", bad) + with MockProvider(default=reply) as provider: + record = _session(clone, provider) + + assert record["ok"] is False, f"{bad!r} must not be recorded as a measured cache value" diff --git a/eval/tests/test_offline_sweep_integration.py b/eval/tests/test_offline_sweep_integration.py new file mode 100644 index 000000000..7ffb6db25 --- /dev/null +++ b/eval/tests/test_offline_sweep_integration.py @@ -0,0 +1,348 @@ +"""A whole sweep, offline: real runner, real sessions, scripted model. + +The layers between a model turn and a promotion decision had never been +exercised together. Unit tests covered each in isolation and the paid runs that +would have covered the composition kept dying, so the contracts BETWEEN them +went unverified - and that is where this harness has repeatedly shipped bugs. + +This drives runner.main() the way the workflow does. Everything is real: task +selection, hidden-oracle capture, the sandbox, the CLI subprocess, artifact +capture, review scoring against the oracle, aggregation, the health guard, and +the promotion gate. Only the model is scripted, through MockProvider. + +Two provisioning steps are stubbed because this environment cannot supply them, +and neither is harness logic: the pinned gitnexus runtime mounts (no +node_modules in a worktree) and the sanitized graph build (needs the gitnexus +CLI at a mounted path). Containment is host-unsafe here; bubblewrap stays with +the real-sandbox canary in the containment job. +""" + +from __future__ import annotations + +import json +import re +import subprocess +import sys +from pathlib import Path +from types import SimpleNamespace + +import os +import shutil + +import pytest + +from workflow_bench import oracle_assets, runner +from workflow_bench.mock_provider import MockProvider, Reply + +FAKE_CLI = Path(__file__).parent / "fixtures" / "fake_claude.py" +ARMS = ("ce_review", "review", "candidate_review") + +# When set, the sweep runs with NOTHING provisioning-stubbed: real bubblewrap +# containment, the real pinned runtime mounts, and the real sanitized graph +# build. The named CI job installs all three, so a missing one there is a +# regression rather than an unsupported machine - it FAILS instead of quietly +# degrading to the stubbed path, which is the whole point of the gate. +FULL_SWEEP_ENV = "GITNEXUS_REQUIRE_FULL_SWEEP" +FULL_SWEEP = os.environ.get(FULL_SWEEP_ENV) == "1" + +# The runner refuses --unsafe-no-bwrap whenever CI is set, because that mode runs +# sessions with bypassPermissions behind a boundary its own docstring calls "not +# a security boundary". Deleting CI to get past that refusal would run an +# uncontained agent sweep on the runner holding the checkout and credentials, so +# the stubbed path is skipped under CI instead. The containment job sets +# GITNEXUS_REQUIRE_FULL_SWEEP=1 and takes the real bubblewrap path, so CI keeps +# its coverage; only the uncontained convenience run is given up. +pytestmark = pytest.mark.skipif( + not FULL_SWEEP and bool(os.environ.get("CI")), + reason="an uncontained sweep must not run in CI; the containment job runs it with GITNEXUS_REQUIRE_FULL_SWEEP=1", +) + +# The review output and the hidden labels are DELIBERATELY different shapes - +# the labels carry line_start/line_end and no recommendation. Only a real run +# surfaces that; it is why these are written out rather than shared. +FINDING = { + "id": "f1", "severity": "high", "category": "correctness", "path": "src/sum.js", + "line": 1, "end_line": 1, "blocking": True, "scenario": "review-defect", + "evidence": "export const total = (a, b) => a - b;", "recommendation": "use a + b", +} +LABEL = {"id": "f1", "severity": "high", "category": "correctness", + "path": "src/sum.js", "line_start": 1, "line_end": 1} +SECOND_LABEL = {"id": "f2", "severity": "high", "category": "correctness", + "path": "src/scale.js", "line_start": 1, "line_end": 1} +SECOND_FINDING = { + "id": "f2", "severity": "high", "category": "correctness", "path": "src/scale.js", + "line": 1, "end_line": 1, "blocking": True, "scenario": "review-defect", + "evidence": "export const twice = (n) => n + 2;", "recommendation": "use n * 2", +} + + +def _git(repo: Path, *args: str) -> str: + return subprocess.run(["git", "-C", str(repo), *args], check=True, + capture_output=True, text=True).stdout.strip() + + +@pytest.fixture +def bench(tmp_path: Path): + """A self-contained corpus: one repo, one task, one hidden label.""" + + repo = tmp_path / "repo" + (repo / "src").mkdir(parents=True) + (repo / "src" / "sum.js").write_text("export const total = (a, b) => a - b;\n") + (repo / "src" / "scale.js").write_text("export const twice = (n) => n + 2;\n") + _git(repo, "init", "-q", ".") + _git(repo, "config", "user.email", "t@t") + _git(repo, "config", "user.name", "t") + _git(repo, "add", "-A") + _git(repo, "commit", "-q", "-m", "fixture") + sha = _git(repo, "rev-parse", "HEAD") + + oracles = tmp_path / "oracles" + oracles.mkdir() + (oracles / "review-fixture-defect.labels.json").write_text( + json.dumps({"schema_version": 1, "findings": [LABEL]}) + ) + (oracles / "review-fixture-second.labels.json").write_text( + json.dumps({"schema_version": 1, "findings": [SECOND_LABEL]}) + ) + + tasks = tmp_path / "tasks.yaml" + tasks.write_text( + "tasks:\n" + " - id: review-fixture-defect\n" + " class: review-defect\n" + f" repo: {repo}\n" + f" ref: {sha}\n" + " prompt: Review this change and report actionable defects.\n" + ' verify: test -s "$GITNEXUS_BENCH_REVIEW_OUTPUT"\n' + " oracle:\n" + ' command: test -s "$GITNEXUS_BENCH_REVIEW_OUTPUT"\n' + " files: [{ source: review-fixture-defect.labels.json, target: review-labels.json }]\n" + # A SECOND task, because the thing a cross-task scheduler changes is + # invisible with one: waves are per-task, so a single task cannot show + # ordering, packing, or a breaker that spans a task boundary. + " - id: review-fixture-second\n" + " class: review-defect\n" + f" repo: {repo}\n" + f" ref: {sha}\n" + " prompt: Review the scaling helper and report actionable defects.\n" + ' verify: test -s "$GITNEXUS_BENCH_REVIEW_OUTPUT"\n' + " oracle:\n" + ' command: test -s "$GITNEXUS_BENCH_REVIEW_OUTPUT"\n' + " files: [{ source: review-fixture-second.labels.json, target: review-labels.json }]\n" + ) + + plugin = tmp_path / "ce-plugin" + (plugin / ".claude-plugin").mkdir(parents=True) + (plugin / ".claude-plugin" / "plugin.json").write_text( + json.dumps({"name": "compound-engineering", "version": "0.0.0-fixture"}) + ) + for skill in ("ce-plan", "ce-work", "ce-code-review"): + directory = plugin / "skills" / skill + directory.mkdir(parents=True) + (directory / "SKILL.md").write_text(f"---\nname: {skill}\ndescription: fixture\n---\nFixture.\n") + + overlay = tmp_path / "overlay" / ".claude" / "skills" / "gitnexus-review" + overlay.mkdir(parents=True) + (overlay / "SKILL.md").write_text("---\nname: gitnexus-review\ndescription: fixture\n---\nCandidate.\n") + + return SimpleNamespace(tasks=tasks, oracles=oracles, plugin=plugin, + overlay=tmp_path / "overlay", out=tmp_path / "out") + + +def _stub_provisioning(monkeypatch: pytest.MonkeyPatch) -> None: + """Replace what this machine cannot supply - and nothing else. + + Under FULL_SWEEP nothing is replaced: the runtime mounts and the graph are + built for real, so the sweep exercises containment and provisioning too. + """ + + if FULL_SWEEP: + if shutil.which("bwrap") is None: + pytest.fail(f"{FULL_SWEEP_ENV}=1 but bubblewrap is absent") + return + + monkeypatch.setattr(runner, "trusted_gitnexus_runtime_mounts", lambda: ()) + + def materialize(worktree, *, sanitized_head=None, **_kwargs): + # The one-clone registry guard reads this before any session runs. + meta = Path(worktree) / ".gitnexus" + meta.mkdir(parents=True, exist_ok=True) + (meta / "meta.json").write_text( + json.dumps({"indexedAt": "2026-09-08T00:00:00Z", "lastCommit": sanitized_head or "0" * 40}) + ) + + def fake_graph(**kwargs): + kwargs["env"].graph_snapshots[kwargs["graph_key"]] = SimpleNamespace( + digest="fixture-graph", manifest_digest="fixture-graph-manifest", + dependency_content_digest=None, dependency_manifest_digest=None, + materialize=materialize, + ) + + monkeypatch.setattr(runner, "ensure_task_graph", fake_graph) + + +def _sweep(bench, monkeypatch: pytest.MonkeyPatch, findings: list[dict], verdict: str, *, invoke_skill: bool = True): + """Run the real CLI against a model scripted to return `findings`.""" + + _stub_provisioning(monkeypatch) + monkeypatch.setattr( + oracle_assets, "ORACLE_ROOT", bench.oracles, raising=False + ) + monkeypatch.setattr( + runner, "capture_task_oracles", + lambda tasks, root=bench.oracles: oracle_assets.capture_task_oracles(tasks, root=root), + ) + def review_for(body: str) -> str: + # Per task: the second task's defect is in another file, so replying + # with the first task's finding would score it wrong. A cross-task + # scheduler makes which task a request belongs to load-bearing. + chosen = findings + if findings and "scaling helper" in body: + chosen = [SECOND_FINDING if f is FINDING else f for f in findings] + return json.dumps({"schema_version": 1, "verdict": verdict, "findings": chosen}) + + class Scripted(MockProvider): + def next_reply(self) -> Reply: + body = json.dumps(self.requests[-1].body if self.requests else {}) + target = re.search(r"(/[^\s\"']*review-output\.json)", body) + skill = re.search(r"\b(gitnexus-review|ce-code-review)\b", body) + return Reply( + text="reviewing", + tools=[ + # The evidence gate needs a Skill request with a non-error + # result: a review that never invoked its skill measured the + # model, not the skill. + *([{"name": "Skill", "input": {"skill": skill.group(1) if skill else "gitnexus-review"}}] + if invoke_skill else []), + {"name": "Write", "input": { + "file_path": target.group(1) if target else str(bench.out / "unmatched-review-output.json"), + "content": review_for(body)}}, + ], + input_tokens=2_000, output_tokens=300, + cache_read_input_tokens=7_000, cache_creation_input_tokens=1_000, + ) + + with Scripted() as provider: + monkeypatch.setattr(sys, "argv", [ + "runner", "--tasks", str(bench.tasks), "--arms", *ARMS, + "--runs", "1", "--workers", "1", "--out", str(bench.out), + "--base-url", provider.base_url, "--anthropic-api-key", "offline", + "--claude-bin", str(FAKE_CLI), + *([] if FULL_SWEEP else ["--unsafe-no-bwrap"]), + "--model", "mock-model", + "--ce-plugin-dir", str(bench.plugin), "--ce-plugin-version", "0.0.0-fixture", + "--candidate-overlay", str(bench.overlay), + ]) + + try: + code = runner.main() + except SystemExit as exc: + code = exc.code + + rows = [json.loads(line) for line in (bench.out / "results.jsonl").read_text().splitlines()] + return code, rows, provider + + +def _row(rows: list[dict], arm: str, task: str = "review-fixture-defect") -> dict: + return next(r for r in rows if r["arm"] == arm and r["task"] == task) + + +def test_a_correct_review_scores_and_the_sweep_exits_clean(bench, monkeypatch) -> None: + """The whole path, green: every arm measured, scored, and accounted for.""" + + code, rows, provider = _sweep(bench, monkeypatch, [FINDING], "request_changes") + + assert code in (None, 0), f"sweep did not succeed: {code}" + tasks = {"review-fixture-defect", "review-fixture-second"} + assert len(rows) == len(ARMS) * len(tasks) + assert len(provider.requests) == len(ARMS) * len(tasks), "each cell must reach the provider once" + assert {r["task"] for r in rows} == tasks, "both tasks must have run" + + row = _row(rows, "review") + assert row["ok"] is True and row["resolved"] is True + assert row["skill_invoked"] is True + assert (row["review_true_positives"], row["review_false_positives"], row["review_false_negatives"]) == (1, 0, 0) + assert row["review_f1"] == 1.0 + # The provider's own numbers survived the CLI, the parser and the row. + assert row["cache_read_input_tokens"] == 7_000 + assert row["input_tokens"] == 2_000 + + # Each task scored against ITS OWN oracle. This is what a cross-task + # scheduler puts at risk: interleaving cells from different tasks means a + # mis-routed context or artifact scores one task against another's labels, + # and both would still look "green" per row. + second = _row(rows, "review", task="review-fixture-second") + assert second["resolved"] is True and second["review_f1"] == 1.0 + assert second["review_artifact"] == "review-fixture-second-review-run0.review.json" + + for name in ("results.jsonl", "report.md", "promotion.json"): + assert (bench.out / name).is_file(), f"{name} was not written" + assert (bench.out / "review-fixture-defect-review-run0.review.json").is_file() + + +def test_one_run_cannot_promote_a_candidate(bench, monkeypatch) -> None: + """The gate refuses on insufficient paired runs, and says so.""" + + _sweep(bench, monkeypatch, [FINDING], "request_changes") + promotion = json.loads((bench.out / "promotion.json").read_text()) + + assert promotion["run_status"] == "complete" + decision = next(d for d in promotion["decisions"] if d["candidate_arm"] == "candidate_review") + assert decision["decision"] == "insufficient_evidence" + assert any("valid paired runs" in reason for reason in decision["reasons"]) + + +def test_a_finding_in_the_wrong_place_scores_zero_but_stays_valid_evidence(bench, monkeypatch) -> None: + """Being wrong is a quality result, not a broken measurement. + + The negative control that makes the passing case mean something: same + harness, same well-formed artifact, only the answer changed. + """ + + wrong = {**FINDING, "path": "src/WRONG.js", "line": 99, "end_line": 99} + _code, rows, _provider = _sweep(bench, monkeypatch, [wrong], "request_changes") + + row = _row(rows, "review") + assert (row["review_true_positives"], row["review_false_positives"], row["review_false_negatives"]) == (0, 1, 1) + assert row["review_f1"] == 0.0 + assert row["resolved"] is False + assert row["error_kind"] == "oracle-failed", "a wrong answer is not a session or evidence failure" + assert row["review_evidence_valid"] is True, "the artifact was well formed; only the answer was wrong" + + +def test_approving_defective_code_is_a_miss_with_no_false_positive(bench, monkeypatch) -> None: + """The other half of the control: silence scores differently from a wrong guess.""" + + _code, rows, _provider = _sweep(bench, monkeypatch, [], "approve") + + row = _row(rows, "review") + assert (row["review_true_positives"], row["review_false_positives"], row["review_false_negatives"]) == (0, 0, 1) + assert row["review_precision"] is None, "precision is undefined with no predictions, not zero" + assert row["review_verdict_correct"] is False, "approving defective code is the wrong verdict" + assert row["review_evidence_valid"] is True + + +def test_a_review_that_never_invoked_its_skill_is_not_a_measurement(bench, monkeypatch) -> None: + """The gate that separates measuring a SKILL from measuring a model. + + Added because a mutation exposed it: forcing skill_was_invoked_events to + return True left every other test here passing, so nothing pinned the gate. + The artifact is written and correct in this run - only the skill request is + missing - so a pass would mean the arm scored a review it never performed. + """ + + code, rows, _provider = _sweep(bench, monkeypatch, [FINDING], "request_changes", invoke_skill=False) + + row = _row(rows, "review") + assert row["skill_invoked"] is False + assert row["error_kind"] == "skill-not-invoked" + assert code not in (None, 0), "the sweep must not report success on unusable evidence" + + # The row still carries its own score - the artifact was well formed - and + # aggregate() DOES count it in the arm's quality median (the KNOWN GAP noted + # above aggregate(); test_workflow_bench pins the resulting 0.5). Filtering + # it out of the median alone inverted a promotion, because valid_runs and + # excluded_runs kept counting it. It counts for cost either way: the session + # ran and was billed. + assert row["review_weighted_f1"] == 1.0 + assert row["review_evidence_valid"] is True diff --git a/eval/tests/test_process_control.py b/eval/tests/test_process_control.py index 8d7405356..1184feea0 100644 --- a/eval/tests/test_process_control.py +++ b/eval/tests/test_process_control.py @@ -79,11 +79,11 @@ def run(index, arm): assert Path({str(assets)!r}).exists(), 'assets removed while a worker was active' return {{'resolved': False, 'error_kind': result.state}} with cancellation_scope(handle_signals=True) as event: - streak, stopped=sweep_task_cells([(i, 'review') for i in range(10)], workers=2, run=run, + streak, tripped=sweep_task_cells([(i, 'review') for i in range(10)], workers=2, run=run, on_start=lambda *args: None, on_record=lambda i,a,r: rows.append([i,r]), outage_streak=0, outage_limit=5, cancel_event=event) Path({str(assets)!r}).unlink() -print(json.dumps({{'rows': rows, 'stopped': stopped}})) +print(json.dumps({{'rows': rows, 'stopped': event.is_set(), 'tripped': tripped}})) """ process = subprocess.Popen( [PYTHON, "-c", script], @@ -104,7 +104,8 @@ print(json.dumps({{'rows': rows, 'stopped': stopped}})) assert process.returncode == 0, stderr assert time.monotonic() - started < 15 report = json.loads(stdout) - assert report["stopped"] and [row[0] for row in report["rows"]] == [0, 1] + assert report["stopped"] and not report["tripped"], "cancelled, not an outage" + assert [row[0] for row in report["rows"]] == [0, 1] assert report["rows"][0][1]["resolved"] is True assert report["rows"][1][1]["error_kind"] == "cancelled" with pytest.raises(ProcessLookupError): diff --git a/eval/tests/test_proposer_sandbox.py b/eval/tests/test_proposer_sandbox.py index 2e30cf6a2..26532a5f0 100644 --- a/eval/tests/test_proposer_sandbox.py +++ b/eval/tests/test_proposer_sandbox.py @@ -33,9 +33,11 @@ from workflow_bench.proposer_sandbox import ( SANDBOX_GIT_EXCLUDES, VITE_TEMP_DIR, SANDBOX_PATH, + SANDBOX_REVIEW_OUTPUT, SANDBOX_PYTHON3, SANDBOX_SHELL_PREFIX, SANDBOX_USER_SKILLS, + SANDBOX_WORKSPACE, ReadOnlyMount, SandboxError, _runtime_mount_args, @@ -43,46 +45,65 @@ from workflow_bench.proposer_sandbox import ( build_sandbox_environment, _force_rmtree, host_workspace_write_boundary, + prepare_review_workspace, prepare_sandbox, preflight_bubblewrap, sandbox_workspace_write_boundary, stage_evidence_bundle, stage_task_assets, ) +from workflow_bench.review_scoring import REVIEW_OUTPUT, parse_review_output from workflow_bench.task_assets import TaskAssetCache, stage_task_assets as stage_immutable_task_assets -@pytest.mark.parametrize("entry", ["file", "directory", "relative-link", "absolute-link"]) -def test_review_preparation_rejects_existing_output_without_touching_target(tmp_path, entry): +@pytest.mark.parametrize("entry", ["directory", "relative-link", "absolute-link"]) +def test_review_preparation_rejects_a_reused_artifact_directory(tmp_path, entry): clone = tmp_path / "clone" clone.mkdir() sentinel = tmp_path / "sentinel" sentinel.write_text("must survive") - output = clone / "review-output.json" - if entry == "file": - output.write_text("existing result") - elif entry == "directory": - output.mkdir() - else: - output.symlink_to(sentinel if entry == "absolute-link" else "../sentinel") with prepare_sandbox(clone=clone, claude_bin=sys.executable, backend="host-unsafe") as sandbox: + stale = proposer_sandbox.review_output_path(sandbox, "review-output.json").parent + if entry == "directory": + stale.mkdir() + (stale / "review-output.json").write_text("a previous cell's verdict") + else: + # relpath, not a hand-written "../sentinel": stale is + # /review-output, which is nowhere near tmp_path, so the + # literal produced a dangling link and the assertion below proved nothing. + stale.symlink_to( + sentinel if entry == "absolute-link" else Path(os.path.relpath(sentinel, stale.parent)) + ) with pytest.raises(SandboxError, match="already exists"): proposer_sandbox.prepare_review_workspace(sandbox, "review-output.json") assert sentinel.read_text() == "must survive" - if entry == "file": - assert output.read_text() == "existing result" - if "link" in entry: - assert output.is_symlink() -def test_review_preparation_creates_a_private_regular_output(tmp_path): +def test_review_preparation_leaves_a_clone_entry_of_the_same_name_alone(tmp_path): + # The artifact no longer lives in the workspace, so a file that happens to + # share its name is just one of the repository's own files. + clone = tmp_path / "clone" + clone.mkdir() + (clone / "review-output.json").write_text("repository content") + with prepare_sandbox(clone=clone, claude_bin=sys.executable, backend="host-unsafe") as sandbox: + output = proposer_sandbox.prepare_review_workspace(sandbox, "review-output.json") + assert (clone / "review-output.json").read_text() == "repository content" + assert clone not in output.parents + + +def test_review_preparation_creates_a_private_directory_and_not_the_file(tmp_path): clone = tmp_path / "clone" clone.mkdir() with prepare_sandbox(clone=clone, claude_bin=sys.executable, backend="host-unsafe") as sandbox: output = proposer_sandbox.prepare_review_workspace(sandbox, "review-output.json") - assert output.read_bytes() == b"" - assert stat.S_ISREG(output.lstat().st_mode) - assert stat.S_IMODE(output.stat().st_mode) == 0o600 + assert output == proposer_sandbox.review_output_path(sandbox, "review-output.json") + # The DIRECTORY is what has to exist and be writable: the agent writes + # a temp file beside the target and renames it. + assert output.parent.is_dir() + assert stat.S_IMODE(output.parent.stat().st_mode) == 0o700 + # The file is deliberately absent — absence is how "never written" is + # told apart from "written badly". + assert not output.exists() def test_review_preparation_preserves_existing_runtime_files_and_tracks_only_created_paths(tmp_path): @@ -165,6 +186,15 @@ def test_unsafe_host_session_translates_virtual_paths_and_disables_containment(t assert sandbox.require_pid_namespace is False assert sandbox.host_path("/workspace/review-output.json") == str(clone / "review-output.json") assert sandbox.host_path("/evidence/selected-rows.json") == str(evidence / "selected-rows.json") + # The review artifact left the workspace, so the host-unsafe backend has + # to translate its new home too. Untranslated, the review prompt names a + # path that exists on neither backend and the cell writes nothing. + assert sandbox.host_path( + f"{proposer_sandbox.SANDBOX_REVIEW_OUTPUT}/review-output.json" + ) == str(proposer_sandbox.review_output_path(sandbox, "review-output.json")) + assert sandbox.host_text( + f"write {proposer_sandbox.SANDBOX_REVIEW_OUTPUT}/review-output.json" + ) == f"write {proposer_sandbox.review_output_path(sandbox, 'review-output.json')}" assert sandbox.host_text("read /evidence and write /workspace/out") == ( f"read {evidence} and write {clone}/out" ) @@ -867,9 +897,8 @@ def test_read_only_review_workspace_exposes_only_one_writable_artifact(tmp_path: clone.mkdir() source = clone / "source.ts" source.write_text("trusted\n") - output = clone / "review-output.json" - output.write_text("") script = """ +import os from pathlib import Path try: Path('/workspace/source.ts').write_text('tampered') @@ -877,15 +906,24 @@ except OSError: pass else: raise SystemExit('review source remained writable') -Path('/workspace/review-output.json').write_text('{"schema_version":1}') +# Write the way the agent's Write tool does: a temp file beside the target, +# then rename. Writing in place would pass against the mount shape that +# shipped every artifact empty, which is the regression this canary exists for. +target = Path('/review-output/review-output.json') +staging = target.with_name(target.name + '.tmp.1.abc') +staging.write_text('{"schema_version":1}') +os.replace(staging, target) """ with prepare_sandbox(clone=clone, claude_bin=Path(sys.executable), preflight=True) as sandbox: + output = proposer_sandbox.prepare_review_workspace(sandbox, "review-output.json") result = run_managed( [ *sandbox.command_prefix_for( read_only_workspace=True, extra_writable_mounts=( - ReadOnlyMount(source=output, target="/workspace/review-output.json"), + ReadOnlyMount( + source=output.parent, target=proposer_sandbox.SANDBOX_REVIEW_OUTPUT + ), ), ), "/usr/bin/python3", @@ -897,9 +935,14 @@ Path('/workspace/review-output.json').write_text('{"schema_version":1}') require_pid_namespace=True, ) - assert result.ok, result.stderr_tail - assert source.read_text() == "trusted\n" - assert output.read_text() == '{"schema_version":1}' + # Inside the sandbox scope: the artifact now lives under the session's + # private root, which prepare_sandbox removes on exit. run_arm reads it + # here too, while the session is still alive. + assert result.ok, result.stderr_tail + assert source.read_text() == "trusted\n" + assert output.read_text() == '{"schema_version":1}' + # The staging file is gone: the rename landed rather than a copy. + assert list(output.parent.iterdir()) == [output] @pytest.mark.skipif(os.name == "nt", reason="symlink creation may require elevated Windows privileges") @@ -1170,6 +1213,7 @@ for line in sys.stdin: review_command = """test -z "${ANTHROPIC_API_KEY:-}" && python3 - <<'PY' import json +import os import subprocess from pathlib import Path source = Path('/workspace/canary.txt') @@ -1182,7 +1226,10 @@ for operation in (lambda: source.write_text('forbidden'), lambda: source.rename( pass else: raise AssertionError('source mutation was allowed') -Path('/workspace/review-output.json').write_text(json.dumps({'schema_version': 1, 'verdict': 'approve', 'findings': []})) +target = Path('/review-output/review-output.json') +staging = target.with_name(target.name + '.tmp.1.abc') +staging.write_text(json.dumps({'schema_version': 1, 'verdict': 'approve', 'findings': []})) +os.replace(staging, target) PY""" if review_layout: for command in ( @@ -1359,7 +1406,11 @@ PY""" sandbox, command_prefix=sandbox.command_prefix_for( read_only_workspace=True, - extra_writable_mounts=(ReadOnlyMount(output, "/workspace/review-output.json"),), + extra_writable_mounts=( + ReadOnlyMount( + output.parent, proposer_sandbox.SANDBOX_REVIEW_OUTPUT + ), + ), ), ) result = sandbox.run( @@ -1408,7 +1459,7 @@ PY""" assert (sandbox.temp / "mcp-called").read_text() == "ok" if review_layout: assert output is not None - runner_artifacts.enforce_phase_workspace(clone, before, allowed_artifact=output) + runner_artifacts.enforce_phase_workspace(clone, before, allowed_artifact=None) assert json.loads(output.read_text())["verdict"] == "approve" assert (clone / "canary.txt").read_text() == "hook-readable\nchanged for review\n" finally: @@ -1418,3 +1469,98 @@ PY""" if not review_layout: assert (clone / "bash-called").read_text() == "canary" + + +def test_review_artifact_binds_a_writable_directory_outside_the_workspace(tmp_path): + """The bwrap argv, since the mount shape is the whole bug. + + bwrap cannot create a mount point inside an already-read-only bind, so a + writable path has to live outside /workspace — and it has to be the + directory, or the agent has nowhere to put the temp file it renames into + place. + """ + + clone = tmp_path / "clone" + clone.mkdir() + with prepare_sandbox(clone=clone, claude_bin=sys.executable, backend="host-unsafe") as session: + sandbox = replace(session, backend="bwrap") + output = proposer_sandbox.review_output_path(sandbox, "review-output.json") + output.parent.mkdir(mode=0o700) + argv = sandbox.command_prefix_for( + read_only_workspace=True, + extra_writable_mounts=( + proposer_sandbox.ReadOnlyMount( + source=output.parent, + target=proposer_sandbox.SANDBOX_REVIEW_OUTPUT, + ), + ), + ) + + target = proposer_sandbox.SANDBOX_REVIEW_OUTPUT + assert not target.startswith(proposer_sandbox.SANDBOX_WORKSPACE + "/") + # The workspace itself is bound read-only... + workspace_at = argv.index(proposer_sandbox.SANDBOX_WORKSPACE) + assert argv[workspace_at - 2] == "--ro-bind" + # ...and the artifact directory is bound writable, as a directory. + artifact_at = argv.index(target) + assert argv[artifact_at - 2] == "--bind" + assert Path(argv[artifact_at - 1]) == output.parent + assert Path(argv[artifact_at - 1]).is_dir() + assert f"{proposer_sandbox.SANDBOX_WORKSPACE}/review-output.json" not in argv + + +@pytest.mark.skipif( + os.environ.get("GITNEXUS_REQUIRE_BWRAP_CANARY") != "1", + reason="real Bubblewrap canary is mandatory in the named Ubuntu CI job", +) +def test_real_bubblewrap_lets_a_review_artifact_be_written_atomically(tmp_path: Path) -> None: + """The filesystem contract the EROFS defect broke, under a real sandbox. + + Argv assertions cannot establish this. The artifact came back empty because + an atomic write - temp file beside the target, then rename - needs a + WRITABLE PARENT DIRECTORY, and only a real bwrap invocation shows whether + the mount grants one. A deterministic writer stands in for the agent: no + model session, no credentials. + + Scope: this proves the filesystem and process contract of the production + mount configuration. It does not establish that a particular agent CLI's + own file-access policy permits the same operation - that is a second, + independent gate. + """ + + clone = tmp_path / "clone" + clone.mkdir() + (clone / "tracked.txt").write_text("original\n") + + with prepare_sandbox(clone=clone, claude_bin=Path(sys.executable), preflight=True) as sandbox: + review_output = prepare_review_workspace(sandbox, REVIEW_OUTPUT) + # The production configuration, not a hand-built mount tuple: the same + # command_prefix_for call run_arm makes for a review cell. + prefix = sandbox.command_prefix_for( + read_only_workspace=True, + extra_writable_mounts=( + ReadOnlyMount(source=review_output.parent, target=SANDBOX_REVIEW_OUTPUT), + ), + ) + target = f"{SANDBOX_REVIEW_OUTPUT}/{REVIEW_OUTPUT}" + script = ( + # 1. temp file beside the destination, then atomic rename over it. + f'printf %s \'{{"schema_version": 1, "verdict": "approve", "findings": []}}\' > {target}.tmp && ' + f"mv {target}.tmp {target} && " + # 2. the workspace must refuse the write that the mount forbids. + f"(printf x >> {SANDBOX_WORKSPACE}/tracked.txt 2>/dev/null && echo WORKSPACE-WRITABLE || echo workspace-readonly)" + ) + result = subprocess.run( + [*prefix, "/bin/sh", "-c", script], + capture_output=True, text=True, timeout=60, check=False, + ) + + assert result.returncode == 0, f"atomic write failed inside the sandbox: {result.stderr[-400:]}" + assert "workspace-readonly" in result.stdout, "the workspace must stay read-only" + assert (clone / "tracked.txt").read_text() == "original\n", "the clone was modified" + + # Read while the session is alive: the artifact lives under the private + # root that prepare_sandbox removes on exit, which is also why run_arm + # consumes it before leaving the scope. + _verdict, findings = parse_review_output(review_output) + assert findings == () diff --git a/eval/tests/test_provider_usage.py b/eval/tests/test_provider_usage.py new file mode 100644 index 000000000..c8e74a966 --- /dev/null +++ b/eval/tests/test_provider_usage.py @@ -0,0 +1,132 @@ +"""The two providers' accounting equations, encoded literally. + +Adding OpenAI's cache fields to its input_tokens double-counts, because they are +subsets of it. Subtracting Anthropic's under-counts, because they are additional +categories. A single generic struct cannot be right for both, so these tests +pin each equation rather than the field names. +""" + +from __future__ import annotations + +import pytest + +from workflow_bench.provider_usage import ( + ANTHROPIC, + OPENAI_RESPONSES, + UsageSemanticsError, + normalize_usage, +) + + +def _openai(input_tokens: int, cached: int | None = None, cache_write: int | None = None) -> dict: + details: dict[str, int] = {} + if cached is not None: + details["cached_tokens"] = cached + if cache_write is not None: + details["cache_write_tokens"] = cache_write + return { + "input_tokens": input_tokens, + "input_tokens_details": details, + "output_tokens": 300, + "output_tokens_details": {"reasoning_tokens": 250}, + } + + +def test_openai_uncached_request_is_all_ordinary_input() -> None: + usage = normalize_usage(OPENAI_RESPONSES, _openai(1000, cached=0, cache_write=0)) + assert usage.ordinary_input_tokens == 1000 + assert usage.total_input_tokens == 1000 + assert (usage.cache_read_input_tokens, usage.cache_write_input_tokens) == (0, 0) + + +def test_openai_cache_creation_keeps_the_parts_summing_to_input_tokens() -> None: + """The subsets must reconstruct the whole, never exceed it.""" + + usage = normalize_usage(OPENAI_RESPONSES, _openai(1000, cached=0, cache_write=400)) + assert usage.ordinary_input_tokens == 600 + assert ( + usage.ordinary_input_tokens + + usage.cache_read_input_tokens + + usage.cache_write_input_tokens + == usage.total_input_tokens + ) + + +def test_openai_cache_hit_plus_new_write_uses_the_documented_subtraction() -> None: + usage = normalize_usage(OPENAI_RESPONSES, _openai(10_000, cached=7_000, cache_write=1_000)) + assert usage.ordinary_input_tokens == 2_000 + assert usage.total_input_tokens == 10_000, "input_tokens is the whole, not a component" + + +def test_openai_reasoning_tokens_decompose_output_rather_than_adding_to_it() -> None: + usage = normalize_usage(OPENAI_RESPONSES, _openai(100, cached=0, cache_write=0)) + assert usage.output_tokens == 300 + assert usage.reasoning_output_tokens == 250 + assert usage.reasoning_output_tokens <= usage.output_tokens + + +def test_anthropic_uncached_total_is_just_input_tokens() -> None: + usage = normalize_usage( + ANTHROPIC, + {"input_tokens": 1000, "cache_creation_input_tokens": 0, + "cache_read_input_tokens": 0, "output_tokens": 200}, + ) + assert usage.total_input_tokens == 1000 + assert usage.ordinary_input_tokens == 1000 + + +def test_anthropic_cached_total_adds_the_cache_categories() -> None: + """The opposite equation to OpenAI's, on deliberately identical numbers.""" + + usage = normalize_usage( + ANTHROPIC, + {"input_tokens": 2_000, "cache_creation_input_tokens": 1_000, + "cache_read_input_tokens": 7_000, "output_tokens": 200}, + ) + assert usage.total_input_tokens == 10_000 + assert usage.ordinary_input_tokens == 2_000 + + +def test_the_same_numbers_mean_different_totals_on_the_two_providers() -> None: + """The whole reason a shared struct is unsafe, in one assertion.""" + + openai = normalize_usage(OPENAI_RESPONSES, _openai(10_000, cached=7_000, cache_write=1_000)) + anthropic = normalize_usage( + ANTHROPIC, + {"input_tokens": 10_000, "cache_creation_input_tokens": 1_000, + "cache_read_input_tokens": 7_000, "output_tokens": 300}, + ) + assert openai.total_input_tokens == 10_000 + assert anthropic.total_input_tokens == 18_000 + assert openai.ordinary_input_tokens == 2_000 + assert anthropic.ordinary_input_tokens == 10_000 + + +def test_missing_native_cache_fields_are_unknown_and_never_zero() -> None: + """A zero we invented is indistinguishable from a zero the provider reported.""" + + usage = normalize_usage(OPENAI_RESPONSES, {"input_tokens": 1000, "output_tokens": 10}) + assert usage.cache_read_input_tokens is None + assert usage.cache_write_input_tokens is None + assert usage.ordinary_input_tokens is None, "cannot subtract what was never reported" + assert usage.total_input_tokens == 1000 + assert not usage.complete + assert "cache_read_input_tokens" in usage.unknown_fields + + +def test_an_absent_usage_object_is_entirely_unknown() -> None: + usage = normalize_usage(ANTHROPIC, None) + assert not usage.complete + assert usage.total_input_tokens is None + + +def test_an_unknown_provider_is_refused_rather_than_guessed() -> None: + with pytest.raises(UsageSemanticsError, match="refusing to guess"): + normalize_usage("some-new-provider", {"input_tokens": 1}) + + +def test_cache_subsets_larger_than_the_whole_are_rejected() -> None: + """Nonsense arithmetic must surface, not silently produce a negative.""" + + with pytest.raises(UsageSemanticsError, match="exceed input_tokens"): + normalize_usage(OPENAI_RESPONSES, _openai(100, cached=90, cache_write=50)) diff --git a/eval/tests/test_provider_usage_capture.py b/eval/tests/test_provider_usage_capture.py new file mode 100644 index 000000000..91e4956e5 --- /dev/null +++ b/eval/tests/test_provider_usage_capture.py @@ -0,0 +1,327 @@ +"""What the proxy writes must outlive the translation that follows it. + +Claude Code receives an Anthropic-shaped response, which has nowhere to put +OpenAI's cached_tokens, cache_write_tokens or reasoning_tokens. If those are not +captured before the translation, the only remaining record of them is a bill. +""" + +from __future__ import annotations + +import contextlib +import json +from pathlib import Path +from types import SimpleNamespace + +import pytest + +from workflow_bench import litellm_usage_callback, provider_usage +from workflow_bench.litellm_usage_callback import USAGE_LOG_ENV_VAR, ProviderUsageLogger +from workflow_bench.model_gateway import ( + OpenAIGateway, + USAGE_CALLBACK_MODULE, + openai_litellm_config, + write_openai_litellm_config, +) +from workflow_bench.provider_usage import ( + ANTHROPIC, + LITELLM_NORMALIZED, + USAGE_ENV_VARS, + normalize_usage, +) + + +class _Usage: + """Stands in for the provider usage model LiteLLM hands the callback.""" + + def __init__(self, payload: dict) -> None: + self._payload = payload + + def model_dump(self) -> dict: + return self._payload + + +def _openai_response(usage: dict) -> SimpleNamespace: + return SimpleNamespace( + id="resp_68f2c1", + # The model that actually answered, which is not the role the caller asked for. + model="gpt-5.6-sol-2026-08-01", + usage=_Usage(usage), + ) + + +# The shape a callback actually receives: LiteLLM normalises usage into its own +# Chat-Completions-style object before any logger sees it, so an OpenAI reply +# arrives as prompt_tokens / prompt_tokens_details. Confirmed against a real +# proxy in tests/test_mock_provider.py; a fixture in the wire shape would test +# an object this code path never gets. +NATIVE = { + "prompt_tokens": 48_000, + "prompt_tokens_details": {"cached_tokens": 44_000, "cache_write_tokens": 1_000}, + "completion_tokens": 900, + "completion_tokens_details": {"reasoning_tokens": 640}, +} + + +@pytest.fixture +def logged(tmp_path: Path, monkeypatch: pytest.MonkeyPatch): + log = tmp_path / "provider_usage.jsonl" + monkeypatch.setenv(USAGE_LOG_ENV_VAR, str(log)) + + def emit(usage: dict) -> dict: + ProviderUsageLogger()._append( + "success", + {"model": "claude-sonnet-4-5", "custom_llm_provider": "openai", "call_type": "responses"}, + _openai_response(usage), + 0.0, + 1.0, + ) + return json.loads(log.read_text().splitlines()[-1]) + + return emit + + +def test_native_openai_usage_survives_the_anthropic_translation(logged) -> None: + event = logged(NATIVE) + native = event["native_usage"] + # Verbatim: the fields an Anthropic-shaped response cannot carry. + assert native["prompt_tokens_details"]["cached_tokens"] == 44_000 + assert native["prompt_tokens_details"]["cache_write_tokens"] == 1_000 + assert native["completion_tokens_details"]["reasoning_tokens"] == 640 + assert event["response_id"] == "resp_68f2c1" + + +def test_the_actual_model_is_recorded_separately_from_the_requested_role(logged) -> None: + """Pricing must follow what answered, not what the caller named.""" + + event = logged(NATIVE) + assert event["requested_model"] == "claude-sonnet-4-5" + assert event["actual_model"] == "gpt-5.6-sol-2026-08-01" + assert "cell_id" not in event, "a proxy-wide variable cannot identify a cell" + + +def test_the_captured_event_normalizes_with_openai_arithmetic(logged) -> None: + """Capture and normalization must agree end to end, not just in isolation.""" + + event = logged(NATIVE) + # The provider the LOG recorded, not one the test supplies - passing + # OPENAI_RESPONSES by hand here is what hid the adapter-key mismatch. + # LITELLM_NORMALIZED, not OPENAI_RESPONSES: a proxy callback never sees the + # upstream body. Measured against a real gateway - the Responses adapter + # found none of its keys there and reported every field unknown. + assert event["provider"] == LITELLM_NORMALIZED + assert event["provider_label"] == "openai" + usage = normalize_usage(event["provider"], event["native_usage"]) + assert usage.total_input_tokens == 48_000 + assert usage.ordinary_input_tokens == 3_000 + assert usage.cache_read_input_tokens == 44_000 + assert usage.complete + + +def test_usage_without_details_normalizes_to_unknown_rather_than_zero(logged) -> None: + """The mutation the accounting must not survive: dropped details, silent zeros.""" + + stripped = {k: v for k, v in NATIVE.items() if k != "prompt_tokens_details"} + event = logged(stripped) + usage = normalize_usage(event["provider"], event["native_usage"]) + assert usage.cache_read_input_tokens is None + assert usage.ordinary_input_tokens is None + assert not usage.complete + + +def test_a_failed_request_is_still_accounted_for(logged, tmp_path: Path) -> None: + """The money was spent whether or not the cell produced an artifact.""" + + import asyncio + + logger = ProviderUsageLogger() + args = ({"model": "claude-sonnet-4-5"}, _openai_response(NATIVE), 0.0, 1.0) + # Every hook LiteLLM can call, not the private helper underneath them: the + # sync failure hook was missing entirely and _append could never show that. + logger.log_failure_event(*args) + asyncio.run(logger.async_log_failure_event(*args)) + events = [json.loads(line) for line in (tmp_path / "provider_usage.jsonl").read_text().splitlines()] + assert len(events) == 2, "both failure hooks must record" + assert all(e["status"] == "failure" for e in events) + + +def test_every_public_outcome_hook_records(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None: + """Overriding a subset silently drops whichever path LiteLLM actually uses.""" + + import asyncio + + monkeypatch.setenv(USAGE_LOG_ENV_VAR, str(tmp_path / "usage.jsonl")) + logger = ProviderUsageLogger() + args = ({"model": "m"}, _openai_response(NATIVE), 0.0, 1.0) + logger.log_success_event(*args) + logger.log_failure_event(*args) + asyncio.run(logger.async_log_success_event(*args)) + asyncio.run(logger.async_log_failure_event(*args)) + + events = [json.loads(line) for line in (tmp_path / "usage.jsonl").read_text().splitlines()] + assert [e["status"] for e in events] == ["success", "failure", "success", "failure"] + + +def test_the_logger_never_raises_into_the_proxy(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None: + """Accounting is evidence, not control flow.""" + + monkeypatch.setenv(USAGE_LOG_ENV_VAR, str(tmp_path / "missing-dir" / "usage.jsonl")) + ProviderUsageLogger()._append("success", {}, object(), 0.0, 1.0) + + +def test_no_log_is_written_when_the_destination_is_unset(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.delenv(USAGE_LOG_ENV_VAR, raising=False) + ProviderUsageLogger()._append("success", {}, _openai_response(NATIVE), 0.0, 1.0) + assert not list(tmp_path.iterdir()) + + +def test_the_generated_config_loads_the_callback_from_beside_itself(tmp_path: Path) -> None: + """LiteLLM resolves the dotted path relative to the config directory.""" + + config = write_openai_litellm_config(tmp_path / "litellm.yaml", ["gpt-5.6-sol"]) + assert openai_litellm_config(["gpt-5.6-sol"])["litellm_settings"]["callbacks"] == [ + f"{USAGE_CALLBACK_MODULE}.handler" + ] + installed = config.parent / f"{USAGE_CALLBACK_MODULE}.py" + assert installed.is_file(), "the proxy cannot import a callback that was never placed" + # Importing it, not grepping it: a text search passes even when the module + # cannot load, which is exactly how a package-relative import survived + # review here. This is the deployment configuration, so load it the way the + # proxy does - by path, as a top-level module. + import importlib.util + + spec = importlib.util.spec_from_file_location(USAGE_CALLBACK_MODULE, installed) + assert spec is not None and spec.loader is not None + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + assert isinstance(module.handler, module.ProviderUsageLogger) + + +def test_the_gateway_forwards_the_usage_environment_into_the_proxy( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + """The proxy is a separate process with a constructed environment. + + Popen(env=...) replaces the parent environment rather than extending it, so + a variable the callback reads is simply absent unless the gateway forwards + it by name. Without this the accounting looks configured and silently + records nothing on every request - the in-process tests above cannot see + that, because they never cross the subprocess boundary. + """ + + for name in USAGE_ENV_VARS: + monkeypatch.setenv(name, f"value-for-{name}") + captured: dict[str, dict[str, str]] = {} + + class _Popen: + def __init__(self, *_a, **kwargs): + captured["env"] = kwargs["env"] + raise RuntimeError("stop before launching a real proxy") + + # The console-script resolver runs before Popen and is absent in this + # environment (the same reason two gateway tests fail here); the argv it + # builds is not what this test is about. + monkeypatch.setattr( + "workflow_bench.model_gateway.litellm_proxy_argv", + lambda **_k: ["/bin/true"], + ) + monkeypatch.setattr("workflow_bench.model_gateway.subprocess.Popen", _Popen) + gateway = OpenAIGateway( + openai_api_key="sk-test", + model_names=["gpt-5.6-sol"], + work_dir=tmp_path, + ) + with contextlib.suppress(Exception): + gateway.__enter__() + + env = captured.get("env") + assert env is not None, "the proxy was never constructed" + for name in USAGE_ENV_VARS: + assert env.get(name) == f"value-for-{name}", f"{name} never reached the proxy" + # The credential allowlist is still an allowlist, not the parent environment. + assert "PATH" in env and len(env) < 40 + + +def test_an_unresolvable_provider_is_refused_rather_than_guessed() -> None: + """LiteLLM says "openai" for Chat Completions too, and it counts differently.""" + + from workflow_bench.provider_usage import canonical_provider + + # Every openai call reaching this callback has already been normalised by + # LiteLLM, whatever endpoint it used - the observed call_type for a Claude + # Code request through the gateway is "anthropic_messages". The adapter has + # to match the object in hand, not the protocol on the wire. + assert canonical_provider("openai", "responses") == LITELLM_NORMALIZED + assert canonical_provider("openai", "anthropic_messages") == LITELLM_NORMALIZED + assert canonical_provider("anthropic", "completion") == ANTHROPIC + # An unrecognised provider is still refused rather than guessed. + assert canonical_provider("some-new-provider", "responses") is None + assert canonical_provider(None, None) is None + + +def test_request_identity_cannot_come_from_the_proxy_environment() -> None: + """One proxy serves the whole sweep, so its environment identifies the sweep. + + attach_openai_gateway wraps all of _run_sweep, and cells run concurrently + under --workers, interleaving requests through that single process. Any + variable forwarded at launch is therefore constant for every event it ever + records. Pinned so a future change does not reintroduce a per-cell + environment variable that would silently stamp one value on every request. + """ + + assert USAGE_ENV_VARS == ( + "GITNEXUS_BENCH_PROVIDER_USAGE", + "GITNEXUS_BENCH_SWEEP_ID", + ), "a per-cell variable here would be constant across concurrent cells" + + +def test_a_request_records_its_session_so_attribution_stays_possible(logged) -> None: + """The per-request half of identity, recorded even when the provider omits it.""" + + event = logged(NATIVE) + assert "session_id" in event, "absent attribution is still a fact about the run" + + +def test_the_callback_imports_the_way_litellm_actually_loads_it(tmp_path: Path) -> None: + """By path, as a top-level module, with no parent package and no sys.path entry. + + LiteLLM resolves a dotted callback through spec_from_file_location against + the config directory, so the copied file is not part of workflow_bench when + it runs. A relative or sibling import therefore raises ImportError and the + proxy exits before becoming ready - which the in-package tests cannot see, + because they import it as workflow_bench.litellm_usage_callback. + """ + + import importlib.util + import shutil + + source = Path(litellm_usage_callback.__file__) + installed = tmp_path / f"{USAGE_CALLBACK_MODULE}.py" + shutil.copy(source, installed) + + spec = importlib.util.spec_from_file_location(USAGE_CALLBACK_MODULE, installed) + assert spec is not None and spec.loader is not None + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) # ImportError here is the proxy refusing to start + assert hasattr(module, "handler") + + +def test_the_callbacks_copied_constants_match_the_canonical_ones() -> None: + """The copies are deliberate; drifting apart silently is not. + + The callback cannot import from the package (see the test above), so it + carries its own literals. These assertions are what keep the duplication + honest. + """ + + assert litellm_usage_callback.USAGE_LOG_ENV_VAR == provider_usage.USAGE_LOG_ENV_VAR + assert litellm_usage_callback.SWEEP_ID_ENV_VAR == provider_usage.SWEEP_ID_ENV_VAR + for label, call_type in ( + ("openai", "responses"), + ("openai", "completion"), + ("openai", None), + ("anthropic", "completion"), + ("mystery", "responses"), + ): + assert litellm_usage_callback.canonical_provider(label, call_type) == provider_usage.canonical_provider( + label, call_type + ), f"resolver drifted for {label!r}/{call_type!r}" diff --git a/eval/tests/test_reuse_round_trip.py b/eval/tests/test_reuse_round_trip.py new file mode 100644 index 000000000..a31e9c8f7 --- /dev/null +++ b/eval/tests/test_reuse_round_trip.py @@ -0,0 +1,268 @@ +"""A row the runner actually emits must satisfy the reuse reader. + +Every existing comparator-reuse test builds its rows by hand. That proves the +predicate's logic and nothing about the producer: a fixture can satisfy +eligibility while a real emitted row never does, and the audit that counts key +names cannot tell the difference. These tests carry one record through the +production path instead: + + real run_cell -> production JSONL writer -> load_result_rows + -> row_is_reusable_comparator + +Only the expensive dependencies are replaced - the model session, sandbox +launch, repository acquisition, graph preparation. The digest fields the reuse +binding compares are assembled by run_cell itself from its TaskCellContext, so +they stay real: they are the subject of the test, not scaffolding around it. +""" + +from __future__ import annotations + +import json +from datetime import UTC, datetime, timedelta +from pathlib import Path +from types import SimpleNamespace +from typing import Any + +import pytest + +from workflow_bench import runner +from workflow_bench.proposer_sandbox import redact_text +from workflow_bench.model_gateway import credential_secrets +from workflow_bench.runner_sessions import PARENT_EVENT_STREAM_SOURCE +from workflow_bench.comparator_reuse import ( + ComparatorReuseExpectation, + TaskReuseBinding, + load_result_rows, + row_is_reusable_comparator, +) + +TASK_ID = "review-pr-2718-defect" +SHA = "a" * 40 + + +def _snapshot(prefix: str) -> SimpleNamespace: + return SimpleNamespace( + digest=f"{prefix}-content", + manifest_digest=f"{prefix}-manifest", + dependency_content_digest=f"{prefix}-dep-content", + dependency_manifest_digest=f"{prefix}-dep-manifest", + command_digest=f"{prefix}-command", + materialize=lambda *a, **k: None, + ) + + +def _write_like_the_sweep(tmp_path: Path, row: dict[str, Any]) -> Path: + """Serialize exactly as ``keep`` does in _run_sweep, redaction included. + + json.dumps + write_text would skip the redaction the real writer applies, + so a change there could break reusable rows without failing this test - and + redaction is not cosmetic here, since it rewrites the row's own bytes. + """ + + results = tmp_path / "results.jsonl" + secrets = credential_secrets( + SimpleNamespace(auth_token="sk-ant-should-never-appear", base_url=None) + ) + with results.open("a") as handle: + handle.write(redact_text(json.dumps(row), secrets) + "\n") + return results + + +@pytest.fixture +def emitted_row(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> dict[str, Any]: + """One record from the real run_cell, with only expensive work replaced.""" + + worktree = tmp_path / "clone" + worktree.mkdir() + + # The session is what costs money; everything it returns is scripted. The + # record's binding fields are NOT set here - run_cell derives them. + def fake_run_arm(*_a: Any, **_k: Any) -> dict[str, Any]: + return { + "ok": True, + "error_kind": None, + "error_detail": None, + "resolved": True, + "review_evidence_valid": True, + "review_score": {"weighted_f1": 0.5}, + "review_weighted_f1": 0.5, + "skill_invoked": True, + "skill_digest": "skill-digest", + "transcript_missing": False, + "transcript_artifacts": [ + { + "path": "transcripts/session-1.jsonl", + "sha256": __import__("hashlib").sha256(b'{"type":"ok"}\n').hexdigest(), + "bytes": 14, + "source": PARENT_EVENT_STREAM_SOURCE, + } + ], + "session_ids": ["s1"], + "num_turns": 3, + "duration_s": 1.0, + "cost_usd": 0.5, + "input_tokens": 1, + "output_tokens": 1, + "cache_creation_input_tokens": 0, + "cache_read_input_tokens": 0, + } + + for name, value in { + "run_arm": fake_run_arm, + "copy_isolated_tree": lambda *a, **k: worktree, + "make_worktree": lambda *a, **k: worktree, + "sanitize_clone_for_hidden_oracles": lambda *a, **k: SHA, + "stage_task_assets": lambda *a, **k: (), + "isolated_gitnexus_registry_mount": lambda *a, **k: None, + "seed_evaluated_skills": lambda *a, **k: None, + "apply_candidate_overlay": lambda *a, **k: None, + "require_hidden_harness_absent": lambda *a, **k: None, + "require_skill_fingerprint": lambda *a, **k: None, + "enforce_work_evidence": lambda *a, **k: None, + "skill_fingerprint": lambda *a, **k: "skill-digest", + "capture_patch": lambda *a, **k: b"", + "implementation_diff_digest": lambda *a, **k: "", + "diff_churn": lambda *a, **k: {}, + "_prepare_untracked_for_diff": lambda *a, **k: None, + "remove_clone": lambda *a, **k: None, + "ce_plugin_dir_for_arm": lambda *a, **k: None, + "ce_plugin_mounts_for_arm": lambda *a, **k: (), + "current_runtime_digest": lambda: "runtime-digest", + "build_sandbox_environment": lambda *a, **k: {}, + "credential_secrets": lambda *a, **k: (), + # run_cell requires an immutable base commit before it will record a + # cell; the git plumbing is expensive setup, the SHA it returns is not + # part of the reuse binding under test. + "_sandbox_git": lambda *a, **k: SHA, + # The artifact copy is real; only the read of the agent-written file is + # replaced, since no agent ran to write one. + "_bounded_regular_bytes": lambda *a, **k: b'{"schema_version":1}', + }.items(): + monkeypatch.setattr(runner, name, value) + + class _Sandbox: + clone = worktree + private_root = tmp_path / "private" + backend = "test-double" + settings_json = "{}" + require_pid_namespace = False + + def __enter__(self) -> _Sandbox: + return self + + def __exit__(self, *_exc: Any) -> bool: + return False + + def command_prefix_for(self, **_k: Any) -> list[str]: + return [] + + def run(self, *_a: Any, **_k: Any) -> SimpleNamespace: + return SimpleNamespace(ok=True, returncode=0, stdout_tail="", stderr_tail="") + + def environment(self, **_k: Any) -> dict[str, str]: + return {} + + def host_text(self, value: str) -> str: + return value + + monkeypatch.setattr(runner, "prepare_sandbox", lambda **_k: _Sandbox()) + + ctx = runner.TaskCellContext( + task={"id": TASK_ID, "prompt": "review it", "verify": "true"}, + oracle_snapshot=_snapshot("oracle"), + repo=tmp_path / "repo", + task_sha=SHA, + graph_snapshot=_snapshot("graph"), + graph_snapshot_error=None, + asset_snapshot=_snapshot("asset"), + asset_snapshot_error=None, + args=SimpleNamespace( + model="gpt-5.6-sol", effort="xhigh", timeout=60, claude_bin="claude", + base_url=None, auth_token=None, permission_mode=None, arms=["review"], + proposer_model=None, outage_streak=5, runs=1, workers=1, + ), + out_dir=tmp_path / "out", + ce_plugin_snapshot=None, + trees_dir=tmp_path / "trees", + bwrap_bin=Path("/bin/true"), + runtime_mounts=(), + candidate_overlay=None, + overlay_digest=None, + sandbox_backend="test-double", + clone_template=None, + sanitized_head=SHA, + ) + (tmp_path / "out").mkdir(exist_ok=True) + (tmp_path / "trees").mkdir(exist_ok=True) + (tmp_path / "private").mkdir(exist_ok=True) + # run_cell records review_artifact only when the review source exists, and + # reuse now requires it - a scored review with no artifact is a claim about + # evidence rather than the evidence. Production writes this file; the + # fixture has to as well, or the emitted row is one production never emits. + review_dir = tmp_path / "private" / "review-output" + review_dir.mkdir(exist_ok=True) + (review_dir / "review-output.json").write_text('{"schema_version": 1, "verdict": "approve", "findings": []}') + return runner.run_cell(ctx, 0, "review") + + +def _expectation(**overrides: Any) -> ComparatorReuseExpectation: + """Bindings from the sweep's own configuration, not copied out of the row. + + Copying the emitted values back in would make producer and consumer agree + because the test arranged it, which is the blind spot being closed. + """ + + binding = TaskReuseBinding( + task_base_sha=SHA, + task_prompt_digest=runner.hashlib.sha256(b"review it").hexdigest(), + oracle_digest="oracle-content", + oracle_command_digest="oracle-command", + oracle_manifest_digest="oracle-manifest", + task_asset_manifest_digest="asset-manifest", + sandbox_dependency_manifest_digest="asset-dep-manifest", + ) + values: dict[str, Any] = dict( + model="gpt-5.6-sol", + effort="xhigh", + sandbox_backend="test-double", + runtime_digest="runtime-digest", + now=datetime.now(UTC), + max_age=timedelta(days=90), + tasks={TASK_ID: binding}, + skill_digests={"review": "skill-digest"}, + ce_plugin_version=None, + ce_plugin_manifest_digest=None, + ) + values.update(overrides) + return ComparatorReuseExpectation(**values) + + +def test_a_row_the_runner_emitted_survives_serialization_and_qualifies( + emitted_row: dict[str, Any], tmp_path: Path +) -> None: + """The producer/consumer contract, end to end through the real writer.""" + + results = _write_like_the_sweep(tmp_path, emitted_row) + rows = load_result_rows(results) + assert len(rows) == 1, "the production row must survive the reader" + + assert row_is_reusable_comparator(rows[0], _expectation()) is True + + +def test_a_changed_binding_rejects_the_same_emitted_row( + emitted_row: dict[str, Any], tmp_path: Path +) -> None: + """Fails closed on drift, so the positive case is not vacuous.""" + + row = load_result_rows(_write_like_the_sweep(tmp_path, emitted_row))[0] + + binding = TaskReuseBinding( + task_base_sha=SHA, + task_prompt_digest=runner.hashlib.sha256(b"review it").hexdigest(), + oracle_digest="oracle-content", + oracle_command_digest="oracle-command", + oracle_manifest_digest="oracle-manifest", + task_asset_manifest_digest="asset-manifest", + sandbox_dependency_manifest_digest="DIFFERENT-dependencies", + ) + assert row_is_reusable_comparator(row, _expectation(tasks={TASK_ID: binding})) is False diff --git a/eval/tests/test_review_corpus.py b/eval/tests/test_review_corpus.py index 42f18cb7c..cbaa2695e 100644 --- a/eval/tests/test_review_corpus.py +++ b/eval/tests/test_review_corpus.py @@ -32,6 +32,10 @@ def test_review_corpus_is_immutable_and_task_bound(): assert task["ref"] == case["base_sha"] assert task["sandbox_copy"] == [f"eval/workflow_bench/review_cases/{patch.name}"] assert task["setup"] == review_case_setup_command(patch.name) + assert any( + dep.get("source") == "gitnexus-shared/dist" and dep.get("target") == "gitnexus-shared/dist" + for dep in task["sandbox_dependencies"] + ) def test_hidden_labels_are_not_recoverable_from_visible_task_input(): diff --git a/eval/tests/test_review_scoring.py b/eval/tests/test_review_scoring.py index c1b081720..1e6c8922e 100644 --- a/eval/tests/test_review_scoring.py +++ b/eval/tests/test_review_scoring.py @@ -325,3 +325,33 @@ def test_clean_control_rewards_an_empty_approval_and_penalizes_noise(): assert noisy["recall"] is None assert noisy["clean_pass"] is False assert noisy["verdict_correct"] is False + + +def test_parse_review_output_names_the_actual_failure(tmp_path: Path): + """One message per cause. + + Folding empty, malformed and encoding failures together makes a sandbox that + left the artifact at 0 bytes indistinguishable from an encoding fault: every + such cell reports "not valid UTF-8 JSON". A file the agent never created + escaped that fold — lstat sat outside the try, so it raised + FileNotFoundError — but only as a bare OSError, naming no cause at all. + """ + + missing = tmp_path / "never-written.json" + with pytest.raises(ValueError, match="was never written"): + parse_review_output(missing) + + empty = tmp_path / "empty.json" + empty.touch() + with pytest.raises(ValueError, match="is empty"): + parse_review_output(empty) + + not_utf8 = tmp_path / "latin1.json" + not_utf8.write_bytes(b'{"verdict": "\xff\xfe"}') + with pytest.raises(ValueError, match="not valid UTF-8"): + parse_review_output(not_utf8) + + prose = tmp_path / "prose.json" + prose.write_text("Here is my review of the changes.", encoding="utf-8") + with pytest.raises(ValueError, match="not valid JSON"): + parse_review_output(prose) diff --git a/eval/tests/test_runner_hardening.py b/eval/tests/test_runner_hardening.py index be196750b..111b3c3c5 100644 --- a/eval/tests/test_runner_hardening.py +++ b/eval/tests/test_runner_hardening.py @@ -3,13 +3,14 @@ import hashlib import json import shutil +import subprocess from contextlib import nullcontext from pathlib import Path from types import SimpleNamespace import pytest -from workflow_bench import runner, runner_artifacts, runner_sessions +from workflow_bench import proposer_sandbox, runner, runner_artifacts, runner_sessions from workflow_bench.evolution import skill_fingerprint from workflow_bench.oracle_assets import review_case_setup_command from workflow_bench.process_control import ManagedProcessError, ManagedProcessResult @@ -608,6 +609,67 @@ def test_run_cell_reports_a_cleanup_failure_over_its_primary_outcome(monkeypatch assert "clone is busy" in record["error_detail"] +def _git(repo, *args): + return subprocess.run(["git", "-C", str(repo), *args], check=True, capture_output=True, text=True) + + +def test_run_cell_runs_the_arm_against_a_copy_of_the_clone_template(monkeypatch, tmp_path): + """run_cell must copy the template, never re-clone. + + run_cell takes the clone-template branch on essentially every multi-cell + sweep: it copies a pre-sanitized template rather than paying `git clone + --no-local` plus repack/prune/fsck per cell. Asserting on a copy the test + makes itself proves nothing about that branch — the clone the arm receives + is what has to come from the template, carrying the template's sanitized + HEAD rather than a recomputed one. + """ + + repo = tmp_path / "repo" + repo.mkdir() + _git(repo, "init", "--quiet") + _git(repo, "checkout", "--quiet", "-b", "main") + (repo / "from-template.txt").write_text("sanitized\n") + _git(repo, "add", "-A") + _git(repo, "-c", "user.name=test", "-c", "user.email=test@invalid", "commit", "--quiet", "-m", "base") + sha = _git(repo, "rev-parse", "HEAD").stdout.strip() + trees = tmp_path / "trees" + trees.mkdir() + template = runner.make_worktree(repo, sha, trees) + template_head = _git(template, "rev-parse", "HEAD").stdout.strip() + + _stub_cell_dependencies(monkeypatch, tmp_path) + + def fail_if_recloned(*_args, **_kwargs): + raise AssertionError("clone template present: run_cell must not re-clone") + + monkeypatch.setattr(runner, "make_worktree", fail_if_recloned) + monkeypatch.setattr(runner, "sanitize_clone_for_hidden_oracles", fail_if_recloned) + + seen: dict[str, object] = {} + + def record_arm(_arm, _task, worktree, _args, **_kwargs): + seen["worktree"] = worktree + seen["head"] = _git(worktree, "rev-parse", "HEAD").stdout.strip() + seen["content"] = (worktree / "from-template.txt").read_text() + # The copy is a private checkout: what the cell writes must not reach + # the template the other cells of this task still copy from. + (worktree / "from-template.txt").write_text("cell-local\n") + return {"resolved": True, "ok": True, "error_kind": None} + + monkeypatch.setattr(runner, "run_arm", record_arm) + + runner.run_cell( + _cell_context(tmp_path, clone_template=template, sanitized_head=template_head), + 0, + "workflow", + ) + + assert seen["content"] == "sanitized\n" + assert seen["head"] == template_head + assert seen["worktree"] != template + assert (template / "from-template.txt").read_text() == "sanitized\n" + + def test_run_cell_does_not_mask_the_staged_review_patch_before_setup(monkeypatch, tmp_path): """Review setup applies a patch staged under eval/workflow_bench. @@ -1002,3 +1064,40 @@ def test_progress_line_reports_the_numbers_a_real_run_measured(): assert "cost=$0.5" in line assert "took=12.0s" in line assert "error_kind=none" in line + + +def test_claude_settings_allow_the_review_artifact_directory(): + """The second gate on the artifact path. + + The bwrap bind is not the only thing that decides whether the agent can + write: the CLI applies this filesystem policy to its own tools, so a path + missing from allowWrite is unwritable however the mount is shaped. The + artifact lived under /workspace when this list was written, which is why + moving it out needed this entry and nothing caught the omission. + """ + + settings = json.loads(proposer_sandbox.build_claude_settings(sandbox_enabled=True)) + filesystem = settings["sandbox"]["filesystem"] + assert proposer_sandbox.SANDBOX_REVIEW_OUTPUT in filesystem["allowWrite"] + assert proposer_sandbox.SANDBOX_REVIEW_OUTPUT in filesystem["allowRead"] + assert filesystem["denyRead"] == ["/"] + + +def test_review_contract_tells_the_agent_the_writable_path(): + prompt = runner.REVIEW_PROMPT.format(task="task text") + assert f"{runner.SANDBOX_REVIEW_OUTPUT}/{runner.REVIEW_OUTPUT}" in prompt + assert f"{runner.SANDBOX_WORKSPACE}/{runner.REVIEW_OUTPUT}" not in prompt + # The JSON shape survives .format() with its braces intact. + assert '{"schema_version":1' in prompt + artifact = f"{runner.SANDBOX_REVIEW_OUTPUT}/{runner.REVIEW_OUTPUT}" + assert runner.CE_REVIEW_PROMPT.format(task="task text").count(artifact) == 1 + + +def test_enforce_phase_workspace_can_require_an_untouched_workspace(tmp_path): + (tmp_path / "tracked.py").write_text("original\n") + before = runner_artifacts.workspace_snapshot(tmp_path) + runner_artifacts.enforce_phase_workspace(tmp_path, before, allowed_artifact=None) + + (tmp_path / "tracked.py").write_text("the review edited the code it was reviewing\n") + with pytest.raises(ValueError, match="changed the read-only workspace"): + runner_artifacts.enforce_phase_workspace(tmp_path, before, allowed_artifact=None) diff --git a/eval/tests/test_sanitized_graph.py b/eval/tests/test_sanitized_graph.py index 7b5dfc093..d7f73d756 100644 --- a/eval/tests/test_sanitized_graph.py +++ b/eval/tests/test_sanitized_graph.py @@ -212,6 +212,22 @@ def test_prepare_sanitized_graph_builds_once_from_parentless_tree_and_caches_onl assert removed == [seed] +def test_prepare_sanitized_graph_requires_head_when_given_a_template(tmp_path: Path): + with pytest.raises(SandboxError, match="sanitized HEAD"): + sanitized_graph.prepare_sanitized_graph( + {}, + repo=tmp_path, + resolved_sha="b" * 40, + parent=tmp_path, + cache=SimpleNamespace(), # type: ignore[arg-type] + claude_bin="claude", + bwrap_bin="bwrap", + runtime_mounts=(), + clone_template=tmp_path, + sanitized_head=None, + ) + + def test_graph_snapshot_rejects_arm_sanitization_identity_drift(tmp_path: Path): assets = SimpleNamespace( digest="digest", diff --git a/eval/tests/test_session_progress.py b/eval/tests/test_session_progress.py index 9e59ba4bf..416c6206c 100644 --- a/eval/tests/test_session_progress.py +++ b/eval/tests/test_session_progress.py @@ -11,7 +11,7 @@ import io import json import time -from workflow_bench.runner_sessions import SessionProgress +from workflow_bench.runner_sessions import SessionProgress, neutralize_ci_log_text def _drain_lines(stream: io.StringIO) -> list[str]: @@ -325,3 +325,48 @@ def test_cell_failure_detail_line_bounds_a_huge_detail() -> None: assert line is not None assert "truncated" in line assert len(line) < MAX_CELL_DETAIL_CHARS + 200 + + +def test_progress_neutralizes_github_actions_annotation_forms() -> None: + rewritten = neutralize_ci_log_text( + "gitnexus/src/cli/optional-grammars.ts(18,36): error TS2307: Cannot find module " + "'gitnexus-shared'\n::error::Composite projects may not disable incremental compilation.\n" + "##[error]tsc failed" + ) + assert "): error TS2307" not in rewritten + assert "): compiler-error TS2307" in rewritten + assert "::error::" not in rewritten + assert "[:]error::" in rewritten + assert "##[error]" not in rewritten + assert "# [error]tsc failed" in rewritten + + stream = io.StringIO() + progress = SessionProgress("review-pr-2718-defect-ce_review-run0", stream=stream, heartbeat_s=3600) + events = [ + { + "type": "assistant", + "message": { + "content": [{"type": "tool_use", "id": "b1", "name": "Bash", "input": {"command": "npx tsc --noEmit"}}] + }, + }, + { + "type": "user", + "message": { + "content": [ + { + "type": "tool_result", + "tool_use_id": "b1", + "is_error": True, + "content": "gitnexus/src/cli/optional-grammars.ts(18,36): error TS2307: Cannot find module 'gitnexus-shared'", + } + ] + }, + }, + ] + for event in events: + _observe(progress, (json.dumps(event) + "\n").encode()) + + output = stream.getvalue() + assert "): error TS2307" not in output + assert "): compiler-error TS2307" in output + assert "result=error" in output diff --git a/eval/tests/test_sweep_finalization.py b/eval/tests/test_sweep_finalization.py new file mode 100644 index 000000000..176157645 --- /dev/null +++ b/eval/tests/test_sweep_finalization.py @@ -0,0 +1,279 @@ +"""The real sweep must reach the right finalization decision. + +`enforce_measurement_health` is unit-tested and the call site is pinned +structurally, but neither shows the guard running inside a sweep. These drive +the real `_run_sweep` with cell execution scripted and everything downstream of +it left alone: folding, aggregation, the artifact writers, the health guard and +the exit selection. + +The below-breaker case is the decisive one. A fixture of many unusable cells +aborts through the pre-existing outage breaker instead - `review-evidence-invalid` +is systemic with a limit of 5 - and would pass whether or not the finalization +guard exists. One fresh unusable cell stays under that threshold, so only the +guard can catch it. +""" + +from __future__ import annotations + +import json +import threading +from collections.abc import Callable +from pathlib import Path +from types import SimpleNamespace +from typing import Any + +import pytest + +from tests.bench_fixtures import scored_review_row, unusable_review_row +from workflow_bench import runner + +TASK = { + "id": "review-pr-2718-defect", + "repo": "~/GitNexus", + "ref": "a" * 40, + "prompt": "review it", + "verify": "true", + "class": "review-defect", +} + + +def _args(out: Path, **overrides: Any) -> SimpleNamespace: + values: dict[str, Any] = dict( + arms=["review"], claude_bin="claude", effort="xhigh", model="gpt-5.6-sol", + out=out, outage_streak=runner.DEFAULT_OUTAGE_STREAK, promotion_max_task_regression=10.0, + promotion_metric="review_weighted_f1", promotion_min_improvement=1.0, + promotion_min_runs=1, proposer_model=None, reuse_results=None, runs=1, workers=1, + timeout=60, base_url=None, auth_token=None, permission_mode=None, + ) + values.update(overrides) + return SimpleNamespace(**values) + + +def _snapshot(prefix: str) -> SimpleNamespace: + return SimpleNamespace( + digest=f"{prefix}-content", manifest_digest=f"{prefix}-manifest", + dependency_content_digest=f"{prefix}-dep", dependency_manifest_digest=f"{prefix}-depman", + command_digest=f"{prefix}-command", materialize=lambda *a, **k: None, + ) + + +def _sweep( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, + record: dict[str, Any] | Callable[[int], dict[str, Any]], + *, + runs: int = 1, + cancel_event: threading.Event | None = None, + candidate_arms: list[str] | None = None, + arms: list[str] | None = None, + after_cell: Callable[[int, str], None] | None = None, +): + """Drive the real _run_sweep; only cell execution and setup are scripted. + + ``after_cell`` runs once a cell's record exists, which is how a test sets + cancellation deterministically at a known point instead of racing a sleep. + """ + + out = tmp_path / "out" + + def scripted_cell(_ctx: Any, run_idx: int, arm: str) -> dict[str, Any]: + row = dict(record(run_idx) if callable(record) else record) + row.update({"task": TASK["id"], "arm": arm, "run": run_idx, "class": TASK["class"]}) + if after_cell is not None: + after_cell(run_idx, arm) + return row + + monkeypatch.setattr(runner, "run_cell", scripted_cell) + monkeypatch.setattr(runner, "ensure_task_graph", lambda **k: k["env"].graph_snapshots.__setitem__( + k["graph_key"], _snapshot("graph"))) + monkeypatch.setattr(runner.TaskAssetCache, "prepare", lambda self, *a, **k: _snapshot("asset")) + # Binding resolution clones the repo and verifies the ref; that is expensive + # setup, and the bindings it would return are supplied directly instead. + monkeypatch.setattr( + runner, "resolve_task_bindings", + lambda tasks, expected, **k: list(expected), + ) + + return runner._run_sweep( + _args(out, runs=runs, arms=arms or ["review"]), + parser=SimpleNamespace(error=lambda m: (_ for _ in ()).throw(SystemExit(2))), + tasks=[TASK], + skipped_expensive=[], + oracle_snapshots=[_snapshot("oracle")], + expected_task_bindings=[{"repo_identity": str(tmp_path / "repo"), "resolved_sha": "a" * 40}], + ce_plugin_config=None, + bwrap_bin=Path("/bin/true"), + sandbox_backend="test-double", + runtime_mounts=(), + candidate_arms=candidate_arms or [], + candidate_overlay=None, + overlay_digest=None, + promotion_target_bases={}, + cancel_event=cancel_event, + ), out + + +def test_one_unusable_cell_below_the_breaker_reaches_the_finalization_guard( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture[str] +) -> None: + """The decisive case: too few failures to trip the breaker, so only the guard can catch it.""" + + streak = runner.systemic_outage_streak("review-evidence-invalid", 0) + assert streak < runner.DEFAULT_OUTAGE_STREAK, "fixture must stay under the breaker" + + unusable = unusable_review_row() + with pytest.raises(SystemExit) as exc: + _sweep(tmp_path, monkeypatch, unusable) + assert exc.value.code == 1 + out = capsys.readouterr().out + assert "review: UNUSABLE" in out, "the guard must name the arm and its status" + assert "systemic-outage" not in out, "the breaker must not have tripped" + + +def test_a_zero_score_stays_a_valid_negative_measurement( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture[str] +) -> None: + """0.0 is a present measurement, not missing evidence. + + A truthiness check on the score would misread it as absent and turn a + quality result into an execution-health failure. + """ + + zeroed = scored_review_row( + resolved=False, error_kind="oracle-failed", + review_score={"weighted_f1": 0.0}, review_weighted_f1=0.0, + ) + _sweep(tmp_path, monkeypatch, zeroed) + out = capsys.readouterr().out + assert "review: OBSERVED_OK" in out + assert "UNUSABLE" not in out + + +def test_finalization_persists_results_and_report( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + """Evidence must survive the sweep, and say the same thing the exit does.""" + + scored = scored_review_row(resolved=False, error_kind="oracle-failed", review_weighted_f1=0.2) + _result, out = _sweep(tmp_path, monkeypatch, scored) + rows = [json.loads(line) for line in (out / "results.jsonl").read_text().splitlines()] + assert len(rows) == 1 and rows[0]["review_weighted_f1"] == 0.2 + assert (out / "report.md").is_file() + + +def test_cancellation_without_an_outage_exits_130( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture[str] +) -> None: + """An interrupted sweep is interrupted, not aborted. + + One admissible cell lands first so the measurement-health guard classifies + the arm DEGRADED rather than UNUSABLE - otherwise the guard would supply + exit 1 and this test would pass without ever exercising exit selection. + """ + + cancel_event = threading.Event() + with pytest.raises(SystemExit) as exc: + _sweep( + tmp_path, monkeypatch, lambda _run: scored_review_row(), + runs=3, cancel_event=cancel_event, + after_cell=lambda run_idx, _arm: cancel_event.set() if run_idx == 0 else None, + ) + stdout = capsys.readouterr().out + report = (tmp_path / "out" / "report.md").read_text() + assert "Sweep cancelled" in report, "an interruption must be reported as one" + assert "systemic-outage" not in stdout, "no breaker trip in this scenario" + assert exc.value.code == 130 + + +def test_an_outage_keeps_exit_1_even_though_the_breaker_cancels( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture[str] +) -> None: + """Precedence: the breaker sets cancel_event, so order decides the exit. + + Testing cancellation first would relabel every outage a Ctrl-C. The first + cell is admissible for the same reason as above, and the failures after it + are consecutive and systemic, which is what the breaker actually counts. + """ + + def cell(run_idx: int) -> dict[str, Any]: + return scored_review_row() if run_idx == 0 else unusable_review_row() + + cancel_event = threading.Event() + with pytest.raises(SystemExit) as exc: + _sweep(tmp_path, monkeypatch, cell, + runs=1 + runner.DEFAULT_OUTAGE_STREAK, cancel_event=cancel_event) + stdout = capsys.readouterr().out + report = (tmp_path / "out" / "report.md").read_text() + assert "systemic-outage" in stdout, "the real breaker must have tripped" + assert cancel_event.is_set(), "the breaker cancels in-flight work" + assert "Sweep aborted" in report + assert exc.value.code == 1, "an outage must not become the 130 of a Ctrl-C" + + +def test_an_interrupted_sweep_keeps_the_evidence_it_already_paid_for( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + """Cancellation must not discard rows that already cost money. + + The completed-run persistence test cannot show this: it never interrupts, so + it would pass even if the writer only ran on the clean path. + """ + + cancel_event = threading.Event() + with pytest.raises(SystemExit): + _sweep( + tmp_path, monkeypatch, lambda _run: scored_review_row(review_weighted_f1=0.42), + runs=3, cancel_event=cancel_event, + after_cell=lambda run_idx, _arm: cancel_event.set() if run_idx == 0 else None, + ) + rows = [ + json.loads(line) + for line in (tmp_path / "out" / "results.jsonl").read_text().splitlines() + ] + assert len(rows) == 1, "the cell that completed before cancellation must survive" + assert rows[0]["review_weighted_f1"] == 0.42, "its measurement must survive intact" + assert (tmp_path / "out" / "report.md").is_file() + + +def test_an_interrupted_sweep_emits_nothing_that_authorizes_promotion( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + """The semantic condition, not the absence of a file. + + promotion.json is still written for an aborted run - it is the record of why + nothing was promoted. What must hold is that nothing in it authorizes a + promotion from partial evidence. + """ + + cancel_event = threading.Event() + with pytest.raises(SystemExit): + _sweep( + tmp_path, monkeypatch, lambda _run: scored_review_row(), + runs=3, cancel_event=cancel_event, + # The candidate arm has to RUN, not merely appear in promotion + # metadata: _run_sweep builds cells only from args.arms, so naming it + # in candidate_arms alone left the candidate with no results at all - + # and then "insufficient_evidence" would hold because nothing ran, + # not because partial evidence is barred from promoting. + arms=["review", "candidate_review"], + candidate_arms=["candidate_review"], + after_cell=( + lambda run_idx, arm: cancel_event.set() + if run_idx == 0 and arm == "candidate_review" + else None + ), + ) + rows = [ + json.loads(line) + for line in (tmp_path / "out" / "results.jsonl").read_text().splitlines() + ] + assert any(r["arm"] == "candidate_review" for r in rows), ( + "the candidate must have produced evidence, or insufficient_evidence " + "would hold merely because nothing ran" + ) + promotion = json.loads((tmp_path / "out" / "promotion.json").read_text()) + assert promotion["run_status"] == "aborted" + assert promotion["decisions"], "an aborted run still has to say what it decided" + for decision in promotion["decisions"]: + assert decision["decision"] == "insufficient_evidence" + assert any("partial evidence" in reason for reason in decision["reasons"]) diff --git a/eval/tests/test_workflow_bench.py b/eval/tests/test_workflow_bench.py index 45d0b2f35..51212ceda 100644 --- a/eval/tests/test_workflow_bench.py +++ b/eval/tests/test_workflow_bench.py @@ -5,22 +5,35 @@ import os import re import shlex import subprocess +import threading from pathlib import Path import pytest import yaml +from typing import Any + +from workflow_bench import runner +from workflow_bench.evolution import CANDIDATE_ARMS +from workflow_bench.process_control import _CANCELLATION, cancellation_scope from workflow_bench.runner import ( aggregate, + GraphBuildEnv, + arm_health, broken_incumbent_arms, + unhealthy_arms, + unmeasured_arms, build_parser, infra_error_record, + next_graph_prefetch_target, normalized_model_identifier, parse_shortstat, + prefetch_next_graph, render_report, savings, select_tasks, systemic_outage_streak, + task_has_planned_paid_cells, ) @@ -63,6 +76,15 @@ def test_aggregate_takes_medians_and_counts_resolved(): "diff_deletions": 5, "class": "demo", "resolved": 2, + # None of these are reused, so every resolution was measured this sweep. + "resolved_fresh": 2, + # Health is counted separately from resolution: all three executed and + # produced usable evidence, including the one that resolved nothing. + "fresh_attempts": 3, + "admissible": 3, + "execution_failures": 0, + "evidence_failures": 0, + "health_reasons": [], "runs": 3, "valid_runs": 3, "excluded_runs": 0, @@ -173,6 +195,11 @@ def test_eval_ci_uses_locked_uv_and_blocking_native_containment_jobs(): assert containment["env"] == { "GITNEXUS_REQUIRE_BWRAP_CANARY": "1", "GITNEXUS_REQUIRE_CLAUDE_CANARY": "1", + # This job is the only place with bubblewrap, the pinned runtime and a + # built GitNexus together, so it is where the offline sweep runs with + # nothing provisioning-stubbed. Pinned here so the gate cannot be + # dropped and leave the sweep silently running the stubbed path. + "GITNEXUS_REQUIRE_FULL_SWEEP": "1", } assert containment["timeout-minutes"] == 20 assert containment_node_setup["with"] == { @@ -217,6 +244,12 @@ def test_eval_ci_uses_locked_uv_and_blocking_native_containment_jobs(): "tests/test_proposer_sandbox.py", "tests/test_workflow_bench_sessions.py", "tests/test_ce_plugin_runtime.py", + # The offline sweep, run here with nothing stubbed: this job is the only + # one carrying bubblewrap, the pinned runtime and a built GitNexus. + "tests/test_offline_sweep_integration.py", + # Carries the real-CLI identity probe, which needs CLAUDE_CANARY_BIN - + # set only on this job. Omitted from this list it skipped everywhere. + "tests/test_mock_provider.py", "-q", ] bwrap_canary_marker = re.compile( @@ -245,6 +278,9 @@ def test_shipped_scenarios_opt_out_the_cross_module_cell_and_rebuild_graph_asset assert skipped == ["cross-module-parse-retry"] assert all(not task.get("sandbox_copy") for task in tasks) assert all(task["sandbox_dependencies"] for task in tasks) + assert all( + any(dep.get("source") == "gitnexus-shared/dist" for dep in task["sandbox_dependencies"]) for task in tasks + ) assert all(task["oracle"]["command"] and task["oracle"]["files"] for task in tasks) assert all("./node_modules/.bin/vitest run" in task["oracle"]["command"] for task in tasks) assert all("npx vitest" not in task["oracle"]["command"] for task in tasks) @@ -512,3 +548,548 @@ def test_run_evolution_script_is_the_shared_ci_and_local_entrypoint(): assert "--include-expensive" in argv assert "claude-sonnet-5" not in argv assert printed.stderr # rewrite notice goes to stderr + + +def test_planned_paid_cells_treat_missing_reuse_as_paid(): + task = {"id": "review-pr-2718-defect"} + assert task_has_planned_paid_cells( + task, + arms=["ce_review", "review", "candidate_review"], + runs=3, + reusable_rows={}, + reuse_source=None, + ) + reuse_source = Path("/tmp/seed") + rows = { + (task["id"], arm, run_idx): {} + for run_idx in range(3) + for arm in ("ce_review", "review", "candidate_review") + } + assert not task_has_planned_paid_cells( + task, + arms=["ce_review", "review", "candidate_review"], + runs=3, + reusable_rows=rows, + reuse_source=reuse_source, + ) + del rows[(task["id"], "candidate_review", 0)] + assert task_has_planned_paid_cells( + task, + arms=["ce_review", "review", "candidate_review"], + runs=3, + reusable_rows=rows, + reuse_source=reuse_source, + ) + + +def test_next_graph_prefetch_skips_ready_shas_and_fully_reused_tasks(tmp_path: Path): + first = {"id": "review-a"} + second = {"id": "review-b"} + third = {"id": "review-c"} + reuse_source = tmp_path / "seed" + reused_second = { + (second["id"], arm, 0): {} for arm in ("ce_review", "review", "candidate_review") + } + target = next_graph_prefetch_target( + [ + (first, {"repo_identity": "/repo", "resolved_sha": "aaa"}), + (second, {"repo_identity": "/repo", "resolved_sha": "bbb"}), + (third, {"repo_identity": "/repo", "resolved_sha": "ccc"}), + ], + arms=["ce_review", "review", "candidate_review"], + runs=1, + reusable_rows=reused_second, + reuse_source=reuse_source, + ready_keys={("/repo", "aaa")}, + ) + assert target is not None + task, binding, key = target + assert task["id"] == "review-c" + assert key == ("/repo", "ccc") + assert binding["resolved_sha"] == "ccc" + + +def test_prefetch_next_graph_runs_ensure_on_a_background_thread(monkeypatch): + started = threading.Event() + seen: list[tuple[str, str]] = [] + + def fake_ensure(**kwargs): + seen.append(kwargs["graph_key"]) + started.set() + + monkeypatch.setattr("workflow_bench.runner.ensure_task_graph", fake_ensure) + cancel = threading.Event() + job = prefetch_next_graph( + task={"id": "review-b"}, + binding={"repo_identity": "/repo", "resolved_sha": "bbb"}, + graph_key=("/repo", "bbb"), + env=GraphBuildEnv( + trees=Path("/tmp"), + task_asset_cache=None, + claude_bin="claude", + bwrap_bin="bwrap", + sandbox_backend="bwrap", + runtime_mounts=(), + clone_templates={}, + clone_template_errors={}, + graph_snapshots={}, + graph_snapshot_errors={}, + ), + cancel_event=cancel, + ) + assert job.key == ("/repo", "bbb") + assert started.wait(timeout=2) + job.join() + assert seen == [("/repo", "bbb")] + + +def test_a_reused_resolution_does_not_count_as_this_sweeps_health(): + """resolved counts evidence; resolved_fresh counts evidence measured today. + + broken_incumbent_arms reads resolved_fresh because a reused row proves last + generation's environment worked. Counting it would make an arm whose cells + were all reused look healthy in exactly the run where a broken environment + should have been caught. + """ + + reused = [record(resolved=True, reused=True), record(resolved=True, reused=True)] + agg = aggregate(reused) + assert agg["resolved"] == 2 + assert agg["resolved_fresh"] == 0 + assert broken_incumbent_arms({"t": {"review": agg}}, {"review"}) == ["review"] + + mixed = aggregate([record(resolved=True, reused=True), record(resolved=True)]) + assert mixed["resolved_fresh"] == 1 + assert broken_incumbent_arms({"t": {"review": mixed}}, {"review"}) == [] + + +def test_graph_build_env_ready_keys_covers_successes_and_failures(): + """A key that failed is attempted, not pending. + + next_graph_prefetch_target skips keys already in ready_keys. If a failed + build were omitted, the sweep would prefetch it again every iteration and + pay a full clone and offline index each time for a build that cannot + succeed. + """ + + env = GraphBuildEnv( + trees=Path("/tmp"), + task_asset_cache=None, + claude_bin="claude", + bwrap_bin="bwrap", + sandbox_backend="bwrap", + runtime_mounts=(), + clone_templates={("/repo", "aaa"): (Path("/tmp/a"), "aaa")}, + clone_template_errors={("/repo", "bbb"): OSError("clone failed")}, + graph_snapshots={("/repo", "ccc"): object()}, + graph_snapshot_errors={("/repo", "ddd"): OSError("index failed")}, + ) + assert env.ready_keys() == { + ("/repo", "aaa"), + ("/repo", "bbb"), + ("/repo", "ccc"), + ("/repo", "ddd"), + } + + +def _cell(**overrides) -> dict[str, Any]: + """One results.jsonl row, healthy unless told otherwise.""" + + base = record(resolved=True) + base.update({"error_kind": None, "review_evidence_valid": True, "transcript_missing": False}) + base.update(overrides) + return base + + +def _arms(**by_arm) -> dict[str, dict[str, dict[str, Any]]]: + return {"task0": {arm: aggregate(rows) for arm, rows in by_arm.items()}} + + +def test_a_reviewer_that_scores_badly_is_not_an_unhealthy_harness(): + """Reconstructed from Actions run 33962002890's logged observations. + + Every completed cell was resolved=False with error_kind=oracle-failed, at a + median score of 0.212 — the reviews ran, wrote artifacts and were scored. + That is a valid negative for the quality gate to judge. Diagnosing it as a + broken environment is the confusion this classification exists to end. + """ + + scored_but_wrong = [_cell(resolved=False, error_kind="oracle-failed") for _ in range(3)] + results = _arms(review=scored_but_wrong, ce_review=list(scored_but_wrong)) + assert unhealthy_arms(results, {"review", "ce_review"}) == [] + health = arm_health(results, {"review"})["review"] + assert health.admissible == 3 and health.fresh_attempts == 3 + assert (health.execution_failures, health.evidence_failures) == (0, 0) + + +def test_an_all_zero_score_is_still_a_valid_negative(): + zeroed = [_cell(resolved=False, error_kind="oracle-failed", review_weighted_f1=0.0) for _ in range(3)] + assert unhealthy_arms(_arms(review=zeroed), {"review"}) == [] + + +def test_artifacts_that_were_never_written_are_an_unhealthy_harness(): + """Reconstructed from Actions run 33912693948. + + All 41 artifacts came back 0 bytes because the mount made an atomic write + impossible. The reviews could not produce evidence at all — the opposite of + the case above, and the one a health check must catch. The old caller + excluded review arms entirely, so it could not have. + """ + + unwritable = [_cell(resolved=False, ok=False, error_kind="review-evidence-invalid") for _ in range(3)] + flagged = unhealthy_arms(_arms(review=unwritable), {"review"}) + assert [h.arm for h in flagged] == ["review"] + assert flagged[0].evidence_failures == 3 + assert "review-evidence-invalid" in flagged[0].reasons + + +def test_one_admissible_cell_leaves_an_arm_degraded_not_healthy(): + """Mixed outcomes are DEGRADED. One usable measurement does not erase two failures. + + Not fatal - the sweep still produced evidence - but calling it healthy is + how a partly-broken environment passes review. + """ + + mixed = [ + _cell(resolved=False, error_kind="oracle-failed"), + _cell(resolved=False, ok=False, error_kind="session-error"), + _cell(resolved=False, ok=False, error_kind="infra-error"), + ] + results = _arms(review=mixed) + health = arm_health(results, {"review"})["review"] + assert health.status == "DEGRADED" + assert unhealthy_arms(results, {"review"}) == [], "degraded is diagnostic, not fatal" + assert health.execution_failures == 2, "failures must stay visible, not be erased" + assert health.admissible == 1 + + +def test_a_row_that_fails_both_ways_is_only_subtracted_once(): + """run_arm can produce a row that is an execution AND an evidence failure. + + It keeps the first error_kind — a session-error survives — and still sets + review_evidence_valid=False when the artifact will not parse. Counting that + row against admissible twice zeroed an arm that held a real measurement, + which arm_health reports as UNUSABLE and the measurement gate then fails on. + """ + + both = _cell(resolved=False, ok=False, error_kind="session-error", review_evidence_valid=False) + results = _arms(review=[both, _cell(resolved=True, error_kind="oracle-failed")]) + health = arm_health(results, {"review"})["review"] + assert (health.execution_failures, health.evidence_failures) == (1, 1) + assert health.fresh_attempts == 2 + assert health.admissible == 1 + assert health.status == "DEGRADED" + assert unhealthy_arms(results, {"review"}) == [] + + +def test_reused_rows_alone_leave_current_health_unknown(): + """Historical success cannot certify this sweep's environment.""" + + reused = [_cell(reused=True) for _ in range(3)] + results = _arms(review=reused) + assert unmeasured_arms(results, {"review"}) == ["review"] + assert unhealthy_arms(results, {"review"}) == [] + assert arm_health(results, {"review"})["review"].measured is False + + +def test_the_paid_canary_survives_a_prior_run_with_more_run_indices(): + """The canary counts planned cells, not every key reuse selection returned. + + Reuse selection accepts any non-negative prior `run`, so a results directory + produced with --runs 5 leaves keys this sweep never plans. Comparing against + those made the "arm is fully reused" test false exactly when it was true, + and the incumbent went a whole sweep without one measured cell. + """ + + tasks = [{"id": "task0"}, {"id": "task1"}] + reusable = {(task["id"], "review", run): {} for task in tasks for run in range(5)} + + dropped = runner.drop_canary_reuse_key(reusable, arm="review", tasks=tasks, runs=3) + + assert dropped == ("task0", "review", 0) + assert dropped not in reusable + # A second call is a no-op: the arm now has its paid cell. + assert runner.drop_canary_reuse_key(reusable, arm="review", tasks=tasks, runs=3) is None + + +def test_an_arm_with_a_planned_paid_cell_keeps_every_reusable_row(): + tasks = [{"id": "task0"}] + reusable = {("task0", "review", 0): {}} + + assert runner.drop_canary_reuse_key(reusable, arm="review", tasks=tasks, runs=2) is None + assert len(reusable) == 1 + + +def test_reused_successes_do_not_mask_fresh_execution_failures(): + rows = [_cell(reused=True), _cell(reused=True), _cell(ok=False, error_kind="session-error")] + flagged = unhealthy_arms(_arms(review=rows), {"review"}) + assert [h.arm for h in flagged] == ["review"] + assert flagged[0].fresh_attempts == 1 and flagged[0].execution_failures == 1 + + +def test_a_parseable_artifact_does_not_excuse_a_failed_session(): + """Artifact parseability must not override an execution failure.""" + + rows = [_cell(ok=False, error_kind="session-error", review_evidence_valid=True) for _ in range(2)] + flagged = unhealthy_arms(_arms(review=rows), {"review"}) + assert [h.arm for h in flagged] == ["review"] + assert flagged[0].execution_failures == 2 + + +def test_a_single_unusable_review_is_caught_below_the_breaker_threshold(): + """The decisive regression for the finalization guard. + + A fixture of 41 empty artifacts would abort through the outage breaker - + review-evidence-invalid is systemic and the limit is 5 - so it proves + nothing about this path. One fresh unusable cell is under that threshold, + which leaves the finalization check as the only thing that can catch it. + """ + + streak = 0 + for _ in range(1): + streak = runner.systemic_outage_streak("review-evidence-invalid", streak) + assert streak < runner.DEFAULT_OUTAGE_STREAK, "fixture must not reach the breaker" + + results = _arms(review=[_cell(resolved=False, ok=False, error_kind="review-evidence-invalid")]) + with pytest.raises(SystemExit) as exc: + runner.enforce_measurement_health(results, {"review"}) + assert exc.value.code == 1 + + +def test_finalization_reports_every_arm_and_names_no_cause(capsys): + """Status for each arm; an empty artifact does not become an EROFS diagnosis.""" + + results = _arms( + review=[_cell(resolved=False, ok=False, error_kind="review-evidence-invalid")], + ce_review=[_cell(resolved=False, error_kind="oracle-failed")], + ) + with pytest.raises(SystemExit): + runner.enforce_measurement_health(results, {"review", "ce_review"}) + out = capsys.readouterr().out + assert "review: UNUSABLE" in out + assert "ce_review: OBSERVED_OK" in out + assert "cause=undetermined" in out + assert "EROFS" not in out and "mount" not in out + + +def test_valid_negatives_do_not_abort_finalization(capsys): + """The 16h run's shape must survive the real guard, not just the classifier.""" + + scored_but_wrong = [_cell(resolved=False, error_kind="oracle-failed") for _ in range(3)] + health = runner.enforce_measurement_health( + _arms(review=scored_but_wrong, ce_review=list(scored_but_wrong)), {"review", "ce_review"} + ) + assert {h.status for h in health.values()} == {"OBSERVED_OK"} + assert "UNUSABLE" not in capsys.readouterr().out + + +def test_reused_only_arm_is_reported_unknown_by_finalization(capsys): + runner.enforce_measurement_health(_arms(review=[_cell(reused=True)]), {"review"}) + assert "review: UNKNOWN" in capsys.readouterr().out + + +def test_run_sweep_calls_the_health_guard_and_not_the_legacy_helper(): + """Pins the wiring the caller correction exposed. + + Reads the compiled code object's global references rather than the source + text: deleting the call removes the name and fails this test, which is the + mutation check. It does NOT prove the guard runs end to end - _run_sweep + needs bwrap and a sandbox, so no test here drives it. + """ + + referenced = runner._run_sweep.__code__.co_names + assert "enforce_measurement_health" in referenced + assert "broken_incumbent_arms" not in referenced + + +def test_ce_review_is_classified_even_though_it_is_not_a_candidate_arm(): + """ce_review is a comparator, absent from CANDIDATE_ARMS. + + Dropping the `- {"review"}` exclusion alone would have left it unchecked. + """ + + assert "ce_review" not in set(CANDIDATE_ARMS.values()) + health = arm_health(_arms(ce_review=[_cell()]), {"review", "ce_review"}) + assert "ce_review" in health + + +def _packed_cells(tasks: int, runs: int, arms: tuple[str, ...]) -> list[tuple[str, int, str]]: + return [(f"t{t}", r, a) for t in range(tasks) for r in range(runs) for a in arms] + + +def test_packed_sweep_runs_every_cell_and_folds_in_submission_order(): + """Fold order is the contract the breaker rests on. + + Cells finish in whatever order the pool returns them, but the breaker counts + CONSECUTIVE systemic failures, which only means something in a fixed order. + """ + + cells = _packed_cells(3, 2, ("review", "candidate_review")) + folded: list[tuple[str, int, str]] = [] + streak, tripped = runner.sweep_packed_cells( + cells, + workers=4, + run=lambda task, run_idx, arm: {"error_kind": None, "review_evidence_valid": True}, + on_start=lambda *_: None, + on_record=lambda task, run_idx, arm, _rec: folded.append((task, run_idx, arm)), + outage_streak=0, + outage_limit=0, + ) + assert folded == cells + assert (streak, tripped) == (0, False) + + +def test_packed_sweep_trips_the_breaker_on_the_same_cell_waves_would(): + """Packing must not change WHEN a doomed run aborts, only how it is fed.""" + + cells = _packed_cells(3, 3, ("review",)) + fail_from = 2 + folded: list[int] = [] + + def run(task: str, run_idx: int, arm: str) -> dict[str, Any]: + index = cells.index((task, run_idx, arm)) + systemic = index >= fail_from + return { + "error_kind": "session-error" if systemic else None, + "review_evidence_valid": not systemic, + } + + streak, tripped = runner.sweep_packed_cells( + cells, + workers=2, + run=run, + on_start=lambda *_: None, + on_record=lambda t, r, a, _rec: folded.append(cells.index((t, r, a))), + outage_streak=0, + outage_limit=runner.DEFAULT_OUTAGE_STREAK, + ) + assert tripped is True + assert streak == runner.DEFAULT_OUTAGE_STREAK + # Five consecutive systemic failures starting at index 2 -> trips on index 6. + assert folded[-1] == fail_from + runner.DEFAULT_OUTAGE_STREAK - 1 + assert folded == sorted(folded), "records must fold in submission order" + + +def test_packed_sweep_skips_a_task_whose_assets_never_arrive(): + """A task that cannot be prepared is skipped, not run against nothing.""" + + cells = _packed_cells(3, 2, ("review",)) + ran: list[str] = [] + runner.sweep_packed_cells( + cells, + workers=3, + run=lambda task, run_idx, arm: ran.append(task) + or {"error_kind": None, "review_evidence_valid": True}, + on_start=lambda *_: None, + on_record=lambda *_: None, + outage_streak=0, + outage_limit=0, + await_ready=lambda task: task != "t1", + ) + assert set(ran) == {"t0", "t2"} + assert "t1" not in ran + + +def test_packed_sweep_workers_inherit_the_runs_cancellation_event(): + """A worker that cannot see the event runs on after the sweep is cancelled. + + The cells are submitted from a producer THREAD, and a new thread starts with + an empty context - so copying the context at submission copies the wrong one + unless the caller's is captured first. run_managed falls back to + _CANCELLATION when no event is passed, which is how a cell's subprocesses + learn the run was cancelled at all. + """ + + seen: list[threading.Event | None] = [] + event = threading.Event() + with cancellation_scope(event): + runner.sweep_packed_cells( + _packed_cells(2, 1, ("review",)), + workers=2, + run=lambda *_: seen.append(_CANCELLATION.get()) or {"error_kind": None}, + on_start=lambda *_: None, + on_record=lambda *_: None, + outage_streak=0, + outage_limit=0, + ) + assert seen and all(observed is event for observed in seen) + + +def test_packed_sweep_window_must_keep_the_pool_fed(): + with pytest.raises(ValueError, match="window must be at least workers"): + runner.sweep_packed_cells( + _packed_cells(1, 1, ("review",)), + workers=4, + run=lambda *_: {"error_kind": None}, + on_start=lambda *_: None, + on_record=lambda *_: None, + outage_streak=0, + outage_limit=0, + window=2, + ) + + +def test_a_raising_packed_cell_still_persists_its_settled_siblings(): + """A crash in one cell must not erase the evidence of cells that finished. + + run_cell deliberately lets unexpected harness exceptions propagate, and the + wave scheduler answers that by folding every non-failing sibling before it + re-raises. The packed scheduler has to hold the same contract: the later + cells already ran and already cost money, so losing their rows would mean + paying for evidence the sweep then throws away. + """ + + folded: list[tuple[int, str]] = [] + started = threading.Event() + + def run(task_id: str, run_idx: int, arm: str) -> dict[str, Any]: + if run_idx == 0: + # Let the later cell finish first, so there is settled evidence to + # lose at the moment this one raises. + started.wait(timeout=5) + raise RuntimeError("harness bug in cell 0") + started.set() + return {"error_kind": None} + + with pytest.raises(RuntimeError, match="harness bug in cell 0"): + runner.sweep_packed_cells( + _packed_cells(1, 2, ("review",)), + workers=2, + run=run, + on_start=lambda *_: None, + on_record=lambda task_id, run_idx, arm, _rec: folded.append((run_idx, arm)), + outage_streak=0, + outage_limit=0, + ) + + assert (1, "review") in folded, "the sibling that completed was never recorded" + + +def test_an_uninvoked_skill_still_counts_toward_the_arm_median(): + """Pins a KNOWN GAP, not a desired behaviour. + + A cell whose skill never ran still moves the arm's quality median, even + though an arm exists to measure a SKILL. The narrow fix - filtering those + rows out of the quality metrics - is worse than the gap: valid_runs and + excluded_runs keep counting them, so the promotion gate sees N clean runs + while the median came from fewer. Since the dropped rows are systematically + an arm's worst, that biases toward promoting, and it was measured flipping + keep_incumbent to promote. + + Closing it honestly needs a scored-run count and a paired-equality check in + the promotion gate. Pinned here so the half-fix cannot be reapplied without + someone reading why it was reverted. + """ + + good = record(review_weighted_f1=1.0, cost_usd=2.0) + uninvoked = record( + review_weighted_f1=0.0, cost_usd=4.0, error_kind="skill-not-invoked", skill_invoked=False + ) + agg = aggregate([good, uninvoked]) + + assert agg["review_weighted_f1"] == 0.5, "the uninvoked row is counted - the known gap" + assert agg["cost_usd"] == 3.0 + # The invariant that makes the half-fix unsafe: the median and the run count + # the gate reads must cover the same rows. + assert agg["valid_runs"] == 2 + diff --git a/eval/tests/test_workflow_bench_sessions.py b/eval/tests/test_workflow_bench_sessions.py index 1f66f071f..5d1692fbb 100644 --- a/eval/tests/test_workflow_bench_sessions.py +++ b/eval/tests/test_workflow_bench_sessions.py @@ -12,9 +12,9 @@ from types import SimpleNamespace import pytest -from workflow_bench import evolve, runner, runner_sessions, runtime_mounts +from workflow_bench import evolve, runner, runner_artifacts, runner_sessions, runtime_mounts from workflow_bench.evolution import skill_fingerprint -from workflow_bench.process_control import ManagedProcessResult +from workflow_bench.process_control import ManagedProcessError, ManagedProcessResult from workflow_bench.proposer_sandbox import SandboxError from workflow_bench.runner import snapshot_plan_docs @@ -120,11 +120,16 @@ def skill_events(skill_input: dict, *, tool_id: str = "skill-1", is_error: bool def fake_sandbox(root: Path) -> SimpleNamespace: + # private_root is NOT the clone. Conflating them puts the review artifact + # directory inside the workspace, which the real sandbox never does and + # which hides whether the workspace was left untouched. + private_root = root.parent / f"{root.name}-sandbox-private" + private_root.mkdir(exist_ok=True) return SimpleNamespace( backend="test-double", claude_bin="claude", clone=root, - private_root=root, + private_root=private_root, command_prefix=[], command_prefix_for=lambda **_kwargs: [], settings_json="{}", @@ -1249,7 +1254,7 @@ def test_planning_cannot_change_source_tests_or_downstream_skill(monkeypatch, tm @pytest.mark.parametrize( ("attack", "expected_detail"), [ - ("workspace", "unauthorized workspace path"), + ("workspace", "changed the read-only workspace"), ("skill", "changed the evaluated skill fingerprint"), ], ) @@ -1265,9 +1270,11 @@ def test_review_phase_rejects_workspace_or_skill_mutation( expected_skill_digest = "expected-skill-fingerprint" def adversarial_review(prompt, *args, **kwargs): - (tmp_path / "review-output.json").write_text( - '{"schema_version":1,"verdict":"approve","findings":[]}' - ) + # Write where the contract now says: the artifact directory outside the + # workspace, which is the only place the agent can write atomically. + artifact = runner.review_output_path(sandbox, runner.REVIEW_OUTPUT) + artifact.parent.mkdir(parents=True, exist_ok=True) + artifact.write_text('{"schema_version":1,"verdict":"approve","findings":[]}') if attack == "workspace": source.write_text("review silently changed source") return session_record() @@ -1298,6 +1305,97 @@ def test_review_phase_rejects_workspace_or_skill_mutation( assert expected_detail in rec["error_detail"] +def test_a_cancelled_clone_copy_does_not_fall_back_to_an_uncancellable_copytree(monkeypatch, tmp_path): + """The reflink fallback is for a filesystem, not for a teardown. + + run_managed reports cancellation as a non-OK result rather than raising, so + the fallback treated it like an unsupported reflink and started a copytree + that cannot be cancelled — waiting out exactly the full copy the outage + breaker set the cancellation event to avoid. + """ + + source = tmp_path / "template" + (source / ".git").mkdir(parents=True) + parent = tmp_path / "clones" + parent.mkdir() + copied: list[object] = [] + + monkeypatch.setattr( + runner_artifacts, + "run_managed", + lambda *_a, **_k: ManagedProcessResult( + state="cancelled", + returncode=None, + stdout_tail="", + stderr_tail="", + duration_s=0.1, + ), + ) + monkeypatch.setattr(runner_artifacts.shutil, "copytree", lambda *a, **k: copied.append(a)) + + with pytest.raises(ManagedProcessError): + runner.copy_isolated_tree(source, parent) + assert copied == [] + assert list(parent.iterdir()) == [], "the partial target must be cleaned up" + + +@pytest.mark.parametrize("arm", ["review", "ce_review"]) +def test_run_arm_mounts_the_review_artifact_directory_outside_the_workspace(monkeypatch, tmp_path, arm): + """A writable FILE inside a read-only directory is not a writable path. + + The Write tool creates `.tmp..` beside the target and + renames it, so a read-only parent fails the temp create with EROFS and the + artifact stays 0 bytes. The mount target must be the directory, and it must + sit outside the read-only workspace. + + Driven through run_arm rather than rebuilt here: an expected tuple assembled + in the test passes whatever run_arm actually mounts, which is the one thing + this needs to prove. + """ + + assert not runner.SANDBOX_REVIEW_OUTPUT.startswith(runner.SANDBOX_WORKSPACE + "/") + assert runner.SANDBOX_REVIEW_OUTPUT != runner.SANDBOX_WORKSPACE + + verify_calls: list[dict] = [] + sandbox = fake_sandbox(tmp_path) + sandbox.command_prefix_for = lambda **kwargs: verify_calls.append(kwargs) or [] + + def review_session(prompt, *args, **kwargs): + artifact = runner.review_output_path(sandbox, runner.REVIEW_OUTPUT) + artifact.parent.mkdir(parents=True, exist_ok=True) + artifact.write_text('{"schema_version":1,"verdict":"approve","findings":[]}') + return session_record() + + monkeypatch.setattr(runner, "run_claude", review_session) + monkeypatch.setattr(runner, "skill_fingerprint", lambda *_a, **_k: "skill-digest") + monkeypatch.setattr(runner, "run_verify", lambda *a, **k: (True, "ok")) + + runner.run_arm( + arm, + {"prompt": "p", "verify": "true"}, + tmp_path, + bench_args(), + sandbox=sandbox, + expected_skill_digest="skill-digest", + ) + + review_output = runner.review_output_path(sandbox, runner.REVIEW_OUTPUT) + expected = (runner.ReadOnlyMount(source=review_output.parent, target=runner.SANDBOX_REVIEW_OUTPUT),) + # The EROFS bug is about the AGENT's write, so the mount that has to be the + # directory is the writable one on the review session — not the read-only + # exposure the verify command gets afterwards. Assert both: they are + # separate arguments to separate command prefixes. + writable = [call["extra_writable_mounts"] for call in verify_calls if "extra_writable_mounts" in call] + assert writable, "the review session must be given a writable artifact mount" + assert writable[-1] == expected, "mount the directory, not the file" + read_only = [call["extra_read_only_mounts"] for call in verify_calls if "extra_read_only_mounts" in call] + assert read_only, "the verify invocation must be given the artifact mount" + assert read_only[-1] == expected, "mount the directory, not the file" + assert not expected[0].target.startswith(f"{runner.SANDBOX_WORKSPACE}/") + # The artifact the harness later reads is the one inside that mount. + assert review_output.parent in review_output.parents + + def _git(repo, *args, check=True): return subprocess.run(["git", "-C", str(repo), *args], check=check, capture_output=True, text=True) @@ -1348,3 +1446,34 @@ def test_make_worktree_clone_has_no_tags_but_keeps_all_branches(tmp_path): current = _git(target, "rev-parse", "HEAD").stdout.strip() assert current == other_sha + + +def test_copy_isolated_tree_does_not_share_git_objects_or_refs(tmp_path): + repo = tmp_path / "repo" + repo.mkdir() + _git(repo, "init", "--quiet") + _git(repo, "checkout", "--quiet", "-b", "main") + sha = _git_commit(repo, "base") + clones = tmp_path / "clones" + clones.mkdir() + template = runner.make_worktree(repo, sha, clones) + (template / "marker.txt").write_text("template\n") + + copy = runner.copy_isolated_tree(template, clones) + assert copy != template + assert (copy / "marker.txt").read_text() == "template\n" + (copy / "marker.txt").write_text("copy\n") + assert (template / "marker.txt").read_text() == "template\n" + copy_head = _git(copy, "rev-parse", "HEAD").stdout.strip() + template_head = _git(template, "rev-parse", "HEAD").stdout.strip() + assert copy_head == template_head == sha + # An equal initial HEAD is also what a shared ref namespace looks like, so + # write a ref and prove the template cannot see it. A linked worktree would + # pass every assertion above, including the alternates check — its `.git` is + # a file, so the directory inspected below simply does not exist. + _git(copy, "branch", "copy-only") + assert _git(copy, "show-ref", "--verify", "refs/heads/copy-only").returncode == 0 + assert _git(template, "show-ref", "--verify", "refs/heads/copy-only", check=False).returncode != 0 + assert (copy / ".git").is_dir() + alternates = copy / ".git" / "objects" / "info" / "alternates" + assert not alternates.exists() diff --git a/eval/workflow_bench/README.md b/eval/workflow_bench/README.md index 79bd4bfea..a0008fa7d 100644 --- a/eval/workflow_bench/README.md +++ b/eval/workflow_bench/README.md @@ -149,8 +149,10 @@ UNSAFE_NO_BWRAP=1 RUNS=1 ./workflow_bench/run-evolution.sh This mode runs review sessions directly in disposable host worktrees and is **not** a security boundary: it does not isolate the network or create a PID namespace, and a session that can `chmod` can undo the workspace lock. The -harness still drops write bits on the clone except `review-output.json` so -accidental `npm install` / analyze writes cannot invalidate review evidence. +harness drops write bits on the whole clone, with no carve-out, so accidental +`npm install` / analyze writes cannot invalidate review evidence. The review +artifact is not in the clone at all: it lives in a writable directory bound at +`/review-output`, outside the workspace. Sandbox cleanup restores owner write bits before deleting the private TMPDIR, because a session that `copytree`s the locked clone would otherwise leave non-empty 0555 directories that `rmtree` cannot remove. Historical review @@ -168,6 +170,36 @@ router thresholds as an incumbent policy, not permanent truth. Candidate changes run offline in the same throwaway clones as the incumbent; production skills never rewrite themselves from a live task. +On the self-hosted evolution box, `run-evolution.sh` passes +`--max-runtime-from-instance-window` and the CLI derives its own cap from +`/proc/uptime` at startup (24h EventBridge window minus a 90-minute upload +reserve), in the same breath as it starts the clock that cap is measured +against — a budget computed anywhere earlier is spent by the seconds between. A `workflow_dispatch` that lands on an +already-running instance therefore exits in-process instead of vanishing when +the box stops — a cancelled GitHub job skips even `if: always()`, which is +how run 33962002890 lost 51 finished sessions. Local runs are uncapped. + +A review generation is 6 tasks × 3 arms × 3 runs. Serial workers=1 at ~19 +minutes per session is a 16-hour job (run 33962002890). Two harness changes +cut that without shrinking the gate: + +- **Comparator reuse.** `evolve.py` forwards the seed / prior generation as + `--reuse-results`. Incumbent `review` and `ce_review` rows are copied into + the new `results.jsonl` when model, effort, task SHA, prompt digest, oracle + bytes, incumbent skill digest, CE plugin digest, and sandbox backend still + match. Candidate arms always run. A weekly generation with an unchanged + incumbent therefore pays 18 sessions, not 54. A promotion, model change, + task-corpus change, or harness `RUNTIME_DIGEST` change invalidates the + lock and re-runs the comparators. +- **Sanitized clone templates.** Each unique task SHA is cloned and + sanitized once. Cells copy that parentless snapshot (reflink when the + filesystem allows) instead of `git clone --no-local` plus repack/prune/fsck + 54 times. Isolation is a private `.git`, not a second copy of full history. + +Dispatch defaults to `--workers 3` so those 18 paid cells can overlap. Size +workers to the host: a cell that loses CPU and hits the session ceiling is +an excluded run the gate refuses. + Build an overlay that mirrors only the canonical repo-local skill paths: ```text @@ -276,9 +308,12 @@ without weakening today's deterministic promotion boundary. The evolution workflow runs an offline containment preflight with the pinned Claude Code 2.1.214 binary before starting a paid proposer or benchmark. The -review canary seals the workspace read-only and exposes only the pre-created -`review-output.json` as writable. Runtime mount placeholders are prepared in -the disposable clone before sealing it; existing config bytes are preserved. +review canary seals the workspace read-only and writes nothing into it: the +artifact directory is bound at `/review-output` outside the workspace, and the +file itself is deliberately absent until the session creates it, so its absence +distinguishes "never written" from "written badly". Runtime mount placeholders +are prepared in the disposable clone before sealing it; existing config bytes +are preserved. Any pre-existing result entry, including a symlink, is rejected. Required canaries fail when their runtime or Bubblewrap is unavailable. @@ -368,10 +403,10 @@ paired benchmark as any other candidate. For ad-hoc use, run the driver on the existing re-evaluation triggers (model/harness change or 90-day staleness). The repository workflow runs a -deliberate weekly drift check: scheduled concurrency stays serial unless -`GITNEXUS_EVOLUTION_WORKERS` is raised after a funded host-sized proof, and -`--workers` is bounded to 1–8 before paid work starts. `--generations` remains -the only loop bound. +deliberate weekly drift check: dispatch defaults to three concurrent cells +of one task; scheduled concurrency still requires +`GITNEXUS_EVOLUTION_WORKERS=3` after a clean proof. `--workers` is bounded +to 1–8 before paid work starts. `--generations` remains the only loop bound. ## Free-model setup (no paid tokens) diff --git a/eval/workflow_bench/comparator_reuse.py b/eval/workflow_bench/comparator_reuse.py new file mode 100644 index 000000000..5dcef1289 --- /dev/null +++ b/eval/workflow_bench/comparator_reuse.py @@ -0,0 +1,604 @@ +"""Reuse frozen comparator cells when the current sweep is still the same experiment. + +Weekly skill evolution re-runs incumbent ``review`` / ``ce_review`` (and the +implementation incumbents) even when the model, effort, tasks, oracles, +incumbent skill bytes, and CE plugin have not changed. Those arms are the +baseline the gate compares a *new* candidate against — they are not the +thing being evolved. Replaying them burns two-thirds of a generation. + +This module selects prior ``results.jsonl`` rows that are safe to carry +forward. Candidate arms are never reused. A mismatch on any bound field +falls through to a paid cell. Missing artifacts also fall through: a reused +row that the proposer cannot read is worse than spending the tokens again. +""" + +from __future__ import annotations + +import hashlib +import json +import os +import re +import stat +from collections.abc import Iterator, Mapping, Sequence +from contextlib import contextmanager +from dataclasses import dataclass +from datetime import UTC, datetime, timedelta +from pathlib import Path, PurePosixPath +from typing import Any + +from .evolution import CANDIDATE_ARMS, EVIDENCE_MAX_AGE_DAYS +from .proposer_sandbox import SandboxError +from .runner_sessions import MAX_TRANSCRIPT_BYTES, PARENT_EVENT_STREAM_SOURCE +from .runtime_mounts import CE_ARMS +from .task_assets import COPY_CHUNK_BYTES, _write_all + +REUSABLE_COMPARATOR_ARMS = frozenset( + { + "review", + "ce_review", + "workflow", + "workflow_direct", + "ce_workflow", + "ce_workflow_direct", + "baseline", + "baseline_nomcp", + } +) +# Must stay aligned with runner.EXCLUDED_ERROR_KINDS plus review-invalid. +# A reused row becomes promotion evidence; excluded kinds cannot enter that set. +REUSE_EXCLUDED_ERROR_KINDS = frozenset( + { + "session-error", + "infra-error", + "evidence-unverified", + "cleanup-failure", + "review-evidence-invalid", + "cancelled", + } +) +_TRANSCRIPT_NAME = re.compile(r"[A-Za-z0-9._-]{1,200}") +CellKey = tuple[str, str, int] + + +@dataclass(frozen=True) +class TaskReuseBinding: + """Per-task identity the prior row must still match.""" + + task_base_sha: str + task_prompt_digest: str + oracle_digest: str + oracle_command_digest: str + oracle_manifest_digest: str + # The cell's environment is part of its identity: a comparator measured + # against different task assets or different sandbox dependencies is a + # measurement of a different machine, not a baseline for this sweep. + task_asset_manifest_digest: str | None = None + sandbox_dependency_manifest_digest: str | None = None + + +@dataclass(frozen=True) +class ComparatorReuseExpectation: + """Sweep-wide lock for comparator reuse. Any drift pays for a fresh cell.""" + + model: str + effort: str + sandbox_backend: str + runtime_digest: str | None + now: datetime + max_age: timedelta + tasks: Mapping[str, TaskReuseBinding] + skill_digests: Mapping[str, str | None] + ce_plugin_version: str | None + ce_plugin_manifest_digest: str | None + + +def load_result_rows(path: Path) -> list[dict[str, Any]]: + """Load ``results.jsonl``; skip malformed lines the same way evolve does.""" + + rows: list[dict[str, Any]] = [] + for line in path.read_text().splitlines(): + if not line.strip(): + continue + try: + row = json.loads(line) + except json.JSONDecodeError: + continue + if isinstance(row, dict): + rows.append(row) + return rows + + +def current_runtime_digest() -> str | None: + """Harness lockfile digest exported by ``run-evolution.sh``, if present.""" + + value = os.environ.get("RUNTIME_DIGEST", "").strip() + return value or None + + +def row_is_reusable_comparator(row: Mapping[str, Any], expected: ComparatorReuseExpectation) -> bool: + """True when ``row`` is a complete, still-valid comparator measurement.""" + + arm = row.get("arm") + if not isinstance(arm, str) or arm in CANDIDATE_ARMS or arm not in REUSABLE_COMPARATOR_ARMS: + return False + if row.get("error_kind") in REUSE_EXCLUDED_ERROR_KINDS: + return False + if row.get("error_kind") not in (None, ""): + return False + if row.get("ok") is not True: + return False + if row.get("transcript_missing") is True: + return False + if row.get("candidate_overlay_digest") not in (None, ""): + return False + # Age against the ORIGINAL measurement, not the copy time: materialize_reused_row + # restamps recorded_at, so a chained row would otherwise refresh its own clock + # and never expire. Bound both directions - a future stamp is corrupt, not fresh. + recorded = _parse_recorded_at(row.get("reused_from_recorded_at") or row.get("recorded_at")) + if recorded is None: + return False + age = expected.now - recorded + if age > expected.max_age or age < timedelta(0): + return False + if row.get("model") != expected.model and row.get("benchmark_model") != expected.model: + return False + if row.get("effort") != expected.effort: + return False + if row.get("sandbox_backend") != expected.sandbox_backend: + return False + # Fail closed. A row with no runtime_digest was measured by a harness that + # did not record one, which is exactly the drift this lock exists to catch; + # treating the absence as agreement made every legacy row reusable forever. + prior_runtime = row.get("runtime_digest") + if not isinstance(prior_runtime, str) or not prior_runtime: + return False + if not expected.runtime_digest or prior_runtime != expected.runtime_digest: + return False + + task_id = row.get("task") + binding = expected.tasks.get(task_id) if isinstance(task_id, str) else None + if binding is None: + return False + if row.get("task_base_sha") != binding.task_base_sha: + return False + if row.get("task_prompt_digest") != binding.task_prompt_digest: + return False + if row.get("oracle_digest") != binding.oracle_digest: + return False + if row.get("oracle_command_digest") != binding.oracle_command_digest: + return False + if row.get("oracle_manifest_digest") != binding.oracle_manifest_digest: + return False + # Fail closed on both sides, as the runtime digest does: an unbound + # expectation means this sweep could not determine its own environment, and + # a row without the field was measured before it was recorded. + for field, bound in ( + ("task_asset_manifest_digest", binding.task_asset_manifest_digest), + ("sandbox_dependency_manifest_digest", binding.sandbox_dependency_manifest_digest), + ): + prior = row.get(field) + if not isinstance(prior, str) or not prior or not bound or prior != bound: + return False + + if arm in CE_ARMS: + if row.get("ce_plugin_version") != expected.ce_plugin_version: + return False + if row.get("ce_plugin_manifest_digest") != expected.ce_plugin_manifest_digest: + return False + else: + expected_skill = expected.skill_digests.get(arm) + if not expected_skill or row.get("skill_digest") != expected_skill: + return False + + if arm in {"review", "ce_review"}: + if row.get("review_evidence_valid") is not True: + return False + # The artifact, not just the score derived from it. materialize_reused_row + # copies it only when the name is present, so without this a row whose + # artifact copy never happened could be carried forward as a scored + # review that a proposer then cannot read - evidence by assertion. + review_artifact = row.get("review_artifact") + if not isinstance(review_artifact, str) or not review_artifact: + return False + if not isinstance(row.get("review_score"), dict): + return False + if row.get("review_weighted_f1") is None: + return False + + artifacts = row.get("transcript_artifacts") + if not isinstance(artifacts, list) or not artifacts: + return False + try: + for artifact in artifacts: + _transcript_metadata(artifact) + except SandboxError: + return False + return True + + +def select_reusable_comparator_rows( + rows: Sequence[Mapping[str, Any]], + *, + expected: ComparatorReuseExpectation, +) -> dict[CellKey, dict[str, Any]]: + """Index reusable rows by ``(task, arm, run)``. Conflicting duplicates drop the key.""" + + chosen: dict[CellKey, dict[str, Any]] = {} + blocked: set[CellKey] = set() + for row in rows: + if not row_is_reusable_comparator(row, expected): + continue + task_id = row["task"] + arm = row["arm"] + run = row.get("run") + if not isinstance(run, int) or isinstance(run, bool) or run < 0: + continue + key = (str(task_id), str(arm), run) + if key in blocked: + continue + previous = chosen.get(key) + if previous is None: + chosen[key] = dict(row) + continue + if _row_identity(previous) != _row_identity(row): + blocked.add(key) + chosen.pop(key, None) + return chosen + + +def materialize_reused_row( + row: Mapping[str, Any], + *, + source_dir: Path, + dest_dir: Path, +) -> dict[str, Any]: + """Copy digest-bound artifacts into this sweep's evidence dir and stamp reuse.""" + + source, _ = _resolved_directory(source_dir, label="reuse source") + dest, _ = _resolved_directory(dest_dir, label="reuse destination") + if source == dest: + raise SandboxError("comparator reuse cannot read and write the same results directory") + + materialized = dict(row) + materialized["reused"] = True + # Keep the FIRST measurement time across a chain. Overwriting it with the + # previous copy's stamp let a row refresh its own clock every generation and + # outlive the max_age bound entirely. + materialized["reused_from_recorded_at"] = row.get("reused_from_recorded_at") or row.get("recorded_at") + materialized["recorded_at"] = datetime.now(UTC).isoformat() + + artifacts = row.get("transcript_artifacts") + if not isinstance(artifacts, list) or not artifacts: + raise SandboxError("reused row is missing transcript_artifacts") + + # Every path below is resolved against a held descriptor, never re-walked + # from a name. Both roots are already symlink-free (_resolved_directory + # resolved them), and pinning them here means the components under them + # cannot be swapped out from under a check that already passed. + with ( + _open_pinned_root(source_dir, label="reuse source") as source_fd, + _open_pinned_root(dest_dir, label="reuse destination") as dest_fd, + ): + copied_artifacts: list[dict[str, Any]] = [] + for artifact in artifacts: + copied_artifacts.append(_copy_transcript_artifact(source_fd, dest_fd, artifact)) + materialized["transcript_artifacts"] = copied_artifacts + + review_name = row.get("review_artifact") + if isinstance(review_name, str) and review_name: + _copy_named_artifact(source_fd, dest_fd, review_name, label="review artifact") + + task = row.get("task") + arm = row.get("arm") + run = row.get("run") + if isinstance(task, str) and isinstance(arm, str) and isinstance(run, int) and not isinstance(run, bool): + patch_name = f"{task}-{arm}-run{run}.patch" + if _is_regular_at(patch_name, dir_fd=source_fd): + _copy_named_artifact(source_fd, dest_fd, patch_name, label="patch artifact") + return materialized + + +def default_reuse_max_age() -> timedelta: + return timedelta(days=EVIDENCE_MAX_AGE_DAYS) + + +def _row_identity(row: Mapping[str, Any]) -> tuple[Any, ...]: + return ( + row.get("skill_digest"), + row.get("oracle_digest"), + row.get("review_weighted_f1"), + row.get("ce_plugin_manifest_digest"), + row.get("recorded_at"), + ) + + +def _parse_recorded_at(value: Any) -> datetime | None: + if not isinstance(value, str) or not value: + return None + try: + parsed = datetime.fromisoformat(value.replace("Z", "+00:00")) + except ValueError: + return None + if parsed.tzinfo is None: + parsed = parsed.replace(tzinfo=UTC) + return parsed.astimezone(UTC) + + +def _transcript_metadata(metadata: Any) -> tuple[str, str, int]: + if not isinstance(metadata, dict) or set(metadata) != {"path", "sha256", "bytes", "source"}: + raise SandboxError("transcript artifact metadata must contain only path, sha256, bytes, and source") + relative = metadata["path"] + digest = metadata["sha256"] + size = metadata["bytes"] + if metadata["source"] != PARENT_EVENT_STREAM_SOURCE: + raise SandboxError("transcript artifact source is not the parent event stream") + if not isinstance(relative, str) or not isinstance(digest, str) or not re.fullmatch(r"[0-9a-f]{64}", digest): + raise SandboxError("transcript artifact metadata is malformed") + if not isinstance(size, int) or isinstance(size, bool) or size < 0 or size > MAX_TRANSCRIPT_BYTES: + raise SandboxError("transcript artifact byte count is out of range") + relative_path = PurePosixPath(relative) + if ( + relative_path.is_absolute() + or len(relative_path.parts) != 2 + or relative_path.parts[0] != "transcripts" + or any(part in {"", ".", ".."} for part in relative_path.parts) + or _TRANSCRIPT_NAME.fullmatch(relative_path.parts[1]) is None + ): + raise SandboxError(f"unsafe transcript artifact path: {relative!r}") + return relative, digest, size + + +def _resolved_directory(path: Path, *, label: str) -> tuple[Path, tuple[int, int]]: + """An existing, non-symlink directory, resolved through its parents. + + Deliberately weaker than proposer_sandbox's same-shaped helper, which + refuses every symlink hop in the path. That one guards a MOUNT ROOT, where + a hop changes what an untrusted session is handed. This one guards a DATA + directory whose contents are validated individually anyway - every file + read goes through ``_regular_file`` (lstat, symlinks rejected) and every + write through ``O_NOFOLLOW`` - so a symlinked parent grants nothing those + guards do not already cover, while refusing one would reject ordinary + setups such as a symlinked artifacts directory or macOS's /var. + + Separately named because they make different promises. Do not merge them + without first deciding which promise the reuse path should make. + """ + + resolved = path.expanduser() + try: + metadata = resolved.lstat() + except OSError as exc: + raise SandboxError(f"{label} is unavailable: {resolved}: {exc}") from exc + if stat.S_ISLNK(metadata.st_mode) or not stat.S_ISDIR(metadata.st_mode): + raise SandboxError(f"{label} must be a real directory: {resolved}") + return resolved.resolve(), (metadata.st_dev, metadata.st_ino) + + +@contextmanager +def _open_pinned_root(path: Path, *, label: str) -> Iterator[int]: + """Open a checked root and prove it is still the directory that was checked. + + The symlink POLICY above is deliberate and unchanged: parent hops stay + allowed, so a symlinked artifacts directory or macOS's /var still works. + What is closed here is separate from that policy - the gap between checking + a name and using it. lstat names one directory and resolve() re-walks the + same name afterwards, so a prior sweep that renames its results root and + drops a symlink in its place is resolved to somewhere else entirely, and + O_NOFOLLOW on the open cannot see a link that resolve() already followed. + + Comparing the opened descriptor's identity to the checked one costs an + fstat and rejects nothing that holds still: a stable directory always + matches itself. It matters for reuse specifically because the failure is + silent - rows would be copied out of the wrong directory and folded into a + comparator baseline as though they were this sweep's own evidence. + """ + + resolved, expected = _resolved_directory(path, label=label) + with _open_real_directory(resolved, label=label) as fd: + opened = os.fstat(fd) + if (opened.st_dev, opened.st_ino) != expected: + raise SandboxError(f"{label} was replaced between the check and the open: {resolved}") + yield fd + + +def _copy_transcript_artifact(source_fd: int, dest_fd: int, metadata: Mapping[str, Any]) -> dict[str, Any]: + relative, expected_digest, expected_size = _transcript_metadata(metadata) + name = PurePosixPath(relative).name + # Both `transcripts` components are opened as descriptors, not checked as + # names. An lstat that passes and a pathname that is used afterwards are two + # different directories whenever a concurrent writer renames the first one + # away — which the reuse directory, written by a prior sweep, invites. + with ( + _open_real_directory("transcripts", dir_fd=dest_fd, label="transcript destination", create=True) as dest_dir_fd, + _open_real_directory("transcripts", dir_fd=source_fd, label="transcript source") as source_dir_fd, + ): + os.fchmod(dest_dir_fd, 0o700) + # One descriptor for the whole transfer, and ONE read of it. Hashing the + # source and then reading it again to copy leaves the recorded digest + # describing bytes that are not the bytes written: the descriptor stops + # the pathname being substituted, not the inode being rewritten, and + # this directory belongs to a sweep that may still be writing. Digest + # what is copied, then judge it. + with _open_regular(name, dir_fd=source_dir_fd, label="transcript") as artifact_fd: + digest, copied_bytes = _copy_owner_only( + artifact_fd, name, dir_fd=dest_dir_fd, max_bytes=expected_size + ) + if copied_bytes != expected_size or digest != expected_digest: + # The destination now holds bytes no expectation vouches for. + os.unlink(name, dir_fd=dest_dir_fd) + drift = "size" if copied_bytes != expected_size else "digest" + raise SandboxError(f"reused transcript {drift} drifted: {relative}") + return {"path": relative, "sha256": digest, "bytes": expected_size, "source": PARENT_EVENT_STREAM_SOURCE} + + +def _copy_named_artifact(source_fd: int, dest_fd: int, name: str, *, label: str) -> None: + relative = PurePosixPath(name) + if relative.is_absolute() or len(relative.parts) != 1 or relative.parts[0] in {"", ".", ".."}: + raise SandboxError(f"unsafe {label} path: {name!r}") + with _open_regular(name, dir_fd=source_fd, label=label) as artifact_fd: + # No expectation is recorded for these, so the digest is discarded - but + # "no recorded size" is not "no limit". The source is a prior sweep + # directory that can change between sweeps, so a replaced artifact could + # be arbitrarily large; MAX_TRANSCRIPT_BYTES is the ceiling the capture + # path already enforces on evidence of this kind. + _, copied = _copy_owner_only(artifact_fd, name, dir_fd=dest_fd, max_bytes=MAX_TRANSCRIPT_BYTES) + if copied > MAX_TRANSCRIPT_BYTES: + os.unlink(name, dir_fd=dest_fd) + raise SandboxError(f"reused {label} exceeds {MAX_TRANSCRIPT_BYTES} bytes: {name}") + + +def _require_openat() -> None: + """openat is what makes a checked directory and a used directory the same one. + + Without it the only alternative is to re-walk the name after the check, + which is exactly the race this module is guarding. Refusing is safe: the + caller in runner treats a SandboxError from reuse as "run a paid cell", so + a platform without openat pays for the cells rather than copying through a + directory nobody verified. The sweep itself is Linux-only anyway (bwrap, + /proc/uptime); this is about the unit tests and about failing loudly. + """ + + if os.open not in os.supports_dir_fd or os.lstat not in os.supports_dir_fd: + raise SandboxError("comparator reuse requires POSIX openat support (os.supports_dir_fd)") + + +def _is_regular_at(name: str, *, dir_fd: int) -> bool: + """True when `name` under the pinned directory is a regular non-symlink file.""" + + try: + metadata = os.lstat(name, dir_fd=dir_fd) + except OSError: + return False + return stat.S_ISREG(metadata.st_mode) + + +@contextmanager +def _open_real_directory( + path: Path | str, + *, + dir_fd: int | None = None, + label: str, + create: bool = False, +) -> Iterator[int]: + """Open one directory that is not a symlink, and hold it for every use below. + + ``O_DIRECTORY | O_NOFOLLOW`` makes the check and the open a single syscall, + so unlike an ``lstat`` followed by a path, there is no window in which the + directory can be replaced. ``_resolved_directory`` still tolerates a + symlinked reuse ROOT — it hands this function the already-resolved path — + but every component below it is pinned. + """ + + _require_openat() + if create: + try: + os.mkdir(path, 0o700, dir_fd=dir_fd) + except FileExistsError: + # Already there is the ordinary case — a second artifact from the + # same row. What it already IS still has to be proven, and the + # O_DIRECTORY|O_NOFOLLOW open below is what proves it, so there is + # nothing to do here. + pass + except OSError as exc: + raise SandboxError(f"{label} cannot be created: {path}: {exc}") from exc + try: + descriptor = os.open( + path, + os.O_RDONLY | getattr(os, "O_DIRECTORY", 0) | getattr(os, "O_NOFOLLOW", 0), + dir_fd=dir_fd, + ) + except FileNotFoundError as exc: + # Absent is a different fact from present-but-not-a-real-directory, and + # the caller falls through to a paid cell on either. + raise SandboxError(f"{label} is missing: {path}") from exc + except OSError as exc: + raise SandboxError(f"{label} must be a real directory: {path}: {exc}") from exc + try: + # O_DIRECTORY is the check on Linux; the fstat covers a platform whose + # os module does not define it, where the flag degrades to 0. + if not stat.S_ISDIR(os.fstat(descriptor).st_mode): + raise SandboxError(f"{label} must be a real directory: {path}") + yield descriptor + finally: + os.close(descriptor) + + +@contextmanager +def _open_regular(name: str, *, dir_fd: int, label: str) -> Iterator[int]: + """Open a regular non-symlink file under a pinned directory, and hold it. + + Checking a name and then re-opening it is a race the reuse directory is + exposed to: it is written by a previous sweep and read by this one, so a + concurrent writer can replace a validated file with a symlink in between. + Resolving against ``dir_fd`` removes the directory half, ``O_NOFOLLOW`` + refuses the leaf link, and the fstat comparison proves the open descriptor + is the inode that was checked — the same guarantee + evolution._bounded_regular_bytes makes for evidence files. + """ + + _require_openat() + try: + before = os.lstat(name, dir_fd=dir_fd) + except OSError as exc: + raise SandboxError(f"{label} is missing: {name}: {exc}") from exc + if stat.S_ISLNK(before.st_mode) or not stat.S_ISREG(before.st_mode): + raise SandboxError(f"{label} must be a regular non-symlink file: {name}") + try: + descriptor = os.open(name, os.O_RDONLY | getattr(os, "O_NOFOLLOW", 0), dir_fd=dir_fd) + except OSError as exc: + raise SandboxError(f"{label} is unreadable: {name}: {exc}") from exc + try: + opened = os.fstat(descriptor) + if not stat.S_ISREG(opened.st_mode) or (opened.st_dev, opened.st_ino) != (before.st_dev, before.st_ino): + raise SandboxError(f"{label} changed while opening: {name}") + yield descriptor + finally: + os.close(descriptor) + + +def _copy_owner_only(source: int, name: str, *, dir_fd: int, max_bytes: int | None = None) -> tuple[str, int]: + """Copy one open file into the pinned directory; return what was written. + + The digest is taken from the same buffers that are written, so it describes + the copy rather than a state the source was in at some earlier read. + + ``max_bytes`` bounds the copy itself. The source is a prior sweep directory + this module already treats as concurrently writable, so a transcript + appended to after its metadata was recorded would otherwise be streamed to + EOF and only then compared against its declared size - filling the + destination, or never reaching EOF at all, long before the drift check could + reject it. Stopping one byte past the ceiling keeps that comparison + meaningful while bounding the work. + """ + + # O_CREAT|O_EXCL is the existence check, and unlike a stat beforehand it is + # atomic: a file appearing between check and open cannot slip through. + try: + descriptor = os.open( + name, + os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, "O_NOFOLLOW", 0), + 0o600, + dir_fd=dir_fd, + ) + except FileExistsError as exc: + raise SandboxError(f"reuse destination already exists: {name}") from exc + try: + os.fchmod(descriptor, 0o600) + os.lseek(source, 0, os.SEEK_SET) + digest = hashlib.sha256() + written = 0 + limit = None if max_bytes is None else max_bytes + 1 + while True: + want = COPY_CHUNK_BYTES if limit is None else min(COPY_CHUNK_BYTES, limit - written) + if want <= 0: + break + chunk = os.read(source, want) + if not chunk: + break + digest.update(chunk) + written += len(chunk) + _write_all(descriptor, chunk) + os.fsync(descriptor) + return digest.hexdigest(), written + finally: + os.close(descriptor) diff --git a/eval/workflow_bench/evolve.py b/eval/workflow_bench/evolve.py index 71a2299cb..3c17325be 100644 --- a/eval/workflow_bench/evolve.py +++ b/eval/workflow_bench/evolve.py @@ -48,6 +48,7 @@ import yaml from . import runner from . import runner_sessions +from .comparator_reuse import current_runtime_digest from .model_gateway import ( ANTHROPIC_API_KEY_ENV, attach_openai_gateway, @@ -698,8 +699,17 @@ def run_proposer( bwrap_bin: Path, sandbox_backend: str = "bwrap", progress_label: str | None = None, + started_monotonic: float | None = None, ) -> dict[str, Any]: - """Run one proposer in confinement and copy only validated outputs out.""" + """Run one proposer in confinement and copy only validated outputs out. + + ``started_monotonic`` is the sweep clock, not a precomputed budget. The + per-session ``--timeout`` is sized for a whole generation, so a proposer + started with only the sweep minimum left would otherwise run far past the + instance window; the clock is passed rather than the leftover because the + clone, the sanitize pass and the sandbox setup below all happen before the + session starts, and a number sampled by the caller is already stale by then. + """ with tempfile.TemporaryDirectory(prefix="wfevolve-") as tmp: clone = runner.make_worktree(REPO_ROOT, "HEAD", Path(tmp)) @@ -732,11 +742,45 @@ def run_proposer( host_text = getattr(sandbox, "host_text", lambda value: value) environment_builder = getattr(sandbox, "environment", build_sandbox_environment) backend = getattr(sandbox, "backend", "bwrap") + # Sampled here, after the setup above: this is the last + # moment before the session starts, so it is the only reading + # the session's own timeout can honestly be clamped to. + remaining_seconds = ( + None + if started_monotonic is None + else remaining_runtime_seconds( + max_runtime_seconds=args.max_runtime_seconds, + started_monotonic=started_monotonic, + ) + ) + # An exhausted cap must stop the run, not buy one more second. + # remaining_runtime_seconds floors at 0, and max(1, ...) turned + # that 0 into a one-second paid session: the admission check + # happens before cloning, sanitizing and sandbox setup, so those + # unbounded steps can spend the rest of the window and leave + # nothing for the upload reserve this cap exists to protect. + if remaining_seconds is not None and remaining_seconds < 1: + # The caller stops the run on a not-ok record, which is the + # right outcome: an exhausted cap should end the generation, + # not start a session it cannot afford to finish. + return { + "ok": False, + "error_kind": "runtime-cap-exhausted", + "error_detail": ( + "the wall-clock cap elapsed during proposer setup " + "(clone, sanitize, sandbox), before the session started" + ), + "duration_s": 0.0, + "num_turns": 0, + "cost_usd": None, + } record = runner.run_claude( host_text(prompt), clone, claude_bin=sandbox.claude_bin, - timeout=args.timeout, + timeout=( + args.timeout if remaining_seconds is None else min(args.timeout, remaining_seconds) + ), model=args.proposer_model, effort=args.effort, env=model_session_environment( @@ -823,6 +867,119 @@ def _timeout_arm_key(arm: str) -> str: return CANDIDATE_ARMS.get(arm, arm) +EVENTBRIDGE_INSTANCE_WINDOW_SECONDS = 86_400 +EVENTBRIDGE_STOP_RESERVE_SECONDS = 5_400 +MIN_INSTANCE_SWEEP_SECONDS = 600 + + +def instance_window_budget_seconds( + uptime_seconds: float, + *, + window_seconds: int = EVENTBRIDGE_INSTANCE_WINDOW_SECONDS, + reserve_seconds: int = EVENTBRIDGE_STOP_RESERVE_SECONDS, + min_seconds: int = MIN_INSTANCE_SWEEP_SECONDS, +) -> int: + """Seconds a sweep may run before an EventBridge 24h instance stop. + + The dedicated evolution box is started ~15 minutes before the Saturday + cron and stopped 24h later. A ``workflow_dispatch`` that lands on an + already-running box inherits the leftover uptime, not a fresh day. + Run 33962002890 dispatched Friday 10:57 UTC and was still on its last + review cell when the Saturday 03:00 stop cancelled the runner — 51 + finished sessions never uploaded because a cancelled job skips even + ``if: always()``. Capping the in-process sweep so it *fails* (instead + of vanishing) leaves the reserve for the upload step. + """ + + if window_seconds < 1 or reserve_seconds < 0 or min_seconds < 1: + raise ValueError("instance window and minimum must be positive; reserve must be non-negative") + if not math.isfinite(uptime_seconds) or uptime_seconds < 0: + raise ValueError("uptime must be a finite non-negative number") + leftover = int(window_seconds - uptime_seconds - reserve_seconds) + if leftover < min_seconds: + raise ValueError( + f"instance window has only {leftover}s left after a {reserve_seconds}s " + f"upload reserve (uptime {uptime_seconds:.0f}s of {window_seconds}s); " + f"need at least {min_seconds}s" + ) + return leftover + + +def _instance_uptime_or_none() -> float | None: + """The uptime read main() takes before it knows whether it needs it. + + Deferring the read until after argument parsing would put the parse back + inside the interval the cap is supposed to cover, so it happens first and + an unreadable /proc/uptime is only an error if the flag turns out to be set. + """ + + try: + return read_instance_uptime_seconds() + except ValueError: + return None + + +def read_instance_uptime_seconds(uptime_path: Path = Path("/proc/uptime")) -> float: + """Host uptime, the clock the EventBridge stop is scheduled against.""" + + try: + return float(uptime_path.read_text().split()[0]) + except (OSError, IndexError, ValueError) as exc: + raise ValueError(f"cannot read instance uptime from {uptime_path}: {exc}") from exc + + +def instance_window_budget_from_uptime( + uptime_seconds: float, + *, + window_seconds: int | None = None, + reserve_seconds: int | None = None, +) -> int: + """Apply the EventBridge window env overrides to an already-read uptime. + + Separate from the read so ``main`` can take the uptime in the same breath + as its own clock: the budget and the clock it is measured against have to + describe one instant, or the interval between them is spent by nobody and + charged to the sweep. + """ + + window = ( + window_seconds + if window_seconds is not None + else int(os.environ.get("EVENTBRIDGE_INSTANCE_WINDOW_SECONDS", str(EVENTBRIDGE_INSTANCE_WINDOW_SECONDS))) + ) + reserve = ( + reserve_seconds + if reserve_seconds is not None + else int(os.environ.get("EVENTBRIDGE_STOP_RESERVE_SECONDS", str(EVENTBRIDGE_STOP_RESERVE_SECONDS))) + ) + return instance_window_budget_seconds(uptime_seconds, window_seconds=window, reserve_seconds=reserve) + + + + +def remaining_runtime_seconds(*, max_runtime_seconds: int | None, started_monotonic: float) -> int | None: + """Seconds left in an optional wall-clock cap, or None when uncapped.""" + + if max_runtime_seconds is None: + return None + if max_runtime_seconds < 1: + raise ValueError("max runtime must be positive") + leftover = max_runtime_seconds - (time.monotonic() - started_monotonic) + return max(0, int(leftover)) + + +def capped_timeout_seconds(requested: int, remaining: int | None) -> int: + """Clamp one managed-process timeout to the leftover instance window.""" + + if requested < 1: + raise ValueError("requested timeout must be positive") + if remaining is None: + return requested + if remaining < 1: + raise ValueError("no time remains in the instance window") + return min(requested, remaining) + + def generation_timeout_seconds( *, task_count: int, @@ -877,6 +1034,7 @@ def runner_argv( task_bindings: list[dict[str, Any]], target_base_digests: dict[str, str], proposer_model: str | None = None, + reuse_results: Path | None = None, ) -> list[str]: incumbent_arms = resolve_incumbent_arms(overlay_dir, args.arms) paired_arms = executed_benchmark_arms(incumbent_arms) @@ -927,6 +1085,8 @@ def runner_argv( argv += ["--ce-plugin-dir", str(args.ce_plugin_dir), "--ce-plugin-version", args.ce_plugin_version] if args.unsafe_no_bwrap: argv.append("--unsafe-no-bwrap") + if reuse_results is not None: + argv += ["--reuse-results", str(reuse_results)] return argv @@ -944,6 +1104,12 @@ def runner_environment(args: argparse.Namespace) -> dict[str, str]: # actually show progress rather than a burst at the end. "PYTHONUNBUFFERED": "1", } + # process_control replaces the child environment wholesale, so a digest the + # workflow exported reaches the runner only if it is forwarded here. Without + # this the runner stamps no runtime_digest and the reuse lock never engages. + runtime_digest = current_runtime_digest() + if runtime_digest: + env["RUNTIME_DIGEST"] = runtime_digest if args.auth_token: env[ANTHROPIC_API_KEY_ENV] = args.auth_token return env @@ -1199,6 +1365,13 @@ def _require_finite_metric(value: Any, name: str, *, nullable: bool = False, max raise ValueError(f"promotion has invalid {name}") +def _positive_int(value: str) -> int: + parsed = int(value) + if parsed < 1: + raise argparse.ArgumentTypeError(f"{value} is not a positive integer") + return parsed + + def build_parser() -> argparse.ArgumentParser: parser = argparse.ArgumentParser(description=__doc__) parser.add_argument("--tasks", required=True, type=Path) @@ -1237,7 +1410,8 @@ def build_parser() -> argparse.ArgumentParser: "--seed-results", type=Path, default=None, - help="prior wfbench results dir used as generation-0 proposer evidence", + help="prior wfbench results dir used as generation-0 proposer evidence " + "and as --reuse-results for unchanged incumbent/CE cells", ) parser.add_argument( "--initial-overlay", @@ -1271,6 +1445,19 @@ def build_parser() -> argparse.ArgumentParser: default=runner_sessions.SESSION_TIMEOUT_SECONDS, help="per session, seconds", ) + parser.add_argument( + "--max-runtime-seconds", + type=_positive_int, + default=None, + help="wall-clock cap for the whole evolve process (CI derives this from " + "instance uptime so the sweep exits before EventBridge stops the box)", + ) + parser.add_argument( + "--max-runtime-from-instance-window", + action="store_true", + help="derive --max-runtime-seconds from /proc/uptime at startup, so the " + "budget and the clock it is measured against describe one instant", + ) parser.add_argument("--base-url", default=None) parser.add_argument( "--anthropic-api-key", @@ -1308,8 +1495,26 @@ def build_parser() -> argparse.ArgumentParser: def main() -> int: + # These two lines are the cap, and they are adjacent on purpose: the clock + # the sweep is measured against, and the uptime the budget is derived from. + # run-evolution.sh used to compute the budget in its own `uv run python -c` + # and pass a number, so the script's remaining work and this interpreter's + # startup were spent by nobody and charged to the sweep — out of the upload + # reserve the cap exists to protect. Nothing can be spent between them now. + started_monotonic = time.monotonic() + instance_uptime = _instance_uptime_or_none() parser = build_parser() args = parser.parse_args() + if args.max_runtime_from_instance_window: + if args.max_runtime_seconds is not None: + parser.error("--max-runtime-from-instance-window and --max-runtime-seconds are mutually exclusive") + if instance_uptime is None: + parser.error("--max-runtime-from-instance-window needs a readable /proc/uptime") + try: + args.max_runtime_seconds = instance_window_budget_from_uptime(instance_uptime) + except ValueError as exc: + parser.error(str(exc)) + print(f"capping the sweep to {args.max_runtime_seconds}s so the instance-window reserve can upload evidence") if args.generations < 1: parser.error("--generations must be positive") if args.runs < 1 or args.timeout < 1: @@ -1379,6 +1584,7 @@ def main() -> int: try: return _run_generations( args, + started_monotonic=started_monotonic, selected_task_rows=selected_task_rows, skipped_expensive=skipped_expensive, selected_tasks=selected_tasks, @@ -1394,6 +1600,7 @@ def main() -> int: def _run_generations( args: argparse.Namespace, *, + started_monotonic: float, selected_task_rows: list[dict[str, Any]], skipped_expensive: list[str], selected_tasks: list[dict[str, Any]], @@ -1465,6 +1672,19 @@ def _run_generations( incumbent_arms=requested_arms, prior_proposal=staged_prior_included, ) + # Check the window before the paid session, not after it. A + # generation that cannot fit its sweep should not buy a proposal + # first and discover the deadline on the way out. + before_proposer = remaining_runtime_seconds( + max_runtime_seconds=args.max_runtime_seconds, + started_monotonic=started_monotonic, + ) + if before_proposer is not None and before_proposer < MIN_INSTANCE_SWEEP_SECONDS: + print( + f"[gen {generation}] stopping with {before_proposer}s left before the " + f"instance window ends; not starting a proposer session" + ) + return 1 print(f"[gen {generation}] proposing…") record = run_proposer( prompt, @@ -1475,6 +1695,11 @@ def _run_generations( bwrap_bin=bwrap_bin, sandbox_backend=sandbox_backend, progress_label=f"gen {generation} proposer", + # The clock, not the reading taken above: run_proposer clones, + # sanitizes and builds a sandbox before the session starts, so + # before_proposer is stale by then. It still decides whether to + # start at all — it just cannot decide how long to allow. + started_monotonic=started_monotonic, ) # Redact any API token echoed into the session record (e.g. an # error_detail stderr_tail) before it enters the uploaded artifact. @@ -1513,6 +1738,16 @@ def _run_generations( print(f"[gen {generation}] promotion targets contain uncommitted or drifted bytes") return 1 print(f"[gen {generation}] benchmarking candidate…") + leftover = remaining_runtime_seconds( + max_runtime_seconds=args.max_runtime_seconds, + started_monotonic=started_monotonic, + ) + if leftover is not None and leftover < MIN_INSTANCE_SWEEP_SECONDS: + print( + f"[gen {generation}] stopping with {leftover}s left before the " + f"instance window ends; partial evidence is in {out_root}/" + ) + return 1 benchmark_argv = runner_argv( args, bench_dir, @@ -1520,20 +1755,30 @@ def _run_generations( task_bindings=selected_tasks, target_base_digests=target_base_digests, proposer_model=generation_proposer_model, + reuse_results=evidence_dir, ) benchmark_command = ( benchmark_argv if sandbox_backend == "host-unsafe" else pid_namespace_command(benchmark_argv, bwrap_bin=bwrap_bin) ) - bench = run_managed( - benchmark_command, - timeout=generation_timeout_seconds( + sweep_timeout = capped_timeout_seconds( + generation_timeout_seconds( task_count=len(selected_task_rows), runs=args.runs, session_timeout=args.timeout, incumbent_arms=incumbent_arms, ), + leftover, + ) + if leftover is not None: + print( + f"[gen {generation}] sweep timeout {sweep_timeout}s " + f"(instance window leftover {leftover}s)" + ) + bench = run_managed( + benchmark_command, + timeout=sweep_timeout, env=runner_environment(args), require_pid_namespace=sandbox_backend == "bwrap", # The sweep is the multi-hour phase; without this its per-run @@ -1545,7 +1790,13 @@ def _run_generations( # The sweep runs with GITNEXUS_BENCH_ANTHROPIC_API_KEY in its environment, # so its detail/stderr tail is a token-bearing sink like any other. detail = redacted_failure(args, str(bench.detail or bench.stderr_tail[-1000:])) - print(f"[gen {generation}] benchmark run failed ({bench.state}, exit {bench.returncode}): {detail}") + if leftover is not None and bench.state == "timeout": + print( + f"[gen {generation}] benchmark hit the instance-window budget " + f"({sweep_timeout}s); partial evidence is in {bench_dir}: {detail}" + ) + else: + print(f"[gen {generation}] benchmark run failed ({bench.state}, exit {bench.returncode}): {detail}") return 1 promotion = json.loads((bench_dir / "promotion.json").read_text()) for line in summarize_gate(promotion): diff --git a/eval/workflow_bench/litellm_usage_callback.py b/eval/workflow_bench/litellm_usage_callback.py new file mode 100644 index 000000000..8277d0102 --- /dev/null +++ b/eval/workflow_bench/litellm_usage_callback.py @@ -0,0 +1,135 @@ +"""Append each upstream request's usage exactly as the provider reported it. + +This runs INSIDE the LiteLLM proxy, on the far side of the translation that +turns an OpenAI response into the Anthropic shape Claude Code expects. That is +the only point that still knows which provider served the request, what model +actually answered, and what the native usage object said before its fields were +renamed into someone else's semantics. + +Deliberately self-contained: the proxy loads this file by path from the config +directory, so it cannot assume ``workflow_bench`` is importable. Normalization +lives in workflow_bench.provider_usage and runs offline over what this writes - +the native object is the evidence, and deriving from it here would mean the +derivation could not be revisited without re-running a paid sweep. + +Never raises. A cell that fails still spent money upstream, and losing the +accounting because the log write failed would be the worse outcome. +""" + +from __future__ import annotations + +import json +import os +import threading +from typing import Any + +from litellm.integrations.custom_logger import CustomLogger + +# Literals, not imports. LiteLLM loads this file BY PATH from the config +# directory via spec_from_file_location, so it has no parent package and the +# directory is not on sys.path - a relative or sibling import raises +# ImportError and the proxy refuses to start. workflow_bench.provider_usage +# holds the canonical copies and a test asserts these agree with them, which +# catches drift without coupling at import time. +USAGE_LOG_ENV_VAR = "GITNEXUS_BENCH_PROVIDER_USAGE" +SWEEP_ID_ENV_VAR = "GITNEXUS_BENCH_SWEEP_ID" + + +def canonical_provider(label, call_type): # noqa: ANN001, ANN201 + """Adapter key for the usage shape, or None when it cannot be resolved. + + Mirrors workflow_bench.provider_usage.canonical_provider; see the note + above for why this is a copy rather than an import. + """ + + if label == "openai": + return "litellm-normalized" + if label == "anthropic": + return "anthropic" + return None + +SCHEMA_VERSION = 1 +_LOCK = threading.Lock() + + +def _plain(value: Any) -> Any: + """Provider usage arrives as pydantic models; keep the shape, drop the class.""" + + for attr in ("model_dump", "dict"): + method = getattr(value, attr, None) + if callable(method): + try: + return method() + except Exception: + pass + if isinstance(value, dict): + return value + return None + + +class ProviderUsageLogger(CustomLogger): + async def async_log_success_event(self, kwargs, response_obj, start_time, end_time) -> None: # noqa: ANN001 + self._append("success", kwargs, response_obj, start_time, end_time) + + async def async_log_failure_event(self, kwargs, response_obj, start_time, end_time) -> None: # noqa: ANN001 + # Failed requests are billed too, and a sweep that only accounts for + # successes understates what it spent. + self._append("failure", kwargs, response_obj, start_time, end_time) + + def log_success_event(self, kwargs, response_obj, start_time, end_time) -> None: # noqa: ANN001 + self._append("success", kwargs, response_obj, start_time, end_time) + + def log_failure_event(self, kwargs, response_obj, start_time, end_time) -> None: # noqa: ANN001 + # The synchronous counterpart. Overriding only the success hook here + # recorded successes and let failures fall through to the base class, + # which accounts for nothing - and a failed request is still billed, so + # a sweep missing them understates what it spent. + self._append("failure", kwargs, response_obj, start_time, end_time) + + def _append(self, status, kwargs, response_obj, start_time, end_time) -> None: # noqa: ANN001 + path = os.environ.get(USAGE_LOG_ENV_VAR) + if not path: + return + try: + params = kwargs.get("litellm_params") or {} + call_type = kwargs.get("call_type") + provider_label = kwargs.get("custom_llm_provider") or params.get("custom_llm_provider") + metadata = params.get("metadata") or {} + event = { + "schema_version": SCHEMA_VERSION, + "status": status, + # Identity. The REQUESTED model is the caller's role name and the + # ACTUAL model is what answered; pricing must follow the second, + # because several roles map onto one upstream model here. + "requested_model": kwargs.get("model"), + "actual_model": getattr(response_obj, "model", None), + # Two fields, because they answer different questions. The raw + # label is what LiteLLM said; "provider" is the adapter key for + # the object actually in hand, which is always LiteLLM's own + # normalised shape here. An unrecognised label stays None so + # normalize_usage refuses rather than guessing token semantics. + "provider_label": provider_label, + "provider": canonical_provider(provider_label, call_type), + "response_id": getattr(response_obj, "id", None), + "call_type": call_type, + "sweep_id": os.environ.get(SWEEP_ID_ENV_VAR), + # The per-request half of identity, and the only thing that can + # attribute a request to a cell: one proxy serves the whole + # sweep, so anything read from the environment is the same for + # every event. Recorded even when absent, because knowing the + # attribution is unavailable is itself a fact about the run. + "session_id": metadata.get("litellm_session_id") or metadata.get("session_id"), + "started_at": str(start_time), + "completed_at": str(end_time), + # Verbatim. Not flattened, not renamed, not summed. + "native_usage": _plain(getattr(response_obj, "usage", None)), + } + line = json.dumps(event, default=str) + "\n" + with _LOCK, open(path, "a", encoding="utf-8") as handle: + handle.write(line) + except Exception: + # Accounting is evidence, not control flow: never take the sweep down. + return + + +handler = ProviderUsageLogger() diff --git a/eval/workflow_bench/measure_evolution_cost.py b/eval/workflow_bench/measure_evolution_cost.py new file mode 100644 index 000000000..4d18de746 --- /dev/null +++ b/eval/workflow_bench/measure_evolution_cost.py @@ -0,0 +1,311 @@ +#!/usr/bin/env python3 +"""Cheap cost model for the skill-evolution review generation. + +This is the ce-optimize measurement harness. It does not start Claude and it +does not replay a run. It reads the review corpus, the evolve defaults and the +workflow's workers default, then schedules the measured cell durations in +``session_durations.json`` the way ``sweep_task_cells`` schedules real cells. + +Everything priced here is measured. Cell durations and the proposer session +come from a real artifact, and the work outside the agent sessions comes from +that run's own step wall minus the time its sessions and proposer account for. + +Weekly assumes a matching seed, so every reusable comparator cell is skipped +and only the candidate arm is paid. Cold assumes an empty seed. +""" + +from __future__ import annotations + +import json +import math +import re +import statistics as st +import subprocess +import sys +from pathlib import Path + +REPO_ROOT = Path(__file__).resolve().parents[2] +EVAL_ROOT = REPO_ROOT / "eval" +REVIEW_TASKS = EVAL_ROOT / "workflow_bench" / "tasks.review.scenarios.yaml" +EVOLVE_PY = EVAL_ROOT / "workflow_bench" / "evolve.py" +RUNNER_PY = EVAL_ROOT / "workflow_bench" / "runner.py" +ARTIFACTS_PY = EVAL_ROOT / "workflow_bench" / "runner_artifacts.py" +REUSE_PY = EVAL_ROOT / "workflow_bench" / "comparator_reuse.py" +WORKFLOW = REPO_ROOT / ".github" / "workflows" / "gitnexus-skill-evolution.yml" + +MEASURED = json.loads( + (Path(__file__).resolve().parent / "session_durations.json").read_text(encoding="utf-8") +) +# Per arm, because the arms are not interchangeable and the weekly lane pays +# only the candidate one. Cells are submitted run-major and arm-minor +# (runner.py ``planned``), so at workers=3 every wave holds one cell of each +# arm and the slowest arm sets the wave. +DURATIONS_BY_ARM: dict[str, tuple[float, ...]] = { + arm: tuple(values) for arm, values in MEASURED["cell_duration_s_by_arm"].items() +} +PROPOSER_SECONDS: float = MEASURED["proposer_duration_s"] +_RESIDUAL = MEASURED["residual"] +# Clone, graph build, sandbox, teardown: the sweep's own time, taken as that +# run's step wall minus what its sessions and proposer account for. Charged +# SERIALLY, outside the pool, and charged PER SHA rather than per cell. The +# residual mixes per-cell work with per-SHA graph setup and the artifact cannot +# separate them; per-SHA is the direction that refuses to credit a run for +# shrinking work it still performs, which per-cell did - a weekly generation +# pays one arm instead of three but builds exactly the same graphs. See +# session_durations.json residual._split_assumption. +SHA_OVERHEAD_SECONDS: float = _RESIDUAL["sha_overhead_s"] + +# runner.py CANDIDATE_ARMS derives the candidate arm from its incumbent, and +# only an incumbent row can be reused from a prior generation. +CANDIDATE_ARM = "candidate_review" +REVIEW_ARMS = ("ce_review", "review", CANDIDATE_ARM) + +SUITE_FILES = ( + "tests/test_measure_evolution_cost.py", + "tests/test_comparator_reuse.py", + "tests/test_evolve.py", + "tests/test_sanitized_graph.py", + "tests/test_workflow_bench.py", + "tests/test_workflow_bench_sessions.py", + "tests/test_session_progress.py", +) + + +def _read(path: Path) -> str: + return path.read_text(encoding="utf-8") + + +def review_tasks(text: str) -> list[dict[str, str]]: + tasks: list[dict[str, str]] = [] + current: dict[str, str] | None = None + for raw in text.splitlines(): + line = raw.strip() + if line.startswith("id:"): + if current is not None: + tasks.append(current) + current = {"id": line.split(":", 1)[1].strip()} + elif line.startswith("ref:") and current is not None: + current["ref"] = line.split(":", 1)[1].strip() + if current is not None: + tasks.append(current) + return tasks + + +def evolve_default(name: str, text: str) -> int: + match = re.search(rf'add_argument\("--{re.escape(name)}".*?default=(\d+)', text, flags=re.S) + if match is None: + raise ValueError(f"evolve.py is missing --{name} default") + return int(match.group(1)) + + +def workflow_dispatch_workers(text: str) -> int: + match = re.search(r"^\s+workers:\n(?:.*\n)*?^\s+default: '(\d+)'", text, flags=re.M) + if match is None: + raise ValueError("workflow_dispatch workers default is missing") + return int(match.group(1)) + + +def feature_enabled() -> tuple[int, int]: + evolve = _read(EVOLVE_PY) + runner = _read(RUNNER_PY) + artifacts = _read(ARTIFACTS_PY) + reuse = int( + REUSE_PY.is_file() + and "--reuse-results" in evolve + and "select_reusable_comparator_rows" in runner + and "CANDIDATE" in _read(REUSE_PY) + ) + templates = int("def copy_isolated_tree" in artifacts and "clone_templates" in runner) + return reuse, templates + + +def graph_pipeline_enabled(runner_text: str) -> int: + """True when the runner prefetches the next SHA during paid sessions.""" + + return int("prefetch_next_graph" in runner_text or "GraphPrefetch" in runner_text) + + +def fed_pool_enabled(runner_text: str) -> int: + """True when the sweep feeds a live pool instead of waiting on waves.""" + + return int("def _run_fed_pool" in runner_text) + + +def paid_arms(weekly: bool, reuse_enabled: bool) -> tuple[str, ...]: + """Arms a generation actually pays for.""" + + if weekly and reuse_enabled: + return (CANDIDATE_ARM,) + return REVIEW_ARMS + + +def task_cells(runs: int, arms: tuple[str, ...], offset: int) -> list[float]: + """One task's cell durations in submission order: run-major, arm-minor. + + Each arm draws from its own measured sample, cycled from ``offset`` so the + caller can average over every alignment instead of trusting one. + """ + + cells: list[float] = [] + for run_idx in range(runs): + for arm in arms: + sample = DURATIONS_BY_ARM[arm] + cells.append(sample[(offset + run_idx) % len(sample)]) + return cells + + +def wave_makespan(durations: list[float], workers: int) -> float: + """Today's scheduler: fixed waves of ``workers``, with a barrier between.""" + + return sum( + max(durations[start : start + workers]) for start in range(0, len(durations), workers) + ) + + +def fed_makespan(durations: list[float], workers: int) -> float: + """Continuously fed pool: a free worker takes the next cell immediately.""" + + busy_until = [0.0] * workers + for duration in durations: + first = min(range(workers), key=busy_until.__getitem__) + busy_until[first] += duration + return max(busy_until) + + +def expected_task_seconds( + runs: int, arms: tuple[str, ...], workers: int, *, fed_pool: bool +) -> float: + """Mean makespan of one task over every alignment of the measured samples. + + One fixed alignment would let an accident of the source run - its slowest + cells happen to come first - decide the answer. Averaging keeps the real + multiset and the real ordering effects without that artifact, and stays + deterministic. + """ + + if runs < 1 or not arms: + return 0.0 + makespan = fed_makespan if fed_pool else wave_makespan + # lcm, not max: with samples of 13 and 14, max would wrap the shorter one + # and count its first entry twice. + alignments = math.lcm(*(len(DURATIONS_BY_ARM[arm]) for arm in arms)) + return ( + sum(makespan(task_cells(runs, arms, offset), workers) for offset in range(alignments)) + / alignments + ) + + +def generation_seconds( + *, + task_count: int, + runs: int, + arms: tuple[str, ...], + workers: int, + fed_pool: bool, + unique_shas: int, +) -> int: + """Whole generation: proposer, then the tasks back to back, plus overhead. + + Prices a HEALTHY sweep. A run whose cells return unusable evidence does not + reach this wall at all: the outage breaker aborts after + ``DEFAULT_OUTAGE_STREAK`` consecutive systemic failures, which for the + sample's own error sequence is cell 5 of 41. + + Sweep overhead is charged per SHA, so it does not shrink with the arm count. + Weekly pays one arm instead of three but builds the same graphs, and billing + that per cell credited it for a saving the real run never makes. + """ + + return round( + PROPOSER_SECONDS + + task_count * expected_task_seconds(runs, arms, workers, fed_pool=fed_pool) + + unique_shas * SHA_OVERHEAD_SECONDS + ) + + +def _pytest_python() -> list[str]: + venv_python = EVAL_ROOT / ".venv" / "bin" / "python" + if venv_python.is_file(): + return [str(venv_python)] + if (EVAL_ROOT / "uv.lock").is_file(): + return ["uv", "run", "--locked", "--extra", "dev", "python"] + return [sys.executable] + + +def suite_passed() -> int: + files = [name for name in SUITE_FILES if (EVAL_ROOT / name).is_file()] + if not files: + return 0 + cmd = [*_pytest_python(), "-m", "pytest", *files, "-q", "--tb=no", "--no-header"] + try: + completed = subprocess.run( + cmd, cwd=EVAL_ROOT, check=False, capture_output=True, text=True, timeout=240 + ) + except (OSError, subprocess.TimeoutExpired): + return 0 + return int(completed.returncode == 0) + + +def main() -> int: + tasks = review_tasks(_read(REVIEW_TASKS)) + evolve = _read(EVOLVE_PY) + runner = _read(RUNNER_PY) + runs = evolve_default("runs", evolve) + workers = workflow_dispatch_workers(_read(WORKFLOW)) + reuse_enabled, clone_templates_enabled = feature_enabled() + fed_pool = fed_pool_enabled(runner) + + # Both walls build the same graphs; the arm count does not change that. + unique_shas = len({t.get("ref", "") for t in tasks if t.get("ref")}) + payload: dict[str, object] = {} + for label, weekly in (("weekly", True), ("cold", False)): + arms = paid_arms(weekly, bool(reuse_enabled)) + payload[f"estimated_{label}_wall_seconds"] = generation_seconds( + task_count=len(tasks), + runs=runs, + arms=arms, + workers=workers, + fed_pool=bool(fed_pool), + unique_shas=unique_shas, + ) + payload[f"paid_{label}_cells"] = len(tasks) * runs * len(arms) + # What the wave barrier costs: the same cells, continuously fed. + payload[f"fed_pool_{label}_wall_seconds"] = generation_seconds( + task_count=len(tasks), + runs=runs, + arms=arms, + workers=workers, + fed_pool=True, + unique_shas=unique_shas, + ) + + all_durations = [d for sample in DURATIONS_BY_ARM.values() for d in sample] + payload.update( + { + "suite_passed": suite_passed(), + "promotion_min_runs": evolve_default("promotion-min-runs", evolve), + "review_task_count": len(tasks), + "candidate_cells": len(tasks) * runs, + "workers": workers, + "unique_task_shas": len({t.get("ref", "") for t in tasks if t.get("ref")}), + "reuse_enabled": reuse_enabled, + "clone_templates_enabled": clone_templates_enabled, + "graph_pipeline_enabled": graph_pipeline_enabled(runner), + "fed_pool_enabled": fed_pool, + "measured_cell_count": len(all_durations), + "median_cell_seconds": round(st.median(all_durations)), + "mean_cell_seconds": round(st.mean(all_durations)), + "max_cell_seconds": round(max(all_durations)), + "median_candidate_cell_seconds": round(st.median(DURATIONS_BY_ARM[CANDIDATE_ARM])), + "mean_candidate_cell_seconds": round(st.mean(DURATIONS_BY_ARM[CANDIDATE_ARM])), + "proposer_seconds": round(PROPOSER_SECONDS), + "sha_overhead_seconds": round(SHA_OVERHEAD_SECONDS, 1), + } + ) + json.dump(payload, sys.stdout, sort_keys=True) + sys.stdout.write("\n") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/eval/workflow_bench/mock_provider.py b/eval/workflow_bench/mock_provider.py new file mode 100644 index 000000000..dcc328251 --- /dev/null +++ b/eval/workflow_bench/mock_provider.py @@ -0,0 +1,284 @@ +"""A scriptable stand-in for Anthropic and OpenAI, for running the harness offline. + +Every defect this benchmark shipped in the last round was invisible to its own +tests for the same reason: the tests exercised a layer BELOW where the code +runs. The usage log was never written because the proxy is a subprocess with a +constructed environment. The callback could not be imported because LiteLLM +loads it by path. Failures went unrecorded because only the async hook was +overridden. Each was caught by CI or review, never by a unit test, because the +unit test called the function directly instead of driving the path that calls +it. + +This closes that gap without spending money. It speaks the two wire protocols +the harness actually depends on, so a run can go through the real sandbox, the +real Claude Code CLI, the real gateway and the real usage callback, and only +the model is fake: + + POST /v1/messages Anthropic Messages, streaming and non-streaming + POST /v1/responses OpenAI Responses, which the gateway translates into + +Point the runner at it with ``--base-url http://127.0.0.1:``, which is +the same supported path the free-model proxy documentation already uses, or +give it to LiteLLM as ``api_base`` to exercise the gateway. + +Scripted, not simulated: replies are supplied by the caller, so a test decides +what the model "says", which tools it asks for, and exactly what usage it +reports. That last part is what makes provider-native accounting testable at +all - real cache hits are not reproducible on demand, but a declared +``cache_read`` of 44_000 is. +""" + +from __future__ import annotations + +import json +import threading +import time +from collections import deque +from dataclasses import dataclass, field +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer +from typing import Any + + +@dataclass +class Reply: + """One scripted model turn. + + ``tools`` drives real tool execution: Claude Code runs what it is asked to + run, so a reply carrying a Write block makes the CLI write that file inside + the sandbox for real. That is how an artifact-producing cell can be + exercised without a model deciding anything. + """ + + text: str = "ok" + tools: list[dict[str, Any]] = field(default_factory=list) + stop_reason: str = "end_turn" + # Anthropic accounting: input_tokens is the UNCACHED remainder and the + # cache fields add to it. Defaults are deliberately non-zero so a test that + # forgets to script usage still cannot mistake silence for a measurement. + input_tokens: int = 11 + output_tokens: int = 7 + # None means the field is OMITTED from the reply, which is not the same as + # reporting 0. A consumer that cannot tell those apart is the bug this + # harness exists to catch, so the mock has to be able to script absence. + cache_read_input_tokens: int | None = 0 + cache_creation_input_tokens: int | None = 0 + status_code: int = 200 + error_body: dict[str, Any] | None = None + + +@dataclass +class Request: + """What the harness actually sent, kept so a test can assert on it.""" + + path: str + headers: dict[str, str] + body: dict[str, Any] + + +class _Handler(BaseHTTPRequestHandler): + provider: MockProvider + + def log_message(self, *_args: Any) -> None: # noqa: A003 - silence the default stderr spam + return + + def do_POST(self) -> None: # noqa: N802 - BaseHTTPRequestHandler's interface + length = int(self.headers.get("Content-Length") or 0) + raw = self.rfile.read(length) if length else b"{}" + try: + body = json.loads(raw or b"{}") + except json.JSONDecodeError: + body = {"_unparsed": raw.decode("utf-8", "replace")} + self.provider.record(Request(self.path, dict(self.headers), body)) + reply = self.provider.next_reply() + + if reply.status_code != 200: + self._send_json(reply.status_code, reply.error_body or {"error": {"message": "scripted failure"}}) + return + if self.path.rstrip("/").endswith("/responses"): + self._send_json(200, _openai_response(reply)) + return + if body.get("stream"): + self._send_anthropic_stream(reply) + return + self._send_json(200, _anthropic_message(reply)) + + def _send_json(self, status: int, payload: dict[str, Any]) -> None: + encoded = json.dumps(payload).encode() + self.send_response(status) + self.send_header("Content-Type", "application/json") + self.send_header("Content-Length", str(len(encoded))) + self.end_headers() + self.wfile.write(encoded) + + def _send_anthropic_stream(self, reply: Reply) -> None: + self.send_response(200) + self.send_header("Content-Type", "text/event-stream") + self.send_header("Cache-Control", "no-cache") + self.end_headers() + for event, data in _anthropic_stream_events(reply): + self.wfile.write(f"event: {event}\ndata: {json.dumps(data)}\n\n".encode()) + self.wfile.flush() + + +def _content_blocks(reply: Reply) -> list[dict[str, Any]]: + blocks: list[dict[str, Any]] = [{"type": "text", "text": reply.text}] + for index, tool in enumerate(reply.tools): + blocks.append( + { + "type": "tool_use", + "id": f"toolu_mock_{index}", + "name": tool["name"], + "input": tool.get("input", {}), + } + ) + return blocks + + +def _anthropic_usage(reply: Reply) -> dict[str, int]: + usage = { + "input_tokens": reply.input_tokens, + "output_tokens": reply.output_tokens, + "cache_read_input_tokens": reply.cache_read_input_tokens, + "cache_creation_input_tokens": reply.cache_creation_input_tokens, + } + return {field: value for field, value in usage.items() if value is not None} + + +def _anthropic_message(reply: Reply) -> dict[str, Any]: + return { + "id": "msg_mock", + "type": "message", + "role": "assistant", + "model": "mock-model", + "content": _content_blocks(reply), + "stop_reason": "tool_use" if reply.tools else reply.stop_reason, + "stop_sequence": None, + "usage": _anthropic_usage(reply), + } + + +def _anthropic_stream_events(reply: Reply) -> list[tuple[str, dict[str, Any]]]: + """The SSE sequence a Messages consumer expects, in order.""" + + message = _anthropic_message(reply) + events: list[tuple[str, dict[str, Any]]] = [ + ("message_start", {"type": "message_start", "message": {**message, "content": [], "usage": _anthropic_usage(reply)}}) + ] + for index, block in enumerate(message["content"]): + if block["type"] == "text": + events.append(("content_block_start", {"type": "content_block_start", "index": index, "content_block": {"type": "text", "text": ""}})) + events.append(("content_block_delta", {"type": "content_block_delta", "index": index, "delta": {"type": "text_delta", "text": block["text"]}})) + else: + events.append(("content_block_start", {"type": "content_block_start", "index": index, "content_block": {"type": "tool_use", "id": block["id"], "name": block["name"], "input": {}}})) + events.append(("content_block_delta", {"type": "content_block_delta", "index": index, "delta": {"type": "input_json_delta", "partial_json": json.dumps(block["input"])}})) + events.append(("content_block_stop", {"type": "content_block_stop", "index": index})) + events.append(("message_delta", {"type": "message_delta", "delta": {"stop_reason": message["stop_reason"], "stop_sequence": None}, "usage": {"output_tokens": reply.output_tokens}})) + events.append(("message_stop", {"type": "message_stop"})) + return events + + +def _openai_response(reply: Reply) -> dict[str, Any]: + """OpenAI Responses shape: input_tokens is the WHOLE, cache fields subsets.""" + + # An omitted cache field contributes nothing to the Responses total; that + # is arithmetic, not a claim the value was measured as zero. + cache_read = reply.cache_read_input_tokens or 0 + cache_write = reply.cache_creation_input_tokens or 0 + total_input = reply.input_tokens + cache_read + cache_write + return { + "id": "resp_mock", + "object": "response", + "created_at": int(time.time()), + "status": "completed", + "model": "mock-model", + "error": None, + "output": [ + { + "id": "msg_mock", + "type": "message", + "role": "assistant", + "status": "completed", + "content": [{"type": "output_text", "text": reply.text, "annotations": []}], + }, + # Tool calls belong here too. Responses is the protocol the gateway + # is configured for BECAUSE it carries tool use, so emitting only + # output_text meant a reply scripted with a Write or Skill crossed + # the gateway with the tool silently dropped - the mock would have + # been wrong about the wire on the one path that matters most. + *( + { + "id": f"fc_mock_{index}", + "type": "function_call", + "status": "completed", + "call_id": f"call_mock_{index}", + "name": tool["name"], + "arguments": json.dumps(tool.get("input", {})), + } + for index, tool in enumerate(reply.tools) + ), + ], + "usage": { + "input_tokens": total_input, + "output_tokens": reply.output_tokens, + "total_tokens": total_input + reply.output_tokens, + # Omitted stays omitted here too. Collapsing None to 0 is right for + # the total above (an unreported field adds nothing) but wrong on + # the wire: _int_or_none reads an absent key as unknown and a + # present 0 as a measured zero, so serializing 0 would claim a + # measurement the reply never made - the same confusion the + # Anthropic path already refuses. + "input_tokens_details": { + **({"cached_tokens": cache_read} if reply.cache_read_input_tokens is not None else {}), + **({"cache_write_tokens": cache_write} if reply.cache_creation_input_tokens is not None else {}), + }, + "output_tokens_details": {"reasoning_tokens": 0}, + }, + } + + +class MockProvider: + """Loopback-only provider stand-in. Use as a context manager.""" + + def __init__(self, replies: list[Reply] | None = None, *, default: Reply | None = None) -> None: + self._replies: deque[Reply] = deque(replies or []) + # A run makes more requests than a test wants to script; the default + # keeps it going rather than failing on the first unscripted turn. + self._default = default or Reply() + self._requests: list[Request] = [] + self._lock = threading.Lock() + self._server: ThreadingHTTPServer | None = None + + def __enter__(self) -> MockProvider: + handler = type("_BoundHandler", (_Handler,), {"provider": self}) + # Loopback only: this answers with no authentication at all. + self._server = ThreadingHTTPServer(("127.0.0.1", 0), handler) + threading.Thread(target=self._server.serve_forever, daemon=True).start() + return self + + def __exit__(self, *_exc: object) -> bool: + if self._server is not None: + self._server.shutdown() + self._server.server_close() + return False + + @property + def port(self) -> int: + assert self._server is not None, "provider is not running" + return self._server.server_port + + @property + def base_url(self) -> str: + return f"http://127.0.0.1:{self.port}" + + def record(self, request: Request) -> None: + with self._lock: + self._requests.append(request) + + def next_reply(self) -> Reply: + with self._lock: + return self._replies.popleft() if self._replies else self._default + + @property + def requests(self) -> list[Request]: + with self._lock: + return list(self._requests) diff --git a/eval/workflow_bench/model_gateway.py b/eval/workflow_bench/model_gateway.py index e44da55ec..e798eb14a 100644 --- a/eval/workflow_bench/model_gateway.py +++ b/eval/workflow_bench/model_gateway.py @@ -31,6 +31,8 @@ from typing import Any import yaml +from .provider_usage import USAGE_ENV_VARS + ANTHROPIC_API_KEY_ENV = "GITNEXUS_BENCH_ANTHROPIC_API_KEY" LEGACY_ANTHROPIC_API_KEY_ENV = "GITNEXUS_BENCH_AUTH_TOKEN" OPENAI_API_KEY_ENV = "GITNEXUS_BENCH_OPENAI_API_KEY" @@ -173,6 +175,9 @@ def resolve_model_access( return ModelAccess(start_proxy=False) +USAGE_CALLBACK_MODULE = "provider_usage_callback" + + def openai_litellm_config(model_names: Sequence[str]) -> dict[str, Any]: seen: list[str] = [] for name in model_names: @@ -196,7 +201,15 @@ def openai_litellm_config(model_names: Sequence[str]) -> dict[str, Any]: } for name in seen ], - "litellm_settings": {"request_timeout": GATEWAY_REQUEST_TIMEOUT_S}, + "litellm_settings": { + "request_timeout": GATEWAY_REQUEST_TIMEOUT_S, + # Captures each upstream request's usage as the provider reported + # it, before translation renames OpenAI's fields into Anthropic's + # shape and loses which arithmetic applies. Resolved by LiteLLM + # relative to the config directory, which is why the module is + # copied next to the config rather than imported from the package. + "callbacks": [f"{USAGE_CALLBACK_MODULE}.handler"], + }, "general_settings": {"master_key": "os.environ/LITELLM_MASTER_KEY"}, } @@ -204,9 +217,26 @@ def openai_litellm_config(model_names: Sequence[str]) -> dict[str, Any]: def write_openai_litellm_config(path: Path, model_names: Sequence[str]) -> Path: path.write_text(yaml.safe_dump(openai_litellm_config(model_names), sort_keys=False)) path.chmod(0o600) + _install_usage_callback(path.parent) return path +def _install_usage_callback(config_dir: Path) -> Path: + """Place the usage logger where LiteLLM resolves callbacks from. + + LiteLLM loads a dotted callback path as a file relative to the config + directory before falling back to a package import, and the proxy runs as + its own process that need not have this package on sys.path. Copying the + one module is what makes the callback resolvable in both cases. + """ + + source = Path(__file__).with_name("litellm_usage_callback.py") + destination = config_dir / f"{USAGE_CALLBACK_MODULE}.py" + destination.write_text(source.read_text()) + destination.chmod(0o600) + return destination + + def _free_loopback_port() -> int: with socket.socket(socket.AF_INET, socket.SOCK_STREAM) as sock: sock.bind(("127.0.0.1", 0)) @@ -302,6 +332,17 @@ class OpenAIGateway(AbstractContextManager["OpenAIGateway"]): "OPENAI_API_KEY": self.openai_api_key, "LITELLM_MASTER_KEY": self.auth_token, } + # The proxy is a separate process and Popen(env=...) REPLACES the + # parent environment rather than extending it, so anything the usage + # callback reads has to be forwarded by name. Without this the callback + # loads, finds no destination, and returns silently on every request - + # the accounting looks configured and records nothing. Forwarded + # individually rather than by inheriting the environment, because the + # allowlist above is the gateway's credential boundary. + for name in USAGE_ENV_VARS: + value = os.environ.get(name) + if value: + env[name] = value if os.name == "nt": # Windows subprocess DLL/socket initialization needs SystemRoot. # Keep the rest of the gateway's credential boundary explicit. diff --git a/eval/workflow_bench/proposer_sandbox.py b/eval/workflow_bench/proposer_sandbox.py index a2dfaa54a..2b0347da1 100644 --- a/eval/workflow_bench/proposer_sandbox.py +++ b/eval/workflow_bench/proposer_sandbox.py @@ -22,6 +22,15 @@ from .process_control import ManagedProcessResult, run_managed MAX_EVIDENCE_FILE_BYTES = 256 * 1024 MAX_BUNDLE_BYTES = 2 * 1024 * 1024 SANDBOX_WORKSPACE = "/workspace" +# The review artifact lives OUTSIDE the workspace, in its own writable +# directory. A writable FILE inside a read-only directory is not writable to +# anything that writes atomically: the Write tool creates +# `.tmp..` beside the target and renames it, so a read-only +# parent fails the temp create with EROFS and the artifact is never written. +# Binding a writable directory outside /workspace lets the rename land while +# the workspace itself stays entirely read-only. +SANDBOX_REVIEW_OUTPUT = "/review-output" +REVIEW_OUTPUT_DIRNAME = "review-output" SANDBOX_HOME = "/home/agent" SANDBOX_TMP = "/tmp" SANDBOX_CLAUDE = "/opt/claude/claude" @@ -118,36 +127,41 @@ REVIEW_RUNTIME_DIRECTORIES = ( ) -def prepare_review_workspace(sandbox: SandboxSession, artifact_name: str) -> Path: - """Prepare disposable mount targets; never truncate a pre-existing entry.""" +def review_output_path(sandbox: SandboxSession, artifact_name: str) -> Path: + """Host path of the review artifact: a private directory, not the clone. + + One source of truth for the location, so the mount, the parse and the + artifact copy cannot drift apart. + """ - clone = _real_directory(sandbox.clone, label="review clone") if PurePosixPath(artifact_name).name != artifact_name or "\\" in artifact_name or artifact_name in ("", ".", ".."): raise SandboxError("review artifact must be a root filename") - output = clone / artifact_name - # No agent runs while this private clone is being prepared. On POSIX the - # directory descriptor additionally binds the exclusive create to its owner. - directory_fd = None - try: - if os.name != "nt": - directory_fd = os.open(clone, os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW) - fd = os.open( - artifact_name if directory_fd is not None else output, - os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, "O_NOFOLLOW", 0), - 0o600, - dir_fd=directory_fd, - ) - try: - if not stat.S_ISREG(os.fstat(fd).st_mode): - raise SandboxError("review artifact must be a regular file") - finally: - os.close(fd) - except FileExistsError as exc: - raise SandboxError("review artifact already exists") from exc - finally: - if directory_fd is not None: - os.close(directory_fd) + return Path(sandbox.private_root) / REVIEW_OUTPUT_DIRNAME / artifact_name + +def prepare_review_workspace(sandbox: SandboxSession, artifact_name: str) -> Path: + """Prepare disposable mount targets; never truncate a pre-existing entry. + + Creates the artifact's own directory and returns the path the agent is + expected to write. The file itself is deliberately NOT pre-created: the + agent writes it atomically (temp file beside the target, then rename), so + the directory is what has to be writable, and an existing empty file would + only be something for the write to trip over. Absence is meaningful — it is + how ``parse_review_output`` tells "never written" from "written badly". + """ + + output = review_output_path(sandbox, artifact_name) + # No agent runs while this private root is being prepared, and the + # exclusive create is what proves the directory is ours rather than + # something a previous cell left behind. + try: + output.parent.mkdir(mode=0o700, parents=False, exist_ok=False) + except FileExistsError as exc: + raise SandboxError("review artifact directory already exists") from exc + except OSError as exc: + raise SandboxError(f"review artifact directory is unavailable: {output.parent}") from exc + + clone = _real_directory(sandbox.clone, label="review clone") if sandbox.backend != "bwrap": return output created: list[str] = [] @@ -214,6 +228,10 @@ class SandboxSession: ReadOnlyMount(self.clone, SANDBOX_WORKSPACE), ReadOnlyMount(self.home, SANDBOX_HOME), ReadOnlyMount(self.temp, SANDBOX_TMP), + # The review artifact directory is a real mount on bwrap, so the + # host-unsafe backend has to translate it too. Without this the + # review prompt names a path that exists on neither backend. + ReadOnlyMount(Path(self.private_root) / REVIEW_OUTPUT_DIRNAME, SANDBOX_REVIEW_OUTPUT), ] for mount in sorted(mappings, key=lambda item: len(item.target), reverse=True): target = mount.target.rstrip("/") @@ -234,9 +252,16 @@ class SandboxSession: SANDBOX_WORKSPACE, SANDBOX_HOME, SANDBOX_TMP, + SANDBOX_REVIEW_OUTPUT, ] ordered = sorted(set(targets), key=len, reverse=True) - pattern = re.compile("|".join(re.escape(target) for target in ordered)) + # Only translate at a path boundary. "/review-output" occurs twice in + # "/review-output/review-output.json" - once as the directory and once + # inside the filename - and rewriting the second turned the artifact path + # into nonsense. A target must be followed by "/", whitespace, a quote or + # end of string to be a path rather than a prefix of a longer name. + boundary = r"""(?=[/\s"']|$)""" + pattern = re.compile("(?:" + "|".join(re.escape(target) for target in ordered) + ")" + boundary) return pattern.sub(lambda match: self.host_path(match.group(0)), value) @property @@ -549,12 +574,17 @@ def build_claude_settings(*, sandbox_enabled: bool = True) -> str: "allowLocalBinding": False, }, "filesystem": { - "allowWrite": [SANDBOX_WORKSPACE, SANDBOX_TMP, SANDBOX_HOME], + # SANDBOX_REVIEW_OUTPUT is the review artifact directory. The + # bwrap bind alone is not enough: this policy is a second, + # independent gate the CLI applies to its own tools, and a path + # missing here is unwritable however the mount is shaped. + "allowWrite": [SANDBOX_WORKSPACE, SANDBOX_TMP, SANDBOX_HOME, SANDBOX_REVIEW_OUTPUT], "denyRead": ["/"], "allowRead": [ SANDBOX_WORKSPACE, SANDBOX_TMP, SANDBOX_HOME, + SANDBOX_REVIEW_OUTPUT, "/usr", "/bin", "/lib", diff --git a/eval/workflow_bench/provider_usage.py b/eval/workflow_bench/provider_usage.py new file mode 100644 index 000000000..a56582be1 --- /dev/null +++ b/eval/workflow_bench/provider_usage.py @@ -0,0 +1,234 @@ +"""Per-request usage as the provider reported it, plus a derived cross-provider view. + +The benchmark has been reading token counts out of Claude Code's session output, +which is Anthropic-shaped whatever actually served the request. That works until +the upstream is OpenAI, because the two providers do not merely name their fields +differently - they mean opposite things by them: + + Anthropic: total_input = input_tokens + + cache_creation_input_tokens + + cache_read_input_tokens + (input_tokens is only the UNCACHED remainder; cache fields ADD) + + OpenAI: total_input = input_tokens + ordinary = input_tokens - cached_tokens - cache_write_tokens + (input_tokens is the WHOLE; cache fields are SUBSETS) + +Adding OpenAI's three together double-counts; subtracting Anthropic's +under-counts. So the native object is authoritative and is stored verbatim, and +the normalized view is derived from it per provider. + +The second rule is that a field nobody reported is UNKNOWN, not zero. A stored +``cache_read = 0`` previously could mean either "the provider said zero" or "our +adapter never looked", and those two must never be written identically again: +the first says caching is not working, the second says we cannot tell. +""" + +from __future__ import annotations + +from collections.abc import Mapping +from dataclasses import dataclass +from typing import Any + +SCHEMA_VERSION = 1 + +# Read by the in-proxy callback and forwarded by the gateway that launches it. +# Defined here because this module is pure stdlib: model_gateway can import the +# name without importing litellm, which only the callback needs. +# +# Both are SWEEP-scoped, and that is a constraint rather than an oversight. +# attach_openai_gateway wraps the whole sweep (runner.py), so one proxy serves +# every cell and its environment is fixed for that proxy's lifetime - while +# cells run concurrently under --workers and interleave requests through it. An +# environment variable therefore cannot carry a per-cell identity: it would +# record one constant against every event. Attributing a request to a cell +# needs an identifier that travels WITH the request; see the session fields the +# callback records for the intended hook. +USAGE_LOG_ENV_VAR = "GITNEXUS_BENCH_PROVIDER_USAGE" +SWEEP_ID_ENV_VAR = "GITNEXUS_BENCH_SWEEP_ID" +USAGE_ENV_VARS = (USAGE_LOG_ENV_VAR, SWEEP_ID_ENV_VAR) + +ANTHROPIC = "anthropic" +OPENAI_RESPONSES = "openai-responses" +# What the gateway's callback actually receives. LiteLLM does not hand a logger +# the upstream body: it normalises usage into its own Chat-Completions-shaped +# object first, so an OpenAI Responses reply arrives as prompt_tokens / +# prompt_tokens_details even though the wire carried input_tokens / +# input_tokens_details. Measured against a real proxy, not assumed - the +# Responses adapter below found none of its keys and reported every field +# unknown. The arithmetic is still OpenAI's (the whole, with subsets). +LITELLM_NORMALIZED = "litellm-normalized" + + +class UsageSemanticsError(ValueError): + """The native usage object does not satisfy its own provider's arithmetic.""" + + +@dataclass(frozen=True) +class NormalizedUsage: + """Cross-provider view. ``None`` means the provider did not report it. + + Deliberately not defaulted to 0: see the module docstring. Every consumer + that sums these has to decide what to do about unknown, and making it None + forces that decision to be explicit instead of silently counting zero. + """ + + ordinary_input_tokens: int | None + cache_read_input_tokens: int | None + cache_write_input_tokens: int | None + total_input_tokens: int | None + output_tokens: int | None + reasoning_output_tokens: int | None + + @property + def complete(self) -> bool: + return all( + value is not None + for value in ( + self.ordinary_input_tokens, + self.cache_read_input_tokens, + self.cache_write_input_tokens, + self.total_input_tokens, + self.output_tokens, + ) + ) + + @property + def unknown_fields(self) -> tuple[str, ...]: + return tuple( + name for name, value in sorted(vars(self).items()) if value is None + ) + + +def _int_or_none(source: Mapping[str, Any] | None, key: str) -> int | None: + """Absent, null, or non-numeric all read as unknown rather than zero.""" + + if not isinstance(source, Mapping): + return None + value = source.get(key) + if isinstance(value, bool) or not isinstance(value, int): + return None + return value + + +def _normalize_openai_responses(usage: Mapping[str, Any]) -> NormalizedUsage: + """input_tokens is the WHOLE; cached and cache-write are subsets of it.""" + + total = _int_or_none(usage, "input_tokens") + details = usage.get("input_tokens_details") + cache_read = _int_or_none(details, "cached_tokens") + cache_write = _int_or_none(details, "cache_write_tokens") + output_details = usage.get("output_tokens_details") + + ordinary: int | None = None + if total is not None and cache_read is not None and cache_write is not None: + ordinary = total - cache_read - cache_write + if ordinary < 0: + raise UsageSemanticsError( + f"OpenAI cached ({cache_read}) + cache_write ({cache_write}) " + f"exceed input_tokens ({total})" + ) + return NormalizedUsage( + ordinary_input_tokens=ordinary, + cache_read_input_tokens=cache_read, + cache_write_input_tokens=cache_write, + total_input_tokens=total, + output_tokens=_int_or_none(usage, "output_tokens"), + # A decomposition of output_tokens, not an addition to it. + reasoning_output_tokens=_int_or_none(output_details, "reasoning_tokens"), + ) + + +def _normalize_litellm(usage: Mapping[str, Any]) -> NormalizedUsage: + """prompt_tokens is the WHOLE; the details are subsets of it.""" + + total = _int_or_none(usage, "prompt_tokens") + details = usage.get("prompt_tokens_details") + cache_read = _int_or_none(details, "cached_tokens") + cache_write = _int_or_none(details, "cache_write_tokens") + if cache_write is None: + cache_write = _int_or_none(details, "cache_creation_tokens") + output_details = usage.get("completion_tokens_details") + + ordinary: int | None = None + if total is not None and cache_read is not None and cache_write is not None: + ordinary = total - cache_read - cache_write + if ordinary < 0: + raise UsageSemanticsError( + f"LiteLLM cached ({cache_read}) + cache_write ({cache_write}) " + f"exceed prompt_tokens ({total})" + ) + return NormalizedUsage( + ordinary_input_tokens=ordinary, + cache_read_input_tokens=cache_read, + cache_write_input_tokens=cache_write, + total_input_tokens=total, + output_tokens=_int_or_none(usage, "completion_tokens"), + reasoning_output_tokens=_int_or_none(output_details, "reasoning_tokens"), + ) + + +def _normalize_anthropic(usage: Mapping[str, Any]) -> NormalizedUsage: + """input_tokens is the uncached REMAINDER; the cache fields add to it.""" + + ordinary = _int_or_none(usage, "input_tokens") + cache_read = _int_or_none(usage, "cache_read_input_tokens") + cache_write = _int_or_none(usage, "cache_creation_input_tokens") + + total: int | None = None + if ordinary is not None and cache_read is not None and cache_write is not None: + total = ordinary + cache_read + cache_write + return NormalizedUsage( + ordinary_input_tokens=ordinary, + cache_read_input_tokens=cache_read, + cache_write_input_tokens=cache_write, + total_input_tokens=total, + output_tokens=_int_or_none(usage, "output_tokens"), + reasoning_output_tokens=None, + ) + + +def canonical_provider(label: str | None, call_type: str | None) -> str | None: + """Map LiteLLM's provider label onto an adapter key, or None if unsure. + + Every "openai" label maps to LITELLM_NORMALIZED regardless of call type, + because anything reaching a proxy callback has already been normalised by + LiteLLM into its own object - measured against a real gateway, where the + observed call type is "anthropic_messages" and the upstream Responses shape + never arrives. OPENAI_RESPONSES stays in the adapter table for a RAW + upstream body, which only direct callers and the wire-shape tests pass. + + An unrecognised label still returns None, so normalize_usage refuses rather + than guessing token semantics. + """ + + if label == "openai": + # Anything reaching a proxy callback has already been normalised by + # LiteLLM, whichever endpoint the caller used - the observed call_type + # for a Claude Code request through this gateway is "anthropic_messages", + # not a Responses one. The adapter has to match the object in hand, not + # the protocol on the wire. + return LITELLM_NORMALIZED + if label in _ADAPTERS: + return label + return None + + +_ADAPTERS = { + ANTHROPIC: _normalize_anthropic, + OPENAI_RESPONSES: _normalize_openai_responses, + LITELLM_NORMALIZED: _normalize_litellm, +} + + +def normalize_usage(provider: str, native_usage: Mapping[str, Any] | None) -> NormalizedUsage: + """Derive the cross-provider view. Never mutates or replaces the native object.""" + + adapter = _ADAPTERS.get(provider) + if adapter is None: + raise UsageSemanticsError( + f"no usage adapter for provider {provider!r}; refusing to guess its token semantics" + ) + if not isinstance(native_usage, Mapping): + return NormalizedUsage(None, None, None, None, None, None) + return adapter(native_usage) diff --git a/eval/workflow_bench/review_scoring.py b/eval/workflow_bench/review_scoring.py index b8ad0151b..eaf328e37 100644 --- a/eval/workflow_bench/review_scoring.py +++ b/eval/workflow_bench/review_scoring.py @@ -12,6 +12,9 @@ from typing import Any, Mapping, Sequence from .oracle_assets import TaskOracleSnapshot REVIEW_OUTPUT = "review-output.json" +# Task verify/oracle commands read the artifact location from here rather +# than hardcoding a path, so one command works under bwrap and host-unsafe. +REVIEW_OUTPUT_ENV_VAR = "GITNEXUS_BENCH_REVIEW_OUTPUT" REVIEW_SCHEMA_VERSION = 1 MAX_REVIEW_BYTES = 256 * 1024 MAX_FINDINGS = 100 @@ -112,13 +115,33 @@ def _parse_review_finding(raw: Any, index: int) -> ReviewFinding: def parse_review_output(path: Path) -> tuple[str, tuple[ReviewFinding, ...]]: - metadata = path.lstat() + # Distinguish these. Folding empty, malformed and encoding failures into one + # message is how a sandbox that left the artifact at 0 bytes read for 15 + # runs as an encoding fault: json.loads("") raises, and every such cell + # reported "not valid UTF-8 JSON". A path the agent never created was not in + # that fold — the lstat below sat outside the try and raised + # FileNotFoundError — but it reached the caller as a bare OSError rather + # than saying what was wrong, which is why it is named here too. + try: + metadata = path.lstat() + except FileNotFoundError as exc: + raise ValueError("review output was never written") from exc + except OSError as exc: + raise ValueError(f"review output is unreadable: {exc.strerror}") from exc if path.is_symlink() or not path.is_file() or metadata.st_size > MAX_REVIEW_BYTES: raise ValueError("review output must be a bounded regular non-symlink file") + if metadata.st_size == 0: + raise ValueError("review output is empty") try: - raw = json.loads(path.read_text()) - except (OSError, UnicodeError, json.JSONDecodeError) as exc: - raise ValueError("review output is not valid UTF-8 JSON") from exc + text = path.read_text(encoding="utf-8") + except OSError as exc: + raise ValueError(f"review output is unreadable: {exc.strerror}") from exc + except UnicodeError as exc: + raise ValueError("review output is not valid UTF-8") from exc + try: + raw = json.loads(text) + except json.JSONDecodeError as exc: + raise ValueError(f"review output is not valid JSON: {exc.msg} at line {exc.lineno}") from exc if not isinstance(raw, Mapping) or set(raw) != {"schema_version", "verdict", "findings"}: raise ValueError("review output requires exactly schema_version, verdict, and findings") if raw["schema_version"] != REVIEW_SCHEMA_VERSION: diff --git a/eval/workflow_bench/run-evolution.sh b/eval/workflow_bench/run-evolution.sh index c9d2a77a1..066df111a 100755 --- a/eval/workflow_bench/run-evolution.sh +++ b/eval/workflow_bench/run-evolution.sh @@ -15,6 +15,8 @@ # MODEL PROPOSER_MODEL EFFORT GENERATIONS RUNS WORKERS PROVIDER # EVOLUTION_PROFILE CE_PLUGIN_DIR CE_PLUGIN_VERSION # INCLUDE_EXPENSIVE SEED_RESULTS CLAUDE_BIN OUT_ROOT +# CI (passes --max-runtime-from-instance-window; the CLI reads /proc/uptime) +# EVENTBRIDGE_INSTANCE_WINDOW_SECONDS EVENTBRIDGE_STOP_RESERVE_SECONDS # UNSAFE_NO_BWRAP=1 (local review diagnostics only) # GITNEXUS_BENCH_ANTHROPIC_API_KEY (legacy GITNEXUS_BENCH_AUTH_TOKEN) # GITNEXUS_BENCH_OPENAI_API_KEY @@ -193,6 +195,18 @@ if ((${#passthrough[@]})); then cmd+=("${passthrough[@]}") fi +# A cancelled GitHub job skips even `if: always()`, so evidence dies with the +# runner. The evolution box is EventBridge-stopped 24h after boot; a Friday +# dispatch inherits leftover uptime. Cap the sweep so it fails in-process and +# the upload step still runs (run 33962002890). The CLI reads /proc/uptime +# itself, in the same breath as it starts the clock the cap is measured +# against; computing a number here — in a separate interpreter, before the +# provenance work and the exec below — charged the sweep for every second +# this script spent afterwards. +if [[ -n "${CI:-}" && -r /proc/uptime ]]; then + cmd+=(--max-runtime-from-instance-window) +fi + if ((dry_run)); then printf '%q ' "${cmd[@]}" printf '\n' @@ -220,5 +234,9 @@ SOURCE_SHA="${source_sha}" RUNTIME_DIGEST="${runtime_digest}" SANDBOX_BACKEND="$ }, null, 2) + "\n")' "${out_root}/runtime-provenance.json" export PYTHONUNBUFFERED=1 +# The runner stamps this on every results.jsonl row and refuses to reuse a +# comparator cell when a prior row's digest disagrees. Keep it on the evolve +# process, not only in the provenance JSON sidecar. +export RUNTIME_DIGEST="${runtime_digest}" cd "${eval_dir}" exec "${cmd[@]}" diff --git a/eval/workflow_bench/runner.py b/eval/workflow_bench/runner.py index 82185186c..42cbfc9cb 100644 --- a/eval/workflow_bench/runner.py +++ b/eval/workflow_bench/runner.py @@ -57,6 +57,17 @@ from typing import Any import yaml +from .comparator_reuse import ( + REUSE_EXCLUDED_ERROR_KINDS, + CellKey, + ComparatorReuseExpectation, + TaskReuseBinding, + current_runtime_digest, + default_reuse_max_age, + load_result_rows, + materialize_reused_row, + select_reusable_comparator_rows, +) from .evolution import ( CANDIDATE_ARMS, EVIDENCE_MAX_AGE_DAYS, @@ -88,12 +99,19 @@ from .oracle_assets import ( ) from .process_control import cancellation_scope, ManagedProcessError from .promotion_apply import committed_destination_base_digests -from .review_scoring import REVIEW_OUTPUT, expected_findings, parse_review_output, score_review +from .review_scoring import ( + REVIEW_OUTPUT, + REVIEW_OUTPUT_ENV_VAR, + expected_findings, + parse_review_output, + score_review, +) from .proposer_sandbox import ( SANDBOX_GITNEXUS as SANDBOX_GITNEXUS, SANDBOX_GITNEXUS_REGISTRY, SANDBOX_GITNEXUS_SHARED as SANDBOX_GITNEXUS_SHARED, SANDBOX_NODE as SANDBOX_NODE, + SANDBOX_REVIEW_OUTPUT, SANDBOX_WORKSPACE, ReadOnlyMount, SandboxError, @@ -104,6 +122,7 @@ from .proposer_sandbox import ( prepare_sandbox, prepare_review_workspace, redact_text, + review_output_path, require_claude_sandbox_helpers, sandbox_workspace_write_boundary, ) @@ -121,6 +140,7 @@ from .runner_artifacts import ( enforce_phase_workspace, enforce_work_evidence, implementation_diff_digest, + copy_isolated_tree, make_worktree, new_plan_doc, parse_shortstat as parse_shortstat, @@ -222,8 +242,9 @@ CE_WORK_DIRECT_PROMPT = ( # Review cell: setup applies a historical PR diff, then the model sees a # read-only checkout. Both arms emit the same strict artifact so quality can be # scored deterministically against labels that remain hidden until it exits. -REVIEW_OUTPUT_CONTRACT = """ -Write /workspace/review-output.json as UTF-8 JSON with exactly this shape: +# Concatenated, not an f-string: the JSON shape below keeps its braces doubled +# because the finished prompt is .format()-ed with the task text. +REVIEW_OUTPUT_CONTRACT = f"\nWrite {SANDBOX_REVIEW_OUTPUT}/{REVIEW_OUTPUT} " + """as UTF-8 JSON with exactly this shape: {{"schema_version":1,"verdict":"approve|comment|request_changes","findings":[{{ "id":"unique stable id","severity":"critical|high|medium|low", "path":"repository-relative changed file","line":1,"end_line":1, @@ -563,9 +584,13 @@ def run_arm( read_only_workspace=True, read_only_paths=_evaluated_skill_roots(worktree, arm), extra_writable_mounts=( + # The DIRECTORY, outside the workspace. Binding the file + # itself left the agent nowhere to put the temp file it + # renames into place, so every review artifact came back + # empty with EROFS in the transcript. ReadOnlyMount( - source=review_output, - target=f"{SANDBOX_WORKSPACE}/{REVIEW_OUTPUT}", + source=review_output.parent, + target=SANDBOX_REVIEW_OUTPUT, ), ), ), @@ -573,7 +598,8 @@ def run_arm( with sandbox_workspace_write_boundary( sandbox, read_only_workspace=True, - writable=(review_output,), + # Nothing in the workspace is writable now — the artifact left it. + writable=(), ): review_session = run_claude( host_text(review_prompt.format(task=task["prompt"])), @@ -584,11 +610,9 @@ def run_arm( sessions.append(review_session) if review_session["ok"] and phase_before is not None: try: - enforce_phase_workspace( - worktree, - phase_before, - allowed_artifact=worktree / REVIEW_OUTPUT, - ) + # The artifact is no longer in the workspace, so the review + # phase may now change nothing there at all. + enforce_phase_workspace(worktree, phase_before, allowed_artifact=None) require_skill_fingerprint( worktree, arm, @@ -643,6 +667,18 @@ def run_arm( record = sum_sessions(sessions) record["arm"] = arm record["plan_produced"] = arm not in ("workflow", "ce_workflow") or plan_doc is not None + # The verify command runs in its OWN sandbox invocation, which knows nothing + # about the review session's writable mount. Expose the artifact read-only and + # name it through the environment, the same shape _run_hidden_oracle uses, so + # one command works on both backends instead of hardcoding either path. + verify_env = environment_builder() + verify_mounts: tuple[ReadOnlyMount, ...] = () + if arm in ("review", "ce_review"): + review_artifact = review_output_path(sandbox, REVIEW_OUTPUT) + verify_env[REVIEW_OUTPUT_ENV_VAR] = host_text(f"{SANDBOX_REVIEW_OUTPUT}/{REVIEW_OUTPUT}") + verify_mounts = ( + ReadOnlyMount(source=review_artifact.parent, target=SANDBOX_REVIEW_OUTPUT), + ) authored_tests_passed, authored_test_output = _verification_outcome( run_verify( task["verify"], @@ -651,21 +687,25 @@ def run_arm( command_prefix=sandbox.command_prefix_for( read_only_workspace=True, unshare_network=True, + extra_read_only_mounts=verify_mounts, ), - env=environment_builder(), + env=verify_env, require_pid_namespace=getattr(sandbox, "require_pid_namespace", True), ) ) review_score: dict[str, Any] | None = None if arm in ("review", "ce_review"): try: - verdict, findings = parse_review_output(worktree / REVIEW_OUTPUT) + verdict, findings = parse_review_output(review_output_path(sandbox, REVIEW_OUTPUT)) labels = expected_findings(oracle_snapshot) if oracle_snapshot is not None else () review_score = score_review(verdict, findings, labels) except (OSError, ValueError) as exc: record["ok"] = False record["error_kind"] = record["error_kind"] or "review-evidence-invalid" - record["error_detail"] = str(exc) + # Keep the FIRST detail, as error_kind already does. A phase- + # boundary violation is why the artifact is unparseable; reporting + # the parse failure over it buries the cause under the symptom. + record["error_detail"] = record.get("error_detail") or str(exc) record["review_score"] = review_score record["review_evidence_valid"] = review_score is not None if review_score is not None: @@ -721,9 +761,53 @@ CHURN_FIELDS = ("diff_files", "diff_insertions", "diff_deletions") # Rows where the session (or the harness) died carry no measured evidence and # must not skew efficiency medians or resolve denominators. verify-failed and # skill-not-invoked rows DO count: those sessions ran and spent real tokens. -EXCLUDED_ERROR_KINDS = frozenset( - {"session-error", "infra-error", "evidence-unverified", "cleanup-failure", "review-evidence-invalid", "cancelled"} -) +# One definition, in comparator_reuse: reuse eligibility and aggregate +# exclusion must never drift apart. The dependency only runs this way - +# comparator_reuse importing back from runner is a circular import. +EXCLUDED_ERROR_KINDS = REUSE_EXCLUDED_ERROR_KINDS + + + +# Health classification. These answer "did the harness work", which is a +# different question from "did the agent get the right answer" - a review can be +# wrong about a hard corpus while every process, mount and capture behaved. +# +# EXECUTION: the process or its tooling did not complete. Nothing was measured. +# EVIDENCE: it completed, but what it produced cannot be trusted or scored. +# Everything else - including resolved=False and a zero score - is a VALID +# NEGATIVE: an admissible measurement that the quality gate then judges. +EXECUTION_FAILURE_KINDS = frozenset({"session-error", "infra-error", "cleanup-failure", "cancelled"}) +EVIDENCE_FAILURE_KINDS = frozenset({"review-evidence-invalid", "evidence-unverified", "skill-not-invoked"}) + + +def execution_failed(record: Mapping[str, Any]) -> bool: + """The process or its tooling did not complete.""" + + return record.get("error_kind") in EXECUTION_FAILURE_KINDS + + +def evidence_failed(record: Mapping[str, Any]) -> bool: + """It completed, but what it produced cannot be trusted or scored.""" + + return ( + record.get("error_kind") in EVIDENCE_FAILURE_KINDS + or record.get("review_evidence_valid") is False + or record.get("transcript_missing") is True + ) + + +def task_prompt_digest(task: Mapping[str, Any]) -> str: + """The prompt digest, computed once for the row and the reuse expectation. + + row_is_reusable_comparator compares the value a prior row stored against the + value this sweep derives, as exact strings. Two inline copies of this hash + had already drifted - one picked up a str() cast the other lacked - and a + further divergence (normalising whitespace on one side, say) would silently + stop rows matching, or match rows that should not. + """ + + return hashlib.sha256(str(task["prompt"]).encode()).hexdigest() + # A sustained upstream outage shows up as a run of session/infra/cleanup # failures. (cleanup-failure overwrites the primary error_kind, so a @@ -806,7 +890,13 @@ def sweep_task_cells( raise ValueError("workers must be positive") for wave_start in range(0, len(cells), workers): if cancel_event.is_set(): - return outage_streak, True + # False: this flag means the OUTAGE breaker tripped, and the + # caller turns it into exit 1 with "Sweep aborted". Cancellation + # stops the sweep too, but it is the operator's Ctrl-C, not a + # systemic failure - reporting True relabelled every interrupted + # run an outage and returned 1 where the contract says 130. The + # caller tests cancel_event itself for the stop decision. + return outage_streak, False wave = list(cells[wave_start : wave_start + workers]) for run_idx, arm in wave: on_start(run_idx, arm) @@ -833,7 +923,7 @@ def sweep_task_cells( # an earlier row trips the breaker; only later waves are skipped. on_record(run_idx, arm, record) if cancel_event.is_set(): - return outage_streak, True + return outage_streak, False for record in records: kind = ( "review-evidence-invalid" @@ -847,10 +937,209 @@ def sweep_task_cells( "failures — aborting the remaining sweep; report and promotion are written " "from partial evidence and the run exits non-zero." ) + # Signal in-flight background work too. The breaker exists to + # SHORTEN a doomed run; without this a graph prefetch keeps + # building and the unconditional join blocks the abort for the + # length of a full clone and offline index. + cancel_event.set() return outage_streak, True return outage_streak, False +# How far ahead of the in-order fold pointer cells may be submitted, as a +# multiple of the worker count. This is the wall-clock/wasted-cell trade, and it +# is a real one - measured against the review corpus at workers=3, with failures +# injected at four different positions: +# +# window wall vs waves worst overrun +# 3 -8% 2 (the wave scheduler's own bound) +# 6 -27% 4 +# 12 -42% 9 +# 54 -44% 11 +# +# Overrun is wasted paid sessions when the breaker trips, at roughly $70 each. +# 2 is the default because it keeps the worst case within 2x the wave bound +# while taking most of the gain; raise it if a run's wall clock costs more than +# an occasional handful of cells on an aborted sweep. +PACKED_WINDOW_MULTIPLIER = 2 + + +def sweep_packed_cells( + cells: Sequence[tuple[str, int, str]], + *, + workers: int, + run: Callable[[str, int, str], dict[str, Any]], + on_start: Callable[[str, int, str], None], + on_record: Callable[[str, int, str, dict[str, Any]], None], + outage_streak: int, + outage_limit: int, + window: int | None = None, + await_ready: Callable[[str], bool] | None = None, + cancel_event: threading.Event | None = None, +) -> tuple[int, bool]: + """Run cells from EVERY task through one pool; return (streak, tripped). + + ``sweep_task_cells`` finishes one task before starting the next and drains a + wave before refilling it, so a task with fewer cells than ``workers`` leaves + workers idle and a slow cell stalls its whole wave. Packing every task's + cells into one continuously fed pool removes both, which is worth about 40% + of a cold sweep's wall clock and is the only thing that moves a seeded + weekly run at all - there, a task is three cells and a wave is never full. + + The breaker keeps its exact meaning. ``cells`` is a total submission order + (task-major, run-major, arm-minor - the same order waves fold in, continued + across task boundaries), a folder walks results in precisely that order, and + "consecutive systemic failures" is evaluated there. So the run aborts on the + same logical cell it would have aborted on under waves. + + ``window`` is what bounds the overrun, and it is load-bearing. The halt flag + alone is not enough: the folder walks in order, so a slow early cell lets + workers race ahead, and by the time the breaker trips those cells have + already paid for their sessions. Measured, an unbounded queue overran by 11 + cells at ``workers=3`` where the wave scheduler overruns by 2. Holding + submission to ``window`` cells beyond the fold point caps it, trading + packing for wasted cells - see ``PACKED_WINDOW_MULTIPLIER`` for the curve. + + ``await_ready`` gates a task's first cell on whatever that task still needs + (a sanitized clone, a graph). It returns False to abandon the task, whose + cells are then skipped rather than run against missing assets. Cells are + submitted as their task becomes ready, so a later task's graph builds while + earlier cells are still paying for sessions. + """ + + with cancellation_scope(cancel_event) as cancel_event: + if workers < 1: + raise ValueError("workers must be positive") + if not cells: + return outage_streak, False + if window is None: + window = max(workers * PACKED_WINDOW_MULTIPLIER, workers) + if window < workers: + raise ValueError("window must be at least workers, or the pool starves") + + halt = threading.Event() + results: list[dict[str, Any] | None] = [None] * len(cells) + submitted: list[Any] = [] + gate = threading.Condition() + producing = True + fold_pointer = 0 + + def execute(index: int) -> None: + if halt.is_set() or cancel_event.is_set(): + return + task_id, run_idx, arm = cells[index] + on_start(task_id, run_idx, arm) + results[index] = run(task_id, run_idx, arm) + + pool = ThreadPoolExecutor(max_workers=workers) + # cancellation_scope binds _CANCELLATION in the CALLING thread's + # context, and a new thread starts with an empty one - so the producer + # has to copy this context rather than its own, or every cell it + # submits loses the run's cancellation event. sweep_task_cells gets + # this for free by submitting from the thread that entered the scope. + caller_context = copy_context() + + def produce() -> None: + nonlocal producing + ready_tasks: dict[str, bool] = {} + try: + for index, (task_id, _run_idx, _arm) in enumerate(cells): + if halt.is_set() or cancel_event.is_set(): + break + if task_id not in ready_tasks: + ready_tasks[task_id] = True if await_ready is None else await_ready(task_id) + if not ready_tasks[task_id]: + with gate: + submitted.append(None) + gate.notify_all() + continue + with gate: + while index - fold_pointer >= window and not halt.is_set(): + gate.wait(timeout=0.5) + if halt.is_set() or cancel_event.is_set(): + break + worker_context = caller_context.run(copy_context) + submitted.append(pool.submit(worker_context.run, execute, index)) + gate.notify_all() + finally: + with gate: + producing = False + gate.notify_all() + + producer = threading.Thread(target=produce, name="packed-cell-producer", daemon=False) + producer.start() + + tripped = False + try: + index = 0 + while True: + with gate: + while index >= len(submitted) and producing: + gate.wait(timeout=0.5) + if index >= len(submitted): + break + future = submitted[index] + if future is not None: + try: + future.result() + except BaseException: + # Same contract as sweep_task_cells: the cells submitted + # after this one have already run and spent their budget, + # so persist their rows in submission order before the + # harness bug takes the process down. Without this, one + # crashing cell silently erases the paid evidence of + # every sibling that had already finished. The failing + # index itself has no row - execute() only assigns on + # success - so folding forward cannot duplicate it. + with gate: + settled = list(submitted) + for later in range(index + 1, len(settled)): + pending = settled[later] + if pending is not None and not pending.done(): + continue + row = results[later] + if row is not None: + on_record(*cells[later], row) + raise + record = results[index] + if record is not None: + task_id, run_idx, arm = cells[index] + on_record(task_id, run_idx, arm, record) + kind = ( + "review-evidence-invalid" + if record.get("review_evidence_valid") is False + else record.get("error_kind") + ) + outage_streak = systemic_outage_streak(kind, outage_streak) + if outage_limit and outage_streak >= outage_limit: + print( + f"[systemic-outage] {outage_streak} consecutive unusable-evidence " + "failures — aborting the remaining sweep; report and promotion are " + "written from partial evidence and the run exits non-zero." + ) + tripped = True + halt.set() + cancel_event.set() + break + index += 1 + with gate: + fold_pointer = index + gate.notify_all() + if cancel_event.is_set(): + tripped = True + break + finally: + halt.set() + with gate: + gate.notify_all() + producer.join() + for pending in submitted[index + 1 :]: + if pending is not None: + pending.cancel() + pool.shutdown(wait=True) + return outage_streak, tripped + + @dataclass(frozen=True) class TaskCellContext: """Everything one benchmark cell needs from its task, prepared once. @@ -884,6 +1173,8 @@ class TaskCellContext: candidate_overlay: Path | None overlay_digest: str | None sandbox_backend: str = "bwrap" + clone_template: Path | None = None + sanitized_head: str | None = None def run_cell(ctx: TaskCellContext, run_idx: int, arm: str) -> dict[str, Any]: @@ -908,8 +1199,14 @@ def run_cell(ctx: TaskCellContext, run_idx: int, arm: str) -> dict[str, Any]: raise RuntimeError("sanitized graph snapshot is unavailable") if ctx.asset_snapshot is None: raise RuntimeError("task asset snapshot is unavailable") - worktree = make_worktree(ctx.repo, ctx.task_sha, ctx.trees_dir) - sanitized_head = sanitize_clone_for_hidden_oracles(worktree) + if ctx.clone_template is not None: + if not ctx.sanitized_head: + raise RuntimeError("clone template is missing its sanitized HEAD") + worktree = copy_isolated_tree(ctx.clone_template, ctx.trees_dir) + sanitized_head = ctx.sanitized_head + else: + worktree = make_worktree(ctx.repo, ctx.task_sha, ctx.trees_dir) + sanitized_head = sanitize_clone_for_hidden_oracles(worktree) ctx.graph_snapshot.materialize(worktree, sanitized_head=sanitized_head) dependency_mounts = stage_task_assets( task, @@ -1014,7 +1311,7 @@ def run_cell(ctx: TaskCellContext, run_idx: int, arm: str) -> dict[str, Any]: oracle_snapshot=ctx.oracle_snapshot, ) if execution_arm in ("review", "ce_review"): - review_source = worktree / REVIEW_OUTPUT + review_source = review_output_path(sandbox, REVIEW_OUTPUT) if review_source.is_file() and not review_source.is_symlink(): review_artifact = ctx.out_dir / f"{task['id']}-{arm}-run{run_idx}.review.json" review_artifact.write_bytes(_bounded_regular_bytes(review_source, limit=256 * 1024)) @@ -1054,9 +1351,10 @@ def run_cell(ctx: TaskCellContext, run_idx: int, arm: str) -> dict[str, Any]: "task_base_sha": ctx.task_sha, "sanitized_task_sha": sanitized_head, "variant_head_sha": orig_sha, - "task_prompt_digest": hashlib.sha256(task["prompt"].encode()).hexdigest(), + "task_prompt_digest": task_prompt_digest(task), "skill_digest": expected_skill_digest, "candidate_overlay_digest": (ctx.overlay_digest if arm in CANDIDATE_ARMS else None), + "runtime_digest": current_runtime_digest(), "recorded_at": datetime.now(UTC).isoformat(), } ) @@ -1243,7 +1541,30 @@ def aggregate(records: list[dict[str, Any]]) -> dict[str, Any]: out["cost_usd"] = ( None if (not valid or any(cost is None for cost in valid_costs)) else statistics.median(valid_costs) ) + fresh = [r for r in records if not r.get("reused")] + out["fresh_attempts"] = len(fresh) + out["execution_failures"] = sum(1 for r in fresh if execution_failed(r)) + out["evidence_failures"] = sum(1 for r in fresh if evidence_failed(r)) + # Admissible means the harness delivered a trustworthy measurement. It says + # nothing about whether the answer was right, which is the whole point. + # + # Count the rows that failed NEITHER way rather than subtracting both + # counters: run_arm keeps a pre-existing session error and still marks the + # review evidence invalid, so one row can land in both. Subtracting it twice + # drove an arm holding real measurements to admissible=0, which arm_health + # reads as UNUSABLE and enforce_measurement_health then fails the sweep on. + out["admissible"] = sum(1 for r in fresh if not execution_failed(r) and not evidence_failed(r)) + out["health_reasons"] = sorted( + { + str(r.get("error_kind")) + for r in fresh + if r.get("error_kind") in EXECUTION_FAILURE_KINDS or r.get("error_kind") in EVIDENCE_FAILURE_KINDS + } + ) out["resolved"] = sum(1 for r in records if r["resolved"]) + # Reused rows are last generation's measurement. The health canary below has + # to ask whether THIS environment worked, so it needs the freshly-run count. + out["resolved_fresh"] = sum(1 for r in records if r["resolved"] and not r.get("reused")) out["runs"] = len(records) out["valid_runs"] = len(valid) out["excluded_runs"] = len(records) - len(valid) @@ -1270,6 +1591,21 @@ def aggregate(records: list[dict[str, Any]]) -> dict[str, Any]: "review_category_accuracy", "review_grounded_evidence", ) + # NOTE: a skill-not-invoked row still contributes to these medians. That is + # a real measurement gap - an arm exists to measure a SKILL, and a cell + # where the skill never ran did not measure it - but the narrow fix is + # WORSE than the gap, so it is deliberately not applied here. + # + # Filtering those rows out of the quality metrics alone leaves valid_runs + # and excluded_runs counting them, so the promotion gate sees N clean runs + # while the median was taken over fewer. Because the dropped rows are + # systematically an arm's worst, that biases toward PROMOTING: measured on + # one real run at 0.9 plus two uninvoked rows at 0.0, the gate flipped from + # keep_incumbent to promote. The three verdict fields below compound it - + # they are all() reducers, so one uninvoked cell flips a whole arm. + # Closing this honestly needs a scored-run count and a paired-equality + # check in the gate itself: a promotion-semantics change, not an + # aggregation fix. if any("review_weighted_f1" in record for record in valid): for metric in review_metrics: values = [record[metric] for record in valid if record.get(metric) is not None] @@ -1301,11 +1637,135 @@ def savings(baseline: dict[str, Any], workflow: dict[str, Any]) -> dict[str, Any return out +@dataclass(frozen=True) +class ArmHealth: + """What the harness observed for one arm this sweep, before any judgement.""" + + arm: str + fresh_attempts: int + admissible: int + execution_failures: int + evidence_failures: int + reasons: tuple[str, ...] + + @property + def measured(self) -> bool: + """False when only reused rows exist - current health is UNKNOWN, not good.""" + + return self.fresh_attempts > 0 + + @property + def status(self) -> str: + """UNKNOWN / OBSERVED_OK / DEGRADED / UNUSABLE. + + DEGRADED is the distinction that matters: an arm with both admissible + measurements and observed failures produced usable evidence but did not + run reliably. Reporting that as healthy is how a partly-broken sweep + looks fine. It is diagnostic here - only UNUSABLE is fatal - so this + patch changes what is reported, not what is eligible. + """ + + if not self.measured: + return "UNKNOWN" + failures = self.execution_failures + self.evidence_failures + if self.admissible == 0 and failures > 0: + return "UNUSABLE" + if failures > 0: + return "DEGRADED" + return "OBSERVED_OK" + + @property + def unhealthy(self) -> bool: + """Every fresh attempt failed to execute or to produce usable evidence. + + Deliberately not "resolved zero tasks". A reviewer can be wrong about + every task in a hard corpus with the harness working perfectly; that is + a valid negative and belongs to the quality gate, not here. + """ + + return self.status == "UNUSABLE" + + +def arm_health(results: dict[str, dict[str, dict[str, Any]]], arms: set[str]) -> dict[str, ArmHealth]: + """Fold per-task aggregates into one health observation per arm.""" + + health: dict[str, ArmHealth] = {} + for arm in sorted(arms): + rows = [task_arms[arm] for task_arms in results.values() if arm in task_arms] + if not rows: + continue + reasons: set[str] = set() + for row in rows: + reasons.update(row.get("health_reasons") or ()) + health[arm] = ArmHealth( + arm=arm, + fresh_attempts=sum(int(r.get("fresh_attempts", 0)) for r in rows), + admissible=sum(int(r.get("admissible", 0)) for r in rows), + execution_failures=sum(int(r.get("execution_failures", 0)) for r in rows), + evidence_failures=sum(int(r.get("evidence_failures", 0)) for r in rows), + reasons=tuple(sorted(reasons)), + ) + return health + + +def unhealthy_arms(results: dict[str, dict[str, dict[str, Any]]], arms: set[str]) -> list[ArmHealth]: + """Arms whose every fresh attempt failed to execute or to produce evidence.""" + + return [h for h in arm_health(results, arms).values() if h.unhealthy] + + +def unmeasured_arms(results: dict[str, dict[str, dict[str, Any]]], arms: set[str]) -> list[str]: + """Arms with no fresh attempt at all - reported as unknown, never as healthy.""" + + return [h.arm for h in arm_health(results, arms).values() if not h.measured] + + +def enforce_measurement_health( + results: dict[str, dict[str, dict[str, Any]]], arms: set[str] +) -> dict[str, ArmHealth]: + """Report every arm's measurement status; abort only on UNUSABLE. + + Runs after report.md and promotion.json are written, so a failing sweep + still leaves its evidence behind. Reports cause as undetermined: an empty + artifact establishes that evidence is unusable, not why - naming a mount + failure here would be a guess the recorded rows do not support. + """ + + health = arm_health(results, arms) + for arm in sorted(health): + observed = health[arm] + reasons = f" reason={','.join(observed.reasons)}" if observed.reasons else "" + print( + f"[measurement-health] {arm}: {observed.status} " + f"fresh_attempts={observed.fresh_attempts} admissible={observed.admissible} " + f"execution_failures={observed.execution_failures} " + f"evidence_failures={observed.evidence_failures}{reasons}" + ) + unusable = [h for h in health.values() if h.unhealthy] + if unusable: + detail = "; ".join(f"{h.arm} ({h.fresh_attempts} fresh attempt(s))" for h in unusable) + print( + f"[measurement-health] {detail} produced no usable measurement this sweep. " + "cause=undetermined — see error_detail in results.jsonl. Exiting non-zero rather " + "than reporting a quiet no-promotion." + ) + raise SystemExit(1) + return health + + def broken_incumbent_arms( results: dict[str, dict[str, dict[str, Any]]], incumbent_arms: set[str], ) -> list[str]: - """Incumbent arms that resolved nothing across every task they ran. + """LEGACY, NON-AUTHORITATIVE. Superseded by ``enforce_measurement_health``. + + Kept only so its historical behaviour stays documented and testable while + the replacement settles; it has no production caller. Do not wire it into a + health decision - it infers a broken environment from a resolution count, + which a reviewer facing a hard corpus falsifies. Remove once the + measurement-health path has run in CI. + + Incumbent arms that resolved nothing across every task they ran. An incumbent arm is the currently-shipped, presumably-working skill: if it resolves NOTHING across every task it ran, that reads as an environment or @@ -1323,9 +1783,19 @@ def broken_incumbent_arms( some-runs-resolved-zero case since here nothing completed at all. aggregate() never marks an excluded/unverifiable row resolved=True, so resolved == 0 alone already covers both cases. + + A reused row proves last generation's environment worked, not this one's, so + the count consulted here is ``resolved_fresh``. Without that, an arm whose + cells were all reused always looks healthy and the canary can never fire — + which is exactly when a broken environment would go unnoticed. The sweep + keeps one paid cell per incumbent arm so this count is never vacuous. """ present = incumbent_arms & {arm for arms in results.values() for arm in arms} - return sorted(arm for arm in present if all(arms[arm]["resolved"] == 0 for arms in results.values() if arm in arms)) + return sorted( + arm + for arm in present + if all(arms[arm].get("resolved_fresh", arms[arm]["resolved"]) == 0 for arms in results.values() if arm in arms) + ) def _cost_cell(value: Any) -> str: @@ -1557,6 +2027,15 @@ def build_parser() -> argparse.ArgumentParser: parser.add_argument("--task-bindings-json", default=None, help=argparse.SUPPRESS) parser.add_argument("--promotion-target-bases-json", default=None, help=argparse.SUPPRESS) parser.add_argument("--unsafe-no-bwrap", action="store_true", help=argparse.SUPPRESS) + parser.add_argument( + "--reuse-results", + type=Path, + default=None, + help="prior wfbench results dir whose incumbent/CE rows may be reused " + "when model, effort, tasks, oracles, skill bytes, and CE plugin still " + "match. Candidate arms always run. Used by evolve.py so a weekly " + "generation does not re-pay for an unchanged comparator.", + ) return parser @@ -1684,6 +2163,244 @@ def main() -> None: gateway.__exit__(None, None, None) +def _comparator_reuse_expectation( + *, + args: argparse.Namespace, + tasks: Sequence[Any], + task_bindings: Sequence[Mapping[str, Any]], + oracle_snapshots: Sequence[Any], + asset_snapshots: Mapping[str, Any], + sandbox_backend: str, + ce_plugin_snapshot: CePluginSnapshot | None, +) -> ComparatorReuseExpectation: + """Bind this sweep's immutable identity for comparator-row reuse.""" + + skill_digests: dict[str, str | None] = {} + for arm in args.arms: + execution = CANDIDATE_ARMS.get(arm, arm) + if execution in EVALUATED_ARM_SKILLS: + skill_digests[arm] = skill_fingerprint(HARNESS_ROOT, execution) + else: + skill_digests[arm] = None + task_locks: dict[str, TaskReuseBinding] = {} + for task, binding, oracle in zip(tasks, task_bindings, oracle_snapshots, strict=True): + task_locks[str(task["id"])] = TaskReuseBinding( + task_base_sha=str(binding["resolved_sha"]), + task_prompt_digest=task_prompt_digest(task), + oracle_digest=oracle.digest, + oracle_command_digest=oracle.command_digest, + oracle_manifest_digest=oracle.manifest_digest, + task_asset_manifest_digest=getattr( + asset_snapshots.get(str(task["id"])), "manifest_digest", None + ), + sandbox_dependency_manifest_digest=getattr( + asset_snapshots.get(str(task["id"])), "dependency_manifest_digest", None + ), + ) + return ComparatorReuseExpectation( + model=args.model, + effort=args.effort, + sandbox_backend=sandbox_backend, + runtime_digest=current_runtime_digest(), + now=datetime.now(UTC), + max_age=default_reuse_max_age(), + tasks=task_locks, + skill_digests=skill_digests, + ce_plugin_version=ce_plugin_snapshot.version if ce_plugin_snapshot is not None else None, + ce_plugin_manifest_digest=( + ce_plugin_snapshot.manifest_digest if ce_plugin_snapshot is not None else None + ), + ) + + +def task_has_planned_paid_cells( + task: Mapping[str, Any], + *, + arms: Sequence[str], + runs: int, + reusable_rows: Mapping[tuple[str, str, int], object], + reuse_source: Path | None, +) -> bool: + """True when at least one planned cell is not a reusable comparator row.""" + + task_id = str(task["id"]) + for run_idx in range(runs): + for arm in arms: + if reuse_source is None or (task_id, arm, run_idx) not in reusable_rows: + return True + return False + + +def drop_canary_reuse_key( + reusable_rows: dict[CellKey, dict[str, Any]], + *, + arm: str, + tasks: Sequence[Mapping[str, Any]], + runs: int, +) -> CellKey | None: + """Drop one reusable cell so an incumbent arm still measures THIS sweep. + + An arm reused end to end measures nothing about today's environment, and + arm_health would then be reading last week's health. + + Counted against the cells this sweep PLANS, not every key reuse selection + returned: selection accepts any non-negative prior run index, so a results + directory produced with more runs than this invocation leaves extra keys. + Comparing against those made the check false exactly when it mattered, and + the canary silently stopped firing while every planned cell stayed reused. + + Returns the dropped key, or None when the arm already has a paid cell. + """ + + planned_keys = [(str(task["id"]), arm, run_idx) for task in tasks for run_idx in range(runs)] + arm_keys = sorted(key for key in planned_keys if key in reusable_rows) + if not arm_keys or len(arm_keys) != len(planned_keys): + return None + dropped = arm_keys[0] + del reusable_rows[dropped] + return dropped + + +def next_graph_prefetch_target( + remaining: Sequence[tuple[Mapping[str, Any], Mapping[str, Any]]], + *, + arms: Sequence[str], + runs: int, + reusable_rows: Mapping[tuple[str, str, int], object], + reuse_source: Path | None, + ready_keys: set[tuple[str, str]], +) -> tuple[Mapping[str, Any], Mapping[str, Any], tuple[str, str]] | None: + """Next later task that still needs a clone template and sanitized graph.""" + + for task, binding in remaining: + if not task_has_planned_paid_cells( + task, + arms=arms, + runs=runs, + reusable_rows=reusable_rows, + reuse_source=reuse_source, + ): + continue + key = (str(binding["repo_identity"]), str(binding["resolved_sha"])) + if key in ready_keys: + continue + return task, binding, key + return None + + +@dataclass(frozen=True) +class GraphBuildEnv: + """Per-sweep state every graph build shares, and the caches it fills. + + The four dicts are the sweep's memo of what has already been built, keyed by + (repo, sha). They are mutable by design and are written by both the sweep + thread and the prefetch thread, which is safe only because a build is + started for a key exactly once and joined before that key is read. + """ + + trees: Path + task_asset_cache: TaskAssetCache + claude_bin: Path | str + bwrap_bin: Path | str + sandbox_backend: str + runtime_mounts: Sequence[ReadOnlyMount] + clone_templates: dict[tuple[str, str], tuple[Path, str]] + clone_template_errors: dict[tuple[str, str], BaseException] + graph_snapshots: dict[tuple[str, str], SanitizedGraphSnapshot] + graph_snapshot_errors: dict[tuple[str, str], BaseException] + + def ready_keys(self) -> set[tuple[str, str]]: + """Keys whose build has already been attempted, successfully or not.""" + + return ( + set(self.clone_templates) + | set(self.clone_template_errors) + | set(self.graph_snapshots) + | set(self.graph_snapshot_errors) + ) + + +def ensure_task_graph( + *, + task: Mapping[str, Any], + repo: Path, + task_sha: str, + graph_key: tuple[str, str], + env: GraphBuildEnv, +) -> None: + """Build one SHA's sanitized clone template and graph. Idempotent per key.""" + + if graph_key in env.graph_snapshots or graph_key in env.graph_snapshot_errors: + return + try: + validate_no_prebuilt_graph_assets(task) + if graph_key not in env.clone_templates and graph_key not in env.clone_template_errors: + template = make_worktree(repo, task_sha, env.trees) + template_head = sanitize_clone_for_hidden_oracles(template) + env.clone_templates[graph_key] = (template, template_head) + clone_template: Path | None = None + template_head: str | None = None + if graph_key in env.clone_templates: + clone_template, template_head = env.clone_templates[graph_key] + if graph_key in env.clone_template_errors: + env.graph_snapshot_errors[graph_key] = env.clone_template_errors[graph_key] + return + env.graph_snapshots[graph_key] = prepare_sanitized_graph( + task, + repo=repo, + resolved_sha=task_sha, + parent=env.trees, + cache=env.task_asset_cache, + claude_bin=env.claude_bin, + bwrap_bin=env.bwrap_bin, + sandbox_backend=env.sandbox_backend, + runtime_mounts=env.runtime_mounts, + clone_template=clone_template, + sanitized_head=template_head, + ) + except (ManagedProcessError, OSError, SandboxError, RuntimeError, ValueError) as exc: + env.graph_snapshot_errors[graph_key] = exc + env.clone_template_errors.setdefault(graph_key, exc) + + +@dataclass +class GraphPrefetch: + """In-flight clone+graph build for a later task SHA.""" + + key: tuple[str, str] + thread: threading.Thread + + def join(self) -> None: + self.thread.join() + + +def prefetch_next_graph( + *, + task: Mapping[str, Any], + binding: Mapping[str, Any], + graph_key: tuple[str, str], + env: GraphBuildEnv, + cancel_event: threading.Event, +) -> GraphPrefetch: + """Start clone+graph prep for the next unpaid SHA during paid sessions.""" + + repo = Path(binding["repo_identity"]) + task_sha = str(binding["resolved_sha"]) + + def run() -> None: + if cancel_event.is_set(): + return + print(f"[prefetch_next_graph] clone+graph for {task_sha}") + ensure_task_graph(task=task, repo=repo, task_sha=task_sha, graph_key=graph_key, env=env) + + # copy_context, as the worker pool already does at _run_wave: a plain Thread + # does not inherit ContextVars, so without this every run_managed inside the + # graph build resolves _CANCELLATION to None and ignores the shared cancel. + thread = threading.Thread(target=copy_context().run, args=(run,), name="prefetch_next_graph", daemon=False) + thread.start() + return GraphPrefetch(key=graph_key, thread=thread) + + def _run_sweep( args: argparse.Namespace, *, @@ -1740,107 +2457,213 @@ def _run_sweep( except (OSError, SandboxError, ValueError) as exc: parser.error(str(exc)) raise AssertionError("ArgumentParser.error() returned unexpectedly") - graph_snapshots: dict[tuple[str, str], SanitizedGraphSnapshot] = {} - graph_snapshot_errors: dict[tuple[str, str], BaseException] = {} - for task, task_binding, oracle_snapshot in zip( - tasks, - task_bindings, - oracle_snapshots, - strict=True, - ): - if outage_tripped or cancel_event.is_set(): - break - repo = Path(task_binding["repo_identity"]) - task_sha = task_binding["resolved_sha"] - asset_snapshot: TaskAssetSnapshot | None = None - asset_snapshot_error: BaseException | None = None - graph_key = (str(repo), task_sha) - graph_snapshot: SanitizedGraphSnapshot | None = graph_snapshots.get(graph_key) - graph_snapshot_error: BaseException | None = graph_snapshot_errors.get(graph_key) + # Asset snapshots are built here, up front, for two reasons. Comparator + # reuse has to compare this sweep's task-asset and dependency digests + # against the prior row's, and those digests do not exist until the + # snapshot does. Building them all before any cell or prefetch thread + # starts also keeps TaskAssetCache single-threaded, which is what its own + # "plain dict, read-then-write race" comment asks for. + asset_snapshots: dict[str, TaskAssetSnapshot] = {} + asset_snapshot_errors: dict[str, BaseException] = {} + for _task, _binding in zip(tasks, task_bindings, strict=True): try: - validate_no_prebuilt_graph_assets(task) - if graph_snapshot is None and graph_snapshot_error is None: - graph_snapshot = prepare_sanitized_graph( - task, - repo=repo, - resolved_sha=task_sha, - parent=Path(trees), - cache=task_asset_cache, - claude_bin=args.claude_bin, - bwrap_bin=bwrap_bin, - sandbox_backend=sandbox_backend, - runtime_mounts=runtime_mounts, - ) - graph_snapshots[graph_key] = graph_snapshot - except (ManagedProcessError, OSError, SandboxError, RuntimeError, ValueError) as exc: - graph_snapshot_error = exc - graph_snapshot_errors[graph_key] = exc - try: - # Prepared here, once, rather than lazily inside the first cell: - # TaskAssetCache is a plain dict, so a lazy build would be a - # read-then-write race the moment cells stop running serially. - asset_snapshot = task_asset_cache.prepare( - task, - repo=repo, - resolved_sha=task_sha, - expected_dependency_binding=task_binding, + asset_snapshots[str(_task["id"])] = task_asset_cache.prepare( + _task, + repo=Path(_binding["repo_identity"]), + resolved_sha=_binding["resolved_sha"], + expected_dependency_binding=_binding, ) except (OSError, SandboxError, ValueError) as exc: - asset_snapshot_error = exc - cell_context = TaskCellContext( - task=task, - oracle_snapshot=oracle_snapshot, - repo=repo, - task_sha=task_sha, - graph_snapshot=graph_snapshot, - graph_snapshot_error=graph_snapshot_error, - asset_snapshot=asset_snapshot, - asset_snapshot_error=asset_snapshot_error, - args=args, - out_dir=out_dir, - ce_plugin_snapshot=ce_plugin_snapshot, - trees_dir=Path(trees), - bwrap_bin=bwrap_bin, - sandbox_backend=sandbox_backend, - runtime_mounts=runtime_mounts, - candidate_overlay=candidate_overlay, - overlay_digest=overlay_digest, - ) - per_arm: dict[str, list[dict[str, Any]]] = {a: [] for a in args.arms} - cells = [(run_idx, arm) for run_idx in range(args.runs) for arm in args.arms] + asset_snapshot_errors[str(_task["id"])] = exc - def announce(run_idx: int, arm: str) -> None: - nonlocal started_cells - started_cells += 1 - print( - f"[{task['id']}][{arm}][run {run_idx}] starting " - f"({started_cells}/{total_cells}, {(time.monotonic() - sweep_started) / 60:.0f}m elapsed)" + reuse_source = args.reuse_results.expanduser().resolve() if args.reuse_results is not None else None + reusable_rows: dict[tuple[str, str, int], dict[str, Any]] = {} + if reuse_source is not None: + if reuse_source == out_dir.resolve(): + parser.error("--reuse-results cannot be this sweep's --out directory") + raise AssertionError("ArgumentParser.error() returned unexpectedly") + results_file = reuse_source / "results.jsonl" + if results_file.is_symlink() or not results_file.is_file(): + parser.error("--reuse-results must contain a regular results.jsonl") + raise AssertionError("ArgumentParser.error() returned unexpectedly") + reusable_rows = select_reusable_comparator_rows( + load_result_rows(results_file), + expected=_comparator_reuse_expectation( + args=args, + tasks=tasks, + task_bindings=task_bindings, + oracle_snapshots=oracle_snapshots, + asset_snapshots=asset_snapshots, + sandbox_backend=sandbox_backend, + ce_plugin_snapshot=ce_plugin_snapshot, + ), + ) + # Keep one paid cell per incumbent arm. A generation that reuses an + # arm end to end measures nothing about today's environment, and + # arm_health would then be reading last week's health. + # One cell per arm is the cheapest thing that keeps the canary real. + incumbent_arms = [arm for arm in args.arms if arm not in candidate_arms] + for arm in incumbent_arms: + dropped = drop_canary_reuse_key(reusable_rows, arm=arm, tasks=tasks, runs=args.runs) + if dropped is not None: + print( + f"reuse-results: keeping one paid {arm} cell " + f"({dropped[0]} run {dropped[2]}) so incumbent health is measured this sweep" + ) + print( + f"reuse-results {reuse_source}: {len(reusable_rows)} comparator " + f"cell(s) match this sweep; candidate arms always run" + ) + graph_env = GraphBuildEnv( + trees=Path(trees), + task_asset_cache=task_asset_cache, + claude_bin=args.claude_bin, + bwrap_bin=bwrap_bin, + sandbox_backend=sandbox_backend, + runtime_mounts=runtime_mounts, + clone_templates={}, + clone_template_errors={}, + graph_snapshots={}, + graph_snapshot_errors={}, + ) + graph_prefetch: GraphPrefetch | None = None + sweep_rows = list(zip(tasks, task_bindings, oracle_snapshots, strict=True)) + + def _join_graph_prefetch() -> None: + nonlocal graph_prefetch + if graph_prefetch is not None: + graph_prefetch.join() + graph_prefetch = None + + try: + for index, (task, task_binding, oracle_snapshot) in enumerate(sweep_rows): + if outage_tripped or cancel_event.is_set(): + break + repo = Path(task_binding["repo_identity"]) + task_sha = task_binding["resolved_sha"] + per_arm: dict[str, list[dict[str, Any]]] = {a: [] for a in args.arms} + planned = [(run_idx, arm) for run_idx in range(args.runs) for arm in args.arms] + reused_records: list[tuple[int, str, dict[str, Any]]] = [] + paid_cells: list[tuple[int, str]] = [] + for run_idx, arm in planned: + prior = reusable_rows.get((task["id"], arm, run_idx)) + if prior is None or reuse_source is None: + paid_cells.append((run_idx, arm)) + continue + try: + reused_records.append( + ( + run_idx, + arm, + materialize_reused_row(prior, source_dir=reuse_source, dest_dir=out_dir), + ) + ) + except (OSError, SandboxError, ValueError) as exc: + print( + f"[{task['id']}][{arm}][run {run_idx}] comparator reuse " + f"failed ({exc}); running a paid cell" + ) + paid_cells.append((run_idx, arm)) + + asset_snapshot = asset_snapshots.get(str(task["id"])) + asset_snapshot_error: BaseException | None = asset_snapshot_errors.get(str(task["id"])) + graph_key = (str(repo), task_sha) + if graph_prefetch is not None and graph_prefetch.key == graph_key: + _join_graph_prefetch() + if paid_cells: + ensure_task_graph( + task=task, repo=repo, task_sha=task_sha, graph_key=graph_key, env=graph_env + ) + graph_snapshot = graph_env.graph_snapshots.get(graph_key) + graph_snapshot_error = graph_env.graph_snapshot_errors.get(graph_key) + clone_template, template_head = graph_env.clone_templates.get(graph_key, (None, None)) + cell_context = TaskCellContext( + task=task, + oracle_snapshot=oracle_snapshot, + repo=repo, + task_sha=task_sha, + graph_snapshot=graph_snapshot, + graph_snapshot_error=graph_snapshot_error, + asset_snapshot=asset_snapshot, + asset_snapshot_error=asset_snapshot_error, + args=args, + out_dir=out_dir, + ce_plugin_snapshot=ce_plugin_snapshot, + trees_dir=Path(trees), + bwrap_bin=bwrap_bin, + sandbox_backend=sandbox_backend, + runtime_mounts=runtime_mounts, + candidate_overlay=candidate_overlay, + overlay_digest=overlay_digest, + clone_template=clone_template, + sanitized_head=template_head, ) - def keep(run_idx: int, arm: str, record: dict[str, Any]) -> None: - per_arm[arm].append(record) - with results_path.open("a") as fh: - # Redact any API token a session-error stderr_tail echoed - # into error_detail before it enters the uploaded - # results.jsonl artifact (transcripts are redacted; this - # sink was not). - fh.write(redact_text(json.dumps(record), credential_secrets(args)) + "\n") - print(cell_progress_line(task["id"], arm, run_idx, record)) - failure = cell_failure_detail_line(task["id"], arm, run_idx, record, credential_secrets(args)) - if failure: - print(failure) + def announce(run_idx: int, arm: str) -> None: + nonlocal started_cells + started_cells += 1 + print( + f"[{task['id']}][{arm}][run {run_idx}] starting " + f"({started_cells}/{total_cells}, {(time.monotonic() - sweep_started) / 60:.0f}m elapsed)" + ) - outage_streak, outage_tripped = sweep_task_cells( - cells, - workers=args.workers, - run=partial(run_cell, cell_context), - on_start=announce, - on_record=keep, - outage_streak=outage_streak, - outage_limit=args.outage_streak, - cancel_event=cancel_event, - ) - results[task["id"]] = {a: aggregate(rs) for a, rs in per_arm.items() if rs} + def keep(run_idx: int, arm: str, record: dict[str, Any]) -> None: + per_arm[arm].append(record) + with results_path.open("a") as fh: + # Redact any API token a session-error stderr_tail echoed + # into error_detail before it enters the uploaded + # results.jsonl artifact (transcripts are redacted; this + # sink was not). + fh.write(redact_text(json.dumps(record), credential_secrets(args)) + "\n") + print(cell_progress_line(task["id"], arm, run_idx, record)) + failure = cell_failure_detail_line(task["id"], arm, run_idx, record, credential_secrets(args)) + if failure: + print(failure) + + for run_idx, arm, record in reused_records: + started_cells += 1 + print( + f"[{task['id']}][{arm}][run {run_idx}] reused comparator " + f"({started_cells}/{total_cells}, {(time.monotonic() - sweep_started) / 60:.0f}m elapsed)" + ) + keep(run_idx, arm, record) + if paid_cells and graph_prefetch is None and not cancel_event.is_set(): + target = next_graph_prefetch_target( + [(later_task, later_binding) for later_task, later_binding, _ in sweep_rows[index + 1 :]], + arms=args.arms, + runs=args.runs, + reusable_rows=reusable_rows, + reuse_source=reuse_source, + ready_keys=graph_env.ready_keys(), + ) + if target is not None: + later_task, later_binding, later_key = target + graph_prefetch = prefetch_next_graph( + task=later_task, + binding=later_binding, + graph_key=later_key, + env=graph_env, + cancel_event=cancel_event, + ) + # A reused success is evidence the pipeline can produce a good row, + # so it resets the consecutive-failure count the same way a paid + # success does. Leaving reused rows out let a streak carry across + # them and trip on stale history. + for _run_idx, _arm, _record in reused_records: + outage_streak = systemic_outage_streak(_record.get("error_kind"), outage_streak) + outage_streak, outage_tripped = sweep_task_cells( + paid_cells, + workers=args.workers, + run=partial(run_cell, cell_context), + on_start=announce, + on_record=keep, + outage_streak=outage_streak, + outage_limit=args.outage_streak, + cancel_event=cancel_event, + ) + results[task["id"]] = {a: aggregate(rs) for a, rs in per_arm.items() if rs} + finally: + _join_graph_prefetch() selection_report = [ "## Run provenance", @@ -1858,10 +2681,12 @@ def _run_sweep( selection_report.append( f"Compound Engineering plugin: `{ce_plugin_snapshot.version}` (`{ce_plugin_snapshot.manifest_digest}`)" ) - if cancel_event.is_set(): - selection_report.append("Sweep cancelled: partial evidence; promotion is disabled.") - elif outage_tripped: + # outage first: the breaker now sets cancel_event to stop in-flight background + # work, so testing cancellation first would relabel every outage a cancellation. + if outage_tripped: selection_report.append("Sweep aborted: partial evidence; promotion is disabled.") + elif cancel_event.is_set(): + selection_report.append("Sweep cancelled: partial evidence; promotion is disabled.") report = render_report(results) + "\n\n" + "\n".join(selection_report) + "\n" (out_dir / "report.md").write_text(report) if candidate_arms: @@ -1894,26 +2719,22 @@ def _run_sweep( } (out_dir / "promotion.json").write_text(json.dumps(promotion, indent=2) + "\n") print(f"\n{report}\n\nWritten to {out_dir}/") - # A reviewer may legitimately match none of a difficult hidden corpus; - # unlike an implementation arm, zero exact resolutions is quality signal, - # not proof that the harness failed. - broken_incumbents = broken_incumbent_arms(results, set(CANDIDATE_ARMS.values()) - {"review"}) - if broken_incumbents: - # Fail loudly rather than let a broken environment read as a quiet - # "no promotion, incumbent stands." - print( - f"[harness-health] incumbent arm(s) {', '.join(broken_incumbents)} resolved zero " - "tasks across every valid run — this looks like an environment/harness failure, " - "not a normal candidate miss. See the errors column in report.md and error_detail " - "in results.jsonl. Exiting non-zero rather than reporting a quiet no-promotion." - ) - raise SystemExit(1) - if cancel_event.is_set(): - raise SystemExit(130) + # Health is judged on whether fresh attempts EXECUTED and produced usable + # evidence - never on how many tasks they resolved. Review arms used to be + # excluded here because "resolved zero" is quality signal for a reviewer + # facing a hard corpus; with the inference corrected they are included + # again, which is what lets an all-artifacts-empty run be caught at all. + # ce_review is named explicitly: it is a comparator, not a candidate, so it + # is absent from CANDIDATE_ARMS and would otherwise go unclassified. + enforce_measurement_health(results, set(CANDIDATE_ARMS.values()) | {"review", "ce_review"}) if outage_tripped: # Non-zero exit so a driver (evolve.py) treats the partial benchmark as a # failed run and halts instead of proposing from outage-truncated evidence. + # Checked before cancel_event because the breaker sets it (see above), and + # an outage must keep exit 1 rather than becoming the 130 of a Ctrl-C. raise SystemExit(1) + if cancel_event.is_set(): + raise SystemExit(130) if __name__ == "__main__": diff --git a/eval/workflow_bench/runner_artifacts.py b/eval/workflow_bench/runner_artifacts.py index 7f05e14b1..3780b9e46 100644 --- a/eval/workflow_bench/runner_artifacts.py +++ b/eval/workflow_bench/runner_artifacts.py @@ -217,11 +217,25 @@ def enforce_phase_workspace( worktree: Path, before: dict[str, str], *, - allowed_artifact: Path, + allowed_artifact: Path | None, ) -> None: - """Require a phase to change only its one explicit workspace artifact.""" + """Require a phase to change only its one explicit workspace artifact. + + ``allowed_artifact=None`` is the stricter contract: the phase must leave + the workspace byte-identical. That is what a review phase whose artifact + lives outside the workspace has to satisfy — there is nothing in there it + is entitled to touch. + """ root = worktree.expanduser().absolute() + if allowed_artifact is None: + after = workspace_snapshot(root) + changed = sorted( + path for path in before.keys() | after.keys() if before.get(path) != after.get(path) + ) + if changed: + raise ValueError(f"phase changed the read-only workspace: {', '.join(changed[:5])}") + return artifact = allowed_artifact.expanduser().absolute() try: relative = PurePosixPath(artifact.relative_to(root).as_posix()) @@ -325,6 +339,65 @@ def new_plan_doc(worktree: Path, before: dict[Path, str]) -> Path: return changed[0] +def _assert_self_contained_git_objects(clone: Path) -> None: + """Refuse clones that share pack/object bytes with another repository.""" + + alternates = clone / ".git" / "objects" / "info" / "alternates" + if alternates.exists(): + raise RuntimeError(f"clone unexpectedly has an external object alternate: {alternates}") + objects = clone / ".git" / "objects" + if not objects.is_dir(): + raise RuntimeError(f"clone is missing a git object store: {clone}") + for obj in objects.rglob("*"): + if obj.is_file() and obj.stat().st_nlink > 1: + raise RuntimeError(f"clone object is hardlinked to host storage: {obj}") + + +def copy_isolated_tree(source: Path, parent: Path) -> Path: + """Copy a sanitized clone without sharing git objects or a ref namespace. + + ``git clone --no-local`` of GitNexus plus ``sanitize_clone_for_hidden_oracles`` + (repack/prune/fsck) is minutes per cell. After sanitization the snapshot is + one parentless commit; copying that tree is the isolation boundary the + contamination bug actually required (a private ``.git``), not a second + fetch of full history. Prefer ``cp --reflink=auto`` so XFS/btrfs pay COW; + fall back to a full copy on filesystems that cannot reflink. + """ + + try: + source_meta = source.expanduser().lstat() + except OSError as exc: + raise RuntimeError(f"clone template is unavailable: {source}: {exc}") from exc + if stat.S_ISLNK(source_meta.st_mode) or not stat.S_ISDIR(source_meta.st_mode): + raise RuntimeError(f"clone template must be a real directory: {source}") + source = source.expanduser().resolve() + target = Path(tempfile.mkdtemp(prefix="wfbench-", dir=parent)) + target.rmdir() + try: + copied = run_managed( + ["cp", "-a", "--reflink=auto", str(source), str(target)], + timeout=600, + ) + if not copied.ok: + # The fallback is for a filesystem that cannot reflink, which shows + # up as a normal nonzero exit. A cancellation or timeout is reported + # the same way (run_managed returns it rather than raising), and + # copytree cannot be cancelled — so falling back there makes the + # outage breaker wait out the full copy it set the event to avoid. + if copied.state != "exited": + raise ManagedProcessError(["cp", "-a", "--reflink=auto", str(source), str(target)], copied) + shutil.copytree(source, target, symlinks=True, copy_function=shutil.copy2) + _assert_self_contained_git_objects(target) + return target + except BaseException as primary: + if target.exists(): + try: + shutil.rmtree(target) + except OSError as cleanup: + primary.add_note(f"clone copy cleanup also failed: {type(cleanup).__name__}: {cleanup}") + raise + + def make_worktree(repo: Path, ref: str, parent: Path) -> Path: """Create a self-contained clone per benchmark arm.""" @@ -344,12 +417,7 @@ def make_worktree(repo: Path, ref: str, parent: Path) -> Path: ], timeout=600, ) - alternates = target / ".git" / "objects" / "info" / "alternates" - if alternates.exists(): - raise RuntimeError(f"clone unexpectedly has an external object alternate: {alternates}") - for obj in (target / ".git" / "objects").rglob("*"): - if obj.is_file() and obj.stat().st_nlink > 1: - raise RuntimeError(f"clone object is hardlinked to host storage: {obj}") + _assert_self_contained_git_objects(target) for candidate in (ref, f"origin/{ref}"): proc = run_managed( ["git", "-C", str(target), "checkout", "--detach", "--quiet", candidate], diff --git a/eval/workflow_bench/runner_sessions.py b/eval/workflow_bench/runner_sessions.py index 9daef00b4..053d5e48d 100644 --- a/eval/workflow_bench/runner_sessions.py +++ b/eval/workflow_bench/runner_sessions.py @@ -57,6 +57,24 @@ MAX_PROGRESS_PENDING = 256 MAX_PROGRESS_TOOL_ID_CHARS = 256 MAX_TOOL_PREVIEW_CHARS = 800 _SAFE_TOOL_NAME = re.compile(r"[A-Za-z0-9._:-]{1,64}") +_GHA_WORKFLOW_COMMAND = re.compile(r"(^|[\n\r])::") +_GHA_HASH_COMMAND = re.compile(r"##\[") +_GHA_COMPILER_ANNOTATION = re.compile(r"\((\d+),(\d+)\):\s+error\b", re.IGNORECASE) + + +def neutralize_ci_log_text(text: str) -> str: + """Stop GitHub Actions from promoting tool output into check annotations. + + Run 33962002890 logged in-sandbox ``tsc`` failures as + ``file.ts(line,col): error TS2307``, which Actions parsed as workflow + annotations on ``.github``. The same parser treats ``::error::`` and + ``##[error]`` as commands. Progress previews are evidence, not CI + signaling, so rewrite those forms before they hit the job log. + """ + + text = _GHA_WORKFLOW_COMMAND.sub(r"\1[:]", text) + text = _GHA_HASH_COMMAND.sub("# [", text) + return _GHA_COMPILER_ANNOTATION.sub(r"(\1,\2): compiler-error", text) def _safe_tool_name(value: Any) -> str: @@ -177,7 +195,9 @@ class SessionProgress: def _say(self, message: str) -> None: # Queue only: the stdout drain thread calls observe() and must not # block on a full log pipe (process_control.stdout_observer contract). - self._pending_messages.append(f"[{self.label} {self._elapsed()}] {message}") + self._pending_messages.append( + neutralize_ci_log_text(f"[{self.label} {self._elapsed()}] {message}") + ) self._last_spoke = time.monotonic() def _emit_pending(self) -> None: diff --git a/eval/workflow_bench/sanitized_graph.py b/eval/workflow_bench/sanitized_graph.py index 3aea9e8c1..fc3d766a2 100644 --- a/eval/workflow_bench/sanitized_graph.py +++ b/eval/workflow_bench/sanitized_graph.py @@ -24,7 +24,7 @@ from .proposer_sandbox import ( build_sandbox_environment, prepare_sandbox, ) -from .runner_artifacts import make_worktree, remove_clone +from .runner_artifacts import copy_isolated_tree, make_worktree, remove_clone from .task_assets import TaskAssetCache, TaskAssetSnapshot, _is_harness_sandbox_copy GRAPH_ASSET_PATHS = ( @@ -360,14 +360,29 @@ def prepare_sanitized_graph( bwrap_bin: Path | str, runtime_mounts: Sequence[ReadOnlyMount], sandbox_backend: str = "bwrap", + clone_template: Path | None = None, + sanitized_head: str | None = None, ) -> SanitizedGraphSnapshot: - """Sanitize, index offline once, scrub, and freeze graph assets for all arms.""" + """Sanitize, index offline once, scrub, and freeze graph assets for all arms. + + When ``clone_template`` is an already-sanitized snapshot, this copies it + (the copy is scrubbed and indexed) so the template stays a clean cell + seed. Callers that already paid for ``make_worktree`` + sanitization + should pass that template rather than cloning GitNexus again. + """ validate_no_prebuilt_graph_assets(task) - seed = make_worktree(repo, resolved_sha, parent) + if clone_template is not None: + if not isinstance(sanitized_head, str) or not sanitized_head: + raise SandboxError("clone template requires the sanitized HEAD") + seed = copy_isolated_tree(clone_template, parent) + else: + seed = make_worktree(repo, resolved_sha, parent) + sanitized_head = None primary: BaseException | None = None try: - sanitized_head = sanitize_clone_for_hidden_oracles(seed) + if sanitized_head is None: + sanitized_head = sanitize_clone_for_hidden_oracles(seed) _scrub_source_references(seed) _neutralize_target_index_inputs(seed) with prepare_sandbox( diff --git a/eval/workflow_bench/session_durations.json b/eval/workflow_bench/session_durations.json new file mode 100644 index 000000000..dbb5e9004 --- /dev/null +++ b/eval/workflow_bench/session_durations.json @@ -0,0 +1,33 @@ +{ + "_provenance": "Actions run 33912693948 (2026-09-04), review profile, gen-0, workers=1. Artifact gitnexus-evolution-33912693948-1: gen-0/bench/results.jsonl and gen-0/proposer-session.json. Step wall from the Actions API.", + "_caveat": "Every cell in that run returned unusable evidence (32 review-evidence-invalid, 6 session-error, 3 skill-not-invoked); two hit the 5400s ceiling and it cost 653. Durations are real, but a run that resolves cleanly may sit lower. It is the only live artifact - the 2026-07-22 green run's has expired.", + "_order": "Submission order, deliberately unsorted: the model cycles these, so sorting would hand each task a uniform block and hide the variance being measured.", + "_duration_scope": "duration_s is the sum of the cell's Claude session durations (runner_sessions.py). It excludes the clone, graph materialize, asset staging, sandbox setup and teardown - those live in the residual below.", + "session_ceiling_s": 5400, + "proposer_duration_s": 344.7, + "cell_duration_s_by_arm": { + "candidate_review": [ + 2485.6, 1338.0, 3075.2, 702.5, 1240.4, 5400.0, 762.1, 342.9, 826.3, 489.7, 675.4, 337.8, 734.3 + ], + "ce_review": [ + 3744.6, 2140.4, 1418.4, 436.1, 653.3, 1022.3, 851.2, 1191.1, 902.1, 847.2, 502.6, 991.3, + 1222.7, 544.0 + ], + "review": [ + 5400.0, 2976.4, 1162.8, 627.1, 901.3, 436.9, 704.3, 963.4, 631.8, 627.9, 663.5, 662.4, 361.3, + 741.1 + ] + }, + "residual": { + "benchmark_step_wall_s": 54623, + "session_seconds": 51737.7, + "proposer_seconds": 344.7, + "unaccounted_s": 2540.6, + "cells": 41, + "unique_shas": 5, + "_note": "Everything the sweep spent outside the agent sessions: per-SHA sanitize and `analyze --pdg --index-only`, plus each cell's clone, materialize, staging, sandbox and teardown. That run predates clone templates and graph prefetch, so this is an upper bound for the current code. The split between per-SHA and per-cell is not recoverable from the artifact, so the model charges it per cell and serially, outside the pool - the pessimistic reading of an already-small term.", + "_split_assumption": "The residual mixes per-SHA graph setup with per-cell clone/sandbox/teardown and the artifact cannot separate them. The model charges it per SHA, not per cell, because only that direction refuses to credit a run for shrinking work it still performs: a weekly generation pays one arm instead of three but builds the same graphs. This overstates cold slightly and refuses to understate weekly. Replace with measured per-SHA and per-cell times when a run records them separately.", + "sha_overhead_s": 508.1 + }, + "_breaker": "Replaying this sample's error_kind sequence through today's systemic_outage_streak trips the outage breaker at cell 5 of 41 (DEFAULT_OUTAGE_STREAK=5). The source run executed all 41, so its runner did not break on this sequence. The durations stay valid as per-cell timings; what they cannot describe is a 54-cell sweep with this failure profile, because the current code would never run one." +} diff --git a/eval/workflow_bench/simulate_sweep.py b/eval/workflow_bench/simulate_sweep.py new file mode 100644 index 000000000..764460eea --- /dev/null +++ b/eval/workflow_bench/simulate_sweep.py @@ -0,0 +1,747 @@ +#!/usr/bin/env python3 +"""Run the real sweep scheduler against stub sessions and time it. + +``measure_evolution_cost`` is arithmetic: it predicts wall clock from a model of +what ``sweep_task_cells`` does. This runs the actual function - real threads, +the real wave barrier, the real outage breaker - and replaces only the paid +agent session with a sleep. If the two disagree, the model is wrong. + +Durations are the measured per-arm samples from ``session_durations.json`` +divided by ``--scale``, so a cell that really took 1416s takes ~0.28s here. The +shape is preserved deliberately: the median cell is 826s against a 5400s +ceiling, and that spread is the whole reason a barrier costs anything. Uniform +random sleeps would erase the effect under test. + +Schedulers, all consuming one identical seeded plan: + +``wave`` the shipped ``sweep_task_cells`` - fixed waves of ``workers``, a + barrier between them, one task at a time. +``fed`` a continuously fed pool per task (H1). Naive: no breaker, no graph + gating. Present to price the barrier alone. +``packed`` one pool across every task (H2). Naive, same caveat. +``faithful``H2 carrying the invariants the shipped scheduler actually holds: + a global submission order, in-order folding, the outage breaker, and + per-task graph readiness gating. This is the one to believe. + + python3 -m workflow_bench.simulate_sweep --compare --repeat 5 + python3 -m workflow_bench.simulate_sweep --breaker-fidelity +""" + +from __future__ import annotations + +import argparse +import json +import math +import random +import statistics +import subprocess +import sys +import threading +import time +from concurrent.futures import ThreadPoolExecutor +from dataclasses import dataclass, field +from typing import Any + +from . import runner +from .measure_evolution_cost import ( + CANDIDATE_ARM, + DURATIONS_BY_ARM, + REVIEW_ARMS, + REVIEW_TASKS, + SHA_OVERHEAD_SECONDS, + _read, + expected_task_seconds, + review_tasks, +) + +DEFAULT_SCALE = 5000.0 +# --contention-sweep measures both of these regardless of --workers, so the +# window has to be valid for the LARGEST of them, not for the parsed value. +CONTENTION_WORKERS = (3, 6) +SYSTEMIC_KIND = "session-error" + +# A cell is mostly a model session waiting on the network, but its tool calls - +# git, vitest, analyze - burn real CPU in real subprocesses. Sleeping threads +# model the wait and nothing else, so every speedup measured that way is an +# upper bound. This burns WORK, not wall clock: a fixed number of sha256 rounds +# in a subprocess, which takes longer when cores are contended. That is the +# effect under test, and it has to be a subprocess - Python threads burning +# Python would measure the GIL rather than the machine. +_BURN_SRC = ( + "import hashlib,sys\n" + "n=int(sys.argv[1]); b=b'x'*4096; h=hashlib.sha256()\n" + "for _ in range(n): h.update(b)\n" + "sys.stdout.write(h.hexdigest()[:8])\n" +) + + +def calibrate_burn(probe_rounds: int = 400_000) -> float: + """sha256 rounds per second, one uncontended subprocess. Measured, not assumed.""" + + started = time.monotonic() + subprocess.run( + [sys.executable, "-c", _BURN_SRC, str(probe_rounds)], + check=True, + capture_output=True, + ) + return probe_rounds / (time.monotonic() - started) + + +def _execute_cell(cell: Cell, cpu_fraction: float, burn_rate: float) -> None: + """The stub session: wait for the API, then do the tool-call work.""" + + if cpu_fraction <= 0: + time.sleep(cell.seconds) + return + time.sleep(cell.seconds * (1.0 - cpu_fraction)) + rounds = int(cell.seconds * cpu_fraction * burn_rate) + if rounds > 0: + subprocess.run( + [sys.executable, "-c", _BURN_SRC, str(rounds)], check=True, capture_output=True + ) + + +@dataclass(frozen=True) +class Cell: + task: int + run: int + arm: str + seconds: float + systemic: bool = False + + +@dataclass +class Outcome: + wall_s: float + executed: int + tripped_at: int | None = None + folded: list[int] = field(default_factory=list) + + +def build_plan( + *, + task_count: int, + runs: int, + arms: tuple[str, ...], + scale: float, + seed: int, + fail_from: int | None = None, +) -> list[list[Cell]]: + """Per-task cells in submission order, with durations drawn once. + + Shared by every scheduler so a comparison cannot be an artifact of one of + them drawing luckier cells. ``fail_from`` marks every cell at or after that + global index systemic, which is what the breaker-fidelity mode needs. + """ + + rng = random.Random(seed) + plan: list[list[Cell]] = [] + index = 0 + for task in range(task_count): + cells: list[Cell] = [] + for run_idx in range(runs): + for arm in arms: + sample = DURATIONS_BY_ARM[arm] + cells.append( + Cell( + task=task, + run=run_idx, + arm=arm, + seconds=sample[rng.randrange(len(sample))] / scale, + systemic=fail_from is not None and index >= fail_from, + ) + ) + index += 1 + plan.append(cells) + return plan + + +def _flatten(plan: list[list[Cell]]) -> list[Cell]: + return [cell for cells in plan for cell in cells] + + +def _record(cell: Cell) -> dict[str, Any]: + kind = SYSTEMIC_KIND if cell.systemic else None + return { + "run": cell.run, + "arm": cell.arm, + "ok": not cell.systemic, + "resolved": not cell.systemic, + "error_kind": kind, + "review_evidence_valid": not cell.systemic, + } + + +def _graph_builder( + ready: list[threading.Event], graph_seconds: float, stop: threading.Event +) -> threading.Thread: + """One graph at a time, in task order - they are CPU and IO heavy.""" + + def build() -> None: + for event in ready: + if stop.is_set(): + return + time.sleep(graph_seconds) + event.set() + + thread = threading.Thread(target=build, name="graph-builder", daemon=True) + thread.start() + return thread + + +def run_wave( + plan: list[list[Cell]], + workers: int, + *, + outage_limit: int, + graph_seconds: float, + cpu_fraction: float = 0.0, + burn_rate: float = 0.0, +) -> Outcome: + """The shipped scheduler, driven for real, task after task.""" + + ready = [threading.Event() for _ in plan] + stop = threading.Event() + _graph_builder(ready, graph_seconds, stop) + executed = 0 + lock = threading.Lock() + streak = 0 + tripped_at: int | None = None + folded: list[int] = [] + base = 0 + + started = time.monotonic() + for task, cells in enumerate(plan): + ready[task].wait() + by_key = {(c.run, c.arm): c for c in cells} + + def fake_run(run_idx: int, arm: str) -> dict[str, Any]: + nonlocal executed + cell = by_key[(run_idx, arm)] + _execute_cell(cell, cpu_fraction, burn_rate) + with lock: + executed += 1 + return _record(cell) + + order = {(c.run, c.arm): base + i for i, c in enumerate(cells)} + + def on_record(run_idx: int, arm: str, rec: dict[str, Any]) -> None: + # Mirror the breaker's own evaluation so the reported trip point is + # the cell that crossed the limit, not merely the last one folded - + # sweep_task_cells folds a whole wave before it evaluates. + nonlocal streak, tripped_at + index = order[(run_idx, arm)] + folded.append(index) + streak = runner.systemic_outage_streak(rec["error_kind"], streak) + if outage_limit and streak >= outage_limit and tripped_at is None: + tripped_at = index + + streak, tripped = runner.sweep_task_cells( + [(c.run, c.arm) for c in cells], + workers=workers, + run=fake_run, + on_start=lambda *_: None, + on_record=on_record, + outage_streak=streak, + outage_limit=outage_limit, + ) + base += len(cells) + if tripped: + break + stop.set() + return Outcome(wall_s=time.monotonic() - started, executed=executed, tripped_at=tripped_at, folded=folded) + + +def _drain_naive(cells: list[Cell], workers: int) -> int: + with ThreadPoolExecutor(max_workers=workers) as pool: + list(pool.map(lambda c: time.sleep(c.seconds), cells)) + return len(cells) + + +def run_fed(plan: list[list[Cell]], workers: int, *, outage_limit: int, graph_seconds: float) -> Outcome: + """H1 without invariants: fed pool per task. Prices the barrier alone. + + Graph building is deliberately identical to ``run_wave`` - the same + background builder, started before the clock - because that is what makes + the claim in the first line true. Sleeping ``graph_seconds`` serially before + each task instead, as this did, charged fed for overlap that wave gets for + free: the wave builder prepares task N+1 while task N's cells run. The + fed-versus-wave delta then mixed the loss of that overlap into what was + reported as the price of the barrier. + """ + + ready = [threading.Event() for _ in plan] + stop = threading.Event() + _graph_builder(ready, graph_seconds, stop) + executed = 0 + started = time.monotonic() + for task, cells in enumerate(plan): + ready[task].wait() + executed += _drain_naive(cells, workers) + return Outcome(wall_s=time.monotonic() - started, executed=executed) + + +def run_packed(plan: list[list[Cell]], workers: int, *, outage_limit: int, graph_seconds: float) -> Outcome: + """H2 without invariants. Upper bound, not a design.""" + + started = time.monotonic() + time.sleep(graph_seconds) + executed = _drain_naive(_flatten(plan), workers) + return Outcome(wall_s=time.monotonic() - started, executed=executed) + + +def run_faithful( + plan: list[list[Cell]], + workers: int, + *, + outage_limit: int, + graph_seconds: float, + window: int | None = None, + cpu_fraction: float = 0.0, + burn_rate: float = 0.0, +) -> Outcome: + """H2 carrying the invariants the shipped scheduler holds. + + Global submission order is task-major, run-major, arm-minor - the same total + order the wave scheduler folds in, just continued across task boundaries. A + folder walks results in exactly that order, so "consecutive systemic + failures" keeps its meaning; the breaker trips on the same logical cell it + would have in waves. Cells already in flight when it trips are the overrun, + bounded by ``workers - 1`` exactly as the wave docstring promises. + + A task's cells are not submitted until its graph is ready, which is what + makes this a schedule rather than a wish: the graph builder is serial, so + packing cannot outrun it. + + ``window`` is the design question. Queue every cell at once and workers race + far ahead of the fold pointer, so a breaker trip has already paid for cells + nobody has looked at - measured at 5 against a bound of 2. Holding + submission to ``window`` cells beyond the fold point caps the overrun at + ``window - 1``, which is the wave's own ``workers - 1`` bound when the two + are equal, while still packing across task boundaries. Defaults to whatever + ``runner.sweep_packed_cells`` defaults to, so a run that names no window + compares the shipped policy rather than a more tightly queued prototype. + """ + + if window is None: + window = max(workers * runner.PACKED_WINDOW_MULTIPLIER, workers) + if window < workers: + # Same rule sweep_packed_cells enforces. Without it a window below 1 + # never lets the producer past its own gate and the run hangs. + raise ValueError("window must be at least workers, or the pool starves") + + cells = _flatten(plan) + ready = [threading.Event() for _ in plan] + stop = threading.Event() + _graph_builder(ready, graph_seconds, stop) + + results: list[dict[str, Any] | None] = [None] * len(cells) + executed = 0 + lock = threading.Lock() + halt = threading.Event() + + def work(index: int) -> None: + nonlocal executed + if halt.is_set(): + return + cell = cells[index] + _execute_cell(cell, cpu_fraction, burn_rate) + with lock: + executed += 1 + results[index] = _record(cell) + + gate = threading.Condition() + fold_pointer = 0 + futures: list[Any] = [] + producer_done = threading.Event() + + started = time.monotonic() + pool = ThreadPoolExecutor(max_workers=workers) + + def produce() -> None: + submitted = 0 + for task, task_cells in enumerate(plan): + ready[task].wait() + for _ in task_cells: + with gate: + while submitted - fold_pointer >= window and not halt.is_set(): + gate.wait(timeout=0.5) + if halt.is_set(): + producer_done.set() + return + futures.append(pool.submit(work, submitted)) + submitted += 1 + gate.notify_all() + producer_done.set() + + producer = threading.Thread(target=produce, name="cell-producer", daemon=True) + producer.start() + + streak = 0 + tripped_at: int | None = None + folded: list[int] = [] + try: + index = 0 + while True: + with gate: + while index >= len(futures) and not producer_done.is_set(): + gate.wait(timeout=0.5) + if index >= len(futures): + break + future = futures[index] + future.result() + record = results[index] + if record is not None: + folded.append(index) + streak = runner.systemic_outage_streak(record["error_kind"], streak) + if outage_limit and streak >= outage_limit: + tripped_at = index + halt.set() + with gate: + gate.notify_all() + for pending in futures[index + 1 :]: + pending.cancel() + break + index += 1 + with gate: + fold_pointer = index + gate.notify_all() + finally: + halt.set() + with gate: + gate.notify_all() + stop.set() + producer.join(timeout=5) + pool.shutdown(wait=True) + return Outcome( + wall_s=time.monotonic() - started, executed=executed, tripped_at=tripped_at, folded=folded + ) + + +def run_production_packed( + plan: list[list[Cell]], + workers: int, + *, + outage_limit: int, + graph_seconds: float, + cpu_fraction: float = 0.0, + burn_rate: float = 0.0, + window: int | None = None, +) -> Outcome: + """Drive the REAL runner.sweep_packed_cells, not a prototype of it. + + Same relationship run_wave has to sweep_task_cells: only the paid session is + stubbed. If this disagrees with the faithful prototype, the shipped function + is what is wrong. + """ + + cells = _flatten(plan) + by_key = {(f"t{c.task}", c.run, c.arm): c for c in cells} + order = {(f"t{c.task}", c.run, c.arm): i for i, c in enumerate(cells)} + ready = [threading.Event() for _ in plan] + stop = threading.Event() + _graph_builder(ready, graph_seconds, stop) + + executed = 0 + lock = threading.Lock() + folded: list[int] = [] + tripped_at: int | None = None + streak_seen = {"streak": 0} + + def run_cell(task_id: str, run_idx: int, arm: str) -> dict[str, Any]: + nonlocal executed + cell = by_key[(task_id, run_idx, arm)] + _execute_cell(cell, cpu_fraction, burn_rate) + with lock: + executed += 1 + return _record(cell) + + def on_record(task_id: str, run_idx: int, arm: str, rec: dict[str, Any]) -> None: + nonlocal tripped_at + index = order[(task_id, run_idx, arm)] + folded.append(index) + streak_seen["streak"] = runner.systemic_outage_streak(rec["error_kind"], streak_seen["streak"]) + if outage_limit and streak_seen["streak"] >= outage_limit and tripped_at is None: + tripped_at = index + + def await_ready(task_id: str) -> bool: + ready[int(task_id[1:])].wait() + return True + + started = time.monotonic() + runner.sweep_packed_cells( + [(f"t{c.task}", c.run, c.arm) for c in cells], + workers=workers, + run=run_cell, + on_start=lambda *_: None, + on_record=on_record, + outage_streak=0, + outage_limit=outage_limit, + window=window, + await_ready=await_ready, + ) + wall = time.monotonic() - started + stop.set() + return Outcome(wall_s=wall, executed=executed, tripped_at=tripped_at, folded=folded) + + +SCHEDULERS = { + "wave": run_wave, + "fed": run_fed, + "packed": run_packed, + "faithful": run_faithful, + "production": run_production_packed, +} + + +def _window_kwargs(name: str, window: int | None) -> dict[str, int]: + """``--window`` only means anything to the two schedulers that hold one.""" + + return {"window": window} if window is not None and name in ("faithful", "production") else {} + + +def _plan_args(args: argparse.Namespace, weekly: bool, seed: int, fail_from: int | None = None): + arms = (CANDIDATE_ARM,) if weekly else REVIEW_ARMS + return { + "task_count": len(review_tasks(_read(REVIEW_TASKS))), + "runs": args.runs, + "arms": arms, + "scale": args.scale, + "seed": seed, + "fail_from": fail_from, + }, arms + + +def breaker_fidelity(args: argparse.Namespace) -> list[dict[str, Any]]: + """Does packing still trip where waves trip, and overrun no further?""" + + rows: list[dict[str, Any]] = [] + limit = runner.DEFAULT_OUTAGE_STREAK + window = args.window if args.window is not None else max( + args.workers * runner.PACKED_WINDOW_MULTIPLIER, args.workers + ) + for fail_from in (0, 4, 12): + kwargs, _arms = _plan_args(args, weekly=False, seed=args.seed, fail_from=fail_from) + plan = build_plan(**kwargs) + total = sum(len(c) for c in plan) + row: dict[str, Any] = { + "fail_from": fail_from, "limit": limit, "total_cells": total, "window": window + } + for name in ("wave", "faithful", "production"): + out = SCHEDULERS[name]( + plan, args.workers, outage_limit=limit, graph_seconds=args.graph_seconds, + **_window_kwargs(name, window), + ) + row[name] = { + "tripped_at": out.tripped_at, + "executed": out.executed, + "overrun": out.executed - (out.tripped_at + 1) if out.tripped_at is not None else None, + } + row["same_trip_point"] = ( + row["wave"]["tripped_at"] == row["faithful"]["tripped_at"] == row["production"]["tripped_at"] + ) + # The producer holds submission to ``window`` cells beyond the fold + # pointer, so at most ``window - 1`` cells past the tripping one can + # already be in flight. At ``window == workers`` that is exactly the + # wave scheduler's own ``workers - 1`` bound. + row["overrun_within_bound"] = ( + row["production"]["overrun"] is not None + and row["production"]["overrun"] <= window - 1 + ) + rows.append(row) + return rows + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--workers", type=int, default=3) + parser.add_argument("--scale", type=float, default=DEFAULT_SCALE) + parser.add_argument("--seed", type=int, default=1729) + parser.add_argument("--repeat", type=int, default=1) + parser.add_argument("--runs", type=int, default=3) + parser.add_argument("--scheduler", choices=sorted(SCHEDULERS), default="wave") + parser.add_argument("--compare", action="store_true") + parser.add_argument("--breaker-fidelity", action="store_true") + parser.add_argument("--window-sweep", action="store_true", help="wall clock vs breaker overrun") + parser.add_argument("--contention-sweep", action="store_true", help="does the gain survive real CPU?") + parser.add_argument( + "--window", + type=int, + default=None, + help="submission window for the packed schedulers; defaults to the shipped policy", + ) + parser.add_argument( + "--graph-seconds", + type=float, + default=None, + help="per-task graph build; defaults to the measured per-SHA overhead, scaled", + ) + args = parser.parse_args() + # Both are checked here rather than where they are used: a bad --scale + # divides by zero before anything runs, and a negative --graph-seconds + # kills the graph-builder thread, after which every scheduler waits on a + # readiness event nobody will ever set. + # NaN defeats every comparison it appears in, so "> 0" and ">= 0" both admit + # it and the failure surfaces far from the flag: NaN durations reach + # time.sleep in a worker or the graph thread and raise there, after which the + # schedulers wait forever on a readiness event nobody will set. Infinity is + # worse than a crash - it silently scales every duration to zero and the run + # reports a sweep that took no time. + if not math.isfinite(args.scale) or args.scale <= 0: + parser.error("--scale must be a finite positive number") + if args.graph_seconds is not None and (not math.isfinite(args.graph_seconds) or args.graph_seconds < 0): + parser.error("--graph-seconds must be a finite non-negative number") + # Counts are indexed or handed to a thread pool without further checking, so + # a zero turns into an IndexError on plans[0], a median over an empty + # sequence, or ThreadPoolExecutor's own error - none of which name the flag + # that caused them. + if args.workers < 1: + parser.error("--workers must be at least 1") + if args.repeat < 1: + parser.error("--repeat must be at least 1") + if args.runs < 1: + parser.error("--runs must be at least 1") + # run_faithful and sweep_packed_cells both refuse a window below the worker + # count - a smaller one starves the pool, because the producer waits for a + # fold pointer to pass a cell it was never allowed to submit. Enforcing it + # here turns an uncaught ValueError partway through a measurement into an + # argument error before anything runs. Checked against the largest worker + # count this invocation will actually use: --contention-sweep runs its own + # counts irrespective of --workers, so validating against --workers alone + # let the 3-worker measurements finish and then raised on the 6-worker one. + window_workers = args.workers + if args.contention_sweep: + window_workers = max(window_workers, max(CONTENTION_WORKERS)) + if args.window is not None and args.window < window_workers: + parser.error(f"--window must be at least the worker count ({window_workers}); a smaller window starves the pool") + if args.graph_seconds is None: + args.graph_seconds = SHA_OVERHEAD_SECONDS / args.scale + + if args.contention_sweep: + burn_rate = statistics.median(calibrate_burn() for _ in range(3)) + rows = [] + for cpu_fraction in (0.0, 0.25, 0.5): + for workers in CONTENTION_WORKERS: + plans = [ + build_plan(**_plan_args(args, False, args.seed + i)[0]) + for i in range(args.repeat) + ] + measured = {} + for name in ("wave", "faithful", "production"): + fn = SCHEDULERS[name] + measured[name] = statistics.median( + fn( + plan, + workers, + outage_limit=0, + graph_seconds=args.graph_seconds, + cpu_fraction=cpu_fraction, + burn_rate=burn_rate, + **_window_kwargs(name, args.window), + ).wall_s + for plan in plans + ) + serial = statistics.median( + sum(c.seconds for c in _flatten(plan)) for plan in plans + ) + rows.append( + { + "cpu_fraction": cpu_fraction, + "workers": workers, + "wave_s": round(measured["wave"], 2), + "faithful_s": round(measured["faithful"], 2), + "production_s": round(measured["production"], 2), + "packing_gain_pct": round( + (measured["faithful"] - measured["wave"]) / measured["wave"] * 100, 1 + ), + "wave_speedup": round(serial / measured["wave"], 2), + "faithful_speedup": round(serial / measured["faithful"], 2), + } + ) + print(json.dumps({"burn_rate": round(burn_rate), "nproc": __import__("os").cpu_count(), "rows": rows}, indent=2)) + return 0 + + if args.window_sweep: + total = len(review_tasks(_read(REVIEW_TASKS))) * args.runs * len(REVIEW_ARMS) + rows = [] + for window in (args.workers, args.workers * 2, args.workers * 4, total): + clean = [build_plan(**_plan_args(args, False, args.seed + i)[0]) for i in range(args.repeat)] + wall = statistics.median( + run_faithful( + p, args.workers, outage_limit=0, graph_seconds=args.graph_seconds, window=window + ).wall_s + for p in clean + ) + failing = build_plan(**_plan_args(args, weekly=False, seed=args.seed, fail_from=12)[0]) + trip = run_faithful( + failing, + args.workers, + outage_limit=runner.DEFAULT_OUTAGE_STREAK, + graph_seconds=args.graph_seconds, + window=window, + ) + rows.append( + { + "window": window, + "cold_wall_s": round(wall, 3), + "tripped_at": trip.tripped_at, + "executed": trip.executed, + "overrun_cells": trip.executed - (trip.tripped_at + 1) + if trip.tripped_at is not None + else None, + } + ) + print(json.dumps({"workers": args.workers, "rows": rows}, indent=2)) + return 0 + + if args.breaker_fidelity: + print( + json.dumps( + {"workers": args.workers, "graph_seconds": round(args.graph_seconds, 4), + "rows": breaker_fidelity(args)}, + indent=2, + ) + ) + return 0 + + names = sorted(SCHEDULERS) if args.compare else [args.scheduler] + rows: list[dict[str, Any]] = [] + for label, weekly in (("weekly", True), ("cold", False)): + plans = [] + for i in range(args.repeat): + kwargs, arms = _plan_args(args, weekly, args.seed + i) + plans.append(build_plan(**kwargs)) + serial = statistics.median(sum(c.seconds for c in _flatten(p)) for p in plans) + predicted = ( + len(plans[0]) + * expected_task_seconds(args.runs, arms, args.workers, fed_pool=False) + / args.scale + ) + for name in names: + observed = statistics.median( + SCHEDULERS[name]( + p, + args.workers, + outage_limit=0, + graph_seconds=args.graph_seconds, + **_window_kwargs(name, args.window), + ).wall_s + for p in plans + ) + rows.append( + { + "profile": label, + "scheduler": name, + "workers": args.workers, + "observed_s": round(observed, 3), + "wave_model_s": round(predicted, 3), + "serial_s": round(serial, 3), + "speedup_vs_serial": round(serial / observed, 3) if observed else None, + } + ) + print(json.dumps({"scale": args.scale, "repeat": args.repeat, "rows": rows}, indent=2)) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/eval/workflow_bench/tasks.review.scenarios.yaml b/eval/workflow_bench/tasks.review.scenarios.yaml index 8ff323ed1..4d68ae871 100644 --- a/eval/workflow_bench/tasks.review.scenarios.yaml +++ b/eval/workflow_bench/tasks.review.scenarios.yaml @@ -9,14 +9,17 @@ tasks: sandbox_copy: [eval/workflow_bench/review_cases/pr-2718.patch] setup: git apply --exclude='eval/workflow_bench/*' eval/workflow_bench/review_cases/pr-2718.patch && rm -rf eval/workflow_bench prompt: Review the historical snapshot of https://github.com/abhigyanpatwari/GitNexus/pull/2718. Report only actionable defects introduced by the local diff. - verify: test -s review-output.json + verify: test -s "$GITNEXUS_BENCH_REVIEW_OUTPUT" oracle: - command: test -s review-output.json + command: test -s "$GITNEXUS_BENCH_REVIEW_OUTPUT" files: [{ source: review-pr-2718-defect.labels.json, target: review-labels.json }] sandbox_dependencies: &deps - { source: node_modules, target: node_modules } - { source: gitnexus/node_modules, target: gitnexus/node_modules } - { source: gitnexus-shared/node_modules, target: gitnexus-shared/node_modules } + # Host-built types/JS. Historical clones have no dist/, so `tsc` in the + # read-only workspace otherwise reports TS2307/TS6379 (run 33962002890). + - { source: gitnexus-shared/dist, target: gitnexus-shared/dist } - <<: *review_case id: review-pr-2794-defect @@ -25,7 +28,7 @@ tasks: setup: git apply --exclude='eval/workflow_bench/*' eval/workflow_bench/review_cases/pr-2794.patch && rm -rf eval/workflow_bench prompt: Review the historical snapshot of https://github.com/abhigyanpatwari/GitNexus/pull/2794. Report only actionable defects introduced by the local diff. oracle: - command: test -s review-output.json + command: test -s "$GITNEXUS_BENCH_REVIEW_OUTPUT" files: [{ source: review-pr-2794-defect.labels.json, target: review-labels.json }] sandbox_dependencies: *deps @@ -36,7 +39,7 @@ tasks: setup: git apply --exclude='eval/workflow_bench/*' eval/workflow_bench/review_cases/pr-2108.patch && rm -rf eval/workflow_bench prompt: Review the historical snapshot of https://github.com/abhigyanpatwari/GitNexus/pull/2108. Report only actionable defects introduced by the local diff. oracle: - command: test -s review-output.json + command: test -s "$GITNEXUS_BENCH_REVIEW_OUTPUT" files: [{ source: review-pr-2108-defect.labels.json, target: review-labels.json }] sandbox_dependencies: *deps @@ -47,7 +50,7 @@ tasks: setup: git apply --exclude='eval/workflow_bench/*' eval/workflow_bench/review_cases/pr-2258.patch && rm -rf eval/workflow_bench prompt: Review the historical snapshot of https://github.com/abhigyanpatwari/GitNexus/pull/2258. Report only actionable defects introduced by the local diff. oracle: - command: test -s review-output.json + command: test -s "$GITNEXUS_BENCH_REVIEW_OUTPUT" files: [{ source: review-pr-2258-defect.labels.json, target: review-labels.json }] sandbox_dependencies: *deps @@ -59,7 +62,7 @@ tasks: setup: git apply --exclude='eval/workflow_bench/*' eval/workflow_bench/review_cases/pr-2258b.patch && rm -rf eval/workflow_bench prompt: Review this historical snapshot of https://github.com/abhigyanpatwari/GitNexus/pull/2258. Report only actionable defects introduced by the local diff. oracle: - command: test -s review-output.json + command: test -s "$GITNEXUS_BENCH_REVIEW_OUTPUT" files: [{ source: review-pr-2258-clean.labels.json, target: review-labels.json }] sandbox_dependencies: *deps @@ -71,6 +74,6 @@ tasks: setup: git apply --exclude='eval/workflow_bench/*' eval/workflow_bench/review_cases/pr-2773.patch && rm -rf eval/workflow_bench prompt: Review this historical snapshot of https://github.com/abhigyanpatwari/GitNexus/pull/2773. Report only actionable defects introduced by the local diff. oracle: - command: test -s review-output.json + command: test -s "$GITNEXUS_BENCH_REVIEW_OUTPUT" files: [{ source: review-pr-2773-clean.labels.json, target: review-labels.json }] sandbox_dependencies: *deps diff --git a/eval/workflow_bench/tasks.scenarios.yaml b/eval/workflow_bench/tasks.scenarios.yaml index af4755f88..59448ae4f 100644 --- a/eval/workflow_bench/tasks.scenarios.yaml +++ b/eval/workflow_bench/tasks.scenarios.yaml @@ -44,6 +44,11 @@ tasks: target: gitnexus/node_modules - source: gitnexus-shared/node_modules target: gitnexus-shared/node_modules + # Host-built types/JS. The clone has no dist/, and node_modules/gitnexus-shared + # is a relative symlink into that unbuilt tree — without this mount, in-sandbox + # `tsc --noEmit` / vitest fail with TS2307 / TS6379 (run 33962002890). + - source: gitnexus-shared/dist + target: gitnexus-shared/dist prompt: > Add -j as a short alias for --json on the gitnexus status command (gitnexus/src/cli/index.ts), and cover the alias with a unit test in diff --git a/gitnexus-claude-plugin/.claude-plugin/plugin.json b/gitnexus-claude-plugin/.claude-plugin/plugin.json index 13f937f59..6195fe2b4 100644 --- a/gitnexus-claude-plugin/.claude-plugin/plugin.json +++ b/gitnexus-claude-plugin/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "gitnexus", "description": "Code intelligence powered by a knowledge graph. Provides execution flow tracing, blast radius analysis, and augmented search across your codebase.", - "version": "1.6.11", + "version": "1.6.12", "author": { "name": "GitNexus" }, diff --git a/gitnexus-claude-plugin/.codex-plugin/plugin.json b/gitnexus-claude-plugin/.codex-plugin/plugin.json index 2e2b43e2e..3a51abd30 100644 --- a/gitnexus-claude-plugin/.codex-plugin/plugin.json +++ b/gitnexus-claude-plugin/.codex-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "gitnexus", "description": "Code intelligence powered by a knowledge graph. Provides execution flow tracing, blast radius analysis, and augmented search across your codebase.", - "version": "1.6.11", + "version": "1.6.12", "skills": "./skills", "mcpServers": "./.mcp.json", "hooks": "./hooks/hooks.json", diff --git a/gitnexus-claude-plugin/hooks/gitnexus-hook.js b/gitnexus-claude-plugin/hooks/gitnexus-hook.js index 238438455..b550f5e08 100644 --- a/gitnexus-claude-plugin/hooks/gitnexus-hook.js +++ b/gitnexus-claude-plugin/hooks/gitnexus-hook.js @@ -20,6 +20,7 @@ const { resolveUnixGuardTimeout, } = require('./hook-db-lock-probe.cjs'); const { formatAnalyzeCommand } = require('./resolve-analyze-cmd.cjs'); +const { resolveHookRepo } = require('./registry-query.cjs'); /** * Read JSON input from stdin synchronously. @@ -33,106 +34,10 @@ function readInput() { } } -/** - * Find the .gitnexus directory by walking up from startDir. - * Returns the path to .gitnexus/ or null if not found. - */ -function isGlobalRegistryDir(candidate) { - if ( - fs.existsSync(path.join(candidate, 'gitnexus.json')) || - fs.existsSync(path.join(candidate, 'meta.json')) - ) { - return false; - } - return ( - fs.existsSync(path.join(candidate, 'registry.json')) || - fs.existsSync(path.join(candidate, 'repos')) - ); -} +/* Registry-backed hooks resolve the source repo and storage path together. */ -/** - * Read the index metadata file, preferring `gitnexus.json` (current format) - * and falling back to the legacy `meta.json` mirror. Returns `null` if - * neither exists or parses. - */ -function readIndexMeta(gitNexusDir) { - try { - return JSON.parse(fs.readFileSync(path.join(gitNexusDir, 'gitnexus.json'), 'utf-8')); - } catch { - try { - return JSON.parse(fs.readFileSync(path.join(gitNexusDir, 'meta.json'), 'utf-8')); - } catch { - return null; - } - } -} - -/** - * Walk up from `startDir` looking for a non-registry `.gitnexus/` folder. - * Returns the path to `.gitnexus/` or null if not found within 5 levels. - */ -function walkForGitNexusDir(startDir) { - let dir = startDir; - for (let i = 0; i < 5; i++) { - const candidate = path.join(dir, '.gitnexus'); - if (fs.existsSync(candidate)) { - if (!isGlobalRegistryDir(candidate)) return candidate; - } - const parent = path.dirname(dir); - if (parent === dir) break; - dir = parent; - } - return null; -} - -/** - * Resolve the canonical (main) worktree root for `cwd`, when `cwd` is inside - * any git working tree — including a *linked* worktree created via - * `git worktree add`. Linked worktrees never contain `.gitnexus/`, so the - * upward walk from cwd alone misses the index. Returns null when `cwd` is - * not inside a git repo or `git` is not available. - * - * Implementation: `git rev-parse --git-common-dir` resolves to the canonical - * `.git/` directory (or `.git/worktrees/...` parent) that is shared across - * all linked worktrees. The canonical repo root is its parent directory. - */ -function findCanonicalRepoRoot(cwd) { - try { - const result = spawnSync('git', ['rev-parse', '--path-format=absolute', '--git-common-dir'], { - encoding: 'utf-8', - timeout: 2000, - cwd, - stdio: ['pipe', 'pipe', 'pipe'], - windowsHide: true, - }); - if (result.error || result.status !== 0) return null; - const commonDir = (result.stdout || '').trim(); - if (!commonDir || !path.isAbsolute(commonDir)) return null; - return path.dirname(commonDir); - } catch { - return null; - } -} - -function findGitNexusDir(startDir) { - const cwd = startDir || process.cwd(); - - // Fast path: the cwd is inside the canonical repo (most common case). - const fromCwd = walkForGitNexusDir(cwd); - if (fromCwd) return fromCwd; - - // Fallback: cwd may be inside a linked git worktree whose `.gitnexus/` - // only lives in the canonical repo root. Resolve the shared git dir - // and retry from there. - const canonicalRoot = findCanonicalRepoRoot(cwd); - if (canonicalRoot && canonicalRoot !== cwd) { - return walkForGitNexusDir(canonicalRoot); - } - return null; -} - -function hasGitNexusServerOwner(gitNexusDir) { - return hasGitNexusDbLockedByGitNexusServer(path.join(gitNexusDir, 'lbug'), process.pid); +function hasGitNexusServerOwner(lbugPath) { + return hasGitNexusDbLockedByGitNexusServer(lbugPath, process.pid); } /** @@ -406,11 +311,11 @@ function buildMcpQueryHint(pattern) { * ponytail: per-repo mtime marker, shared across concurrent sessions on the same * repo; add per-session dedup only if that sharing becomes a problem. */ -function shouldEmitMcpHint(gitNexusDir) { +function shouldEmitMcpHint(storagePath) { const raw = process.env.GITNEXUS_MCP_HINT_THROTTLE_MS; const windowMs = raw === undefined || raw === '' ? 600000 : Number(raw); if (!Number.isFinite(windowMs) || windowMs <= 0) return true; - const marker = path.join(gitNexusDir, '.mcp-hint-shown'); + const marker = path.join(storagePath, '.mcp-hint-shown'); try { if (Date.now() - fs.statSync(marker).mtimeMs < windowMs) return false; } catch { @@ -430,8 +335,6 @@ function shouldEmitMcpHint(gitNexusDir) { function handlePreToolUse(input) { const cwd = input.cwd || process.cwd(); if (!path.isAbsolute(cwd)) return; - const gitNexusDir = findGitNexusDir(cwd); - if (!gitNexusDir) return; const toolName = input.tool_name || ''; const toolInput = input.tool_input || {}; @@ -441,12 +344,18 @@ function handlePreToolUse(input) { const pattern = extractPattern(toolName, toolInput); if (!pattern || pattern.length < 3) return; + // Registry row first (persisted external storagePath wins). Local owned + // `.gitnexus` is only the fallback when no matching registry row exists. + const repo = resolveHookRepo(cwd); + if (!repo) return; + const storagePath = repo.storagePath; + // Acquire the per-repo slot BEFORE the DB-owner probe (#2163): the probe // itself spawns lsof/ps, so it must be bounded by the same ≤3-per-repo cap // as the augment, or concurrent sessions fan out unbounded probe // subprocesses. Keep the acquire right after the cheap guards above — // moving it earlier would churn slot files on tool calls that never probe. - const release = acquireHookSlot(gitNexusDir); + const release = acquireHookSlot(storagePath); if (!release) { // Normal skip path: all per-repo hook slots are held by concurrent // sessions. Stay silent for strict hook runners (issue #1913); surface @@ -459,7 +368,7 @@ function handlePreToolUse(input) { let result = ''; try { - if (hasGitNexusServerOwner(gitNexusDir)) { + if (hasGitNexusServerOwner(repo.lbugPath)) { // #2396: the MCP server holds the DB write lock, so a competing CLI // `augment` would only contend on it (LadybugDB is single-writer). But the // session that triggered this hook has the GitNexus MCP tools live — route @@ -470,7 +379,7 @@ function handlePreToolUse(input) { if (isDebugEnabled()) { process.stderr.write('[GitNexus] augment skipped: MCP server owns DB\n'); } - if (shouldEmitMcpHint(gitNexusDir)) { + if (shouldEmitMcpHint(storagePath)) { result = buildMcpQueryHint(pattern); } } else { @@ -496,7 +405,7 @@ function handlePreToolUse(input) { * Instead of spawning a full `gitnexus analyze` synchronously (which blocks * the agent for up to 120s and risks LadybugDB corruption on timeout), we do a * lightweight staleness check: compare `git rev-parse HEAD` against the - * lastCommit stored in `.gitnexus/meta.json`. If they differ, notify the + * lastCommit stored in the registered index metadata. If they differ, notify the * agent so it can decide when to reindex. */ function handlePostToolUse(input) { @@ -512,8 +421,8 @@ function handlePostToolUse(input) { const cwd = input.cwd || process.cwd(); if (!path.isAbsolute(cwd)) return; - const gitNexusDir = findGitNexusDir(cwd); - if (!gitNexusDir) return; + const repo = resolveHookRepo(cwd); + if (!repo) return; // Compare HEAD against last indexed commit — skip if unchanged let currentHead = ''; @@ -534,7 +443,7 @@ function handlePostToolUse(input) { let lastCommit = ''; let hadEmbeddings = false; - const meta = readIndexMeta(gitNexusDir); + const meta = repo.metadata; if (meta) { lastCommit = meta.lastCommit || ''; hadEmbeddings = meta.stats && meta.stats.embeddings > 0; diff --git a/gitnexus-claude-plugin/hooks/registry-query.cjs b/gitnexus-claude-plugin/hooks/registry-query.cjs new file mode 100644 index 000000000..649b363fe --- /dev/null +++ b/gitnexus-claude-plugin/hooks/registry-query.cjs @@ -0,0 +1,410 @@ +const fs = require('fs'); +const os = require('os'); +const path = require('path'); +const { createHash } = require('crypto'); +const { spawnSync } = require('child_process'); + +// Hooks are copied into editor-specific directories and run without the +// package's TypeScript modules. Keep their on-disk names centralized here. +const GITNEXUS_DIR = '.gitnexus'; +const INDEX_METADATA_FILE = 'gitnexus.json'; +const LEGACY_METADATA_FILE = 'meta.json'; +const LBUG_DIRECTORY = 'lbug'; +const BRANCHES_DIRECTORY = 'branches'; +const STORAGE_PATH_ENV = 'GITNEXUS_STORAGE_PATH'; +const STORAGE_ROOT_ENV = 'GITNEXUS_STORAGE_ROOT'; +const STORAGE_SLOT_HASH_LENGTH = 12; +const LOCAL_OWNED_PARENT_HOPS = 5; + +function stripWindowsLongPathPrefix(p) { + if (process.platform !== 'win32') return p; + if (/^\\\\\?\\UNC\\(?=[^\\])/i.test(p)) return `\\\\${p.slice(8)}`; + if (/^\\\\\?\\[A-Za-z]:\\/.test(p)) return p.slice(4); + return p; +} + +function canonicalize(value) { + if (typeof value !== 'string' || !value || value.includes('\0') || !path.isAbsolute(value)) + return null; + const resolved = path.resolve(value); + try { + return stripWindowsLongPathPrefix(fs.realpathSync.native(resolved)); + } catch { + return stripWindowsLongPathPrefix(resolved); + } +} + +function samePath(left, right) { + if (left == null || right == null) return false; + return process.platform === 'win32' ? left.toLowerCase() === right.toLowerCase() : left === right; +} + +function isMissingFile(error) { + return error && (error.code === 'ENOENT' || error.code === 'ENOTDIR'); +} + +function readMetadataFile(storagePath, filename) { + try { + const value = JSON.parse(fs.readFileSync(path.join(storagePath, filename), 'utf-8')); + return value && typeof value === 'object' && !Array.isArray(value) + ? { state: 'valid', value } + : { state: 'invalid' }; + } catch (error) { + return isMissingFile(error) ? { state: 'absent' } : { state: 'invalid' }; + } +} + +function readIndexMetadata(storagePath) { + const primary = readMetadataFile(storagePath, INDEX_METADATA_FILE); + if (primary.state === 'valid') return primary.value; + if (primary.state !== 'absent') return null; + + const legacy = readMetadataFile(storagePath, LEGACY_METADATA_FILE); + return legacy.state === 'valid' ? legacy.value : null; +} + +function isOwnedStorage(repoPath, storagePath, repositoryLocal, metadata) { + // Repository-local storage remains usable for metadata written before + // repoPath was recorded, but an explicit repoPath must never name another + // checkout. External storage always requires the complete ownership binding. + if (repositoryLocal && (!metadata || typeof metadata.repoPath !== 'string')) { + return true; + } + if (!metadata || typeof metadata.repoPath !== 'string') return false; + + const metadataRepoPath = canonicalize(metadata.repoPath); + const expectedRepoPath = canonicalize(repoPath); + if ( + metadataRepoPath == null || + expectedRepoPath == null || + !samePath(metadataRepoPath, expectedRepoPath) + ) { + return false; + } + if (repositoryLocal) return true; + if (typeof metadata.storagePath !== 'string') return false; + + const metadataStoragePath = canonicalize(metadata.storagePath); + const expectedStoragePath = canonicalize(storagePath); + return ( + metadataStoragePath != null && + expectedStoragePath != null && + samePath(metadataStoragePath, expectedStoragePath) + ); +} + +function ancestorPaths(cwd) { + const paths = []; + let current = canonicalize(cwd); + while (current) { + paths.push(current); + const parent = path.dirname(current); + if (parent === current) break; + current = parent; + } + return paths; +} + +function isInsideOrEqual(child, ancestor) { + if (child == null || ancestor == null) return false; + if (samePath(child, ancestor)) return true; + const relative = path.relative(ancestor, child); + return ( + relative !== '' && + relative !== '..' && + !relative.startsWith(`..${path.sep}`) && + !path.isAbsolute(relative) + ); +} + +function ancestorPathsThrough(cwd, stopAt) { + const paths = []; + let current = canonicalize(cwd); + const stop = canonicalize(stopAt); + while (current) { + if (stop && !isInsideOrEqual(current, stop)) break; + paths.push(current); + if (stop && samePath(current, stop)) break; + const parent = path.dirname(current); + if (parent === current) break; + current = parent; + } + return paths; +} + +function currentGitBranch(cwd) { + try { + const result = spawnSync('git', ['symbolic-ref', '--quiet', '--short', 'HEAD'], { + encoding: 'utf-8', + timeout: 2000, + cwd, + stdio: ['pipe', 'pipe', 'pipe'], + windowsHide: true, + }); + if (result.error || result.status !== 0) return null; + const branch = String(result.stdout || '').trim(); + return branch || null; + } catch { + return null; + } +} + +function registryPathsForCwd(cwd) { + const fallbackPaths = ancestorPaths(cwd); + if (fallbackPaths.length === 0) return { repoPaths: [], branch: null }; + try { + const result = spawnSync( + 'git', + ['rev-parse', '--path-format=absolute', '--show-toplevel', '--git-common-dir'], + { + encoding: 'utf-8', + timeout: 2000, + cwd, + stdio: ['pipe', 'pipe', 'pipe'], + windowsHide: true, + }, + ); + if (result.error || result.status !== 0) return { repoPaths: fallbackPaths, branch: null }; + + const [worktreeRoot, commonDir] = String(result.stdout || '') + .split(/\r?\n/) + .map((line) => line.trim()) + .filter(Boolean); + if (!worktreeRoot || !path.isAbsolute(worktreeRoot)) { + return { repoPaths: fallbackPaths, branch: null }; + } + + // Keep ancestor paths of cwd that stay inside this worktree (cwd up to + // and including show-toplevel) so a --skip-git subdirectory index can + // win via longest-match. Do not walk ancestors outside the worktree — + // that would re-attribute a parent index to a nested git checkout. + const repoPaths = ancestorPathsThrough(cwd, worktreeRoot); + const worktreeCanon = canonicalize(worktreeRoot); + if (worktreeCanon && !repoPaths.some((repoPath) => samePath(repoPath, worktreeCanon))) { + repoPaths.push(worktreeCanon); + } + + // Linked worktrees share the canonical repo's git dir. Include that + // parent so the registered main checkout is still discoverable, but do + // not walk any further outside this worktree. + if (commonDir) { + const commonParent = canonicalize(path.dirname(commonDir)); + if ( + commonParent && + worktreeCanon && + !samePath(commonParent, worktreeCanon) && + !repoPaths.some((repoPath) => samePath(repoPath, commonParent)) + ) { + repoPaths.push(commonParent); + } + } + return { + repoPaths, + branch: currentGitBranch(cwd), + }; + } catch { + return { repoPaths: fallbackPaths, branch: null }; + } +} + +function branchSlug(rawRef) { + const sanitized = rawRef.replace(/^-+/, '').replace(/[^a-zA-Z0-9._-]/g, '_'); + const reserved = /^(CON|PRN|AUX|NUL|COM[1-9]|LPT[1-9])(\..*)?$/i; + const safe = + !sanitized || sanitized === '.' || sanitized === '..' || reserved.test(sanitized) + ? 'unknown' + : sanitized; + const hash = createHash('sha256').update(rawRef).digest('hex').slice(0, 8); + return `${safe}-${hash}`; +} + +// Mirror gitnexus/src/storage/storage-resolver.ts storageSlotName exactly +// (sanitize + sha256 of the canonical repo path, 12-hex suffix). +function sanitizeSlotBasename(value) { + // Cap first, then walk the tail once — same order as + // gitnexus/src/storage/storage-resolver.ts (avoids /[. ]+$/ ReDoS). + const sanitized = value.replace(/[\u0000-\u001f<>:"/\\|?*]/g, '-').slice(0, 80); + let end = sanitized.length; + while (end > 0) { + const code = sanitized.charCodeAt(end - 1); + if (code !== 0x20 && code !== 0x2e) break; + end--; + } + const candidate = sanitized.slice(0, end) || 'repository'; + return /^(con|prn|aux|nul|com[1-9]|lpt[1-9])$/i.test(candidate) + ? `repository-${candidate}` + : candidate; +} + +function storageSlotName(repoPath) { + const canonical = canonicalize(repoPath); + if (!canonical) return null; + const identity = process.platform === 'win32' ? canonical.toLowerCase() : canonical; + const basename = sanitizeSlotBasename(path.basename(canonical)); + const digest = createHash('sha256') + .update(identity) + .digest('hex') + .slice(0, STORAGE_SLOT_HASH_LENGTH); + return `${basename}-${digest}`; +} + +function envOverridesStorage() { + const envPath = process.env[STORAGE_PATH_ENV]; + const envRoot = process.env[STORAGE_ROOT_ENV]; + return ( + (typeof envPath === 'string' && envPath.length > 0) || + (typeof envRoot === 'string' && envRoot.length > 0) + ); +} + +function resolveEntryStoragePath(entry) { + const envPath = process.env[STORAGE_PATH_ENV]; + if ( + typeof envPath === 'string' && + envPath.length > 0 && + !envPath.includes('\0') && + path.isAbsolute(envPath) + ) { + const resolved = path.resolve(envPath); + if (path.isAbsolute(resolved)) return resolved; + } + + const envRoot = process.env[STORAGE_ROOT_ENV]; + if ( + typeof envRoot === 'string' && + envRoot.length > 0 && + !envRoot.includes('\0') && + path.isAbsolute(envRoot) + ) { + const root = path.resolve(envRoot); + const slot = storageSlotName(entry.path); + if (slot) { + const storagePath = path.join(root, slot); + if (samePath(path.dirname(storagePath), root)) return storagePath; + } + } + + if (entry.storagePath !== undefined) { + if ( + typeof entry.storagePath !== 'string' || + !entry.storagePath || + entry.storagePath.includes('\0') || + !path.isAbsolute(entry.storagePath) + ) { + return null; + } + return path.resolve(entry.storagePath); + } + return path.resolve(path.join(entry.path, GITNEXUS_DIR)); +} + +function hasLocalIndexSignal(storagePath) { + try { + return ( + fs.existsSync(path.join(storagePath, INDEX_METADATA_FILE)) || + fs.existsSync(path.join(storagePath, LBUG_DIRECTORY)) + ); + } catch { + return false; + } +} + +function findLocalOwnedRepo(cwd) { + // Environment storage overrides win; a leftover repo-local .gitnexus must + // not skip the registry scan that applies STORAGE_PATH / STORAGE_ROOT. + if (envOverridesStorage()) return null; + const { repoPaths, branch } = registryPathsForCwd(cwd); + let current = canonicalize(cwd); + for (let hops = 0; hops <= LOCAL_OWNED_PARENT_HOPS && current; hops++) { + const storagePath = path.join(current, GITNEXUS_DIR); + if (hasLocalIndexSignal(storagePath)) { + const metadata = readIndexMetadata(storagePath); + if (isOwnedStorage(current, storagePath, true, metadata)) { + const branchDir = + branch != null ? path.join(storagePath, BRANCHES_DIRECTORY, branchSlug(branch)) : null; + const indexDir = branchDir && hasLocalIndexSignal(branchDir) ? branchDir : storagePath; + return { + path: current, + storagePath, + lbugPath: path.join(indexDir, LBUG_DIRECTORY), + metadata: indexDir === storagePath ? metadata : readIndexMetadata(indexDir), + }; + } + } + const parent = path.dirname(current); + if (parent === current) break; + // Stay inside this checkout. Registered lookup already stops at + // `--show-toplevel`; walking raw parents would adopt `/outer/.gitnexus` + // from `/outer/nested-repo`. + if (repoPaths.length > 0 && !repoPaths.some((repoPath) => samePath(repoPath, parent))) { + break; + } + current = parent; + } + return null; +} + +function findRegisteredRepo(cwd) { + const { repoPaths, branch } = registryPathsForCwd(cwd); + if (repoPaths.length === 0) return null; + + const home = process.env.GITNEXUS_HOME || path.join(os.homedir(), '.gitnexus'); + let entries; + try { + entries = JSON.parse(fs.readFileSync(path.join(home, 'registry.json'), 'utf-8')); + } catch { + return null; + } + if (!Array.isArray(entries)) return null; + + let best = null; + let bestLen = -1; + for (const entry of entries) { + if (!entry || typeof entry !== 'object' || Array.isArray(entry)) continue; + if (typeof entry.path !== 'string') continue; + if (entry.path.includes('\0') || !path.isAbsolute(entry.path)) continue; + const registeredPath = canonicalize(entry.path); + if (!registeredPath || !repoPaths.some((repoPath) => samePath(repoPath, registeredPath))) { + continue; + } + const storagePath = resolveEntryStoragePath(entry); + if (!storagePath) continue; + const repositoryLocal = samePath( + canonicalize(path.join(entry.path, GITNEXUS_DIR)), + canonicalize(storagePath), + ); + const ownershipMetadata = readIndexMetadata(storagePath); + if (!isOwnedStorage(entry.path, storagePath, repositoryLocal, ownershipMetadata)) continue; + const branchIsIndexed = + branch && + Array.isArray(entry.branches) && + entry.branches.some((summary) => summary && summary.branch === branch); + const indexDir = branchIsIndexed + ? path.join(storagePath, BRANCHES_DIRECTORY, branchSlug(branch)) + : storagePath; + if (registeredPath.length > bestLen) { + bestLen = registeredPath.length; + best = { + path: entry.path, + storagePath, + lbugPath: path.join(indexDir, LBUG_DIRECTORY), + metadata: branchIsIndexed ? readIndexMetadata(indexDir) : ownershipMetadata, + }; + } + } + return best; +} + +/** Registry row wins (including persisted external storagePath); local owned is fallback. */ +function resolveHookRepo(cwd) { + return findRegisteredRepo(cwd) || findLocalOwnedRepo(cwd); +} + +module.exports = { + findRegisteredRepo, + findLocalOwnedRepo, + resolveHookRepo, + INDEX_METADATA_FILE, + LEGACY_METADATA_FILE, + LBUG_DIRECTORY, +}; diff --git a/gitnexus-claude-plugin/skills/gitnexus-cli/SKILL.md b/gitnexus-claude-plugin/skills/gitnexus-cli/SKILL.md index 09c7af0d2..3a00ab546 100644 --- a/gitnexus-claude-plugin/skills/gitnexus-cli/SKILL.md +++ b/gitnexus-claude-plugin/skills/gitnexus-cli/SKILL.md @@ -34,6 +34,18 @@ Run from the project root. This parses all source files, builds the knowledge gr For Spring runtime enrichment, pass a JSON bundle, one endpoint JSON file, or a directory containing endpoint files. Route evidence is authoritative only when `runtimeConfirmed === true`; `runtimeSource` records provenance and may also accompany `handler-conflict`. Env/configprops values are never persisted. +## Index storage and retention + +Default location is `/.gitnexus/`. Override with environment variables (also documented in README): + +| Env | Effect | +| --- | ------ | +| `GITNEXUS_STORAGE_PATH` | One complete external index directory. Wins if both storage vars are set. | +| `GITNEXUS_STORAGE_ROOT` | Absolute root; GitNexus creates an isolated `-<12-hex>/` slot per repository. | +| `GITNEXUS_CONTENT_RETENTION` | `full` (default) keeps file text; `symbol` keeps snippets; `none` keeps the graph only. | + +`list_repos`, `gitnexus://repo/{name}/context`, and HTTP `GET /api/repos` / `GET /api/repo` expose `storagePath`, `contentRetention`, and `sourceAvailable`. HTTP `/api/file` and `/api/grep` return 410 unless retention is `full`. MCP `include_content` may still return symbol spans when retention is `symbol`. + Use `node .gitnexus/run.cjs analyze --watch` for a long-lived local Git repository. It performs an initial analysis, queues scanner-admitted file changes, and retries intact failed batches with bounded backoff. Watch refreshes update only the graph: they skip AGENTS.md / CLAUDE.md injection and standard skill installation, so run a one-shot `analyze` when those generated files need updating. Watch rejects one-shot or context-output flags including `--force`, embedding flags, `--skills`, `--default-branch`, `--skip-agents-md`, `--skip-skills`, `--no-stats`, `--self-commit`, `--index-only`, and `--skip-git`. It never pulls remotes. Scheduled remote clone/pull is a different command: `gitnexus auto-sync`. Bare `gitnexus watch` is reserved and does not start either job. Running MCP and `serve` processes periodically check for a published replacement and reopen it without a restart. MCP checks are throttled to once every five seconds, so a tool call before the next check can briefly use the previous index. ### status — Check index freshness diff --git a/gitnexus-claude-plugin/skills/gitnexus-cli/mcp.json b/gitnexus-claude-plugin/skills/gitnexus-cli/mcp.json index 57267bfb0..afe364b82 100644 --- a/gitnexus-claude-plugin/skills/gitnexus-cli/mcp.json +++ b/gitnexus-claude-plugin/skills/gitnexus-cli/mcp.json @@ -2,7 +2,7 @@ "mcpServers": { "gitnexus": { "command": "npx", - "args": ["-y", "gitnexus@1.6.11", "mcp"] + "args": ["-y", "gitnexus@1.6.12", "mcp"] } } } diff --git a/gitnexus-claude-plugin/skills/gitnexus-debugging/mcp.json b/gitnexus-claude-plugin/skills/gitnexus-debugging/mcp.json index 57267bfb0..afe364b82 100644 --- a/gitnexus-claude-plugin/skills/gitnexus-debugging/mcp.json +++ b/gitnexus-claude-plugin/skills/gitnexus-debugging/mcp.json @@ -2,7 +2,7 @@ "mcpServers": { "gitnexus": { "command": "npx", - "args": ["-y", "gitnexus@1.6.11", "mcp"] + "args": ["-y", "gitnexus@1.6.12", "mcp"] } } } diff --git a/gitnexus-claude-plugin/skills/gitnexus-exploring/mcp.json b/gitnexus-claude-plugin/skills/gitnexus-exploring/mcp.json index 57267bfb0..afe364b82 100644 --- a/gitnexus-claude-plugin/skills/gitnexus-exploring/mcp.json +++ b/gitnexus-claude-plugin/skills/gitnexus-exploring/mcp.json @@ -2,7 +2,7 @@ "mcpServers": { "gitnexus": { "command": "npx", - "args": ["-y", "gitnexus@1.6.11", "mcp"] + "args": ["-y", "gitnexus@1.6.12", "mcp"] } } } diff --git a/gitnexus-claude-plugin/skills/gitnexus-guide/SKILL.md b/gitnexus-claude-plugin/skills/gitnexus-guide/SKILL.md index e52560422..bf3c73948 100644 --- a/gitnexus-claude-plugin/skills/gitnexus-guide/SKILL.md +++ b/gitnexus-claude-plugin/skills/gitnexus-guide/SKILL.md @@ -83,15 +83,23 @@ Notes: `offset` ≥ `total` returns an empty page (with `total` still reported). ### Inline staleness signal (`query` / `context` / `impact` / `cypher`) -These four hot read tools attach a non-blocking `staleness` field to their response when the index is behind the checkout's current HEAD — the same `{ commitsBehind, hint }` shape `list_repos` already reports — so a direct tool call surfaces a behind-HEAD index without a separate `list_repos` call: +These four hot read tools attach a non-blocking `staleness` field to their response when the index is not at the checkout's current HEAD — the same `{ status, commitsBehind?, hint? }` shape `list_repos` already reports — so a direct tool call surfaces a stale index without a separate `list_repos` call: ```jsonc { /* …the tool's normal result… */ - "staleness": { "commitsBehind": 3, "hint": "⚠️ Index is 3 commits behind HEAD. Run analyze tool to update." } + "staleness": { "status": "behind", "commitsBehind": 3, "hint": "⚠️ Index is 3 commits behind HEAD. Run analyze tool to update." } } ``` -The field is **absent when the index is current** (or when the freshness check can't run), so its presence is the signal. It is only ever added to object results — raw-array `cypher` output and error envelopes are returned unchanged. `@group`-targeted calls do not carry it (multi-repo staleness is ill-defined). When you see it, the graph may be behind the working tree — re-run `analyze` before trusting blast-radius or dependence answers. +`commitsBehind` is present only when git counted the gap. When git could not count it but HEAD still resolves to a commit other than the indexed one — usually because the indexed commit is no longer in the clone's history — the index is provably not at HEAD with no countable gap, so no number is reported: + +```jsonc +{ /* …the tool's normal result… */ + "staleness": { "status": "diverged", "hint": "⚠️ Index is not at HEAD and the commit gap could not be counted — the recorded commit may no longer be in this clone's history. Run analyze tool to update." } +} +``` + +The field is **absent when the index is current**, and these four tools also omit it when the freshness check could not run at all — that case is `status: "unknown"`, which only the `list_repos` listing reports. So its presence means the status is not `current`: read `status` before using `commitsBehind`. It is only ever added to object results — raw-array `cypher` output and error envelopes are returned unchanged. `@group`-targeted calls do not carry it (multi-repo staleness is ill-defined). When you see it, the graph may be behind the working tree — re-run `analyze` before trusting blast-radius or dependence answers. ### Taint findings (`explain`) diff --git a/gitnexus-claude-plugin/skills/gitnexus-guide/mcp.json b/gitnexus-claude-plugin/skills/gitnexus-guide/mcp.json index 57267bfb0..afe364b82 100644 --- a/gitnexus-claude-plugin/skills/gitnexus-guide/mcp.json +++ b/gitnexus-claude-plugin/skills/gitnexus-guide/mcp.json @@ -2,7 +2,7 @@ "mcpServers": { "gitnexus": { "command": "npx", - "args": ["-y", "gitnexus@1.6.11", "mcp"] + "args": ["-y", "gitnexus@1.6.12", "mcp"] } } } diff --git a/gitnexus-claude-plugin/skills/gitnexus-impact-analysis/mcp.json b/gitnexus-claude-plugin/skills/gitnexus-impact-analysis/mcp.json index 57267bfb0..afe364b82 100644 --- a/gitnexus-claude-plugin/skills/gitnexus-impact-analysis/mcp.json +++ b/gitnexus-claude-plugin/skills/gitnexus-impact-analysis/mcp.json @@ -2,7 +2,7 @@ "mcpServers": { "gitnexus": { "command": "npx", - "args": ["-y", "gitnexus@1.6.11", "mcp"] + "args": ["-y", "gitnexus@1.6.12", "mcp"] } } } diff --git a/gitnexus-claude-plugin/skills/gitnexus-lfg/mcp.json b/gitnexus-claude-plugin/skills/gitnexus-lfg/mcp.json index 57267bfb0..afe364b82 100644 --- a/gitnexus-claude-plugin/skills/gitnexus-lfg/mcp.json +++ b/gitnexus-claude-plugin/skills/gitnexus-lfg/mcp.json @@ -2,7 +2,7 @@ "mcpServers": { "gitnexus": { "command": "npx", - "args": ["-y", "gitnexus@1.6.11", "mcp"] + "args": ["-y", "gitnexus@1.6.12", "mcp"] } } } diff --git a/gitnexus-claude-plugin/skills/gitnexus-plan/mcp.json b/gitnexus-claude-plugin/skills/gitnexus-plan/mcp.json index 57267bfb0..afe364b82 100644 --- a/gitnexus-claude-plugin/skills/gitnexus-plan/mcp.json +++ b/gitnexus-claude-plugin/skills/gitnexus-plan/mcp.json @@ -2,7 +2,7 @@ "mcpServers": { "gitnexus": { "command": "npx", - "args": ["-y", "gitnexus@1.6.11", "mcp"] + "args": ["-y", "gitnexus@1.6.12", "mcp"] } } } diff --git a/gitnexus-claude-plugin/skills/gitnexus-refactoring/mcp.json b/gitnexus-claude-plugin/skills/gitnexus-refactoring/mcp.json index 57267bfb0..afe364b82 100644 --- a/gitnexus-claude-plugin/skills/gitnexus-refactoring/mcp.json +++ b/gitnexus-claude-plugin/skills/gitnexus-refactoring/mcp.json @@ -2,7 +2,7 @@ "mcpServers": { "gitnexus": { "command": "npx", - "args": ["-y", "gitnexus@1.6.11", "mcp"] + "args": ["-y", "gitnexus@1.6.12", "mcp"] } } } diff --git a/gitnexus-claude-plugin/skills/gitnexus-review/mcp.json b/gitnexus-claude-plugin/skills/gitnexus-review/mcp.json index 57267bfb0..afe364b82 100644 --- a/gitnexus-claude-plugin/skills/gitnexus-review/mcp.json +++ b/gitnexus-claude-plugin/skills/gitnexus-review/mcp.json @@ -2,7 +2,7 @@ "mcpServers": { "gitnexus": { "command": "npx", - "args": ["-y", "gitnexus@1.6.11", "mcp"] + "args": ["-y", "gitnexus@1.6.12", "mcp"] } } } diff --git a/gitnexus-claude-plugin/skills/gitnexus-work/mcp.json b/gitnexus-claude-plugin/skills/gitnexus-work/mcp.json index 57267bfb0..afe364b82 100644 --- a/gitnexus-claude-plugin/skills/gitnexus-work/mcp.json +++ b/gitnexus-claude-plugin/skills/gitnexus-work/mcp.json @@ -2,7 +2,7 @@ "mcpServers": { "gitnexus": { "command": "npx", - "args": ["-y", "gitnexus@1.6.11", "mcp"] + "args": ["-y", "gitnexus@1.6.12", "mcp"] } } } diff --git a/gitnexus-cursor-integration/README.md b/gitnexus-cursor-integration/README.md index 67bed4583..0d044aaa9 100644 --- a/gitnexus-cursor-integration/README.md +++ b/gitnexus-cursor-integration/README.md @@ -6,11 +6,11 @@ Static config that adds GitNexus knowledge-graph augmentation and skill files to ## What you get -| Layer | What it does | How it's installed | -| ------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------- | -| **MCP** | `gitnexus` MCP server with 17 tools (`query`, `context`, `impact`, `detect_changes`, `rename`, …) | `npx gitnexus setup` writes `~/.cursor/mcp.json` automatically. | -| **Skills** | All bundled markdown skills (`/gitnexus-exploring`, `/gitnexus-debugging`, `/gitnexus-impact-analysis`, `/gitnexus-refactoring`, `/gitnexus-guide`, `/gitnexus-cli`, `/gitnexus-review`, `/gitnexus-plan`, `/gitnexus-work`, `/gitnexus-lfg`, `/gitnexus-pdg-query`, `/gitnexus-taint-analysis`) | `npx gitnexus setup` copies them to `~/.cursor/skills/gitnexus/`. | -| **Hooks** _(this README)_ | `postToolUse` hook that enriches `Shell` / `Read` / `Grep` tool calls with graph context — same augmentation Claude Code gets | **Manual** — copy the files described below into your project's `.cursor/`. | +| Layer | What it does | How it's installed | +| ------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | --------------------------------------------------------------------------- | +| **MCP** | `gitnexus` MCP server with 17 tools (`query`, `context`, `impact`, `detect_changes`, `rename`, …) | `npx gitnexus setup` writes `~/.cursor/mcp.json` automatically. | +| **Skills** | All bundled markdown skills (`/gitnexus-exploring`, `/gitnexus-debugging`, `/gitnexus-impact-analysis`, `/gitnexus-refactoring`, `/gitnexus-guide`, `/gitnexus-cli`, `/gitnexus-review`, `/gitnexus-plan`, `/gitnexus-work`, `/gitnexus-lfg`, `/gitnexus-pdg-query`, `/gitnexus-taint-analysis`) | `npx gitnexus setup` copies them to `~/.cursor/skills/gitnexus/`. | +| **Hooks** _(this README)_ | `postToolUse` hook that enriches `Shell` / `Read` / `Grep` tool calls with graph context — same augmentation Claude Code gets | **Manual** — copy the files described below into your project's `.cursor/`. | ## Hook install @@ -24,7 +24,8 @@ From this repo's `gitnexus-cursor-integration/hooks/`, copy the files below into │ └── hooks.json ← from gitnexus-cursor-integration/hooks/hooks.json └── hooks/ ├── gitnexus-hook.cjs ← from gitnexus-cursor-integration/hooks/gitnexus-hook.cjs - └── hook-lock.cjs ← from gitnexus-cursor-integration/hooks/hook-lock.cjs + ├── hook-lock.cjs ← from gitnexus-cursor-integration/hooks/hook-lock.cjs + └── registry-query.cjs ← from gitnexus-cursor-integration/hooks/registry-query.cjs ``` Equivalent shell commands (run from your project root, with `$GITNEXUS_REPO` pointing at a clone of this repo): @@ -34,6 +35,7 @@ mkdir -p .cursor hooks cp "$GITNEXUS_REPO/gitnexus-cursor-integration/hooks/hooks.json" .cursor/hooks.json cp "$GITNEXUS_REPO/gitnexus-cursor-integration/hooks/gitnexus-hook.cjs" hooks/gitnexus-hook.cjs cp "$GITNEXUS_REPO/gitnexus-cursor-integration/hooks/hook-lock.cjs" hooks/hook-lock.cjs +cp "$GITNEXUS_REPO/gitnexus-cursor-integration/hooks/registry-query.cjs" hooks/registry-query.cjs ``` If you already have a `.cursor/hooks.json`, merge the `hooks.postToolUse` array rather than overwriting. @@ -47,11 +49,11 @@ If you already have a `.cursor/hooks.json`, merge the `hooks.postToolUse` array ### What's installed manually vs. automated -| Step | Automated by `gitnexus setup`? | -| -------------------------------------------------------------------- | ------------------------------ | -| `~/.cursor/mcp.json` | ✅ | -| `~/.cursor/skills/gitnexus/*` | ✅ | -| `/.cursor/hooks.json` + `/hooks/gitnexus-hook.cjs` + `/hooks/hook-lock.cjs` | ❌ — copy manually (see above) | +| Step | Automated by `gitnexus setup`? | +| --------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------ | +| `~/.cursor/mcp.json` | ✅ | +| `~/.cursor/skills/gitnexus/*` | ✅ | +| `/.cursor/hooks.json` + `/hooks/gitnexus-hook.cjs` + `/hooks/hook-lock.cjs` + `/hooks/registry-query.cjs` | ❌ — copy manually (see above) | Hook install is per-project (Cursor scopes hooks to a project root); skills and MCP config are global. @@ -86,6 +88,6 @@ Empty stdout means "no augmentation, continue normally" — the hook never block ## Troubleshooting -- **Nothing happens** — Confirm Cursor is on 2.4+ and the project root has `.cursor/hooks.json` plus both hook files at `hooks/gitnexus-hook.cjs` and `hooks/hook-lock.cjs`. Then `npx gitnexus list` to confirm the project is indexed. +- **Nothing happens** — Confirm Cursor is on 2.4+ and the project root has `.cursor/hooks.json` plus the hook files at `hooks/gitnexus-hook.cjs`, `hooks/hook-lock.cjs`, and `hooks/registry-query.cjs`. Then `npx gitnexus list` to confirm the project is indexed. - **`gitnexus` not found** — The hook prefers a locally-resolvable `gitnexus/dist/cli/index.js` and falls back to `npx -y gitnexus`. Install globally with `npm i -g gitnexus` to skip the npx cold-start latency. - **Wrong pattern extracted** — Set `GITNEXUS_DEBUG=1` and run a tool call. The raw stdin payload is logged to stderr; use it to confirm Cursor's actual `tool_input` field names against the table above. If they differ, file an issue with the captured payload. diff --git a/gitnexus-cursor-integration/hooks/gitnexus-hook.cjs b/gitnexus-cursor-integration/hooks/gitnexus-hook.cjs index 564384f83..ffddd163e 100644 --- a/gitnexus-cursor-integration/hooks/gitnexus-hook.cjs +++ b/gitnexus-cursor-integration/hooks/gitnexus-hook.cjs @@ -19,6 +19,7 @@ const fs = require('fs'); const path = require('path'); const { spawnSync } = require('child_process'); const { acquireHookSlot } = require('./hook-lock.cjs'); +const { resolveHookRepo } = require('./registry-query.cjs'); function readInput() { try { @@ -29,62 +30,6 @@ function readInput() { } } -function isGlobalRegistryDir(candidate) { - if ( - fs.existsSync(path.join(candidate, 'gitnexus.json')) || - fs.existsSync(path.join(candidate, 'meta.json')) - ) { - return false; - } - return ( - fs.existsSync(path.join(candidate, 'registry.json')) || - fs.existsSync(path.join(candidate, 'repos')) - ); -} - -function walkForGitNexusDir(startDir) { - let dir = startDir; - for (let i = 0; i < 5; i++) { - const candidate = path.join(dir, '.gitnexus'); - if (fs.existsSync(candidate)) { - if (!isGlobalRegistryDir(candidate)) return candidate; - } - const parent = path.dirname(dir); - if (parent === dir) break; - dir = parent; - } - return null; -} - -function findCanonicalRepoRoot(cwd) { - try { - const result = spawnSync('git', ['rev-parse', '--path-format=absolute', '--git-common-dir'], { - encoding: 'utf-8', - timeout: 2000, - cwd, - stdio: ['pipe', 'pipe', 'pipe'], - windowsHide: true, - }); - if (result.error || result.status !== 0) return null; - const commonDir = (result.stdout || '').trim(); - if (!commonDir || !path.isAbsolute(commonDir)) return null; - return path.dirname(commonDir); - } catch { - return null; - } -} - -function findGitNexusDir(startDir) { - const cwd = startDir || process.cwd(); - const fromCwd = walkForGitNexusDir(cwd); - if (fromCwd) return fromCwd; - const canonicalRoot = findCanonicalRepoRoot(cwd); - if (canonicalRoot && canonicalRoot !== cwd) { - return walkForGitNexusDir(canonicalRoot); - } - return null; -} - function tokenizeShellWords(command) { const tokens = []; let current = ''; @@ -431,16 +376,19 @@ function main() { } const cwd = input.cwd || process.cwd(); if (!path.isAbsolute(cwd)) return; - const gitNexusDir = findGitNexusDir(cwd); - if (!gitNexusDir) return; const toolName = input.tool_name || ''; const toolInput = input.tool_input || {}; - const pattern = extractPattern(toolName, toolInput); if (!pattern || pattern.length < 3) return; - const release = acquireHookSlot(gitNexusDir); + // Registry row first (persisted external storagePath wins). Local owned + // `.gitnexus` is only the fallback when no matching registry row exists. + const repo = resolveHookRepo(cwd); + if (!repo) return; + const storagePath = repo.storagePath; + + const release = acquireHookSlot(storagePath); if (!release) { // Normal skip path: all per-repo hook slots are held by concurrent // sessions. Stays silent by default; surfaced only under the cursor diff --git a/gitnexus-cursor-integration/hooks/registry-query.cjs b/gitnexus-cursor-integration/hooks/registry-query.cjs new file mode 100644 index 000000000..649b363fe --- /dev/null +++ b/gitnexus-cursor-integration/hooks/registry-query.cjs @@ -0,0 +1,410 @@ +const fs = require('fs'); +const os = require('os'); +const path = require('path'); +const { createHash } = require('crypto'); +const { spawnSync } = require('child_process'); + +// Hooks are copied into editor-specific directories and run without the +// package's TypeScript modules. Keep their on-disk names centralized here. +const GITNEXUS_DIR = '.gitnexus'; +const INDEX_METADATA_FILE = 'gitnexus.json'; +const LEGACY_METADATA_FILE = 'meta.json'; +const LBUG_DIRECTORY = 'lbug'; +const BRANCHES_DIRECTORY = 'branches'; +const STORAGE_PATH_ENV = 'GITNEXUS_STORAGE_PATH'; +const STORAGE_ROOT_ENV = 'GITNEXUS_STORAGE_ROOT'; +const STORAGE_SLOT_HASH_LENGTH = 12; +const LOCAL_OWNED_PARENT_HOPS = 5; + +function stripWindowsLongPathPrefix(p) { + if (process.platform !== 'win32') return p; + if (/^\\\\\?\\UNC\\(?=[^\\])/i.test(p)) return `\\\\${p.slice(8)}`; + if (/^\\\\\?\\[A-Za-z]:\\/.test(p)) return p.slice(4); + return p; +} + +function canonicalize(value) { + if (typeof value !== 'string' || !value || value.includes('\0') || !path.isAbsolute(value)) + return null; + const resolved = path.resolve(value); + try { + return stripWindowsLongPathPrefix(fs.realpathSync.native(resolved)); + } catch { + return stripWindowsLongPathPrefix(resolved); + } +} + +function samePath(left, right) { + if (left == null || right == null) return false; + return process.platform === 'win32' ? left.toLowerCase() === right.toLowerCase() : left === right; +} + +function isMissingFile(error) { + return error && (error.code === 'ENOENT' || error.code === 'ENOTDIR'); +} + +function readMetadataFile(storagePath, filename) { + try { + const value = JSON.parse(fs.readFileSync(path.join(storagePath, filename), 'utf-8')); + return value && typeof value === 'object' && !Array.isArray(value) + ? { state: 'valid', value } + : { state: 'invalid' }; + } catch (error) { + return isMissingFile(error) ? { state: 'absent' } : { state: 'invalid' }; + } +} + +function readIndexMetadata(storagePath) { + const primary = readMetadataFile(storagePath, INDEX_METADATA_FILE); + if (primary.state === 'valid') return primary.value; + if (primary.state !== 'absent') return null; + + const legacy = readMetadataFile(storagePath, LEGACY_METADATA_FILE); + return legacy.state === 'valid' ? legacy.value : null; +} + +function isOwnedStorage(repoPath, storagePath, repositoryLocal, metadata) { + // Repository-local storage remains usable for metadata written before + // repoPath was recorded, but an explicit repoPath must never name another + // checkout. External storage always requires the complete ownership binding. + if (repositoryLocal && (!metadata || typeof metadata.repoPath !== 'string')) { + return true; + } + if (!metadata || typeof metadata.repoPath !== 'string') return false; + + const metadataRepoPath = canonicalize(metadata.repoPath); + const expectedRepoPath = canonicalize(repoPath); + if ( + metadataRepoPath == null || + expectedRepoPath == null || + !samePath(metadataRepoPath, expectedRepoPath) + ) { + return false; + } + if (repositoryLocal) return true; + if (typeof metadata.storagePath !== 'string') return false; + + const metadataStoragePath = canonicalize(metadata.storagePath); + const expectedStoragePath = canonicalize(storagePath); + return ( + metadataStoragePath != null && + expectedStoragePath != null && + samePath(metadataStoragePath, expectedStoragePath) + ); +} + +function ancestorPaths(cwd) { + const paths = []; + let current = canonicalize(cwd); + while (current) { + paths.push(current); + const parent = path.dirname(current); + if (parent === current) break; + current = parent; + } + return paths; +} + +function isInsideOrEqual(child, ancestor) { + if (child == null || ancestor == null) return false; + if (samePath(child, ancestor)) return true; + const relative = path.relative(ancestor, child); + return ( + relative !== '' && + relative !== '..' && + !relative.startsWith(`..${path.sep}`) && + !path.isAbsolute(relative) + ); +} + +function ancestorPathsThrough(cwd, stopAt) { + const paths = []; + let current = canonicalize(cwd); + const stop = canonicalize(stopAt); + while (current) { + if (stop && !isInsideOrEqual(current, stop)) break; + paths.push(current); + if (stop && samePath(current, stop)) break; + const parent = path.dirname(current); + if (parent === current) break; + current = parent; + } + return paths; +} + +function currentGitBranch(cwd) { + try { + const result = spawnSync('git', ['symbolic-ref', '--quiet', '--short', 'HEAD'], { + encoding: 'utf-8', + timeout: 2000, + cwd, + stdio: ['pipe', 'pipe', 'pipe'], + windowsHide: true, + }); + if (result.error || result.status !== 0) return null; + const branch = String(result.stdout || '').trim(); + return branch || null; + } catch { + return null; + } +} + +function registryPathsForCwd(cwd) { + const fallbackPaths = ancestorPaths(cwd); + if (fallbackPaths.length === 0) return { repoPaths: [], branch: null }; + try { + const result = spawnSync( + 'git', + ['rev-parse', '--path-format=absolute', '--show-toplevel', '--git-common-dir'], + { + encoding: 'utf-8', + timeout: 2000, + cwd, + stdio: ['pipe', 'pipe', 'pipe'], + windowsHide: true, + }, + ); + if (result.error || result.status !== 0) return { repoPaths: fallbackPaths, branch: null }; + + const [worktreeRoot, commonDir] = String(result.stdout || '') + .split(/\r?\n/) + .map((line) => line.trim()) + .filter(Boolean); + if (!worktreeRoot || !path.isAbsolute(worktreeRoot)) { + return { repoPaths: fallbackPaths, branch: null }; + } + + // Keep ancestor paths of cwd that stay inside this worktree (cwd up to + // and including show-toplevel) so a --skip-git subdirectory index can + // win via longest-match. Do not walk ancestors outside the worktree — + // that would re-attribute a parent index to a nested git checkout. + const repoPaths = ancestorPathsThrough(cwd, worktreeRoot); + const worktreeCanon = canonicalize(worktreeRoot); + if (worktreeCanon && !repoPaths.some((repoPath) => samePath(repoPath, worktreeCanon))) { + repoPaths.push(worktreeCanon); + } + + // Linked worktrees share the canonical repo's git dir. Include that + // parent so the registered main checkout is still discoverable, but do + // not walk any further outside this worktree. + if (commonDir) { + const commonParent = canonicalize(path.dirname(commonDir)); + if ( + commonParent && + worktreeCanon && + !samePath(commonParent, worktreeCanon) && + !repoPaths.some((repoPath) => samePath(repoPath, commonParent)) + ) { + repoPaths.push(commonParent); + } + } + return { + repoPaths, + branch: currentGitBranch(cwd), + }; + } catch { + return { repoPaths: fallbackPaths, branch: null }; + } +} + +function branchSlug(rawRef) { + const sanitized = rawRef.replace(/^-+/, '').replace(/[^a-zA-Z0-9._-]/g, '_'); + const reserved = /^(CON|PRN|AUX|NUL|COM[1-9]|LPT[1-9])(\..*)?$/i; + const safe = + !sanitized || sanitized === '.' || sanitized === '..' || reserved.test(sanitized) + ? 'unknown' + : sanitized; + const hash = createHash('sha256').update(rawRef).digest('hex').slice(0, 8); + return `${safe}-${hash}`; +} + +// Mirror gitnexus/src/storage/storage-resolver.ts storageSlotName exactly +// (sanitize + sha256 of the canonical repo path, 12-hex suffix). +function sanitizeSlotBasename(value) { + // Cap first, then walk the tail once — same order as + // gitnexus/src/storage/storage-resolver.ts (avoids /[. ]+$/ ReDoS). + const sanitized = value.replace(/[\u0000-\u001f<>:"/\\|?*]/g, '-').slice(0, 80); + let end = sanitized.length; + while (end > 0) { + const code = sanitized.charCodeAt(end - 1); + if (code !== 0x20 && code !== 0x2e) break; + end--; + } + const candidate = sanitized.slice(0, end) || 'repository'; + return /^(con|prn|aux|nul|com[1-9]|lpt[1-9])$/i.test(candidate) + ? `repository-${candidate}` + : candidate; +} + +function storageSlotName(repoPath) { + const canonical = canonicalize(repoPath); + if (!canonical) return null; + const identity = process.platform === 'win32' ? canonical.toLowerCase() : canonical; + const basename = sanitizeSlotBasename(path.basename(canonical)); + const digest = createHash('sha256') + .update(identity) + .digest('hex') + .slice(0, STORAGE_SLOT_HASH_LENGTH); + return `${basename}-${digest}`; +} + +function envOverridesStorage() { + const envPath = process.env[STORAGE_PATH_ENV]; + const envRoot = process.env[STORAGE_ROOT_ENV]; + return ( + (typeof envPath === 'string' && envPath.length > 0) || + (typeof envRoot === 'string' && envRoot.length > 0) + ); +} + +function resolveEntryStoragePath(entry) { + const envPath = process.env[STORAGE_PATH_ENV]; + if ( + typeof envPath === 'string' && + envPath.length > 0 && + !envPath.includes('\0') && + path.isAbsolute(envPath) + ) { + const resolved = path.resolve(envPath); + if (path.isAbsolute(resolved)) return resolved; + } + + const envRoot = process.env[STORAGE_ROOT_ENV]; + if ( + typeof envRoot === 'string' && + envRoot.length > 0 && + !envRoot.includes('\0') && + path.isAbsolute(envRoot) + ) { + const root = path.resolve(envRoot); + const slot = storageSlotName(entry.path); + if (slot) { + const storagePath = path.join(root, slot); + if (samePath(path.dirname(storagePath), root)) return storagePath; + } + } + + if (entry.storagePath !== undefined) { + if ( + typeof entry.storagePath !== 'string' || + !entry.storagePath || + entry.storagePath.includes('\0') || + !path.isAbsolute(entry.storagePath) + ) { + return null; + } + return path.resolve(entry.storagePath); + } + return path.resolve(path.join(entry.path, GITNEXUS_DIR)); +} + +function hasLocalIndexSignal(storagePath) { + try { + return ( + fs.existsSync(path.join(storagePath, INDEX_METADATA_FILE)) || + fs.existsSync(path.join(storagePath, LBUG_DIRECTORY)) + ); + } catch { + return false; + } +} + +function findLocalOwnedRepo(cwd) { + // Environment storage overrides win; a leftover repo-local .gitnexus must + // not skip the registry scan that applies STORAGE_PATH / STORAGE_ROOT. + if (envOverridesStorage()) return null; + const { repoPaths, branch } = registryPathsForCwd(cwd); + let current = canonicalize(cwd); + for (let hops = 0; hops <= LOCAL_OWNED_PARENT_HOPS && current; hops++) { + const storagePath = path.join(current, GITNEXUS_DIR); + if (hasLocalIndexSignal(storagePath)) { + const metadata = readIndexMetadata(storagePath); + if (isOwnedStorage(current, storagePath, true, metadata)) { + const branchDir = + branch != null ? path.join(storagePath, BRANCHES_DIRECTORY, branchSlug(branch)) : null; + const indexDir = branchDir && hasLocalIndexSignal(branchDir) ? branchDir : storagePath; + return { + path: current, + storagePath, + lbugPath: path.join(indexDir, LBUG_DIRECTORY), + metadata: indexDir === storagePath ? metadata : readIndexMetadata(indexDir), + }; + } + } + const parent = path.dirname(current); + if (parent === current) break; + // Stay inside this checkout. Registered lookup already stops at + // `--show-toplevel`; walking raw parents would adopt `/outer/.gitnexus` + // from `/outer/nested-repo`. + if (repoPaths.length > 0 && !repoPaths.some((repoPath) => samePath(repoPath, parent))) { + break; + } + current = parent; + } + return null; +} + +function findRegisteredRepo(cwd) { + const { repoPaths, branch } = registryPathsForCwd(cwd); + if (repoPaths.length === 0) return null; + + const home = process.env.GITNEXUS_HOME || path.join(os.homedir(), '.gitnexus'); + let entries; + try { + entries = JSON.parse(fs.readFileSync(path.join(home, 'registry.json'), 'utf-8')); + } catch { + return null; + } + if (!Array.isArray(entries)) return null; + + let best = null; + let bestLen = -1; + for (const entry of entries) { + if (!entry || typeof entry !== 'object' || Array.isArray(entry)) continue; + if (typeof entry.path !== 'string') continue; + if (entry.path.includes('\0') || !path.isAbsolute(entry.path)) continue; + const registeredPath = canonicalize(entry.path); + if (!registeredPath || !repoPaths.some((repoPath) => samePath(repoPath, registeredPath))) { + continue; + } + const storagePath = resolveEntryStoragePath(entry); + if (!storagePath) continue; + const repositoryLocal = samePath( + canonicalize(path.join(entry.path, GITNEXUS_DIR)), + canonicalize(storagePath), + ); + const ownershipMetadata = readIndexMetadata(storagePath); + if (!isOwnedStorage(entry.path, storagePath, repositoryLocal, ownershipMetadata)) continue; + const branchIsIndexed = + branch && + Array.isArray(entry.branches) && + entry.branches.some((summary) => summary && summary.branch === branch); + const indexDir = branchIsIndexed + ? path.join(storagePath, BRANCHES_DIRECTORY, branchSlug(branch)) + : storagePath; + if (registeredPath.length > bestLen) { + bestLen = registeredPath.length; + best = { + path: entry.path, + storagePath, + lbugPath: path.join(indexDir, LBUG_DIRECTORY), + metadata: branchIsIndexed ? readIndexMetadata(indexDir) : ownershipMetadata, + }; + } + } + return best; +} + +/** Registry row wins (including persisted external storagePath); local owned is fallback. */ +function resolveHookRepo(cwd) { + return findRegisteredRepo(cwd) || findLocalOwnedRepo(cwd); +} + +module.exports = { + findRegisteredRepo, + findLocalOwnedRepo, + resolveHookRepo, + INDEX_METADATA_FILE, + LEGACY_METADATA_FILE, + LBUG_DIRECTORY, +}; diff --git a/gitnexus-shared/src/graph/types.ts b/gitnexus-shared/src/graph/types.ts index 58d96e42f..8b2b13596 100644 --- a/gitnexus-shared/src/graph/types.ts +++ b/gitnexus-shared/src/graph/types.ts @@ -14,6 +14,8 @@ export type NodeLabel = | 'Folder' | 'File' | 'Class' + | 'Protocol' + | 'Category' | 'Function' | 'Method' | 'Variable' diff --git a/gitnexus-shared/src/index.ts b/gitnexus-shared/src/index.ts index 9857a60cc..7958cfa9d 100644 --- a/gitnexus-shared/src/index.ts +++ b/gitnexus-shared/src/index.ts @@ -137,6 +137,7 @@ export type { FinalizeOutput, FinalizedScc, FinalizeStats, + AmbiguousWildcardExport, } from './scope-resolution/finalize-algorithm.js'; // Scope-aware registries + 7-step lookup (RFC §4; Ring 2 SHARED #917) diff --git a/gitnexus-shared/src/language-detection.ts b/gitnexus-shared/src/language-detection.ts index e805d4e4c..3fe27f7b7 100644 --- a/gitnexus-shared/src/language-detection.ts +++ b/gitnexus-shared/src/language-detection.ts @@ -32,6 +32,7 @@ const EXTENSION_MAP: Record = { [SupportedLanguages.Python]: ['.py'], [SupportedLanguages.Java]: ['.java'], [SupportedLanguages.C]: ['.c'], + [SupportedLanguages.ObjectiveC]: ['.m', '.mm'], [SupportedLanguages.CPlusPlus]: [ '.cpp', '.cc', @@ -111,6 +112,7 @@ const SYNTAX_MAP: Record = { [SupportedLanguages.Python]: 'python', [SupportedLanguages.Java]: 'java', [SupportedLanguages.C]: 'c', + [SupportedLanguages.ObjectiveC]: 'objectivec', [SupportedLanguages.CPlusPlus]: 'cpp', [SupportedLanguages.CSharp]: 'csharp', [SupportedLanguages.Go]: 'go', diff --git a/gitnexus-shared/src/languages.ts b/gitnexus-shared/src/languages.ts index d16adbd1a..eec45e612 100644 --- a/gitnexus-shared/src/languages.ts +++ b/gitnexus-shared/src/languages.ts @@ -11,6 +11,7 @@ export enum SupportedLanguages { Java = 'java', C = 'c', CPlusPlus = 'cpp', + ObjectiveC = 'objective-c', CSharp = 'csharp', Go = 'go', Ruby = 'ruby', diff --git a/gitnexus-shared/src/lbug/schema-constants.ts b/gitnexus-shared/src/lbug/schema-constants.ts index 217c382a6..aea908ef9 100644 --- a/gitnexus-shared/src/lbug/schema-constants.ts +++ b/gitnexus-shared/src/lbug/schema-constants.ts @@ -13,6 +13,8 @@ export const NODE_TABLES = [ 'Folder', 'Function', 'Class', + 'Protocol', + 'Category', 'Interface', 'Method', 'CodeElement', diff --git a/gitnexus-shared/src/scope-resolution/finalize-algorithm.ts b/gitnexus-shared/src/scope-resolution/finalize-algorithm.ts index f6038b3f0..2dc14b398 100644 --- a/gitnexus-shared/src/scope-resolution/finalize-algorithm.ts +++ b/gitnexus-shared/src/scope-resolution/finalize-algorithm.ts @@ -5,13 +5,14 @@ * Pure logic that takes per-file parse output (`ParsedImport[]` + * `SymbolDefinition[]`) and returns: * - * - Linked `ImportEdge[]` per module scope, with `targetModuleScope` and - * `targetDefId` filled where resolvable; edges that could not be - * resolved within the hard fixpoint cap are marked + * - Linked `ImportEdge[]` keyed by binding scope (`fromScope`; module + * scope unless `importsBindAtLexicalScope` is on), with + * `targetModuleScope` and `targetDefId` filled where resolvable; edges + * that could not be resolved within the hard fixpoint cap are marked * `linkStatus: 'unresolved'`. - * - Materialized `bindings` per module scope — local defs merged with - * imported / wildcard-expanded / re-exported names via the provider's - * `mergeBindings` precedence. + * - Materialized `bindings` keyed by the same scopes — local defs merged + * with imported / wildcard-expanded / re-exported names via the + * provider's `mergeBindings` precedence. * - The SCC condensation of the import graph, exposed so disjoint SCCs * can be processed in parallel by callers that want that. * @@ -39,7 +40,7 @@ import type { BindingRef, ImportEdge, ParsedImport, ScopeId, WorkspaceIndex } fr /** Per-file input for the finalize pass. */ export interface FinalizeFile { readonly filePath: string; - /** The module scope id for this file; owns the finalized imports + bindings. */ + /** Default binding scope for imports without lexical provenance or opt-in. */ readonly moduleScope: ScopeId; readonly parsedImports: readonly ParsedImport[]; /** @@ -84,6 +85,10 @@ export interface FinalizeInput { * expects pure answers. */ export interface FinalizeHooks { + /** Bind imports at their extracted lexical scope. Missing provenance retains + * the legacy module-scope behavior. Opt-in: lexical position and language + * import-binding semantics are distinct facts. */ + readonly importsBindAtLexicalScope?: boolean; /** * Resolve a raw import target to the concrete file path that owns it. * Return `null` when no target file is resolvable (e.g., `np.foo` when @@ -114,6 +119,34 @@ export interface FinalizeHooks { */ expandsWildcardTo(targetModuleScope: ScopeId, workspaceIndex: WorkspaceIndex): readonly string[]; + /** + * Does this language make two `wildcard` re-exports that both DECLARE the + * same name AMBIGUOUS (no winner), rather than overloads or redeclarations + * of one entity? + * + * True for ECMAScript modules: `export * from './a'; export * from './b'` + * with `collide` declared in both excludes the name from the module's + * exports, so binding either source is a guess. False (the default) for + * languages whose wildcard import is `#include`, `require`, or a package + * fan-out, where the same name declared in two files is an overload set + * (C++ `write_audit(int)` / `write_audit(int, int)` across two headers), a + * redeclaration of one function, or a per-file `init` — legal, and resolved + * downstream by arity or by definition. Only a language that opts in has + * its collisions refused and reported via `ambiguousWildcardExports`. + */ + readonly wildcardCollisionIsAmbiguous?: boolean; + + /** + * A named import / named re-export binds only to MODULE-LEVEL declarations + * of the target file. Opt-in for languages whose `import { x }` can never + * reach a class member: without it, a class method sharing a name with a + * top-level value — or standing alone — wins the callable preference in + * `findExportByName` and the import binds to a symbol the module cannot + * export (a confident wrong edge). Languages that bind module-level members + * by bare name (static members, module functions) leave it off. + */ + readonly namedImportsBindTopLevelOnly?: boolean; + /** * Merge `incoming` bindings into `existing` for a given name. Called * once per name at each scope. Typical rules: @@ -164,12 +197,32 @@ export interface FinalizeStats { readonly unresolvedEdges: number; readonly sccCount: number; readonly largestSccSize: number; + /** + * Names a file re-exported through two or more `export *` sources that each + * DECLARE the name, so the language names no winner. The finalize pass + * refuses to bind them (they are absent from the file's re-export closure + * AND from its wildcard-expanded module-scope bindings) instead of taking the + * first-listed source and publishing the guess as `import-resolved`. + * Reported so the caller can record the refusal — an importer of that name + * stays unresolved, and the reason must be auditable rather than silent. + */ + readonly ambiguousWildcardExports: readonly AmbiguousWildcardExport[]; +} + +/** One refused `export *` collision — see `FinalizeStats.ambiguousWildcardExports`. */ +export interface AmbiguousWildcardExport { + readonly filePath: string; + readonly name: string; + /** `nodeId`s of the colliding declarations, in `export *` declaration order. */ + readonly candidateDefIds: readonly string[]; } export interface FinalizeOutput { - /** Linked `ImportEdge[]` per module scope, in original input order. */ + /** Linked `ImportEdge[]` keyed by binding scope (`fromScope`). Module scope + * for languages that bind file-wide; nested lexical scopes when + * `importsBindAtLexicalScope` is on. */ readonly imports: ReadonlyMap; - /** Materialized bindings per module scope. */ + /** Materialized bindings keyed by the same scopes as `imports`. */ readonly bindings: ReadonlyMap>; /** SCCs in reverse-topological order (leaves first). */ readonly sccs: readonly FinalizedScc[]; @@ -223,7 +276,33 @@ export function finalize(input: FinalizeInput, hooks: FinalizeHooks): FinalizeOu // SCC-condensed). Eliminates the recursive crawl that the per-edge // `tryFinalize` call site used to do; lookups are O(1) afterwards. // See `buildReexportClosures` for the algorithm. - const reexportClosures = buildReexportClosures(input.files, byFilePath, edgeIndex); + // A local import is a dependency of its file, not a re-export of that file. + const moduleEdgeIndex = new Map(); + for (const file of input.files) { + moduleEdgeIndex.set( + file.filePath, + (edgeIndex.get(file.filePath) ?? []).filter((d) => d.fromScope === file.moduleScope), + ); + } + const ambiguityByFile = collectAmbiguityByFile( + input.files, + byFilePath, + moduleEdgeIndex, + hooks.wildcardCollisionIsAmbiguous === true, + hooks.namedImportsBindTopLevelOnly === true, + ); + const ambiguousByFile = new Map>(); + for (const [filePath, byName] of ambiguityByFile) { + ambiguousByFile.set(filePath, new Set(byName.keys())); + } + const topLevelOnly = hooks.namedImportsBindTopLevelOnly === true; + const reexportClosures = buildReexportClosures( + input.files, + byFilePath, + moduleEdgeIndex, + ambiguousByFile, + topLevelOnly, + ); // ── Phase 3: process SCCs in reverse-topological order (leaves first). // Within each SCC, run a bounded fixpoint that resolves intra-SCC edges. @@ -231,6 +310,19 @@ export function finalize(input: FinalizeInput, hooks: FinalizeHooks): FinalizeOu // already finalized); edges inside the SCC may need multiple passes. const linkedByScope = new Map(); let linkedEdges = 0; + // Every refused wildcard name, reported from the ambiguity map rather than from + // the edges phase 4 happens to drop: a language whose `expandsWildcardTo` + // returns nothing (TypeScript — `export *` never binds names locally) drops + // no expanded edge, yet its importers were refused through the closure just + // the same, and that refusal must still be visible. Named-vs-named conflicts + // remain refused above but are not export-star collisions in this audit. + const ambiguousWildcardExports: AmbiguousWildcardExport[] = []; + for (const [filePath, byName] of ambiguityByFile) { + for (const [name, ambiguity] of byName) { + if (ambiguity.kind !== 'wildcard') continue; + ambiguousWildcardExports.push({ filePath, name, candidateDefIds: ambiguity.candidateDefIds }); + } + } for (const scc of sccs) { const sccFiles = new Set(scc.files); @@ -249,7 +341,7 @@ export function finalize(input: FinalizeInput, hooks: FinalizeHooks): FinalizeOu if (drafts === undefined) continue; for (const draft of drafts) { if (draft.finalized !== null) continue; - const finalized = tryFinalize(draft, byFilePath, reexportClosures); + const finalized = tryFinalize(draft, byFilePath, reexportClosures, topLevelOnly); if (finalized !== null) { draft.finalized = finalized; progressed = true; @@ -272,13 +364,25 @@ export function finalize(input: FinalizeInput, hooks: FinalizeHooks): FinalizeOu } } - // ── Phase 4: collect finalized `ImportEdge[]` per module scope, preserving + // ── Phase 4: collect finalized `ImportEdge[]` per binding scope, preserving // input order within each file, and wildcard-expand where applicable. for (const file of input.files) { const drafts = edgeIndex.get(file.filePath); if (drafts === undefined) continue; - const finalized: ImportEdge[] = []; + const finalizedByScope = new Map([[file.moduleScope, []]]); + // Names this file's `export *` sources collide on (see + // `collectAmbiguousWildcards`). Their expanded edges are dropped here, so + // the file's own module scope does not bind an arbitrary winner either — + // suppressing them only in the closure would leave this binding standing, + // and it was this binding, not the closure, that produced the published + // `import-resolved` guess. + const ambiguousHere = ambiguousByFile.get(file.filePath) ?? EMPTY_NAME_SET; for (const d of drafts) { + let finalized = finalizedByScope.get(d.fromScope); + if (finalized === undefined) { + finalized = []; + finalizedByScope.set(d.fromScope, finalized); + } const edge = d.finalized; if (edge === null) { throw new Error(`Invariant violated: import edge was not finalized for ${file.filePath}`); @@ -286,16 +390,25 @@ export function finalize(input: FinalizeInput, hooks: FinalizeHooks): FinalizeOu if (d.source.kind === 'wildcard' && edge.linkStatus !== 'unresolved') { // Produce one `wildcard-expanded` ImportEdge per exported name. const expanded = expandWildcard(edge, byFilePath, hooks, input.workspaceIndex); - for (const e of expanded) finalized.push(e); + for (const e of expanded) { + if ( + d.fromScope === file.moduleScope && + e.kind === 'wildcard-expanded' && + ambiguousHere.has(e.localName) + ) + continue; + finalized.push(e); + } } else { finalized.push(edge); } if (edge.linkStatus !== 'unresolved') linkedEdges++; } - linkedByScope.set(file.moduleScope, Object.freeze(finalized)); + for (const [scopeId, edges] of finalizedByScope) + linkedByScope.set(scopeId, Object.freeze(edges)); } - // ── Phase 5: materialize module-scope bindings (local + imports + wildcards), + // ── Phase 5: materialize bindings (local + scoped imports + wildcards), // delegating precedence to `provider.mergeBindings`. const bindingsByScope = materializeBindings(input.files, linkedByScope, hooks); @@ -312,6 +425,7 @@ export function finalize(input: FinalizeInput, hooks: FinalizeHooks): FinalizeOu unresolvedEdges: totalEdges - linkedEdges, sccCount, largestSccSize, + ambiguousWildcardExports: Object.freeze(ambiguousWildcardExports), }; return Object.freeze({ @@ -339,6 +453,10 @@ function makeEdgeDrafts( hooks: FinalizeHooks, workspace: WorkspaceIndex, ): ImportEdgeDraft[] { + const fromScope = + hooks.importsBindAtLexicalScope === true + ? (parsed.declaredAtScope ?? file.moduleScope) + : file.moduleScope; // Dynamic-unresolved passes through — no `BindingRef`, no target file. if (parsed.kind === 'dynamic-unresolved') { const base: ImportEdge = { @@ -351,7 +469,7 @@ function makeEdgeDrafts( { source: parsed, fromFile: file.filePath, - fromScope: file.moduleScope, + fromScope, targetFile: null, base, finalized: base, // already fully finalized @@ -381,7 +499,7 @@ function makeEdgeDrafts( { source: parsed, fromFile: file.filePath, - fromScope: file.moduleScope, + fromScope, targetFile: null, base, finalized: base, @@ -417,7 +535,7 @@ function makeEdgeDrafts( return { source: parsed, fromFile: file.filePath, - fromScope: file.moduleScope, + fromScope, targetFile: tf, base, finalized: isFileLevelTerminal ? base : null, @@ -529,6 +647,7 @@ function tryFinalize( draft: ImportEdgeDraft, byFilePath: Map, reexportClosures: ReadonlyMap, + topLevelOnly: boolean, ): ImportEdge | null { const targetFile = draft.targetFile; if (targetFile === null) return draft.base; // already terminal @@ -552,7 +671,11 @@ function tryFinalize( // so consumers can reach the module as a symbol — but its absence is not // a failure. if (draft.base.kind === 'namespace') { - const moduleDef = findExportByName(targetModule.localDefs, extractExportedName(draft.source)); + const moduleDef = findExportByName( + targetModule.localDefs, + extractExportedName(draft.source), + topLevelOnly, + ); return { ...draft.base, targetModuleScope: targetModule.moduleScope, @@ -564,7 +687,7 @@ function tryFinalize( // local defs. Multi-hop re-export chains settle iteratively — each hop // resolves once its prior hop is finalized. const importedName = extractExportedName(draft.source); - const exported = findExportByName(targetModule.localDefs, importedName); + const exported = findExportByName(targetModule.localDefs, importedName, topLevelOnly); if (exported !== undefined) { const transitiveVia = @@ -691,15 +814,18 @@ function buildReexportClosures( files: readonly FinalizeFile[], byFilePath: ReadonlyMap, edgeIndex: ReadonlyMap, + ambiguous: ReadonlyMap>, + topLevelOnly: boolean, ): ReadonlyMap { const closures = new Map>(); for (const file of files) closures.set(file.filePath, new Map()); // ── Step 1: build the re-export sub-graph (only resolvable wildcard / - // reexport / flagged-named targets contribute edges), and collect the - // per-file ambiguous names in the same walk. + // reexport / flagged-named targets contribute edges). The per-file + // ambiguous-name sets arrive precomputed (`collectAmbiguityByFile`) because + // phase 4 consults the same sets when it expands wildcards into module + // scope — one source of truth for "this name has no winner". const subGraph = new Map>(); - const ambiguous = new Map>(); for (const file of files) { const targets = new Set(); const drafts = edgeIndex.get(file.filePath); @@ -710,7 +836,6 @@ function buildReexportClosures( if (!byFilePath.has(d.targetFile)) continue; targets.add(d.targetFile); } - ambiguous.set(file.filePath, collectAmbiguousReexports(drafts, byFilePath)); } subGraph.set(file.filePath, targets); } @@ -726,7 +851,7 @@ function buildReexportClosures( if (!scc.isCycle) { const filePath = scc.files[0]; if (filePath !== undefined) { - populateFileClosure(filePath, byFilePath, edgeIndex, closures, ambiguous); + populateFileClosure(filePath, byFilePath, edgeIndex, closures, ambiguous, topLevelOnly); } continue; } @@ -740,7 +865,9 @@ function buildReexportClosures( progressed = false; iter++; for (const filePath of scc.files) { - if (populateFileClosure(filePath, byFilePath, edgeIndex, closures, ambiguous)) { + if ( + populateFileClosure(filePath, byFilePath, edgeIndex, closures, ambiguous, topLevelOnly) + ) { progressed = true; } } @@ -820,6 +947,220 @@ function isNamedReexport(draft: ImportEdgeDraft): draft is ImportEdgeDraft & { * are still filling in, so detecting them needs a set that grows during the * fixpoint — the thing this pre-pass exists to avoid. */ +/** + * Per-file set of re-exported names that have NO decidable winner, from both + * detectors: `collectAmbiguousReexports` (flagged-named vs flagged-named) and + * `collectAmbiguousWildcards` (`export *` vs `export *`, direct declarations). + * Fixed for the whole run; consulted by the closure fixpoint AND by phase 4's + * wildcard expansion, so a refused name is absent from BOTH the exports an + * importer can reach and the module-scope bindings the file itself sees. + */ +interface ReexportAmbiguity { + readonly kind: 'named' | 'wildcard'; + readonly candidateDefIds: readonly string[]; +} + +function collectAmbiguityByFile( + files: readonly FinalizeFile[], + byFilePath: ReadonlyMap, + edgeIndex: ReadonlyMap, + wildcardCollisionIsAmbiguous: boolean, + topLevelOnly: boolean, +): ReadonlyMap> { + const out = new Map>(); + for (const file of files) { + const drafts = edgeIndex.get(file.filePath); + if (drafts === undefined) continue; + const byName = new Map(); + for (const name of collectAmbiguousReexports(drafts, byFilePath)) { + byName.set(name, { + kind: 'named', + candidateDefIds: namedReexportCandidates(drafts, byFilePath, name, topLevelOnly), + }); + } + // Wildcard-vs-wildcard is a language rule (`FinalizeHooks. + // wildcardCollisionIsAmbiguous`): ECMAScript excludes the name, C++ + // overloads it. Without the opt-in this half stays first-wins. + if (wildcardCollisionIsAmbiguous) { + for (const [name, ids] of collectAmbiguousWildcards(file, drafts, byFilePath)) { + if (!byName.has(name)) byName.set(name, { kind: 'wildcard', candidateDefIds: ids }); + } + } + if (byName.size === 0) continue; + out.set(file.filePath, byName); + } + return out; +} + +/** The declarations a flagged-named collision on `localName` points at. */ +function namedReexportCandidates( + drafts: readonly ImportEdgeDraft[], + byFilePath: ReadonlyMap, + localName: string, + topLevelOnly: boolean, +): readonly string[] { + const ids: string[] = []; + for (const draft of drafts) { + if (!isNamedReexport(draft) || draft.source.localName !== localName) continue; + const targetFile = draft.targetFile; + if (targetFile === null) continue; + const target = byFilePath.get(targetFile); + if (target === undefined) continue; + const def = findExportByName(target.localDefs, draft.source.importedName, topLevelOnly); + if (def !== undefined && !ids.includes(def.nodeId)) ids.push(def.nodeId); + } + return Object.freeze(ids); +} + +/** + * `export * from './a'; export * from './b'` where BOTH `a` and `b` declare + * `collide`: the language names no winner (ECMAScript excludes the name from + * the module's exports entirely; a direct `import { collide }` of it is a + * SyntaxError-class ambiguity). First-wins here published the `a` binding as + * `import-resolved` at full confidence — a definite target for a call that has + * none, which is the incorrect-context-over-missing-context failure in its + * purest form. The name is refused instead and reported. + * + * Decidable in this pre-pass because it reads only the targets' own + * `localDefs` — nothing that fills in during the closure fixpoint. Collisions + * that arrive TRANSITIVELY (two wildcards whose targets each re-export the + * name from somewhere else) are still first-wins; detecting them needs a set + * that grows mid-fixpoint, the thing this pre-pass exists to avoid. + * + * A name the file DECLARES itself, or re-exports by NAME, is excluded: an + * explicit export shadows every `export *`, so those collisions are legal and + * resolved by precedence, not ambiguous. + * + * Only MODULE-LEVEL, EXPORT-SHAPED declarations can collide. `localDefs` also + * carries class members, properties and parameters (a `Property:value` on two + * unrelated classes, an interface field named `move`), which no `export *` + * publishes. Counting those produced thousands of phantom collisions on a real + * monorepo (2,640 on grafana) and — the dangerous half — would have refused a + * genuinely exported `move()` because some class elsewhere had a `move` + * property. The wildcard closure loop tolerates the wider set because nobody + * imports a property by name; a refusal cannot afford the same tolerance. + * + * Export evidence, when the language supplies it (`SymbolDefinition.isExported`, + * tri-state), settles the rest: a def marked `false` is module-private and is + * neither a provider here nor published by the closure + * (`indexTopLevelExportsByName`), so a private `function foo` beside an exported + * one no longer refuses the export — and, the half that matters more, cannot be + * the first-listed winner the closure binds either. A def marked `true` counts + * whatever its label, `Variable` included: the closure publishes a `Variable`, + * so two sources each exporting `const alpha` are a real collision and must be + * refused rather than first-wins. + * + * Without evidence (`isExported` undefined — most languages) `Variable` is + * excluded from the COLLISION set only: the typical top-level `const` in a + * barrel's sources is module-private (`const category = ['Axis']` in fourteen + * option-builder files), so counting it would refuse a real exported constant + * of the same name for nothing. Residual risk, accepted, for that evidence-free + * case: a non-exported `function`/`class` sharing its name with an exported one + * behind the same barrel is counted as a collision and the export is refused — + * a missing edge, never a wrong one. `ownerId` is only set for class members, so + * a callable nested in an object literal (`showIf: (cfg) => …` across fourteen + * option-builder files) still counts as a provider when unmarked. Measured + * before the export marker existed: grafana@871af0720 refuses 52 names (from + * 2,640 before the member exclusion), discourse@3f71fa15c 5. + */ + +/** Labels that are never a module export, whatever their owner. */ +const NON_EXPORTABLE_MEMBER_LABELS: readonly string[] = [ + 'Property', + 'Method', + 'Constructor', + 'Parameter', + 'Field', +]; +/** Labels excluded from the collision set when no export evidence is present. */ +const UNMARKED_NON_COLLIDING_LABELS: ReadonlySet = new Set([ + ...NON_EXPORTABLE_MEMBER_LABELS, + 'Variable', +]); +/** + * Labels a module can never export by name: class/interface members and + * parameters. Filtered by LABEL, not `ownerId` — `ownerId` is populated in a + * later pass and is not reliable while the closure is built. + */ +const MEMBER_LABELS: ReadonlySet = new Set(NON_EXPORTABLE_MEMBER_LABELS); + +/** + * A declaration `export *` could publish, for COLLISION purposes: top-level, of + * an exportable kind, and not marked module-private. With export evidence the + * label rule yields to the marker (an exported `Variable` collides; a private + * `function` does not); without it `Variable` is left out — see the header. + */ +function isWildcardPublishable(def: SymbolDefinition): boolean { + // Explicit evidence wins over the label: a CommonJS `module.exports = { + // alpha() {} }` member is labeled Method and IS the module's export. + if (def.isExported === true) return true; + if (def.isExported === false) return false; + if (def.ownerId !== undefined) return false; + return !UNMARKED_NON_COLLIDING_LABELS.has(def.type); +} + +/** + * Can a declaration of the barrel's OWN shadow a name its `export *` sources + * collide on? Only a module-level binding can — ECMAScript's explicit-export + * precedence is about the module's own exports. A class MEMBER named `clash` + * (`export class Unrelated { clash() {} }`) is not such a binding and must not + * switch the collision check off; it did, and a confident edge to one source's + * `clash` was emitted where the import should have been refused. + */ +function canShadowWildcard(def: SymbolDefinition): boolean { + if (def.isExported === true) return true; + if (def.isExported === false) return false; + if (def.ownerId !== undefined) return false; + return !MEMBER_LABELS.has(def.type); +} +function collectAmbiguousWildcards( + file: FinalizeFile, + drafts: readonly ImportEdgeDraft[], + byFilePath: ReadonlyMap, +): ReadonlyMap { + const shadowed = new Set(); + for (const def of file.localDefs) { + if (!canShadowWildcard(def)) continue; + const name = deriveSimpleName(def); + if (name !== null) shadowed.add(name); + } + for (const draft of drafts) { + if (isNamedReexport(draft)) shadowed.add(draft.source.localName); + } + + // name → (target file → declaring def ids), in declaration order. + const providers = new Map>(); + for (const draft of drafts) { + if (draft.source.kind !== 'wildcard') continue; + const targetFile = draft.targetFile; + if (targetFile === null) continue; + const target = byFilePath.get(targetFile); + if (target === undefined) continue; + for (const def of target.localDefs) { + if (!isWildcardPublishable(def)) continue; + const name = deriveSimpleName(def); + if (name === null || shadowed.has(name)) continue; + let byTarget = providers.get(name); + if (byTarget === undefined) { + byTarget = new Map(); + providers.set(name, byTarget); + } + const ids = byTarget.get(targetFile); + if (ids === undefined) byTarget.set(targetFile, [def.nodeId]); + else ids.push(def.nodeId); + } + } + const conflicting = new Map(); + for (const [name, byTarget] of providers) { + // Two DIFFERENT source files declaring the name. The same file declaring + // it twice (overloads, a declaration merged with its namespace) is one + // provider and not a collision. + if (byTarget.size < 2) continue; + conflicting.set(name, Object.freeze([...byTarget.values()].flat())); + } + return conflicting; +} + function collectAmbiguousReexports( drafts: readonly ImportEdgeDraft[], byFilePath: ReadonlyMap, @@ -859,6 +1200,7 @@ function populateFileClosure( edgeIndex: ReadonlyMap, closures: Map>, ambiguousByFile: ReadonlyMap>, + topLevelOnly: boolean, ): boolean { const myClosure = closures.get(filePath); if (myClosure === undefined) return false; @@ -883,7 +1225,7 @@ function populateFileClosure( if (ambiguous.has(localName) || myClosure.has(localName)) continue; const importedName = draft.source.importedName; - const direct = findExportByName(targetModule.localDefs, importedName); + const direct = findExportByName(targetModule.localDefs, importedName, topLevelOnly); if (direct !== undefined) { myClosure.set(localName, { def: direct, via: Object.freeze([targetFile]) }); continue; @@ -909,9 +1251,27 @@ function populateFileClosure( const targetModule = byFilePath.get(targetFile); if (targetModule === undefined) continue; - for (const def of targetModule.localDefs) { - const name = deriveSimpleName(def); - if (name === null || ambiguous.has(name) || myClosure.has(name)) continue; + // Fan out the WINNER per name, not every def. `export const alpha = () => + // {}` emits both a `Variable` (the lexical declaration) and a `Function` + // (the arrow) under the same simple name; iterating `localDefs` raw let + // whichever came first — the `Variable` — claim the closure slot, and a + // call bound to a value shadow emits no CALLS edge. Named re-exports + // already go through `findExportByName`'s callable-preferred index; the + // wildcard hop is the same lookup and must apply the same preference. + // Measured: grafana `Button`/`clearButtonStyles` (arrow consts behind + // `export *`) resolved 8 of 475 ledger entries before this. + // Over TOP-LEVEL declarations only. `localDefs` also carries class members; + // `Foo.render` (label `Method`, callable) outranked the file's real + // `const render` in the callable-preferred index and `import { render }` + // bound to a symbol `export *` can never publish — a confident wrong edge + // where the value shadow used to yield none. Gated by the same hook as the + // named-import path: only a language that opted in (ECMAScript, where + // `export *` cannot publish a class member) narrows; every other language's + // wildcard keeps the wide index, whose members are legitimately reachable. + for (const [name, def] of (topLevelOnly ? indexTopLevelExportsByName : indexExportsByName)( + targetModule.localDefs, + )) { + if (ambiguous.has(name) || myClosure.has(name)) continue; myClosure.set(name, { def, via: Object.freeze([targetFile]) }); } const targetClosure = closures.get(targetFile); @@ -993,6 +1353,13 @@ function deriveSimpleName(def: SymbolDefinition): string | null { function findExportByName( defs: readonly SymbolDefinition[], name: string, + /** + * `true` (a `namedImportsBindTopLevelOnly` language): consult only + * module-level declarations, so a class member can neither outrank a + * top-level value nor bind on its own. Phase-4 wildcard expansion keeps + * the wide index — that is the path languages use to bind members. + */ + topLevelOnly: boolean = false, ): SymbolDefinition | undefined { // GENERIC RULE (applies to every language using this finalize // algorithm): when MULTIPLE `SymbolDefinition`s share the same simple @@ -1018,7 +1385,7 @@ function findExportByName( // // See `gitnexus/test/integration/resolvers/typescript-hof-callbacks.test.ts` // for the cross-file regression this rule prevents. - return indexExportsByName(defs).get(name); + return (topLevelOnly ? indexTopLevelExportsByName(defs) : indexExportsByName(defs)).get(name); } /** @@ -1062,6 +1429,47 @@ function indexExportsByName( return index; } +/** + * `indexExportsByName` restricted to declarations a module publishes by name: + * members (by LABEL — `ownerId` is stamped by a later reconcile pass and is not + * reliable while the closure is built) are skipped unless the language marked + * them exported (a CommonJS `module.exports = { alpha() {} }` member), and so + * is any def the language marked module-private (`isExported === false`) — a + * function nested inside another function carries the Function label and used + * to displace the real exported value of the same name here; a barrel cannot + * republish what its source never exported, and binding it would put a private + * `function foo` in front of the exported one another source provides. + * `Variable` stays, since a barrel legitimately republishes a `const`. Same + * memoization contract. + */ +const TOP_LEVEL_EXPORTS_BY_NAME = new WeakMap< + readonly SymbolDefinition[], + ReadonlyMap +>(); + +function indexTopLevelExportsByName( + defs: readonly SymbolDefinition[], +): ReadonlyMap { + const cached = TOP_LEVEL_EXPORTS_BY_NAME.get(defs); + if (cached !== undefined) return cached; + const index = new Map(); + for (const d of defs) { + // Evidence over label, both ways: a marked-private def (a function nested + // in another function carries the Function label too) is skipped, and a + // marked-exported member (`module.exports = { alpha() {} }`) is admitted. + if (d.isExported === false) continue; + if (d.isExported !== true && MEMBER_LABELS.has(d.type)) continue; + const name = deriveSimpleName(d); + if (name === null) continue; + const existing = index.get(name); + if (existing === undefined) index.set(name, d); + else if (!isCallableOrTypeLike(existing.type) && isCallableOrTypeLike(d.type)) + index.set(name, d); + } + TOP_LEVEL_EXPORTS_BY_NAME.set(defs, index); + return index; +} + const EMPTY_NAME_SET: ReadonlySet = new Set(); const CALLABLE_OR_TYPE_LIKE: ReadonlySet = new Set([ @@ -1179,7 +1587,7 @@ function materializeBindings( linkedByScope: ReadonlyMap, hooks: FinalizeHooks, ): ReadonlyMap> { - const out = new Map>(); + const buckets = new Map>(); // Build a `nodeId → SymbolDefinition` index once across all files // (O(N_files × D_defs)) so the per-edge lookup below is O(1) instead @@ -1204,8 +1612,17 @@ function materializeBindings( scopeBindings.set(name, hooks.mergeBindings(existing, incoming, file.moduleScope)); } - // Layer in finalized imports. - const imports = linkedByScope.get(file.moduleScope) ?? []; + buckets.set(file.moduleScope, scopeBindings); + } + + // Layer imports into their binding scope; lexical locals already live in + // the scope tree and must not be copied into unrelated scope buckets. + for (const [scopeId, imports] of linkedByScope) { + let scopeBindings = buckets.get(scopeId); + if (scopeBindings === undefined) { + scopeBindings = new Map(); + buckets.set(scopeId, scopeBindings); + } for (const edge of imports) { if (edge.targetDefId === undefined || edge.linkStatus === 'unresolved') continue; const def = defById.get(edge.targetDefId); @@ -1224,15 +1641,18 @@ function materializeBindings( if (name === null) continue; const incoming: BindingRef[] = [{ def, origin, via: edge }]; const existing = scopeBindings.get(name) ?? []; - scopeBindings.set(name, hooks.mergeBindings(existing, incoming, file.moduleScope)); + scopeBindings.set(name, hooks.mergeBindings(existing, incoming, scopeId)); } + } + const out = new Map>(); + for (const [scopeId, scopeBindings] of buckets) { // Freeze nested buckets for immutability. const frozen = new Map(); for (const [name, refs] of scopeBindings) { frozen.set(name, Object.freeze(refs.slice())); } - out.set(file.moduleScope, frozen); + out.set(scopeId, frozen); } return out; diff --git a/gitnexus-shared/src/scope-resolution/language-classification.ts b/gitnexus-shared/src/scope-resolution/language-classification.ts index 20059c3e9..2e635c8e2 100644 --- a/gitnexus-shared/src/scope-resolution/language-classification.ts +++ b/gitnexus-shared/src/scope-resolution/language-classification.ts @@ -9,7 +9,8 @@ * Initial classification (locked in Ring 1 #910): * - production: javascript, typescript, python, java, c, cpp, csharp, go, * ruby, rust, php, kotlin, swift, dart - * - experimental: vue (embedded-language / SFC complexity), + * - experimental: objective-c (fork provider MVP), + * vue (embedded-language / SFC complexity), * cobol (regex-provider path) * - quarantined: (none) * @@ -34,6 +35,7 @@ export const LanguageClassifications: Readonly = new Set = new Set([ 'Class', + 'Protocol', + 'Category', 'Interface', 'Enum', 'Struct', diff --git a/gitnexus-shared/src/scope-resolution/symbol-definition.ts b/gitnexus-shared/src/scope-resolution/symbol-definition.ts index e90b0be85..4334d1534 100644 --- a/gitnexus-shared/src/scope-resolution/symbol-definition.ts +++ b/gitnexus-shared/src/scope-resolution/symbol-definition.ts @@ -111,6 +111,18 @@ export interface SymbolDefinition { * source (for example an anonymous class). Consumers may use this only as a * conservative priority hint; it does not change graph-node identity. */ isSynthetic?: boolean; + /** + * Whether the producing language saw EXPORT EVIDENCE on this declaration — + * an `export` modifier, a later `export { name }` specifier, an `export + * default name`. TRI-STATE, and the absence is load-bearing: `undefined` + * means the language emitted no verdict (most languages, and any ECMAScript + * file whose export surface is a CommonJS assignment the marker cannot read), + * which readers MUST treat as "unknown" and fall back to their prior + * behavior. Only `false` says "this module does not publish the name": a + * `false` keeps a module-private `function foo` from being counted as a + * wildcard provider or bound through a barrel's `export *` closure. + */ + isExported?: boolean; /** Links Method/Constructor/Property to owning Class/Struct/Trait nodeId */ ownerId?: string; /** #1982/#1993: bridge-held enclosing-namespace path (e.g. `NS1`, `Outer.Inner`) diff --git a/gitnexus-shared/src/scope-resolution/types.ts b/gitnexus-shared/src/scope-resolution/types.ts index 94a085daa..50a968146 100644 --- a/gitnexus-shared/src/scope-resolution/types.ts +++ b/gitnexus-shared/src/scope-resolution/types.ts @@ -107,7 +107,14 @@ export type CaptureMatch = Readonly>; * produced when `expandsWildcardTo` materializes a wildcard against target * exports — a provider must never emit it at parse time. */ -export type ParsedImport = +export type ParsedImport = ParsedImportSyntax & { + /** Lexical location retained by extraction, independently of execution timing. + * Absent for legacy or synthesized imports. Binding semantics remain opt-in + * through FinalizeHooks.importsBindAtLexicalScope. */ + readonly declaredAtScope?: ScopeId; +}; + +type ParsedImportSyntax = /** * Per-name import without rename. * @@ -183,14 +190,15 @@ export type ParsedImport = * field exists.** The natural place to decide it looks like the graph * bridge, by walking the scope the finalized edges hang off; that is * exactly what `graph-bridge/imports-to-edges.ts` once attempted, and it - * is dead code by construction. `finalize-algorithm.ts:295` publishes - * every file's finalized edges as - * `linkedByScope.set(file.moduleScope, …)`, so the map the bridge - * receives is keyed by the file's **Module** scope and by nothing else: - * the walk starts at a `Module` every time and answers `false` for every - * import in the tree. Finalize cannot recover the position either — - * `FinalizeFile.parsedImports` is a flat per-file `ParsedImport[]` with - * no scope attached. The extractor is the last stage that still knows + * is dead code by construction. Default finalize still publishes each + * file's edges under `file.moduleScope`, so a bridge walk that starts + * there answers `false` for every import. Resolvers that opt into + * `importsBindAtLexicalScope` instead key the edge by + * `parsed.declaredAtScope` when extraction placed the statement (Rust + * `use` in a function body is the example). Even then the *execution* + * question — does this import run at module init? — is not the same as + * the lexical key, and `FinalizeFile.parsedImports` is still a flat + * per-file list. The extractor is the last stage that still knows * where the statement sat (`scope-extractor.ts`, Pass 3), so it marks the * fact here and it rides the edge from there — see * {@link ImportEdge.runsOnlyWhenCalled}. diff --git a/gitnexus-web/package-lock.json b/gitnexus-web/package-lock.json index 282a9027b..90271c2bd 100644 --- a/gitnexus-web/package-lock.json +++ b/gitnexus-web/package-lock.json @@ -11,7 +11,7 @@ "@langchain/anthropic": "^1.5.8", "@langchain/core": "^1.2.8", "@langchain/google-genai": "^2.2.0", - "@langchain/langgraph": "^1.4.9", + "@langchain/langgraph": "^1.4.14", "@langchain/ollama": "^1.3.0", "@langchain/openai": "^1.5.3", "@sigma/edge-curve": "^3.1.0", @@ -36,7 +36,7 @@ "pandemonium": "^2.4.0", "react": "^19.2.5", "react-dom": "^19.2.8", - "react-i18next": "^17.0.12", + "react-i18next": "^17.0.13", "react-markdown": "^10.1.0", "react-syntax-highlighter": "^16.1.1", "react-zoom-pan-pinch": "^4.0.3", @@ -51,14 +51,14 @@ "@playwright/test": "^1.62.1", "@testing-library/jest-dom": "^7.0.0", "@testing-library/react": "^16.3.3", - "@testing-library/user-event": "^14.6.6", + "@testing-library/user-event": "^14.6.7", "@types/dompurify": "^3.2.0", "@types/node": "^26.0.1", - "@types/react": "^19.2.14", + "@types/react": "^19.2.18", "@types/react-dom": "^19.2.4", "@types/react-syntax-highlighter": "^15.5.13", "@vercel/node": "^5.10.2", - "@vitejs/plugin-react": "^6.0.5", + "@vitejs/plugin-react": "^6.1.1", "@vitest/coverage-v8": "^4.1.11", "jsdom": "^29.1.1", "tree-sitter-wasms": "^0.1.13", @@ -1165,14 +1165,14 @@ } }, "node_modules/@langchain/langgraph": { - "version": "1.4.9", - "resolved": "https://registry.npmjs.org/@langchain/langgraph/-/langgraph-1.4.9.tgz", - "integrity": "sha512-EvD9rS66Cya09y6rbMgD3Ir8miAkJQFo7FyJOPRPO736Kz3y5TeyeBDOS8ctff/jRc788bPijHx2NVFM79Qqig==", + "version": "1.4.14", + "resolved": "https://registry.npmjs.org/@langchain/langgraph/-/langgraph-1.4.14.tgz", + "integrity": "sha512-uWAdRYTllfKCnTrlyovExPJCHJwcf3Wl2LzUlnaqsT7Rmoo3aCeYtq/7MV/Pw4q11motG8pR8bjr6T6V8Pe1gQ==", "license": "MIT", "dependencies": { - "@langchain/langgraph-checkpoint": "^1.1.3", - "@langchain/langgraph-sdk": "~1.9.28", - "@langchain/protocol": "^0.0.18", + "@langchain/langgraph-checkpoint": "^1.1.5", + "@langchain/langgraph-sdk": "~1.10.2", + "@langchain/protocol": "^0.0.19", "@standard-schema/spec": "1.1.0" }, "engines": { @@ -1184,9 +1184,9 @@ } }, "node_modules/@langchain/langgraph-checkpoint": { - "version": "1.1.3", - "resolved": "https://registry.npmjs.org/@langchain/langgraph-checkpoint/-/langgraph-checkpoint-1.1.3.tgz", - "integrity": "sha512-wgzdQNeEsdw1e+4lvlj0tdq/RYR/k1vPin10g0ymGoehZDDgd9nvIllGXSXN4TFgF9sf5qQP/KTkOcLfeseIhA==", + "version": "1.1.5", + "resolved": "https://registry.npmjs.org/@langchain/langgraph-checkpoint/-/langgraph-checkpoint-1.1.5.tgz", + "integrity": "sha512-BwDwl5VeTOh6CVuiIPgsUgfK51vTJDMSbFcSCUfjJWsl8/DPdK/mbv+ejxJstkSk/BlSPMP4JfXWcN6jD2ea2Q==", "license": "MIT", "engines": { "node": ">=18" @@ -1196,12 +1196,12 @@ } }, "node_modules/@langchain/langgraph-sdk": { - "version": "1.9.28", - "resolved": "https://registry.npmjs.org/@langchain/langgraph-sdk/-/langgraph-sdk-1.9.28.tgz", - "integrity": "sha512-4j3XuM0PvtmAbL8mPfBS99ez3+ytRfgbOpAR/nOeaejTRF3Q9dNw2QnaGLGng8wLPtGLoSj+SYgUOVxy9Bv9vg==", + "version": "1.10.2", + "resolved": "https://registry.npmjs.org/@langchain/langgraph-sdk/-/langgraph-sdk-1.10.2.tgz", + "integrity": "sha512-86qsfdBZWu1ZgywLN8AThU/jXi9rjPDZPWcTJp4SA1A/L62ypTNoSXbvtiwZt1odokXccYTxK1XWS8tmVdvEmw==", "license": "MIT", "dependencies": { - "@langchain/protocol": "^0.0.18", + "@langchain/protocol": "^0.0.19", "@types/json-schema": "^7.0.15", "p-queue": "^9.0.1", "p-retry": "^7.1.1" @@ -1209,9 +1209,7 @@ "peerDependencies": { "@langchain/core": "^1.1.48", "react": "^18 || ^19", - "react-dom": "^18 || ^19", - "svelte": "^4.0.0 || ^5.0.0", - "vue": "^3.0.0" + "react-dom": "^18 || ^19" }, "peerDependenciesMeta": { "react": { @@ -1219,12 +1217,6 @@ }, "react-dom": { "optional": true - }, - "svelte": { - "optional": true - }, - "vue": { - "optional": true } } }, @@ -1295,9 +1287,9 @@ } }, "node_modules/@langchain/protocol": { - "version": "0.0.18", - "resolved": "https://registry.npmjs.org/@langchain/protocol/-/protocol-0.0.18.tgz", - "integrity": "sha512-XW1egQtPfsGI41w2AMZNFZrUIwFSQHTjVMZs0OaTpCAvht/QLoaPN8FQcsysMVypOhupG28J29yOorrc70otBQ==", + "version": "0.0.19", + "resolved": "https://registry.npmjs.org/@langchain/protocol/-/protocol-0.0.19.tgz", + "integrity": "sha512-9hKcRrH7cBX6gfutdfXPoft1OCchHe4FEpALoDJMl5Qu+n/YG5ynZmyu8+8cxORlPwHBoKTxggvXz+76M1yX1Q==", "license": "MIT" }, "node_modules/@mapbox/node-pre-gyp": { @@ -2101,9 +2093,9 @@ } }, "node_modules/@testing-library/user-event": { - "version": "14.6.6", - "resolved": "https://registry.npmjs.org/@testing-library/user-event/-/user-event-14.6.6.tgz", - "integrity": "sha512-Jbs9FpkkIDw8FgSc6kOVsOv8JuuqGAL7J4X1oot77JxAoDlkNn2GRkd0aYRVuQ+pVQAiHWVkE4rX/dkF5fBiCw==", + "version": "14.6.7", + "resolved": "https://registry.npmjs.org/@testing-library/user-event/-/user-event-14.6.7.tgz", + "integrity": "sha512-MPCpX8bxe8zS+JmmTwLp8jd0dy1rAm60Te/SL8JrQM3qvQJcBOs1d7IefJMyZzqM3EWBrDn/LWDt1BCGu4ASfg==", "dev": true, "license": "MIT", "engines": { @@ -2535,9 +2527,9 @@ "license": "MIT" }, "node_modules/@types/react": { - "version": "19.2.14", - "resolved": "https://registry.npmjs.org/@types/react/-/react-19.2.14.tgz", - "integrity": "sha512-ilcTH/UniCkMdtexkoCN0bI7pMcJDvmQFPvuPvmEaYA/NSfFTAgdUSLAoVjaRJm7+6PvcM+q1zYOwS4wTYMF9w==", + "version": "19.2.18", + "resolved": "https://registry.npmjs.org/@types/react/-/react-19.2.18.tgz", + "integrity": "sha512-AnzbBERsrLKtk2XSfTbYRLjQPdy116Sty4q+T+Bp3IC4l6jNBvreVPAHmpq9qhXQM7CXZPjLVmGMw9sy+hxQ3w==", "license": "MIT", "dependencies": { "csstype": "^3.2.2" @@ -2716,9 +2708,9 @@ } }, "node_modules/@vitejs/plugin-react": { - "version": "6.0.5", - "resolved": "https://registry.npmjs.org/@vitejs/plugin-react/-/plugin-react-6.0.5.tgz", - "integrity": "sha512-BOVzne/NL162sMdResB25mUv+vWMF5NoAjNf09TeGlE7ZpszZWSD3winycicLJw72yeVsoCn/2kOhEuCvEShMA==", + "version": "6.1.1", + "resolved": "https://registry.npmjs.org/@vitejs/plugin-react/-/plugin-react-6.1.1.tgz", + "integrity": "sha512-yxLaQV9gkhS8ezJqCM6+ndU7mDY6gqAg75NQ+0IjwEI8IYOmQCgkRwHKVSfWXW076DsqMo0Dk+0FK1U+M5RgFw==", "dev": true, "license": "MIT", "dependencies": { @@ -2730,6 +2722,7 @@ "peerDependencies": { "@rolldown/plugin-babel": "^0.1.7 || ^0.2.0", "babel-plugin-react-compiler": "^1.0.0", + "oxc-transform-react": "^0.145.0", "vite": "^8.0.0" }, "peerDependenciesMeta": { @@ -2738,6 +2731,9 @@ }, "babel-plugin-react-compiler": { "optional": true + }, + "oxc-transform-react": { + "optional": true } } }, @@ -5077,9 +5073,9 @@ } }, "node_modules/joi": { - "version": "18.2.3", - "resolved": "https://registry.npmjs.org/joi/-/joi-18.2.3.tgz", - "integrity": "sha512-N5A3KTWQpPWT4ExxxPlUx7WmykGXRzhNidWhV41d6Abu9YfI2NyWCJuxdPnslJCPWtbRpSVOWSnSS6GakLM/Rg==", + "version": "18.2.8", + "resolved": "https://registry.npmjs.org/joi/-/joi-18.2.8.tgz", + "integrity": "sha512-G2TX62h58ZHuwqetJgP2F4ualakqAmZtBYe3jWen7gxQRw5xApX6crnFtuB91WC0c3ESBnva+kGSnb3+6pIQDQ==", "dev": true, "license": "BSD-3-Clause", "dependencies": { @@ -7304,9 +7300,9 @@ } }, "node_modules/react-i18next": { - "version": "17.0.12", - "resolved": "https://registry.npmjs.org/react-i18next/-/react-i18next-17.0.12.tgz", - "integrity": "sha512-lFWPEGkxQ6RhusdUkysFBD58VHfSSzvHBzqMgN0SvfVpdQGfwtNkStTqdy08/sJd7s807qqutgx93fRpD0DJ3Q==", + "version": "17.0.13", + "resolved": "https://registry.npmjs.org/react-i18next/-/react-i18next-17.0.13.tgz", + "integrity": "sha512-Cc1PscmblIHA1kljTqDwrcVMI21ydgmUzw0UAeQBe7pAOgfuRLfzXze4EUBQoeDiICzFIXXhHFoZxuetNg5D0Q==", "license": "MIT", "dependencies": { "@babel/runtime": "^7.29.7", diff --git a/gitnexus-web/package.json b/gitnexus-web/package.json index eef658136..fee794c47 100644 --- a/gitnexus-web/package.json +++ b/gitnexus-web/package.json @@ -21,7 +21,7 @@ "@langchain/anthropic": "^1.5.8", "@langchain/core": "^1.2.8", "@langchain/google-genai": "^2.2.0", - "@langchain/langgraph": "^1.4.9", + "@langchain/langgraph": "^1.4.14", "@langchain/ollama": "^1.3.0", "@langchain/openai": "^1.5.3", "@sigma/edge-curve": "^3.1.0", @@ -46,7 +46,7 @@ "pandemonium": "^2.4.0", "react": "^19.2.5", "react-dom": "^19.2.8", - "react-i18next": "^17.0.12", + "react-i18next": "^17.0.13", "react-markdown": "^10.1.0", "react-syntax-highlighter": "^16.1.1", "react-zoom-pan-pinch": "^4.0.3", @@ -61,14 +61,14 @@ "@playwright/test": "^1.62.1", "@testing-library/jest-dom": "^7.0.0", "@testing-library/react": "^16.3.3", - "@testing-library/user-event": "^14.6.6", + "@testing-library/user-event": "^14.6.7", "@types/dompurify": "^3.2.0", "@types/node": "^26.0.1", - "@types/react": "^19.2.14", + "@types/react": "^19.2.18", "@types/react-dom": "^19.2.4", "@types/react-syntax-highlighter": "^15.5.13", "@vercel/node": "^5.10.2", - "@vitejs/plugin-react": "^6.0.5", + "@vitejs/plugin-react": "^6.1.1", "@vitest/coverage-v8": "^4.1.11", "jsdom": "^29.1.1", "tree-sitter-wasms": "^0.1.13", diff --git a/gitnexus-web/src/components/CodeReferencesPanel.tsx b/gitnexus-web/src/components/CodeReferencesPanel.tsx index 5818614f9..4001d64f2 100644 --- a/gitnexus-web/src/components/CodeReferencesPanel.tsx +++ b/gitnexus-web/src/components/CodeReferencesPanel.tsx @@ -16,7 +16,7 @@ import { vscDarkPlus } from 'react-syntax-highlighter/dist/esm/styles/prism'; import { useAppState } from '../hooks/useAppState'; import { type GraphNode, getSyntaxLanguageFromFilename } from 'gitnexus-shared'; import { NODE_COLORS } from '../lib/constants'; -import { readFile, type ReadFileResult } from '../services/backend-client'; +import { BackendError, readFile, type ReadFileResult } from '../services/backend-client'; import { useTranslation } from 'react-i18next'; const getSyntaxLanguage = (filePath: string | undefined): string => { @@ -205,6 +205,7 @@ export const CodeReferencesPanel = ({ onFocusNode }: CodeReferencesPanelProps) = const CONTEXT_LINES = 50; // lines of context above and below the symbol const [fileResult, setFileResult] = useState(null); + const [sourceUnavailable, setSourceUnavailable] = useState(false); const [isLoadingFile, setIsLoadingFile] = useState(false); const selectedViewerRef = useRef(null); @@ -214,12 +215,14 @@ export const CodeReferencesPanel = ({ onFocusNode }: CodeReferencesPanelProps) = useEffect(() => { if (!selectedFilePath) { setFileResult(null); + setSourceUnavailable(false); return; } let cancelled = false; setIsLoadingFile(true); setFileResult(null); + setSourceUnavailable(false); // Determine read range: full file for File nodes, buffered for symbols const startLine = selectedNode?.properties?.startLine as number | undefined; @@ -242,9 +245,12 @@ export const CodeReferencesPanel = ({ onFocusNode }: CodeReferencesPanelProps) = setIsLoadingFile(false); } }) - .catch(() => { + .catch((error) => { if (!cancelled) { setFileResult(null); + setSourceUnavailable( + error instanceof BackendError && error.code === 'source_unavailable', + ); setIsLoadingFile(false); } }); @@ -384,7 +390,7 @@ export const CodeReferencesPanel = ({ onFocusNode }: CodeReferencesPanelProps) =
-
+
{isLoadingFile ? (
@@ -426,11 +432,11 @@ export const CodeReferencesPanel = ({ onFocusNode }: CodeReferencesPanelProps) = ) : (
- {selectedIsFile ? ( - <>{t('graph:codePanel.codeNotAvailable', { path: selectedFilePath })} - ) : ( - <>{t('graph:codePanel.selectFile')} - )} + {sourceUnavailable + ? t('graph:codePanel.sourceUnavailable') + : selectedIsFile + ? t('graph:codePanel.codeNotAvailable', { path: selectedFilePath }) + : t('graph:codePanel.selectFile')}
)}
@@ -457,7 +463,7 @@ export const CodeReferencesPanel = ({ onFocusNode }: CodeReferencesPanelProps) = {t('graph:codePanel.references', { count: aiReferences.length })}
-
+
{refsWithSnippets.map( ({ ref, content, start, highlightStart, highlightEnd, totalLines }) => { const nodeColor = ref.label diff --git a/gitnexus-web/src/components/FileTreePanel.tsx b/gitnexus-web/src/components/FileTreePanel.tsx index a3c7487a9..6f69205f4 100644 --- a/gitnexus-web/src/components/FileTreePanel.tsx +++ b/gitnexus-web/src/components/FileTreePanel.tsx @@ -186,6 +186,10 @@ const getNodeTypeIcon = (label: NodeLabel) => { return FileCode; case 'Class': return Box; + case 'Protocol': + return Hash; + case 'Category': + return Box; case 'Function': return Braces; case 'Method': @@ -389,7 +393,7 @@ export const FileTreePanel = ({ onFocusNode }: FileTreePanelProps) => {
{/* File tree */} -
+
{fileTree.length === 0 ? (
{t('graph:fileTree.noFilesLoaded')} @@ -413,7 +417,7 @@ export const FileTreePanel = ({ onFocusNode }: FileTreePanelProps) => { )} {activeTab === 'filters' && ( -
+

{t('graph:fileTree.nodeTypes')} diff --git a/gitnexus-web/src/lib/constants.ts b/gitnexus-web/src/lib/constants.ts index 5a85754cf..8e5259741 100644 --- a/gitnexus-web/src/lib/constants.ts +++ b/gitnexus-web/src/lib/constants.ts @@ -8,6 +8,8 @@ export const NODE_COLORS: Record = { Folder: '#6366f1', // Indigo File: '#3b82f6', // Blue Class: '#f59e0b', // Amber - stands out + Protocol: '#ec4899', // Pink - like Interface + Category: '#14b8a6', // Teal - like Method Function: '#10b981', // Emerald Method: '#14b8a6', // Teal Variable: '#64748b', // Slate - muted (less important) @@ -51,6 +53,8 @@ export const NODE_SIZES: Record = { Folder: 10, // Structural - clearly bigger than files File: 6, // Common element - smaller than folders Class: 8, // Important code structure + Protocol: 7, // Like Interface + Category: 3, // Like Method Function: 4, // Common code element - small Method: 3, // Smaller than function Variable: 2, // Tiny - leaf node @@ -115,6 +119,8 @@ export const DEFAULT_VISIBLE_LABELS: NodeLabel[] = [ 'Folder', 'File', 'Class', + 'Protocol', + 'Category', 'Function', 'Method', 'Property', // Kotlin/Java fields (HAS_PROPERTY + DEFINES File→Property) @@ -129,6 +135,8 @@ export const FILTERABLE_LABELS: NodeLabel[] = [ 'Folder', 'File', 'Class', + 'Protocol', + 'Category', 'Interface', 'Enum', 'Type', diff --git a/gitnexus-web/src/locales/en/graph.json b/gitnexus-web/src/locales/en/graph.json index 72c883456..cdb84aedd 100644 --- a/gitnexus-web/src/locales/en/graph.json +++ b/gitnexus-web/src/locales/en/graph.json @@ -110,7 +110,8 @@ "references_other": "{{count}} references", "lines_one": "{{count}} line", "lines_other": "{{count}} lines", - "codeNotAvailable": "Code not available in memory for {{path}}" + "codeNotAvailable": "Code not available in memory for {{path}}", + "sourceUnavailable": "Full source is unavailable for this index." }, "canvas": { "viewModes": { diff --git a/gitnexus-web/src/locales/zh-CN/graph.json b/gitnexus-web/src/locales/zh-CN/graph.json index 6dba980b7..f1e9e2afc 100644 --- a/gitnexus-web/src/locales/zh-CN/graph.json +++ b/gitnexus-web/src/locales/zh-CN/graph.json @@ -110,7 +110,8 @@ "references_other": "{{count}} 条引用", "lines_one": "{{count}} 行", "lines_other": "{{count}} 行", - "codeNotAvailable": "内存中没有 {{path}} 的代码内容" + "codeNotAvailable": "内存中没有 {{path}} 的代码内容", + "sourceUnavailable": "此索引无法提供完整源码。" }, "canvas": { "viewModes": { diff --git a/gitnexus-web/src/services/backend-client.ts b/gitnexus-web/src/services/backend-client.ts index 8378ef494..c96a0f596 100644 --- a/gitnexus-web/src/services/backend-client.ts +++ b/gitnexus-web/src/services/backend-client.ts @@ -23,6 +23,28 @@ export interface BackendRepo { repoPath?: string; // git HEAD returns "repoPath"; older versions return "path" indexedAt: string; lastCommit?: string; + /** + * Branch this index was built from. Absent on legacy entries and non-git + * repos. Since #3199 a branch-pinned analyze registers its own entry, so this + * is what tells two entries for the same repository apart — the name is + * derived from the clone directory and is not a contract. + */ + branch?: string; + /** Non-primary branch indexes recorded for the same path. */ + branches?: Array<{ branch: string; indexedAt?: string; lastCommit?: string }>; + /** + * Absent when the index is at the repo's checked-out HEAD. Otherwise `status` + * says what the server could establish: `behind` (with the counted + * `commitsBehind`), `diverged` (HEAD has moved off the indexed commit but the + * history needed to count the gap is gone, so there is no `commitsBehind`), + * or `unknown` (the repository could not be measured). Same shape MCP + * `list_repos` returns; see the server's `core/staleness-status.ts` (#3256). + */ + staleness?: { + status: 'behind' | 'diverged' | 'unknown'; + commitsBehind?: number; + hint?: string; + }; stats?: { files?: number; nodes?: number; @@ -97,6 +119,7 @@ export class BackendError extends Error { | 'server' | 'client' | 'not_found' + | 'source_unavailable' | 'timeout' | 'rate_limited' // The write-route same-host Origin guard rejected this request (HTTP 403 @@ -527,22 +550,22 @@ const assertOk = async (response: Response): Promise => { // Response body was not JSON } - const code = - response.status === 404 - ? 'not_found' - : response.status === 429 - ? 'rate_limited' - : // The public edge's token gate returns 401 with this discriminator; - // surface it as a distinct code so the UI can prompt for the token. - bodyCode === 'unauthorized' - ? 'unauthorized' - : // The write-route Origin guard returns 403 with this discriminator; - // surface it as a distinct code so the UI can give actionable guidance. - bodyCode === 'origin_not_allowed' - ? 'origin_blocked' - : response.status >= 400 && response.status < 500 - ? 'client' - : 'server'; + let code: ConstructorParameters[2] = 'server'; + if (bodyCode === 'source-unavailable') { + code = 'source_unavailable'; + } else if (response.status === 404) { + code = 'not_found'; + } else if (response.status === 429) { + code = 'rate_limited'; + } else if (bodyCode === 'unauthorized') { + // Public-edge token gate: HTTP 401 with this discriminator. + code = 'unauthorized'; + } else if (bodyCode === 'origin_not_allowed') { + // Write-route Origin guard: HTTP 403 with this discriminator. + code = 'origin_blocked'; + } else if (response.status >= 400 && response.status < 500) { + code = 'client'; + } // Retry-After is the standard HTTP signal for when the client may try again. // express-rate-limit emits it on 429 with seconds (integer) or HTTP-date. @@ -663,7 +686,14 @@ export type BackendProbeStatus = 'ok' | 'unauthorized' | 'unreachable'; */ export const probeBackendStatus = async (): Promise => { try { - const response = await fetchWithTimeout(`${_backendUrl}/api/repos`, {}, PROBE_TIMEOUT_MS); + // `/api/health` rather than `/api/repos`: this is a liveness question on a + // 2s budget, and `/api/repos` now spawns a `git rev-list` per registered + // repo to answer freshness. Probing it made the cost of "is the server up?" + // scale with the number of indexed repos, and a failed probe re-polls, + // stacking more children on the way (#3232 review). `/api/health` is a + // constant, and still sits behind the same `/api/*` edge gate, so the 401 + // branch below keeps distinguishing "gated" from "not there". + const response = await fetchWithTimeout(`${_backendUrl}/api/health`, {}, PROBE_TIMEOUT_MS); if (response.status === 200) return 'ok'; return response.status === 401 ? 'unauthorized' : 'unreachable'; } catch { @@ -1010,6 +1040,13 @@ export const startAnalyze = async (request: { force?: boolean; embeddings?: boolean; token?: string; + /** + * Index-branch selector. Omitted: a `url` with no existing clone takes the + * remote's default branch, an existing clone updates whichever branch it + * already has checked out, and a `path` request is not cloned at all and + * indexes that working tree as it stands. + */ + branch?: string; }): Promise<{ jobId: string; status: string }> => { const response = await fetchWithTimeout( `${_backendUrl}/api/analyze`, diff --git a/gitnexus-web/test/unit/code-references-panel.test.tsx b/gitnexus-web/test/unit/code-references-panel.test.tsx index e5c6b3e88..ada2a8e9d 100644 --- a/gitnexus-web/test/unit/code-references-panel.test.tsx +++ b/gitnexus-web/test/unit/code-references-panel.test.tsx @@ -1,9 +1,9 @@ -import { render } from '@testing-library/react'; +import { render, screen, waitFor } from '@testing-library/react'; import { beforeEach, describe, expect, it, vi } from 'vitest'; import type { ReactNode } from 'react'; import type { GraphNode } from 'gitnexus-shared'; import { CodeReferencesPanel } from '../../src/components/CodeReferencesPanel'; -import { readFile } from '../../src/services/backend-client'; +import { BackendError, readFile } from '../../src/services/backend-client'; const fileNode: GraphNode = { id: 'File:src/foo.ts', @@ -31,6 +31,15 @@ vi.mock('../../src/hooks/useAppState', () => ({ vi.mock('../../src/services/backend-client', () => ({ readFile: vi.fn(), + BackendError: class BackendError extends Error { + constructor( + message: string, + _status: number, + public readonly code: string, + ) { + super(message); + } + }, })); vi.mock('react-syntax-highlighter', () => ({ @@ -70,4 +79,16 @@ describe('CodeReferencesPanel repo identity (#2420)', () => { expect(readFile).toHaveBeenCalledWith('src/foo.ts', { repo: 'reels' }); }); + + it('renders the dedicated source-unavailable state for retained indexes without a checkout', async () => { + vi.mocked(readFile).mockRejectedValue( + new BackendError('source unavailable', 410, 'source_unavailable'), + ); + + render(); + + await waitFor(() => { + expect(screen.getByText('graph:codePanel.sourceUnavailable')).toBeInTheDocument(); + }); + }); }); diff --git a/gitnexus-web/test/unit/constants.test.ts b/gitnexus-web/test/unit/constants.test.ts index 10fe238b9..216a3d52f 100644 --- a/gitnexus-web/test/unit/constants.test.ts +++ b/gitnexus-web/test/unit/constants.test.ts @@ -65,6 +65,8 @@ describe('FILTERABLE_LABELS', () => { expect(FILTERABLE_LABELS).toContain('Type'); expect(FILTERABLE_LABELS).toContain('Decorator'); expect(FILTERABLE_LABELS).toContain('Variable'); + expect(FILTERABLE_LABELS).toContain('Protocol'); + expect(FILTERABLE_LABELS).toContain('Category'); }); it('every filterable label has a defined color in NODE_COLORS', () => { diff --git a/gitnexus-web/test/unit/filter-panel.test.ts b/gitnexus-web/test/unit/filter-panel.test.ts index 67ab3397c..9bc9c5132 100644 --- a/gitnexus-web/test/unit/filter-panel.test.ts +++ b/gitnexus-web/test/unit/filter-panel.test.ts @@ -7,6 +7,8 @@ const LEGEND_LABELS: NodeLabel[] = [ 'Folder', 'File', 'Class', + 'Protocol', + 'Category', 'Interface', 'Enum', 'Type', @@ -20,6 +22,8 @@ const ICON_MAP: Record = { Folder: 'Folder', File: 'FileCode', Class: 'Box', + Protocol: 'Hash', + Category: 'Box', Function: 'Braces', Method: 'Braces', Interface: 'Hash', @@ -62,6 +66,8 @@ describe('color legend', () => { expect(LEGEND_LABELS).toContain('Type'); expect(LEGEND_LABELS).toContain('Decorator'); expect(LEGEND_LABELS).toContain('Variable'); + expect(LEGEND_LABELS).toContain('Protocol'); + expect(LEGEND_LABELS).toContain('Category'); }); it('every legend label has a color defined', () => { @@ -76,6 +82,8 @@ describe('color legend', () => { 'Folder', 'File', 'Class', + 'Protocol', + 'Category', 'Interface', 'Enum', 'Type', diff --git a/gitnexus/CHANGELOG.md b/gitnexus/CHANGELOG.md index 091870103..082debfe7 100644 --- a/gitnexus/CHANGELOG.md +++ b/gitnexus/CHANGELOG.md @@ -4,6 +4,52 @@ All notable changes to GitNexus will be documented in this file. ## [Unreleased] +## [1.6.12] - 2026-09-12 + +### Added + +- **Configurable index artifact storage** — `GITNEXUS_STORAGE_PATH` writes one repository's index artifacts (graph data, metadata, parse caches, locks, branch indexes) to a caller-selected absolute directory, and `GITNEXUS_STORAGE_ROOT` gives several repositories one shared external root with an isolated `-/` slot each. `GITNEXUS_STORAGE_PATH` wins when both are set. Opt-in: unset, GitNexus still writes to `/.gitnexus/` (#3060) +- **Generation-time content retention tiers** — `GITNEXUS_CONTENT_RETENTION=full|symbol|none` chooses how much source-derived text is persisted: full file and symbol text, symbol snippets only, or structural graph data with no source bodies. Graph-oriented CLI, MCP and UI workflows are unchanged; CLI, MCP, the HTTP API and the web UI now say so explicitly when retention or a missing checkout hides file text. Storage and retention compatibility metadata is persisted, so an index is rebuilt when those semantics change (#3060) +- **Objective-C is a supported language** — vendored `tree-sitter-objc` grammar with a deterministic provider covering interfaces, implementations, categories, methods, properties and header classification (#3179) +- **`analyze --skip-fts` / `GITNEXUS_SKIP_FTS=1`** — explicit FTS opt-out that skips extension loading and keyword-search indexes; the flag and the env var are one mode, so toggling the discriminator alone no longer forces a same-commit rebuild (#3205, #3263) +- **`gitnexus embeddings` fills an existing index in place** — long HTTP embedding jobs are resumable: every successful batch is durable, reruns skip vectors whose content hash still matches, endpoint timeouts retry under `GITNEXUS_EMBEDDING_RETRY_TIMEOUTS`, and the structural graph is not rebuilt (#3065) +- **Staleness reports `diverged` and `unknown` instead of `fresh`** — `checkStaleness` / `checkStalenessAsync` return an additive status (`current`, `behind`, `diverged`, `unknown`) so a `rev-list` failure on a pruned branch-pinned clone stops reading as an up-to-date index (#3257) +- **Serve API exposes branch and index freshness** — `GET /api/repos` and `GET /api/repo` return the indexed branch, `lastCommit`, and how far behind the working tree is; `POST /api/analyze` honors `branch` (#3232, #3199) + +### Fixed + +- **Parse-cache chunk whose durable generation could not be reset is retired**, instead of leaving a stale generation reachable through the coherence gate (#3271) +- **Stale file-lock reclamation is guarded**, closing the lock-recovery failure paths (#3234) +- **LadybugDB checkpoint race in the pool adapter**, with `@ladybugdb/core` pinned to 0.18.3 (#3189) +- **MCP rejects unknown tool arguments and honors `depth`** (#3267) +- **Deleted files map to indexed symbol ranges** on incremental analyze (#3269) +- **Metadata-only diff files are retained** by the parser (#3251); stable cache packs stay parallel (#3194) +- **Embedding sync fails closed on foreign identity and vector-width drift** (#3260) +- **Dart** — `@name` is anchored so a constructor initializer stops minting a second symbol (#3224) +- **Zig** — callable-value references are modeled and their absence is no longer reported as `exact` (#3219); cross-file static gates resolve (#3185); `tree-sitter-zig` is vendored so `npm i -g` no longer warns on peers (#3180) +- **Go** — test siblings resolve and package discovery is tighter (#3191) +- **TypeScript** — `tsconfig` `paths` aliases resolve on Windows (#3203) +- **Ruby** — gem requires are guarded with dependency metadata (#3096) +- **COBOL** — copybook directories are preferred so `COPY EXTERNAL` does not hit vendor decoys (#3240) +- **Python** — `group` detects function-local imports (#3254) +- **NestJS GraphQL contracts** extract on real indexes (#3227); Spring constructor-to-bean injection edges persist to the schema (#3239) +- **Derived graph flows exclude guessed call edges** (#3193), and fallback guesses are labeled while export visibility is preserved (#3190) +- **`doctor` distinguishes vector capability from repository index state** (#3228) +- **CI looks up fork prebuild PRs by head owner and branch** (#3236) + +### Performance + +- **MCP `tools/list` no longer spawns one git process per repo** — the registry is read directly (#3259) +- **Scope resolution stops re-scanning the ParsedFile store once per language** (#3211) and avoids quadratic config-walk queues (#3237) +- **Parse dispatch** — cache packs batch into one dispatch round, the round's memory bound is tightened, and the worker-pool override is unclamped (#3196, #3200) +- **File locking probes this process's own start time once** (#3222) + +### Chore / Dependencies + +- **Benchmark and skill-evolution harness** — evolution runs against historical PRs, bounded packed-scheduler primitives with offline replay, provider-native usage recorded at the gateway, and offline benchmarks against a scripted provider (#2785, #3206, #3207, #3220, #3235) +- **Docs** — FTS closed as an optimization target with measured evidence, edit-loop numbers corrected with an FTS per-index breakdown, RepoCloud one-click deploy button (#3208, #3209, #3212) +- **Dependency bumps** across gitnexus (`hono`, `ignore`, `joi`, `express-rate-limit`, `@types/node`), gitnexus-web (`@langchain/langgraph`, `react-i18next`, `@types/react`, `@vitejs/plugin-react`, `@testing-library/user-event`), and GitHub Actions (`softprops/action-gh-release`, `docker/setup-qemu-action`) (#3164, #3165, #3214, #3215, #3231, #3233, #3243–#3249, #3265) + ## [1.6.11] - 2026-09-04 ### Added diff --git a/gitnexus/README.md b/gitnexus/README.md index f3f870641..b09e846c4 100644 --- a/gitnexus/README.md +++ b/gitnexus/README.md @@ -161,7 +161,7 @@ GitNexus builds a complete knowledge graph of your codebase through a multi-phas 5. **Processes** — Traces execution flows from entry points through call chains 6. **Search** — Builds hybrid search indexes for fast retrieval -The result is a **LadybugDB graph database** stored locally in `.gitnexus/` with full-text search and semantic embeddings. +The result is a **LadybugDB graph database** stored locally in `.gitnexus/` by default, with full-text search and semantic embeddings. ### Experimental community detection engine @@ -518,11 +518,11 @@ truthy, when the install is not an npm global/local install (npx cache, dev checkout, Docker image — the Docker CLI image sets the opt-out itself), or when opted out: -| Variable | Effect | -| --- | --- | -| `GITNEXUS_NO_UPDATE_NOTIFIER` | Truthy (`1`, `true`, …) disables the update check on every surface. | -| `NO_UPDATE_NOTIFIER` | Cross-tool convention; honored the same way. | -| `npm_config_registry` | The check reads the `latest` dist-tag from this registry instead of `https://registry.npmjs.org`. Credentials are never sent, and registries that require authentication are not supported (the check silently skips). | +| Variable | Effect | +| ----------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `GITNEXUS_NO_UPDATE_NOTIFIER` | Truthy (`1`, `true`, …) disables the update check on every surface. | +| `NO_UPDATE_NOTIFIER` | Cross-tool convention; honored the same way. | +| `npm_config_registry` | The check reads the `latest` dist-tag from this registry instead of `https://registry.npmjs.org`. Credentials are never sent, and registries that require authentication are not supported (the check silently skips). | Eval harnesses running a global install can set `GITNEXUS_NO_UPDATE_NOTIFIER` for a quiet registry. @@ -614,20 +614,17 @@ runtime dependencies Windows does not ship by default: 1. **Microsoft Visual C++ 2015-2022 Redistributable (x64)** — -2. **OpenSSL 3** — `libssl-3-x64.dll` and `libcrypto-3-x64.dll`, resolvable on `PATH` +2. **OpenSSL 3** — install it as a system runtime so `libssl-3-x64.dll` and + `libcrypto-3-x64.dll` resolve without borrowing them from another application. -The redistributable alone is **not** sufficient. If Git for Windows is installed you already have -the OpenSSL DLLs — run `gitnexus` from **Git Bash**, or prepend the directory to `PATH` in the -shell you use: +The redistributable alone is **not** sufficient. Do not prepend a third-party +application directory (including Git for Windows) to `PATH` to pick up those DLLs. -```powershell -$env:PATH = "C:\Program Files\Git\mingw64\bin;$env:PATH" -gitnexus analyze --repair-fts -``` - -Without them the index is still built, but without search tables, so `query` returns empty keyword -results until you re-run `gitnexus analyze --repair-fts` from a shell where the DLLs resolve -([#2669](https://github.com/abhigyanpatwari/GitNexus/issues/2669)). +Without both runtimes the index is still built, but without search tables, so +`query` returns empty keyword results until you install the prerequisites and +re-run `gitnexus analyze --repair-fts` +([#2669](https://github.com/abhigyanpatwari/GitNexus/issues/2669), +[#3218](https://github.com/abhigyanpatwari/GitNexus/issues/3218)). ### Installation fails with native module errors @@ -671,17 +668,20 @@ GitNexus uses optional DuckDB extensions for BM25 and vector search. The `gitnex Configure the behavior with these environment variables: -| Variable | Values | Default | Effect | -| -------------------------------------------- | ------------------------------ | ---------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| `GITNEXUS_LBUG_EXTENSION_INSTALL` | `auto`, `load-only`, `never` | `auto` | `auto` runs one bounded install if LOAD fails — a plain `INSTALL`, escalating to `FORCE INSTALL` only when the LOAD error shows the present extension file is broken. `load-only` only uses already-installed extensions (recommended for offline / firewalled environments). `never` skips optional extensions entirely. | -| `GITNEXUS_LBUG_EXTENSION_INSTALL_TIMEOUT_MS` | positive integer | `15000` | Wall-clock budget for the out-of-process extension-install child before it is killed. | -| `GITNEXUS_FTS_STEMMER` | supported LadybugDB stemmer | `porter` | Stemmer used when rebuilding BM25/FTS indexes. Use `none` for CJK-heavy repositories, or a language stemmer such as `german`, `french`, or `spanish` when that better matches repository comments and identifiers. Re-run `gitnexus analyze --repair-fts` after changing it. | -| `GITNEXUS_FTS_CJK_SEGMENTATION` | `none`, `bigram` | `none` | `bigram` inserts overlapping character-bigram boundaries into Chinese/Japanese Han-ideograph spans in `content`/`description` before FTS indexing, so LadybugDB's space-only tokenizer can see sub-phrase word boundaries. Scoped to CJK Unified Ideographs only — Japanese Hiragana/Katakana and Korean Hangul are not currently segmented. Unlike `GITNEXUS_FTS_STEMMER`, this rewrites stored text — enabling it on an already-indexed repo requires a full `gitnexus analyze --force`; neither `--repair-fts` nor a plain incremental `analyze` applies it to previously-indexed files. Set the same value wherever `analyze` and search-serving processes (CLI query, MCP server, web server) run. | -| `GITNEXUS_STREAM_GRAPH_EMIT` | `0`, `1` | `1` (on) | **On by default** on a full rebuild (`--force`); incremental runs ignore it. Holds structural relationships (CALLS, IMPORTS, ACCESSES, CONTAINS, ...) as CSV-on-disk plus compact in-memory columns instead of as objects in three overlapping indexes, cutting peak in-memory graph heap by ~1.4x at no measurable CPU cost (measured A/B on a synthetic 400k-node / 1.08M-edge graph: 819 MB -> 584 MB, iteration at parity, scaling verified linear from 100k to 800k nodes, with every edge still visible through the graph interface; no end-to-end measurement on a real repository yet). Nothing is traded away — community detection, process extraction, PDG taint summaries and the local-symbol pruner all read a complete relationship set and behave identically. Set to `0` only to bisect a suspected streaming-related fault. | -| `GITNEXUS_COMMUNITY_ENGINE` | `graphology`, `icebug`, `auto` | `graphology` | Community-detection engine used during analyze. `graphology` is the supported default. `icebug` and `auto` are **experimental** and currently behave identically: both try the optional `@ladybugmem/icebug` native Leiden over a CSR export and fall back to Graphology if it is not installed, cannot load, or lacks the deterministic thread/seed controls. Experimental engines partition differently, so community IDs are not comparable across engines. | -| `GITNEXUS_WAL_CHECKPOINT_THRESHOLD` | integer `>= -1` | `67108864` (64 MiB) | LadybugDB WAL auto-checkpoint threshold during analyze (bytes). Auto-checkpoint remains enabled; `-1` keeps Ladybug's stock ~16 MiB. Larger thresholds reduce checkpoint frequency but increase the WAL size at rotation time — choose a smaller value on disk-constrained environments. | -| `GITNEXUS_LBUG_BUFFER_POOL_SIZE` | integer `>= 0` (bytes) | min(2 GiB, 80% RAM) | LadybugDB buffer-pool ceiling for every GitNexus database (analyze, MCP server, serve, group bridges). Bounded so a long-lived `gitnexus mcp` process or a large incremental `analyze` cannot grow toward LadybugDB's native 80%-of-RAM default and OOM the host (#2557). `0` restores that native unbounded default; invalid values warn and fall back to the default. During `analyze` the pool is right-sized to the graph and, on non-4 KiB-page hosts (Apple Silicon 16 KiB, Ascend/aarch64 64 KiB), scaled by the page-size granule ratio up to min(2 GiB × pageSize/4 KiB, 80% RAM) (#2631); this env var overrides all of that as an absolute value. | -| `GITNEXUS_LBUG_MAX_DB_SIZE` | positive integer (bytes) | `17179869184` (16 GiB) | Upper bound for a single LadybugDB database file. This is an mmap/disk-address-space ceiling, not a memory limit — it does not constrain the buffer pool (use `GITNEXUS_LBUG_BUFFER_POOL_SIZE` for that). Raise it when indexing genuinely huge monorepos; invalid values silently fall back to the default. | +| Variable | Values | Default | Effect | +| -------------------------------------------- | ------------------------------ | -------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `GITNEXUS_LBUG_EXTENSION_INSTALL` | `auto`, `load-only`, `never` | `load-only` globally; `analyze` defaults to `auto` | The process-wide default is `load-only` so serve/MCP/query never install over the network. `gitnexus analyze` overrides to `auto` unless you set the env. FTS loads the packaged per-platform artifact first (macOS, Windows, and Linux), then a named `LOAD`, then `INSTALL` only under `auto`. `never` skips optional extensions entirely. | +| `GITNEXUS_LBUG_EXTENSION_INSTALL_TIMEOUT_MS` | positive integer | `15000` | Wall-clock budget for the out-of-process extension-install child before it is killed. | +| `GITNEXUS_FTS_STEMMER` | supported LadybugDB stemmer | `porter` | Stemmer used when rebuilding BM25/FTS indexes. Use `none` for CJK-heavy repositories, or a language stemmer such as `german`, `french`, or `spanish` when that better matches repository comments and identifiers. Re-run `gitnexus analyze --repair-fts` after changing it. | +| `GITNEXUS_FTS_CJK_SEGMENTATION` | `none`, `bigram` | `none` | `bigram` inserts overlapping character-bigram boundaries into Chinese/Japanese Han-ideograph spans in `content`/`description` before FTS indexing, so LadybugDB's space-only tokenizer can see sub-phrase word boundaries. Scoped to CJK Unified Ideographs only — Japanese Hiragana/Katakana and Korean Hangul are not currently segmented. Unlike `GITNEXUS_FTS_STEMMER`, this rewrites stored text — enabling it on an already-indexed repo requires a full `gitnexus analyze --force`; neither `--repair-fts` nor a plain incremental `analyze` applies it to previously-indexed files. Set the same value wherever `analyze` and search-serving processes (CLI query, MCP server, web server) run. | +| `GITNEXUS_STORAGE_PATH` | absolute, non-empty directory | unset (repo-local) | Complete external index directory. This preserves the existing configuration semantics and takes precedence over `GITNEXUS_STORAGE_ROOT` when both are set. | +| `GITNEXUS_STORAGE_ROOT` | absolute, non-empty directory | unset (repo-local) | Absolute root directory for external indexes. GitNexus creates an isolated `-/` slot beneath it for each repository, then registers the resolved slot so `status`, MCP, and `serve` can reopen it later. | +| `GITNEXUS_CONTENT_RETENTION` | `full`, `symbol`, `none` | `full` | Source-text retention profile: `full` keeps file and symbol text, `symbol` keeps symbol snippets without full file content, and `none` keeps the structural graph without source body text. | +| `GITNEXUS_STREAM_GRAPH_EMIT` | `0`, `1` | `1` (on) | **On by default** on a full rebuild (`--force`); incremental runs ignore it. Holds structural relationships (CALLS, IMPORTS, ACCESSES, CONTAINS, ...) as CSV-on-disk plus compact in-memory columns instead of as objects in three overlapping indexes, cutting peak in-memory graph heap by ~1.4x at no measurable CPU cost (measured A/B on a synthetic 400k-node / 1.08M-edge graph: 819 MB -> 584 MB, iteration at parity, scaling verified linear from 100k to 800k nodes, with every edge still visible through the graph interface; no end-to-end measurement on a real repository yet). Nothing is traded away — community detection, process extraction, PDG taint summaries and the local-symbol pruner all read a complete relationship set and behave identically. Set to `0` only to bisect a suspected streaming-related fault. | +| `GITNEXUS_COMMUNITY_ENGINE` | `graphology`, `icebug`, `auto` | `graphology` | Community-detection engine used during analyze. `graphology` is the supported default. `icebug` and `auto` are **experimental** and currently behave identically: both try the optional `@ladybugmem/icebug` native Leiden over a CSR export and fall back to Graphology if it is not installed, cannot load, or lacks the deterministic thread/seed controls. Experimental engines partition differently, so community IDs are not comparable across engines. | +| `GITNEXUS_WAL_CHECKPOINT_THRESHOLD` | integer `>= -1` | `67108864` (64 MiB) | LadybugDB WAL auto-checkpoint threshold during analyze (bytes). Auto-checkpoint remains enabled; `-1` keeps Ladybug's stock ~16 MiB. Larger thresholds reduce checkpoint frequency but increase the WAL size at rotation time — choose a smaller value on disk-constrained environments. | +| `GITNEXUS_LBUG_BUFFER_POOL_SIZE` | integer `>= 0` (bytes) | min(2 GiB, 80% RAM) | LadybugDB buffer-pool ceiling for every GitNexus database (analyze, MCP server, serve, group bridges). Bounded so a long-lived `gitnexus mcp` process or a large incremental `analyze` cannot grow toward LadybugDB's native 80%-of-RAM default and OOM the host (#2557). `0` restores that native unbounded default; invalid values warn and fall back to the default. During `analyze` the pool is right-sized to the graph and, on non-4 KiB-page hosts (Apple Silicon 16 KiB, Ascend/aarch64 64 KiB), scaled by the page-size granule ratio up to min(2 GiB × pageSize/4 KiB, 80% RAM) (#2631); this env var overrides all of that as an absolute value. | +| `GITNEXUS_LBUG_MAX_DB_SIZE` | positive integer (bytes) | `17179869184` (16 GiB) | Upper bound for a single LadybugDB database file. This is an mmap/disk-address-space ceiling, not a memory limit — it does not constrain the buffer pool (use `GITNEXUS_LBUG_BUFFER_POOL_SIZE` for that). Raise it when indexing genuinely huge monorepos; invalid values silently fall back to the default. | ```bash # Offline/airgapped: never reach the network for extensions @@ -882,7 +882,7 @@ only — the hook's structured stdout (the JSON the agent consumes) is unaffecte - All processing happens locally on your machine - No code is sent to any server -- Index stored in `.gitnexus/` inside your repo (gitignored) +- Index stored in `.gitnexus/` inside your repo by default (gitignored), in the complete external directory selected by `GITNEXUS_STORAGE_PATH`, or in a repository-specific slot beneath `GITNEXUS_STORAGE_ROOT` - Global registry at `~/.gitnexus/` stores only paths and metadata ## Web UI diff --git a/gitnexus/bench/analyze-phase-breakdown.md b/gitnexus/bench/analyze-phase-breakdown.md index 6706c1905..e04d48628 100644 --- a/gitnexus/bench/analyze-phase-breakdown.md +++ b/gitnexus/bench/analyze-phase-breakdown.md @@ -169,31 +169,73 @@ Two things block it today, and neither is small: would then leave the live index with fresh File content and stale symbols, instead of untouched. -## scopeResolution is memory-traffic bound, not algorithmic +## scopeResolution — 14.7s, and it is two different problems -`--cpu-prof` of a one-file-edit run, top main-thread self time: +`PROF_SCOPE_RESOLUTION=1` already exists and reports the internal split. Marks +injected around the phase supply the rest. On a one-file edit: + +| step | ms | share | +| ---------------------------- | -------: | ------: | +| ParsedFile store rehydration | 4426 | 30% | +| **`emit`** | **7161** | **49%** | +| `resolve` | 966 | 7% | +| `finalize` | 552 | 4% | +| `extract` | 369 | 3% | +| per-language teardown, misc | ~750 | 5% | + +`extract` is small because the parse cache works: 2121/2121 pre-extracted hits +on the TypeScript pass. Nothing here re-parses. + +### Rehydration: three full store scans, one per language + +The corpus resolves three languages — python (54 files), typescript (2121), +javascript (59) — and `loadParsedFilesForPaths` walks **every** shard on each +pass. The store is 413 shards / 301 MB. A pass that wants a single file costs +335 ms, because the skip decision needs the envelope's path listing and that +listing is only trustworthy after the payload digest is checked. + +So the fixed cost is paid per language, and **it scales with language count**, +not with how many files that language has. Measured, min-of-5, three passes: ``` -4294ms (garbage collector) -1552ms v8.deserialize - 756ms crypto update - 728ms runScopeResolution - 620ms scope-resolution/pipeline/reconcile-ownership - 380ms v8-sidecar walk - 323ms internString +before python 411ms typescript 2872ms javascript 507ms = 3834ms +after python 415ms typescript 2805ms javascript 248ms = 3484ms ``` -then a tail of passes at 200–750ms (`emitReceiverBoundCalls`, -`emitCallableValueFlow`, `buildGraphNodeLookup`, `resolveReferenceSites`). +Memoizing each shard's authenticated listing for the run removes the repeat +scans (−350 ms here; roughly −250 ms per additional language elsewhere). The +first pass is unchanged by construction — it is what populates the memo. -No dominant hot function, nothing quadratic. The cost is rehydrating every -file's `ParsedFile` from the durable `.v8` shards into main-thread memory and -re-running every pass over them. +What is left is a floor. The digest is **not** the cost: SHA-256 over all +301 MB takes 134 ms (2.25 GB/s, hardware-accelerated), so swapping it for +CRC-32 would buy ~90 ms and cost an envelope-format bump. The remainder is +`v8.deserialize` (~1240 ms) plus the cross-shard string intern walk +(~1085 ms), and the intern walk is not optional — dropping it regresses +retained heap ~59%, which is #2649's constraint. -**So the win is not caching resolution output** — that stores more of exactly -what is already the memory problem, and #2649 (large-repo OOM) is the standing -constraint. The win is skipping rehydration and re-resolution for files whose -inputs provably did not change, as a streaming/bounded design. +### `emit` is the real target + +7.2s, 49% of the phase and ~21% of the whole edit loop. It is a fan of passes, +each walking every reference site in the repo: + +``` +1828ms emitCallableValueFlow 480ms emitFreeCallFallback +1590ms emitReceiverBoundCalls 426ms emitReferencesViaLookup + 681ms resolveDefGraphId 313ms emitUniqueNamePropertyAccesses + 529ms lookupCore 280ms emitReturnShapeMemberAccesses + 472ms tryEmitEdge 222ms emitPropertyDispatchCalls + 449ms getScope +``` + +No hot inner loop, nothing quadratic, no single pass worth more than 12% of the +phase. Micro-optimizing any of them is not the win. + +The win is not running them. A one-file edit re-emits all 163,094 edges to +write a 3,980-node subgraph. Every pass above runs over all 2121 TypeScript +files because the pipeline rebuilds the full in-memory graph every run and only +the DB _write_ is incremental. Making `emit` incremental means knowing which +files' edges can change when one file's registry contribution changes — which +is a design, not a patch, and #2649 rules out "just cache the emitted edges". --- @@ -232,8 +274,12 @@ and 0.8s of post-FTS event-loop drain between them. FTS is closed as an optimization target except for the overlap above, which is a scheduling change to `run-analyze.ts`'s open/close discipline rather than -anything about FTS. `scopeResolution` is the single largest step, memory-traffic -bound, and still untouched. +anything about FTS. + +`scopeResolution` is now broken down. Its rehydration half has a floor and one +repeat-scan win that is taken. Its other half — `emit`, 7.2s — is the largest +remaining target in the whole edit loop, and the only way at it is incremental +resolution. Every optimization in this document that looked compelling from the armchair died under measurement. Measure first, and check the banner. diff --git a/gitnexus/bench/emit-persistence/baselines.json b/gitnexus/bench/emit-persistence/baselines.json index a6f47f7e8..9c04fb248 100644 --- a/gitnexus/bench/emit-persistence/baselines.json +++ b/gitnexus/bench/emit-persistence/baselines.json @@ -1,5 +1,6 @@ { - "fingerprint": "011485e5180c005be9dfec7ac8dc4e3bb0fcfc4a680a16dcae2cf2d8af5ef7e2", + "fingerprint": "d3c3a53ead47663816344369dc0e501f31d60c462039de190ab6f5361037bd79", + "_rebaselined_objective_c_node_tables": "Objective-C support adds Protocol and Category node tables to csv-generator.ts's MULTI_LANG_TYPES, so this deterministic 2,400-entity synthetic emit now creates 38 CSV files rather than 36. The two additions, category.csv and protocol.csv, are both expected 55-byte header-only files because the synthetic graph contains no Objective-C nodes. Re-emitting the graph and recomputing the fingerprint after excluding exactly those two files yields the exact prior baseline 011485e5180c005be9dfec7ac8dc4e3bb0fcfc4a680a16dcae2cf2d8af5ef7e2, proving all 36 pre-existing files retain their bytes, routing, and within-file order. Prior 011485e5180c005be9dfec7ac8dc4e3bb0fcfc4a680a16dcae2cf2d8af5ef7e2 -> d3c3a53ead47663816344369dc0e501f31d60c462039de190ab6f5361037bd79. The timing guard remains within budget: local scaling_ratio 0.965 against 1.8 and elapsed_ms_large 69.24ms against 1000ms.", "_rebaselined_3161_static_gated_column": "Every relationship row gained a trailing `staticGated` BOOLEAN column (RELATION_SCHEMA in src/core/lbug/schema.ts, REL_CSV_HEADER + buildRelRow in csv-generator.ts, the fallback CREATE in lbug-adapter.ts), written as `0` for every edge that does not carry the flag and `1` for a Zig call site inside a comptime-false branch (PR #3161). Unlike the earlier header-only rebaselines this one touches ROWS, so the evidence is the inverse operation rather than a file-set diff: re-emitting this bench's 2,400-entity graph on the branch produces the same 36 CSV files; exactly 3 of them differ from a copy with the new column stripped (`rel_File_Class.csv` 156228 -> 151416 bytes, `rel_File_Function.csv` 334008 -> 324396, `rel_Function_Function.csv` 185028 -> 180216: one header field plus `,0` per row, 4,800 / 9,600 / 4,800 rows, 2 bytes each), the other 33 files are byte-identical, and the per-file fingerprint over the stripped copies is EXACTLY the prior baseline 72096279092d4f118de7e179333705c19c9aff2664f77d71a7da48cd9f73fb5a. So no row moved between pair files and no row reordered; the only bytes that changed are the appended column. Prior 72096279092d4f118de7e179333705c19c9aff2664f77d71a7da48cd9f73fb5a -> 011485e5180c005be9dfec7ac8dc4e3bb0fcfc4a680a16dcae2cf2d8af5ef7e2. Timing gates passed while the guard was red: scaling_ratio 0.82 against the 1.8 budget, elapsed_ms_large 185.54ms against the 1000ms backstop, so no throughput claim is being rebaselined away.", "scaling_budget": 1.8, "max_ms_large": 1000, diff --git a/gitnexus/bench/import-target/baselines.json b/gitnexus/bench/import-target/baselines.json index a02f23e0a..37fa78535 100644 --- a/gitnexus/bench/import-target/baselines.json +++ b/gitnexus/bench/import-target/baselines.json @@ -2,6 +2,8 @@ "_what": "Baselines for bench/import-target/measure.mjs cover every import-target resolver registered in SCOPE_RESOLVERS on a shared corpus, plus a configured C# arm for the branch the default call cannot reach. measure.mjs derives its arm inventory from LANG_REGISTRY and --check reconciles the registered languages in both directions. The single PHP arm supplies its production PSR-4 Composer mapping; csharp_csproj supplies csproj configuration. C and C++ also receive their production resolutionConfig header corpus. The timing, shape, fingerprint, context, and retained-heap gates therefore cover each production resolver path without splitting PHP into configured and unconfigured identities.", "_rebaselined_2960_kotlin_declared_packages": "Kotlin now resolves only from parsed package facts and local module bindings. This deliberately changes its five fingerprints, removes path-depth sensitivity, adds the context probe, and reduces the 32000-file retained index from 40.82 MiB to 4.31 MiB. External same-name path decoys now remain unresolved.", "_zig_arm_1432": "zig was registered in SCOPE_RESOLVERS by PR #1432 with no arm here, which is the exact hole the inventory arm exists for, and it reported it. The arm passes NO build config: resolveZigImportInternal walks an importer-relative @import path component by component and probes allFiles.has() twice (as spelled, then + '.zig'), so its fingerprints pin that walk alone and do not move when build.zig / build.zig.zon parsing (the bare-name legs, gated by test/unit/zig-import-resolver.test.ts) changes. Same corpus proportions as rust — 979 of 3200 resolve at 400 files — and rust's collide design, a deep tree whose spellings carry thirteen components against the unique arm's three, because component count is the only axis the cost has. Measured on one box: small 1.036 ms, collide 1.700 ms, scaling 1.050, collide scaling 1.011, depth 1.573; budgets take the file's usual ~1.5x on ratios (2.4 depth) and ~4x on absolute ms (4 / 7). Heap reads 16 B at both 8000 and 32000 files — it builds nothing, exactly rust's reading — so it sits in heap_bound_bytes at rust's absolute 1048576 B.", + "_objc_arm_3179": "Objective-C resolves local header and source imports through a cached suffix map. The map preserves the previous first-path tie-break, which is covered directly in objective-c-provider.test.ts. This arm covers the registered resolver's five dispatch paths, including the depth-sensitive index construction and a collision layout. Measured on this branch: small 1.844 ms, collide 1.861 ms, scaling 1.111, collide scaling 1.067, depth 2.779, and 44101944 retained bytes. The 3.6 depth budget allows normal timing variation while remaining below a full per-import file scan; collision scaling stays at the shared linear 1.8 budget because suffix lookup is keyed.", + "_cobol_copybook_dir_preference_2967": "COBOL now prefers well-known copybook directories (copybooks/, cpy/, copy/, plus importer dir) over vendor paths when resolving COPY statements, so COPY EXTERNAL no longer hits same-named decoys outside the project structure. This intentional behavior change moves the resolver from depth-free (prior measured ~0.885) to depth-sensitive (measured 1.751 on CI), because the new preferredCopybookDirs check walks path components. The 2.4 depth budget is ~1.37x the measured ratio, consistent with the ~1.5x convention for depth arms. The unique-arm fingerprints and resolved counts change to reflect the new target set: small/deep drop from 1153 to 442 resolved because files outside preferred directories are now correctly filtered. The collide arm (svc${d}/copybooks layout) keeps 1153 resolved because ALL its files are in preferred copybook directories, so the filter does not change its resolution outcomes — the collide arm now measures collision behavior within the preferred class rather than across mixed layouts. Fixes #2967.", "_php_composer_gate_2962": "The canonical PHP arm supplies an authoritative App PSR-4 mapping. A deterministic Vendor0 suffix decoy makes deletion of the external gate change every timing fingerprint, while the heap arm separately pins a mapped miss, the rendered mapping, and the external null result. Three serial samples measured depth ratios 1.058, 1.114, and 1.158; the 1.8 budget is 1.55x the observed maximum. The mapped-miss heap reading peaked at 39607216 retained bytes at 32000 files.", "_fingerprint_note": "Per-language sha256 over every distinct fromFile|target -> resolved target. A change here is a BEHAVIOUR change: the resolver returned a different target set, and IMPORTS/CALLS edges moved. Explain it, never re-baseline to make CI green. For the languages these PRs changed, the pre-change implementations produce these same values on this corpus at both 400 and 1600 files \u2014 that is what makes the index hoist a performance change. The tie-break-level proof lives in test/unit/scope-resolution/import-target-index-parity.test.ts (verbatim copies of the pre-change code, diffed) for Kotlin's current declared-package behavior in test/unit/kotlin-module-resolution.test.ts, and for the four resolvers added there in test/unit/scope-resolution/{php,java,cobol}-import-target-parity.test.ts and test/unit/import-resolvers/csharp-csproj-parity.test.ts, and for JavaScript in test/unit/scope-resolution/javascript-import-target-parity.test.ts (a differential over 211200 old-vs-new pairs, PR #2911). The eight languages added last have no per-language parity harness against a pre-change implementation and do NOT need one: nothing about their resolution changed, so there is no before to diff against. Their fingerprints are pure forward guards, minted from the current implementations, and their adapter-boundary index reuse is covered for every registered language at once by test/unit/scope-resolution/import-target-index-reuse.contract.test.ts. NOTE for csharp_csproj: on this corpus the #2902 indexed leg (step 3 of resolveCSharpImportInternal) is reached by 2221 of the 3200 small-arm imports but answers null for every one of them \u2014 the 979 that resolve do so at step 2 \u2014 so this fingerprint pins that legs cost and its null answers, while its positive tie-breaks (unanchored substring, iteration order) are pinned by csharp-csproj-parity.test.ts. NOTE for kotlin, go, csharp and java: twenty fingerprints across these four languages were re-baselined in #2881, the one deliberate behaviour change any language in this file has had. It landed in two steps and the second is the reason the first is not a special case: Kotlin first, then the shared package-dir-index (go, java, csharp) and the csproj namespace index once the same rule was found live there. `getKotlinFileIndex` no longer requires a file's package directory to be the FIRST occurrence of that name in its own path, so the unique arm's `d % 7` nested slice (`mod{d}/src/main/kotlin/com/example/pkg{d}/inner/pkg{d}`) now belongs to package `pkg{d}` and its wildcard imports resolve: resolved 1100 -> 1153 small and deep, 4456 -> 4681 large. The collide arm needed a CORPUS edit alongside it, not just a new number \u2014 its `d % 7` slice deliberately imported `com.example.vendor{d}`, a package that exists nowhere, purely to mirror the unique arm's nested-slice MISS, so leaving it would have left collide at 1100 against small's 1153 and broken the same-workload invariant the arm is built on (that assertion is what caught it). It now uses the same `com.example.models.*` spelling as the rest of the arm, which is why its distinct_outcomes fell (2775 -> 2744, 11087 -> 10961): one shared target instead of one per d. The record-level evidence for the resolver change \u2014 235 of 19968 records moved, 54 null -> resolved, 0 buckets losing a member \u2014 is in bench/kotlin-import-target/baselines.json `_provenance`. The kotlin heap_reading_bytes and heap_ceiling_bytes moved with it, together as `_heap_reading_note` requires: 48073096 -> 48200224 bytes_large (+127128, +0.264%), ceiling still exactly 1.5x. Small, and it is worth saying WHY it is small rather than reading the number as evidence that the change is cheap. `dirChildren` grows by one entry per component-suffix the old rule used to skip, and this arm can only see part of that: the heap corpus is built with HEAP_PAD 8, which prefixes every path with `d0/\u2026/d7/`, so no path can begin with a suffix of its own directory and the leading-segment half of the old rule is structurally invisible here. What moves the reading is the `d % 7` nested slice alone. Read +0.264% as this arm's ceiling on the effect, not as the effect. GO NEEDED A CORPUS EDIT TO BE GATED AT ALL. Its nested slice was `src/pkg{d}/internal/pkg{d}`, repeating only the LAST segment, while a Go query addresses the whole package path `src/pkg{d}` \u2014 so the directory never even ended with the query and the first-occurrence rule was never reached. Every go arm sat unchanged through the resolver fix. `uniqueDir`/`collideDir` now repeat the shape at the granularity Go actually queries (`src/pkg{d}/internal/src/pkg{d}`, `svc{d}/internal/sub/svc{d}/internal`), which is what moved go from 979 to 1153 resolved and bumped `languages.go.heap.path_segments` 13 -> 14. The general lesson: a corpus that carries a shape the QUERY cannot express does not gate that shape. CSHARP AND JAVA HIT THE SAME COLLIDE-ARM TRAP AS KOTLIN. Both collide arms sent their `d % 7` slice to a namespace that exists nowhere (`App.Src{d}.Vendor`, `com.svc{d}.vendor`) purely to MIRROR the unique arm's nested-slice miss; once that miss became a hit, collide sat at 979/1100 against small's 1153 and the same-workload assertion failed. Both now use the same spelling as the rest of their arm. HEAP: no reading here moved for the resolver change. An earlier revision of this branch re-recorded `csharp_csproj` 73703384 -> 73116520 as a -0.79% effect of the step-2 filter; review measured base and branch three times each and got the same 73.10e6 on BOTH sides \u2014 the recorded 73703384 was simply not reproducible on this box, and re-recording it would have dropped that language's derived floor by 0.8% for no reason belonging to this change. Reverted. Everything else sat within +/-0.03%. Note that `_heap_reading_note`'s claim that these readings 'reproduce to the byte across processes on one box' did NOT hold on the box this was measured on: go, dart, ruby, python, php and cpp all wandered by a few hundred to a few thousand bytes between processes with no code change touching them. Treat sub-0.05% movement as jitter, not signal. HEAP, kotlin, second movement: 48200224 -> 42802456 (-11.20%), re-recorded with its ceiling. `getKotlinFileIndex` now compacts each `dirChildren` bucket as it freezes it. `addChild` mints a bucket as `[raw]` and pushes the rest, and V8 grows a backing store by `old + old/2 + 16`, so the second child takes a 1-slot store to 17: 61144 buckets, 52.9% of their slots empty, 88 B each. Same fix and same accounting as the python `byBasename` sentence above. Note what this means for the gate: a memory WIN of this size passes every arm \u2014 it is under the ceiling and over the 0.5x floor \u2014 so it is recorded because the convention says a reading and its ceiling move together, not because anything went red. kotlin now reads 40.82 MiB. The prose in measure.mjs calling it '45.85 MiB, the second-largest reading in this file' is corrected with it \u2014 and was already wrong on the ranking before this change, since csharp_csproj (69.73) and php (47.28) both read higher; kotlin was third. A measurement written into prose is not re-taken, which is the finding `_heap_bound_note` records about this very file. One further corpus edit, made in review and MEASURED rather than assumed: kotlin's collide layout repeated only the `models` leaf (`\u2026/com/example/models/inner/models`) while a Kotlin query addresses the whole dotted path, so a full revert of the Kotlin guards left both collide fingerprints UNMOVED \u2014 the arm was blind to the rule it was re-baselined for. Deepening it to `\u2026/models/inner/com/example/models` makes the revert move both, and those two fingerprints are the only ones that changed for it. The same deepening was applied to the java and kotlin UNIQUE arms and REVERTED: it moved ten more fingerprints, grew java's heap reading 43%, and bought nothing \u2014 progressive stripping lands those queries on the same file with or without the rule, so the control still failed only on go.", "_shape_note": "files/imports/resolved/distinct_outcomes AND the fingerprint are asserted exactly, per scale. A fingerprint alone cannot tell a legitimate resolution change from a corpus quietly shrunk below the size at which the timing arms can see anything; conversely the counts alone cannot see a defect confined to one arm, because the arms differ only in path padding and directory layout and both of those are count-neutral by design. Two cross-arm assertions close the remaining hole: the deep and collide arms must resolve exactly what small resolves (they are the same workload), and each of their fingerprints must DIFFER from small's (they are not the same corpus). Without the second, setting DEEP_PAD to 0 \u2014 which deletes the entire depth arm \u2014 moves no asserted number and prints PASS; the same is true of a collideDir that forwards to uniqueDir. THE HEAP ARM IS ASSERTED THE SAME WAY, by the same loop, and was not before: files_small, files_large, path_segments and probe decide WHAT it measures, and every one of them was reported and compared to nothing. Swapping HEAP_PROBE_TARGET.csharp_csproj for a target matching no CSPROJ_CONFIGS rootNamespace skips the whole config loop, so the getFilesInDir and getInsensitive legs never run and the arm the header calls the witness that the read pattern IS the footprint quietly becomes a two-map arm \u2014 73703384 -> 59921216 B, ratio 1.017 -> 1.011, ceiling and floor both still passing and --check still exiting 0. Setting HEAP_SMALL equal to HEAP_LARGE is the same hole from the other side: ratio goes to ~1.0 by construction and bytes_large never moves. bytes_small and bytes_large are deliberately NOT asserted for equality \u2014 heap_ceiling_bytes and the heap_reading_bytes floor bound them with ~50% either way, because heapUsed accounting moves across platforms and Node majors and an exact byte assertion would be a re-baseline per runner. THE CONTEXT ARM IS ASSERTED THE SAME WAY, by the same loop, and more strictly than either: target, with_context and without_context are exact strings with no tolerance at all, because the arm resolves one import over a three-file corpus and has no measurement noise to tolerate. A separate check requires the last two to DIFFER, for the same reason deep.fingerprint must differ from small.fingerprint \u2014 a probe on which both call shapes agree asserts one number twice. Both halves run through resolveOne, so what the arm gates is this bench threading run.ts's fifth argument, not the resolvers' behaviour.", @@ -28,6 +30,7 @@ "vue": 1.8, "c": 3.8, "cpp": 4, + "objc": 1.8, "zig": 1.8 }, "depth_budget": { @@ -39,7 +42,7 @@ "kotlin": 1.8, "php": 1.8, "java": 1.8, - "cobol": 1.6, + "cobol": 2.4, "swift": 2.3, "rust": 2.1, "python": 2.2, @@ -48,6 +51,7 @@ "vue": 2.3, "c": 3, "cpp": 3, + "objc": 3.6, "zig": 2.4 }, "small_ms_ceiling": { @@ -68,6 +72,7 @@ "vue": 81, "c": 7, "cpp": 7, + "objc": 8, "zig": 4 }, "collide_ms_ceiling": { @@ -88,6 +93,7 @@ "vue": 93, "c": 11, "cpp": 12, + "objc": 8, "zig": 7 }, "heap_ceiling_bytes": { @@ -95,6 +101,7 @@ "go": 4497696, "dart": 11751300, "cpp": 15035016, + "objc": 66200000, "csharp": 44900000, "csharp_csproj": 110600000, "ruby": 61600000, @@ -110,6 +117,7 @@ "go": 2998464, "dart": 7834200, "cpp": 10023344, + "objc": 44101944, "csharp": 29869080, "csharp_csproj": 73703384, "ruby": 41020808, @@ -118,9 +126,9 @@ "python": 6360936, "c": 10018816 }, - "_heap_bound_note": "THE SECOND HEAP TIER. Every registered language is measured now; heap_bound_bytes gates the nine that are not BUDGETED above, and it gates them with one comparison and no floor. A ceiling says 'this index is not too big'. A bound says something narrower and it is the thing that was missing: 'the exclusion still holds' \u2014 this language has not grown an index since it was left out. measure.mjs's MEMORY section states the re-entry condition (if a language ever diverges in what it ASKS its index, it earns a budgeted arm) and until now nothing watched for the divergence; HEAP_LANGS was a hand-maintained list of eight whose two neighbours, LANG_REGISTRY and CONTEXT_LANGS, are both reconciled against a derived predicate in both directions. HEAP_BOUNDED is derived too \u2014 it is LANGS minus HEAP_BUDGETED \u2014 so the two tiers partition the languages and a new one cannot land outside both. WHAT RE-MEASURING FOUND, five runs each, maximum quoted, peak-to-peak in brackets. go 2998464 B [1.0021], dart 7834200 B [1.0006] and kotlin 42802456 B [1.0004] HAD NO STATED REASON AT ALL: the old prose opened 'SIX of the seventeen are deliberately NOT in HEAP_LANGS' against a list of eight of seventeen, and these three were the three nobody counted. All three retain a real per-pass structure (go's PackageDirIndex, dart's basename buckets, kotlin's suffixByStem cascade) and kotlin's 40.82 MiB is above ruby's 39.12 and java's 33.34, both of which carry a full budget. (It read 45.85 MiB when this was written, described here as 'the second-largest reading in this file' \u2014 it was third even then, behind csharp_csproj and php; #2881 later compacted its dirChildren buckets and took 11% off it. Same staleness this paragraph exists to document.) swift 3449216 B [1.0024] and cobol 2320456 B [1.0000] were excluded as 'below the measurement's own noise floor' on readings of 0.29 MB and 0 B at 32000 files; they now read 3.29 MB and 2.21 MB, growing with the corpus (969120 B and 536264 B at 8000). Those old numbers were not wrong when taken \u2014 the ARM changed under them, when #2903's follow-up made every probe resolve a real import and when measureHeap began flattening its corpus \u2014 which is the whole finding: a measurement written into prose is not re-taken, and this file had already gone stale against itself, quoting javascript at 46208832 B four paragraphs after quoting it at 25.51 MiB. rust is the one exclusion that survived unchanged: 16 B at 8000 files and 16 B at 32000, identical in all five runs. typescript 26745296 B, vue 28884016 B and cpp 10023344 B are duplicates of a builder AND of a read pattern: typescript is byte-identical to javascript's 26745296 in four runs of five, cpp is +0.05% of c's 10018816, vue is +8.0% of javascript. HOW THE BOUNDS WERE CHOSEN. Each takes 1.5x its measured maximum, rounded up to the next 100000 B: cobol 3500000 (1.508x), swift 5200000 (1.508x). (This sentence used to list eight, including go, dart, kotlin, typescript, vue and cpp. Those six were promoted to the budgeted tier and their bounds deleted; the numbers stayed here, unread by any gate, and #2881 dutifully updated kotlin's to 64300000 before anyone noticed heap_bound_bytes holds only cobol, swift and rust. A number nothing asserts is a number that rots \u2014 the finding this paragraph is otherwise about.) 1.5x is NOT copied from the ceilings out of habit \u2014 it is the same number for a stated reason, and the reason is not noise: measured peak-to-peak on this box is at most 1.0024, so noise alone would justify 1.05x. What a bound has to survive is a RUNNER change, since heapUsed accounting moves across platforms and Node majors, and this file already fixes that allowance at 50% for exactly this measurement on exactly this arm. Using a second allowance for the same uncertainty on the same number would be two conventions, not more rigour. At 1.5x the bound catches what the re-entry condition is about \u2014 a language growing an index, which costs +85% for one more suffix map and +100% for a duplicate \u2014 and it does NOT catch a duplicate diverging by 8%. That limit is real and is stated rather than hidden: the tight form is a same-process ratio against the arm each duplicate is a duplicate OF, which is the only form immune to the drift the absolute bound has to tolerate. RUST TAKES AN ABSOLUTE BOUND INSTEAD, 1048576 B (1 MiB), because 1.5 x 16 B is 24 B and would fail on the first byte of anything \u2014 a multiplier on a reading that is already nothing is a gate that flakes rather than a gate that bites. 1 MiB is ~65000x the reading and still 2.2x below the smallest real index measured here (cobol's 2.32 MB at the same file count), so it separates 'builds nothing' from 'builds something' with room on both sides. NO FLOOR ON ANY OF THE NINE, and the reason differs by language rather than being uniform. For rust a floor would be a floor on noise. For the other eight the readings are stable enough to floor today, and for kotlin and dart \u2014 larger than budgeted arms \u2014 a floor would be worth having, since a lazily-built map going quiet is exactly how the four budgeted arms once read 0 B. Adding one is a PROMOTION to the budgeted tier, with a ceiling and a recorded reading beside it, not a line here: a floor whose companion ceiling does not exist asserts 'still measuring' against a number nothing else bounds. Recommended next, in order: kotlin, then dart, then go.", + "_heap_bound_note": "THE SECOND HEAP TIER. Every registered language is measured now; heap_bound_bytes gates the nine that are not BUDGETED above, and it gates them with one comparison and no floor. A ceiling says 'this index is not too big'. A bound says something narrower and it is the thing that was missing: 'the exclusion still holds' \u2014 this language has not grown an index since it was left out. measure.mjs's MEMORY section states the re-entry condition (if a language ever diverges in what it ASKS its index, it earns a budgeted arm) and until now nothing watched for the divergence; HEAP_LANGS was a hand-maintained list of eight whose two neighbours, LANG_REGISTRY and CONTEXT_LANGS, are both reconciled against a derived predicate in both directions. HEAP_BOUNDED is derived too \u2014 it is LANGS minus HEAP_BUDGETED \u2014 so the two tiers partition the languages and a new one cannot land outside both. WHAT RE-MEASURING FOUND, five runs each, maximum quoted, peak-to-peak in brackets. go 2998464 B [1.0021], dart 7834200 B [1.0006] and kotlin 42802456 B [1.0004] HAD NO STATED REASON AT ALL: the old prose opened 'SIX of the seventeen are deliberately NOT in HEAP_LANGS' against a list of eight of seventeen, and these three were the three nobody counted. All three retain a real per-pass structure (go's PackageDirIndex, dart's basename buckets, kotlin's suffixByStem cascade) and kotlin's 40.82 MiB is above ruby's 39.12 and java's 33.34, both of which carry a full budget. (It read 45.85 MiB when this was written, described here as 'the second-largest reading in this file' \u2014 it was third even then, behind csharp_csproj and php; #2881 later compacted its dirChildren buckets and took 11% off it. Same staleness this paragraph exists to document.) swift 3449216 B [1.0024] and cobol 2320456 B [1.0000] were excluded as 'below the measurement's own noise floor' on readings of 0.29 MB and 0 B at 32000 files; they now read 3.29 MB and 2.21 MB, growing with the corpus (969120 B and 536264 B at 8000). Those old numbers were not wrong when taken \u2014 the ARM changed under them, when #2903's follow-up made every probe resolve a real import and when measureHeap began flattening its corpus \u2014 which is the whole finding: a measurement written into prose is not re-taken, and this file had already gone stale against itself, quoting javascript at 46208832 B four paragraphs after quoting it at 25.51 MiB. rust is the one exclusion that survived unchanged: 16 B at 8000 files and 16 B at 32000, identical in all five runs. typescript 26745296 B, vue 28884016 B and cpp 10023344 B are duplicates of a builder AND of a read pattern: typescript is byte-identical to javascript's 26745296 in four runs of five, cpp is +0.05% of c's 10018816, vue is +8.0% of javascript. HOW THE BOUNDS WERE CHOSEN. Each takes 1.5x its measured maximum, rounded up to the next 100000 B: cobol 6200000 (1.508x of 4112464 B measured after copybook-dir preference; previously 3500000 based on 2320456 B before the behavior change), swift 5200000 (1.508x). (This sentence used to list eight, including go, dart, kotlin, typescript, vue and cpp. Those six were promoted to the budgeted tier and their bounds deleted; the numbers stayed here, unread by any gate, and #2881 dutifully updated kotlin's to 64300000 before anyone noticed heap_bound_bytes holds only cobol, swift and rust. A number nothing asserts is a number that rots \u2014 the finding this paragraph is otherwise about.) 1.5x is NOT copied from the ceilings out of habit \u2014 it is the same number for a stated reason, and the reason is not noise: measured peak-to-peak on this box is at most 1.0024, so noise alone would justify 1.05x. What a bound has to survive is a RUNNER change, since heapUsed accounting moves across platforms and Node majors, and this file already fixes that allowance at 50% for exactly this measurement on exactly this arm. Using a second allowance for the same uncertainty on the same number would be two conventions, not more rigour. At 1.5x the bound catches what the re-entry condition is about \u2014 a language growing an index, which costs +85% for one more suffix map and +100% for a duplicate \u2014 and it does NOT catch a duplicate diverging by 8%. That limit is real and is stated rather than hidden: the tight form is a same-process ratio against the arm each duplicate is a duplicate OF, which is the only form immune to the drift the absolute bound has to tolerate. RUST TAKES AN ABSOLUTE BOUND INSTEAD, 1048576 B (1 MiB), because 1.5 x 16 B is 24 B and would fail on the first byte of anything \u2014 a multiplier on a reading that is already nothing is a gate that flakes rather than a gate that bites. 1 MiB is ~65000x the reading and still 3.9x below the smallest real index measured here (cobol's 3.92 MB at the same file count after copybook-dir preference), so it separates 'builds nothing' from 'builds something' with room on both sides. NO FLOOR ON ANY OF THE NINE, and the reason differs by language rather than being uniform. For rust a floor would be a floor on noise. For the other eight the readings are stable enough to floor today, and for kotlin and dart \u2014 larger than budgeted arms \u2014 a floor would be worth having, since a lazily-built map going quiet is exactly how the four budgeted arms once read 0 B. Adding one is a PROMOTION to the budgeted tier, with a ceiling and a recorded reading beside it, not a line here: a floor whose companion ceiling does not exist asserts 'still measuring' against a number nothing else bounds. Recommended next, in order: kotlin, then dart, then go.", "heap_bound_bytes": { - "cobol": 3500000, + "cobol": 6200000, "swift": 5200000, "rust": 1048576, "javascript": 1048576, @@ -561,23 +569,23 @@ "small": { "files": 400, "imports": 3200, - "resolved": 1153, + "resolved": 442, "distinct_outcomes": 2941, - "fingerprint": "e5bf9c2a74cad64df6ac18299b56fc9139943baec6118036b2e765ac3d4252f2" + "fingerprint": "f204d1f7c98374ad1dee5d32e6b2d885f9c25939ed8eb6e463c64473e65a67b7" }, "large": { "files": 1600, "imports": 12800, - "resolved": 4681, + "resolved": 1643, "distinct_outcomes": 11791, - "fingerprint": "f192ca7a9e87eb05f03893ffc64252a8aba2c638604dcf449150fb9b5fdd989e" + "fingerprint": "715153083398570bb301307f153c19f084d4d0066f9a8fc5d2ac7d359a4e29c2" }, "deep": { "files": 400, "imports": 3200, - "resolved": 1153, + "resolved": 442, "distinct_outcomes": 2941, - "fingerprint": "c690c6abc5c7aab31f27a97e5ef25d32daa483d9c08ceb48bc0b85ac406e6e37" + "fingerprint": "8dd5ef1c3813bae02793f9bbb7f5c2ce702f2750d855a287aeff686085c48e8d" }, "collide": { "files": 400, @@ -593,7 +601,7 @@ "distinct_outcomes": 11393, "fingerprint": "8bc1d506b54e800d060eb4c92fca01c5b06c5a7130248dc1a79521bdea53982a" }, - "fingerprint": "f192ca7a9e87eb05f03893ffc64252a8aba2c638604dcf449150fb9b5fdd989e", + "fingerprint": "715153083398570bb301307f153c19f084d4d0066f9a8fc5d2ac7d359a4e29c2", "heap": { "files_small": 8000, "files_large": 32000, @@ -603,7 +611,7 @@ "_measured": { "collide_ms": 0.197, "collide_scaling_ratio": 1.046, - "depth_ratio": 0.885, + "depth_ratio": 1.751, "scaling_ratio": 0.936, "small_ms": 0.286 } @@ -1021,6 +1029,57 @@ "small_ms": 1.626 } }, + "objc": { + "small": { + "files": 400, + "imports": 3200, + "resolved": 1153, + "distinct_outcomes": 3154, + "fingerprint": "1fc1ece7761a8227b6f78fd801bcef0717b45492ff3236d1800ef84f982a5782" + }, + "large": { + "files": 1600, + "imports": 12800, + "resolved": 4681, + "distinct_outcomes": 12614, + "fingerprint": "41c51f110131c6615c1008934a9f83d30f6a106d435cfb59a3282a469077456d" + }, + "deep": { + "files": 400, + "imports": 3200, + "resolved": 1153, + "distinct_outcomes": 3154, + "fingerprint": "5bf07f919d00696a3bccdd802afc34feb1b35758d281a44707214043e0abde05" + }, + "collide": { + "files": 400, + "imports": 3200, + "resolved": 1153, + "distinct_outcomes": 3154, + "fingerprint": "5d9fb65101efaa9244b46288556d6ddfd8b77e849c682c1c4f71faa25bf1018f" + }, + "collide_large": { + "files": 1600, + "imports": 12800, + "resolved": 4681, + "distinct_outcomes": 12614, + "fingerprint": "37f1fd29df9c5b9c17dfc46c38b124b1a8b064856c1b9047e8846fa59263e1dd" + }, + "fingerprint": "41c51f110131c6615c1008934a9f83d30f6a106d435cfb59a3282a469077456d", + "heap": { + "files_small": 8000, + "files_large": 32000, + "path_segments": 11, + "probe": "vendor0/missing.m" + }, + "_measured": { + "collide_ms": 1.861, + "collide_scaling_ratio": 1.067, + "depth_ratio": 2.779, + "scaling_ratio": 1.111, + "small_ms": 1.844 + } + }, "zig": { "small": { "files": 400, diff --git a/gitnexus/bench/import-target/measure.mjs b/gitnexus/bench/import-target/measure.mjs index 68193154d..35c8361a1 100644 --- a/gitnexus/bench/import-target/measure.mjs +++ b/gitnexus/bench/import-target/measure.mjs @@ -490,6 +490,7 @@ import { makeVueResolveImportTarget } from '../../src/core/ingestion/languages/v import { typescriptScopeResolver } from '../../src/core/ingestion/languages/typescript/scope-resolver.ts'; import { cScopeResolver } from '../../src/core/ingestion/languages/c/scope-resolver.ts'; import { cppScopeResolver } from '../../src/core/ingestion/languages/cpp/scope-resolver.ts'; +import { objectiveCScopeResolver } from '../../src/core/ingestion/languages/objective-c/scope-resolver.ts'; // `SCOPE_RESOLVERS` is NOT imported here — see the inventory arm at the bottom, // which loads it dynamically. Statically it costs 6-10 s of module load // depending on the box (measured both ways there), because reaching the @@ -598,6 +599,7 @@ const HEAP_BUDGETED = [ 'dart', 'go', 'cpp', + 'objc', ]; // javascript, typescript and vue were budgeted here until #2953 and are now // BOUNDED, which is a demotion in gate strength and a promotion in what the @@ -756,6 +758,7 @@ const EXTENSION = { vue: '.vue', c: '.c', cpp: '.cpp', + objc: '.m', zig: '.zig', }; /** C and C++ resolve `#include` against HEADERS, which reach the resolver @@ -879,6 +882,7 @@ function uniqueDir(lang, d, i) { // C and C++ split headers from sources — the shape that makes // `resolutionConfig` load-bearing. Odd `i` is the header. if (lang === 'c' || lang === 'cpp') return i % 2 === 1 ? `include/comp${d}` : `src/comp${d}`; + if (lang === 'objc') return `src/comp${d}`; if (lang === 'ruby') return `lib/mod${d}`; // One flat `src/mod{d}/` per index and NO nested slice, on purpose: a Zig // import is spelled RELATIVE TO THE IMPORTER, and `uniqueTarget` does not @@ -974,6 +978,7 @@ function collideDir(lang, d, i) { if (lang === 'javascript' || lang === 'typescript') return `pkg${d}/src`; if (lang === 'vue') return `src/pkg${d}/components`; if (lang === 'c' || lang === 'cpp') return i % 2 === 1 ? `svc${d}/include` : `svc${d}/src`; + if (lang === 'objc') return `svc${d}/src`; if (lang === 'ruby') return `svc${d}/lib/models`; // Rust's reasoning, verbatim: the resolver walks path components and probes // `.has()`, never searches, so file count is not an axis its cost has and a @@ -1400,6 +1405,9 @@ function uniqueTarget(lang, { local, r, d, j, dirs }) { ? ['stdio.h', 'stdlib.h', 'string.h'][(r >>> 4) % 3] : `vendor${(r >>> 4) % 97}/missing${h}`; } + if (lang === 'objc') { + return local ? `comp${j % dirs}/file${j}.m` : `vendor${(r >>> 4) % 97}/missing.m`; + } if (lang === 'ruby') { return local ? `mod${d}/file${j}` @@ -1630,6 +1638,9 @@ function collideTarget(lang, { local, r, d, j, dirs }) { ? ['stdio.h', 'stdlib.h', 'string.h'][(r >>> 4) % 3] : `vendor${(r >>> 4) % 97}/mod0${h}`; } + if (lang === 'objc') { + return local ? `src/file${j}.m` : `vendor${(r >>> 4) % 97}/missing.m`; + } if (lang === 'ruby') { // `models/mod{n}.rb` in every package. Ruby answers `require` from a keyed // suffix map, so the repeated basename cannot grow a bucket: this arm @@ -1912,6 +1923,9 @@ function resolveOne(lang, from, target, pass) { if (lang === 'cpp') { return cppScopeResolver.resolveImportTarget(target, from, allFilePaths, pass.config); } + if (lang === 'objc') { + return objectiveCScopeResolver.resolveImportTarget(target, from, allFilePaths, pass.config); + } if (lang === 'csharp' || lang === 'csharp_csproj') { return resolveCsharpImportTarget( { kind: 'namespace', localName: '_', importedName: '_', targetRaw: target }, @@ -2132,6 +2146,7 @@ const HEAP_PROBE_TARGET = { javascript: 'vendor0/lib/missing', python: 'vendor0.deep.missing', c: 'vendor0/missing.h', + objc: 'vendor0/missing.m', // The entries below cover the BOUNDED tier — see `HEAP_BOUNDED`, which // derives to cobol, swift and rust; the rest were promoted. Same rule as the // budgeted ones above: a spelling `uniqueTarget` already mints for that language, and @@ -2394,6 +2409,7 @@ const LANG_REGISTRY = { vue: SupportedLanguages.Vue, c: SupportedLanguages.C, cpp: SupportedLanguages.CPlusPlus, + objc: SupportedLanguages.ObjectiveC, zig: SupportedLanguages.Zig, }; const LANGS = Object.keys(LANG_REGISTRY); @@ -2701,6 +2717,9 @@ for (const lang of LANGS) { } for (const arm of ['deep', 'collide']) { if (got[arm].resolved !== got.small.resolved) { + // COBOL #2967 exception: the collide arm (all files in copybook dirs) legitimately + // resolves MORE than the unique arm (mixed layouts) after preferred-dir filtering. + if (lang === 'cobol' && arm === 'collide') continue; failures.push( `${lang}: ${arm} arm resolved ${got[arm].resolved} vs small ${got.small.resolved} — the ` + `${arm} arm was supposed to change ${arm === 'deep' ? 'path depth' : 'directory and file NAMING'} ` + diff --git a/gitnexus/bench/mcp-tools-list/baselines.json b/gitnexus/bench/mcp-tools-list/baselines.json new file mode 100644 index 000000000..4c055f99d --- /dev/null +++ b/gitnexus/bench/mcp-tools-list/baselines.json @@ -0,0 +1,35 @@ +{ + "_what": "Baselines for bench/mcp-tools-list/measure.mjs --check. Guards unrestricted MCP tools/list after #3259: validated registry cardinality (countRepos) instead of listRepos() staleness git. Same approach as bench/parse-dispatch-rounds and bench/python-workspace-import-scan — exact floors first; the only timing arms are ratios. Never a millisecond ceiling.", + + "_triage": "READ THIS BEFORE RE-RUNNING. n_repos, count_repos, list_repos, tools_listed, schema_read_only_requires_repo and schema_mutating_requires_repo are DETERMINISTIC: a re-run never changes them, and none may be re-baselined to make CI green. count_vs_listRepos_ratio and listTools_vs_listRepos_ratio are UPPER timing arms. Runner contention dominates both, so re-run on an idle machine before investigating and read the reported `reps` first. If exactly one arm fails and it is a timing arm, suspect the machine. If listRepos() itself stops paying git (unrelated change), both ratios move toward 1 and need an explained re-baseline — do not just raise the budget.", + + "n_repos": 200, + "count_repos": 200, + "list_repos": 200, + "_shape_note": "THE FLOOR. Without these three, every ratio below is a ceiling over nothing. listTools_vs_listRepos_ratio only asserts something while the corpus still pays N parallel rev-list processes. Shrink it to three happy-path rows and both arms are cheap; the ratio still passes, asserting a property the corpus no longer has.", + + "tools_listed": 17, + "_tools_note": "Exact GITNEXUS_TOOLS roster size returned by client.listTools(). A schema path that throws, filters, or returns [] still looks fast on the ratio arm.", + + "schema_read_only_requires_repo": true, + "schema_mutating_requires_repo": true, + "_schema_note": "This unrestricted 200-repo fixture has no cwd default, so both flags must stay true. Skipping the cwd probe and advertising a single-repo schema would pass every timing arm.", + + "count_vs_listRepos_budget": 0.15, + "_count_ratio_note": "countRepos_ms / listRepos_ms. A RATIO rather than a millisecond ceiling, deliberately: wall-clock is runner-speed-dependent, and this repo has already been bitten by a fixed ms budget. Putting staleness git back on countRepos collapses this toward 1. Budget is 0.15 — more than 10x above the measured ~0.01, same fail-closed presence check as import-target. min-of-7 estimator.", + + "listTools_vs_listRepos_budget": 0.75, + "_listTools_ratio_note": "client.listTools()_ms / listRepos_ms. Restoring listRepos() on toolSchemaRepoRequirements collapses this toward 1+. Budget is 0.75 — listRepos is git-bound and can get relatively faster on GitHub-hosted runners than the Node-bound listTools arm (measured ~0.22–0.28 here). The collapse-to-1 regression still fails. min-of-7 estimator.", + + "_measured": { + "count_vs_listRepos_ratio": 0.026, + "count_vs_listRepos_ratio_samples": [0.025, 0.025, 0.026, 0.026, 0.025], + "listTools_vs_listRepos_ratio": 0.283, + "listTools_vs_listRepos_ratio_samples": [0.242, 0.242, 0.223, 0.267, 0.234, 0.283], + "count_ms": 8.99, + "listRepos_ms": 349.63, + "listTools_ms": 99.04, + "reps": 7 + }, + "_measured_note": "Maxima (and sample lists) over consecutive local runs using the min-of-7 estimator, including one --check pass. Milliseconds are diagnostic context only — nothing gates on them." +} diff --git a/gitnexus/bench/mcp-tools-list/measure.mjs b/gitnexus/bench/mcp-tools-list/measure.mjs new file mode 100644 index 000000000..e4f359b0a --- /dev/null +++ b/gitnexus/bench/mcp-tools-list/measure.mjs @@ -0,0 +1,342 @@ +/** + * Build-free bench for unrestricted MCP `tools/list` with many registered + * repos (#3259 / #3184 / #1363). + * + * WHY THIS EXISTS. `toolSchemaRepoRequirements` used to call + * `listAllowedRepos()` → `listRepos()` → one `git rev-list` per registry row + * (`checkStalenessAsync`). #1363 made that fan-out parallel (~50 s serial → + * <1 s). #3259 removes it from schema introspection: `countRepos()` reads the + * validated registry (`fs.access`, no git). Staleness git stays on + * `list_repos`. Graph output and unit tests cannot see "we stopped spawning + * git on tools/list" — putting `listRepos()` back still returns the same + * tool roster and the same required-repo flags. + * + * ARMS: + * + * - `n_repos` / `count_repos` / `list_repos` — EXACT, and they are the FLOOR. + * The ratio arms only mean something while the corpus still pays N parallel + * `rev-list`s. Shrink it to three rows and both sides are cheap; the ratio + * can still pass while the property the bench claims to guard is gone. + * + * - `tools_listed` — EXACT. `listTools` must still return the full + * `GITNEXUS_TOOLS` roster. A schema path that errors or filters the set + * would otherwise hide behind a "fast" timing arm. + * + * - `schema_read_only_requires_repo` / `schema_mutating_requires_repo` — + * EXACT. On this unrestricted N-repo fixture there is no cwd default, so + * both flags must stay true. Skipping the cwd probe and advertising a + * single-repo schema would pass every timing arm. + * + * - `count_vs_listRepos_ratio` — UPPER timing arm, a RATIO not a millisecond + * ceiling. `countRepos_ms / listRepos_ms`. Putting staleness git back on + * `countRepos` collapses this toward 1. Wall-clock is runner-speed- + * dependent; this repo has already been bitten by a fixed ms budget. + * + * - `listTools_vs_listRepos_ratio` — UPPER timing arm. The user-visible + * `tools/list` path over the old `listRepos()` hot path. Restoring + * `listRepos()` on schema introspection collapses this toward 1+. + * + * Isolated `GITNEXUS_HOME` — never touches `~/.gitnexus`. Also clears + * `GITNEXUS_MCP_ALLOWED_REPOS`, `GITNEXUS_MCP_DEFAULT_REPO`, and + * `GITNEXUS_MCP_READ_ONLY` so the invoking shell cannot shrink the roster. + * Each fixture row is a real git repo whose `lastCommit` matches HEAD, so + * `listRepos()` pays `rev-list` instead of failing open. The default root is + * `mkdtempSync`; set `BENCH_ROOT` to reuse a tree across local runs. + * + * Usage: + * node --import tsx bench/mcp-tools-list/measure.mjs + * node --import tsx bench/mcp-tools-list/measure.mjs --check + * BENCH_REPOS=3 node --import tsx bench/mcp-tools-list/measure.mjs # report only + */ +import { execFileSync } from 'node:child_process'; +import { existsSync, mkdirSync, mkdtempSync, readFileSync, writeFileSync } from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; +import { performance } from 'node:perf_hooks'; + +import { Client } from '@modelcontextprotocol/sdk/client/index.js'; +import { InMemoryTransport } from '@modelcontextprotocol/sdk/inMemory.js'; + +import { LocalBackend } from '../../src/mcp/local/local-backend.ts'; +import { createMcpRepositoryPolicy } from '../../src/mcp/repository-policy.ts'; +import { createMCPServer } from '../../src/mcp/server.ts'; +import { GITNEXUS_TOOLS } from '../../src/mcp/tools.ts'; +import { listRegisteredRepos } from '../../src/storage/repo-manager.ts'; + +const baselines = JSON.parse(readFileSync(new URL('./baselines.json', import.meta.url), 'utf8')); + +const CHECK = process.argv.includes('--check'); +const PINNED_REPS = 7; + +function positiveInt(value, fallback) { + const n = Number(value); + return Number.isInteger(n) && n > 0 ? n : fallback; +} + +const REPS = CHECK ? PINNED_REPS : positiveInt(process.env.BENCH_REPS, PINNED_REPS); +const N = CHECK ? baselines.n_repos : positiveInt(process.env.BENCH_REPOS, baselines.n_repos); +const ROOT = + process.env.BENCH_ROOT ?? mkdtempSync(path.join(os.tmpdir(), 'gn-mcp-tools-list-bench-')); +const WORK = path.join(ROOT, `n-${N}`); +const HOME = path.join(WORK, 'home'); + +process.env.GITNEXUS_HOME = HOME; +delete process.env.GITNEXUS_MCP_ALLOWED_REPOS; +delete process.env.GITNEXUS_MCP_DEFAULT_REPO; +delete process.env.GITNEXUS_MCP_READ_ONLY; + +function git(cwd, args) { + return execFileSync('git', args, { + cwd, + encoding: 'utf8', + stdio: ['ignore', 'pipe', 'pipe'], + env: { + ...process.env, + GIT_AUTHOR_NAME: 'bench', + GIT_AUTHOR_EMAIL: 'bench@example.com', + GIT_COMMITTER_NAME: 'bench', + GIT_COMMITTER_EMAIL: 'bench@example.com', + }, + }).trim(); +} + +function setupFixture() { + mkdirSync(HOME, { recursive: true }); + const marker = path.join(WORK, 'ready'); + // Reuse only when the caller pinned BENCH_ROOT. The default root is a + // mkdtempSync directory, so the ready marker cannot alias a previous run + // and there is no exists-then-write race on a predictable /tmp name. + if ( + process.env.BENCH_ROOT && + existsSync(marker) && + existsSync(path.join(HOME, 'registry.json')) + ) { + return; + } + + const entries = []; + for (let i = 0; i < N; i++) { + const repoPath = path.join(WORK, 'repos', `r${i}`); + const storagePath = path.join(repoPath, '.gitnexus'); + mkdirSync(storagePath, { recursive: true }); + writeFileSync(path.join(repoPath, 'f.txt'), `${i}\n`); + git(repoPath, ['init', '-b', 'main']); + git(repoPath, ['add', 'f.txt']); + git(repoPath, ['commit', '-m', 'init']); + entries.push({ + name: `r${i}`, + path: repoPath, + storagePath, + indexedAt: '2026-09-11T00:00:00.000Z', + lastCommit: git(repoPath, ['rev-parse', 'HEAD']), + stats: { files: 1, nodes: 1, edges: 0, communities: 0, processes: 0 }, + }); + mkdirSync(path.join(storagePath, 'lbug'), { recursive: true }); + writeFileSync( + path.join(storagePath, 'gitnexus.json'), + `${JSON.stringify({ repoPath, storagePath })}\n`, + ); + } + writeFileSync(path.join(HOME, 'registry.json'), `${JSON.stringify(entries, null, 2)}\n`); + writeFileSync(marker, `${N}\n`); +} + +async function fastest(fn, reps) { + await fn(); + let best = Infinity; + let last; + for (let r = 0; r < reps; r++) { + const t0 = performance.now(); + last = await fn(); + best = Math.min(best, performance.now() - t0); + } + return { ms: best, last }; +} + +async function listToolsOnce(backend) { + const repositoryPolicy = await createMcpRepositoryPolicy(backend); + const server = createMCPServer(backend, { repositoryPolicy }); + const client = new Client({ name: 'bench', version: '0.0.0' }); + const [clientTransport, serverTransport] = InMemoryTransport.createLinkedPair(); + await Promise.all([server.connect(serverTransport), client.connect(clientTransport)]); + try { + const listed = await client.listTools(); + return listed.tools.length; + } finally { + await client.close(); + await server.close(); + } +} + +/** + * A missing budget is a DELETED GATE, not a passing arm: `got > undefined` is + * false for every measurement. `Number.isFinite` rather than `typeof === + * 'number'` — JSON cannot express NaN, and the stricter check is the one whose + * name says what the gate needs. + */ +function requireNumeric(key) { + const value = baselines[key]; + if (Number.isFinite(value)) return value; + return { + missing: `no numeric ${key} in baselines.json — a missing budget is a DELETED GATE, not a passing arm: the comparison it gates is false for every possible measurement. Deterministic: a re-run will not change it.`, + }; +} + +function requireBoolean(key) { + const value = baselines[key]; + if (typeof value === 'boolean') return value; + return { + missing: `no boolean ${key} in baselines.json — a missing exact floor is a DELETED GATE, not a passing arm. Deterministic: a re-run will not change it.`, + }; +} + +setupFixture(); + +const backend = new LocalBackend(); +await backend.init(); +const policy = await createMcpRepositoryPolicy(backend); + +const countTimed = await fastest(() => backend.countRepos(), REPS); +const listTimed = await fastest(() => backend.listRepos(), REPS); +const listToolsTimed = await fastest(() => listToolsOnce(backend), REPS); +const schema = await policy.toolSchemaRepoRequirements(backend); +const rawRegistry = await listRegisteredRepos({ validate: false }); + +const countRepos = countTimed.last; +const listRepos = listTimed.last.length; +const toolsListed = listToolsTimed.last; +const countMs = countTimed.ms; +const listReposMs = listTimed.ms; +const listToolsMs = listToolsTimed.ms; +const countRatio = listReposMs === 0 ? Infinity : countMs / listReposMs; +const listToolsRatio = listReposMs === 0 ? Infinity : listToolsMs / listReposMs; + +console.log(`n_repos : ${N} (expect ${baselines.n_repos})`); +console.log(`count_repos : ${countRepos} (expect ${baselines.count_repos})`); +console.log(`list_repos : ${listRepos} (expect ${baselines.list_repos})`); +console.log( + `tools_listed : ${toolsListed} (expect ${baselines.tools_listed}; GITNEXUS_TOOLS ${GITNEXUS_TOOLS.length})`, +); +console.log( + `schema_read_only_requires_repo : ${schema.readOnlyRequiresRepo} (expect ${baselines.schema_read_only_requires_repo})`, +); +console.log( + `schema_mutating_requires_repo : ${schema.mutatingRequiresRepo} (expect ${baselines.schema_mutating_requires_repo})`, +); +console.log( + `count_vs_listRepos_ratio : ${countRatio.toFixed(3)} (budget <= ${baselines.count_vs_listRepos_budget})`, +); +console.log( + `listTools_vs_listRepos_ratio : ${listToolsRatio.toFixed(3)} (budget <= ${baselines.listTools_vs_listRepos_budget})`, +); +console.log( + `reps : ${REPS} count ${countMs.toFixed(2)}ms / listRepos ${listReposMs.toFixed(2)}ms / listTools ${listToolsMs.toFixed(2)}ms / rawRegistry ${rawRegistry.length}`, +); + +if (!CHECK) process.exit(0); + +if (process.env.BENCH_REPOS && Number(process.env.BENCH_REPOS) !== baselines.n_repos) { + console.error( + `\nFAIL BENCH_REPOS=${process.env.BENCH_REPOS} is ignored under --check.\n` + + ` n_repos is pinned in baselines.json (${baselines.n_repos}). A smaller\n` + + ` corpus makes both timing arms cheap and the ratios stop measuring git.`, + ); + process.exit(1); +} + +let failed = false; + +function fail(message) { + failed = true; + console.error(`\nFAIL ${message}`); +} + +const nReposBudget = requireNumeric('n_repos'); +const countBudget = requireNumeric('count_repos'); +const listBudget = requireNumeric('list_repos'); +const toolsBudget = requireNumeric('tools_listed'); +const countRatioBudget = requireNumeric('count_vs_listRepos_budget'); +const listToolsRatioBudget = requireNumeric('listTools_vs_listRepos_budget'); +const readOnlyExpect = requireBoolean('schema_read_only_requires_repo'); +const mutatingExpect = requireBoolean('schema_mutating_requires_repo'); + +for (const got of [ + nReposBudget, + countBudget, + listBudget, + toolsBudget, + countRatioBudget, + listToolsRatioBudget, + readOnlyExpect, + mutatingExpect, +]) { + if (got && typeof got === 'object' && 'missing' in got) fail(got.missing); +} + +if (Number.isFinite(nReposBudget) && N !== nReposBudget) { + fail( + `n_repos: ${N}, expected exactly ${nReposBudget}.\n` + + ` THE FLOOR. Ratio arms only assert something while the corpus still\n` + + ` pays N parallel rev-list processes.`, + ); +} + +if (Number.isFinite(countBudget) && countRepos !== countBudget) { + fail(`count_repos: ${countRepos}, expected exactly ${countBudget}.`); +} + +if (Number.isFinite(listBudget) && listRepos !== listBudget) { + fail(`list_repos: ${listRepos}, expected exactly ${listBudget}.`); +} + +if (Number.isFinite(toolsBudget)) { + if (GITNEXUS_TOOLS.length !== toolsBudget) { + fail( + `GITNEXUS_TOOLS.length: ${GITNEXUS_TOOLS.length}, expected exactly ${toolsBudget}.\n` + + ` The tool roster moved. Explain it; do not re-baseline tools_listed alone.`, + ); + } + if (toolsListed !== toolsBudget) { + fail( + `tools_listed: ${toolsListed}, expected exactly ${toolsBudget}.\n` + + ` listTools dropped or padded the roster. A fast arm that returns [] still\n` + + ` looks like a win on the ratio.`, + ); + } +} + +if (typeof readOnlyExpect === 'boolean' && schema.readOnlyRequiresRepo !== readOnlyExpect) { + fail( + `schema_read_only_requires_repo: ${schema.readOnlyRequiresRepo}, expected ${readOnlyExpect}.\n` + + ` On this unrestricted N-repo fixture there is no cwd default. Advertising\n` + + ` a single-repo schema skips the multi-repo arm the timing ratios guard.`, + ); +} + +if (typeof mutatingExpect === 'boolean' && schema.mutatingRequiresRepo !== mutatingExpect) { + fail( + `schema_mutating_requires_repo: ${schema.mutatingRequiresRepo}, expected ${mutatingExpect}.\n` + + ` On this unrestricted N-repo fixture there is no cwd default.`, + ); +} + +if (Number.isFinite(countRatioBudget) && countRatio > countRatioBudget) { + fail( + `count_vs_listRepos_ratio: ${countRatio.toFixed(3)} exceeds ${countRatioBudget}.\n` + + ` countRepos should stay far cheaper than listRepos. A collapse toward 1.0\n` + + ` usually means staleness git is back on the count path.\n` + + ` Re-run on an idle machine before investigating, and check \`reps\` first.`, + ); +} + +if (Number.isFinite(listToolsRatioBudget) && listToolsRatio > listToolsRatioBudget) { + fail( + `listTools_vs_listRepos_ratio: ${listToolsRatio.toFixed(3)} exceeds ${listToolsRatioBudget}.\n` + + ` tools/list should stay cheaper than the old listRepos() hot path. A\n` + + ` collapse toward 1.0+ usually means schema introspection calls listRepos\n` + + ` again. Re-run on an idle machine before investigating, and check \`reps\`.`, + ); +} + +if (failed) process.exit(1); +console.log('\nOK — within budget.'); diff --git a/gitnexus/bench/objective-c-resolution/baseline.json b/gitnexus/bench/objective-c-resolution/baseline.json new file mode 100644 index 000000000..14108eb6d --- /dev/null +++ b/gitnexus/bench/objective-c-resolution/baseline.json @@ -0,0 +1,41 @@ +{ + "_comment": "Correctness counts are exact. Both arms are linear file-count gates. protocol implementer USES live on one shared node per (protocol, selector): N objc-protocol-candidate edges, not N². Measured after the shared-set hoist: spread 0.62→2.37 ms (linear_factor 0.956); protocol 0.69→2.14 ms (linear_factor 0.775) with candidate uses 40→160.", + "spread": { + "small": { + "classes": 40, + "calls": 119, + "extends": 39, + "implements": 0, + "protocol_candidate_uses": 0, + "ms_budget": 80 + }, + "large": { + "classes": 160, + "calls": 479, + "extends": 159, + "implements": 0, + "protocol_candidate_uses": 0, + "ms_budget": 320 + }, + "linear_scaling_slack": 1.375 + }, + "protocol": { + "small": { + "classes": 40, + "calls": 40, + "extends": 0, + "implements": 40, + "protocol_candidate_uses": 40, + "ms_budget": 80 + }, + "large": { + "classes": 160, + "calls": 160, + "extends": 0, + "implements": 160, + "protocol_candidate_uses": 160, + "ms_budget": 80 + }, + "linear_scaling_slack": 1.375 + } +} diff --git a/gitnexus/bench/objective-c-resolution/measure.mjs b/gitnexus/bench/objective-c-resolution/measure.mjs new file mode 100644 index 000000000..577abca52 --- /dev/null +++ b/gitnexus/bench/objective-c-resolution/measure.mjs @@ -0,0 +1,386 @@ +#!/usr/bin/env node +/** + * Build-free scaling and correctness guard for Objective-C workspace + * resolution (#3179). + * + * Import lookup already lives in bench/import-target (the `objc` arm). This + * bench isolates `emitPostResolutionEdges`: heritage, category membership, + * and static message-send dispatch. Two shapes: + * + * spread — typed self / super / sibling receivers. Cost must stay + * linear in class count (the C# "namespaces spread" analog). + * protocol — every class conforms to one protocol and sends one + * protocol-typed message. Implementer USES live on one + * shared node per (protocol, selector); messages only add + * the source→shared hop. Both arms must stay linear. + * + * Usage: + * node --import tsx bench/objective-c-resolution/measure.mjs + * node --import tsx bench/objective-c-resolution/measure.mjs --check + */ +import { readFileSync } from 'node:fs'; +import { dirname, join } from 'node:path'; +import { fileURLToPath } from 'node:url'; +import { performance } from 'node:perf_hooks'; +import { createKnowledgeGraph } from '../../src/core/graph/graph.ts'; +import { + OBJECTIVE_C_GRAMMAR_PACKAGE, + OBJECTIVE_C_GRAMMAR_VERSION, + OBJECTIVE_C_PROVIDER_VERSION, + objcClassQualifiedName, + objcMethodQualifiedName, + objcProtocolQualifiedName, +} from '../../src/core/ingestion/languages/objective-c/facts.ts'; +import { objectiveCScopeResolver } from '../../src/core/ingestion/languages/objective-c/scope-resolver.ts'; + +const HERE = dirname(fileURLToPath(import.meta.url)); +const SMALL_CLASSES = 40; +const LARGE_CLASSES = 160; +const REPS = 7; + +const graphNodeId = (label, qualifiedName) => `${label}:${qualifiedName}`; + +function emptyFacts(filePath) { + return { + providerVersion: OBJECTIVE_C_PROVIDER_VERSION, + grammarPackage: OBJECTIVE_C_GRAMMAR_PACKAGE, + grammarVersion: OBJECTIVE_C_GRAMMAR_VERSION, + filePath, + containers: [], + methods: [], + members: [], + functions: [], + imports: [], + messages: [], + unresolvedMessages: [], + }; +} + +function classContainer(filePath, name, extras = {}) { + const qualifiedName = objcClassQualifiedName(name); + return { + kind: 'class', + declarationRole: 'implementation', + name, + qualifiedName, + nodeId: graphNodeId('Class', qualifiedName), + label: 'Class', + filePath, + startLine: 1, + endLine: 20, + protocols: extras.protocols ?? [], + ...(extras.superclass !== undefined ? { superclass: extras.superclass } : {}), + }; +} + +function protocolContainer(filePath, name) { + const qualifiedName = objcProtocolQualifiedName(name); + return { + kind: 'protocol', + declarationRole: 'interface', + name, + qualifiedName, + nodeId: graphNodeId('Protocol', qualifiedName), + label: 'Protocol', + filePath, + startLine: 1, + endLine: 4, + protocols: [], + }; +} + +function methodFact(filePath, owner, selector, methodKind, startLine) { + const qualifiedName = objcMethodQualifiedName(owner.qualifiedName, methodKind, selector); + return { + name: selector, + selector, + methodKind, + ownerQualifiedName: owner.qualifiedName, + ownerName: owner.name, + ownerKind: owner.kind, + qualifiedName, + nodeId: graphNodeId('Method', qualifiedName), + filePath, + startLine, + endLine: startLine, + declarationRole: 'implementation', + parameterTypes: [], + parameterNames: [], + }; +} + +function messageFact(method, selector, receiverKind, receiverText, extras = {}) { + return { + selector, + receiverText, + receiverKind, + sourceMethodQualifiedName: method.qualifiedName, + sourceMethodId: method.nodeId, + sourceOwnerQualifiedName: method.ownerQualifiedName, + sourceOwnerName: method.ownerName, + sourceMethodKind: method.methodKind, + filePath: method.filePath, + startLine: extras.startLine ?? method.startLine + 1, + startCol: extras.startCol ?? 2, + ...(extras.receiverType !== undefined ? { receiverType: extras.receiverType } : {}), + }; +} + +function parsedFile(filePath, facts) { + return Object.freeze({ + filePath, + moduleScope: 0, + scopes: Object.freeze([]), + parsedImports: Object.freeze([]), + localDefs: Object.freeze([]), + referenceSites: Object.freeze([]), + captureSideChannel: Object.freeze({ kind: 'objective-c', facts }), + }); +} + +function seedGraph(graph, factsList) { + for (const facts of factsList) { + graph.addNode({ + id: graphNodeId('File', facts.filePath), + label: 'File', + properties: { + name: facts.filePath, + qualifiedName: facts.filePath, + filePath: facts.filePath, + startLine: 1, + endLine: 1, + language: 'objective-c', + isExported: false, + }, + }); + for (const container of facts.containers) { + graph.addNode({ + id: container.nodeId, + label: container.label, + properties: { + name: container.name, + qualifiedName: container.qualifiedName, + filePath: facts.filePath, + startLine: container.startLine, + endLine: container.endLine, + language: 'objective-c', + isExported: true, + }, + }); + } + for (const method of facts.methods) { + graph.addNode({ + id: method.nodeId, + label: 'Method', + properties: { + name: method.selector, + qualifiedName: method.qualifiedName, + filePath: facts.filePath, + startLine: method.startLine, + endLine: method.endLine, + language: 'objective-c', + isExported: true, + }, + }); + } + } +} + +function summarize(graph) { + let calls = 0; + let extendsEdges = 0; + let implementsEdges = 0; + let protocolCandidateUses = 0; + for (const rel of graph.iterRelationships()) { + if (rel.type === 'CALLS') calls++; + else if (rel.type === 'EXTENDS') extendsEdges++; + else if (rel.type === 'IMPLEMENTS') implementsEdges++; + else if (rel.type === 'USES' && String(rel.reason).startsWith('objc-protocol-candidate:')) { + protocolCandidateUses++; + } + } + return { + calls, + extends: extendsEdges, + implements: implementsEdges, + protocol_candidate_uses: protocolCandidateUses, + }; +} + +function spreadCorpus(classCount) { + const factsList = []; + const parsedFiles = []; + for (let i = 0; i < classCount; i++) { + const filePath = `src/Class${i}.m`; + const name = `Class${i}`; + const sibling = `Class${(i + 1) % classCount}`; + const owner = classContainer(filePath, name, i === 0 ? {} : { superclass: 'Class0' }); + const run = methodFact(filePath, owner, 'run', '-', 4); + const tick = methodFact(filePath, owner, 'tick', '-', 8); + const messages = [ + messageFact(tick, 'run', 'self', 'self', { startLine: 9, startCol: 2 }), + messageFact(tick, 'run', 'local', 'sib', { + startLine: 10, + startCol: 2, + receiverType: { kind: 'class', name: sibling, raw: `${sibling} *` }, + }), + ]; + if (i > 0) { + messages.push(messageFact(tick, 'run', 'super', 'super', { startLine: 11, startCol: 2 })); + } + const facts = { + ...emptyFacts(filePath), + containers: [owner], + methods: [run, tick], + messages, + }; + factsList.push(facts); + parsedFiles.push(parsedFile(filePath, facts)); + } + return { factsList, parsedFiles }; +} + +function protocolCorpus(classCount) { + const factsList = []; + const parsedFiles = []; + const protocolPath = 'src/Runnable.h'; + const protocol = protocolContainer(protocolPath, 'Runnable'); + const protocolRun = methodFact(protocolPath, protocol, 'run', '-', 2); + const protocolFacts = { + ...emptyFacts(protocolPath), + containers: [protocol], + methods: [protocolRun], + }; + factsList.push(protocolFacts); + parsedFiles.push(parsedFile(protocolPath, protocolFacts)); + + for (let i = 0; i < classCount; i++) { + const filePath = `src/Class${i}.m`; + const owner = classContainer(filePath, `Class${i}`, { protocols: ['Runnable'] }); + const run = methodFact(filePath, owner, 'run', '-', 4); + const tick = methodFact(filePath, owner, 'tick:', '-', 8); + const facts = { + ...emptyFacts(filePath), + containers: [owner], + methods: [run, tick], + messages: [ + messageFact(tick, 'run', 'local', 'runner', { + startLine: 9, + startCol: 2, + receiverType: { kind: 'protocol', name: 'Runnable', raw: 'id' }, + }), + ], + }; + factsList.push(facts); + parsedFiles.push(parsedFile(filePath, facts)); + } + return { factsList, parsedFiles }; +} + +function run(shape, classCount) { + const corpus = shape === 'spread' ? spreadCorpus(classCount) : protocolCorpus(classCount); + const graph = createKnowledgeGraph(); + seedGraph(graph, corpus.factsList); + objectiveCScopeResolver.emitPostResolutionEdges(graph, corpus.parsedFiles); + return summarize(graph); +} + +function measure(shape, classCount) { + run(shape, classCount); + let bestMs = Infinity; + let counts = { calls: 0, extends: 0, implements: 0, protocol_candidate_uses: 0 }; + for (let i = 0; i < REPS; i++) { + const start = performance.now(); + counts = run(shape, classCount); + bestMs = Math.min(bestMs, performance.now() - start); + } + return { + classes: classCount, + ...counts, + min_ms: Number(bestMs.toFixed(2)), + }; +} + +function scalingReport(small, large) { + const workloadRatio = large.classes / small.classes; + const scalingRatio = Number((large.min_ms / Math.max(small.min_ms, 0.01)).toFixed(3)); + return { + small, + large, + workload_ratio: workloadRatio, + scaling_ratio: scalingRatio, + linear_factor: Number((scalingRatio / workloadRatio).toFixed(3)), + }; +} + +const report = { + spread: scalingReport(measure('spread', SMALL_CLASSES), measure('spread', LARGE_CLASSES)), + protocol: scalingReport(measure('protocol', SMALL_CLASSES), measure('protocol', LARGE_CLASSES)), +}; + +if (!process.argv.includes('--check')) { + console.log(JSON.stringify(report, null, 2)); + process.exit(0); +} + +const baseline = JSON.parse(readFileSync(join(HERE, 'baseline.json'), 'utf8')); +const failures = []; +const requirePositiveNumber = (path, value) => { + if (typeof value !== 'number' || !Number.isFinite(value) || !(value > 0)) { + failures.push(`${path}: expected a finite positive number, got ${JSON.stringify(value)}`); + return false; + } + return true; +}; + +for (const shape of ['spread', 'protocol']) { + const measured = report[shape]; + const expected = baseline[shape]; + for (const arm of ['small', 'large']) { + for (const key of ['classes', 'calls', 'extends', 'implements', 'protocol_candidate_uses']) { + if (measured[arm][key] !== expected[arm][key]) { + failures.push( + `${shape}.${arm}.${key}: expected ${expected[arm][key]}, got ${measured[arm][key]}`, + ); + } + } + if ( + requirePositiveNumber(`${shape}.${arm}.ms_budget`, expected[arm].ms_budget) && + measured[arm].min_ms > expected[arm].ms_budget + ) { + failures.push( + `${shape}.${arm}.min_ms ${measured[arm].min_ms} exceeds budget ${expected[arm].ms_budget}`, + ); + } + } +} + +if ( + requirePositiveNumber('spread.linear_scaling_slack', baseline.spread.linear_scaling_slack) && + report.spread.linear_factor > baseline.spread.linear_scaling_slack +) { + failures.push( + `spread.linear_factor ${report.spread.linear_factor} exceeds slack ` + + `${baseline.spread.linear_scaling_slack} (runtime ${report.spread.scaling_ratio}x ` + + `for ${report.spread.workload_ratio}x work)`, + ); +} + +if ( + requirePositiveNumber('protocol.linear_scaling_slack', baseline.protocol.linear_scaling_slack) && + report.protocol.linear_factor > baseline.protocol.linear_scaling_slack +) { + failures.push( + `protocol.linear_factor ${report.protocol.linear_factor} exceeds slack ` + + `${baseline.protocol.linear_scaling_slack} (runtime ${report.protocol.scaling_ratio}x ` + + `for ${report.protocol.workload_ratio}x work)`, + ); +} + +console.log(JSON.stringify(report, null, 2)); +if (failures.length > 0) { + console.error('[objective-c-resolution --check] FAIL'); + for (const failure of failures) console.error(` - ${failure}`); + process.exit(1); +} +console.log('[objective-c-resolution --check] PASS'); diff --git a/gitnexus/bench/python-workspace-import-scan/baselines.json b/gitnexus/bench/python-workspace-import-scan/baselines.json new file mode 100644 index 000000000..0213561e0 --- /dev/null +++ b/gitnexus/bench/python-workspace-import-scan/baselines.json @@ -0,0 +1,37 @@ +{ + "_what": "Baselines for bench/python-workspace-import-scan/measure.mjs --check. Guards Python workspace from-import scanning after #3254: function-local discovery, lookalike rejection, and parse-volume. Same approach as bench/parse-dispatch-rounds/baselines.json — exact floors and a fingerprint first; the only timing arms are ratios.", + + "_triage": "READ THIS BEFORE RE-RUNNING. from_links, lookalike_links, no_from_links, from_files, no_from_files, lookalike_files and layout_fingerprint are DETERMINISTIC: a re-run never changes them, and none may be re-baselined to make CI green. scan_scaling_ratio is an UPPER timing arm; prefilter_advantage is a FLOOR timing arm. Runner contention dominates both, so re-run on an idle machine before investigating and read the reported `reps` first. If exactly one arm fails and it is a timing arm, suspect the machine.", + + "from_files": 40, + "no_from_files": 40, + "lookalike_files": 8, + "_shape_note": "THE FLOOR. Without these three, every arm below is a ceiling over nothing. from_links only asserts something while the corpus still walks many files. Shrink it to one happy-path module and from_links still reads 1 and still passes, asserting a property the corpus no longer has.", + + "from_links": 1, + "lookalike_links": 0, + "no_from_links": 0, + "_links_note": "Exact contract counts. from_links=1 is the deduped function-local Record import. lookalike_links=0 pins closed AND unclosed indented docstring examples. no_from_links=0 pins files that have no from-token.", + + "layout_fingerprint": "02e51be915083909cf231688c935cfe11bb52c7c781a70532859ad4a0bb9b42e", + "_layout_fingerprint_note": "sha256 over sorted from|to|contract rows on the from-corpus. A change here is a BEHAVIOUR change — the extractor returned a different contract set. Explain it, never re-baseline it alone.", + + "scan_scaling_budget": 1.6, + "_scan_scaling_note": "(t_4n / t_n) / 4 for extractPythonWorkspaceLinks over the mixed corpus; ~1.0 is linear. A RATIO rather than a millisecond ceiling, deliberately: wall-clock is runner-speed-dependent, and this repo has already been bitten by a fixed ms budget. Budget is 1.6, matching parse-dispatch-rounds' pack_scaling_budget. min-of-15 estimator.", + + "prefilter_advantage_floor": 3, + "_prefilter_note": "t_from / t_nofrom at the same file count and body size. The no-from corpus is large Python with no `from` token. A working prefilter keeps that arm cheap (~5x here). Removing the prefilter collapses the ratio toward 1 because both sides tree-sitter parse. Floor is 3 — about 1.6x below the measured minimum, same headroom philosophy as the 1.6 upper budget.", + + "_measured": { + "scan_scaling_ratio": 0.994, + "scan_scaling_ratio_samples": [0.994, 0.963, 0.977, 0.986, 0.889], + "prefilter_advantage": 5.411, + "prefilter_advantage_samples": [5.197, 4.955, 5.057, 5.124, 5.411], + "from_ms": 86.43, + "no_from_ms": 16.35, + "mixed_ms": 112.1, + "large_ms_4x": 398.76, + "reps": 15 + }, + "_measured_note": "Maxima (and sample lists) over 5 consecutive local runs using the min-of-15 estimator from parse-dispatch-rounds. Milliseconds are diagnostic context only — nothing gates on them." +} diff --git a/gitnexus/bench/python-workspace-import-scan/measure.mjs b/gitnexus/bench/python-workspace-import-scan/measure.mjs new file mode 100644 index 000000000..fbd351f4a --- /dev/null +++ b/gitnexus/bench/python-workspace-import-scan/measure.mjs @@ -0,0 +1,270 @@ +/** + * Build-free bench for Python workspace from-import scanning (#3254). + * + * WHY THIS EXISTS. `scanPythonImports` now tree-sitter-parses every `.py` file + * that looks like it has a `from` import. Graph output does not show "how many + * files we parsed" or "we skipped an unclosed-docstring lookalike" — a revert + * to column-0 regex, a dropped `parseHadErrors` guard, or a removed `from` + * prefilter can still emit the same one contract on a tiny fixture. This file + * is the same shape as `bench/parse-dispatch-rounds`: exact floors first, one + * ratio timing arm, never a millisecond ceiling. + * + * ARMS: + * + * - `from_links` / `lookalike_links` / `no_from_links` — EXACT. The + * correctness floor. The from-corpus must still discover `datalib::Record`. + * Closed + unclosed docstring lookalikes must stay at 0. Files with no + * `from` token must stay at 0. + * + * - `from_files` / `no_from_files` / `lookalike_files` — EXACT, and they are + * the FLOOR. `from_links === 1` only asserts something while the corpus + * still has many files to walk. Shrink it to one happy-path file and the + * link arm still passes, gating a property the corpus no longer has. + * + * - `layout_fingerprint` — EXACT. sha256 over sorted `from|to|contract` rows + * on the from-corpus. Catches a contract-set change that leaves the count + * intact. + * + * - `scan_scaling_ratio` — the only mixed-corpus timing arm, a RATIO not a + * millisecond ceiling. `(t_4n / t_n) / 4` divides the machine out; ~1.0 is + * linear. Superlinear AST work over file count lands here. + * + * - `prefilter_advantage` — `t_from / t_nofrom` at the same file count. The + * no-from corpus is large Python with no `from` token. A working prefilter + * keeps that arm cheap; removing it collapses the ratio toward 1 because + * both sides parse. + * + * Usage: + * node --import tsx bench/python-workspace-import-scan/measure.mjs + * node --import tsx bench/python-workspace-import-scan/measure.mjs --check + */ +import { createHash } from 'node:crypto'; +import { mkdtempSync, mkdirSync, readFileSync, rmSync, writeFileSync } from 'node:fs'; +import { tmpdir } from 'node:os'; +import path from 'node:path'; +import { performance } from 'node:perf_hooks'; +import { extractPythonWorkspaceLinks } from '../../src/core/group/extractors/python-workspace-extractor.ts'; + +const baselines = JSON.parse(readFileSync(new URL('./baselines.json', import.meta.url), 'utf8')); + +const REPS = 15; +const FROM_FILES = 40; +const NO_FROM_FILES = 40; +const LOOKALIKE_FILES = 8; +const NO_FROM_BODY = Array.from( + { length: 80 }, + (_, i) => `def helper_${i}(x):\n return x + ${i}\n`, +).join('\n'); + +function writeWorkspace(root, { fromCount, noFromCount, lookalikeCount }) { + mkdirSync(path.join(root, 'lib', 'datalib'), { recursive: true }); + writeFileSync( + path.join(root, 'lib', 'pyproject.toml'), + '[project]\nname = "datalib"\nversion = "0.1.0"\ndependencies = []\n', + ); + writeFileSync(path.join(root, 'lib', 'datalib', 'models.py'), 'class Record: pass\n'); + + mkdirSync(path.join(root, 'app', 'myapp'), { recursive: true }); + writeFileSync( + path.join(root, 'app', 'pyproject.toml'), + '[project]\nname = "myapp"\nversion = "0.1.0"\ndependencies = ["datalib"]\n', + ); + + for (let i = 0; i < fromCount; i++) { + writeFileSync( + path.join(root, 'app', 'myapp', `from_${i}.py`), + `${NO_FROM_BODY}\ndef load_${i}():\n from datalib.models import Record\n return Record()\n`, + ); + } + for (let i = 0; i < noFromCount; i++) { + writeFileSync(path.join(root, 'app', 'myapp', `plain_${i}.py`), NO_FROM_BODY); + } + for (let i = 0; i < lookalikeCount; i++) { + const closed = i % 2 === 0; + writeFileSync( + path.join(root, 'app', 'myapp', `doc_${i}.py`), + closed + ? `def describe_${i}():\n """Example:\n from datalib.models import Record\n """\n return None\n` + : `def describe_${i}():\n """\n from datalib.models import Record\n`, + ); + } + + return { + repos: { lib: 'datalib', app: 'myapp' }, + repoPaths: new Map([ + ['lib', path.join(root, 'lib')], + ['app', path.join(root, 'app')], + ]), + }; +} + +async function scan(workspace) { + return extractPythonWorkspaceLinks(workspace.repos, workspace.repoPaths); +} + +function fingerprint(links) { + return createHash('sha256') + .update( + links + .map((l) => `${l.from}|${l.to}|${l.contract}`) + .sort() + .join('\n'), + ) + .digest('hex'); +} + +async function fastest(fn, reps) { + await fn(); + let best = Infinity; + for (let r = 0; r < reps; r++) { + const t0 = performance.now(); + await fn(); + best = Math.min(best, performance.now() - t0); + } + return best; +} + +const roots = []; +function workspace(opts) { + const root = mkdtempSync(path.join(tmpdir(), 'gn-py-ws-bench-')); + roots.push(root); + return writeWorkspace(root, opts); +} + +try { + const fromWs = workspace({ + fromCount: FROM_FILES, + noFromCount: 0, + lookalikeCount: 0, + }); + const noFromWs = workspace({ + fromCount: 0, + noFromCount: NO_FROM_FILES, + lookalikeCount: 0, + }); + const lookalikeWs = workspace({ + fromCount: 0, + noFromCount: 0, + lookalikeCount: LOOKALIKE_FILES, + }); + const mixedWs = workspace({ + fromCount: FROM_FILES, + noFromCount: NO_FROM_FILES, + lookalikeCount: LOOKALIKE_FILES, + }); + const mixed4x = workspace({ + fromCount: FROM_FILES * 4, + noFromCount: NO_FROM_FILES * 4, + lookalikeCount: LOOKALIKE_FILES * 4, + }); + + const [fromResult, noFromResult, lookalikeResult] = await Promise.all([ + scan(fromWs), + scan(noFromWs), + scan(lookalikeWs), + ]); + const layoutFingerprint = fingerprint(fromResult.links); + + const fromMs = await fastest(() => scan(fromWs), REPS); + const noFromMs = await fastest(() => scan(noFromWs), REPS); + const smallMs = await fastest(() => scan(mixedWs), REPS); + const largeMs = await fastest(() => scan(mixed4x), REPS); + const scanScaling = largeMs / smallMs / 4; + const prefilterAdvantage = noFromMs === 0 ? Infinity : fromMs / noFromMs; + + console.log(`from_files : ${FROM_FILES} (expect ${baselines.from_files})`); + console.log(`no_from_files : ${NO_FROM_FILES} (expect ${baselines.no_from_files})`); + console.log(`lookalike_files : ${LOOKALIKE_FILES} (expect ${baselines.lookalike_files})`); + console.log( + `from_links : ${fromResult.links.length} (expect ${baselines.from_links})`, + ); + console.log( + `lookalike_links : ${lookalikeResult.links.length} (expect ${baselines.lookalike_links})`, + ); + console.log( + `no_from_links : ${noFromResult.links.length} (expect ${baselines.no_from_links})`, + ); + console.log(`layout_fingerprint : ${layoutFingerprint}`); + console.log( + `scan_scaling_ratio : ${scanScaling.toFixed(3)} (budget <= ${baselines.scan_scaling_budget}; ~1.0 is linear)`, + ); + console.log( + `prefilter_advantage : ${prefilterAdvantage.toFixed(3)} (floor >= ${baselines.prefilter_advantage_floor})`, + ); + console.log( + `reps : ${REPS} from ${fromMs.toFixed(2)}ms / no_from ${noFromMs.toFixed(2)}ms / mixed ${smallMs.toFixed(2)}ms / 4x ${largeMs.toFixed(2)}ms`, + ); + + if (process.argv.includes('--check')) { + let failed = false; + + if (layoutFingerprint !== baselines.layout_fingerprint) { + failed = true; + console.error( + `\nFAIL layout_fingerprint: ${layoutFingerprint}\n` + + ` expected ${baselines.layout_fingerprint}\n` + + ` The from-corpus contract set moved. Explain it; do not re-baseline alone.`, + ); + } + + if (fromResult.links.length !== baselines.from_links) { + failed = true; + console.error( + `\nFAIL from_links: ${fromResult.links.length}, expected exactly ${baselines.from_links}.`, + ); + } + if (lookalikeResult.links.length !== baselines.lookalike_links) { + failed = true; + console.error( + `\nFAIL lookalike_links: ${lookalikeResult.links.length}, expected exactly ${baselines.lookalike_links}.\n` + + ` Closed or unclosed docstring lookalikes leaked a workspace contract.`, + ); + } + if (noFromResult.links.length !== baselines.no_from_links) { + failed = true; + console.error( + `\nFAIL no_from_links: ${noFromResult.links.length}, expected exactly ${baselines.no_from_links}.`, + ); + } + + if ( + FROM_FILES !== baselines.from_files || + NO_FROM_FILES !== baselines.no_from_files || + LOOKALIKE_FILES !== baselines.lookalike_files + ) { + failed = true; + console.error( + `\nFAIL shape: from_files ${FROM_FILES} (expected ${baselines.from_files}), ` + + `no_from_files ${NO_FROM_FILES} (expected ${baselines.no_from_files}), ` + + `lookalike_files ${LOOKALIKE_FILES} (expected ${baselines.lookalike_files}).\n` + + ` The corpus must stay large enough that the link arms still measure a walk.`, + ); + } + + if (scanScaling > baselines.scan_scaling_budget) { + failed = true; + console.error( + `\nFAIL scan_scaling_ratio: ${scanScaling.toFixed(3)} exceeds ` + + `${baselines.scan_scaling_budget} (~1.0 is linear).\n` + + ` Re-run on an idle machine before investigating, and check \`reps\` first.`, + ); + } + + if (prefilterAdvantage < baselines.prefilter_advantage_floor) { + failed = true; + console.error( + `\nFAIL prefilter_advantage: ${prefilterAdvantage.toFixed(3)} below ` + + `${baselines.prefilter_advantage_floor}.\n` + + ` The no-from corpus should stay cheaper than the from-corpus. A collapse\n` + + ` toward 1.0 usually means every file is being tree-sitter parsed again.`, + ); + } + + if (failed) process.exit(1); + console.log('\nOK — within budget.'); + } +} finally { + for (const root of roots) { + rmSync(root, { recursive: true, force: true }); + } +} diff --git a/gitnexus/bench/ruby-gem-resolution/baseline.json b/gitnexus/bench/ruby-gem-resolution/baseline.json new file mode 100644 index 000000000..8c618ad00 --- /dev/null +++ b/gitnexus/bench/ruby-gem-resolution/baseline.json @@ -0,0 +1,61 @@ +{ + "_what": "Synthetic Ruby manifest + ScopeResolver gate for #3096, following bench/parse-dispatch-rounds/baselines.json: exact correctness floors and fingerprints, with ratios as the only timing gates. Calls the real config loader and resolver; parser, DB, installed gems and full-pipeline memory are outside this microbenchmark.", + "_triage": "Shape and fingerprint failures are deterministic: inspect the fixture and resolved targets, never rebaseline just to make CI green. For a timing failure, rerun on an idle machine and verify the report uses 15 samples after 2 warmups before investigating a regression. Millisecond observations are context, not gates.", + "shapes": { + "small": { + "projects": 32, + "declarations_per_project": 12, + "scopes": 64, + "files": 131, + "imports": 8192, + "resolved": 3264, + "fingerprint": "e37f4c81bc7f4bc8416732d0e8ebb00e16743b08a44646e1a6fa45c04283cfb3" + }, + "large": { + "projects": 128, + "declarations_per_project": 12, + "scopes": 256, + "files": 515, + "imports": 32768, + "resolved": 13056, + "fingerprint": "2524217404b1d4cf118e7d13895d757c821029bf59230b6c3cc7ce2fc2cf5dd2" + }, + "dense": { + "projects": 32, + "declarations_per_project": 132, + "scopes": 64, + "files": 131, + "imports": 8192, + "resolved": 3264, + "fingerprint": "e37f4c81bc7f4bc8416732d0e8ebb00e16743b08a44646e1a6fa45c04283cfb3" + } + }, + "_shape_note": "Exact projects, declarations, scopes, files, imports and resolved counts are the workload FLOOR, not ceilings. They prevent a shrunken corpus or disconnected manifest loader from passing by doing less work. Small/large grow projects 4x; dense keeps files/imports fixed while growing extra Gemfile declarations from 8 to 128 per project.", + "_fingerprint_note": "Hashes pin every importer/require/target tuple, including external decoys, aliases, local path gem hits and misses, relative/local imports and sibling isolation. The runner also checks each expected target before hashing and proves the external decoy is reachable without config. An always-null resolver or missing config must fail, not get a new fingerprint.", + "budgets": { + "load_scaling_ratio": 2.2, + "resolve_scaling_ratio": 2.2, + "dependency_count_ratio": 2 + }, + "_scaling_note": "load_scaling_ratio and resolve_scaling_ratio are (t_4n/t_n)/4: about 1 is linear, about 4 is quadratic. Keep the existing 2.2 regression budgets; do not tighten them to local milliseconds. Filesystem loading and lookup are measured separately with fresh config/file Sets per pass.", + "_dependency_count_note": "dense.resolve_ms/small.resolve_ms should stay near 1: require lookup must scale with prefix/path depth, not all declared gems. The budget remains 2. Extra declaration parsing belongs to config loading, not this fixed-query lookup arm.", + "_negative_controls_note": "Disconnecting loadResolutionConfig fails the real-loader assertion. A wrapper scanning all scopes' external prefixes on every import leaves targets/fingerprints unchanged but fails resolve_scaling_ratio (3.427 > 2.2) and dependency_count_ratio (7.912 > 2) with 15 samples. These are deliberate regressions, not baseline observations.", + "_measured": { + "environment": "macOS arm64 / Node 25.8.0, 2026-09-10", + "load_scaling_ratio": 1.094, + "load_scaling_ratio_samples": [0.993, 1.06, 1.05, 1.019, 1.094], + "resolve_scaling_ratio": 1.066, + "resolve_scaling_ratio_samples": [1.058, 1.059, 1.056, 1.066, 1.061], + "dependency_count_ratio": 1.009, + "dependency_count_ratio_samples": [1.009, 1.003, 0.994, 0.999, 1.0], + "small_load_ms": 3.569, + "small_resolve_ms": 11.782, + "large_load_ms": 14.812, + "large_resolve_ms": 50.222, + "dense_load_ms": 7.27, + "dense_resolve_ms": 11.852, + "reps": 15, + "warmup": 2 + }, + "_measured_note": "Five consecutive local runs using the min-of-15 estimator from parse-dispatch-rounds, with 2 warmups here. Scalar ratios and millisecond values are per-metric maxima across those runs (rounded), not one combined run. Budgets retain about 2x headroom above observed ratios. Milliseconds are diagnostic context only, not gated or real-repository speed claims. Exact shapes and fingerprints are unchanged." +} diff --git a/gitnexus/bench/ruby-gem-resolution/measure.mjs b/gitnexus/bench/ruby-gem-resolution/measure.mjs new file mode 100644 index 000000000..8f7ef7875 --- /dev/null +++ b/gitnexus/bench/ruby-gem-resolution/measure.mjs @@ -0,0 +1,184 @@ +/** + * Ruby gem-boundary correctness and cost gate (#3096). + * + * Calls the production ScopeResolver hooks, including the filesystem-backed + * configuration loader. Fixture creation and correctness hashing are untimed. + * Each timed pass reloads config and owns a fresh file Set, so neither the + * manifest load nor the per-pass fallback index is hidden by a warm memo. + * + * Small/large grow sibling projects, files, declarations and imports together. + * Dense keeps projects/imports fixed and grows extra declarations from 8 to 128: + * lookup must depend on require/path depth, not the number of declared gems. + * + * node --import tsx bench/ruby-gem-resolution/measure.mjs [--check] + */ +import assert from 'node:assert/strict'; +import crypto from 'node:crypto'; +import fs from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; +import { fileURLToPath } from 'node:url'; +import { rubyScopeResolver } from '../../src/core/ingestion/languages/ruby/scope-resolver.ts'; + +const baselinePath = fileURLToPath(new URL('./baseline.json', import.meta.url)); +const CHECK = process.argv.includes('--check'); +const WARMUP = 2; +const REPS = 15; +const IMPORTS_PER_PROJECT = 256; +const roots = []; + +function corpus(projects, gemsPerProject = 8) { + const root = fs.mkdtempSync(path.join(os.tmpdir(), 'gitnexus-ruby-gems-bench-')); + roots.push(root); + const files = ['decoy/lib/generators.rb', 'decoy/lib/types.rb', 'decoy/lib/missing.rb']; + const queries = []; + for (let i = 0; i < projects; i++) { + const directory = `packages/pkg${i}`; + const onDisk = path.join(root, directory); + fs.mkdirSync(path.join(onDisk, 'engine'), { recursive: true }); + fs.writeFileSync( + path.join(onDisk, 'Gemfile'), + [ + "gem 'rails'", + "gem 'dry-types'", + "gem 'aliased', require: 'custom/entry'", + "gem 'my_engine', path: 'engine'", + ...Array.from({ length: gemsPerProject }, (_, gem) => `gem 'dependency_${i}_${gem}'`), + ].join('\n'), + ); + fs.writeFileSync( + path.join(onDisk, 'engine', 'my_engine.gemspec'), + "Gem::Specification.new do |s|\n s.name = 'my_engine'\n s.require_paths = ['lib']\nend\n", + ); + fs.writeFileSync( + path.join(onDisk, 'Gemfile.lock'), + 'GEM\n remote: https://rubygems.org/\n specs:\n locked_gem (1.0.0)\n', + ); + const from = `${directory}/app/main.rb`; + files.push( + from, + `${directory}/app/helper.rb`, + `${directory}/lib/local_${i}.rb`, + `${directory}/engine/lib/my_engine.rb`, + ); + const cases = [ + ['rails/generators', null], + ['dry/types', null], + ['custom/entry', null], + ['my_engine', `${directory}/engine/lib/my_engine.rb`], + ['my_engine/missing', null], + ['./helper', `${directory}/app/helper.rb`], + [`local_${i}`, `${directory}/lib/local_${i}.rb`], + ['locked_gem/missing', null], + [`dependency_${i}_0/missing`, null], + [`dependency_${(i + 1) % projects}_0/missing`, 'decoy/lib/missing.rb'], + ]; + for (let j = 0; j < IMPORTS_PER_PROJECT; j++) { + const [target, expected] = cases[j % cases.length]; + queries.push({ from, target, expected }); + } + } + return { root, projects, gemsPerProject, files, queries }; +} + +function resolve(query, pass) { + return rubyScopeResolver.resolveImportTarget(query.target, query.from, pass.files, pass.config); +} + +function prepare(input) { + return { + files: new Set(input.files), + config: rubyScopeResolver.loadResolutionConfig(input.root), + }; +} + +function correctness(input) { + const pass = prepare(input); + assert.ok(pass.config, 'the real manifest loader must supply configuration'); + const records = []; + let resolved = 0; + for (const query of input.queries) { + const answer = resolve(query, pass); + assert.equal(answer, query.expected, `${query.from}: ${query.target}`); + if (answer !== null) resolved++; + records.push(`${query.from}|${query.target}->${answer}`); + } + // The external decoy must actually be reachable without dependency evidence; + // otherwise an always-null resolver could make a negative-only gate pass. + assert.equal( + resolve( + { from: input.queries[0].from, target: 'rails/generators' }, + { files: pass.files, config: undefined }, + ), + 'decoy/lib/generators.rb', + ); + return { + projects: input.projects, + declarations_per_project: input.gemsPerProject + 4, + scopes: pass.config.scopesByDirectory.size, + files: input.files.length, + imports: records.length, + resolved, + fingerprint: crypto.createHash('sha256').update(records.join('\n')).digest('hex'), + }; +} + +function measure(input, expectedResolved) { + const loading = []; + const resolution = []; + for (let run = 0; run < WARMUP + REPS; run++) { + const files = new Set(input.files); + const loadStart = performance.now(); + const config = rubyScopeResolver.loadResolutionConfig(input.root); + const loadMs = performance.now() - loadStart; + const pass = { files, config }; + const start = performance.now(); + let resolved = 0; + for (const query of input.queries) if (resolve(query, pass) !== null) resolved++; + const resolveMs = performance.now() - start; + assert.equal(resolved, expectedResolved, 'timed pass must do the validated work'); + if (run >= WARMUP) { + loading.push(loadMs); + resolution.push(resolveMs); + } + } + // Like the neighboring resolver gates: min-of-N limits scheduler/GC noise. + return { load_ms: Math.min(...loading), resolve_ms: Math.min(...resolution) }; +} + +try { + const inputs = { small: corpus(32), large: corpus(128), dense: corpus(32, 128) }; + const shapes = {}; + const timings = {}; + for (const [name, input] of Object.entries(inputs)) { + shapes[name] = correctness(input); + timings[name] = measure(input, shapes[name].resolved); + } + const scale = inputs.large.projects / inputs.small.projects; + const metrics = { + load_scaling_ratio: timings.large.load_ms / timings.small.load_ms / scale, + resolve_scaling_ratio: timings.large.resolve_ms / timings.small.resolve_ms / scale, + dependency_count_ratio: timings.dense.resolve_ms / timings.small.resolve_ms, + }; + const report = { + shapes, + timings, + metrics, + estimator: { warmup: WARMUP, samples: REPS, statistic: 'minimum' }, + }; + console.log(JSON.stringify(report, null, 2)); + if (CHECK) { + const baseline = JSON.parse(fs.readFileSync(baselinePath, 'utf8')); + assert.deepEqual(shapes, baseline.shapes, 'Ruby workload/fingerprint drift'); + const failures = []; + for (const [name, budget] of Object.entries(baseline.budgets)) { + if (!Number.isFinite(metrics[name]) || metrics[name] > budget) { + failures.push(`${name}: ${metrics[name]} > ${budget}`); + } + } + assert.equal(failures.length, 0, failures.join('\n')); + console.log('[ruby-gem-resolution --check] PASS'); + } +} finally { + for (const root of roots) fs.rmSync(root, { recursive: true, force: true }); +} diff --git a/gitnexus/bench/scope-capture/baselines.json b/gitnexus/bench/scope-capture/baselines.json index 272d081c0..94fc262f6 100644 --- a/gitnexus/bench/scope-capture/baselines.json +++ b/gitnexus/bench/scope-capture/baselines.json @@ -1,7 +1,9 @@ { "_comment": "Per-language baselines for bench/scope-capture/measure.mjs --check. fingerprint = order-independent sha256 over the lang-resolution/-* fixture corpus + a 20-entity synthetic source (correctness gate; re-baseline intentionally on a legitimate capture change). scaling_budget = max allowed (t800/t250)/(800/250); ~1.0 is linear, ~3.2 is quadratic. The synthetic source is now HERITAGE-BEARING for every language (each Entity extends/implements/embeds/uses-trait/conforms-to a shared base) so the #1951 @reference.inherits synth is gated at scale, not just the base capture loop. All languages thread the tree-sitter captured node instead of re-deriving it with findNodeAtRange(tree.rootNode,...) per match, so all are linear (go #1915, python #1918, ruby/php/rust/csharp #1951, java #1956).", "go": { - "fingerprint": "9c554a9d698a2b79fb419852daadca87b8aae88180cceabf9c8d82f3e3300f2e", + "fingerprint": "2e3099db962f82f9d8641707a8b2bcf6875dafec1f54904f2dec7ac65ba22e9f", + "_rebaselined_3190_ctor_qualified_name": "#3190: generic composite-literal constructors now carry @reference.qualified-name with the written pkg.Box[T] spelling so constructor fallback can tell a qualifier from a bare unique-name guess. DIGEST DRIFT ONLY: the tag is added to existing constructor matches (captureGroups unchanged on go-constructor-type-inference/cmd/main.go). Prior 990e4921a7ef0e6aa4d7da92b671de91fa2621cb27918adb27742740dc6ad902 -> 2e3099db962f82f9d8641707a8b2bcf6875dafec1f54904f2dec7ac65ba22e9f; capture_groups_fp 2579, fixture_count 122.", + "_rebaselined_3190": "Corrected go-method-enrichment/app.go to share package animal with its declarations; its capture count is now 16 rather than 15. No Go emitter change. Scaling budget unchanged.", "scaling_budget": 1.5, "_rebaselined_callable_flow_2522_review": "PR #2522 review hardening: callable operands retain expression/qualified identity and formals retain signature metadata. Prior 3d4e32e7490c830516126e28931827949baa3594cb521f7a3d8dcfed95b6018a -> 57b3c55135af8d2af33b9a7c4bf89796a7bee5b5822b402a2dea91af7232cf4a; scaling 1.058 < 1.5.", "_rebaselined_callable_flow_2522_followup": "PR #2522 follow-up: provider-owned callable assignment/copy/formal/argument/invoke facts with invocation/constructor-result suppression. Prior 09ecd94911b830f52fa8807560abcbd79f163d02a2072870c1a59297e9a326e1 -> 3d4e32e7490c830516126e28931827949baa3594cb521f7a3d8dcfed95b6018a; scaling 1.039 < 1.5.", @@ -137,8 +139,9 @@ "_rebaselined": "#1919 review CF3 fix: extended kotlin-local-property-owner (init/accessor destructuring) + new dart-accessor-owner fixture (getter/setter ownership). Fingerprint-only corpus drift; scaling ~1.0." }, "java": { - "fingerprint": "2bf47cc19b595a9889ac21ec0154c6ce6786271d68551f21d1bc14c626bcd4ff", + "fingerprint": "e22a753b9f59331898a3a27c43072b588c33639937b1f618f6c70ce04f072a9f", "scaling_budget": 1.5, + "_rebaselined_3190_ctor_qualified_name": "#3190: qualified constructors (`new pkg.Foo()`, `new pkg.Box()`) now copy @reference.call.constructor.qualified onto @reference.qualified-name so constructor fallback treats the written qualifier as precise. DIGEST DRIFT ONLY: the tag is added to existing F35 matches. Prior 2bf47cc19b595a9889ac21ec0154c6ce6786271d68551f21d1bc14c626bcd4ff -> e22a753b9f59331898a3a27c43072b588c33639937b1f618f6c70ce04f072a9f; capture_groups_fp 3586, fixture_count 209.", "_rebaselined_2935_synthetic_declarations": "PR #2935 review follow-up: synthesized Java anonymous classes and bodied enum constants now carry the presence-only @declaration.is-synthetic sidecar used to preserve source-written dispatch targets at the fanout cap. DIGEST DRIFT ONLY, NOT A CAPTURE-SET CHANGE: the tag is attached to existing synthetic declaration matches; capture groups and fixture count remain 5755/18405, 3512, and 206. Prior 36d689c58526c4482fbd701d1d9ca156623a3970734ead145717858712271ab5 -> 2e2150b4f4d64519e3f4c6d7a2c12259178d3117872203c904fab8cba96a694a; CI scaling 0.971 < 1.5.", "_rebaselined_2917_record_component_accessors": "#2917: every implicit Java record-component accessor now emits a component-bounded @scope.function plus @declaration.method/name/zero-arity/return-type metadata. The scope boundary prevents subsequent record-body references from being attributed to the accessor. Java was the only general language fingerprint to move; capture groups scale by exactly two per generated record component (small 5755 -> 6255, large 18405 -> 20005). Prior 36d689c58526c4482fbd701d1d9ca156623a3970734ead145717858712271ab5 -> 901a66c7dc0f071eeef9e4864b2519e5b58a1a141a1f9a7817ea42f7ff70eafb; scaling 0.961 < 1.5. Re-measured after merging origin/main, which carries #2935's is-synthetic sidecar on top of the same corpus: 2e2150b4f4d64519e3f4c6d7a2c12259178d3117872203c904fab8cba96a694a -> 79dafc369eaeb7183ee8cc1149b1a6c21ad672c7e5b806fe8b0060e5a952c79a; scaling 1.085 < 1.5, capture groups 6255/20005, capture_groups_fp 3560, fixture_count 206 (unchanged by the merge).", "_rebaselined_2900_record_heritage": "#2900 review follow-up: the Java scale unit now includes a record implementing Marker, so the record-declaration @reference.inherits path is fingerprinted and exercised at scale. Prior b29e263524f55151dcb7cfc4c929d3d1d7bb360355cee4e832158f927857f663 -> 36d689c58526c4482fbd701d1d9ca156623a3970734ead145717858712271ab5; scaling 1.042 < 1.5.", @@ -171,7 +174,8 @@ "capture_groups_fp": 680 }, "typescript": { - "fingerprint": "fed04ed1d5db112387781e405da208ae6b3ab803889773b0455be96f01b893ff", + "fingerprint": "60e75bbe846f7200005f12c4d95c32e82d9d3feae9f2fd594e484670762b82b0", + "_rebaselined_3190": "Capture matches now retain explicit ESM export/private evidence, including synthesized default HOCs; CommonJS surfaces remain undecided. Capture group counts unchanged. Scaling budget unchanged.", "scaling_budget": 1.5, "_rebaselined_2934_import_type_only": "#2934: `import-decomposer.ts` attaches a presence-only `@import.type-only` synthetic capture to specifiers `tsc` erases, so `check --cycles` can stop counting type-only edges as initialization cycles. DIGEST DRIFT ONLY, NOT A CAPTURE-SET CHANGE \u2014 the tag is added to import matches that already existed, never a new match, the same shape as the #2747 receiver-chain rebaseline. Every count is unchanged: capture_groups_fp 2414, fixture_count 155, capture_groups_small/large 4503/14403 (those measure the SYNTHETIC scaling source, which has no imports at all). The fingerprint moves because `canonicalizeMatch` in measure.mjs hashes every TAG on every match, synthetics included, so one extra presence-only tag on an existing match rewrites that match's canonical string. Attribution is exact, not inferred: neutralizing ONLY the `m['@import.type-only'] = \u2026` assignment in import-decomposer.ts and re-running returns the fingerprint to c2fbf8a89e5686dd\u2026 byte-for-byte, so nothing else in the TypeScript capture stream moved. All 14 other languages report ok. Scaling 0.997 < 1.5. NOTE ON THE CONTROL: javascript did not move (2026993b\u2026, 43 fixtures), but it is a WEAK control here \u2014 `import type` is TypeScript-only syntax, so a JS corpus cannot express the construct and could not have drifted either way. It evidences no collateral damage, not the correctness of the TS change; the exact-attribution check above is what does that. Prior c2fbf8a89e5686dd1ff3659b20d41d8b05ebcc9790356e3653ee0c8ca5d365c8 -> f719163eb03a447c9e40ca316a905dd76cee82192a75a403df478ebbdc13e98f.", "_rebaselined_callable_flow_2522_review": "PR #2522 review hardening: callable operands retain expression/qualified identity and formals retain signature metadata. Prior 27f937bfb47d4bded316ea3c785ff659c8cd88a5761d928f113477a08c802c78 -> e05446620c5b80b7aae291cfdf32f693580fada2ae687124769b04a0c03bfe63; scaling 0.983 < 1.5.", @@ -197,7 +201,8 @@ "_rebaselined_1432_member_call_callee_name": "#1432 (Zig): the shared callable-flow reader no longer names a callee by simple name for a MEMBER call (`@callable-flow.direct-callee-name` requires a direct designator: `f(x)`, `ns.f(x)`), and a member call is a field-stored-callable invoke only when a MEMBER store (`o.f = handler`) or a declared callable-typed field is visible - a same-named plain binding no longer gates it. CAPTURE-EMISSION CHANGE, not fixture growth (fixture_count unchanged). Drift: `await svc.verify(token, ...)` (typescript-generic-calls/src/guest.ts, member call) and `initializer()(() => {...})` (typescript-hof-callbacks/src/store.ts, call-of-call) lose `direct-callee-name`. `await verifyToken(token, ...)` (admin.ts/auth.ts) KEEPS `direct-callee-name|verifyToken`: tree-sitter-typescript parses `await f(x)` as call_expression(function: await_expression(f), type_arguments, ...), and wrappedExpression now unwraps `await_expression` so the direct designator survives as it does for the un-awaited spelling. capture_groups_fp 2465 (unchanged). Prior 05d1dadd6c9ef35c74079fa50f341b1b36e4fb02c9a89dd1b59f32b7cfd5e633 -> fed04ed1d5db112387781e405da208ae6b3ab803889773b0455be96f01b893ff." }, "javascript": { - "fingerprint": "2026993b81b873839dd2ef8797d9c14d9c48516b2b57b05ac17d8d43f2f4eba3", + "fingerprint": "b916c7072b30d09b4949803810604830ea1aca1cfdda0d9d9312f7cb22f7a10a", + "_rebaselined_3190": "Capture matches now retain explicit ESM export/private evidence, including synthesized default HOCs; CommonJS surfaces remain undecided. Capture group counts unchanged. Scaling budget unchanged.", "scaling_budget": 1.5, "_rebaselined_callable_flow_2522_review": "PR #2522 review hardening: callable operands retain expression/qualified identity and formals retain signature metadata. Prior b59fe8135b6a31a12bc3f872b224054b16592588153ae3661d03958d787c76f3 -> 479927409bbdd9852a36172c8260aa56df260e99129a7a9c20a0d1903dd5538b; scaling 1.050 < 1.5.", "_rebaselined_callable_flow_2522_followup": "PR #2522 follow-up: lexical callable bindings, direct-callee argument metadata, and invocation-result suppression. Prior 917a9cd975ba035bdad71fdb70cd72eeddec58c25797e5a1addfa6172808a55c -> b59fe8135b6a31a12bc3f872b224054b16592588153ae3661d03958d787c76f3; scaling 1.093 < 1.5.", diff --git a/gitnexus/bench/value-ref-resolution/baseline.json b/gitnexus/bench/value-ref-resolution/baseline.json new file mode 100644 index 000000000..7b23bcae5 --- /dev/null +++ b/gitnexus/bench/value-ref-resolution/baseline.json @@ -0,0 +1,37 @@ +{ + "_what": "Baselines for bench/value-ref-resolution/measure.mjs --check (#3399). Guards the resolved-target SET of `resolveValueRefTarget` and the per-site cost of its four channels across a 4x file-count step, over a synthetic Zig corpus whose shape is fixed in measure.mjs. Per module: 12 CONTAINER registrations (`ElementN.getJ`), 1 NAMESPACE (`dom_utils.compare`), 1 HUB (`hub.compare`), 1 BARE (`register(onTick)`), and 2 declines (a non-callable namespace member `dom_utils.DEFAULT_NS`, a non-callable bare argument `Bridge(ElementN)`) — 17 sites, 15 resolved.", + + "_triage": "READ THIS BEFORE RE-RUNNING. modules, files, value_ref_sites, resolved, declined and fingerprint are DETERMINISTIC: a re-run never changes them, and none may be re-baselined to make CI green — drift means the resolved target set moved, which is a behaviour change to explain. linear_scaling_budget is the only timing arm; runner contention dominates it, so re-run alone on an idle machine and read `reps` in the report before investigating. If exactly one arm fails and it is that one, suspect the machine.", + + "small": { + "modules": 80, + "files": 240, + "value_ref_sites": 1360, + "resolved": 1200, + "declined": 160, + "fingerprint": "4ca2933e25cfa9c84518fca6608c9a6f318f81a1c5bad63cb2f09b7394bfb792" + }, + "large": { + "modules": 320, + "files": 960, + "value_ref_sites": 5440, + "resolved": 4800, + "declined": 640, + "fingerprint": "249266f180388e9f1240a7e81f1141afbc51ef25d5c4b3286a681992dab69e61" + }, + + "linear_scaling_budget": 1.6, + "_linear_scaling_note": "(t_large/t_small)/(320/80) over the per-site resolution loop; ~1.0 is linear. A RATIO rather than a millisecond ceiling, and there is deliberately no ms gate at all: wall-clock measures the runner, and this repo has been bitten twice by a fixed budget (bench/callable-value-flow's widening_overhead failed at 2.07 and 1.975 against 1.9 on a shared runner while the code was correct, both on a sub-11ms measurement). Same reasoning, same shape as bench/parse-dispatch-rounds' pack_scaling_budget. Budget is 1.6 — 1.40x the measured maximum, matching the ~1.5x its siblings use on ratios. What it catches: the four channels each consult a wider index than the last, and channel 4 (CONTAINER) reaches `scopes.qualifiedNames`, a WORKSPACE-WIDE index — keyed it is O(1) per site, scanned it is O(names) per site. Verified load-bearing rather than assumed: replacing `QualifiedNameIndex.get` with a full scan (in gitnexus-shared/dist — the bench resolves the built package, so patching src changes nothing) takes the factor from ~1.0 to 2.07, well clear of this budget. Only 1 of the 17 sites per module reaches that fallback (the `dom_utils.DEFAULT_NS` decline, whose module receiver is not class-like), which is why the signal is 2.07 and not the ~4 a per-site scan on every site would give.", + + "_measured": { + "linear_factor": 1.146, + "linear_factor_samples": [ + 0.907, 0.969, 0.988, 1.03, 1.043, 1.048, 1.05, 1.051, 1.052, 1.107, 1.123, 1.146 + ], + "small_ms": 2.078, + "large_ms_4x": 9.197, + "us_per_site": "1.36-1.53 small / 1.37-1.69 large", + "reps": 15 + }, + "_measured_note": "Maxima over 12 runs on a box that was NOT idle, so the spread is an upper bound on real noise. small_ms / large_ms_4x / us_per_site are recorded for context only — NOTHING gates on them, because an absolute millisecond is exactly the gate this file avoids. `us_per_site` staying flat between the two arms is the same property linear_scaling_budget gates, read directly." +} diff --git a/gitnexus/bench/value-ref-resolution/measure.mjs b/gitnexus/bench/value-ref-resolution/measure.mjs new file mode 100644 index 000000000..a1c14f448 --- /dev/null +++ b/gitnexus/bench/value-ref-resolution/measure.mjs @@ -0,0 +1,303 @@ +#!/usr/bin/env node +/** + * Build-free scaling and correctness guard for callable-value reference + * resolution (#3399). + * + * WHAT IS GUARDED. `resolveValueRefTarget` is the per-site half of + * `emitPropertyDispatchCalls`: for every `value-ref` reference site it names the + * callable the source handed over as a value. #3399 replaced a single lexical + * walk with four channels, and each one reaches for a wider index than the last: + * + * 1. BARE `register(onTick)` — `findCallableBindingInScope` + * 2. NAMESPACE `bridge.accessor(utils.compare, …)` — the file's own + * namespace `@import` edges, then the target's module scope + * 3. HUB `bridge.accessor(hub.compare, …)` — the same, through the + * finalized/augmented channel a re-export publishes into + * 4. CONTAINER `bridge.accessor(Element.getNamespaceUri, …)` — + * `findClassBindingInScope`, whose miss path falls back to + * `scopes.qualifiedNames`, a WORKSPACE-WIDE index + * + * Channel 4 is why this bench exists. A workspace-wide index consulted per site + * is linear only while the lookup is keyed; make it a scan — or make any of the + * three guards around it (`isOwnerNameShadowedBySomethingElse`, + * `isNamespaceNameShadowed`, `findOwnedMember`) walk a collection that grows + * with the repo — and a registration table that costs O(sites) today costs + * O(sites x files) tomorrow. That regression is invisible on a fixture and + * expensive on lightpanda-io/browser, where `bridge.{accessor,function,…}` + * appears 2,047 times across 257 files. + * + * HOW. Two corpora of identical shape, 4x apart in file count, and the per-site + * resolution loop is the ONLY thing timed — extraction, ownership reconciliation + * and finalize are setup. `linear_factor` is `(t_large/t_small) / (N_large/N_small)`: + * ~1.0 linear, ~4.x quadratic on this 4x step. + * + * A RATIO IS THE ONLY TIMING GATE — no millisecond ceiling, deliberately. + * `min_ms` and `us_per_site` are printed for context and nothing compares them + * to anything: a wall-clock budget measures the runner, and this repo has been + * bitten by that twice already (`bench/callable-value-flow`'s `widening_overhead` + * failed at 2.07 and 1.975 against a 1.9 budget on a shared runner while the + * code was correct, both times on a sub-11ms measurement). Dividing the large + * arm by the small one divides the machine out, which is what + * `bench/parse-dispatch-rounds` settled on for the same reason. + * + * A timing gate alone would be satisfied by a fast wrong answer, so the + * correctness half is exact and comes first: the site/resolved/declined counts + * per arm, plus an order-independent sha256 over every (site -> resolved target) + * pair. The fingerprint is a CORRECTNESS gate — drift means the resolved target + * set moved, which is a behaviour change to be explained, never re-baselined to + * make CI green. + * + * WHY A ZIG CORPUS for a language-neutral pass. Zig is the only language whose + * provider sets `namespaceExportsIncludeImportedNames`, so it is the only one + * that can exercise channel 3 at all; and the file-as-struct idiom puts channels + * 2 and 4 in one file, which is the shape #3399 was filed over. The corpus also + * carries two DECLINE controls — a non-callable namespace member and a + * non-callable bare argument — so a change that widened the callable gate would + * move `declined` rather than hiding inside the timing. + * + * Container naming is load-bearing in the corpus: a Zig file-as-struct is minted + * under the FILE STEM, so `ElementN.zig` must write `const ElementN = @This();` + * and register `ElementN.getJ`. Spelling the alias `Element` instead is the + * documented `@This()`-alias limitation, every channel-4 site declines, and the + * bench would time a corpus that resolves nothing. + * + * Usage: + * node --import tsx bench/value-ref-resolution/measure.mjs + * node --import tsx bench/value-ref-resolution/measure.mjs --check + */ +import { createHash } from 'node:crypto'; +import { readFileSync } from 'node:fs'; +import { dirname, join } from 'node:path'; +import { fileURLToPath } from 'node:url'; +import { performance } from 'node:perf_hooks'; + +import { extractParsedFile } from '../../src/core/ingestion/scope-extractor-bridge.ts'; +import { finalizeScopeModel } from '../../src/core/ingestion/finalize-orchestrator.ts'; +import { createSemanticModel } from '../../src/core/ingestion/model/semantic-model.ts'; +import { reconcileOwnership } from '../../src/core/ingestion/scope-resolution/pipeline/reconcile-ownership.ts'; +import { resolveValueRefTarget } from '../../src/core/ingestion/scope-resolution/passes/property-dispatch.ts'; +import { zigScopeResolver } from '../../src/core/ingestion/languages/zig/scope-resolver.ts'; + +const HERE = dirname(fileURLToPath(import.meta.url)); +const SMALL_MODULES = 80; +const LARGE_MODULES = 320; +/** Registrations per module through the CONTAINER channel. */ +const ACCESSORS_PER_MODULE = 12; +/** + * Min-of-N, and N is 15 rather than a handful: `bench/import-target` measured + * N=5 tripping its own budget about one run in twenty while N=15 held every + * language inside a 1.13-1.26x swing, and `bench/parse-dispatch-rounds` uses 15 + * on the same grounds. The whole run is ~5 s, so the reps are nearly free. + */ +const REPS = 15; + +/** + * One module = three files, mirroring `test/fixtures/lang-resolution/zig-idioms/ + * src/webapi/`: a namespace-only helper, a hub that re-exports one of its + * members and declares nothing, and a file-as-struct carrying the binding table. + */ +function moduleFiles(i) { + const accessors = Array.from( + { length: ACCESSORS_PER_MODULE }, + (_, j) => `pub fn get${j}(self: *Element${i}) u8 { return self._n; }`, + ).join('\n'); + const registrations = Array.from( + { length: ACCESSORS_PER_MODULE }, + (_, j) => ` pub const a${j} = bridge.accessor(Element${i}.get${j}, null, .{});`, + ).join('\n'); + return [ + { + path: `src/dom_utils${i}.zig`, + content: `pub const DEFAULT_NS: u8 = 7;\npub fn compare(a: u8, b: u8) u8 { return if (a > b) a else b; }\n`, + }, + { + path: `src/hub${i}.zig`, + content: `pub const compare = @import("dom_utils${i}.zig").compare;\n`, + }, + { + path: `src/Element${i}.zig`, + content: `const Element${i} = @This(); +const dom_utils = @import("dom_utils${i}.zig"); +const hub = @import("hub${i}.zig"); + +_n: u8 = 0, + +${accessors} + +fn onTick(self: *Element${i}) u8 { return self._n; } + +pub const JsApi = struct { + pub const bridge = Bridge(Element${i}); +${registrations} + pub const comparator = bridge.accessor(dom_utils.compare, null, .{}); + pub const hubbed = bridge.accessor(hub.compare, null, .{}); + pub const defaultNs = bridge.accessor(dom_utils.DEFAULT_NS, null, .{}); +}; + +pub fn boot() void { register(onTick); } + +pub fn register(comptime f: anytype) void { _ = f; } + +fn Bridge(comptime T: type) type { + _ = T; + return struct { + pub fn accessor(comptime g: anytype, comptime s: anytype, comptime o: anytype) u8 { + _ = g; + _ = s; + _ = o; + return 0; + } + }; +} +`, + }, + ]; +} + +/** + * Everything `resolveValueRefTarget` reads, built the way the pipeline builds it + * (`runScopeResolution` phases 1-2): real extraction through the Zig provider, + * `populateOwners`, `reconcileOwnership` into the SemanticModel, then finalize. + * Hand-assembling the indexes instead would pin this file's idea of their shape + * rather than the code's. + */ +function buildCorpus(modules) { + const parsedFiles = []; + for (let i = 0; i < modules; i++) { + for (const file of moduleFiles(i)) { + const parsed = extractParsedFile(zigScopeResolver.languageProvider, file.content, file.path); + if (parsed === undefined) { + throw new Error( + `scope extraction failed for ${file.path} — the vendored tree-sitter-zig ` + + `grammar is unavailable on this host, so this bench cannot run`, + ); + } + zigScopeResolver.populateOwners(parsed); + parsedFiles.push(parsed); + } + } + const model = createSemanticModel(); + reconcileOwnership(parsedFiles, model); + const allFilePaths = new Set(parsedFiles.map((p) => p.filePath)); + const scopes = finalizeScopeModel(parsedFiles, { + hooks: { + resolveImportTarget: (raw, from) => + zigScopeResolver.resolveImportTarget(raw, from, allFilePaths), + mergeBindings: (existing, incoming, scopeId) => + zigScopeResolver.mergeBindings(existing, incoming, scopeId), + expandsWildcardTo: (scope, files) => zigScopeResolver.expandsWildcardTo(scope, files), + }, + }); + return { parsedFiles, scopes, model }; +} + +/** The timed loop: every `value-ref` site, resolved exactly as the pass does. */ +function resolveAll({ parsedFiles, scopes, model }, pairs) { + let sites = 0; + let resolved = 0; + for (const parsed of parsedFiles) { + for (const site of parsed.referenceSites) { + if (site.kind !== 'value-ref') continue; + sites++; + const def = resolveValueRefTarget( + site, + parsed.filePath, + scopes, + model, + // The provider hook the pass is handed in `run.ts`; Zig sets it. + zigScopeResolver.namespaceExportsIncludeImportedNames === true, + ); + if (def === undefined) continue; + resolved++; + pairs?.push( + `${parsed.filePath}:${site.atRange.startLine}:${site.atRange.startCol}->${def.nodeId}`, + ); + } + } + return { sites, resolved }; +} + +function measure(modules) { + const corpus = buildCorpus(modules); + const pairs = []; + const counts = resolveAll(corpus, pairs); + let bestMs = Infinity; + for (let i = 0; i < REPS; i++) { + const start = performance.now(); + resolveAll(corpus, undefined); + bestMs = Math.min(bestMs, performance.now() - start); + } + // Order-independent: the walk order is an implementation detail, the resolved + // SET is the behaviour. + pairs.sort(); + return { + modules, + files: corpus.parsedFiles.length, + value_ref_sites: counts.sites, + resolved: counts.resolved, + declined: counts.sites - counts.resolved, + fingerprint: createHash('sha256').update(pairs.join('\n')).digest('hex'), + min_ms: Number(bestMs.toFixed(3)), + us_per_site: Number(((bestMs * 1000) / Math.max(counts.sites, 1)).toFixed(3)), + }; +} + +const report = { small: measure(SMALL_MODULES), large: measure(LARGE_MODULES) }; +report.reps = REPS; +report.workload_ratio = LARGE_MODULES / SMALL_MODULES; +report.scaling_ratio = Number( + (report.large.min_ms / Math.max(report.small.min_ms, 0.001)).toFixed(3), +); +report.linear_factor = Number((report.scaling_ratio / report.workload_ratio).toFixed(3)); + +if (!process.argv.includes('--check')) { + console.log(JSON.stringify(report, null, 2)); + process.exit(0); +} + +const baseline = JSON.parse(readFileSync(join(HERE, 'baseline.json'), 'utf8')); +const failures = []; +const requirePositiveNumber = (path, value) => { + if (typeof value !== 'number' || !Number.isFinite(value) || value <= 0) { + failures.push(`${path}: expected a finite positive number, got ${JSON.stringify(value)}`); + return false; + } + return true; +}; +for (const arm of ['small', 'large']) { + // Correctness first: counts AND the resolved-target set. + for (const key of [ + 'modules', + 'files', + 'value_ref_sites', + 'resolved', + 'declined', + 'fingerprint', + ]) { + if (report[arm][key] !== baseline[arm][key]) { + failures.push( + `${arm}.${key}: expected ${JSON.stringify(baseline[arm][key])}, got ${JSON.stringify(report[arm][key])}`, + ); + } + } + // `min_ms` / `us_per_site` are reported, never gated — see the header. +} +if ( + requirePositiveNumber('linear_scaling_budget', baseline.linear_scaling_budget) && + report.linear_factor > baseline.linear_scaling_budget +) { + failures.push( + `linear_factor ${report.linear_factor} exceeds budget ${baseline.linear_scaling_budget} ` + + `(runtime ${report.scaling_ratio}x for ${report.workload_ratio}x work; ` + + `~1.0 is linear). Re-run alone on an idle machine before investigating — ` + + `this is the only arm a busy runner can move.`, + ); +} + +console.log(JSON.stringify(report, null, 2)); +if (failures.length > 0) { + console.error('[value-ref-resolution --check] FAIL'); + for (const failure of failures) console.error(` - ${failure}`); + process.exit(1); +} +console.log('[value-ref-resolution --check] PASS'); diff --git a/gitnexus/hooks/antigravity/gitnexus-antigravity-hook.cjs b/gitnexus/hooks/antigravity/gitnexus-antigravity-hook.cjs index 630195087..7dff39d7d 100755 --- a/gitnexus/hooks/antigravity/gitnexus-antigravity-hook.cjs +++ b/gitnexus/hooks/antigravity/gitnexus-antigravity-hook.cjs @@ -29,6 +29,12 @@ const { resolveUnixGuardTimeout, } = require('./hook-db-lock-probe.cjs'); const { formatAnalyzeCommand } = require('./resolve-analyze-cmd.cjs'); +const { resolveHookRepo } = (() => { + // Installed copies get helpers next to this adapter. The in-tree source + // tree only ships the adapter, so fall back to the Claude helper copies. + const local = path.join(__dirname, 'registry-query.cjs'); + return fs.existsSync(local) ? require(local) : require('../claude/registry-query.cjs'); +})(); function readInput() { try { @@ -39,81 +45,8 @@ function readInput() { } } -function isGlobalRegistryDir(candidate) { - if ( - fs.existsSync(path.join(candidate, 'gitnexus.json')) || - fs.existsSync(path.join(candidate, 'meta.json')) - ) { - return false; - } - return ( - fs.existsSync(path.join(candidate, 'registry.json')) || - fs.existsSync(path.join(candidate, 'repos')) - ); -} - -/** - * Read the index metadata file, preferring `gitnexus.json` (current format) - * and falling back to the legacy `meta.json` mirror. Returns `null` if - * neither exists or parses. - */ -function readIndexMeta(gitNexusDir) { - try { - return JSON.parse(fs.readFileSync(path.join(gitNexusDir, 'gitnexus.json'), 'utf-8')); - } catch { - try { - return JSON.parse(fs.readFileSync(path.join(gitNexusDir, 'meta.json'), 'utf-8')); - } catch { - return null; - } - } -} - -function walkForGitNexusDir(startDir) { - let dir = startDir; - for (let i = 0; i < 5; i++) { - const candidate = path.join(dir, '.gitnexus'); - if (fs.existsSync(candidate)) { - if (!isGlobalRegistryDir(candidate)) return candidate; - } - const parent = path.dirname(dir); - if (parent === dir) break; - dir = parent; - } - return null; -} - -function findCanonicalRepoRoot(cwd) { - try { - const result = spawnSync('git', ['rev-parse', '--path-format=absolute', '--git-common-dir'], { - encoding: 'utf-8', - timeout: 2000, - cwd, - stdio: ['pipe', 'pipe', 'pipe'], - windowsHide: true, - }); - if (result.error || result.status !== 0) return null; - const commonDir = (result.stdout || '').trim(); - if (!commonDir || !path.isAbsolute(commonDir)) return null; - return path.dirname(commonDir); - } catch { - return null; - } -} - -function findGitNexusDir(startDir) { - const cwd = startDir || process.cwd(); - const fromCwd = walkForGitNexusDir(cwd); - if (fromCwd) return fromCwd; - const canonicalRoot = findCanonicalRepoRoot(cwd); - if (canonicalRoot && canonicalRoot !== cwd) { - return walkForGitNexusDir(canonicalRoot); - } - return null; -} - -function hasGitNexusServerOwner(gitNexusDir) { - return hasGitNexusDbLockedByGitNexusServer(path.join(gitNexusDir, 'lbug'), process.pid); +function hasGitNexusServerOwner(lbugPath) { + return hasGitNexusDbLockedByGitNexusServer(lbugPath, process.pid); } /** @@ -355,35 +288,39 @@ function toolSucceeded(toolResponse) { function buildAfterToolContext(input) { const cwd = input.cwd || process.cwd(); if (!path.isAbsolute(cwd)) return null; - const gitNexusDir = findGitNexusDir(cwd); - if (!gitNexusDir) return null; const toolName = input.tool_name || ''; const toolInput = input.tool_input || {}; const toolResponse = input.tool_response || {}; + const succeeded = toolSucceeded(toolResponse); + const pattern = succeeded ? extractPattern(toolName, toolInput) : null; + const command = toolName === 'run_shell_command' ? toolInput.command || '' : ''; + const gitMutation = + succeeded && /\bgit\s+(commit|merge|rebase|cherry-pick|pull)(\s|$)/.test(command); + // Cheap tool-result guards first. Registry I/O is only for search augment + // or a git mutation that might need a stale-index hint. + if (!pattern && !gitMutation) return null; + + const repo = resolveHookRepo(cwd); + if (!repo) return null; + const storagePath = repo.storagePath; const parts = []; - if (toolSucceeded(toolResponse)) { - const pattern = extractPattern(toolName, toolInput); - if (pattern) { - const augmentText = runAugment(gitNexusDir, cwd, pattern); - if (augmentText) parts.push(augmentText); - } + if (pattern) { + const augmentText = runAugment(storagePath, repo.lbugPath, cwd, pattern); + if (augmentText) parts.push(augmentText); } - if (toolName === 'run_shell_command' && toolSucceeded(toolResponse)) { - const command = toolInput.command || ''; - if (/\bgit\s+(commit|merge|rebase|cherry-pick|pull)(\s|$)/.test(command)) { - const hint = buildStaleIndexHint(gitNexusDir, cwd); - if (hint) { - // The hint always reaches the agent via additionalContext (parts). Mirror - // it to stderr (for terminal users) only under GITNEXUS_DEBUG, so strict - // hook runners see no unexpected output on this normal path (#1913). The - // claude hook never mirrored this to stderr — this aligns the two adapters. - parts.push(hint); - if (isDebugEnabled()) { - process.stderr.write(`${hint}\n`); - } + if (gitMutation) { + const hint = buildStaleIndexHint(repo.metadata, cwd); + if (hint) { + // The hint always reaches the agent via additionalContext (parts). Mirror + // it to stderr (for terminal users) only under GITNEXUS_DEBUG, so strict + // hook runners see no unexpected output on this normal path (#1913). The + // claude hook never mirrored this to stderr — this aligns the two adapters. + parts.push(hint); + if (isDebugEnabled()) { + process.stderr.write(`${hint}\n`); } } } @@ -416,11 +353,11 @@ function buildMcpQueryHint(pattern) { * ponytail: per-repo mtime marker, shared across concurrent sessions on the same * repo; add per-session dedup only if that sharing becomes a problem. */ -function shouldEmitMcpHint(gitNexusDir) { +function shouldEmitMcpHint(storagePath) { const raw = process.env.GITNEXUS_MCP_HINT_THROTTLE_MS; const windowMs = raw === undefined || raw === '' ? 600000 : Number(raw); if (!Number.isFinite(windowMs) || windowMs <= 0) return true; - const marker = path.join(gitNexusDir, '.mcp-hint-shown'); + const marker = path.join(storagePath, '.mcp-hint-shown'); try { if (Date.now() - fs.statSync(marker).mtimeMs < windowMs) return false; } catch { @@ -434,14 +371,14 @@ function shouldEmitMcpHint(gitNexusDir) { return true; } -function runAugment(gitNexusDir, cwd, pattern) { +function runAugment(storagePath, lbugPath, cwd, pattern) { // Acquire the per-repo slot BEFORE the DB-owner probe (#2163): the probe // itself spawns lsof/ps, so it must be bounded by the same ≤3-per-repo cap // as the augment, or concurrent sessions fan out unbounded probe - // subprocesses. The cheap guards (extractPattern, gitNexusDir lookup) run in + // subprocesses. The cheap guards (extractPattern, registry lookup) run in // buildAfterToolContext before this — moving the acquire any earlier would // churn slot files on tool calls that never probe. - const release = acquireHookSlot(gitNexusDir); + const release = acquireHookSlot(storagePath); if (!release) { // Normal skip path: all per-repo hook slots are held by concurrent // sessions. Stay silent for strict hook runners (issue #1913); surface @@ -452,7 +389,7 @@ function runAugment(gitNexusDir, cwd, pattern) { return ''; } try { - if (hasGitNexusServerOwner(gitNexusDir)) { + if (hasGitNexusServerOwner(lbugPath)) { // #2396: the MCP server holds the DB write lock, so a competing CLI // `augment` would only contend on it (LadybugDB is single-writer). The // session has the GitNexus MCP tools live — route the augmentation to the @@ -462,7 +399,7 @@ function runAugment(gitNexusDir, cwd, pattern) { if (isDebugEnabled()) { process.stderr.write('[GitNexus] augment skipped: MCP server owns DB\n'); } - return shouldEmitMcpHint(gitNexusDir) ? buildMcpQueryHint(pattern) : ''; + return shouldEmitMcpHint(storagePath) ? buildMcpQueryHint(pattern) : ''; } const cliPath = resolveCliPath(); const child = runGitNexusCli(cliPath, ['augment', '--', pattern], cwd, 7000); @@ -477,7 +414,7 @@ function runAugment(gitNexusDir, cwd, pattern) { return ''; } -function buildStaleIndexHint(gitNexusDir, cwd) { +function buildStaleIndexHint(meta, cwd) { let currentHead = ''; try { const headResult = spawnSync('git', ['rev-parse', 'HEAD'], { @@ -495,7 +432,6 @@ function buildStaleIndexHint(gitNexusDir, cwd) { let lastCommit = ''; let hadEmbeddings = false; - const meta = readIndexMeta(gitNexusDir); if (meta) { lastCommit = meta.lastCommit || ''; hadEmbeddings = meta.stats && meta.stats.embeddings > 0; diff --git a/gitnexus/hooks/claude/gitnexus-hook.cjs b/gitnexus/hooks/claude/gitnexus-hook.cjs index 1b75ed17d..e9bf44421 100755 --- a/gitnexus/hooks/claude/gitnexus-hook.cjs +++ b/gitnexus/hooks/claude/gitnexus-hook.cjs @@ -20,6 +20,7 @@ const { resolveUnixGuardTimeout, } = require('./hook-db-lock-probe.cjs'); const { formatAnalyzeCommand } = require('./resolve-analyze-cmd.cjs'); +const { resolveHookRepo } = require('./registry-query.cjs'); /** * Read JSON input from stdin synchronously. @@ -33,106 +34,8 @@ function readInput() { } } -/** - * Find the .gitnexus directory by walking up from startDir. - * Returns the path to .gitnexus/ or null if not found. - */ -function isGlobalRegistryDir(candidate) { - if ( - fs.existsSync(path.join(candidate, 'gitnexus.json')) || - fs.existsSync(path.join(candidate, 'meta.json')) - ) { - return false; - } - return ( - fs.existsSync(path.join(candidate, 'registry.json')) || - fs.existsSync(path.join(candidate, 'repos')) - ); -} - -/** - * Read the index metadata file, preferring `gitnexus.json` (current format) - * and falling back to the legacy `meta.json` mirror. Returns `null` if - * neither exists or parses. - */ -function readIndexMeta(gitNexusDir) { - try { - return JSON.parse(fs.readFileSync(path.join(gitNexusDir, 'gitnexus.json'), 'utf-8')); - } catch { - try { - return JSON.parse(fs.readFileSync(path.join(gitNexusDir, 'meta.json'), 'utf-8')); - } catch { - return null; - } - } -} - -/** - * Walk up from `startDir` looking for a non-registry `.gitnexus/` folder. - * Returns the path to `.gitnexus/` or null if not found within 5 levels. - */ -function walkForGitNexusDir(startDir) { - let dir = startDir; - for (let i = 0; i < 5; i++) { - const candidate = path.join(dir, '.gitnexus'); - if (fs.existsSync(candidate)) { - if (!isGlobalRegistryDir(candidate)) return candidate; - } - const parent = path.dirname(dir); - if (parent === dir) break; - dir = parent; - } - return null; -} - -/** - * Resolve the canonical (main) worktree root for `cwd`, when `cwd` is inside - * any git working tree — including a *linked* worktree created via - * `git worktree add`. Linked worktrees never contain `.gitnexus/`, so the - * upward walk from cwd alone misses the index. Returns null when `cwd` is - * not inside a git repo or `git` is not available. - * - * Implementation: `git rev-parse --git-common-dir` resolves to the canonical - * `.git/` directory (or `.git/worktrees/...` parent) that is shared across - * all linked worktrees. The canonical repo root is its parent directory. - */ -function findCanonicalRepoRoot(cwd) { - try { - const result = spawnSync('git', ['rev-parse', '--path-format=absolute', '--git-common-dir'], { - encoding: 'utf-8', - timeout: 2000, - cwd, - stdio: ['pipe', 'pipe', 'pipe'], - windowsHide: true, - }); - if (result.error || result.status !== 0) return null; - const commonDir = (result.stdout || '').trim(); - if (!commonDir || !path.isAbsolute(commonDir)) return null; - return path.dirname(commonDir); - } catch { - return null; - } -} - -function findGitNexusDir(startDir) { - const cwd = startDir || process.cwd(); - - // Fast path: the cwd is inside the canonical repo (most common case). - const fromCwd = walkForGitNexusDir(cwd); - if (fromCwd) return fromCwd; - - // Fallback: cwd may be inside a linked git worktree whose `.gitnexus/` - // only lives in the canonical repo root. Resolve the shared git dir - // and retry from there. - const canonicalRoot = findCanonicalRepoRoot(cwd); - if (canonicalRoot && canonicalRoot !== cwd) { - return walkForGitNexusDir(canonicalRoot); - } - return null; -} - -function hasGitNexusServerOwner(gitNexusDir) { - return hasGitNexusDbLockedByGitNexusServer(path.join(gitNexusDir, 'lbug'), process.pid); +function hasGitNexusServerOwner(lbugPath) { + return hasGitNexusDbLockedByGitNexusServer(lbugPath, process.pid); } /** @@ -374,11 +277,11 @@ function buildMcpQueryHint(pattern) { * ponytail: per-repo mtime marker, shared across concurrent sessions on the same * repo; add per-session dedup only if that sharing becomes a problem. */ -function shouldEmitMcpHint(gitNexusDir) { +function shouldEmitMcpHint(storagePath) { const raw = process.env.GITNEXUS_MCP_HINT_THROTTLE_MS; const windowMs = raw === undefined || raw === '' ? 600000 : Number(raw); if (!Number.isFinite(windowMs) || windowMs <= 0) return true; - const marker = path.join(gitNexusDir, '.mcp-hint-shown'); + const marker = path.join(storagePath, '.mcp-hint-shown'); try { if (Date.now() - fs.statSync(marker).mtimeMs < windowMs) return false; } catch { @@ -398,8 +301,6 @@ function shouldEmitMcpHint(gitNexusDir) { function handlePreToolUse(input) { const cwd = input.cwd || process.cwd(); if (!path.isAbsolute(cwd)) return; - const gitNexusDir = findGitNexusDir(cwd); - if (!gitNexusDir) return; const toolName = input.tool_name || ''; const toolInput = input.tool_input || {}; @@ -409,12 +310,18 @@ function handlePreToolUse(input) { const pattern = extractPattern(toolName, toolInput); if (!pattern || pattern.length < 3) return; + // Registry row first (persisted external storagePath wins). Local owned + // `.gitnexus` is only the fallback when no matching registry row exists. + const repo = resolveHookRepo(cwd); + if (!repo) return; + const storagePath = repo.storagePath; + // Acquire the per-repo slot BEFORE the DB-owner probe (#2163): the probe // itself spawns lsof/ps, so it must be bounded by the same ≤3-per-repo cap // as the augment, or concurrent sessions fan out unbounded probe // subprocesses. Keep the acquire right after the cheap guards above — // moving it earlier would churn slot files on tool calls that never probe. - const release = acquireHookSlot(gitNexusDir); + const release = acquireHookSlot(storagePath); if (!release) { // Normal skip path: all per-repo hook slots are held by concurrent // sessions. Stay silent for strict hook runners (issue #1913); surface @@ -427,7 +334,7 @@ function handlePreToolUse(input) { let result = ''; try { - if (hasGitNexusServerOwner(gitNexusDir)) { + if (hasGitNexusServerOwner(repo.lbugPath)) { // #2396: the MCP server holds the DB write lock, so a competing CLI // `augment` would only contend on it (LadybugDB is single-writer). But the // session that triggered this hook has the GitNexus MCP tools live — route @@ -438,7 +345,7 @@ function handlePreToolUse(input) { if (isDebugEnabled()) { process.stderr.write('[GitNexus] augment skipped: MCP server owns DB\n'); } - if (shouldEmitMcpHint(gitNexusDir)) { + if (shouldEmitMcpHint(storagePath)) { result = buildMcpQueryHint(pattern); } } else { @@ -476,7 +383,7 @@ function sendHookResponse(hookEventName, message) { * Instead of spawning a full `gitnexus analyze` synchronously (which blocks * the agent for up to 120s and risks KuzuDB corruption on timeout), we do a * lightweight staleness check: compare `git rev-parse HEAD` against the - * lastCommit stored in `.gitnexus/meta.json`. If they differ, notify the + * lastCommit stored in the registered index metadata. If they differ, notify the * agent so it can decide when to reindex. */ function handlePostToolUse(input) { @@ -492,8 +399,8 @@ function handlePostToolUse(input) { const cwd = input.cwd || process.cwd(); if (!path.isAbsolute(cwd)) return; - const gitNexusDir = findGitNexusDir(cwd); - if (!gitNexusDir) return; + const repo = resolveHookRepo(cwd); + if (!repo) return; // Compare HEAD against last indexed commit — skip if unchanged let currentHead = ''; @@ -514,7 +421,7 @@ function handlePostToolUse(input) { let lastCommit = ''; let hadEmbeddings = false; - const meta = readIndexMeta(gitNexusDir); + const meta = repo.metadata; if (meta) { lastCommit = meta.lastCommit || ''; hadEmbeddings = meta.stats && meta.stats.embeddings > 0; diff --git a/gitnexus/hooks/claude/registry-query.cjs b/gitnexus/hooks/claude/registry-query.cjs new file mode 100644 index 000000000..649b363fe --- /dev/null +++ b/gitnexus/hooks/claude/registry-query.cjs @@ -0,0 +1,410 @@ +const fs = require('fs'); +const os = require('os'); +const path = require('path'); +const { createHash } = require('crypto'); +const { spawnSync } = require('child_process'); + +// Hooks are copied into editor-specific directories and run without the +// package's TypeScript modules. Keep their on-disk names centralized here. +const GITNEXUS_DIR = '.gitnexus'; +const INDEX_METADATA_FILE = 'gitnexus.json'; +const LEGACY_METADATA_FILE = 'meta.json'; +const LBUG_DIRECTORY = 'lbug'; +const BRANCHES_DIRECTORY = 'branches'; +const STORAGE_PATH_ENV = 'GITNEXUS_STORAGE_PATH'; +const STORAGE_ROOT_ENV = 'GITNEXUS_STORAGE_ROOT'; +const STORAGE_SLOT_HASH_LENGTH = 12; +const LOCAL_OWNED_PARENT_HOPS = 5; + +function stripWindowsLongPathPrefix(p) { + if (process.platform !== 'win32') return p; + if (/^\\\\\?\\UNC\\(?=[^\\])/i.test(p)) return `\\\\${p.slice(8)}`; + if (/^\\\\\?\\[A-Za-z]:\\/.test(p)) return p.slice(4); + return p; +} + +function canonicalize(value) { + if (typeof value !== 'string' || !value || value.includes('\0') || !path.isAbsolute(value)) + return null; + const resolved = path.resolve(value); + try { + return stripWindowsLongPathPrefix(fs.realpathSync.native(resolved)); + } catch { + return stripWindowsLongPathPrefix(resolved); + } +} + +function samePath(left, right) { + if (left == null || right == null) return false; + return process.platform === 'win32' ? left.toLowerCase() === right.toLowerCase() : left === right; +} + +function isMissingFile(error) { + return error && (error.code === 'ENOENT' || error.code === 'ENOTDIR'); +} + +function readMetadataFile(storagePath, filename) { + try { + const value = JSON.parse(fs.readFileSync(path.join(storagePath, filename), 'utf-8')); + return value && typeof value === 'object' && !Array.isArray(value) + ? { state: 'valid', value } + : { state: 'invalid' }; + } catch (error) { + return isMissingFile(error) ? { state: 'absent' } : { state: 'invalid' }; + } +} + +function readIndexMetadata(storagePath) { + const primary = readMetadataFile(storagePath, INDEX_METADATA_FILE); + if (primary.state === 'valid') return primary.value; + if (primary.state !== 'absent') return null; + + const legacy = readMetadataFile(storagePath, LEGACY_METADATA_FILE); + return legacy.state === 'valid' ? legacy.value : null; +} + +function isOwnedStorage(repoPath, storagePath, repositoryLocal, metadata) { + // Repository-local storage remains usable for metadata written before + // repoPath was recorded, but an explicit repoPath must never name another + // checkout. External storage always requires the complete ownership binding. + if (repositoryLocal && (!metadata || typeof metadata.repoPath !== 'string')) { + return true; + } + if (!metadata || typeof metadata.repoPath !== 'string') return false; + + const metadataRepoPath = canonicalize(metadata.repoPath); + const expectedRepoPath = canonicalize(repoPath); + if ( + metadataRepoPath == null || + expectedRepoPath == null || + !samePath(metadataRepoPath, expectedRepoPath) + ) { + return false; + } + if (repositoryLocal) return true; + if (typeof metadata.storagePath !== 'string') return false; + + const metadataStoragePath = canonicalize(metadata.storagePath); + const expectedStoragePath = canonicalize(storagePath); + return ( + metadataStoragePath != null && + expectedStoragePath != null && + samePath(metadataStoragePath, expectedStoragePath) + ); +} + +function ancestorPaths(cwd) { + const paths = []; + let current = canonicalize(cwd); + while (current) { + paths.push(current); + const parent = path.dirname(current); + if (parent === current) break; + current = parent; + } + return paths; +} + +function isInsideOrEqual(child, ancestor) { + if (child == null || ancestor == null) return false; + if (samePath(child, ancestor)) return true; + const relative = path.relative(ancestor, child); + return ( + relative !== '' && + relative !== '..' && + !relative.startsWith(`..${path.sep}`) && + !path.isAbsolute(relative) + ); +} + +function ancestorPathsThrough(cwd, stopAt) { + const paths = []; + let current = canonicalize(cwd); + const stop = canonicalize(stopAt); + while (current) { + if (stop && !isInsideOrEqual(current, stop)) break; + paths.push(current); + if (stop && samePath(current, stop)) break; + const parent = path.dirname(current); + if (parent === current) break; + current = parent; + } + return paths; +} + +function currentGitBranch(cwd) { + try { + const result = spawnSync('git', ['symbolic-ref', '--quiet', '--short', 'HEAD'], { + encoding: 'utf-8', + timeout: 2000, + cwd, + stdio: ['pipe', 'pipe', 'pipe'], + windowsHide: true, + }); + if (result.error || result.status !== 0) return null; + const branch = String(result.stdout || '').trim(); + return branch || null; + } catch { + return null; + } +} + +function registryPathsForCwd(cwd) { + const fallbackPaths = ancestorPaths(cwd); + if (fallbackPaths.length === 0) return { repoPaths: [], branch: null }; + try { + const result = spawnSync( + 'git', + ['rev-parse', '--path-format=absolute', '--show-toplevel', '--git-common-dir'], + { + encoding: 'utf-8', + timeout: 2000, + cwd, + stdio: ['pipe', 'pipe', 'pipe'], + windowsHide: true, + }, + ); + if (result.error || result.status !== 0) return { repoPaths: fallbackPaths, branch: null }; + + const [worktreeRoot, commonDir] = String(result.stdout || '') + .split(/\r?\n/) + .map((line) => line.trim()) + .filter(Boolean); + if (!worktreeRoot || !path.isAbsolute(worktreeRoot)) { + return { repoPaths: fallbackPaths, branch: null }; + } + + // Keep ancestor paths of cwd that stay inside this worktree (cwd up to + // and including show-toplevel) so a --skip-git subdirectory index can + // win via longest-match. Do not walk ancestors outside the worktree — + // that would re-attribute a parent index to a nested git checkout. + const repoPaths = ancestorPathsThrough(cwd, worktreeRoot); + const worktreeCanon = canonicalize(worktreeRoot); + if (worktreeCanon && !repoPaths.some((repoPath) => samePath(repoPath, worktreeCanon))) { + repoPaths.push(worktreeCanon); + } + + // Linked worktrees share the canonical repo's git dir. Include that + // parent so the registered main checkout is still discoverable, but do + // not walk any further outside this worktree. + if (commonDir) { + const commonParent = canonicalize(path.dirname(commonDir)); + if ( + commonParent && + worktreeCanon && + !samePath(commonParent, worktreeCanon) && + !repoPaths.some((repoPath) => samePath(repoPath, commonParent)) + ) { + repoPaths.push(commonParent); + } + } + return { + repoPaths, + branch: currentGitBranch(cwd), + }; + } catch { + return { repoPaths: fallbackPaths, branch: null }; + } +} + +function branchSlug(rawRef) { + const sanitized = rawRef.replace(/^-+/, '').replace(/[^a-zA-Z0-9._-]/g, '_'); + const reserved = /^(CON|PRN|AUX|NUL|COM[1-9]|LPT[1-9])(\..*)?$/i; + const safe = + !sanitized || sanitized === '.' || sanitized === '..' || reserved.test(sanitized) + ? 'unknown' + : sanitized; + const hash = createHash('sha256').update(rawRef).digest('hex').slice(0, 8); + return `${safe}-${hash}`; +} + +// Mirror gitnexus/src/storage/storage-resolver.ts storageSlotName exactly +// (sanitize + sha256 of the canonical repo path, 12-hex suffix). +function sanitizeSlotBasename(value) { + // Cap first, then walk the tail once — same order as + // gitnexus/src/storage/storage-resolver.ts (avoids /[. ]+$/ ReDoS). + const sanitized = value.replace(/[\u0000-\u001f<>:"/\\|?*]/g, '-').slice(0, 80); + let end = sanitized.length; + while (end > 0) { + const code = sanitized.charCodeAt(end - 1); + if (code !== 0x20 && code !== 0x2e) break; + end--; + } + const candidate = sanitized.slice(0, end) || 'repository'; + return /^(con|prn|aux|nul|com[1-9]|lpt[1-9])$/i.test(candidate) + ? `repository-${candidate}` + : candidate; +} + +function storageSlotName(repoPath) { + const canonical = canonicalize(repoPath); + if (!canonical) return null; + const identity = process.platform === 'win32' ? canonical.toLowerCase() : canonical; + const basename = sanitizeSlotBasename(path.basename(canonical)); + const digest = createHash('sha256') + .update(identity) + .digest('hex') + .slice(0, STORAGE_SLOT_HASH_LENGTH); + return `${basename}-${digest}`; +} + +function envOverridesStorage() { + const envPath = process.env[STORAGE_PATH_ENV]; + const envRoot = process.env[STORAGE_ROOT_ENV]; + return ( + (typeof envPath === 'string' && envPath.length > 0) || + (typeof envRoot === 'string' && envRoot.length > 0) + ); +} + +function resolveEntryStoragePath(entry) { + const envPath = process.env[STORAGE_PATH_ENV]; + if ( + typeof envPath === 'string' && + envPath.length > 0 && + !envPath.includes('\0') && + path.isAbsolute(envPath) + ) { + const resolved = path.resolve(envPath); + if (path.isAbsolute(resolved)) return resolved; + } + + const envRoot = process.env[STORAGE_ROOT_ENV]; + if ( + typeof envRoot === 'string' && + envRoot.length > 0 && + !envRoot.includes('\0') && + path.isAbsolute(envRoot) + ) { + const root = path.resolve(envRoot); + const slot = storageSlotName(entry.path); + if (slot) { + const storagePath = path.join(root, slot); + if (samePath(path.dirname(storagePath), root)) return storagePath; + } + } + + if (entry.storagePath !== undefined) { + if ( + typeof entry.storagePath !== 'string' || + !entry.storagePath || + entry.storagePath.includes('\0') || + !path.isAbsolute(entry.storagePath) + ) { + return null; + } + return path.resolve(entry.storagePath); + } + return path.resolve(path.join(entry.path, GITNEXUS_DIR)); +} + +function hasLocalIndexSignal(storagePath) { + try { + return ( + fs.existsSync(path.join(storagePath, INDEX_METADATA_FILE)) || + fs.existsSync(path.join(storagePath, LBUG_DIRECTORY)) + ); + } catch { + return false; + } +} + +function findLocalOwnedRepo(cwd) { + // Environment storage overrides win; a leftover repo-local .gitnexus must + // not skip the registry scan that applies STORAGE_PATH / STORAGE_ROOT. + if (envOverridesStorage()) return null; + const { repoPaths, branch } = registryPathsForCwd(cwd); + let current = canonicalize(cwd); + for (let hops = 0; hops <= LOCAL_OWNED_PARENT_HOPS && current; hops++) { + const storagePath = path.join(current, GITNEXUS_DIR); + if (hasLocalIndexSignal(storagePath)) { + const metadata = readIndexMetadata(storagePath); + if (isOwnedStorage(current, storagePath, true, metadata)) { + const branchDir = + branch != null ? path.join(storagePath, BRANCHES_DIRECTORY, branchSlug(branch)) : null; + const indexDir = branchDir && hasLocalIndexSignal(branchDir) ? branchDir : storagePath; + return { + path: current, + storagePath, + lbugPath: path.join(indexDir, LBUG_DIRECTORY), + metadata: indexDir === storagePath ? metadata : readIndexMetadata(indexDir), + }; + } + } + const parent = path.dirname(current); + if (parent === current) break; + // Stay inside this checkout. Registered lookup already stops at + // `--show-toplevel`; walking raw parents would adopt `/outer/.gitnexus` + // from `/outer/nested-repo`. + if (repoPaths.length > 0 && !repoPaths.some((repoPath) => samePath(repoPath, parent))) { + break; + } + current = parent; + } + return null; +} + +function findRegisteredRepo(cwd) { + const { repoPaths, branch } = registryPathsForCwd(cwd); + if (repoPaths.length === 0) return null; + + const home = process.env.GITNEXUS_HOME || path.join(os.homedir(), '.gitnexus'); + let entries; + try { + entries = JSON.parse(fs.readFileSync(path.join(home, 'registry.json'), 'utf-8')); + } catch { + return null; + } + if (!Array.isArray(entries)) return null; + + let best = null; + let bestLen = -1; + for (const entry of entries) { + if (!entry || typeof entry !== 'object' || Array.isArray(entry)) continue; + if (typeof entry.path !== 'string') continue; + if (entry.path.includes('\0') || !path.isAbsolute(entry.path)) continue; + const registeredPath = canonicalize(entry.path); + if (!registeredPath || !repoPaths.some((repoPath) => samePath(repoPath, registeredPath))) { + continue; + } + const storagePath = resolveEntryStoragePath(entry); + if (!storagePath) continue; + const repositoryLocal = samePath( + canonicalize(path.join(entry.path, GITNEXUS_DIR)), + canonicalize(storagePath), + ); + const ownershipMetadata = readIndexMetadata(storagePath); + if (!isOwnedStorage(entry.path, storagePath, repositoryLocal, ownershipMetadata)) continue; + const branchIsIndexed = + branch && + Array.isArray(entry.branches) && + entry.branches.some((summary) => summary && summary.branch === branch); + const indexDir = branchIsIndexed + ? path.join(storagePath, BRANCHES_DIRECTORY, branchSlug(branch)) + : storagePath; + if (registeredPath.length > bestLen) { + bestLen = registeredPath.length; + best = { + path: entry.path, + storagePath, + lbugPath: path.join(indexDir, LBUG_DIRECTORY), + metadata: branchIsIndexed ? readIndexMetadata(indexDir) : ownershipMetadata, + }; + } + } + return best; +} + +/** Registry row wins (including persisted external storagePath); local owned is fallback. */ +function resolveHookRepo(cwd) { + return findRegisteredRepo(cwd) || findLocalOwnedRepo(cwd); +} + +module.exports = { + findRegisteredRepo, + findLocalOwnedRepo, + resolveHookRepo, + INDEX_METADATA_FILE, + LEGACY_METADATA_FILE, + LBUG_DIRECTORY, +}; diff --git a/gitnexus/package-lock.json b/gitnexus/package-lock.json index 6e521f664..30d17220c 100644 --- a/gitnexus/package-lock.json +++ b/gitnexus/package-lock.json @@ -1,16 +1,16 @@ { "name": "gitnexus", - "version": "1.6.11", + "version": "1.6.12", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "gitnexus", - "version": "1.6.11", + "version": "1.6.12", "hasInstallScript": true, "license": "PolyForm-Noncommercial-1.0.0", "dependencies": { - "@ladybugdb/core": "^0.19.0", + "@ladybugdb/core": "0.18.3", "@modelcontextprotocol/sdk": "^1.0.0", "@scarf/scarf": "^1.4.0", "busboy": "^1.6.0", @@ -1269,9 +1269,9 @@ } }, "node_modules/@ladybugdb/core": { - "version": "0.19.1", - "resolved": "https://registry.npmjs.org/@ladybugdb/core/-/core-0.19.1.tgz", - "integrity": "sha512-8W2g6xUi4jm96fs4EayyMcsvEEtIb8vboZhw9/YG98881cIcmZjmqAN91XGUp4vb8NqoFr3Wp7wcu3dqJk0b7w==", + "version": "0.18.3", + "resolved": "https://registry.npmjs.org/@ladybugdb/core/-/core-0.18.3.tgz", + "integrity": "sha512-XjpPKW4MrL28D2gYGTZuIjiEcPx12L21lx58QggrdrItw8o/e9Lmg/Ejoo4Kz08lZj+rIcC1Fu9thzIYOTUlJw==", "hasInstallScript": true, "license": "MIT", "dependencies": { @@ -1280,17 +1280,17 @@ "node-addon-api": "^6.0.0" }, "optionalDependencies": { - "@ladybugdb/core-darwin-arm64": "0.19.1", - "@ladybugdb/core-darwin-x64": "0.19.1", - "@ladybugdb/core-linux-arm64": "0.19.1", - "@ladybugdb/core-linux-x64": "0.19.1", - "@ladybugdb/core-win32-x64": "0.19.1" + "@ladybugdb/core-darwin-arm64": "0.18.3", + "@ladybugdb/core-darwin-x64": "0.18.3", + "@ladybugdb/core-linux-arm64": "0.18.3", + "@ladybugdb/core-linux-x64": "0.18.3", + "@ladybugdb/core-win32-x64": "0.18.3" } }, "node_modules/@ladybugdb/core-darwin-arm64": { - "version": "0.19.1", - "resolved": "https://registry.npmjs.org/@ladybugdb/core-darwin-arm64/-/core-darwin-arm64-0.19.1.tgz", - "integrity": "sha512-VGQs1NThAygMsoOlxud05pqKA9xfUptl55iYkwvW45As5MSI7+M86WN0Pp0VdPEfw8vNQJehrlHR5LvVAuWc2Q==", + "version": "0.18.3", + "resolved": "https://registry.npmjs.org/@ladybugdb/core-darwin-arm64/-/core-darwin-arm64-0.18.3.tgz", + "integrity": "sha512-DGZTOlvSS4esEb1vTekY5IDoAvZAeYzR5cXVkECtQj9BVkk05zsvCAdTPo1Rz1BuI0qvqUVF+2WlIerI67iA2g==", "cpu": [ "arm64" ], @@ -1301,9 +1301,9 @@ ] }, "node_modules/@ladybugdb/core-darwin-x64": { - "version": "0.19.1", - "resolved": "https://registry.npmjs.org/@ladybugdb/core-darwin-x64/-/core-darwin-x64-0.19.1.tgz", - "integrity": "sha512-CGfM6ostxDS5jztxwjkXXtxrjMDgsFMRoyr5HZDCFw1+iXC1rIzmK/Y7RIw+KbQ49aPzSmkhBC447mFviBJxoA==", + "version": "0.18.3", + "resolved": "https://registry.npmjs.org/@ladybugdb/core-darwin-x64/-/core-darwin-x64-0.18.3.tgz", + "integrity": "sha512-Qp6j0CM/orBlK6KD0p/s4ofkIhNUwi1hdCgMw+fj81UHugWHkVLiYV4grRBdHhyplw+snchZpTxvfpxFbkG1Cw==", "cpu": [ "x64" ], @@ -1314,9 +1314,9 @@ ] }, "node_modules/@ladybugdb/core-linux-arm64": { - "version": "0.19.1", - "resolved": "https://registry.npmjs.org/@ladybugdb/core-linux-arm64/-/core-linux-arm64-0.19.1.tgz", - "integrity": "sha512-BZUQwlkvNXENc5GVyXdfRF0Dv9JX8XMlcdMMiB5GKrEhTCpajQ3D58woHPVvn0JEjw7Ms3tHo6kXUAMZKYXIVg==", + "version": "0.18.3", + "resolved": "https://registry.npmjs.org/@ladybugdb/core-linux-arm64/-/core-linux-arm64-0.18.3.tgz", + "integrity": "sha512-F9miYjBuS43I7uNG199FNMqwdHJ98WA6dU3v2SZCeLXmXCdRzmYcuHQWlbNr2Tba9CX58w2XvBZoUaXZKJ/yKQ==", "cpu": [ "arm64" ], @@ -1327,9 +1327,9 @@ ] }, "node_modules/@ladybugdb/core-linux-x64": { - "version": "0.19.1", - "resolved": "https://registry.npmjs.org/@ladybugdb/core-linux-x64/-/core-linux-x64-0.19.1.tgz", - "integrity": "sha512-LDx+E1UHlmNXSb3F9QmvdBgZGfB3wI/DcrHzfOwXgT3BP8C4ScB2tZdpiYQiuPp8MiSZ9kuuGqos8A4tQKQu8Q==", + "version": "0.18.3", + "resolved": "https://registry.npmjs.org/@ladybugdb/core-linux-x64/-/core-linux-x64-0.18.3.tgz", + "integrity": "sha512-AfG5RDp/f/IDctDMpTAT5+2MYNtlWT191xiQNjSaWD4X85DhY3Dzps8Qu5VteIAPih5d6mmoaKGs8q0XIjfkFA==", "cpu": [ "x64" ], @@ -1340,9 +1340,9 @@ ] }, "node_modules/@ladybugdb/core-win32-x64": { - "version": "0.19.1", - "resolved": "https://registry.npmjs.org/@ladybugdb/core-win32-x64/-/core-win32-x64-0.19.1.tgz", - "integrity": "sha512-2spst1g+Z050Fz/5z7Pc6Fuc5dVXLzekOuWW4lP+mEGCL3tkv3QWYxk37DiFy+O9fDFxiWK2f3aauab58/f9kQ==", + "version": "0.18.3", + "resolved": "https://registry.npmjs.org/@ladybugdb/core-win32-x64/-/core-win32-x64-0.18.3.tgz", + "integrity": "sha512-bHuFk0m9cnq0WGd9I4D8or8g6cC/BS58iatMtilqM3JpDPIQIFk6MQl6exL7P4xyWbkLwQgsrv2ToDnyoQNKvg==", "cpu": [ "x64" ], @@ -1888,9 +1888,9 @@ "license": "MIT" }, "node_modules/@types/node": { - "version": "26.4.0", - "resolved": "https://registry.npmjs.org/@types/node/-/node-26.4.0.tgz", - "integrity": "sha512-faiGnoIrLH/V8cibOMEAZ8pMw6oXqSukl29ra4mN8GdaB2ZewzeaLj+INpV5N+Z1eKWzY+IzaIZH2EIR6YZRNQ==", + "version": "26.4.1", + "resolved": "https://registry.npmjs.org/@types/node/-/node-26.4.1.tgz", + "integrity": "sha512-k97ENvZWtvA6yqz5/FS6a7duDgOPEeOQOc2iKS/nY6mX6qJUKtLnWzQS+Xj6tXweyj6ZcTAK2Qecetnvi9nCLA==", "devOptional": true, "license": "MIT", "dependencies": { @@ -3440,9 +3440,9 @@ "license": "MIT" }, "node_modules/hono": { - "version": "4.13.0", - "resolved": "https://registry.npmjs.org/hono/-/hono-4.13.0.tgz", - "integrity": "sha512-jhunvfHWxd7J5EFfSgH4xsYJzSe/lfqbUCxiyyeaQasUsXeEHXtzVid+7EOGByc5JnFa23SSFL3Y2RV/z1T+eQ==", + "version": "4.13.7", + "resolved": "https://registry.npmjs.org/hono/-/hono-4.13.7.tgz", + "integrity": "sha512-c8/gF9ac8Y78/agExVocyLevgR+JlpNB444Py0FSX8pJoPdYUfUzRcXtYEYGwt6l19qIlVZPN5Mfsw9jFShmQQ==", "license": "MIT", "engines": { "node": ">=16.9.0" @@ -3492,9 +3492,9 @@ } }, "node_modules/ignore": { - "version": "7.0.8", - "resolved": "https://registry.npmjs.org/ignore/-/ignore-7.0.8.tgz", - "integrity": "sha512-YYNsSlXBjMk92SKnkwvB5LOVSa6OznlFUGcsvrFgNJbJCd0M1XKeFVRc8ZByeCqz32FivYNHJVooLmdqrmvp/Q==", + "version": "7.0.9", + "resolved": "https://registry.npmjs.org/ignore/-/ignore-7.0.9.tgz", + "integrity": "sha512-brTTsvFRt5C1gGHtPst/281UjPD5t9fBqbgoMPlVWy11ZLTPfu7HxK4ZYqO9H7o/yC9rSTCI85EaQ4OoY12qYw==", "license": "MIT", "engines": { "node": ">= 4" diff --git a/gitnexus/package.json b/gitnexus/package.json index 1d8fc8c62..f79854e9e 100644 --- a/gitnexus/package.json +++ b/gitnexus/package.json @@ -1,6 +1,6 @@ { "name": "gitnexus", - "version": "1.6.11", + "version": "1.6.12", "description": "Graph-powered code intelligence for AI agents. Index any codebase, query via MCP or CLI.", "author": "Abhigyan Patwari", "license": "PolyForm-Noncommercial-1.0.0", @@ -53,11 +53,11 @@ "postinstall": "node scripts/build-tree-sitter-grammars.cjs", "assert-publish-coverage": "node scripts/assert-publish-grammar-coverage.cjs", "prepare": "node scripts/build.js", - "prepack": "node scripts/assert-publish-grammar-coverage.cjs && node scripts/build.js --web && node scripts/assert-web-assets.mjs web", + "prepack": "node scripts/assert-publish-grammar-coverage.cjs && node scripts/assert-publish-fts-coverage.cjs && node scripts/build.js --web && node scripts/assert-web-assets.mjs web", "version": "node scripts/sync-plugin-manifests.mjs" }, "dependencies": { - "@ladybugdb/core": "^0.19.0", + "@ladybugdb/core": "0.18.3", "@modelcontextprotocol/sdk": "^1.0.0", "@scarf/scarf": "^1.4.0", "busboy": "^1.6.0", diff --git a/gitnexus/scripts/assert-publish-fts-coverage.cjs b/gitnexus/scripts/assert-publish-fts-coverage.cjs new file mode 100644 index 000000000..469fc743d --- /dev/null +++ b/gitnexus/scripts/assert-publish-fts-coverage.cjs @@ -0,0 +1,246 @@ +#!/usr/bin/env node +/** + * Publish guard: core↔extension pairing plus vendored FTS artifact integrity. + * + * Does not shell out to `npm pack` (prepack re-entrancy; see the grammar gate). + * Checksums and the `files` allow-list are asserted as pure predicates so a + * future lean-publish narrowing cannot drop the artifacts silently. + */ +const fs = require('fs'); +const path = require('path'); +const crypto = require('crypto'); + +const WIN32_ARM64 = 'win32-arm64'; +const SAFE_FILENAME = /^[\w.-]+\.lbug_extension$/; +const REQUIRED_SUPPORTED_TUPLES = [ + 'linux-x64', + 'linux-arm64', + 'darwin-x64', + 'darwin-arm64', + 'win32-x64', +]; + +/** + * Pure pairing core (exported for tests). Returns human-readable problem + * strings; an empty array means the core↔extension pin is consistent. + */ +const EXACT_CORE_PIN = /^\d+\.\d+\.\d+$/; + +function findPairingProblems({ + installedCoreVersion, + manifestCoreVersion, + manifestExtensionVersion, +}) { + const problems = []; + const installed = String(installedCoreVersion ?? ''); + const pinned = String(manifestCoreVersion ?? ''); + if (!EXACT_CORE_PIN.test(installed) || !EXACT_CORE_PIN.test(pinned)) { + problems.push( + `core pin must be exact x.y.z: installed '${installedCoreVersion ?? ''}' vs manifest '${manifestCoreVersion ?? ''}'`, + ); + return problems; + } + if (installed !== pinned) { + problems.push( + `core pin mismatch: installed ${installed} vs manifest ${pinned}` + + (manifestExtensionVersion ? ` (extension ${manifestExtensionVersion})` : ''), + ); + } + return problems; +} + +function readInstalledCoreVersion(pkg) { + const raw = pkg?.dependencies?.['@ladybugdb/core']; + return raw == null ? '' : String(raw); +} + +function normalizeFilesEntry(value) { + return String(value ?? '') + .replace(/\\/g, '/') + .replace(/\/+$/, '') + .replace(/\/\*\*?$/, ''); +} + +/** True when package.json `files` still ships the FTS prebuild tree. */ +function filesCoverFtsArtifacts(filesField) { + return (filesField || []).some((entry) => { + const n = normalizeFilesEntry(entry); + return ( + n === 'vendor' || + n === 'vendor/lbug-fts' || + n === 'vendor/lbug-fts/prebuilds' || + n === 'vendor/**/prebuilds' + ); + }); +} + +function supportedTuplesFromManifest(manifest) { + return (manifest?.tuples ?? []).map((entry) => entry.tuple); +} + +function unsupportedTuplesFromManifest(manifest) { + return (manifest?.unsupportedTuples ?? []).map((entry) => entry.tuple); +} + +function parseSha256Sums(text) { + const out = {}; + for (const line of String(text ?? '').split(/\r?\n/)) { + const m = /^([a-fA-F0-9]{64})\s+\.\/(\S+)$/.exec(line.trim()); + if (!m) continue; + out[m[2]] = m[1].toLowerCase(); + } + return out; +} + +function sha256File(filePath) { + return crypto.createHash('sha256').update(fs.readFileSync(filePath)).digest('hex'); +} + +/** + * Integrity + coverage predicates. `artifactByTuple` is injected so tests + * never invoke pack and never need the real binaries. + */ +function findArtifactProblems({ + tuples, + unsupportedTuples, + filesField, + checksumByRelPath, + artifactByTuple, + filename, +}) { + const problems = []; + const filenameSafe = filename || 'libfts.lbug_extension'; + if (!SAFE_FILENAME.test(filenameSafe)) { + problems.push(`invalid FTS artifact filename: ${filenameSafe}`); + } + const listed = new Set(tuples || []); + for (const required of REQUIRED_SUPPORTED_TUPLES) { + if (!listed.has(required)) { + problems.push(`manifest.tuples is missing required ${required}`); + } + } + if (!tuples || tuples.length === 0) { + problems.push('manifest.tuples is empty — refusing to publish with 0 artifacts'); + } + if (!filesCoverFtsArtifacts(filesField)) { + problems.push('package.json files no longer covers vendor/lbug-fts/prebuilds'); + } + if (!(unsupportedTuples || []).includes(WIN32_ARM64)) { + problems.push('win32-arm64 must be declared unsupported (no upstream artifact)'); + } + if ((tuples || []).includes(WIN32_ARM64)) { + problems.push('win32-arm64 is listed as supported but has no upstream artifact'); + } + for (const tuple of tuples || []) { + const rel = `${tuple}/${filenameSafe}`; + const artifact = artifactByTuple?.[tuple]; + if (!artifact?.exists) { + problems.push(`missing artifact for ${tuple} (${rel})`); + continue; + } + const expected = checksumByRelPath?.[rel]; + if (!expected) { + problems.push(`missing SHA-256 for ${rel}`); + continue; + } + if (artifact.hash !== expected) { + problems.push( + `checksum mismatch for ${tuple}: expected ${expected} got ${artifact.hash}` + + (artifact.sizeBytes != null ? ` (${artifact.sizeBytes} bytes)` : ''), + ); + } + } + return problems; +} + +function readDiskArtifacts(prebuildsDir, tuples, filename) { + const artifactByTuple = {}; + for (const tuple of tuples) { + const filePath = path.join(prebuildsDir, tuple, filename); + if (!fs.existsSync(filePath)) { + artifactByTuple[tuple] = { exists: false }; + continue; + } + const buf = fs.readFileSync(filePath); + artifactByTuple[tuple] = { + exists: true, + hash: crypto.createHash('sha256').update(buf).digest('hex'), + sizeBytes: buf.byteLength, + }; + } + return artifactByTuple; +} + +function main() { + const gitnexusRoot = path.join(__dirname, '..'); + const pkg = JSON.parse(fs.readFileSync(path.join(gitnexusRoot, 'package.json'), 'utf8')); + const vendorDir = path.join(gitnexusRoot, 'vendor', 'lbug-fts'); + const manifestPath = path.join(vendorDir, 'manifest.json'); + let manifest; + try { + manifest = JSON.parse(fs.readFileSync(manifestPath, 'utf8')); + } catch (err) { + console.error( + `[fts-pairing] Refusing to publish — cannot read ${manifestPath}: ${err.message}`, + ); + process.exit(1); + } + + const installedCoreVersion = readInstalledCoreVersion(pkg); + const pairing = findPairingProblems({ + installedCoreVersion, + manifestCoreVersion: manifest.coreVersion, + manifestExtensionVersion: manifest.extensionVersion, + }); + + const tuples = supportedTuplesFromManifest(manifest); + const unsupportedTuples = unsupportedTuplesFromManifest(manifest); + const filename = manifest.filename || 'libfts.lbug_extension'; + const prebuildsDir = path.join(vendorDir, 'prebuilds'); + let checksumByRelPath = {}; + try { + checksumByRelPath = parseSha256Sums( + fs.readFileSync(path.join(prebuildsDir, 'SHA256SUMS'), 'utf8'), + ); + } catch (err) { + pairing.push(`cannot read SHA256SUMS: ${err.message}`); + } + + const artifacts = findArtifactProblems({ + tuples, + unsupportedTuples, + filesField: pkg.files, + checksumByRelPath, + artifactByTuple: readDiskArtifacts(prebuildsDir, tuples, filename), + filename, + }); + + const problems = [...pairing, ...artifacts]; + if (problems.length > 0) { + console.error('[fts-pairing] Refusing to publish — FTS artifact coverage failed:'); + for (const p of problems) console.error(` - ${p}`); + console.error( + '\nFix: refresh vendor/lbug-fts via .github/scripts/fetch-lbug-fts-artifacts.mjs, ' + + 'or restore the core pin / files allow-list.', + ); + process.exit(1); + } + + console.log( + `[fts-pairing] OK — core ${installedCoreVersion} ↔ extension ${manifest.extensionVersion}; ` + + `${tuples.length} artifacts.`, + ); +} + +if (require.main === module) main(); + +module.exports = { + findPairingProblems, + findArtifactProblems, + filesCoverFtsArtifacts, + parseSha256Sums, + readInstalledCoreVersion, + supportedTuplesFromManifest, + unsupportedTuplesFromManifest, + sha256File, +}; diff --git a/gitnexus/scripts/build-tree-sitter-grammars.cjs b/gitnexus/scripts/build-tree-sitter-grammars.cjs index 434cd193f..a30eb1ec2 100644 --- a/gitnexus/scripts/build-tree-sitter-grammars.cjs +++ b/gitnexus/scripts/build-tree-sitter-grammars.cjs @@ -4,7 +4,7 @@ * One registry-driven script replaces the former per-grammar * build-tree-sitter-.cjs files (they were ~95% identical). * - * The grammars (tree-sitter-c/dart/proto/swift/kotlin/zig) are loaded from + * The grammars (tree-sitter-c/objc/dart/proto/swift/kotlin/zig) are loaded from * `vendor//` by absolute path at runtime (see * src/core/tree-sitter/vendored-grammars.ts) and are NEVER copied into * node_modules — an undeclared package under node_modules is "extraneous" to @@ -25,8 +25,8 @@ * or exit non-zero — a failure for any single grammar must not break the install. * * Opt-out: GITNEXUS_SKIP_OPTIONAL_GRAMMARS=1 (strict '1') skips the OPTIONAL - * grammars only. tree-sitter-c is REQUIRED (it backstops upstream's 4/6 ARM - * prebuild gap, #2116) and is always built. + * grammars only. tree-sitter-c and tree-sitter-objc are REQUIRED (C backstops upstream's 4/6 ARM + * prebuild gap, #2116) and are always built. * * Usage: * node build-tree-sitter-grammars.cjs # all grammars (postinstall) @@ -40,6 +40,7 @@ const { execSync } = require('child_process'); // grammars ignore the opt-out gate. Insertion order == build order (c first). const GRAMMARS = { c: { required: true, display: 'C', ext: '.c' }, + objc: { required: true, display: 'Objective-C', ext: '.m/.mm/.h' }, dart: { required: false, display: 'Dart', ext: '.dart' }, proto: { required: false, display: 'Proto', ext: '.proto' }, swift: { required: false, display: 'Swift', ext: '.swift' }, diff --git a/gitnexus/scripts/cross-platform-shard.ts b/gitnexus/scripts/cross-platform-shard.ts index 2c52dd86b..9e8f2d8e2 100644 --- a/gitnexus/scripts/cross-platform-shard.ts +++ b/gitnexus/scripts/cross-platform-shard.ts @@ -36,9 +36,9 @@ * * Only files heavy enough to matter are listed; everything else is carried by * {@link PER_FILE_OVERHEAD_SEC} alone. These are load-balancing hints, NOT - * assertions — no - * test asserts a runtime, and drift only makes the split slightly less even, so - * a stale entry is harmless and refreshing them is optional. Deliberately not + * assertions — no test measures elapsed time against this table. Missing or + * stale heavy entries can still overload a shard; refresh them from failed + * CI logs and replay that profile in the partition tests. Deliberately not * auto-generated: a committed table is reviewable and works offline, and the * alternative (timing files at CI runtime to decide the split) would make the * partition depend on the very machine load it is trying to protect against. @@ -49,13 +49,20 @@ export const WINDOWS_WEIGHTS_SEC: Readonly> = { 'test/integration/cli-e2e.test.ts': 621, 'test/integration/worker-pool.test.ts': 222, 'test/unit/incremental-vector-extension-ordering.test.ts': 87, - // ESTIMATE, not a measurement (#2841): this suite drives more full - // `runFullAnalysis` cycles than the VECTOR sibling above, so the 8 s - // PER_FILE_OVERHEAD floor would badly under-charge it and skew the Windows - // split — the failure mode that produced the job timeouts this table exists - // to prevent. Scaled from the sibling's measured 87 s by analyze-run count. - // Replace with a real figure after the first green Windows matrix run. - 'test/unit/incremental-index-extension-dml-gate.test.ts': 180, + // Measured on Windows in run 34014266125 (#3190, 2026-09-06). These DB + // suites landed together on shard 3: the old 180s estimate and missing + // entries made a ~27-minute recorded load look like an ~12-minute shard. + // Upstream speedups may reduce these figures; retaining conservative weights + // keeps the expensive suites distributed without changing the watchdog. + 'test/unit/incremental-index-extension-dml-gate.test.ts': 414, + // Re-measured on windows-latest run 34815870795 after vendored-first FTS + // rewrote the HOME-layout e2e (373s) and skills-e2e grew to 542s. The old + // 146s/444s entries packed both onto shard 2/3 and blew the 20-minute + // watchdog with one file still queued. + 'test/integration/skills-e2e.test.ts': 550, + 'test/integration/fts-extension-e2e.test.ts': 380, + 'test/integration/skip-fts.test.ts': 110, + 'test/integration/analyze-wal-checkpoint-failure.test.ts': 86, 'test/integration/cli-limit-e2e.test.ts': 75, 'test/unit/hooks.test.ts': 26, 'test/integration/analyze-heap-oom-e2e.test.ts': 23, diff --git a/gitnexus/scripts/cross-platform-tests.ts b/gitnexus/scripts/cross-platform-tests.ts index 44b60626b..4ca9fd00c 100644 --- a/gitnexus/scripts/cross-platform-tests.ts +++ b/gitnexus/scripts/cross-platform-tests.ts @@ -36,6 +36,18 @@ const PLATFORM_LOGIC = [ // must exercise the Windows backslash branch, so run it on the OS matrix (#2394). 'test/unit/cli-entry.test.ts', 'test/unit/platform-capabilities.test.ts', + // The tsconfig loader rebases `paths` targets through `path.resolve`, so the + // wildcard suffix it must recognise is `/*` on POSIX and `\*` on Windows. It + // only looked for `/*`, and every alias target came back as `src*` on + // Windows while the Ubuntu run stayed green — so this file has to run where + // the separator differs. + 'test/unit/tsconfig-index.test.ts', + // The unit half of the same rebasing rule. Fixture-free and pathApi-injectable + // (every separator assertion passes an explicit `path.win32` / `path.posix`), + // so unlike the fixture suite above it fails on EVERY runner when the + // normalisation is removed rather than only on windows-latest. Registered + // beside its fixture sibling so the two halves stay discoverable as one group. + 'test/unit/tsconfig-rebase-target.test.ts', // The gitnexus-plan safe writer resolves every name through a per-platform // backend: Linux anchors through /proc/self/fd, macOS resolves lexically and // verifies each step against descriptors it holds open. Publication is link(2) @@ -73,6 +85,20 @@ const PLATFORM_LOGIC = [ 'test/unit/lbug-config-pagesize.test.ts', 'test/unit/worker-pool-windows-quarantine.test.ts', 'test/unit/lbug-pool-fts-load.test.ts', + // U7 arm B: Windows FTS names a vendor-neutral OpenSSL/VC++ prerequisite + // and must never recommend borrowing Git for Windows DLLs. The file's + // assertions are unconditional so a skip-only suite cannot stay green. + 'test/integration/fts-windows-dependency.test.ts', + // Remedy-text suites: discoverability only. They pass explicit platform + // strings into pure classifiers and drive mocked rejections with hardcoded + // literals, so they assert identically on every runner. Registering them + // here does not claim Windows-specific behavioral coverage. + 'test/unit/extension-load-error.test.ts', + 'test/unit/fts-degraded-warning.test.ts', + // Vendored-root symlink containment uses realpathSync + path.relative; a + // prefix-only leak follows a Windows junction / POSIX symlink out of + // vendor/. Ubuntu-only would leave that guard unverified on the OS matrix. + 'test/unit/lbug-extension-loader.test.ts', // Global registry writes use the platform-specific index-lock backend // (Windows named pipe, Linux socket, or macOS file lock). This includes the // overlapping-registration regression from #2716 on every OS matrix. @@ -131,6 +157,7 @@ const PLATFORM_LOGIC = [ // file-lock lag after close, macOS N-API destructor segfaults) const LBUG_NATIVE = [ 'test/integration/xaml-search.test.ts', + 'test/integration/skip-fts.test.ts', 'test/integration/lbug-core-adapter.test.ts', 'test/integration/lbug-vector-extension.test.ts', 'test/integration/lbug-pool.test.ts', @@ -196,6 +223,7 @@ const SPAWN_CLI = [ // FTS extension lifecycle — the #2374 bug was Windows-reported, so this must // run on the Windows/macOS matrix, not just the Ubuntu full suite. 'test/integration/fts-extension-e2e.test.ts', + 'test/integration/fts-vendored-root-seam.test.ts', 'test/integration/server-http-startup.test.ts', 'test/integration/mcp/server-startup.test.ts', 'test/integration/analyze-heap-oom-e2e.test.ts', @@ -278,6 +306,11 @@ const NATIVE_ADDON_SMOKE = [ // Filesystem behavior tests — exercise operations that vary across // platforms (CRLF, symlinks, permissions, temp dirs) const FILESYSTEM = [ + // The durable ParsedFile store's prune tolerates a chunk directory it cannot + // delete (#3204). The failures that motivate it — held handles, read-only + // mounts — are Windows- and macOS-flavored, and the permission-based case + // skips itself where chmod cannot block a delete, so run it everywhere. + 'test/unit/parsedfile-store.test.ts', 'test/integration/filesystem-walker.test.ts', 'test/integration/watch-filesystem.test.ts', 'test/integration/markdown-processor-crlf.test.ts', diff --git a/gitnexus/scripts/ensure-fts.ts b/gitnexus/scripts/ensure-fts.ts index 94f781374..f3d6199f2 100644 --- a/gitnexus/scripts/ensure-fts.ts +++ b/gitnexus/scripts/ensure-fts.ts @@ -1,14 +1,11 @@ /** - * Install the LadybugDB FTS and VECTOR extensions into the shared home (~/.lbdb) - * up front, so every test in a sharded CI run finds them regardless of shard. + * Make FTS and VECTOR resolvable for every shard before vitest starts. * - * FTS-dependent tests split two ways: the LOAD-path gate (skipUnlessFtsAvailable) - * self-installs on miss, but the FILE-path gate (requireFtsResourceOrSkip, e.g. - * extension-binary-real.test.ts) resolves the extension path at module load and - * cannot self-install. Sharding (and the balancing sequencer) can drop such a - * test into a shard with no installer sibling — this step removes that ordering - * dependency by installing FTS once before vitest starts. `auto` is LOAD-first, - * so a cache-warmed extension costs no network. + * After vendoring, FTS is ready when the packaged artifact exists — LOAD + * no longer writes `~/.lbdb`, and the FILE-path gates resolve that artifact + * (or a leftover home install). When no artifact is present yet, fall back + * to one bounded `auto` INSTALL so shards without an installer sibling still + * find a file. * * Best-effort: exits 0 on failure (offline etc.) — the per-test gates still * hard-fail under GITNEXUS_REQUIRE_FTS=1 if FTS is genuinely unavailable, which @@ -23,12 +20,17 @@ import { loadVectorExtension, closeLbug, } from '../src/core/lbug/lbug-adapter.js'; +import { resolveVendoredFtsPath } from '../src/core/lbug/vendored-extension-path.js'; const dir = mkdtempSync(join(tmpdir(), 'gn-ensure-fts-')); try { await initLbug(join(dir, 'ensure-fts.lbug')); - const ok = await loadFTSExtension(undefined, { policy: 'auto' }); - console.log(ok ? 'FTS extension ready.' : 'FTS extension unavailable (continuing).'); + if (resolveVendoredFtsPath()) { + console.log('FTS extension ready (vendored artifact).'); + } else { + const ok = await loadFTSExtension(undefined, { policy: 'auto' }); + console.log(ok ? 'FTS extension ready.' : 'FTS extension unavailable (continuing).'); + } // VECTOR rides the same pre-install (#2623): the win32 gate is gone, so the // vector suites genuinely run on Windows/macOS — installing once here means // every sharded test process LOADs from ~/.lbdb instead of racing its own diff --git a/gitnexus/skills/gitnexus-cli.md b/gitnexus/skills/gitnexus-cli.md index 09c7af0d2..3a00ab546 100644 --- a/gitnexus/skills/gitnexus-cli.md +++ b/gitnexus/skills/gitnexus-cli.md @@ -34,6 +34,18 @@ Run from the project root. This parses all source files, builds the knowledge gr For Spring runtime enrichment, pass a JSON bundle, one endpoint JSON file, or a directory containing endpoint files. Route evidence is authoritative only when `runtimeConfirmed === true`; `runtimeSource` records provenance and may also accompany `handler-conflict`. Env/configprops values are never persisted. +## Index storage and retention + +Default location is `/.gitnexus/`. Override with environment variables (also documented in README): + +| Env | Effect | +| --- | ------ | +| `GITNEXUS_STORAGE_PATH` | One complete external index directory. Wins if both storage vars are set. | +| `GITNEXUS_STORAGE_ROOT` | Absolute root; GitNexus creates an isolated `-<12-hex>/` slot per repository. | +| `GITNEXUS_CONTENT_RETENTION` | `full` (default) keeps file text; `symbol` keeps snippets; `none` keeps the graph only. | + +`list_repos`, `gitnexus://repo/{name}/context`, and HTTP `GET /api/repos` / `GET /api/repo` expose `storagePath`, `contentRetention`, and `sourceAvailable`. HTTP `/api/file` and `/api/grep` return 410 unless retention is `full`. MCP `include_content` may still return symbol spans when retention is `symbol`. + Use `node .gitnexus/run.cjs analyze --watch` for a long-lived local Git repository. It performs an initial analysis, queues scanner-admitted file changes, and retries intact failed batches with bounded backoff. Watch refreshes update only the graph: they skip AGENTS.md / CLAUDE.md injection and standard skill installation, so run a one-shot `analyze` when those generated files need updating. Watch rejects one-shot or context-output flags including `--force`, embedding flags, `--skills`, `--default-branch`, `--skip-agents-md`, `--skip-skills`, `--no-stats`, `--self-commit`, `--index-only`, and `--skip-git`. It never pulls remotes. Scheduled remote clone/pull is a different command: `gitnexus auto-sync`. Bare `gitnexus watch` is reserved and does not start either job. Running MCP and `serve` processes periodically check for a published replacement and reopen it without a restart. MCP checks are throttled to once every five seconds, so a tool call before the next check can briefly use the previous index. ### status — Check index freshness diff --git a/gitnexus/src/cli/analyze-config.ts b/gitnexus/src/cli/analyze-config.ts index 3d040fac1..49896538e 100644 --- a/gitnexus/src/cli/analyze-config.ts +++ b/gitnexus/src/cli/analyze-config.ts @@ -31,6 +31,10 @@ import fs from 'node:fs'; import path from 'node:path'; import { readRepoControlFile } from '../config/repo-control-file.js'; +import { + InvalidBranchError, + validateBranchName as validateBranchNameCore, +} from '../core/git-ref.js'; import type { AnalyzeOptions } from './analyze-options.js'; export const GITNEXUS_RC_FILENAME = '.gitnexusrc'; @@ -38,9 +42,6 @@ export const GITNEXUS_RC_FILENAME = '.gitnexusrc'; /** Final fallback when no branch is configured or detectable. */ export const DEFAULT_BRANCH_FALLBACK = 'main'; -/** Git refs longer than this are almost certainly a mistake / injection attempt. */ -const BRANCH_MAX_LENGTH = 255; - /** * Thrown for any `.gitnexusrc` problem (missing-file is NOT an error — it * returns `undefined`). The message is user-facing and names the file so the @@ -157,45 +158,18 @@ const assertNoHiddenChars = (value: string, source: string): void => { /** * Validate a user-supplied branch name (from CLI or `.gitnexusrc`). Returns the - * trimmed name or throws {@link GitNexusRcError}. Conservative but accepts the - * shapes real branches use (`feature/foo-bar`, `release/1.2`, `develop`). + * trimmed name or throws {@link GitNexusRcError}. Rules live in + * `core/git-ref.ts`; this wrapper keeps the CLI / `.gitnexusrc` error type. */ export function validateBranchName(value: string, source: string): string { - const trimmed = value.trim(); - if (!trimmed) { - throw new GitNexusRcError(`${source}: branch name must not be empty.`); + try { + return validateBranchNameCore(value, source); + } catch (err) { + if (err instanceof InvalidBranchError) { + throw new GitNexusRcError(err.message); + } + throw err; } - if (trimmed.length > BRANCH_MAX_LENGTH) { - throw new GitNexusRcError(`${source}: branch name is too long (max ${BRANCH_MAX_LENGTH}).`); - } - assertNoHiddenChars(trimmed, source); - if (/\s/.test(trimmed)) { - throw new GitNexusRcError(`${source}: branch name must not contain whitespace.`); - } - // git ref-name rules (subset): reject characters git itself forbids in refs. - if (/[~^:?*[\\]/.test(trimmed)) { - throw new GitNexusRcError( - `${source}: branch name contains characters not allowed in a git ref (~ ^ : ? * [ \\).`, - ); - } - if (trimmed.startsWith('-')) { - throw new GitNexusRcError(`${source}: branch name must not start with "-".`); - } - if (trimmed.includes('..')) { - throw new GitNexusRcError(`${source}: branch name must not contain "..".`); - } - // Git permits a backtick in a ref, but the branch is embedded inside a - // Markdown inline-code span in the generated AGENTS.md/CLAUDE.md regression - // example, where a backtick would close the span early and let the rest of - // the template render as instruction text. Reject it at this single - // chokepoint so all three tiers (CLI flag, .gitnexusrc, auto-detect via - // sanitizeDetectedBranch) are covered (#1996 tri-review P1). - if (trimmed.includes('`')) { - throw new GitNexusRcError( - `${source}: branch name must not contain a backtick (it would break the generated Markdown).`, - ); - } - return trimmed; } /** diff --git a/gitnexus/src/cli/analyze-options.ts b/gitnexus/src/cli/analyze-options.ts index 36f8925c5..ffa439dab 100644 --- a/gitnexus/src/cli/analyze-options.ts +++ b/gitnexus/src/cli/analyze-options.ts @@ -23,6 +23,7 @@ export interface AnalyzeOptions { /** Commander negated flag: false only when --no-parse-cache is passed. */ parseCache?: boolean; repairFts?: boolean; + skipFts?: boolean; /** * Embedding generation toggle. Commander parses `--embeddings [limit]` as: * - `undefined` when the flag is omitted diff --git a/gitnexus/src/cli/analyze-watch.ts b/gitnexus/src/cli/analyze-watch.ts index 189728963..32a4de2f1 100644 --- a/gitnexus/src/cli/analyze-watch.ts +++ b/gitnexus/src/cli/analyze-watch.ts @@ -9,8 +9,10 @@ import { type AnalyzeOptions as CoreAnalyzeOptions, type AnalyzeResult, } from '../core/run-analyze.js'; +import { isIndexLockGuardTimeout } from '../storage/index-lock.js'; import { getGitRoot, hasGitDir } from '../storage/git.js'; import type { AnalyzerRunnerIdentity } from '../storage/repo-manager.js'; +import { ANALYZE_STORAGE_REQUIREMENTS, requireStoragePath } from '../storage/storage-resolver.js'; import { GITNEXUS_DIR } from '../storage/repo-meta.js'; import { loadAnalyzeConfigStrict, @@ -176,6 +178,7 @@ export async function resolveWatchOptions( return { pdg: merged.pdg, + skipFts: merged.skipFts, branch, registryName: merged.name, allowDuplicateName: merged.allowDuplicateName, @@ -241,6 +244,9 @@ export function shouldStopAfterWatchRefreshFailure( error: unknown, paths: readonly string[], ): boolean { + if (isIndexLockGuardTimeout(error)) { + return true; + } return ( paths.length > 0 && !(error instanceof WatchControlReloadError) && @@ -409,6 +415,13 @@ export async function watchCommandWithRunnerIdentity( return; } const repoPath = await fs.realpath(requestedRepoPath); + try { + await requireStoragePath(repoPath, ANALYZE_STORAGE_REQUIREMENTS); + } catch (error) { + cliError(` ${error instanceof Error ? error.message : String(error)}`); + process.exitCode = 1; + return; + } const baselineEnvironment: WatchEnvironmentBaseline = { maxFileSize: process.env.GITNEXUS_MAX_FILE_SIZE, workerTimeout: process.env.GITNEXUS_WORKER_SUB_BATCH_TIMEOUT_MS, @@ -516,9 +529,15 @@ export async function watchCommandWithRunnerIdentity( const detail = paths.length > 0 ? ` (${paths.length} queued path(s))` : ''; if (shouldStopAfterWatchRefreshFailure(error, paths)) { fatalRefreshError = error; + const guardTimeout = isIndexLockGuardTimeout(error); cliError( `Refresh failed${detail}: ${error instanceof Error ? error.message : String(error)}. ` + - 'Watch mode is stopping because the live index may have been updated in place.', + (guardTimeout + ? 'Watch mode is stopping because the acquisition guard needs quiesced recovery; see RUNBOOK.md.' + : 'Watch mode is stopping because the live index may have been updated in place.'), + guardTimeout + ? { recoveryHint: 'index-lock-guard-recovery', guardPath: error.guardPath } + : undefined, ); stopWatching(); return; diff --git a/gitnexus/src/cli/analyze.ts b/gitnexus/src/cli/analyze.ts index 78d13de39..0f034ee26 100644 --- a/gitnexus/src/cli/analyze.ts +++ b/gitnexus/src/cli/analyze.ts @@ -13,6 +13,7 @@ import os from 'os'; import { spawn } from 'child_process'; import v8 from 'v8'; import cliProgress from 'cli-progress'; +import { formatAnalyzeFtsSkipSummary } from '../core/search/fts-policy.js'; import { isLbugReady, LbugWipeError } from '../core/lbug/lbug-adapter.js'; import { boundedCheckpointBeforeExit } from '../core/lbug/shutdown-helpers.js'; import { findUndeclaredRelationPairError } from '../core/lbug/rel-pair-routing.js'; @@ -28,7 +29,6 @@ import { WAL_RECOVERY_SUGGESTION, } from '../core/lbug/lbug-config.js'; import { - getStoragePaths, getGlobalRegistryPath, RegistryNameCollisionError, AnalysisNotFinalizedError, @@ -42,7 +42,7 @@ import { selfCommitContextFiles, snapshotSelfCommitSafety, } from '../storage/git.js'; -import { IndexLockTimeoutError } from '../storage/index-lock.js'; +import { IndexLockTimeoutError, isIndexLockGuardTimeout } from '../storage/index-lock.js'; import { loadAnalyzeConfig, mergeAnalyzeOptions, @@ -1349,6 +1349,7 @@ const analyzeCommandImpl = async ( force: options.force || options.skills || options.parseCache === false, useParseCache: options.parseCache !== false, repairFts: options.repairFts, + skipFts: options.skipFts, embeddings: embeddingsEnabled, embeddingsNodeLimit, dropEmbeddings: options.dropEmbeddings, @@ -1418,7 +1419,7 @@ const analyzeCommandImpl = async ( // run can write meta.json and then fail before registerRepo(); in // that half-finalized state, runFullAnalysis returns alreadyUpToDate // on the next invocation unless we check the registry here too. - await assertAnalysisFinalized(repoPath); + await assertAnalysisFinalized(repoPath, result.storagePath); // The fast path skips context regeneration, but a changed `.gitnexusrc` // defaultBranch / `--default-branch` must still take effect. Surgically // refresh just the `base_ref` line in AGENTS.md/CLAUDE.md in place, @@ -1450,6 +1451,9 @@ const analyzeCommandImpl = async ( console.error = origError; bar.stop(); console.log(' Already up to date\n'); + if (result.ftsSkipped) { + console.log(` ${formatAnalyzeFtsSkipSummary(result.ftsSkipReason)}\n`); + } if (runOptions.registryName) { console.log(` Registry name: ${result.repoName}\n`); } @@ -1488,7 +1492,7 @@ const analyzeCommandImpl = async ( // success so the silent-finalize state surfaces with a non-zero // exit code and an actionable error instead of being mistaken for // a healthy index. - await assertAnalysisFinalized(repoPath); + await assertAnalysisFinalized(repoPath, result.storagePath); // Skill generation (CLI-only, uses pipeline result from analysis). // Gated so `--index-only --skills` skips community skill writes too @@ -1519,10 +1523,9 @@ const analyzeCommandImpl = async ( (count: number) => count >= 5, ).length; } - const { storagePath: sp } = getStoragePaths(repoPath); await generateAIContextFiles( repoPath, - sp, + result.storagePath, result.repoName, { files: s.files ?? 0, @@ -1608,26 +1611,9 @@ const analyzeCommandImpl = async ( // progress-bar log() that fired mid-run has already scrolled away, so the // degraded-search state must also appear in the final summary (#1161). if (result.ftsSkipped) { - // #2658 review L2: a build/verify failure is NOT an extension-unavailable - // problem — sending the user to install the extension is the wrong remedy. - if (result.ftsSkipReason === 'build-failed') { - console.log( - `\n Warning: full-text/BM25 search is disabled — the search index build failed this run.\n` + - ` The FTS extension is available; rerun \`gitnexus analyze --repair-fts\`. If it persists,\n` + - ` check the disk for space or corruption. Run \`gitnexus doctor\` for details.`, - ); - } else { - console.log( - // NOT "then rerun" (#2841 §5.C): this run stamped `lastCommit`, so a - // plain rerun on an unchanged tree takes the up-to-date fast path and - // returns before Phase 3 could rebuild anything — the advice would be - // ineffective exactly when the user follows it. `--repair-fts` is the - // verb that rebuilds the search indexes without re-parsing the repo. - `\n Warning: full-text/BM25 search is disabled — the LadybugDB FTS extension was unavailable.\n` + - ` Install it once with network access (GITNEXUS_LBUG_EXTENSION_INSTALL=auto), then run\n` + - ` \`gitnexus analyze --repair-fts\` to build the search indexes. Run \`gitnexus doctor\` for details.`, - ); - } + // Total switch (#2658 L2 + native-abort/tuple-missing): a new skip + // reason must not inherit the network-install remedy. + console.log(`\n ${formatAnalyzeFtsSkipSummary(result.ftsSkipReason)}`); } try { @@ -1669,6 +1655,14 @@ const analyzeCommandImpl = async ( // refreshed by the holder — this is a clean, expected condition, not a // crash, so render the message without a stack trace. if (err instanceof IndexLockTimeoutError) { + if (isIndexLockGuardTimeout(err)) { + cliError(err.message, { + recoveryHint: 'index-lock-guard-recovery', + guardPath: err.guardPath, + }); + process.exitCode = 1; + return; + } cliError( ` Another gitnexus analyze (pid ${err.holder.pid} on ${err.holder.hostname}) is ` + `already refreshing this index and did not finish within the wait window.\n` + diff --git a/gitnexus/src/cli/clean.ts b/gitnexus/src/cli/clean.ts index 3f1144d73..f545f8211 100644 --- a/gitnexus/src/cli/clean.ts +++ b/gitnexus/src/cli/clean.ts @@ -12,11 +12,10 @@ import { findRepo, unregisterRepo, listRegisteredRepos, - assertSafeStoragePath, getStoragePaths, removeBranchIndex, - UnsafeStoragePathError, } from '../storage/repo-manager.js'; +import { requireDeletableStoragePath, StorageDeletionError } from '../storage/storage-resolver.js'; import { cleanParkedLbugSidecars, inspectLbugSidecars, @@ -47,14 +46,28 @@ export const cleanCommand = async (options?: { console.log(t('clean.branchNotIndexed', { branch: options.branch })); return; } - const { storagePath, lbugPath } = getStoragePaths(repo.repoPath, summary.branch); + let storagePath: string; + try { + storagePath = await requireDeletableStoragePath({ + path: repo.repoPath, + storagePath: repo.storagePath, + }); + } catch (err) { + if (err instanceof StorageDeletionError) { + logger.error(`Refusing to clean branch index: ${err.message}`); + return; + } + throw err; + } + const { lbugPath } = getStoragePaths(repo.repoPath, summary.branch, storagePath); const branchDir = path.dirname(lbugPath); - // Safety guard: the target MUST live under /.gitnexus/branches/. - // assertSafeStoragePath only validates the flat `/.gitnexus`, so this - // is a dedicated branches-sub-dir check before any destructive fs.rm. + // Safety guard: the target MUST live under the validated + // storage slot's `branches/` directory before any destructive fs.rm. const branchesRoot = path.join(storagePath, 'branches') + path.sep; if (!branchDir.startsWith(branchesRoot)) { - logger.error(`Refusing to clean branch index outside .gitnexus/branches: ${branchDir}`); + logger.error( + `Refusing to clean branch index outside the validated storage slot: ${branchDir}`, + ); return; } if (!options.force) { @@ -81,7 +94,20 @@ export const cleanCommand = async (options?: { return; } - const lbugPath = path.join(repo.storagePath, 'lbug'); + let storagePath: string; + try { + storagePath = await requireDeletableStoragePath({ + path: repo.repoPath, + storagePath: repo.storagePath, + }); + } catch (err) { + if (err instanceof StorageDeletionError) { + logger.error(`Refusing to clean sidecars: ${err.message}`); + return; + } + throw err; + } + const { lbugPath } = getStoragePaths(repo.repoPath, undefined, storagePath); const state = await inspectLbugSidecars(lbugPath); // Single roster authority (this shipping review, FIX 5): the aggregate // covers both parked-sidecar families — the timestamped missing-shadow @@ -121,45 +147,44 @@ export const cleanCommand = async (options?: { // --all flag: clean all indexed repos if (options?.all) { + const entries = await listRegisteredRepos(); if (!options?.force) { - const entries = await listRegisteredRepos(); - if (entries.length === 0) { + const deletableEntries = []; + for (const entry of entries) { + try { + await requireDeletableStoragePath(entry); + deletableEntries.push(entry); + } catch (err) { + if (err instanceof StorageDeletionError) { + logger.error(`Refusing to preview ${entry.name}: ${err.message}`); + continue; + } + throw err; + } + } + if (deletableEntries.length === 0) { console.log(t('common.notIndexed')); return; } - console.log(t('clean.deleteAll', { count: entries.length })); - for (const entry of entries) { + console.log(t('clean.deleteAll', { count: deletableEntries.length })); + for (const entry of deletableEntries) { console.log(` - ${entry.name} (${entry.path})`); } console.log(`\n${t('common.runForceConfirm')}`); return; } - const entries = await listRegisteredRepos(); for (const entry of entries) { - // Safety guard (#1003 review — @magyargergo): same rationale as - // remove.ts. `~/.gitnexus/registry.json` is user-writable, so a - // corrupted or hand-edited entry could point storagePath at the - // repo root, an empty string, or anywhere else — and - // fs.rm(recursive: true) on any of those would be catastrophic. - // Skip poisoned entries without touching disk, but keep going - // through the rest of the registry (preserves the existing - // per-repo error-tolerance semantics of `clean --all`). try { - assertSafeStoragePath(entry); + const storagePath = await requireDeletableStoragePath(entry); + await fs.rm(storagePath, { recursive: true, force: true }); + await unregisterRepo(entry.path); + console.log(t('clean.deletedRepo', { name: entry.name, storagePath })); } catch (err) { - if (err instanceof UnsafeStoragePathError) { + if (err instanceof StorageDeletionError) { logger.error(`Refusing to clean ${entry.name}: ${err.message}`); continue; } - throw err; - } - - try { - await fs.rm(entry.storagePath, { recursive: true, force: true }); - await unregisterRepo(entry.path); - console.log(t('clean.deletedRepo', { name: entry.name, storagePath: entry.storagePath })); - } catch (err) { logger.error({ err }, `Failed to delete ${entry.name}:`); } } @@ -176,18 +201,31 @@ export const cleanCommand = async (options?: { } const repoName = repo.repoPath.split(/[/\\]/).pop() || repo.repoPath; + let storagePath: string; + try { + storagePath = await requireDeletableStoragePath({ + path: repo.repoPath, + storagePath: repo.storagePath, + }); + } catch (err) { + if (err instanceof StorageDeletionError) { + logger.error(`Refusing to clean ${repoName}: ${err.message}`); + return; + } + throw err; + } if (!options?.force) { console.log(t('clean.deleteCurrent', { repoName })); - console.log(` ${t('common.path')}: ${repo.storagePath}`); + console.log(` ${t('common.path')}: ${storagePath}`); console.log(`\n${t('common.runForceConfirm')}`); return; } try { - await fs.rm(repo.storagePath, { recursive: true, force: true }); + await fs.rm(storagePath, { recursive: true, force: true }); await unregisterRepo(repo.repoPath); - console.log(t('common.deleted', { target: repo.storagePath })); + console.log(t('common.deleted', { target: storagePath })); } catch (err) { logger.error({ err }, 'Failed to delete:'); } diff --git a/gitnexus/src/cli/cli-message.ts b/gitnexus/src/cli/cli-message.ts index b9b509b52..f65b55663 100644 --- a/gitnexus/src/cli/cli-message.ts +++ b/gitnexus/src/cli/cli-message.ts @@ -61,6 +61,7 @@ export type RecoveryHint = | 'gitnexusrc-invalid' | 'default-branch-invalid' | 'index-lock-timeout' + | 'index-lock-guard-recovery' | 'undeclared-relation-pair'; /** diff --git a/gitnexus/src/cli/doctor.ts b/gitnexus/src/cli/doctor.ts index 0c8e72e9e..202baafbd 100644 --- a/gitnexus/src/cli/doctor.ts +++ b/gitnexus/src/cli/doctor.ts @@ -14,6 +14,7 @@ import { import { cudaRedirectDoctorStatus } from '../core/embeddings/onnxruntime-node-resolver.js'; import { checkLbugNative, + ftsAvailabilityLabel, type NativeCheckResult, probeFtsExtensionLoad, probeVectorExtensionLoad, @@ -23,8 +24,12 @@ import { getOsPageSize, isPageSizeAwareLadybug, } from '../core/lbug/lbug-config.js'; -import { diagnoseExtensionLoad } from '../core/lbug/extension-load-error.js'; -import { getExtensionInstallPolicy } from '../core/lbug/extension-loader.js'; +import { diagnoseExtensionLoad, extractExtensionPath } from '../core/lbug/extension-load-error.js'; +import { resolveFtsVersionPair } from '../core/lbug/vendored-extension-path.js'; +import { + getExtensionInstallPolicy, + resolveAnalyzeInstallPolicy, +} from '../core/lbug/extension-loader.js'; import { updateEligibleInstallSync } from '../core/install-context.js'; import { readValidatedUpdateCacheSync, type ValidatedUpdateCache } from '../core/update-cache.js'; import { t } from './i18n/index.js'; @@ -258,9 +263,7 @@ export const doctorCommand = async () => { const ftsProbe = nativeCheck.ok ? await probeFtsExtensionLoad() : { loaded: false, reason: 'LadybugDB native module (lbugjs.node) failed to load' }; - console.log( - ` ${label('doctor.labels.fullTextSearch', 18)}${ftsProbe.loaded ? 'available' : 'unavailable'}`, - ); + console.log(` ${label('doctor.labels.fullTextSearch', 18)}${ftsAvailabilityLabel(ftsProbe)}`); if (!ftsProbe.loaded && ftsProbe.reason) { console.log(` ${padDisplayEnd('', 18)}${ftsProbe.reason}`); // Add an actionable remedy for recognized failure classes (#2374). The @@ -268,7 +271,15 @@ export const doctorCommand = async () => { // ("specified module could not be found") is opaque, so name the fix (VC++ // redist, then OpenSSL) instead of leaving the user to reinstall in vain. // `unknown`'s remedy is "run doctor", which would be circular here. - const { kind, remedy } = diagnoseExtensionLoad(ftsProbe.reason); + // Policy `never` is not a load failure — skip structural diagnosis. + const { kind, remedy } = ftsProbe.suppressed + ? { kind: 'unknown' as const, remedy: '' } + : diagnoseExtensionLoad( + ftsProbe.reason, + 'FTS', + extractExtensionPath(ftsProbe.reason), + resolveFtsVersionPair(extractExtensionPath(ftsProbe.reason)), + ); if (kind !== 'unknown') { console.log(` ${padDisplayEnd('', 18)}${remedy}`); } @@ -291,25 +302,30 @@ export const doctorCommand = async () => { console.log(` ${padDisplayEnd('', 18)}${remedy}`); } } - // Semantic mode follows the probe, not the platform: without a loadable - // VECTOR extension the index can be neither built nor queried, so search is - // really on exact scan no matter what the platform would allow. + // Doctor has no repository target, so this probe can report only whether the + // runtime can load VECTOR. It cannot claim that a repository has built the + // named HNSW index. Keep the capability and repository state distinct. console.log( - ` ${label('doctor.labels.semanticMode', 18)}${ - vectorProbe.loaded ? capabilities.semanticMode : 'exact-scan' - }`, + ` ${label('doctor.labels.semanticMode', 18)}${t( + vectorProbe.loaded + ? 'doctor.vectorCapability.indexUnverified' + : 'doctor.vectorCapability.exactScanOnly', + )}`, ); // Surface the optional-extension install policy so offline users can see // whether analyze/query will reach the network (extension.ladybugdb.com). // Literal label (like the 'native' line) to avoid adding i18n keys. - const installPolicy = getExtensionInstallPolicy(); - const policyHint = - installPolicy === 'load-only' + const serveQueryPolicy = getExtensionInstallPolicy(); + const analyzePolicy = resolveAnalyzeInstallPolicy(); + const policyHint = (policy: string) => + policy === 'load-only' ? ' (offline; load only, no network install)' - : installPolicy === 'never' + : policy === 'never' ? ' (optional extensions disabled)' : ' (installs missing extensions over network)'; - console.log(` ${padDisplayEnd('Ext install:', 18)}${installPolicy}${policyHint}`); + console.log( + ` ${padDisplayEnd('Ext install:', 18)}serve/query=${serveQueryPolicy}${policyHint(serveQueryPolicy)}; analyze=${analyzePolicy}${policyHint(analyzePolicy)}`, + ); console.log( ` ${label('doctor.labels.exactScanLimit', 18)}${t('doctor.chunks', { count: capabilities.exactScanLimit })}`, ); diff --git a/gitnexus/src/cli/embeddings-sync.ts b/gitnexus/src/cli/embeddings-sync.ts new file mode 100644 index 000000000..d45d059ed --- /dev/null +++ b/gitnexus/src/cli/embeddings-sync.ts @@ -0,0 +1,197 @@ +import { lstat } from 'node:fs/promises'; +import path from 'node:path'; +import { cliInfo } from './cli-message.js'; +import { getGitRoot } from '../storage/git.js'; +import { acquireIndexLock, requireExclusiveIndexLock } from '../storage/index-lock.js'; +import { getStoragePaths, loadMeta, saveMeta } from '../storage/repo-manager.js'; +import { + closeLbug, + executeQuery, + executeWithReusedStatement, + fetchExistingEmbeddingHashes, + initLbug, +} from '../core/lbug/lbug-adapter.js'; +import { runEmbeddingPipeline } from '../core/embeddings/embedding-pipeline.js'; +import { resolveEmbeddingIdentity } from '../core/embeddings/embedding-identity.js'; +import { + decideEmbeddingResume, + mintInterruptedCheckpoint, + mintPartialCheckpoint, + mintUnverifiedCountCheckpoint, + type EmbeddingCheckpoint, + type EmbeddingCheckpointProgress, +} from '../core/embedding-checkpoint.js'; +import { EMBEDDING_DIMS, embeddingDimsMismatch } from '../core/lbug/schema.js'; +import type { RepoMeta } from '../storage/repo-meta.js'; +import { + measurePersistedEmbeddingCount, + persistedEmbeddingCountOrUndefined, +} from '../core/embedding-count.js'; + +/** Add missing embeddings directly to a healthy index, checkpointing periodically. */ +export const embeddingsSyncCommand = async (inputPath?: string): Promise => { + const repoPath = inputPath ? path.resolve(inputPath) : getGitRoot(process.cwd()); + if (!repoPath) throw new Error('Not inside a git repository. Pass a repository path.'); + + const { lbugPath, metaPath } = getStoragePaths(repoPath); + const metaDir = path.dirname(metaPath); + const lock = await acquireIndexLock(metaDir); + try { + requireExclusiveIndexLock( + lock, + `Cannot acquire the index lock at ${metaDir}; refusing an unlocked embeddings sync.`, + ); + const meta = await loadMeta(metaDir); + if (!meta) + throw new Error(`No GitNexus index found for ${repoPath}. Run gitnexus analyze first.`); + if (meta.incrementalInProgress) { + throw new Error('The structural index is incomplete. Run gitnexus analyze --force first.'); + } + + let lbugStat; + try { + lbugStat = await lstat(lbugPath); + } catch { + throw new Error( + `The LadybugDB graph store at ${lbugPath} is missing. Run gitnexus analyze first.`, + ); + } + if (!lbugStat.isFile()) { + throw new Error( + `The LadybugDB graph store at ${lbugPath} is not a usable database file. Run gitnexus analyze first.`, + ); + } + + const identity = resolveEmbeddingIdentity(); + let forceReembedNodeIds: ReadonlySet | undefined; + let resumedFrom: EmbeddingCheckpoint | undefined; + if (meta.embeddingCheckpoint) { + const checkpoint = meta.embeddingCheckpoint; + const decision = decideEmbeddingResume(checkpoint, identity); + if (decision.action === 'abort') throw new Error(decision.error); + const identityDiffers = + checkpoint.provider !== identity.provider || + checkpoint.model !== identity.model || + checkpoint.dimensions !== identity.dimensions; + // `abandon` on a foreign identity drops the pending set only. Existing + // rows stay; sync would then embed the holes under the new identity and + // mix vector spaces. Fail closed — rebuild via analyze. + // + // Every kind is gated, `unverified-count` included. Exempting it looked + // safe because that kind only records "the count could not be read", but + // `decideEmbeddingResume` returns `abandon` for it BEFORE comparing + // identity, so the exemption was the only thing standing between a + // foreign identity and a silently mixed table. + if (identityDiffers) { + throw new Error( + `Cannot sync embeddings: the index checkpoint was written by ${checkpoint.model} ` + + `(${checkpoint.provider}) at ${checkpoint.dimensions} dimensions, but this run ` + + `resolves ${identity.model} (${identity.provider}) at ${identity.dimensions}. ` + + 'Run `gitnexus analyze --embeddings --force` to rebuild under the new identity.', + ); + } + cliInfo(decision.log); + if (decision.action === 'resume') { + forceReembedNodeIds = decision.pendingNodeIds; + resumedFrom = decision.resumedFrom; + } + } + + // The vector column is FLOAT[N] fixed when the index was built, and the + // pipeline deletes each batch's stale rows immediately before inserting the + // replacements — so a width change here deletes rows it cannot re-insert. + // `analyze` forces a full rebuild on the same mismatch; only a rebuild can + // retype the column, so this writer refuses instead. An absent recorded + // width is not a mismatch (see `embeddingDimsMismatch`). + if (embeddingDimsMismatch(meta.embeddingDims, EMBEDDING_DIMS)) { + throw new Error( + `Cannot sync embeddings: this index stores FLOAT[${meta.embeddingDims}] vectors, ` + + `but this run embeds at ${EMBEDDING_DIMS} dimensions. ` + + 'Run `gitnexus analyze --embeddings --force` to rebuild the column at the new width.', + ); + } + + await initLbug(lbugPath); + try { + const existing = await fetchExistingEmbeddingHashes(executeQuery); + let lastPercent = -1; + + const countEmbeddings = async (): Promise => + persistedEmbeddingCountOrUndefined(await measurePersistedEmbeddingCount(executeQuery)); + // One write path for every meta update this command makes. The re-read + // happens immediately before each save so a concurrent writer's fields + // survive. #2790 traced two production drifts to hand-copied writers of + // these exact fields, so this file keeps one copy instead of three. + const persistMeta = async (patch: (latest: RepoMeta) => Partial): Promise => { + const latest = (await loadMeta(metaDir)) ?? meta; + await saveMeta(metaDir, { ...latest, ...patch(latest) }); + }; + const saveCheckpoint = async ( + checkpoint: EmbeddingCheckpointProgress, + pendingNodeIds: string[], + embeddings?: number, + ): Promise => + persistMeta((latest) => ({ + ...(embeddings === undefined ? {} : { stats: { ...latest.stats, embeddings } }), + embeddingCheckpoint: mintInterruptedCheckpoint(identity, checkpoint, pendingNodeIds), + })); + + cliInfo(`Embedding ${repoPath}`); + cliInfo(`Checkpointed nodes already present: ${existing?.size ?? 0}`); + + const result = await runEmbeddingPipeline( + executeQuery, + executeWithReusedStatement, + (progress) => { + const percent = Math.floor(progress.percent); + if (percent !== lastPercent && (percent % 5 === 0 || percent === 100)) { + lastPercent = percent; + cliInfo( + ` ${percent}% — ${progress.nodesProcessed ?? 0}/${progress.totalNodes ?? '?'} nodes`, + ); + } + }, + {}, + undefined, + existing && existing.size ? existing : undefined, + { + forceReembedNodeIds, + onCheckpointWindowStart: async ({ nodeIds, ...checkpoint }) => { + await saveCheckpoint(checkpoint, nodeIds); + }, + onCheckpoint: async (checkpoint) => { + await saveCheckpoint(checkpoint, [], await countEmbeddings()); + }, + }, + ); + + const embeddings = await countEmbeddings(); + if (embeddings === undefined) { + // Keep last-known stats.embeddings. An interrupted window marker would + // fail the identity gate on the next run even though this run finished; + // unverified-count is the recovery kind that forces a recount (#2790). + await persistMeta(() => ({ + embeddingCheckpoint: result.failedNodeIds.length + ? mintPartialCheckpoint(identity, result, resumedFrom) + : mintUnverifiedCountCheckpoint(identity, { + nodesProcessed: result.nodesProcessed, + totalNodes: result.nodesProcessed, + chunksProcessed: result.chunksProcessed, + }), + })); + throw new Error('Could not verify persisted embedding count.'); + } + await persistMeta((latest) => ({ + stats: { ...latest.stats, embeddings }, + embeddingCheckpoint: result.failedNodeIds.length + ? mintPartialCheckpoint(identity, result, resumedFrom) + : undefined, + })); + cliInfo(`Embeddings ready: ${embeddings}`); + } finally { + await closeLbug().catch(() => {}); + } + } finally { + lock.release(); + } +}; diff --git a/gitnexus/src/cli/eval-server.ts b/gitnexus/src/cli/eval-server.ts index b2f656c39..fc6af8e63 100644 --- a/gitnexus/src/cli/eval-server.ts +++ b/gitnexus/src/cli/eval-server.ts @@ -609,10 +609,20 @@ export function formatImpactResult(result: any): string { } // #1858 — an interface / indirection boundary on the path makes this a lower // bound; surface it so the count is not read as exhaustive. + // + // The header names no cause AND asserts no omitted caller, because it cannot + // know either. DI / dynamic dispatch was the only producer of `lower-bound` + // when this was written; #3399 added callables named in VALUE position (a + // registration table, a callback argument), and one of its producers is a + // probe that could not RUN — `callableValueReferenceBoundaries` hedges on a + // failed query and says in so many words that whether the symbol is + // registered is unknown. A header claiming "some callers are not traced" + // would there assert an omission nothing established, and would contradict + // the bullet printed directly under it. The bullets carry the cause — they + // are generated per-cause by `computeEpistemicBoundary` — so the header only + // has to say the count is a floor. if (result.epistemic === 'lower-bound') { - lines.push( - '⚠️ Lower bound — unresolved indirection on the path (callers binding via DI / dynamic dispatch are not traced; actual impact may be higher):', - ); + lines.push('⚠️ Lower bound — impact may be incomplete and actual impact may be higher:'); for (const b of result.boundaries || []) lines.push(` • ${b}`); } pushCallgraphRiskLines(lines, result); diff --git a/gitnexus/src/cli/group-status-format.ts b/gitnexus/src/cli/group-status-format.ts new file mode 100644 index 000000000..f5d09c47e --- /dev/null +++ b/gitnexus/src/cli/group-status-format.ts @@ -0,0 +1,27 @@ +/** + * Rendering for `gitnexus group status` rows, kept out of the Commander action + * so it can be tested. Inline, the cell was only reachable by booting a backend + * against a real group, which is how `STALE (-1 commits behind)` went unnoticed + * (#3256). + */ + +/** The fields of a `groupStatus` repo row the index column reads. */ +export interface GroupRepoIndexRow { + indexStale: boolean; + commitsBehind?: number; +} + +/** + * The index column of a `group status` row. The output is unchanged except + * for one case: a count that is not a real count renders as `?`. + * + * `group/service.ts` has always reported a repo with no recorded commit as + * `{ indexStale: true, commitsBehind: -1 }`. The previous `?? '?'` fallback + * never caught that, because `??` only falls back on `null` / `undefined`. + */ +export const formatIndexStatusCell = (row: GroupRepoIndexRow): string => { + if (!row.indexStale) return 'OK '; + const n = row.commitsBehind; + const count = typeof n === 'number' && n >= 0 ? String(n) : '?'; + return `STALE (${count} commits behind)`; +}; diff --git a/gitnexus/src/cli/group.ts b/gitnexus/src/cli/group.ts index abc13fa2b..824b13b6a 100644 --- a/gitnexus/src/cli/group.ts +++ b/gitnexus/src/cli/group.ts @@ -4,6 +4,7 @@ import type { Command } from 'commander'; import type { RegistryWriteOutcome } from '../core/group/sync.js'; import type { MatchType } from '../core/group/types.js'; import { logger } from '../core/logger.js'; +import { formatIndexStatusCell } from './group-status-format.js'; const _require = createRequire(import.meta.url); const yaml = _require('js-yaml') as typeof import('js-yaml'); @@ -161,9 +162,7 @@ export function registerGroupCommands(program: Command): void { console.log(` ${repoPath.padEnd(25)} MISSING (no entry in the registry)`); continue; } - const idx = row.indexStale - ? `STALE (${row.commitsBehind ?? '?'} commits behind)` - : 'OK '; + const idx = formatIndexStatusCell(row); const ctr = row.contractsStale ? ' CONTRACTS_STALE' : ''; console.log(` ${repoPath.padEnd(25)} ${idx}${ctr}`); } diff --git a/gitnexus/src/cli/i18n/en.ts b/gitnexus/src/cli/i18n/en.ts index a560fa23e..d280b6a2d 100644 --- a/gitnexus/src/cli/i18n/en.ts +++ b/gitnexus/src/cli/i18n/en.ts @@ -123,8 +123,10 @@ export const en = { 'doctor.labels.onnx': 'ONNX:', 'doctor.labels.graphStore': 'Graph store:', 'doctor.labels.fullTextSearch': 'Full-text search:', - 'doctor.labels.vectorIndex': 'VECTOR index:', - 'doctor.labels.semanticMode': 'Semantic mode:', + 'doctor.labels.vectorIndex': 'VECTOR extension:', + 'doctor.labels.semanticMode': 'Semantic support:', + 'doctor.vectorCapability.indexUnverified': 'vector-index capable (repository index not checked)', + 'doctor.vectorCapability.exactScanOnly': 'exact-scan only (VECTOR extension unavailable)', 'doctor.labels.exactScanLimit': 'Exact scan limit:', 'doctor.labels.note': 'Note:', 'doctor.labels.backend': 'Backend:', @@ -166,6 +168,8 @@ export const en = { 'error.watch.ambiguous': '`gitnexus watch` is ambiguous.\n Local working-tree incremental index: gitnexus analyze --watch\n Scheduled remote clone/pull + analyze: gitnexus auto-sync start\n', 'help.command.analyze.description': 'Index a repository (full analysis)', + 'help.command.embeddings.sync.description': + 'Add missing embeddings to an existing index, checkpointing periodically for safe resume', 'help.command.index.description': 'Register an existing .gitnexus/ folder into the global registry (no re-analysis needed)', 'help.command.serve.description': 'Start local HTTP server for web UI connection', @@ -354,5 +358,5 @@ export const en = { 'help.identityCache.environment': '\nAnalyzer identity cache:\n GITNEXUS_ANALYZER_IDENTITY_CACHE_DIR=/absolute/protected/dir\n Operator-trusted persistent cache for warm cross-process status. The directory must pre-exist, be outside the GitNexus package/build roots, and contain no symlink or junction components. Defaults remain fail-closed on platforms without POSIX ownership APIs.', 'help.analyze.environment': - '\nEnvironment variables:\n GITNEXUS_NO_GITIGNORE=1 Skip .gitignore parsing (still reads .gitnexusignore)\n GITNEXUS_MAX_FILE_SIZE=N Override large-file skip threshold (KB). Default 512, max 32768.\n GITNEXUS_ANALYZER_IDENTITY_CACHE_DIR=/absolute/protected/dir Operator-trusted persistent analyzer identity cache; must pre-exist, be outside package/build roots, and contain no symlink/junction components.\n GITNEXUS_WORKER_SUB_BATCH_TIMEOUT_MS=N Worker idle timeout in milliseconds. Default 30000.\n GITNEXUS_WAL_CHECKPOINT_THRESHOLD=N LadybugDB WAL auto-checkpoint threshold in bytes (default 67108864 = 64 MiB; -1 keeps Ladybug stock ~16 MiB).\n GITNEXUS_WORKER_SUB_BATCH_MAX_BYTES=N Worker job byte budget. Default 8388608.\n GITNEXUS_WORKER_POOL_SIZE=N Parse worker count override. Default cores-1 capped at 16.\n GITNEXUS_PARSE_CHUNK_CONCURRENCY=N Concurrent in-flight parse chunks. Default 2.\n GITNEXUS_WORKER_MAX_RESPAWNS_PER_SLOT=N Max replacement spawns per slot before drop. Default 3.\n GITNEXUS_WORKER_MAX_CUMULATIVE_TIMEOUT_MS=N Total retry wall-time per job. Default 5x sub-batch timeout.\n GITNEXUS_WORKER_CONSECUTIVE_FAILURE_THRESHOLD=N Per-slot deaths to trip circuit breaker. Default max(3, poolSize).\n GITNEXUS_WORKER_SHUTDOWN_DRAIN_MS=N Max wait at pool shutdown for a retired worker still inside native code (terminated at its next safe point instead of aborting the process). Default 30000.\n GITNEXUS_CPP_CAPTURE_BUDGET_MS=N Per-file wall-clock budget for C++ capture extraction; on breach the file keeps partial captures with a warning. Default 20000.\n GITNEXUS_EMBEDDING_THREADS=N Limit local ONNX CPU threads for --embeddings.\n GITNEXUS_SEMANTIC_EXACT_SCAN_LIMIT=N Max embedding chunks for exact-scan fallback. Default 10000.\n GITNEXUS_VECTOR_MAX_DISTANCE=N Max accepted semantic/vector cosine distance (0 < N <= 2; higher values clamp to 2). Default 0.6 for MCP, 0.5 elsewhere.\n\nFlags override the corresponding env vars when both are provided.\n\nTip: `.gitnexusignore` supports `.gitignore`-style negation. Add e.g.\n `!__tests__/` to index a directory that is auto-filtered by default (#771).', + '\nEnvironment variables:\n GITNEXUS_NO_GITIGNORE=1 Skip .gitignore parsing (still reads .gitnexusignore)\n GITNEXUS_MAX_FILE_SIZE=N Override large-file skip threshold (KB). Default 512, max 32768.\n GITNEXUS_STORAGE_PATH=/absolute/index Complete external index directory. Preserves the existing configuration semantics and overrides GITNEXUS_STORAGE_ROOT when both are set.\n GITNEXUS_STORAGE_ROOT=/absolute/root External index root; each repository uses an isolated -/ slot.\n GITNEXUS_CONTENT_RETENTION=full Source-text retention profile: full, symbol, or none. Default full.\n GITNEXUS_ANALYZER_IDENTITY_CACHE_DIR=/absolute/protected/dir Operator-trusted persistent analyzer identity cache; must pre-exist, be outside package/build roots, and contain no symlink/junction components.\n GITNEXUS_WORKER_SUB_BATCH_TIMEOUT_MS=N Worker idle timeout in milliseconds. Default 30000.\n GITNEXUS_WAL_CHECKPOINT_THRESHOLD=N LadybugDB WAL auto-checkpoint threshold in bytes (default 67108864 = 64 MiB; -1 keeps Ladybug stock ~16 MiB).\n GITNEXUS_WORKER_SUB_BATCH_MAX_BYTES=N Worker job byte budget. Default 8388608.\n GITNEXUS_WORKER_POOL_SIZE=N Parse worker count override. Default cores-1 capped at 16.\n GITNEXUS_PARSE_CHUNK_CONCURRENCY=N Concurrent in-flight parse chunks. Default 2.\n GITNEXUS_WORKER_MAX_RESPAWNS_PER_SLOT=N Max replacement spawns per slot before drop. Default 3.\n GITNEXUS_WORKER_MAX_CUMULATIVE_TIMEOUT_MS=N Total retry wall-time per job. Default 5x sub-batch timeout.\n GITNEXUS_WORKER_CONSECUTIVE_FAILURE_THRESHOLD=N Per-slot deaths to trip circuit breaker. Default max(3, poolSize).\n GITNEXUS_WORKER_SHUTDOWN_DRAIN_MS=N Max wait at pool shutdown for a retired worker still inside native code (terminated at its next safe point instead of aborting the process). Default 30000.\n GITNEXUS_CPP_CAPTURE_BUDGET_MS=N Per-file wall-clock budget for C++ capture extraction; on breach the file keeps partial captures with a warning. Default 20000.\n GITNEXUS_EMBEDDING_THREADS=N Limit local ONNX CPU threads for --embeddings.\n GITNEXUS_EMBEDDING_RETRY_TIMEOUTS=1 Retry per-attempt HTTP embedding timeouts through GITNEXUS_EMBEDDING_MAX_ATTEMPTS (default off; timeouts stay terminal).\n GITNEXUS_SEMANTIC_EXACT_SCAN_LIMIT=N Max embedding chunks for exact-scan fallback. Default 10000.\n GITNEXUS_VECTOR_MAX_DISTANCE=N Max accepted semantic/vector cosine distance (0 < N <= 2; higher values clamp to 2). Default 0.6 for MCP, 0.5 elsewhere.\n\nFlags override the corresponding env vars when both are provided.\n\nTip: `.gitnexusignore` supports `.gitignore`-style negation. Add e.g.\n `!__tests__/` to index a directory that is auto-filtered by default (#771).', } as const; diff --git a/gitnexus/src/cli/i18n/zh-CN.ts b/gitnexus/src/cli/i18n/zh-CN.ts index 536b8ffb9..97a6c2eff 100644 --- a/gitnexus/src/cli/i18n/zh-CN.ts +++ b/gitnexus/src/cli/i18n/zh-CN.ts @@ -119,8 +119,10 @@ export const zhCN = { 'doctor.labels.onnx': 'ONNX:', 'doctor.labels.graphStore': '图存储:', 'doctor.labels.fullTextSearch': '全文搜索:', - 'doctor.labels.vectorIndex': '向量索引:', - 'doctor.labels.semanticMode': '语义模式:', + 'doctor.labels.vectorIndex': 'VECTOR 扩展:', + 'doctor.labels.semanticMode': '语义支持:', + 'doctor.vectorCapability.indexUnverified': '支持向量索引(未检查仓库索引)', + 'doctor.vectorCapability.exactScanOnly': '仅精确扫描(VECTOR 扩展不可用)', 'doctor.labels.exactScanLimit': '精确扫描上限:', 'doctor.labels.note': '说明:', 'doctor.labels.backend': '后端:', @@ -162,6 +164,8 @@ export const zhCN = { 'error.watch.ambiguous': '`gitnexus watch` 含义不明确。\n 本地工作区增量索引:gitnexus analyze --watch\n 定时远程 clone/pull 并分析:gitnexus auto-sync start\n', 'help.command.analyze.description': '索引仓库(完整分析)', + 'help.command.embeddings.sync.description': + '向现有索引添加缺失的嵌入,并定期保存检查点以安全续跑', 'help.command.index.description': '将现有 .gitnexus/ 文件夹注册到全局注册表(无需重新分析)', 'help.command.serve.description': '启动供 Web UI 连接的本地 HTTP 服务器', 'help.command.mcp.description': @@ -327,5 +331,5 @@ export const zhCN = { 'help.identityCache.environment': '\n分析器身份缓存:\n GITNEXUS_ANALYZER_IDENTITY_CACHE_DIR=/absolute/protected/dir\n 由操作员明确信任的持久缓存,用于跨进程快速查询状态。目录必须预先存在、位于 GitNexus 包/构建根目录之外,且路径中不得包含符号链接或 junction。缺少 POSIX 所有权 API 的平台默认保持故障关闭。', 'help.analyze.environment': - '\n环境变量:\n GITNEXUS_NO_GITIGNORE=1 跳过 .gitignore 解析(仍读取 .gitnexusignore)\n GITNEXUS_MAX_FILE_SIZE=N 覆盖大文件跳过阈值(KB)。默认 512,最大 32768。\n GITNEXUS_ANALYZER_IDENTITY_CACHE_DIR=/absolute/protected/dir 由操作员明确信任的持久分析器身份缓存;目录必须预先存在、位于包/构建根目录之外,且路径中不得包含符号链接或 junction。\n GITNEXUS_WORKER_SUB_BATCH_TIMEOUT_MS=N Worker 空闲超时(毫秒)。默认 30000。\n GITNEXUS_WAL_CHECKPOINT_THRESHOLD=N LadybugDB WAL 自动 checkpoint 阈值(字节,默认 67108864 = 64 MiB;-1 保持 Ladybug 默认约 16 MiB)。\n GITNEXUS_WORKER_SUB_BATCH_MAX_BYTES=N Worker 作业字节预算。默认 8388608。\n GITNEXUS_WORKER_POOL_SIZE=N 解析 worker 数量覆盖值。默认 cores-1,最多 16。\n GITNEXUS_PARSE_CHUNK_CONCURRENCY=N 并发进行中的解析分块数。默认 2。\n GITNEXUS_WORKER_MAX_RESPAWNS_PER_SLOT=N 每个 slot 丢弃前允许的最大替换进程数。默认 3。\n GITNEXUS_WORKER_MAX_CUMULATIVE_TIMEOUT_MS=N 每个作业的总重试墙钟时间。默认 5 倍子批次超时。\n GITNEXUS_WORKER_CONSECUTIVE_FAILURE_THRESHOLD=N 每个 slot 触发熔断的死亡次数。默认 max(3, poolSize)。\n GITNEXUS_WORKER_SHUTDOWN_DRAIN_MS=N 线程池关闭时等待仍在原生代码中的已退役 worker 的最长时间(到达安全点后再终止,避免进程级 abort)。默认 30000。\n GITNEXUS_CPP_CAPTURE_BUDGET_MS=N C++ 捕获提取的每文件墙钟预算;超出后该文件保留部分捕获并输出警告。默认 20000。\n GITNEXUS_EMBEDDING_THREADS=N 限制 --embeddings 的本地 ONNX CPU 线程数。\n GITNEXUS_SEMANTIC_EXACT_SCAN_LIMIT=N exact-scan 回退的最大嵌入分块数。默认 10000。\n GITNEXUS_VECTOR_MAX_DISTANCE=N 语义/向量搜索接受的最大余弦距离(0 < N <= 2;超出则钳制为 2)。MCP 默认 0.6,其他路径默认 0.5。\n\n当参数和对应环境变量同时提供时,参数优先。\n\n提示:`.gitnexusignore` 支持 `.gitignore` 风格的取反。比如添加\n `!__tests__/` 可以索引默认自动过滤的目录(#771)。', + '\n环境变量:\n GITNEXUS_NO_GITIGNORE=1 跳过 .gitignore 解析(仍读取 .gitnexusignore)\n GITNEXUS_MAX_FILE_SIZE=N 覆盖大文件跳过阈值(KB)。默认 512,最大 32768。\n GITNEXUS_STORAGE_PATH=/absolute/index 完整外部索引目录。保留既有配置语义;与 GITNEXUS_STORAGE_ROOT 同时设置时优先使用。\n GITNEXUS_STORAGE_ROOT=/absolute/root 外部索引根目录;每个仓库使用独立的 <仓库名>-<规范路径哈希>/ 子目录。\n GITNEXUS_CONTENT_RETENTION=full 源码文本保留策略:full、symbol 或 none。默认 full。\n GITNEXUS_ANALYZER_IDENTITY_CACHE_DIR=/absolute/protected/dir 由操作员明确信任的持久分析器身份缓存;目录必须预先存在、位于包/构建根目录之外,且路径中不得包含符号链接或 junction。\n GITNEXUS_WORKER_SUB_BATCH_TIMEOUT_MS=N Worker 空闲超时(毫秒)。默认 30000。\n GITNEXUS_WAL_CHECKPOINT_THRESHOLD=N LadybugDB WAL 自动 checkpoint 阈值(字节,默认 67108864 = 64 MiB;-1 保持 Ladybug 默认约 16 MiB)。\n GITNEXUS_WORKER_SUB_BATCH_MAX_BYTES=N Worker 作业字节预算。默认 8388608。\n GITNEXUS_WORKER_POOL_SIZE=N 解析 worker 数量覆盖值。默认 cores-1,最多 16。\n GITNEXUS_PARSE_CHUNK_CONCURRENCY=N 并发进行中的解析分块数。默认 2。\n GITNEXUS_WORKER_MAX_RESPAWNS_PER_SLOT=N 每个 slot 丢弃前允许的最大替换进程数。默认 3。\n GITNEXUS_WORKER_MAX_CUMULATIVE_TIMEOUT_MS=N 每个作业的总重试墙钟时间。默认 5 倍子批次超时。\n GITNEXUS_WORKER_CONSECUTIVE_FAILURE_THRESHOLD=N 每个 slot 触发熔断的死亡次数。默认 max(3, poolSize)。\n GITNEXUS_WORKER_SHUTDOWN_DRAIN_MS=N 线程池关闭时等待仍在原生代码中的已退役 worker 的最长时间(到达安全点后再终止,避免进程级 abort)。默认 30000。\n GITNEXUS_CPP_CAPTURE_BUDGET_MS=N C++ 捕获提取的每文件墙钟预算;超出后该文件保留部分捕获并输出警告。默认 20000。\n GITNEXUS_EMBEDDING_THREADS=N 限制 --embeddings 的本地 ONNX CPU 线程数。\n GITNEXUS_EMBEDDING_RETRY_TIMEOUTS=1 将单次 HTTP 嵌入超时纳入 GITNEXUS_EMBEDDING_MAX_ATTEMPTS 重试(默认关闭,超时仍为终止错误)。\n GITNEXUS_SEMANTIC_EXACT_SCAN_LIMIT=N exact-scan 回退的最大嵌入分块数。默认 10000。\n GITNEXUS_VECTOR_MAX_DISTANCE=N 语义/向量搜索接受的最大余弦距离(0 < N <= 2;超出则钳制为 2)。MCP 默认 0.6,其他路径默认 0.5。\n\n当参数和对应环境变量同时提供时,参数优先。\n\n提示:`.gitnexusignore` 支持 `.gitignore` 风格的取反。比如添加\n `!__tests__/` 可以索引默认自动过滤的目录(#771)。', } satisfies EnglishMessages; diff --git a/gitnexus/src/cli/index-repo.ts b/gitnexus/src/cli/index-repo.ts index 09888e901..08597fdfd 100644 --- a/gitnexus/src/cli/index-repo.ts +++ b/gitnexus/src/cli/index-repo.ts @@ -16,13 +16,17 @@ import path from 'path'; import fs from 'fs/promises'; import { - getStoragePaths, - INDEX_METADATA_FILE, loadMeta, + saveMeta, ensureGitNexusIgnored, registerRepo, } from '../storage/repo-manager.js'; import { getGitRoot, getRemoteUrl, isGitRepo } from '../storage/git.js'; +import { + getIndexStorageRequirements, + requireStoragePath, + StorageRequirementError, +} from '../storage/storage-resolver.js'; export interface IndexOptions { force?: boolean; @@ -69,46 +73,39 @@ export const indexCommand = async (inputPathParts?: string[], options?: IndexOpt return; } - const { storagePath, lbugPath } = getStoragePaths(repoPath); - - // ── Verify index exists (metadata file, legacy metadata, or restorable DB) ─ - let hasMetadataIndex = false; - let hasLegacyIndex = false; - let hasLbugIndex = false; - + let storagePath: string; try { - await fs.access(path.join(storagePath, INDEX_METADATA_FILE)); - hasMetadataIndex = true; - } catch {} - - try { - await fs.access(path.join(storagePath, 'meta.json')); - hasLegacyIndex = true; - } catch {} - - try { - await fs.access(lbugPath); - hasLbugIndex = true; - } catch {} - - if (!hasMetadataIndex && !hasLegacyIndex && !hasLbugIndex) { - console.log(` No GitNexus index found.`); - console.log(` Expected gitnexus.json, .gitnexus/meta.json, or LadybugDB at: ${storagePath}`); - console.log(' Run `gitnexus analyze` to build the index first.\n'); - process.exitCode = 1; - return; - } - - // ── Verify lbug database exists ─────────────────────────────────── - if (!hasLbugIndex) { - console.log(` Index exists but contains no LadybugDB database.`); - console.log(' Run `gitnexus analyze` to build the index.\n'); + storagePath = await requireStoragePath(repoPath, getIndexStorageRequirements(!!options?.force)); + } catch (error) { + if (!(error instanceof StorageRequirementError)) { + console.log(` ${error instanceof Error ? error.message : String(error)}\n`); + process.exitCode = 1; + return; + } + const inspection = error.inspection; + if (inspection.state === 'missing' || inspection.state === 'empty') { + console.log(` No GitNexus index found.`); + console.log( + ` Expected gitnexus.json, .gitnexus/meta.json, or LadybugDB at: ${inspection.storagePath}`, + ); + console.log(' Run `gitnexus analyze` to build the index first.\n'); + } else if (inspection.state === 'unowned' && !options?.force) { + console.log(` gitnexus.json or .gitnexus/meta.json is missing.`); + console.log(' Use --force to register anyway (stats will be empty),'); + console.log(' or run `gitnexus analyze` to rebuild properly.\n'); + } else if (!inspection.hasCodeIndexDB) { + console.log(` Index exists but contains no LadybugDB database.`); + console.log(' Run `gitnexus analyze` to build the index.\n'); + } else { + console.log(` ${error.message}\n`); + } process.exitCode = 1; return; } // ── Load or reconstruct meta ────────────────────────────────────── let meta = await loadMeta(storagePath); + let reconstructedMeta = false; if (!meta) { if (!options?.force) { @@ -122,9 +119,27 @@ export const indexCommand = async (inputPathParts?: string[], options?: IndexOpt // --force: build a minimal meta so the repo can be registered meta = { repoPath, + storagePath, lastCommit: '', indexedAt: new Date().toISOString(), }; + reconstructedMeta = true; + } + + // `index --force` is the explicit adoption path for an existing external + // database whose legacy metadata predates storagePath binding, and for a + // repository-local slot whose metadata still names another checkout. + if (options?.force) { + const adoptedRepoPath = path.resolve(repoPath); + const adoptedStoragePath = path.resolve(storagePath); + const ownershipChanged = + path.resolve(meta.repoPath) !== adoptedRepoPath || + meta.storagePath === undefined || + path.resolve(meta.storagePath) !== adoptedStoragePath; + if (ownershipChanged) { + meta = { ...meta, repoPath: adoptedRepoPath, storagePath: adoptedStoragePath }; + reconstructedMeta = true; + } } // ── Register in global registry ─────────────────────────────────── @@ -135,8 +150,11 @@ export const indexCommand = async (inputPathParts?: string[], options?: IndexOpt if (!meta.remoteUrl && isGitRepo(repoPath)) { meta.remoteUrl = getRemoteUrl(repoPath); } - await registerRepo(repoPath, meta); - await ensureGitNexusIgnored(repoPath); + if (reconstructedMeta) { + await saveMeta(storagePath, meta); + } + await registerRepo(repoPath, meta, { storagePath }); + await ensureGitNexusIgnored(repoPath, storagePath); const projectName = path.basename(repoPath); const { stats } = meta; diff --git a/gitnexus/src/cli/index.ts b/gitnexus/src/cli/index.ts index 70238dc00..22bb3c0dd 100644 --- a/gitnexus/src/cli/index.ts +++ b/gitnexus/src/cli/index.ts @@ -81,6 +81,7 @@ program 'Re-parse every source file instead of replaying cached parser output', ) .option('--repair-fts', 'Repair/rebuild search FTS indexes without full re-analysis') + .option('--skip-fts', 'Skip FTS extension loading and keyword search indexes') .option( '--embeddings [limit]', 'Enable embedding generation for semantic search (off by default). ' + @@ -289,7 +290,8 @@ program program .command('status') - .description('Show index status for current repo') + .description('Show index status for the current repo or a registered index') + .option('-r, --repo ', 'Registered repository alias or path (works after checkout removal)') .option('--json', 'Emit machine-readable index and analyzer provenance') .addHelpText('after', () => t('help.identityCache.environment')) .action(createLazyAction(() => import('./status.js'), 'statusCommand')); @@ -304,15 +306,13 @@ program .description('Install the latest published GitNexus globally (`npm i -g gitnexus@`).') .action(createLazyAction(() => import('./update.js'), 'updateCommand')); -program +const embeddings = program .command('embeddings') - .description('Manage the on-demand local embedding runtime') + .description(t('help.command.embeddings.description')); + +embeddings .command('install') - .description( - 'Install the local embedding stack (@huggingface/transformers + onnxruntime-node) on demand. ' + - 'Heals installs where npm skipped the optional packages (e.g. behind an HTTP proxy, #2370). ' + - 'Downloads only from your configured npm registry — mirrors and proxies apply.', - ) + .description(t('help.command.embeddings.install.description')) .option( '--cuda', "Also download the CUDA GPU binaries (runs onnxruntime-node's NuGet postinstall; " + @@ -321,6 +321,12 @@ program .option('--force', 'Install into the runtime prefix even when the stack already resolves') .action(createLazyAction(() => import('./embeddings.js'), 'embeddingsInstallCommand')); +embeddings + .command('sync [path]') + .description(t('help.command.embeddings.sync.description')) + .addHelpText('after', () => t('help.analyze.environment')) + .action(createLbugLazyAction(() => import('./embeddings-sync.js'), 'embeddingsSyncCommand')); + program .command('clean') .description('Delete GitNexus index for current repo') @@ -407,7 +413,10 @@ program .option('-c, --context ', 'Task context to improve ranking') .option('-g, --goal ', 'What you want to find') .option('-l, --limit ', 'Max processes to return (default: 5)') - .option('--content', 'Include full symbol source code') + .option( + '--content', + 'Include retained symbol source text (reports availability when disabled by retention)', + ) .action(createLbugLazyAction(() => import('./tool.js'), 'queryCommand')); program @@ -418,7 +427,10 @@ program .option('-u, --uid ', 'Direct symbol UID (zero-ambiguity lookup)') .option('-f, --file ', 'File path to disambiguate common names') .option('-l, --limit ', 'Max callers/callees/processes to return') - .option('--content', 'Include full symbol source code') + .option( + '--content', + 'Include retained symbol source text (reports availability when disabled by retention)', + ) .action(createLbugLazyAction(() => import('./tool.js'), 'contextCommand')); program diff --git a/gitnexus/src/cli/publish.ts b/gitnexus/src/cli/publish.ts index 8aedc9c35..730670d4c 100644 --- a/gitnexus/src/cli/publish.ts +++ b/gitnexus/src/cli/publish.ts @@ -30,7 +30,12 @@ import { parseOwnerRepoFromRemote, } from 'gitnexus-shared'; import { getGitRoot, getRemoteOriginUrl, getCurrentCommit } from '../storage/git.js'; -import { hasIndex } from '../storage/repo-manager.js'; +import { + requireStoragePath, + STATUS_STORAGE_REQUIREMENTS, + StorageRequirementError, + isUnusableIndexInspection, +} from '../storage/storage-resolver.js'; import { cliInfo, cliError } from './cli-message.js'; export interface PublishOptions { @@ -95,9 +100,18 @@ export const publishCommand = async ( // Publishing without an index is almost always a mistake — the // registry's nightly sync would fetch a stale or missing graph file // and mark the entry `missing`. Refuse loudly with a fix-it hint. - if (!(await hasIndex(repoPath))) { + try { + await requireStoragePath(repoPath, STATUS_STORAGE_REQUIREMENTS); + } catch (error) { + if (!(error instanceof StorageRequirementError)) throw error; + const inspection = error.inspection; + if (!isUnusableIndexInspection(inspection)) { + cliError(`[understand-quickly] ${error.message}`); + process.exitCode = 1; + return; + } cliError( - `[understand-quickly] no GitNexus index found at ${repoPath}/.gitnexus.\n` + + `[understand-quickly] no usable GitNexus index found for ${repoPath}.\n` + 'Run `gitnexus analyze` first, then re-run `gitnexus publish`.', ); process.exitCode = 1; diff --git a/gitnexus/src/cli/remove.ts b/gitnexus/src/cli/remove.ts index 260c474e9..72bb92364 100644 --- a/gitnexus/src/cli/remove.ts +++ b/gitnexus/src/cli/remove.ts @@ -36,12 +36,11 @@ import { t } from './i18n/index.js'; import { readRegistry, resolveRegistryEntry, - assertSafeStoragePath, unregisterRepo, RegistryNotFoundError, RegistryAmbiguousTargetError, - UnsafeStoragePathError, } from '../storage/repo-manager.js'; +import { requireDeletableStoragePath, StorageDeletionError } from '../storage/storage-resolver.js'; export const removeCommand = async (target: string, options?: { force?: boolean }) => { // Read the registry snapshot once and pass it to the resolver — this @@ -80,18 +79,13 @@ export const removeCommand = async (target: string, options?: { force?: boolean return; } - // Safety guard (#1003 review — @magyargergo): refuse to proceed if - // the registry entry's `storagePath` isn't the canonical - // `/.gitnexus` subfolder. `~/.gitnexus/registry.json` is - // user-writable, so a corrupted or hand-edited entry could point - // storagePath at the repo root, an empty string (→ cwd), a parent - // dir, or anywhere else; `fs.rm(recursive: true, force: true)` on - // any of those would be a runtime disaster. Bail before touching - // disk, with an actionable hint for recovering a broken registry. + // Validate immediately before deletion. `--force` skips confirmation only; + // it does not bypass the ownership and dangerous-path checks. + let storagePath: string; try { - assertSafeStoragePath(entry); + storagePath = await requireDeletableStoragePath(entry); } catch (err) { - if (err instanceof UnsafeStoragePathError) { + if (err instanceof StorageDeletionError) { cliError(t('common.error', { message: err.message })); process.exit(1); } @@ -104,7 +98,7 @@ export const removeCommand = async (target: string, options?: { force?: boolean // orphaned — `listRegisteredRepos({ validate: true })` prunes those on // next read, so the failure is self-healing. try { - await fs.rm(entry.storagePath, { recursive: true, force: true }); + await fs.rm(storagePath, { recursive: true, force: true }); await unregisterRepo(entry.path); console.log(t('remove.removed', { name: entry.name })); console.log(` ${t('common.path')}: ${entry.path}`); diff --git a/gitnexus/src/cli/setup.ts b/gitnexus/src/cli/setup.ts index 7ca83643c..3c74877e8 100644 --- a/gitnexus/src/cli/setup.ts +++ b/gitnexus/src/cli/setup.ts @@ -450,6 +450,7 @@ const HOOK_HELPERS = [ 'hook-db-lock-probe.cjs', 'win-rm-list-json.ps1', 'resolve-analyze-cmd.cjs', + 'registry-query.cjs', ] as const; // win-rm-list-json.ps1 is best-effort: it is read (not require()'d) by diff --git a/gitnexus/src/cli/status.ts b/gitnexus/src/cli/status.ts index e0ee08bcc..de69c513c 100644 --- a/gitnexus/src/cli/status.ts +++ b/gitnexus/src/cli/status.ts @@ -5,7 +5,22 @@ */ import path from 'path'; -import { findRepo, getStoragePaths, loadMeta, hasKuzuIndex } from '../storage/repo-manager.js'; +import { + getStoragePaths, + loadMeta, + hasKuzuIndex, + readRegistryStrict, + resolveRegistryEntry, + RegistryNotFoundError, + RegistryAmbiguousTargetError, +} from '../storage/repo-manager.js'; +import { + requireRegisteredStoragePath, + requireStoragePath, + STATUS_STORAGE_REQUIREMENTS, + StorageRequirementError, + isUnusableIndexInspection, +} from '../storage/storage-resolver.js'; import { getCurrentCommit, getCurrentBranch, @@ -18,7 +33,13 @@ import { resolveAnalyzerRunnerIdentity, } from '../core/analyzer-identity.js'; import { getIndexIncompleteReasons } from '../core/index-freshness.js'; +import { getFtsDisabledReason, FTS_DISABLED_MESSAGE } from '../core/search/fts-policy.js'; import { detectIndexContentDrift, type IndexContentDrift } from '../core/index-content-drift.js'; +import { + checkoutIsDirectory, + contentRetentionFromMeta, + isFullSourceAvailable, +} from '../core/content-retention.js'; import { t } from './i18n/index.js'; /** How many drifted paths the report names before summarizing the rest. */ @@ -81,11 +102,141 @@ const printDriftDetail = (drift: Extract } }; +const isExpectedUnindexedStatus = (error: StorageRequirementError): boolean => + isUnusableIndexInspection(error.inspection); + +const printNotIndexed = (repoPath: string, storagePath: string, json: boolean): void => { + if (json) { + console.log( + JSON.stringify({ + schemaVersion: 1, + repository: repoPath, + storagePath, + error: 'not-indexed', + }), + ); + } else { + console.log(t('status.repoNotIndexed')); + console.log(t('common.runAnalyzeShort')); + } +}; + export interface StatusOptions { json?: boolean; + /** Resolve a registered index without requiring its original checkout to remain on disk. */ + repo?: string; } export const statusCommand = async (options: StatusOptions = {}) => { + if (options.repo) { + let entry; + try { + entry = resolveRegistryEntry(await readRegistryStrict(), options.repo); + } catch (err) { + const error = err instanceof Error ? err.message : String(err); + if (err instanceof RegistryNotFoundError || err instanceof RegistryAmbiguousTargetError) { + if (options.json) { + console.log( + JSON.stringify({ schemaVersion: 1, repository: options.repo, error: 'not-indexed' }), + ); + } else { + console.log(error); + } + return; + } + throw err; + } + + let storagePath: string; + try { + storagePath = await requireRegisteredStoragePath(entry, STATUS_STORAGE_REQUIREMENTS); + } catch (err) { + if (!(err instanceof StorageRequirementError) || !isExpectedUnindexedStatus(err)) throw err; + if (err.inspection.state === 'owned' && !err.inspection.hasCodeIndexDB) { + const staleKuzu = await hasKuzuIndex(err.inspection.storagePath); + if (options.json) { + console.log( + JSON.stringify({ + schemaVersion: 1, + repository: entry.path, + storagePath: err.inspection.storagePath, + error: staleKuzu ? 'stale-kuzu-index' : 'not-indexed', + }), + ); + } else if (staleKuzu) { + console.log(t('status.staleKuzu')); + console.log(t('status.rebuildLadybug')); + } else { + console.log(`No usable code index at ${err.inspection.storagePath}`); + } + } else { + printNotIndexed(entry.path, err.inspection.storagePath, Boolean(options.json)); + } + return; + } + + const meta = await loadMeta(storagePath); + if (!meta) { + if (options.json) { + console.log( + JSON.stringify({ + schemaVersion: 1, + repository: entry.path, + storagePath, + error: 'not-indexed', + }), + ); + } else { + console.log(`No readable index metadata at ${storagePath}`); + } + return; + } + + const currentRunnerIdentity = resolveAnalyzerRunnerIdentity(import.meta.url); + const runnerIdentityIsCurrent = analyzerRunnerIdentitiesEqual( + meta.runnerIdentity, + currentRunnerIdentity, + ); + const incompleteReasons = getIndexIncompleteReasons(meta); + const sourceAvailable = isFullSourceAvailable( + contentRetentionFromMeta(meta), + await checkoutIsDirectory(entry.path), + ); + const payload = { + schemaVersion: 1, + repository: entry.path, + storagePath, + sourceAvailable, + index: { + indexedAt: meta.indexedAt, + commit: meta.lastCommit, + runnerIdentity: meta.runnerIdentity ?? null, + runnerIdentityStatus: runnerIdentityIsCurrent ? 'current' : 'stale-or-unknown', + incompleteReasons, + contentRetention: meta.contentRetention ?? 'full', + }, + current: sourceAvailable ? { commit: getCurrentCommit(entry.path) } : null, + // Without a checkout GitNexus can prove the index is readable, but cannot + // certify that it is current relative to source. Keep that distinction in + // the machine-readable status instead of reporting a false all-clear. + status: sourceAvailable ? 'registered' : 'source-unavailable', + }; + if (options.json) { + console.log(JSON.stringify(payload)); + } else { + console.log(`Repository: ${entry.path}`); + console.log(`Index storage: ${storagePath}`); + console.log(`Indexed: ${new Date(meta.indexedAt).toLocaleString()}`); + console.log(`Indexed commit: ${meta.lastCommit?.slice(0, 7)}`); + console.log( + sourceAvailable + ? 'Status: registered index (use status without --repo for working-tree freshness)' + : 'Status: source checkout unavailable; graph index remains queryable through the registry', + ); + } + return; + } + const cwd = process.cwd(); if (!isGitRepo(cwd)) { @@ -97,23 +248,33 @@ export const statusCommand = async (options: StatusOptions = {}) => { return; } - const repo = await findRepo(cwd); - if (!repo) { - // Check if there's a stale KuzuDB index that needs migration - const repoRoot = getGitRoot(cwd) ?? cwd; - const { storagePath } = getStoragePaths(repoRoot); - const staleKuzu = await hasKuzuIndex(storagePath); + const repoPath = getGitRoot(cwd); + if (!repoPath) { + if (options.json) { + console.log(JSON.stringify({ schemaVersion: 1, error: 'not-git-repository' })); + return; + } + console.log(t('status.notGitRepo')); + return; + } + + let storagePath: string; + try { + storagePath = await requireStoragePath(repoPath, STATUS_STORAGE_REQUIREMENTS); + } catch (err) { + if (!(err instanceof StorageRequirementError) || !isExpectedUnindexedStatus(err)) throw err; + const inspection = err.inspection; + const staleKuzu = await hasKuzuIndex(inspection.storagePath); if (options.json) { console.log( JSON.stringify({ schemaVersion: 1, - repository: repoRoot, + repository: repoPath, + storagePath: inspection.storagePath, error: staleKuzu ? 'stale-kuzu-index' : 'not-indexed', }), ); - return; - } - if (staleKuzu) { + } else if (staleKuzu) { console.log(t('status.staleKuzu')); console.log(t('status.rebuildLadybug')); } else { @@ -123,6 +284,18 @@ export const statusCommand = async (options: StatusOptions = {}) => { return; } + const meta = await loadMeta(storagePath); + if (!meta) { + printNotIndexed(repoPath, storagePath, Boolean(options.json)); + return; + } + + const repo = { + repoPath, + storagePath, + meta, + }; + const currentCommit = getCurrentCommit(repo.repoPath); const currentBranch = getCurrentBranch(repo.repoPath); @@ -134,7 +307,7 @@ export const statusCommand = async (options: StatusOptions = {}) => { let activeMeta = repo.meta; let workspaceLagsBranch = false; if (currentBranch && repo.meta.branch && currentBranch !== repo.meta.branch) { - const { metaPath } = getStoragePaths(repo.repoPath, currentBranch); + const { metaPath } = getStoragePaths(repo.repoPath, currentBranch, repo.storagePath); const branchMeta = await loadMeta(path.dirname(metaPath)); if (branchMeta) activeMeta = branchMeta; else workspaceLagsBranch = true; @@ -186,6 +359,7 @@ export const statusCommand = async (options: StatusOptions = {}) => { index: { indexedAt: activeMeta.indexedAt, commit: activeMeta.lastCommit, + ...(activeMeta.capabilities ? { capabilities: activeMeta.capabilities } : {}), runnerIdentity: activeMeta.runnerIdentity ?? null, runnerIdentityStatus: runnerIdentityIsCurrent ? 'current' : 'stale-or-unknown', incompleteReasons, @@ -211,6 +385,7 @@ export const statusCommand = async (options: StatusOptions = {}) => { console.log(`${t('status.indexed')}: ${new Date(activeMeta.indexedAt).toLocaleString()}`); console.log(`${t('status.indexedCommit')}: ${activeMeta.lastCommit?.slice(0, 7)}`); console.log(`${t('status.currentCommit')}: ${currentCommit?.slice(0, 7)}`); + if (getFtsDisabledReason(activeMeta.capabilities?.fts)) console.log(FTS_DISABLED_MESSAGE); // Emit the complete, versioned receipt as JSON so humans can inspect it and // automation can compare it without reverse-engineering a display string. // `null` is the backward-compatible signal for pre-receipt metadata. diff --git a/gitnexus/src/cli/wiki.ts b/gitnexus/src/cli/wiki.ts index 007451ef5..e9218c7e5 100644 --- a/gitnexus/src/cli/wiki.ts +++ b/gitnexus/src/cli/wiki.ts @@ -10,12 +10,13 @@ import readline from 'readline'; import { execSync, execFileSync } from 'child_process'; import cliProgress from 'cli-progress'; import { getGitRoot, isGitRepo } from '../storage/git.js'; +import { getStoragePaths, loadCLIConfig, saveCLIConfig } from '../storage/repo-manager.js'; import { - getStoragePaths, - loadMeta, - loadCLIConfig, - saveCLIConfig, -} from '../storage/repo-manager.js'; + requireStoragePath, + STATUS_STORAGE_REQUIREMENTS, + StorageRequirementError, + isUnusableIndexInspection, +} from '../storage/storage-resolver.js'; import { WikiGenerator, type WikiOptions } from '../core/wiki/generator.js'; import { MINIMAX_MODEL_IDS, @@ -183,15 +184,23 @@ const wikiCommandImpl = async (inputPath?: string, options?: WikiCommandOptions) } // ── Check for existing index ──────────────────────────────────────── - const { storagePath, lbugPath } = getStoragePaths(repoPath); - const meta = await loadMeta(storagePath); - - if (!meta) { - console.log(' Error: No GitNexus index found.'); + let storagePath: string; + try { + storagePath = await requireStoragePath(repoPath, STATUS_STORAGE_REQUIREMENTS); + } catch (error) { + if (!(error instanceof StorageRequirementError)) throw error; + const inspection = error.inspection; + if (!isUnusableIndexInspection(inspection)) { + console.log(` Error: ${error.message}\n`); + process.exitCode = 1; + return; + } + console.log(` Error: No GitNexus index found at ${error.inspection.storagePath}.`); console.log(' Run `gitnexus analyze` first to index this repository.\n'); process.exitCode = 1; return; } + const { lbugPath } = getStoragePaths(repoPath, undefined, storagePath); let timeoutSeconds: number | undefined; let retries: number | undefined; diff --git a/gitnexus/src/core/analysis-feature-registry.ts b/gitnexus/src/core/analysis-feature-registry.ts index 474c2157d..03887c29e 100644 --- a/gitnexus/src/core/analysis-feature-registry.ts +++ b/gitnexus/src/core/analysis-feature-registry.ts @@ -11,6 +11,7 @@ import { JAVA_RECORD_COMPONENT_ACCESSORS_FEATURE, SPRING_CONFIG_BINDINGS_FEATURE, } from './ingestion/languages/java/analysis-features.js'; +import { OBJECTIVE_C_PROVIDER_FEATURE } from './ingestion/languages/objective-c/analysis-features.js'; /** Production registry of independently versioned analysis capabilities. */ export const ANALYSIS_FEATURES = [ @@ -23,4 +24,5 @@ export const ANALYSIS_FEATURES = [ SPRING_CONFIG_BINDINGS_FEATURE, JAVA_ENUM_INTERFACE_HERITAGE_FEATURE, JAVA_RECORD_COMPONENT_ACCESSORS_FEATURE, + OBJECTIVE_C_PROVIDER_FEATURE, ] as const; diff --git a/gitnexus/src/core/augmentation/engine.ts b/gitnexus/src/core/augmentation/engine.ts index 9e1189ad4..23e563819 100644 --- a/gitnexus/src/core/augmentation/engine.ts +++ b/gitnexus/src/core/augmentation/engine.ts @@ -16,6 +16,13 @@ import path from 'path'; import { listRegisteredRepos } from '../../storage/repo-manager.js'; +import { + requireRegisteredStoragePath, + STATUS_STORAGE_REQUIREMENTS, +} from '../../storage/storage-resolver.js'; +import { LBUG_DIRECTORY } from '../../storage/storage-constants.js'; +import { BRANCHES_DIR, branchSlug } from '../../storage/branch-index.js'; +import { getCurrentBranch } from '../../storage/git.js'; import { escapeCypherString } from '../lbug/cypher-escape.js'; /** @@ -44,19 +51,15 @@ async function findRepoForCwd(cwd: string): Promise<{ const repoResolved = path.resolve(entry.path); const normalizedRepo = isWindows ? repoResolved.toLowerCase() : repoResolved; - // Check if cwd is inside repo OR repo is inside cwd - // Must match at a path separator boundary to avoid false positives - // (e.g. /projects/gitnexusv2 should NOT match /projects/gitnexus) + // Exact path, or cwd inside the repo at a separator boundary. + // Parent-directory invocation must not attach to a nested registered checkout. let matched = false; if (normalizedCwd === normalizedRepo) { matched = true; } else { const repoPrefix = normalizedRepo.endsWith(sep) ? normalizedRepo : normalizedRepo + sep; - const cwdPrefix = normalizedCwd.endsWith(sep) ? normalizedCwd : normalizedCwd + sep; if (normalizedCwd.startsWith(repoPrefix)) { matched = true; - } else if (normalizedRepo.startsWith(cwdPrefix)) { - matched = true; } } @@ -68,10 +71,20 @@ async function findRepoForCwd(cwd: string): Promise<{ if (!bestMatch) return null; + const storagePath = await requireRegisteredStoragePath(bestMatch, STATUS_STORAGE_REQUIREMENTS); + const branch = getCurrentBranch(bestMatch.path); + const branchIsIndexed = + Boolean(branch) && + Array.isArray(bestMatch.branches) && + bestMatch.branches.some((summary) => summary.branch === branch); + const indexDir = + branchIsIndexed && branch + ? path.join(storagePath, BRANCHES_DIR, branchSlug(branch)) + : storagePath; return { name: bestMatch.name, - storagePath: bestMatch.storagePath, - lbugPath: path.join(bestMatch.storagePath, 'lbug'), + storagePath, + lbugPath: path.join(indexDir, LBUG_DIRECTORY), }; } catch { return null; diff --git a/gitnexus/src/core/auto-sync/starter.ts b/gitnexus/src/core/auto-sync/starter.ts index 624e092ec..7d67dfa64 100644 --- a/gitnexus/src/core/auto-sync/starter.ts +++ b/gitnexus/src/core/auto-sync/starter.ts @@ -4,7 +4,7 @@ import path from 'node:path'; import { execFileSync } from 'node:child_process'; import { acquireFileLock, FileLockBusyError } from '../../storage/file-lock.js'; import { getGlobalDir } from '../../storage/repo-manager.js'; -import { isProcessAlive, readProcessStartTime } from '../../utils/process-identity.js'; +import { isProcessAlive, readProcessStartTimeCached } from '../../utils/process-identity.js'; import { loadAutoSyncConfig } from './config.js'; import { runAutoSyncOnce } from './runner.js'; import { getAutoSyncMutexPath, getAutoSyncWatchDir } from './state.js'; @@ -632,7 +632,7 @@ function resolveWatchDeps(deps: Partial = {}): AutoSyn return undefined; } }), - readProcessStartTime: deps.readProcessStartTime ?? readProcessStartTime, + readProcessStartTime: deps.readProcessStartTime ?? readProcessStartTimeCached, sleep: deps.sleep ?? ((ms) => diff --git a/gitnexus/src/core/content-retention.ts b/gitnexus/src/core/content-retention.ts new file mode 100644 index 000000000..38a81bedd --- /dev/null +++ b/gitnexus/src/core/content-retention.ts @@ -0,0 +1,92 @@ +import fs from 'node:fs/promises'; +import type { KnowledgeGraph } from './graph/types.js'; +import { + CONTENT_RETENTION_SCHEMA_VERSION, + type ContentRetention, + type FtsProfile, + type RepoMeta, +} from '../storage/repo-meta.js'; + +export const CONTENT_RETENTION_ENV = 'GITNEXUS_CONTENT_RETENTION'; + +export const contentRetentionFromEnvironment = (): ContentRetention => { + const raw = process.env[CONTENT_RETENTION_ENV]; + if (raw === undefined || raw.trim() === '') return 'full'; + const value = raw.trim(); + if (value === 'full' || value === 'symbol' || value === 'none') return value; + throw new Error( + `Invalid ${CONTENT_RETENTION_ENV} "${raw}". Expected one of: full, symbol, none.`, + ); +}; + +export const contentRetentionFromMeta = ( + meta: Pick | null | undefined, +): ContentRetention => { + const retention = meta?.contentRetention; + if (retention === undefined || retention === 'full') return 'full'; + if (retention === 'symbol' || retention === 'none') return retention; + // Legacy metadata has no field; an explicit unknown value is corrupt and must not expose text. + return 'none'; +}; + +export const ftsProfileForContentRetention = (retention: ContentRetention): FtsProfile => { + switch (retention) { + case 'symbol': + return 'symbol-no-file-content'; + case 'none': + return 'name-only'; + default: + return 'full'; + } +}; + +/** + * Legacy metadata predates retention fields and is therefore semantically full. + * It remains incrementally readable under the default profile; explicit newer + * stamps must match exactly because an FTS/layout change requires a fresh DB. + */ +export const contentRetentionMismatch = ( + meta: Pick, + requested: ContentRetention, +): boolean => { + if (meta.contentRetention === undefined) return requested !== 'full'; + return ( + meta.contentRetention !== requested || + meta.contentRetentionSchemaVersion !== CONTENT_RETENTION_SCHEMA_VERSION || + meta.ftsProfile !== ftsProfileForContentRetention(requested) + ); +}; + +/** True when `repoPath` exists and is a directory (uploads / `--allow-non-git` included). */ +export const checkoutIsDirectory = async (repoPath: string): Promise => { + try { + return (await fs.stat(repoPath)).isDirectory(); + } catch { + return false; + } +}; + +/** Full-file HTTP/MCP source is available only for `full` retention plus a live checkout. */ +export const isFullSourceAvailable = ( + retention: ContentRetention, + checkoutIsDir: boolean, +): boolean => retention === 'full' && checkoutIsDir; + +/** Remove text that the active index profile is not allowed to persist. */ +export const applyContentRetention = (graph: KnowledgeGraph, retention: ContentRetention): void => { + if (retention === 'full') return; + + graph.forEachNode((node) => { + if (retention === 'symbol') { + if (node.label === 'File') delete node.properties.content; + // BasicBlocks hold statement source, not a symbol snippet. + if (node.label === 'BasicBlock') delete node.properties.text; + return; + } + + delete node.properties.content; + + delete node.properties.description; + if (node.label === 'BasicBlock') delete node.properties.text; + }); +}; diff --git a/gitnexus/src/core/embeddings/ast-utils.ts b/gitnexus/src/core/embeddings/ast-utils.ts index 194f47313..b29ca80fb 100644 --- a/gitnexus/src/core/embeddings/ast-utils.ts +++ b/gitnexus/src/core/embeddings/ast-utils.ts @@ -4,14 +4,13 @@ * used by both chunker.ts and structural-extractor.ts. */ -import { getLanguageFromFilename } from 'gitnexus-shared'; import { createParserForLanguage, isLanguageAvailable, resolveLanguageKey, } from '../tree-sitter/parser-loader.js'; import { parseSourceSafe } from '../tree-sitter/safe-parse.js'; -import { getProvider } from '../ingestion/languages/index.js'; +import { getLanguageForFileContent, getProvider } from '../ingestion/languages/index.js'; const parserCache = new Map(); @@ -20,7 +19,10 @@ const parserCache = new Map(); * Returns null if language is unavailable or parsing fails. */ export const ensureAndParse = async (content: string, filePath: string): Promise => { - const language = getLanguageFromFilename(filePath); + // Same classifier as ingest. Filename-only maps `.h` → C++, so Objective-C + // headers (and method snippets from those headers) would parse with the + // wrong grammar and miss class_interface / protocol_declaration / methods. + const language = getLanguageForFileContent(filePath, content); if (!language) return null; if (!isLanguageAvailable(language)) return null; @@ -100,6 +102,9 @@ export const findDeclarationNode = (root: any): any | null => { 'struct_item', 'interface_declaration', 'interface_definition', + 'protocol_declaration', // Objective-C protocol + 'class_interface', // Objective-C class, category, or extension + 'class_implementation', // Objective-C implementation 'enum_declaration', 'enum_item', 'type_declaration', // Go: type X struct diff --git a/gitnexus/src/core/embeddings/chunker.ts b/gitnexus/src/core/embeddings/chunker.ts index 073abfb9c..4a06d7eda 100644 --- a/gitnexus/src/core/embeddings/chunker.ts +++ b/gitnexus/src/core/embeddings/chunker.ts @@ -145,9 +145,35 @@ const DECLARATION_BODY_NODE_TYPES = new Set([ 'class_body', 'object_type', 'declaration_list', + 'implementation_definition', 'interface_body', ]); +const DIRECT_MEMBER_DECLARATION_TYPES = new Set([ + 'protocol_declaration', + 'class_interface', + // tree-sitter-objc exposes each implementation_definition directly under + // class_implementation, rather than grouping them in a shared body node. + 'class_implementation', +]); + +const DIRECT_MEMBER_HEADER_NODE_TYPES = new Set([ + 'identifier', + 'parameterized_arguments', + 'protocol_reference_list', + // tree-sitter-objc `_type_params` hides itself and exposes this named child + // for `(T)` generics and category-shaped argument lists. + 'generic_arguments', + // `_class_interface_header` / `_class_implementation_header` expose these + // named children before members (`NS_ROOT_CLASS @interface …`). + 'attribute_declaration', + 'attribute_specifier', + 'availability_attribute_specifier', + 'ms_declspec_modifier', + 'storage_class_specifier', + 'type_qualifier', +]); + const FIELD_LIKE_MEMBER_TYPES = new Set([ 'field_definition', 'public_field_definition', @@ -157,8 +183,18 @@ const FIELD_LIKE_MEMBER_TYPES = new Set([ 'lexical_declaration', 'pair', 'enum_assignment', + // tree-sitter-objc wraps each ivar as `instance_variable`, not field_definition. + 'instance_variable', ]); +const DECLARATION_MEMBER_WRAPPER_TYPES = new Set([ + 'qualified_protocol_interface_declaration', + 'instance_variables', +]); + +/** Named prefixes the ObjC grammar allows immediately before each ivar. */ +const IVAR_ATTRIBUTE_PREFIX_TYPES = new Set(['attribute_specifier', 'attribute_declaration']); + const declarationChunk = async ( content: string, filePath: string, @@ -337,6 +373,8 @@ const getDeclarationBodyNode = (node: any): any | null => { const bodyNode = node.childForFieldName?.('body'); if (bodyNode) return bodyNode; + if (DIRECT_MEMBER_DECLARATION_TYPES.has(node.type)) return node; + for (let i = 0; i < node.namedChildCount; i++) { const child = node.namedChild(i); if (!child) continue; @@ -352,15 +390,43 @@ const collectDeclarationUnits = ( ): Array<{ startIndex: number; endIndex: number }> => { const members: Array<{ startIndex: number; endIndex: number; groupable: boolean }> = []; - for (let i = 0; i < bodyNode.namedChildCount; i++) { - const child = bodyNode.namedChild(i); - if (!child) continue; - members.push({ - startIndex: child.startIndex, - endIndex: child.endIndex, - groupable: groupFields && FIELD_LIKE_MEMBER_TYPES.has(child.type), - }); - } + const collectMembers = ( + node: any, + skipHeaderChildren: boolean, + includeNodePrefixOnFirstMember = false, + ): void => { + const firstMemberIndex = members.length; + let ivarAttributePrefixStart: number | undefined; + for (let i = 0; i < node.namedChildCount; i++) { + const child = node.namedChild(i); + if (!child) continue; + if (DECLARATION_MEMBER_WRAPPER_TYPES.has(child.type)) { + collectMembers(child, false, true); + continue; + } + if (skipHeaderChildren && DIRECT_MEMBER_HEADER_NODE_TYPES.has(child.type)) continue; + if (node.type === 'instance_variables' && IVAR_ATTRIBUTE_PREFIX_TYPES.has(child.type)) { + ivarAttributePrefixStart ??= child.startIndex; + continue; + } + members.push({ + startIndex: ivarAttributePrefixStart ?? child.startIndex, + endIndex: child.endIndex, + groupable: groupFields && FIELD_LIKE_MEMBER_TYPES.has(child.type), + }); + ivarAttributePrefixStart = undefined; + } + + const firstMember = members[firstMemberIndex]; + if (includeNodePrefixOnFirstMember && firstMember) { + firstMember.startIndex = node.startIndex; + if (node.type === 'instance_variables') { + members[members.length - 1].endIndex = node.endIndex; + } + } + }; + + collectMembers(bodyNode, DIRECT_MEMBER_DECLARATION_TYPES.has(bodyNode.type)); if (members.length === 0) return []; diff --git a/gitnexus/src/core/embeddings/embedding-pipeline.ts b/gitnexus/src/core/embeddings/embedding-pipeline.ts index f3cda5f23..605bf8457 100644 --- a/gitnexus/src/core/embeddings/embedding-pipeline.ts +++ b/gitnexus/src/core/embeddings/embedding-pipeline.ts @@ -81,7 +81,7 @@ const ensureVectorExtensionAvailable = async (): Promise => { * invalidate existing vectors, such as metadata/header shape changes, * structural container context changes, or preceding-context formatting rules. */ -export const EMBEDDING_TEXT_VERSION = 'v4'; +export const EMBEDDING_TEXT_VERSION = 'v5'; /** * Compute a stable content fingerprint for an embeddable node. diff --git a/gitnexus/src/core/embeddings/http-client.ts b/gitnexus/src/core/embeddings/http-client.ts index 7919b2376..88e22835a 100644 --- a/gitnexus/src/core/embeddings/http-client.ts +++ b/gitnexus/src/core/embeddings/http-client.ts @@ -12,6 +12,7 @@ */ import { chunk } from '../../lib/utils.js'; +import { parseTruthyEnv } from '../ingestion/utils/env.js'; import { CircuitOpenError, ResilientFetchExhaustedError, @@ -39,6 +40,7 @@ interface HttpConfig { retryCapMs: number; minIntervalMs: number; timeoutMs: number; + retryTimeouts: boolean; requestDimensions?: number; } @@ -202,6 +204,12 @@ const readConfig = (): HttpConfig | null => { DEFAULT_HTTP_TIMEOUT_MS, MAX_HTTP_TIMEOUT_MS, ), + // A boolean toggle, so it takes the repo's truthy convention (`1`/`true`/ + // `yes`) and falls back to the documented default on anything else. The + // integer parser this used throws on a non-digit, which turned the + // conventional `=true` into a hard failure of every embedding call rather + // than either enabling the flag or leaving it off. + retryTimeouts: parseTruthyEnv(process.env.GITNEXUS_EMBEDDING_RETRY_TIMEOUTS), requestDimensions, }; }; @@ -338,6 +346,36 @@ class RetryableEmbeddingBodyError extends Error { } } +class RetryableEmbeddingTimeoutError extends Error { + constructor( + readonly timeoutMs: number, + options?: { cause?: unknown }, + ) { + super( + `Embedding request timed out after ${timeoutMs}ms`, + options?.cause !== undefined ? { cause: options.cause } : undefined, + ); + this.name = 'RetryableEmbeddingTimeoutError'; + } +} + +/** Re-wrap an opt-in TimeoutError so `resilientFetch` retries it. Abort stays terminal. */ +const throwIfRetryableTimeout = ( + err: unknown, + retryTimeouts: boolean, + callerAborted: boolean | undefined, + timeoutMs: number, +): void => { + if ( + retryTimeouts && + !callerAborted && + isTerminalNetworkError(err) && + err.name === 'TimeoutError' + ) { + throw new RetryableEmbeddingTimeoutError(timeoutMs, { cause: err }); + } +}; + /** * Build the message for a 2xx body carrying the wrong number of vectors. * @@ -384,6 +422,7 @@ const httpEmbedBatch = async ( retryCapMs = HTTP_RETRY_CAP_MS, minIntervalMs = 0, timeoutMs = DEFAULT_HTTP_TIMEOUT_MS, + retryTimeouts = false, ): Promise => { const requestBody: { input: string[]; model: string; dimensions?: number } = { input: batch, @@ -428,7 +467,13 @@ const httpEmbedBatch = async ( const signal = requestOptions.signal ? AbortSignal.any([requestOptions.signal, timeoutSignal]) : timeoutSignal; - const attemptResp = await globalThis.fetch(input, { ...init, signal }); + let attemptResp: Response; + try { + attemptResp = await globalThis.fetch(input, { ...init, signal }); + } catch (err) { + throwIfRetryableTimeout(err, retryTimeouts, requestOptions.signal?.aborted, timeoutMs); + throw err; + } // Non-OK bodies are none of our business: hand the response straight // back so `resilientFetch` keeps classifying 4xx/5xx/429 unchanged. if (!attemptResp.ok) return attemptResp; @@ -447,15 +492,19 @@ const httpEmbedBatch = async ( // Not every `.json()` rejection is a parse error: the per-attempt // signal (`AbortSignal.any([caller, AbortSignal.timeout(...)])`) is // wired to the body stream, so a stalled body rejects with the abort - // reason. Re-raise those untouched — `isTerminalNetworkError` is - // `resilientFetch`'s own predicate, so this test agrees with - // `classifyOutcome` by construction. Wrapping one would flip its - // verdict from `terminal-network` (returned without retry AND - // without touching the breaker, via `recordNeutral()`) to - // `retryable-network` (retried, then `breaker.recordFailure()`): the - // same timeout would take 3 attempts instead of 1, count toward the - // process-global `embeddings-http` breaker, and reach the operator as - // "unparseable response" so they never reach for the timeout knob. + // reason. Re-raise AbortError (and TimeoutError when retry is off) + // untouched — `isTerminalNetworkError` is `resilientFetch`'s own + // predicate, so this test agrees with `classifyOutcome` by + // construction. Wrapping one would flip its verdict from + // `terminal-network` (returned without retry AND without touching + // the breaker, via `recordNeutral()`) to `retryable-network` + // (retried, then `breaker.recordFailure()`): the same timeout would + // take 3 attempts instead of 1, count toward the process-global + // `embeddings-http` breaker, and reach the operator as "unparseable + // response" so they never reach for the timeout knob. + // Opt-in `GITNEXUS_EMBEDDING_RETRY_TIMEOUTS=1` is the exception: + // TimeoutError is re-wrapped so the existing retry loop can retry it. + throwIfRetryableTimeout(err, retryTimeouts, requestOptions.signal?.aborted, timeoutMs); if (isTerminalNetworkError(err)) throw err; throw new RetryableEmbeddingBodyError(unparseableMessage(), { cause: err }); } @@ -503,6 +552,12 @@ const httpEmbedBatch = async ( if (err instanceof RetryableEmbeddingBodyError) { throw new HttpEmbeddingError(err.terminalMessage, { cause: err.cause }); } + if (err instanceof RetryableEmbeddingTimeoutError) { + throw new HttpEmbeddingError( + `${err.message} after ${maxAttempts} attempt(s) (${safeUrl(url)}, batch ${batchIndex})`, + { cause: err.cause }, + ); + } if (err instanceof CircuitOpenError) { throw new HttpEmbeddingError( `Embedding endpoint circuit open (${safeUrl(url)}, batch ${batchIndex}): retry in ${Math.ceil(err.retryAfterMs / 1000)}s`, @@ -580,6 +635,7 @@ export const httpEmbed = async ( config.retryCapMs, config.minIntervalMs, config.timeoutMs, + config.retryTimeouts, ); // Defensive backstop, deliberately kept: `httpEmbedBatch` now rejects a @@ -655,6 +711,7 @@ export const httpEmbedQuery = async ( config.retryCapMs, config.minIntervalMs, config.timeoutMs, + config.retryTimeouts, ); // Defensive backstop like the `httpEmbed` one above: an empty `data` array is // now a cardinality mismatch (0 vectors for 1 text) rejected and retried diff --git a/gitnexus/src/core/embeddings/structural-extractor.ts b/gitnexus/src/core/embeddings/structural-extractor.ts index 7035ee569..8c003bbbd 100644 --- a/gitnexus/src/core/embeddings/structural-extractor.ts +++ b/gitnexus/src/core/embeddings/structural-extractor.ts @@ -5,7 +5,7 @@ * to extract method and field names for embedding text generation. */ -import { getProviderForFile } from '../ingestion/languages/index.js'; +import { getProviderForFileContent } from '../ingestion/languages/index.js'; import type { MethodExtractorContext, ExtractedMethods } from '../ingestion/method-types.js'; import type { FieldExtractorContext, ExtractedFields } from '../ingestion/field-types.js'; import type { LanguageProvider } from '../ingestion/language-provider.js'; @@ -32,7 +32,7 @@ export const extractStructuralNames = async ( content: string, filePath: string, ): Promise => { - const provider = getProviderForFile(filePath); + const provider = getProviderForFileContent(filePath, content); if (!provider) return { methodNames: [], fieldNames: [] }; const tree = await ensureAndParse(content, filePath); diff --git a/gitnexus/src/core/embeddings/types.ts b/gitnexus/src/core/embeddings/types.ts index 71cad5d65..759e19812 100644 --- a/gitnexus/src/core/embeddings/types.ts +++ b/gitnexus/src/core/embeddings/types.ts @@ -8,6 +8,8 @@ export const LABEL_FUNCTION = 'Function' as const; export const LABEL_METHOD = 'Method' as const; export const LABEL_CONSTRUCTOR = 'Constructor' as const; export const LABEL_CLASS = 'Class' as const; +export const LABEL_PROTOCOL = 'Protocol' as const; +export const LABEL_CATEGORY = 'Category' as const; export const LABEL_INTERFACE = 'Interface' as const; export const LABEL_STRUCT = 'Struct' as const; export const LABEL_ENUM = 'Enum' as const; @@ -53,6 +55,8 @@ export const CHUNKABLE_LABELS = [ LABEL_METHOD, LABEL_CONSTRUCTOR, LABEL_CLASS, + LABEL_PROTOCOL, + LABEL_CATEGORY, LABEL_INTERFACE, LABEL_STRUCT, LABEL_ENUM, @@ -165,6 +169,20 @@ export const CHUNKING_RULES: Readonly