diff --git a/.claude/skills/gitnexus-cli/SKILL.md b/.claude/skills/gitnexus-cli/SKILL.md index 342e8b08f..853d44860 100644 --- a/.claude/skills/gitnexus-cli/SKILL.md +++ b/.claude/skills/gitnexus-cli/SKILL.md @@ -60,7 +60,7 @@ Generates repository documentation from the knowledge graph using an LLM. Requir | Flag | Effect | | ------------------- | ----------------------------------------- | | `--force` | Force full regeneration | -| `--model ` | LLM model (default: minimax/minimax-m2.5) | +| `--model ` | LLM model (default: MiniMax-M3) | | `--base-url ` | LLM API base URL | | `--api-key ` | LLM API key | | `--concurrency ` | Parallel LLM calls (default: 3) | diff --git a/.claude/skills/gitnexus-impact-analysis/SKILL.md b/.claude/skills/gitnexus-impact-analysis/SKILL.md index 0b81795de..ee1cd3496 100644 --- a/.claude/skills/gitnexus-impact-analysis/SKILL.md +++ b/.claude/skills/gitnexus-impact-analysis/SKILL.md @@ -53,6 +53,14 @@ description: "Use when the user wants to know what will break if they change som | 5-15 symbols, 2-5 processes | MEDIUM | | >15 symbols or many processes | HIGH | | Critical path (auth, payments) | CRITICAL | +| **Zero callers found** | **UNKNOWN** | + +`UNKNOWN` is not a low rung on this scale — it means the walk could not answer. +An empty caller set is equally consistent with "genuinely unused" and "the +callers are not resolvable by the index" (plain-object property access, dynamic +dispatch, cross-language calls), so few-callers ⇒ LOW does **not** apply. The +result carries a `riskNote` saying so. Confirm with a text search before +treating the symbol as safe to change or delete. ## Tools @@ -84,6 +92,11 @@ detect_changes({scope: "all"}) → Risk: MEDIUM ``` +`partial: true` (a graph query failed) or `truncated: true` (the changed-symbol +listing was capped) means the result is short of the truth, and reads like +`UNKNOWN` above: a zero there means unseen, not unaffected. Re-run it rather +than tick the pre-commit check. + ## Example: "What breaks if I change validateUser?" ``` diff --git a/.claude/skills/gitnexus-plan/README.md b/.claude/skills/gitnexus-plan/README.md index f7fe58ab9..153374bb7 100644 --- a/.claude/skills/gitnexus-plan/README.md +++ b/.claude/skills/gitnexus-plan/README.md @@ -124,12 +124,17 @@ phase that needs them. statement-level claims (never reconstructs fake edges). - No GitNexus at all → fallback mode: targeted grep/read exploration, findings labelled **source-derived**, with a recommendation to index. -- Reading or publishing a plan requires Linux `/proc/self/fd`, `O_DIRECTORY`, - and `O_NOFOLLOW`; publication also requires a validated absolute Python 3 - PATH candidate with libc `renameat2(RENAME_NOREPLACE)` support, a - writable target repository, and a shared filesystem for the plan and - Git-admin vault. The writer fails closed when those guarantees are - unavailable; it never redirects the plan elsewhere. +- Reading or publishing a plan requires `O_DIRECTORY` and `O_NOFOLLOW`, plus + `/proc/self/fd` on Linux; every other platform is refused. No interpreter is + spawned and no native code is loaded. Publication is `link(2)`, which fails + rather than replaces when the destination name is taken. Linux resolves every + name against a held descriptor, so a parent swapped mid-write cannot redirect + the operation; macOS has no equivalent path and instead pins each directory + with an open descriptor and re-proves the chain either side of every step, + which detects such a swap and aborts. Publishing also needs a writable target + repository and a shared filesystem for the plan and Git-admin vault. The + writer fails closed when those guarantees are unavailable; it never redirects + the plan elsewhere. ## Limitations diff --git a/.claude/skills/gitnexus-plan/references/evidence-provenance.md b/.claude/skills/gitnexus-plan/references/evidence-provenance.md index c686599da..3df5a046d 100644 --- a/.claude/skills/gitnexus-plan/references/evidence-provenance.md +++ b/.claude/skills/gitnexus-plan/references/evidence-provenance.md @@ -98,8 +98,11 @@ excluded. ## Safe existing-plan read contract -`read-plan` fails closed unless Linux `/proc/self/fd`, `O_DIRECTORY`, and -`O_NOFOLLOW` are available. It resolves the exact Git top-level, opens the +`read-plan` fails closed unless the host platform can resolve names against a +held directory descriptor: Linux `/proc/self/fd` with `O_DIRECTORY` and +`O_NOFOLLOW`, or macOS `O_DIRECTORY`/`O_NOFOLLOW`. Every other platform is +refused outright — an unverified read is not a degraded read, it is a different, +racy operation. It resolves the exact Git top-level, opens the repository root and every plan parent as held no-follow directory descriptors, rejects missing, symlink, non-directory, and escaping parents, and opens the leaf with `O_NOFOLLOW`. It reads at most 16 MiB from that held file descriptor, @@ -109,13 +112,17 @@ Neither Deepen nor work may parse bytes obtained before or outside this receipt. ## Safe generated-plan write contract -The writer fails closed unless Linux `/proc/self/fd`, `O_DIRECTORY`, -`O_NOFOLLOW`, and Python 3 with libc `renameat2(RENAME_NOREPLACE)` support are -available. Python may live in `/usr/local`, a Nix profile, or another absolute -PATH directory, but the helper accepts only a resolved executable and -containing directory owned by root or the current user and not writable by -group/other. The resolved executable is opened without following links and -invoked through that held descriptor. Relative PATH entries are ignored. The plan parent and the +The writer fails closed unless the host platform offers `O_DIRECTORY` and +`O_NOFOLLOW`, plus `/proc/self/fd` on Linux. It spawns no interpreter and loads +no native code: publication is `link(2)`, which is atomic, fails `EEXIST` when +the destination name is taken, and refuses a symlinked destination without +following it — the same no-replace guarantee `renameat2(RENAME_NOREPLACE)` and +`renameatx_np(RENAME_EXCL)` provide, available through `fs.linkSync` on every +supported platform. The temporary name is unlinked once the link succeeds; the +published file is the same inode the writer created and verified, so every +identity check downstream holds by construction. A link that succeeds followed +by an unlink that fails leaves the plan published and is reported as success, +because it is one. The plan parent and the repository's Git-admin directory must also share a filesystem. It resolves the target repository's exact Git top-level, opens that root and every destination parent as held no-follow directory descriptors, creates missing @@ -128,15 +135,45 @@ The writer creates a random exclusive temporary file relative to the held final parent descriptor and keeps its no-follow descriptor open. It writes and flushes the bytes, binds the temporary name to the opened inode, and hashes the open file before publication. Immediately before publication it revalidates -the parent and the temporary path, inode, size, and digest. Publication uses an -atomic no-replace move relative to the held directory descriptor. Initial mode -therefore cannot overwrite a destination that appears after the absent check. +the parent and the temporary path, inode, size, and digest. Publication links +the temporary name to the destination relative to the held directory +descriptor, which fails rather than replaces if the destination is taken. +Initial mode therefore cannot overwrite a destination that appears after the +absent check. The writer then flushes the directory and revalidates the committed path by opening it with `O_NOFOLLOW`, hashing both the original temporary fd and the path-bound fd, and performing a second descriptor-anchored path identity check after hashing. A detected mutation or replacement aborts instead of accepting mixed-era output. +### Linux anchors, macOS verifies + +The two platforms reach the same destination by different proofs, and the +difference is real enough to state rather than smooth over. + +On Linux every name resolves through `/proc/self/fd//`, a magic link +the kernel resolves against the inode the descriptor already holds. The names +above it are never re-walked, so an attacker who renames a parent between the +check and the use cannot redirect the operation. The race is impossible, not +merely detected. + +macOS has no such path. `/dev/fd/` is a devfs node, not a magic link: it can +be opened, but nothing can be resolved through it. `open("/dev/fd//child")` +returns `ENOENT`, and `realpath` of it returns `/dev/fd/` rather than the +directory's path — measured on macOS 26, not inferred. Node exposes no `openat`, +no `dir_fd` parameter, and no FFI, so on macOS the writer resolves names +lexically with `O_NOFOLLOW` at every component, holds an open descriptor on +every directory in the chain for the whole operation, and proves before *and* +after each step that the chain still names exactly the inodes it is holding. +Holding the descriptors is what makes the recorded inode numbers trustworthy: +an open descriptor pins its inode, so a freed number cannot be recycled beneath +the walk. + +What that buys is detection rather than prevention. A parent swapped inside the +window between a check and its use is caught by the check that follows, and the +operation aborts having written nothing — but on Linux it could not have +happened at all. No published byte escapes verification on either platform. + `--replace` accepts only a pre-existing regular file and is reserved for Deepen; without it, accidental overwrite is rejected. It also requires the exact canonical `generated_plan_path` and `plan_digest` from the same session's diff --git a/.claude/skills/gitnexus-plan/scripts/evidence-provenance.mjs b/.claude/skills/gitnexus-plan/scripts/evidence-provenance.mjs index 181d2120b..793fe4cd8 100644 --- a/.claude/skills/gitnexus-plan/scripts/evidence-provenance.mjs +++ b/.claude/skills/gitnexus-plan/scripts/evidence-provenance.mjs @@ -479,11 +479,11 @@ function resolveOwnGitTopLevel(absolute) { if (result.status !== 0) return null; let topLevel; try { - topLevel = fs.realpathSync(decodeUtf8(result.stdout, 'nested repository root').trim()); + topLevel = fs.realpathSync.native(decodeUtf8(result.stdout, 'nested repository root').trim()); } catch { return null; } - return topLevel === fs.realpathSync(absolute) ? topLevel : null; + return topLevel === fs.realpathSync.native(absolute) ? topLevel : null; } function readOwnGitlinkHead(absolute) { @@ -616,17 +616,30 @@ function filesystemObject(absolute, expectedKind, mutationGuards, testHooks) { throw new Error(`Unsupported filesystem object at ${absolute}`); } -function guardPathParents(repo, repoPath, mutationGuards) { +// Every dirty path re-walks its own parents, and dirty paths overwhelmingly +// share them — the repository root is re-stat'ed once per path. `guarded` is +// per-snapshot and remembers which absolute directories already carry a guard, +// so each distinct directory is stat'ed and guarded exactly once. +// +// Keeping the first-seen identity is the conservative choice: verifyGuards +// re-checks every guard against the filesystem at the end, so a directory that +// changes after it was guarded still fails there. Skipping a re-stat cannot hide +// a change; it only avoids recording the same directory twice. +function guardPathParents(repo, repoPath, mutationGuards, guarded) { const components = repoPath.split('/'); let current = repo; - const rootStat = fs.lstatSync(repo, { bigint: true }); - mutationGuards.push({ - type: 'directory', - absolute: repo, - identity: stableDirectoryIdentity(rootStat), - }); + if (!guarded.has(repo)) { + guarded.add(repo); + mutationGuards.push({ + type: 'directory', + absolute: repo, + identity: stableDirectoryIdentity(fs.lstatSync(repo, { bigint: true })), + }); + } for (const component of components.slice(0, -1)) { current = path.join(current, component); + // Already proved a real directory and already guarded on an earlier path. + if (guarded.has(current)) continue; let stat; try { stat = fs.lstatSync(current, { bigint: true }); @@ -638,6 +651,7 @@ function guardPathParents(repo, repoPath, mutationGuards) { throw new Error(`Refusing to traverse symlink parent for ${repoPath}`); } if (!stat.isDirectory()) return; + guarded.add(current); mutationGuards.push({ type: 'directory', absolute: current, @@ -646,81 +660,153 @@ function guardPathParents(repo, repoPath, mutationGuards) { } } -function recordAnchoredAbsence(repo, repoPath, mutationGuards) { - requireDescriptorAnchoring(); - const flags = - fs.constants.O_RDONLY | - fs.constants.O_DIRECTORY | - fs.constants.O_NOFOLLOW | - (fs.constants.O_CLOEXEC ?? 0); - const descriptors = []; - let retainedFd; - try { - let currentFd = fs.openSync(repo, flags); - descriptors.push(currentFd); - const components = repoPath.split('/'); - for (let index = 0; index < components.length; index += 1) { - const component = components[index]; - const child = descriptorPath(currentFd, component); - let childStat; - try { - childStat = fs.lstatSync(child, { bigint: true }); - } catch (error) { - if (error?.code !== 'ENOENT' && error?.code !== 'ENOTDIR') throw error; - const parentStat = fs.fstatSync(currentFd, { bigint: true }); - if (!parentStat.isDirectory()) { - throw new Error(`Absence parent is no longer a directory for ${repoPath}`); - } - retainedFd = currentFd; - mutationGuards.push({ - type: 'absence', - fd: retainedFd, - childName: component, - repoPath, - parentIdentity: stableDirectoryIdentity(parentStat), - parentMutationIdentity: statIdentity(parentStat), - }); - for (const fd of descriptors) { - if (fd !== retainedFd) fs.closeSync(fd); - } - return; - } - if (index === components.length - 1) { - throw new Error(`${repoPath} appeared while its absence was being anchored`); - } - if (childStat.isSymbolicLink() || !childStat.isDirectory()) { - throw new Error(`Refusing a non-directory parent while anchoring absence for ${repoPath}`); - } - const nextFd = fs.openSync(child, flags); - descriptors.push(nextFd); - currentFd = nextFd; - } - throw new Error(`Could not anchor absence for ${repoPath}`); - } catch (error) { - for (const fd of descriptors) { - if (fd === retainedFd) continue; - try { - fs.closeSync(fd); - } catch { - // Preserve the primary absence-anchoring error. - } - } - throw error; +// A bound, not a bug: the absence cache deduplicates correctly and leaks nothing, +// but citedPaths is caller-supplied and unbounded, so a pathological snapshot +// could hold more descriptors than the process is allowed (macOS +// kern.maxfilesperproc is 24576). The peak precedes a `git` spawn, so exhaustion +// would surface as a git failure misreported as evidence instability. +// +// Refuse rather than evict: closing a cached descriptor would silently break the +// pinned chain of an absence guard that was already recorded against it, which is +// exactly the inode-recycling hole the pins exist to close. +const ABSENCE_ANCHOR_LIMITS = Object.freeze({ maxPinnedDirectories: 4096 }); + +// Every no-follow read and every exclusive create in this file uses one of these +// two, so a change lands in one place rather than in seven. +const VERIFIED_READ_FLAGS = + fs.constants.O_RDONLY | fs.constants.O_NOFOLLOW | (fs.constants.O_CLOEXEC ?? 0); +const VERIFIED_CREATE_FLAGS = + fs.constants.O_RDWR | + fs.constants.O_CREAT | + fs.constants.O_EXCL | + fs.constants.O_NOFOLLOW | + (fs.constants.O_CLOEXEC ?? 0); + +function requireAbsenceAnchorCapacity(cache) { + if (cache.size >= ABSENCE_ANCHOR_LIMITS.maxPinnedDirectories) { + throw new Error( + `Absence anchoring exceeds ${ABSENCE_ANCHOR_LIMITS.maxPinnedDirectories} pinned directories`, + ); } } -function materializeRecord(repo, statusRecord, layers, mutationGuards, testHooks) { +const ANCHORED_DIRECTORY_FLAGS = + fs.constants.O_RDONLY | + fs.constants.O_DIRECTORY | + fs.constants.O_NOFOLLOW | + (fs.constants.O_CLOEXEC ?? 0); + +// Every absence receipt is verified long after its walk returns, so the chain +// that produced it has to stay pinned until the snapshot ends — an unpinned inode +// number can be recycled by a replacement directory that then reproduces the +// recorded identity exactly. Absent cited paths overwhelmingly share prefixes, so +// the walked directories are cached per snapshot and keyed by repo-relative +// prefix: one open descriptor and one anchored walk per distinct directory rather +// than per path. snapshotEvidence owns every descriptor in this cache and closes +// each exactly once; guards only borrow them for verification. +function anchoredAbsenceRoot(repo, cache) { + const cached = cache.get(''); + if (cached) return cached; + requireAbsenceAnchorCapacity(cache); + const fd = openVerifiedDirectory(repo, ANCHORED_DIRECTORY_FLAGS); + const handle = { + fd, + expectedPath: repo, + chain: [ + { expectedPath: repo, identity: stableDirectoryIdentity(fs.fstatSync(fd, { bigint: true })) }, + ], + descriptors: [fd], + }; + cache.set('', handle); + return handle; +} + +function recordAnchoredAbsence(repo, repoPath, mutationGuards, cache) { + requireDescriptorAnchoring(); + const components = repoPath.split('/'); + let handle = anchoredAbsenceRoot(repo, cache); + let prefix = ''; + for (let index = 0; index < components.length; index += 1) { + const component = components[index]; + const isFinal = index === components.length - 1; + prefix = prefix === '' ? component : `${prefix}/${component}`; + // The final component is always re-checked against the filesystem: it is the + // one whose absence is being recorded, and a cached answer would be a stale + // one. Only the prefix directories are reused. + const cached = isFinal ? undefined : cache.get(prefix); + if (cached) { + handle = cached; + continue; + } + const child = anchoredChild(handle, component); + let childStat; + try { + childStat = lstatChild(child); + } catch (error) { + if (error?.code !== 'ENOENT' && error?.code !== 'ENOTDIR') throw error; + const parentStat = fs.fstatSync(handle.fd, { bigint: true }); + if (!parentStat.isDirectory()) { + throw new Error(`Absence parent is no longer a directory for ${repoPath}`); + } + mutationGuards.push({ + type: 'absence', + // The handle is the holder the guard verifies against, and `ref` is the + // child path already built through the anchoredChild chokepoint — the + // guard must never re-derive that name itself. + handle, + ref: child, + fd: handle.fd, + repoPath, + parentMutationIdentity: statIdentity(parentStat), + }); + return; + } + if (isFinal) { + throw new Error(`${repoPath} appeared while its absence was being anchored`); + } + if (childStat.isSymbolicLink() || !childStat.isDirectory()) { + throw new Error(`Refusing a non-directory parent while anchoring absence for ${repoPath}`); + } + requireAbsenceAnchorCapacity(cache); + const childFd = openVerifiedDirectory(child.path, ANCHORED_DIRECTORY_FLAGS); + const expectedPath = path.join(handle.expectedPath, component); + let next; + try { + if (!anchoringBackend().descriptorMatchesChild(childFd, expectedPath, childStat)) { + throw new Error( + `Absence parent descriptor does not match its verified inode for ${repoPath}`, + ); + } + next = { + fd: childFd, + expectedPath, + chain: [...handle.chain, { expectedPath, identity: stableDirectoryIdentity(childStat) }], + descriptors: [...handle.descriptors, childFd], + }; + } catch (error) { + fs.closeSync(childFd); + throw error; + } + cache.set(prefix, next); + handle = next; + } + throw new Error(`Could not anchor absence for ${repoPath}`); +} + +function materializeRecord(repo, statusRecord, layers, mutationGuards, testHooks, walkState) { const head = layers.head(statusRecord.path); const index = layers.index(statusRecord.path); const expectedKind = index.kind === 'gitlink' || head.kind === 'gitlink' ? 'gitlink' : null; - guardPathParents(repo, statusRecord.path, mutationGuards); + guardPathParents(repo, statusRecord.path, mutationGuards, walkState.guardedDirectories); const filesystem = filesystemObject( path.join(repo, ...statusRecord.path.split('/')), expectedKind, mutationGuards, testHooks, ); - if (filesystem.kind === ABSENT) recordAnchoredAbsence(repo, statusRecord.path, mutationGuards); + if (filesystem.kind === ABSENT) { + recordAnchoredAbsence(repo, statusRecord.path, mutationGuards, walkState.absenceCache); + } if (statusRecord.directory_hint && filesystem.kind !== 'directory') { throw new Error( `Git reported an embedded directory but found ${filesystem.kind}: ${statusRecord.path}`, @@ -789,9 +875,15 @@ export function serializeDirtyRecords(entries) { } function assertRepository(repoInput) { - const repo = fs.realpathSync(requireString(repoInput, 'repo')); + // realpathSync.native, not realpathSync: the JS resolver preserves a Windows + // 8.3 short component (C:\Users\RUNNER~1\...) while git always reports the long + // form, so the two would never compare equal and every caller would be told the + // worktree root is not the worktree root it just named. + const repo = fs.realpathSync.native(requireString(repoInput, 'repo')); const topLevelResult = git(repo, ['rev-parse', '--show-toplevel']); - const topLevel = fs.realpathSync(decodeUtf8(topLevelResult.stdout, 'repository root').trim()); + const topLevel = fs.realpathSync.native( + decodeUtf8(topLevelResult.stdout, 'repository root').trim(), + ); if (topLevel !== repo) throw new Error(`--repo must be the Git worktree root (${topLevel})`); return repo; } @@ -882,17 +974,48 @@ function stableFileIdentity(stat) { return [stat.dev, stat.ino, stat.mode, stat.size].map(String).join(':'); } +// The two backends below differ in one decisive way, and it is worth stating +// plainly because the security properties are not the same. +// +// Linux ANCHORS. A name is resolved through /proc/self/fd//, which +// starts the walk at the inode the descriptor holds, so a parent that is renamed +// away cannot be traversed at all: the descriptor keeps pointing at the original +// directory and the impostor planted at the same name is simply never reached. +// +// macOS VERIFIES. Node cannot resolve a name relative to a descriptor there — +// /dev/fd/ is not a magic link (it stats as the directory but every attempt +// to traverse a child through it returns ENOENT), and fcntl F_GETPATH is a +// name-cache snapshot rather than a live anchor. So the Darwin backend resolves +// lexically, holds an open descriptor on every element of the chain, and proves +// before and after each operation that the path chain still names exactly the +// inodes it is holding. That DETECTS a swapped parent and aborts the write; it +// does not make the swap impossible the way the Linux path does. A swap landing +// inside the window between a check and the call it guards is caught by the +// following check, after the fact, rather than being unreachable. +// +// Every other platform gets neither and is refused outright. function requireDescriptorAnchoring() { - if ( - process.platform !== 'linux' || - fs.constants.O_DIRECTORY === undefined || - fs.constants.O_NOFOLLOW === undefined || - !fs.existsSync('/proc/self/fd') - ) { - throw new Error( - 'Safe generated-plan writes require Linux /proc/self/fd and O_DIRECTORY/O_NOFOLLOW; refusing an unanchored write', - ); + const directoryFlagsAvailable = + fs.constants.O_DIRECTORY !== undefined && fs.constants.O_NOFOLLOW !== undefined; + if (process.platform === 'linux') { + if (!directoryFlagsAvailable || !fs.existsSync('/proc/self/fd')) { + throw new Error( + 'Safe generated-plan writes require Linux /proc/self/fd and O_DIRECTORY/O_NOFOLLOW; refusing an unanchored write', + ); + } + return; } + if (process.platform === 'darwin') { + if (!directoryFlagsAvailable) { + throw new Error( + 'Safe generated-plan writes require macOS O_DIRECTORY/O_NOFOLLOW; refusing an unverified write', + ); + } + return; + } + throw new Error( + `Safe generated-plan writes require Linux /proc/self/fd or macOS O_DIRECTORY/O_NOFOLLOW; ${process.platform} offers neither, so refusing an unanchored write`, + ); } function descriptorPath(fd, childName) { @@ -900,157 +1023,352 @@ function descriptorPath(fd, childName) { return childName === undefined ? base : path.join(base, childName); } -function externalDescriptorPath(fd, childName) { - const base = `/proc/${process.pid}/fd/${fd}`; - return childName === undefined ? base : path.join(base, childName); +// Directory opens are plain O_RDONLY|O_DIRECTORY|O_NOFOLLOW|O_CLOEXEC on both +// platforms, and deliberately nothing else. +// +// O_NOFOLLOW_ANY (macOS 11+) used to be ORed in here on the theory that XNU +// ignores unrecognized open flag bits, so it would be inert where unsupported. +// That was wrong: combined with O_DIRECTORY macOS rejects it outright with +// EINVAL, and every directory open on Darwin failed. It is gone and is not +// coming back behind a probe or a degrade-on-EINVAL path — the per-component +// O_NOFOLLOW walk is what delivers the guarantee. Rust's cap-std, the closest +// reference implementation of this problem, has not adopted O_NOFOLLOW_ANY +// either (their issue #179 is still open). +function openVerifiedDirectory(absolute, flags) { + return fs.openSync(absolute, flags); } -const RENAME_NOREPLACE_SCRIPT = String.raw` -import ctypes -import errno -import os -import sys - -libc = ctypes.CDLL(None, use_errno=True) -try: - renameat2 = libc.renameat2 -except AttributeError: - print("libc does not expose renameat2", file=sys.stderr) - raise SystemExit(125) - -renameat2.argtypes = [ctypes.c_int, ctypes.c_char_p, ctypes.c_int, ctypes.c_char_p, ctypes.c_uint] -renameat2.restype = ctypes.c_int -result = renameat2(-100, os.fsencode(sys.argv[1]), -100, os.fsencode(sys.argv[2]), 1) -if result != 0: - error_number = ctypes.get_errno() - error_name = errno.errorcode.get(error_number, "UNKNOWN") - print(f"renameat2 RENAME_NOREPLACE failed: {error_name}: {os.strerror(error_number)}", file=sys.stderr) - raise SystemExit(17 if error_number == errno.EEXIST else 126) -`; - -let atomicMoverPath; - -function spawnHeldExecutable(executable, args, options) { - const before = fs.fstatSync(executable.fd, { bigint: true }); - if (!before.isFile() || statIdentity(before) !== executable.identity) { - throw new Error('Validated Python executable changed before invocation'); - } - const result = spawnSync('/proc/self/fd/3', args, { - ...options, - stdio: ['ignore', 'pipe', 'pipe', executable.fd], - }); - const after = fs.fstatSync(executable.fd, { bigint: true }); - assertStableIdentity(before, after, 'validated Python executable'); - return result; +// File opens additionally get O_NONBLOCK, which directory opens do not need: +// it stops a FIFO swapped in at the target name from wedging the process on +// open. The identity comparison that follows rejects the FIFO anyway, but only +// if we ever get as far as running it. +function openVerifiedFile(absolute, flags, mode) { + const nonBlocking = flags | (fs.constants.O_NONBLOCK ?? 0); + return mode === undefined + ? fs.openSync(absolute, nonBlocking) + : fs.openSync(absolute, nonBlocking, mode); } -function validatedPathExecutable(candidate) { - if (!path.isAbsolute(candidate)) return null; - const candidateDirectory = path.dirname(candidate); - let resolvedDirectory; - let resolved; - let directoryStats; - let executableStat; +// The publish primitive, identical on both platforms. +// +// link() is the portable no-replace publish: it fails with EEXIST if the +// destination name is taken — by a regular file, by a directory, or by a symlink, +// live or dangling — and it never follows that symlink to clobber its target. +// It also works where renameat2(RENAME_NOREPLACE) does not, notably v9fs, which +// is why the WSL2 9p case that used to fail every time now works. +// +// The published file is the same inode as the temporary, so every identity +// comparison the callers already make still holds, and validateCommittedPlan +// becomes strictly stronger: it compares the destination against the exact inode +// whose bytes were fsynced. +// +// On Linux both paths are /proc/self/fd//, so the publish is anchored +// to the held parent descriptors exactly like every other operation. +// link(2) BUGS: "On NFS filesystems, the return code may be wrong in case the NFS +// server performs the link creation and dies before it can say so. Use stat(2) to +// find out if the link got created." open(2) NOTES gives the remedy this +// implements: on a reported failure, stat the source and see whether its link +// count reached 2. A false positive would need someone to have hardlinked a +// 16-random-byte name inside a directory we hold open — and validateCommittedPlan +// still proves the destination is the exact temporary inode afterwards. +function linkCreatedDespiteError(sourcePath) { try { - resolvedDirectory = fs.realpathSync(candidateDirectory); - resolved = fs.realpathSync(candidate); - const resolvedExecutableDirectory = fs.realpathSync(path.dirname(resolved)); - directoryStats = [...new Set([resolvedDirectory, resolvedExecutableDirectory])].map( - (directory) => fs.statSync(directory), - ); - executableStat = fs.lstatSync(resolved); - fs.accessSync(resolved, fs.constants.X_OK); + return fs.statSync(sourcePath, { bigint: true }).nlink === 2n; } catch { - return null; + return false; } - if ( - directoryStats.some((stat) => !stat.isDirectory()) || - !executableStat.isFile() || - executableStat.isSymbolicLink() - ) { - return null; - } - const uid = typeof process.getuid === 'function' ? process.getuid() : null; - const trustedOwner = (stat) => uid === null || stat.uid === 0 || stat.uid === uid; - if ( - directoryStats.some((stat) => !trustedOwner(stat) || (stat.mode & 0o022) !== 0) || - !trustedOwner(executableStat) || - (executableStat.mode & 0o022) !== 0 - ) { - return null; - } - return resolved; } -function resolveAtomicMover() { - if (atomicMoverPath) return atomicMoverPath; - const candidates = new Set(); - for (const entry of (process.env.PATH ?? '').split(path.delimiter)) { - if (entry && path.isAbsolute(entry)) candidates.add(path.join(entry, 'python3')); - } - for (const entry of ['/usr/local/bin/python3', '/usr/bin/python3', '/bin/python3']) { - candidates.add(entry); - } - for (const candidate of candidates) { - const resolved = validatedPathExecutable(candidate); - if (!resolved) continue; - let fd; - try { - fd = fs.openSync( - resolved, - fs.constants.O_RDONLY | fs.constants.O_NOFOLLOW | (fs.constants.O_CLOEXEC ?? 0), - ); - } catch { - continue; +function linkNoReplace(sourcePath, destinationPath) { + try { + fs.linkSync(sourcePath, destinationPath); + } catch (error) { + // Callers treat "destination taken" as a distinct outcome, not a failure. + if (error?.code === 'EEXIST') return false; + if (!linkCreatedDespiteError(sourcePath)) { + // FAT, Coda, and some SMB/FUSE/virtiofs mounts have no hardlinks at all. + // Git falls back to rename here, but git can afford to lose collision + // detection because its objects are content-addressed; a plan destination + // is a plain name, so a replacing rename would silently clobber whatever + // is already there. Refuse loudly instead. + if (error?.code === 'EPERM' || error?.code === 'ENOTSUP' || error?.code === 'EMLINK') { + throw new Error( + `Generated-plan publication requires hard links, which this filesystem refused (${error.code}); refusing to fall back to a replacing rename`, + ); + } + throw error; } - const opened = fs.fstatSync(fd, { bigint: true }); - const executable = { fd, identity: statIdentity(opened), resolved }; - const version = spawnHeldExecutable( - executable, - ['-I', '-S', '-c', 'import sys; print(sys.version_info[0])'], - { - encoding: 'utf8', - env: { ...process.env, LANG: 'C', LC_ALL: 'C' }, - timeout: 10_000, - windowsHide: true, - }, - ); - if (version.status === 0 && version.stdout.trim() === '3') { - atomicMoverPath = executable; - return executable; - } - fs.closeSync(fd); } - throw new Error( - 'Safe generated-plan publication requires a trusted absolute Python 3 PATH candidate with libc renameat2 support', - ); -} - -function atomicMoveNoReplace(source, destination) { - const mover = resolveAtomicMover(); - const result = spawnHeldExecutable( - mover, - ['-I', '-S', '-c', RENAME_NOREPLACE_SCRIPT, source, destination], - { - encoding: 'utf8', - env: { ...process.env, LANG: 'C', LC_ALL: 'C' }, - timeout: 10_000, - windowsHide: true, - }, - ); - if (result.error) throw result.error; - if (result.status === 17) return false; - if (result.status !== 0) { - throw new Error( - `Atomic no-replace move failed (${result.status}): ${(result.stderr ?? '').trim()}`, - ); + try { + fs.unlinkSync(sourcePath); + } catch { + // The link succeeded, so the plan IS published. A temporary name left behind + // is a stray file, not an unpublished plan: reporting it as a failure would + // be a lie, and rolling back would unpublish a plan that is already live. } return true; } -function lstatOptional(absolute) { +// A directory holder is anything that owns a verified chain: a plan-parent +// handle, a ref's parent directory, or an absence guard. Two arrays describe it, +// both root-first and the same length — `chain` records each element's expected +// path and dev/ino/mode, and `descriptors` holds an open descriptor on each. +// +// Holding those descriptors is load-bearing rather than decorative. dev/ino/mode +// is unique only among *live* inodes: an inode number freed by an rmdir is handed +// straight back to the next mkdir, so a replacement directory can reproduce a +// recorded identity exactly. An open descriptor pins the inode, so the number +// cannot be recycled for as long as the holder exists. +function verifyPinnedDescriptors(holder) { + const { chain, descriptors } = holder; + if (!Array.isArray(descriptors) || descriptors.length !== chain.length) { + throw new Error('Generated-plan parent chain is missing the descriptors that pin it'); + } + chain.forEach((item, index) => { + const pinned = fs.fstatSync(descriptors[index], { bigint: true }); + if (!pinned.isDirectory() || stableDirectoryIdentity(pinned) !== item.identity) { + throw new Error('Generated-plan parent descriptor changed during the write'); + } + }); +} + +function verifyLexicalChain(holder) { + for (const item of holder.chain) { + let lexical; + try { + lexical = fs.lstatSync(item.expectedPath, { bigint: true }); + } catch (error) { + if (error?.code !== 'ENOENT' && error?.code !== 'ENOTDIR') throw error; + // A parent renamed out from under us is a mismatch, not a missing file: + // reporting the raw ENOENT would leak an unrelated-looking error out of a + // check whose whole job is to say the chain no longer holds. + throw new Error('Generated-plan lexical parent no longer matches its directory descriptor'); + } + if ( + lexical.isSymbolicLink() || + !lexical.isDirectory() || + stableDirectoryIdentity(lexical) !== item.identity + ) { + throw new Error('Generated-plan lexical parent no longer matches its directory descriptor'); + } + } +} + +// The whole platform seam, in five methods. Everything else an operation does is +// identical on both platforms and lives in the shared functions below. +// +// Only two things actually differ: how a name becomes a path, and what guard +// wraps the operation that uses it. +// +// Linux ANCHORS. /proc/self/fd// starts the walk at the inode the +// descriptor holds, so a parent renamed away cannot be traversed at all and the +// guard is a no-op — there is nothing left to verify. +// +// macOS VERIFIES. It resolves lexically, so before and after every operation it +// proves that each element of the path chain still names the exact inode being +// held for it. That DETECTS a swapped parent and aborts; it does not make the +// swap impossible. A swap landing inside the window is caught by the trailing +// check, after the fact, rather than being unreachable. The check runs after a +// failure too, because a verdict observed through a chain that has since changed +// is not a verdict. +const LINUX_ANCHORING = { + childPath(dirHandle, childName) { + return descriptorPath(dirHandle.fd, childName); + }, + verified(holders, run) { + return run(); + }, + descriptorMatchesChild(fd, expectedPath) { + return fs.realpathSync.native(descriptorPath(fd)) === expectedPath; + }, + parentStillResolves(parentHandle) { + return fs.realpathSync.native(descriptorPath(parentHandle.fd)) === parentHandle.expectedPath; + }, + verifyAbsentChild(guard) { + if (absentChildIsPresent(guard.ref)) { + throw new Error(`${guard.repoPath} appeared before evidence materialization completed`); + } + }, +}; + +const DARWIN_ANCHORING = { + childPath(dirHandle, childName) { + return path.join(dirHandle.expectedPath, childName); + }, + verified(holders, run) { + const list = Array.isArray(holders) ? holders : [holders]; + const proveChain = () => { + for (const holder of list) { + verifyPinnedDescriptors(holder); + verifyLexicalChain(holder); + } + }; + proveChain(); + let value; + try { + value = run(); + } catch (error) { + proveChain(); + throw error; + } + proveChain(); + return value; + }, + descriptorMatchesChild(fd, _expectedPath, childStat) { + // There is no live fd-to-path oracle on macOS (F_GETPATH is a name-cache + // snapshot, not an anchor), so escape is decided the other way round: the + // name was just resolved under a verified chain, and the descriptor opened + // from it counts only if it is that same inode. + const opened = fs.fstatSync(fd, { bigint: true }); + return ( + opened.isDirectory() && stableDirectoryIdentity(opened) === stableDirectoryIdentity(childStat) + ); + }, + parentStillResolves(parentHandle) { + // Both halves are needed: a directory renamed away keeps its inode, so the + // descriptors alone still match and only the lexical half notices it moved. + try { + verifyPinnedDescriptors(parentHandle); + verifyLexicalChain(parentHandle); + } catch { + return false; + } + return true; + }, + verifyAbsentChild(guard) { + let present; + try { + present = DARWIN_ANCHORING.verified(guard.handle, () => absentChildIsPresent(guard.ref)); + } catch (error) { + // A chain that no longer holds makes the absence verdict meaningless, and + // the caller reports that as the anchor changing rather than as a stray + // parent-descriptor error. Linux cannot reach this: its guard is a no-op. + throw new Error( + `Absence anchor changed for ${guard.repoPath}: ${ + error instanceof Error ? error.message : String(error) + }`, + ); + } + if (present) { + throw new Error(`${guard.repoPath} appeared before evidence materialization completed`); + } + }, +}; + +const ANCHORING_BACKENDS = new Map([ + ['linux', LINUX_ANCHORING], + ['darwin', DARWIN_ANCHORING], +]); + +function anchoringBackend() { + const backend = ANCHORING_BACKENDS.get(process.platform); + if (!backend) { + // requireDescriptorAnchoring normally refuses first; this is the same answer + // from the other side, so an unsupported platform can never fall through to + // whichever backend happened to be the ternary's default. + throw new Error( + `No generated-plan anchoring backend for ${process.platform}; refusing an unanchored write`, + ); + } + return backend; +} + +// Open, fstat, compare, close on mismatch. The descriptor never escapes this +// function unless it refers to the inode the caller already verified by name, so +// a lexical open that landed anywhere else cannot be used by accident. On Linux +// the comparison passes trivially — the /proc walk already resolved from the +// held parent — and costs one fstat to keep the guarantee structural rather than +// dependent on which backend is in play. +function adoptVerifiedFile(ref, expectedStat, flags) { + const fd = openVerifiedFile(ref.path, flags); + let opened; try { - return fs.lstatSync(absolute, { bigint: true }); + opened = fs.fstatSync(fd, { bigint: true }); + } catch (error) { + fs.closeSync(fd); + throw error; + } + if (stableFileIdentity(opened) !== stableFileIdentity(expectedStat)) { + fs.closeSync(fd); + return null; + } + return fd; +} + +function absentChildIsPresent(ref) { + try { + fs.lstatSync(ref.path, { bigint: true }); + } catch (error) { + if (error?.code === 'ENOENT') return false; + throw error; + } + return true; +} + +// The operations. Each is the same on both platforms; only the guard differs. +function lstatChild(ref) { + return anchoringBackend().verified(ref.dir, () => fs.lstatSync(ref.path, { bigint: true })); +} + +function openChildRead(ref, flags, expectedStat) { + return anchoringBackend().verified(ref.dir, () => { + const fd = adoptVerifiedFile(ref, expectedStat, flags); + if (fd === null) { + throw new Error(`${ref.name} was replaced between its verified stat and its no-follow open`); + } + return fd; + }); +} + +function createChild(ref, flags, mode) { + // O_CREAT|O_EXCL|O_NOFOLLOW is atomic at the leaf, so the only thing the guard + // has to cover is which directory the leaf landed in. + return anchoringBackend().verified(ref.dir, () => openVerifiedFile(ref.path, flags, mode)); +} + +function mkdirChild(ref, mode) { + anchoringBackend().verified(ref.dir, () => fs.mkdirSync(ref.path, { mode })); +} + +function publishNoReplace(sourceRef, destinationRef) { + return anchoringBackend().verified([sourceRef.dir, destinationRef.dir], () => + linkNoReplace(sourceRef.path, destinationRef.path), + ); +} + +// The single place a name becomes a path, and therefore the right place to +// enforce that a name is one ordinary component. +// +// A trailing separator is the sharp edge here, not a tidiness concern: +// open(path, O_NOFOLLOW) FOLLOWS a symlink when path ends in "/" — the trap +// behind CVE-2026-39822 / golang/go#79005, which let os.Root escape its own +// root. path.join preserves that trailing slash, so a component carrying one +// would turn every no-follow open in this file into a following one. +// normalizeRepoPath already rejects such components upstream; this is the +// chokepoint that makes it true for every caller, including the generated +// temporary and vault names that never pass through it. +function anchoredChild(dirHandle, childName) { + if ( + typeof childName !== 'string' || + childName === '' || + childName === '.' || + childName === '..' || + childName.includes('/') || + childName.includes('\\') || + childName.includes('\0') + ) { + throw new Error(`Refusing to resolve ${JSON.stringify(childName)} as a single path component`); + } + return { + dir: dirHandle, + name: childName, + path: anchoringBackend().childPath(dirHandle, childName), + }; +} + +function lstatAnchoredOptional(ref) { + try { + return lstatChild(ref); } catch (error) { if (error?.code === 'ENOENT' || error?.code === 'ENOTDIR') return null; throw error; @@ -1063,39 +1381,37 @@ function openPlanParent( { createMissing = true, purpose = 'Generated-plan' } = {}, ) { requireDescriptorAnchoring(); - const flags = - fs.constants.O_RDONLY | - fs.constants.O_DIRECTORY | - fs.constants.O_NOFOLLOW | - (fs.constants.O_CLOEXEC ?? 0); + // Root-first and index-aligned with `chain`: verifyPinnedDescriptors relies on + // that, and the descriptors are what pin each recorded inode against reuse. const descriptors = []; try { - let currentFd = fs.openSync(repo, flags); + let currentFd = openVerifiedDirectory(repo, ANCHORED_DIRECTORY_FLAGS); descriptors.push(currentFd); const rootStat = fs.fstatSync(currentFd, { bigint: true }); const chain = [{ expectedPath: repo, identity: stableDirectoryIdentity(rootStat) }]; + let currentHandle = { fd: currentFd, expectedPath: repo, chain, descriptors }; const traversed = []; for (const component of parentComponents) { traversed.push(component); - const anchoredChild = descriptorPath(currentFd, component); + const child = anchoredChild(currentHandle, component); let childStat; let created = false; try { - childStat = fs.lstatSync(anchoredChild, { bigint: true }); + childStat = lstatChild(child); } catch (error) { if (error?.code !== 'ENOENT' && error?.code !== 'ENOTDIR') throw error; if (!createMissing) { throw new Error(`${purpose} parent does not exist: ${traversed.join('/')}`); } - fs.mkdirSync(anchoredChild, { mode: 0o755 }); - childStat = fs.lstatSync(anchoredChild, { bigint: true }); + mkdirChild(child, 0o755); + childStat = lstatChild(child); created = true; } if (childStat.isSymbolicLink() || !childStat.isDirectory()) { throw new Error(`${purpose} parent is not a real directory: ${traversed.join('/')}`); } const parentFd = currentFd; - const childFd = fs.openSync(anchoredChild, flags); + const childFd = openVerifiedDirectory(child.path, ANCHORED_DIRECTORY_FLAGS); descriptors.push(childFd); currentFd = childFd; if (created) { @@ -1103,18 +1419,16 @@ function openPlanParent( fs.fsyncSync(parentFd); } const expected = path.join(repo, ...traversed); - const actual = fs.realpathSync(descriptorPath(currentFd)); - if (actual !== expected) { + if (!anchoringBackend().descriptorMatchesChild(currentFd, expected, childStat)) { throw new Error(`${purpose} parent escaped the repository: ${traversed.join('/')}`); } const openedStat = fs.fstatSync(currentFd, { bigint: true }); chain.push({ expectedPath: expected, identity: stableDirectoryIdentity(openedStat) }); + currentHandle = { fd: currentFd, expectedPath: expected, chain, descriptors }; } - const stat = fs.fstatSync(currentFd, { bigint: true }); return { descriptors, fd: currentFd, - identity: stableDirectoryIdentity(stat), expectedPath: path.join(repo, ...parentComponents), chain, }; @@ -1134,9 +1448,16 @@ function closeDescriptors(descriptors) { } } +// A handle's identity IS its chain leaf's identity. Storing it twice meant two +// fstats a line apart and a re-stamp helper to keep them agreeing; deriving it +// removes both. +function handleIdentity(handle) { + return handle.chain[handle.chain.length - 1].identity; +} + function resolveGitDirectory(repo) { const result = git(repo, ['rev-parse', '--absolute-git-dir']); - return fs.realpathSync(decodeUtf8(result.stdout, 'Git administrative directory').trim()); + return fs.realpathSync.native(decodeUtf8(result.stdout, 'Git administrative directory').trim()); } function openBackupVault(repo, { createMissing = true } = {}) { @@ -1147,9 +1468,12 @@ function openBackupVault(repo, { createMissing = true } = {}) { }); fs.fchmodSync(handle.fd, 0o700); fs.fsyncSync(handle.fd); - const stat = fs.fstatSync(handle.fd, { bigint: true }); - handle.identity = stableDirectoryIdentity(stat); - handle.chain[handle.chain.length - 1].identity = handle.identity; + // mode is part of every directory identity, so hardening the vault changes the + // identity the chain recorded for it; without this the next verification would + // reject the directory it just hardened. + handle.chain[handle.chain.length - 1].identity = stableDirectoryIdentity( + fs.fstatSync(handle.fd, { bigint: true }), + ); return { ...handle, gitDirectory }; } @@ -1157,33 +1481,28 @@ function validatePlanParent(parentHandle) { const descriptorStat = fs.fstatSync(parentHandle.fd, { bigint: true }); if ( !descriptorStat.isDirectory() || - stableDirectoryIdentity(descriptorStat) !== parentHandle.identity + stableDirectoryIdentity(descriptorStat) !== handleIdentity(parentHandle) ) { throw new Error('Generated-plan parent descriptor changed during the write'); } - const descriptorRealPath = fs.realpathSync(descriptorPath(parentHandle.fd)); - if (descriptorRealPath !== parentHandle.expectedPath) { + if (!anchoringBackend().parentStillResolves(parentHandle)) { throw new Error('Generated-plan parent moved or was replaced during the write'); } - for (const item of parentHandle.chain) { - const lexicalStat = fs.lstatSync(item.expectedPath, { bigint: true }); - if ( - lexicalStat.isSymbolicLink() || - !lexicalStat.isDirectory() || - stableDirectoryIdentity(lexicalStat) !== item.identity - ) { - throw new Error('Generated-plan lexical parent no longer matches its directory descriptor'); - } - } + // Both halves come from the shared helpers rather than being restated here: an + // earlier hand-copy of the lexical loop lost verifyLexicalChain's ENOENT/ENOTDIR + // translation, so a renamed parent could surface a raw errno from a function + // with a dozen call sites. + verifyPinnedDescriptors(parentHandle); + verifyLexicalChain(parentHandle); } function inspectPlanDestination( - finalPath, + finalRef, { replace, expectedIdentity, mustBeAbsent = false } = {}, ) { let stat; try { - stat = fs.lstatSync(finalPath, { bigint: true }); + stat = lstatChild(finalRef); } catch (error) { if (error?.code === 'ENOENT') { if (expectedIdentity) throw new Error('Generated plan disappeared during the write'); @@ -1201,19 +1520,17 @@ function inspectPlanDestination( if (expectedIdentity && identity !== expectedIdentity) { throw new Error('Generated plan changed during the write'); } - return identity; + return stat; } -function openExistingPlanDestination(finalPath, replace) { - const identity = inspectPlanDestination(finalPath, { replace }); - if (identity === null) { +function openExistingPlanDestination(finalRef, replace) { + const stat = inspectPlanDestination(finalRef, { replace }); + if (stat === null) { if (replace) throw new Error('Deepen mode requires an existing generated plan to replace'); return { fd: undefined, identity: null, stableIdentity: null }; } - const fd = fs.openSync( - finalPath, - fs.constants.O_RDONLY | fs.constants.O_NOFOLLOW | (fs.constants.O_CLOEXEC ?? 0), - ); + const identity = statIdentity(stat); + const fd = openChildRead(finalRef, VERIFIED_READ_FLAGS, stat); try { const opened = fs.fstatSync(fd, { bigint: true }); if (!opened.isFile() || statIdentity(opened) !== identity) { @@ -1264,8 +1581,8 @@ function hashOpenFile(fd, label) { }; } -function validateCommittedPlan(finalPath, tempFd, expectedTemp, testHooks) { - const before = fs.lstatSync(finalPath, { bigint: true }); +function validateCommittedPlan(finalRef, tempFd, expectedTemp, testHooks) { + const before = lstatChild(finalRef); if ( before.isSymbolicLink() || !before.isFile() || @@ -1273,19 +1590,16 @@ function validateCommittedPlan(finalPath, tempFd, expectedTemp, testHooks) { ) { throw new Error('Generated-plan destination failed its first post-write identity check'); } - const finalFd = fs.openSync( - finalPath, - fs.constants.O_RDONLY | fs.constants.O_NOFOLLOW | (fs.constants.O_CLOEXEC ?? 0), - ); + const finalFd = openChildRead(finalRef, VERIFIED_READ_FLAGS, before); try { const opened = fs.fstatSync(finalFd, { bigint: true }); if (!opened.isFile() || stableFileIdentity(opened) !== expectedTemp.identity) { throw new Error('Generated-plan destination changed while its no-follow descriptor opened'); } - testHooks?.afterFinalOpen?.({ fd: finalFd, finalPath }); + testHooks?.afterFinalOpen?.({ fd: finalFd, finalPath: finalRef.path }); const committedViaTemp = hashOpenFile(tempFd, 'generated-plan committed file'); const committedViaPath = hashOpenFile(finalFd, 'generated-plan destination descriptor'); - const after = fs.lstatSync(finalPath, { bigint: true }); + const after = lstatChild(finalRef); const openedAfter = fs.fstatSync(finalFd, { bigint: true }); if ( after.isSymbolicLink() || @@ -1320,22 +1634,19 @@ function copyOpenFile(sourceFd, destinationFd, label) { return after; } -function openVerifiedPathFile(absolute, label) { - const before = fs.lstatSync(absolute, { bigint: true }); +function openVerifiedAnchoredFile(ref, label, knownStat) { + const before = knownStat ?? lstatChild(ref); if (before.isSymbolicLink() || !before.isFile()) { throw new Error(`${label} is not a regular no-follow file`); } - const fd = fs.openSync( - absolute, - fs.constants.O_RDONLY | fs.constants.O_NOFOLLOW | (fs.constants.O_CLOEXEC ?? 0), - ); + const fd = openChildRead(ref, VERIFIED_READ_FLAGS, before); try { const opened = fs.fstatSync(fd, { bigint: true }); if (!opened.isFile() || stableFileIdentity(opened) !== stableFileIdentity(before)) { throw new Error(`${label} changed while its descriptor opened`); } const layer = hashOpenFile(fd, label); - const after = fs.lstatSync(absolute, { bigint: true }); + const after = lstatChild(ref); if (after.isSymbolicLink() || !after.isFile() || stableFileIdentity(after) !== layer.identity) { throw new Error(`${label} changed after verification`); } @@ -1358,10 +1669,10 @@ export function readPlanSafely({ repo: repoInput, generatedPlanPath, testHooks } let fd; try { validatePlanParent(parentHandle); - const finalPath = descriptorPath(parentHandle.fd, finalName); + const finalRef = anchoredChild(parentHandle, finalName); let before; try { - before = fs.lstatSync(finalPath, { bigint: true }); + before = lstatChild(finalRef); } catch (error) { if (error?.code === 'ENOENT' || error?.code === 'ENOTDIR') { throw new Error(`Loaded plan does not exist: ${generatedPlan}`); @@ -1371,15 +1682,12 @@ export function readPlanSafely({ repo: repoInput, generatedPlanPath, testHooks } if (before.isSymbolicLink() || !before.isFile()) { throw new Error('Loaded plan must be a regular file, never a symlink'); } - fd = fs.openSync( - finalPath, - fs.constants.O_RDONLY | fs.constants.O_NOFOLLOW | (fs.constants.O_CLOEXEC ?? 0), - ); + fd = openChildRead(finalRef, VERIFIED_READ_FLAGS, before); const opened = fs.fstatSync(fd, { bigint: true }); if (!opened.isFile() || statIdentity(opened) !== statIdentity(before)) { throw new Error('Loaded plan changed while its no-follow descriptor opened'); } - testHooks?.afterPlanOpen?.({ fd, finalPath }); + testHooks?.afterPlanOpen?.({ fd, finalPath: finalRef.path }); const chunks = []; let total = 0; const buffer = Buffer.allocUnsafe(64 * 1024); @@ -1394,7 +1702,7 @@ export function readPlanSafely({ repo: repoInput, generatedPlanPath, testHooks } decodeUtf8(contents, 'loaded plan'); const after = fs.fstatSync(fd, { bigint: true }); assertStableIdentity(opened, after, 'loaded plan'); - const pathAfter = fs.lstatSync(finalPath, { bigint: true }); + const pathAfter = lstatChild(finalRef); if ( pathAfter.isSymbolicLink() || !pathAfter.isFile() || @@ -1419,24 +1727,22 @@ function artifactGitPath(name) { return `gitnexus-plan-backups/${name}`; } -function verifyVaultArtifactFromFreshRoot(repo, gitPath, expectedLayer) { - const components = gitPath.split('/'); - if (components.length !== 2 || components[0] !== 'gitnexus-plan-backups') { - throw new Error(`Invalid Git-admin artifact path: ${gitPath}`); - } +function verifyVaultArtifactFromFreshRoot(repo, name, expectedLayer) { const freshVault = openBackupVault(repo, { createMissing: false }); try { validatePlanParent(freshVault); - const opened = openVerifiedPathFile( - descriptorPath(freshVault.fd, components[1]), - `Git-admin artifact ${gitPath}`, + const opened = openVerifiedAnchoredFile( + anchoredChild(freshVault, name), + `Git-admin artifact ${artifactGitPath(name)}`, ); try { if ( opened.layer.identity !== expectedLayer.identity || opened.layer.digest !== expectedLayer.digest ) { - throw new Error(`Git-admin artifact changed before fresh-root verification: ${gitPath}`); + throw new Error( + `Git-admin artifact changed before fresh-root verification: ${artifactGitPath(name)}`, + ); } } finally { fs.closeSync(opened.fd); @@ -1449,16 +1755,8 @@ function verifyVaultArtifactFromFreshRoot(repo, gitPath, expectedLayer) { function createVaultCopyFromFd(repo, vault, sourceFd, role) { validatePlanParent(vault); const name = `.gitnexus-plan-${role}-${process.pid}-${randomBytes(16).toString('hex')}.bak`; - const absolute = descriptorPath(vault.fd, name); - const destinationFd = fs.openSync( - absolute, - fs.constants.O_RDWR | - fs.constants.O_CREAT | - fs.constants.O_EXCL | - fs.constants.O_NOFOLLOW | - (fs.constants.O_CLOEXEC ?? 0), - 0o600, - ); + const artifact = anchoredChild(vault, name); + const destinationFd = createChild(artifact, VERIFIED_CREATE_FLAGS, 0o600); let destination; try { const sourceStat = copyOpenFile(sourceFd, destinationFd, role); @@ -1469,7 +1767,7 @@ function createVaultCopyFromFd(repo, vault, sourceFd, role) { if (source.size !== destination.size || source.digest !== destination.digest) { throw new Error(`${role} vault copy does not match its held source descriptor`); } - const pathStat = fs.lstatSync(absolute, { bigint: true }); + const pathStat = lstatChild(artifact); if ( pathStat.isSymbolicLink() || !pathStat.isFile() || @@ -1481,24 +1779,15 @@ function createVaultCopyFromFd(repo, vault, sourceFd, role) { } finally { fs.closeSync(destinationFd); } - const gitPath = artifactGitPath(name); - verifyVaultArtifactFromFreshRoot(repo, gitPath, destination); - return { role, gitPath, layer: destination }; + verifyVaultArtifactFromFreshRoot(repo, name, destination); + return { role, gitPath: artifactGitPath(name), layer: destination }; } function createVaultCopyFromBytes(repo, vault, contents, role) { validatePlanParent(vault); const name = `.gitnexus-plan-${role}-${process.pid}-${randomBytes(16).toString('hex')}.bak`; - const absolute = descriptorPath(vault.fd, name); - const fd = fs.openSync( - absolute, - fs.constants.O_RDWR | - fs.constants.O_CREAT | - fs.constants.O_EXCL | - fs.constants.O_NOFOLLOW | - (fs.constants.O_CLOEXEC ?? 0), - 0o600, - ); + const artifact = anchoredChild(vault, name); + const fd = createChild(artifact, VERIFIED_CREATE_FLAGS, 0o600); let layer; try { writeAll(fd, contents); @@ -1508,7 +1797,7 @@ function createVaultCopyFromBytes(repo, vault, contents, role) { if (layer.size !== BigInt(contents.length) || layer.digest !== sha256(contents)) { throw new Error(`${role} vault copy does not match the intended plan bytes`); } - const pathStat = fs.lstatSync(absolute, { bigint: true }); + const pathStat = lstatChild(artifact); if ( pathStat.isSymbolicLink() || !pathStat.isFile() || @@ -1520,32 +1809,31 @@ function createVaultCopyFromBytes(repo, vault, contents, role) { } finally { fs.closeSync(fd); } - const gitPath = artifactGitPath(name); - verifyVaultArtifactFromFreshRoot(repo, gitPath, layer); - return { role, gitPath, layer }; + verifyVaultArtifactFromFreshRoot(repo, name, layer); + return { role, gitPath: artifactGitPath(name), layer }; } function movePathToVault(repo, sourceHandle, sourceName, vault, role) { - const source = descriptorPath(sourceHandle.fd, sourceName); - if (!lstatOptional(source)) return null; + const source = anchoredChild(sourceHandle, sourceName); + if (!lstatAnchoredOptional(source)) return null; const name = `.gitnexus-plan-${role}-${process.pid}-${randomBytes(16).toString('hex')}.bak`; - const destination = descriptorPath(vault.fd, name); - const moved = atomicMoveNoReplace( - externalDescriptorPath(sourceHandle.fd, sourceName), - externalDescriptorPath(vault.fd, name), - ); + const destination = anchoredChild(vault, name); + const moved = publishNoReplace(source, destination); if (!moved) throw new Error(`${role} preservation destination unexpectedly exists`); fs.fsyncSync(sourceHandle.fd); if (vault.fd !== sourceHandle.fd) fs.fsyncSync(vault.fd); - const sourceAfter = lstatOptional(source); - const destinationAfter = lstatOptional(destination); + const sourceAfter = lstatAnchoredOptional(source); + const destinationAfter = lstatAnchoredOptional(destination); if (sourceAfter || !destinationAfter) { throw new Error(`${role} could not be atomically moved into the Git-admin vault`); } - const opened = openVerifiedPathFile(destination, `${role} Git-admin artifact`); - const gitPath = artifactGitPath(name); - verifyVaultArtifactFromFreshRoot(repo, gitPath, opened.layer); - return { role, gitPath, layer: opened.layer, fd: opened.fd }; + const opened = openVerifiedAnchoredFile( + destination, + `${role} Git-admin artifact`, + destinationAfter, + ); + verifyVaultArtifactFromFreshRoot(repo, name, opened.layer); + return { role, gitPath: artifactGitPath(name), layer: opened.layer, fd: opened.fd }; } function formatPreservedArtifacts(artifacts) { @@ -1600,10 +1888,10 @@ export function writePlanSafely({ const finalName = components.pop(); let parentHandle; let vaultHandle; - let tempPath; + let tempRef; let tempName; let tempFd; - let finalPath; + let finalRef; let expectedTemp; let originalDestination; let priorBackup; @@ -1611,7 +1899,6 @@ export function writePlanSafely({ try { parentHandle = openPlanParent(repo, components); vaultHandle = openBackupVault(repo); - resolveAtomicMover(); const parentDevice = fs.fstatSync(parentHandle.fd, { bigint: true }).dev; const vaultDevice = fs.fstatSync(vaultHandle.fd, { bigint: true }).dev; if (parentDevice !== vaultDevice) { @@ -1622,19 +1909,11 @@ export function writePlanSafely({ testHooks?.afterParentOpen?.({ fd: parentHandle.fd, path: parentHandle.expectedPath }); validatePlanParent(parentHandle); validatePlanParent(vaultHandle); - finalPath = descriptorPath(parentHandle.fd, finalName); - originalDestination = openExistingPlanDestination(finalPath, shouldReplace); + finalRef = anchoredChild(parentHandle, finalName); + originalDestination = openExistingPlanDestination(finalRef, shouldReplace); tempName = `.gitnexus-plan-${process.pid}-${randomBytes(16).toString('hex')}.tmp`; - tempPath = descriptorPath(parentHandle.fd, tempName); - tempFd = fs.openSync( - tempPath, - fs.constants.O_RDWR | - fs.constants.O_CREAT | - fs.constants.O_EXCL | - fs.constants.O_NOFOLLOW | - (fs.constants.O_CLOEXEC ?? 0), - 0o600, - ); + tempRef = anchoredChild(parentHandle, tempName); + tempFd = createChild(tempRef, VERIFIED_CREATE_FLAGS, 0o600); writeAll(tempFd, contents); fs.fchmodSync(tempFd, 0o644); fs.fsyncSync(tempFd); @@ -1646,12 +1925,12 @@ export function writePlanSafely({ testHooks?.beforeRename?.({ fd: parentHandle.fd, path: parentHandle.expectedPath, - tempPath, + tempPath: tempRef.path, }); validatePlanParent(parentHandle); validatePlanParent(vaultHandle); validateOpenPlanDestination(originalDestination); - const tempPathStat = fs.lstatSync(tempPath, { bigint: true }); + const tempPathStat = lstatChild(tempRef); const currentTemp = hashOpenFile(tempFd, 'generated-plan temporary file'); if ( tempPathStat.isSymbolicLink() || @@ -1664,7 +1943,7 @@ export function writePlanSafely({ } if (shouldReplace) { - testHooks?.beforeBackupMove?.({ fd: parentHandle.fd, finalPath }); + testHooks?.beforeBackupMove?.({ fd: parentHandle.fd, finalPath: finalRef.path }); const originalLayer = hashOpenFile(originalDestination.fd, 'prior generated plan'); if (originalLayer.digest !== expectedDigest) { throw new Error( @@ -1673,7 +1952,7 @@ export function writePlanSafely({ } validatePlanParent(parentHandle); validateOpenPlanDestination(originalDestination); - inspectPlanDestination(finalPath, { + inspectPlanDestination(finalRef, { replace: true, expectedIdentity: originalDestination.identity, }); @@ -1691,20 +1970,20 @@ export function writePlanSafely({ ); throw new Error('Destination raced while the prior plan was moved into preservation'); } - if (lstatOptional(finalPath)) { + if (lstatAnchoredOptional(finalRef)) { throw new Error('Destination reappeared after the prior plan was preserved'); } } testHooks?.beforePublication?.({ fd: parentHandle.fd, - finalPath, - tempPath, + finalPath: finalRef.path, + tempPath: tempRef.path, replace: shouldReplace, }); validatePlanParent(parentHandle); validatePlanParent(vaultHandle); - const finalTempPathStat = fs.lstatSync(tempPath, { bigint: true }); + const finalTempPathStat = lstatChild(tempRef); const finalTemp = hashOpenFile(tempFd, 'generated-plan temporary file'); if ( finalTempPathStat.isSymbolicLink() || @@ -1715,19 +1994,25 @@ export function writePlanSafely({ ) { throw new Error('Generated-plan temporary path or content changed at publication'); } - atomicMoveNoReplace( - externalDescriptorPath(parentHandle.fd, tempName), - externalDescriptorPath(parentHandle.fd, finalName), - ); - if (lstatOptional(tempPath) || !lstatOptional(finalPath)) { + // link() reports the race itself; re-deriving that verdict from a later pair + // of stats would be both slower and weaker. + if (!publishNoReplace(tempRef, finalRef)) { throw new Error('Generated-plan publication was refused because the destination raced'); } + // link() creates a directory entry, so it needs the parent fsync that rename + // needed: the file's own bytes were fsynced through tempFd before this point, + // and this makes the name that now reaches them durable too. Skipping it is + // the step write-file-atomic omits and maildir, git and atomicwrites all + // mandate. + // + // Honest limitation: on macOS fsync is not a write barrier — the durable + // primitive there is fcntl(F_FULLFSYNC), which Node does not expose. A + // macOS plan write is therefore as durable as fsync makes it and no more. fs.fsyncSync(parentHandle.fd); - testHooks?.afterPublication?.({ fd: parentHandle.fd, finalPath }); - testHooks?.afterRename?.({ fd: parentHandle.fd, finalPath }); + testHooks?.afterPublication?.({ fd: parentHandle.fd, finalPath: finalRef.path }); validatePlanParent(parentHandle); validatePlanParent(vaultHandle); - validateCommittedPlan(finalPath, tempFd, expectedTemp, testHooks); + validateCommittedPlan(finalRef, tempFd, expectedTemp, testHooks); const receipt = { generated_plan_path: generatedPlan, bytes_written: contents.length }; if (priorBackup) receipt.prior_plan_backup_git_path = priorBackup.gitPath; return receipt; @@ -1848,6 +2133,11 @@ export function snapshotEvidence({ const headGuards = captureHeadGuards(repo); const dirty = initialDirty.records; const mutationGuards = []; + // Per-snapshot walk state: `absenceCache` owns every descriptor an absence + // anchor holds, deduplicated by repo-relative prefix and closed exactly once + // below; `guardedDirectories` keeps parent guarding to one stat per directory. + const absenceCache = new Map(); + const walkState = { absenceCache, guardedDirectories: new Set() }; try { testHooks?.afterAnchorCapture?.({ headCommit: head }); @@ -1862,7 +2152,9 @@ export function snapshotEvidence({ testHooks?.afterGitLayerLoad?.({ headCommit: head }); const globalEntries = [...dirty.values()] .filter((record) => record.path !== generatedPlan) - .map((record) => materializeRecord(repo, record, layers, mutationGuards, testHooks)); + .map((record) => + materializeRecord(repo, record, layers, mutationGuards, testHooks, walkState), + ); const citedEntries = [...normalizedCitations].sort(compareUtf8).map((repoPath) => { const status = dirty.get(repoPath) ?? { path: repoPath, @@ -1871,7 +2163,7 @@ export function snapshotEvidence({ rename_to: null, has_untracked: false, }; - const entry = materializeRecord(repo, status, layers, mutationGuards, testHooks); + const entry = materializeRecord(repo, status, layers, mutationGuards, testHooks, walkState); const present = Object.values(entry.object_kind).some((kind) => kind !== ABSENT); if (!present) entry.state = ABSENT; else if (entry.state === 'clean' && entry.object_kind.untracked !== ABSENT) { @@ -1906,21 +2198,13 @@ export function snapshotEvidence({ throw new Error(`${guard.absolute} changed before evidence materialization completed`); } } else if (guard.type === 'absence') { + // statIdentity is a strict superset of stableDirectoryIdentity on the + // same stat, so comparing both could only ever fire together. const parent = fs.fstatSync(guard.fd, { bigint: true }); - if ( - !parent.isDirectory() || - stableDirectoryIdentity(parent) !== guard.parentIdentity || - statIdentity(parent) !== guard.parentMutationIdentity - ) { + if (!parent.isDirectory() || statIdentity(parent) !== guard.parentMutationIdentity) { throw new Error(`Absence anchor changed for ${guard.repoPath}`); } - try { - fs.lstatSync(descriptorPath(guard.fd, guard.childName), { bigint: true }); - } catch (error) { - if (error?.code === 'ENOENT') continue; - throw error; - } - throw new Error(`${guard.repoPath} appeared before evidence materialization completed`); + anchoringBackend().verifyAbsentChild(guard); } } for (const guard of headGuards) verifyControlFile(guard); @@ -1955,12 +2239,10 @@ export function snapshotEvidence({ cited_path_manifest: citedEntries, }; } finally { - const closed = new Set(); - for (const guard of mutationGuards) { - if (guard.type !== 'absence' || closed.has(guard.fd)) continue; - closed.add(guard.fd); + // One entry per distinct anchored directory, so one close per descriptor. + for (const handle of absenceCache.values()) { try { - fs.closeSync(guard.fd); + fs.closeSync(handle.fd); } catch { // Preserve the primary snapshot result/error. } diff --git a/.claude/skills/gitnexus-refactoring/SKILL.md b/.claude/skills/gitnexus-refactoring/SKILL.md index 2dbb71ca0..4f10bbc6a 100644 --- a/.claude/skills/gitnexus-refactoring/SKILL.md +++ b/.claude/skills/gitnexus-refactoring/SKILL.md @@ -87,6 +87,11 @@ detect_changes({scope: "all"}) → Risk: MEDIUM ``` +`partial: true` (a graph query failed) or `truncated: true` (the changed-symbol +listing was capped) means the result is short of the truth: a short or empty +list is not proof that only the expected files changed. Re-run it rather than +treat the refactor as verified. + **cypher** — custom reference queries: ```cypher diff --git a/.claude/skills/gitnexus-work/SKILL.md b/.claude/skills/gitnexus-work/SKILL.md index 4f7856ea7..f9baab16a 100644 --- a/.claude/skills/gitnexus-work/SKILL.md +++ b/.claude/skills/gitnexus-work/SKILL.md @@ -216,7 +216,10 @@ Work through plan §7 step by step, in order. For each step: `detect_changes` → commit as one unbroken sequence from the repository root — interleaving other work between the gate and the commit is how the gate gets skipped. Unexpected - affected flows → investigate before committing, not after. + affected flows → investigate before committing, not after. A result + flagged `partial` (a graph query failed) or `truncated` (the symbol + listing was capped) blocks the commit the same way: the gate did not + see every changed symbol, so re-run it rather than read it as clean. A relationship-affecting implementation edit or commit invalidates the procedure's prior proof. The next step must perform the required inter-step diff --git a/.claude/skills/gitnexus-work/references/evidence-provenance.md b/.claude/skills/gitnexus-work/references/evidence-provenance.md index c686599da..3df5a046d 100644 --- a/.claude/skills/gitnexus-work/references/evidence-provenance.md +++ b/.claude/skills/gitnexus-work/references/evidence-provenance.md @@ -98,8 +98,11 @@ excluded. ## Safe existing-plan read contract -`read-plan` fails closed unless Linux `/proc/self/fd`, `O_DIRECTORY`, and -`O_NOFOLLOW` are available. It resolves the exact Git top-level, opens the +`read-plan` fails closed unless the host platform can resolve names against a +held directory descriptor: Linux `/proc/self/fd` with `O_DIRECTORY` and +`O_NOFOLLOW`, or macOS `O_DIRECTORY`/`O_NOFOLLOW`. Every other platform is +refused outright — an unverified read is not a degraded read, it is a different, +racy operation. It resolves the exact Git top-level, opens the repository root and every plan parent as held no-follow directory descriptors, rejects missing, symlink, non-directory, and escaping parents, and opens the leaf with `O_NOFOLLOW`. It reads at most 16 MiB from that held file descriptor, @@ -109,13 +112,17 @@ Neither Deepen nor work may parse bytes obtained before or outside this receipt. ## Safe generated-plan write contract -The writer fails closed unless Linux `/proc/self/fd`, `O_DIRECTORY`, -`O_NOFOLLOW`, and Python 3 with libc `renameat2(RENAME_NOREPLACE)` support are -available. Python may live in `/usr/local`, a Nix profile, or another absolute -PATH directory, but the helper accepts only a resolved executable and -containing directory owned by root or the current user and not writable by -group/other. The resolved executable is opened without following links and -invoked through that held descriptor. Relative PATH entries are ignored. The plan parent and the +The writer fails closed unless the host platform offers `O_DIRECTORY` and +`O_NOFOLLOW`, plus `/proc/self/fd` on Linux. It spawns no interpreter and loads +no native code: publication is `link(2)`, which is atomic, fails `EEXIST` when +the destination name is taken, and refuses a symlinked destination without +following it — the same no-replace guarantee `renameat2(RENAME_NOREPLACE)` and +`renameatx_np(RENAME_EXCL)` provide, available through `fs.linkSync` on every +supported platform. The temporary name is unlinked once the link succeeds; the +published file is the same inode the writer created and verified, so every +identity check downstream holds by construction. A link that succeeds followed +by an unlink that fails leaves the plan published and is reported as success, +because it is one. The plan parent and the repository's Git-admin directory must also share a filesystem. It resolves the target repository's exact Git top-level, opens that root and every destination parent as held no-follow directory descriptors, creates missing @@ -128,15 +135,45 @@ The writer creates a random exclusive temporary file relative to the held final parent descriptor and keeps its no-follow descriptor open. It writes and flushes the bytes, binds the temporary name to the opened inode, and hashes the open file before publication. Immediately before publication it revalidates -the parent and the temporary path, inode, size, and digest. Publication uses an -atomic no-replace move relative to the held directory descriptor. Initial mode -therefore cannot overwrite a destination that appears after the absent check. +the parent and the temporary path, inode, size, and digest. Publication links +the temporary name to the destination relative to the held directory +descriptor, which fails rather than replaces if the destination is taken. +Initial mode therefore cannot overwrite a destination that appears after the +absent check. The writer then flushes the directory and revalidates the committed path by opening it with `O_NOFOLLOW`, hashing both the original temporary fd and the path-bound fd, and performing a second descriptor-anchored path identity check after hashing. A detected mutation or replacement aborts instead of accepting mixed-era output. +### Linux anchors, macOS verifies + +The two platforms reach the same destination by different proofs, and the +difference is real enough to state rather than smooth over. + +On Linux every name resolves through `/proc/self/fd//`, a magic link +the kernel resolves against the inode the descriptor already holds. The names +above it are never re-walked, so an attacker who renames a parent between the +check and the use cannot redirect the operation. The race is impossible, not +merely detected. + +macOS has no such path. `/dev/fd/` is a devfs node, not a magic link: it can +be opened, but nothing can be resolved through it. `open("/dev/fd//child")` +returns `ENOENT`, and `realpath` of it returns `/dev/fd/` rather than the +directory's path — measured on macOS 26, not inferred. Node exposes no `openat`, +no `dir_fd` parameter, and no FFI, so on macOS the writer resolves names +lexically with `O_NOFOLLOW` at every component, holds an open descriptor on +every directory in the chain for the whole operation, and proves before *and* +after each step that the chain still names exactly the inodes it is holding. +Holding the descriptors is what makes the recorded inode numbers trustworthy: +an open descriptor pins its inode, so a freed number cannot be recycled beneath +the walk. + +What that buys is detection rather than prevention. A parent swapped inside the +window between a check and its use is caught by the check that follows, and the +operation aborts having written nothing — but on Linux it could not have +happened at all. No published byte escapes verification on either platform. + `--replace` accepts only a pre-existing regular file and is reserved for Deepen; without it, accidental overwrite is rejected. It also requires the exact canonical `generated_plan_path` and `plan_digest` from the same session's diff --git a/.claude/skills/gitnexus-work/scripts/evidence-provenance.mjs b/.claude/skills/gitnexus-work/scripts/evidence-provenance.mjs index 181d2120b..793fe4cd8 100644 --- a/.claude/skills/gitnexus-work/scripts/evidence-provenance.mjs +++ b/.claude/skills/gitnexus-work/scripts/evidence-provenance.mjs @@ -479,11 +479,11 @@ function resolveOwnGitTopLevel(absolute) { if (result.status !== 0) return null; let topLevel; try { - topLevel = fs.realpathSync(decodeUtf8(result.stdout, 'nested repository root').trim()); + topLevel = fs.realpathSync.native(decodeUtf8(result.stdout, 'nested repository root').trim()); } catch { return null; } - return topLevel === fs.realpathSync(absolute) ? topLevel : null; + return topLevel === fs.realpathSync.native(absolute) ? topLevel : null; } function readOwnGitlinkHead(absolute) { @@ -616,17 +616,30 @@ function filesystemObject(absolute, expectedKind, mutationGuards, testHooks) { throw new Error(`Unsupported filesystem object at ${absolute}`); } -function guardPathParents(repo, repoPath, mutationGuards) { +// Every dirty path re-walks its own parents, and dirty paths overwhelmingly +// share them — the repository root is re-stat'ed once per path. `guarded` is +// per-snapshot and remembers which absolute directories already carry a guard, +// so each distinct directory is stat'ed and guarded exactly once. +// +// Keeping the first-seen identity is the conservative choice: verifyGuards +// re-checks every guard against the filesystem at the end, so a directory that +// changes after it was guarded still fails there. Skipping a re-stat cannot hide +// a change; it only avoids recording the same directory twice. +function guardPathParents(repo, repoPath, mutationGuards, guarded) { const components = repoPath.split('/'); let current = repo; - const rootStat = fs.lstatSync(repo, { bigint: true }); - mutationGuards.push({ - type: 'directory', - absolute: repo, - identity: stableDirectoryIdentity(rootStat), - }); + if (!guarded.has(repo)) { + guarded.add(repo); + mutationGuards.push({ + type: 'directory', + absolute: repo, + identity: stableDirectoryIdentity(fs.lstatSync(repo, { bigint: true })), + }); + } for (const component of components.slice(0, -1)) { current = path.join(current, component); + // Already proved a real directory and already guarded on an earlier path. + if (guarded.has(current)) continue; let stat; try { stat = fs.lstatSync(current, { bigint: true }); @@ -638,6 +651,7 @@ function guardPathParents(repo, repoPath, mutationGuards) { throw new Error(`Refusing to traverse symlink parent for ${repoPath}`); } if (!stat.isDirectory()) return; + guarded.add(current); mutationGuards.push({ type: 'directory', absolute: current, @@ -646,81 +660,153 @@ function guardPathParents(repo, repoPath, mutationGuards) { } } -function recordAnchoredAbsence(repo, repoPath, mutationGuards) { - requireDescriptorAnchoring(); - const flags = - fs.constants.O_RDONLY | - fs.constants.O_DIRECTORY | - fs.constants.O_NOFOLLOW | - (fs.constants.O_CLOEXEC ?? 0); - const descriptors = []; - let retainedFd; - try { - let currentFd = fs.openSync(repo, flags); - descriptors.push(currentFd); - const components = repoPath.split('/'); - for (let index = 0; index < components.length; index += 1) { - const component = components[index]; - const child = descriptorPath(currentFd, component); - let childStat; - try { - childStat = fs.lstatSync(child, { bigint: true }); - } catch (error) { - if (error?.code !== 'ENOENT' && error?.code !== 'ENOTDIR') throw error; - const parentStat = fs.fstatSync(currentFd, { bigint: true }); - if (!parentStat.isDirectory()) { - throw new Error(`Absence parent is no longer a directory for ${repoPath}`); - } - retainedFd = currentFd; - mutationGuards.push({ - type: 'absence', - fd: retainedFd, - childName: component, - repoPath, - parentIdentity: stableDirectoryIdentity(parentStat), - parentMutationIdentity: statIdentity(parentStat), - }); - for (const fd of descriptors) { - if (fd !== retainedFd) fs.closeSync(fd); - } - return; - } - if (index === components.length - 1) { - throw new Error(`${repoPath} appeared while its absence was being anchored`); - } - if (childStat.isSymbolicLink() || !childStat.isDirectory()) { - throw new Error(`Refusing a non-directory parent while anchoring absence for ${repoPath}`); - } - const nextFd = fs.openSync(child, flags); - descriptors.push(nextFd); - currentFd = nextFd; - } - throw new Error(`Could not anchor absence for ${repoPath}`); - } catch (error) { - for (const fd of descriptors) { - if (fd === retainedFd) continue; - try { - fs.closeSync(fd); - } catch { - // Preserve the primary absence-anchoring error. - } - } - throw error; +// A bound, not a bug: the absence cache deduplicates correctly and leaks nothing, +// but citedPaths is caller-supplied and unbounded, so a pathological snapshot +// could hold more descriptors than the process is allowed (macOS +// kern.maxfilesperproc is 24576). The peak precedes a `git` spawn, so exhaustion +// would surface as a git failure misreported as evidence instability. +// +// Refuse rather than evict: closing a cached descriptor would silently break the +// pinned chain of an absence guard that was already recorded against it, which is +// exactly the inode-recycling hole the pins exist to close. +const ABSENCE_ANCHOR_LIMITS = Object.freeze({ maxPinnedDirectories: 4096 }); + +// Every no-follow read and every exclusive create in this file uses one of these +// two, so a change lands in one place rather than in seven. +const VERIFIED_READ_FLAGS = + fs.constants.O_RDONLY | fs.constants.O_NOFOLLOW | (fs.constants.O_CLOEXEC ?? 0); +const VERIFIED_CREATE_FLAGS = + fs.constants.O_RDWR | + fs.constants.O_CREAT | + fs.constants.O_EXCL | + fs.constants.O_NOFOLLOW | + (fs.constants.O_CLOEXEC ?? 0); + +function requireAbsenceAnchorCapacity(cache) { + if (cache.size >= ABSENCE_ANCHOR_LIMITS.maxPinnedDirectories) { + throw new Error( + `Absence anchoring exceeds ${ABSENCE_ANCHOR_LIMITS.maxPinnedDirectories} pinned directories`, + ); } } -function materializeRecord(repo, statusRecord, layers, mutationGuards, testHooks) { +const ANCHORED_DIRECTORY_FLAGS = + fs.constants.O_RDONLY | + fs.constants.O_DIRECTORY | + fs.constants.O_NOFOLLOW | + (fs.constants.O_CLOEXEC ?? 0); + +// Every absence receipt is verified long after its walk returns, so the chain +// that produced it has to stay pinned until the snapshot ends — an unpinned inode +// number can be recycled by a replacement directory that then reproduces the +// recorded identity exactly. Absent cited paths overwhelmingly share prefixes, so +// the walked directories are cached per snapshot and keyed by repo-relative +// prefix: one open descriptor and one anchored walk per distinct directory rather +// than per path. snapshotEvidence owns every descriptor in this cache and closes +// each exactly once; guards only borrow them for verification. +function anchoredAbsenceRoot(repo, cache) { + const cached = cache.get(''); + if (cached) return cached; + requireAbsenceAnchorCapacity(cache); + const fd = openVerifiedDirectory(repo, ANCHORED_DIRECTORY_FLAGS); + const handle = { + fd, + expectedPath: repo, + chain: [ + { expectedPath: repo, identity: stableDirectoryIdentity(fs.fstatSync(fd, { bigint: true })) }, + ], + descriptors: [fd], + }; + cache.set('', handle); + return handle; +} + +function recordAnchoredAbsence(repo, repoPath, mutationGuards, cache) { + requireDescriptorAnchoring(); + const components = repoPath.split('/'); + let handle = anchoredAbsenceRoot(repo, cache); + let prefix = ''; + for (let index = 0; index < components.length; index += 1) { + const component = components[index]; + const isFinal = index === components.length - 1; + prefix = prefix === '' ? component : `${prefix}/${component}`; + // The final component is always re-checked against the filesystem: it is the + // one whose absence is being recorded, and a cached answer would be a stale + // one. Only the prefix directories are reused. + const cached = isFinal ? undefined : cache.get(prefix); + if (cached) { + handle = cached; + continue; + } + const child = anchoredChild(handle, component); + let childStat; + try { + childStat = lstatChild(child); + } catch (error) { + if (error?.code !== 'ENOENT' && error?.code !== 'ENOTDIR') throw error; + const parentStat = fs.fstatSync(handle.fd, { bigint: true }); + if (!parentStat.isDirectory()) { + throw new Error(`Absence parent is no longer a directory for ${repoPath}`); + } + mutationGuards.push({ + type: 'absence', + // The handle is the holder the guard verifies against, and `ref` is the + // child path already built through the anchoredChild chokepoint — the + // guard must never re-derive that name itself. + handle, + ref: child, + fd: handle.fd, + repoPath, + parentMutationIdentity: statIdentity(parentStat), + }); + return; + } + if (isFinal) { + throw new Error(`${repoPath} appeared while its absence was being anchored`); + } + if (childStat.isSymbolicLink() || !childStat.isDirectory()) { + throw new Error(`Refusing a non-directory parent while anchoring absence for ${repoPath}`); + } + requireAbsenceAnchorCapacity(cache); + const childFd = openVerifiedDirectory(child.path, ANCHORED_DIRECTORY_FLAGS); + const expectedPath = path.join(handle.expectedPath, component); + let next; + try { + if (!anchoringBackend().descriptorMatchesChild(childFd, expectedPath, childStat)) { + throw new Error( + `Absence parent descriptor does not match its verified inode for ${repoPath}`, + ); + } + next = { + fd: childFd, + expectedPath, + chain: [...handle.chain, { expectedPath, identity: stableDirectoryIdentity(childStat) }], + descriptors: [...handle.descriptors, childFd], + }; + } catch (error) { + fs.closeSync(childFd); + throw error; + } + cache.set(prefix, next); + handle = next; + } + throw new Error(`Could not anchor absence for ${repoPath}`); +} + +function materializeRecord(repo, statusRecord, layers, mutationGuards, testHooks, walkState) { const head = layers.head(statusRecord.path); const index = layers.index(statusRecord.path); const expectedKind = index.kind === 'gitlink' || head.kind === 'gitlink' ? 'gitlink' : null; - guardPathParents(repo, statusRecord.path, mutationGuards); + guardPathParents(repo, statusRecord.path, mutationGuards, walkState.guardedDirectories); const filesystem = filesystemObject( path.join(repo, ...statusRecord.path.split('/')), expectedKind, mutationGuards, testHooks, ); - if (filesystem.kind === ABSENT) recordAnchoredAbsence(repo, statusRecord.path, mutationGuards); + if (filesystem.kind === ABSENT) { + recordAnchoredAbsence(repo, statusRecord.path, mutationGuards, walkState.absenceCache); + } if (statusRecord.directory_hint && filesystem.kind !== 'directory') { throw new Error( `Git reported an embedded directory but found ${filesystem.kind}: ${statusRecord.path}`, @@ -789,9 +875,15 @@ export function serializeDirtyRecords(entries) { } function assertRepository(repoInput) { - const repo = fs.realpathSync(requireString(repoInput, 'repo')); + // realpathSync.native, not realpathSync: the JS resolver preserves a Windows + // 8.3 short component (C:\Users\RUNNER~1\...) while git always reports the long + // form, so the two would never compare equal and every caller would be told the + // worktree root is not the worktree root it just named. + const repo = fs.realpathSync.native(requireString(repoInput, 'repo')); const topLevelResult = git(repo, ['rev-parse', '--show-toplevel']); - const topLevel = fs.realpathSync(decodeUtf8(topLevelResult.stdout, 'repository root').trim()); + const topLevel = fs.realpathSync.native( + decodeUtf8(topLevelResult.stdout, 'repository root').trim(), + ); if (topLevel !== repo) throw new Error(`--repo must be the Git worktree root (${topLevel})`); return repo; } @@ -882,17 +974,48 @@ function stableFileIdentity(stat) { return [stat.dev, stat.ino, stat.mode, stat.size].map(String).join(':'); } +// The two backends below differ in one decisive way, and it is worth stating +// plainly because the security properties are not the same. +// +// Linux ANCHORS. A name is resolved through /proc/self/fd//, which +// starts the walk at the inode the descriptor holds, so a parent that is renamed +// away cannot be traversed at all: the descriptor keeps pointing at the original +// directory and the impostor planted at the same name is simply never reached. +// +// macOS VERIFIES. Node cannot resolve a name relative to a descriptor there — +// /dev/fd/ is not a magic link (it stats as the directory but every attempt +// to traverse a child through it returns ENOENT), and fcntl F_GETPATH is a +// name-cache snapshot rather than a live anchor. So the Darwin backend resolves +// lexically, holds an open descriptor on every element of the chain, and proves +// before and after each operation that the path chain still names exactly the +// inodes it is holding. That DETECTS a swapped parent and aborts the write; it +// does not make the swap impossible the way the Linux path does. A swap landing +// inside the window between a check and the call it guards is caught by the +// following check, after the fact, rather than being unreachable. +// +// Every other platform gets neither and is refused outright. function requireDescriptorAnchoring() { - if ( - process.platform !== 'linux' || - fs.constants.O_DIRECTORY === undefined || - fs.constants.O_NOFOLLOW === undefined || - !fs.existsSync('/proc/self/fd') - ) { - throw new Error( - 'Safe generated-plan writes require Linux /proc/self/fd and O_DIRECTORY/O_NOFOLLOW; refusing an unanchored write', - ); + const directoryFlagsAvailable = + fs.constants.O_DIRECTORY !== undefined && fs.constants.O_NOFOLLOW !== undefined; + if (process.platform === 'linux') { + if (!directoryFlagsAvailable || !fs.existsSync('/proc/self/fd')) { + throw new Error( + 'Safe generated-plan writes require Linux /proc/self/fd and O_DIRECTORY/O_NOFOLLOW; refusing an unanchored write', + ); + } + return; } + if (process.platform === 'darwin') { + if (!directoryFlagsAvailable) { + throw new Error( + 'Safe generated-plan writes require macOS O_DIRECTORY/O_NOFOLLOW; refusing an unverified write', + ); + } + return; + } + throw new Error( + `Safe generated-plan writes require Linux /proc/self/fd or macOS O_DIRECTORY/O_NOFOLLOW; ${process.platform} offers neither, so refusing an unanchored write`, + ); } function descriptorPath(fd, childName) { @@ -900,157 +1023,352 @@ function descriptorPath(fd, childName) { return childName === undefined ? base : path.join(base, childName); } -function externalDescriptorPath(fd, childName) { - const base = `/proc/${process.pid}/fd/${fd}`; - return childName === undefined ? base : path.join(base, childName); +// Directory opens are plain O_RDONLY|O_DIRECTORY|O_NOFOLLOW|O_CLOEXEC on both +// platforms, and deliberately nothing else. +// +// O_NOFOLLOW_ANY (macOS 11+) used to be ORed in here on the theory that XNU +// ignores unrecognized open flag bits, so it would be inert where unsupported. +// That was wrong: combined with O_DIRECTORY macOS rejects it outright with +// EINVAL, and every directory open on Darwin failed. It is gone and is not +// coming back behind a probe or a degrade-on-EINVAL path — the per-component +// O_NOFOLLOW walk is what delivers the guarantee. Rust's cap-std, the closest +// reference implementation of this problem, has not adopted O_NOFOLLOW_ANY +// either (their issue #179 is still open). +function openVerifiedDirectory(absolute, flags) { + return fs.openSync(absolute, flags); } -const RENAME_NOREPLACE_SCRIPT = String.raw` -import ctypes -import errno -import os -import sys - -libc = ctypes.CDLL(None, use_errno=True) -try: - renameat2 = libc.renameat2 -except AttributeError: - print("libc does not expose renameat2", file=sys.stderr) - raise SystemExit(125) - -renameat2.argtypes = [ctypes.c_int, ctypes.c_char_p, ctypes.c_int, ctypes.c_char_p, ctypes.c_uint] -renameat2.restype = ctypes.c_int -result = renameat2(-100, os.fsencode(sys.argv[1]), -100, os.fsencode(sys.argv[2]), 1) -if result != 0: - error_number = ctypes.get_errno() - error_name = errno.errorcode.get(error_number, "UNKNOWN") - print(f"renameat2 RENAME_NOREPLACE failed: {error_name}: {os.strerror(error_number)}", file=sys.stderr) - raise SystemExit(17 if error_number == errno.EEXIST else 126) -`; - -let atomicMoverPath; - -function spawnHeldExecutable(executable, args, options) { - const before = fs.fstatSync(executable.fd, { bigint: true }); - if (!before.isFile() || statIdentity(before) !== executable.identity) { - throw new Error('Validated Python executable changed before invocation'); - } - const result = spawnSync('/proc/self/fd/3', args, { - ...options, - stdio: ['ignore', 'pipe', 'pipe', executable.fd], - }); - const after = fs.fstatSync(executable.fd, { bigint: true }); - assertStableIdentity(before, after, 'validated Python executable'); - return result; +// File opens additionally get O_NONBLOCK, which directory opens do not need: +// it stops a FIFO swapped in at the target name from wedging the process on +// open. The identity comparison that follows rejects the FIFO anyway, but only +// if we ever get as far as running it. +function openVerifiedFile(absolute, flags, mode) { + const nonBlocking = flags | (fs.constants.O_NONBLOCK ?? 0); + return mode === undefined + ? fs.openSync(absolute, nonBlocking) + : fs.openSync(absolute, nonBlocking, mode); } -function validatedPathExecutable(candidate) { - if (!path.isAbsolute(candidate)) return null; - const candidateDirectory = path.dirname(candidate); - let resolvedDirectory; - let resolved; - let directoryStats; - let executableStat; +// The publish primitive, identical on both platforms. +// +// link() is the portable no-replace publish: it fails with EEXIST if the +// destination name is taken — by a regular file, by a directory, or by a symlink, +// live or dangling — and it never follows that symlink to clobber its target. +// It also works where renameat2(RENAME_NOREPLACE) does not, notably v9fs, which +// is why the WSL2 9p case that used to fail every time now works. +// +// The published file is the same inode as the temporary, so every identity +// comparison the callers already make still holds, and validateCommittedPlan +// becomes strictly stronger: it compares the destination against the exact inode +// whose bytes were fsynced. +// +// On Linux both paths are /proc/self/fd//, so the publish is anchored +// to the held parent descriptors exactly like every other operation. +// link(2) BUGS: "On NFS filesystems, the return code may be wrong in case the NFS +// server performs the link creation and dies before it can say so. Use stat(2) to +// find out if the link got created." open(2) NOTES gives the remedy this +// implements: on a reported failure, stat the source and see whether its link +// count reached 2. A false positive would need someone to have hardlinked a +// 16-random-byte name inside a directory we hold open — and validateCommittedPlan +// still proves the destination is the exact temporary inode afterwards. +function linkCreatedDespiteError(sourcePath) { try { - resolvedDirectory = fs.realpathSync(candidateDirectory); - resolved = fs.realpathSync(candidate); - const resolvedExecutableDirectory = fs.realpathSync(path.dirname(resolved)); - directoryStats = [...new Set([resolvedDirectory, resolvedExecutableDirectory])].map( - (directory) => fs.statSync(directory), - ); - executableStat = fs.lstatSync(resolved); - fs.accessSync(resolved, fs.constants.X_OK); + return fs.statSync(sourcePath, { bigint: true }).nlink === 2n; } catch { - return null; + return false; } - if ( - directoryStats.some((stat) => !stat.isDirectory()) || - !executableStat.isFile() || - executableStat.isSymbolicLink() - ) { - return null; - } - const uid = typeof process.getuid === 'function' ? process.getuid() : null; - const trustedOwner = (stat) => uid === null || stat.uid === 0 || stat.uid === uid; - if ( - directoryStats.some((stat) => !trustedOwner(stat) || (stat.mode & 0o022) !== 0) || - !trustedOwner(executableStat) || - (executableStat.mode & 0o022) !== 0 - ) { - return null; - } - return resolved; } -function resolveAtomicMover() { - if (atomicMoverPath) return atomicMoverPath; - const candidates = new Set(); - for (const entry of (process.env.PATH ?? '').split(path.delimiter)) { - if (entry && path.isAbsolute(entry)) candidates.add(path.join(entry, 'python3')); - } - for (const entry of ['/usr/local/bin/python3', '/usr/bin/python3', '/bin/python3']) { - candidates.add(entry); - } - for (const candidate of candidates) { - const resolved = validatedPathExecutable(candidate); - if (!resolved) continue; - let fd; - try { - fd = fs.openSync( - resolved, - fs.constants.O_RDONLY | fs.constants.O_NOFOLLOW | (fs.constants.O_CLOEXEC ?? 0), - ); - } catch { - continue; +function linkNoReplace(sourcePath, destinationPath) { + try { + fs.linkSync(sourcePath, destinationPath); + } catch (error) { + // Callers treat "destination taken" as a distinct outcome, not a failure. + if (error?.code === 'EEXIST') return false; + if (!linkCreatedDespiteError(sourcePath)) { + // FAT, Coda, and some SMB/FUSE/virtiofs mounts have no hardlinks at all. + // Git falls back to rename here, but git can afford to lose collision + // detection because its objects are content-addressed; a plan destination + // is a plain name, so a replacing rename would silently clobber whatever + // is already there. Refuse loudly instead. + if (error?.code === 'EPERM' || error?.code === 'ENOTSUP' || error?.code === 'EMLINK') { + throw new Error( + `Generated-plan publication requires hard links, which this filesystem refused (${error.code}); refusing to fall back to a replacing rename`, + ); + } + throw error; } - const opened = fs.fstatSync(fd, { bigint: true }); - const executable = { fd, identity: statIdentity(opened), resolved }; - const version = spawnHeldExecutable( - executable, - ['-I', '-S', '-c', 'import sys; print(sys.version_info[0])'], - { - encoding: 'utf8', - env: { ...process.env, LANG: 'C', LC_ALL: 'C' }, - timeout: 10_000, - windowsHide: true, - }, - ); - if (version.status === 0 && version.stdout.trim() === '3') { - atomicMoverPath = executable; - return executable; - } - fs.closeSync(fd); } - throw new Error( - 'Safe generated-plan publication requires a trusted absolute Python 3 PATH candidate with libc renameat2 support', - ); -} - -function atomicMoveNoReplace(source, destination) { - const mover = resolveAtomicMover(); - const result = spawnHeldExecutable( - mover, - ['-I', '-S', '-c', RENAME_NOREPLACE_SCRIPT, source, destination], - { - encoding: 'utf8', - env: { ...process.env, LANG: 'C', LC_ALL: 'C' }, - timeout: 10_000, - windowsHide: true, - }, - ); - if (result.error) throw result.error; - if (result.status === 17) return false; - if (result.status !== 0) { - throw new Error( - `Atomic no-replace move failed (${result.status}): ${(result.stderr ?? '').trim()}`, - ); + try { + fs.unlinkSync(sourcePath); + } catch { + // The link succeeded, so the plan IS published. A temporary name left behind + // is a stray file, not an unpublished plan: reporting it as a failure would + // be a lie, and rolling back would unpublish a plan that is already live. } return true; } -function lstatOptional(absolute) { +// A directory holder is anything that owns a verified chain: a plan-parent +// handle, a ref's parent directory, or an absence guard. Two arrays describe it, +// both root-first and the same length — `chain` records each element's expected +// path and dev/ino/mode, and `descriptors` holds an open descriptor on each. +// +// Holding those descriptors is load-bearing rather than decorative. dev/ino/mode +// is unique only among *live* inodes: an inode number freed by an rmdir is handed +// straight back to the next mkdir, so a replacement directory can reproduce a +// recorded identity exactly. An open descriptor pins the inode, so the number +// cannot be recycled for as long as the holder exists. +function verifyPinnedDescriptors(holder) { + const { chain, descriptors } = holder; + if (!Array.isArray(descriptors) || descriptors.length !== chain.length) { + throw new Error('Generated-plan parent chain is missing the descriptors that pin it'); + } + chain.forEach((item, index) => { + const pinned = fs.fstatSync(descriptors[index], { bigint: true }); + if (!pinned.isDirectory() || stableDirectoryIdentity(pinned) !== item.identity) { + throw new Error('Generated-plan parent descriptor changed during the write'); + } + }); +} + +function verifyLexicalChain(holder) { + for (const item of holder.chain) { + let lexical; + try { + lexical = fs.lstatSync(item.expectedPath, { bigint: true }); + } catch (error) { + if (error?.code !== 'ENOENT' && error?.code !== 'ENOTDIR') throw error; + // A parent renamed out from under us is a mismatch, not a missing file: + // reporting the raw ENOENT would leak an unrelated-looking error out of a + // check whose whole job is to say the chain no longer holds. + throw new Error('Generated-plan lexical parent no longer matches its directory descriptor'); + } + if ( + lexical.isSymbolicLink() || + !lexical.isDirectory() || + stableDirectoryIdentity(lexical) !== item.identity + ) { + throw new Error('Generated-plan lexical parent no longer matches its directory descriptor'); + } + } +} + +// The whole platform seam, in five methods. Everything else an operation does is +// identical on both platforms and lives in the shared functions below. +// +// Only two things actually differ: how a name becomes a path, and what guard +// wraps the operation that uses it. +// +// Linux ANCHORS. /proc/self/fd// starts the walk at the inode the +// descriptor holds, so a parent renamed away cannot be traversed at all and the +// guard is a no-op — there is nothing left to verify. +// +// macOS VERIFIES. It resolves lexically, so before and after every operation it +// proves that each element of the path chain still names the exact inode being +// held for it. That DETECTS a swapped parent and aborts; it does not make the +// swap impossible. A swap landing inside the window is caught by the trailing +// check, after the fact, rather than being unreachable. The check runs after a +// failure too, because a verdict observed through a chain that has since changed +// is not a verdict. +const LINUX_ANCHORING = { + childPath(dirHandle, childName) { + return descriptorPath(dirHandle.fd, childName); + }, + verified(holders, run) { + return run(); + }, + descriptorMatchesChild(fd, expectedPath) { + return fs.realpathSync.native(descriptorPath(fd)) === expectedPath; + }, + parentStillResolves(parentHandle) { + return fs.realpathSync.native(descriptorPath(parentHandle.fd)) === parentHandle.expectedPath; + }, + verifyAbsentChild(guard) { + if (absentChildIsPresent(guard.ref)) { + throw new Error(`${guard.repoPath} appeared before evidence materialization completed`); + } + }, +}; + +const DARWIN_ANCHORING = { + childPath(dirHandle, childName) { + return path.join(dirHandle.expectedPath, childName); + }, + verified(holders, run) { + const list = Array.isArray(holders) ? holders : [holders]; + const proveChain = () => { + for (const holder of list) { + verifyPinnedDescriptors(holder); + verifyLexicalChain(holder); + } + }; + proveChain(); + let value; + try { + value = run(); + } catch (error) { + proveChain(); + throw error; + } + proveChain(); + return value; + }, + descriptorMatchesChild(fd, _expectedPath, childStat) { + // There is no live fd-to-path oracle on macOS (F_GETPATH is a name-cache + // snapshot, not an anchor), so escape is decided the other way round: the + // name was just resolved under a verified chain, and the descriptor opened + // from it counts only if it is that same inode. + const opened = fs.fstatSync(fd, { bigint: true }); + return ( + opened.isDirectory() && stableDirectoryIdentity(opened) === stableDirectoryIdentity(childStat) + ); + }, + parentStillResolves(parentHandle) { + // Both halves are needed: a directory renamed away keeps its inode, so the + // descriptors alone still match and only the lexical half notices it moved. + try { + verifyPinnedDescriptors(parentHandle); + verifyLexicalChain(parentHandle); + } catch { + return false; + } + return true; + }, + verifyAbsentChild(guard) { + let present; + try { + present = DARWIN_ANCHORING.verified(guard.handle, () => absentChildIsPresent(guard.ref)); + } catch (error) { + // A chain that no longer holds makes the absence verdict meaningless, and + // the caller reports that as the anchor changing rather than as a stray + // parent-descriptor error. Linux cannot reach this: its guard is a no-op. + throw new Error( + `Absence anchor changed for ${guard.repoPath}: ${ + error instanceof Error ? error.message : String(error) + }`, + ); + } + if (present) { + throw new Error(`${guard.repoPath} appeared before evidence materialization completed`); + } + }, +}; + +const ANCHORING_BACKENDS = new Map([ + ['linux', LINUX_ANCHORING], + ['darwin', DARWIN_ANCHORING], +]); + +function anchoringBackend() { + const backend = ANCHORING_BACKENDS.get(process.platform); + if (!backend) { + // requireDescriptorAnchoring normally refuses first; this is the same answer + // from the other side, so an unsupported platform can never fall through to + // whichever backend happened to be the ternary's default. + throw new Error( + `No generated-plan anchoring backend for ${process.platform}; refusing an unanchored write`, + ); + } + return backend; +} + +// Open, fstat, compare, close on mismatch. The descriptor never escapes this +// function unless it refers to the inode the caller already verified by name, so +// a lexical open that landed anywhere else cannot be used by accident. On Linux +// the comparison passes trivially — the /proc walk already resolved from the +// held parent — and costs one fstat to keep the guarantee structural rather than +// dependent on which backend is in play. +function adoptVerifiedFile(ref, expectedStat, flags) { + const fd = openVerifiedFile(ref.path, flags); + let opened; try { - return fs.lstatSync(absolute, { bigint: true }); + opened = fs.fstatSync(fd, { bigint: true }); + } catch (error) { + fs.closeSync(fd); + throw error; + } + if (stableFileIdentity(opened) !== stableFileIdentity(expectedStat)) { + fs.closeSync(fd); + return null; + } + return fd; +} + +function absentChildIsPresent(ref) { + try { + fs.lstatSync(ref.path, { bigint: true }); + } catch (error) { + if (error?.code === 'ENOENT') return false; + throw error; + } + return true; +} + +// The operations. Each is the same on both platforms; only the guard differs. +function lstatChild(ref) { + return anchoringBackend().verified(ref.dir, () => fs.lstatSync(ref.path, { bigint: true })); +} + +function openChildRead(ref, flags, expectedStat) { + return anchoringBackend().verified(ref.dir, () => { + const fd = adoptVerifiedFile(ref, expectedStat, flags); + if (fd === null) { + throw new Error(`${ref.name} was replaced between its verified stat and its no-follow open`); + } + return fd; + }); +} + +function createChild(ref, flags, mode) { + // O_CREAT|O_EXCL|O_NOFOLLOW is atomic at the leaf, so the only thing the guard + // has to cover is which directory the leaf landed in. + return anchoringBackend().verified(ref.dir, () => openVerifiedFile(ref.path, flags, mode)); +} + +function mkdirChild(ref, mode) { + anchoringBackend().verified(ref.dir, () => fs.mkdirSync(ref.path, { mode })); +} + +function publishNoReplace(sourceRef, destinationRef) { + return anchoringBackend().verified([sourceRef.dir, destinationRef.dir], () => + linkNoReplace(sourceRef.path, destinationRef.path), + ); +} + +// The single place a name becomes a path, and therefore the right place to +// enforce that a name is one ordinary component. +// +// A trailing separator is the sharp edge here, not a tidiness concern: +// open(path, O_NOFOLLOW) FOLLOWS a symlink when path ends in "/" — the trap +// behind CVE-2026-39822 / golang/go#79005, which let os.Root escape its own +// root. path.join preserves that trailing slash, so a component carrying one +// would turn every no-follow open in this file into a following one. +// normalizeRepoPath already rejects such components upstream; this is the +// chokepoint that makes it true for every caller, including the generated +// temporary and vault names that never pass through it. +function anchoredChild(dirHandle, childName) { + if ( + typeof childName !== 'string' || + childName === '' || + childName === '.' || + childName === '..' || + childName.includes('/') || + childName.includes('\\') || + childName.includes('\0') + ) { + throw new Error(`Refusing to resolve ${JSON.stringify(childName)} as a single path component`); + } + return { + dir: dirHandle, + name: childName, + path: anchoringBackend().childPath(dirHandle, childName), + }; +} + +function lstatAnchoredOptional(ref) { + try { + return lstatChild(ref); } catch (error) { if (error?.code === 'ENOENT' || error?.code === 'ENOTDIR') return null; throw error; @@ -1063,39 +1381,37 @@ function openPlanParent( { createMissing = true, purpose = 'Generated-plan' } = {}, ) { requireDescriptorAnchoring(); - const flags = - fs.constants.O_RDONLY | - fs.constants.O_DIRECTORY | - fs.constants.O_NOFOLLOW | - (fs.constants.O_CLOEXEC ?? 0); + // Root-first and index-aligned with `chain`: verifyPinnedDescriptors relies on + // that, and the descriptors are what pin each recorded inode against reuse. const descriptors = []; try { - let currentFd = fs.openSync(repo, flags); + let currentFd = openVerifiedDirectory(repo, ANCHORED_DIRECTORY_FLAGS); descriptors.push(currentFd); const rootStat = fs.fstatSync(currentFd, { bigint: true }); const chain = [{ expectedPath: repo, identity: stableDirectoryIdentity(rootStat) }]; + let currentHandle = { fd: currentFd, expectedPath: repo, chain, descriptors }; const traversed = []; for (const component of parentComponents) { traversed.push(component); - const anchoredChild = descriptorPath(currentFd, component); + const child = anchoredChild(currentHandle, component); let childStat; let created = false; try { - childStat = fs.lstatSync(anchoredChild, { bigint: true }); + childStat = lstatChild(child); } catch (error) { if (error?.code !== 'ENOENT' && error?.code !== 'ENOTDIR') throw error; if (!createMissing) { throw new Error(`${purpose} parent does not exist: ${traversed.join('/')}`); } - fs.mkdirSync(anchoredChild, { mode: 0o755 }); - childStat = fs.lstatSync(anchoredChild, { bigint: true }); + mkdirChild(child, 0o755); + childStat = lstatChild(child); created = true; } if (childStat.isSymbolicLink() || !childStat.isDirectory()) { throw new Error(`${purpose} parent is not a real directory: ${traversed.join('/')}`); } const parentFd = currentFd; - const childFd = fs.openSync(anchoredChild, flags); + const childFd = openVerifiedDirectory(child.path, ANCHORED_DIRECTORY_FLAGS); descriptors.push(childFd); currentFd = childFd; if (created) { @@ -1103,18 +1419,16 @@ function openPlanParent( fs.fsyncSync(parentFd); } const expected = path.join(repo, ...traversed); - const actual = fs.realpathSync(descriptorPath(currentFd)); - if (actual !== expected) { + if (!anchoringBackend().descriptorMatchesChild(currentFd, expected, childStat)) { throw new Error(`${purpose} parent escaped the repository: ${traversed.join('/')}`); } const openedStat = fs.fstatSync(currentFd, { bigint: true }); chain.push({ expectedPath: expected, identity: stableDirectoryIdentity(openedStat) }); + currentHandle = { fd: currentFd, expectedPath: expected, chain, descriptors }; } - const stat = fs.fstatSync(currentFd, { bigint: true }); return { descriptors, fd: currentFd, - identity: stableDirectoryIdentity(stat), expectedPath: path.join(repo, ...parentComponents), chain, }; @@ -1134,9 +1448,16 @@ function closeDescriptors(descriptors) { } } +// A handle's identity IS its chain leaf's identity. Storing it twice meant two +// fstats a line apart and a re-stamp helper to keep them agreeing; deriving it +// removes both. +function handleIdentity(handle) { + return handle.chain[handle.chain.length - 1].identity; +} + function resolveGitDirectory(repo) { const result = git(repo, ['rev-parse', '--absolute-git-dir']); - return fs.realpathSync(decodeUtf8(result.stdout, 'Git administrative directory').trim()); + return fs.realpathSync.native(decodeUtf8(result.stdout, 'Git administrative directory').trim()); } function openBackupVault(repo, { createMissing = true } = {}) { @@ -1147,9 +1468,12 @@ function openBackupVault(repo, { createMissing = true } = {}) { }); fs.fchmodSync(handle.fd, 0o700); fs.fsyncSync(handle.fd); - const stat = fs.fstatSync(handle.fd, { bigint: true }); - handle.identity = stableDirectoryIdentity(stat); - handle.chain[handle.chain.length - 1].identity = handle.identity; + // mode is part of every directory identity, so hardening the vault changes the + // identity the chain recorded for it; without this the next verification would + // reject the directory it just hardened. + handle.chain[handle.chain.length - 1].identity = stableDirectoryIdentity( + fs.fstatSync(handle.fd, { bigint: true }), + ); return { ...handle, gitDirectory }; } @@ -1157,33 +1481,28 @@ function validatePlanParent(parentHandle) { const descriptorStat = fs.fstatSync(parentHandle.fd, { bigint: true }); if ( !descriptorStat.isDirectory() || - stableDirectoryIdentity(descriptorStat) !== parentHandle.identity + stableDirectoryIdentity(descriptorStat) !== handleIdentity(parentHandle) ) { throw new Error('Generated-plan parent descriptor changed during the write'); } - const descriptorRealPath = fs.realpathSync(descriptorPath(parentHandle.fd)); - if (descriptorRealPath !== parentHandle.expectedPath) { + if (!anchoringBackend().parentStillResolves(parentHandle)) { throw new Error('Generated-plan parent moved or was replaced during the write'); } - for (const item of parentHandle.chain) { - const lexicalStat = fs.lstatSync(item.expectedPath, { bigint: true }); - if ( - lexicalStat.isSymbolicLink() || - !lexicalStat.isDirectory() || - stableDirectoryIdentity(lexicalStat) !== item.identity - ) { - throw new Error('Generated-plan lexical parent no longer matches its directory descriptor'); - } - } + // Both halves come from the shared helpers rather than being restated here: an + // earlier hand-copy of the lexical loop lost verifyLexicalChain's ENOENT/ENOTDIR + // translation, so a renamed parent could surface a raw errno from a function + // with a dozen call sites. + verifyPinnedDescriptors(parentHandle); + verifyLexicalChain(parentHandle); } function inspectPlanDestination( - finalPath, + finalRef, { replace, expectedIdentity, mustBeAbsent = false } = {}, ) { let stat; try { - stat = fs.lstatSync(finalPath, { bigint: true }); + stat = lstatChild(finalRef); } catch (error) { if (error?.code === 'ENOENT') { if (expectedIdentity) throw new Error('Generated plan disappeared during the write'); @@ -1201,19 +1520,17 @@ function inspectPlanDestination( if (expectedIdentity && identity !== expectedIdentity) { throw new Error('Generated plan changed during the write'); } - return identity; + return stat; } -function openExistingPlanDestination(finalPath, replace) { - const identity = inspectPlanDestination(finalPath, { replace }); - if (identity === null) { +function openExistingPlanDestination(finalRef, replace) { + const stat = inspectPlanDestination(finalRef, { replace }); + if (stat === null) { if (replace) throw new Error('Deepen mode requires an existing generated plan to replace'); return { fd: undefined, identity: null, stableIdentity: null }; } - const fd = fs.openSync( - finalPath, - fs.constants.O_RDONLY | fs.constants.O_NOFOLLOW | (fs.constants.O_CLOEXEC ?? 0), - ); + const identity = statIdentity(stat); + const fd = openChildRead(finalRef, VERIFIED_READ_FLAGS, stat); try { const opened = fs.fstatSync(fd, { bigint: true }); if (!opened.isFile() || statIdentity(opened) !== identity) { @@ -1264,8 +1581,8 @@ function hashOpenFile(fd, label) { }; } -function validateCommittedPlan(finalPath, tempFd, expectedTemp, testHooks) { - const before = fs.lstatSync(finalPath, { bigint: true }); +function validateCommittedPlan(finalRef, tempFd, expectedTemp, testHooks) { + const before = lstatChild(finalRef); if ( before.isSymbolicLink() || !before.isFile() || @@ -1273,19 +1590,16 @@ function validateCommittedPlan(finalPath, tempFd, expectedTemp, testHooks) { ) { throw new Error('Generated-plan destination failed its first post-write identity check'); } - const finalFd = fs.openSync( - finalPath, - fs.constants.O_RDONLY | fs.constants.O_NOFOLLOW | (fs.constants.O_CLOEXEC ?? 0), - ); + const finalFd = openChildRead(finalRef, VERIFIED_READ_FLAGS, before); try { const opened = fs.fstatSync(finalFd, { bigint: true }); if (!opened.isFile() || stableFileIdentity(opened) !== expectedTemp.identity) { throw new Error('Generated-plan destination changed while its no-follow descriptor opened'); } - testHooks?.afterFinalOpen?.({ fd: finalFd, finalPath }); + testHooks?.afterFinalOpen?.({ fd: finalFd, finalPath: finalRef.path }); const committedViaTemp = hashOpenFile(tempFd, 'generated-plan committed file'); const committedViaPath = hashOpenFile(finalFd, 'generated-plan destination descriptor'); - const after = fs.lstatSync(finalPath, { bigint: true }); + const after = lstatChild(finalRef); const openedAfter = fs.fstatSync(finalFd, { bigint: true }); if ( after.isSymbolicLink() || @@ -1320,22 +1634,19 @@ function copyOpenFile(sourceFd, destinationFd, label) { return after; } -function openVerifiedPathFile(absolute, label) { - const before = fs.lstatSync(absolute, { bigint: true }); +function openVerifiedAnchoredFile(ref, label, knownStat) { + const before = knownStat ?? lstatChild(ref); if (before.isSymbolicLink() || !before.isFile()) { throw new Error(`${label} is not a regular no-follow file`); } - const fd = fs.openSync( - absolute, - fs.constants.O_RDONLY | fs.constants.O_NOFOLLOW | (fs.constants.O_CLOEXEC ?? 0), - ); + const fd = openChildRead(ref, VERIFIED_READ_FLAGS, before); try { const opened = fs.fstatSync(fd, { bigint: true }); if (!opened.isFile() || stableFileIdentity(opened) !== stableFileIdentity(before)) { throw new Error(`${label} changed while its descriptor opened`); } const layer = hashOpenFile(fd, label); - const after = fs.lstatSync(absolute, { bigint: true }); + const after = lstatChild(ref); if (after.isSymbolicLink() || !after.isFile() || stableFileIdentity(after) !== layer.identity) { throw new Error(`${label} changed after verification`); } @@ -1358,10 +1669,10 @@ export function readPlanSafely({ repo: repoInput, generatedPlanPath, testHooks } let fd; try { validatePlanParent(parentHandle); - const finalPath = descriptorPath(parentHandle.fd, finalName); + const finalRef = anchoredChild(parentHandle, finalName); let before; try { - before = fs.lstatSync(finalPath, { bigint: true }); + before = lstatChild(finalRef); } catch (error) { if (error?.code === 'ENOENT' || error?.code === 'ENOTDIR') { throw new Error(`Loaded plan does not exist: ${generatedPlan}`); @@ -1371,15 +1682,12 @@ export function readPlanSafely({ repo: repoInput, generatedPlanPath, testHooks } if (before.isSymbolicLink() || !before.isFile()) { throw new Error('Loaded plan must be a regular file, never a symlink'); } - fd = fs.openSync( - finalPath, - fs.constants.O_RDONLY | fs.constants.O_NOFOLLOW | (fs.constants.O_CLOEXEC ?? 0), - ); + fd = openChildRead(finalRef, VERIFIED_READ_FLAGS, before); const opened = fs.fstatSync(fd, { bigint: true }); if (!opened.isFile() || statIdentity(opened) !== statIdentity(before)) { throw new Error('Loaded plan changed while its no-follow descriptor opened'); } - testHooks?.afterPlanOpen?.({ fd, finalPath }); + testHooks?.afterPlanOpen?.({ fd, finalPath: finalRef.path }); const chunks = []; let total = 0; const buffer = Buffer.allocUnsafe(64 * 1024); @@ -1394,7 +1702,7 @@ export function readPlanSafely({ repo: repoInput, generatedPlanPath, testHooks } decodeUtf8(contents, 'loaded plan'); const after = fs.fstatSync(fd, { bigint: true }); assertStableIdentity(opened, after, 'loaded plan'); - const pathAfter = fs.lstatSync(finalPath, { bigint: true }); + const pathAfter = lstatChild(finalRef); if ( pathAfter.isSymbolicLink() || !pathAfter.isFile() || @@ -1419,24 +1727,22 @@ function artifactGitPath(name) { return `gitnexus-plan-backups/${name}`; } -function verifyVaultArtifactFromFreshRoot(repo, gitPath, expectedLayer) { - const components = gitPath.split('/'); - if (components.length !== 2 || components[0] !== 'gitnexus-plan-backups') { - throw new Error(`Invalid Git-admin artifact path: ${gitPath}`); - } +function verifyVaultArtifactFromFreshRoot(repo, name, expectedLayer) { const freshVault = openBackupVault(repo, { createMissing: false }); try { validatePlanParent(freshVault); - const opened = openVerifiedPathFile( - descriptorPath(freshVault.fd, components[1]), - `Git-admin artifact ${gitPath}`, + const opened = openVerifiedAnchoredFile( + anchoredChild(freshVault, name), + `Git-admin artifact ${artifactGitPath(name)}`, ); try { if ( opened.layer.identity !== expectedLayer.identity || opened.layer.digest !== expectedLayer.digest ) { - throw new Error(`Git-admin artifact changed before fresh-root verification: ${gitPath}`); + throw new Error( + `Git-admin artifact changed before fresh-root verification: ${artifactGitPath(name)}`, + ); } } finally { fs.closeSync(opened.fd); @@ -1449,16 +1755,8 @@ function verifyVaultArtifactFromFreshRoot(repo, gitPath, expectedLayer) { function createVaultCopyFromFd(repo, vault, sourceFd, role) { validatePlanParent(vault); const name = `.gitnexus-plan-${role}-${process.pid}-${randomBytes(16).toString('hex')}.bak`; - const absolute = descriptorPath(vault.fd, name); - const destinationFd = fs.openSync( - absolute, - fs.constants.O_RDWR | - fs.constants.O_CREAT | - fs.constants.O_EXCL | - fs.constants.O_NOFOLLOW | - (fs.constants.O_CLOEXEC ?? 0), - 0o600, - ); + const artifact = anchoredChild(vault, name); + const destinationFd = createChild(artifact, VERIFIED_CREATE_FLAGS, 0o600); let destination; try { const sourceStat = copyOpenFile(sourceFd, destinationFd, role); @@ -1469,7 +1767,7 @@ function createVaultCopyFromFd(repo, vault, sourceFd, role) { if (source.size !== destination.size || source.digest !== destination.digest) { throw new Error(`${role} vault copy does not match its held source descriptor`); } - const pathStat = fs.lstatSync(absolute, { bigint: true }); + const pathStat = lstatChild(artifact); if ( pathStat.isSymbolicLink() || !pathStat.isFile() || @@ -1481,24 +1779,15 @@ function createVaultCopyFromFd(repo, vault, sourceFd, role) { } finally { fs.closeSync(destinationFd); } - const gitPath = artifactGitPath(name); - verifyVaultArtifactFromFreshRoot(repo, gitPath, destination); - return { role, gitPath, layer: destination }; + verifyVaultArtifactFromFreshRoot(repo, name, destination); + return { role, gitPath: artifactGitPath(name), layer: destination }; } function createVaultCopyFromBytes(repo, vault, contents, role) { validatePlanParent(vault); const name = `.gitnexus-plan-${role}-${process.pid}-${randomBytes(16).toString('hex')}.bak`; - const absolute = descriptorPath(vault.fd, name); - const fd = fs.openSync( - absolute, - fs.constants.O_RDWR | - fs.constants.O_CREAT | - fs.constants.O_EXCL | - fs.constants.O_NOFOLLOW | - (fs.constants.O_CLOEXEC ?? 0), - 0o600, - ); + const artifact = anchoredChild(vault, name); + const fd = createChild(artifact, VERIFIED_CREATE_FLAGS, 0o600); let layer; try { writeAll(fd, contents); @@ -1508,7 +1797,7 @@ function createVaultCopyFromBytes(repo, vault, contents, role) { if (layer.size !== BigInt(contents.length) || layer.digest !== sha256(contents)) { throw new Error(`${role} vault copy does not match the intended plan bytes`); } - const pathStat = fs.lstatSync(absolute, { bigint: true }); + const pathStat = lstatChild(artifact); if ( pathStat.isSymbolicLink() || !pathStat.isFile() || @@ -1520,32 +1809,31 @@ function createVaultCopyFromBytes(repo, vault, contents, role) { } finally { fs.closeSync(fd); } - const gitPath = artifactGitPath(name); - verifyVaultArtifactFromFreshRoot(repo, gitPath, layer); - return { role, gitPath, layer }; + verifyVaultArtifactFromFreshRoot(repo, name, layer); + return { role, gitPath: artifactGitPath(name), layer }; } function movePathToVault(repo, sourceHandle, sourceName, vault, role) { - const source = descriptorPath(sourceHandle.fd, sourceName); - if (!lstatOptional(source)) return null; + const source = anchoredChild(sourceHandle, sourceName); + if (!lstatAnchoredOptional(source)) return null; const name = `.gitnexus-plan-${role}-${process.pid}-${randomBytes(16).toString('hex')}.bak`; - const destination = descriptorPath(vault.fd, name); - const moved = atomicMoveNoReplace( - externalDescriptorPath(sourceHandle.fd, sourceName), - externalDescriptorPath(vault.fd, name), - ); + const destination = anchoredChild(vault, name); + const moved = publishNoReplace(source, destination); if (!moved) throw new Error(`${role} preservation destination unexpectedly exists`); fs.fsyncSync(sourceHandle.fd); if (vault.fd !== sourceHandle.fd) fs.fsyncSync(vault.fd); - const sourceAfter = lstatOptional(source); - const destinationAfter = lstatOptional(destination); + const sourceAfter = lstatAnchoredOptional(source); + const destinationAfter = lstatAnchoredOptional(destination); if (sourceAfter || !destinationAfter) { throw new Error(`${role} could not be atomically moved into the Git-admin vault`); } - const opened = openVerifiedPathFile(destination, `${role} Git-admin artifact`); - const gitPath = artifactGitPath(name); - verifyVaultArtifactFromFreshRoot(repo, gitPath, opened.layer); - return { role, gitPath, layer: opened.layer, fd: opened.fd }; + const opened = openVerifiedAnchoredFile( + destination, + `${role} Git-admin artifact`, + destinationAfter, + ); + verifyVaultArtifactFromFreshRoot(repo, name, opened.layer); + return { role, gitPath: artifactGitPath(name), layer: opened.layer, fd: opened.fd }; } function formatPreservedArtifacts(artifacts) { @@ -1600,10 +1888,10 @@ export function writePlanSafely({ const finalName = components.pop(); let parentHandle; let vaultHandle; - let tempPath; + let tempRef; let tempName; let tempFd; - let finalPath; + let finalRef; let expectedTemp; let originalDestination; let priorBackup; @@ -1611,7 +1899,6 @@ export function writePlanSafely({ try { parentHandle = openPlanParent(repo, components); vaultHandle = openBackupVault(repo); - resolveAtomicMover(); const parentDevice = fs.fstatSync(parentHandle.fd, { bigint: true }).dev; const vaultDevice = fs.fstatSync(vaultHandle.fd, { bigint: true }).dev; if (parentDevice !== vaultDevice) { @@ -1622,19 +1909,11 @@ export function writePlanSafely({ testHooks?.afterParentOpen?.({ fd: parentHandle.fd, path: parentHandle.expectedPath }); validatePlanParent(parentHandle); validatePlanParent(vaultHandle); - finalPath = descriptorPath(parentHandle.fd, finalName); - originalDestination = openExistingPlanDestination(finalPath, shouldReplace); + finalRef = anchoredChild(parentHandle, finalName); + originalDestination = openExistingPlanDestination(finalRef, shouldReplace); tempName = `.gitnexus-plan-${process.pid}-${randomBytes(16).toString('hex')}.tmp`; - tempPath = descriptorPath(parentHandle.fd, tempName); - tempFd = fs.openSync( - tempPath, - fs.constants.O_RDWR | - fs.constants.O_CREAT | - fs.constants.O_EXCL | - fs.constants.O_NOFOLLOW | - (fs.constants.O_CLOEXEC ?? 0), - 0o600, - ); + tempRef = anchoredChild(parentHandle, tempName); + tempFd = createChild(tempRef, VERIFIED_CREATE_FLAGS, 0o600); writeAll(tempFd, contents); fs.fchmodSync(tempFd, 0o644); fs.fsyncSync(tempFd); @@ -1646,12 +1925,12 @@ export function writePlanSafely({ testHooks?.beforeRename?.({ fd: parentHandle.fd, path: parentHandle.expectedPath, - tempPath, + tempPath: tempRef.path, }); validatePlanParent(parentHandle); validatePlanParent(vaultHandle); validateOpenPlanDestination(originalDestination); - const tempPathStat = fs.lstatSync(tempPath, { bigint: true }); + const tempPathStat = lstatChild(tempRef); const currentTemp = hashOpenFile(tempFd, 'generated-plan temporary file'); if ( tempPathStat.isSymbolicLink() || @@ -1664,7 +1943,7 @@ export function writePlanSafely({ } if (shouldReplace) { - testHooks?.beforeBackupMove?.({ fd: parentHandle.fd, finalPath }); + testHooks?.beforeBackupMove?.({ fd: parentHandle.fd, finalPath: finalRef.path }); const originalLayer = hashOpenFile(originalDestination.fd, 'prior generated plan'); if (originalLayer.digest !== expectedDigest) { throw new Error( @@ -1673,7 +1952,7 @@ export function writePlanSafely({ } validatePlanParent(parentHandle); validateOpenPlanDestination(originalDestination); - inspectPlanDestination(finalPath, { + inspectPlanDestination(finalRef, { replace: true, expectedIdentity: originalDestination.identity, }); @@ -1691,20 +1970,20 @@ export function writePlanSafely({ ); throw new Error('Destination raced while the prior plan was moved into preservation'); } - if (lstatOptional(finalPath)) { + if (lstatAnchoredOptional(finalRef)) { throw new Error('Destination reappeared after the prior plan was preserved'); } } testHooks?.beforePublication?.({ fd: parentHandle.fd, - finalPath, - tempPath, + finalPath: finalRef.path, + tempPath: tempRef.path, replace: shouldReplace, }); validatePlanParent(parentHandle); validatePlanParent(vaultHandle); - const finalTempPathStat = fs.lstatSync(tempPath, { bigint: true }); + const finalTempPathStat = lstatChild(tempRef); const finalTemp = hashOpenFile(tempFd, 'generated-plan temporary file'); if ( finalTempPathStat.isSymbolicLink() || @@ -1715,19 +1994,25 @@ export function writePlanSafely({ ) { throw new Error('Generated-plan temporary path or content changed at publication'); } - atomicMoveNoReplace( - externalDescriptorPath(parentHandle.fd, tempName), - externalDescriptorPath(parentHandle.fd, finalName), - ); - if (lstatOptional(tempPath) || !lstatOptional(finalPath)) { + // link() reports the race itself; re-deriving that verdict from a later pair + // of stats would be both slower and weaker. + if (!publishNoReplace(tempRef, finalRef)) { throw new Error('Generated-plan publication was refused because the destination raced'); } + // link() creates a directory entry, so it needs the parent fsync that rename + // needed: the file's own bytes were fsynced through tempFd before this point, + // and this makes the name that now reaches them durable too. Skipping it is + // the step write-file-atomic omits and maildir, git and atomicwrites all + // mandate. + // + // Honest limitation: on macOS fsync is not a write barrier — the durable + // primitive there is fcntl(F_FULLFSYNC), which Node does not expose. A + // macOS plan write is therefore as durable as fsync makes it and no more. fs.fsyncSync(parentHandle.fd); - testHooks?.afterPublication?.({ fd: parentHandle.fd, finalPath }); - testHooks?.afterRename?.({ fd: parentHandle.fd, finalPath }); + testHooks?.afterPublication?.({ fd: parentHandle.fd, finalPath: finalRef.path }); validatePlanParent(parentHandle); validatePlanParent(vaultHandle); - validateCommittedPlan(finalPath, tempFd, expectedTemp, testHooks); + validateCommittedPlan(finalRef, tempFd, expectedTemp, testHooks); const receipt = { generated_plan_path: generatedPlan, bytes_written: contents.length }; if (priorBackup) receipt.prior_plan_backup_git_path = priorBackup.gitPath; return receipt; @@ -1848,6 +2133,11 @@ export function snapshotEvidence({ const headGuards = captureHeadGuards(repo); const dirty = initialDirty.records; const mutationGuards = []; + // Per-snapshot walk state: `absenceCache` owns every descriptor an absence + // anchor holds, deduplicated by repo-relative prefix and closed exactly once + // below; `guardedDirectories` keeps parent guarding to one stat per directory. + const absenceCache = new Map(); + const walkState = { absenceCache, guardedDirectories: new Set() }; try { testHooks?.afterAnchorCapture?.({ headCommit: head }); @@ -1862,7 +2152,9 @@ export function snapshotEvidence({ testHooks?.afterGitLayerLoad?.({ headCommit: head }); const globalEntries = [...dirty.values()] .filter((record) => record.path !== generatedPlan) - .map((record) => materializeRecord(repo, record, layers, mutationGuards, testHooks)); + .map((record) => + materializeRecord(repo, record, layers, mutationGuards, testHooks, walkState), + ); const citedEntries = [...normalizedCitations].sort(compareUtf8).map((repoPath) => { const status = dirty.get(repoPath) ?? { path: repoPath, @@ -1871,7 +2163,7 @@ export function snapshotEvidence({ rename_to: null, has_untracked: false, }; - const entry = materializeRecord(repo, status, layers, mutationGuards, testHooks); + const entry = materializeRecord(repo, status, layers, mutationGuards, testHooks, walkState); const present = Object.values(entry.object_kind).some((kind) => kind !== ABSENT); if (!present) entry.state = ABSENT; else if (entry.state === 'clean' && entry.object_kind.untracked !== ABSENT) { @@ -1906,21 +2198,13 @@ export function snapshotEvidence({ throw new Error(`${guard.absolute} changed before evidence materialization completed`); } } else if (guard.type === 'absence') { + // statIdentity is a strict superset of stableDirectoryIdentity on the + // same stat, so comparing both could only ever fire together. const parent = fs.fstatSync(guard.fd, { bigint: true }); - if ( - !parent.isDirectory() || - stableDirectoryIdentity(parent) !== guard.parentIdentity || - statIdentity(parent) !== guard.parentMutationIdentity - ) { + if (!parent.isDirectory() || statIdentity(parent) !== guard.parentMutationIdentity) { throw new Error(`Absence anchor changed for ${guard.repoPath}`); } - try { - fs.lstatSync(descriptorPath(guard.fd, guard.childName), { bigint: true }); - } catch (error) { - if (error?.code === 'ENOENT') continue; - throw error; - } - throw new Error(`${guard.repoPath} appeared before evidence materialization completed`); + anchoringBackend().verifyAbsentChild(guard); } } for (const guard of headGuards) verifyControlFile(guard); @@ -1955,12 +2239,10 @@ export function snapshotEvidence({ cited_path_manifest: citedEntries, }; } finally { - const closed = new Set(); - for (const guard of mutationGuards) { - if (guard.type !== 'absence' || closed.has(guard.fd)) continue; - closed.add(guard.fd); + // One entry per distinct anchored directory, so one close per descriptor. + for (const handle of absenceCache.values()) { try { - fs.closeSync(guard.fd); + fs.closeSync(handle.fd); } catch { // Preserve the primary snapshot result/error. } diff --git a/.github/workflows/ci-e2e.yml b/.github/workflows/ci-e2e.yml index b40ccc19f..a30371637 100644 --- a/.github/workflows/ci-e2e.yml +++ b/.github/workflows/ci-e2e.yml @@ -17,7 +17,7 @@ jobs: - uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0 with: persist-credentials: false - - uses: dorny/paths-filter@7b450fff21473bca461d4b92ce414b9d0420d706 # v3 + - uses: dorny/paths-filter@ceb8a2b8f2d89434be7ff52d3de7ec3738c5cc9d # v3 id: filter with: filters: | diff --git a/.github/workflows/ci-tests.yml b/.github/workflows/ci-tests.yml index 87d2c4377..f211b12c3 100644 --- a/.github/workflows/ci-tests.yml +++ b/.github/workflows/ci-tests.yml @@ -482,6 +482,14 @@ jobs: working-directory: gitnexus - name: Cross-language scope-capture fingerprint + scaling guards + # Runs even after an earlier guard fails (#2895). Every step here was + # fail-fast, so the FIRST failing --check aborted the job and every guard + # after it reported `skipped` — which reads identically to "nothing to do". + # Audited across 13 benchmark runs on #2856: the job succeeded zero times + # and the last two guards executed zero times for the life of the PR, while + # two reviews read the checks summary and saw nothing wrong. `!cancelled()` + # rather than `always()` so an explicit cancel still stops the job. + if: ${{ !cancelled() }} # Build-free: asserts emitScopeCaptures output is unchanged # (fingerprint) and stays linear (scaling < 1.5) for go/csharp/rust/php/ # ruby/cobol. Catches an O(n^2) re-regression without the worker pool. @@ -489,6 +497,7 @@ jobs: working-directory: gitnexus - name: Callable-value-flow target-index guards (#2693) + if: ${{ !cancelled() }} # Build-free: asserts buildGraphTargetIndex resolves an unchanged target # set (fingerprint), stays linear in def count, and that the #2693 # widened gate — which now considers VALUE bindings, a population that @@ -500,7 +509,19 @@ jobs: run: node --import tsx bench/callable-value-flow/measure.mjs --check working-directory: gitnexus + - name: Re-export closure scaling guards (#2864) + # Build-free: asserts buildReexportClosures stays linear in chain depth + # and within an absolute ceiling on a wide package corpus. #2864 changed + # this pass's input class from TypeScript barrels (a handful of shallow + # edges) to every module-level Python `from m import x`, which is where + # its two quadratic corners became reachable. The depth arm specifically + # guards MAX_VIA_LENGTH — the bound that was removed once already, in + # fc919ad6, and stayed invisible for as long as the input was shallow. + run: node --import tsx bench/finalize-reexport/measure.mjs --check + working-directory: gitnexus + - name: C++ qualified-namespace resolution guards (#2788) + if: ${{ !cancelled() }} # Build-free: asserts resolveCppQualifiedNamespaceMember resolves an # unchanged symbol set (fingerprint) and that per-call-site cost stays # independent of corpus size. Rationale and history: see the header of @@ -508,7 +529,120 @@ jobs: run: node --import tsx bench/cpp-qualified-ns/measure.mjs --check working-directory: gitnexus + - name: Import-target resolution guards (every registered language, #2877–#2909, PR #2911) + if: ${{ !cancelled() }} + # Build-free: runs EVERY import-target resolver registered in + # SCOPE_RESOLVERS — plus C# a second time WITH csproj configs, over the + # identical corpus, because the no-csproj arm returns before it can + # reach the leg #2902 indexed. One arm per registered language over ONE + # shared corpus, and no registered language ungated. That is enforced, + # not enumerated: measure.mjs derives its list from a LANG_REGISTRY + # table and its --check inventory arm reconciles that table against + # SCOPE_RESOLVERS in both directions, so a language roster typed out + # here would only be a second copy that can go stale — this one did. + # A C/C++ #include is an import site for this purpose and is gated like + # every other registered language (its headers arrive through + # resolutionConfig rather than allFilePaths, which is the one structural + # difference — see `newPass`). + # + # Asserts each returns an unchanged target set (a fingerprint per + # language AND per arm), that per-import cost stays independent of + # corpus size AND of path depth, that the absolute small-arm cost holds + # — a constant-factor regression that grows both scale arms equally + # passes every ratio — and that the per-pass index eight of them retain + # stays within an absolute byte ceiling. The corpus SHAPE is asserted + # too: a fingerprint alone cannot tell a legitimate resolution change + # from a corpus quietly shrunk below the size the timing arms need. + # + # Several arms exist because an arm that stops MEASURING otherwise + # passes. The heap arms drive real resolvers and carry a FLOOR as well + # as a ceiling: when buildSuffixIndex's suffix maps went lazy, four arms + # that called the builder directly read 0 B, and 0 B is under every + # ceiling. EVERY budget is checked for PRESENCE first, timing and heap + # alike, because `got > undefined` is false and `got < ceiling * + # undefined` is false too, so deleting a budget key deleted its gate — + # and the two heap scalars gate all eight heap arms at once. The heap + # arm's own corpus shape (its two file counts, its path depth and the + # probe it resolves) is asserted by the same loop as the timing arms, + # because those four decide WHAT it measures. And an inventory arm + # reconciles the bench's language table against SCOPE_RESOLVERS itself, + # so a newly registered resolver cannot ship ungated the way JavaScript + # did. + # + # The resolvers gated first were added as their own O(imports × files) + # scans were indexed away (Ruby rebuilt a suffix index per `require`; + # COBOL scanned twice per `COPY`), and the same corpus shape scores >3.3 + # against those pre-fix implementations. The rest were ungated until + # this PR, which is not a theoretical gap: PR #2911 found JavaScript + # reaching suffixResolve with no index at all — 25 972 µs per import at + # 8000 files, protected only by unit tests. This step is what stops the + # next one shipping. + # + # SCOPE: "independent of corpus size" holds for UNIQUE-LEAF layouts, + # where no two directories share a last segment and no two files share a + # basename — which is what the small/large/deep arms are, and where + # every index bucket holds exactly one entry. The `collide` arm runs the + # identical workload on the layout these languages are actually written + # in (svcN/internal, SrcN/Models, a repeated basename per package, four + # SPM modules instead of fifty); there the bucket grows with the file + # count by construction and go, csharp, dart, java, swift and c/cpp + # legitimately score 2.1–3.9, so that arm carries its own per-language + # budget. It is a scope limit, not a regression — the indexed code is + # still faster on that shape than the pre-change scan. Rust is the one + # language whose collide arm is NOT a shared-leaf layout: it probes + # candidate paths and is provably flat in the file count, so its arm is + # a deep module tree that varies `::` segment count instead — the axis + # its cost actually has. + # + # --expose-gc enables the retained-heap arm; --check REFUSES to run + # without it rather than passing with the memory gate silently skipped. + # ~44–45 s, which is essentially unchanged from the ~46 s it cost + # before: the timing phase did fall from 39.8 s to 28.7 s when the + # min-of-N estimator became per-language, but the inventory arm's one + # dynamic import (pipeline/registry.ts pulls in every registered + # provider) costs 6–10 s depending on the box and consumes almost all of + # that. Report mode, which does not load the registry, is the mode that + # got faster: ~33–35 s. Kept as-is because this job runs minutes clear + # of the sharded coverage job that gates the merge, so the seconds buy + # no merge latency — see COST in the bench header. The ts + # family (javascript/typescript/vue) is still the largest block, 8.8 s, + # because suffixResolve probes ~39 extensions per path part on a miss. + # If this ever has to shrink, drop collide/collide_large for typescript + # and vue (−3.9 s) — the only cut that removes near-duplicate work + # rather than coverage. N is 15 (matching bench/cfg) for every language + # whose cheapest arm is under 5 ms, because depth_ratio divides two + # sub-3 ms numbers and at 5 or 7 it tripped its own budget roughly 1 run + # in 20; the six languages whose cheapest arm is 20-28 ms drop to 7-8, + # where the measured overshoot is at most 6.3%. The estimator was fixed + # rather than the budget widened; distributions in _arms_note. + # The Kotlin arm here is a second corpus, not a replacement for the + # kotlin-import-target bench below, which carries tie-break probes (both + # file-set iteration orders, the four-tier cascade) this one does not. + # It sits with the other resolver-index guards rather than at the end of + # the job: parking a new gate last is not safety, it is the slot least + # likely to execute (#2895 measured the last two guards running zero + # times in 13 runs). #2899 landed the `if: ${{ !cancelled() }}` below, + # which is what makes position irrelevant — a failing step no longer + # aborts the ones after it. + # Rationale, budgets and the measured blind spot: see the header of + # measure.mjs and _blind_spot in baselines.json. + run: node --expose-gc --import tsx bench/import-target/measure.mjs --check + working-directory: gitnexus + + - name: Kotlin import-resolution identity + scaling guards + if: ${{ !cancelled() }} + # Build-free: asserts resolveKotlinImportTarget resolves an unchanged + # file set (fingerprint, in both file-set iteration orders — every + # tie-break in that resolver is expressed only through iteration order) + # and that per-import cost stays independent of workspace size. The + # pre-index implementation scores 3.737 on this corpus against 0.99 for + # the index, so the gate separates them by a wide margin. Rationale and + # history: see the header of bench/kotlin-import-target/measure.mjs. + run: node --import tsx bench/kotlin-import-target/measure.mjs --check + working-directory: gitnexus + - name: Receiver-resolution drop guards + if: ${{ !cancelled() }} # NOT build-free: this one runs the real pipeline, so it needs dist/ # (the setup action above builds). ~2m15s. # @@ -534,6 +668,7 @@ jobs: working-directory: gitnexus - name: Scope-emission guards (#2699) + if: ${{ !cancelled() }} # Build-free: asserts the JS/TS scope set is unchanged. Block scopes are # what make `let`/`const` in sibling blocks distinct bindings, but a # scope per `statement_block` triples the count and deepens every @@ -546,6 +681,7 @@ jobs: working-directory: gitnexus - name: CFG construction time / disk / memory guards (#2081 M1) + if: ${{ !cancelled() }} # Build-free: asserts collectFunctionCfgs output is unchanged # (fingerprint) and that wall-time, cfgSideChannel disk bytes, AND # retained heap all stay sub-quadratic for the straight-line / @@ -556,6 +692,7 @@ jobs: working-directory: gitnexus - name: Emit-persistence throughput / byte-identity guards (#2203) + if: ${{ !cancelled() }} # Build-free: asserts streamAllCSVsToDisk output is byte-identical # (order-independent CSV-line fingerprint — the #2203 U2/U3 emit # optimisations must not change graph content) and that emit wall-time @@ -565,6 +702,7 @@ jobs: working-directory: gitnexus - name: Streaming PDG-emit byte-identity / bounded-RSS guards (#2202) + if: ${{ !cancelled() }} # Build-free: asserts the streaming PdgEmitSink emits a CSV row SET # byte-identical to the whole-graph streamAllCSVsToDisk emit, AND that # the in-memory graph retains zero BasicBlock nodes (the O(chunk) peak-RSS @@ -574,6 +712,7 @@ jobs: working-directory: gitnexus - name: Cross-language pipeline benchmarks (GITNEXUS_BENCH, serial) + if: ${{ !cancelled() }} # cpp-adl-benchmark.test.ts is not a `*-pipeline-benchmark.test.ts` but # belongs here for the same reason: it is skipIf-gated on GITNEXUS_BENCH, # so it had never run in CI and the PR #1990 ADL emit-scaling guard it diff --git a/.github/workflows/codeql.yml b/.github/workflows/codeql.yml index d47d25a78..16d3e15cc 100644 --- a/.github/workflows/codeql.yml +++ b/.github/workflows/codeql.yml @@ -48,7 +48,7 @@ jobs: persist-credentials: false - name: Initialize CodeQL - uses: github/codeql-action/init@e4fba868fa4b1b91e1fdab776edc8cfbe6e9fb81 # v4.37.3 + uses: github/codeql-action/init@5595ccaf912efad79be6eef63a5619ff05969be3 # v4.37.6 with: languages: ${{ matrix.language }} queries: security-and-quality @@ -73,6 +73,6 @@ jobs: - '**/test/**/fixtures/**' - name: Perform CodeQL Analysis - uses: github/codeql-action/analyze@e4fba868fa4b1b91e1fdab776edc8cfbe6e9fb81 # v4.37.3 + uses: github/codeql-action/analyze@5595ccaf912efad79be6eef63a5619ff05969be3 # v4.37.6 with: category: '/language:${{ matrix.language }}' diff --git a/.github/workflows/scorecard.yml b/.github/workflows/scorecard.yml index f511d4d51..04d722160 100644 --- a/.github/workflows/scorecard.yml +++ b/.github/workflows/scorecard.yml @@ -53,6 +53,6 @@ jobs: retention-days: 5 - name: Upload to Security tab - uses: github/codeql-action/upload-sarif@e4fba868fa4b1b91e1fdab776edc8cfbe6e9fb81 # v4.37.3 + uses: github/codeql-action/upload-sarif@5595ccaf912efad79be6eef63a5619ff05969be3 # v4.37.6 with: sarif_file: results.sarif diff --git a/.github/workflows/trivy.yml b/.github/workflows/trivy.yml index ecb394c99..1e2b2d1e3 100644 --- a/.github/workflows/trivy.yml +++ b/.github/workflows/trivy.yml @@ -76,7 +76,7 @@ jobs: exit-code: '0' - name: Upload to Security tab - uses: github/codeql-action/upload-sarif@e4fba868fa4b1b91e1fdab776edc8cfbe6e9fb81 # v4.37.3 + uses: github/codeql-action/upload-sarif@5595ccaf912efad79be6eef63a5619ff05969be3 # v4.37.6 with: sarif_file: trivy-${{ matrix.image.name }}.sarif category: trivy-${{ matrix.image.name }} diff --git a/.github/workflows/workflow-lint.yml b/.github/workflows/workflow-lint.yml index e013a0bf6..f771406a0 100644 --- a/.github/workflows/workflow-lint.yml +++ b/.github/workflows/workflow-lint.yml @@ -76,7 +76,7 @@ jobs: continue-on-error: true - name: Upload SARIF - uses: github/codeql-action/upload-sarif@e4fba868fa4b1b91e1fdab776edc8cfbe6e9fb81 # v4.37.3 + uses: github/codeql-action/upload-sarif@5595ccaf912efad79be6eef63a5619ff05969be3 # v4.37.6 with: sarif_file: zizmor.sarif category: zizmor diff --git a/AGENTS.md b/AGENTS.md index 032a3a6b6..c83a2e909 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -111,15 +111,16 @@ mirror. `gitnexus/test/unit/shipped-skills-sync.test.ts` guards the copies. Toke # GitNexus — Code Intelligence -This project is indexed by GitNexus as **GitNexus** (248612 symbols, 565510 relationships, 918 execution flows). Use GitNexus graph tools to understand code, assess impact, and navigate safely. +This project is indexed by GitNexus as **GitNexus** (248612 symbols, 565510 relationships, 918 execution flows). -> Index stale? Run `node .gitnexus/run.cjs analyze` from the project root — it auto-selects an available runner. No `.gitnexus/run.cjs` yet? Bootstrap with `npx`, `bunx`, or `pnpm dlx` — e.g. `bunx gitnexus@latest analyze` (npm 11 npx crash; #1939). +> Index stale? Run `node .gitnexus/run.cjs analyze --index-only` from the project root — it auto-selects an available runner. No `.gitnexus/run.cjs` yet? Bootstrap with `npx`, `bunx`, or `pnpm dlx` — e.g. `bunx gitnexus@latest analyze` (npm 11 npx crash; #1939). ## Always Do - **MUST run impact analysis before editing.** Use `impact({target: "symbolName", direction: "upstream"})` (MCP) or `node .gitnexus/run.cjs impact "symbolName" --direction upstream --repo .` (CLI fallback); report callers, processes, and risk. Never substitute grep for graph analysis. For unified PDG impact, add `mode: "pdg"` with optional `line: ` — it returns statement-level `affectedStatements` over CDG + REACHING_DEF and inter-procedural symbols in `interproceduralByDepth`/`byDepth`; no-layer/degraded PDG results are UNKNOWN-risk notes (`--pdg` layer). CLI equivalent: `node .gitnexus/run.cjs impact "symbolName" --direction upstream --mode pdg --line --repo .`. -- **MUST analyze graph changes before committing.** Use `detect_changes({scope: "all"})` (MCP) or `node .gitnexus/run.cjs detect-changes --scope all --repo .` (CLI fallback). For regression review: `detect_changes({scope: "compare", base_ref: "main"})` or `node .gitnexus/run.cjs detect-changes --scope compare --base-ref "main" --repo .`. +- **MUST analyze graph changes before committing.** Use `detect_changes({scope: "all"})` (MCP) or `node .gitnexus/run.cjs detect-changes --scope all --repo .` (CLI fallback). `partial: true` or `truncated: true` is not a clean check — a zero means unseen, not unaffected; re-run it. For regression review: `detect_changes({scope: "compare", base_ref: "main"})` or `node .gitnexus/run.cjs detect-changes --scope compare --base-ref "main" --repo .`. - **MUST warn the user** if impact analysis returns HIGH or CRITICAL risk before proceeding with edits. +- **MUST treat `risk: UNKNOWN` as unresolved, not as low.** An empty caller set is not evidence the symbol is unused — it can also mean the callers are not resolvable by the index (plain-object property access, dynamic dispatch, cross-language calls). `impact` pairs `UNKNOWN` with a `riskNote` saying so. Confirm with a text search before treating the symbol as safe to change or delete; do not proceed on the strength of a zero. - When exploring unfamiliar code, use `query({search_query: "concept"})` to find execution flows instead of grepping. It returns process-grouped results ranked by relevance. - When you need full context on a specific symbol — callers, callees, which execution flows it participates in — use `context({name: "symbolName"})`. - For security review, `explain({target: "fileOrSymbol"})` lists taint findings (source→sink flows; needs `analyze --pdg`). @@ -128,7 +129,7 @@ This project is indexed by GitNexus as **GitNexus** (248612 symbols, 565510 rela ## Never Do - NEVER edit a function, class, or method before MCP/CLI impact analysis. -- NEVER ignore HIGH or CRITICAL risk warnings from impact analysis. +- NEVER ignore HIGH or CRITICAL risk warnings from impact analysis, and never read `UNKNOWN` as an all-clear — it means the walk could not answer, which is the one verdict that requires confirming by other means. - NEVER rename symbols with find-and-replace — use `rename` which understands the call graph. - NEVER commit before MCP/CLI graph change analysis. diff --git a/ARCHITECTURE.md b/ARCHITECTURE.md index 42160627e..138530d24 100644 --- a/ARCHITECTURE.md +++ b/ARCHITECTURE.md @@ -98,7 +98,7 @@ scan → structure → [springConfig, markdown, cobol] → parse → [routes, to | `markdown` | `markdown.ts` | `structure` | Section nodes, cross-link edges from .md/.mdx | | `cobol` | `cobol.ts` | `structure` | COBOL program/paragraph/section nodes (regex, no tree-sitter) | | `parse` | `parse.ts` + `parse-impl.ts` | `structure`, `markdown`, `cobol` | Symbol nodes, IMPORTS/CALLS/EXTENDS edges, extracted routes/tools/ORM queries | -| `routes` | `routes.ts` | `parse` | Route nodes + HANDLES_ROUTE edges (Next.js, Expo, PHP, decorators) | +| `routes` | `routes.ts` | `parse` | Route nodes + HANDLES_ROUTE edges (Next.js, Expo, PHP, decorators, and JS/TS dispatch guards — see below) | | `tools` | `tools.ts` | `parse` | Tool nodes + HANDLES_TOOL edges | | `orm` | `orm.ts` | `parse` | QUERIES edges (Prisma, Supabase) | | `crossFile` | `cross-file.ts` + `cross-file-impl.ts` | `parse`, `routes`, `tools`, `orm` | Cross-file type propagation in topological import order | @@ -164,6 +164,48 @@ export const myPhase: PipelinePhase = { }; ``` +### Where routes come from + +`route-extractors/` holds four independent ways a route can be discovered, all +converging on the routes phase's `(method, url)` registry: + +| Source | Shape | Examples | +| --- | --- | --- | +| Filesystem convention | path → URL, no parsing | Next.js `app/`, Expo, PHP | +| Single-file framework route | `isRouteFile` + worker extraction | Laravel `routes/*.php` | +| Cross-file framework route | `discoverRootRouteFiles` + `extractRoutes` | Django `urlpatterns` | +| AST-level route in a normal file | `extractDecoratorRoutes` | Spring, FastAPI, NestJS, **JS/TS dispatch guards** | + +The last row is the one whose name undersells it. A route is DECLARED by a +decorator, but it can also be **inferred** from a raw `node:http` server's own +dispatch — `if (req.method === 'GET' && pathname === '/api/x')` is a route with +a path, a verb and a handler, and nothing else in the pipeline could see it. +`route-extractors/dispatch-guard.ts` reads that shape; the transport, dedup and +handler resolution are shared with decorator routes, and +`ExtractedDecoratorRoute.source` carries the provenance difference through to +the `HANDLES_ROUTE` edge. + +That extractor is deliberately **precision-weighted**: `route_map` presents its +output as fact, so a `startsWith` namespace test, a bare `pathname === '/'` +without a verb, and any regex it cannot translate exactly are all dropped rather +than guessed at. A missing route is a coverage limit; an invented one is a lie. + +Two rules there need more than one comparison to decide, and are worth knowing +about before changing either: + +- **Same-file constant folding.** `` pathname === `${basePath}/rules` `` is + common enough that refusing it loses whole route modules — and loses them + invisibly, since a module with unfoldable paths and a module with no routes + produce the same empty answer. Folding is same-file, string literals only, one + alias hop, and refuses on ambiguity (a name declared twice with different + values is dropped, never guessed). +- **Whole-repo reconciliation** (`reconcileDispatchGuardRoutes`, applied in the + routes phase). A split route table — one module listing every path it + recognises so the dispatcher can 404 early, handlers in others — otherwise + lists every route twice, once verb-less with the table as its "handler". It + applies to dispatch-guard routes only: a framework route with no verb is + method-agnostic *by declaration*, which is a fact, not a weaker observation. + --- ## Semantic model @@ -214,6 +256,9 @@ Language-agnostic scope-resolution resolver. This is the resolution path for eve │ emitReferencesViaLookup ── uses handledSites + deferred-site skip set │ emitPropertyDispatchCalls ── registration USES + conservative CALLS │ emitCallableValueFlow ── assigned/passed callable invocation CALLS + │ emitImportedValueReferences ── cross-file value reads via finalized imports + │ emitUniqueNamePropertyAccesses ── LAST-RESORT property reads by name, + │ narrowed same-file → direct-import, refusing to choose otherwise │ emitImportEdges ▼ KnowledgeGraph (IMPORTS / CALLS / ACCESSES / INHERITS / USES) @@ -232,6 +277,12 @@ The solver is flow-insensitive but bounded: dependency-indexed work items rerun Property-key dispatch remains a separate conservative fallback. Its per-key fan-out cap is 32; capped keys synthesize no partial calls and are reported at warning level with language, skipped-key count, dropped key names (bounded), and cap; the count also travels in `RunScopeResolutionStats.propertyDispatchSkippedKeys`. +Interface-dispatch fan-out walks the subtype closure of the receiver's interface and is **generic-instantiation aware** (#2912): a call through `IValidator` must not reach an implementor of `IValidator`, which shares its declaration and therefore its subtype list. Each heritage clause's arguments reach resolution by one of three routes — read off the `@reference.inherits` anchor's own spelling where that anchor spans the whole base (most languages, no query change), through the `@reference.type-arguments` sub-tag where the anchor is the bare name and moving it would renumber inheritance edge ids (Rust `impl T for S`, Dart `extends`), or on a heritage MARKER payload for clauses that never become reference sites (Dart `implements`/`with`). Whichever pass emits the edge records the pair through one sink: `preEmitInheritanceEdges` for heritage clauses, `ScopeResolver.emitHeritageEdges` for the rest. + +The walk then carries a substitution: a subtype's own type parameters bind to the receiver's arguments, so `class Wrapper : IValidator` stays reachable from every instantiation while `class IntValidator : IValidator` is pruned from the `string` one. Receiver arguments come from the declared type (Case 4), a class-level field's declared type (Case 6), or — for a compound receiver such as `this._repo` — the spelling the compound fold typed that position from, reported back through `recordReceiverType` and accepted only when it names the class the fold returned. + +The filter prunes only on positive evidence: an unknown instantiation on either side, an argument list whose arity does not line up, a name that may be a type variable the language's captures never recorded, or an unresolved spelling whose simple name matches all keep the target. A type parameter of the declaration ENCLOSING either side is recognised as such and never compared — `void Run(IValidator v)` writes a receiver with no known instantiation, so it keeps the unfiltered fan-out. That recognition is what generic METHODS now carry `@declaration.type-parameters` for in C#, Java and Kotlin (TypeScript already did): without it an unbounded `T` grounds to nothing and a bounded one grounds to its BOUND, and both compare unequal to an implementor's concrete argument. Languages that capture neither type arguments nor type parameters therefore emit exactly the pre-#2912 fan-out. The fan-out cap (32, `GITNEXUS_MAX_INTERFACE_DISPATCH_FANOUT`) and its skipped-target reporting are unchanged and apply after filtering. Note the fan-out itself still fires only for a receiver whose folded type is an `Interface` symbol, so a Rust `Trait` or a Dart abstract `Class` receiver emits no secondary targets to filter in the first place. + Standalone (regex-based) providers such as COBOL participate via `ScopeResolver.scopeResolutionEdgeMode: 'callable-flow-only'`: `runScopeResolution` runs for them, but every ordinary emission path — heritage, interface implementations, receiver-bound, free-call fallback, reference/import edges, post-resolution hooks — is gated off, so their legacy phase (e.g. `cobolPhase`) remains the sole owner of structural edges and the callable solver's `CALLS` are purely additive. A callable-flow-only provider whose files emitted no callable facts exits early, before finalize, keeping the opt-in proportional to source scanning. ### Receiver chains and the drop census (#2766) diff --git a/CLAUDE.md b/CLAUDE.md index af4473f38..55b84d583 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -62,15 +62,16 @@ See the ` … ` block in **[AGENTS.m # GitNexus — Code Intelligence -This project is indexed by GitNexus as **GitNexus** (248612 symbols, 565510 relationships, 918 execution flows). Use GitNexus graph tools to understand code, assess impact, and navigate safely. +This project is indexed by GitNexus as **GitNexus** (248612 symbols, 565510 relationships, 918 execution flows). -> Index stale? Run `node .gitnexus/run.cjs analyze` from the project root — it auto-selects an available runner. No `.gitnexus/run.cjs` yet? Bootstrap with `npx`, `bunx`, or `pnpm dlx` — e.g. `bunx gitnexus@latest analyze` (npm 11 npx crash; #1939). +> Index stale? Run `node .gitnexus/run.cjs analyze --index-only` from the project root — it auto-selects an available runner. No `.gitnexus/run.cjs` yet? Bootstrap with `npx`, `bunx`, or `pnpm dlx` — e.g. `bunx gitnexus@latest analyze` (npm 11 npx crash; #1939). ## Always Do - **MUST run impact analysis before editing.** Use `impact({target: "symbolName", direction: "upstream"})` (MCP) or `node .gitnexus/run.cjs impact "symbolName" --direction upstream --repo .` (CLI fallback); report callers, processes, and risk. Never substitute grep for graph analysis. For unified PDG impact, add `mode: "pdg"` with optional `line: ` — it returns statement-level `affectedStatements` over CDG + REACHING_DEF and inter-procedural symbols in `interproceduralByDepth`/`byDepth`; no-layer/degraded PDG results are UNKNOWN-risk notes (`--pdg` layer). CLI equivalent: `node .gitnexus/run.cjs impact "symbolName" --direction upstream --mode pdg --line --repo .`. -- **MUST analyze graph changes before committing.** Use `detect_changes({scope: "all"})` (MCP) or `node .gitnexus/run.cjs detect-changes --scope all --repo .` (CLI fallback). For regression review: `detect_changes({scope: "compare", base_ref: "main"})` or `node .gitnexus/run.cjs detect-changes --scope compare --base-ref "main" --repo .`. +- **MUST analyze graph changes before committing.** Use `detect_changes({scope: "all"})` (MCP) or `node .gitnexus/run.cjs detect-changes --scope all --repo .` (CLI fallback). `partial: true` or `truncated: true` is not a clean check — a zero means unseen, not unaffected; re-run it. For regression review: `detect_changes({scope: "compare", base_ref: "main"})` or `node .gitnexus/run.cjs detect-changes --scope compare --base-ref "main" --repo .`. - **MUST warn the user** if impact analysis returns HIGH or CRITICAL risk before proceeding with edits. +- **MUST treat `risk: UNKNOWN` as unresolved, not as low.** An empty caller set is not evidence the symbol is unused — it can also mean the callers are not resolvable by the index (plain-object property access, dynamic dispatch, cross-language calls). `impact` pairs `UNKNOWN` with a `riskNote` saying so. Confirm with a text search before treating the symbol as safe to change or delete; do not proceed on the strength of a zero. - When exploring unfamiliar code, use `query({search_query: "concept"})` to find execution flows instead of grepping. It returns process-grouped results ranked by relevance. - When you need full context on a specific symbol — callers, callees, which execution flows it participates in — use `context({name: "symbolName"})`. - For security review, `explain({target: "fileOrSymbol"})` lists taint findings (source→sink flows; needs `analyze --pdg`). @@ -79,7 +80,7 @@ This project is indexed by GitNexus as **GitNexus** (248612 symbols, 565510 rela ## Never Do - NEVER edit a function, class, or method before MCP/CLI impact analysis. -- NEVER ignore HIGH or CRITICAL risk warnings from impact analysis. +- NEVER ignore HIGH or CRITICAL risk warnings from impact analysis, and never read `UNKNOWN` as an all-clear — it means the walk could not answer, which is the one verdict that requires confirming by other means. - NEVER rename symbols with find-and-replace — use `rename` which understands the call graph. - NEVER commit before MCP/CLI graph change analysis. diff --git a/GUARDRAILS.md b/GUARDRAILS.md index 3fe36a875..e157ade1e 100644 --- a/GUARDRAILS.md +++ b/GUARDRAILS.md @@ -31,7 +31,7 @@ Format: **Trigger → Instruction → Reason**. Append new Signs when the same m ### Stale graph after edits - **Trigger:** MCP warns index is behind `HEAD`, or search doesn't match latest commit. -- **Do:** `npx gitnexus analyze` (plus `--embeddings` if used). Runs incrementally by default — the pipeline parses every file every run (cross-file resolution requires it), but tree-sitter dispatch is skipped for unchanged file chunks via the content-addressed cache, and only changed-file rows (plus their importers, transitively) are rewritten in LadybugDB. When the effective write set exceeds ~50% of the repo's files (minimum 50 files), the run transparently switches to the full wipe + bulk-COPY write plan and logs "switching to a full DB write" — expected behavior, not a bug, and file-level bookkeeping stays incremental. +- **Do:** `npx gitnexus analyze` (plus `--embeddings` if used). Runs incrementally by default — the pipeline parses every file every run (cross-file resolution requires it), but tree-sitter dispatch is skipped for unchanged file chunks via the content-addressed cache, and only changed-file rows (plus their importers, transitively) are rewritten in LadybugDB. When the effective write set exceeds ~50% of the repo's files (minimum 50 files), the run transparently switches to the full wipe + bulk-COPY write plan and logs "switching to a full DB write" — expected behavior, not a bug, and file-level bookkeeping stays incremental. That same line also appears — regardless of write-set size, even for a one-file change — when a LadybugDB extension the existing index depends on cannot load on this machine (VECTOR, #2623; FTS, #2841), because a DB carrying those indexes refuses all row-level DML until the extension is loaded; run `gitnexus doctor` for live extension status and re-run with `GITNEXUS_LBUG_EXTENSION_INSTALL=auto` (with network access) to allow one bounded install attempt. The rebuild is one-shot: it clears the indexes, so the next run goes back to the incremental plan. - **Why:** Tools query LadybugDB from last analyze; git changes are invisible until re-indexed. ### Index seems corrupt or "incremental" is misbehaving @@ -52,6 +52,12 @@ Format: **Trigger → Instruction → Reason**. Append new Signs when the same m - **Do:** Re-run plain `npx gitnexus analyze` — no `--embeddings` flag needed. A retained `embeddingCheckpoint` in the index metadata forces embedding generation for exactly the pending nodes regardless of flags, and clears once they succeed. `--drop-embeddings` abandons the pending nodes instead of retrying them; `--force` also discards the checkpoint (with a warning) and rebuilds without resuming it. - **Why:** A long analyze run against a flaky HTTP embedding endpoint tolerates bounded sub-batch failures instead of aborting the whole run: it deletes the affected nodes' embedding rows (so they hold zero rows, never a partial set) and records those nodes as pending in `embeddingCheckpoint`. `stats.embeddings` stays an honest, non-zero count of everything that did succeed, so this state never trips the "Embeddings vanished" Sign above — `embedding-checkpoint-pending` is the only reliable signal. +### Analyze reports INCOMPLETE with a collapsed graph write + +- **Trigger:** `npx gitnexus status` reports `incompleteReasons: ["graph-write-collapsed"]`; the analyze summary printed `Repository indexed INCOMPLETELY` naming an expected and a persisted relationship count, and the CLI exited non-zero. +- **Do:** Re-run `npx gitnexus analyze --force`. If it recurs, check free disk space on the volume holding `.gitnexus/`, confirm no second `analyze` is running against the same repo (both stage through `.gitnexus/csv`), then run `npx gitnexus doctor`. +- **Why:** The run finished and wrote metadata, but far fewer relationships are readable back than the pipeline produced. Nothing throws: the DB holds rows and the metadata is valid, so every query answers with missing edges rather than an error — a confident empty answer, which is worse than a failure because it looks like a result. Unlike `incremental-in-progress` and `embedding-checkpoint-pending`, which describe a run that did what it said and left work for next time, this one means most of your edges are gone, so it is the one incomplete reason that also fails the exit code. The check compares in-memory totals (including rows streamed out of the heap) against the post-write count, refuses to answer when the count cannot be read, and is skipped on incremental runs where whole-scope counts are not comparable. + ### MCP lists no repos - **Trigger:** MCP stderr says no indexed repos. diff --git a/MIGRATION.md b/MIGRATION.md index 9c6c1de2d..f63d18c87 100644 --- a/MIGRATION.md +++ b/MIGRATION.md @@ -106,6 +106,19 @@ Running `npx gitnexus analyze` writes both `gitnexus.json` and `meta.json` with identical content. A pre-existing repo that only has `meta.json` gets `gitnexus.json` bootstrapped from it on the first run. +### Process ids are not stable across this release + +`Process` ids are positional (`proc__`), and this release changes +both which execution flows are detected and the order they are selected in: +tracing is depth-first, sibling branches follow source order, and selection +round-robins across terminals so one flow cannot take every slot. A given +`proc_7_handle` before the upgrade is not the same flow afterwards. + +Nothing in GitNexus persists or joins on a raw process id across a re-index — +the MCP resource keys by label — so this is one-time index churn rather than a +broken consumer. If you have external tooling that stored a process id, re- +resolve it by label after the next analyze. + ### What about rollback? Downgrading to an older GitNexus version is safe: `meta.json` is always diff --git a/RUNBOOK.md b/RUNBOOK.md index e34d20c2b..0f5c8b7bb 100644 --- a/RUNBOOK.md +++ b/RUNBOOK.md @@ -66,6 +66,16 @@ npx gitnexus analyze No `--embeddings` flag needed — a retained checkpoint forces embedding generation for the pending nodes regardless of flags, and clears once they succeed. `--drop-embeddings` abandons the pending nodes instead of retrying them; `--force` also discards the checkpoint (with a warning) and rebuilds without resuming it. +**Collapsed graph write (analyze exits NON-ZERO and says INCOMPLETE):** A run can finish writing metadata while only a fraction of the relationships it produced are readable back from the index — edges collapsing to a small share of what was built, or a `CodeRelation` table that never materialized (which reads as a persisted count of zero). Because the metadata IS written and the DB does hold rows, nothing looks broken: queries answer with missing edges rather than an error, which is a confident empty answer rather than a failure. `npx gitnexus status` reports `incompleteReasons: ["graph-write-collapsed"]`, the analyze summary prints `Repository indexed INCOMPLETELY` with the expected and persisted counts, and the CLI exits non-zero so automation is not told an unusable index is fine. + +Recovery is a full rebuild: + +```bash +npx gitnexus analyze --force +``` + +If it recurs, the cause is almost always environmental rather than a code defect: check free disk space on the volume holding `.gitnexus/`, make sure no second `analyze` is running against the same repo (both use `.gitnexus/csv` for staging), then run `npx gitnexus doctor`. The check compares in-memory relationship totals (including streamed rows) against what the DB hands back, and is deliberately skipped on incremental runs, where the two counts are not comparable. + **Large repos:** Analyze may skip or limit embedding work when node counts are very high; watch CLI output. --- diff --git a/gitnexus-claude-plugin/hooks/gitnexus-hook.js b/gitnexus-claude-plugin/hooks/gitnexus-hook.js index a53d79f29..238438455 100644 --- a/gitnexus-claude-plugin/hooks/gitnexus-hook.js +++ b/gitnexus-claude-plugin/hooks/gitnexus-hook.js @@ -543,7 +543,7 @@ function handlePostToolUse(input) { // If HEAD matches last indexed commit, no reindex needed if (currentHead && currentHead === lastCommit) return; - const analyzeCmd = formatAnalyzeCommand({ embeddings: hadEmbeddings }); + const analyzeCmd = formatAnalyzeCommand({ embeddings: hadEmbeddings, indexOnly: true }); sendHookResponse( 'PostToolUse', `GitNexus index is stale (last indexed: ${lastCommit ? lastCommit.slice(0, 7) : 'never'}). ` + diff --git a/gitnexus-claude-plugin/hooks/resolve-analyze-cmd.cjs b/gitnexus-claude-plugin/hooks/resolve-analyze-cmd.cjs index 56f5235fb..c74f03f5d 100644 --- a/gitnexus-claude-plugin/hooks/resolve-analyze-cmd.cjs +++ b/gitnexus-claude-plugin/hooks/resolve-analyze-cmd.cjs @@ -276,7 +276,13 @@ function formatBunxCommand(gitnexusArgs) { } function formatAnalyzeCommand(options = {}, deps = {}) { - const suffix = options.embeddings ? ' --embeddings' : ''; + // `--index-only` is what a routine "your index is stale" nudge wants: it + // reindexes without rewriting AGENTS.md / CLAUDE.md / skills, so an agent + // following the nudge on every commit cannot churn the tracked agent guides + // (#2907). Callers that actually want the docs refreshed omit it. + const suffix = `${options.indexOnly ? ' --index-only' : ''}${ + options.embeddings ? ' --embeddings' : '' + }`; // Keep the stale-index hook budget tight by querying each tool at most once. // The memoized `probe` is a spawn-free PATH scan (resolveOnPath) shared with // resolveInvocationMode, so `gitnexus` is scanned only once and no subprocess diff --git a/gitnexus-claude-plugin/skills/gitnexus-cli/SKILL.md b/gitnexus-claude-plugin/skills/gitnexus-cli/SKILL.md index 91c1ae992..9c7a1b599 100644 --- a/gitnexus-claude-plugin/skills/gitnexus-cli/SKILL.md +++ b/gitnexus-claude-plugin/skills/gitnexus-cli/SKILL.md @@ -60,7 +60,7 @@ Generates repository documentation from the knowledge graph using an LLM. Requir | Flag | Effect | |------|--------| | `--force` | Force full regeneration, also required to re-gerenate an existing wiki in a different language | -| `--model ` | LLM model (default: minimax/minimax-m2.5) | +| `--model ` | LLM model (default: MiniMax-M3) | | `--base-url ` | LLM API base URL | | `--api-key ` | LLM API key | | `--concurrency ` | Parallel LLM calls (default: 3) | diff --git a/gitnexus-claude-plugin/skills/gitnexus-impact-analysis/SKILL.md b/gitnexus-claude-plugin/skills/gitnexus-impact-analysis/SKILL.md index 0b81795de..ee1cd3496 100644 --- a/gitnexus-claude-plugin/skills/gitnexus-impact-analysis/SKILL.md +++ b/gitnexus-claude-plugin/skills/gitnexus-impact-analysis/SKILL.md @@ -53,6 +53,14 @@ description: "Use when the user wants to know what will break if they change som | 5-15 symbols, 2-5 processes | MEDIUM | | >15 symbols or many processes | HIGH | | Critical path (auth, payments) | CRITICAL | +| **Zero callers found** | **UNKNOWN** | + +`UNKNOWN` is not a low rung on this scale — it means the walk could not answer. +An empty caller set is equally consistent with "genuinely unused" and "the +callers are not resolvable by the index" (plain-object property access, dynamic +dispatch, cross-language calls), so few-callers ⇒ LOW does **not** apply. The +result carries a `riskNote` saying so. Confirm with a text search before +treating the symbol as safe to change or delete. ## Tools @@ -84,6 +92,11 @@ detect_changes({scope: "all"}) → Risk: MEDIUM ``` +`partial: true` (a graph query failed) or `truncated: true` (the changed-symbol +listing was capped) means the result is short of the truth, and reads like +`UNKNOWN` above: a zero there means unseen, not unaffected. Re-run it rather +than tick the pre-commit check. + ## Example: "What breaks if I change validateUser?" ``` diff --git a/gitnexus-claude-plugin/skills/gitnexus-plan/README.md b/gitnexus-claude-plugin/skills/gitnexus-plan/README.md index f7fe58ab9..153374bb7 100644 --- a/gitnexus-claude-plugin/skills/gitnexus-plan/README.md +++ b/gitnexus-claude-plugin/skills/gitnexus-plan/README.md @@ -124,12 +124,17 @@ phase that needs them. statement-level claims (never reconstructs fake edges). - No GitNexus at all → fallback mode: targeted grep/read exploration, findings labelled **source-derived**, with a recommendation to index. -- Reading or publishing a plan requires Linux `/proc/self/fd`, `O_DIRECTORY`, - and `O_NOFOLLOW`; publication also requires a validated absolute Python 3 - PATH candidate with libc `renameat2(RENAME_NOREPLACE)` support, a - writable target repository, and a shared filesystem for the plan and - Git-admin vault. The writer fails closed when those guarantees are - unavailable; it never redirects the plan elsewhere. +- Reading or publishing a plan requires `O_DIRECTORY` and `O_NOFOLLOW`, plus + `/proc/self/fd` on Linux; every other platform is refused. No interpreter is + spawned and no native code is loaded. Publication is `link(2)`, which fails + rather than replaces when the destination name is taken. Linux resolves every + name against a held descriptor, so a parent swapped mid-write cannot redirect + the operation; macOS has no equivalent path and instead pins each directory + with an open descriptor and re-proves the chain either side of every step, + which detects such a swap and aborts. Publishing also needs a writable target + repository and a shared filesystem for the plan and Git-admin vault. The + writer fails closed when those guarantees are unavailable; it never redirects + the plan elsewhere. ## Limitations diff --git a/gitnexus-claude-plugin/skills/gitnexus-plan/references/evidence-provenance.md b/gitnexus-claude-plugin/skills/gitnexus-plan/references/evidence-provenance.md index c686599da..3df5a046d 100644 --- a/gitnexus-claude-plugin/skills/gitnexus-plan/references/evidence-provenance.md +++ b/gitnexus-claude-plugin/skills/gitnexus-plan/references/evidence-provenance.md @@ -98,8 +98,11 @@ excluded. ## Safe existing-plan read contract -`read-plan` fails closed unless Linux `/proc/self/fd`, `O_DIRECTORY`, and -`O_NOFOLLOW` are available. It resolves the exact Git top-level, opens the +`read-plan` fails closed unless the host platform can resolve names against a +held directory descriptor: Linux `/proc/self/fd` with `O_DIRECTORY` and +`O_NOFOLLOW`, or macOS `O_DIRECTORY`/`O_NOFOLLOW`. Every other platform is +refused outright — an unverified read is not a degraded read, it is a different, +racy operation. It resolves the exact Git top-level, opens the repository root and every plan parent as held no-follow directory descriptors, rejects missing, symlink, non-directory, and escaping parents, and opens the leaf with `O_NOFOLLOW`. It reads at most 16 MiB from that held file descriptor, @@ -109,13 +112,17 @@ Neither Deepen nor work may parse bytes obtained before or outside this receipt. ## Safe generated-plan write contract -The writer fails closed unless Linux `/proc/self/fd`, `O_DIRECTORY`, -`O_NOFOLLOW`, and Python 3 with libc `renameat2(RENAME_NOREPLACE)` support are -available. Python may live in `/usr/local`, a Nix profile, or another absolute -PATH directory, but the helper accepts only a resolved executable and -containing directory owned by root or the current user and not writable by -group/other. The resolved executable is opened without following links and -invoked through that held descriptor. Relative PATH entries are ignored. The plan parent and the +The writer fails closed unless the host platform offers `O_DIRECTORY` and +`O_NOFOLLOW`, plus `/proc/self/fd` on Linux. It spawns no interpreter and loads +no native code: publication is `link(2)`, which is atomic, fails `EEXIST` when +the destination name is taken, and refuses a symlinked destination without +following it — the same no-replace guarantee `renameat2(RENAME_NOREPLACE)` and +`renameatx_np(RENAME_EXCL)` provide, available through `fs.linkSync` on every +supported platform. The temporary name is unlinked once the link succeeds; the +published file is the same inode the writer created and verified, so every +identity check downstream holds by construction. A link that succeeds followed +by an unlink that fails leaves the plan published and is reported as success, +because it is one. The plan parent and the repository's Git-admin directory must also share a filesystem. It resolves the target repository's exact Git top-level, opens that root and every destination parent as held no-follow directory descriptors, creates missing @@ -128,15 +135,45 @@ The writer creates a random exclusive temporary file relative to the held final parent descriptor and keeps its no-follow descriptor open. It writes and flushes the bytes, binds the temporary name to the opened inode, and hashes the open file before publication. Immediately before publication it revalidates -the parent and the temporary path, inode, size, and digest. Publication uses an -atomic no-replace move relative to the held directory descriptor. Initial mode -therefore cannot overwrite a destination that appears after the absent check. +the parent and the temporary path, inode, size, and digest. Publication links +the temporary name to the destination relative to the held directory +descriptor, which fails rather than replaces if the destination is taken. +Initial mode therefore cannot overwrite a destination that appears after the +absent check. The writer then flushes the directory and revalidates the committed path by opening it with `O_NOFOLLOW`, hashing both the original temporary fd and the path-bound fd, and performing a second descriptor-anchored path identity check after hashing. A detected mutation or replacement aborts instead of accepting mixed-era output. +### Linux anchors, macOS verifies + +The two platforms reach the same destination by different proofs, and the +difference is real enough to state rather than smooth over. + +On Linux every name resolves through `/proc/self/fd//`, a magic link +the kernel resolves against the inode the descriptor already holds. The names +above it are never re-walked, so an attacker who renames a parent between the +check and the use cannot redirect the operation. The race is impossible, not +merely detected. + +macOS has no such path. `/dev/fd/` is a devfs node, not a magic link: it can +be opened, but nothing can be resolved through it. `open("/dev/fd//child")` +returns `ENOENT`, and `realpath` of it returns `/dev/fd/` rather than the +directory's path — measured on macOS 26, not inferred. Node exposes no `openat`, +no `dir_fd` parameter, and no FFI, so on macOS the writer resolves names +lexically with `O_NOFOLLOW` at every component, holds an open descriptor on +every directory in the chain for the whole operation, and proves before *and* +after each step that the chain still names exactly the inodes it is holding. +Holding the descriptors is what makes the recorded inode numbers trustworthy: +an open descriptor pins its inode, so a freed number cannot be recycled beneath +the walk. + +What that buys is detection rather than prevention. A parent swapped inside the +window between a check and its use is caught by the check that follows, and the +operation aborts having written nothing — but on Linux it could not have +happened at all. No published byte escapes verification on either platform. + `--replace` accepts only a pre-existing regular file and is reserved for Deepen; without it, accidental overwrite is rejected. It also requires the exact canonical `generated_plan_path` and `plan_digest` from the same session's diff --git a/gitnexus-claude-plugin/skills/gitnexus-plan/scripts/evidence-provenance.mjs b/gitnexus-claude-plugin/skills/gitnexus-plan/scripts/evidence-provenance.mjs index 181d2120b..793fe4cd8 100644 --- a/gitnexus-claude-plugin/skills/gitnexus-plan/scripts/evidence-provenance.mjs +++ b/gitnexus-claude-plugin/skills/gitnexus-plan/scripts/evidence-provenance.mjs @@ -479,11 +479,11 @@ function resolveOwnGitTopLevel(absolute) { if (result.status !== 0) return null; let topLevel; try { - topLevel = fs.realpathSync(decodeUtf8(result.stdout, 'nested repository root').trim()); + topLevel = fs.realpathSync.native(decodeUtf8(result.stdout, 'nested repository root').trim()); } catch { return null; } - return topLevel === fs.realpathSync(absolute) ? topLevel : null; + return topLevel === fs.realpathSync.native(absolute) ? topLevel : null; } function readOwnGitlinkHead(absolute) { @@ -616,17 +616,30 @@ function filesystemObject(absolute, expectedKind, mutationGuards, testHooks) { throw new Error(`Unsupported filesystem object at ${absolute}`); } -function guardPathParents(repo, repoPath, mutationGuards) { +// Every dirty path re-walks its own parents, and dirty paths overwhelmingly +// share them — the repository root is re-stat'ed once per path. `guarded` is +// per-snapshot and remembers which absolute directories already carry a guard, +// so each distinct directory is stat'ed and guarded exactly once. +// +// Keeping the first-seen identity is the conservative choice: verifyGuards +// re-checks every guard against the filesystem at the end, so a directory that +// changes after it was guarded still fails there. Skipping a re-stat cannot hide +// a change; it only avoids recording the same directory twice. +function guardPathParents(repo, repoPath, mutationGuards, guarded) { const components = repoPath.split('/'); let current = repo; - const rootStat = fs.lstatSync(repo, { bigint: true }); - mutationGuards.push({ - type: 'directory', - absolute: repo, - identity: stableDirectoryIdentity(rootStat), - }); + if (!guarded.has(repo)) { + guarded.add(repo); + mutationGuards.push({ + type: 'directory', + absolute: repo, + identity: stableDirectoryIdentity(fs.lstatSync(repo, { bigint: true })), + }); + } for (const component of components.slice(0, -1)) { current = path.join(current, component); + // Already proved a real directory and already guarded on an earlier path. + if (guarded.has(current)) continue; let stat; try { stat = fs.lstatSync(current, { bigint: true }); @@ -638,6 +651,7 @@ function guardPathParents(repo, repoPath, mutationGuards) { throw new Error(`Refusing to traverse symlink parent for ${repoPath}`); } if (!stat.isDirectory()) return; + guarded.add(current); mutationGuards.push({ type: 'directory', absolute: current, @@ -646,81 +660,153 @@ function guardPathParents(repo, repoPath, mutationGuards) { } } -function recordAnchoredAbsence(repo, repoPath, mutationGuards) { - requireDescriptorAnchoring(); - const flags = - fs.constants.O_RDONLY | - fs.constants.O_DIRECTORY | - fs.constants.O_NOFOLLOW | - (fs.constants.O_CLOEXEC ?? 0); - const descriptors = []; - let retainedFd; - try { - let currentFd = fs.openSync(repo, flags); - descriptors.push(currentFd); - const components = repoPath.split('/'); - for (let index = 0; index < components.length; index += 1) { - const component = components[index]; - const child = descriptorPath(currentFd, component); - let childStat; - try { - childStat = fs.lstatSync(child, { bigint: true }); - } catch (error) { - if (error?.code !== 'ENOENT' && error?.code !== 'ENOTDIR') throw error; - const parentStat = fs.fstatSync(currentFd, { bigint: true }); - if (!parentStat.isDirectory()) { - throw new Error(`Absence parent is no longer a directory for ${repoPath}`); - } - retainedFd = currentFd; - mutationGuards.push({ - type: 'absence', - fd: retainedFd, - childName: component, - repoPath, - parentIdentity: stableDirectoryIdentity(parentStat), - parentMutationIdentity: statIdentity(parentStat), - }); - for (const fd of descriptors) { - if (fd !== retainedFd) fs.closeSync(fd); - } - return; - } - if (index === components.length - 1) { - throw new Error(`${repoPath} appeared while its absence was being anchored`); - } - if (childStat.isSymbolicLink() || !childStat.isDirectory()) { - throw new Error(`Refusing a non-directory parent while anchoring absence for ${repoPath}`); - } - const nextFd = fs.openSync(child, flags); - descriptors.push(nextFd); - currentFd = nextFd; - } - throw new Error(`Could not anchor absence for ${repoPath}`); - } catch (error) { - for (const fd of descriptors) { - if (fd === retainedFd) continue; - try { - fs.closeSync(fd); - } catch { - // Preserve the primary absence-anchoring error. - } - } - throw error; +// A bound, not a bug: the absence cache deduplicates correctly and leaks nothing, +// but citedPaths is caller-supplied and unbounded, so a pathological snapshot +// could hold more descriptors than the process is allowed (macOS +// kern.maxfilesperproc is 24576). The peak precedes a `git` spawn, so exhaustion +// would surface as a git failure misreported as evidence instability. +// +// Refuse rather than evict: closing a cached descriptor would silently break the +// pinned chain of an absence guard that was already recorded against it, which is +// exactly the inode-recycling hole the pins exist to close. +const ABSENCE_ANCHOR_LIMITS = Object.freeze({ maxPinnedDirectories: 4096 }); + +// Every no-follow read and every exclusive create in this file uses one of these +// two, so a change lands in one place rather than in seven. +const VERIFIED_READ_FLAGS = + fs.constants.O_RDONLY | fs.constants.O_NOFOLLOW | (fs.constants.O_CLOEXEC ?? 0); +const VERIFIED_CREATE_FLAGS = + fs.constants.O_RDWR | + fs.constants.O_CREAT | + fs.constants.O_EXCL | + fs.constants.O_NOFOLLOW | + (fs.constants.O_CLOEXEC ?? 0); + +function requireAbsenceAnchorCapacity(cache) { + if (cache.size >= ABSENCE_ANCHOR_LIMITS.maxPinnedDirectories) { + throw new Error( + `Absence anchoring exceeds ${ABSENCE_ANCHOR_LIMITS.maxPinnedDirectories} pinned directories`, + ); } } -function materializeRecord(repo, statusRecord, layers, mutationGuards, testHooks) { +const ANCHORED_DIRECTORY_FLAGS = + fs.constants.O_RDONLY | + fs.constants.O_DIRECTORY | + fs.constants.O_NOFOLLOW | + (fs.constants.O_CLOEXEC ?? 0); + +// Every absence receipt is verified long after its walk returns, so the chain +// that produced it has to stay pinned until the snapshot ends — an unpinned inode +// number can be recycled by a replacement directory that then reproduces the +// recorded identity exactly. Absent cited paths overwhelmingly share prefixes, so +// the walked directories are cached per snapshot and keyed by repo-relative +// prefix: one open descriptor and one anchored walk per distinct directory rather +// than per path. snapshotEvidence owns every descriptor in this cache and closes +// each exactly once; guards only borrow them for verification. +function anchoredAbsenceRoot(repo, cache) { + const cached = cache.get(''); + if (cached) return cached; + requireAbsenceAnchorCapacity(cache); + const fd = openVerifiedDirectory(repo, ANCHORED_DIRECTORY_FLAGS); + const handle = { + fd, + expectedPath: repo, + chain: [ + { expectedPath: repo, identity: stableDirectoryIdentity(fs.fstatSync(fd, { bigint: true })) }, + ], + descriptors: [fd], + }; + cache.set('', handle); + return handle; +} + +function recordAnchoredAbsence(repo, repoPath, mutationGuards, cache) { + requireDescriptorAnchoring(); + const components = repoPath.split('/'); + let handle = anchoredAbsenceRoot(repo, cache); + let prefix = ''; + for (let index = 0; index < components.length; index += 1) { + const component = components[index]; + const isFinal = index === components.length - 1; + prefix = prefix === '' ? component : `${prefix}/${component}`; + // The final component is always re-checked against the filesystem: it is the + // one whose absence is being recorded, and a cached answer would be a stale + // one. Only the prefix directories are reused. + const cached = isFinal ? undefined : cache.get(prefix); + if (cached) { + handle = cached; + continue; + } + const child = anchoredChild(handle, component); + let childStat; + try { + childStat = lstatChild(child); + } catch (error) { + if (error?.code !== 'ENOENT' && error?.code !== 'ENOTDIR') throw error; + const parentStat = fs.fstatSync(handle.fd, { bigint: true }); + if (!parentStat.isDirectory()) { + throw new Error(`Absence parent is no longer a directory for ${repoPath}`); + } + mutationGuards.push({ + type: 'absence', + // The handle is the holder the guard verifies against, and `ref` is the + // child path already built through the anchoredChild chokepoint — the + // guard must never re-derive that name itself. + handle, + ref: child, + fd: handle.fd, + repoPath, + parentMutationIdentity: statIdentity(parentStat), + }); + return; + } + if (isFinal) { + throw new Error(`${repoPath} appeared while its absence was being anchored`); + } + if (childStat.isSymbolicLink() || !childStat.isDirectory()) { + throw new Error(`Refusing a non-directory parent while anchoring absence for ${repoPath}`); + } + requireAbsenceAnchorCapacity(cache); + const childFd = openVerifiedDirectory(child.path, ANCHORED_DIRECTORY_FLAGS); + const expectedPath = path.join(handle.expectedPath, component); + let next; + try { + if (!anchoringBackend().descriptorMatchesChild(childFd, expectedPath, childStat)) { + throw new Error( + `Absence parent descriptor does not match its verified inode for ${repoPath}`, + ); + } + next = { + fd: childFd, + expectedPath, + chain: [...handle.chain, { expectedPath, identity: stableDirectoryIdentity(childStat) }], + descriptors: [...handle.descriptors, childFd], + }; + } catch (error) { + fs.closeSync(childFd); + throw error; + } + cache.set(prefix, next); + handle = next; + } + throw new Error(`Could not anchor absence for ${repoPath}`); +} + +function materializeRecord(repo, statusRecord, layers, mutationGuards, testHooks, walkState) { const head = layers.head(statusRecord.path); const index = layers.index(statusRecord.path); const expectedKind = index.kind === 'gitlink' || head.kind === 'gitlink' ? 'gitlink' : null; - guardPathParents(repo, statusRecord.path, mutationGuards); + guardPathParents(repo, statusRecord.path, mutationGuards, walkState.guardedDirectories); const filesystem = filesystemObject( path.join(repo, ...statusRecord.path.split('/')), expectedKind, mutationGuards, testHooks, ); - if (filesystem.kind === ABSENT) recordAnchoredAbsence(repo, statusRecord.path, mutationGuards); + if (filesystem.kind === ABSENT) { + recordAnchoredAbsence(repo, statusRecord.path, mutationGuards, walkState.absenceCache); + } if (statusRecord.directory_hint && filesystem.kind !== 'directory') { throw new Error( `Git reported an embedded directory but found ${filesystem.kind}: ${statusRecord.path}`, @@ -789,9 +875,15 @@ export function serializeDirtyRecords(entries) { } function assertRepository(repoInput) { - const repo = fs.realpathSync(requireString(repoInput, 'repo')); + // realpathSync.native, not realpathSync: the JS resolver preserves a Windows + // 8.3 short component (C:\Users\RUNNER~1\...) while git always reports the long + // form, so the two would never compare equal and every caller would be told the + // worktree root is not the worktree root it just named. + const repo = fs.realpathSync.native(requireString(repoInput, 'repo')); const topLevelResult = git(repo, ['rev-parse', '--show-toplevel']); - const topLevel = fs.realpathSync(decodeUtf8(topLevelResult.stdout, 'repository root').trim()); + const topLevel = fs.realpathSync.native( + decodeUtf8(topLevelResult.stdout, 'repository root').trim(), + ); if (topLevel !== repo) throw new Error(`--repo must be the Git worktree root (${topLevel})`); return repo; } @@ -882,17 +974,48 @@ function stableFileIdentity(stat) { return [stat.dev, stat.ino, stat.mode, stat.size].map(String).join(':'); } +// The two backends below differ in one decisive way, and it is worth stating +// plainly because the security properties are not the same. +// +// Linux ANCHORS. A name is resolved through /proc/self/fd//, which +// starts the walk at the inode the descriptor holds, so a parent that is renamed +// away cannot be traversed at all: the descriptor keeps pointing at the original +// directory and the impostor planted at the same name is simply never reached. +// +// macOS VERIFIES. Node cannot resolve a name relative to a descriptor there — +// /dev/fd/ is not a magic link (it stats as the directory but every attempt +// to traverse a child through it returns ENOENT), and fcntl F_GETPATH is a +// name-cache snapshot rather than a live anchor. So the Darwin backend resolves +// lexically, holds an open descriptor on every element of the chain, and proves +// before and after each operation that the path chain still names exactly the +// inodes it is holding. That DETECTS a swapped parent and aborts the write; it +// does not make the swap impossible the way the Linux path does. A swap landing +// inside the window between a check and the call it guards is caught by the +// following check, after the fact, rather than being unreachable. +// +// Every other platform gets neither and is refused outright. function requireDescriptorAnchoring() { - if ( - process.platform !== 'linux' || - fs.constants.O_DIRECTORY === undefined || - fs.constants.O_NOFOLLOW === undefined || - !fs.existsSync('/proc/self/fd') - ) { - throw new Error( - 'Safe generated-plan writes require Linux /proc/self/fd and O_DIRECTORY/O_NOFOLLOW; refusing an unanchored write', - ); + const directoryFlagsAvailable = + fs.constants.O_DIRECTORY !== undefined && fs.constants.O_NOFOLLOW !== undefined; + if (process.platform === 'linux') { + if (!directoryFlagsAvailable || !fs.existsSync('/proc/self/fd')) { + throw new Error( + 'Safe generated-plan writes require Linux /proc/self/fd and O_DIRECTORY/O_NOFOLLOW; refusing an unanchored write', + ); + } + return; } + if (process.platform === 'darwin') { + if (!directoryFlagsAvailable) { + throw new Error( + 'Safe generated-plan writes require macOS O_DIRECTORY/O_NOFOLLOW; refusing an unverified write', + ); + } + return; + } + throw new Error( + `Safe generated-plan writes require Linux /proc/self/fd or macOS O_DIRECTORY/O_NOFOLLOW; ${process.platform} offers neither, so refusing an unanchored write`, + ); } function descriptorPath(fd, childName) { @@ -900,157 +1023,352 @@ function descriptorPath(fd, childName) { return childName === undefined ? base : path.join(base, childName); } -function externalDescriptorPath(fd, childName) { - const base = `/proc/${process.pid}/fd/${fd}`; - return childName === undefined ? base : path.join(base, childName); +// Directory opens are plain O_RDONLY|O_DIRECTORY|O_NOFOLLOW|O_CLOEXEC on both +// platforms, and deliberately nothing else. +// +// O_NOFOLLOW_ANY (macOS 11+) used to be ORed in here on the theory that XNU +// ignores unrecognized open flag bits, so it would be inert where unsupported. +// That was wrong: combined with O_DIRECTORY macOS rejects it outright with +// EINVAL, and every directory open on Darwin failed. It is gone and is not +// coming back behind a probe or a degrade-on-EINVAL path — the per-component +// O_NOFOLLOW walk is what delivers the guarantee. Rust's cap-std, the closest +// reference implementation of this problem, has not adopted O_NOFOLLOW_ANY +// either (their issue #179 is still open). +function openVerifiedDirectory(absolute, flags) { + return fs.openSync(absolute, flags); } -const RENAME_NOREPLACE_SCRIPT = String.raw` -import ctypes -import errno -import os -import sys - -libc = ctypes.CDLL(None, use_errno=True) -try: - renameat2 = libc.renameat2 -except AttributeError: - print("libc does not expose renameat2", file=sys.stderr) - raise SystemExit(125) - -renameat2.argtypes = [ctypes.c_int, ctypes.c_char_p, ctypes.c_int, ctypes.c_char_p, ctypes.c_uint] -renameat2.restype = ctypes.c_int -result = renameat2(-100, os.fsencode(sys.argv[1]), -100, os.fsencode(sys.argv[2]), 1) -if result != 0: - error_number = ctypes.get_errno() - error_name = errno.errorcode.get(error_number, "UNKNOWN") - print(f"renameat2 RENAME_NOREPLACE failed: {error_name}: {os.strerror(error_number)}", file=sys.stderr) - raise SystemExit(17 if error_number == errno.EEXIST else 126) -`; - -let atomicMoverPath; - -function spawnHeldExecutable(executable, args, options) { - const before = fs.fstatSync(executable.fd, { bigint: true }); - if (!before.isFile() || statIdentity(before) !== executable.identity) { - throw new Error('Validated Python executable changed before invocation'); - } - const result = spawnSync('/proc/self/fd/3', args, { - ...options, - stdio: ['ignore', 'pipe', 'pipe', executable.fd], - }); - const after = fs.fstatSync(executable.fd, { bigint: true }); - assertStableIdentity(before, after, 'validated Python executable'); - return result; +// File opens additionally get O_NONBLOCK, which directory opens do not need: +// it stops a FIFO swapped in at the target name from wedging the process on +// open. The identity comparison that follows rejects the FIFO anyway, but only +// if we ever get as far as running it. +function openVerifiedFile(absolute, flags, mode) { + const nonBlocking = flags | (fs.constants.O_NONBLOCK ?? 0); + return mode === undefined + ? fs.openSync(absolute, nonBlocking) + : fs.openSync(absolute, nonBlocking, mode); } -function validatedPathExecutable(candidate) { - if (!path.isAbsolute(candidate)) return null; - const candidateDirectory = path.dirname(candidate); - let resolvedDirectory; - let resolved; - let directoryStats; - let executableStat; +// The publish primitive, identical on both platforms. +// +// link() is the portable no-replace publish: it fails with EEXIST if the +// destination name is taken — by a regular file, by a directory, or by a symlink, +// live or dangling — and it never follows that symlink to clobber its target. +// It also works where renameat2(RENAME_NOREPLACE) does not, notably v9fs, which +// is why the WSL2 9p case that used to fail every time now works. +// +// The published file is the same inode as the temporary, so every identity +// comparison the callers already make still holds, and validateCommittedPlan +// becomes strictly stronger: it compares the destination against the exact inode +// whose bytes were fsynced. +// +// On Linux both paths are /proc/self/fd//, so the publish is anchored +// to the held parent descriptors exactly like every other operation. +// link(2) BUGS: "On NFS filesystems, the return code may be wrong in case the NFS +// server performs the link creation and dies before it can say so. Use stat(2) to +// find out if the link got created." open(2) NOTES gives the remedy this +// implements: on a reported failure, stat the source and see whether its link +// count reached 2. A false positive would need someone to have hardlinked a +// 16-random-byte name inside a directory we hold open — and validateCommittedPlan +// still proves the destination is the exact temporary inode afterwards. +function linkCreatedDespiteError(sourcePath) { try { - resolvedDirectory = fs.realpathSync(candidateDirectory); - resolved = fs.realpathSync(candidate); - const resolvedExecutableDirectory = fs.realpathSync(path.dirname(resolved)); - directoryStats = [...new Set([resolvedDirectory, resolvedExecutableDirectory])].map( - (directory) => fs.statSync(directory), - ); - executableStat = fs.lstatSync(resolved); - fs.accessSync(resolved, fs.constants.X_OK); + return fs.statSync(sourcePath, { bigint: true }).nlink === 2n; } catch { - return null; + return false; } - if ( - directoryStats.some((stat) => !stat.isDirectory()) || - !executableStat.isFile() || - executableStat.isSymbolicLink() - ) { - return null; - } - const uid = typeof process.getuid === 'function' ? process.getuid() : null; - const trustedOwner = (stat) => uid === null || stat.uid === 0 || stat.uid === uid; - if ( - directoryStats.some((stat) => !trustedOwner(stat) || (stat.mode & 0o022) !== 0) || - !trustedOwner(executableStat) || - (executableStat.mode & 0o022) !== 0 - ) { - return null; - } - return resolved; } -function resolveAtomicMover() { - if (atomicMoverPath) return atomicMoverPath; - const candidates = new Set(); - for (const entry of (process.env.PATH ?? '').split(path.delimiter)) { - if (entry && path.isAbsolute(entry)) candidates.add(path.join(entry, 'python3')); - } - for (const entry of ['/usr/local/bin/python3', '/usr/bin/python3', '/bin/python3']) { - candidates.add(entry); - } - for (const candidate of candidates) { - const resolved = validatedPathExecutable(candidate); - if (!resolved) continue; - let fd; - try { - fd = fs.openSync( - resolved, - fs.constants.O_RDONLY | fs.constants.O_NOFOLLOW | (fs.constants.O_CLOEXEC ?? 0), - ); - } catch { - continue; +function linkNoReplace(sourcePath, destinationPath) { + try { + fs.linkSync(sourcePath, destinationPath); + } catch (error) { + // Callers treat "destination taken" as a distinct outcome, not a failure. + if (error?.code === 'EEXIST') return false; + if (!linkCreatedDespiteError(sourcePath)) { + // FAT, Coda, and some SMB/FUSE/virtiofs mounts have no hardlinks at all. + // Git falls back to rename here, but git can afford to lose collision + // detection because its objects are content-addressed; a plan destination + // is a plain name, so a replacing rename would silently clobber whatever + // is already there. Refuse loudly instead. + if (error?.code === 'EPERM' || error?.code === 'ENOTSUP' || error?.code === 'EMLINK') { + throw new Error( + `Generated-plan publication requires hard links, which this filesystem refused (${error.code}); refusing to fall back to a replacing rename`, + ); + } + throw error; } - const opened = fs.fstatSync(fd, { bigint: true }); - const executable = { fd, identity: statIdentity(opened), resolved }; - const version = spawnHeldExecutable( - executable, - ['-I', '-S', '-c', 'import sys; print(sys.version_info[0])'], - { - encoding: 'utf8', - env: { ...process.env, LANG: 'C', LC_ALL: 'C' }, - timeout: 10_000, - windowsHide: true, - }, - ); - if (version.status === 0 && version.stdout.trim() === '3') { - atomicMoverPath = executable; - return executable; - } - fs.closeSync(fd); } - throw new Error( - 'Safe generated-plan publication requires a trusted absolute Python 3 PATH candidate with libc renameat2 support', - ); -} - -function atomicMoveNoReplace(source, destination) { - const mover = resolveAtomicMover(); - const result = spawnHeldExecutable( - mover, - ['-I', '-S', '-c', RENAME_NOREPLACE_SCRIPT, source, destination], - { - encoding: 'utf8', - env: { ...process.env, LANG: 'C', LC_ALL: 'C' }, - timeout: 10_000, - windowsHide: true, - }, - ); - if (result.error) throw result.error; - if (result.status === 17) return false; - if (result.status !== 0) { - throw new Error( - `Atomic no-replace move failed (${result.status}): ${(result.stderr ?? '').trim()}`, - ); + try { + fs.unlinkSync(sourcePath); + } catch { + // The link succeeded, so the plan IS published. A temporary name left behind + // is a stray file, not an unpublished plan: reporting it as a failure would + // be a lie, and rolling back would unpublish a plan that is already live. } return true; } -function lstatOptional(absolute) { +// A directory holder is anything that owns a verified chain: a plan-parent +// handle, a ref's parent directory, or an absence guard. Two arrays describe it, +// both root-first and the same length — `chain` records each element's expected +// path and dev/ino/mode, and `descriptors` holds an open descriptor on each. +// +// Holding those descriptors is load-bearing rather than decorative. dev/ino/mode +// is unique only among *live* inodes: an inode number freed by an rmdir is handed +// straight back to the next mkdir, so a replacement directory can reproduce a +// recorded identity exactly. An open descriptor pins the inode, so the number +// cannot be recycled for as long as the holder exists. +function verifyPinnedDescriptors(holder) { + const { chain, descriptors } = holder; + if (!Array.isArray(descriptors) || descriptors.length !== chain.length) { + throw new Error('Generated-plan parent chain is missing the descriptors that pin it'); + } + chain.forEach((item, index) => { + const pinned = fs.fstatSync(descriptors[index], { bigint: true }); + if (!pinned.isDirectory() || stableDirectoryIdentity(pinned) !== item.identity) { + throw new Error('Generated-plan parent descriptor changed during the write'); + } + }); +} + +function verifyLexicalChain(holder) { + for (const item of holder.chain) { + let lexical; + try { + lexical = fs.lstatSync(item.expectedPath, { bigint: true }); + } catch (error) { + if (error?.code !== 'ENOENT' && error?.code !== 'ENOTDIR') throw error; + // A parent renamed out from under us is a mismatch, not a missing file: + // reporting the raw ENOENT would leak an unrelated-looking error out of a + // check whose whole job is to say the chain no longer holds. + throw new Error('Generated-plan lexical parent no longer matches its directory descriptor'); + } + if ( + lexical.isSymbolicLink() || + !lexical.isDirectory() || + stableDirectoryIdentity(lexical) !== item.identity + ) { + throw new Error('Generated-plan lexical parent no longer matches its directory descriptor'); + } + } +} + +// The whole platform seam, in five methods. Everything else an operation does is +// identical on both platforms and lives in the shared functions below. +// +// Only two things actually differ: how a name becomes a path, and what guard +// wraps the operation that uses it. +// +// Linux ANCHORS. /proc/self/fd// starts the walk at the inode the +// descriptor holds, so a parent renamed away cannot be traversed at all and the +// guard is a no-op — there is nothing left to verify. +// +// macOS VERIFIES. It resolves lexically, so before and after every operation it +// proves that each element of the path chain still names the exact inode being +// held for it. That DETECTS a swapped parent and aborts; it does not make the +// swap impossible. A swap landing inside the window is caught by the trailing +// check, after the fact, rather than being unreachable. The check runs after a +// failure too, because a verdict observed through a chain that has since changed +// is not a verdict. +const LINUX_ANCHORING = { + childPath(dirHandle, childName) { + return descriptorPath(dirHandle.fd, childName); + }, + verified(holders, run) { + return run(); + }, + descriptorMatchesChild(fd, expectedPath) { + return fs.realpathSync.native(descriptorPath(fd)) === expectedPath; + }, + parentStillResolves(parentHandle) { + return fs.realpathSync.native(descriptorPath(parentHandle.fd)) === parentHandle.expectedPath; + }, + verifyAbsentChild(guard) { + if (absentChildIsPresent(guard.ref)) { + throw new Error(`${guard.repoPath} appeared before evidence materialization completed`); + } + }, +}; + +const DARWIN_ANCHORING = { + childPath(dirHandle, childName) { + return path.join(dirHandle.expectedPath, childName); + }, + verified(holders, run) { + const list = Array.isArray(holders) ? holders : [holders]; + const proveChain = () => { + for (const holder of list) { + verifyPinnedDescriptors(holder); + verifyLexicalChain(holder); + } + }; + proveChain(); + let value; + try { + value = run(); + } catch (error) { + proveChain(); + throw error; + } + proveChain(); + return value; + }, + descriptorMatchesChild(fd, _expectedPath, childStat) { + // There is no live fd-to-path oracle on macOS (F_GETPATH is a name-cache + // snapshot, not an anchor), so escape is decided the other way round: the + // name was just resolved under a verified chain, and the descriptor opened + // from it counts only if it is that same inode. + const opened = fs.fstatSync(fd, { bigint: true }); + return ( + opened.isDirectory() && stableDirectoryIdentity(opened) === stableDirectoryIdentity(childStat) + ); + }, + parentStillResolves(parentHandle) { + // Both halves are needed: a directory renamed away keeps its inode, so the + // descriptors alone still match and only the lexical half notices it moved. + try { + verifyPinnedDescriptors(parentHandle); + verifyLexicalChain(parentHandle); + } catch { + return false; + } + return true; + }, + verifyAbsentChild(guard) { + let present; + try { + present = DARWIN_ANCHORING.verified(guard.handle, () => absentChildIsPresent(guard.ref)); + } catch (error) { + // A chain that no longer holds makes the absence verdict meaningless, and + // the caller reports that as the anchor changing rather than as a stray + // parent-descriptor error. Linux cannot reach this: its guard is a no-op. + throw new Error( + `Absence anchor changed for ${guard.repoPath}: ${ + error instanceof Error ? error.message : String(error) + }`, + ); + } + if (present) { + throw new Error(`${guard.repoPath} appeared before evidence materialization completed`); + } + }, +}; + +const ANCHORING_BACKENDS = new Map([ + ['linux', LINUX_ANCHORING], + ['darwin', DARWIN_ANCHORING], +]); + +function anchoringBackend() { + const backend = ANCHORING_BACKENDS.get(process.platform); + if (!backend) { + // requireDescriptorAnchoring normally refuses first; this is the same answer + // from the other side, so an unsupported platform can never fall through to + // whichever backend happened to be the ternary's default. + throw new Error( + `No generated-plan anchoring backend for ${process.platform}; refusing an unanchored write`, + ); + } + return backend; +} + +// Open, fstat, compare, close on mismatch. The descriptor never escapes this +// function unless it refers to the inode the caller already verified by name, so +// a lexical open that landed anywhere else cannot be used by accident. On Linux +// the comparison passes trivially — the /proc walk already resolved from the +// held parent — and costs one fstat to keep the guarantee structural rather than +// dependent on which backend is in play. +function adoptVerifiedFile(ref, expectedStat, flags) { + const fd = openVerifiedFile(ref.path, flags); + let opened; try { - return fs.lstatSync(absolute, { bigint: true }); + opened = fs.fstatSync(fd, { bigint: true }); + } catch (error) { + fs.closeSync(fd); + throw error; + } + if (stableFileIdentity(opened) !== stableFileIdentity(expectedStat)) { + fs.closeSync(fd); + return null; + } + return fd; +} + +function absentChildIsPresent(ref) { + try { + fs.lstatSync(ref.path, { bigint: true }); + } catch (error) { + if (error?.code === 'ENOENT') return false; + throw error; + } + return true; +} + +// The operations. Each is the same on both platforms; only the guard differs. +function lstatChild(ref) { + return anchoringBackend().verified(ref.dir, () => fs.lstatSync(ref.path, { bigint: true })); +} + +function openChildRead(ref, flags, expectedStat) { + return anchoringBackend().verified(ref.dir, () => { + const fd = adoptVerifiedFile(ref, expectedStat, flags); + if (fd === null) { + throw new Error(`${ref.name} was replaced between its verified stat and its no-follow open`); + } + return fd; + }); +} + +function createChild(ref, flags, mode) { + // O_CREAT|O_EXCL|O_NOFOLLOW is atomic at the leaf, so the only thing the guard + // has to cover is which directory the leaf landed in. + return anchoringBackend().verified(ref.dir, () => openVerifiedFile(ref.path, flags, mode)); +} + +function mkdirChild(ref, mode) { + anchoringBackend().verified(ref.dir, () => fs.mkdirSync(ref.path, { mode })); +} + +function publishNoReplace(sourceRef, destinationRef) { + return anchoringBackend().verified([sourceRef.dir, destinationRef.dir], () => + linkNoReplace(sourceRef.path, destinationRef.path), + ); +} + +// The single place a name becomes a path, and therefore the right place to +// enforce that a name is one ordinary component. +// +// A trailing separator is the sharp edge here, not a tidiness concern: +// open(path, O_NOFOLLOW) FOLLOWS a symlink when path ends in "/" — the trap +// behind CVE-2026-39822 / golang/go#79005, which let os.Root escape its own +// root. path.join preserves that trailing slash, so a component carrying one +// would turn every no-follow open in this file into a following one. +// normalizeRepoPath already rejects such components upstream; this is the +// chokepoint that makes it true for every caller, including the generated +// temporary and vault names that never pass through it. +function anchoredChild(dirHandle, childName) { + if ( + typeof childName !== 'string' || + childName === '' || + childName === '.' || + childName === '..' || + childName.includes('/') || + childName.includes('\\') || + childName.includes('\0') + ) { + throw new Error(`Refusing to resolve ${JSON.stringify(childName)} as a single path component`); + } + return { + dir: dirHandle, + name: childName, + path: anchoringBackend().childPath(dirHandle, childName), + }; +} + +function lstatAnchoredOptional(ref) { + try { + return lstatChild(ref); } catch (error) { if (error?.code === 'ENOENT' || error?.code === 'ENOTDIR') return null; throw error; @@ -1063,39 +1381,37 @@ function openPlanParent( { createMissing = true, purpose = 'Generated-plan' } = {}, ) { requireDescriptorAnchoring(); - const flags = - fs.constants.O_RDONLY | - fs.constants.O_DIRECTORY | - fs.constants.O_NOFOLLOW | - (fs.constants.O_CLOEXEC ?? 0); + // Root-first and index-aligned with `chain`: verifyPinnedDescriptors relies on + // that, and the descriptors are what pin each recorded inode against reuse. const descriptors = []; try { - let currentFd = fs.openSync(repo, flags); + let currentFd = openVerifiedDirectory(repo, ANCHORED_DIRECTORY_FLAGS); descriptors.push(currentFd); const rootStat = fs.fstatSync(currentFd, { bigint: true }); const chain = [{ expectedPath: repo, identity: stableDirectoryIdentity(rootStat) }]; + let currentHandle = { fd: currentFd, expectedPath: repo, chain, descriptors }; const traversed = []; for (const component of parentComponents) { traversed.push(component); - const anchoredChild = descriptorPath(currentFd, component); + const child = anchoredChild(currentHandle, component); let childStat; let created = false; try { - childStat = fs.lstatSync(anchoredChild, { bigint: true }); + childStat = lstatChild(child); } catch (error) { if (error?.code !== 'ENOENT' && error?.code !== 'ENOTDIR') throw error; if (!createMissing) { throw new Error(`${purpose} parent does not exist: ${traversed.join('/')}`); } - fs.mkdirSync(anchoredChild, { mode: 0o755 }); - childStat = fs.lstatSync(anchoredChild, { bigint: true }); + mkdirChild(child, 0o755); + childStat = lstatChild(child); created = true; } if (childStat.isSymbolicLink() || !childStat.isDirectory()) { throw new Error(`${purpose} parent is not a real directory: ${traversed.join('/')}`); } const parentFd = currentFd; - const childFd = fs.openSync(anchoredChild, flags); + const childFd = openVerifiedDirectory(child.path, ANCHORED_DIRECTORY_FLAGS); descriptors.push(childFd); currentFd = childFd; if (created) { @@ -1103,18 +1419,16 @@ function openPlanParent( fs.fsyncSync(parentFd); } const expected = path.join(repo, ...traversed); - const actual = fs.realpathSync(descriptorPath(currentFd)); - if (actual !== expected) { + if (!anchoringBackend().descriptorMatchesChild(currentFd, expected, childStat)) { throw new Error(`${purpose} parent escaped the repository: ${traversed.join('/')}`); } const openedStat = fs.fstatSync(currentFd, { bigint: true }); chain.push({ expectedPath: expected, identity: stableDirectoryIdentity(openedStat) }); + currentHandle = { fd: currentFd, expectedPath: expected, chain, descriptors }; } - const stat = fs.fstatSync(currentFd, { bigint: true }); return { descriptors, fd: currentFd, - identity: stableDirectoryIdentity(stat), expectedPath: path.join(repo, ...parentComponents), chain, }; @@ -1134,9 +1448,16 @@ function closeDescriptors(descriptors) { } } +// A handle's identity IS its chain leaf's identity. Storing it twice meant two +// fstats a line apart and a re-stamp helper to keep them agreeing; deriving it +// removes both. +function handleIdentity(handle) { + return handle.chain[handle.chain.length - 1].identity; +} + function resolveGitDirectory(repo) { const result = git(repo, ['rev-parse', '--absolute-git-dir']); - return fs.realpathSync(decodeUtf8(result.stdout, 'Git administrative directory').trim()); + return fs.realpathSync.native(decodeUtf8(result.stdout, 'Git administrative directory').trim()); } function openBackupVault(repo, { createMissing = true } = {}) { @@ -1147,9 +1468,12 @@ function openBackupVault(repo, { createMissing = true } = {}) { }); fs.fchmodSync(handle.fd, 0o700); fs.fsyncSync(handle.fd); - const stat = fs.fstatSync(handle.fd, { bigint: true }); - handle.identity = stableDirectoryIdentity(stat); - handle.chain[handle.chain.length - 1].identity = handle.identity; + // mode is part of every directory identity, so hardening the vault changes the + // identity the chain recorded for it; without this the next verification would + // reject the directory it just hardened. + handle.chain[handle.chain.length - 1].identity = stableDirectoryIdentity( + fs.fstatSync(handle.fd, { bigint: true }), + ); return { ...handle, gitDirectory }; } @@ -1157,33 +1481,28 @@ function validatePlanParent(parentHandle) { const descriptorStat = fs.fstatSync(parentHandle.fd, { bigint: true }); if ( !descriptorStat.isDirectory() || - stableDirectoryIdentity(descriptorStat) !== parentHandle.identity + stableDirectoryIdentity(descriptorStat) !== handleIdentity(parentHandle) ) { throw new Error('Generated-plan parent descriptor changed during the write'); } - const descriptorRealPath = fs.realpathSync(descriptorPath(parentHandle.fd)); - if (descriptorRealPath !== parentHandle.expectedPath) { + if (!anchoringBackend().parentStillResolves(parentHandle)) { throw new Error('Generated-plan parent moved or was replaced during the write'); } - for (const item of parentHandle.chain) { - const lexicalStat = fs.lstatSync(item.expectedPath, { bigint: true }); - if ( - lexicalStat.isSymbolicLink() || - !lexicalStat.isDirectory() || - stableDirectoryIdentity(lexicalStat) !== item.identity - ) { - throw new Error('Generated-plan lexical parent no longer matches its directory descriptor'); - } - } + // Both halves come from the shared helpers rather than being restated here: an + // earlier hand-copy of the lexical loop lost verifyLexicalChain's ENOENT/ENOTDIR + // translation, so a renamed parent could surface a raw errno from a function + // with a dozen call sites. + verifyPinnedDescriptors(parentHandle); + verifyLexicalChain(parentHandle); } function inspectPlanDestination( - finalPath, + finalRef, { replace, expectedIdentity, mustBeAbsent = false } = {}, ) { let stat; try { - stat = fs.lstatSync(finalPath, { bigint: true }); + stat = lstatChild(finalRef); } catch (error) { if (error?.code === 'ENOENT') { if (expectedIdentity) throw new Error('Generated plan disappeared during the write'); @@ -1201,19 +1520,17 @@ function inspectPlanDestination( if (expectedIdentity && identity !== expectedIdentity) { throw new Error('Generated plan changed during the write'); } - return identity; + return stat; } -function openExistingPlanDestination(finalPath, replace) { - const identity = inspectPlanDestination(finalPath, { replace }); - if (identity === null) { +function openExistingPlanDestination(finalRef, replace) { + const stat = inspectPlanDestination(finalRef, { replace }); + if (stat === null) { if (replace) throw new Error('Deepen mode requires an existing generated plan to replace'); return { fd: undefined, identity: null, stableIdentity: null }; } - const fd = fs.openSync( - finalPath, - fs.constants.O_RDONLY | fs.constants.O_NOFOLLOW | (fs.constants.O_CLOEXEC ?? 0), - ); + const identity = statIdentity(stat); + const fd = openChildRead(finalRef, VERIFIED_READ_FLAGS, stat); try { const opened = fs.fstatSync(fd, { bigint: true }); if (!opened.isFile() || statIdentity(opened) !== identity) { @@ -1264,8 +1581,8 @@ function hashOpenFile(fd, label) { }; } -function validateCommittedPlan(finalPath, tempFd, expectedTemp, testHooks) { - const before = fs.lstatSync(finalPath, { bigint: true }); +function validateCommittedPlan(finalRef, tempFd, expectedTemp, testHooks) { + const before = lstatChild(finalRef); if ( before.isSymbolicLink() || !before.isFile() || @@ -1273,19 +1590,16 @@ function validateCommittedPlan(finalPath, tempFd, expectedTemp, testHooks) { ) { throw new Error('Generated-plan destination failed its first post-write identity check'); } - const finalFd = fs.openSync( - finalPath, - fs.constants.O_RDONLY | fs.constants.O_NOFOLLOW | (fs.constants.O_CLOEXEC ?? 0), - ); + const finalFd = openChildRead(finalRef, VERIFIED_READ_FLAGS, before); try { const opened = fs.fstatSync(finalFd, { bigint: true }); if (!opened.isFile() || stableFileIdentity(opened) !== expectedTemp.identity) { throw new Error('Generated-plan destination changed while its no-follow descriptor opened'); } - testHooks?.afterFinalOpen?.({ fd: finalFd, finalPath }); + testHooks?.afterFinalOpen?.({ fd: finalFd, finalPath: finalRef.path }); const committedViaTemp = hashOpenFile(tempFd, 'generated-plan committed file'); const committedViaPath = hashOpenFile(finalFd, 'generated-plan destination descriptor'); - const after = fs.lstatSync(finalPath, { bigint: true }); + const after = lstatChild(finalRef); const openedAfter = fs.fstatSync(finalFd, { bigint: true }); if ( after.isSymbolicLink() || @@ -1320,22 +1634,19 @@ function copyOpenFile(sourceFd, destinationFd, label) { return after; } -function openVerifiedPathFile(absolute, label) { - const before = fs.lstatSync(absolute, { bigint: true }); +function openVerifiedAnchoredFile(ref, label, knownStat) { + const before = knownStat ?? lstatChild(ref); if (before.isSymbolicLink() || !before.isFile()) { throw new Error(`${label} is not a regular no-follow file`); } - const fd = fs.openSync( - absolute, - fs.constants.O_RDONLY | fs.constants.O_NOFOLLOW | (fs.constants.O_CLOEXEC ?? 0), - ); + const fd = openChildRead(ref, VERIFIED_READ_FLAGS, before); try { const opened = fs.fstatSync(fd, { bigint: true }); if (!opened.isFile() || stableFileIdentity(opened) !== stableFileIdentity(before)) { throw new Error(`${label} changed while its descriptor opened`); } const layer = hashOpenFile(fd, label); - const after = fs.lstatSync(absolute, { bigint: true }); + const after = lstatChild(ref); if (after.isSymbolicLink() || !after.isFile() || stableFileIdentity(after) !== layer.identity) { throw new Error(`${label} changed after verification`); } @@ -1358,10 +1669,10 @@ export function readPlanSafely({ repo: repoInput, generatedPlanPath, testHooks } let fd; try { validatePlanParent(parentHandle); - const finalPath = descriptorPath(parentHandle.fd, finalName); + const finalRef = anchoredChild(parentHandle, finalName); let before; try { - before = fs.lstatSync(finalPath, { bigint: true }); + before = lstatChild(finalRef); } catch (error) { if (error?.code === 'ENOENT' || error?.code === 'ENOTDIR') { throw new Error(`Loaded plan does not exist: ${generatedPlan}`); @@ -1371,15 +1682,12 @@ export function readPlanSafely({ repo: repoInput, generatedPlanPath, testHooks } if (before.isSymbolicLink() || !before.isFile()) { throw new Error('Loaded plan must be a regular file, never a symlink'); } - fd = fs.openSync( - finalPath, - fs.constants.O_RDONLY | fs.constants.O_NOFOLLOW | (fs.constants.O_CLOEXEC ?? 0), - ); + fd = openChildRead(finalRef, VERIFIED_READ_FLAGS, before); const opened = fs.fstatSync(fd, { bigint: true }); if (!opened.isFile() || statIdentity(opened) !== statIdentity(before)) { throw new Error('Loaded plan changed while its no-follow descriptor opened'); } - testHooks?.afterPlanOpen?.({ fd, finalPath }); + testHooks?.afterPlanOpen?.({ fd, finalPath: finalRef.path }); const chunks = []; let total = 0; const buffer = Buffer.allocUnsafe(64 * 1024); @@ -1394,7 +1702,7 @@ export function readPlanSafely({ repo: repoInput, generatedPlanPath, testHooks } decodeUtf8(contents, 'loaded plan'); const after = fs.fstatSync(fd, { bigint: true }); assertStableIdentity(opened, after, 'loaded plan'); - const pathAfter = fs.lstatSync(finalPath, { bigint: true }); + const pathAfter = lstatChild(finalRef); if ( pathAfter.isSymbolicLink() || !pathAfter.isFile() || @@ -1419,24 +1727,22 @@ function artifactGitPath(name) { return `gitnexus-plan-backups/${name}`; } -function verifyVaultArtifactFromFreshRoot(repo, gitPath, expectedLayer) { - const components = gitPath.split('/'); - if (components.length !== 2 || components[0] !== 'gitnexus-plan-backups') { - throw new Error(`Invalid Git-admin artifact path: ${gitPath}`); - } +function verifyVaultArtifactFromFreshRoot(repo, name, expectedLayer) { const freshVault = openBackupVault(repo, { createMissing: false }); try { validatePlanParent(freshVault); - const opened = openVerifiedPathFile( - descriptorPath(freshVault.fd, components[1]), - `Git-admin artifact ${gitPath}`, + const opened = openVerifiedAnchoredFile( + anchoredChild(freshVault, name), + `Git-admin artifact ${artifactGitPath(name)}`, ); try { if ( opened.layer.identity !== expectedLayer.identity || opened.layer.digest !== expectedLayer.digest ) { - throw new Error(`Git-admin artifact changed before fresh-root verification: ${gitPath}`); + throw new Error( + `Git-admin artifact changed before fresh-root verification: ${artifactGitPath(name)}`, + ); } } finally { fs.closeSync(opened.fd); @@ -1449,16 +1755,8 @@ function verifyVaultArtifactFromFreshRoot(repo, gitPath, expectedLayer) { function createVaultCopyFromFd(repo, vault, sourceFd, role) { validatePlanParent(vault); const name = `.gitnexus-plan-${role}-${process.pid}-${randomBytes(16).toString('hex')}.bak`; - const absolute = descriptorPath(vault.fd, name); - const destinationFd = fs.openSync( - absolute, - fs.constants.O_RDWR | - fs.constants.O_CREAT | - fs.constants.O_EXCL | - fs.constants.O_NOFOLLOW | - (fs.constants.O_CLOEXEC ?? 0), - 0o600, - ); + const artifact = anchoredChild(vault, name); + const destinationFd = createChild(artifact, VERIFIED_CREATE_FLAGS, 0o600); let destination; try { const sourceStat = copyOpenFile(sourceFd, destinationFd, role); @@ -1469,7 +1767,7 @@ function createVaultCopyFromFd(repo, vault, sourceFd, role) { if (source.size !== destination.size || source.digest !== destination.digest) { throw new Error(`${role} vault copy does not match its held source descriptor`); } - const pathStat = fs.lstatSync(absolute, { bigint: true }); + const pathStat = lstatChild(artifact); if ( pathStat.isSymbolicLink() || !pathStat.isFile() || @@ -1481,24 +1779,15 @@ function createVaultCopyFromFd(repo, vault, sourceFd, role) { } finally { fs.closeSync(destinationFd); } - const gitPath = artifactGitPath(name); - verifyVaultArtifactFromFreshRoot(repo, gitPath, destination); - return { role, gitPath, layer: destination }; + verifyVaultArtifactFromFreshRoot(repo, name, destination); + return { role, gitPath: artifactGitPath(name), layer: destination }; } function createVaultCopyFromBytes(repo, vault, contents, role) { validatePlanParent(vault); const name = `.gitnexus-plan-${role}-${process.pid}-${randomBytes(16).toString('hex')}.bak`; - const absolute = descriptorPath(vault.fd, name); - const fd = fs.openSync( - absolute, - fs.constants.O_RDWR | - fs.constants.O_CREAT | - fs.constants.O_EXCL | - fs.constants.O_NOFOLLOW | - (fs.constants.O_CLOEXEC ?? 0), - 0o600, - ); + const artifact = anchoredChild(vault, name); + const fd = createChild(artifact, VERIFIED_CREATE_FLAGS, 0o600); let layer; try { writeAll(fd, contents); @@ -1508,7 +1797,7 @@ function createVaultCopyFromBytes(repo, vault, contents, role) { if (layer.size !== BigInt(contents.length) || layer.digest !== sha256(contents)) { throw new Error(`${role} vault copy does not match the intended plan bytes`); } - const pathStat = fs.lstatSync(absolute, { bigint: true }); + const pathStat = lstatChild(artifact); if ( pathStat.isSymbolicLink() || !pathStat.isFile() || @@ -1520,32 +1809,31 @@ function createVaultCopyFromBytes(repo, vault, contents, role) { } finally { fs.closeSync(fd); } - const gitPath = artifactGitPath(name); - verifyVaultArtifactFromFreshRoot(repo, gitPath, layer); - return { role, gitPath, layer }; + verifyVaultArtifactFromFreshRoot(repo, name, layer); + return { role, gitPath: artifactGitPath(name), layer }; } function movePathToVault(repo, sourceHandle, sourceName, vault, role) { - const source = descriptorPath(sourceHandle.fd, sourceName); - if (!lstatOptional(source)) return null; + const source = anchoredChild(sourceHandle, sourceName); + if (!lstatAnchoredOptional(source)) return null; const name = `.gitnexus-plan-${role}-${process.pid}-${randomBytes(16).toString('hex')}.bak`; - const destination = descriptorPath(vault.fd, name); - const moved = atomicMoveNoReplace( - externalDescriptorPath(sourceHandle.fd, sourceName), - externalDescriptorPath(vault.fd, name), - ); + const destination = anchoredChild(vault, name); + const moved = publishNoReplace(source, destination); if (!moved) throw new Error(`${role} preservation destination unexpectedly exists`); fs.fsyncSync(sourceHandle.fd); if (vault.fd !== sourceHandle.fd) fs.fsyncSync(vault.fd); - const sourceAfter = lstatOptional(source); - const destinationAfter = lstatOptional(destination); + const sourceAfter = lstatAnchoredOptional(source); + const destinationAfter = lstatAnchoredOptional(destination); if (sourceAfter || !destinationAfter) { throw new Error(`${role} could not be atomically moved into the Git-admin vault`); } - const opened = openVerifiedPathFile(destination, `${role} Git-admin artifact`); - const gitPath = artifactGitPath(name); - verifyVaultArtifactFromFreshRoot(repo, gitPath, opened.layer); - return { role, gitPath, layer: opened.layer, fd: opened.fd }; + const opened = openVerifiedAnchoredFile( + destination, + `${role} Git-admin artifact`, + destinationAfter, + ); + verifyVaultArtifactFromFreshRoot(repo, name, opened.layer); + return { role, gitPath: artifactGitPath(name), layer: opened.layer, fd: opened.fd }; } function formatPreservedArtifacts(artifacts) { @@ -1600,10 +1888,10 @@ export function writePlanSafely({ const finalName = components.pop(); let parentHandle; let vaultHandle; - let tempPath; + let tempRef; let tempName; let tempFd; - let finalPath; + let finalRef; let expectedTemp; let originalDestination; let priorBackup; @@ -1611,7 +1899,6 @@ export function writePlanSafely({ try { parentHandle = openPlanParent(repo, components); vaultHandle = openBackupVault(repo); - resolveAtomicMover(); const parentDevice = fs.fstatSync(parentHandle.fd, { bigint: true }).dev; const vaultDevice = fs.fstatSync(vaultHandle.fd, { bigint: true }).dev; if (parentDevice !== vaultDevice) { @@ -1622,19 +1909,11 @@ export function writePlanSafely({ testHooks?.afterParentOpen?.({ fd: parentHandle.fd, path: parentHandle.expectedPath }); validatePlanParent(parentHandle); validatePlanParent(vaultHandle); - finalPath = descriptorPath(parentHandle.fd, finalName); - originalDestination = openExistingPlanDestination(finalPath, shouldReplace); + finalRef = anchoredChild(parentHandle, finalName); + originalDestination = openExistingPlanDestination(finalRef, shouldReplace); tempName = `.gitnexus-plan-${process.pid}-${randomBytes(16).toString('hex')}.tmp`; - tempPath = descriptorPath(parentHandle.fd, tempName); - tempFd = fs.openSync( - tempPath, - fs.constants.O_RDWR | - fs.constants.O_CREAT | - fs.constants.O_EXCL | - fs.constants.O_NOFOLLOW | - (fs.constants.O_CLOEXEC ?? 0), - 0o600, - ); + tempRef = anchoredChild(parentHandle, tempName); + tempFd = createChild(tempRef, VERIFIED_CREATE_FLAGS, 0o600); writeAll(tempFd, contents); fs.fchmodSync(tempFd, 0o644); fs.fsyncSync(tempFd); @@ -1646,12 +1925,12 @@ export function writePlanSafely({ testHooks?.beforeRename?.({ fd: parentHandle.fd, path: parentHandle.expectedPath, - tempPath, + tempPath: tempRef.path, }); validatePlanParent(parentHandle); validatePlanParent(vaultHandle); validateOpenPlanDestination(originalDestination); - const tempPathStat = fs.lstatSync(tempPath, { bigint: true }); + const tempPathStat = lstatChild(tempRef); const currentTemp = hashOpenFile(tempFd, 'generated-plan temporary file'); if ( tempPathStat.isSymbolicLink() || @@ -1664,7 +1943,7 @@ export function writePlanSafely({ } if (shouldReplace) { - testHooks?.beforeBackupMove?.({ fd: parentHandle.fd, finalPath }); + testHooks?.beforeBackupMove?.({ fd: parentHandle.fd, finalPath: finalRef.path }); const originalLayer = hashOpenFile(originalDestination.fd, 'prior generated plan'); if (originalLayer.digest !== expectedDigest) { throw new Error( @@ -1673,7 +1952,7 @@ export function writePlanSafely({ } validatePlanParent(parentHandle); validateOpenPlanDestination(originalDestination); - inspectPlanDestination(finalPath, { + inspectPlanDestination(finalRef, { replace: true, expectedIdentity: originalDestination.identity, }); @@ -1691,20 +1970,20 @@ export function writePlanSafely({ ); throw new Error('Destination raced while the prior plan was moved into preservation'); } - if (lstatOptional(finalPath)) { + if (lstatAnchoredOptional(finalRef)) { throw new Error('Destination reappeared after the prior plan was preserved'); } } testHooks?.beforePublication?.({ fd: parentHandle.fd, - finalPath, - tempPath, + finalPath: finalRef.path, + tempPath: tempRef.path, replace: shouldReplace, }); validatePlanParent(parentHandle); validatePlanParent(vaultHandle); - const finalTempPathStat = fs.lstatSync(tempPath, { bigint: true }); + const finalTempPathStat = lstatChild(tempRef); const finalTemp = hashOpenFile(tempFd, 'generated-plan temporary file'); if ( finalTempPathStat.isSymbolicLink() || @@ -1715,19 +1994,25 @@ export function writePlanSafely({ ) { throw new Error('Generated-plan temporary path or content changed at publication'); } - atomicMoveNoReplace( - externalDescriptorPath(parentHandle.fd, tempName), - externalDescriptorPath(parentHandle.fd, finalName), - ); - if (lstatOptional(tempPath) || !lstatOptional(finalPath)) { + // link() reports the race itself; re-deriving that verdict from a later pair + // of stats would be both slower and weaker. + if (!publishNoReplace(tempRef, finalRef)) { throw new Error('Generated-plan publication was refused because the destination raced'); } + // link() creates a directory entry, so it needs the parent fsync that rename + // needed: the file's own bytes were fsynced through tempFd before this point, + // and this makes the name that now reaches them durable too. Skipping it is + // the step write-file-atomic omits and maildir, git and atomicwrites all + // mandate. + // + // Honest limitation: on macOS fsync is not a write barrier — the durable + // primitive there is fcntl(F_FULLFSYNC), which Node does not expose. A + // macOS plan write is therefore as durable as fsync makes it and no more. fs.fsyncSync(parentHandle.fd); - testHooks?.afterPublication?.({ fd: parentHandle.fd, finalPath }); - testHooks?.afterRename?.({ fd: parentHandle.fd, finalPath }); + testHooks?.afterPublication?.({ fd: parentHandle.fd, finalPath: finalRef.path }); validatePlanParent(parentHandle); validatePlanParent(vaultHandle); - validateCommittedPlan(finalPath, tempFd, expectedTemp, testHooks); + validateCommittedPlan(finalRef, tempFd, expectedTemp, testHooks); const receipt = { generated_plan_path: generatedPlan, bytes_written: contents.length }; if (priorBackup) receipt.prior_plan_backup_git_path = priorBackup.gitPath; return receipt; @@ -1848,6 +2133,11 @@ export function snapshotEvidence({ const headGuards = captureHeadGuards(repo); const dirty = initialDirty.records; const mutationGuards = []; + // Per-snapshot walk state: `absenceCache` owns every descriptor an absence + // anchor holds, deduplicated by repo-relative prefix and closed exactly once + // below; `guardedDirectories` keeps parent guarding to one stat per directory. + const absenceCache = new Map(); + const walkState = { absenceCache, guardedDirectories: new Set() }; try { testHooks?.afterAnchorCapture?.({ headCommit: head }); @@ -1862,7 +2152,9 @@ export function snapshotEvidence({ testHooks?.afterGitLayerLoad?.({ headCommit: head }); const globalEntries = [...dirty.values()] .filter((record) => record.path !== generatedPlan) - .map((record) => materializeRecord(repo, record, layers, mutationGuards, testHooks)); + .map((record) => + materializeRecord(repo, record, layers, mutationGuards, testHooks, walkState), + ); const citedEntries = [...normalizedCitations].sort(compareUtf8).map((repoPath) => { const status = dirty.get(repoPath) ?? { path: repoPath, @@ -1871,7 +2163,7 @@ export function snapshotEvidence({ rename_to: null, has_untracked: false, }; - const entry = materializeRecord(repo, status, layers, mutationGuards, testHooks); + const entry = materializeRecord(repo, status, layers, mutationGuards, testHooks, walkState); const present = Object.values(entry.object_kind).some((kind) => kind !== ABSENT); if (!present) entry.state = ABSENT; else if (entry.state === 'clean' && entry.object_kind.untracked !== ABSENT) { @@ -1906,21 +2198,13 @@ export function snapshotEvidence({ throw new Error(`${guard.absolute} changed before evidence materialization completed`); } } else if (guard.type === 'absence') { + // statIdentity is a strict superset of stableDirectoryIdentity on the + // same stat, so comparing both could only ever fire together. const parent = fs.fstatSync(guard.fd, { bigint: true }); - if ( - !parent.isDirectory() || - stableDirectoryIdentity(parent) !== guard.parentIdentity || - statIdentity(parent) !== guard.parentMutationIdentity - ) { + if (!parent.isDirectory() || statIdentity(parent) !== guard.parentMutationIdentity) { throw new Error(`Absence anchor changed for ${guard.repoPath}`); } - try { - fs.lstatSync(descriptorPath(guard.fd, guard.childName), { bigint: true }); - } catch (error) { - if (error?.code === 'ENOENT') continue; - throw error; - } - throw new Error(`${guard.repoPath} appeared before evidence materialization completed`); + anchoringBackend().verifyAbsentChild(guard); } } for (const guard of headGuards) verifyControlFile(guard); @@ -1955,12 +2239,10 @@ export function snapshotEvidence({ cited_path_manifest: citedEntries, }; } finally { - const closed = new Set(); - for (const guard of mutationGuards) { - if (guard.type !== 'absence' || closed.has(guard.fd)) continue; - closed.add(guard.fd); + // One entry per distinct anchored directory, so one close per descriptor. + for (const handle of absenceCache.values()) { try { - fs.closeSync(guard.fd); + fs.closeSync(handle.fd); } catch { // Preserve the primary snapshot result/error. } diff --git a/gitnexus-claude-plugin/skills/gitnexus-refactoring/SKILL.md b/gitnexus-claude-plugin/skills/gitnexus-refactoring/SKILL.md index 2dbb71ca0..4f10bbc6a 100644 --- a/gitnexus-claude-plugin/skills/gitnexus-refactoring/SKILL.md +++ b/gitnexus-claude-plugin/skills/gitnexus-refactoring/SKILL.md @@ -87,6 +87,11 @@ detect_changes({scope: "all"}) → Risk: MEDIUM ``` +`partial: true` (a graph query failed) or `truncated: true` (the changed-symbol +listing was capped) means the result is short of the truth: a short or empty +list is not proof that only the expected files changed. Re-run it rather than +treat the refactor as verified. + **cypher** — custom reference queries: ```cypher diff --git a/gitnexus-claude-plugin/skills/gitnexus-work/SKILL.md b/gitnexus-claude-plugin/skills/gitnexus-work/SKILL.md index 4f7856ea7..f9baab16a 100644 --- a/gitnexus-claude-plugin/skills/gitnexus-work/SKILL.md +++ b/gitnexus-claude-plugin/skills/gitnexus-work/SKILL.md @@ -216,7 +216,10 @@ Work through plan §7 step by step, in order. For each step: `detect_changes` → commit as one unbroken sequence from the repository root — interleaving other work between the gate and the commit is how the gate gets skipped. Unexpected - affected flows → investigate before committing, not after. + affected flows → investigate before committing, not after. A result + flagged `partial` (a graph query failed) or `truncated` (the symbol + listing was capped) blocks the commit the same way: the gate did not + see every changed symbol, so re-run it rather than read it as clean. A relationship-affecting implementation edit or commit invalidates the procedure's prior proof. The next step must perform the required inter-step diff --git a/gitnexus-claude-plugin/skills/gitnexus-work/references/evidence-provenance.md b/gitnexus-claude-plugin/skills/gitnexus-work/references/evidence-provenance.md index c686599da..3df5a046d 100644 --- a/gitnexus-claude-plugin/skills/gitnexus-work/references/evidence-provenance.md +++ b/gitnexus-claude-plugin/skills/gitnexus-work/references/evidence-provenance.md @@ -98,8 +98,11 @@ excluded. ## Safe existing-plan read contract -`read-plan` fails closed unless Linux `/proc/self/fd`, `O_DIRECTORY`, and -`O_NOFOLLOW` are available. It resolves the exact Git top-level, opens the +`read-plan` fails closed unless the host platform can resolve names against a +held directory descriptor: Linux `/proc/self/fd` with `O_DIRECTORY` and +`O_NOFOLLOW`, or macOS `O_DIRECTORY`/`O_NOFOLLOW`. Every other platform is +refused outright — an unverified read is not a degraded read, it is a different, +racy operation. It resolves the exact Git top-level, opens the repository root and every plan parent as held no-follow directory descriptors, rejects missing, symlink, non-directory, and escaping parents, and opens the leaf with `O_NOFOLLOW`. It reads at most 16 MiB from that held file descriptor, @@ -109,13 +112,17 @@ Neither Deepen nor work may parse bytes obtained before or outside this receipt. ## Safe generated-plan write contract -The writer fails closed unless Linux `/proc/self/fd`, `O_DIRECTORY`, -`O_NOFOLLOW`, and Python 3 with libc `renameat2(RENAME_NOREPLACE)` support are -available. Python may live in `/usr/local`, a Nix profile, or another absolute -PATH directory, but the helper accepts only a resolved executable and -containing directory owned by root or the current user and not writable by -group/other. The resolved executable is opened without following links and -invoked through that held descriptor. Relative PATH entries are ignored. The plan parent and the +The writer fails closed unless the host platform offers `O_DIRECTORY` and +`O_NOFOLLOW`, plus `/proc/self/fd` on Linux. It spawns no interpreter and loads +no native code: publication is `link(2)`, which is atomic, fails `EEXIST` when +the destination name is taken, and refuses a symlinked destination without +following it — the same no-replace guarantee `renameat2(RENAME_NOREPLACE)` and +`renameatx_np(RENAME_EXCL)` provide, available through `fs.linkSync` on every +supported platform. The temporary name is unlinked once the link succeeds; the +published file is the same inode the writer created and verified, so every +identity check downstream holds by construction. A link that succeeds followed +by an unlink that fails leaves the plan published and is reported as success, +because it is one. The plan parent and the repository's Git-admin directory must also share a filesystem. It resolves the target repository's exact Git top-level, opens that root and every destination parent as held no-follow directory descriptors, creates missing @@ -128,15 +135,45 @@ The writer creates a random exclusive temporary file relative to the held final parent descriptor and keeps its no-follow descriptor open. It writes and flushes the bytes, binds the temporary name to the opened inode, and hashes the open file before publication. Immediately before publication it revalidates -the parent and the temporary path, inode, size, and digest. Publication uses an -atomic no-replace move relative to the held directory descriptor. Initial mode -therefore cannot overwrite a destination that appears after the absent check. +the parent and the temporary path, inode, size, and digest. Publication links +the temporary name to the destination relative to the held directory +descriptor, which fails rather than replaces if the destination is taken. +Initial mode therefore cannot overwrite a destination that appears after the +absent check. The writer then flushes the directory and revalidates the committed path by opening it with `O_NOFOLLOW`, hashing both the original temporary fd and the path-bound fd, and performing a second descriptor-anchored path identity check after hashing. A detected mutation or replacement aborts instead of accepting mixed-era output. +### Linux anchors, macOS verifies + +The two platforms reach the same destination by different proofs, and the +difference is real enough to state rather than smooth over. + +On Linux every name resolves through `/proc/self/fd//`, a magic link +the kernel resolves against the inode the descriptor already holds. The names +above it are never re-walked, so an attacker who renames a parent between the +check and the use cannot redirect the operation. The race is impossible, not +merely detected. + +macOS has no such path. `/dev/fd/` is a devfs node, not a magic link: it can +be opened, but nothing can be resolved through it. `open("/dev/fd//child")` +returns `ENOENT`, and `realpath` of it returns `/dev/fd/` rather than the +directory's path — measured on macOS 26, not inferred. Node exposes no `openat`, +no `dir_fd` parameter, and no FFI, so on macOS the writer resolves names +lexically with `O_NOFOLLOW` at every component, holds an open descriptor on +every directory in the chain for the whole operation, and proves before *and* +after each step that the chain still names exactly the inodes it is holding. +Holding the descriptors is what makes the recorded inode numbers trustworthy: +an open descriptor pins its inode, so a freed number cannot be recycled beneath +the walk. + +What that buys is detection rather than prevention. A parent swapped inside the +window between a check and its use is caught by the check that follows, and the +operation aborts having written nothing — but on Linux it could not have +happened at all. No published byte escapes verification on either platform. + `--replace` accepts only a pre-existing regular file and is reserved for Deepen; without it, accidental overwrite is rejected. It also requires the exact canonical `generated_plan_path` and `plan_digest` from the same session's diff --git a/gitnexus-claude-plugin/skills/gitnexus-work/scripts/evidence-provenance.mjs b/gitnexus-claude-plugin/skills/gitnexus-work/scripts/evidence-provenance.mjs index 181d2120b..793fe4cd8 100644 --- a/gitnexus-claude-plugin/skills/gitnexus-work/scripts/evidence-provenance.mjs +++ b/gitnexus-claude-plugin/skills/gitnexus-work/scripts/evidence-provenance.mjs @@ -479,11 +479,11 @@ function resolveOwnGitTopLevel(absolute) { if (result.status !== 0) return null; let topLevel; try { - topLevel = fs.realpathSync(decodeUtf8(result.stdout, 'nested repository root').trim()); + topLevel = fs.realpathSync.native(decodeUtf8(result.stdout, 'nested repository root').trim()); } catch { return null; } - return topLevel === fs.realpathSync(absolute) ? topLevel : null; + return topLevel === fs.realpathSync.native(absolute) ? topLevel : null; } function readOwnGitlinkHead(absolute) { @@ -616,17 +616,30 @@ function filesystemObject(absolute, expectedKind, mutationGuards, testHooks) { throw new Error(`Unsupported filesystem object at ${absolute}`); } -function guardPathParents(repo, repoPath, mutationGuards) { +// Every dirty path re-walks its own parents, and dirty paths overwhelmingly +// share them — the repository root is re-stat'ed once per path. `guarded` is +// per-snapshot and remembers which absolute directories already carry a guard, +// so each distinct directory is stat'ed and guarded exactly once. +// +// Keeping the first-seen identity is the conservative choice: verifyGuards +// re-checks every guard against the filesystem at the end, so a directory that +// changes after it was guarded still fails there. Skipping a re-stat cannot hide +// a change; it only avoids recording the same directory twice. +function guardPathParents(repo, repoPath, mutationGuards, guarded) { const components = repoPath.split('/'); let current = repo; - const rootStat = fs.lstatSync(repo, { bigint: true }); - mutationGuards.push({ - type: 'directory', - absolute: repo, - identity: stableDirectoryIdentity(rootStat), - }); + if (!guarded.has(repo)) { + guarded.add(repo); + mutationGuards.push({ + type: 'directory', + absolute: repo, + identity: stableDirectoryIdentity(fs.lstatSync(repo, { bigint: true })), + }); + } for (const component of components.slice(0, -1)) { current = path.join(current, component); + // Already proved a real directory and already guarded on an earlier path. + if (guarded.has(current)) continue; let stat; try { stat = fs.lstatSync(current, { bigint: true }); @@ -638,6 +651,7 @@ function guardPathParents(repo, repoPath, mutationGuards) { throw new Error(`Refusing to traverse symlink parent for ${repoPath}`); } if (!stat.isDirectory()) return; + guarded.add(current); mutationGuards.push({ type: 'directory', absolute: current, @@ -646,81 +660,153 @@ function guardPathParents(repo, repoPath, mutationGuards) { } } -function recordAnchoredAbsence(repo, repoPath, mutationGuards) { - requireDescriptorAnchoring(); - const flags = - fs.constants.O_RDONLY | - fs.constants.O_DIRECTORY | - fs.constants.O_NOFOLLOW | - (fs.constants.O_CLOEXEC ?? 0); - const descriptors = []; - let retainedFd; - try { - let currentFd = fs.openSync(repo, flags); - descriptors.push(currentFd); - const components = repoPath.split('/'); - for (let index = 0; index < components.length; index += 1) { - const component = components[index]; - const child = descriptorPath(currentFd, component); - let childStat; - try { - childStat = fs.lstatSync(child, { bigint: true }); - } catch (error) { - if (error?.code !== 'ENOENT' && error?.code !== 'ENOTDIR') throw error; - const parentStat = fs.fstatSync(currentFd, { bigint: true }); - if (!parentStat.isDirectory()) { - throw new Error(`Absence parent is no longer a directory for ${repoPath}`); - } - retainedFd = currentFd; - mutationGuards.push({ - type: 'absence', - fd: retainedFd, - childName: component, - repoPath, - parentIdentity: stableDirectoryIdentity(parentStat), - parentMutationIdentity: statIdentity(parentStat), - }); - for (const fd of descriptors) { - if (fd !== retainedFd) fs.closeSync(fd); - } - return; - } - if (index === components.length - 1) { - throw new Error(`${repoPath} appeared while its absence was being anchored`); - } - if (childStat.isSymbolicLink() || !childStat.isDirectory()) { - throw new Error(`Refusing a non-directory parent while anchoring absence for ${repoPath}`); - } - const nextFd = fs.openSync(child, flags); - descriptors.push(nextFd); - currentFd = nextFd; - } - throw new Error(`Could not anchor absence for ${repoPath}`); - } catch (error) { - for (const fd of descriptors) { - if (fd === retainedFd) continue; - try { - fs.closeSync(fd); - } catch { - // Preserve the primary absence-anchoring error. - } - } - throw error; +// A bound, not a bug: the absence cache deduplicates correctly and leaks nothing, +// but citedPaths is caller-supplied and unbounded, so a pathological snapshot +// could hold more descriptors than the process is allowed (macOS +// kern.maxfilesperproc is 24576). The peak precedes a `git` spawn, so exhaustion +// would surface as a git failure misreported as evidence instability. +// +// Refuse rather than evict: closing a cached descriptor would silently break the +// pinned chain of an absence guard that was already recorded against it, which is +// exactly the inode-recycling hole the pins exist to close. +const ABSENCE_ANCHOR_LIMITS = Object.freeze({ maxPinnedDirectories: 4096 }); + +// Every no-follow read and every exclusive create in this file uses one of these +// two, so a change lands in one place rather than in seven. +const VERIFIED_READ_FLAGS = + fs.constants.O_RDONLY | fs.constants.O_NOFOLLOW | (fs.constants.O_CLOEXEC ?? 0); +const VERIFIED_CREATE_FLAGS = + fs.constants.O_RDWR | + fs.constants.O_CREAT | + fs.constants.O_EXCL | + fs.constants.O_NOFOLLOW | + (fs.constants.O_CLOEXEC ?? 0); + +function requireAbsenceAnchorCapacity(cache) { + if (cache.size >= ABSENCE_ANCHOR_LIMITS.maxPinnedDirectories) { + throw new Error( + `Absence anchoring exceeds ${ABSENCE_ANCHOR_LIMITS.maxPinnedDirectories} pinned directories`, + ); } } -function materializeRecord(repo, statusRecord, layers, mutationGuards, testHooks) { +const ANCHORED_DIRECTORY_FLAGS = + fs.constants.O_RDONLY | + fs.constants.O_DIRECTORY | + fs.constants.O_NOFOLLOW | + (fs.constants.O_CLOEXEC ?? 0); + +// Every absence receipt is verified long after its walk returns, so the chain +// that produced it has to stay pinned until the snapshot ends — an unpinned inode +// number can be recycled by a replacement directory that then reproduces the +// recorded identity exactly. Absent cited paths overwhelmingly share prefixes, so +// the walked directories are cached per snapshot and keyed by repo-relative +// prefix: one open descriptor and one anchored walk per distinct directory rather +// than per path. snapshotEvidence owns every descriptor in this cache and closes +// each exactly once; guards only borrow them for verification. +function anchoredAbsenceRoot(repo, cache) { + const cached = cache.get(''); + if (cached) return cached; + requireAbsenceAnchorCapacity(cache); + const fd = openVerifiedDirectory(repo, ANCHORED_DIRECTORY_FLAGS); + const handle = { + fd, + expectedPath: repo, + chain: [ + { expectedPath: repo, identity: stableDirectoryIdentity(fs.fstatSync(fd, { bigint: true })) }, + ], + descriptors: [fd], + }; + cache.set('', handle); + return handle; +} + +function recordAnchoredAbsence(repo, repoPath, mutationGuards, cache) { + requireDescriptorAnchoring(); + const components = repoPath.split('/'); + let handle = anchoredAbsenceRoot(repo, cache); + let prefix = ''; + for (let index = 0; index < components.length; index += 1) { + const component = components[index]; + const isFinal = index === components.length - 1; + prefix = prefix === '' ? component : `${prefix}/${component}`; + // The final component is always re-checked against the filesystem: it is the + // one whose absence is being recorded, and a cached answer would be a stale + // one. Only the prefix directories are reused. + const cached = isFinal ? undefined : cache.get(prefix); + if (cached) { + handle = cached; + continue; + } + const child = anchoredChild(handle, component); + let childStat; + try { + childStat = lstatChild(child); + } catch (error) { + if (error?.code !== 'ENOENT' && error?.code !== 'ENOTDIR') throw error; + const parentStat = fs.fstatSync(handle.fd, { bigint: true }); + if (!parentStat.isDirectory()) { + throw new Error(`Absence parent is no longer a directory for ${repoPath}`); + } + mutationGuards.push({ + type: 'absence', + // The handle is the holder the guard verifies against, and `ref` is the + // child path already built through the anchoredChild chokepoint — the + // guard must never re-derive that name itself. + handle, + ref: child, + fd: handle.fd, + repoPath, + parentMutationIdentity: statIdentity(parentStat), + }); + return; + } + if (isFinal) { + throw new Error(`${repoPath} appeared while its absence was being anchored`); + } + if (childStat.isSymbolicLink() || !childStat.isDirectory()) { + throw new Error(`Refusing a non-directory parent while anchoring absence for ${repoPath}`); + } + requireAbsenceAnchorCapacity(cache); + const childFd = openVerifiedDirectory(child.path, ANCHORED_DIRECTORY_FLAGS); + const expectedPath = path.join(handle.expectedPath, component); + let next; + try { + if (!anchoringBackend().descriptorMatchesChild(childFd, expectedPath, childStat)) { + throw new Error( + `Absence parent descriptor does not match its verified inode for ${repoPath}`, + ); + } + next = { + fd: childFd, + expectedPath, + chain: [...handle.chain, { expectedPath, identity: stableDirectoryIdentity(childStat) }], + descriptors: [...handle.descriptors, childFd], + }; + } catch (error) { + fs.closeSync(childFd); + throw error; + } + cache.set(prefix, next); + handle = next; + } + throw new Error(`Could not anchor absence for ${repoPath}`); +} + +function materializeRecord(repo, statusRecord, layers, mutationGuards, testHooks, walkState) { const head = layers.head(statusRecord.path); const index = layers.index(statusRecord.path); const expectedKind = index.kind === 'gitlink' || head.kind === 'gitlink' ? 'gitlink' : null; - guardPathParents(repo, statusRecord.path, mutationGuards); + guardPathParents(repo, statusRecord.path, mutationGuards, walkState.guardedDirectories); const filesystem = filesystemObject( path.join(repo, ...statusRecord.path.split('/')), expectedKind, mutationGuards, testHooks, ); - if (filesystem.kind === ABSENT) recordAnchoredAbsence(repo, statusRecord.path, mutationGuards); + if (filesystem.kind === ABSENT) { + recordAnchoredAbsence(repo, statusRecord.path, mutationGuards, walkState.absenceCache); + } if (statusRecord.directory_hint && filesystem.kind !== 'directory') { throw new Error( `Git reported an embedded directory but found ${filesystem.kind}: ${statusRecord.path}`, @@ -789,9 +875,15 @@ export function serializeDirtyRecords(entries) { } function assertRepository(repoInput) { - const repo = fs.realpathSync(requireString(repoInput, 'repo')); + // realpathSync.native, not realpathSync: the JS resolver preserves a Windows + // 8.3 short component (C:\Users\RUNNER~1\...) while git always reports the long + // form, so the two would never compare equal and every caller would be told the + // worktree root is not the worktree root it just named. + const repo = fs.realpathSync.native(requireString(repoInput, 'repo')); const topLevelResult = git(repo, ['rev-parse', '--show-toplevel']); - const topLevel = fs.realpathSync(decodeUtf8(topLevelResult.stdout, 'repository root').trim()); + const topLevel = fs.realpathSync.native( + decodeUtf8(topLevelResult.stdout, 'repository root').trim(), + ); if (topLevel !== repo) throw new Error(`--repo must be the Git worktree root (${topLevel})`); return repo; } @@ -882,17 +974,48 @@ function stableFileIdentity(stat) { return [stat.dev, stat.ino, stat.mode, stat.size].map(String).join(':'); } +// The two backends below differ in one decisive way, and it is worth stating +// plainly because the security properties are not the same. +// +// Linux ANCHORS. A name is resolved through /proc/self/fd//, which +// starts the walk at the inode the descriptor holds, so a parent that is renamed +// away cannot be traversed at all: the descriptor keeps pointing at the original +// directory and the impostor planted at the same name is simply never reached. +// +// macOS VERIFIES. Node cannot resolve a name relative to a descriptor there — +// /dev/fd/ is not a magic link (it stats as the directory but every attempt +// to traverse a child through it returns ENOENT), and fcntl F_GETPATH is a +// name-cache snapshot rather than a live anchor. So the Darwin backend resolves +// lexically, holds an open descriptor on every element of the chain, and proves +// before and after each operation that the path chain still names exactly the +// inodes it is holding. That DETECTS a swapped parent and aborts the write; it +// does not make the swap impossible the way the Linux path does. A swap landing +// inside the window between a check and the call it guards is caught by the +// following check, after the fact, rather than being unreachable. +// +// Every other platform gets neither and is refused outright. function requireDescriptorAnchoring() { - if ( - process.platform !== 'linux' || - fs.constants.O_DIRECTORY === undefined || - fs.constants.O_NOFOLLOW === undefined || - !fs.existsSync('/proc/self/fd') - ) { - throw new Error( - 'Safe generated-plan writes require Linux /proc/self/fd and O_DIRECTORY/O_NOFOLLOW; refusing an unanchored write', - ); + const directoryFlagsAvailable = + fs.constants.O_DIRECTORY !== undefined && fs.constants.O_NOFOLLOW !== undefined; + if (process.platform === 'linux') { + if (!directoryFlagsAvailable || !fs.existsSync('/proc/self/fd')) { + throw new Error( + 'Safe generated-plan writes require Linux /proc/self/fd and O_DIRECTORY/O_NOFOLLOW; refusing an unanchored write', + ); + } + return; } + if (process.platform === 'darwin') { + if (!directoryFlagsAvailable) { + throw new Error( + 'Safe generated-plan writes require macOS O_DIRECTORY/O_NOFOLLOW; refusing an unverified write', + ); + } + return; + } + throw new Error( + `Safe generated-plan writes require Linux /proc/self/fd or macOS O_DIRECTORY/O_NOFOLLOW; ${process.platform} offers neither, so refusing an unanchored write`, + ); } function descriptorPath(fd, childName) { @@ -900,157 +1023,352 @@ function descriptorPath(fd, childName) { return childName === undefined ? base : path.join(base, childName); } -function externalDescriptorPath(fd, childName) { - const base = `/proc/${process.pid}/fd/${fd}`; - return childName === undefined ? base : path.join(base, childName); +// Directory opens are plain O_RDONLY|O_DIRECTORY|O_NOFOLLOW|O_CLOEXEC on both +// platforms, and deliberately nothing else. +// +// O_NOFOLLOW_ANY (macOS 11+) used to be ORed in here on the theory that XNU +// ignores unrecognized open flag bits, so it would be inert where unsupported. +// That was wrong: combined with O_DIRECTORY macOS rejects it outright with +// EINVAL, and every directory open on Darwin failed. It is gone and is not +// coming back behind a probe or a degrade-on-EINVAL path — the per-component +// O_NOFOLLOW walk is what delivers the guarantee. Rust's cap-std, the closest +// reference implementation of this problem, has not adopted O_NOFOLLOW_ANY +// either (their issue #179 is still open). +function openVerifiedDirectory(absolute, flags) { + return fs.openSync(absolute, flags); } -const RENAME_NOREPLACE_SCRIPT = String.raw` -import ctypes -import errno -import os -import sys - -libc = ctypes.CDLL(None, use_errno=True) -try: - renameat2 = libc.renameat2 -except AttributeError: - print("libc does not expose renameat2", file=sys.stderr) - raise SystemExit(125) - -renameat2.argtypes = [ctypes.c_int, ctypes.c_char_p, ctypes.c_int, ctypes.c_char_p, ctypes.c_uint] -renameat2.restype = ctypes.c_int -result = renameat2(-100, os.fsencode(sys.argv[1]), -100, os.fsencode(sys.argv[2]), 1) -if result != 0: - error_number = ctypes.get_errno() - error_name = errno.errorcode.get(error_number, "UNKNOWN") - print(f"renameat2 RENAME_NOREPLACE failed: {error_name}: {os.strerror(error_number)}", file=sys.stderr) - raise SystemExit(17 if error_number == errno.EEXIST else 126) -`; - -let atomicMoverPath; - -function spawnHeldExecutable(executable, args, options) { - const before = fs.fstatSync(executable.fd, { bigint: true }); - if (!before.isFile() || statIdentity(before) !== executable.identity) { - throw new Error('Validated Python executable changed before invocation'); - } - const result = spawnSync('/proc/self/fd/3', args, { - ...options, - stdio: ['ignore', 'pipe', 'pipe', executable.fd], - }); - const after = fs.fstatSync(executable.fd, { bigint: true }); - assertStableIdentity(before, after, 'validated Python executable'); - return result; +// File opens additionally get O_NONBLOCK, which directory opens do not need: +// it stops a FIFO swapped in at the target name from wedging the process on +// open. The identity comparison that follows rejects the FIFO anyway, but only +// if we ever get as far as running it. +function openVerifiedFile(absolute, flags, mode) { + const nonBlocking = flags | (fs.constants.O_NONBLOCK ?? 0); + return mode === undefined + ? fs.openSync(absolute, nonBlocking) + : fs.openSync(absolute, nonBlocking, mode); } -function validatedPathExecutable(candidate) { - if (!path.isAbsolute(candidate)) return null; - const candidateDirectory = path.dirname(candidate); - let resolvedDirectory; - let resolved; - let directoryStats; - let executableStat; +// The publish primitive, identical on both platforms. +// +// link() is the portable no-replace publish: it fails with EEXIST if the +// destination name is taken — by a regular file, by a directory, or by a symlink, +// live or dangling — and it never follows that symlink to clobber its target. +// It also works where renameat2(RENAME_NOREPLACE) does not, notably v9fs, which +// is why the WSL2 9p case that used to fail every time now works. +// +// The published file is the same inode as the temporary, so every identity +// comparison the callers already make still holds, and validateCommittedPlan +// becomes strictly stronger: it compares the destination against the exact inode +// whose bytes were fsynced. +// +// On Linux both paths are /proc/self/fd//, so the publish is anchored +// to the held parent descriptors exactly like every other operation. +// link(2) BUGS: "On NFS filesystems, the return code may be wrong in case the NFS +// server performs the link creation and dies before it can say so. Use stat(2) to +// find out if the link got created." open(2) NOTES gives the remedy this +// implements: on a reported failure, stat the source and see whether its link +// count reached 2. A false positive would need someone to have hardlinked a +// 16-random-byte name inside a directory we hold open — and validateCommittedPlan +// still proves the destination is the exact temporary inode afterwards. +function linkCreatedDespiteError(sourcePath) { try { - resolvedDirectory = fs.realpathSync(candidateDirectory); - resolved = fs.realpathSync(candidate); - const resolvedExecutableDirectory = fs.realpathSync(path.dirname(resolved)); - directoryStats = [...new Set([resolvedDirectory, resolvedExecutableDirectory])].map( - (directory) => fs.statSync(directory), - ); - executableStat = fs.lstatSync(resolved); - fs.accessSync(resolved, fs.constants.X_OK); + return fs.statSync(sourcePath, { bigint: true }).nlink === 2n; } catch { - return null; + return false; } - if ( - directoryStats.some((stat) => !stat.isDirectory()) || - !executableStat.isFile() || - executableStat.isSymbolicLink() - ) { - return null; - } - const uid = typeof process.getuid === 'function' ? process.getuid() : null; - const trustedOwner = (stat) => uid === null || stat.uid === 0 || stat.uid === uid; - if ( - directoryStats.some((stat) => !trustedOwner(stat) || (stat.mode & 0o022) !== 0) || - !trustedOwner(executableStat) || - (executableStat.mode & 0o022) !== 0 - ) { - return null; - } - return resolved; } -function resolveAtomicMover() { - if (atomicMoverPath) return atomicMoverPath; - const candidates = new Set(); - for (const entry of (process.env.PATH ?? '').split(path.delimiter)) { - if (entry && path.isAbsolute(entry)) candidates.add(path.join(entry, 'python3')); - } - for (const entry of ['/usr/local/bin/python3', '/usr/bin/python3', '/bin/python3']) { - candidates.add(entry); - } - for (const candidate of candidates) { - const resolved = validatedPathExecutable(candidate); - if (!resolved) continue; - let fd; - try { - fd = fs.openSync( - resolved, - fs.constants.O_RDONLY | fs.constants.O_NOFOLLOW | (fs.constants.O_CLOEXEC ?? 0), - ); - } catch { - continue; +function linkNoReplace(sourcePath, destinationPath) { + try { + fs.linkSync(sourcePath, destinationPath); + } catch (error) { + // Callers treat "destination taken" as a distinct outcome, not a failure. + if (error?.code === 'EEXIST') return false; + if (!linkCreatedDespiteError(sourcePath)) { + // FAT, Coda, and some SMB/FUSE/virtiofs mounts have no hardlinks at all. + // Git falls back to rename here, but git can afford to lose collision + // detection because its objects are content-addressed; a plan destination + // is a plain name, so a replacing rename would silently clobber whatever + // is already there. Refuse loudly instead. + if (error?.code === 'EPERM' || error?.code === 'ENOTSUP' || error?.code === 'EMLINK') { + throw new Error( + `Generated-plan publication requires hard links, which this filesystem refused (${error.code}); refusing to fall back to a replacing rename`, + ); + } + throw error; } - const opened = fs.fstatSync(fd, { bigint: true }); - const executable = { fd, identity: statIdentity(opened), resolved }; - const version = spawnHeldExecutable( - executable, - ['-I', '-S', '-c', 'import sys; print(sys.version_info[0])'], - { - encoding: 'utf8', - env: { ...process.env, LANG: 'C', LC_ALL: 'C' }, - timeout: 10_000, - windowsHide: true, - }, - ); - if (version.status === 0 && version.stdout.trim() === '3') { - atomicMoverPath = executable; - return executable; - } - fs.closeSync(fd); } - throw new Error( - 'Safe generated-plan publication requires a trusted absolute Python 3 PATH candidate with libc renameat2 support', - ); -} - -function atomicMoveNoReplace(source, destination) { - const mover = resolveAtomicMover(); - const result = spawnHeldExecutable( - mover, - ['-I', '-S', '-c', RENAME_NOREPLACE_SCRIPT, source, destination], - { - encoding: 'utf8', - env: { ...process.env, LANG: 'C', LC_ALL: 'C' }, - timeout: 10_000, - windowsHide: true, - }, - ); - if (result.error) throw result.error; - if (result.status === 17) return false; - if (result.status !== 0) { - throw new Error( - `Atomic no-replace move failed (${result.status}): ${(result.stderr ?? '').trim()}`, - ); + try { + fs.unlinkSync(sourcePath); + } catch { + // The link succeeded, so the plan IS published. A temporary name left behind + // is a stray file, not an unpublished plan: reporting it as a failure would + // be a lie, and rolling back would unpublish a plan that is already live. } return true; } -function lstatOptional(absolute) { +// A directory holder is anything that owns a verified chain: a plan-parent +// handle, a ref's parent directory, or an absence guard. Two arrays describe it, +// both root-first and the same length — `chain` records each element's expected +// path and dev/ino/mode, and `descriptors` holds an open descriptor on each. +// +// Holding those descriptors is load-bearing rather than decorative. dev/ino/mode +// is unique only among *live* inodes: an inode number freed by an rmdir is handed +// straight back to the next mkdir, so a replacement directory can reproduce a +// recorded identity exactly. An open descriptor pins the inode, so the number +// cannot be recycled for as long as the holder exists. +function verifyPinnedDescriptors(holder) { + const { chain, descriptors } = holder; + if (!Array.isArray(descriptors) || descriptors.length !== chain.length) { + throw new Error('Generated-plan parent chain is missing the descriptors that pin it'); + } + chain.forEach((item, index) => { + const pinned = fs.fstatSync(descriptors[index], { bigint: true }); + if (!pinned.isDirectory() || stableDirectoryIdentity(pinned) !== item.identity) { + throw new Error('Generated-plan parent descriptor changed during the write'); + } + }); +} + +function verifyLexicalChain(holder) { + for (const item of holder.chain) { + let lexical; + try { + lexical = fs.lstatSync(item.expectedPath, { bigint: true }); + } catch (error) { + if (error?.code !== 'ENOENT' && error?.code !== 'ENOTDIR') throw error; + // A parent renamed out from under us is a mismatch, not a missing file: + // reporting the raw ENOENT would leak an unrelated-looking error out of a + // check whose whole job is to say the chain no longer holds. + throw new Error('Generated-plan lexical parent no longer matches its directory descriptor'); + } + if ( + lexical.isSymbolicLink() || + !lexical.isDirectory() || + stableDirectoryIdentity(lexical) !== item.identity + ) { + throw new Error('Generated-plan lexical parent no longer matches its directory descriptor'); + } + } +} + +// The whole platform seam, in five methods. Everything else an operation does is +// identical on both platforms and lives in the shared functions below. +// +// Only two things actually differ: how a name becomes a path, and what guard +// wraps the operation that uses it. +// +// Linux ANCHORS. /proc/self/fd// starts the walk at the inode the +// descriptor holds, so a parent renamed away cannot be traversed at all and the +// guard is a no-op — there is nothing left to verify. +// +// macOS VERIFIES. It resolves lexically, so before and after every operation it +// proves that each element of the path chain still names the exact inode being +// held for it. That DETECTS a swapped parent and aborts; it does not make the +// swap impossible. A swap landing inside the window is caught by the trailing +// check, after the fact, rather than being unreachable. The check runs after a +// failure too, because a verdict observed through a chain that has since changed +// is not a verdict. +const LINUX_ANCHORING = { + childPath(dirHandle, childName) { + return descriptorPath(dirHandle.fd, childName); + }, + verified(holders, run) { + return run(); + }, + descriptorMatchesChild(fd, expectedPath) { + return fs.realpathSync.native(descriptorPath(fd)) === expectedPath; + }, + parentStillResolves(parentHandle) { + return fs.realpathSync.native(descriptorPath(parentHandle.fd)) === parentHandle.expectedPath; + }, + verifyAbsentChild(guard) { + if (absentChildIsPresent(guard.ref)) { + throw new Error(`${guard.repoPath} appeared before evidence materialization completed`); + } + }, +}; + +const DARWIN_ANCHORING = { + childPath(dirHandle, childName) { + return path.join(dirHandle.expectedPath, childName); + }, + verified(holders, run) { + const list = Array.isArray(holders) ? holders : [holders]; + const proveChain = () => { + for (const holder of list) { + verifyPinnedDescriptors(holder); + verifyLexicalChain(holder); + } + }; + proveChain(); + let value; + try { + value = run(); + } catch (error) { + proveChain(); + throw error; + } + proveChain(); + return value; + }, + descriptorMatchesChild(fd, _expectedPath, childStat) { + // There is no live fd-to-path oracle on macOS (F_GETPATH is a name-cache + // snapshot, not an anchor), so escape is decided the other way round: the + // name was just resolved under a verified chain, and the descriptor opened + // from it counts only if it is that same inode. + const opened = fs.fstatSync(fd, { bigint: true }); + return ( + opened.isDirectory() && stableDirectoryIdentity(opened) === stableDirectoryIdentity(childStat) + ); + }, + parentStillResolves(parentHandle) { + // Both halves are needed: a directory renamed away keeps its inode, so the + // descriptors alone still match and only the lexical half notices it moved. + try { + verifyPinnedDescriptors(parentHandle); + verifyLexicalChain(parentHandle); + } catch { + return false; + } + return true; + }, + verifyAbsentChild(guard) { + let present; + try { + present = DARWIN_ANCHORING.verified(guard.handle, () => absentChildIsPresent(guard.ref)); + } catch (error) { + // A chain that no longer holds makes the absence verdict meaningless, and + // the caller reports that as the anchor changing rather than as a stray + // parent-descriptor error. Linux cannot reach this: its guard is a no-op. + throw new Error( + `Absence anchor changed for ${guard.repoPath}: ${ + error instanceof Error ? error.message : String(error) + }`, + ); + } + if (present) { + throw new Error(`${guard.repoPath} appeared before evidence materialization completed`); + } + }, +}; + +const ANCHORING_BACKENDS = new Map([ + ['linux', LINUX_ANCHORING], + ['darwin', DARWIN_ANCHORING], +]); + +function anchoringBackend() { + const backend = ANCHORING_BACKENDS.get(process.platform); + if (!backend) { + // requireDescriptorAnchoring normally refuses first; this is the same answer + // from the other side, so an unsupported platform can never fall through to + // whichever backend happened to be the ternary's default. + throw new Error( + `No generated-plan anchoring backend for ${process.platform}; refusing an unanchored write`, + ); + } + return backend; +} + +// Open, fstat, compare, close on mismatch. The descriptor never escapes this +// function unless it refers to the inode the caller already verified by name, so +// a lexical open that landed anywhere else cannot be used by accident. On Linux +// the comparison passes trivially — the /proc walk already resolved from the +// held parent — and costs one fstat to keep the guarantee structural rather than +// dependent on which backend is in play. +function adoptVerifiedFile(ref, expectedStat, flags) { + const fd = openVerifiedFile(ref.path, flags); + let opened; try { - return fs.lstatSync(absolute, { bigint: true }); + opened = fs.fstatSync(fd, { bigint: true }); + } catch (error) { + fs.closeSync(fd); + throw error; + } + if (stableFileIdentity(opened) !== stableFileIdentity(expectedStat)) { + fs.closeSync(fd); + return null; + } + return fd; +} + +function absentChildIsPresent(ref) { + try { + fs.lstatSync(ref.path, { bigint: true }); + } catch (error) { + if (error?.code === 'ENOENT') return false; + throw error; + } + return true; +} + +// The operations. Each is the same on both platforms; only the guard differs. +function lstatChild(ref) { + return anchoringBackend().verified(ref.dir, () => fs.lstatSync(ref.path, { bigint: true })); +} + +function openChildRead(ref, flags, expectedStat) { + return anchoringBackend().verified(ref.dir, () => { + const fd = adoptVerifiedFile(ref, expectedStat, flags); + if (fd === null) { + throw new Error(`${ref.name} was replaced between its verified stat and its no-follow open`); + } + return fd; + }); +} + +function createChild(ref, flags, mode) { + // O_CREAT|O_EXCL|O_NOFOLLOW is atomic at the leaf, so the only thing the guard + // has to cover is which directory the leaf landed in. + return anchoringBackend().verified(ref.dir, () => openVerifiedFile(ref.path, flags, mode)); +} + +function mkdirChild(ref, mode) { + anchoringBackend().verified(ref.dir, () => fs.mkdirSync(ref.path, { mode })); +} + +function publishNoReplace(sourceRef, destinationRef) { + return anchoringBackend().verified([sourceRef.dir, destinationRef.dir], () => + linkNoReplace(sourceRef.path, destinationRef.path), + ); +} + +// The single place a name becomes a path, and therefore the right place to +// enforce that a name is one ordinary component. +// +// A trailing separator is the sharp edge here, not a tidiness concern: +// open(path, O_NOFOLLOW) FOLLOWS a symlink when path ends in "/" — the trap +// behind CVE-2026-39822 / golang/go#79005, which let os.Root escape its own +// root. path.join preserves that trailing slash, so a component carrying one +// would turn every no-follow open in this file into a following one. +// normalizeRepoPath already rejects such components upstream; this is the +// chokepoint that makes it true for every caller, including the generated +// temporary and vault names that never pass through it. +function anchoredChild(dirHandle, childName) { + if ( + typeof childName !== 'string' || + childName === '' || + childName === '.' || + childName === '..' || + childName.includes('/') || + childName.includes('\\') || + childName.includes('\0') + ) { + throw new Error(`Refusing to resolve ${JSON.stringify(childName)} as a single path component`); + } + return { + dir: dirHandle, + name: childName, + path: anchoringBackend().childPath(dirHandle, childName), + }; +} + +function lstatAnchoredOptional(ref) { + try { + return lstatChild(ref); } catch (error) { if (error?.code === 'ENOENT' || error?.code === 'ENOTDIR') return null; throw error; @@ -1063,39 +1381,37 @@ function openPlanParent( { createMissing = true, purpose = 'Generated-plan' } = {}, ) { requireDescriptorAnchoring(); - const flags = - fs.constants.O_RDONLY | - fs.constants.O_DIRECTORY | - fs.constants.O_NOFOLLOW | - (fs.constants.O_CLOEXEC ?? 0); + // Root-first and index-aligned with `chain`: verifyPinnedDescriptors relies on + // that, and the descriptors are what pin each recorded inode against reuse. const descriptors = []; try { - let currentFd = fs.openSync(repo, flags); + let currentFd = openVerifiedDirectory(repo, ANCHORED_DIRECTORY_FLAGS); descriptors.push(currentFd); const rootStat = fs.fstatSync(currentFd, { bigint: true }); const chain = [{ expectedPath: repo, identity: stableDirectoryIdentity(rootStat) }]; + let currentHandle = { fd: currentFd, expectedPath: repo, chain, descriptors }; const traversed = []; for (const component of parentComponents) { traversed.push(component); - const anchoredChild = descriptorPath(currentFd, component); + const child = anchoredChild(currentHandle, component); let childStat; let created = false; try { - childStat = fs.lstatSync(anchoredChild, { bigint: true }); + childStat = lstatChild(child); } catch (error) { if (error?.code !== 'ENOENT' && error?.code !== 'ENOTDIR') throw error; if (!createMissing) { throw new Error(`${purpose} parent does not exist: ${traversed.join('/')}`); } - fs.mkdirSync(anchoredChild, { mode: 0o755 }); - childStat = fs.lstatSync(anchoredChild, { bigint: true }); + mkdirChild(child, 0o755); + childStat = lstatChild(child); created = true; } if (childStat.isSymbolicLink() || !childStat.isDirectory()) { throw new Error(`${purpose} parent is not a real directory: ${traversed.join('/')}`); } const parentFd = currentFd; - const childFd = fs.openSync(anchoredChild, flags); + const childFd = openVerifiedDirectory(child.path, ANCHORED_DIRECTORY_FLAGS); descriptors.push(childFd); currentFd = childFd; if (created) { @@ -1103,18 +1419,16 @@ function openPlanParent( fs.fsyncSync(parentFd); } const expected = path.join(repo, ...traversed); - const actual = fs.realpathSync(descriptorPath(currentFd)); - if (actual !== expected) { + if (!anchoringBackend().descriptorMatchesChild(currentFd, expected, childStat)) { throw new Error(`${purpose} parent escaped the repository: ${traversed.join('/')}`); } const openedStat = fs.fstatSync(currentFd, { bigint: true }); chain.push({ expectedPath: expected, identity: stableDirectoryIdentity(openedStat) }); + currentHandle = { fd: currentFd, expectedPath: expected, chain, descriptors }; } - const stat = fs.fstatSync(currentFd, { bigint: true }); return { descriptors, fd: currentFd, - identity: stableDirectoryIdentity(stat), expectedPath: path.join(repo, ...parentComponents), chain, }; @@ -1134,9 +1448,16 @@ function closeDescriptors(descriptors) { } } +// A handle's identity IS its chain leaf's identity. Storing it twice meant two +// fstats a line apart and a re-stamp helper to keep them agreeing; deriving it +// removes both. +function handleIdentity(handle) { + return handle.chain[handle.chain.length - 1].identity; +} + function resolveGitDirectory(repo) { const result = git(repo, ['rev-parse', '--absolute-git-dir']); - return fs.realpathSync(decodeUtf8(result.stdout, 'Git administrative directory').trim()); + return fs.realpathSync.native(decodeUtf8(result.stdout, 'Git administrative directory').trim()); } function openBackupVault(repo, { createMissing = true } = {}) { @@ -1147,9 +1468,12 @@ function openBackupVault(repo, { createMissing = true } = {}) { }); fs.fchmodSync(handle.fd, 0o700); fs.fsyncSync(handle.fd); - const stat = fs.fstatSync(handle.fd, { bigint: true }); - handle.identity = stableDirectoryIdentity(stat); - handle.chain[handle.chain.length - 1].identity = handle.identity; + // mode is part of every directory identity, so hardening the vault changes the + // identity the chain recorded for it; without this the next verification would + // reject the directory it just hardened. + handle.chain[handle.chain.length - 1].identity = stableDirectoryIdentity( + fs.fstatSync(handle.fd, { bigint: true }), + ); return { ...handle, gitDirectory }; } @@ -1157,33 +1481,28 @@ function validatePlanParent(parentHandle) { const descriptorStat = fs.fstatSync(parentHandle.fd, { bigint: true }); if ( !descriptorStat.isDirectory() || - stableDirectoryIdentity(descriptorStat) !== parentHandle.identity + stableDirectoryIdentity(descriptorStat) !== handleIdentity(parentHandle) ) { throw new Error('Generated-plan parent descriptor changed during the write'); } - const descriptorRealPath = fs.realpathSync(descriptorPath(parentHandle.fd)); - if (descriptorRealPath !== parentHandle.expectedPath) { + if (!anchoringBackend().parentStillResolves(parentHandle)) { throw new Error('Generated-plan parent moved or was replaced during the write'); } - for (const item of parentHandle.chain) { - const lexicalStat = fs.lstatSync(item.expectedPath, { bigint: true }); - if ( - lexicalStat.isSymbolicLink() || - !lexicalStat.isDirectory() || - stableDirectoryIdentity(lexicalStat) !== item.identity - ) { - throw new Error('Generated-plan lexical parent no longer matches its directory descriptor'); - } - } + // Both halves come from the shared helpers rather than being restated here: an + // earlier hand-copy of the lexical loop lost verifyLexicalChain's ENOENT/ENOTDIR + // translation, so a renamed parent could surface a raw errno from a function + // with a dozen call sites. + verifyPinnedDescriptors(parentHandle); + verifyLexicalChain(parentHandle); } function inspectPlanDestination( - finalPath, + finalRef, { replace, expectedIdentity, mustBeAbsent = false } = {}, ) { let stat; try { - stat = fs.lstatSync(finalPath, { bigint: true }); + stat = lstatChild(finalRef); } catch (error) { if (error?.code === 'ENOENT') { if (expectedIdentity) throw new Error('Generated plan disappeared during the write'); @@ -1201,19 +1520,17 @@ function inspectPlanDestination( if (expectedIdentity && identity !== expectedIdentity) { throw new Error('Generated plan changed during the write'); } - return identity; + return stat; } -function openExistingPlanDestination(finalPath, replace) { - const identity = inspectPlanDestination(finalPath, { replace }); - if (identity === null) { +function openExistingPlanDestination(finalRef, replace) { + const stat = inspectPlanDestination(finalRef, { replace }); + if (stat === null) { if (replace) throw new Error('Deepen mode requires an existing generated plan to replace'); return { fd: undefined, identity: null, stableIdentity: null }; } - const fd = fs.openSync( - finalPath, - fs.constants.O_RDONLY | fs.constants.O_NOFOLLOW | (fs.constants.O_CLOEXEC ?? 0), - ); + const identity = statIdentity(stat); + const fd = openChildRead(finalRef, VERIFIED_READ_FLAGS, stat); try { const opened = fs.fstatSync(fd, { bigint: true }); if (!opened.isFile() || statIdentity(opened) !== identity) { @@ -1264,8 +1581,8 @@ function hashOpenFile(fd, label) { }; } -function validateCommittedPlan(finalPath, tempFd, expectedTemp, testHooks) { - const before = fs.lstatSync(finalPath, { bigint: true }); +function validateCommittedPlan(finalRef, tempFd, expectedTemp, testHooks) { + const before = lstatChild(finalRef); if ( before.isSymbolicLink() || !before.isFile() || @@ -1273,19 +1590,16 @@ function validateCommittedPlan(finalPath, tempFd, expectedTemp, testHooks) { ) { throw new Error('Generated-plan destination failed its first post-write identity check'); } - const finalFd = fs.openSync( - finalPath, - fs.constants.O_RDONLY | fs.constants.O_NOFOLLOW | (fs.constants.O_CLOEXEC ?? 0), - ); + const finalFd = openChildRead(finalRef, VERIFIED_READ_FLAGS, before); try { const opened = fs.fstatSync(finalFd, { bigint: true }); if (!opened.isFile() || stableFileIdentity(opened) !== expectedTemp.identity) { throw new Error('Generated-plan destination changed while its no-follow descriptor opened'); } - testHooks?.afterFinalOpen?.({ fd: finalFd, finalPath }); + testHooks?.afterFinalOpen?.({ fd: finalFd, finalPath: finalRef.path }); const committedViaTemp = hashOpenFile(tempFd, 'generated-plan committed file'); const committedViaPath = hashOpenFile(finalFd, 'generated-plan destination descriptor'); - const after = fs.lstatSync(finalPath, { bigint: true }); + const after = lstatChild(finalRef); const openedAfter = fs.fstatSync(finalFd, { bigint: true }); if ( after.isSymbolicLink() || @@ -1320,22 +1634,19 @@ function copyOpenFile(sourceFd, destinationFd, label) { return after; } -function openVerifiedPathFile(absolute, label) { - const before = fs.lstatSync(absolute, { bigint: true }); +function openVerifiedAnchoredFile(ref, label, knownStat) { + const before = knownStat ?? lstatChild(ref); if (before.isSymbolicLink() || !before.isFile()) { throw new Error(`${label} is not a regular no-follow file`); } - const fd = fs.openSync( - absolute, - fs.constants.O_RDONLY | fs.constants.O_NOFOLLOW | (fs.constants.O_CLOEXEC ?? 0), - ); + const fd = openChildRead(ref, VERIFIED_READ_FLAGS, before); try { const opened = fs.fstatSync(fd, { bigint: true }); if (!opened.isFile() || stableFileIdentity(opened) !== stableFileIdentity(before)) { throw new Error(`${label} changed while its descriptor opened`); } const layer = hashOpenFile(fd, label); - const after = fs.lstatSync(absolute, { bigint: true }); + const after = lstatChild(ref); if (after.isSymbolicLink() || !after.isFile() || stableFileIdentity(after) !== layer.identity) { throw new Error(`${label} changed after verification`); } @@ -1358,10 +1669,10 @@ export function readPlanSafely({ repo: repoInput, generatedPlanPath, testHooks } let fd; try { validatePlanParent(parentHandle); - const finalPath = descriptorPath(parentHandle.fd, finalName); + const finalRef = anchoredChild(parentHandle, finalName); let before; try { - before = fs.lstatSync(finalPath, { bigint: true }); + before = lstatChild(finalRef); } catch (error) { if (error?.code === 'ENOENT' || error?.code === 'ENOTDIR') { throw new Error(`Loaded plan does not exist: ${generatedPlan}`); @@ -1371,15 +1682,12 @@ export function readPlanSafely({ repo: repoInput, generatedPlanPath, testHooks } if (before.isSymbolicLink() || !before.isFile()) { throw new Error('Loaded plan must be a regular file, never a symlink'); } - fd = fs.openSync( - finalPath, - fs.constants.O_RDONLY | fs.constants.O_NOFOLLOW | (fs.constants.O_CLOEXEC ?? 0), - ); + fd = openChildRead(finalRef, VERIFIED_READ_FLAGS, before); const opened = fs.fstatSync(fd, { bigint: true }); if (!opened.isFile() || statIdentity(opened) !== statIdentity(before)) { throw new Error('Loaded plan changed while its no-follow descriptor opened'); } - testHooks?.afterPlanOpen?.({ fd, finalPath }); + testHooks?.afterPlanOpen?.({ fd, finalPath: finalRef.path }); const chunks = []; let total = 0; const buffer = Buffer.allocUnsafe(64 * 1024); @@ -1394,7 +1702,7 @@ export function readPlanSafely({ repo: repoInput, generatedPlanPath, testHooks } decodeUtf8(contents, 'loaded plan'); const after = fs.fstatSync(fd, { bigint: true }); assertStableIdentity(opened, after, 'loaded plan'); - const pathAfter = fs.lstatSync(finalPath, { bigint: true }); + const pathAfter = lstatChild(finalRef); if ( pathAfter.isSymbolicLink() || !pathAfter.isFile() || @@ -1419,24 +1727,22 @@ function artifactGitPath(name) { return `gitnexus-plan-backups/${name}`; } -function verifyVaultArtifactFromFreshRoot(repo, gitPath, expectedLayer) { - const components = gitPath.split('/'); - if (components.length !== 2 || components[0] !== 'gitnexus-plan-backups') { - throw new Error(`Invalid Git-admin artifact path: ${gitPath}`); - } +function verifyVaultArtifactFromFreshRoot(repo, name, expectedLayer) { const freshVault = openBackupVault(repo, { createMissing: false }); try { validatePlanParent(freshVault); - const opened = openVerifiedPathFile( - descriptorPath(freshVault.fd, components[1]), - `Git-admin artifact ${gitPath}`, + const opened = openVerifiedAnchoredFile( + anchoredChild(freshVault, name), + `Git-admin artifact ${artifactGitPath(name)}`, ); try { if ( opened.layer.identity !== expectedLayer.identity || opened.layer.digest !== expectedLayer.digest ) { - throw new Error(`Git-admin artifact changed before fresh-root verification: ${gitPath}`); + throw new Error( + `Git-admin artifact changed before fresh-root verification: ${artifactGitPath(name)}`, + ); } } finally { fs.closeSync(opened.fd); @@ -1449,16 +1755,8 @@ function verifyVaultArtifactFromFreshRoot(repo, gitPath, expectedLayer) { function createVaultCopyFromFd(repo, vault, sourceFd, role) { validatePlanParent(vault); const name = `.gitnexus-plan-${role}-${process.pid}-${randomBytes(16).toString('hex')}.bak`; - const absolute = descriptorPath(vault.fd, name); - const destinationFd = fs.openSync( - absolute, - fs.constants.O_RDWR | - fs.constants.O_CREAT | - fs.constants.O_EXCL | - fs.constants.O_NOFOLLOW | - (fs.constants.O_CLOEXEC ?? 0), - 0o600, - ); + const artifact = anchoredChild(vault, name); + const destinationFd = createChild(artifact, VERIFIED_CREATE_FLAGS, 0o600); let destination; try { const sourceStat = copyOpenFile(sourceFd, destinationFd, role); @@ -1469,7 +1767,7 @@ function createVaultCopyFromFd(repo, vault, sourceFd, role) { if (source.size !== destination.size || source.digest !== destination.digest) { throw new Error(`${role} vault copy does not match its held source descriptor`); } - const pathStat = fs.lstatSync(absolute, { bigint: true }); + const pathStat = lstatChild(artifact); if ( pathStat.isSymbolicLink() || !pathStat.isFile() || @@ -1481,24 +1779,15 @@ function createVaultCopyFromFd(repo, vault, sourceFd, role) { } finally { fs.closeSync(destinationFd); } - const gitPath = artifactGitPath(name); - verifyVaultArtifactFromFreshRoot(repo, gitPath, destination); - return { role, gitPath, layer: destination }; + verifyVaultArtifactFromFreshRoot(repo, name, destination); + return { role, gitPath: artifactGitPath(name), layer: destination }; } function createVaultCopyFromBytes(repo, vault, contents, role) { validatePlanParent(vault); const name = `.gitnexus-plan-${role}-${process.pid}-${randomBytes(16).toString('hex')}.bak`; - const absolute = descriptorPath(vault.fd, name); - const fd = fs.openSync( - absolute, - fs.constants.O_RDWR | - fs.constants.O_CREAT | - fs.constants.O_EXCL | - fs.constants.O_NOFOLLOW | - (fs.constants.O_CLOEXEC ?? 0), - 0o600, - ); + const artifact = anchoredChild(vault, name); + const fd = createChild(artifact, VERIFIED_CREATE_FLAGS, 0o600); let layer; try { writeAll(fd, contents); @@ -1508,7 +1797,7 @@ function createVaultCopyFromBytes(repo, vault, contents, role) { if (layer.size !== BigInt(contents.length) || layer.digest !== sha256(contents)) { throw new Error(`${role} vault copy does not match the intended plan bytes`); } - const pathStat = fs.lstatSync(absolute, { bigint: true }); + const pathStat = lstatChild(artifact); if ( pathStat.isSymbolicLink() || !pathStat.isFile() || @@ -1520,32 +1809,31 @@ function createVaultCopyFromBytes(repo, vault, contents, role) { } finally { fs.closeSync(fd); } - const gitPath = artifactGitPath(name); - verifyVaultArtifactFromFreshRoot(repo, gitPath, layer); - return { role, gitPath, layer }; + verifyVaultArtifactFromFreshRoot(repo, name, layer); + return { role, gitPath: artifactGitPath(name), layer }; } function movePathToVault(repo, sourceHandle, sourceName, vault, role) { - const source = descriptorPath(sourceHandle.fd, sourceName); - if (!lstatOptional(source)) return null; + const source = anchoredChild(sourceHandle, sourceName); + if (!lstatAnchoredOptional(source)) return null; const name = `.gitnexus-plan-${role}-${process.pid}-${randomBytes(16).toString('hex')}.bak`; - const destination = descriptorPath(vault.fd, name); - const moved = atomicMoveNoReplace( - externalDescriptorPath(sourceHandle.fd, sourceName), - externalDescriptorPath(vault.fd, name), - ); + const destination = anchoredChild(vault, name); + const moved = publishNoReplace(source, destination); if (!moved) throw new Error(`${role} preservation destination unexpectedly exists`); fs.fsyncSync(sourceHandle.fd); if (vault.fd !== sourceHandle.fd) fs.fsyncSync(vault.fd); - const sourceAfter = lstatOptional(source); - const destinationAfter = lstatOptional(destination); + const sourceAfter = lstatAnchoredOptional(source); + const destinationAfter = lstatAnchoredOptional(destination); if (sourceAfter || !destinationAfter) { throw new Error(`${role} could not be atomically moved into the Git-admin vault`); } - const opened = openVerifiedPathFile(destination, `${role} Git-admin artifact`); - const gitPath = artifactGitPath(name); - verifyVaultArtifactFromFreshRoot(repo, gitPath, opened.layer); - return { role, gitPath, layer: opened.layer, fd: opened.fd }; + const opened = openVerifiedAnchoredFile( + destination, + `${role} Git-admin artifact`, + destinationAfter, + ); + verifyVaultArtifactFromFreshRoot(repo, name, opened.layer); + return { role, gitPath: artifactGitPath(name), layer: opened.layer, fd: opened.fd }; } function formatPreservedArtifacts(artifacts) { @@ -1600,10 +1888,10 @@ export function writePlanSafely({ const finalName = components.pop(); let parentHandle; let vaultHandle; - let tempPath; + let tempRef; let tempName; let tempFd; - let finalPath; + let finalRef; let expectedTemp; let originalDestination; let priorBackup; @@ -1611,7 +1899,6 @@ export function writePlanSafely({ try { parentHandle = openPlanParent(repo, components); vaultHandle = openBackupVault(repo); - resolveAtomicMover(); const parentDevice = fs.fstatSync(parentHandle.fd, { bigint: true }).dev; const vaultDevice = fs.fstatSync(vaultHandle.fd, { bigint: true }).dev; if (parentDevice !== vaultDevice) { @@ -1622,19 +1909,11 @@ export function writePlanSafely({ testHooks?.afterParentOpen?.({ fd: parentHandle.fd, path: parentHandle.expectedPath }); validatePlanParent(parentHandle); validatePlanParent(vaultHandle); - finalPath = descriptorPath(parentHandle.fd, finalName); - originalDestination = openExistingPlanDestination(finalPath, shouldReplace); + finalRef = anchoredChild(parentHandle, finalName); + originalDestination = openExistingPlanDestination(finalRef, shouldReplace); tempName = `.gitnexus-plan-${process.pid}-${randomBytes(16).toString('hex')}.tmp`; - tempPath = descriptorPath(parentHandle.fd, tempName); - tempFd = fs.openSync( - tempPath, - fs.constants.O_RDWR | - fs.constants.O_CREAT | - fs.constants.O_EXCL | - fs.constants.O_NOFOLLOW | - (fs.constants.O_CLOEXEC ?? 0), - 0o600, - ); + tempRef = anchoredChild(parentHandle, tempName); + tempFd = createChild(tempRef, VERIFIED_CREATE_FLAGS, 0o600); writeAll(tempFd, contents); fs.fchmodSync(tempFd, 0o644); fs.fsyncSync(tempFd); @@ -1646,12 +1925,12 @@ export function writePlanSafely({ testHooks?.beforeRename?.({ fd: parentHandle.fd, path: parentHandle.expectedPath, - tempPath, + tempPath: tempRef.path, }); validatePlanParent(parentHandle); validatePlanParent(vaultHandle); validateOpenPlanDestination(originalDestination); - const tempPathStat = fs.lstatSync(tempPath, { bigint: true }); + const tempPathStat = lstatChild(tempRef); const currentTemp = hashOpenFile(tempFd, 'generated-plan temporary file'); if ( tempPathStat.isSymbolicLink() || @@ -1664,7 +1943,7 @@ export function writePlanSafely({ } if (shouldReplace) { - testHooks?.beforeBackupMove?.({ fd: parentHandle.fd, finalPath }); + testHooks?.beforeBackupMove?.({ fd: parentHandle.fd, finalPath: finalRef.path }); const originalLayer = hashOpenFile(originalDestination.fd, 'prior generated plan'); if (originalLayer.digest !== expectedDigest) { throw new Error( @@ -1673,7 +1952,7 @@ export function writePlanSafely({ } validatePlanParent(parentHandle); validateOpenPlanDestination(originalDestination); - inspectPlanDestination(finalPath, { + inspectPlanDestination(finalRef, { replace: true, expectedIdentity: originalDestination.identity, }); @@ -1691,20 +1970,20 @@ export function writePlanSafely({ ); throw new Error('Destination raced while the prior plan was moved into preservation'); } - if (lstatOptional(finalPath)) { + if (lstatAnchoredOptional(finalRef)) { throw new Error('Destination reappeared after the prior plan was preserved'); } } testHooks?.beforePublication?.({ fd: parentHandle.fd, - finalPath, - tempPath, + finalPath: finalRef.path, + tempPath: tempRef.path, replace: shouldReplace, }); validatePlanParent(parentHandle); validatePlanParent(vaultHandle); - const finalTempPathStat = fs.lstatSync(tempPath, { bigint: true }); + const finalTempPathStat = lstatChild(tempRef); const finalTemp = hashOpenFile(tempFd, 'generated-plan temporary file'); if ( finalTempPathStat.isSymbolicLink() || @@ -1715,19 +1994,25 @@ export function writePlanSafely({ ) { throw new Error('Generated-plan temporary path or content changed at publication'); } - atomicMoveNoReplace( - externalDescriptorPath(parentHandle.fd, tempName), - externalDescriptorPath(parentHandle.fd, finalName), - ); - if (lstatOptional(tempPath) || !lstatOptional(finalPath)) { + // link() reports the race itself; re-deriving that verdict from a later pair + // of stats would be both slower and weaker. + if (!publishNoReplace(tempRef, finalRef)) { throw new Error('Generated-plan publication was refused because the destination raced'); } + // link() creates a directory entry, so it needs the parent fsync that rename + // needed: the file's own bytes were fsynced through tempFd before this point, + // and this makes the name that now reaches them durable too. Skipping it is + // the step write-file-atomic omits and maildir, git and atomicwrites all + // mandate. + // + // Honest limitation: on macOS fsync is not a write barrier — the durable + // primitive there is fcntl(F_FULLFSYNC), which Node does not expose. A + // macOS plan write is therefore as durable as fsync makes it and no more. fs.fsyncSync(parentHandle.fd); - testHooks?.afterPublication?.({ fd: parentHandle.fd, finalPath }); - testHooks?.afterRename?.({ fd: parentHandle.fd, finalPath }); + testHooks?.afterPublication?.({ fd: parentHandle.fd, finalPath: finalRef.path }); validatePlanParent(parentHandle); validatePlanParent(vaultHandle); - validateCommittedPlan(finalPath, tempFd, expectedTemp, testHooks); + validateCommittedPlan(finalRef, tempFd, expectedTemp, testHooks); const receipt = { generated_plan_path: generatedPlan, bytes_written: contents.length }; if (priorBackup) receipt.prior_plan_backup_git_path = priorBackup.gitPath; return receipt; @@ -1848,6 +2133,11 @@ export function snapshotEvidence({ const headGuards = captureHeadGuards(repo); const dirty = initialDirty.records; const mutationGuards = []; + // Per-snapshot walk state: `absenceCache` owns every descriptor an absence + // anchor holds, deduplicated by repo-relative prefix and closed exactly once + // below; `guardedDirectories` keeps parent guarding to one stat per directory. + const absenceCache = new Map(); + const walkState = { absenceCache, guardedDirectories: new Set() }; try { testHooks?.afterAnchorCapture?.({ headCommit: head }); @@ -1862,7 +2152,9 @@ export function snapshotEvidence({ testHooks?.afterGitLayerLoad?.({ headCommit: head }); const globalEntries = [...dirty.values()] .filter((record) => record.path !== generatedPlan) - .map((record) => materializeRecord(repo, record, layers, mutationGuards, testHooks)); + .map((record) => + materializeRecord(repo, record, layers, mutationGuards, testHooks, walkState), + ); const citedEntries = [...normalizedCitations].sort(compareUtf8).map((repoPath) => { const status = dirty.get(repoPath) ?? { path: repoPath, @@ -1871,7 +2163,7 @@ export function snapshotEvidence({ rename_to: null, has_untracked: false, }; - const entry = materializeRecord(repo, status, layers, mutationGuards, testHooks); + const entry = materializeRecord(repo, status, layers, mutationGuards, testHooks, walkState); const present = Object.values(entry.object_kind).some((kind) => kind !== ABSENT); if (!present) entry.state = ABSENT; else if (entry.state === 'clean' && entry.object_kind.untracked !== ABSENT) { @@ -1906,21 +2198,13 @@ export function snapshotEvidence({ throw new Error(`${guard.absolute} changed before evidence materialization completed`); } } else if (guard.type === 'absence') { + // statIdentity is a strict superset of stableDirectoryIdentity on the + // same stat, so comparing both could only ever fire together. const parent = fs.fstatSync(guard.fd, { bigint: true }); - if ( - !parent.isDirectory() || - stableDirectoryIdentity(parent) !== guard.parentIdentity || - statIdentity(parent) !== guard.parentMutationIdentity - ) { + if (!parent.isDirectory() || statIdentity(parent) !== guard.parentMutationIdentity) { throw new Error(`Absence anchor changed for ${guard.repoPath}`); } - try { - fs.lstatSync(descriptorPath(guard.fd, guard.childName), { bigint: true }); - } catch (error) { - if (error?.code === 'ENOENT') continue; - throw error; - } - throw new Error(`${guard.repoPath} appeared before evidence materialization completed`); + anchoringBackend().verifyAbsentChild(guard); } } for (const guard of headGuards) verifyControlFile(guard); @@ -1955,12 +2239,10 @@ export function snapshotEvidence({ cited_path_manifest: citedEntries, }; } finally { - const closed = new Set(); - for (const guard of mutationGuards) { - if (guard.type !== 'absence' || closed.has(guard.fd)) continue; - closed.add(guard.fd); + // One entry per distinct anchored directory, so one close per descriptor. + for (const handle of absenceCache.values()) { try { - fs.closeSync(guard.fd); + fs.closeSync(handle.fd); } catch { // Preserve the primary snapshot result/error. } diff --git a/gitnexus-cursor-integration/skills/gitnexus-impact-analysis/SKILL.md b/gitnexus-cursor-integration/skills/gitnexus-impact-analysis/SKILL.md index 8f9f1d1e7..e3817d111 100644 --- a/gitnexus-cursor-integration/skills/gitnexus-impact-analysis/SKILL.md +++ b/gitnexus-cursor-integration/skills/gitnexus-impact-analysis/SKILL.md @@ -36,6 +36,8 @@ description: Analyze blast radius before making code changes - [ ] Assess risk level and report to user ``` +> `partial: true` (a graph query failed) or `truncated: true` (the changed-symbol listing was capped) means the result is short of the truth: a zero there means unseen, not unaffected. Re-run it rather than tick the pre-commit check. + ## Understanding Output | Depth | Risk Level | Meaning | @@ -52,6 +54,14 @@ description: Analyze blast radius before making code changes | 5-15 symbols, 2-5 processes | MEDIUM | | >15 symbols or many processes | HIGH | | Critical path (auth, payments) | CRITICAL | +| **Zero callers found** | **UNKNOWN** | + +`UNKNOWN` is not a low rung on this scale — it means the walk could not answer. +An empty caller set is equally consistent with "genuinely unused" and "the +callers are not resolvable by the index" (plain-object property access, dynamic +dispatch, cross-language calls), so few-callers ⇒ LOW does **not** apply. The +result carries a `riskNote` saying so. Confirm with a text search before +treating the symbol as safe to change or delete. ## Tools diff --git a/gitnexus-cursor-integration/skills/gitnexus-refactoring/SKILL.md b/gitnexus-cursor-integration/skills/gitnexus-refactoring/SKILL.md index 9495a19d5..66f2c2982 100644 --- a/gitnexus-cursor-integration/skills/gitnexus-refactoring/SKILL.md +++ b/gitnexus-cursor-integration/skills/gitnexus-refactoring/SKILL.md @@ -23,6 +23,8 @@ description: Plan safe refactors using blast radius and dependency mapping > If "Index is stale" → run `node .gitnexus/run.cjs analyze` in terminal. +> Every `detect_changes()` below: `partial: true` (a graph query failed) or `truncated: true` (the changed-symbol listing was capped) means the result is short of the truth — a short or empty list is not proof that only the expected files changed. Re-run it rather than treat the refactor as verified. + ## Checklists ### Rename Symbol diff --git a/gitnexus-shared/package-lock.json b/gitnexus-shared/package-lock.json index 0fee05147..4359ce75c 100644 --- a/gitnexus-shared/package-lock.json +++ b/gitnexus-shared/package-lock.json @@ -8,21 +8,382 @@ "name": "gitnexus-shared", "version": "1.0.0", "devDependencies": { - "typescript": "^6.0.3" + "typescript": "^7.0.2" + } + }, + "node_modules/@typescript/typescript-aix-ppc64": { + "version": "7.0.2", + "resolved": "https://registry.npmjs.org/@typescript/typescript-aix-ppc64/-/typescript-aix-ppc64-7.0.2.tgz", + "integrity": "sha512-MTKKkWB7p/0E9xi1d1tHtZ5PiLkGEMIq88pK2CubZjOsLtYTLqhgIgi6zepFa+9GHZ6h05NMCkQxGKiPXMxXtQ==", + "cpu": [ + "ppc64" + ], + "dev": true, + "license": "Apache-2.0", + "optional": true, + "os": [ + "aix" + ], + "engines": { + "node": ">=16.20.0" + } + }, + "node_modules/@typescript/typescript-darwin-arm64": { + "version": "7.0.2", + "resolved": "https://registry.npmjs.org/@typescript/typescript-darwin-arm64/-/typescript-darwin-arm64-7.0.2.tgz", + "integrity": "sha512-gowzar9MwS/aRWp6f3a4KUqzRjAZjOsmGNCM6LcTgXum+dBfgsBVMN+AgvOCCbguXyick6LJhpBszxMebJ8syA==", + "cpu": [ + "arm64" + ], + "dev": true, + "license": "Apache-2.0", + "optional": true, + "os": [ + "darwin" + ], + "engines": { + "node": ">=16.20.0" + } + }, + "node_modules/@typescript/typescript-darwin-x64": { + "version": "7.0.2", + "resolved": "https://registry.npmjs.org/@typescript/typescript-darwin-x64/-/typescript-darwin-x64-7.0.2.tgz", + "integrity": "sha512-SZ9xZInqApNlNGc9s0W1VSsktYSOe9cFqNOIqmN1Gs8SmkjKZYFt017G4VwPxASInODuAdbTW7sXiFUf893RgA==", + "cpu": [ + "x64" + ], + "dev": true, + "license": "Apache-2.0", + "optional": true, + "os": [ + "darwin" + ], + "engines": { + "node": ">=16.20.0" + } + }, + "node_modules/@typescript/typescript-freebsd-arm64": { + "version": "7.0.2", + "resolved": "https://registry.npmjs.org/@typescript/typescript-freebsd-arm64/-/typescript-freebsd-arm64-7.0.2.tgz", + "integrity": "sha512-W5NH4y/J0plIIS5b2xvTEkU7JFxyqdMAOgf+Ilhl0vHQXKO5dZoxd+C/jEtq56c4F3wk71RB4BMRQ2XdI+bwYQ==", + "cpu": [ + "arm64" + ], + "dev": true, + "license": "Apache-2.0", + "optional": true, + "os": [ + "freebsd" + ], + "engines": { + "node": ">=16.20.0" + } + }, + "node_modules/@typescript/typescript-freebsd-x64": { + "version": "7.0.2", + "resolved": "https://registry.npmjs.org/@typescript/typescript-freebsd-x64/-/typescript-freebsd-x64-7.0.2.tgz", + "integrity": "sha512-UMGDx5sTpzNw3WiPebH7l90IWfJggEd+egHt/q6p7/Cm3zqoV7VxkGXt+3DxPIw8CcmvAB0j3sVVfbhX+M4Tpw==", + "cpu": [ + "x64" + ], + "dev": true, + "license": "Apache-2.0", + "optional": true, + "os": [ + "freebsd" + ], + "engines": { + "node": ">=16.20.0" + } + }, + "node_modules/@typescript/typescript-linux-arm": { + "version": "7.0.2", + "resolved": "https://registry.npmjs.org/@typescript/typescript-linux-arm/-/typescript-linux-arm-7.0.2.tgz", + "integrity": "sha512-gffT3xPz9sR7j/YJExkyPntrI0P2EP9XbOyWzth2/Gs0RstK+90RBcO0ncXoXy/beYll1SXw846Nf2zdnEz0QQ==", + "cpu": [ + "arm" + ], + "dev": true, + "license": "Apache-2.0", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">=16.20.0" + } + }, + "node_modules/@typescript/typescript-linux-arm64": { + "version": "7.0.2", + "resolved": "https://registry.npmjs.org/@typescript/typescript-linux-arm64/-/typescript-linux-arm64-7.0.2.tgz", + "integrity": "sha512-Qh4eU4/y3yDjnfjjyPYihMj5/ODIlmt+Bzu17OI+fiSRDW57QmU5SiN63exPRNJPKUzcc1INa1NXdrJ+MqHjUQ==", + "cpu": [ + "arm64" + ], + "dev": true, + "license": "Apache-2.0", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">=16.20.0" + } + }, + "node_modules/@typescript/typescript-linux-loong64": { + "version": "7.0.2", + "resolved": "https://registry.npmjs.org/@typescript/typescript-linux-loong64/-/typescript-linux-loong64-7.0.2.tgz", + "integrity": "sha512-uEHck9i8hoAzXPiYRib1O7miOnz23SxIeVl6F4LXox+qov1K35jHcEW6VHKvZI+pyvl7fZEP4MCU5LYvIq1GuQ==", + "cpu": [ + "loong64" + ], + "dev": true, + "license": "Apache-2.0", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">=16.20.0" + } + }, + "node_modules/@typescript/typescript-linux-mips64el": { + "version": "7.0.2", + "resolved": "https://registry.npmjs.org/@typescript/typescript-linux-mips64el/-/typescript-linux-mips64el-7.0.2.tgz", + "integrity": "sha512-R4KvAMnE43W5Qeqb0Ly56O3mWMWIAgsMyz36DCaycd5nbg/9kzm0liw3JocfRqyJY0KPmzFjbswozXyW0DnIYA==", + "cpu": [ + "mips64el" + ], + "dev": true, + "license": "Apache-2.0", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">=16.20.0" + } + }, + "node_modules/@typescript/typescript-linux-ppc64": { + "version": "7.0.2", + "resolved": "https://registry.npmjs.org/@typescript/typescript-linux-ppc64/-/typescript-linux-ppc64-7.0.2.tgz", + "integrity": "sha512-DORx5b3sd/4S7eayxm4FQv+A7CrkUIGRaHiwI8oiHTAI1fAPWhF4J0vAlkC8biAlHSVVwxMQ3tjZ2/DVbnQiiA==", + "cpu": [ + "ppc64" + ], + "dev": true, + "license": "Apache-2.0", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">=16.20.0" + } + }, + "node_modules/@typescript/typescript-linux-riscv64": { + "version": "7.0.2", + "resolved": "https://registry.npmjs.org/@typescript/typescript-linux-riscv64/-/typescript-linux-riscv64-7.0.2.tgz", + "integrity": "sha512-wf0jqEDOjrPRnKwYRyyJDRo11KMbvMFrU+q4zqKyChODBzvlkbhNQfKvLxQCcwTpdDaXSHZTVuh0JoCrKCUMHQ==", + "cpu": [ + "riscv64" + ], + "dev": true, + "license": "Apache-2.0", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">=16.20.0" + } + }, + "node_modules/@typescript/typescript-linux-s390x": { + "version": "7.0.2", + "resolved": "https://registry.npmjs.org/@typescript/typescript-linux-s390x/-/typescript-linux-s390x-7.0.2.tgz", + "integrity": "sha512-IkwJc3L7yhytWd/ewjyxNDfOmswCm9GWMJT/ue/dU4aZNbwZeYAetq42VyLmsmSjvoX7z74X6ZaYCtzAr0EuGw==", + "cpu": [ + "s390x" + ], + "dev": true, + "license": "Apache-2.0", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">=16.20.0" + } + }, + "node_modules/@typescript/typescript-linux-x64": { + "version": "7.0.2", + "resolved": "https://registry.npmjs.org/@typescript/typescript-linux-x64/-/typescript-linux-x64-7.0.2.tgz", + "integrity": "sha512-EYdf2cNg7rgCWJnxCdJ+F3V39O8ihb37eHAu1LK8oAFizgTQbPOK7zHHXbPt8rX24COqODXeI3sIf0fCXG7H/A==", + "cpu": [ + "x64" + ], + "dev": true, + "license": "Apache-2.0", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">=16.20.0" + } + }, + "node_modules/@typescript/typescript-netbsd-arm64": { + "version": "7.0.2", + "resolved": "https://registry.npmjs.org/@typescript/typescript-netbsd-arm64/-/typescript-netbsd-arm64-7.0.2.tgz", + "integrity": "sha512-+polYF4MF04aPpO5FTkHran9yUQDSXqy5GiSDKpsll5jy3l3+g9QLhpf39T+ePtefhXLOGrLl0QIjkQP6VnelA==", + "cpu": [ + "arm64" + ], + "dev": true, + "license": "Apache-2.0", + "optional": true, + "os": [ + "netbsd" + ], + "engines": { + "node": ">=16.20.0" + } + }, + "node_modules/@typescript/typescript-netbsd-x64": { + "version": "7.0.2", + "resolved": "https://registry.npmjs.org/@typescript/typescript-netbsd-x64/-/typescript-netbsd-x64-7.0.2.tgz", + "integrity": "sha512-8YIT0EHM/3dq10ZOVF/A7pc/YSMtbcecct4rWtexrnSCHOPcpC2KTLXfTCR6vDpnSiY12heNb1GiN/wu+T/FyA==", + "cpu": [ + "x64" + ], + "dev": true, + "license": "Apache-2.0", + "optional": true, + "os": [ + "netbsd" + ], + "engines": { + "node": ">=16.20.0" + } + }, + "node_modules/@typescript/typescript-openbsd-arm64": { + "version": "7.0.2", + "resolved": "https://registry.npmjs.org/@typescript/typescript-openbsd-arm64/-/typescript-openbsd-arm64-7.0.2.tgz", + "integrity": "sha512-APT8+ClYnuYm1u9+kgGXoMj2VzWzcymwh2gNSQVySHfkRDGOTVkoWLjCmOQSaO+PoqQ57B0flRp9SA+7GnnkzQ==", + "cpu": [ + "arm64" + ], + "dev": true, + "license": "Apache-2.0", + "optional": true, + "os": [ + "openbsd" + ], + "engines": { + "node": ">=16.20.0" + } + }, + "node_modules/@typescript/typescript-openbsd-x64": { + "version": "7.0.2", + "resolved": "https://registry.npmjs.org/@typescript/typescript-openbsd-x64/-/typescript-openbsd-x64-7.0.2.tgz", + "integrity": "sha512-yX7s+Q0Dln0Dt9tEzZsAjXXR/+ytBM7AlglaqyeMPxQszJ1JhlJdZ6jLA+IzldHtflX81em7lDao1xXu+aRRkg==", + "cpu": [ + "x64" + ], + "dev": true, + "license": "Apache-2.0", + "optional": true, + "os": [ + "openbsd" + ], + "engines": { + "node": ">=16.20.0" + } + }, + "node_modules/@typescript/typescript-sunos-x64": { + "version": "7.0.2", + "resolved": "https://registry.npmjs.org/@typescript/typescript-sunos-x64/-/typescript-sunos-x64-7.0.2.tgz", + "integrity": "sha512-dLJDGaLZ1D4HPQn62u1n8mBDkJREwMsAkCdkwd4Ieqw+x3TUyTsqY0YiBCtE6H6OzzgGk3iuZ3vFWRS+E8/d1g==", + "cpu": [ + "x64" + ], + "dev": true, + "license": "Apache-2.0", + "optional": true, + "os": [ + "sunos" + ], + "engines": { + "node": ">=16.20.0" + } + }, + "node_modules/@typescript/typescript-win32-arm64": { + "version": "7.0.2", + "resolved": "https://registry.npmjs.org/@typescript/typescript-win32-arm64/-/typescript-win32-arm64-7.0.2.tgz", + "integrity": "sha512-Gyl1Vy6OsWesLzmq+EP0Fb7b4Nid5232AvcA2SFcdYreldpNtYFFofPjnt62y9hQy7VTaZp65ICJjuAQRaVcIQ==", + "cpu": [ + "arm64" + ], + "dev": true, + "license": "Apache-2.0", + "optional": true, + "os": [ + "win32" + ], + "engines": { + "node": ">=16.20.0" + } + }, + "node_modules/@typescript/typescript-win32-x64": { + "version": "7.0.2", + "resolved": "https://registry.npmjs.org/@typescript/typescript-win32-x64/-/typescript-win32-x64-7.0.2.tgz", + "integrity": "sha512-0BQ3HkAHHlKLSp1qRvf3SUhGpGsDuhB/jgFw75guyqbxJqEaS0Cw/VFO8i2nHglJUzQCRtMMR/IBAKE3ETMC4g==", + "cpu": [ + "x64" + ], + "dev": true, + "license": "Apache-2.0", + "optional": true, + "os": [ + "win32" + ], + "engines": { + "node": ">=16.20.0" } }, "node_modules/typescript": { - "version": "6.0.3", - "resolved": "https://registry.npmjs.org/typescript/-/typescript-6.0.3.tgz", - "integrity": "sha512-y2TvuxSZPDyQakkFRPZHKFm+KKVqIisdg9/CZwm9ftvKXLP8NRWj38/ODjNbr43SsoXqNuAisEf1GdCxqWcdBw==", + "version": "7.0.2", + "resolved": "https://registry.npmjs.org/typescript/-/typescript-7.0.2.tgz", + "integrity": "sha512-8FYau96o3NKOhbjKi/qNvG/W5jhzxkbdm5sj9AbZ/5T5sWqn3hJgLfGx27sRKZWTvyzCP8dLRBTf5tBTSRVUNA==", "dev": true, "license": "Apache-2.0", "bin": { - "tsc": "bin/tsc", - "tsserver": "bin/tsserver" + "tsc": "bin/tsc" }, "engines": { - "node": ">=14.17" + "node": ">=16.20.0" + }, + "optionalDependencies": { + "@typescript/typescript-aix-ppc64": "7.0.2", + "@typescript/typescript-darwin-arm64": "7.0.2", + "@typescript/typescript-darwin-x64": "7.0.2", + "@typescript/typescript-freebsd-arm64": "7.0.2", + "@typescript/typescript-freebsd-x64": "7.0.2", + "@typescript/typescript-linux-arm": "7.0.2", + "@typescript/typescript-linux-arm64": "7.0.2", + "@typescript/typescript-linux-loong64": "7.0.2", + "@typescript/typescript-linux-mips64el": "7.0.2", + "@typescript/typescript-linux-ppc64": "7.0.2", + "@typescript/typescript-linux-riscv64": "7.0.2", + "@typescript/typescript-linux-s390x": "7.0.2", + "@typescript/typescript-linux-x64": "7.0.2", + "@typescript/typescript-netbsd-arm64": "7.0.2", + "@typescript/typescript-netbsd-x64": "7.0.2", + "@typescript/typescript-openbsd-arm64": "7.0.2", + "@typescript/typescript-openbsd-x64": "7.0.2", + "@typescript/typescript-sunos-x64": "7.0.2", + "@typescript/typescript-win32-arm64": "7.0.2", + "@typescript/typescript-win32-x64": "7.0.2" } } } diff --git a/gitnexus-shared/package.json b/gitnexus-shared/package.json index 0a5d7a2db..a60f54ccb 100644 --- a/gitnexus-shared/package.json +++ b/gitnexus-shared/package.json @@ -24,6 +24,6 @@ "src" ], "devDependencies": { - "typescript": "^6.0.3" + "typescript": "^7.0.2" } } diff --git a/gitnexus-shared/src/index.ts b/gitnexus-shared/src/index.ts index 284e94268..13c2eac5a 100644 --- a/gitnexus-shared/src/index.ts +++ b/gitnexus-shared/src/index.ts @@ -30,7 +30,11 @@ export type { PipelinePhase, PipelineProgress } from './pipeline.js'; // ─── Scope-based resolution — RFC #909 (Ring 1 #910) ──────────────────────── // Data model (RFC §2) -export type { ParameterTypeClass, SymbolDefinition } from './scope-resolution/symbol-definition.js'; +export type { + ParameterTypeClass, + SymbolDefinition, + TypeParameter, +} from './scope-resolution/symbol-definition.js'; export type { ScopeId, DefId, diff --git a/gitnexus-shared/src/scope-resolution/finalize-algorithm.ts b/gitnexus-shared/src/scope-resolution/finalize-algorithm.ts index e50337af3..e2a90c253 100644 --- a/gitnexus-shared/src/scope-resolution/finalize-algorithm.ts +++ b/gitnexus-shared/src/scope-resolution/finalize-algorithm.ts @@ -373,6 +373,8 @@ function makeEdgeDrafts( targetFile: null, targetExportedName: extractExportedName(parsed), kind: edgeKindFor(parsed), + ...typeOnlyFor(parsed), + ...runsOnlyWhenCalledFor(parsed), linkStatus: 'unresolved', }; return [ @@ -392,7 +394,13 @@ function makeEdgeDrafts( // and resolved-dynamic imports are terminal at the file level — no // `targetDefId` needed since they materialize no `BindingRef`. Pre- // finalize them here so the fixpoint loop skips them entirely. - const targetFiles = Array.isArray(targetFile) ? targetFile : [targetFile]; + // Annotated rather than inferred: `isArray`'s `arg is any[]` predicate widens + // the true branch to a MUTABLE array, and a resolver may hand back a cached, + // frozen candidate list (Kotlin's `dirChildren` buckets do). Only `.map` is + // wanted here, so pinning `readonly` makes an in-place `.sort()`/`.push()` — + // which would reorder that resolver's index for the rest of the run — a + // compile error rather than a runtime TypeError. + const targetFiles: readonly string[] = Array.isArray(targetFile) ? targetFile : [targetFile]; const isFileLevelTerminal = parsed.kind === 'side-effect' || parsed.kind === 'dynamic-resolved'; return targetFiles.map((tf) => { const base: ImportEdge = { @@ -403,6 +411,8 @@ function makeEdgeDrafts( hooks.isNamespaceImport?.(parsed, tf, file.filePath) === true ? 'namespace' : edgeKindFor(parsed), + ...typeOnlyFor(parsed), + ...runsOnlyWhenCalledFor(parsed), }; return { source: parsed, @@ -420,6 +430,73 @@ function edgeKindFor(parsed: ParsedImport): ImportEdge['kind'] { return parsed.kind; } +/** + * Carry `ParsedImport.typeOnly` onto the edge — the erasure fact `check + * --cycles` needs and cannot re-derive, because `kind` is identical for the + * erased and the runtime spelling of the same import (`import type D` and + * `import D` both arrive as `alias`). + * + * `'typeOnly' in parsed` rather than a switch over the erasable kinds: only + * four variants declare the property, so `parsed.typeOnly` does not compile + * against the whole union, and `in` narrows it without naming them. That is + * also the safer shape — an enumeration has to be updated when a variant gains + * the property or the fact silently stops reaching the edge, while this form + * handles a new variant correctly whether or not it declares one. + * + * Returns a spreadable object rather than a `boolean` so an edge that is not + * type-only keeps the exact property set it had before this field existed. + * Every `finalized` edge is built by spreading `base`, so setting it here is + * enough for all of them. + */ +function typeOnlyFor(parsed: ParsedImport): { typeOnly?: true } { + return 'typeOnly' in parsed && parsed.typeOnly === true ? { typeOnly: true } : {}; +} + +/** + * Re-carry both runtime-presence flags from an existing edge onto a derived + * one. + * + * `expandWildcard` builds each `wildcard-expanded` edge from scratch rather + * than spreading the source (three fields differ per exported name), so every + * field it does not name is dropped. That is exactly how both flags were lost + * once already. Naming the pair here keeps "these two travel together" in one + * place, so a third presence flag is added in one place too. + */ +function carriedPresenceFlags(edge: Pick): { + typeOnly?: true; + runsOnlyWhenCalled?: true; +} { + return { + ...(edge.typeOnly === true ? { typeOnly: true } : {}), + ...(edge.runsOnlyWhenCalled === true ? { runsOnlyWhenCalled: true } : {}), + }; +} + +/** + * Carry `ParsedImport.runsOnlyWhenCalled` onto the edge — the position fact + * `check --cycles` needs and, unlike every other property of an import, cannot + * look up for itself. + * + * The scope an import was written in does not survive to here: + * `FinalizeFile.parsedImports` is a flat per-file list, and Phase 4 publishes + * the finalized edges under `file.moduleScope` (see `linkedByScope.set` above), + * so the consumer's map is keyed by the Module scope for every file. Walking + * that map's key to look for an enclosing `Function` therefore always starts — + * and ends — at a `Module`. Only the extractor still knows, so the edge has to + * carry what it decided. + * + * No `in` guard, unlike {@link typeOnlyFor}: position is a property of where + * the statement sits, so every variant declares `runsOnlyWhenCalled` and + * `parsed.runsOnlyWhenCalled` compiles against the whole union. A new variant + * that omits it is a build break here, which is the right outcome. + * + * Returns a spreadable object rather than a `boolean` so an edge that is not + * deferred keeps the exact property set it had before this field existed. + */ +function runsOnlyWhenCalledFor(parsed: ParsedImport): { runsOnlyWhenCalled?: true } { + return parsed.runsOnlyWhenCalled === true ? { runsOnlyWhenCalled: true } : {}; +} + function extractLocalName(parsed: ParsedImport): string { switch (parsed.kind) { case 'wildcard': @@ -515,9 +592,11 @@ function tryFinalize( return null; } - const viaFiles = [targetFile, ...followed.via]; + // Capped here too, not just inside the closure: this is the last hop, the + // one the emitted edge carries. + const viaFiles = extendVia(targetFile, followed.via); const transitiveVia = - draft.source.kind === 'reexport' || viaFiles.length > 1 ? Object.freeze(viaFiles) : undefined; + draft.source.kind === 'reexport' || viaFiles.length > 1 ? viaFiles : undefined; return { ...draft.base, @@ -549,11 +628,19 @@ type FileReexportClosure = ReadonlyMap; * level import graph. Replaces the legacy recursive * `followReexportChain` crawl with a bounded, stack-safe pass: * - * 1. **Sub-graph.** Build a directed graph whose edges are - * `reexport` and `wildcard` drafts only (regular imports do not - * contribute to the export surface, and `namespace`/ - * `reexport-namespace` are terminal — their target def lives in - * `localDefs`). + * 1. **Sub-graph.** Build a directed graph whose edges are `wildcard` + * drafts, `reexport` drafts, and `named`/`alias` drafts flagged + * `reexportsName` by their provider. `namespace`/`reexport-namespace` + * are terminal — their target def lives in `localDefs` — and are + * excluded on `base.kind`, after any `isNamespaceImport` + * reclassification. + * + * The flagged-named case is what languages with no dedicated + * re-export form need (today: Python, whose module-level + * `from m import x` both binds and republishes). For those providers + * the sub-graph is close to the file-level named-import graph, NOT a + * sparse barrel graph — measured ~20× more edges on the CPython + * stdlib — so read every bound below with that input class in mind. * 2. **SCC condensation.** Run the same iterative `tarjanSccs` over * the sub-graph. Output is in reverse-topological order (leaves * first), so when we process an SCC every out-of-SCC neighbor @@ -567,21 +654,34 @@ type FileReexportClosure = ReadonlyMap; * the cycle; first-wins precedence keeps the map monotone * so the fixpoint converges in at most |SCC| hops). * - * **Precedence semantics — preserved from the recursive crawl.** + * **Precedence semantics.** * * Named re-exports take precedence over wildcards. * * Within each kind, declaration order wins (first match for a - * given exported name is kept; later drafts skip). + * given exported name is kept; later drafts skip). This is only sound + * where the language makes a duplicate export illegal — true for TS + * and Rust `kind: 'reexport'`, false for the flagged-named form, where + * the module namespace rebinds (last write wins) and `if`/`try` pairs + * execute exactly one branch. For those, an in-file collision on the + * same published name with two different in-workspace targets is + * genuinely ambiguous and is dropped instead of guessed — see + * `collectAmbiguousReexports`. * * **Complexity.** * * Pre-pass: O(V + E_re) for SCC, plus O(|SCC| × Σ drafts) per cyclic - * SCC. For tree-shaped barrel graphs (the common case) it - * collapses to O(E_re) total. - * * Per-edge lookup at finalize time: O(1). + * SCC. Tree-shaped barrel graphs collapse to O(E_re) total; the + * flagged-named input class does not — the CPython stdlib produces 10 + * cyclic SCCs here where TypeScript-shaped input produced none. + * * Per-edge lookup at finalize time: O(1). Target `localDefs` are + * indexed by simple name on first use (`findExportByName`), so the + * per-hop cost is O(1) rather than a linear scan of the target file. * * `transitiveVia` preserves the exact file path chain for diagnostics * and graph provenance. Building those arrays copies the inherited path, - * which is O(depth²) in a pathological single-name barrel chain; practical - * TypeScript barrel chains are shallow enough that we keep exact paths - * instead of capping or summarizing them. + * which is Θ(depth²) in a single-name chain, and Θ(|SCC|²) for a cyclic + * SCC whose chain tracks the cycle. `MAX_REEXPORT_DEPTH = 100` bounded + * this until it was removed in `fc919ad6` for shallow TypeScript + * barrels; **nothing bounds it now**, and the flagged-named class feeds + * it far deeper input. Real `__init__.py` chains measure ≤ ~6, so this + * is a known unenforced assumption, not a live regression. * * Pathological deep chains that previously needed * `MAX_REEXPORT_DEPTH=100` to bound stack growth now resolve * in full and are bounded only by available memory — the @@ -595,19 +695,22 @@ function buildReexportClosures( const closures = new Map>(); for (const file of files) closures.set(file.filePath, new Map()); - // ── Step 1: build the re-export sub-graph (only resolvable - // reexport/wildcard targets contribute edges). + // ── Step 1: build the re-export sub-graph (only resolvable wildcard / + // reexport / flagged-named targets contribute edges), and collect the + // per-file ambiguous names in the same walk. const subGraph = new Map>(); + const ambiguous = new Map>(); for (const file of files) { const targets = new Set(); const drafts = edgeIndex.get(file.filePath); if (drafts !== undefined) { for (const d of drafts) { - if (d.source.kind !== 'reexport' && d.source.kind !== 'wildcard') continue; + if (!contributesReexportEdge(d)) continue; if (d.targetFile === null) continue; if (!byFilePath.has(d.targetFile)) continue; targets.add(d.targetFile); } + ambiguous.set(file.filePath, collectAmbiguousReexports(drafts, byFilePath)); } subGraph.set(file.filePath, targets); } @@ -623,7 +726,7 @@ function buildReexportClosures( if (!scc.isCycle) { const filePath = scc.files[0]; if (filePath !== undefined) { - populateFileClosure(filePath, byFilePath, edgeIndex, closures); + populateFileClosure(filePath, byFilePath, edgeIndex, closures, ambiguous); } continue; } @@ -637,7 +740,7 @@ function buildReexportClosures( progressed = false; iter++; for (const filePath of scc.files) { - if (populateFileClosure(filePath, byFilePath, edgeIndex, closures)) { + if (populateFileClosure(filePath, byFilePath, edgeIndex, closures, ambiguous)) { progressed = true; } } @@ -647,6 +750,95 @@ function buildReexportClosures( return closures; } +/** + * Does this import republish names from its target under the *importing* file, + * making it an edge in the re-export sub-graph? + * + * `reexport` and `wildcard` are the explicit forms; `named`/`alias` drafts + * flagged `reexportsName` cover providers whose ordinary import syntax also + * republishes (see that field on `ParsedImport` for the contract). + * + * Tested on `base.kind`, not `source.kind`: `isNamespaceImport` can reclassify + * a `named` draft to `namespace` (Python's `from . import submodule`), and a + * namespace import aliases the target *module* — it publishes no name, so + * admitting it would republish whatever def happens to share the module's + * simple name. + */ +function contributesReexportEdge(draft: ImportEdgeDraft): boolean { + if (draft.base.kind === 'namespace') return false; + if (draft.source.kind === 'wildcard') return true; + return isNamedReexport(draft); +} + +/** + * Named (non-wildcard) re-export. The narrowed type lets `populateFileClosure` + * read `localName` (the name this file publishes) and `importedName` (the name + * the target exports) without re-discriminating on `kind`. + */ +function isNamedReexport(draft: ImportEdgeDraft): draft is ImportEdgeDraft & { + readonly source: Extract; +} { + if (draft.base.kind === 'namespace') return false; + const source = draft.source; + if (source.kind === 'reexport') return true; + return (source.kind === 'named' || source.kind === 'alias') && source.reexportsName === true; +} + +/** + * Names this file publishes ambiguously, which the closure must decline to + * answer for rather than guess at. + * + * Declaration-order first-wins is sound only where a duplicate export is + * illegal — two `export { X } from …` is a TypeScript compile error, so the + * rule never fires. The flagged-named form has no such guarantee: CPython's + * module namespace rebinds, so + * + * from .v1 import Client # legacy, left behind + * from .v2 import Client # the actual public Client + * + * binds `v2`, and first-wins would attribute every `from pkg import Client` in + * the repo to the dead implementation. Last-wins is not the answer either — + * for the equally common `try:`/`except ImportError:` and `if + * sys.version_info` pairs exactly one branch runs, and which one is not + * decidable here. So both directions are wrong on real code and the entry is + * dropped: the importer stays unresolved, which is exactly the pre-#2864 + * answer, and the file-level IMPORTS edge is unaffected. + * + * Computed once per file from data phase 0 froze (`edgeIndex`, `targetFile`) + * and never revised, so the closure map stays monotone and the `|SCC| + 1` + * fixpoint cap keeps the meaning it has above. A set that could grow mid- + * fixpoint would need retraction to propagate to files that already inherited + * the name, and would break both. + * + * Only two flagged drafts resolving to two *different in-workspace files* + * count. Duplicates of the same target are harmless, and an unresolvable + * target (`null` — the `try: import ujson / except: import json` shape, both + * external) never entered the closure to begin with. + * + * ponytail: named-vs-named only. Wildcard-vs-wildcard collisions are also + * first-wins today, but their inherited half depends on target closures that + * are still filling in, so detecting them needs a set that grows during the + * fixpoint — the thing this pre-pass exists to avoid. + */ +function collectAmbiguousReexports( + drafts: readonly ImportEdgeDraft[], + byFilePath: ReadonlyMap, +): ReadonlySet { + const firstTarget = new Map(); + const conflicting = new Set(); + for (const draft of drafts) { + if (!isNamedReexport(draft)) continue; + if (draft.source.kind === 'reexport') continue; // explicit form: duplicates are illegal upstream + const targetFile = draft.targetFile; + if (targetFile === null || !byFilePath.has(targetFile)) continue; + const localName = draft.source.localName; + const seen = firstTarget.get(localName); + if (seen === undefined) firstTarget.set(localName, targetFile); + else if (seen !== targetFile) conflicting.add(localName); + } + return conflicting; +} + /** * Populate one file's re-export closure for one pass. Returns `true` * iff the closure grew (signalling fixpoint progress to the caller). @@ -666,24 +858,29 @@ function populateFileClosure( byFilePath: ReadonlyMap, edgeIndex: ReadonlyMap, closures: Map>, + ambiguousByFile: ReadonlyMap>, ): boolean { const myClosure = closures.get(filePath); if (myClosure === undefined) return false; const before = myClosure.size; const drafts = edgeIndex.get(filePath); if (drafts === undefined) return false; + // Fixed for the whole run — see `collectAmbiguousReexports`. Consulted in + // both loops below: suppressing only the named one would let a later + // `import *` refill the name and reinstate an arbitrary winner. + const ambiguous = ambiguousByFile.get(filePath) ?? EMPTY_NAME_SET; // Named re-exports — precedence over wildcards, declaration order // first-wins for duplicates of the same exported name. for (const draft of drafts) { - if (draft.source.kind !== 'reexport') continue; + if (!isNamedReexport(draft)) continue; const targetFile = draft.targetFile; if (targetFile === null) continue; const targetModule = byFilePath.get(targetFile); if (targetModule === undefined) continue; const localName = draft.source.localName; - if (myClosure.has(localName)) continue; + if (ambiguous.has(localName) || myClosure.has(localName)) continue; const importedName = draft.source.importedName; const direct = findExportByName(targetModule.localDefs, importedName); @@ -695,7 +892,7 @@ function populateFileClosure( if (inherited !== undefined) { myClosure.set(localName, { def: inherited.def, - via: Object.freeze([targetFile, ...inherited.via]), + via: extendVia(targetFile, inherited.via), }); } // Else: target's closure is still empty (in-SCC, awaiting next @@ -714,16 +911,16 @@ function populateFileClosure( for (const def of targetModule.localDefs) { const name = deriveSimpleName(def); - if (name === null || myClosure.has(name)) continue; + if (name === null || ambiguous.has(name) || myClosure.has(name)) continue; myClosure.set(name, { def, via: Object.freeze([targetFile]) }); } const targetClosure = closures.get(targetFile); if (targetClosure !== undefined) { for (const [name, entry] of targetClosure) { - if (myClosure.has(name)) continue; + if (ambiguous.has(name) || myClosure.has(name)) continue; myClosure.set(name, { def: entry.def, - via: Object.freeze([targetFile, ...entry.via]), + via: extendVia(targetFile, entry.via), }); } } @@ -732,6 +929,35 @@ function populateFileClosure( return myClosure.size > before; } +/** + * Longest `transitiveVia` chain kept intact. Beyond this the tail is replaced + * by {@link VIA_TRUNCATED}, so the entry still says "this came through a long + * chain" without carrying it. + * + * Reinstates a bound the algorithm lost. Each hop copies the inherited path, + * so an uncapped chain is Θ(depth²) in both time and retained memory, and + * Θ(|SCC|²) for a cycle whose chain tracks it. `MAX_REEXPORT_DEPTH = 100` + * covered this until `fc919ad6` removed it — correctly, for the TypeScript + * barrels that were then the only input, which are shallow. Admitting + * flagged-named imports changes the input class, so the bound comes back. + * + * 32 against a measured real-world worst case of ~6 for `__init__.py` chains: + * five times the deepest chain anyone has, and it turns the quadratic into + * O(depth × 32). Safe to truncate because `ImportEdge.transitiveVia` has no + * production reader — it is diagnostic provenance, emitted and typed but not + * consumed by graph emission (`emitImportEdges` dedups on source→target and + * drops it). + */ +const MAX_VIA_LENGTH = 32; +const VIA_TRUNCATED = '…'; + +function extendVia(head: string, inherited: readonly string[]): readonly string[] { + if (inherited.length + 1 <= MAX_VIA_LENGTH) return Object.freeze([head, ...inherited]); + // Already truncated one hop down: re-truncating keeps the array at the cap + // rather than growing it by one per hop, which is the whole point. + return Object.freeze([head, ...inherited.slice(0, MAX_VIA_LENGTH - 2), VIA_TRUNCATED]); +} + /** * O(1) lookup into a precomputed re-export closure. Replaces the legacy * recursive `followReexportChain` traversal with a single map indexing. @@ -792,15 +1018,52 @@ function findExportByName( // // See `gitnexus/test/integration/resolvers/typescript-hof-callbacks.test.ts` // for the cross-file regression this rule prevents. - let fallback: SymbolDefinition | undefined; - for (const d of defs) { - if (deriveSimpleName(d) !== name) continue; - if (isCallableOrTypeLike(d.type)) return d; - if (fallback === undefined) fallback = d; - } - return fallback; + return indexExportsByName(defs).get(name); } +/** + * `simple name → winning def` for one file's `localDefs`, built once and + * memoized on the array itself. + * + * Every caller of `findExportByName` sits in a loop that revisits the same + * target files: the phase-3 fixpoint rescans a target once per iteration, and + * `populateFileClosure` scans once per admitted re-export — which for a + * provider setting `reexportsName` is every named import in the file, where it + * used to be zero. Keeping the scan turned that into O(edges × defs). + * + * Safe to key on identity because `FinalizeFile.localDefs` is documented static + * input that the fixpoint never mutates; a `WeakMap` ties each index to its + * array's lifetime with no cross-pass state to invalidate. Same shape as the + * `defById` map `materializeBindings` already builds for the same reason. + */ +const EXPORTS_BY_NAME = new WeakMap< + readonly SymbolDefinition[], + ReadonlyMap +>(); + +function indexExportsByName( + defs: readonly SymbolDefinition[], +): ReadonlyMap { + const cached = EXPORTS_BY_NAME.get(defs); + if (cached !== undefined) return cached; + const index = new Map(); + for (const d of defs) { + const name = deriveSimpleName(d); + if (name === null) continue; + const existing = index.get(name); + // First match wins within a tier; a callable displaces a stored value + // shadow but never another callable — identical to the linear scan's + // "first callable if any, else first match". + if (existing === undefined) index.set(name, d); + else if (!isCallableOrTypeLike(existing.type) && isCallableOrTypeLike(d.type)) + index.set(name, d); + } + EXPORTS_BY_NAME.set(defs, index); + return index; +} + +const EMPTY_NAME_SET: ReadonlySet = new Set(); + const CALLABLE_OR_TYPE_LIKE: ReadonlySet = new Set([ 'Function', 'Method', @@ -874,6 +1137,35 @@ function expandWildcard( kind: 'wildcard-expanded', targetModuleScope: edge.targetModuleScope, targetDefId: def.nodeId, + // Every expanded edge inherits the presence facts of the ONE statement it + // came from. They are built fresh rather than spread from `edge` because + // `localName`, `targetExportedName` and `targetDefId` all differ per name + // — which is exactly how a property added to the wildcard edge upstream + // gets silently dropped here, and how `runsOnlyWhenCalled` was. + // + // `runsOnlyWhenCalled`: Ruby's `def f; require './m'; end` is one + // statement inside one method body — and every Ruby `require` is a + // wildcard, since the required file's whole surface becomes visible — so + // each name it brings in is bound only when `f` runs. Losing the flag + // here re-reports the pair as an initialization dependency and + // suppresses nothing — it INVENTS a cycle (`check --cycles`), which is + // why this is carried and not derived. + // + // Ruby is the reachable spelling. Python has no function-local + // `from x import *` — it is a SyntaxError — and Rust's `fn f() { use + // m::*; }`, which IS legal, is not deferred at all: `use` is a + // compile-time path alias, so the Rust provider opts out of the position + // rule (`LanguageProvider.importsExecuteWhereWritten`). + // + // `typeOnly`: unreachable today and deliberately kept. No provider emits + // a type-only wildcard — `reexport-wildcard` returns `kind: 'wildcard'` + // with no `typeOnly` because `export type *` is unparseable by the + // vendored grammar (documented on `ParsedImport`'s `wildcard` variant). + // It is propagated so the day that gap closes does not silently + // reintroduce this same defect for erasure. Do not delete it as dead + // code; `typeOnlyFor` is the gate that decides whether it can ever be + // set, and it is where the correspondence is enforced. + ...carriedPresenceFlags(edge), }); } return expanded; diff --git a/gitnexus-shared/src/scope-resolution/reference-site.ts b/gitnexus-shared/src/scope-resolution/reference-site.ts index 6629dacd3..b559d32e3 100644 --- a/gitnexus-shared/src/scope-resolution/reference-site.ts +++ b/gitnexus-shared/src/scope-resolution/reference-site.ts @@ -82,6 +82,28 @@ export interface ReferenceSite { * otherwise, in which case resolution is unchanged. */ readonly rawQualifiedName?: string; + /** + * Top-level generic/template arguments the source wrote ON this reference — + * `class UserValidator : IValidator` yields `['string']` on the + * `inherits` site whose `name` is `IValidator`. + * + * `name` is the BASE name and stays that way: every lookup in resolution is + * keyed by it, and one declaration answers for every instantiation of itself. + * This records what the erasure threw away, so a consumer that needs the + * INSTANTIATION — receiver-bound interface dispatch, which must not fan a + * `IValidator` receiver out to an `IValidator` implementor + * (#2912) — can ask for it without re-parsing the source. + * + * Derived generically from the anchor capture's own text (see + * `collectReferenceSites`), so no language query change is needed: an emitter + * whose `@reference.inherits` anchor spans the whole base gets this for free, + * and one whose anchor is the bare name simply leaves it absent. + * + * ABSENT MEANS UNKNOWN, never "not generic" — the two are indistinguishable + * here, and only the first is safe to act on. Consumers must fail OPEN on + * absence (keep the target), matching `SymbolDefinition.typeParameters`. + */ + readonly typeArguments?: readonly string[]; /** Source-text range of this reference. */ readonly atRange: Range; /** diff --git a/gitnexus-shared/src/scope-resolution/symbol-definition.ts b/gitnexus-shared/src/scope-resolution/symbol-definition.ts index 64fbce93a..e90b0be85 100644 --- a/gitnexus-shared/src/scope-resolution/symbol-definition.ts +++ b/gitnexus-shared/src/scope-resolution/symbol-definition.ts @@ -24,6 +24,38 @@ export interface ParameterTypeClass { templateArguments?: string[]; } +/** + * One declared generic/template TYPE PARAMETER — `T` in `class Box`, `template struct Vec`, `interface Repo`. + * + * NOT the same axis as `SymbolDefinition.templateArguments`, and conflating the + * two is the defect this shape exists to end. `templateArguments` records the + * arguments a declaration was written AGAINST (`template <> struct Vec` → + * `['bool']`); `typeParameters` records the parameters it was written IN TERMS + * OF. A declaration can carry both — a C++ partial specialization + * `template struct Vec` has `templateArguments: ['T*']` AND + * `typeParameters: [{name: 'T'}]` — and that pairing is precisely what tells a + * partial specialization apart from the full specialization `template <> struct + * Vec`, which carries the identical `templateArguments` and NO parameters. + */ +export interface TypeParameter { + /** The parameter's declared name, exactly as written (`T`, `Ts`, `TKey`). */ + name: string; + /** + * The declared upper bound / constraint, verbatim and un-split, when the + * declaration states one inline: `T extends Repo` → `Repo`, `T : Repo` → + * `Repo`, `T extends Repo & Closeable` → `Repo & Closeable`. + * + * VERBATIM because the intersection/compound spellings differ per language + * and a shared consumer that wants the first bound can take the first token + * itself, while one that wants to round-trip the source cannot recover what a + * split threw away. Absent when the parameter is unbounded, and absent when + * the bound is declared OUT OF LINE (C# `where T : IRepo`, Kotlin/Rust + * `where` clauses) — see `parseTypeParameterList`. + */ + bound?: string; +} + export interface SymbolDefinition { nodeId: string; filePath: string; @@ -48,6 +80,18 @@ export interface SymbolDefinition { declaredType?: string; /** Generic/template specialization arguments for class-like symbols (e.g. ['User'], ['T*']). */ templateArguments?: string[]; + /** + * Declared generic/template TYPE PARAMETERS, in DECLARATION ORDER — see + * {@link TypeParameter} for how this differs from `templateArguments`. + * + * ORDER IS LOAD-BEARING: substitution is positional (`Repo` binds the + * FIRST parameter), so a set or a name-keyed map would discard exactly the + * information this carries. Absent for a non-generic declaration and for every + * language whose captures do not populate it, so a reader MUST treat absence + * as "unknown", never as "not generic" — the two are indistinguishable here + * and only the first is safe to act on. + */ + typeParameters?: TypeParameter[]; /** Per-language constraint payload for template / generic overloads * (e.g. C++ `enable_if_t` predicate trees, C++20 `requires` clauses). * Opaque to shared code — the producing language adapter owns the shape @@ -63,6 +107,10 @@ export interface SymbolDefinition { * Unavailable callables still participate in overload selection, but a * selected unavailable target must suppress edge emission. */ isDeleted?: boolean; + /** True when the declaration identity was synthesized rather than written in + * source (for example an anonymous class). Consumers may use this only as a + * conservative priority hint; it does not change graph-node identity. */ + isSynthetic?: boolean; /** Links Method/Constructor/Property to owning Class/Struct/Trait nodeId */ ownerId?: string; /** #1982/#1993: bridge-held enclosing-namespace path (e.g. `NS1`, `Outer.Inner`) diff --git a/gitnexus-shared/src/scope-resolution/types.ts b/gitnexus-shared/src/scope-resolution/types.ts index dcd074400..80b961bda 100644 --- a/gitnexus-shared/src/scope-resolution/types.ts +++ b/gitnexus-shared/src/scope-resolution/types.ts @@ -119,8 +119,96 @@ export type ParsedImport = readonly importedName: string; readonly targetRaw: string; /** Provider-specific imported symbol category when module and symbol - * namespaces have distinct resolution rules (for example PHP). */ + * namespaces have distinct resolution rules (for example PHP). + * + * **Not** the same fact as {@link ParsedImport.typeOnly} — see the note + * on `typeOnly` below, which is documented on this variant. */ readonly importedSymbolKind?: 'type' | 'function' | 'const'; + /** + * Is this import ERASED before the module ever runs? + * + * TypeScript `import type { X } from './m'` and `import { type X }` are + * deleted by `tsc`: no `require`/`import` for `./m` survives in the + * emitted JavaScript, so the pair cannot force a module-INITIALIZATION + * order and cannot participate in an init cycle. That is the one thing + * `check --cycles` exists to find, so the fact has to survive from the + * syntax down to the emitted `IMPORTS` edge — see `ImportEdge.typeOnly` + * and `graph-bridge/imports-to-edges.ts`. + * + * **Distinct from `importedSymbolKind: 'type'`, which is NOT a substitute.** + * That field is a resolution-NAMESPACE category (PHP's `use function` / + * `use const` split), it exists only on this variant, and it says "the + * thing imported is a type". A symbol being a type says nothing about + * whether the import STATEMENT is erased, and PHP erases nothing at all. + * This field is about the statement's runtime existence, not the symbol's + * category. + * + * Set only by providers whose syntax marks it. Absent everywhere else, + * which reads as "not erased" — the fail-safe direction, since it only + * makes `check --cycles` over-report. + * + * That fail-safe matters more than it first looks, because an explicit + * `type` is a SUFFICIENT signal of erasure and not a necessary one. With + * neither `verbatimModuleSyntax` nor `importsNotUsedAsValues: preserve` + * set — this repo sets neither — `tsc` also elides a plain + * `import { SomeInterface }` whose bindings are every one of them used in + * type position. Those statements are erased at run time and carry no + * marker, so they stay tagged as initializing and `check --cycles` can + * still report a cycle that cannot exist. Closing that gap needs + * whole-program binding USE information, not import syntax, which is why + * this field stops at what the syntax states. + */ + readonly typeOnly?: boolean; + /** + * Was this import written inside a function body — so that it runs only + * when something CALLS that function, never while the module itself is + * initializing? + * + * Python's `def f(): from x import Y` and a CommonJS + * `function f() { const { Y } = require('./x'); }` are the spellings. + * Both are syntactically ordinary imports — no `kind` tells them apart + * from a top-level one, and nothing about the target does either. Only + * their POSITION defers them. + * + * Not every language's imports are like that, and the rule is wrong for + * the ones that are not: Rust's `use` and C/C++'s `#include` are legal + * in a function body and are deferred by NOTHING, because neither is an + * executed statement. Those providers opt out — see + * `LanguageProvider.importsExecuteWhereWritten`, below. + * + * **Why this cannot be re-derived downstream — the whole reason the + * field exists.** The natural place to decide it looks like the graph + * bridge, by walking the scope the finalized edges hang off; that is + * exactly what `graph-bridge/imports-to-edges.ts` once attempted, and it + * is dead code by construction. `finalize-algorithm.ts:295` publishes + * every file's finalized edges as + * `linkedByScope.set(file.moduleScope, …)`, so the map the bridge + * receives is keyed by the file's **Module** scope and by nothing else: + * the walk starts at a `Module` every time and answers `false` for every + * import in the tree. Finalize cannot recover the position either — + * `FinalizeFile.parsedImports` is a flat per-file `ParsedImport[]` with + * no scope attached. The extractor is the last stage that still knows + * where the statement sat (`scope-extractor.ts`, Pass 3), so it marks the + * fact here and it rides the edge from there — see + * {@link ImportEdge.runsOnlyWhenCalled}. + * + * Consumed by `check --cycles`, which asks "can these modules be + * initialized in any order?". A deferred import carries no + * initialization order, and deferring one is the standard way to BREAK + * an init cycle, so counting it reports the fix as the bug. + * + * Set by the central extractor for every language, not by providers — + * except that a provider may declare that its imports do not execute + * where they are written (`LanguageProvider.importsExecuteWhereWritten: + * false`) and be skipped entirely. C, C++, Rust and COBOL do. A `#include` + * or a `use` inside a function body is not deferred: the header is + * spliced and the path alias is resolved before anything runs, so the + * pair really is a dependency and the cycle it can form is real. + * + * Absent reads as "runs at initialization" — the fail-safe direction, + * since it only makes `check --cycles` over-report. + */ + readonly runsOnlyWhenCalled?: boolean; /** * Set by providers when `targetRaw` already names the imported symbol * rather than only its containing module. Consumers that compose @@ -128,6 +216,40 @@ export type ParsedImport = * duplicating `importedName`. */ readonly targetIncludesImportedName?: boolean; + /** + * Set by providers whose import syntax *also* republishes the name from + * the importing module, so a third file can import it from there. + * + * Python has no dedicated re-export form: a module-level + * `from pkg.impl import X` binds `X` locally **and** publishes it as + * `pkg.X`, which is the standard way a package `__init__.py` declares + * its public surface. Languages with an explicit form (TS `export … from`, + * Rust `pub use`) emit `kind: 'reexport'` instead and leave this unset. + * + * **The flag must track actual republication, not syntax.** Only a + * module-level statement publishes: the same `from m import X` inside a + * `def` or `class` body binds locally and puts nothing in the module + * namespace, so flagging it fabricates a re-export of a name no importer + * can reach. `if` / `try` / `for` / `with` do not suppress it — Python + * has no block scope. A provider that cannot tell these apart at + * interpret time must carry the fact down from its capture emitter, + * where the syntax node is still available. + * + * **Why not `kind: 'reexport'`.** Not because that form drops the local + * binding — `materializeBindings` creates a module-scope `BindingRef` + * for every linked edge, re-export included. It is that `reexport` + * changes what the binding *is*: `origin` flips to `'reexport'`, which + * carries different evidence weight and `ORIGIN_PRIORITY`, and it + * misreports the parse-time syntax Python actually wrote. A flag adds + * the export-surface fact without restating the import as something the + * source does not say. + * + * Consumed by `buildReexportClosures` (`finalize-algorithm.ts`), which + * also documents how ambiguous duplicates of one published name are + * handled — the precedence rules that hold for an explicit re-export do + * not carry over. + */ + readonly reexportsName?: boolean; } /** * Per-name import with rename. @@ -144,8 +266,18 @@ export type ParsedImport = readonly targetRaw: string; /** See the same field on the `named` variant. */ readonly importedSymbolKind?: 'type' | 'function' | 'const'; + /** See the same field on the `named` variant — including why it is not + * interchangeable with `importedSymbolKind`. Reaches this variant from + * `import type D from './m'` and `import { type X as Y } from './m'`. */ + readonly typeOnly?: boolean; + /** See the same field on the `named` variant. Reaches this variant from + * Python's `def f(): from x import Y as Z` and a CommonJS + * `function f() { const { Y: Z } = require('./x'); }`. */ + readonly runsOnlyWhenCalled?: boolean; /** See the same field on the `named` variant. */ readonly targetIncludesImportedName?: boolean; + /** See the same field on the `named` variant. */ + readonly reexportsName?: boolean; } /** * Qualified module handle, with or without rename. `importedName` is the @@ -165,6 +297,12 @@ export type ParsedImport = /** Module being aliased (e.g. `numpy` in `import numpy as np`). */ readonly importedName: string; readonly targetRaw: string; + /** See the same field on the `named` variant. Reaches this variant from + * TypeScript `import type * as N from './m'`. */ + readonly typeOnly?: boolean; + /** See the same field on the `named` variant. Reaches this variant from + * Python's `def f(): import numpy as np`. */ + readonly runsOnlyWhenCalled?: boolean; } /** * Syntactically-detectable parse-time re-export. Finalize may still produce @@ -186,6 +324,19 @@ export type ParsedImport = readonly targetRaw: string; /** Set when the re-export renames the symbol (e.g. `export { X as Y } from './y'`). */ readonly alias?: string; + /** See the same field on the `named` variant. Reaches this variant from + * TypeScript `export type { X } from './y'` and `export { type X } from './y'`. */ + readonly typeOnly?: boolean; + /** See the same field on the `named` variant. NO spelling reaches this + * variant today: the two providers that emit `reexport` are TypeScript + * / JavaScript, whose `export … from` is a module-top-level-only + * declaration, and Rust, whose `pub use` is a compile-time path alias + * that its provider exempts from the position rule outright + * (`LanguageProvider.importsExecuteWhereWritten`). Kept because the + * extractor sets the field with no `switch` on `kind`, so a re-export + * form that IS an executed statement would be tagged the moment one + * appears — not because anything sets it now. */ + readonly runsOnlyWhenCalled?: boolean; } /** * Wildcard import — brings every exported name from the target module into @@ -197,10 +348,26 @@ export type ParsedImport = * - Python `from foo import *` → `{ kind: 'wildcard', targetRaw: 'foo' }` * - JS `export * from './foo'` → `{ kind: 'wildcard', targetRaw: './foo' }` * - Rust `pub use foo::*` → `{ kind: 'wildcard', targetRaw: 'foo' }` + * + * No `typeOnly` here on purpose. The one syntax that would set it, + * TypeScript 5.0's `export type * from './m'`, is not parsed by the + * vendored tree-sitter-typescript grammar — it yields an `ERROR` node + * holding the bare `type` token, so the fact is not readable at the + * statement level (see `typescript/import-decomposer.ts`). Add the field + * with the grammar that can express it, not before. */ | { readonly kind: 'wildcard'; readonly targetRaw: string; + /** See the same field on the `named` variant. Present here although + * `typeOnly` is not: erasure is a syntactic fact this spelling cannot + * express, but POSITION is not — Ruby's `def f; require './m'; end` is + * a wildcard (everything in the required file becomes visible) and IS + * deferred. Python cannot reach it: `from x import *` inside a `def` is + * a SyntaxError. Rust's fn-local `use foo::*` is legal but not + * deferred — `use` does not execute + * (`LanguageProvider.importsExecuteWhereWritten`). */ + readonly runsOnlyWhenCalled?: boolean; } /** * Runtime-computed target — the import path is not a static literal at @@ -217,6 +384,9 @@ export type ParsedImport = readonly localName: string; /** Source text of the unresolved expression when available; `null` otherwise. */ readonly targetRaw: string | null; + /** See the same field on the `named` variant. Set by position like every + * other variant; this kind links no target, so nothing reads it here. */ + readonly runsOnlyWhenCalled?: boolean; } /** * Lazy / dynamic import whose target IS a static string literal at parse @@ -238,6 +408,10 @@ export type ParsedImport = | { readonly kind: 'dynamic-resolved'; readonly targetRaw: string; + /** See the same field on the `named` variant. Redundant on this kind — + * `import()` is already deferred wherever it is written — but set + * uniformly, because position is decided without consulting `kind`. */ + readonly runsOnlyWhenCalled?: boolean; } /** * Bare-source / side-effect import that introduces no local name binding @@ -253,6 +427,10 @@ export type ParsedImport = | { readonly kind: 'side-effect'; readonly targetRaw: string; + /** See the same field on the `named` variant. Reaches this variant from + * a bare CommonJS `function f() { require('./polyfill'); }` — the ESM + * spelling `import './polyfill'` cannot, being top-level only. */ + readonly runsOnlyWhenCalled?: boolean; }; /** @@ -348,6 +526,37 @@ export interface ImportEdge { | 'side-effect'; /** Re-export chain, for provenance (e.g., `['./y']` when re-exported via `./y`). */ readonly transitiveVia?: readonly string[]; + /** + * The import is erased before the module runs — see `ParsedImport`'s + * `typeOnly` on the `named` variant for the full note, including why + * `importedSymbolKind: 'type'` is a different fact and not a substitute. + * + * Carried straight from the `ParsedImport` by `makeEdgeDrafts`. The edge is + * still emitted: a type-only import is a real source-level dependency that + * `impact` and `trace` must see, and editing the target still breaks the + * importer's typecheck. What the flag removes is the claim that the pair + * forces an INITIALIZATION order. + */ + readonly typeOnly?: boolean; + /** + * The import was written inside a function body, so it runs only when that + * function is called — never during module initialization. See + * `ParsedImport`'s `runsOnlyWhenCalled` on the `named` variant for the full + * note, including why the consumer cannot re-derive this from the scope tree + * and therefore has to be told (`finalize-algorithm.ts:295`). + * + * Carried straight from the `ParsedImport` by `makeEdgeDrafts`, for the same + * reason `typeOnly` is: the edge is where `graph-bridge/imports-to-edges.ts` + * can still see it. The edge is still emitted either way — a deferred import + * is a real dependency. What the flag removes is the claim that the pair + * forces an INITIALIZATION order. + * + * Distinct from `kind === 'dynamic-resolved'`, which records the OTHER way an + * import can be deferred (`import('./m')`). Neither implies the other: a + * top-level `import()` is deferred with this flag unset, and a function-local + * `from x import Y` is deferred with an ordinary `named` kind. + */ + readonly runsOnlyWhenCalled?: boolean; /** Set to `'unresolved'` when the SCC fixpoint could not link this edge. */ readonly linkStatus?: 'unresolved'; } diff --git a/gitnexus-web/package-lock.json b/gitnexus-web/package-lock.json index 9261cc32a..a877b7be0 100644 --- a/gitnexus-web/package-lock.json +++ b/gitnexus-web/package-lock.json @@ -11,14 +11,14 @@ "@langchain/anthropic": "^1.5.1", "@langchain/core": "^1.2.3", "@langchain/google-genai": "^2.2.0", - "@langchain/langgraph": "^1.4.8", + "@langchain/langgraph": "^1.4.9", "@langchain/ollama": "^1.3.0", "@langchain/openai": "^1.5.3", "@sigma/edge-curve": "^3.1.0", "@tailwindcss/vite": "^4.3.3", "axios": "^1.18.1", "d3": "^7.9.0", - "dompurify": "^3.4.12", + "dompurify": "^3.4.13", "gitnexus-shared": "file:../gitnexus-shared", "graphology": "^0.26.0", "graphology-indices": "^0.17.0", @@ -28,14 +28,14 @@ "graphology-utils": "^2.3.0", "i18next": "^26.3.6", "i18next-browser-languagedetector": "^8.2.1", - "langchain": "^1.4.6", + "langchain": "^1.5.4", "lru-cache": "^11.5.2", - "lucide-react": "^1.23.0", - "mermaid": "^11.15.0", + "lucide-react": "^1.28.0", + "mermaid": "^11.16.1", "mnemonist": "^0.40.4", "pandemonium": "^2.4.0", "react": "^19.2.5", - "react-dom": "^19.2.7", + "react-dom": "^19.2.8", "react-i18next": "^17.0.11", "react-markdown": "^10.1.0", "react-syntax-highlighter": "^16.1.1", @@ -55,10 +55,10 @@ "@types/dompurify": "^3.2.0", "@types/node": "^26.0.1", "@types/react": "^19.2.14", - "@types/react-dom": "^19.2.3", + "@types/react-dom": "^19.2.4", "@types/react-syntax-highlighter": "^15.5.13", "@vercel/node": "^5.8.23", - "@vitejs/plugin-react": "^6.0.4", + "@vitejs/plugin-react": "^6.0.5", "@vitest/coverage-v8": "^4.1.9", "jsdom": "^29.1.1", "tree-sitter-wasms": "^0.1.13", @@ -289,9 +289,9 @@ } }, "node_modules/@braintree/sanitize-url": { - "version": "7.1.1", - "resolved": "https://registry.npmjs.org/@braintree/sanitize-url/-/sanitize-url-7.1.1.tgz", - "integrity": "sha512-i1L7noDNxtFyL5DmZafWy1wRVhGehQmzZaz1HiN5e7iylJMSZR7ekOV7NsIqa5qBldlLrsKv4HbgFUVlQrz8Mw==", + "version": "7.1.2", + "resolved": "https://registry.npmjs.org/@braintree/sanitize-url/-/sanitize-url-7.1.2.tgz", + "integrity": "sha512-jigsZK+sMF/cuiB7sERuo9V7N9jx+dhmHHnQyDSVdpZwVutaBu7WvNYqMDLSgFgfB30n452TP3vjDAvFC973mA==", "license": "MIT" }, "node_modules/@bramus/specificity": { @@ -1172,13 +1172,13 @@ } }, "node_modules/@langchain/langgraph": { - "version": "1.4.8", - "resolved": "https://registry.npmjs.org/@langchain/langgraph/-/langgraph-1.4.8.tgz", - "integrity": "sha512-DN1Np1XefdBEbp1qBKlt39cwoL743AAGpR5Ipja0gY2YbWvsoQnOTIrjnj/orSAhaUYsdTKS8VSWdFzsHZo6Ig==", + "version": "1.4.9", + "resolved": "https://registry.npmjs.org/@langchain/langgraph/-/langgraph-1.4.9.tgz", + "integrity": "sha512-EvD9rS66Cya09y6rbMgD3Ir8miAkJQFo7FyJOPRPO736Kz3y5TeyeBDOS8ctff/jRc788bPijHx2NVFM79Qqig==", "license": "MIT", "dependencies": { "@langchain/langgraph-checkpoint": "^1.1.3", - "@langchain/langgraph-sdk": "~1.9.26", + "@langchain/langgraph-sdk": "~1.9.28", "@langchain/protocol": "^0.0.18", "@standard-schema/spec": "1.1.0" }, @@ -1330,12 +1330,12 @@ } }, "node_modules/@mermaid-js/parser": { - "version": "1.1.1", - "resolved": "https://registry.npmjs.org/@mermaid-js/parser/-/parser-1.1.1.tgz", - "integrity": "sha512-VuHdsYMK1bT6X2JbcAaWAhugTRvRBRyuZgd+c22swUeI9g/ntaxF7CY7dYarhZovofCbUNO0G7JesfmNtjYOCw==", + "version": "1.2.0", + "resolved": "https://registry.npmjs.org/@mermaid-js/parser/-/parser-1.2.0.tgz", + "integrity": "sha512-oYPyv8A4As1yH5Bx+04iQEQxXuIQDe0GKCNSRgao6z8AM9jixXIfP0vsppRLvGf+nKIOb9/LdpWA4YuJiVvESA==", "license": "MIT", "dependencies": { - "@chevrotain/types": "~11.1.1" + "@chevrotain/types": "~11.1.2" } }, "node_modules/@napi-rs/wasm-runtime": { @@ -1755,12 +1755,6 @@ "tailwindcss": "4.3.3" } }, - "node_modules/@tailwindcss/node/node_modules/tailwindcss": { - "version": "4.3.2", - "resolved": "https://registry.npmjs.org/tailwindcss/-/tailwindcss-4.3.2.tgz", - "integrity": "sha512-WtctNNSH8A9jlMIqxzuYumOHU5uGZyRv0Q5svQl+oEPy5w84YpBxdb7MdqyiSPQge5jTJ6zFQLq0PFygdccSBA==", - "license": "MIT" - }, "node_modules/@tailwindcss/oxide": { "version": "4.3.3", "resolved": "https://registry.npmjs.org/@tailwindcss/oxide/-/oxide-4.3.3.tgz", @@ -2075,12 +2069,6 @@ "vite": "^5.2.0 || ^6 || ^7 || ^8" } }, - "node_modules/@tailwindcss/vite/node_modules/tailwindcss": { - "version": "4.3.2", - "resolved": "https://registry.npmjs.org/tailwindcss/-/tailwindcss-4.3.2.tgz", - "integrity": "sha512-WtctNNSH8A9jlMIqxzuYumOHU5uGZyRv0Q5svQl+oEPy5w84YpBxdb7MdqyiSPQge5jTJ6zFQLq0PFygdccSBA==", - "license": "MIT" - }, "node_modules/@testing-library/dom": { "version": "10.4.1", "resolved": "https://registry.npmjs.org/@testing-library/dom/-/dom-10.4.1.tgz", @@ -2601,9 +2589,9 @@ } }, "node_modules/@types/react-dom": { - "version": "19.2.3", - "resolved": "https://registry.npmjs.org/@types/react-dom/-/react-dom-19.2.3.tgz", - "integrity": "sha512-jp2L/eY6fn+KgVVQAOqYItbF0VY/YApe5Mz2F0aykSO8gx31bYCZyvSeYxCHKvzHG5eZjc+zyaS5BrBWya2+kQ==", + "version": "19.2.4", + "resolved": "https://registry.npmjs.org/@types/react-dom/-/react-dom-19.2.4.tgz", + "integrity": "sha512-Bsc+QHgp+P/F02XDzNCY9jnZNCUuLki36KT7VKrTXXLdHf+vHMNZnW1rVu5DNW/rCK+fya3DATySbLM4yhtKUw==", "dev": true, "license": "MIT", "peerDependencies": { @@ -2800,9 +2788,9 @@ } }, "node_modules/@vitejs/plugin-react": { - "version": "6.0.4", - "resolved": "https://registry.npmjs.org/@vitejs/plugin-react/-/plugin-react-6.0.4.tgz", - "integrity": "sha512-XcCQz0TBpBgljhj0gMuuDj49i6Ytqh5q1osT/Gp5uAVJUCTWxyskk/l1jwYYiu2xcNHHipdMz40EGfM1VdamVg==", + "version": "6.0.5", + "resolved": "https://registry.npmjs.org/@vitejs/plugin-react/-/plugin-react-6.0.5.tgz", + "integrity": "sha512-BOVzne/NL162sMdResB25mUv+vWMF5NoAjNf09TeGlE7ZpszZWSD3winycicLJw72yeVsoCn/2kOhEuCvEShMA==", "dev": true, "license": "MIT", "dependencies": { @@ -3470,9 +3458,9 @@ "license": "MIT" }, "node_modules/cytoscape": { - "version": "3.33.1", - "resolved": "https://registry.npmjs.org/cytoscape/-/cytoscape-3.33.1.tgz", - "integrity": "sha512-iJc4TwyANnOGR1OmWhsS9ayRS3s+XQ185FmuHObThD+5AeJCakAAbWv8KimMTt08xCCLNgneQwFp+JRJOr9qGQ==", + "version": "3.34.0", + "resolved": "https://registry.npmjs.org/cytoscape/-/cytoscape-3.34.0.tgz", + "integrity": "sha512-62rNSrioXw93uliKFBwjukeQyeWwH2PqDrTac31r2P6464u3AUvTk0xS4LVvT251g7IgkFunrI48ZEZGjywSOg==", "license": "MIT", "engines": { "node": ">=0.10" @@ -4021,9 +4009,9 @@ } }, "node_modules/dayjs": { - "version": "1.11.19", - "resolved": "https://registry.npmjs.org/dayjs/-/dayjs-1.11.19.tgz", - "integrity": "sha512-t5EcLVS6QPBNqM2z8fakk/NKel+Xzshgt8FFKAn+qwlD1pzZWxh0nVCrvFK7ZDb6XucZeF9z8C7CBWTRIVApAw==", + "version": "1.11.21", + "resolved": "https://registry.npmjs.org/dayjs/-/dayjs-1.11.21.tgz", + "integrity": "sha512-98IT+HOahAisibz/yjKbzuOBwYcjJ7BCLPzARyHiyEBmRz4fatF+KPJszEHXsGYjUG234aH/cOjW1wwTbKUZlA==", "license": "MIT" }, "node_modules/debug": { @@ -4121,9 +4109,9 @@ "peer": true }, "node_modules/dompurify": { - "version": "3.4.12", - "resolved": "https://registry.npmjs.org/dompurify/-/dompurify-3.4.12.tgz", - "integrity": "sha512-zQvGet8Z2sWbQhCmfFz/T5QWH2oBmjnqK3qvOjaqaNLrLEF912WamU+ohnTp0TCep/MFVHpdJuCZEdFOdTnEFg==", + "version": "3.4.13", + "resolved": "https://registry.npmjs.org/dompurify/-/dompurify-3.4.13.tgz", + "integrity": "sha512-2vmYIoqjze2d+kakP8S/nS5shfsl587kzwEjcGlTdiksUVgFHnFCsLYDVj/JNqJVOQZGSYBTmuycv0PodwmnMQ==", "license": "(MPL-2.0 OR Apache-2.0)", "optionalDependencies": { "@types/trusted-types": "^2.0.7" @@ -5345,9 +5333,9 @@ } }, "node_modules/katex": { - "version": "0.16.27", - "resolved": "https://registry.npmjs.org/katex/-/katex-0.16.27.tgz", - "integrity": "sha512-aeQoDkuRWSqQN6nSvVCEFvfXdqo1OQiCmmW1kc9xSdjutPv7BGO7pqY9sQRJpMOGrEdfDgF2TfRXe5eUAD2Waw==", + "version": "0.16.47", + "resolved": "https://registry.npmjs.org/katex/-/katex-0.16.47.tgz", + "integrity": "sha512-Eeo8Ys1doU1z+x8AZsPpQu+p/QcZBI5PeOo7QGQdy2x2m0MU/hYagBbGOmXwr5KVbEfVuWv9LpnQWeehogurjg==", "funding": [ "https://opencollective.com/katex", "https://github.com/sponsors/katex" @@ -5375,13 +5363,13 @@ "integrity": "sha512-Ls993zuzfayK269Svk9hzpeGUKob/sIgZzyHYdjQoAdQetRKpOLj+k/QQQ/6Qi0Yz65mlROrfd+Ev+1+7dz9Kw==" }, "node_modules/langchain": { - "version": "1.4.6", - "resolved": "https://registry.npmjs.org/langchain/-/langchain-1.4.6.tgz", - "integrity": "sha512-pwuFmGOyiMezptLVLrpb5jILirvYPGHI5uJCFHL5K5WPxMy2XuPLI5QNMKtoHkdiL6a2dLebqugKw87cneaESw==", + "version": "1.5.4", + "resolved": "https://registry.npmjs.org/langchain/-/langchain-1.5.4.tgz", + "integrity": "sha512-9Rq6Ih77UOy3+7bCbxMJS16MRUJwfxuljU0yW2KOXDgEKWE8cmaZJE6ONEy4HdWGMsbj3qyv3vD5UvV7fvNksg==", "license": "MIT", "dependencies": { - "@langchain/langgraph": "^1.3.4", - "@langchain/langgraph-checkpoint": "^1.0.4", + "@langchain/langgraph": "^1.4.7", + "@langchain/langgraph-checkpoint": "^1.1.3", "langsmith": ">=0.5.0 <1.0.0", "zod": "^3.25.76 || ^4" }, @@ -5389,7 +5377,7 @@ "node": ">=20" }, "peerDependencies": { - "@langchain/core": "^1.2.0" + "@langchain/core": "^1.2.3" } }, "node_modules/langsmith": { @@ -5727,9 +5715,9 @@ } }, "node_modules/lucide-react": { - "version": "1.23.0", - "resolved": "https://registry.npmjs.org/lucide-react/-/lucide-react-1.23.0.tgz", - "integrity": "sha512-38BpJcD0JhFosxHApP/BYsBetLpQFRoTRzEzstM/XCc3jsAG7wqaY1lgVwxiUe3xqYE+lNxo2PkCmYwXWrwwIw==", + "version": "1.28.0", + "resolved": "https://registry.npmjs.org/lucide-react/-/lucide-react-1.28.0.tgz", + "integrity": "sha512-fARAFJULsGuDDydjp6+6blekG/sBIM29TerzLjc9bQUKAcEfrSc4ZQKb25KRz4OMKd87cZTb5dgq0w/T6KufVg==", "license": "ISC", "peerDependencies": { "react": "^16.5.1 || ^17.0.0 || ^18.0.0 || ^19.0.0" @@ -6138,26 +6126,26 @@ } }, "node_modules/mermaid": { - "version": "11.15.0", - "resolved": "https://registry.npmjs.org/mermaid/-/mermaid-11.15.0.tgz", - "integrity": "sha512-pTMbcf3rWdtLiYGpmoTjHEpeY8seiy6sR+9nD7LOs8KfUbHE4lOUAprTRqRAcWSQ6MQpdX+YEsxShtGsINtPtw==", + "version": "11.16.1", + "resolved": "https://registry.npmjs.org/mermaid/-/mermaid-11.16.1.tgz", + "integrity": "sha512-TQsq6u22fAn3rek5VOubrhKPo1g5hwC3FXUN9hiyupTckcYiGuuKGkNQrKYwGJkXUxZdojwRG46gsSCFZMDp4g==", "license": "MIT", "dependencies": { - "@braintree/sanitize-url": "^7.1.1", + "@braintree/sanitize-url": "^7.1.2", "@iconify/utils": "^3.0.2", - "@mermaid-js/parser": "^1.1.1", + "@mermaid-js/parser": "^1.2.0", "@types/d3": "^7.4.3", "@upsetjs/venn.js": "^2.0.0", - "cytoscape": "^3.33.1", + "cytoscape": "^3.33.3", "cytoscape-cose-bilkent": "^4.1.0", "cytoscape-fcose": "^2.2.0", "d3": "^7.9.0", "d3-sankey": "^0.12.3", "dagre-d3-es": "7.0.14", - "dayjs": "^1.11.19", - "dompurify": "^3.3.1", + "dayjs": "^1.11.20", + "dompurify": "^3.3.3", "es-toolkit": "^1.45.1", - "katex": "^0.16.25", + "katex": "^0.16.45", "khroma": "^2.1.0", "marked": "^16.3.0", "roughjs": "^4.6.6", @@ -7405,24 +7393,24 @@ "license": "MIT" }, "node_modules/react": { - "version": "19.2.7", - "resolved": "https://registry.npmjs.org/react/-/react-19.2.7.tgz", - "integrity": "sha512-HNe9WslTbXmFK8o8cmwgAeJFSBvt1bPdHCVKtaaV+WlAN36mpT4hcRpwbf3fY56ar2oIXzsBpOAiIRHAdY0OlQ==", + "version": "19.2.8", + "resolved": "https://registry.npmjs.org/react/-/react-19.2.8.tgz", + "integrity": "sha512-PWaYA1L/q9u2u7xYQi+Y3L3Yfnie7XyLeaJICV1MGD6LprsBxcAqGjYyr0eY3p+QdsA+x/Irkt4Qif8D63+Sbw==", "license": "MIT", "engines": { "node": ">=0.10.0" } }, "node_modules/react-dom": { - "version": "19.2.7", - "resolved": "https://registry.npmjs.org/react-dom/-/react-dom-19.2.7.tgz", - "integrity": "sha512-t0BRVXvbiE/o20Hfw669rLbMCDWtYZLvmJigy2f0MxsXF+71pxhR3xOkspmsO8h3ZlNzyibAmtCa3l4lYKk6gQ==", + "version": "19.2.8", + "resolved": "https://registry.npmjs.org/react-dom/-/react-dom-19.2.8.tgz", + "integrity": "sha512-rVprimfGBG3DR+Tq0IQG2DT5PxKth1WIGDmj5yPmlzr4YBe7uyE+Du4oVqTDXZSHGGGXRtTJEGSSePyQCMBglQ==", "license": "MIT", "dependencies": { "scheduler": "^0.27.0" }, "peerDependencies": { - "react": "^19.2.7" + "react": "^19.2.8" } }, "node_modules/react-i18next": { diff --git a/gitnexus-web/package.json b/gitnexus-web/package.json index 70f6469bb..493546aee 100644 --- a/gitnexus-web/package.json +++ b/gitnexus-web/package.json @@ -21,14 +21,14 @@ "@langchain/anthropic": "^1.5.1", "@langchain/core": "^1.2.3", "@langchain/google-genai": "^2.2.0", - "@langchain/langgraph": "^1.4.8", + "@langchain/langgraph": "^1.4.9", "@langchain/ollama": "^1.3.0", "@langchain/openai": "^1.5.3", "@sigma/edge-curve": "^3.1.0", "@tailwindcss/vite": "^4.3.3", "axios": "^1.18.1", "d3": "^7.9.0", - "dompurify": "^3.4.12", + "dompurify": "^3.4.13", "gitnexus-shared": "file:../gitnexus-shared", "graphology": "^0.26.0", "graphology-indices": "^0.17.0", @@ -38,14 +38,14 @@ "graphology-utils": "^2.3.0", "i18next": "^26.3.6", "i18next-browser-languagedetector": "^8.2.1", - "langchain": "^1.4.6", + "langchain": "^1.5.4", "lru-cache": "^11.5.2", - "lucide-react": "^1.23.0", - "mermaid": "^11.15.0", + "lucide-react": "^1.28.0", + "mermaid": "^11.16.1", "mnemonist": "^0.40.4", "pandemonium": "^2.4.0", "react": "^19.2.5", - "react-dom": "^19.2.7", + "react-dom": "^19.2.8", "react-i18next": "^17.0.11", "react-markdown": "^10.1.0", "react-syntax-highlighter": "^16.1.1", @@ -65,10 +65,10 @@ "@types/dompurify": "^3.2.0", "@types/node": "^26.0.1", "@types/react": "^19.2.14", - "@types/react-dom": "^19.2.3", + "@types/react-dom": "^19.2.4", "@types/react-syntax-highlighter": "^15.5.13", "@vercel/node": "^5.8.23", - "@vitejs/plugin-react": "^6.0.4", + "@vitejs/plugin-react": "^6.0.5", "@vitest/coverage-v8": "^4.1.9", "jsdom": "^29.1.1", "tree-sitter-wasms": "^0.1.13", diff --git a/gitnexus-web/src/components/SettingsPanel.tsx b/gitnexus-web/src/components/SettingsPanel.tsx index 0c3a22aec..9b32ffd92 100644 --- a/gitnexus-web/src/components/SettingsPanel.tsx +++ b/gitnexus-web/src/components/SettingsPanel.tsx @@ -21,7 +21,13 @@ import { fetchOpenRouterModels, } from '../core/llm/settings-service'; import { getAuthToken, setAuthToken } from '../services/backend-client'; -import type { LLMSettings, LLMProvider } from '../core/llm/types'; +import type { LLMSettings, LLMProvider, MiniMaxThinkingMode } from '../core/llm/types'; +import { + getMiniMaxModelCapabilities, + MINIMAX_ANTHROPIC_BASE_URLS, + MINIMAX_DOCS_ROOTS, + MINIMAX_MODEL_IDS, +} from '../core/llm/types'; import { DEFAULT_OLLAMA_BASE_URL } from '../config/ui-constants'; import { ProviderConfigCard } from './settings/ProviderConfigCard'; import { SecretInput } from './settings/SecretInput'; @@ -341,6 +347,20 @@ export const SettingsPanel = ({ if (!isOpen) return null; + const miniMaxModel = settings.minimax?.model ?? MINIMAX_MODEL_IDS[0]; + const miniMaxCapabilities = getMiniMaxModelCapabilities(miniMaxModel); + const configuredMiniMaxThinkingMode = settings.minimax?.thinkingMode; + const miniMaxThinkingMode = + configuredMiniMaxThinkingMode && + miniMaxCapabilities?.thinkingModes.includes(configuredMiniMaxThinkingMode) + ? configuredMiniMaxThinkingMode + : (miniMaxCapabilities?.thinkingModes[0] ?? configuredMiniMaxThinkingMode ?? 'adaptive'); + const miniMaxBaseUrl = settings.minimax?.baseUrl ?? MINIMAX_ANTHROPIC_BASE_URLS.global_en; + const miniMaxDocsRoot = + miniMaxBaseUrl === MINIMAX_ANTHROPIC_BASE_URLS.cn_zh + ? MINIMAX_DOCS_ROOTS.cn_zh + : MINIMAX_DOCS_ROOTS.global_en; + const providers: LLMProvider[] = [ 'openai', 'gemini', @@ -864,7 +884,7 @@ export const SettingsPanel = ({ value: settings.minimax?.apiKey ?? '', placeholder: t('settings:providers.minimax.apiKeyPlaceholder'), helperText: t('settings:providers.minimax.helperText'), - helperLink: 'https://platform.minimax.io', + helperLink: miniMaxDocsRoot, helperLinkLabel: t('settings:providers.minimax.helperLinkLabel'), isVisible: !!showApiKey['minimax'], onChange: (value) => @@ -875,16 +895,79 @@ export const SettingsPanel = ({ onToggleVisibility: () => toggleApiKeyVisibility('minimax'), }} model={{ - value: settings.minimax?.model ?? 'MiniMax-M2.5', + value: miniMaxModel, placeholder: t('settings:providers.minimax.modelPlaceholder'), onChange: (value) => setSettings((prev) => ({ ...prev, - minimax: { ...prev.minimax!, model: value }, + minimax: { + ...prev.minimax!, + model: value, + thinkingMode: + getMiniMaxModelCapabilities(value)?.thinkingModes[0] ?? + prev.minimax?.thinkingMode, + }, })), helperText: t('settings:providers.minimax.helperModel'), }} - /> + > +
+ + +
+ +
+ + + {miniMaxCapabilities && ( +

+ {t('settings:providers.minimax.capabilities', { + contextWindow: miniMaxCapabilities.contextWindow.toLocaleString(), + modalities: miniMaxCapabilities.inputModalities.join(', '), + })} +

+ )} +
+ )} {/* DeepSeek Settings */} diff --git a/gitnexus-web/src/core/llm/agent.ts b/gitnexus-web/src/core/llm/agent.ts index c10748fd0..555cf0d10 100644 --- a/gitnexus-web/src/core/llm/agent.ts +++ b/gitnexus-web/src/core/llm/agent.ts @@ -20,6 +20,7 @@ import { ChatOllama } from '@langchain/ollama'; import type { BaseChatModel } from '@langchain/core/language_models/chat_models'; import { createGraphRAGTools, type GraphRAGBackend } from './tools'; import type { + AgentUserContent, ProviderConfig, OpenAIConfig, AzureOpenAIConfig, @@ -32,7 +33,9 @@ import type { DeepSeekConfig, AgentStreamChunk, AgentHistoryMessage, + MiniMaxThinkingMode, } from './types'; +import { getMiniMaxModelCapabilities, MINIMAX_ANTHROPIC_BASE_URLS } from './types'; import { type CodebaseContext, buildDynamicSystemPrompt, @@ -275,14 +278,28 @@ export const createChatModel = (config: ProviderConfig): BaseChatModel => { throw new Error('MiniMax API key is required but was not provided'); } + const capabilities = getMiniMaxModelCapabilities(minimaxConfig.model); + const requestedThinkingMode = minimaxConfig.thinkingMode; + const thinkingMode: MiniMaxThinkingMode | undefined = + requestedThinkingMode && capabilities?.thinkingModes.includes(requestedThinkingMode) + ? requestedThinkingMode + : (capabilities?.thinkingModes[0] ?? requestedThinkingMode); + const thinking = + thinkingMode && thinkingMode !== 'always_on' ? { type: thinkingMode } : undefined; + const temperature = + thinkingMode === 'adaptive' || thinkingMode === 'always_on' + ? undefined + : (minimaxConfig.temperature ?? 0.1); + return new ChatAnthropic({ anthropicApiKey: minimaxConfig.apiKey, model: minimaxConfig.model, - temperature: minimaxConfig.temperature ?? 0.1, + ...(temperature !== undefined ? { temperature } : {}), maxTokens: minimaxConfig.maxTokens ?? 8192, streaming: true, + ...(thinking ? { thinking } : {}), clientOptions: { - baseURL: 'https://api.minimax.io/anthropic', + baseURL: minimaxConfig.baseUrl ?? MINIMAX_ANTHROPIC_BASE_URLS.global_en, }, }); } @@ -393,7 +410,7 @@ export const createGraphRAGAgent = ( /** * Message type for agent conversation */ -export type AgentMessage = { role: 'user'; content: string } | AgentHistoryMessage; +export type AgentMessage = { role: 'user'; content: AgentUserContent } | AgentHistoryMessage; export interface AgentRuntimeOptions { /** Capture assistant/tool messages for providers that require exact transcript replay. */ @@ -412,7 +429,9 @@ const isAbortError = (error: unknown, signal?: AbortSignal): boolean => { export const buildLangChainMessages = (messages: AgentMessage[]): BaseMessage[] => messages.map((message) => { if (message.role === 'user') { - return new HumanMessage(message.content); + return typeof message.content === 'string' + ? new HumanMessage(message.content) + : new HumanMessage({ content: message.content as any }); } if (message.role === 'tool') { return new ToolMessage({ @@ -542,6 +561,7 @@ export async function* streamAgentResponse( // Handle content that can be string or array of content blocks let content: string = ''; + let thinkingContent: string = ''; if (typeof rawContent === 'string') { content = rawContent; } else if (Array.isArray(rawContent)) { @@ -550,6 +570,14 @@ export async function* streamAgentResponse( .filter((block: any) => block.type === 'text' || typeof block === 'string') .map((block: any) => (typeof block === 'string' ? block : block.text || '')) .join(''); + thinkingContent = rawContent + .filter((block: any) => block?.type === 'thinking') + .map((block: any) => block.thinking || '') + .join(''); + } + + if (thinkingContent) { + yield { type: 'reasoning', reasoning: thinkingContent }; } // If chunk has content, stream it diff --git a/gitnexus-web/src/core/llm/settings-service.ts b/gitnexus-web/src/core/llm/settings-service.ts index 79a7a4309..fb2591172 100644 --- a/gitnexus-web/src/core/llm/settings-service.ts +++ b/gitnexus-web/src/core/llm/settings-service.ts @@ -19,12 +19,32 @@ import { GLMConfig, DeepSeekConfig, ProviderConfig, + MINIMAX_MODEL_IDS, } from './types'; import { DEFAULT_OPENROUTER_BASE_URL, DEFAULT_OLLAMA_BASE_URL } from '../../config/ui-constants'; import { resilientFetch } from 'gitnexus-shared'; const STORAGE_KEY = 'gitnexus-llm-settings'; +const mergeMiniMaxSettings = ( + stored?: LLMSettings['minimax'], +): NonNullable => { + const merged = { + ...DEFAULT_LLM_SETTINGS.minimax, + ...stored, + }; + + if (!(MINIMAX_MODEL_IDS as readonly string[]).includes(merged.model ?? '')) { + return { + ...merged, + model: DEFAULT_LLM_SETTINGS.minimax?.model, + thinkingMode: DEFAULT_LLM_SETTINGS.minimax?.thinkingMode, + }; + } + + return merged; +}; + const mergeWithDefaults = (parsed?: Partial | null): LLMSettings => ({ ...DEFAULT_LLM_SETTINGS, ...parsed, @@ -52,10 +72,7 @@ const mergeWithDefaults = (parsed?: Partial | null): LLMSettings => ...DEFAULT_LLM_SETTINGS.openrouter, ...parsed?.openrouter, }, - minimax: { - ...DEFAULT_LLM_SETTINGS.minimax, - ...parsed?.minimax, - }, + minimax: mergeMiniMaxSettings(parsed?.minimax), glm: { ...DEFAULT_LLM_SETTINGS.glm, ...parsed?.glm, @@ -437,7 +454,7 @@ export const getAvailableModels = (provider: LLMProvider): string[] => { case 'ollama': return ['llama3.2', 'llama3.1', 'mistral', 'codellama', 'deepseek-coder']; case 'minimax': - return ['MiniMax-M2.5', 'MiniMax-M2.5-highspeed']; + return [...MINIMAX_MODEL_IDS]; case 'glm': return ['GLM-5', 'GLM-5-Turbo', 'GLM-4.7', 'GLM-4.5']; case 'deepseek': diff --git a/gitnexus-web/src/core/llm/tools.ts b/gitnexus-web/src/core/llm/tools.ts index 9f6f34291..a702e7db8 100644 --- a/gitnexus-web/src/core/llm/tools.ts +++ b/gitnexus-web/src/core/llm/tools.ts @@ -1233,7 +1233,20 @@ MATCH (n:Function {id: emb.nodeId}) RETURN n`, } } - return `No ${direction} dependencies found for "${target}" (types: ${activeRelTypes.join(', ')}). This code appears to be ${direction === 'upstream' ? 'unused (not called by anything)' : 'self-contained (no outgoing dependencies)'}.${multipleMatchWarning}`; + // An empty UPSTREAM walk is not evidence of disuse — it is the absence + // of evidence. The symbol may be reached only through a reference class + // the index does not record (a property access on a plain object, a + // dynamic dispatch, a call from a language whose resolver is weaker + // here). The Node/MCP path reports `risk: UNKNOWN` with a `riskNote` + // for exactly this case; this surface answers in prose rather than an + // enum, so it carries the same MEANING rather than the same field — + // saying "appears to be unused" here is the identical false certainty. + // + // Downstream keeps its wording: no outgoing dependencies really does + // describe the symbol itself, not a claim about the rest of the repo. + return direction === 'upstream' + ? `No ${direction} dependencies found for "${target}" (types: ${activeRelTypes.join(', ')}). This does NOT establish the symbol is unused — an empty caller set can also mean the callers are not resolvable by the index (plain-object property access, dynamic dispatch, cross-language calls). Confirm with a text search before treating it as dead code.${multipleMatchWarning}` + : `No ${direction} dependencies found for "${target}" (types: ${activeRelTypes.join(', ')}). This code appears to be self-contained (no outgoing dependencies).${multipleMatchWarning}`; } const depth1 = byDepth.get(1) || []; diff --git a/gitnexus-web/src/core/llm/types.ts b/gitnexus-web/src/core/llm/types.ts index b7727da10..c5198bd16 100644 --- a/gitnexus-web/src/core/llm/types.ts +++ b/gitnexus-web/src/core/llm/types.ts @@ -20,6 +20,71 @@ export type LLMProvider = | 'glm' | 'deepseek'; +export const MINIMAX_ANTHROPIC_BASE_URLS = { + global_en: 'https://api.minimax.io/anthropic', + cn_zh: 'https://api.minimaxi.com/anthropic', +} as const; + +export const MINIMAX_DOCS_ROOTS = { + global_en: 'https://platform.minimax.io/docs', + cn_zh: 'https://platform.minimaxi.com/docs', +} as const; + +export const MINIMAX_MODEL_IDS = ['MiniMax-M3', 'MiniMax-M2.7'] as const; + +export type MiniMaxModelId = (typeof MINIMAX_MODEL_IDS)[number]; +export type MiniMaxThinkingMode = 'adaptive' | 'disabled' | 'always_on'; +export type MiniMaxInputModality = 'text' | 'image' | 'video'; + +export interface MiniMaxModelCapabilities { + contextWindow: number; + inputModalities: readonly MiniMaxInputModality[]; + thinkingModes: readonly MiniMaxThinkingMode[]; +} + +export const MINIMAX_MODEL_CAPABILITIES: Record = { + 'MiniMax-M3': { + contextWindow: 1_000_000, + inputModalities: ['text', 'image', 'video'], + thinkingModes: ['adaptive', 'disabled'], + }, + 'MiniMax-M2.7': { + contextWindow: 204_800, + inputModalities: ['text'], + thinkingModes: ['always_on'], + }, +}; + +export const getMiniMaxModelCapabilities = (model: string): MiniMaxModelCapabilities | undefined => + MINIMAX_MODEL_CAPABILITIES[model as MiniMaxModelId]; + +export type MiniMaxMediaDetail = 'low' | 'default' | 'high'; + +export type MiniMaxMediaSource = + | { + type: 'url'; + url: string; + detail?: MiniMaxMediaDetail; + fps?: number; + max_long_side_pixel?: number; + } + | { + type: 'base64'; + media_type: string; + data: string; + detail?: MiniMaxMediaDetail; + fps?: number; + max_long_side_pixel?: number; + }; + +export type AgentUserContent = + | string + | Array< + | { type: 'text'; text: string } + | { type: 'image'; source: MiniMaxMediaSource } + | { type: 'video'; source: MiniMaxMediaSource } + >; + /** * Base configuration shared by all providers */ @@ -94,7 +159,9 @@ export interface OpenRouterConfig extends BaseProviderConfig { export interface MiniMaxConfig extends BaseProviderConfig { provider: 'minimax'; apiKey: string; - model: string; // e.g., 'MiniMax-M2.5', 'MiniMax-M2.5-highspeed' + model: string; + baseUrl?: string; + thinkingMode?: MiniMaxThinkingMode; } /** @@ -200,7 +267,9 @@ export const DEFAULT_LLM_SETTINGS: LLMSettings = { }, minimax: { apiKey: '', - model: 'MiniMax-M2.5', + model: MINIMAX_MODEL_IDS[0], + baseUrl: MINIMAX_ANTHROPIC_BASE_URLS.global_en, + thinkingMode: 'adaptive', temperature: 0.1, }, glm: { diff --git a/gitnexus-web/src/locales/en/settings.json b/gitnexus-web/src/locales/en/settings.json index cf9746c72..91c6b2a68 100644 --- a/gitnexus-web/src/locales/en/settings.json +++ b/gitnexus-web/src/locales/en/settings.json @@ -76,8 +76,20 @@ "apiKeyPlaceholder": "Enter your MiniMax API key", "helperText": "Get your API key from", "helperLinkLabel": "MiniMax Platform", - "modelPlaceholder": "e.g., MiniMax-M2.5, MiniMax-M2.5-highspeed", - "helperModel": "Available: MiniMax-M2.5 (default), MiniMax-M2.5-highspeed (faster)" + "modelPlaceholder": "e.g., MiniMax-M3 or MiniMax-M2.7", + "helperModel": "Available: MiniMax-M3 (default) and MiniMax-M2.7", + "endpoint": "Regional endpoint", + "endpoints": { + "global": "Global (api.minimax.io)", + "china": "China (api.minimaxi.com)" + }, + "thinking": "Thinking mode", + "thinkingModes": { + "adaptive": "Adaptive", + "disabled": "Disabled", + "always_on": "Always on" + }, + "capabilities": "{{contextWindow}} token context | Inputs: {{modalities}}" }, "glm": { "apiKeyPlaceholder": "Enter your Z.AI API key" diff --git a/gitnexus-web/src/locales/zh-CN/settings.json b/gitnexus-web/src/locales/zh-CN/settings.json index 0efe220d4..4bc4fb05b 100644 --- a/gitnexus-web/src/locales/zh-CN/settings.json +++ b/gitnexus-web/src/locales/zh-CN/settings.json @@ -76,8 +76,20 @@ "apiKeyPlaceholder": "输入 MiniMax API Key", "helperText": "从这里获取 API Key:", "helperLinkLabel": "MiniMax Platform", - "modelPlaceholder": "例如:MiniMax-M2.5、MiniMax-M2.5-highspeed", - "helperModel": "可用:MiniMax-M2.5(默认)、MiniMax-M2.5-highspeed(更快)" + "modelPlaceholder": "例如:MiniMax-M3 或 MiniMax-M2.7", + "helperModel": "可用:MiniMax-M3(默认)和 MiniMax-M2.7", + "endpoint": "区域端点", + "endpoints": { + "global": "全球(api.minimax.io)", + "china": "中国(api.minimaxi.com)" + }, + "thinking": "思考模式", + "thinkingModes": { + "adaptive": "自适应", + "disabled": "关闭", + "always_on": "始终开启" + }, + "capabilities": "{{contextWindow}} token 上下文 | 输入:{{modalities}}" }, "glm": { "apiKeyPlaceholder": "输入 Z.AI API Key" diff --git a/gitnexus-web/test/unit/agent-abort.test.ts b/gitnexus-web/test/unit/agent-abort.test.ts index 2a8475e34..92af12a0e 100644 --- a/gitnexus-web/test/unit/agent-abort.test.ts +++ b/gitnexus-web/test/unit/agent-abort.test.ts @@ -95,3 +95,34 @@ describe('streamAgentResponse abort', () => { expect(chunks).toEqual([{ type: 'error', error: 'Cannot abort the current transaction' }]); }); }); + +describe('streamAgentResponse content blocks', () => { + const userMessage: AgentMessage[] = [{ role: 'user', content: 'hello' }]; + + it('emits thinking blocks as reasoning', async () => { + const agent = { + stream: async function* () { + yield [ + 'messages', + [ + { + _getType: () => 'ai', + content: [{ type: 'thinking', thinking: 'Reviewing the repository context.' }], + tool_calls: [], + }, + ], + ]; + }, + }; + + const chunks = []; + for await (const chunk of streamAgentResponse(agent as any, userMessage)) { + chunks.push(chunk); + } + + expect(chunks).toEqual([ + { type: 'reasoning', reasoning: 'Reviewing the repository context.' }, + { type: 'done', historyMessages: undefined }, + ]); + }); +}); diff --git a/gitnexus-web/test/unit/agent-history.test.ts b/gitnexus-web/test/unit/agent-history.test.ts index 756534b25..f672e8a9d 100644 --- a/gitnexus-web/test/unit/agent-history.test.ts +++ b/gitnexus-web/test/unit/agent-history.test.ts @@ -10,6 +10,7 @@ import { DeepSeekChatOpenAI, DeepSeekChatOpenAICompletions, } from '../../src/core/llm/deepseek-chat-model'; +import { MINIMAX_ANTHROPIC_BASE_URLS, MINIMAX_MODEL_IDS } from '../../src/core/llm/types'; describe('buildLangChainMessages', () => { it('reconstructs assistant tool-call turns for replay', () => { @@ -50,6 +51,24 @@ describe('buildLangChainMessages', () => { ]); expect((langChainMessages[2] as any).tool_call_id).toBe('call_weather'); }); + + it('preserves MiniMax image and video content blocks', () => { + const content = [ + { type: 'text' as const, text: 'Compare these inputs.' }, + { + type: 'image' as const, + source: { type: 'url' as const, url: 'https://example.com/image.png' }, + }, + { + type: 'video' as const, + source: { type: 'url' as const, url: 'https://example.com/video.mp4', fps: 1 }, + }, + ]; + + const [message] = buildLangChainMessages([{ role: 'user', content }]); + + expect((message as any).content).toEqual(content); + }); }); describe('serializeAgentHistoryMessages', () => { @@ -206,6 +225,48 @@ it('drops reasoningContent from serialized assistant messages without tool calls }); describe('createChatModel', () => { + it('configures MiniMax-M3 adaptive thinking on the China endpoint', () => { + const model = createChatModel({ + provider: 'minimax', + apiKey: 'minimax-test-key', + model: MINIMAX_MODEL_IDS[0], + baseUrl: MINIMAX_ANTHROPIC_BASE_URLS.cn_zh, + thinkingMode: 'adaptive', + temperature: 0.1, + } as any) as any; + + expect(model.model).toBe(MINIMAX_MODEL_IDS[0]); + expect(model.clientOptions.baseURL).toBe(MINIMAX_ANTHROPIC_BASE_URLS.cn_zh); + expect(model.thinking).toEqual({ type: 'adaptive' }); + expect(model.temperature).toBeUndefined(); + }); + + it('supports disabled thinking for MiniMax-M3', () => { + const model = createChatModel({ + provider: 'minimax', + apiKey: 'minimax-test-key', + model: MINIMAX_MODEL_IDS[0], + thinkingMode: 'disabled', + temperature: 0.1, + } as any) as any; + + expect(model.thinking).toEqual({ type: 'disabled' }); + expect(model.temperature).toBe(0.1); + }); + + it('keeps MiniMax-M2.7 thinking always on', () => { + const model = createChatModel({ + provider: 'minimax', + apiKey: 'minimax-test-key', + model: MINIMAX_MODEL_IDS[1], + thinkingMode: 'disabled', + temperature: 0.1, + } as any) as any; + + expect(model.invocationParams({}).thinking).toBeUndefined(); + expect(model.temperature).toBeUndefined(); + }); + it('keeps DeepSeek model subclasses on withConfig clones used for tool binding', () => { const model = createChatModel({ provider: 'deepseek', diff --git a/gitnexus-web/test/unit/settings-service.test.ts b/gitnexus-web/test/unit/settings-service.test.ts index a9ded356f..b0762604f 100644 --- a/gitnexus-web/test/unit/settings-service.test.ts +++ b/gitnexus-web/test/unit/settings-service.test.ts @@ -10,6 +10,12 @@ import { getAvailableModels, getProviderCapabilities, } from '../../src/core/llm/settings-service'; +import { + getMiniMaxModelCapabilities, + MINIMAX_ANTHROPIC_BASE_URLS, + MINIMAX_MODEL_IDS, +} from '../../src/core/llm/types'; +import { createChatModel } from '../../src/core/llm/agent'; describe('loadSettings', () => { it('returns defaults when nothing is stored', () => { @@ -17,6 +23,11 @@ describe('loadSettings', () => { expect(settings.activeProvider).toBeDefined(); expect(settings.openai).toBeDefined(); expect(settings.ollama).toBeDefined(); + expect(settings.minimax).toMatchObject({ + model: MINIMAX_MODEL_IDS[0], + baseUrl: MINIMAX_ANTHROPIC_BASE_URLS.global_en, + thinkingMode: 'adaptive', + }); }); it('merges stored values with defaults', () => { @@ -35,6 +46,30 @@ describe('loadSettings', () => { expect(settings.openai).toBeDefined(); }); + it('migrates unsupported legacy MiniMax models to the current default', () => { + sessionStorage.setItem( + 'gitnexus-llm-settings', + JSON.stringify({ + activeProvider: 'minimax', + minimax: { + apiKey: 'minimax-test-key', + model: 'MiniMax-M2.5', + temperature: 0.1, + }, + }), + ); + + const settings = loadSettings(); + expect(settings.minimax).toMatchObject({ + model: MINIMAX_MODEL_IDS[0], + thinkingMode: 'adaptive', + }); + + const model = createChatModel(getActiveProviderConfig()!) as any; + expect(model.model).toBe(MINIMAX_MODEL_IDS[0]); + expect(model.thinking).toEqual({ type: 'adaptive' }); + }); + it('returns defaults on corrupted JSON', () => { sessionStorage.setItem('gitnexus-llm-settings', 'not-json{{{'); const settings = loadSettings(); @@ -116,6 +151,26 @@ describe('getActiveProviderConfig', () => { expect(config!.provider).toBe('deepseek'); }); + it('returns the regional endpoint and thinking mode for MiniMax', () => { + const settings = loadSettings(); + settings.activeProvider = 'minimax'; + settings.minimax = { + ...settings.minimax, + apiKey: 'minimax-test-key', + model: MINIMAX_MODEL_IDS[0], + baseUrl: MINIMAX_ANTHROPIC_BASE_URLS.cn_zh, + thinkingMode: 'disabled', + }; + saveSettings(settings); + + expect(getActiveProviderConfig()).toMatchObject({ + provider: 'minimax', + model: MINIMAX_MODEL_IDS[0], + baseUrl: MINIMAX_ANTHROPIC_BASE_URLS.cn_zh, + thinkingMode: 'disabled', + }); + }); + it('returns null for openrouter with empty API key', () => { const settings = loadSettings(); settings.activeProvider = 'openrouter'; @@ -161,6 +216,20 @@ describe('getAvailableModels', () => { expect(getAvailableModels('ollama').length).toBeGreaterThan(0); expect(getAvailableModels('anthropic')).toContain('claude-sonnet-4-20250514'); expect(getAvailableModels('deepseek')).toContain('deepseek-v4-flash'); + expect(getAvailableModels('minimax')).toEqual([...MINIMAX_MODEL_IDS]); + }); + + it('describes MiniMax model input and thinking capabilities', () => { + expect(getMiniMaxModelCapabilities(MINIMAX_MODEL_IDS[0])).toEqual({ + contextWindow: 1_000_000, + inputModalities: ['text', 'image', 'video'], + thinkingModes: ['adaptive', 'disabled'], + }); + expect(getMiniMaxModelCapabilities(MINIMAX_MODEL_IDS[1])).toEqual({ + contextWindow: 204_800, + inputModalities: ['text'], + thinkingModes: ['always_on'], + }); }); it('returns empty array for unknown provider', () => { diff --git a/gitnexus/bench/emit-persistence/baselines.json b/gitnexus/bench/emit-persistence/baselines.json index 402c9b06c..1d295bd19 100644 --- a/gitnexus/bench/emit-persistence/baselines.json +++ b/gitnexus/bench/emit-persistence/baselines.json @@ -1,6 +1,7 @@ { - "fingerprint": "69e9182ae205183ade24c3d8ad5d7292aea677144b1cbe443dd631bc25b0cafe", + "fingerprint": "4ee15e742a9839671a900df4f57c1c91196c64256c8cab2ac445bec605a092d5", "scaling_budget": 1.8, "max_ms_large": 1000, - "_note": "fingerprint = sha256 over per-file digests (filename + sha256(file bytes)), entry list sorted — binds each emitted line to its file so a row routed to the WRONG pair file changes the hash, AND catches within-file row reordering (file bytes hashed as-written). Byte-identity gate for #2203 U2/U3. NOTE: a future change that legitimately reorders emit (without changing the node/edge SET) will trip --check; regenerate then. scaling_budget bounds (t_large/t_small)/(LARGE/SMALL): observed ~0.95-1.05 (linear); 1.8 tolerates disk-I/O timing noise on CI while still catching an O(n^2) re-regression (~4x). max_ms_large=1000ms is a coarse absolute backstop (observed ~200ms) that catches a gross uniform slowdown the ratio gate misses; generous so CI host noise won't flake it. Regenerate via `node --import tsx bench/emit-persistence/measure.mjs`." + "_rebaselined_2856_property_is_detail": "Third and last of the bench guards this branch left red. The Property node table gained an `isDetail` BOOLEAN column (see PROPERTY_SCHEMA in src/core/lbug/schema.ts), so `streamAllCSVsToDisk` writes one more header field and one more cell per Property row — csv-generator.ts `propertyHeader` and the `node.label === 'Property'` tail. Verified to be header-only drift rather than a change in what is emitted: dumping every CSV this bench produces on `origin/main` and on this branch and diffing per-file (filename, byte length, sha256) shows the file SET is identical at 35 CSVs on both sides, 34 of the 35 are byte-identical, and the sole difference is `property.csv` growing 68 -> 77 bytes, `id,name,filePath,startLine,endLine,content,description,declaredType` -> `...,declaredType,isDetail`. The synthetic graph has no Property nodes, so no ROW moved at all. That is the check that matters here: a row routed to the wrong pair file, or a within-file reordering, is what this fingerprint exists to catch, and neither happened. Prior 69e9182ae205183ade24c3d8ad5d7292aea677144b1cbe443dd631bc25b0cafe -> 4ee15e742a9839671a900df4f57c1c91196c64256c8cab2ac445bec605a092d5. Both timing gates passed unchanged while this was red (scaling_ratio 0.783 vs budget 1.8, elapsed_ms_large 229ms vs the 1000ms backstop), so no throughput claim is being rebaselined away.", + "_note": "fingerprint = sha256 over per-file digests (filename + sha256(file bytes)), entry list sorted — binds each emitted line to its file so a row routed to the WRONG pair file changes the hash, AND catches within-file row reordering (file bytes hashed as-written). Byte-identity gate for #2203 U2/U3. NOTE: a future change that legitimately reorders emit (without changing the node/edge SET) will trip --check; regenerate then, and record WHY in a `_rebaselined_` key alongside — bench/scope-capture/baselines.json sets that convention and it is what makes a regenerated hash reviewable. scaling_budget bounds (t_large/t_small)/(LARGE/SMALL): observed ~0.95-1.05 (linear); 1.8 tolerates disk-I/O timing noise on CI while still catching an O(n^2) re-regression (~4x). max_ms_large=1000ms is a coarse absolute backstop (observed ~200ms) that catches a gross uniform slowdown the ratio gate misses; generous so CI host noise won't flake it. Regenerate via `node --import tsx bench/emit-persistence/measure.mjs`." } diff --git a/gitnexus/bench/finalize-reexport/measure.mjs b/gitnexus/bench/finalize-reexport/measure.mjs new file mode 100644 index 000000000..e0ad67109 --- /dev/null +++ b/gitnexus/bench/finalize-reexport/measure.mjs @@ -0,0 +1,246 @@ +/** + * Build-free scaling bench for `buildReexportClosures`, the re-export closure + * pass inside `finalize`. + * + * WHY THIS EXISTS. Until #2864 the closure sub-graph admitted only `reexport` + * and `wildcard` drafts, so its input was TypeScript barrel files: a handful + * of edges, shallow chains. #2864 admits `named`/`alias` drafts flagged + * `reexportsName`, which for Python is every module-level `from m import x` — + * measured ~20x more edges on the CPython stdlib, and cyclic SCCs where there + * were none. The pass went from "rarely runs" to "runs over the whole named + * import graph", and nothing measured it. + * + * The specific regression this guards is a QUADRATIC, and it has already + * happened once. `populateFileClosure` copies the inherited `via` array at + * every hop, so an unbounded chain is Theta(depth^2) in time AND retained + * memory. `MAX_REEXPORT_DEPTH = 100` bounded it until commit `fc919ad6` + * removed it — a correct call for shallow TS barrels, invisible for years, + * and wrong the moment the input class changed. `MAX_VIA_LENGTH` restores the + * bound; this bench is what notices if it goes away again. Measured at + * depth 400: 67 ms / 145 MB uncapped vs 25 ms / 40 MB capped. + * + * TWO ARMS, deliberately not one, and only one of them is a timing arm: + * + * - `max_via_len` — EXACT and deterministic. Builds a chain far deeper than + * the cap and asserts the longest emitted `transitiveVia` is exactly + * `MAX_VIA_LENGTH`. Removing the cap is directly observable as a longer + * array, so this catches it with zero flake. + * + * This started life as a `depth_ratio` timing arm and that was a BAD GATE. + * Sampled five times capped it scored 2.71-3.52, and three times uncapped + * it scored 5.87-7.65 — the ranges nearly touch, and one uncapped run came + * in UNDER the budget. A gate that passes a third of the time on a broken + * build is worse than no gate, because it is read as evidence. The + * quadratic is real, but at these depths the pass's linear work dilutes it + * enough that wall-clock cannot separate the two cleanly. The structural + * assertion can, so it is the one that gates. + * + * - `width_ms` — an absolute ceiling on a wide, shallow, realistic package + * corpus (the shape a real Python repo actually has). Structural checks + * cannot see a constant factor: reintroducing a per-lookup linear scan of + * a target's `localDefs` leaves every array length untouched while making + * every real analyze slower. This arm IS timing-sensitive — re-run on an + * idle machine before investigating. Its budget is deliberately loose; it + * is here to catch a doubling, not to police drift. + * + * Both arms feed `finalize` through INDEXED hooks. The obvious mistake is to + * reuse the unit tests' `defaultHooks`, whose `resolveImportTarget` does + * `files.some(...)` per import — that is O(imports x files) in the FIXTURE, + * and it swamps the pass under test so completely that removing the cap + * measures as no change at all. + * + * Usage: + * node --import tsx bench/finalize-reexport/measure.mjs # report + * node --import tsx bench/finalize-reexport/measure.mjs --check # CI gate + */ +import { performance } from 'node:perf_hooks'; +import { finalize } from 'gitnexus-shared'; + +/** Must equal `MAX_VIA_LENGTH` in `gitnexus-shared`'s finalize-algorithm.ts. */ +const EXPECTED_MAX_VIA = 32; +// Generous absolute ceiling — this arm exists to catch a restored O(n^2) +// scan (which more than doubles it), not to police small drift. +const WIDTH_MS_BUDGET = 1200; + +const PROBE_DEPTH = 400; + +const deriveSimple = (d) => { + const q = d.qualifiedName; + if (q === undefined || q.length === 0) return null; + const dot = q.lastIndexOf('.'); + return dot === -1 ? q : q.slice(dot + 1); +}; + +function hooksFor(files) { + const byPath = new Map(files.map((f) => [f.filePath, f])); + const byScope = new Map(files.map((f) => [f.moduleScope, f])); + return { + resolveImportTarget: (raw) => (raw !== null && byPath.has(raw) ? raw : null), + expandsWildcardTo: (scope) => { + const t = byScope.get(scope); + return t === undefined ? [] : t.localDefs.map(deriveSimple).filter((n) => n !== null); + }, + mergeBindings: (existing, incoming) => [...existing, ...incoming], + }; +} + +const mkFile = (filePath, localDefs, parsedImports) => ({ + filePath, + moduleScope: `scope:${filePath}#1:0-9999:0:Module`, + localDefs, + parsedImports, +}); +const mkDef = (qn) => ({ nodeId: `def:${qn}`, filePath: 'x', type: 'Function', qualifiedName: qn }); +const reexporting = (name, targetRaw) => ({ + kind: 'named', + localName: name, + importedName: name, + targetRaw, + reexportsName: true, +}); + +/** A `__init__.py` chain N deep, each hop republishing the same names. */ +function chainCorpus(depth, names = 20) { + const files = [ + mkFile( + 'leaf.py', + Array.from({ length: names }, (_, j) => mkDef(`leaf.fn${j}`)), + [], + ), + ]; + let prev = 'leaf.py'; + for (let d = 0; d < depth; d++) { + const p = `hop${d}.py`; + files.push( + mkFile( + p, + [], + Array.from({ length: names }, (_, j) => reexporting(`fn${j}`, prev)), + ), + ); + prev = p; + } + files.push( + mkFile( + 'app.py', + [], + Array.from({ length: names }, (_, j) => ({ + kind: 'named', + localName: `fn${j}`, + importedName: `fn${j}`, + targetRaw: prev, + })), + ), + ); + return files; +} + +/** Wide and shallow: the layout a real Python repo has. */ +function packageCorpus({ leaves, defsPerLeaf, pkgSize, consumers, importsPerConsumer }) { + const files = []; + const leafPaths = []; + for (let i = 0; i < leaves; i++) { + const p = `pkg${Math.floor(i / pkgSize)}/mod${i}.py`; + leafPaths.push(p); + files.push( + mkFile( + p, + Array.from({ length: defsPerLeaf }, (_, j) => mkDef(`mod${i}.fn${j}`)), + [], + ), + ); + } + const initPaths = []; + for (let g = 0; g < Math.ceil(leaves / pkgSize); g++) { + const p = `pkg${g}/__init__.py`; + initPaths.push(p); + const imports = []; + for (let i = g * pkgSize; i < Math.min((g + 1) * pkgSize, leaves); i++) { + for (let j = 0; j < defsPerLeaf; j++) imports.push(reexporting(`fn${j}_${i}`, leafPaths[i])); + } + files.push(mkFile(p, [], imports)); + } + for (let c = 0; c < consumers; c++) { + const imports = []; + for (let k = 0; k < importsPerConsumer; k++) { + const g = (c * 7 + k) % initPaths.length; + imports.push({ + kind: 'named', + localName: `fn0_${g * pkgSize}`, + importedName: `fn0_${g * pkgSize}`, + targetRaw: initPaths[g], + }); + } + files.push(mkFile(`app/consumer${c}.py`, [], imports)); + } + return files; +} + +function timeMedian(files, reps = 5) { + const hooks = hooksFor(files); + finalize({ files, workspaceIndex: undefined }, hooks); // warm + const times = []; + for (let r = 0; r < reps; r++) { + const t0 = performance.now(); + finalize({ files, workspaceIndex: undefined }, hooks); + times.push(performance.now() - t0); + } + times.sort((a, b) => a - b); + return times[Math.floor(times.length / 2)]; +} + +/** Longest `transitiveVia` any edge in this graph carries. */ +function maxViaLength(files) { + const out = finalize({ files, workspaceIndex: undefined }, hooksFor(files)); + let max = 0; + for (const edges of out.imports.values()) { + for (const e of edges) { + if (e.transitiveVia !== undefined) max = Math.max(max, e.transitiveVia.length); + } + } + return max; +} + +const deepChain = chainCorpus(PROBE_DEPTH); +const maxVia = maxViaLength(deepChain); +const chainMs = timeMedian(deepChain); +const widthMs = timeMedian( + packageCorpus({ + leaves: 6000, + defsPerLeaf: 8, + pkgSize: 12, + consumers: 3000, + importsPerConsumer: 15, + }), + 3, +); + +console.log(`chain depth ${PROBE_DEPTH} : ${chainMs.toFixed(1)} ms`); +console.log(`max_via_len : ${maxVia} (must equal ${EXPECTED_MAX_VIA})`); +console.log(`width_ms : ${widthMs.toFixed(1)} (budget <= ${WIDTH_MS_BUDGET})`); + +if (process.argv.includes('--check')) { + let failed = false; + if (maxVia !== EXPECTED_MAX_VIA) { + failed = true; + console.error( + `\nFAIL max_via_len: ${maxVia}, expected exactly ${EXPECTED_MAX_VIA}.\n` + + `A LARGER value means the \`via\` chain copy lost its bound — see ` + + `MAX_VIA_LENGTH in gitnexus-shared/src/scope-resolution/finalize-algorithm.ts. ` + + `Each hop copies the inherited path, so an unbounded chain is O(depth^2) ` + + `in time and retained memory (measured 67 ms / 145 MB vs 25 ms / 40 MB at ` + + `depth ${PROBE_DEPTH}).\nA SMALLER value means the cap moved; update ` + + `EXPECTED_MAX_VIA here and the two finalize-algorithm tests that pin it.`, + ); + } + if (widthMs > WIDTH_MS_BUDGET) { + failed = true; + console.error( + `\nFAIL width_ms: ${widthMs.toFixed(1)} exceeds budget ${WIDTH_MS_BUDGET}. ` + + `With max_via_len healthy this points at a per-lookup linear scan coming ` + + `back (see indexExportsByName). Re-run on an idle machine first.`, + ); + } + if (failed) process.exit(1); + console.log('\nOK — within budget.'); +} diff --git a/gitnexus/bench/import-target/baselines.json b/gitnexus/bench/import-target/baselines.json new file mode 100644 index 000000000..4c8e75f9f --- /dev/null +++ b/gitnexus/bench/import-target/baselines.json @@ -0,0 +1,1012 @@ +{ + "_what": "Baselines for bench/import-target/measure.mjs \u2014 EVERY import-target resolver registered in SCOPE_RESOLVERS, on one shared corpus, plus csharp a second time WITH csproj configs. One entry per registered language and one more for the csproj arm, no registered language ungated \u2014 and that is ASSERTED rather than asserted-in-a-comment, which is also why no roster of language names is kept in this prose to go stale: measure.mjs derives its language list from a LANG_REGISTRY table and a --check inventory arm reconciles that table against SCOPE_RESOLVERS in both directions. A C/C++ #include is an import site for this purpose and is gated like every other registered language. csharp and csharp_csproj resolve the IDENTICAL file corpus (buildFiles aliases the two) and differ in exactly one thing: whether csharpConfigs is supplied. Without that second arm the csproj namespace-directory index ships unmeasured, because every C# import in the no-csproj arm returns before reaching it. C and C++ follow that same precedent for a different context \u2014 their HEADERS arrive through resolutionConfig rather than through allFilePaths, and augmentedFilePaths unions the two once per pass, so the corpus is split at newPass rather than pre-merged. The first nine were added as their own O(imports x files) scans were indexed away (#2877/#2878/#2879/#2880, #2872, #2901, #2902, #2908) and this is the forward guard on each; the other eight were ungated until now, and PR #2911 \u2014 JavaScript reaching suffixResolve with no index at all, 25972 us per import at 8000 files \u2014 is what that costs.", + "_fingerprint_note": "Per-language sha256 over every distinct fromFile|target -> resolved target. A change here is a BEHAVIOUR change: the resolver returned a different target set, and IMPORTS/CALLS edges moved. Explain it, never re-baseline to make CI green. For the languages these PRs changed, the pre-change implementations produce these same values on this corpus at both 400 and 1600 files \u2014 that is what makes the index hoist a performance change. The tie-break-level proof lives in test/unit/scope-resolution/import-target-index-parity.test.ts (verbatim copies of the pre-change code, diffed) for Kotlin in test/unit/scope-resolution/kotlin/kotlin-import-target-parity.test.ts, and for the four resolvers added there in test/unit/scope-resolution/{php,java,cobol}-import-target-parity.test.ts and test/unit/import-resolvers/csharp-csproj-parity.test.ts, and for JavaScript in test/unit/scope-resolution/javascript-import-target-parity.test.ts (a differential over 211200 old-vs-new pairs, PR #2911). The eight languages added last have no per-language parity harness against a pre-change implementation and do NOT need one: nothing about their resolution changed, so there is no before to diff against. Their fingerprints are pure forward guards, minted from the current implementations, and their adapter-boundary index reuse is covered for every registered language at once by test/unit/scope-resolution/import-target-index-reuse.contract.test.ts. NOTE for csharp_csproj: on this corpus the #2902 indexed leg (step 3 of resolveCSharpImportInternal) is reached by 2221 of the 3200 small-arm imports but answers null for every one of them \u2014 the 979 that resolve do so at step 2 \u2014 so this fingerprint pins that legs cost and its null answers, while its positive tie-breaks (unanchored substring, iteration order) are pinned by csharp-csproj-parity.test.ts. NOTE for kotlin, go, csharp and java: twenty fingerprints across these four languages were re-baselined in #2881, the one deliberate behaviour change any language in this file has had. It landed in two steps and the second is the reason the first is not a special case: Kotlin first, then the shared package-dir-index (go, java, csharp) and the csproj namespace index once the same rule was found live there. `getKotlinFileIndex` no longer requires a file's package directory to be the FIRST occurrence of that name in its own path, so the unique arm's `d % 7` nested slice (`mod{d}/src/main/kotlin/com/example/pkg{d}/inner/pkg{d}`) now belongs to package `pkg{d}` and its wildcard imports resolve: resolved 1100 -> 1153 small and deep, 4456 -> 4681 large. The collide arm needed a CORPUS edit alongside it, not just a new number \u2014 its `d % 7` slice deliberately imported `com.example.vendor{d}`, a package that exists nowhere, purely to mirror the unique arm's nested-slice MISS, so leaving it would have left collide at 1100 against small's 1153 and broken the same-workload invariant the arm is built on (that assertion is what caught it). It now uses the same `com.example.models.*` spelling as the rest of the arm, which is why its distinct_outcomes fell (2775 -> 2744, 11087 -> 10961): one shared target instead of one per d. The record-level evidence for the resolver change \u2014 235 of 19968 records moved, 54 null -> resolved, 0 buckets losing a member \u2014 is in bench/kotlin-import-target/baselines.json `_provenance`. The kotlin heap_reading_bytes and heap_ceiling_bytes moved with it, together as `_heap_reading_note` requires: 48073096 -> 48200224 bytes_large (+127128, +0.264%), ceiling still exactly 1.5x. Small, and it is worth saying WHY it is small rather than reading the number as evidence that the change is cheap. `dirChildren` grows by one entry per component-suffix the old rule used to skip, and this arm can only see part of that: the heap corpus is built with HEAP_PAD 8, which prefixes every path with `d0/\u2026/d7/`, so no path can begin with a suffix of its own directory and the leading-segment half of the old rule is structurally invisible here. What moves the reading is the `d % 7` nested slice alone. Read +0.264% as this arm's ceiling on the effect, not as the effect. GO NEEDED A CORPUS EDIT TO BE GATED AT ALL. Its nested slice was `src/pkg{d}/internal/pkg{d}`, repeating only the LAST segment, while a Go query addresses the whole package path `src/pkg{d}` \u2014 so the directory never even ended with the query and the first-occurrence rule was never reached. Every go arm sat unchanged through the resolver fix. `uniqueDir`/`collideDir` now repeat the shape at the granularity Go actually queries (`src/pkg{d}/internal/src/pkg{d}`, `svc{d}/internal/sub/svc{d}/internal`), which is what moved go from 979 to 1153 resolved and bumped `languages.go.heap.path_segments` 13 -> 14. The general lesson: a corpus that carries a shape the QUERY cannot express does not gate that shape. CSHARP AND JAVA HIT THE SAME COLLIDE-ARM TRAP AS KOTLIN. Both collide arms sent their `d % 7` slice to a namespace that exists nowhere (`App.Src{d}.Vendor`, `com.svc{d}.vendor`) purely to MIRROR the unique arm's nested-slice miss; once that miss became a hit, collide sat at 979/1100 against small's 1153 and the same-workload assertion failed. Both now use the same spelling as the rest of their arm. HEAP: no reading here moved for the resolver change. An earlier revision of this branch re-recorded `csharp_csproj` 73703384 -> 73116520 as a -0.79% effect of the step-2 filter; review measured base and branch three times each and got the same 73.10e6 on BOTH sides \u2014 the recorded 73703384 was simply not reproducible on this box, and re-recording it would have dropped that language's derived floor by 0.8% for no reason belonging to this change. Reverted. Everything else sat within +/-0.03%. Note that `_heap_reading_note`'s claim that these readings 'reproduce to the byte across processes on one box' did NOT hold on the box this was measured on: go, dart, ruby, python, php and cpp all wandered by a few hundred to a few thousand bytes between processes with no code change touching them. Treat sub-0.05% movement as jitter, not signal. HEAP, kotlin, second movement: 48200224 -> 42802456 (-11.20%), re-recorded with its ceiling. `getKotlinFileIndex` now compacts each `dirChildren` bucket as it freezes it. `addChild` mints a bucket as `[raw]` and pushes the rest, and V8 grows a backing store by `old + old/2 + 16`, so the second child takes a 1-slot store to 17: 61144 buckets, 52.9% of their slots empty, 88 B each. Same fix and same accounting as the python `byBasename` sentence above. Note what this means for the gate: a memory WIN of this size passes every arm \u2014 it is under the ceiling and over the 0.5x floor \u2014 so it is recorded because the convention says a reading and its ceiling move together, not because anything went red. kotlin now reads 40.82 MiB. The prose in measure.mjs calling it '45.85 MiB, the second-largest reading in this file' is corrected with it \u2014 and was already wrong on the ranking before this change, since csharp_csproj (69.73) and php (47.28) both read higher; kotlin was third. A measurement written into prose is not re-taken, which is the finding `_heap_bound_note` records about this very file. One further corpus edit, made in review and MEASURED rather than assumed: kotlin's collide layout repeated only the `models` leaf (`\u2026/com/example/models/inner/models`) while a Kotlin query addresses the whole dotted path, so a full revert of the Kotlin guards left both collide fingerprints UNMOVED \u2014 the arm was blind to the rule it was re-baselined for. Deepening it to `\u2026/models/inner/com/example/models` makes the revert move both, and those two fingerprints are the only ones that changed for it. The same deepening was applied to the java and kotlin UNIQUE arms and REVERTED: it moved ten more fingerprints, grew java's heap reading 43%, and bought nothing \u2014 progressive stripping lands those queries on the same file with or without the rule, so the control still failed only on go.", + "_shape_note": "files/imports/resolved/distinct_outcomes AND the fingerprint are asserted exactly, per scale. A fingerprint alone cannot tell a legitimate resolution change from a corpus quietly shrunk below the size at which the timing arms can see anything; conversely the counts alone cannot see a defect confined to one arm, because the arms differ only in path padding and directory layout and both of those are count-neutral by design. Two cross-arm assertions close the remaining hole: the deep and collide arms must resolve exactly what small resolves (they are the same workload), and each of their fingerprints must DIFFER from small's (they are not the same corpus). Without the second, setting DEEP_PAD to 0 \u2014 which deletes the entire depth arm \u2014 moves no asserted number and prints PASS; the same is true of a collideDir that forwards to uniqueDir. THE HEAP ARM IS ASSERTED THE SAME WAY, by the same loop, and was not before: files_small, files_large, path_segments and probe decide WHAT it measures, and every one of them was reported and compared to nothing. Swapping HEAP_PROBE_TARGET.csharp_csproj for a target matching no CSPROJ_CONFIGS rootNamespace skips the whole config loop, so the getFilesInDir and getInsensitive legs never run and the arm the header calls the witness that the read pattern IS the footprint quietly becomes a two-map arm \u2014 73703384 -> 59921216 B, ratio 1.017 -> 1.011, ceiling and floor both still passing and --check still exiting 0. Setting HEAP_SMALL equal to HEAP_LARGE is the same hole from the other side: ratio goes to ~1.0 by construction and bytes_large never moves. bytes_small and bytes_large are deliberately NOT asserted for equality \u2014 heap_ceiling_bytes and the heap_reading_bytes floor bound them with ~50% either way, because heapUsed accounting moves across platforms and Node majors and an exact byte assertion would be a re-baseline per runner. THE CONTEXT ARM IS ASSERTED THE SAME WAY, by the same loop, and more strictly than either: target, with_context and without_context are exact strings with no tolerance at all, because the arm resolves one import over a three-file corpus and has no measurement noise to tolerate. A separate check requires the last two to DIFFER, for the same reason deep.fingerprint must differ from small.fingerprint \u2014 a probe on which both call shapes agree asserts one number twice. Both halves run through resolveOne, so what the arm gates is this bench threading run.ts's fifth argument, not the resolvers' behaviour.", + "_arms_note": "Five timing arms, one memory arm and one deterministic arm elsewhere, because none of them gates alone. scaling_ratio (t_large/t_small)/(1600/400) catches cost growing with FILE COUNT \u2014 the #2877-#2880, #2901, #2902 and #2908 regressions themselves; every one of those legs was Theta(files) per import, so a revert scores ~4 here by construction. depth_ratio (t_deep/t_small at a FIXED file count, ~6x the path components) catches cost growing with path DEPTH, which scaling_ratio divides out and structurally cannot see; buildSuffixIndex (C#, Ruby, PHP, Java) and Kotlin suffixByStem emit one entry per component, so they legitimately sit above 1.0 while Go, Dart and COBOL, whose indexes are depth-free, sit at ~1.0. csharp's depth_budget has now been retightened twice for the same reason, and the second time it did lock the win in. It was 5 against a then-measured 3.318; #2903 made buildSuffixIndex's dirMap lazy and it became 3.5 against 2.31, with the file stating plainly that 3.5 did NOT lock that win in because a revert to an eager dirMap scores 3.318 and passes. Extending the laziness to the two SUFFIX maps drops it again, to 1.438 (java likewise 2.214 -> 1.402), because the deep arm has ~6x the path components and an O(files x depth) build of a map the no-csproj leg never reads is exactly the cost that scales with depth. Both are now 2.2, which is this file's 1.5x convention against measurements whose own peak-to-peak over 4 runs is 1.04x and 1.07x \u2014 and 2.2 DOES lock it in: an eager rebuild scores 2.3+ and fails. The other fifteen depth budgets sit at 1.37-1.75x measured and are unchanged. collide_scaling_ratio is the same measurement on a SHARED-LEAF layout (svcN/internal, SrcN/Models, com/example/model in every service, a repeated mod0.dart/mod0.rb/Mod0.cpy basename) carrying an identical file, import and resolved count: the small/large/deep arms mint one directory name per index, so every index bucket in them holds exactly ONE entry (measured: max last-segment bucket 1 and max matching directories 1 for go and csharp at 400 and 1600 files; max basename bucket 1 for dart and ruby), and bucket cardinality is the only non-constant term the new indexes have. On the shared-leaf shape go, csharp, dart and java legitimately score 2.1-3.9 because the bucket grows with the file count BY CONSTRUCTION \u2014 this is a limit on the SCOPE of the \"independent of corpus size\" claim, not a regression (the indexed code is still faster there than the pre-change full scan); their collide budgets say so honestly instead of pretending 1.8. Ruby, Kotlin, PHP and COBOL answer from keyed maps and are collision-immune, so they keep the linear 1.8 budget and that immunity is the assertion. csharp_csproj is the one arm that runs the other way: its shared leaf collapses dirsByLastSegment to the single key Models, so the slash-free sweep (see CSPROJ_CONFIGS) is CHEAPER on the collide layout than on the unique one and its expensive scale arm is large, not collide_large. Its 1.8 collide budget is therefore the linear one, and the arm that carries its real cost is the unique one. The collide arm is also the only arm that reaches filesDirectlyInPkgDir's dirCount > 1 merge (go: 388 multi-directory calls at 400 files, up to 9 directories; 1517 at 1600 files, up to 34) and the only one that reaches COBOL's copybook-over-source tier tie-break, which needs one bookname to name two files. small_ms_ceiling and collide_ms_ceiling are ABSOLUTE (~4x the measured arm), because a constant-factor regression that grows both scale arms equally passes every ratio. The five arms added here use 4.2x, the middle of the 3.7-4.6x the original five already carry; the two COBOL arms use ~5x, the multiplier dart's sub-1 ms arm has always carried, because a fixed scheduler hiccup is a larger fraction of a smaller number \u2014 measured over 8 runs they sat at 0.25-0.37 ms and 0.18-0.30 ms, and the pre-#2908 two-scans-per-COPY implementation costs ~300 ms on the same arm, so 2.0 and 1.5 still separate fixed from broken by two orders of magnitude. NOISE, measured rather than assumed: depth_ratio divides two sub-3 ms numbers (Dart's are sub-1 ms) and is by far the noisiest arm here, so it set N for the whole file. fastest() is a min-of-N estimator, so N is the knob. Over 22 --check runs on an idle box, peak-to-peak: at N=5 go ran 0.757-1.748 (2.31x) and tripped its own 1.6 budget about 1 run in 20; at N=7 (the kotlin-import-target setting) Dart still ran 0.678-2.043 (3.01x) and tripped once; at N=15 (bench/cfg, bench/schema-pairs, bench/callable-value-flow) every language collapsed to a 1.13-1.26x swing with 22/22 passing. The budgets were NOT widened; the estimator was fixed instead, which is why the headroom above is real rather than granted. N IS NOW PER LANGUAGE, and that is a refinement of the same finding rather than a retreat from it. The overshoot of min-of-K against min-of-15 is a function of the CELL's absolute duration, not of the language: replayed against two independent runs' full sample sets, the worst overshoots at K=7 land on swift.small (0.43 ms, 31.8%) and dart.collide (1.5 ms, 37.6%), while every cell at or above 10 ms overshoots by at most 6.3%. So repsFor() keeps 15 while a language's cheapest arm is under 5 ms and otherwise spends ~150 ms per cell, floored at 7 \u2014 15 for go, csharp, dart, kotlin, java, cobol, swift, rust, python, c and cpp (every language the flakiness above was ever about, cheapest arm 0.19-3.2 ms) and 7-8 for csharp_csproj, ruby, php, javascript, typescript and vue (cheapest arm 20-28 ms). Per LANGUAGE, not per cell, so all five arms of a language share one estimator and the four ratios stay comparisons of like with like. The replay passed all 85 cells on all five gates at 0.4-0.7 of budget and saved 12.8 s and 12.4 s of a 46 s run; min-of-7 also reads slightly HIGHER than min-of-15, so the ceilings get marginally more sensitive rather than less. Confirmed on 4 fresh runs with the adaptive estimator live: every small arm inside 1.12x peak-to-peak and every collide arm inside 1.07x, with the six 7-8 rep languages at 1.008-1.071 \u2014 no worse than the 11 that kept 15. The chosen N is reported per language as `reps`. heap_ceiling_bytes bounds the retained per-pass import index, the only arm here that can see memory: buildSuffixIndex emits maps at O(files x depth), the profile package-dir-index.ts cites #2649 to avoid for itself, and csharp, ruby, php and java all retained NOTHING across imports at BASE (C#'s no-csproj leg and PHP's and Java's every leg re-scanned the raw Set; Ruby rebuilt and discarded a suffix index per require). It is measured at 8000 and 32000 files at HEAP_PAD depth rather than at the timing arms' sizes, because the finding is an ABSOLUTE footprint at repository scale. THE ARM NOW READS WHAT THE LANGUAGE READS, and that change is the whole reason this file was re-baselined. Four of these arms used to call getWorkspaceFileIndex(set) directly and then read index.all.length, which asks no suffix question at all \u2014 harmless only while buildSuffixIndex built both maps eagerly. The moment they went lazy the direct call built NO map, csharp, ruby, php and java each reported 0 B at 32000 files, and 0 B is under every ceiling: --check printed PASS over four gates that had silently become ceilings over nothing, which is precisely the failure this file's own header warns about for rust and cobol. Every arm now resolves a real MISSING import through the real resolver (HEAP_PROBE_TARGET, asserted to miss), so the maps it forces are the maps production forces, and a resolver that starts asking a new question moves the number without anyone editing the bench. That makes the READ PATTERN the dominant term, and the eight numbers say so: java 34958600 B and csharp 29862200 B ask index.get and never getInsensitive; php 37579888 B asks getInsensitive and never get, plus its own first-proper-suffix map; ruby 41025360 B and javascript 26745296 B read get(s) || getInsensitive(s) and pay for both, the second DERIVED from the first; and csharp_csproj 73705944 B additionally asks getFilesInDir. csharp_csproj IS NOW GATED, reversing the earlier decision that it would be 'a ceiling on a duplicate': at +20.8% of the C# index it was one, and at 2.47x of it \u2014 same corpus, same getWorkspaceFileIndex, three maps instead of one \u2014 it is the witness that the read pattern is the footprint. The old RESIDUAL note is superseded by that number: a dirMap-sized addition is no longer +18%, and a consumer that asks all three questions blows csharp's ceiling by 1.64x rather than sliding under it. A SECOND MEASUREMENT BIAS was removed at the same time and it moved every figure here, so do not read these against the old ones as if only the read pattern changed. buildFiles mints paths with template literals, which V8 keeps as ropes; the first traversal that slices one flattens it, allocating the flat string and dropping the rope's pieces, so a build measured over an unflattened corpus reports the index MINUS that net release \u2014 11% low, uniformly. bytes_small was read over a corpus a discarded warm-up pass had already flattened and bytes_large over a fresh one, so every ratio read ~0.85-0.89 for structures that are exactly linear in the file count. measureHeap now flattens each corpus before measuring it; all eight ratios read 0.998-1.017, and the warm-up pass is gone because with the corpus flat a language's first and second reads agree to within 0.3%. python's figure rises from 7624992 to 10362976 for this reason and not because anything regressed, and then to 10543152 (+1.7%) because #2913's nestedDirNames set is retained for the pass, and then FALLS to 6360936 (-39.7%) for a reason worth knowing: byBasename holds roughly one bucket per file, and building each with `[]` followed by `push` made V8 grow the backing store to its 16-slot minimum, so every single-file bucket retained 15 empty pointer slots. Constructing the one-element buckets directly (`set(base, [entry])`) is byte-identical in contents and 3.9 MiB smaller at 32000 paths \u2014 37% of what this arm used to read was empty array slots \u2014 the ancestorsByDir memo itself is NOT in this reading, because python's probe target misses at the nested-name rejection and never reaches the walk, so this arm does not bound that memo; measured separately with a probe that does reach it, a 32000-file corpus with every file in its own 10-deep directory retains ~19 MB, which would clear this ceiling, so repointing python's heap probe at a walking spelling means re-recording the ceiling in the same change, and c is unchanged at 10018816 because its basename map does not slice paths. Its ceiling is 1.5x the measured arm, and the DIFFERENCE FROM THE 4x TIMING CONVENTION IS DELIBERATE \u2014 do not harmonise it back. 4x exists because runner contention dominates a wall-clock number; this one has essentially no measurement noise (across 4 runs the widest spread was 0.11% on python, 0.03% on csharp_csproj and 0.00% \u2014 identical to the byte \u2014 on ruby, php, java, javascript and c, and the same holds across separate processes), so 4x would throw away almost all of the gate's power and sail straight past the regression this arm exists to catch. 1.5x still tolerates ~50% of cross-platform and Node-version drift, far more than a Node major bump plausibly moves heapUsed accounting; it catches a duplicated index (+100%) or a second exactMap-sized suffix map (+~85%). heap_floor_fraction is the arm the 0 B incident proved was missing. A ceiling can only say 'not too big'; nothing said 'still measuring something', which is why four dead arms passed. The floor is 0.5 x each language's RECORDED READING (heap_reading_bytes), which is half the measured size and says so. It used to be 0.33 x the CEILING, described the same way \u2014 true only while every ceiling stayed at exactly 1.5x its reading, a convention this file states and nothing enforces, so re-tuning one ceiling upward would have loosened that language's floor by the same factor in the one direction a floor exists to watch. The two forms agree to within 0.8% for all eight today, so this is a correction of derivation, not of strength. It sits ~400x above the readings' own reproducibility and far below any collapse. A genuine 2x memory WIN trips it too, and that is intended: like a fingerprint move, it must be explained and re-baselined rather than absorbed. COBOL is left out for the opposite reason: its index is two Map, O(files) with no depth term, and at 32000 files its retained delta does not clear the noise of the measurement itself. heap_ratio_budget, the linear-growth check across the 4x file-count gap, is the orthogonal arm: it sees per-file and per-depth growth but not a constant factor. ---- THE EIGHT LANGUAGES ADDED LAST (swift, rust, python, javascript, typescript, vue, c, cpp) ---- They carry the SAME five arms and the same gates; what differs is which arm can actually fail for each, because each resolver has a different cost axis, and the budgets below say so instead of copying a number across. Every figure quoted is the MAXIMUM over 5 full runs on an idle box, and the peak-to-peak of every one of these arms stayed inside 1.10x over those runs \u2014 tighter than the 1.13-1.26x the original nine record, because none of these arms divides two sub-1 ms numbers the way dart depth_ratio does. depth_budget is ~1.5x measured throughout: swift 2.3 (1.487), rust 2.1 (1.377), javascript 2.1 (1.376), typescript 2.1 (1.381), vue 2.3 (1.563), c 3.0 (1.990), cpp 3.0 (1.999). PYTHON WAS 11 AGAINST 7.389 AND IS NOW 2.6 AGAINST 1.872, because #2913 fixed the resolver rather than the budget. Its INDEX was always depth-free; hasRepoCandidate and resolveAbsoluteFromFiles each rebuilt one ancestor prefix per directory component of the importer on EVERY import, and the index's own dirPrefixes build inserted one entry per component per file, so the resolver was quadratic in path depth where every other language here is linear or flat. The prefixes are a pure function of the importer's DIRECTORY, so they are now memoized per directory inside getPythonFileIndex (ancestorsByDir), the leading segment is rejected up front against a set of nested directory names, the module and package buckets are consulted before the walk rather than inside it, and the dirPrefixes build stops at the first ancestor already stored. All five fingerprints are byte-identical, so it is a hoist. The budget is 2.2, and BOTH numbers behind it were re-measured on a quiet box AFTER the context leg below started being measured, because that change moved the arm: the work it adds is depth-FLAT, so python's absolute cost more than doubled while depth_ratio FELL to 1.405-1.563 over 5 serial runs (peak-to-peak 1.11x). A budget carried over from before that change would have been slack against a smaller ratio. 2.2 is 1.41x the measured maximum, inside the 1.37-1.75x band the other fifteen sit in, and it LOCKS THE WIN IN: reverting the per-directory ancestor memo alone scores 2.524 and reverting the nested-name rejection alone scores 2.553, both measured under the current call shape, so each fails at 2.2 with 13% to spare. Do not read those two figures as the pre-#2913 cost \u2014 7.239 was that, and the gap closed because the bare-import tier stopped walking at all (see below). The other two parts of the fix are not gated by this arm and are not meant to be: reverting the bucket prune or the dirPrefixes early break lands under any budget this arm's noise supports, so they are gated deterministically instead, by the prefix-parity and package-probe arms of test/unit/scope-resolution/python/python-importer-ancestors.test.ts and python-import-target-parity.test.ts, which go red on exactly those two mutations. A timing budget catches what it can measure; the counts catch the rest. THE BARE-IMPORT TIER (`import os`, single segment, no dot) was a separate O(depth) walk in import-resolvers/python.ts that this bench cannot see at all, because every python arm here spells its imports with a dot and returns at the `pathLike.includes('/')` guard before reaching it. It ran TWICE per `from x import y` \u2014 the package probe's recursion re-ran the whole tail on identical inputs \u2014 and is now one memoized chain plus an O(1) proof-of-absence against the index's basename buckets: 12/24/72 Set probes at depth 1/4/16 became a flat 2, and 11.615 us/import at 18 path components became 0.740. Gated by probe COUNT in test/unit/scope-resolution/python/python-import-probe-count.test.ts, not here. collide_scaling_budget splits three ways. Three languages scan a bucket that grows with the corpus and get their measured value x1.5: swift 4.9 (3.279 \u2014 its bucket is the module file list it RETURNS, and its collide arm is four modules instead of dirs of them so that bucket is fileCount/4, i.e. 100 files at 400 and 400 at 1600), c 3.8 (2.535) and cpp 4.0 (2.639, the same basename bucket its suffix fallback walks). Four answer from keyed maps and keep the linear 1.8 \u2014 python 1.097, javascript 1.083, typescript 1.053, vue 1.079 \u2014 and that immunity IS the assertion, exactly as for ruby, kotlin, php and cobol. RUST IS THE ONE ARM THAT WAS REDESIGNED RATHER THAN BUDGETED. It resolves by probing candidate paths with allFilePaths.has(...) and never searches, so its cost is O(path segments) and provably flat in the file count (1.095 scaling, 1.061 collide scaling): a shared-leaf collide arm for rust would have asserted nothing, which is worse than no arm. Its collide corpus is instead a deep module tree (src/l0/l1/l2/l3/l4/mod{d}) whose targets carry ~2x the :: segments, so the arm exercises the axis that CAN grow, its 1.8 budget asserts the flatness across file counts, and collide_ms_ceiling 19 bounds the absolute cost of the long-path probe. small_ms_ceiling and collide_ms_ceiling are ~4x measured as everywhere else: rust 10/19 (2.609/4.704), python 7/8 (1.76/1.929, retightened from 12/15 against 3.044/3.771 by #2913), javascript 85/89 (21.254/22.145), typescript 85/86 (21.250/21.464), vue 81/93 (20.164/23.227), c 7/11 (1.620/2.850), cpp 7/12 (1.581/3.009). Swift takes ~5x (2 against 0.421 and 4 against 0.821) \u2014 the multiplier dart and cobol already carry, because a fixed scheduler hiccup is a larger fraction of a sub-1 ms number. ONE CAVEAT ON THE THREE ts-FAMILY MS NUMBERS, stated because nothing else in this file would reveal it: resolveTsTarget carries a per-pass resolveCache keyed currentFile::importPath, which no other resolver here has, and ~10% of this corpus is repeat pairs. Their us/import is therefore a slight underestimate of a cold resolve. It is left in rather than defeated because it is what the real pipeline does, and it is identical across all three so the arms stay comparable. HEAP for the eight: rust, swift, typescript, vue, cpp and cobol are still NOT gated, all of them measured before being left out. rust builds no index on this hook (16 B at 8000 files, 0 B at 32000); swift holds one pointer per file-times-segment and mints no strings, reading 0.98 MB at 8000 files against 0.29 MB at 32000 \u2014 a 4x larger corpus reading 3x SMALLER, which is what a measurement below its own noise floor looks like, and the same reading cobol gives (0.54 MB then 0 B); typescript and vue duplicate javascript through the same builder over the same-shaped corpus, and cpp duplicates c (10021320 against 10016960, 0.04% apart). Those four duplications are the ONLY exclusions that still rest on 'it would be a duplicate', and they are duplicates of a builder AND of a read pattern, which is the pairing csharp_csproj failed once the read pattern started to matter \u2014 if any of the four ever diverges in what it ASKS the index, it earns an arm the same way csharp_csproj just did. All eight gated arms are read the same way now (retainedPassBytes, one real import), so unlike before they are directly comparable to one another. WALL CLOCK \u2014 ~33-35 s in report mode, down from ~46 s, and ~44-45 s for --check, which is essentially UNCHANGED from ~46 s. Only report mode got faster; do not read the pair as 46 -> 42. The breakdown is worth having before anyone trims it. Timing arms: go 2.02, csharp 1.09, csharp_csproj 3.22, dart 0.41, ruby 2.90, kotlin 0.85, php 3.46, java 1.57, cobol 0.09, swift 0.46, rust 0.85, python 1.22, javascript 3.23, typescript 2.72, vue 2.89, c 0.86, cpp 0.91 (28.7 s, from 39.8 s: repsFor() accounts for all of it, and every second of it comes from the six languages whose cheapest cell is 20-28 ms); heap arms 3.43 s for SEVENTEEN languages, from 2.06 s for eight (every registered language is measured now; the nine added cost 1.37 s, of which kotlin alone is 0.57 s \u2014 see _heap_bound_note), and 2.1 s came from 3.0 s for seven when flattening retired the warm-up pass; module load 3.9 s. --check pays one import that report mode does not: the inventory arm loads pipeline/registry.ts, which drags in every registered scope resolver and its providers. Measured in isolation with the bench's own static imports already resident, that import costs 6.3-6.5 s on one box and 9.3-10.0 s on another \u2014 i.e. it consumes almost the whole repsFor win, which is why --check did not get faster. It is loaded dynamically at the point of use rather than at the top of the file, so report mode does not pay it and both modes take their measurements in the same module state. IT WAS WEIGHED AND KEPT, on the number that decides it: the benchmarks job is not CI's critical path. On the last green run of main it took 9 m 23 s against 12 m 58 s for the sharded coverage job that gates the merge, so ~4 m 40 s of slack sits above this bench and those seconds buy zero merge latency. Moving the arm to a vitest file would move the registry load ONTO the critical path, and would weaken it as well: this reconciles LANG_REGISTRY's SupportedLanguages values, which are what the five dispatcher branches key off, whereas a test that cannot import measure.mjs can only reconcile this file's arm NAMES plus a hand-written rule for de-aliasing csharp_csproj. The contract test import-target-index-reuse.contract.test.ts already covers the ADAPTER-boundary contract for every registered resolver; this arm covers a different claim, that the BENCH covers the pipeline. The ts family is still the largest single block of the timing phase (8.8 s) \u2014 its cost is suffixResolve probing ~39 extensions per path part on a miss, which is the real resolver and cannot be tuned away from the bench side. IF IT HAS TO SHRINK, drop collide and collide_large for typescript and vue and nothing else: -3.9 s, and it is the only cut that removes near-duplicate work rather than coverage, because all three run the same resolveTsTarget over the same buildSuffixIndex and javascript keeps the collide arm that covers their shared collision axis. Do NOT reach for REPS_MAX: it is 15 because depth_ratio tripped its own budget about 1 run in 20 at 5 and once at 7, and lowering it would re-open that for the eleven languages whose cheapest cell is sub-5 ms \u2014 which is where every recorded trip happened. The six languages it was safe to lower have already been lowered, per language and from a measurement, by repsFor(). ---- THE FIFTH ARGUMENT (context) AND THE TWO ARMS IT MOVED ---- resolveOne now makes run.ts's five-argument call for the two hooks that declare a fifth parameter, so php and python time the legs behind it. Nothing else moved: the other fifteen arms are handed no context and build no ParsedFile[] at all, and over five runs their five ms numbers and four ratios sit exactly where they did. Both languages' ten fingerprints, resolved counts and distinct_outcomes are IDENTICAL \u2014 the leg AGREES with the cascade on this corpus, which is the whole reason the context arm had to be added rather than leaving the fingerprint to notice. PHP: small_ms 27.762 -> 35.125 (+26.5%) and collide_ms 29.407 -> 36.182 (+23.0%), which is filesByDirectory plus, on every import that resolves, a candidate gather over the resolved file's directory and a localDefs filter; the ms ceilings keep PHP's own 4.21x and 4.26x multipliers (117 -> 148, 125 -> 154). depth_ratio 1.144 -> 1.283 and the 1.9 budget is UNCHANGED, which makes it 1.48x measured rather than 1.66x: directoryAliases emits one entry per path segment, so filesByDirectory is O(files x depth) and the depth arm is the only one that can see it \u2014 that budget got TIGHTER relative to its measurement, not looser, and 1.48x sits inside the 1.37-1.75x band the other sixteen carry. Its heap reading rises 37576816 -> 49574008 (+31.9%) for the same structure, and the reading is the MEMO rather than the workspace it indexes: newPass allocates the ParsedFile objects before retainedPassBytes takes its baseline sample, so they sit outside the delta. PYTHON, WHOSE FIGURES ARE THE LEAST SETTLED THING IN THIS FILE AND ARE RECORDED IN TWO SNAPSHOTS BECAUSE OF IT. A named import is the only spelling that reads context.parsedFiles, and it costs up to three entries into the resolver per import (package probe, exports check, submodule probe) where the synthetic namespace spelling this arm used to pass costs one. Against the resolver as it stood when the call shape changed that read small_ms 1.76 -> 5.751 and collide_ms 1.929 -> 5.894, ~3.1x. Against the resolver a few commits later \u2014 which stopped re-running the whole tail after a null package probe, a double-probe this bench could not previously see because the namespace spelling never entered that branch \u2014 the same arms read 4.404 and 4.505. The ceilings are 18 and 19, chosen to clear BOTH: 4.09x and 4.22x of the current numbers, 3.13x and 3.22x of the higher ones, so neither state is red. Retighten toward 4x once that resolver settles. ITS DEPTH ARM WAS DILUTED AND THE BUDGET IS RETIGHTENED TO MATCH, which is the one thing here worth arguing about: the added work is depth-FLAT, so depth_ratio FALLS 1.872 -> 1.478 while the absolute cost more than doubles, and 2.6 against 1.478 would be 1.76x \u2014 far looser than the 1.39x #2913 chose deliberately to lock its own fix in. 2.1 restores that multiplier (1.42x). THE TWO MUTATION SCORES #2913 RECORDED (3.123 for reverting the per-directory memo, 2.734 for reverting the nested-name rejection) WERE TAKEN AGAINST THE OLD CALL SHAPE AND HAVE NOT BEEN RE-TAKEN. Modelled forward, with the depth-quadratic term reappearing in every resolver entry so its absolute contribution scales with the entry count, they land near 2.8 and 2.4 \u2014 both above 2.1, and the second BELOW 2.6, which is the arithmetic that decided the budget. Re-run the two mutations before trusting the lock-in claim above. python's heap reading is unchanged (10543152 recorded; 10529848-10544616 across eight runs) because its probe misses before the branch that reads parsedFiles \u2014 see _blind_spot for why no probe can reach that memo. Every figure in this section is the MAXIMUM over its snapshot's runs (five, then three), with peak-to-peak 1.031-1.058 on php and 1.019-1.081 on python, taken on a box that was NOT idle and with another change landing in python's resolver mid-measurement. Re-take them serially before merging.", + "_triage": "Every ratio and ms ceiling here is a TIMING signal \u2014 re-run on an idle machine before investigating; runner contention dominates. depth_ratio is the noisiest of them by a wide margin (it divides two sub-3 ms numbers, and Dart's are sub-1 ms): if exactly one arm fails and it is that one, suspect the machine first. N is 15 for every language whose cheapest arm is under 5 ms, rather than this bench's original 5, specifically to hold that arm's peak-to-peak swing under 1.26x \u2014 see _arms_note for the measured distributions and for why the six languages that drop to 7-8 are the ones where cell size makes it safe \u2014 so a depth_ratio failure that REPRODUCES is a real signal, not noise. Each language's chosen N is printed as `reps`; read it before blaming the estimator. The fingerprint, shape and heap arms are the opposite: deterministic (over 4 runs the heap arm's widest spread was 0.11% on python and 0.00% on java, javascript and c), a re-run never changes them, and they must never be wished away. TWO heap failures mean the arm STOPPED MEASURING rather than that memory grew, and both are deterministic: a heap floor failure says the probe no longer forces the index it used to (this is how four arms read 0 B when buildSuffixIndex went lazy, and 0 B passes every ceiling), and a `heap probe ... resolved` throw says a probe target that must MISS now hits, so the reading is a materialized answer and the legs past it were never reached. A heap BOUND failure is deterministic in the same way and means one specific thing: a language excluded from the budgeted tier has grown a structure, or started asking its index a question it did not ask when the exclusion was recorded \u2014 never a timing signal, never a re-run, and never fixed by raising the bound without saying what grew. The context arm is deterministic too, and a failure there means one specific thing rather than a range of them: run.ts's fifth argument is not reaching that resolver from this bench, or the leg behind it stopped running. Never a timing signal, never a re-run. TIGHTENED IN #2881, because the measurements they bound got faster and a budget left alone while its reading falls is a gate loosening without anyone deciding to. Each new value holds the headroom the old one expressed over the old reading, computed from `_measured` on both sides: kotlin depth 3.4 -> 2.8 (reading 2.219 -> 1.813), go depth 1.6 -> 1.4 (1.169 -> 0.999), csharp depth 2.2 -> 2.0 (1.438 -> 1.279), java depth 2.2 -> 2.1 (1.402 -> 1.354), kotlin collide_scaling 1.8 -> 1.65 (1.179 -> 1.081), go collide_scaling 5.5 -> 5.1 (3.763 -> 3.465). The ABSOLUTE ms ceilings were deliberately NOT tightened by the same reasoning: they carry runner-contention headroom rather than measurement headroom, and a ratio is runner-speed-invariant where a millisecond is not.", + "_floor": "Measured against the pre-change implementations on THIS corpus at 150/600 files: go 3.36, csharp 4.10, dart 3.32, ruby 3.87. The issues report 4.00 / 3.43 / 4.05 on their own corpora; those are DIFFERENT numbers from different repositories and are not reproduced here \u2014 what they and these share is that both independently land in the quadratic band, well clear of the ~1.0 a linear result gives. Note also that this floor was taken at 150/600 while the gate runs at 400/1600, so it is a lower bound on what the pre-change code would score today. Kotlin's own bench measured its pre-index floor at 3.737. The four resolvers added later were NOT re-floored on this corpus, and the reason is that they do not need to be: every one of their pre-change legs walked the whole file set per import (PHP one findIndex per path part per extension, Java one scan per stripped prefix, COBOL two full scans per COPY, C# csproj one normalizedFileList pass per import per matching config), so their scaling_ratio is ~4 by construction rather than by measurement. Their per-import costs were measured on their own issue corpora instead: PHP 96.40 ms -> 0.036 ms, Java 8.05 ms -> 0.62 ms, COBOL 3879 us -> 10.5 us, C# csproj 1103 us -> 7.6 us. The 1.8 budget sits well above the linear result and well below every one of those. The eight languages added last were NOT floored either, and for a different reason again: they are not fixes, so there is no pre-change implementation to floor against. Their scaling budgets are the global linear 1.8 and the point of the arms is to hold the current numbers (measured 1.01-1.13) rather than to separate a fix from a break. The one exception is javascript, which IS a fix and does have a floor: 6448.9 us per import at 2000 files and 25972.6 us at 8000 \u2014 4.12x the per-import cost for 4x the files, i.e. O(imports x files) \u2014 against 28.5 / 27.4 us with the index PR #2911 gave it, and 25.0 / 27.0 us for TypeScript over the identical corpus.", + "_rebaselined_2910_java_declared_packages": "#2910 replaces Java path-suffix fallback with declared-package resolution. The benchmark now restores package capture side channels, threads parsedFiles through javaScopeResolver, proves the context leg with a positive path/package-mismatch probe, and models the collide arm as one package declared across service paths. External imports now remain unresolved; local exact and wildcard imports preserve the 1153/4681 workload. Java's index is package/type maps rather than suffix maps: bytes_large 34958600 -> 3676984, with its floor and ceiling re-recorded together. Depth and collision scaling budgets tighten to the shared linear 1.8 gate.", + "scaling_budget": 1.8, + "collide_scaling_budget": { + "go": 5.1, + "csharp": 3.4, + "csharp_csproj": 1.8, + "dart": 3.3, + "ruby": 1.8, + "kotlin": 1.65, + "php": 1.8, + "java": 1.8, + "cobol": 1.8, + "swift": 4.9, + "rust": 1.8, + "python": 1.8, + "javascript": 1.8, + "typescript": 1.8, + "vue": 1.8, + "c": 3.8, + "cpp": 4 + }, + "depth_budget": { + "go": 1.4, + "csharp": 2.0, + "csharp_csproj": 2.3, + "dart": 1.6, + "ruby": 2.2, + "kotlin": 2.8, + "php": 1.9, + "java": 1.8, + "cobol": 1.6, + "swift": 2.3, + "rust": 2.1, + "python": 2.2, + "javascript": 2.6, + "typescript": 2.6, + "vue": 2.3, + "c": 3, + "cpp": 3 + }, + "small_ms_ceiling": { + "go": 7, + "csharp": 11, + "csharp_csproj": 97, + "dart": 3, + "ruby": 77, + "kotlin": 12, + "php": 148, + "java": 17, + "cobol": 2, + "swift": 2, + "rust": 10, + "python": 18, + "javascript": 85, + "typescript": 85, + "vue": 81, + "c": 7, + "cpp": 7 + }, + "collide_ms_ceiling": { + "go": 28, + "csharp": 22, + "csharp_csproj": 105, + "dart": 6, + "ruby": 95, + "kotlin": 12, + "php": 154, + "java": 26, + "cobol": 1.5, + "swift": 4, + "rust": 19, + "python": 19, + "javascript": 89, + "typescript": 86, + "vue": 93, + "c": 11, + "cpp": 12 + }, + "heap_ceiling_bytes": { + "kotlin": 46000000, + "go": 4497696, + "dart": 11751300, + "cpp": 15035016, + "csharp": 44900000, + "csharp_csproj": 110600000, + "ruby": 61600000, + "php": 74400000, + "java": 5600000, + "python": 9541404, + "c": 15000000 + }, + "_heap_reading_note": "The measured bytes_large each heap_ceiling_bytes entry above is 1.5x \u2014 every entry except kotlin's, which is 1.0747x for a stated reason (see _heap_compaction_gate). Recorded so the FLOOR can be derived from the reading instead of from the ceiling. It used to be 0.33 x the ceiling, described as 'half the measured size' \u2014 which held only while every ceiling stayed at exactly 1.5x its reading, a convention this file states and nothing enforces, so re-tuning one ceiling upward would have loosened that language's floor by the same factor in the one direction a floor exists to watch. 0.5 x the reading is the same effective floor to within 0.8% for all eight and says what it means \u2014 and it is what let kotlin's ceiling be tightened to 1.0747x without moving kotlin's floor by a byte, which is exactly the independence this key was introduced for. These are NOT asserted for equality: they reproduce to the byte across processes on one box, but a Node major or a different platform moves heapUsed accounting, and the ceiling/floor pair is what tolerates that (+50%/-50%, and +7.5%/-50% for kotlin). Re-baseline a ceiling and re-baseline the reading with it \u2014 they are two views of one measurement.", + "_heap_compaction_gate": "WHY KOTLIN'S CEILING IS TIGHT AND EVERY OTHER ONE IS 1.5x. It is the only ceiling in this file that gates a size REDUCTION being preserved rather than a footprint not growing: #2881 compacts getKotlinFileIndex's dirChildren buckets (`bucket.slice()` before the freeze), and until this entry existed nothing anywhere could see that compaction disappear. MEASURED, not assumed \u2014 head against a copy of languages/kotlin/import-target.ts with the slice deleted and Object.freeze kept, one process, the same corpus this arm builds: bytes_large 42805256 -> 48184784 (+12.57%), bytes_small 10676432 -> 12020736 (+12.6%), byte-identical over three runs. NOTHING ELSE MOVES for that mutation. Every arm of bench/kotlin-import-target is output-identical (its fingerprint, cases and non_null cannot see an array's spare capacity); test/unit/scope-resolution/kotlin/kotlin-index-internals.test.ts stays green and says so in its own header, because a JS array's backing-store capacity has no reflective surface; heap ratio is 1.002 either way, since both scales grow together and a ratio divides the growth out; and at the old ceiling of 64203684 the heap arm passed with 25% to spare. DIRECTION MATTERS: compaction RECLAIMS, so losing it makes the reading GROW. The gate is therefore the CEILING. A floor cannot see this mutation in any sizing, and kotlin's floor stays the file-wide 0.5 x reading. WHERE THE 5.4 MB COMES FROM, so the number can be re-derived rather than trusted: the heap corpus is 32000 files over 4000 directories, 8 files each, and a directory contributes one dirChildren key per component-suffix of its path (~15.3 keys at HEAP_PAD 8), so ~61000 buckets of length 8. On this repo's Node a bucket minted as [raw] and pushed to 8 sits in a 19-slot backing store \u2014 the capacity steps kotlin-index-internals.test.ts records \u2014 leaving 11 slots, 88 B, of retained slack per bucket. 61000 x 88 B is ~5.4 MB, which is the delta. HOW 46000000 WAS CHOSEN: reading 42802456, plus 7.5% is 46012640, rounded down to 46000000 (1.0747x). PROVEN both ways through the real gate, not argued: a full `--check` over a copy of measure.mjs whose only difference is the kotlin import, pointed at a resolver with the slice deleted, reads 48203376 B (45.97 MiB) and fails on THIS ARM ALONE \u2014 every fingerprint, every corpus count, every timing ratio and the heap ratio all stay green, which is the claim 'nothing else moves' turned into a run. That is 4.8% clear above the ceiling. Both margins are three orders of magnitude larger than the measurement's own spread (peak-to-peak 1.0001 over three runs of the isolated arm, 1.0004 over the five runs _heap_bound_note records). The 7.5% is also an order of magnitude above the widest cross-run movement any heap arm in this file shows on this box: kotlin itself reads 42802456 B in a full `--check`, byte-identical to the recorded value, and the noisiest reading here \u2014 csharp_csproj, the one prior sessions found unreproducible \u2014 moves 0.78% between runs. A LOADED RUNNER DOES NOT MOVE THIS NUMBER and the tolerance is not for one: this is a forced-GC heapUsed delta over structures held alive across the window (see HEAP_RETAINED), so scheduler contention has no term in it. What can move it is heapUsed ACCOUNTING \u2014 a Node major, a heap above the pointer-compression cage, a 32-bit platform. TRIAGE, and it is what makes the tight ceiling safe to run: that class of change moves EVERY reading in the run, so compare kotlin against the other 13 budgeted readings in the SAME run before touching this key. kotlin alone over its ceiling with the rest of the file at its recorded values is a lost compaction; everything moving together is a runner change and a whole-file re-baseline. WHAT IT DOES NOT CATCH: any regression under 7.5%, and a compaction that still runs while something else in the index grows to fill the headroom.", + "heap_reading_bytes": { + "kotlin": 42802456, + "go": 2998464, + "dart": 7834200, + "cpp": 10023344, + "csharp": 29869080, + "csharp_csproj": 73703384, + "ruby": 41020808, + "php": 49574008, + "java": 3676984, + "python": 6360936, + "c": 10018816 + }, + "_heap_bound_note": "THE SECOND HEAP TIER. Every registered language is measured now; heap_bound_bytes gates the nine that are not BUDGETED above, and it gates them with one comparison and no floor. A ceiling says 'this index is not too big'. A bound says something narrower and it is the thing that was missing: 'the exclusion still holds' \u2014 this language has not grown an index since it was left out. measure.mjs's MEMORY section states the re-entry condition (if a language ever diverges in what it ASKS its index, it earns a budgeted arm) and until now nothing watched for the divergence; HEAP_LANGS was a hand-maintained list of eight whose two neighbours, LANG_REGISTRY and CONTEXT_LANGS, are both reconciled against a derived predicate in both directions. HEAP_BOUNDED is derived too \u2014 it is LANGS minus HEAP_BUDGETED \u2014 so the two tiers partition the languages and a new one cannot land outside both. WHAT RE-MEASURING FOUND, five runs each, maximum quoted, peak-to-peak in brackets. go 2998464 B [1.0021], dart 7834200 B [1.0006] and kotlin 42802456 B [1.0004] HAD NO STATED REASON AT ALL: the old prose opened 'SIX of the seventeen are deliberately NOT in HEAP_LANGS' against a list of eight of seventeen, and these three were the three nobody counted. All three retain a real per-pass structure (go's PackageDirIndex, dart's basename buckets, kotlin's suffixByStem cascade) and kotlin's 40.82 MiB is above ruby's 39.12 and java's 33.34, both of which carry a full budget. (It read 45.85 MiB when this was written, described here as 'the second-largest reading in this file' \u2014 it was third even then, behind csharp_csproj and php; #2881 later compacted its dirChildren buckets and took 11% off it. Same staleness this paragraph exists to document.) swift 3449216 B [1.0024] and cobol 2320456 B [1.0000] were excluded as 'below the measurement's own noise floor' on readings of 0.29 MB and 0 B at 32000 files; they now read 3.29 MB and 2.21 MB, growing with the corpus (969120 B and 536264 B at 8000). Those old numbers were not wrong when taken \u2014 the ARM changed under them, when #2903's follow-up made every probe resolve a real import and when measureHeap began flattening its corpus \u2014 which is the whole finding: a measurement written into prose is not re-taken, and this file had already gone stale against itself, quoting javascript at 46208832 B four paragraphs after quoting it at 25.51 MiB. rust is the one exclusion that survived unchanged: 16 B at 8000 files and 16 B at 32000, identical in all five runs. typescript 26745296 B, vue 28884016 B and cpp 10023344 B are duplicates of a builder AND of a read pattern: typescript is byte-identical to javascript's 26745296 in four runs of five, cpp is +0.05% of c's 10018816, vue is +8.0% of javascript. HOW THE BOUNDS WERE CHOSEN. Each takes 1.5x its measured maximum, rounded up to the next 100000 B: cobol 3500000 (1.508x), swift 5200000 (1.508x). (This sentence used to list eight, including go, dart, kotlin, typescript, vue and cpp. Those six were promoted to the budgeted tier and their bounds deleted; the numbers stayed here, unread by any gate, and #2881 dutifully updated kotlin's to 64300000 before anyone noticed heap_bound_bytes holds only cobol, swift and rust. A number nothing asserts is a number that rots \u2014 the finding this paragraph is otherwise about.) 1.5x is NOT copied from the ceilings out of habit \u2014 it is the same number for a stated reason, and the reason is not noise: measured peak-to-peak on this box is at most 1.0024, so noise alone would justify 1.05x. What a bound has to survive is a RUNNER change, since heapUsed accounting moves across platforms and Node majors, and this file already fixes that allowance at 50% for exactly this measurement on exactly this arm. Using a second allowance for the same uncertainty on the same number would be two conventions, not more rigour. At 1.5x the bound catches what the re-entry condition is about \u2014 a language growing an index, which costs +85% for one more suffix map and +100% for a duplicate \u2014 and it does NOT catch a duplicate diverging by 8%. That limit is real and is stated rather than hidden: the tight form is a same-process ratio against the arm each duplicate is a duplicate OF, which is the only form immune to the drift the absolute bound has to tolerate. RUST TAKES AN ABSOLUTE BOUND INSTEAD, 1048576 B (1 MiB), because 1.5 x 16 B is 24 B and would fail on the first byte of anything \u2014 a multiplier on a reading that is already nothing is a gate that flakes rather than a gate that bites. 1 MiB is ~65000x the reading and still 2.2x below the smallest real index measured here (cobol's 2.32 MB at the same file count), so it separates 'builds nothing' from 'builds something' with room on both sides. NO FLOOR ON ANY OF THE NINE, and the reason differs by language rather than being uniform. For rust a floor would be a floor on noise. For the other eight the readings are stable enough to floor today, and for kotlin and dart \u2014 larger than budgeted arms \u2014 a floor would be worth having, since a lazily-built map going quiet is exactly how the four budgeted arms once read 0 B. Adding one is a PROMOTION to the budgeted tier, with a ceiling and a recorded reading beside it, not a line here: a floor whose companion ceiling does not exist asserts 'still measuring' against a number nothing else bounds. Recommended next, in order: kotlin, then dart, then go.", + "heap_bound_bytes": { + "cobol": 3500000, + "swift": 5200000, + "rust": 1048576, + "javascript": 1048576, + "typescript": 1048576, + "vue": 1048576 + }, + "heap_floor_fraction": 0.5, + "heap_ratio_budget": 1.25, + "languages": { + "go": { + "small": { + "files": 400, + "imports": 3200, + "resolved": 1153, + "distinct_outcomes": 2913, + "fingerprint": "f2ff032eb7dc4d8f37ecfc9b56d7fc846c5a78cf9a4dfd9f78f4e04588fd0a90" + }, + "large": { + "files": 1600, + "imports": 12800, + "resolved": 4681, + "distinct_outcomes": 11709, + "fingerprint": "19bd34ab249a95fe93843cfb3f5ab84f8215ed0eaccef30ca50298e8dcde1d87" + }, + "deep": { + "files": 400, + "imports": 3200, + "resolved": 1153, + "distinct_outcomes": 2913, + "fingerprint": "67f8fa6657625080e912e417047320928f23f87ecd7a3f45283ad31684752195" + }, + "collide": { + "files": 400, + "imports": 3200, + "resolved": 1153, + "distinct_outcomes": 2868, + "fingerprint": "8d82320278f74c0ebf7ba3e58fd49fde13e9927956f284774cbf39c3b8ca34a8" + }, + "collide_large": { + "files": 1600, + "imports": 12800, + "resolved": 4681, + "distinct_outcomes": 11570, + "fingerprint": "f9c777dc06e32edd30570a5f9316531481941d1f91930e86e2f2bc8dbdf7d6a7" + }, + "fingerprint": "19bd34ab249a95fe93843cfb3f5ab84f8215ed0eaccef30ca50298e8dcde1d87", + "heap": { + "files_small": 8000, + "files_large": 32000, + "path_segments": 14, + "probe": "github.com/org/repo0/pkg/util" + }, + "_measured": { + "collide_ms": 7.15, + "collide_scaling_ratio": 3.465, + "depth_ratio": 0.999, + "scaling_ratio": 0.985, + "small_ms": 1.475 + } + }, + "csharp": { + "small": { + "files": 400, + "imports": 3200, + "resolved": 1153, + "distinct_outcomes": 2844, + "fingerprint": "503dbb3c2fd97d2fa380bc7d77d11706b42878a034a92dbc9a55620f88c53c76" + }, + "large": { + "files": 1600, + "imports": 12800, + "resolved": 4681, + "distinct_outcomes": 11440, + "fingerprint": "1145ce5736bfea02dcd948eac9e1d263d67470f661b7bafb30061887745d2bf5" + }, + "deep": { + "files": 400, + "imports": 3200, + "resolved": 1153, + "distinct_outcomes": 2844, + "fingerprint": "36df304d03f1d0e05e4883e69d91b95728468fdc7c7147eaa357c0ad1d022fd6" + }, + "collide": { + "files": 400, + "imports": 3200, + "resolved": 1153, + "distinct_outcomes": 2844, + "fingerprint": "557c92c82c8960723f0d3ce4bf13f7d661e822d98d9597fe0f5e7c6eaf988f68" + }, + "collide_large": { + "files": 1600, + "imports": 12800, + "resolved": 4681, + "distinct_outcomes": 11440, + "fingerprint": "dbcab955f88895058613b0fb5b9ac81504c7bdc27344e3eab6a6272a14127796" + }, + "fingerprint": "1145ce5736bfea02dcd948eac9e1d263d67470f661b7bafb30061887745d2bf5", + "heap": { + "files_small": 8000, + "files_large": 32000, + "path_segments": 13, + "probe": "Ghost0.Deep.Missing" + }, + "_measured": { + "collide_ms": 5.167, + "collide_scaling_ratio": 2.265, + "depth_ratio": 1.279, + "scaling_ratio": 1.043, + "small_ms": 2.45 + } + }, + "csharp_csproj": { + "small": { + "files": 400, + "imports": 3200, + "resolved": 979, + "distinct_outcomes": 2983, + "fingerprint": "b63d7f2b8078cce64d87a6c93e331db0a6045686cdc0dd70abe3ac2b0bab19d2" + }, + "large": { + "files": 1600, + "imports": 12800, + "resolved": 4064, + "distinct_outcomes": 12029, + "fingerprint": "d9f161410c06c0e73e18ca0f27d6e253402dfe06c9918673a52f04daecb23e36" + }, + "deep": { + "files": 400, + "imports": 3200, + "resolved": 979, + "distinct_outcomes": 2983, + "fingerprint": "7ba2a8ff5151911aa556d809219e9ba5b64c7a022bbfa7f225f08e5d60ab2c62" + }, + "collide": { + "files": 400, + "imports": 3200, + "resolved": 979, + "distinct_outcomes": 2983, + "fingerprint": "fb815bbcfeb4f1049d63f38487ca9e3ada2fcc14ba8478bcc2976e2f632697e1" + }, + "collide_large": { + "files": 1600, + "imports": 12800, + "resolved": 4064, + "distinct_outcomes": 12029, + "fingerprint": "f06217a605d83cf66076a0d55a202dfa13fd4adb55f5605b6154135d0ff745dc" + }, + "fingerprint": "d9f161410c06c0e73e18ca0f27d6e253402dfe06c9918673a52f04daecb23e36", + "heap": { + "files_small": 8000, + "files_large": 32000, + "path_segments": 13, + "probe": "App.Missing0" + }, + "_measured": { + "collide_ms": 25.185, + "collide_scaling_ratio": 1.146, + "depth_ratio": 1.394, + "scaling_ratio": 1.182, + "small_ms": 23.634 + } + }, + "dart": { + "small": { + "files": 400, + "imports": 3200, + "resolved": 1153, + "distinct_outcomes": 2987, + "fingerprint": "318084f48ffa4eeae4a5b7fc25916d4ad78673d92dba62b7e462d1bb87ca553a" + }, + "large": { + "files": 1600, + "imports": 12800, + "resolved": 4681, + "distinct_outcomes": 11875, + "fingerprint": "5151cd2498bd4b7698dc9309e2539977d306f9ba82a388c630c89b51fc4a3187" + }, + "deep": { + "files": 400, + "imports": 3200, + "resolved": 1153, + "distinct_outcomes": 2987, + "fingerprint": "79776ec1c22afa619fd31aeb05dcac567723b436d0461a5782693b9e929f6f74" + }, + "collide": { + "files": 400, + "imports": 3200, + "resolved": 1153, + "distinct_outcomes": 2999, + "fingerprint": "1b145a4c3b41ffc4efa26f74449c3d44646d6d59728163ac896fc0ee25c6d608" + }, + "collide_large": { + "files": 1600, + "imports": 12800, + "resolved": 4681, + "distinct_outcomes": 11948, + "fingerprint": "b7e5303220b8fa64e85c7e17622961018316a10c5ef921a02584864309748b52" + }, + "fingerprint": "5151cd2498bd4b7698dc9309e2539977d306f9ba82a388c630c89b51fc4a3187", + "heap": { + "files_small": 8000, + "files_large": 32000, + "path_segments": 11, + "probe": "package:ext0/src/thing.dart" + }, + "_measured": { + "collide_ms": 1.511, + "collide_scaling_ratio": 2.319, + "depth_ratio": 1.169, + "scaling_ratio": 1.071, + "small_ms": 0.542 + } + }, + "ruby": { + "small": { + "files": 400, + "imports": 3200, + "resolved": 1153, + "distinct_outcomes": 2936, + "fingerprint": "54abc79cc3fd4bfbf341119f3c511c2d64d55556a0c23984032f34e259283e46" + }, + "large": { + "files": 1600, + "imports": 12800, + "resolved": 4681, + "distinct_outcomes": 11786, + "fingerprint": "31804ae9633d51ce7597d886f9c2230fec3448b0aa00d9ab953086e393cd28a9" + }, + "deep": { + "files": 400, + "imports": 3200, + "resolved": 1153, + "distinct_outcomes": 2936, + "fingerprint": "a6dac0609e6800571bcc19ee30818ab571549f275d691f98e4c238aaa4fb1362" + }, + "collide": { + "files": 400, + "imports": 3200, + "resolved": 1153, + "distinct_outcomes": 2936, + "fingerprint": "065fb3a97e1fa01416396b128b1a03d4ea5b49634cb3ffed4b81a07b251c2f2e" + }, + "collide_large": { + "files": 1600, + "imports": 12800, + "resolved": 4681, + "distinct_outcomes": 11786, + "fingerprint": "55a3afc06a48334a6dd2f29c730ae0cfd3a6d54f3013c0853a310af2bbcba277" + }, + "fingerprint": "31804ae9633d51ce7597d886f9c2230fec3448b0aa00d9ab953086e393cd28a9", + "heap": { + "files_small": 8000, + "files_large": 32000, + "path_segments": 11, + "probe": "gem0/missing/thing" + }, + "_measured": { + "collide_ms": 20.732, + "collide_scaling_ratio": 1.119, + "depth_ratio": 1.257, + "scaling_ratio": 1.133, + "small_ms": 19.994 + } + }, + "kotlin": { + "small": { + "files": 400, + "imports": 3200, + "resolved": 1153, + "distinct_outcomes": 2868, + "fingerprint": "1b3cd628bb069a3388af655c301984635cdc00a5d54b0e0ffd2b12849eb7fe40" + }, + "large": { + "files": 1600, + "imports": 12800, + "resolved": 4681, + "distinct_outcomes": 11512, + "fingerprint": "8763c5ea18a25663deb30b2e065d2fde1b0df7bad5b19e46a9ea0d30d8780995" + }, + "deep": { + "files": 400, + "imports": 3200, + "resolved": 1153, + "distinct_outcomes": 2868, + "fingerprint": "b71c0f96506a5775f2a92e035c203a085bb211f93bc45be4183ebefc87ba412f" + }, + "collide": { + "files": 400, + "imports": 3200, + "resolved": 1153, + "distinct_outcomes": 2744, + "fingerprint": "d3453a77f76e4cc59d3479af80d05b358abe9c1277a6aa92611d0f20cac80eea" + }, + "collide_large": { + "files": 1600, + "imports": 12800, + "resolved": 4681, + "distinct_outcomes": 10961, + "fingerprint": "335eaaa54b2663caee5587b66d96d0669106e5a10e4d70ae2ee67443d939890e" + }, + "fingerprint": "8763c5ea18a25663deb30b2e065d2fde1b0df7bad5b19e46a9ea0d30d8780995", + "heap": { + "files_small": 8000, + "files_large": 32000, + "path_segments": 18, + "probe": "com.ghost0.deep.Missing" + }, + "_measured": { + "collide_ms": 2.448, + "collide_scaling_ratio": 1.081, + "depth_ratio": 1.813, + "scaling_ratio": 1.097, + "small_ms": 2.62 + } + }, + "php": { + "small": { + "files": 400, + "imports": 3200, + "resolved": 1153, + "distinct_outcomes": 2867, + "fingerprint": "3bb31eb4cd444b240e56b151007004f2f810bb5ee3f111b7b57738ea17c819b2" + }, + "large": { + "files": 1600, + "imports": 12800, + "resolved": 4681, + "distinct_outcomes": 11517, + "fingerprint": "1c313a83acf55ec58994fc55016754488ae2d352aefaeb84a2e3ecbb928b3479" + }, + "deep": { + "files": 400, + "imports": 3200, + "resolved": 1153, + "distinct_outcomes": 2867, + "fingerprint": "94bdf5cb27b7a1bb0d24e2ba0157ba71dcf61ec726059dd5a0462377a1d0180b" + }, + "collide": { + "files": 400, + "imports": 3200, + "resolved": 1153, + "distinct_outcomes": 2695, + "fingerprint": "61038746f1386bfc747784e7ce6bc52522bc4585259668e22e29f93291b0b3a5" + }, + "collide_large": { + "files": 1600, + "imports": 12800, + "resolved": 4681, + "distinct_outcomes": 10845, + "fingerprint": "c41d254ce8703339576e5642f67dfef81c97445c75db184bb65dc26b4d4715ef" + }, + "fingerprint": "1c313a83acf55ec58994fc55016754488ae2d352aefaeb84a2e3ecbb928b3479", + "heap": { + "files_small": 8000, + "files_large": 32000, + "path_segments": 14, + "probe": "Vendor0\\Ghost\\Missing" + }, + "context": { + "target": "App\\Ns0\\Dup", + "with_context": "src/App/Ns0/Helpers.php", + "without_context": "src/App/Ns0/Dup.php" + }, + "_measured": { + "collide_ms": 35.91, + "collide_scaling_ratio": 1.068, + "depth_ratio": 1.268, + "scaling_ratio": 1.079, + "small_ms": 34.023 + } + }, + "java": { + "small": { + "files": 400, + "imports": 3200, + "resolved": 1153, + "distinct_outcomes": 2868, + "fingerprint": "8e347c485c47a1a9f67ae0183b68327b40b1e5d3510160fda2a2fe54c0f8a453" + }, + "large": { + "files": 1600, + "imports": 12800, + "resolved": 4681, + "distinct_outcomes": 11512, + "fingerprint": "6773de19833d9936cb098c5897b5a44ba19b07bd8990af5a8b70fc03b309794f" + }, + "deep": { + "files": 400, + "imports": 3200, + "resolved": 1153, + "distinct_outcomes": 2868, + "fingerprint": "0ba22e27f87395535533bac3270c481ea50aa4cd7eba4958325ef792f308bfc0" + }, + "collide": { + "files": 400, + "imports": 3200, + "resolved": 1153, + "distinct_outcomes": 2744, + "fingerprint": "2ab215bd5109f13c0f15513c3bf578ca467896dd28d489c9ac4477d2b647c39f" + }, + "collide_large": { + "files": 1600, + "imports": 12800, + "resolved": 4681, + "distinct_outcomes": 10961, + "fingerprint": "31e762ed838528a426e5cc4510956661fc2aef7aa7c743291d75fcec54e46235" + }, + "fingerprint": "6773de19833d9936cb098c5897b5a44ba19b07bd8990af5a8b70fc03b309794f", + "heap": { + "files_small": 8000, + "files_large": 32000, + "path_segments": 18, + "probe": "com.google.common.vendor0.Missing" + }, + "context": { + "target": "com.example.model.User", + "with_context": "weird/path/User.java", + "without_context": "" + }, + "_measured": { + "collide_ms": 5.219, + "collide_scaling_ratio": 1.07, + "depth_ratio": 0.978, + "scaling_ratio": 1.0, + "small_ms": 5.674 + } + }, + "cobol": { + "small": { + "files": 400, + "imports": 3200, + "resolved": 1153, + "distinct_outcomes": 2941, + "fingerprint": "e5bf9c2a74cad64df6ac18299b56fc9139943baec6118036b2e765ac3d4252f2" + }, + "large": { + "files": 1600, + "imports": 12800, + "resolved": 4681, + "distinct_outcomes": 11791, + "fingerprint": "f192ca7a9e87eb05f03893ffc64252a8aba2c638604dcf449150fb9b5fdd989e" + }, + "deep": { + "files": 400, + "imports": 3200, + "resolved": 1153, + "distinct_outcomes": 2941, + "fingerprint": "c690c6abc5c7aab31f27a97e5ef25d32daa483d9c08ceb48bc0b85ac406e6e37" + }, + "collide": { + "files": 400, + "imports": 3200, + "resolved": 1153, + "distinct_outcomes": 2827, + "fingerprint": "c487db2efbf7a683674de84430d88e7a4c75e9427a53cccae934fd8440b85d87" + }, + "collide_large": { + "files": 1600, + "imports": 12800, + "resolved": 4681, + "distinct_outcomes": 11393, + "fingerprint": "8bc1d506b54e800d060eb4c92fca01c5b06c5a7130248dc1a79521bdea53982a" + }, + "fingerprint": "f192ca7a9e87eb05f03893ffc64252a8aba2c638604dcf449150fb9b5fdd989e", + "heap": { + "files_small": 8000, + "files_large": 32000, + "path_segments": 11, + "probe": "VENDOR0" + }, + "_measured": { + "collide_ms": 0.197, + "collide_scaling_ratio": 1.046, + "depth_ratio": 0.885, + "scaling_ratio": 0.936, + "small_ms": 0.286 + } + }, + "swift": { + "small": { + "files": 400, + "imports": 3200, + "resolved": 1153, + "distinct_outcomes": 2913, + "fingerprint": "91c5172b994270807f7fdcf80ac545edd50d3dc87c67290c9aabaed8bb65d594" + }, + "large": { + "files": 1600, + "imports": 12800, + "resolved": 4681, + "distinct_outcomes": 11709, + "fingerprint": "16f80a95e52ad1057cf369b7816ce704684fc5b5663a39f23c9149e9221c6170" + }, + "deep": { + "files": 400, + "imports": 3200, + "resolved": 1153, + "distinct_outcomes": 2913, + "fingerprint": "22ef94a6e087d7ac909733ef32da2ddf292fa8c82b05b70b5910c307fceca1b4" + }, + "collide": { + "files": 400, + "imports": 3200, + "resolved": 1153, + "distinct_outcomes": 2606, + "fingerprint": "4d2c41ba5f8230ab6b9faade80f293ddde153f4ff1d5dd4473f1cd699d2808fd" + }, + "collide_large": { + "files": 1600, + "imports": 12800, + "resolved": 4681, + "distinct_outcomes": 10184, + "fingerprint": "d27b6070f5ad93020bf762221e798c382327406f3f2cca2f7aab3e9ac56faef4" + }, + "fingerprint": "16f80a95e52ad1057cf369b7816ce704684fc5b5663a39f23c9149e9221c6170", + "heap": { + "files_small": 8000, + "files_large": 32000, + "path_segments": 13, + "probe": "ExternalPkg0" + }, + "_measured": { + "collide_ms": 0.819, + "collide_scaling_ratio": 3.454, + "depth_ratio": 1.496, + "scaling_ratio": 1.063, + "small_ms": 0.385 + } + }, + "rust": { + "small": { + "files": 400, + "imports": 3200, + "resolved": 979, + "distinct_outcomes": 2844, + "fingerprint": "6a2435149e055e6903aab2dd3fa2a0986d8d1d7933bccb0b7f311448372f548c" + }, + "large": { + "files": 1600, + "imports": 12800, + "resolved": 4064, + "distinct_outcomes": 11440, + "fingerprint": "442f9124ebeb052557413d9dbb7c5e467ffc69da6357d0d8d9bcd2232ba27092" + }, + "deep": { + "files": 400, + "imports": 3200, + "resolved": 979, + "distinct_outcomes": 2844, + "fingerprint": "aa32ea032a548df09554c40a8b0679f11bc4d4cefc1dcad941283036dbae7c8e" + }, + "collide": { + "files": 400, + "imports": 3200, + "resolved": 979, + "distinct_outcomes": 2844, + "fingerprint": "4d1e28e318a04ee0e2065b5f9a2c971765e2f60310b0e612cfd327c76b6344b6" + }, + "collide_large": { + "files": 1600, + "imports": 12800, + "resolved": 4064, + "distinct_outcomes": 11440, + "fingerprint": "3de5234747493741abbf170e164ef80188092747150cfb39ef2cacd5658effc0" + }, + "fingerprint": "442f9124ebeb052557413d9dbb7c5e467ffc69da6357d0d8d9bcd2232ba27092", + "heap": { + "files_small": 8000, + "files_large": 32000, + "path_segments": 12, + "probe": "ghost0::Missing" + }, + "_measured": { + "collide_ms": 4.767, + "collide_scaling_ratio": 1.042, + "depth_ratio": 1.371, + "scaling_ratio": 1.097, + "small_ms": 2.523 + } + }, + "python": { + "small": { + "files": 400, + "imports": 3200, + "resolved": 845, + "distinct_outcomes": 2867, + "fingerprint": "7a458789903c904968af8f9f851656ea33c1c446959ca79a0222ec65d3e809ed" + }, + "large": { + "files": 1600, + "imports": 12800, + "resolved": 3556, + "distinct_outcomes": 11517, + "fingerprint": "98f99b9eaa3fcc3c58c4be0116853789c8e1299c187088a8b04d28f1885c944a" + }, + "deep": { + "files": 400, + "imports": 3200, + "resolved": 845, + "distinct_outcomes": 2867, + "fingerprint": "c099814a70bbb63471fecc6e9527632e83b9954618963e77648f893ddfe65286" + }, + "collide": { + "files": 400, + "imports": 3200, + "resolved": 845, + "distinct_outcomes": 2867, + "fingerprint": "038c097cb628f6c65c1a228a5df3bb29a81eb3d4d7f297cfe86c3c4c6323c7c0" + }, + "collide_large": { + "files": 1600, + "imports": 12800, + "resolved": 3556, + "distinct_outcomes": 11517, + "fingerprint": "94cd4994ce690db215028ff42f06aa1fd142bd26d62a290f4849bbff36c294f5" + }, + "fingerprint": "98f99b9eaa3fcc3c58c4be0116853789c8e1299c187088a8b04d28f1885c944a", + "heap": { + "files_small": 8000, + "files_large": 32000, + "path_segments": 11, + "probe": "vendor0.deep.missing" + }, + "context": { + "target": "pkg", + "with_context": "pkg/__init__.py", + "without_context": "pkg/X.py" + }, + "_measured": { + "collide_ms": 4.521, + "collide_scaling_ratio": 1.085, + "depth_ratio": 1.563, + "scaling_ratio": 1.144, + "small_ms": 4.431 + } + }, + "javascript": { + "small": { + "files": 400, + "imports": 3200, + "resolved": 1153, + "distinct_outcomes": 2867, + "fingerprint": "4a80c7b940a6c39d0ebd109980469d2398b417a4479f5d03f35abc482fa76122" + }, + "large": { + "files": 1600, + "imports": 12800, + "resolved": 4681, + "distinct_outcomes": 11517, + "fingerprint": "827ac421e8958ff686b2877efa60fbaf1a1c661698cde8f0c8c931919e0f35bd" + }, + "deep": { + "files": 400, + "imports": 3200, + "resolved": 1153, + "distinct_outcomes": 2867, + "fingerprint": "4d0551e044b19bccd9775879b1425f3adecf0cdf1e90a5d8a9607ef0a51af880" + }, + "collide": { + "files": 400, + "imports": 3200, + "resolved": 1153, + "distinct_outcomes": 2871, + "fingerprint": "45b4bb3b2f9797e21029ef1eef7247702813cac39ac430f9a999d3326596a7a1" + }, + "collide_large": { + "files": 1600, + "imports": 12800, + "resolved": 4681, + "distinct_outcomes": 11548, + "fingerprint": "304ed93e83b397b4aa9520750739ffdcf3a9b353ae0ad8fe88c0368c63c85533" + }, + "fingerprint": "827ac421e8958ff686b2877efa60fbaf1a1c661698cde8f0c8c931919e0f35bd", + "heap": { + "files_small": 8000, + "files_large": 32000, + "path_segments": 11, + "probe": "vendor0/lib/missing" + }, + "_measured": { + "collide_ms": 22.96, + "collide_scaling_ratio": 1.077, + "depth_ratio": 1.213, + "scaling_ratio": 1.093, + "small_ms": 22.762 + } + }, + "typescript": { + "small": { + "files": 400, + "imports": 3200, + "resolved": 1153, + "distinct_outcomes": 2867, + "fingerprint": "fe8fcf81efa0fa3a77894edc6bd0b9ec4ff0bf92f24e116ddf108dc70cdcd97e" + }, + "large": { + "files": 1600, + "imports": 12800, + "resolved": 4681, + "distinct_outcomes": 11517, + "fingerprint": "24e36ebfc1c482643812f1ef400e8cb387dae11954531407e113d4e6c3fa2a6d" + }, + "deep": { + "files": 400, + "imports": 3200, + "resolved": 1153, + "distinct_outcomes": 2867, + "fingerprint": "e9faf31f6b1299394760e27ff9e04af1a8b4ddca0370db62fd2a59af7a4f5d05" + }, + "collide": { + "files": 400, + "imports": 3200, + "resolved": 1153, + "distinct_outcomes": 2871, + "fingerprint": "dbb68b66e8d136140f4a7bc024c97f5de02d1c8b6f3a07efa271dcb73366eeb5" + }, + "collide_large": { + "files": 1600, + "imports": 12800, + "resolved": 4681, + "distinct_outcomes": 11548, + "fingerprint": "3129a6f1f25bd5568058682f99812ad35e38231b92025184344b74b86fa2e910" + }, + "fingerprint": "24e36ebfc1c482643812f1ef400e8cb387dae11954531407e113d4e6c3fa2a6d", + "heap": { + "files_small": 8000, + "files_large": 32000, + "path_segments": 11, + "probe": "vendor0/lib/missing" + }, + "_measured": { + "collide_ms": 22.324, + "collide_scaling_ratio": 1.059, + "depth_ratio": 1.25, + "scaling_ratio": 1.079, + "small_ms": 20.882 + } + }, + "vue": { + "small": { + "files": 400, + "imports": 3200, + "resolved": 1153, + "distinct_outcomes": 2867, + "fingerprint": "88e85b85f7158cc87d770f119c992d71801c4c692da13064cbc8b95718517fe4" + }, + "large": { + "files": 1600, + "imports": 12800, + "resolved": 4681, + "distinct_outcomes": 11517, + "fingerprint": "4d62ac179e4371d3f41b691cea725b90272f72da625e914d2f16d323a1e940c8" + }, + "deep": { + "files": 400, + "imports": 3200, + "resolved": 1153, + "distinct_outcomes": 2867, + "fingerprint": "786c801ad824c3e49f05aaf37a63c3bfe7dbe6dd2fb44cfef180099f4fcdd401" + }, + "collide": { + "files": 400, + "imports": 3200, + "resolved": 1153, + "distinct_outcomes": 2871, + "fingerprint": "8a01ee06ddf2eeb0db72dd8b73544180bf48d8cb82b6e73b1969233842553636" + }, + "collide_large": { + "files": 1600, + "imports": 12800, + "resolved": 4681, + "distinct_outcomes": 11548, + "fingerprint": "841b48a4c46fc56cd700ff7c07a515da1139cc4528d479a47876cbed291d91a4" + }, + "fingerprint": "4d62ac179e4371d3f41b691cea725b90272f72da625e914d2f16d323a1e940c8", + "heap": { + "files_small": 8000, + "files_large": 32000, + "path_segments": 11, + "probe": "vendor0/lib/Missing.vue" + }, + "_measured": { + "collide_ms": 24.453, + "collide_scaling_ratio": 1.095, + "depth_ratio": 1.384, + "scaling_ratio": 1.071, + "small_ms": 21.765 + } + }, + "c": { + "small": { + "files": 400, + "imports": 3200, + "resolved": 1153, + "distinct_outcomes": 2863, + "fingerprint": "4dd05ba9a0c6731d449ec555f56d2f7cb05cdcdaa01a8acc242e3709d305184f" + }, + "large": { + "files": 1600, + "imports": 12800, + "resolved": 4681, + "distinct_outcomes": 11512, + "fingerprint": "70d6064fdc08e86036ced58393585afc3693ee527f00983847299e390b413d87" + }, + "deep": { + "files": 400, + "imports": 3200, + "resolved": 1153, + "distinct_outcomes": 2863, + "fingerprint": "43707e57b1079e7f01cc84ea5ab891cf77c395e2d52e7fbb7eee30c058c1d667" + }, + "collide": { + "files": 400, + "imports": 3200, + "resolved": 1153, + "distinct_outcomes": 2695, + "fingerprint": "982b925fdbbc59d05ae52be1f405f3cbb6fd554390ee38eeff869df9316ffaf2" + }, + "collide_large": { + "files": 1600, + "imports": 12800, + "resolved": 4681, + "distinct_outcomes": 10845, + "fingerprint": "70b30b89ae671208bd836693fbc87d6059656cf347b9397d3b905d24e31912cf" + }, + "fingerprint": "70d6064fdc08e86036ced58393585afc3693ee527f00983847299e390b413d87", + "heap": { + "files_small": 8000, + "files_large": 32000, + "path_segments": 11, + "probe": "vendor0/missing.h" + }, + "_measured": { + "collide_ms": 2.924, + "collide_scaling_ratio": 2.575, + "depth_ratio": 1.938, + "scaling_ratio": 1.033, + "small_ms": 1.649 + } + }, + "cpp": { + "small": { + "files": 400, + "imports": 3200, + "resolved": 1153, + "distinct_outcomes": 2863, + "fingerprint": "6c199e829226c1cdd86e74611b159ff2e83d4c17da2a72552503a7c4518182e4" + }, + "large": { + "files": 1600, + "imports": 12800, + "resolved": 4681, + "distinct_outcomes": 11512, + "fingerprint": "191bddd6f77a10ab6bced04e5c5f55e0af4481563ef3cb57e08c5dfa6c86454e" + }, + "deep": { + "files": 400, + "imports": 3200, + "resolved": 1153, + "distinct_outcomes": 2863, + "fingerprint": "5603c080739321bb6204c153ec6214dc185e436b5aadb5ec13738e2a759a84f3" + }, + "collide": { + "files": 400, + "imports": 3200, + "resolved": 1153, + "distinct_outcomes": 2695, + "fingerprint": "cbf2fece6338725205beaf87058ce32ae1ba0860d14cedd29b7904b2f3a63726" + }, + "collide_large": { + "files": 1600, + "imports": 12800, + "resolved": 4681, + "distinct_outcomes": 10845, + "fingerprint": "094f7fe2aace191d7e53c2d4ecd7e063cf15bd66643e6201fd46e45b6b63ab6e" + }, + "fingerprint": "191bddd6f77a10ab6bced04e5c5f55e0af4481563ef3cb57e08c5dfa6c86454e", + "heap": { + "files_small": 8000, + "files_large": 32000, + "path_segments": 11, + "probe": "vendor0/missing.hpp" + }, + "_measured": { + "collide_ms": 3.035, + "collide_scaling_ratio": 2.566, + "depth_ratio": 2.064, + "scaling_ratio": 1.167, + "small_ms": 1.626 + } + } + }, + "_blind_spot": "MEASURED, so nobody has to rediscover it: a full workspace scan reintroduced on 1-in-32 imports passes EVERY arm here \u2014 dart scored 1.458 scaling and 1.736 ms against the 1.8 budget and 4 ms ceiling of an earlier revision. At 1-in-8 the scaling arm catches it (2.414). The gate that NARROWS this is not a timing gate at all: test/unit/scope-resolution/import-target-index-parity.test.ts counts iterations of the file-set Set and reads 14 instead of 1 for that same 1-in-32 mutation, deterministically and for all five languages. It does NOT close it. The counter watches the Set, and the resolvers no longer read the Set \u2014 they read materialized copies of the same file list: WorkspaceFileIndex.normalized and .all (C#, Ruby), Dart's byBasename buckets, and PackageDirIndex.filesByDir (Go, C#). A 1-in-32 scan over any of those three touches the Set zero extra times, so it passes the parity test AND passes --check. Closing it would take an iteration counter on the materialized arrays themselves. Read the two gates together; tightening these ceilings toward the noise floor to chase that case would only buy flaky CI. CONFIRMED THE HARD WAY by PR #2911: JavaScript resolution was scanning ImportPassCache.normalizedFileList on every import \u2014 a materialized array, not the Set \u2014 at 25972 us per import at 8000 files, and no instrument on the #2901-#2909 branch could see it. It took a differential parity test over 211200 old-vs-new pairs to find. The arms added here would have caught THAT one on absolute ms (85 ms budget against a 20 ms arm; the unindexed resolver costs ~83000 ms on the same corpus), which is the argument for gating every registered language rather than only the ones a PR happens to touch. THE SECOND BLIND SPOT IS CLOSED, and this records what closing it changed. This harness used to call the inner resolvers with the NO-CONTEXT shape: run.ts calls provider.resolveImportTarget with five arguments, the fifth being { parsedFiles, parsedImport }, and resolveOne supplied three. resolveOne now makes the production call, newPass mints the ParsedFile[] FIRST and derives the path set from it exactly as run.ts does, and both legs behind the argument run on every import of their arms \u2014 PHP's named/alias function-or-const leg over filesByDirectory(context.parsedFiles), whose memo defeated measures 197.0 us -> 9976.2 us per import (50.6x), and Python's from-import submodule-precedence branch, the only spelling that reads context.parsedFiles at all. Fifteen of the seventeen arms cannot observe a context (their hooks declare three or four parameters) and are handed none, so their numbers did not move; which two CAN is now reconciled against SCOPE_RESOLVERS' hook arity rather than asserted in prose. NOTHING ELSE IN THIS FILE COULD HAVE GATED IT, which is why the context arm exists: on this corpus the leg AGREES with the cascade for every import, so all ten of PHP's and Python's fingerprints, their resolved counts and their distinct_outcomes are unchanged; a dropped context makes the timing arms FASTER and no arm here has a lower bound on ms; and the heap floor (0.5 x 49573840 = 24.8 MB) still passes the 37576816 B a no-context PHP pass reads. The arm is one import per language resolved through resolveOne twice, with and without the pass's parsedFiles, whose two answers must DIFFER and must both match what is recorded. WHAT REMAINS UNMEASURED, narrowed rather than deleted: Python's parsedFileByPath memo is exercised by the five timing arms and cannot be reached by the heap arm at all, because retainedPassBytes requires a probe that MISSES while every path that builds that memo returns a non-null packageTarget \u2014 so no ceiling bounds that Map (one pointer per parsed file, O(files), no depth term) and the contract test's count gate is what holds it to one build per pass. PHP's leg is measured with NO composer.json, so namespaceDirectories only ever returns the directory of an already-resolved file and the PSR-4 mapping branch stays unreached, exactly as csharp cannot reach the csproj leg; closing that is a second PHP arm on the csharp_csproj precedent, not a parameter. And the const tail of PHP's leg is a different ANSWER at the same cost \u2014 it runs the identical candidate gather and localDefs filter and diverges in the last two lines \u2014 so it is gated by count in test/unit/scope-resolution/import-target-index-reuse.contract.test.ts, which stays the gate to read alongside this file.", + "_depth_budget_note_2953": "javascript/typescript/vue moved from 2.0-2.1 to ~2.2 in #2953 and their budgets were raised to 2.6, which is a real shift with an understood cause rather than a loosened guard. Declared resolution never walks path components, so the deep arm's uniform d0/../d15/ prefix reaches these resolvers as the tsconfig baseUrl (see tsBaseUrlFor in measure.mjs) and every candidate string carries it: resolveFile probes ~11 extensions plus their /index forms, and hashing a 60-character path costs more than hashing a 12-character one. The growth is linear in path LENGTH and independent of file COUNT, which is what the ratio exists to bound - a resolver that started walking the corpus again would move scaling_ratio, not just this. Measured over three runs on a loaded box: js 2.109/2.257/2.240, ts 2.129/2.222/2.467, vue 2.116/2.151/2.102.", + "_heap_bound_note_2953": "javascript, typescript and vue moved from heap_reading_bytes/heap_ceiling_bytes to heap_bound_bytes in #2953. They retained 26745296 B (js, ts) and 28884016 B (vue) at 32000 files for a per-pass SuffixIndex over the whole file list; they now build no per-pass structure at all and read 0-16 B, because declared resolution derives nothing from the file set. That is a real saving rather than an arm that stopped measuring - the distinction this floor exists to make - and the evidence it is real is that the resolver fingerprints did NOT move: the same corpus resolves to the same targets, once the config it always implied is passed explicitly. The 1048576 B bound is rust's, chosen the same way: far above a 16 B reading, far below the index whose return it must catch." +} diff --git a/gitnexus/bench/import-target/measure.mjs b/gitnexus/bench/import-target/measure.mjs new file mode 100644 index 000000000..61aed7775 --- /dev/null +++ b/gitnexus/bench/import-target/measure.mjs @@ -0,0 +1,2923 @@ +/** + * Build-free scaling + identity bench for EVERY import-target resolver in + * `SCOPE_RESOLVERS` — the registry decides which, not a list kept here, and the + * `--check` inventory arm at the foot of this file fails when the two disagree + * — over ONE shared corpus so the arms are directly comparable. One arm per + * registered language, plus a second `csharp` arm carrying csproj configs + * (#2902), so there is one more arm than there are languages. + * + * NO LANGUAGE IS OMITTED, and that is the point of the list rather than an + * accident of it. Nine of these arms (go, csharp, csharp_csproj, dart, ruby, + * kotlin, php, java, cobol) were added as their own O(imports × files) scans + * were indexed away — #2877/#2878/#2879/#2880, #2872, #2901, #2902, #2908 — and + * the bench is the forward guard on each. The eight added alongside them + * (swift, rust, python, javascript, typescript, vue, c, cpp) resolve imports + * through the same registered hook with the same per-run memoized indexes, and + * were ungated: nothing pinned their output and nothing pinned their scaling. + * One of them was not hypothetical — JavaScript reached `suffixResolve` with no + * index at all and measured 25 972 µs per import at 8000 files (PR #2911) — + * which is exactly the class of defect the other seven were one commit away + * from. + * + * A C or C++ `#include` is an import site for this purpose and is gated like + * every other registered language. See `newPass` for the one structural thing + * those two need that no other language does. + * + * Kotlin also has `bench/kotlin-import-target/`, and this does not replace it: + * that bench fingerprints both file-set iteration orders and probes the + * four-tier cascade shape by shape, which this corpus does not. What Kotlin + * gains here is a second corpus and the arms below that its own bench predates. + * + * Each of the first nine resolvers answered its lookups with a full + * `allFilePaths` scan per import before its fix, so import resolution cost + * O(imports × files): + * + * - Go: `findRootPackageFiles` / `findAllFilesInPkgDir`, the latter once per + * path segment on the GOPATH fallback — several full scans per import; + * - C#: the no-csproj leg took the raw Set past the memoized index the csproj + * leg was already using — up to eight passes for a four-segment `using`; + * - Dart: one full scan per candidate path, and for an external package both + * candidates miss, so both always ran to completion; + * - Ruby: a complete `buildSuffixIndex` rebuilt and discarded per `require`; + * - PHP: two materialized arrays per import and then no index at all, which + * dropped `suffixResolve` onto a linear `findIndex` — one full pass per path + * part per extension, and there are ~50 extensions (96.40 ms per import at + * 20k files, now 0.036 ms); + * - Java: one scan for the direct match plus one more per stripped package + * prefix, and a JDK or third-party import runs the loop to the end (8.05 ms + * per import, now 0.62 ms); + * - COBOL: two scans per `COPY` — one per extension tier — each calling + * `extname` + `basename` + `toUpperCase` on every path, both always running + * to completion because vendor copybooks live outside the repo (3879 µs per + * import, now 10.5 µs); + * - C# csproj: the namespace-directory fallback re-scanned + * `normalizedFileList` per import per matching config (1103 µs, now 7.6 µs). + * `csharp` here builds its context with NO `csharpConfigs`, so it can never + * reach that leg — `csharp_csproj` is the same corpus with the configs + * supplied, and it exists because without it #2902 ships unmeasured. + * + * The eight added afterwards are not a second class of arm — they carry the + * same five timing arms, the same per-scale fingerprint and shape gates and the + * same budgets. What differs is what each one's cost is a function of, because + * that decides which arm can actually fail for it: + * + * - swift: `getSwiftModuleIndex` buckets a file under EVERY interior + * directory segment, so `Sources/Models/User.swift` answers to `Sources` + * and to `Models`. A miss is a Map miss and flat; a HIT returns the whole + * module bucket minus the importer, so its cost is the BUCKET size. Nothing + * in the unique layout produces a large bucket, which is why its collide + * arm is four modules instead of `dirs` of them (`SWIFT_COLLIDE_MODULES`); + * measured 3.28 there against 0.90 on file count. + * - rust: probes candidate paths with `allFilePaths.has(...)` and never + * searches, so its cost is O(path SEGMENTS) and is provably flat in the + * file count — measured 1.10 scaling, 1.06 collide scaling. That flatness + * IS the assertion, and it is why its collide arm is a deep module tree + * with ~2x the `::` segments rather than a shared-leaf layout: a collide + * arm built on file count would have been an arm that cannot fail. Note + * that `buildRustModuleIndex` lives on a DIFFERENT hook + * (`qualified-call.ts::moduleIndexFor`) and is not on this path at all. + * - python: `getPythonFileIndex` is keyed and flat on both file count and + * bucket cardinality (1.11 / 1.10), but `hasRepoCandidate` and + * `resolveAbsoluteFromFiles` each rebuild one ancestor prefix per directory + * component of the IMPORTER, so per-import cost is quadratic in path depth: + * measured depth_ratio 7.39, by far the largest here, and the reason its + * depth budget is 11 rather than the ~2 most languages carry. + * - javascript, typescript, vue: one resolver (`resolveTsModule`) behind + * three adapters, so the three corpora are the same shape and differ only + * in what actually differs — the extension list (`.js` vs `.ts`) and which + * config leg the arm exercises (`tsBaseUrlConfig` vs `vueTsconfig`). All + * three are miss-dominated bare specifiers. + * + * What they measure CHANGED with #2953. The leg used to be `suffixResolve`, + * a repo-wide search for a path ending in the specifier; these three no + * longer have it, and resolve only against a declared tsconfig mapping or a + * package manifest. Two consequences the numbers show: + * + * - the arms need a `resolutionConfig` to resolve anything at all. With + * none they all reported `resolved: 0` — every import correctly + * external — while still printing a clean scaling ratio, which is a + * bench measuring an empty branch and passing exactly like one + * measuring a full one. + * - `depth_ratio` is now structurally flat for them, and that is the + * result rather than a weakened arm: declared resolution never walks + * path components, so the `deep` arm's uniform prefix reaches the + * config (see `tsBaseUrlFor`) and its cost is the same keyed lookup the + * other arms pay. + * - c, cpp: `resolveCppImportTarget` delegates to `resolveCImportTarget`, so + * the two share a resolver and differ in extension set and in which adapter + * builds the augmented set. Cost is a basename bucket walk with a + * depth-then-lexicographic tie-break, so the collide arm (a `mod{n}` header + * in every service's `include/`) is where it grows: 2.54 / 2.64 against + * 1.06 on file count. + * + * Two properties of the corpus are load-bearing and must not be "simplified": + * + * 1. **Most imports are unresolvable.** In real source the majority of imports + * name the stdlib or a third-party package, and those run every leg of the + * cascade to completion before returning null — the fast paths never fire. + * A corpus of mostly-resolving imports measures the wrong half of the + * function and would score a reintroduced scan as linear. + * 2. **Import count scales WITH file count.** The regression is quadratic in + * `imports × files`; holding imports fixed while files grow would halve the + * exponent and let a per-import scan pass the budget. + * + * Reports per language and scale: + * - `ms`: fastest of REPS full passes, INCLUDING the one-time index build — + * hiding the build would let an index that is itself quadratic pass; + * - `scaling_ratio` `(t_large/t_small)/(LARGE/SMALL)`: ~1.0 linear, ~4.x + * quadratic at this scale gap; + * - `depth_ratio` `t_deep/t_small` at a FIXED file count with ~6x the path + * components. `scaling_ratio` divides the file count out, so it is + * scale-invariant and structurally cannot see a cost that grows with path + * DEPTH instead — and `buildSuffixIndex` (C#, Ruby, PHP, Java) and Kotlin's + * `suffixByStem` each emit one entry per component. Go, Dart and COBOL, + * whose indexes are depth-free (COBOL's are keyed on the basename and + * nothing else), sit at ~0.9-1.0; the others sit legitimately above 1.0, + * which is why the budget is per language. Python is the extreme and the + * reason the spread is worth a per-language number at all: its index is + * depth-free, but `hasRepoCandidate` and `resolveAbsoluteFromFiles` rebuild + * an ancestor prefix per importer directory component ON EVERY IMPORT, so + * the RESOLVER, not the index, is quadratic in depth — 7.39; + * - `collide_scaling_ratio`, the same measurement on a corpus whose + * directories SHARE their last segment and whose files share basenames — + * see the `collide` section below; + * - `heap` (all 17): retained bytes of the per-pass import index, read by + * resolving one real import — see the `heap` section below. Eight carry a + * ceiling, a floor and a ratio; the other nine carry an upper bound only; + * - a sha256 over every distinct `fromFile | target → result`, as the + * correctness gate. The tie-break-level proof that this PR's index + * reproduces the scans lives in + * `test/unit/scope-resolution/import-target-index-parity.test.ts`, which + * diffs against verbatim copies of the pre-change implementations; this + * fingerprint is the forward guard that keeps the output pinned from here. + * Every scale's fingerprint is asserted, not just `large`'s: the arms + * differ only in layout and padding, so a per-scale-only defect (a resolver + * bug that corrupts deep paths, or a corpus edit that quietly deletes the + * depth padding) moves no asserted count and would otherwise print PASS. + * + * `--check` adds arms that no ratio can carry: + * - the corpus SHAPE (files, imports, resolved, distinct outcomes, per scale), + * so a future edit cannot quietly shrink the corpus below the sizes that + * make the timing arms meaningful and still print PASS. The `deep` and + * `collide` arms must also resolve exactly what `small` resolves — padding + * and re-layout were supposed to change path depth and directory naming and + * nothing else; + * - `deep.fingerprint !== small.fingerprint` and + * `collide.fingerprint !== small.fingerprint`, so those two arms' EFFECT is + * pinned rather than only their output. Both are count-neutral by + * construction, so neutering either one (`DEEP_PAD = 0`, a `collideDir` + * that forwards to `uniqueDir`) leaves every asserted number untouched; + * comparing the arms to `small` is the only thing that notices; + * - `small_ms_ceiling` and `collide_ms_ceiling`, ABSOLUTE bounds, because a + * constant-factor regression that grows both scale arms equally passes + * every ratio; + * - a heap FLOOR beside every heap ceiling, and a presence check in front of + * every timing budget. Both exist because the same failure has now happened + * twice in this file's short life: an arm that stops measuring passes. A + * lazy `buildSuffixIndex` made four heap arms read 0 B, and 0 B is under + * every ceiling; a deleted budget key makes `got > undefined` false, which + * is a deleted gate wearing a passing arm's clothes; + * - an INVENTORY arm against `SCOPE_RESOLVERS` itself. `LANG_REGISTRY` claims + * to cover every registered resolver; this is what makes the claim true + * rather than commented, and it is the arm that would have caught PR #2911's + * language shipping unmeasured. + * + * SCOPE OF THE "independent of corpus size" CLAIM — the `collide` arm. + * `small`/`large`/`deep` mint one directory name per index (`src/pkg7`, + * `src/Ns7`, `lib/feature7`) and one basename per file, so every index bucket + * in them holds exactly ONE entry: measured, max last-segment bucket = 1 and + * max matching directories = 1 for go and csharp at both 400 and 1600 files, + * max basename bucket = 1 for dart and ruby. Bucket cardinality is the only + * non-constant term the new indexes have, so those arms certify the headline + * claim on the one shape where that term cannot appear. `collide` is the same + * workload — identical file, import and resolved counts — laid out the way + * these languages are actually written: `svcN/internal/`, `SrcN/Models/`, a + * `mod0.dart`/`mod0.rb` in every package. Measured on that shape the per-import + * cost is NOT corpus-size-independent for the four resolvers that scan a + * bucket: + * + * - go, csharp and java walk `PackageDirIndex.dirsByLastSegment[seg]`, which + * now holds every directory; + * - dart walks its basename bucket, which now holds every same-named file; + * - ruby, kotlin, php and cobol answer from keyed maps and are collision- + * IMMUNE, so their collide budgets are the linear ones — that immunity is + * the assertion, and for cobol the arm is also the only one that reaches + * the copybook-over-source tier tie-break, which needs one bookname to name + * two files; + * - csharp_csproj runs the OTHER way: its shared leaf collapses + * `dirsByLastSegment` to a single key, which makes the slash-free sweep + * (see `CSPROJ_CONFIGS`) cheaper on the collide layout than on the unique + * one, so its expensive scale arm is `large`, not `collide_large`; + * - of the eight added later, swift (3.28) and c/cpp (2.54/2.64) are the two + * that scan a bucket, and they scan DIFFERENT buckets: swift's is the + * module's own file list, which it returns, and C's is the basename bucket + * its suffix fallback walks. python answers from keyed maps and sits at + * 1.03-1.10, and javascript, typescript and vue answer from a declared + * config (#2953) — so all four keep the linear budget and that immunity is + * their assertion, exactly as for ruby and kotlin; + * - rust's collide arm is the one that is NOT a shared-leaf layout, and the + * reason is in the list above: file count is not an axis its cost has, so a + * shared-leaf rust arm would have been an arm that cannot fail. Its collide + * corpus is a deep module tree whose targets carry ~2x the `::` segments, + * which is the axis that CAN grow; the ratio across file counts staying at + * 1.06 on it is the assertion, and `collide_ms_ceiling` bounds the absolute + * cost of the long-path probe. + * + * This is a scope-of-claim limit, not a regression: on the MISS path with a + * shared leaf name the bucket grows with the file count BY CONSTRUCTION, and + * the indexed code is still faster there than the pre-change full scan. The arm + * exists so the real shape is measured and pinned, and so nobody reads the 1.8 + * budget as covering it. Narrowing it would mean a reversed-path prefix-range + * structure, which trades against the O(files × depth) memory + * `package-dir-index.ts` cites #2649 to avoid — a design change, not a tune. + * + * MEMORY — the `heap` arm. C#, Ruby, PHP and Java all resolve through the + * shared `WorkspaceFileIndex`, and `buildSuffixIndex` under it emits maps at + * O(files × depth): exactly the profile `package-dir-index.ts` cites #2649 to + * avoid for itself. That is why this is gated rather than noted — all four + * retained NOTHING across imports at BASE. C#'s `getWorkspaceFileIndex` was + * reached only from the csproj branch while the no-csproj leg scanned the Set; + * PHP and Java scanned on every leg; Ruby rebuilt and discarded a suffix index + * per `require`. Every other arm here is time or count, and no ratio can see a + * footprint. Measured in ABSOLUTE bytes, not only as a ratio: the finding is + * about the footprint itself, and a ratio alone hides a large constant. + * + * Four more are gated for the same reason as those four. JavaScript is the + * clearest case in the file: before PR #2911 it retained NOTHING because it + * built no index at all, and it now retains 25.51 MiB at 32 000 files through + * the + * same `buildSuffixIndex`. `csharp_csproj` is the newest and the one that + * proves the arm's design: same corpus and same `getWorkspaceFileIndex` as + * `csharp`, but its csproj leg asks all three questions instead of one, and it + * retains 70.29 MiB against C#'s 28.48. Python's `getPythonFileIndex` + * (9.88 MiB) and C's basename map (9.55 MiB) are an order of magnitude smaller + * but are the only structure either language keeps, and both are one careless + * edit — a stored `split('/')` array instead of a depth NUMBER — away from the + * O(files × depth) shape this arm exists to catch. + * + * WHAT THE ARM MEASURES IS NOW THE READ PATTERN, and that is the correction + * this file most needed. `buildSuffixIndex`'s two suffix maps became lazy + * (#2903 extended past `dirMap`), and the four original arms — which called + * `getWorkspaceFileIndex(set)` directly and read `index.all.length` — stopped + * asking any suffix question, built no map, and reported 0 B at 32 000 files. + * 0 B is under every ceiling, so `--check` PASSED with four gates that had + * become ceilings over nothing. Every arm now resolves one real MISSING import + * through the real resolver, so the maps it forces are the maps production + * forces; `HEAP_PROBE_TARGET` and `retainedPassBytes` carry the details, and + * `heap_floor_fraction` is the gate that would have caught the 0 B. + * + * EVERY LANGUAGE IS MEASURED, and the eight-entry list this arm ran on is now + * the BUDGET tier rather than the measurement tier. That list — `HEAP_LANGS`, + * now `HEAP_BUDGETED` — was reconciled bidirectionally against its two budget + * maps and every entry had to produce a reading, but nothing tied it to the + * property it stood for, "the languages that retain a per-pass index". Its two + * neighbours in this file do not have that gap: `LANG_REGISTRY` is reconciled + * against `SCOPE_RESOLVERS.keys()` and `CONTEXT_LANGS` against hook arity, both + * directions, both derived. Nine languages were excluded on readings taken once + * and written into this prose, and the paragraph below states the re-entry + * condition ("if any of the four ever diverges in what it ASKS, it earns an arm + * the same way") with nothing watching for the divergence. + * + * Re-measured — all seventeen, five runs each, one probe per language through + * the same `retainedPassBytes` — the prose was wrong in three separate ways: + * + * 1. THREE OF THE NINE HAD NO STATED REASON AT ALL. The old paragraph opened + * "SIX of the seventeen are deliberately NOT in HEAP_LANGS" against a list + * of eight, so go, dart and kotlin were excluded silently. All three + * retain a real per-pass structure: go's `PackageDirIndex` reads + * 2 998 464 B, dart's basename buckets 7 834 200 B, and kotlin's + * `suffixByStem` cascade 42 802 456 B (40.82 MiB) — above ruby's 39.12 and + * java's 33.34, both of which carry a full budget. (Read 48 073 096 B when + * this paragraph was written and described as "the second-largest reading + * in this file", which it was not even then: csharp_csproj and php both + * read higher. #2881 then compacted kotlin's `dirChildren` buckets and + * took 11% off it.) + * 2. TWO OF THE STATED REASONS NO LONGER HOLD. swift was excluded as "below + * its own noise floor" on 0.98 MB at 8000 files against 0.29 MB at 32 000; + * it now reads 969 120 B and 3 449 216 B, growing the right way. COBOL was + * excluded "for the same reason" on 0.54 MB then 0 B; it now reads + * 536 264 B and 2 320 456 B, ratio 1.082. Neither number moved because + * either index changed — the ARM changed, twice, when it started resolving + * a real import (#2903) and when `measureHeap` began flattening its + * corpus. Both re-measure to within 0.24% peak-to-peak over five runs, + * which is not a noise floor. + * 3. THE PROSE HAD GONE STALE AGAINST ITSELF. It quoted javascript at + * 46 208 832 B four paragraphs after quoting it at 25.51 MiB + * (26 745 296 B), because one number was re-taken with the arm and the + * other was only ever written down. + * + * Only rust's exclusion survived unchanged: 16 B at 8000 files and 16 B at + * 32 000, identical in all five runs, because it probes candidate paths with + * `allFilePaths.has(...)` and builds nothing. + * + * So the nine are still not BUDGETED — their ceilings, floors and ratio arms + * are not this change to write — but they are all measured and all bounded. See + * `HEAP_BOUNDED` for the gate and `_heap_bound_note` in baselines.json for each + * language's reading and its own reason, which are not one reason: rust builds + * nothing; go, dart, kotlin, swift and cobol build something this file has + * never bounded; and typescript, vue and cpp are duplicates of a BUILDER and of + * a READ PATTERN, both halves of which have to hold — `csharp_csproj` was + * excluded on the first half alone, at +20.8% of the C# index, and reads 2.47x + * of it now that the second half decides the number. Measured here: typescript + * 26 745 296 B against javascript's 26 745 296 B (byte-identical in four runs + * of five), cpp 10 023 344 B against c's 10 018 816 B (+0.05%), vue + * 28 884 016 B (+8.0%, what `.vue` instead of `.ts` buys on two thirds of the + * paths). The bound is what watches for the divergence the re-entry condition + * names — and it watches at 1.5x, so it catches a language GROWING an index, + * not a duplicate drifting by 8%. That limit is stated rather than papered + * over: the tight form is a same-process ratio against the arm each one + * duplicates, which is the only form immune to the cross-runner heapUsed drift + * an absolute bound has to tolerate. + * + * KNOWN BLIND SPOT, measured: a full workspace scan reintroduced on 1-in-32 + * imports passes every arm here (dart scored 1.458 scaling, 1.736 ms). The gate + * that NARROWS it is not a timing gate — the parity test above counts + * iterations of the file-set Set and reads 14 instead of 1 for that same + * mutation. It does not CLOSE it: the counter watches the Set, while the + * resolvers hold materialized arrays of the same file list + * (`WorkspaceFileIndex.normalized`/`.all`, Dart's basename buckets, + * `PackageDirIndex.filesByDir`, PHP's `filesByRawDirectory`, COBOL's two tier + * maps), and a 1-in-32 scan over one of THOSE passes + * both the parity test and `--check`. Chasing it by tightening these ceilings + * toward the noise floor would only buy flaky CI; see `_blind_spot` in + * baselines.json. PR #2911 is the proof that this blind spot is real rather + * than theoretical: JavaScript's missing index was a scan of + * `ImportPassCache.normalizedFileList` on EVERY import, which the Set counter + * could not see, and it took a differential parity test over 211 200 pairs plus + * this bench's arrival to pin it. + * + * THE FIFTH ARGUMENT — `context`, and exactly how much of it is measured. This + * harness used to call the inner resolvers with THREE arguments while `run.ts` + * calls `provider.resolveImportTarget` with FIVE, the fifth being + * `{ parsedFiles, parsedImport }`. Every arm was therefore a measurement of a + * call shape production never makes, and that is not a cheap thing to get + * wrong: defeating the `perFileSet` memo behind PHP's `filesByDirectory` + * measures 197.0 µs -> 9976.2 µs per import (50.6x) with every test still + * green, and nothing here could see it. + * + * `resolveOne` now makes the production call. Only TWO of the seventeen arms + * can observe it — PHP and Python are the only registered hooks that declare a + * fifth parameter — and that is ASSERTED rather than asserted-in-a-comment: the + * inventory arm at the foot of the file reads + * `SCOPE_RESOLVERS.get(language).resolveImportTarget.length` and reconciles it + * against `CONTEXT_LANGS` in both directions, so a language that grows a + * context leg cannot ship with the leg unmeasured. The other fourteen are handed + * nothing and build no `ParsedFile[]` at all, so their numbers are unmoved. + * + * `newPass` mints the `ParsedFile[]` FIRST and derives the path set from it + * (`new Set(parsedFiles.map(f => f.filePath))`), because that is what `run.ts` + * does — two independently built lists are a shape the pipeline cannot produce + * and would let the two memos disagree about which files exist. Both are fresh + * per pass for the reason the Set always was: `filesByDirectory` (PHP) and + * `parsedFileByPath` (Python) are `perFileSet` memos keyed on the ARRAY's + * identity, so a reused array would hide their build from rep 2 onward and + * `fastest()` takes the minimum. + * + * THE LEGS ACTUALLY RUN, which is what a "context is threaded" claim is worth + * nothing without — a leg that returns early measures nothing, the exact + * failure the four 0 B heap arms already demonstrated in this file. PHP's needs + * `parsedImport.kind` to be `named` or `alias` AND `importedSymbolKind` to be + * `function` or `const`; Python's needs a `named`/`alias` import too, because + * the synthetic `namespace` spelling this file used to pass makes + * `pythonImportedSubmoduleTarget` return null and `context.parsedFiles` is then + * never read at all. A deterministic `context` arm pins both per language: a + * three-file corpus resolved through `resolveOne` twice, once with the pass's + * `parsedFiles` and once without, whose two answers must DIFFER and must both + * equal what baselines.json records. Dropping the fifth argument, dropping + * `importedSymbolKind`, or reverting Python to `namespace` collapses the two + * onto one value and fails. PHP and Python still agree with their fallback on + * the main corpus; Java deliberately has no context-free fallback, so its + * fingerprints move to the declared-package answers recorded here. + * + * WHAT IS STILL NOT MEASURED, narrowed rather than deleted: + * + * - Python's `parsedFileByPath` memo is exercised by the five timing arms and + * NOT by the heap arm, and structurally cannot be. `retainedPassBytes` + * requires its probe to MISS, while every path that builds that memo runs + * through a non-null `packageTarget` which `resolvePythonImportTarget` then + * returns. So nothing here bounds that Map's footprint; it is one pointer + * per parsed file, O(files) with no depth term, and the count gate in + * import-target-index-reuse.contract.test.ts is what holds it to one build + * per pass; + * - PHP's leg is measured with NO composer.json — `resolutionConfig` is + * undefined here, as it always has been — so `namespaceDirectories` only + * ever returns the directory of an already-resolved file and the PSR-4 + * mapping branch stays unreached, exactly as `csharp` cannot reach the + * csproj leg. Closing that is a second PHP arm on the `csharp_csproj` + * precedent, not a parameter; + * - the `const` tail of PHP's leg (`candidateFiles.length === 1`) is a + * different ANSWER, not a different cost: `function` runs the identical + * candidate gather and `localDefs` filter and diverges only in the last two + * lines. It is gated by count in the contract test above. + * + * COST, and the honest version of it. REPORT mode is ~33-35 s, down from ~46 s: + * the timing phase fell from 39.8 s to 28.7 s when `REPS` became per-language + * (see `repsFor`), and that win is real. `--check` is ~44-45 s, which is + * ESSENTIALLY UNCHANGED from the ~46 s it cost before, because the inventory + * arm added here loads `pipeline/registry.ts` and that one dynamic import + * consumes almost the whole `repsFor` win — measured 6.3-6.5 s on one box and + * 9.3-10.0 s on another, in isolation and after this file's own static imports + * are already resident. Do not read the two modes as "~46 → ~42": only report + * mode got faster. + * + * MEASURING ALL SEVENTEEN HEAP ARMS instead of eight costs 1.37 s, and that is + * a measured number rather than the "seconds are free here" the paragraph below + * would have let it be. Timed per language with the phase instrumented, twice: + * the heap phase goes 2.06 s -> 3.43 s (1.377 s and 1.370 s added over the two + * runs). The nine are kotlin 0.57 s — it retains the largest index of the nine + * and builds all three of its maps eagerly — then vue 0.18, typescript 0.17, + * cpp 0.09, swift 0.08, dart 0.08, go 0.07, cobol 0.07, rust 0.06. End to end + * that is report mode 33.76 s -> 34.93 s (min of three runs each, +1.17 s, + * consistent with the phase measurement inside run-to-run noise). `--check` was + * 41.60 s before and reads 41.48-43.56 s after, i.e. the whole-run difference + * is INSIDE the registry import's own 6.3-10.0 s spread and cannot be resolved + * at that level — the +1.37 s phase number is the one to quote. + * + * That cost was weighed and KEPT, on the one number that decides it: the + * `benchmarks` job is not CI's critical path. On the last green run of main it + * finished in 9 m 23 s against 12 m 58 s for the sharded coverage job that + * gates the merge, so ~4 m 40 s of slack sits above this bench and ten seconds + * of it buys zero merge latency. Moving the arm into a vitest file would move + * the registry load ONTO that critical path, and would weaken it besides: from + * `LANG_REGISTRY`'s `SupportedLanguages` values, which are what the five + * dispatcher branches key off, down to baselines.json's arm NAMES plus a + * hand-written rule for de-aliasing `csharp_csproj`. See the wall-clock note in + * `_arms_note` for the per-language breakdown and for what to drop first if + * that stops fitting the job. + * + * Run: + * node --expose-gc --import tsx bench/import-target/measure.mjs # report + * node --expose-gc --import tsx bench/import-target/measure.mjs --check # CI gate + */ +import fs from 'node:fs'; +import path from 'node:path'; +import crypto from 'node:crypto'; +import { fileURLToPath } from 'node:url'; + +import { SupportedLanguages } from 'gitnexus-shared'; + +import { resolveGoImportTarget } from '../../src/core/ingestion/languages/go/import-target.ts'; +import { resolveDartImportTarget } from '../../src/core/ingestion/languages/dart/import-target.ts'; +import { resolveRubyImportTarget } from '../../src/core/ingestion/languages/ruby/import-target.ts'; +import { resolveCsharpImportTarget } from '../../src/core/ingestion/languages/csharp/import-target.ts'; +import { resolveKotlinImportTarget } from '../../src/core/ingestion/languages/kotlin/import-target.ts'; +import { resolvePhpImportTargetInternal } from '../../src/core/ingestion/languages/php/import-target.ts'; +import { javaScopeResolver } from '../../src/core/ingestion/languages/java/scope-resolver.ts'; +import { cobolScopeResolver } from '../../src/core/ingestion/languages/cobol/scope-resolver.ts'; +import { resolveSwiftImportTarget } from '../../src/core/ingestion/languages/swift/import-target.ts'; +import { resolveRustImportTarget } from '../../src/core/ingestion/languages/rust/import-target.ts'; +import { resolvePythonImportTarget } from '../../src/core/ingestion/languages/python/import-target.ts'; +import { makeJsResolveImportTarget } from '../../src/core/ingestion/languages/javascript/import-target.ts'; +import { makeVueResolveImportTarget } from '../../src/core/ingestion/languages/vue/import-target.ts'; +// The two `ScopeResolver`s, not their inner resolvers — see `RESOLVE_HOOK`. +import { typescriptScopeResolver } from '../../src/core/ingestion/languages/typescript/scope-resolver.ts'; +import { cScopeResolver } from '../../src/core/ingestion/languages/c/scope-resolver.ts'; +import { cppScopeResolver } from '../../src/core/ingestion/languages/cpp/scope-resolver.ts'; +// `SCOPE_RESOLVERS` is NOT imported here — see the inventory arm at the bottom, +// which loads it dynamically. Statically it costs 6-10 s of module load +// depending on the box (measured both ways there), because reaching the +// registry pulls in every registered provider and everything under them, and it +// is wanted by one `--check` arm that runs after the last measurement. + +/** The JS and Vue adapter FACTORIES return a closure; the memo they read is + * module-level, so one instance per process is both correct and what the + * registry does (`resolveImportTarget: makeJsResolveImportTarget()`). */ +const jsResolveImportTarget = makeJsResolveImportTarget(); +const vueResolveImportTarget = makeVueResolveImportTarget(); + +const __dirname = path.dirname(fileURLToPath(import.meta.url)); +const BASELINE_PATH = path.resolve(__dirname, 'baselines.json'); + +const SMALL = 400; +const LARGE = 1600; +const IMPORTS_PER_FILE = 8; +/** Extra directory components prepended in the `deep` arm — see `depth_ratio`. */ +const DEEP_PAD = 16; +/** + * `fastest()` below is a min-of-N estimator, so N is the noise knob: raising it + * lowers and stabilises the minimum. `depth_ratio` divides two sub-3 ms + * measurements, and Dart's are sub-1 ms, so it is by far the noisiest number + * here. Measured over 22 `--check` runs on an idle box: at N=5 it tripped its + * own budget ~1 run in 20, at N=7 Dart still swung 3.0x peak-to-peak and + * tripped once. N=15 (which matches `bench/cfg`, `bench/schema-pairs` and + * `bench/callable-value-flow`) collapsed every language to a 1.13-1.26x swing + * with 22/22 passing; the distributions are recorded in `_arms_note`. + * + * N used to be 15 for EVERY arm, set globally by the noisiest cell. That paid + * the noisiest cell's insurance premium on cells a thousand times its size: + * the recorded overshoot of min-of-K against min-of-15 is a function of the + * cell's absolute duration, not of the language — 31.8% on `swift.small` + * (0.43 ms) and 37.6% on `dart.collide` (1.5 ms) at the extreme, but at most + * 6.3% at K=7 for every cell at or above 10 ms. + * + * So N is picked PER LANGUAGE, from the cost of its cheapest arm: 15 while that + * is under `REPS_CHEAP_MS`, and `~REPS_BUDGET_MS` worth of samples above it, + * floored at `REPS_MIN`. Per language rather than per cell so all five arms of + * a language share one estimator and the four ratios stay comparisons of like + * with like. In practice that is still 15 for go, csharp, dart, kotlin, java, + * cobol, swift, rust, python, c and cpp — every language the flakiness above + * was ever about — and 7-8 for php, csharp_csproj, ruby, javascript, typescript + * and vue, whose cheapest cell is 20-28 ms. Replayed against two independent + * runs' sample sets it saved 12.8 s and 12.4 s of a 46 s run with all 85 cells + * passing all five gates at 0.4-0.7 of budget, and min-of-7 reads slightly + * HIGHER than min-of-15, so the gates get marginally more sensitive rather than + * less. The chosen N is reported per language as `reps`. + */ +const REPS_MAX = 15; +const REPS_MIN = 7; +/** Sampling budget per cell for the languages that do not get `REPS_MAX`. */ +const REPS_BUDGET_MS = 150; +/** Below this a cell is small enough for the min-of-N estimator itself to be + * the dominant error, so it gets the full `REPS_MAX` regardless of budget. The + * nearest language on either side of it is 3.2 ms and 20.0 ms, so nothing sits + * near the boundary. */ +const REPS_CHEAP_MS = 5; +const WARMUP = 2; + +/** N for one language, from one warmed pass of its cheapest arm. */ +function repsFor(probeMs) { + if (probeMs < REPS_CHEAP_MS) return REPS_MAX; + return Math.min(REPS_MAX, Math.max(REPS_MIN, Math.ceil(REPS_BUDGET_MS / probeMs))); +} + +/** Heap arm. Far more files than the timing arms because the finding + * is an ABSOLUTE footprint at repository scale, and 1600 files would report a + * fraction of a MiB — a number no ceiling could usefully bound. `HEAP_PAD` + * keeps the paths at a plausible monorepo depth: `buildSuffixIndex` is + * O(files × depth), so a flat corpus would understate it by ~4x. */ +const HEAP_SMALL = 8000; +const HEAP_LARGE = 32000; +const HEAP_PAD = 8; +/** The languages whose retained per-pass index carries a BUDGET — a ceiling, a + * floor derived from `heap_reading_bytes`, and the linear-growth ratio arm. + * All eight are measured the same way as the other nine (`retainedPassBytes`, + * one real import through the real resolver); what this list decides is which + * GATE a reading gets, not whether it is taken. The first five reach the shared + * `WorkspaceFileIndex` and retained NOTHING at BASE; `csharp_csproj` is the + * same corpus through the same index under the csproj context, and it is here + * rather than excluded as a duplicate because after #2903 its READ PATTERN, + * not its corpus, decides the number. + * + * The remaining three are `HEAP_BOUNDED`, DERIVED from this list rather than + * written beside it, and they carry an upper bound and NO floor. That asymmetry + * is the point: a bound catches "this language grew an index", which is the + * re-entry condition, while a floor over a reading at or below its own noise + * would gate the noise. rust reads 16 B at both scales; swift's ratio is 0.888 + * and cobol's 1.082, both outside the linearity every budgeted arm shows, so a + * floor and a ratio arm would be measuring the measurement. See the MEMORY + * section of the header for what re-measuring all seventeen found. */ +const HEAP_BUDGETED = [ + 'csharp', + 'csharp_csproj', + 'ruby', + 'php', + 'java', + 'python', + 'c', + // Promoted once every language was actually measured. Each retains a real + // per-pass structure and each grows LINEARLY with the file count (ratio + // 0.996-1.004 against a 1.25 budget over 8000 -> 32000 files), so each can + // carry the full ceiling + floor + ratio set rather than a bound alone. + // kotlin's 40.82 MiB is larger than ruby's and java's, both of which were + // budgeted from the start, and it had no stated exclusion reason at all. + // + // Its ceiling is also the one TIGHT ceiling in this file — 1.0747x its + // reading where every other is 1.5x — because it is the only one gating a + // size REDUCTION being preserved rather than a footprint not growing. + // #2881 compacts `dirChildren`'s buckets, and deleting that `slice()` is + // invisible to every other instrument in the repository: output-identical, + // so no fingerprint moves; capacity has no reflective surface, so no unit + // assertion moves; and both heap scales grow together, so `heap_ratio_budget` + // divides it out. It shows up here and nowhere else, at +12.57%. See + // `_heap_compaction_gate` in baselines.json for the measurement, the + // arithmetic behind the 5.4 MB, and how to tell a lost compaction from a + // runner's heapUsed accounting moving under the whole file. + 'kotlin', + 'dart', + 'go', + 'cpp', +]; +// javascript, typescript and vue were budgeted here until #2953 and are now +// BOUNDED, which is a demotion in gate strength and a promotion in what the +// number means. They retained ~26.7 MiB each because they built a per-pass +// `SuffixIndex` over the whole file list; they no longer build one at all, +// because declared resolution derives nothing from the file set — a candidate +// comes from a tsconfig mapping or a manifest and is checked with one +// `Set.has`. The readings are 0-16 B. +// +// A floor over a reading at or below its own noise gates the noise, which is +// the same reason rust sits in this tier at 16 B — so they take a bound and no +// floor. The bound is what still matters: it catches these three growing an +// index again, which is the re-entry condition for the cost #2911 and #1918 +// were about. + +/** + * The arms handed the fifth `context` argument — `{ parsedFiles, parsedImport }` + * — because their registered hook DECLARES it. Three of seventeen, and the + * inventory arm at the foot of this file reconciles that claim against + * `SCOPE_RESOLVERS` in both directions rather than trusting this line. + * + * These are also the only arms for which `newPass` builds a `ParsedFile[]` at + * all. Building one for the other fourteen would cost their timed loop an + * O(files) allocation per pass that no resolver of theirs can even observe — + * their hooks declare three or four parameters — so their numbers stay exactly + * where they were. + */ +const CONTEXT_LANGS = ['php', 'java', 'python']; + +/** + * Needs `node --expose-gc` to force collection for a clean delta; without it + * the heap metric is reported as null and its `--check` gate would be skipped, + * which is why `--check` refuses to run without the flag (see below). + * + * TWO cycles because the value is a `WeakMap`'s: the first clears the entry + * once its key is unreachable, the second collects what the entry held. That is + * not always enough — PHP reaches the shared index through a second per-file- + * set memo of its own (`getPhpWorkspaceIndex` wraps `getWorkspaceFileIndex`, + * both keyed on the same Set) and that chain measured FOUR cycles to release, + * with two leaving 9.3 MB of the previous read still counted live. The answer + * to that is `HEAP_RETAINED`, which removes the need to release anything inside + * a measurement window, plus the deeper drain `measureHeap` runs between + * languages where a late free costs nothing. Cycles are not the knob: with + * `HEAP_RETAINED` in place, two and four produce byte-identical readings, and + * four cost 4.5 s of wall clock over a retained heap this size. + */ +const GC = typeof global.gc === 'function' ? () => (global.gc(), global.gc()) : null; + +/** Deterministic 32-bit avalanche (murmur3 finalizer) — no `Math.random()`, so + * the corpus and therefore the fingerprint are byte-reproducible. */ +function mix(n) { + let x = n >>> 0; + x = Math.imul(x ^ (x >>> 16), 0x85ebca6b) >>> 0; + x = Math.imul(x ^ (x >>> 13), 0xc2b2ae35) >>> 0; + return (x ^ (x >>> 16)) >>> 0; +} + +const GO_MODULE = { modulePath: 'example.com/mod' }; +/** + * The `csharp_csproj` arm's project configs — the whole reason that arm exists. + * + * `csharp` builds its context with NO `csharpConfigs`, so every one of its + * imports takes the no-csproj branch and the csproj leg's namespace-directory + * index (#2902) would ship unmeasured. Two configs rather than one because the + * leg's cost is a function of `dirPrefix`'s SHAPE, and one config cannot + * produce all three: + * - `App` + `projectDir: 'src'` gives `dirPrefix = 'src/'`, which + * CONTAINS a slash, so `candidateDirs` answers from the last-segment bucket; + * - `Lib` + `projectDir: ''` gives `dirPrefix = ''`, slash-FREE, the + * one leg that sweeps the last-segment KEYS and so is not constant-time; + * - `Lib` itself (the import IS the root namespace, no `projectDir` to stand + * in) gives an EMPTY `dirPrefix`, answered from `singleSegmentDirs`. + * All three were a full `normalizedFileList` pass per import before #2902. + */ +const CSPROJ_CONFIGS = [ + { rootNamespace: 'App', projectDir: 'src' }, + { rootNamespace: 'Lib', projectDir: '' }, +]; +/** + * The `resolutionConfig` the ts-family arms thread (#2953). + * + * These three used to run with `tsconfigPaths: null` for javascript and + * typescript and an alias map for vue, because the leg being measured was + * `suffixResolve` — a repo-wide search for a path ending in the specifier, + * which needs no configuration to answer and answered even when nothing + * declared the import. #2953 deleted that leg for the ts family: a specifier + * now resolves only against a declared tsconfig mapping or a package manifest. + * + * With no config, therefore, all three arms resolve NOTHING — every import is + * correctly external — and the bench measures an empty branch while reporting a + * perfect scaling ratio. A bench that measures nothing passes exactly like one + * that measures something, so each arm is given the config its corpus is + * spelled for, and the two configs cover the two legs the new resolver has: + * + * - `TS_BASE_URL` — `baseUrl` at the repo root, so `src/mod3/file7` resolves + * the way a `baseUrl` project's absolute import does. Used by javascript and + * typescript. + * - `vueTsconfig` — a `paths` PATTERN, which is a different branch: + * longest-prefix selection and `*` substitution, then a candidate probe per + * target. Every local Vue import below is spelled `@/…`, so the vue arm + * stays a third measurement rather than a third copy — the same role it had + * before, now against the branch that replaced the alias rewrite. + * + * Not covered here: the workspace-manifest leg + * (`node-workspace-packages.ts`), which is a `Map.get` on a package name and + * does not scale with the file set. + */ +const tsBaseUrlConfig = (baseUrl) => ({ + tsconfigs: { scopes: [{ dir: '', baseUrl, paths: [] }] }, + nodeWorkspacePackages: null, +}); +const vueTsconfig = (baseUrl) => ({ + tsconfigs: { + scopes: [ + { dir: '', baseUrl, paths: [{ pattern: '@/*', targets: [joinBase(baseUrl, 'src/*')] }] }, + ], + }, + nodeWorkspacePackages: null, +}); +const joinBase = (baseUrl, rest) => (baseUrl === '' ? rest : `${baseUrl}/${rest}`); +/** + * The `deep` arm prepends a UNIFORM `d0/…/d15/` prefix to every path + * (`buildFiles`), and the import spellings do not change. Under the old suffix + * matcher that was the point: the resolver walked path components, so depth was + * the cost. Declared resolution never walks — the config names an exact base — + * so the prefix has to reach the config or the whole arm resolves nothing and + * measures the miss path at depth instead of the hit path at depth. + */ +const tsBaseUrlFor = (pad) => + pad === 0 ? '' : Array.from({ length: pad }, (_, n) => `d${n}`).join('/'); +/** Keyed by LAYOUT name, so there is no `csharp_csproj` row: `buildFiles` + * aliases that arm to `csharp` before this table is read. */ +const EXTENSION = { + go: '.go', + csharp: '.cs', + dart: '.dart', + ruby: '.rb', + kotlin: '.kt', + php: '.php', + java: '.java', + cobol: '.cbl', + swift: '.swift', + rust: '.rs', + python: '.py', + javascript: '.js', + typescript: '.ts', + vue: '.vue', + c: '.c', + cpp: '.cpp', +}; +/** C and C++ resolve `#include` against HEADERS, which reach the resolver + * through `resolutionConfig` rather than through `allFilePaths` — see + * `newPass`. Half of each corpus is headers; this is their extension. */ +const HEADER_EXTENSION = { c: '.h', cpp: '.hpp' }; +/** Directory fan-out. Shared because `buildRepo`'s collide targets address + * files by `j % dirs` / `Math.floor(j / dirs)` and must agree with the layout + * `buildFiles` produced. */ +const dirsFor = (fileCount) => Math.max(4, Math.floor(fileCount / 8)); +/** Swift's collide arm, and the ONLY place a bucket size is pinned by a + * constant rather than by `dirsFor`. A module bucket is what Swift returns, so + * its cardinality has to grow with the corpus for the arm to measure anything: + * four modules means fileCount/4 per bucket (100 at `collide`, 400 at + * `collide_large`), which is the shape a small SPM package actually has. */ +const SWIFT_COLLIDE_MODULES = 4; +/** File stems follow each language's own naming convention, because C#'s and + * PHP's suffix maps carry a case-insensitive tier and a lower-cased corpus + * would leave it answering the same question twice. Keyed by LAYOUT name, like + * `EXTENSION` — no `csharp_csproj` row, for the same reason. */ +const PASCAL_CASE_FILES = new Set([ + 'csharp', + 'kotlin', + 'php', + 'java', + 'cobol', + // Swift types and Vue SFCs are PascalCase by universal convention. + 'swift', + 'vue', +]); +/** Rust and Python name a DIRECTORY as a module through a well-known file, so + * the first file minted in each directory is that file rather than a numbered + * one. Every in-repo target below resolves to one of them. */ +const PACKAGE_STEM = { rust: 'mod', python: '__init__' }; + +/** + * The end of a per-language dispatcher, where five of them used to fall through + * to a bare `return`. + * + * Four of those fallthroughs meant "ruby" and the fifth meant "csharp". So + * `ruby` appeared nowhere in this file except `EXTENSION` and the language + * list, and — the part that matters — a language added to the list but missed + * in the dispatchers would have been benchmarked as RUBY'S CORPUS RESOLVED BY + * C#'S RESOLVER: five plausible timings, a stable fingerprint, and a permanent + * pass over a language nobody had measured. Every dispatcher now names its last + * branch and throws here instead, so the missing wiring is a crash on the first + * run rather than a green gate. + */ +function unwiredLanguage(where, lang) { + return new Error( + `bench: ${where} has no branch for '${lang}'. Every language in LANG_REGISTRY needs one in ` + + `uniqueDir, collideDir, uniqueTarget, collideTarget and resolveOne. (uniqueDir and ` + + `collideDir see the LAYOUT name, which is never 'csharp_csproj' — buildFiles aliases it ` + + `to 'csharp'.) Falling through here used to hand the language another one's corpus or ` + + `another one's resolver, and nothing in --check could tell.`, + ); +} + +/** + * UNIQUE-LEAF layout: one directory name per index, so no two directories share + * a last segment and no two files share a basename. Every index bucket holds + * exactly one entry. A nested same-name directory in one repo slice is the + * shape the first-`indexOf` tie-break used to reject (see package-dir-index.ts); + * #2881 removed that tie-break from every resolver that had it, so the go, + * csharp, java and kotlin arms all resolve their `d % 7` slice now. + * + * A repeat the query cannot ask about leaves the arm blind, which is why go's + * slice repeats the WHOLE package path: a Go import addresses `src/pkg{d}`, and + * `…/internal/pkg{d}` does not end with that, so the old rule was never even + * reached and every go arm sat still through the fix. Java, C# and Kotlin query + * the whole dotted path FIRST and only fall back to the tail through + * progressive stripping, so their slices — which repeat the last segment only — + * move through that fallback rather than the primary query. The consequence is + * measured and worth knowing: a partial revert that reinstates first-occurrence + * only for multi-segment package paths is caught on the go arm alone. + */ +function uniqueDir(lang, d, i) { + // Go's nested slice repeats the WHOLE queried path (`src/pkg{d}`), not just + // its last segment. `src/pkg{d}/internal/pkg{d}` repeated only `pkg{d}`, so + // the query `src/pkg{d}` failed on "the directory ends with the package path" + // and never reached the first-occurrence rule at all — Go's arms did not move + // when #2881 removed that rule, which would have shipped a widened bucket + // with no bench coverage while C# and Java were re-baselined for it. + if (lang === 'go') return d % 7 === 0 ? `src/pkg${d}/internal/src/pkg${d}` : `src/pkg${d}`; + // Leaf-only repeat, deliberately: this layout is shared with the + // `csharp_csproj` arm, whose configs mint `dirPrefix` against `src/Ns{d}`, so + // deepening it to the full `App/Ns{d}` query path resolves that arm to ZERO + // and breaks its same-workload invariant. C# therefore exercises the removed + // rule through progressive stripping rather than through its primary query. + if (lang === 'csharp') return d % 7 === 0 ? `src/Ns${d}/Sub/Ns${d}` : `src/Ns${d}`; + if (lang === 'dart') return d % 3 === 0 ? `lib/feature${d}` : `pkg/feature${d}`; + if (lang === 'kotlin') { + return d % 7 === 0 + ? `mod${d}/src/main/kotlin/com/example/pkg${d}/inner/pkg${d}` + : `mod${d}/src/main/kotlin/com/example/pkg${d}`; + } + if (lang === 'php') return d % 7 === 0 ? `src/App/Ns${d}/Sub/Ns${d}` : `src/App/Ns${d}`; + if (lang === 'java') { + return d % 7 === 0 + ? `mod${d}/src/main/java/com/example/pkg${d}/inner/pkg${d}` + : `mod${d}/src/main/java/com/example/pkg${d}`; + } + // COBOL resolves on the BASENAME alone (`path.basename(fp, ext)`), so its + // directories are pure realism — a copybook library beside the programs. + if (lang === 'cobol') return d % 3 === 0 ? `copybooks/grp${d}` : `src/prog${d}`; + // SPM. The nested slice makes one file's interior segments repeat + // (`Sources/Mod7/Internal/Mod7/File7.swift`), and `getSwiftModuleIndex` + // pushes once per segment, so that file appears TWICE in module `Mod7`'s + // returned list. Real layout, real output; the fingerprint pins it. + if (lang === 'swift') { + return d % 7 === 0 ? `Sources/Mod${d}/Internal/Mod${d}` : `Sources/Mod${d}`; + } + // Cargo. The nested slice has NO `mod{d}/mod.rs`, so `crate::mod{d}::thing` + // misses there — the same resolves/misses split every other unique arm has. + if (lang === 'rust') return d % 7 === 0 ? `src/mod${d}/inner` : `src/mod${d}`; + if (lang === 'python') return d % 7 === 0 ? `pkg${d}/inner` : `pkg${d}`; + if (lang === 'javascript' || lang === 'typescript') return `src/mod${d}`; + // Vue's local imports are all `@/…`, which the alias rewrites to `src/…`, so + // the whole corpus must live under `src/` for that branch to hit. + if (lang === 'vue') return `src/mod${d}`; + // C and C++ split headers from sources — the shape that makes + // `resolutionConfig` load-bearing. Odd `i` is the header. + if (lang === 'c' || lang === 'cpp') return i % 2 === 1 ? `include/comp${d}` : `src/comp${d}`; + if (lang === 'ruby') return `lib/mod${d}`; + throw unwiredLanguage('uniqueDir', lang); +} + +/** + * SHARED-LEAF layout: every directory ends in the SAME segment, so one bucket + * holds all of them. The `d % 7` slice keeps the nested same-name directory of + * the unique layout, and Go additionally replicates one package (`internal/ + * shared`) across services — the monorepo shape `filesDirectlyInPkgDir`'s merge + * exists for, and the only arm in this bench that reaches `dirCount > 1`. + * + * Each language's local import spelling is chosen so this arm resolves exactly + * as many imports as `small` does (asserted): same workload, different layout. + */ +function collideDir(lang, d, i) { + if (lang === 'go') { + // `…/sub/internal` repeats only the last segment, which the ends-with test + // answers on its own; `…/internal/sub/svc{d}/internal` is the shape the + // removed first-occurrence rule used to reject (see `uniqueDir`). + if (d % 7 === 0) return `svc${d}/internal/sub/svc${d}/internal`; + return d % 5 === 1 ? `svc${d}/internal/shared` : `svc${d}/internal`; + } + // Leaf-only repeat here too, and unlike the kotlin arm below that is not a + // blind spot — measured, base against head over this exact corpus. C#'s match + // test is an unanchored ends-with and its cascade strips leading segments, so + // `App.Src{d}.Models` reaches `Models` after two strips and finds + // `Src{d}/Models/Inner/Models`, whose FIRST `/Models/` is not its last: the + // removed first-occurrence rule rejected it and the current one takes it. The + // `csharp` collide fingerprint therefore moves across #2881 (03c9afe33276 + // head, 89d0a054b617 base) with the resolved count unchanged at 1153 — the + // arm sees the change, it just sees it as different ANSWERS rather than more + // of them. Deepening the slice to `Src{d}/Models/Inner/Src{d}/Models` only + // moves which strip level finds it; both layouts move base -> head, so it + // buys nothing here. + // + // And it costs, because the `csharp_csproj` constraint binds this arm too — + // differently from the way it binds `uniqueDir`. There, deepening resolves + // that arm to ZERO. Here it resolves MORE: `Lib` has `projectDir: ''`, so its + // `dirPrefix` is `Src{d}/Models`, which is not a segment suffix of + // `…/Inner/Models` and is one of `…/Inner/Src{d}/Models`. Measured, the + // csproj arm's collide `resolved` goes 979 -> 1153 against its `small` 979, + // which is the same-workload invariant `--check` asserts. (Worth recording + // while it is measured: with the shipped layout BOTH csproj arms are blind to + // #2881 — unique and collide fingerprints identical base and head — because + // `getFilesInDir` is keyed on segment-aligned directory SUFFIXES and neither + // nested slice is one. Closing that is the deepening plus a mirrored miss for + // the csproj arm's `d % 7` slice, i.e. a corpus redesign and four + // re-baselines, not this edit.) + if (lang === 'csharp') return d % 7 === 0 ? `Src${d}/Models/Inner/Models` : `Src${d}/Models`; + if (lang === 'dart') return `pkg${d}/lib/src`; + if (lang === 'kotlin') { + return d % 7 === 0 + ? // Repeats the WHOLE queried path (`com.example.models`), not just the + // `models` leaf. With a leaf-only repeat this arm was structurally + // blind to the #2881 rule: a full revert of the Kotlin guards left both + // collide fingerprints unmoved, because `com/example/models` is not a + // suffix of `…/models/inner/models` and the query never reached the + // rule. Deepening it is the only corpus edit in this file that buys + // coverage — the same deepening applied to the java and kotlin UNIQUE + // arms was measured and reverted, because progressive stripping lands + // those queries on the same file either way. + `mod${d}/src/main/kotlin/com/example/models/inner/com/example/models` + : `mod${d}/src/main/kotlin/com/example/models`; + } + if (lang === 'php') return `svc${d}/src/Models`; + if (lang === 'java') { + return d % 7 === 0 + ? `svc${d}/src/main/java/com/example/model/inner/model` + : `svc${d}/src/main/java/com/example/model`; + } + if (lang === 'cobol') return `svc${d}/copybooks`; + // Swift's collision axis is neither a shared directory name nor a shared + // basename: `byModule` is KEYED on the module name, so what grows a bucket is + // FEWER modules holding MORE files. `SWIFT_COLLIDE_MODULES` of them, so the + // bucket a hit returns is fileCount/4 — 100 entries at 400 files and 400 at + // 1600 — and a hit copies that whole bucket minus the importer. + if (lang === 'swift') return `Sources/Mod${d % SWIFT_COLLIDE_MODULES}`; + // Rust's cost is O(path SEGMENTS), not O(files) — it probes candidate paths + // with `.has()` and never searches. So its collide arm is a deep module tree + // whose targets carry ~2x the `::` segments, which is the axis that CAN grow; + // that the ratio across file counts stays flat on it is the assertion. + if (lang === 'rust') return `src/l0/l1/l2/l3/l4/mod${d}`; + // The `inner` slice mirrors the unique arm's, and for the same reason: it is + // where the in-repo target misses, so both arms resolve the same count. + if (lang === 'python') return d % 7 === 0 ? `svc${d}/models/inner` : `svc${d}/models`; + if (lang === 'javascript' || lang === 'typescript') return `pkg${d}/src`; + if (lang === 'vue') return `src/pkg${d}/components`; + if (lang === 'c' || lang === 'cpp') return i % 2 === 1 ? `svc${d}/include` : `svc${d}/src`; + if (lang === 'ruby') return `svc${d}/lib/models`; + throw unwiredLanguage('collideDir', lang); +} + +/** + * The file paths of one synthetic repository. `dirs` grows with the file count + * so directory fan-out is realistic at both scales rather than collapsing onto + * a handful of buckets. + * + * `pad` prepends that many extra directory components to every path. Every + * language's in-repo target resolves through a path SUFFIX (Go's module leg + * against the package dir, C#'s progressive strip, Dart's `lib/`, Ruby's + * suffix match, Kotlin's `suffixByStem`), so the padding changes path depth + * without changing what resolves — which is what makes the `deep` arm a clean + * depth measurement rather than a different corpus. + * + * Split out from `buildRepo` so the heap arm can build 32k paths without also + * minting 256k import tuples it would never resolve. + */ +function buildFiles(lang, fileCount, pad, shape) { + const dirs = dirsFor(fileCount); + const files = []; + // `csharp_csproj` is `csharp` with a different CONTEXT and nothing else. The + // alias is here, in the one place that mints paths, rather than as a second + // copy of the same layout in `uniqueDir`/`collideDir`: it makes the two arms' + // corpora identical by construction, so a later edit to C#'s layout cannot + // silently desynchronize them and turn the comparison into two experiments. + const layout = lang === 'csharp_csproj' ? 'csharp' : lang; + const ext = EXTENSION[layout]; + const prefix = pad === 0 ? '' : Array.from({ length: pad }, (_, n) => `d${n}`).join('/') + '/'; + for (let i = 0; i < fileCount; i++) { + const d = i % dirs; + const dir = shape === 'collide' ? collideDir(layout, d, i) : uniqueDir(layout, d, i); + // In the collide shape Dart, Ruby, PHP and COBOL carry a REPEATED basename + // — the term their indexes bucket or key on (COBOL's two tier maps are + // keyed on the uppercased basename and NOTHING else). `i / dirs` is unique + // within a directory (8 files land in each) and identical across + // directories, which is exactly the `models.dart` / `models.rb`-in-every- + // package convention. Go, C#, Kotlin and Java bucket on the DIRECTORY + // instead, so their stems stay unique and the shared leaf segment is what + // collides for them. + const collideStem = + layout === 'dart' || + layout === 'ruby' || + layout === 'cobol' || + layout === 'php' || + // The three added later that also bucket or key on the BASENAME: + // JS/TS `buildSuffixIndex` (one entry per path suffix, so the last + // component is the shortest key), Vue through the same index, Python's + // `byBasename`, and C/C++'s basename map — for the last, only the header + // half is addressable, so only it repeats (see below). + layout === 'javascript' || + layout === 'typescript' || + layout === 'vue' || + layout === 'python' || + layout === 'c' || + layout === 'cpp'; + const [fileStem, modStem] = PASCAL_CASE_FILES.has(layout) ? ['File', 'Mod'] : ['file', 'mod']; + let stem = + shape === 'collide' && collideStem ? `${modStem}${Math.floor(i / dirs)}` : `${fileStem}${i}`; + // Rust's `mod.rs` and Python's `__init__.py`: one per directory, and the + // file every in-repo target of theirs resolves to. Minted at the first file + // of each directory (`i < dirs`, so `d === i`), which is why both arms' + // resolved counts are the count of in-repo imports either way. + if (PACKAGE_STEM[layout] !== undefined && i < dirs) stem = PACKAGE_STEM[layout]; + // C and C++ address only HEADERS, so only their stems repeat in the collide + // shape; the sources stay unique and are pure corpus weight, exactly as in a + // real tree where nobody `#include`s a `.c`. + if ((layout === 'c' || layout === 'cpp') && i % 2 === 0) stem = `src${i}`; + // Go's package leg must exclude `_test.go`; keep a real share of them. + // Kotlin resolves `.kt` and `.kts` through the same stem maps; keep both. + // COBOL's copybook tier (`.cpy`) BEATS its source tier (`.cbl`) on the same + // bookname, so both extensions have to be present for that tie-break to be + // reachable at all — and in the collide shape, where basenames repeat, one + // bookname really does land in both tiers. + // A Vue repo is `.vue` SFCs plus plain `.ts` modules, and only the second + // kind reaches the extension-guessing leg (SFC imports carry `.vue` + // explicitly), so both have to be present for both legs to be measured. + // C/C++ alternate header and source; the header half is the addressable one. + const suffix = + layout === 'go' && i % 6 === 0 + ? '_test.go' + : layout === 'kotlin' && i % 11 === 0 + ? '.kts' + : layout === 'cobol' && i % 3 === 0 + ? '.cpy' + : layout === 'vue' && i % 3 === 0 + ? '.ts' + : HEADER_EXTENSION[layout] !== undefined && i % 2 === 1 + ? HEADER_EXTENSION[layout] + : ext; + files.push(`${prefix}${dir}/${stem}${suffix}`); + } + return files; +} + +/** + * ONE `ParsedFile`, and the ONE place in this file that spells that shape. + * + * CARRIES THE FIELDS THE RESOLVERS READ AND NOTHING ELSE, deliberately. + * `filesByDirectory` reads `filePath`; PHP's declaring-file filter reads + * `localDefs[].type` and `localDefs[].qualifiedName`; Python's + * `pythonFileExportsName` reads `localDefs[].qualifiedName`. `scopes`, + * `parsedImports` and `referenceSites` are on the real shape and are inert on + * this path, and the timed corpora are rebuilt inside every pass (see + * `newPass`), so filling them would charge the RESOLUTION arms for extraction + * work that happens in another phase entirely. + * + * `nodeId` is inert as well — checked, not assumed: neither + * `php/import-target.ts` nor `python/import-target.ts` mentions it, and they are + * the two modules `resolveOne` enters. It is minted anyway because it is on the + * real shape, and its spelling is therefore free to be uniform. + * + * Both callers come through here — `buildParsedFiles` for the timed and heap + * corpora, `CONTEXT_PROBE` for the `context` arm's hand-built ones. It used to + * be spelled out twice, ~900 lines apart, differing only in that `nodeId`; this + * is an untyped `.mjs`, so nothing would have failed at build if `ParsedFile` + * grew a field and only one of the two copies learned about it. + */ +const probeFile = (filePath, defs) => ({ + filePath, + moduleScope: filePath, + scopes: [], + parsedImports: [], + localDefs: defs.map(([type, qualifiedName], n) => ({ + nodeId: `${filePath}#${n}`, + filePath, + type, + qualifiedName, + })), + referenceSites: [], +}); + +const javaProbeFile = (filePath, packageName) => ({ + ...probeFile(filePath, []), + captureSideChannel: { + kind: 'java', + packageFact: { status: 'known', packageName }, + classAnnotations: [], + }, +}); + +function javaBenchmarkPackage(filePath) { + const uniquePackage = /\/com\/example\/(pkg\d+)(?:\/|$)/.exec(`/${filePath}`)?.[1]; + if (uniquePackage !== undefined) return `com.example.${uniquePackage}`; + + return /\/svc\d+\/.*\/com\/example\/model(?:\/|$)/.test(`/${filePath}`) + ? 'com.example.model' + : ''; +} + +/** + * The `ParsedFile[]` the orchestrator threads beside the path set, for the three + * languages whose hook declares a `context` — see `CONTEXT_LANGS`. + * + * Two defs per file, and both are real shapes rather than padding. PHP keeps + * classes and functions in SEPARATE symbol tables, so `App\Ns7\File7` naming + * both a class and a function is ordinary PHP — and it is what makes the leg's + * two halves reachable on the same corpus: the class def exercises the + * `def.type !== expectedType` reject (which returns before the split) and the + * function def exercises the `split(/[\\.]/).at(-1)` compare that decides the + * match. The qualified name carries two separators because that split's cost is + * a function of how many there are, and a one-segment name would understate it. + * + * The owner segment is the file's own directory name (`Ns7`, `Models`, `pkg7`), + * which is stable across the `small`, `deep` and `collide` arms — so the `deep` + * arm differs from `small` in path DEPTH alone, exactly as it does for the path + * set. That matters here: `directoryAliases` emits one entry per path segment, + * so `filesByDirectory` is O(files × depth) and the depth arm is the only one + * that can see it. + */ +function buildParsedFiles(lang, files) { + const parsedFiles = []; + for (const filePath of files) { + if (lang === 'java') { + parsedFiles.push(javaProbeFile(filePath, javaBenchmarkPackage(filePath))); + continue; + } + const slash = filePath.lastIndexOf('/'); + const stem = filePath.slice(slash + 1, filePath.lastIndexOf('.')); + const parent = slash < 0 ? '' : filePath.slice(0, slash); + const owner = parent.slice(parent.lastIndexOf('/') + 1); + const qualifiedName = lang === 'php' ? `App\\${owner}\\${stem}` : `${owner}.${stem}`; + parsedFiles.push( + probeFile(filePath, [ + ['Class', qualifiedName], + ['Function', qualifiedName], + ]), + ); + } + return parsedFiles; +} + +/** + * The import one file issues in the UNIQUE-LEAF layout `uniqueDir` produced. + * + * The TARGET axis is split from the DIRECTORY axis exactly the way `uniqueDir` + * and `collideDir` split it above — two flat functions, selected once — rather + * than a `collide ?` ternary threaded through seventeen languages' `local ? …` + * ladders. `local` picks in-repo vs external; the handful of MISS lines that + * are identical between the two shapes are duplicated on purpose, because the + * alternative is four levels of nesting in a single expression. + */ +function uniqueTarget(lang, { local, r, d, j, dirs }) { + if (lang === 'go') { + return local + ? `${GO_MODULE.modulePath}/src/pkg${d}` + : (r >>> 3) % 2 === 0 + ? ['fmt', 'os', 'net/http', 'encoding/json'][(r >>> 4) % 4] + : `github.com/org/repo${(r >>> 4) % 97}/pkg/util`; + } + if (lang === 'csharp') { + return local + ? `App.Ns${d}` + : (r >>> 3) % 2 === 0 + ? ['System', 'System.Threading.Tasks', 'System.Collections.Generic'][(r >>> 4) % 3] + : `Ghost${(r >>> 4) % 97}.Deep.Missing`; + } + if (lang === 'csharp_csproj') { + // The mix is the arm. `System` and `Ghost{n}.Deep.Missing` match NEITHER + // root namespace, so they `continue` straight out of the config loop + // (csharp.ts:231-241) and never reach the indexed leg at all — an arm built + // on the no-csproj arm's spelling mix would measure #2902 not at all. They + // are kept as the fast-`continue` control at 1 slot in 8; the other four + // external slots address a root namespace on purpose. + if (local) return `App.Ns${d}`; + const leg = (r >>> 3) % 5; + // Matches `App`, misses every directory: `dirPrefix = 'src/Missing{n}'`, + // whose last segment buckets to nothing. 2 slots in 8. + if (leg < 2) return `App.Missing${(r >>> 4) % 97}`; + // Matches `Lib`, whose `projectDir` is empty, so `dirPrefix` is slash-FREE + // and `candidateDirs` sweeps the last-segment keys — the one leg of the + // three whose cost is not constant in the corpus. See `_arms_note`. + if (leg === 2) return `Lib.Missing${(r >>> 4) % 97}`; + // The import IS a root namespace with no `projectDir`: `dirPrefix` is + // EMPTY, the query no last-segment bucket expresses, answered from + // `singleSegmentDirs`. + if (leg === 3) return 'Lib'; + return (r >>> 4) % 2 === 0 + ? ['System', 'System.Threading.Tasks', 'System.Collections.Generic'][(r >>> 5) % 3] + : `Ghost${(r >>> 4) % 97}.Deep.Missing`; + } + if (lang === 'dart') { + return local + ? `package:app/feature${d}/file${j}.dart` + : (r >>> 3) % 3 === 0 + ? ['dart:core', 'dart:async', 'dart:io'][(r >>> 4) % 3] + : `package:ext${(r >>> 4) % 97}/src/thing.dart`; + } + if (lang === 'kotlin') { + // A share of wildcard imports: `.*` lands on the package fan-out tier, + // which returns a LIST and is the only tier whose output is order-bearing. + return local + ? (r >>> 3) % 3 === 0 + ? `com.example.pkg${d}.*` + : `com.example.pkg${d}.File${j}` + : (r >>> 3) % 2 === 0 + ? ['java.util.List', 'kotlin.collections.Map', 'kotlinx.coroutines.flow.Flow'][ + (r >>> 4) % 3 + ] + : `com.ghost${(r >>> 4) % 97}.deep.Missing`; + } + if (lang === 'php') { + // Backslash-separated, the way a `use` statement is actually written; the + // resolver normalizes them. No composer.json is threaded (the adapter's + // `resolutionConfig` is left undefined), so every one of these lands on + // `suffixResolve` — the leg that ran one `findIndex` over every file per + // path part per extension, ~50 of them, and measured 96.40 ms per import at + // 20k files before #2901. + return local + ? `App\\Ns${d}\\File${j}` + : (r >>> 3) % 2 === 0 + ? [ + 'Psr\\Log\\LoggerInterface', + 'Symfony\\Component\\Console\\Command', + 'Doctrine\\ORM\\EntityManager', + ][(r >>> 4) % 3] + : `Vendor${(r >>> 4) % 97}\\Ghost\\Missing`; + } + if (lang === 'java') { + // Java has NO in-repo-namespace gate (#2910 is filed for it), so a JDK + // import genuinely can resolve to a local file — `java.util.List` would + // answer to a `util/List.java` anywhere in the repo, and the progressive + // stripping loop would find it by its bare basename. These spellings are + // chosen to miss on THIS corpus (whose files are all `File{i}.java` under + // `…/pkg{d}/`) and the resolved count is asserted, not assumed. + return local + ? (r >>> 3) % 3 === 0 + ? `com.example.pkg${d}.*` + : `com.example.pkg${d}.File${j}` + : (r >>> 3) % 2 === 0 + ? ['java.util.List', 'java.io.IOException', 'java.util.concurrent.ConcurrentHashMap'][ + (r >>> 4) % 3 + ] + : `com.google.common.vendor${(r >>> 4) % 97}.Missing`; + } + if (lang === 'cobol') { + // `COPY` takes a bare bookname. A share of the local ones is spelled in + // lower case: COBOL is case-insensitive and the resolver upper-cases the + // target, so those must resolve to the same file — free coverage of the + // one transformation on the lookup path. + return local + ? (r >>> 3) % 3 === 0 + ? `file${j}` + : `File${j}` + : (r >>> 3) % 2 === 0 + ? ['DFHAID', 'DFHBMSCA', 'SQLCA', 'CICSDEF'][(r >>> 4) % 4] + : `VENDOR${(r >>> 4) % 97}`; + } + if (lang === 'swift') { + // `import X` names an SPM MODULE, never a file, so there is no `.File{j}` + // spelling to mint: the target is the module and the answer is its whole + // file list. The misses are the frameworks that ship with the platform and + // the SPM packages that live in `.build/`, i.e. outside the corpus. + return local + ? `Mod${d}` + : (r >>> 3) % 2 === 0 + ? ['Foundation', 'UIKit', 'Combine', 'SwiftUI'][(r >>> 4) % 4] + : `ExternalPkg${(r >>> 4) % 97}`; + } + if (lang === 'rust') { + // `crate::mod{d}::thing` resolves by PROBING: `src/mod{d}/thing.rs`, + // `src/mod{d}/thing/mod.rs`, `src/mod{d}.rs`, then `src/mod{d}/mod.rs`, + // which hits. The `d % 7` slice has no `mod.rs` at that path and misses, + // which is where the resolved count comes from. + return local + ? `crate::mod${d}::thing` + : (r >>> 3) % 2 === 0 + ? ['std::collections::HashMap', 'tokio::sync::mpsc', 'serde::Deserialize'][(r >>> 4) % 3] + : `ghost${(r >>> 4) % 97}::Missing`; + } + if (lang === 'python') { + // Dotted absolute imports. The stdlib spellings and the unknown + // distributions both die at `hasRepoCandidate`, which is the gate that + // keeps `django.apps` off a local `accounts/apps.py`. + return local + ? `pkg${d}.file${j}` + : (r >>> 3) % 2 === 0 + ? ['os.path', 'collections.abc', 'django.db.models'][(r >>> 4) % 3] + : `vendor${(r >>> 4) % 97}.deep.missing`; + } + if (lang === 'javascript' || lang === 'typescript') { + // BARE specifiers, not relative ones. A relative import resolves by exact + // `Set.has` and never reaches `suffixResolve` — the leg that had no index + // for JavaScript until PR #2911 and cost 25 972 µs per import at 8000 + // files — so a corpus of `./sibling` imports would measure the wrong one. + return local + ? `src/mod${d}/file${j}` + : (r >>> 3) % 2 === 0 + ? ['react', 'lodash/fp', '@scope/ui/dist/index'][(r >>> 4) % 3] + : `vendor${(r >>> 4) % 97}/lib/missing`; + } + if (lang === 'vue') { + // Every in-repo import is `@/…`, so the alias branch runs on all of them. + // The `.vue` share carries its extension (SFC imports always do) and takes + // the exact-path leg; the `.ts` share omits it and takes the guessing leg. + return local + ? j % 3 === 0 + ? `@/mod${d}/File${j}` + : `@/mod${d}/File${j}.vue` + : (r >>> 3) % 2 === 0 + ? ['vue', 'pinia', '@vueuse/core'][(r >>> 4) % 3] + : `vendor${(r >>> 4) % 97}/lib/Missing.vue`; + } + if (lang === 'c' || lang === 'cpp') { + // `#include "comp{d}/file{j}.h"`. `j | 1` picks the HEADER half of the + // corpus — the even half is `.c`/`.cpp` and nothing includes those. The + // misses are the two kinds a real tree has: a system header that is not in + // the repo at all, and a vendored path that does not exist. + const h = HEADER_EXTENSION[lang]; + const jj = j | 1; + return local + ? `comp${jj % dirs}/file${jj}${h}` + : (r >>> 3) % 2 === 0 + ? ['stdio.h', 'stdlib.h', 'string.h'][(r >>> 4) % 3] + : `vendor${(r >>> 4) % 97}/missing${h}`; + } + if (lang === 'ruby') { + return local + ? `mod${d}/file${j}` + : (r >>> 3) % 2 === 0 + ? ['json', 'set', 'net/http', 'digest'][(r >>> 4) % 4] + : `gem${(r >>> 4) % 97}/missing/thing`; + } + throw unwiredLanguage('uniqueTarget', lang); +} + +/** + * The same import in the SHARED-LEAF layout `collideDir` produced. Each + * language's local spelling is chosen so this arm resolves exactly as many + * imports as the unique arm does (asserted): same workload, different layout. + */ +function collideTarget(lang, { local, r, d, j, dirs }) { + if (lang === 'go') { + return local + ? // The replicated package is addressed by the path it shares, so the + // module leg matches every service at once (`dirCount > 1`). + d % 5 === 1 && d % 7 !== 0 + ? `${GO_MODULE.modulePath}/internal/shared` + : `${GO_MODULE.modulePath}/svc${d}/internal` + : (r >>> 3) % 2 === 0 + ? ['fmt', 'os', 'net/http', 'encoding/json'][(r >>> 4) % 4] + : // Ends in the shared segment, so the GOPATH fallback walks the whole + // bucket three times and still returns null: the MISS path this arm + // exists to measure. + `github.com/org/repo${(r >>> 4) % 97}/internal`; + } + if (lang === 'csharp') { + return local + ? // This used to send the `d % 7` slice to `App.Src{d}.Vendor`, a + // namespace with no directory anywhere, to mirror the unique arm's + // nested-same-name slice, which also resolved to nothing. #2881 made + // that slice resolve, so the mirror has to as well — otherwise this arm + // stops resolving as many imports as `small`, which is the invariant + // that makes the two timings comparable and is asserted below. + `App.Src${d}.Models` + : (r >>> 3) % 2 === 0 + ? ['System', 'System.Threading.Tasks', 'System.Collections.Generic'][(r >>> 4) % 3] + : `Ghost${(r >>> 4) % 97}.Deep.Missing`; + } + if (lang === 'csharp_csproj') { + // Same five families as the unique arm, in the same proportions, so the + // resolved count is identical by construction (asserted). Two things change. + // + // The local spelling moves onto the SECOND config: the collide layout puts + // nothing under `src/`, so `projectDir: 'src'` addresses no directory here + // and `App.Src{d}.Models` would resolve nothing. `Lib` (`projectDir: ''`) + // addresses `Src{d}/Models` directly — the same relayout-not-reworkload + // substitution every other language makes in this function. + // + // And `dirsByLastSegment` collapses from one key per directory to the + // single key `Models`, which makes the slash-free SWEEP cheaper here than + // on the unique layout while making the bucket the nested slice walks hold + // every directory — the inverse of the go/csharp/dart collide arms, whose + // every term gets worse. See `_arms_note`. + if (local) return `Lib.Src${d}.Models`; + const leg = (r >>> 3) % 5; + if (leg < 2) return `App.Missing${(r >>> 4) % 97}`; + if (leg === 2) return `Lib.Missing${(r >>> 4) % 97}`; + if (leg === 3) return 'Lib'; + return (r >>> 4) % 2 === 0 + ? ['System', 'System.Threading.Tasks', 'System.Collections.Generic'][(r >>> 5) % 3] + : `Ghost${(r >>> 4) % 97}.Deep.Missing`; + } + if (lang === 'dart') { + return local + ? `package:app/pkg${j % dirs}/lib/src/mod${Math.floor(j / dirs)}.dart` + : (r >>> 3) % 3 === 0 + ? ['dart:core', 'dart:async', 'dart:io'][(r >>> 4) % 3] + : // A repeated basename under a directory nothing carries: both + // candidates walk the whole basename bucket and miss. + `package:ext${(r >>> 4) % 97}/other/mod${(r >>> 4) % 8}.dart`; + } + if (lang === 'kotlin') { + // Same wildcard share as the unique arm. This used to send the `d % 7` + // nested slice to `com.example.vendor${d}`, a package that exists nowhere, + // to mirror the unique arm's nested slice — which missed, because + // `dirChildren` required the parent to be the FIRST occurrence of its own + // name and `…/com/example/pkg${d}/inner/pkg${d}` therefore did not belong to + // `pkg${d}`. #2881 removed that rule, so the unique arm's nested wildcards + // resolve and the mirror has to as well, or this arm stops resolving as + // many imports as `small` — which is the invariant that makes the two + // timings comparable, and it is asserted below. + return local + ? (r >>> 3) % 3 === 0 + ? `com.example.models.*` + : `com.example.models.File${j}` + : (r >>> 3) % 2 === 0 + ? ['java.util.List', 'kotlin.collections.Map', 'kotlinx.coroutines.flow.Flow'][ + (r >>> 4) % 3 + ] + : `com.ghost${(r >>> 4) % 97}.deep.Missing`; + } + if (lang === 'php') { + // `Models\Mod{n}` is carried by every service, so the segment-suffix key it + // resolves through holds one entry no matter how many files exist: PHP + // answers from keyed maps and is collision-IMMUNE, which is what this arm + // asserts. The local spelling still always resolves, as it does on the + // unique layout — PHP's cascade strips leading segments, so even the + // nested-same-name slice is reachable by a shorter suffix. + return local + ? `App\\Models\\Mod${Math.floor(j / dirs)}` + : (r >>> 3) % 2 === 0 + ? [ + 'Psr\\Log\\LoggerInterface', + 'Symfony\\Component\\Console\\Command', + 'Doctrine\\ORM\\EntityManager', + ][(r >>> 4) % 3] + : `Vendor${(r >>> 4) % 97}\\Ghost\\Missing`; + } + if (lang === 'java') { + // Every file declares the same package despite living under different + // service paths. Exact and wildcard imports therefore exercise one growing + // declared-package bucket without relying on directory layout. + return local + ? (r >>> 3) % 3 === 0 + ? 'com.example.model.*' + : `com.example.model.File${j}` + : (r >>> 3) % 2 === 0 + ? ['java.util.List', 'java.io.IOException', 'java.util.concurrent.ConcurrentHashMap'][ + (r >>> 4) % 3 + ] + : `com.google.common.vendor${(r >>> 4) % 97}.Missing`; + } + if (lang === 'cobol') { + // The repeated basename is COBOL's ONLY collision axis, and its index is a + // keyed map, so this arm asserts immunity. It also reaches the tier + // tie-break the unique arm cannot: `Mod{n}` now names both a `.cpy` and a + // `.cbl`, and the copybook must win regardless of Set-iteration order. + return local + ? (r >>> 3) % 3 === 0 + ? `mod${Math.floor(j / dirs)}` + : `Mod${Math.floor(j / dirs)}` + : (r >>> 3) % 2 === 0 + ? ['DFHAID', 'DFHBMSCA', 'SQLCA', 'CICSDEF'][(r >>> 4) % 4] + : `VENDOR${(r >>> 4) % 97}`; + } + if (lang === 'swift') { + // Four modules instead of `dirs` of them, so the bucket a hit returns holds + // fileCount/4 files and grows with the corpus. Same in-repo share, same + // resolved count; the only thing that changed is bucket cardinality. + return local + ? `Mod${d % SWIFT_COLLIDE_MODULES}` + : (r >>> 3) % 2 === 0 + ? ['Foundation', 'UIKit', 'Combine', 'SwiftUI'][(r >>> 4) % 4] + : `ExternalPkg${(r >>> 4) % 97}`; + } + if (lang === 'rust') { + // ~2x the `::` segments of the unique arm, in both the hits and the misses, + // because SEGMENT COUNT is the only axis this resolver's cost has. The + // `d % 7` slice names a module that exists nowhere, mirroring the unique + // arm's `inner` slice, so the resolved count is unchanged. The external + // spellings run the prefix-shortening loop in `resolveModulePath` to the + // end — two `.has()` probes per shortened prefix — which is the longest + // path through the function and the one worth an absolute ceiling. + return local + ? d % 7 === 0 + ? `crate::l0::l1::l2::l3::l4::vendor${d}::thing::Inner` + : `crate::l0::l1::l2::l3::l4::mod${d}::thing::Inner` + : (r >>> 3) % 2 === 0 + ? [ + 'std::collections::hash_map::HashMap', + 'tokio::sync::mpsc::channel', + 'serde::de::value::MapDeserializer', + ][(r >>> 4) % 3] + : `ghost${(r >>> 4) % 97}::deep::nested::more::Missing`; + } + if (lang === 'python') { + // A `models` package in every service and a repeated `mod{n}.py` inside it, + // so `byBasename` holds one entry per service for each stem and the + // fewest-segments-then-lexicographic tie-break in `resolveAbsoluteFromFiles` + // actually has something to break. The external spelling shares the + // basename and still misses — `vendor{n}` fails `hasRepoCandidate`. + return local + ? `svc${j % dirs}.models.mod${Math.floor(j / dirs)}` + : (r >>> 3) % 2 === 0 + ? ['os.path', 'collections.abc', 'django.db.models'][(r >>> 4) % 3] + : `vendor${(r >>> 4) % 97}.models.mod0`; + } + if (lang === 'javascript' || lang === 'typescript') { + // `pkg{n}/src/mod{m}` in every package. `buildSuffixIndex` is a KEYED map + // that keeps one path per suffix, so this is the arm that asserts the + // ts-family resolver is collision-immune. The external spelling must not + // share the repeated stem, or it would suffix-match a real file and the + // corpus would stop being miss-heavy (measured: 67% resolved instead of + // 36% when it was `vendor{n}/src/mod{m}`). + return local + ? `pkg${j % dirs}/src/mod${Math.floor(j / dirs)}` + : (r >>> 3) % 2 === 0 + ? ['react', 'lodash/fp', '@scope/ui/dist/index'][(r >>> 4) % 3] + : `vendor${(r >>> 4) % 97}/src/ghost${(r >>> 4) % 8}`; + } + if (lang === 'vue') { + return local + ? j % 3 === 0 + ? `@/pkg${j % dirs}/components/Mod${Math.floor(j / dirs)}` + : `@/pkg${j % dirs}/components/Mod${Math.floor(j / dirs)}.vue` + : (r >>> 3) % 2 === 0 + ? ['vue', 'pinia', '@vueuse/core'][(r >>> 4) % 3] + : `vendor${(r >>> 4) % 97}/components/Ghost${(r >>> 4) % 8}.vue`; + } + if (lang === 'c' || lang === 'cpp') { + // A `mod{n}` header in every service's `include/`, which is what a C tree + // looks like. The basename bucket the suffix fallback walks now holds one + // candidate per service, so the depth-then-lexicographic tie-break decides + // — and the bucket grows with the corpus, which is why this arm carries its + // own scaling budget. + const h = HEADER_EXTENSION[lang]; + const jj = j | 1; + return local + ? `include/mod${Math.floor(jj / dirs)}${h}` + : (r >>> 3) % 2 === 0 + ? ['stdio.h', 'stdlib.h', 'string.h'][(r >>> 4) % 3] + : `vendor${(r >>> 4) % 97}/mod0${h}`; + } + if (lang === 'ruby') { + // `models/mod{n}.rb` in every package. Ruby answers `require` from a keyed + // suffix map, so the repeated basename cannot grow a bucket: this arm + // asserts that immunity, which is why its collide budget is the linear one. + return local + ? `svc${j % dirs}/lib/models/mod${Math.floor(j / dirs)}` + : (r >>> 3) % 2 === 0 + ? ['json', 'set', 'net/http', 'digest'][(r >>> 4) % 4] + : `gem${(r >>> 4) % 97}/missing/thing`; + } + throw unwiredLanguage('collideTarget', lang); +} + +/** + * One synthetic repository per language: the file set plus the import list each + * file issues. + */ +function buildRepo(lang, fileCount, pad = 0, shape = 'unique') { + const dirs = dirsFor(fileCount); + const files = buildFiles(lang, fileCount, pad, shape); + const mintTarget = shape === 'collide' ? collideTarget : uniqueTarget; + + const imports = []; + for (let i = 0; i < fileCount; i++) { + const from = files[i]; + for (let k = 0; k < IMPORTS_PER_FILE; k++) { + const r = mix(i * 65599 + k); + // ~3 in 8 imports resolve in-repo; the rest are external and run the + // whole cascade to completion (corpus property 1). + const local = r % 8 < 3; + const d = r % dirs; + const j = r % fileCount; + imports.push([from, mintTarget(lang, { local, r, d, j, dirs })]); + } + } + return { files, imports }; +} + +/** + * The per-pass state one resolver sees: the file set it is handed, and the + * `resolutionConfig` the orchestrator threads beside it. + * + * `allFilePaths` is a FRESH Set per pass on purpose — every per-file-set memo + * in `import-resolvers/per-file-set.ts` is keyed on that object's identity, so + * reusing one Set across passes would hide the index build after the first and + * let a rebuilt-per-import index look free from rep 2 onward. + * + * `config` is why this exists as a function rather than a `new Set(files)` at + * three call sites. Three languages here need one and they need three different + * things: + * + * - C and C++ take their HEADERS through `resolutionConfig`, not through + * `allFilePaths`. The phase hands the C resolver the `.c` files it + * classified and the header scan separately, and + * `augmentedFilePathsFor(allFilePaths)(headerPaths)` unions the two ONCE per + * pass — a two-input memo, so both inputs have to be pass-stable or it + * rebuilds an O(files) Set per include. Splitting the corpus here is what + * makes that union reachable at all; handing the resolver one pre-merged set + * would leave the memo, and the shape it exists for, unmeasured. + * - Vue takes `tsconfigPaths`, and the alias branch is the one leg of the + * shared ts-family resolver its arm covers that the other two do not. + * + * `csharp_csproj` is the precedent and stays where it is: a per-language + * CONTEXT over a corpus aliased to another language's, rather than a new axis. + * + * `parsedFiles` is the third pass-stable object, present for `CONTEXT_LANGS` + * and undefined for everyone else. It is built BEFORE the path set and the path + * set is derived FROM it, which is not a stylistic choice: `run.ts` does + * `new Set(parsedFiles.map((f) => f.filePath))`, so two independently built + * lists would be a shape the pipeline cannot produce. Fresh per pass for + * exactly the reason the Set is — `filesByDirectory` and `parsedFileByPath` are + * `perFileSet` memos keyed on this ARRAY's identity, so reusing one array would + * hide their build from rep 2 onward and `fastest()` reports the minimum. + */ +function newPass(lang, files, pad = 0) { + if (HEADER_EXTENSION[lang] !== undefined) { + const sources = []; + const headers = []; + for (const f of files) (f.endsWith(HEADER_EXTENSION[lang]) ? headers : sources).push(f); + return { allFilePaths: new Set(sources), config: new Set(headers) }; + } + if (lang === 'vue') { + return { allFilePaths: new Set(files), config: vueTsconfig(tsBaseUrlFor(pad)) }; + } + // javascript and typescript resolve their `src/mod{d}/file{j}` locals through + // `baseUrl`; without a config every arm would correctly resolve nothing and + // measure an empty branch (#2953 — see `tsBaseUrlConfig`). + if (lang === 'javascript' || lang === 'typescript') { + return { allFilePaths: new Set(files), config: tsBaseUrlConfig(tsBaseUrlFor(pad)) }; + } + if (CONTEXT_LANGS.includes(lang)) { + const parsedFiles = buildParsedFiles(lang, files); + restoreBenchmarkSideChannels(lang, parsedFiles); + return { + allFilePaths: new Set(parsedFiles.map((f) => f.filePath)), + config: undefined, + parsedFiles, + }; + } + return { allFilePaths: new Set(files), config: undefined }; +} + +/** + * The `{ parsedFiles, parsedImport }` object `run.ts` mints per import — per + * import there too, so this allocation is production's, not the bench's. + * + * `undefined` when the pass carries no parsed workspace, which happens in + * exactly one place: the CONTROL half of the `context` arm, whose whole job is + * to prove the arm can tell the two call shapes apart. + */ +const contextFor = (pass, parsedImport) => + pass.parsedFiles === undefined + ? undefined + : { parsedFiles: pass.parsedFiles, parsedImport, filesSkipped: 0 }; + +function restoreBenchmarkSideChannels(lang, parsedFiles) { + if (lang !== 'java') return; + javaScopeResolver.loadResolutionConfig?.(''); + for (const parsed of parsedFiles) javaScopeResolver.applyCaptureSideChannel?.(parsed); +} + +/** The timed loop. One `newPass` per pass, so every pass pays exactly one index + * build — see `newPass`. */ +function resolveAll(lang, files, imports, pad = 0) { + const pass = newPass(lang, files, pad); + let sink = 0; + for (const [from, target] of imports) { + const hit = resolveOne(lang, from, target, pass); + if (hit !== null) sink++; + } + return sink; +} + +function resolveOne(lang, from, target, pass) { + const allFilePaths = pass.allFilePaths; + if (lang === 'go') return resolveGoImportTarget(target, from, allFilePaths, GO_MODULE); + if (lang === 'dart') return resolveDartImportTarget(target, from, allFilePaths); + if (lang === 'ruby') return resolveRubyImportTarget(target, from, allFilePaths); + if (lang === 'kotlin') { + return resolveKotlinImportTarget( + { kind: 'named', localName: 'X', importedName: 'X', targetRaw: target }, + { fromFile: from, allFilePaths }, + ); + } + // `pass.config` is undefined for PHP, so no composer.json: the PSR-4 mapping + // legs are skipped and every import lands on the suffix cascade #2901 + // indexed. The FIFTH argument is the production one, and + // `importedSymbolKind: 'function'` is what opens the named/alias leg over + // `filesByDirectory(context.parsedFiles)` — see THE FIFTH ARGUMENT. It runs + // on every import rather than on a share of them because the leg is the point + // of the arm and it costs the cascade nothing: `resolvePhpImportInternal` + // has already returned by the time the leg is consulted, so this arm still + // measures everything it measured before, plus the leg. + // + // `importedName` is inert here and stays 'X' like the java and kotlin arms: + // the leg derives the name it matches on from `targetRaw` itself, so + // computing a real one would be a split per import charged to the timed loop + // for a field nothing reads. + if (lang === 'php') { + return resolvePhpImportTargetInternal( + target, + from, + allFilePaths, + pass.config, + contextFor(pass, { + kind: 'named', + localName: 'X', + importedName: 'X', + targetRaw: target, + importedSymbolKind: 'function', + }), + ); + } + if (lang === 'java') { + const parsedImport = { + kind: 'named', + localName: 'X', + importedName: 'X', + targetRaw: target, + }; + return javaScopeResolver.resolveImportTarget( + target, + from, + allFilePaths, + pass.config, + contextFor(pass, parsedImport), + ); + } + // The `ScopeResolver` hook itself — COBOL's copy index has no other export. + if (lang === 'cobol') return cobolScopeResolver.resolveImportTarget(target, from, allFilePaths); + if (lang === 'swift') { + return resolveSwiftImportTarget( + { kind: 'namespace', localName: 'X', importedName: 'X', targetRaw: target }, + { fromFile: from, allFilePaths }, + ); + } + if (lang === 'rust') return resolveRustImportTarget(target, from, allFilePaths, undefined); + if (lang === 'python') { + // `from import X` — the spelling the orchestrator actually hands + // the provider, and the ONLY one that reads `context.parsedFiles`: a + // `namespace` import makes `pythonImportedSubmoduleTarget` return null, the + // submodule-precedence branch never runs and the field is dead. This arm + // used to pass that synthetic namespace spelling and skipped the branch + // for exactly that reason, which is what made the field unmeasurable. + // + // So the arm now pays the branch: a package probe, a `parsedFileByPath` + // lookup over the resolved package's `localDefs`, and a submodule probe — + // up to three entries into the resolver per import, which is why its ms + // numbers are several times what the namespace spelling read. That IS the + // per-import cost of a `from … import …` in production. + // + // 'X' names nothing the corpus declares, on purpose: `pythonFileExportsName` + // then scans the whole `localDefs` list and returns false, so every + // resolving import runs the submodule probe too. That is the expensive + // half of the branch — a name the package DOES export short-circuits at + // the first def — and matches corpus property 1 above. + // + // The `{ fromFile, allFilePaths, parsedFiles }` shape is exactly what + // `pythonScopeResolver` builds from the context before calling this. + return resolvePythonImportTarget( + { kind: 'named', localName: 'X', importedName: 'X', targetRaw: target }, + { fromFile: from, allFilePaths, parsedFiles: pass.parsedFiles }, + ); + } + if (lang === 'javascript') { + return jsResolveImportTarget(target, from, allFilePaths, pass.config); + } + if (lang === 'vue') return vueResolveImportTarget(target, from, allFilePaths, pass.config); + // TypeScript, C and C++ go through the registered `ScopeResolver` hook rather + // than an inner resolver, because for all three the thing under test lives IN + // the adapter: TypeScript's `tsPassCacheFor` memo is private to + // `typescript/scope-resolver.ts`, and C's and C++'s `augmentedFilePathsFor` + // is private to theirs. Calling past it would benchmark a copy of the adapter + // instead of the adapter. + if (lang === 'typescript') { + return typescriptScopeResolver.resolveImportTarget(target, from, allFilePaths, pass.config); + } + if (lang === 'c') { + return cScopeResolver.resolveImportTarget(target, from, allFilePaths, pass.config); + } + if (lang === 'cpp') { + return cppScopeResolver.resolveImportTarget(target, from, allFilePaths, pass.config); + } + if (lang === 'csharp' || lang === 'csharp_csproj') { + return resolveCsharpImportTarget( + { kind: 'namespace', localName: '_', importedName: '_', targetRaw: target }, + { + fromFile: from, + allFilePaths, + // The ONLY difference between the two C# arms. Present, the adapter + // takes the csproj branch and never falls through to the no-csproj legs. + ...(lang === 'csharp_csproj' ? { csharpConfigs: CSPROJ_CONFIGS } : {}), + }, + ); + } + throw unwiredLanguage('resolveOne', lang); +} + +/** The single untimed identity pass, producing BOTH non-timing results: the + * distinct `from|target → result` set the fingerprint hashes, and `resolved` + * counted over every import. One resolve per DISTINCT pair — on a fixed file + * set the resolvers are pure, so a repeated pair can only re-derive what the + * first occurrence already recorded, and the memoized `key → wasNull` answers + * the count for the repeat. Merged from two passes that each walked the whole + * corpus; the duplicate resolves measured ~1.15 s of an 11.4 s run. + * + * Deliberately NOT shared with `resolveAll`, which is the TIMED loop: the memo + * that makes this pass cheap is exactly what would hide the cost that loop + * exists to measure. */ +function identityPass(lang, files, imports, pad = 0) { + const pass = newPass(lang, files, pad); + const outcomes = new Set(); + const wasNullByKey = new Map(); + let resolved = 0; + for (const [from, target] of imports) { + const key = `${from}\u0000${target}`; + let wasNull = wasNullByKey.get(key); + if (wasNull !== undefined) { + if (!wasNull) resolved++; + continue; + } + const hit = resolveOne(lang, from, target, pass); + wasNull = hit === null; + wasNullByKey.set(key, wasNull); + if (!wasNull) resolved++; + const rendered = renderResolved(hit); + outcomes.add(`${key}\u0000${rendered}`); + } + return { outcomes, resolved }; +} + +/** One resolver answer as a comparable string. Kotlin and Java can return a + * LIST (their wildcard tier), so the array form is part of the shape, and + * `` keeps a miss distinct from a resolver that answered the empty + * string. Shared by the fingerprint above and by the `context` arm below, so + * the two never drift into reporting one answer two ways. */ +function renderResolved(hit) { + if (hit === null) return ''; + return Array.isArray(hit) ? hit.join(',') : hit; +} + +/** MIN, not median: both scales are timed in one process and every error source + * (GC, scheduler preemption, a noisy CI neighbour) is additive, so the fastest + * observed pass is the closest estimate of the uncontended cost. */ +function fastest(values) { + return Math.min(...values); +} + +function timeResolution(lang, files, imports, reps, pad = 0) { + for (let w = 0; w < WARMUP; w++) resolveAll(lang, files, imports, pad); + const samples = []; + for (let r = 0; r < reps; r++) { + const t0 = performance.now(); + resolveAll(lang, files, imports, pad); + samples.push(performance.now() - t0); + } + return fastest(samples); +} + +/** + * One WARMED pass, used only to size `reps` for the language. + * + * Run on the `small` arm, and `small` is measurably the cheapest of the five + * for every language where the answer can differ — all six that come out below + * `REPS_MAX` (csharp_csproj, ruby, php, javascript, typescript, vue). Three + * languages do have a cheaper arm — cobol's `collide` by 45%, kotlin's by 7%, + * dart's `deep` by a few percent — and all three sit so far under + * `REPS_CHEAP_MS` that either reading returns 15. `small` is also the arm + * `small_ms_ceiling` bounds, so it is the one number here a reader already has + * an intuition for. + * + * Warmed rather than taken from the WARMUP passes themselves: an unwarmed pass + * reads several times high, which would push the expensive languages to + * `REPS_MIN` for the wrong reason. + */ +function probeMs(lang, files, imports, pad = 0) { + for (let w = 0; w < WARMUP; w++) resolveAll(lang, files, imports, pad); + const t0 = performance.now(); + resolveAll(lang, files, imports, pad); + return performance.now() - t0; +} + +/** + * Retained JS heap of everything one language derives from one file set — + * measured by RESOLVING AN IMPORT through it, never by calling a builder. + * + * THE ARM READS WHAT THE LANGUAGE READS, and that is now the whole design. + * Until #2903 was extended to the two suffix maps, four of these arms called + * `getWorkspaceFileIndex(set)` directly and read `index.all.length`, which asks + * no suffix question at all. That was harmless only while `buildSuffixIndex` + * built both maps eagerly. The moment they went lazy the direct call built NO + * map, all four arms reported 0 B at 32 000 files, and 0 B is under every + * ceiling — four gates silently became ceilings over nothing, which is exactly + * the failure this file's header warns about for rust and cobol. Driving the + * real resolver cannot fail that way: whatever maps the language forces are the + * maps it forces in production, and if a resolver starts asking a new question + * the number moves on its own instead of needing this file edited. + * + * It also happens to be the only form available for half of these languages — + * Swift's `getSwiftModuleIndex`, Python's `getPythonFileIndex`, C's + * `suffixIndex` and the ts-family `passCacheFor` are private to their modules, + * and exporting four builders to feed a bench would widen four module surfaces + * for a measurement's convenience. Now that all eight arms use one form, the + * readings ARE comparable to one another (they were not before). + * + * GROWTH form, not the release form `bench/cfg/measure.mjs` uses: every index + * here is memoized in a `WeakMap` keyed on the Set, so releasing it means + * releasing the Set too, which would fold the Set's own cost into the delta. + * Here the pass is live across BOTH samples and the `files` array holds the + * path strings, so the delta is the derived structures' own footprint and not + * the paths they point at. For C and C++ it legitimately includes the augmented + * Set, which is part of what they hold; for every language it includes the one + * or two resolve-cache entries the probe leaves behind. + */ +function retainedPassBytes(lang, files, probeTarget, pad = 0) { + const pass = newPass(lang, files, pad); + // See `HEAP_RETAINED`: nothing built for this language is released until the + // next one starts, so no deferred collection can land between the two samples + // below and cancel part of the delta. + HEAP_RETAINED.push(pass); + GC(); + const before = process.memoryUsage().heapUsed; + const hit = resolveOne(lang, files[0], probeTarget, pass); + GC(); + const after = process.memoryUsage().heapUsed; + // A HIT would mean the reading is a materialized answer rather than the + // index, and — for the languages whose cascade returns early — that the legs + // past the hit were never reached and their structures never built. + if (hit !== null) { + throw new Error(`heap probe '${probeTarget}' resolved for ${lang}; it must MISS: ${hit}`); + } + // Fails loudly if the corpus ever stops being one distinct path per file, + // which would silently shrink every reading here. + const size = pass.allFilePaths.size + (pass.config instanceof Set ? pass.config.size : 0); + if (size !== files.length) { + throw new Error(`heap arm corpus is not distinct: ${size} of ${files.length}`); + } + return Math.max(0, after - before); +} + +/** + * Every pass this arm builds, held alive ON PURPOSE until the next language + * starts. + * + * A `heapUsed` delta is only the new structures if nothing OLD is released + * between its two samples, and that is not a property a forced GC can be + * trusted to establish: measured, the previous read's index survived a + * two-cycle collect at the next read's baseline and was dropped by the collect + * before its second sample, so the two cancelled and the arm reported 249 200 B + * for a 9.3 MB index (PHP) and 329 064 B for a 6.7 MB one (JavaScript, once, + * non-reproducibly — the same defect with a different language's timing). + * + * Holding the passes removes the precondition instead of tuning it: nothing a + * measurement window depends on is ever collectable inside it, so the delta + * cannot absorb a late free no matter how many cycles the collector needs. + * Byte-identical readings at two and at four `gc()` cycles are the evidence + * that it works, where without it the two disagree by 9 MB. + * + * Emptied once per language, in `measureHeap`, which is the one place a late + * free is harmless: it happens before that language's first baseline and + * outside both of its measurement windows, and it is followed by a drain deeper + * than any chain here has needed. Never emptying at all also works and is what + * this was first measured with, but it peaks at ~380 MB and costs 4.5 s, + * because every forced collection from that point on has to mark it. + */ +const HEAP_RETAINED = []; + +/** + * The import each heap language resolves to force its build. A MISS in every + * case (asserted above), so the reading is the index and not a materialized + * answer, and so the cascade runs to completion instead of returning at the + * first leg. + * + * Each spelling is one the language's own corpus already mints in + * `uniqueTarget`, so the arm forces the same read pattern the timing arms do — + * which after #2903 is what decides the number: + * + * - `csharp` and `java` ask `index.get` and never `getInsensitive`, so the + * case-folded map is never built (49.6% of the eager Java index was dead); + * - `php` asks `getInsensitive` and never `get` (49.4% dead), and builds its + * own first-proper-suffix map on top; + * - `ruby` and the ts family read `get(s) || getInsensitive(s)`, so they pay + * for both — the second one DERIVED from the first, which is why they cost + * less than two independent traversals; + * - `csharp_csproj` additionally asks `getFilesInDir`, forcing the `dirMap` + * #2903 made lazy. It is the witness that the read pattern IS the + * footprint: same corpus and same `getWorkspaceFileIndex` as `csharp`, + * three times the retained bytes. + */ +const HEAP_PROBE_TARGET = { + csharp: 'Ghost0.Deep.Missing', + // Matches the `App` root namespace and no directory, so it runs the config + // loop's single-file leg (`get` + `getInsensitive`) AND its directory leg + // (`getFilesInDir`) before answering null — the three-map read pattern. + csharp_csproj: 'App.Missing0', + ruby: 'gem0/missing/thing', + php: 'Vendor0\\Ghost\\Missing', + java: 'com.google.common.vendor0.Missing', + javascript: 'vendor0/lib/missing', + python: 'vendor0.deep.missing', + c: 'vendor0/missing.h', + // The entries below cover the BOUNDED tier — see `HEAP_BOUNDED`, which + // derives to cobol, swift and rust; the rest were promoted. Same rule as the + // budgeted ones above: a spelling `uniqueTarget` already mints for that language, and + // one that MISSES, so the reading is the index and the cascade runs to the + // end. Chosen from the miss family that reaches furthest into each cascade: + // - `go` takes the GOPATH fallback, one `filesDirectlyInPkgDir` per path + // segment, which is the leg that forces `PackageDirIndex`; + // - `dart` is an external package, so BOTH candidate paths miss and both + // walk the basename bucket to completion; + // - `kotlin` misses in `suffixByStem`, the map its four-tier cascade builds; + // - `cobol` misses in both tier maps, `swift` in `byModule`, and `rust` + // probes candidate paths and builds nothing — that last is the reading + // the exclusion rests on; + // - `typescript`, `vue` and `cpp` carry the same spelling shape as the + // `javascript` and `c` arms they are excluded as duplicates OF, so the + // bound compares like with like. `vue`'s is bare rather than `@/…` + // because the alias branch rewrites to `src/` and would resolve. + go: 'github.com/org/repo0/pkg/util', + dart: 'package:ext0/src/thing.dart', + kotlin: 'com.ghost0.deep.Missing', + cobol: 'VENDOR0', + swift: 'ExternalPkg0', + rust: 'ghost0::Missing', + typescript: 'vendor0/lib/missing', + vue: 'vendor0/lib/Missing.vue', + cpp: 'vendor0/missing.hpp', +}; + +/** + * `buildFiles` mints every path with a template literal, and V8 represents + * those as ROPES — the concatenation is not materialized until something forces + * it. The first traversal that slices a path (`lastIndexOf('/')`, `toLowerCase`, + * every index builder here) flattens it, which allocates the flat string AND + * drops the rope's now-unreachable pieces, so a build measured over an + * unflattened corpus reports the index MINUS that net release: measured 11% + * low, uniformly, on every language whose index slices paths. + * + * It biased the arm in the one direction that matters. `bytes_small` was read + * over a corpus a discarded warm-up pass had already flattened and + * `bytes_large` over a fresh one, so every `ratio` here was ~0.85-0.89 for + * structures that are exactly linear in the file count — the ratio budget was + * bounding an artefact. Flattened first, all eight read 0.99-1.02. + * + * It also retires the warm-up pass, which was never about JIT: with the corpus + * flat, a language's first and second reads of the same file count agree to + * within 0.3%. + */ +function flatten(files) { + for (const file of files) file.lastIndexOf('/'); + return files; +} + +function measureHeap(lang) { + if (GC === null) return null; + // Release the PREVIOUS language's passes here and nowhere else, then drain + // them twice over. This is the one point at which a deferred collection is + // free: it is before this language's first baseline and outside both of its + // measurement windows, so however many cycles the release needs, it cannot + // land between a `before` and an `after`. + HEAP_RETAINED.length = 0; + GC(); + GC(); + const probe = HEAP_PROBE_TARGET[lang]; + const read = (files) => retainedPassBytes(lang, files, probe); + const small = flatten(buildFiles(lang, HEAP_SMALL, HEAP_PAD, 'unique')); + const bytesSmall = read(small); + const large = flatten(buildFiles(lang, HEAP_LARGE, HEAP_PAD, 'unique')); + const bytesLarge = read(large); + return { + files_small: HEAP_SMALL, + files_large: HEAP_LARGE, + path_segments: small[0].split('/').length, + probe, + bytes_small: bytesSmall, + bytes_large: bytesLarge, + mib_large: Number((bytesLarge / 1024 / 1024).toFixed(2)), + ratio: Number((bytesLarge / bytesSmall / (HEAP_LARGE / HEAP_SMALL)).toFixed(3)), + }; +} + +/** + * The `context` arm's corpora — one per `CONTEXT_LANGS` entry, each a handful + * of files carrying ONE import whose answer DIFFERS between the production + * five-argument call and the three-argument one this harness used to make. + * + * That difference is the whole arm. PHP and Python need it because their main + * corpus answers agree with the fallback. Java's main fingerprint also catches + * a dropped context, but this tiny positive probe isolates the adapter contract + * from aggregate corpus changes. Timing cannot prove any of these; a dropped + * context makes the arms faster, and nothing here has a lower bound on ms. + * + * Both are resolved THROUGH `resolveOne`, not through the resolvers directly, + * because what is under test is this file's threading rather than the + * resolvers' behaviour. The control differs in exactly one thing: + * `pass.parsedFiles` is undefined, which `contextFor` turns into no fifth + * argument at all. + */ +const CONTEXT_PROBE = { + /** + * `use function App\Ns0\Dup;` where the CLASS `Dup` lives in `Dup.php` and + * the FUNCTION `Dup` lives in `Helpers.php`. PHP keeps the two in separate + * symbol tables and PSR-4 maps only the class, which is the case the leg + * exists for: the suffix cascade answers the file whose NAME matches the last + * segment, the leg answers the file that DECLARES the function. Two distinct + * non-null paths, so neither half of the arm can be mistaken for a miss, and + * `Alpha.php` is a third file in the same directory so the candidate gather + * has something to reject. + */ + php: { + from: 'src/App/Ns0/Alpha.php', + target: 'App\\Ns0\\Dup', + parsedFiles: [ + probeFile('src/App/Ns0/Alpha.php', [['Class', 'App\\Ns0\\Alpha']]), + probeFile('src/App/Ns0/Dup.php', [['Class', 'App\\Ns0\\Dup']]), + probeFile('src/App/Ns0/Helpers.php', [['Function', 'App\\Ns0\\Dup']]), + ], + }, + /** A declared package resolves its type only when the parsed workspace arrives. */ + java: { + from: 'app/Main.java', + target: 'com.example.model.User', + parsedFiles: [ + javaProbeFile('app/Main.java', 'app'), + javaProbeFile('weird/path/User.java', 'com.example.model'), + ], + }, + /** + * `from pkg import X`, with `pkg/__init__.py` exporting `X` AND a same-named + * submodule `pkg/X.py` beside it — the precedence CPython documents and the + * one `pythonFileExportsName` exists to reproduce. With the parsed workspace + * the package's own export wins (`pkg/__init__.py`); without it the export is + * invisible, the submodule probe runs and `pkg/X.py` wins. + * + * `X` rather than a prettier name because `resolveOne` passes `importedName: + * 'X'`: the probe is tied to the spelling the timing arms use, so changing + * one without the other fails here. + * + * This corpus also catches a revert to the synthetic `namespace` spelling, + * which no exact-value assertion could: that spelling never reads + * `parsedFiles`, so BOTH halves answer `pkg/__init__.py` and the + * with/without inequality below is what notices. + */ + python: { + from: 'app/main.py', + target: 'pkg', + parsedFiles: [ + probeFile('pkg/__init__.py', [['Function', 'pkg.X']]), + probeFile('pkg/X.py', [['Function', 'pkg.X.run']]), + probeFile('app/main.py', [['Function', 'app.main.run']]), + ], + }, +}; + +/** Resolve the probe twice through `resolveOne` — once with the pass's parsed + * workspace, once without — and report both answers. Deterministic and + * microseconds, so it runs in report mode too. */ +function measureContext(lang) { + const { from, target, parsedFiles } = CONTEXT_PROBE[lang]; + const allFilePaths = new Set(parsedFiles.map((f) => f.filePath)); + const answer = (files) => { + restoreBenchmarkSideChannels(lang, files ?? []); + return renderResolved( + resolveOne(lang, from, target, { allFilePaths, config: undefined, parsedFiles: files }), + ); + }; + return { + target, + with_context: answer(parsedFiles), + without_context: answer(undefined), + }; +} + +function fingerprint(outcomes) { + return crypto + .createHash('sha256') + .update([...outcomes].sort().join('\n')) + .digest('hex'); +} + +const CHECK = process.argv.includes('--check'); + +// The heap arm is a primary regression detector, but it can only be measured +// with a forced GC. Rather than let `--check` silently PASS with the heap gate +// skipped (a green no-op if someone drops --expose-gc), fail loudly. +if (CHECK && GC === null) { + process.stderr.write( + '[import-target --check] FAIL: the retained-heap arm requires --expose-gc. ' + + 'Run: node --expose-gc --import tsx bench/import-target/measure.mjs --check\n', + ); + process.exit(1); +} + +/** + * Every arm, and the registered language each one exercises. + * + * This used to be a hand-written list of seventeen strings under a comment + * claiming it was "every language in `SCOPE_RESOLVERS`" — a claim nothing in + * the file could check, because the file never imported the registry. Adding a + * resolver to `pipeline/registry.ts` is two lines, neither of which is this + * one, so a seventeenth registered language would have shipped ungated and + * printed PASS. That is not a hypothetical failure mode: JavaScript reached + * `suffixResolve` with no index at all and measured 25 972 µs per import at + * 8000 files (PR #2911) for exactly as long as nothing gated it. + * + * So the list is DERIVED and the claim is ASSERTED. `LANGS` is this table's + * keys, and the `--check` inventory arm below fails when a registered resolver + * has no arm here (or an arm names a language the registry does not have) — + * the same shape `test/unit/scope-resolution/import-target-index-reuse.contract.test.ts` + * uses ten files away, and the same "one row per language" table + * `bench/cfg/measure.mjs` keeps. + * + * The mapping is many-to-one on purpose: `csharp` and `csharp_csproj` are two + * arms over one registered resolver, differing only in whether `csharpConfigs` + * is supplied, because the no-csproj arm returns before it can reach the leg + * #2902 indexed. + */ +const LANG_REGISTRY = { + go: SupportedLanguages.Go, + csharp: SupportedLanguages.CSharp, + csharp_csproj: SupportedLanguages.CSharp, + dart: SupportedLanguages.Dart, + ruby: SupportedLanguages.Ruby, + kotlin: SupportedLanguages.Kotlin, + php: SupportedLanguages.PHP, + java: SupportedLanguages.Java, + cobol: SupportedLanguages.Cobol, + swift: SupportedLanguages.Swift, + rust: SupportedLanguages.Rust, + python: SupportedLanguages.Python, + javascript: SupportedLanguages.JavaScript, + typescript: SupportedLanguages.TypeScript, + vue: SupportedLanguages.Vue, + c: SupportedLanguages.C, + cpp: SupportedLanguages.CPlusPlus, +}; +const LANGS = Object.keys(LANG_REGISTRY); +/** + * The heap arm's SECOND tier: every arm that is not budgeted, and the reason it + * is a `filter` over `LANGS` rather than a second list beside `HEAP_BUDGETED`. + * + * The two tiers partition `LANGS` by construction, so there is no third state a + * language can be in — the state the nine spent this file's whole life in, + * where "not budgeted" and "not measured" were the same thing and neither was + * derived from anything. Adding a registered language now costs a bound whether + * or not anyone thinks about memory: the inventory arm gives it a `LANGS` row, + * this line gives it a tier, and the presence check below fails until it has a + * key. Deriving it also means the two tiers cannot overlap or leave a gap, which + * two hand-written lists could do in either direction. + * + * A bound and NOT a floor, deliberately, and the boundary is the one thing here + * worth re-reading before moving a language across it: a floor asserts "this + * arm is still measuring something", which is a claim about an index the file + * has budgeted, and rust's 16 B cannot carry it. What every one of the nine CAN + * carry is "the exclusion still holds" — that this language has not grown an + * index since it was left out. See the TIER TWO loop at the foot of the file, + * and `_heap_bound_note` in baselines.json for each language's reason. + */ +const HEAP_BOUNDED = LANGS.filter((lang) => !HEAP_BUDGETED.includes(lang)); +/** name, file count, depth padding, directory/basename layout. */ +const ARMS = [ + ['small', SMALL, 0, 'unique'], + ['large', LARGE, 0, 'unique'], + ['deep', SMALL, DEEP_PAD, 'unique'], + ['collide', SMALL, 0, 'collide'], + ['collide_large', LARGE, 0, 'collide'], +]; +/** Derived, never hand-written: the shape/fingerprint gate below iterates these + * names, so a new arm is asserted by construction rather than measured, + * printed and silently left out of the gate. */ +const SCALES = ARMS.map(([name]) => name); +const report = {}; +for (const lang of LANGS) { + const scales = {}; + // Sized once per language, from the FIRST arm — `small`, the cheapest — so + // all five arms share one estimator and the four ratios below stay + // comparisons of like with like. See `repsFor`. + let reps = null; + for (const [name, fileCount, pad, shape] of ARMS) { + const { files, imports } = buildRepo(lang, fileCount, pad, shape); + const { outcomes, resolved } = identityPass(lang, files, imports, pad); + if (reps === null) reps = repsFor(probeMs(lang, files, imports, pad)); + scales[name] = { + files: files.length, + imports: imports.length, + // Reported, not asserted on its own: a corpus edit that collapsed the + // resolved share would still produce a "valid" fingerprint over far less. + resolved, + distinct_outcomes: outcomes.size, + ms: Number(timeResolution(lang, files, imports, reps, pad).toFixed(3)), + fingerprint: fingerprint(outcomes), + }; + } + report[lang] = { + ...scales, + // Reported so a triager can see which estimator produced the five ms + // numbers above; environment-derived, so never asserted. + reps, + scaling_ratio: Number((scales.large.ms / scales.small.ms / (LARGE / SMALL)).toFixed(3)), + // `scaling_ratio` divides the file count out, so it is scale-invariant and + // structurally cannot see a cost that grows with path DEPTH instead — and + // `buildSuffixIndex` (C#, Ruby, PHP, Java, and the whole ts family) and + // Kotlin's `suffixByStem` all emit one entry per '/' in a path, and + // Python's ancestor walk rebuilds one prefix per component PER IMPORT. + // Same file count, ~6x the components. + depth_ratio: Number((scales.deep.ms / scales.small.ms).toFixed(3)), + // Same measurement on the shared-leaf layout. Legitimately above the 1.8 + // budget for go/csharp/dart — see the scope-of-claim note in the header. + collide_scaling_ratio: Number( + (scales.collide_large.ms / scales.collide.ms / (LARGE / SMALL)).toFixed(3), + ), + fingerprint: scales.large.fingerprint, + }; +} + +// AFTER every timing arm, never interleaved with them, and now for a second +// reason as well as the first. The first: the heap arm allocates a 32k-path +// corpus and a ~70 MiB index per language, and leaving that behind for the next +// language's timed loop to collect would tax an arm it has nothing to do with. +// The second: `HEAP_RETAINED` holds a language's whole corpus and index alive +// across both of its reads — up to ~92 MiB for `csharp_csproj` — and that must +// not overlap a measurement of time. +// +// `LANGS`, not `HEAP_BUDGETED`: which tier a language is in decides its GATE, +// not whether it is read. Measured cost of the nine extra arms is 1.37 s — this +// phase goes 2.06 s -> 3.43 s, of which kotlin alone is 0.57 s. See COST. +for (const lang of LANGS) report[lang].heap = measureHeap(lang); + +// Deterministic and microseconds — it resolves six imports over three tiny +// corpora — so unlike the heap arm it neither needs nor deserves isolation from +// the timing phase. It runs last only because it reads best beside the heap arm +// in the report. +for (const lang of CONTEXT_LANGS) report[lang].context = measureContext(lang); + +if (!CHECK) { + console.log(JSON.stringify(report, null, 2)); + process.exit(0); +} + +const baseline = JSON.parse(fs.readFileSync(BASELINE_PATH, 'utf-8')); +const failures = []; + +/** + * PRESENCE, for one budget, in the one place that spells the reason. + * + * A missing budget is a DELETED GATE, not a passing arm: `got > undefined` is + * `false`, `ceiling * undefined` is `NaN` and `bytes < NaN` is `false`, so every + * comparison in this file answers "within budget" for every possible + * measurement the moment its key stops being a number. Each of the three call + * sites below is one deleted key away from a silent no-op, and the run still + * prints PASS. + * + * `Number.isFinite` rather than `typeof === 'number'`: over JSON input the two + * agree (JSON cannot express NaN or Infinity), and the stricter one is the one + * whose name says what the gate needs. + * + * The two per-site facts stay the caller's, because they are what a triager acts + * on: `reads` is the comparison that silently stopped gating, quoted, and + * `scope` is what deleting this one key actually costs — a single arm, or all + * eight at once. Only the shared framing and the shared trailing sentence live + * here. Returns the message rather than pushing it, so the timing loop can + * `continue` past a budget it must not then compare against. + */ +const requireNumericBudget = ({ key, value, reads, scope }) => + Number.isFinite(value) + ? null + : `no numeric ${key} in baselines.json — a missing budget is a DELETED GATE, not a passing ` + + `arm: the comparison it gates reads \`${reads}\`, which is false for every possible ` + + `measurement. ${scope} Deterministic: a re-run will not change it.`; + +/** + * The REVERSE direction of a reconciliation: every key declared in `label` that + * `codeList` does not name. + * + * The forward direction ("the code has an arm with no budget") is a presence + * check inside whichever loop iterates the code's list. This is the other way + * round — a budget, a baseline block or a registry row for an arm that is never + * measured — and no forward check can see it, because the thing it names is + * exactly the thing nothing iterates. + * + * `codeListName` and `why` stay the caller's: which list is authoritative and + * what the orphan costs are the two facts that differ between the three arms, + * and flattening them would leave a triager with a name and no reading of it. + */ +function expectNoOrphanKeys(label, declaredKeys, codeList, codeListName, why) { + for (const key of declaredKeys) { + if (codeList.includes(key)) continue; + failures.push( + `${label} has an entry for '${key}', which is not in ${codeListName} — ${why} ` + + `Deterministic: a re-run will not change it.`, + ); + } +} + +/** The corpus-shape facts asserted for one timing scale. */ +const SCALE_SHAPE = { + fields: ['files', 'imports', 'resolved', 'distinct_outcomes', 'fingerprint'], + why: + 'the corpus changed shape or the resolver changed its answer for this arm. Every scale is ' + + 'asserted separately: the arms differ only in padding and layout, so a defect that touches ' + + 'one of them alone moves nothing in the others.', +}; +/** The same, for the heap arm — the four inputs that decide what it measures. + * Asserted for all seventeen, budgeted tier and bounded tier alike, and it is + * the bounded tier that needs it most: a bound is a single comparison, so a + * probe swapped for one that reaches less is a bound over a smaller workload + * and there is no floor beside it to notice. + * `bytes_small`/`bytes_large` are deliberately NOT here: they are bounded by + * `heap_ceiling_bytes` and `heap_reading_bytes` with ~50% of slack either way + * (`heap_bound_bytes` with 50% on the one side), because a Node major or a + * different platform moves heapUsed accounting and an exact-equality arm on a + * byte count would be a re-baseline per runner. */ +const HEAP_SHAPE = { + fields: ['files_small', 'files_large', 'path_segments', 'probe'], + why: + 'these four decide WHAT the heap arm measures and nothing else here can see them move — a ' + + 'probe that stops reaching a leg, or two file counts collapsed onto one, leaves every ' + + 'ceiling, floor, bound and ratio passing over an arm that changed workload. Deterministic: ' + + 'a re-run will not change it.', +}; +/** The same, for the `context` arm. All three fields are exact strings, not + * bounds: this arm has no measurement noise at all — it resolves one import + * two ways over a three-file corpus — so anything less than equality would be + * slack for nothing. */ +const CONTEXT_SHAPE = { + fields: ['target', 'with_context', 'without_context'], + why: + 'the fifth `context` argument stopped reaching this resolver, reached it in a different ' + + 'shape, or the resolver changed what it does with it. `with_context` is what the five-argument ' + + 'call `run.ts` makes answers and `without_context` is what the three-argument one this bench ' + + 'used to make answers; both are pinned, so a change is attributed rather than guessed. ' + + 'Deterministic: a re-run will not change it.', +}; +/** Every asserted arm for one language, derived so a new scale is covered by + * construction. The heap arm is present for EVERY language now — it used to be + * conditional on `HEAP_LANGS`, which is what let the other nine be measured by + * nothing and pinned by nothing; the two tiers below decide which gate the + * reading gets. The context arm is still conditional, on `CONTEXT_LANGS`, which + * the registry-arity arm at the foot of the file pins to the hooks that DECLARE + * a fifth parameter. */ +const armShapes = (lang) => [ + ...SCALES.map((scale) => [scale, SCALE_SHAPE]), + ['heap', HEAP_SHAPE], + ...(CONTEXT_LANGS.includes(lang) ? [['context', CONTEXT_SHAPE]] : []), +]; + +for (const lang of LANGS) { + const got = report[lang]; + const want = baseline.languages[lang]; + if (got.fingerprint !== want.fingerprint) { + failures.push( + `${lang}: fingerprint drift ${got.fingerprint} != ${want.fingerprint} — the resolver ` + + `returned a DIFFERENT target set. That is a behaviour change, not a perf one; see the ` + + `parity harnesses in test/unit/scope-resolution/*-import-target-parity.test.ts and the ` + + `all-languages adapter guard in import-target-index-reuse.contract.test.ts.`, + ); + } + // One shape, five facts, so the five budgets read side by side and the shared + // trailing sentence exists once instead of drifting into five wordings. Each + // `why` stays the arm's OWN: it is what tells a triager which corpus shape + // regressed, and flattening it would cost the message its whole value. + // `key` is the baselines.json path the budget came from, so the presence + // check below can name it. + const timingChecks = [ + { + label: 'scaling', + key: 'scaling_budget', + got: got.scaling_ratio, + budget: baseline.scaling_budget, + why: 'per-import cost grows with corpus size again.', + }, + { + label: 'depth', + key: `depth_budget.${lang}`, + got: got.depth_ratio, + budget: baseline.depth_budget?.[lang], + why: + 'cost grows with path DEPTH at a fixed file count, which scaling_ratio divides out and ' + + 'cannot see.', + }, + { + label: 'collide scaling', + key: `collide_scaling_budget.${lang}`, + got: got.collide_scaling_ratio, + budget: baseline.collide_scaling_budget?.[lang], + why: + 'on the SHARED-LEAF layout (svcN/internal, SrcN/Models, a repeated basename per package) ' + + 'per-import cost grew beyond what this shape already costs by construction.', + }, + { + label: 'small arm ms', + key: `small_ms_ceiling.${lang}`, + got: got.small.ms, + budget: baseline.small_ms_ceiling?.[lang], + why: + 'an ABSOLUTE bound, because a constant-factor regression that grows both arms equally ' + + 'passes the ratio.', + }, + { + label: 'collide arm ms', + key: `collide_ms_ceiling.${lang}`, + got: got.collide.ms, + budget: baseline.collide_ms_ceiling?.[lang], + why: 'the ABSOLUTE bound on the shared-leaf layout.', + }, + ]; + for (const check of timingChecks) { + // PRESENCE FIRST — see `requireNumericBudget` for why. All five maps are + // complete today, which is exactly when the check is worth having: every one + // of the four per-language lookups above is one deleted key away from a + // silent no-op. The heap arm HAD THE SAME HOLE and the comment here used to + // deny it: iterating the BASELINE's keys protects that loop against a + // deleted MEASUREMENT, which is a different thing from a deleted BUDGET. + // See `heapBudgetChecks`. + const missing = requireNumericBudget({ + key: check.key, + value: check.budget, + reads: `${check.got} > undefined`, + scope: `That leaves ${lang}'s ${check.label} arm ungated.`, + }); + if (missing !== null) { + failures.push(`${lang}: ${missing}`); + continue; + } + if (check.got > check.budget) { + failures.push( + `${lang}: ${check.label} ${check.got} > budget ${check.budget} — ${check.why} ` + + `Timing arm: re-run on an idle machine before investigating.`, + ); + } + } + for (const arm of ['deep', 'collide']) { + if (got[arm].resolved !== got.small.resolved) { + failures.push( + `${lang}: ${arm} arm resolved ${got[arm].resolved} vs small ${got.small.resolved} — the ` + + `${arm} arm was supposed to change ${arm === 'deep' ? 'path depth' : 'directory and file NAMING'} ` + + `and nothing else, so that it times the same workload. An arm that stopped resolving ` + + `would be timing the null path and its ratio would mean nothing.`, + ); + } + // Count-neutral by design, so neutering the arm (DEEP_PAD = 0, a collideDir + // that forwards to uniqueDir) moves NO asserted count. Comparing the two + // fingerprints is the only arm that notices. + if (got[arm].fingerprint === got.small.fingerprint) { + failures.push( + `${lang}: ${arm}.fingerprint equals small.fingerprint — the ${arm} arm is resolving the ` + + `IDENTICAL corpus, so it measures nothing. ` + + `${arm === 'deep' ? 'DEEP_PAD is 0 or the padding stopped reaching buildFiles' : 'collideDir is returning the uniqueDir layout'}. ` + + `This is a deterministic arm: a re-run will not change it.`, + ); + } + } + // The `context` arm's own discriminator, and the same shape of gate as the + // deep/collide fingerprint comparison above: `armShapes` pins WHAT the two + // call shapes answer, and this pins that they still answer DIFFERENTLY. + // Without it the arm degrades exactly the way `DEEP_PAD = 0` degrades the + // depth arm — a probe on which both halves agree asserts two copies of one + // number. Deleting the fifth argument from `resolveOne`, deleting + // `importedSymbolKind` from PHP's import, or reverting Python to the + // `namespace` spelling all land here, and NOTHING else in this file would + // notice: on the main corpus the leg agrees with the cascade, so the + // fingerprints do not move, and a dropped context only makes the timing arms + // faster. + if (CONTEXT_LANGS.includes(lang) && got.context.with_context === got.context.without_context) { + failures.push( + `${lang}: context arm answers '${got.context.with_context}' with AND without the pass's ` + + `parsedFiles — the fifth argument is not reaching the resolver, or the leg behind it no ` + + `longer runs (PHP needs parsedImport.kind named|alias AND importedSymbolKind ` + + `function|const; Python needs named|alias, since a namespace import never reads ` + + `parsedFiles). run.ts calls resolveImportTarget with five arguments and this bench must ` + + `too. Deterministic: a re-run will not change it.`, + ); + } + // ONE loop for every arm's corpus shape, timing and heap alike. The heap arm + // was reported here and asserted nowhere, which made the four fields that + // decide WHAT it measures free to move: `HEAP_PROBE_TARGET.csharp_csproj` + // swapped for a target matching no `CSPROJ_CONFIGS` rootNamespace skips the + // whole config loop, so the `getFilesInDir` and `getInsensitive` legs never + // run, and the arm the MEMORY section calls "the witness that the read + // pattern IS the footprint" quietly becomes a two-map arm — measured + // 73 703 384 -> 59 921 216 B, ratio 1.017 -> 1.011, ceiling and floor both + // still passing. `HEAP_SMALL` set equal to `HEAP_LARGE` is the same shape of + // hole: it makes `ratio` identically ~1.0 and leaves `bytes_large` untouched. + for (const [arm, shape] of armShapes(lang)) { + for (const field of shape.fields) { + if (got[arm][field] !== want[arm]?.[field]) { + failures.push( + `${lang}.${arm}.${field}: ${got[arm][field]} != ${want[arm]?.[field]} — ${shape.why}`, + ); + } + } + } +} + +// PRESENCE FIRST for the two SCALAR heap budgets, for exactly the reason the +// five timing budgets get it — and the reason the comment up there used to give +// for the heap arm not needing it was wrong. Iterating the baseline's keys +// protects the loop below against a deleted MEASUREMENT (`heap == null`, right +// there); it does nothing about a deleted BUDGET. These two keys are scalars +// rather than per-language maps, so deleting either is one keystroke that +// silently disables that arm for ALL EIGHT languages at once. That makes them +// the widest-blast-radius keys in this file, not the safest — which is what +// their `scope` sentence says and the per-language ones do not. +const heapBudgetChecks = [ + { key: 'heap_floor_fraction', value: baseline.heap_floor_fraction, reads: 'bytes_large < NaN' }, + { key: 'heap_ratio_budget', value: baseline.heap_ratio_budget, reads: 'ratio > undefined' }, +]; +const heapArmScope = `This one key gates all ${HEAP_BUDGETED.length} budgeted heap arms at once.`; +for (const check of heapBudgetChecks) { + const missing = requireNumericBudget({ ...check, scope: heapArmScope }); + if (missing !== null) failures.push(missing); +} + +// And EXACT KEY EQUALITY between each tier's CODE list and the baseline maps +// that gate it, because the loops below iterate the baseline: delete one +// language's ceiling and that language drops out of the loop entirely — still +// measured, still printed, never checked. Both directions, the same shape as the +// LANG_REGISTRY/SCOPE_RESOLVERS inventory arm at the bottom of the file. The +// forward direction (a language with no budget) is the per-language presence +// check inside each loop; this is the reverse (a budget with no arm). +// +// Three maps rather than two: `heap_bound_bytes` is reconciled against +// `HEAP_BOUNDED` exactly as the other two are against `HEAP_BUDGETED`, so a +// language promoted from bounded to budgeted has to move its key in the same +// edit — leave the bound behind and it is an orphan here, take the bound away +// without adding a ceiling and the presence check fires there. +const heapBudgetMaps = [ + ['heap_ceiling_bytes', baseline.heap_ceiling_bytes, HEAP_BUDGETED, 'HEAP_BUDGETED'], + ['heap_reading_bytes', baseline.heap_reading_bytes, HEAP_BUDGETED, 'HEAP_BUDGETED'], + ['heap_bound_bytes', baseline.heap_bound_bytes, HEAP_BOUNDED, 'HEAP_BOUNDED'], +]; +for (const [key, map, codeList, codeListName] of heapBudgetMaps) { + expectNoOrphanKeys( + `baselines.json ${key}`, + Object.keys(map ?? {}), + codeList, + codeListName, + 'the bench budgets a heap arm it does not measure.', + ); +} + +// The two heap tiers are a PARTITION of LANGS by construction (`HEAP_BOUNDED` +// is a filter over it), so the only way a name can be in neither is for +// `HEAP_BUDGETED` to hold one `LANGS` does not — a typo, or a language dropped +// from the registry with its budget left behind. That name would then be +// measured by nothing, and the loop below would report it as a missing arm +// without ever saying why; this says why. +expectNoOrphanKeys( + 'HEAP_BUDGETED', + HEAP_BUDGETED, + LANGS, + 'LANGS', + 'that name is in neither heap tier, because HEAP_BOUNDED is derived as the languages LANGS ' + + 'has and this list does not — so its budget gates nothing and its language, if it has one, ' + + 'is bounded by nothing.', +); +// The same, for the probe map. The forward direction — a language with no probe +// — is caught by the `heap.probe` shape assertion (`undefined` never equals a +// recorded string), so what is left is a probe kept for an arm that no longer +// runs, which reads as coverage and is not. +expectNoOrphanKeys( + 'HEAP_PROBE_TARGET', + Object.keys(HEAP_PROBE_TARGET), + LANGS, + 'LANGS', + 'the bench carries a heap probe for a language it does not benchmark.', +); + +// The same reverse direction for the context arm. The forward direction (a +// language in CONTEXT_LANGS with no baseline block) is `armShapes`, which +// compares against `want.context?.[field]` and fails on undefined; this is the +// other way round — a baseline block for a language the bench hands no context +// is a gate over an arm that is never measured, and `armShapes` would never +// look at it. +expectNoOrphanKeys( + 'baselines.json languages.*.context', + Object.keys(baseline.languages).filter((lang) => baseline.languages[lang].context !== undefined), + CONTEXT_LANGS, + 'CONTEXT_LANGS', + 'the bench pins an arm it does not run.', +); + +// TIER ONE, the budgeted arms: ceiling, floor and ratio, all three unchanged. +// +// Driven by HEAP_BUDGETED, the CODE's list, exactly as the timing arms iterate +// LANGS — so a deleted budget key is a presence failure rather than a language +// that quietly stops being iterated. A deleted MEASUREMENT still fails too: +// `measureHeap` now runs for every language, so a `heap == null` here is the arm +// having been removed or skipped. +for (const lang of HEAP_BUDGETED) { + const ceiling = baseline.heap_ceiling_bytes?.[lang]; + const reading = baseline.heap_reading_bytes?.[lang]; + // `reads` names the comparison each key gates further down: the ceiling is + // compared directly, the reading only after `reading * heap_floor_fraction` + // has turned a missing one into `NaN`. + for (const [key, value, reads] of [ + ['heap_ceiling_bytes', ceiling, 'bytes_large > undefined'], + ['heap_reading_bytes', reading, 'bytes_large < NaN'], + ]) { + const missing = requireNumericBudget({ + key: `${key}.${lang}`, + value, + reads, + scope: + `This loop iterates HEAP_BUDGETED precisely so that deleting the key fails here instead ` + + `of dropping ${lang} out of the gate.`, + }); + if (missing !== null) failures.push(`${lang}: ${missing}`); + } + const heap = report[lang]?.heap; + if (heap == null) { + failures.push( + `${lang}: heap arm missing though HEAP_BUDGETED names it — the retained-index measurement ` + + `was removed or skipped. It is the only arm that can see memory.`, + ); + continue; + } + if (heap.bytes_large > ceiling) { + failures.push( + `${lang}: retained per-pass import index ${heap.mib_large} MiB at ${heap.files_large} ` + + `files (${heap.bytes_large} B) > ceiling ${ceiling} B — these indexes are built at ` + + `O(files × depth) and this is the ABSOLUTE bound on that (#2649). Deterministic: a ` + + `re-run will not change it.`, + ); + } + // A FLOOR as well as a ceiling, and it is the arm that would have caught the + // one defect this whole block exists for. When `buildSuffixIndex` went lazy, + // these four arms stopped asking a suffix question, built no map and reported + // 0 B at 32 000 files — and 0 B is under every ceiling, so `--check` printed + // PASS over four gates that had become ceilings over nothing. A ceiling can + // only ever say "not too big"; nothing said "still measuring something". + // + // Taken as a fraction of the RECORDED READING, not of the ceiling. It used to + // be 0.33 x the ceiling, with the comment claiming that put it "at half the + // measured size" — true only for as long as every ceiling stayed at exactly + // 1.5x its reading, which is a convention this file states and nothing + // enforces. Re-tuning one ceiling upward would have loosened that language's + // floor by the same factor, in the one direction the floor exists to watch. + // 0.5 x the reading is the same effective floor today (within 0.8% for all + // eight) and says what it means. `heap_reading_bytes` is the measurement the + // ceiling is derived from too, so the pair still moves together on a + // re-baseline — far below any plausible drift (the readings reproduce to the + // byte across processes) and far above the collapse it watches for. A genuine + // 2x memory WIN trips it too, and that is intended: it must be explained and + // re-baselined, exactly like a fingerprint move. + const floor = reading * baseline.heap_floor_fraction; + if (heap.bytes_large < floor) { + failures.push( + `${lang}: retained per-pass import index ${heap.bytes_large} B at ${heap.files_large} ` + + `files < floor ${Math.round(floor)} B (${baseline.heap_floor_fraction} x recorded ` + + `reading ${reading}) — this arm has almost certainly stopped MEASURING rather than started ` + + `saving. Probe '${heap.probe}' resolves through the real resolver; if a leg it used to ` + + `reach now returns earlier, or an index it forced is now built lazily behind a question ` + + `nobody asks, the arm reads ~0 and every ceiling above passes. Deterministic: a re-run ` + + `will not change it.`, + ); + } + if (heap.ratio > baseline.heap_ratio_budget) { + failures.push( + `${lang}: retained-heap ratio ${heap.ratio} > budget ${baseline.heap_ratio_budget} ` + + `(${heap.bytes_small} B at ${heap.files_small} files -> ${heap.bytes_large} B at ` + + `${heap.files_large}) — the index stopped growing linearly in the file count.`, + ); + } +} + +/** + * TIER TWO, the bounded arms: ONE comparison, and what it is a comparison FOR. + * + * `heap_bound_bytes` is the "exclusion still holds" bound. It does not claim + * these indexes are small enough, which is what a ceiling claims about a + * budgeted one; it claims each is still the SIZE the decision to leave it out + * was taken on. `HEAP_BOUNDED` derives to THREE today — cobol, swift, rust. + * The prose below still counts nine because six were promoted to tier one + * after it was written; read the counts as history, and `HEAP_BOUNDED` itself + * as the answer. The re-entry condition the MEMORY section states — "if any of + * the four ever diverges in what it ASKS, it earns an arm the same way" — is a + * claim about growth, and this is the only thing in the file that can see it. + * + * NO FLOOR, and the reason is per language rather than uniform. rust reads 16 B + * because it builds nothing, so any floor at all would be a floor on noise and + * `1.5 x 0 B` is 0 — its bound is ABSOLUTE (1 MiB) for the same reason: a + * multiplier on 16 B fails on the first byte of anything. The other eight are + * stable enough today to floor (0.24% peak-to-peak at worst over five runs). + * The two this paragraph named as floor candidates, kotlin and dart, TOOK that + * promotion: both now carry a ceiling and a recorded reading in tier one, which + * is what the paragraph said the promotion had to be. What this tier is NOT is a + * weaker version of tier one — it is a different question, asked of the + * languages tier one does not ask it of. + */ +const heapBoundScope = + `That leaves the arm bounded by nothing, which is the state all nine of these were in before ` + + `they were measured.`; +for (const lang of HEAP_BOUNDED) { + const bound = baseline.heap_bound_bytes?.[lang]; + const missing = requireNumericBudget({ + key: `heap_bound_bytes.${lang}`, + value: bound, + reads: 'bytes_large > undefined', + scope: heapBoundScope, + }); + if (missing !== null) failures.push(`${lang}: ${missing}`); + const heap = report[lang]?.heap; + if (heap == null) { + failures.push( + `${lang}: heap arm missing though HEAP_BOUNDED names it — every registered language is ` + + `measured now, and the tier only decides which gate the reading gets.`, + ); + continue; + } + if (missing === null && heap.bytes_large > bound) { + failures.push( + `${lang}: retained per-pass import index ${heap.mib_large} MiB at ${heap.files_large} ` + + `files (${heap.bytes_large} B) > bound ${bound} B — this language is EXCLUDED from the ` + + `budgeted heap tier, and the bound is what says the exclusion still holds. It has grown ` + + `a structure, or started asking its index a question it did not ask when the exclusion ` + + `was recorded. Read _heap_bound_note in baselines.json for this language's reason and ` + + `its recorded reading, then either explain the growth or promote it to HEAP_BUDGETED ` + + `with a ceiling, a reading and a floor. Deterministic: a re-run will not change it.`, + ); + } +} + +// INVENTORY, the arm that makes "every registered language is gated" a checked +// claim instead of a comment. `LANG_REGISTRY` is a hand-written table — it has +// to be, since each row also implies five dispatcher branches — but which +// languages it must contain is not a judgement call, and this is where the two +// are reconciled. Both directions: a resolver registered with no arm here is +// the PR #2911 hole (a language shipping unmeasured), and an arm naming a +// language the registry does not have is a bench measuring something the +// pipeline no longer runs. +// +// Loaded HERE, after the last measurement, rather than imported at the top. +// Reaching `pipeline/registry.ts` drags in every registered scope resolver and +// its providers, and this arm is the only thing in the file that wants it. The +// side benefit is that both modes now measure in the same module state: report +// mode never loads the registry, and `--check` loads it only once every number +// has been taken. +// +// It is NOT cheap and the header says so plainly rather than rounding it down: +// 6.3-6.5 s on one box and 9.3-10.0 s on another, measured in isolation with +// this file's own static imports already resident, which is most of the +// `repsFor` win and the whole reason `--check` did not get faster. Kept anyway, +// because the `benchmarks` job runs ~4.5 minutes clear of CI's critical path, +// so the seconds buy nothing, and because the alternative reconciles arm NAMES +// where this reconciles the `SupportedLanguages` values the dispatchers key +// off. See COST in the header. +const { SCOPE_RESOLVERS } = + await import('../../src/core/ingestion/scope-resolution/pipeline/registry.ts'); +const registeredLanguages = [...SCOPE_RESOLVERS.keys()].sort(); +const benchedLanguages = [...new Set(Object.values(LANG_REGISTRY))].sort(); +for (const language of registeredLanguages) { + if (benchedLanguages.includes(language)) continue; + failures.push( + `${language} is registered in SCOPE_RESOLVERS but has no arm in LANG_REGISTRY — its ` + + `import-target resolver is ungated: nothing pins its output and nothing pins its scaling. ` + + `That is the state JavaScript was in at 25 972 µs per import (PR #2911). Add a row, then ` + + `the five dispatcher branches it needs (uniqueDir, collideDir, uniqueTarget, collideTarget, ` + + `resolveOne) and a baselines.json entry. Deterministic: a re-run will not change it.`, + ); +} +// The reverse half is the same loop as the two above it, so it goes through the +// same helper. Only the FORWARD half stays written out: its message is a +// five-step remediation for adding a language, which no shared framing carries. +expectNoOrphanKeys( + 'LANG_REGISTRY', + benchedLanguages, + registeredLanguages, + 'SCOPE_RESOLVERS', + 'this bench is gating a resolver the pipeline no longer registers.', +); + +// The SAME reconciliation for `CONTEXT_LANGS`, against the registry rather than +// against a claim in a comment. `run.ts` passes the fifth argument to every +// provider; which ones can OBSERVE it is decided by how many parameters each +// hook declares, and that is a number the registry can be asked for. Today +// exactly three answer 5 (php, java, python) and the other fourteen answer 3 or 4 — +// which is why fourteen arms could ignore this whole question and their numbers +// did not move when it was fixed. +// +// `Function.length` stops at the first defaulted or rest parameter, so a hook +// written as `(a, b, c, d, context = {})` would read 4 and slip past this arm. +// The shared contract declares the parameter as `context?:`, which compiles to +// a plain parameter, so every resolver written against it counts — and one that +// is not is one this arm asks you to look at. +const CONTEXT_PARAM_COUNT = 5; +const contextLanguages = new Set(CONTEXT_LANGS.map((lang) => LANG_REGISTRY[lang])); +for (const [language, resolver] of SCOPE_RESOLVERS) { + const declares = resolver.resolveImportTarget.length >= CONTEXT_PARAM_COUNT; + const benched = contextLanguages.has(language); + if (declares === benched) continue; + failures.push( + declares + ? `${language}'s resolveImportTarget declares ${resolver.resolveImportTarget.length} ` + + `parameters, so it can read the { parsedFiles, parsedImport } context run.ts passes, ` + + `but no arm here supplies one — that leg is measured by nothing. Add the language to ` + + `CONTEXT_LANGS, thread the context in resolveOne, and give it a CONTEXT_PROBE whose ` + + `two answers differ. Deterministic: a re-run will not change it.` + : `CONTEXT_LANGS names '${language}', whose resolveImportTarget declares only ` + + `${resolver.resolveImportTarget.length} parameters — it cannot observe a context, so ` + + `this bench is building a ParsedFile[] per pass that nothing reads and asserting a ` + + `context arm that cannot fail. Deterministic: a re-run will not change it.`, + ); +} + +console.log(JSON.stringify(report, null, 2)); +if (failures.length > 0) { + console.error(`[import-target --check] FAIL\n - ${failures.join('\n - ')}`); + process.exit(1); +} +console.log('[import-target --check] PASS'); diff --git a/gitnexus/bench/kotlin-import-target/baselines.json b/gitnexus/bench/kotlin-import-target/baselines.json new file mode 100644 index 000000000..e89eaaec7 --- /dev/null +++ b/gitnexus/bench/kotlin-import-target/baselines.json @@ -0,0 +1,15 @@ +{ + "_comment": "Baselines for bench/kotlin-import-target/measure.mjs --check. `fingerprint` is a sha256 over every `fileSet | fromFile | targetRaw -> result` record the correctness corpus resolves, in BOTH file-set iteration orders; it is a CORRECTNESS gate, so drift means Kotlin import resolution started returning a different file set and IMPORTS/CALLS edges moved in every Kotlin repository. Explain it, never re-baseline to make CI green. `cases` and `non_null` are asserted beside it because a shrunken or hollowed corpus produces a perfectly valid fingerprint over a smaller surface — all three are one re-baseline, never separate ones. `scaling_budget`, `depth_budget` and `small_ms_ceiling` are timing gates and carry deliberate headroom for shared CI runners.", + "_provenance": "RE-BASELINED ONCE, DELIBERATELY, IN #2881. The previous value ebf1790bf1d42dad483a51f2cbdeb2351e493b9e8236e4eedeef592dd81e2c5c (13256 non-null) was the PRE-INDEX implementation's, and the index that replaced it in #2872 reproduced it byte for byte — that is what made #2872 a performance change. #2881 changes resolution on purpose: `getKotlinFileIndex` no longer requires the parent directory to be the FIRST occurrence of that name in the path, so a file whose package directory name repeats higher in its own path is now a child of that package (`data/src/main/kotlin/com/example/data/Repo.kt` IS a child of `data`, and `import data.helper` resolves instead of returning null). The drift was not read off the new code and accepted; the corpus was dumped from both implementations and diffed record by record. That census was RE-RUN with a shape classifier after review found its taxonomy — 54 NULL -> resolved, 149 reselections, 32 wider fan-outs — was entirely SHAPE-PRESERVING and so had no bucket for a class this change introduces. Both ends of the re-run are validated against numbers this file already publishes, so the census is provably over the surface they describe: driven over this bench's own corpus, the BASE resolver reproduces ebf1790bf1… at 13256 non_null and the HEAD resolver reproduces d91110bee3… at 13310, across 20106 records of which 19968 are distinct. 235 distinct records moved, classified by SHAPE rather than by null-ness: null -> string 38 and null -> array 16, which together are the first census's '54 NULL -> resolved' and exactly the +54 in non_null; string -> string (a different member of a now-wider bucket) 149; array -> array (the fan-out grew) 32; string -> array 0; and ZERO of every other transition — nothing went string -> null, array -> null or array -> string, and no array shrank or reordered. So the old three buckets reappear inside the shape taxonomy exactly, and its two structural claims hold when checked directly instead of inferred: all 32 growths are order-preserving SUPERSETS of the base answer, no record lost a member, and in all 149 reselections the new answer's parent directory is named by a segment of the import and carries the same directory NAME the base answer's parent did. WHAT THE OLD TAXONOMY HAD NO BUCKET FOR is `string -> array`, and it is the one class here that is not shape-preserving: it is a RESOLVED -> UNRESOLVED transition. Tier 3 (`findKotlinPackageFiles`) runs before tier 4 (`findByProgressivePrefixStrip`), so a bucket the removed guards left empty returned null and let tier 4 answer with a single BOUND file; a now-populated bucket stops tier 4 running at all and hands back a fan-out array that need not contain the imported name at all. Two files reproduce it, in both iteration orders: ['data/src/main/kotlin/com/example/data/Repo.kt', 'common/helper.kt'] with `import data.helper` answers 'common/helper.kt' at base and ['data/src/main/kotlin/com/example/data/Repo.kt'] at head. ITS COUNT OVER THIS CORPUS IS 0, AND THAT IS A FACT ABOUT THE CORPUS RATHER THAN ABOUT THE CLASS. This file's own fuzz generator, run at ten times the repositories (4000, ~198600 distinct records), hits the class 12, 4, 10 and 10 times over four seeds — ~5e-5 per record, an expectation of about ONE over the 19968 records here — so 0 is this corpus being an order of magnitude too small to reach it, not the shape being unreachable. The consequence is worth stating plainly: the fingerprint below is blind to a resolved -> unresolved class this change introduces, by corpus SIZE and not by construction, and no arm in this bench gates it today. Adding a hand-written case for it is a deliberate fingerprint move and a fourth re-baseline of this file; it is worth doing and it is not this change. The corpus itself is untouched, which is why `cases` is unchanged at 20106 — the fingerprint is over the same surface as the value it replaces.", + "_gate_controls": "The gate is only worth its baseline if a plausible regression moves it, so each arm was checked against the mutation it exists to catch, with the resolver otherwise untouched. All values below are against the CURRENT baseline (#2881, guards removed + per-directory key memo + bucket compaction). Caught: skipping the dirChildren component walk above depth 8 (fingerprint 41bb550b76d4…, non_null 13310 -> 12800); capping a dirChildren bucket at 17 entries (a7681945b752…, non_null UNCHANGED — the fingerprint is the only arm that sees it, and note the compaction pass now rewrites those same buckets, so this control was re-run after it); capping suffixByStem key depth at 7 (d24b8a2bd822…, non_null unchanged); and a HALF fix that drops only the `startsWith` guard while keeping the `indexOf` first-occurrence check (836977b83bf0…, non_null 13310 -> 13282), which leaves every mid-path repeat such as `top/data/mid/data/Repo.kt` broken and is the mutation #2881 itself makes plausible. Added with the memo: keying `dirKeys` on the directory's LAST SEGMENT instead of its full path (36a4e9dad313…, non_null 13310 -> 13305) — the memo's whole safety argument is that its key determines the key SET a directory contributes, so a coarser key silently hands one directory another's bucket list, and that is the one way this optimization can move an answer. Also caught, with the RESOLVER untouched and only the corpus edited: dropping the competing file from the exact-beats-earlier-suffix case and emptying the repeated-directory case (44df5093ee…). All of these passed silently before this corpus carried deep paths, packages above 16 files, queries against suffix keys deeper than 7, and the file set inside the hashed record. Re-check them after any corpus edit — a corpus that stops spanning an axis takes the gate with it. NOTE what no fingerprint control here can catch: the memo and the compaction are both invisible to this bench by design (identical output), so no arm in this file gates either one, and the honest version of where they ARE gated is narrower than a claim about comparing the three maps would suggest. The memo's gate is test/unit/scope-resolution/kotlin/kotlin-index-internals.test.ts, which drives the resolver's OBSERVABLE SURFACE rather than the built index — the index is module-private — and reconstructs what it needs from the tiers. It pins: bucket CONTENTS and ORDER, read back from the fan-out tier, which hands out the bucket array itself; that the first-child tier reads position 0 of that SAME array; bucket IDENTITY across two calls on one Set, which is what proves the memo's hit path ran at all, since only a second file in the same directory reaches it; the frozen state of the array actually handed out, on the multi-child path, on the `length === 1` skip path, and once per key of a multi-key directory; that the memo keys on the NORMALIZED directory while storing the raw path; and the one mutation that can move an answer — keying `dirKeys` on the directory's last segment instead of the whole `dir` — which fails three of its arms. KEY INSERTION ORDER is unasserted there BY DESIGN and not by omission: `dirChildren` is only ever read by `.get(key)`, so key order has no consumer, and that file says so. The COMPACTION is unasserted there too and cannot be asserted there at all — a JS array's backing-store capacity has no reflective surface, so deleting `bucket.slice()` and freezing the grown bucket in place leaves every arm in that file green, `Object.isFrozen` included. Its only instrument is the retained-heap arm in bench/import-target, whose kotlin ceiling was tightened to 1.0747x its recorded reading precisely so that the +12.57% the slice reclaims fails `--check`; see `_heap_compaction_gate` in bench/import-target/baselines.json for the measurement and for how to tell that failure apart from a runner's heapUsed accounting moving under the whole file.", + "fingerprint": "d91110bee389891c313811c5b4bae61d909561156e1458d38d487be969f0059c", + "cases": 20106, + "non_null": 13310, + "scaling_budget": 1.6, + "depth_budget": 2.0, + "small_ms_ceiling": 40, + "_scaling_note": "(t_large/t_small)/(1600/400). ~1.0 is linear. OBSERVED BAND: 0.99-1.04 on a 12-core dev box, small arm ~6 ms. Read that band as a floor, not a spec — independent runs on other hardware during review came out 0.954-1.014, 0.965-1.036 and ~0.95-1.08, so a 1.2 reading is noise and should be re-run, not investigated. IMPORTS_PER_FILE is sized so the small arm lands in the ms rather than the ~2 ms a first revision measured, where timer granularity and JIT warm-up, not scaling, set the number; bench/cpp-qualified-ns documents the same artifact. TRIAGE: every timing arm here is a TIMING signal — RE-RUN IT on an idle machine before investigating; runner contention dominates. The fingerprint arm is the opposite: deterministic, a re-run never changes it, and it must never be wished away. FLOOR CHECK: the pre-index implementation — i.e. exactly the regression this gate exists to catch — measures ratio 3.737 on this corpus (2207.8 ms small, 33003.5 ms large, one cold run) against ~1.0 for the index. Independent review runs measured its floor at 3.905-4.297. Treat the absolute times as an order of magnitude only: the floor arm is one cold run because best-of-seven against a quadratic implementation costs minutes, while the index arm is best-of-seven after two warmups.", + "_depth_note": "deep_ms/shallow_ms at a FIXED file count, paths 24 components against 8. scaling_ratio divides the file count out, so it is scale-invariant and structurally cannot see a cost that grows with path depth instead — and the two loops the index is built from are depth loops (one suffixByStem entry per '/' in a stem, one dirChildren pass per component of dir). OBSERVED BAND, five runs each on one box: 1.44-1.51 before #2881; 1.27-1.40 after its guard removal, which deleted two string comparisons per component of every dir; 1.20-1.26 after the same issue's per-directory key memo, which turns that whole component walk from once-per-FILE into once-per-DIRECTORY. Both movements are per-depth work, which is why this arm sees them and the file-count arm does not. The BUDGET moved with the band both times — 2.4 -> 2.2 -> 2.0 — holding the ~1.6x headroom over the band's top that 2.4 expressed against the original; left at 2.4 it would quietly have become 1.9x, which is how a gate goes slack without anyone deciding to loosen it. Note what this budget is NOT for: a revert of #2881 scores ~1.5 and passes at any of those numbers, and that is correct — reverting it restores a resolution bug, which is the FINGERPRINT's job to catch, not a timing arm's. It sits above 1.0 legitimately: 3x the depth is 3x the suffix keys per file, so the build genuinely does more work; what the budget forbids is that growing faster than the depth ratio itself.", + "_ceiling_note": "small_ms_ceiling is an ABSOLUTE bound, because scaling_ratio is a ratio and a constant-factor regression that grows both arms equally passes it. Measured during review: a full workspace scan reintroduced on 1-in-16 imports is caught by the ratio (1.814), but at 1-in-32 it passes at 1.490 while running 2.8x slower in absolute terms. 40 ms against an observed 5.9-6.1 ms leaves ~6x of headroom for a loaded shared runner while still catching that shape.", + "_blind_spot": "WHAT THIS BENCH CANNOT SEE, measured rather than guessed. Its scaling corpus gives every module a UNIQUE package leaf (`com/example/mod{N}`), so a `dirChildren` query matches exactly one directory. That makes it blind to any cost that grows with the number of DIRECTORIES sharing a queried segment — the shape a real Kotlin monorepo has, where 200 modules each hold `data`, `ui` and `domain`. Established by building the reuse this file's memo argues against: swapping `dirChildren` for the shared `import-resolvers/package-dir-index.ts` (with its first-occurrence rule off) is OUTPUT-IDENTICAL — same fingerprint, same cases, same non_null, 0 divergences over 107948 answers — and on THIS corpus it costs only 1.37x-1.50x and passes every arm here. On a repeated-leaf corpus the same swap measures 13.5x per first-child query, 409x per fan-out, and 8114x on `import data.*` at 200 matching directories (it merges and SORTS every candidate, per import), for 3.1x-5.7x end to end and a bench-style scaling_ratio of 3.465 against this file's 1.6 budget — i.e. back to the pre-index quadratic floor of 3.737. A change that regresses this resolver to the very shape the bench exists to catch would go GREEN here. The trade it buys is real and also measured: 26.2% less retained memory, 12.18 MiB at 32000 files. If that memory is ever wanted, the shape to build is per-suffix keys -> DIRECTORY lists plus files-per-directory (8.29 MiB against 15.87 measured, single-directory query still one hash lookup) — and the repeated-leaf arm to measure it against already exists one directory over: bench/import-target's kotlin `collide` layout puts `com/example/models` under 200 modules at the 1600-file scale, with `collide_scaling_budget` 1.8 against a measured 1.081. The swap scores 3.465 there. So the gate for this decision is that arm, not a new one here; what this file lacks is only a repeated-leaf arm of its own, which would be duplicated coverage." +} diff --git a/gitnexus/bench/kotlin-import-target/measure.mjs b/gitnexus/bench/kotlin-import-target/measure.mjs new file mode 100644 index 000000000..5d1868910 --- /dev/null +++ b/gitnexus/bench/kotlin-import-target/measure.mjs @@ -0,0 +1,539 @@ +/** + * Build-free identity + scaling bench for `resolveKotlinImportTarget`, the + * Kotlin import resolver. + * + * Before this bench's companion change the resolver walked the ENTIRE + * `allFilePaths` Set on every import. Its four tiers — exact/suffix, + * directory child, package fan-out, progressive prefix strip — each ran + * `for (const raw of allFilePaths)` with a `replace(/\\/g, '/')` and several + * string comparisons per entry, and they are tried in cascade, so one + * unresolved import cost two to four full passes. Resolution was therefore + * O(imports x files). Once a repository reaches tens of thousands of Kotlin + * files that is on the order of 10^10 string operations on one thread: + * `analyze` sits at exactly 1.00 core with a flat heap and emits nothing for + * hours, because every allocation is a short-lived string and nothing + * accumulates to hint at progress. + * + * This is the same shape #1918 fixed for Python and #2788 for C++, and it + * returns the same way: someone adds a tier, reaches for `allFilePaths`, and + * writes a loop. Neither existing gate can catch it here — + * `bench/python-scope/import-target-fingerprint.mjs` drives the Python + * resolver only, and `bench/scope-capture/measure.mjs` fingerprints + * `emitScopeCaptures`, a different function that never calls import + * resolution. Hence this bench, in an always-on CI step. + * + * TWO ARMS, and they fail for opposite reasons: + * + * - `fingerprint` — a sha256 over every `fromFile | targetRaw -> result` + * triple the correctness corpus resolves (an exhaustive branch matrix plus + * a deterministic fuzz). This is a CORRECTNESS gate. Drift means Kotlin + * imports started resolving a DIFFERENT file set, i.e. CALLS/IMPORTS edges + * moved in every Kotlin repository. It is deterministic: a re-run never + * changes it, and it must never be re-baselined to make CI green. This + * value is the one the pre-index implementation produced — see + * `_provenance` in baselines.json. + * + * - `scaling_ratio` — `(t_large/t_small)/(LARGE/SMALL)` over a synthetic + * Kotlin monorepo at two scales, timing the index build TOGETHER with + * resolving every import. ~1.0 is linear; a reintroduced per-import scan + * measures ~4 at this scale gap. This is a TIMING gate: re-run it on an + * idle machine before investigating. + * + * A ratio cannot see a constant factor and a file-count ratio cannot see a + * depth cost, so `--check` also asserts a DEPTH ratio (file count fixed, paths + * ~3x deeper) and an absolute ceiling on the small arm. A full workspace scan + * reintroduced on 1-in-32 imports scores 1.490 — inside the scaling budget — + * while running 2.8x slower; the ceiling is what catches that shape. + * + * One honest limit: at a very small import count the index loses. Building it + * is one workspace pass, so a single import into a 100k-file workspace costs + * ~0.8 s against ~0 for a scan that returns on its first hit. It inverts at + * roughly 15 imports, and in the polyglot case that motivates the worry — + * 100k files, 5% Kotlin, a couple of imports — the index already wins, because + * the build skips non-`.kt` entries as cheaply as the scan did. + * + * Five properties of the corpora are load-bearing and must not be + * "simplified" away: + * + * 1. **The correctness corpus fuzzes each file set in BOTH iteration + * orders.** Every tie-break in this resolver is expressed only through + * Set-iteration order — "first suffix match wins", and the two stem maps + * keeping the FIRST path inserted per key. A single-order corpus scores an + * implementation that keeps the LAST match identically. + * 2. **The correctness corpus contains repeated directory names at BOTH the + * leading and the mid-path position** (`data/src/main/kotlin/com/example/ + * data/Repo.kt` and `top/data/mid/data/Repo.kt`). Until #2881 the resolver + * required a file's package directory to be the FIRST occurrence of that + * name in its own path, so neither file was a child of `data`; both are + * now, and that is what the fingerprint pins. Two positions, not one, + * because the old rule was two guards and a half fix that drops only the + * leading-position one still leaves the mid-path shape broken — see + * `_gate_controls` in baselines.json. Without these shapes the fingerprint + * cannot tell the current rule from either predecessor. + * 3. **~40% of the scaling corpus's imports are unresolvable.** The old cost + * was worst when nothing matched, because only then did all four tiers + * run. A corpus where every import hits tier 1 exits after one pass and + * scores a per-import scan far closer to linear. + * 4. **The hashed record includes the FILE SET, not just the query and the + * result.** Otherwise a corpus edit that swaps the workspace under a case + * while leaving its result string alone is invisible: dropping the + * competing file from the "exact beats an earlier suffix" case, or + * emptying the repeated-directory negative case, each leaves `cases`, + * `non_null` and the fingerprint byte-identical and the gate green. + * 5. **Path depth and package size are spanned, not pinned.** Both loops this + * change added are driven by depth — one `suffixByStem` entry per '/' in a + * stem, one `dirChildren` pass per component of `dir` — and the fan-out + * tier returns a bucket whose length is the package size. While the corpus + * capped depth at 8 components and packages at 16 files, three plausible + * follow-up guards (cap suffix depth at 7, skip the `dirChildren` suffix + * loop above depth 8, cap a bucket at 17) all passed `--check` with a + * byte-identical fingerprint — while on a standard Gradle layout the depth + * skip resolved EVERY package import to null and the bucket cap truncated + * fan-out by 58%. Import ARITY, by contrast, was never blind: a tier-4 cap + * at 4 dotted segments already failed the gate, because the branch matrix + * carries 6- and 8-segment cases. + * + * Run: + * node --import tsx bench/kotlin-import-target/measure.mjs # report + * node --import tsx bench/kotlin-import-target/measure.mjs --check # CI gate + */ +import crypto from 'node:crypto'; +import fs from 'node:fs'; +import path from 'node:path'; +import { fileURLToPath } from 'node:url'; +import { resolveKotlinImportTarget } from '../../src/core/ingestion/languages/kotlin/import-target.ts'; + +const __dirname = path.dirname(fileURLToPath(import.meta.url)); +const BASELINE_PATH = path.resolve(__dirname, 'baselines.json'); + +const SMALL = 400; +const LARGE = 1600; +/** Imports per file. Keeps the import count proportional to the file count, so + * a per-import workspace scan shows up as a quadratic ratio rather than being + * amortized away by a fixed import budget. Sized so the SMALL arm measures in + * the tens of ms: at ~2 ms timer granularity and JIT warm-up, not scaling, set + * the ratio — the same artifact bench/cpp-qualified-ns documents. */ +const IMPORTS_PER_FILE = 32; +/** Depth arm: same file count either side, ~3x the path depth on one side. */ +const DEPTH_FILES = 800; +const DEPTH_PAD = 16; +const WARMUP = 2; +const REPS = 7; + +// --------------------------------------------------------------------------- +// Correctness arm +// --------------------------------------------------------------------------- + +const lines = []; +let nonNull = 0; + +function resolve(files, targetRaw, fromFile) { + return resolveKotlinImportTarget( + { kind: 'named', localName: 'X', importedName: 'X', targetRaw }, + { fromFile, allFilePaths: new Set(files) }, + ); +} + +/** Record one case in BOTH file-set iteration orders — see header property 1. */ +function record(files, targetRaw, fromFile = 'App.kt') { + for (const [order, list] of [ + ['fwd', files], + ['rev', [...files].reverse()], + ]) { + const r = resolve(list, targetRaw, fromFile); + if (r !== null) nonNull++; + const rendered = r === null ? 'NULL' : Array.isArray(r) ? `[${r.join(',')}]` : r; + // The FILE SET is part of the hashed record, not just the query and the + // result — see header property 4. + lines.push(`${order}\t${list.join('|')}\t${fromFile}\t${targetRaw}\t${rendered}`); + } +} + +// ---- 1. Exhaustive branch matrix ------------------------------------------ + +// Tier 1, exact. +record(['util/User.kt', 'util/Repo.kt'], 'util.User'); +// Tier 1, suffix (import is not workspace-rooted). +record(['src/main/kotlin/util/User.kt'], 'util.User'); +// Exact anywhere beats a suffix found earlier. +record(['deep/util/User.kt', 'util/User.kt'], 'util.User'); +// No exact match: first suffix in iteration order wins. +record(['a/util/User.kt', 'b/util/User.kt'], 'util.User'); +// .kt / .kts sharing a stem. +record(['dup/Thing.kt', 'dup/Thing.kts'], 'dup.Thing'); +// Multi-segment suffix query. +record(['src/main/com/example/User.kt'], 'com.example.User'); +record(['a/b/com/example/User.kt', 'com/example/User.kt'], 'com.example.User'); +// Tier 2: stripped path matches a file (class-or-object holding the member). +record(['util/OneArg.kt'], 'util.OneArg.writeAudit'); +record(['src/main/kotlin/util/OneArg.kt'], 'util.OneArg.writeAudit'); +// Tier 3: package fan-out to every direct child, in order. +record(['models/User.kt', 'models/Repo.kt', 'models/sub/Deep.kt'], 'models.getRepo'); +record(['models/User.kt', 'models/sub/Deep.kt', 'models/Repo.kt'], 'models.getRepo'); +// Fan-out where the package directory is reached by suffix, not at the root. +record(['app/src/main/kotlin/models/User.kt', 'app/src/main/kotlin/models/Repo.kt'], 'models.get'); +// Tier 4: progressive prefix strip, one and several skip levels. +record(['x/y/z/Deep.kt'], 'com.example.z.Deep'); +record(['z/Deep.kt'], 'a.b.c.d.z.Deep'); +record(['q/Deep.kt'], 'a.b.c.d.e.f.q.Deep'); +// Tier 4 reaching the fan-out tier after stripping. +record(['pkg/A.kt', 'pkg/B.kt'], 'com.example.pkg.someFunction'); +// Backslash normalization. +record(['win\\pkg\\A.kt'], 'win.pkg.A'); +record(['win\\pkg\\A.kt', 'win\\pkg\\B.kt'], 'win.pkg.someFunction'); +// Non-Kotlin files never resolve. +record(['pkg/A.java', 'pkg/A.md', 'pkg/A.kt.txt'], 'pkg.A'); +// Kotlin file alongside non-Kotlin noise of the same stem. +record(['pkg/A.java', 'pkg/A.kt'], 'pkg.A'); +// Header property 2: repeated directory name — a child of the repeated package +// since #2881, at the leading position here and mid-path below. +record(['data/src/main/kotlin/com/example/data/Repo.kt'], 'data.something'); +record(['data/src/main/kotlin/com/example/data/Repo.kt'], 'data.Repo'); +record(['a/c/b/c/File.kt'], 'c.X'); +record(['c/b/c/File.kt'], 'c.X'); +// Doubly nested same-name directory, both below the root. +record(['top/data/mid/data/Repo.kt'], 'data.something'); +// A path starting with the directory name is not its child unless direct. +record(['data/sub/Repo.kt'], 'data.something'); +record(['data/Repo.kt'], 'data.something'); +// Repo-root file has no package directory. +record(['Root.kt'], 'Root'); +record(['Root.kt', 'pkg/Root.kt'], 'Root'); +// Wildcard: `.*` is stripped and lands on the single-file tier, not fan-out. +record(['models/User.kt', 'models/Repo.kt'], 'models.*'); +record(['models/Repo.kt', 'models/User.kt'], 'models.*'); +record(['util/User.kt'], 'util.User.*'); +// Unknown target. +record(['pkg/A.kt'], 'nowhere.Thing'); +// Single-segment target with no directory anywhere. +record(['pkg/A.kt'], 'A'); +// Empty-ish and degenerate targets. +record(['pkg/A.kt'], '*'); +record(['pkg/A.kt'], 'pkg.'); +// fromFile variation must not change the outcome (this resolver ignores it) — +// pinned so a future change that starts consulting it is visible here. +record(['util/User.kt'], 'util.User', 'deep/nested/Caller.kt'); + +// ---- 1b. Depth and package size, the two axes the loops scale on ---------- +// +// Header property 5. The index writes one `suffixByStem` entry per '/' in a +// stem and walks `dir` once per component, so DEPTH is what those two loops +// cost, and `dirChildren` bucket length is what the fan-out tier returns. A +// corpus that pins either as a constant cannot see a guard on it: capping +// suffix-key depth at 7, skipping the `dirChildren` suffix loop above depth 8, +// or capping a bucket at 17 entries all left the fingerprint, `cases` and +// `non_null` byte-identical before these cases existed — while, on a standard +// Gradle layout, the depth skip resolved EVERY package import to null and the +// bucket cap silently truncated fan-out by 58%. +const DEEP = 'core/data/src/main/kotlin/com/example/core/data/repository'; +// 11 components — ordinary for Android/Gradle source, which runs 9-12. +record([`${DEEP}/UserRepository.kt`], 'com.example.core.data.repository.UserRepository'); +record([`${DEEP}/UserRepository.kt`], 'repository.UserRepository'); +record([`${DEEP}/UserRepository.kt`], 'core.data.repository.UserRepository'); +record([`${DEEP}/UserRepository.kt`, `${DEEP}/PostRepository.kt`], 'repository.findAll'); +record([`${DEEP}/UserRepository.kt`, `${DEEP}/PostRepository.kt`], 'core.data.repository.findAll'); +// Deeper still, and with the repeated-name shape at depth. +const DEEPER = 'feature/home/src/main/kotlin/com/example/feature/home/data/local/dao'; +record([`${DEEPER}/UserDao.kt`], 'dao.UserDao'); +record([`${DEEPER}/UserDao.kt`, `${DEEPER}/PostDao.kt`], 'dao.insertAll'); +record([`${DEEPER}/UserDao.kt`], 'home.data.local.dao.UserDao'); +// Suffix keys deeper than 7 components. Depth in the FILE is not enough on its +// own: a cap on how many component-suffixes a stem contributes stays invisible +// unless something QUERIES one of the deep keys, and every Gradle-shaped import +// above is 6 segments or fewer. These reach the top of the stem. +record( + [`${DEEP}/UserRepository.kt`], + 'src.main.kotlin.com.example.core.data.repository.UserRepository', +); +record( + [`${DEEP}/UserRepository.kt`], + 'data.src.main.kotlin.com.example.core.data.repository.UserRepository', +); +record([`${DEEPER}/UserDao.kt`], 'src.main.kotlin.com.example.feature.home.data.local.dao.UserDao'); +record( + [`${DEEPER}/UserDao.kt`], + 'home.src.main.kotlin.com.example.feature.home.data.local.dao.UserDao', +); +record( + [`${DEEP}/UserRepository.kt`, `${DEEP}/PostRepository.kt`], + 'src.main.kotlin.com.example.core.data.repository.findAll', +); + +// A package larger than any plausible bucket cap. 40 files in one package is +// ordinary; a silent sibling cap is exactly what #2732 shipped on the JVM side. +const BIG_PACKAGE = Array.from({ length: 40 }, (_, i) => `${DEEP}/Item${i}.kt`); +record(BIG_PACKAGE, 'repository.someTopLevelFun'); +record(BIG_PACKAGE, 'com.example.core.data.repository.someTopLevelFun'); +record([...BIG_PACKAGE, `${DEEP}/sub/Nested.kt`], 'repository.someTopLevelFun'); + +// ---- 2. Deterministic fuzz ------------------------------------------------- + +/** xorshift32 — seeded, so the corpus is identical on every machine. */ +let seed = 0x9e3779b9; +function rnd() { + seed ^= seed << 13; + seed ^= seed >>> 17; + seed ^= seed << 5; + seed >>>= 0; + return seed / 0x100000000; +} +function pick(arr) { + return arr[Math.floor(rnd() * arr.length)]; +} + +const DIRS = [ + '', + 'app', + 'core', + 'data', + 'feature/home', + 'lib/data', + 'src/main/kotlin', + 'src/main/kotlin/com/example', + 'module/src/main/kotlin/com/example/data', + 'data/src/main/kotlin/com/example/data', + 'top/data/mid/data', + 'win\\pkg', + // Depth beyond the Gradle norm, so the fuzz spans the axis too rather than + // leaving it to the hand-written cases above. + 'core/data/src/main/kotlin/com/example/core/data/repository', + 'feature/home/src/main/kotlin/com/example/feature/home/data/local/dao', + 'a/b/c/d/e/f/g/h/i/j/k/l', +]; +// Segment alphabet overlaps the DIRS entries on purpose: a random dotted target +// only exercises a deep suffix key if its segments can actually align with a +// deep path. +const SEGS = [ + 'User', + 'Repo', + 'Util', + 'Service', + 'Model', + 'data', + 'core', + 'api', + 'store', + 'sub', + 'src', + 'main', + 'kotlin', + 'com', + 'example', + 'repository', + 'dao', +]; +const EXTS = ['.kt', '.kt', '.kt', '.kts', '.java', '.md']; + +function randPath() { + const dir = pick(DIRS); + const base = pick(SEGS); + const file = `${base}${pick(EXTS)}`; + if (dir === '') return file; + return dir.includes('\\') ? `${dir}\\${file}` : `${dir}/${file}`; +} +function randDotted() { + // Up to 9 segments, not 4: import arity is the one axis the branch matrix + // already spanned, but the fuzz should cover it too now that the corpus + // carries paths deep enough for a long target to align with one. + const n = 1 + Math.floor(rnd() * 9); + const parts = []; + for (let i = 0; i < n; i++) parts.push(pick(SEGS)); + return rnd() < 0.12 ? `${parts.join('.')}.*` : parts.join('.'); +} + +// File counts run to 45, not 16: a package that never exceeds 16 direct +// children cannot distinguish an uncapped `dirChildren` bucket from one capped +// at 17 (header property 5). +for (let repo = 0; repo < 400; repo++) { + const fileCount = 3 + Math.floor(rnd() * 43); + const files = []; + for (let i = 0; i < fileCount; i++) files.push(randPath()); + const fromFile = randPath(); + for (let imp = 0; imp < 25; imp++) record(files, randDotted(), fromFile); +} + +const correctnessFingerprint = crypto + .createHash('sha256') + .update([...lines].sort().join('\n')) + .digest('hex'); + +// --------------------------------------------------------------------------- +// Scaling arm +// --------------------------------------------------------------------------- + +/** A synthetic Kotlin monorepo: Gradle-module roots over a shared package + * namespace, at the path depth real Kotlin source has (the index stores one + * suffix entry per '/' in a stem and walks `dir` once per component, so depth + * is a cost driver and a flat corpus would understate the build). + * + * `padDepth` inserts filler segments so the depth arm below can hold the file + * count fixed and vary only depth — the scaling ratio is scale-invariant in + * FILE COUNT and would otherwise never see a depth-driven cost regression. */ +function buildCorpus(fileCount, padDepth = 0) { + const pad = Array.from({ length: padDepth }, (_, d) => `p${d}`).join('/'); + const files = []; + for (let i = 0; i < fileCount; i++) { + const mod = i % 16; + const root = pad === '' ? `lib${mod}` : `lib${mod}/${pad}`; + files.push(`${root}/src/main/kotlin/com/example/mod${mod}/Class${i}.kt`); + } + return files; +} + +/** Import targets for the corpus, ~40% of them unresolvable — see header + * property 3: only a miss drives all four tiers, which is where the + * per-import scan was worst. */ +function buildImports(fileCount) { + const imports = []; + for (let i = 0; i < fileCount * IMPORTS_PER_FILE; i++) { + const kind = i % 5; + const mod = i % 16; + if (kind === 0) + imports.push(`com.example.mod${mod}.Class${i % fileCount}`); // tier 1 hit + else if (kind === 1) + imports.push(`com.example.mod${mod}.someFunction`); // fan-out + else if (kind === 2) + imports.push(`mod${mod}.Class${i % fileCount}`); // suffix + else imports.push(`org.absent.pkg${mod}.Missing${i}`); // full cascade, no hit + } + return imports; +} + +function fastest(values) { + return Math.min(...values); +} + +/** + * Time one full pass: the index build PLUS resolving every import. The build is + * the work the per-import scan was traded for, so hiding it would let an index + * that is itself quadratic pass. Each pass gets its own Set object, because the + * index is memoized on Set identity and a shared Set would build once and make + * every later pass free. The Sets are constructed OUTSIDE the timer so their + * own O(files) cost never lands in the measurement. + */ +function timeResolution(files, imports) { + const sets = []; + for (let i = 0; i < WARMUP + REPS; i++) sets.push(new Set(files)); + const fromFile = files[0]; + + for (let w = 0; w < WARMUP; w++) { + for (const t of imports) { + resolveKotlinImportTarget( + { kind: 'named', localName: 'X', importedName: 'X', targetRaw: t }, + { fromFile, allFilePaths: sets[w] }, + ); + } + } + const samples = []; + for (let r = 0; r < REPS; r++) { + const set = sets[WARMUP + r]; + const t0 = performance.now(); + for (const t of imports) { + resolveKotlinImportTarget( + { kind: 'named', localName: 'X', importedName: 'X', targetRaw: t }, + { fromFile, allFilePaths: set }, + ); + } + samples.push(performance.now() - t0); + } + return fastest(samples); +} + +const scales = {}; +for (const [name, fileCount] of [ + ['small', SMALL], + ['large', LARGE], +]) { + const files = buildCorpus(fileCount); + const imports = buildImports(fileCount); + scales[name] = { + files: fileCount, + imports: imports.length, + ms: Number(timeResolution(files, imports).toFixed(3)), + }; +} + +const scalingRatio = scales.large.ms / scales.small.ms / (LARGE / SMALL); + +// Depth arm: file count fixed, depth roughly tripled. `scaling_ratio` divides +// out the file count, so it is scale-INVARIANT and structurally cannot see a +// cost that grows with path depth instead — and both loops this PR added are +// depth loops. Same corpus size, same imports, only the paths get longer. +const depthFiles = buildCorpus(DEPTH_FILES, 0); +const depthFilesPadded = buildCorpus(DEPTH_FILES, DEPTH_PAD); +const depthImports = buildImports(DEPTH_FILES); +const shallowMs = timeResolution(depthFiles, depthImports); +const deepMs = timeResolution(depthFilesPadded, depthImports); +const depthRatio = deepMs / shallowMs; + +const report = { + small: scales.small, + large: scales.large, + scaling_ratio: Number(scalingRatio.toFixed(3)), + depth: { + files: DEPTH_FILES, + shallow_components: 8, + deep_components: 8 + DEPTH_PAD, + shallow_ms: Number(shallowMs.toFixed(3)), + deep_ms: Number(deepMs.toFixed(3)), + }, + depth_ratio: Number(depthRatio.toFixed(3)), + cases: lines.length, + non_null: nonNull, + fingerprint: correctnessFingerprint, +}; + +if (!process.argv.includes('--check')) { + console.log(JSON.stringify(report, null, 2)); + process.exit(0); +} + +const baseline = JSON.parse(fs.readFileSync(BASELINE_PATH, 'utf-8')); +const failures = []; +if (report.fingerprint !== baseline.fingerprint) { + failures.push( + `fingerprint drift: ${report.fingerprint} != ${baseline.fingerprint} — Kotlin import ` + + `resolution returned a DIFFERENT file set. That is a behaviour change, not a perf one: ` + + `IMPORTS/CALLS edges move in every Kotlin repository. Explain it, never re-baseline to ` + + `make CI green.`, + ); +} +for (const field of ['cases', 'non_null']) { + if (report[field] !== baseline[field]) { + failures.push( + `${field} ${report[field]} != ${baseline[field]} — the corpus itself changed, so the ` + + `fingerprint above is computed over a different surface and proves nothing about the ` + + `resolver. Re-baseline every corpus field together, deliberately.`, + ); + } +} +if (report.scaling_ratio > baseline.scaling_budget) { + failures.push( + `scaling ${report.scaling_ratio} > budget ${baseline.scaling_budget} — per-import cost grows ` + + `with workspace size again, i.e. a tier went back to walking allFilePaths. Timing arm: ` + + `re-run on an idle machine before investigating (see _scaling_note in baselines.json); the ` + + `fingerprint arm is deterministic and never warrants a re-run.`, + ); +} +if (report.depth_ratio > baseline.depth_budget) { + failures.push( + `depth ratio ${report.depth_ratio} > budget ${baseline.depth_budget} — cost now grows with ` + + `PATH DEPTH at a fixed file count. scaling_ratio divides the file count out and cannot ` + + `see this. Timing arm: re-run on an idle machine first.`, + ); +} +if (report.small.ms > baseline.small_ms_ceiling) { + failures.push( + `small arm ${report.small.ms} ms > ceiling ${baseline.small_ms_ceiling} ms — scaling_ratio is ` + + `a RATIO, so a constant-factor regression that grows both arms equally passes it (a full ` + + `scan reintroduced on 1-in-32 imports measured 1.490, inside the budget, while running ` + + `2.8x slower). This ceiling is what catches that. Timing arm: re-run on an idle machine.`, + ); +} + +console.log(JSON.stringify(report, null, 2)); +if (failures.length > 0) { + console.error(`[kotlin-import-target --check] FAIL\n - ${failures.join('\n - ')}`); + process.exit(1); +} +console.log('[kotlin-import-target --check] PASS'); diff --git a/gitnexus/bench/python-scope/baseline-fingerprint.txt b/gitnexus/bench/python-scope/baseline-fingerprint.txt index 49a146016..aff56e0d5 100644 --- a/gitnexus/bench/python-scope/baseline-fingerprint.txt +++ b/gitnexus/bench/python-scope/baseline-fingerprint.txt @@ -1 +1 @@ -a0da3e7c00f603e4bdad91a376b3fc181577a73c2ca1719ab7449d3463c671e0 +2600a1f6f8a042eb4f520a7870c34d9ca292765824537c3bc861b40dac8769a8 diff --git a/gitnexus/bench/receiver-resolution/BASELINE.md b/gitnexus/bench/receiver-resolution/BASELINE.md index f612258bf..329270eaa 100644 --- a/gitnexus/bench/receiver-resolution/BASELINE.md +++ b/gitnexus/bench/receiver-resolution/BASELINE.md @@ -364,8 +364,13 @@ also resolves, so PHP nullable field types already work. **C++ — the base already resolves, but `this->` field receivers do not.** `pointerArrowChain` and `valueDotChain` both RESOLVE, so a decorated C++ base is -not a gap. But `this->repo.save()` and `this->repo->save()` are both -INVISIBLE-GAP — a distinct defect, not a decoration one. +not a gap. `this->repo.save()` and `this->repo->save()` were both INVISIBLE-GAP +when this was written — a distinct defect, not a decoration one — and #2833 +closed it: a language that declares `this` IS the enclosing class +(`resolveThisViaEnclosingClass`) synthesizes no `this` typeBinding anywhere, so +a chain whose BASE is `this` could never seed its head. It was never a generics +gap; the NON-generic control failed identically. C++'s `fieldReceiverCall` and +`decoratedFieldType` cells moved INVISIBLE-GAP -> RESOLVES with it. **Rust — the decorated receiver is NOT a gap.** `&mut self` resolves, so Go is the only language whose method receiver decoration defeats the lookup. Rust's diff --git a/gitnexus/bench/receiver-resolution/baseline.json b/gitnexus/bench/receiver-resolution/baseline.json index d4136c53c..cf8b8abeb 100644 --- a/gitnexus/bench/receiver-resolution/baseline.json +++ b/gitnexus/bench/receiver-resolution/baseline.json @@ -61,9 +61,9 @@ "awaitParen": "N/A", "explicitTypeArgs": "VISIBLE-GAP", "indexElement": "RESOLVES", - "fieldReceiverCall": "INVISIBLE-GAP", + "fieldReceiverCall": "RESOLVES", "decoratedReceiverBase": "N/A", - "decoratedFieldType": "INVISIBLE-GAP" + "decoratedFieldType": "RESOLVES" }, "go": { "plainChain": "RESOLVES", @@ -200,10 +200,11 @@ }, "countArm": { "callDrops": 102, - "totalDropsAllKinds": 129, + "totalDropsAllKinds": 148, "bySiteKind": { "call": 102, - "read": 27 + "read": 27, + "write": 19 }, "callDropsByExtension": { ".java": 49, diff --git a/gitnexus/bench/schema-pairs/README.md b/gitnexus/bench/schema-pairs/README.md index 29594b9d7..a60f89b20 100644 --- a/gitnexus/bench/schema-pairs/README.md +++ b/gitnexus/bench/schema-pairs/README.md @@ -13,9 +13,10 @@ node --import tsx bench/schema-pairs/measure.mjs --check # gate vs baselines. `src/core/lbug/schema.ts` generates its relation pairs from two cross products, and declines to add a third one **on the strength of a number** — roughly 1.04× -at 450 declared pairs, 1.6× at 786, 2.1× at 1024. That measurement used to live -in a scratch directory, so nobody proposing a third rule could re-run it. This -harness is that measurement, committed — and it reproduces those figures. +near production's pair count, 1.6× at 786, 2.1× at 1024. That measurement used +to live in a scratch directory, so nobody proposing a third rule could re-run +it. This harness is that measurement, committed — and it reproduces those +figures. Run it before widening a rule, and quote the new ratio in the review. @@ -29,8 +30,27 @@ Observed on the reference box, **four runs** (ratios vs the 332-pair list): | 786 | 1.52–1.75× | 1.19–1.31× | | 1024 | 2.03–2.34× | 1.31–1.57× | -Production's 450 came out _faster_ than 332 on three of the four runs, so at this -size the pair count is inside run-to-run noise. Everything past ~640 is not. +Production's former 450-pair surface came out _faster_ than 332 on three of the +four runs, so at this size the pair count is inside run-to-run noise. Everything +past ~640 is not. + +#2801 remeasured the new 461-pair production surface on Windows six times: + +| run | untyped ratio | typed ratio | interpretation | +| --- | ------------- | ----------- | ----------------------------------- | +| 1 | 1.101× | 1.157× | noise-dominated (`typed > untyped`) | +| 2 | 1.324× | 1.122× | below the operational budget | +| 3 | 2.705× | 1.089× | exceeds the operational budget | +| 4 | 1.417× | 1.065× | below the operational budget | +| 5 | 1.196× | 1.050× | below the operational budget | +| 6 | 1.585× | 2.577× | noise-dominated (`typed > untyped`) | + +The three comparable Windows runs below the 1.5× operational ceiling span +**1.20–1.42× untyped / 1.05–1.12× typed**. Run 3 is published rather than +silently discarded: no pre-registered rule excludes it, and `--check` would +correctly reject it. These Windows measurements are not combined with the +historical reference-box rows to infer cross-size ordering. + **Quote the range, not a single run** — one run is not evidence here. ## What it measures @@ -52,8 +72,8 @@ data**, then times two query shapes over 40 anchors × 15 reps (median): control; the real cost of widening sits between it and `ratio_*`. A run where `typed_ratio` moves _more_ than `ratio` is noise-dominated and should be rerun. -- **`ratio_`** — `untyped_ms_ / untyped_ms_332`. `ratio_450` is the - figure `schema.ts` quotes. +- **`ratio_`** — `untyped_ms_ / untyped_ms_332`. The + production-size ratio is the figure `schema.ts` quotes. ### Sizes @@ -66,7 +86,8 @@ harness fails if the row counts ever differ across sizes. | size | what it is | | ---- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | | 332 | the pre-#2792 hand-written list — the reference for every ratio | -| 450 | production today (two cross products + 72 hand-declared pairs) | +| 450 | production before Record became linkable (#2801) | +| 461 | production today (two cross products + 69 hand-declared pairs) | | 641 | the third cross product `schema.ts` defers (`DEFINITION_ANCHOR_LABELS × {CodeElement, Section, Typedef, Union, Namespace, Impl, TypeAlias, Static, Template}`), which would leave ~29 hand-declared lines | | 786 | the size an earlier revision of that comment attributed to the third rule — it is 641; kept as a measured waypoint | | 1024 | the full cross product, the ceiling | @@ -77,11 +98,13 @@ Before timing anything, the harness round-trips the **real** `SCHEMA_QUERIES` through a real database and asserts that `CALL SHOW_CONNECTION('CodeRelation')` reports exactly the pairs `parseRelationSchemaPairs` finds in `RELATION_SCHEMA`. -No magic number is baked in: the invariant is that the DDL LadybugDB _accepted_ -carries the pair set our own parser believes it declares. The absolute count is -reported as `declared_pairs`. A pair declared twice would not reach this check at -all — LadybugDB rejects the `CREATE REL TABLE` outright, which is why a duplicate -kills every `analyze` rather than one repository's. +No production-size magic number is baked in: the measured production size and +budget key are derived from that parsed DDL count. A missing `ratio__budget` +entry makes `--check` fail closed. The invariant is that the DDL LadybugDB +_accepted_ carries the pair set our own parser believes it declares. The +absolute count is reported as `declared_pairs`. A pair declared twice would not +reach this check at all — LadybugDB rejects the `CREATE REL TABLE` outright, +which is why a duplicate kills every `analyze` rather than one repository's. ## What it does NOT measure @@ -93,8 +116,11 @@ kills every `analyze` rather than one repository's. ## Regenerating the baseline -`baselines.json` holds one budget, `ratio_450_budget` — the ceiling on what +`baselines.json` holds one production-size budget — the ceiling on what production's own pair count may cost relative to the 332-pair hand-list it replaced. Re-run without `--check` **several times** and copy the top of the -observed `ratio_450` range plus headroom — the spread between runs on this box -is wider than the effect being measured at 450, so a single run cannot set it. +observed production-size ratio range plus headroom — the spread between runs on +this box is wider than the effect being measured near production, so a single +run cannot set it. Publish the raw ratios and apply only the pre-registered +`typed_ratio > ratio` noise rule; do not silently discard another run to make a +budget pass. diff --git a/gitnexus/bench/schema-pairs/baselines.json b/gitnexus/bench/schema-pairs/baselines.json index 41a12d485..b7852d2b0 100644 --- a/gitnexus/bench/schema-pairs/baselines.json +++ b/gitnexus/bench/schema-pairs/baselines.json @@ -1,4 +1,4 @@ { - "_comment": "ratio_450_budget — ceiling on what production's 450-pair set may cost on untyped-endpoint anchored queries, relative to the 332-pair hand-list it replaced. Observed 0.94x and 1.05x across two runs on the reference box (i.e. inside run-to-run noise; it came out faster than 332 once). The budget carries headroom for that spread — compare typed_ratio_450 (1.10-1.17x) for this box's floor. Raise it only with a measured range, never a single run.", - "ratio_450_budget": 1.3 + "_comment": "ratio_461_budget — operational ceiling on what production's 461-pair set may cost on untyped-endpoint anchored queries, relative to the 332-pair hand-list it replaced. #2801 measured 1.20-1.42x across three comparable Windows runs (typed floor 1.05-1.12x), so 1.5x leaves explicit host headroom. All six raw runs are published in README.md; two meet the pre-registered typed_ratio > ratio noise rule, while one additional 2.705x run is not silently discarded and would fail this gate. Raise the budget only with a published measured range, never a single run.", + "ratio_461_budget": 1.5 } diff --git a/gitnexus/bench/schema-pairs/measure.mjs b/gitnexus/bench/schema-pairs/measure.mjs index 59fe92db7..b5ccaa04c 100644 --- a/gitnexus/bench/schema-pairs/measure.mjs +++ b/gitnexus/bench/schema-pairs/measure.mjs @@ -3,10 +3,10 @@ * * `src/core/lbug/schema.ts` declares its relation pairs from two cross products * plus a small hand-written remainder, and it justifies NOT adding a third cross - * product with a number: anchored queries cost ~1.04× at 450 declared pairs but - * 1.6× at 786 and 2.1× at 1024. That measurement previously lived in a scratch - * directory, so the claim could not be re-checked when someone proposed - * widening a rule. This is it, committed. + * product with a number: anchored queries cost ~1.04× near production's pair + * count but 1.6× at 786 and 2.1× at 1024. That measurement previously lived in + * a scratch directory, so the claim could not be re-checked when someone + * proposed widening a rule. This is it, committed. * * WHAT IT MEASURES. Against a real `@ladybugdb/core` database, with byte-identical * DATA at every size, it times the query shape whose plan actually depends on the @@ -34,7 +34,8 @@ * same query at every size, and the only variable is how many UNUSED pairs the * table declares: * - 332 — the pre-#2792 hand-written list (the historical baseline); - * - 450 — production today (two cross products + 72 hand-declared); + * - 450 — production before Record became linkable (#2801); + * - 461 — production today (two cross products + 69 hand-declared); * - 641 — the third cross product schema.ts defers * (`DEFINITION_ANCHOR_LABELS × {CodeElement, Section, Typedef, Union, * Namespace, Impl, TypeAlias, Static, Template}`), which would leave @@ -43,8 +44,8 @@ * third rule (it is 641; 786 is kept as a measured waypoint); * - 1024 — the full cross product, the ceiling. * - * Ratios are reported against 332, the smallest size — `ratio_450` is the - * number schema.ts quotes. + * Ratios are reported against 332, the smallest size — the production-size + * ratio is the number schema.ts quotes. * * CORRECTNESS GATE. Before timing anything it round-trips the REAL * `SCHEMA_QUERIES` through a real database and asserts that @@ -61,9 +62,9 @@ * node --import tsx bench/schema-pairs/measure.mjs # print JSON lines * node --import tsx bench/schema-pairs/measure.mjs --check # gate vs baselines.json * - * `--check` fails if the correctness gate breaks, or if `ratio_450` exceeds its - * budget — i.e. if production's own pair count starts costing materially more - * than the hand-written list it replaced. + * `--check` fails if the correctness gate breaks, or if the production-size + * ratio exceeds its budget — i.e. if production's own pair count starts + * costing materially more than the hand-written list it replaced. */ import fs from 'node:fs'; import os from 'node:os'; @@ -85,9 +86,14 @@ const lbug = (await import('@ladybugdb/core')).default; // ---- sizes + the pair enumeration every size is a prefix of ---- -const SIZES = [332, 450, 641, 786, 1024]; const REFERENCE_SIZE = 332; // ratios are relative to this -const PRODUCTION_SIZE = 450; // the size schema.ts ships +// Derive production from the same executable DDL the correctness gate +// round-trips. A LINKABLE_LABELS widening must not require a second copied +// count here — and cannot silently select a stale/missing budget key. +const PRODUCTION_SIZE = parseRelationSchemaPairs(RELATION_SCHEMA).size; +const SIZES = [...new Set([REFERENCE_SIZE, 450, PRODUCTION_SIZE, 641, 786, 1024])].sort( + (a, b) => a - b, +); // The four pairs the synthetic data uses. Pinned to the FRONT of the // enumeration so they are declared at every size — otherwise a smaller pair set @@ -338,8 +344,13 @@ if (!CHECK) { process.stdout.write(JSON.stringify(summary) + '\n'); } else { const baselines = JSON.parse(fs.readFileSync(BASELINE_PATH, 'utf8')); - const budget = baselines[`ratio_${PRODUCTION_SIZE}_budget`]; - if (budget !== undefined && summary[`ratio_${PRODUCTION_SIZE}`] >= budget) { + const budgetKey = `ratio_${PRODUCTION_SIZE}_budget`; + const budget = baselines[budgetKey]; + if (budget === undefined) { + failures.push(`no ${budgetKey} in baselines.json — the production gate is disarmed`); + } else if (typeof budget !== 'number' || !Number.isFinite(budget)) { + failures.push(`${budgetKey} must be a finite number (got ${JSON.stringify(budget)})`); + } else if (summary[`ratio_${PRODUCTION_SIZE}`] >= budget) { failures.push( `production pair set (${PRODUCTION_SIZE}) costs ${summary[`ratio_${PRODUCTION_SIZE}`]}× vs ` + `${REFERENCE_SIZE} pairs, >= budget ${budget} (untyped ${reference.untyped_ms}ms -> ` + diff --git a/gitnexus/bench/scope-capture/baselines.json b/gitnexus/bench/scope-capture/baselines.json index fe25691f5..9cc89b431 100644 --- a/gitnexus/bench/scope-capture/baselines.json +++ b/gitnexus/bench/scope-capture/baselines.json @@ -1,27 +1,28 @@ { "_comment": "Per-language baselines for bench/scope-capture/measure.mjs --check. fingerprint = order-independent sha256 over the lang-resolution/-* fixture corpus + a 20-entity synthetic source (correctness gate; re-baseline intentionally on a legitimate capture change). scaling_budget = max allowed (t800/t250)/(800/250); ~1.0 is linear, ~3.2 is quadratic. The synthetic source is now HERITAGE-BEARING for every language (each Entity extends/implements/embeds/uses-trait/conforms-to a shared base) so the #1951 @reference.inherits synth is gated at scale, not just the base capture loop. All languages thread the tree-sitter captured node instead of re-deriving it with findNodeAtRange(tree.rootNode,...) per match, so all are linear (go #1915, python #1918, ruby/php/rust/csharp #1951, java #1956).", "go": { - "fingerprint": "e386598526e502d131e52a17d219635b3a4196d94f1ebdd25922a2582c985d18", + "fingerprint": "9c554a9d698a2b79fb419852daadca87b8aae88180cceabf9c8d82f3e3300f2e", "scaling_budget": 1.5, "_rebaselined_callable_flow_2522_review": "PR #2522 review hardening: callable operands retain expression/qualified identity and formals retain signature metadata. Prior 3d4e32e7490c830516126e28931827949baa3594cb521f7a3d8dcfed95b6018a -> 57b3c55135af8d2af33b9a7c4bf89796a7bee5b5822b402a2dea91af7232cf4a; scaling 1.058 < 1.5.", "_rebaselined_callable_flow_2522_followup": "PR #2522 follow-up: provider-owned callable assignment/copy/formal/argument/invoke facts with invocation/constructor-result suppression. Prior 09ecd94911b830f52fa8807560abcbd79f163d02a2072870c1a59297e9a326e1 -> 3d4e32e7490c830516126e28931827949baa3594cb521f7a3d8dcfed95b6018a; scaling 1.039 < 1.5.", "_rebaselined": "#1976: F33 generic composite literal constructor inference adds generic_type captures in composite_literal patterns; fingerprint drift expected.", - "_rebaselined_receiver_chain_2747": "#2747 receiver-chain rollout: call matches whose receiver is itself an expression now carry `@reference.receiver-chain`, a compact encoding of the receiver's structure, so resolution types it by folding instead of re-parsing receiver source text. Capture GROUP counts are unchanged — the tag is added to existing call matches, never a new match — so this is digest drift only. Prior 57b3c55135af8d2af33b9a7c4bf89796a7bee5b5822b402a2dea91af7232cf4a -> 5d6c59c2f2c0dd937c53bf5d736e0f8376b2899a381e488a33aec23524823efb.", + "_rebaselined_receiver_chain_2747": "#2747 receiver-chain rollout: call matches whose receiver is itself an expression now carry `@reference.receiver-chain`, a compact encoding of the receiver's structure, so resolution types it by folding instead of re-parsing receiver source text. Capture GROUP counts are unchanged \u2014 the tag is added to existing call matches, never a new match \u2014 so this is digest drift only. Prior 57b3c55135af8d2af33b9a7c4bf89796a7bee5b5822b402a2dea91af7232cf4a -> 5d6c59c2f2c0dd937c53bf5d736e0f8376b2899a381e488a33aec23524823efb.", "_rebaselined_2766_go_pointer_receiver_fixture": "#2766: added test/fixtures/lang-resolution/go-pointer-receiver-field-chain/ (2 Go files) as the committed regression fixture for pointer-receiver base resolution. Go fixture_count 100 -> 102. Prior 5d6c59c2f2c0dd937c53bf5d736e0f8376b2899a381e488a33aec23524823efb -> 8cba537ff211fab3bac5fb4456cd1ffba14d6a2db75c40acae28ab8bf29f3d2e. FIXTURE-CORPUS GROWTH, NOT A CAPTURE CHANGE: the accompanying fix is a resolution-time lookup fallback (stripTypePreservingDecoration) and cannot move capture output; go was the ONLY language whose fingerprint drifted, and every other language matched its baseline on the same run.", - "_rebaselined_2766_receiver_chain_wire_v2": "#2766: receiver-chain wire format v1 -> v2 (name-free `await` / `index` step kinds). The VERSION prefix is part of every emitted `@reference.receiver-chain` capture, so every chain-minting language's capture text changed. WIRE-FORMAT CHANGE, NOT A CAPTURE-SET CHANGE: the same chains are minted for the same sites, spelled `2|…` instead of `1|…`. Exactly the 12 chain-minting languages drifted; c, cobol and dart did not, which is the check that this is the prefix and not a capture regression. Accompanied by SCHEMA_BUMP 34 -> 37 and INCREMENTAL_SCHEMA_VERSION 28 -> 31 so a stale index is rejected rather than replaying chains a v2 decoder refuses. Prior 8cba537ff211fab3bac5fb4456cd1ffba14d6a2db75c40acae28ab8bf29f3d2e -> 8162272bb897b0b89472c406321cf8d88a5ae4ea83ea9e3c45f8e817041bff9f.", - "_rebaselined_2766_await_subscript_emission": "#2766: extractMixedChain now walks THROUGH await and subscript nodes and peels transparent wrappers at loop entry, so sites whose receiver is `repos[0]` or `(await f())` mint a receiver chain where they previously minted none. EMISSION CHANGE: more sites carry `@reference.receiver-chain`; no existing chain changed shape. Only go and kotlin drifted of 15 — the two whose fixture corpora contain such receivers. Prior 8162272bb897b0b89472c406321cf8d88a5ae4ea83ea9e3c45f8e817041bff9f -> c9c908f441e3be12fad2448120ed3ea35dc235a12b3f63b0ec532ffdae11d9e9.", - "_rebaselined_2766_phantom_callee_read_site": "#2766: Go's `@reference.read` pattern matches EVERY selector_expression, so a member call `h.dep.Work()` minted THREE sites — the call, the genuine `h.dep` field read, and a PHANTOM read on the callee `h.dep.Work`. The phantom resolved through findOwnedMember (which prefers methods over fields) and emitted an ACCESSES edge to the METHOD duplicating the CALLS edge at the same position; visible today on any receiver the text cascade can type (`RunFromValueReceiver -> DoWork`). The emitter now drops a read match whose selector is in FUNCTION position. FEWER capture matches for Go, no other language affected — go was the only fingerprint of 15 that moved. A method VALUE (`f := h.dep.Work`) is not in function position and is untouched. Prior c9c908f441e3be12fad2448120ed3ea35dc235a12b3f63b0ec532ffdae11d9e9 -> 7bb524a32a2eed57a15b454e3a33480e92a496c683e6856ef02179693c0e02e3.", + "_rebaselined_2766_receiver_chain_wire_v2": "#2766: receiver-chain wire format v1 -> v2 (name-free `await` / `index` step kinds). The VERSION prefix is part of every emitted `@reference.receiver-chain` capture, so every chain-minting language's capture text changed. WIRE-FORMAT CHANGE, NOT A CAPTURE-SET CHANGE: the same chains are minted for the same sites, spelled `2|\u2026` instead of `1|\u2026`. Exactly the 12 chain-minting languages drifted; c, cobol and dart did not, which is the check that this is the prefix and not a capture regression. Accompanied by SCHEMA_BUMP 34 -> 37 and INCREMENTAL_SCHEMA_VERSION 28 -> 31 so a stale index is rejected rather than replaying chains a v2 decoder refuses. Prior 8cba537ff211fab3bac5fb4456cd1ffba14d6a2db75c40acae28ab8bf29f3d2e -> 8162272bb897b0b89472c406321cf8d88a5ae4ea83ea9e3c45f8e817041bff9f.", + "_rebaselined_2766_await_subscript_emission": "#2766: extractMixedChain now walks THROUGH await and subscript nodes and peels transparent wrappers at loop entry, so sites whose receiver is `repos[0]` or `(await f())` mint a receiver chain where they previously minted none. EMISSION CHANGE: more sites carry `@reference.receiver-chain`; no existing chain changed shape. Only go and kotlin drifted of 15 \u2014 the two whose fixture corpora contain such receivers. Prior 8162272bb897b0b89472c406321cf8d88a5ae4ea83ea9e3c45f8e817041bff9f -> c9c908f441e3be12fad2448120ed3ea35dc235a12b3f63b0ec532ffdae11d9e9.", + "_rebaselined_2766_phantom_callee_read_site": "#2766: Go's `@reference.read` pattern matches EVERY selector_expression, so a member call `h.dep.Work()` minted THREE sites \u2014 the call, the genuine `h.dep` field read, and a PHANTOM read on the callee `h.dep.Work`. The phantom resolved through findOwnedMember (which prefers methods over fields) and emitted an ACCESSES edge to the METHOD duplicating the CALLS edge at the same position; visible today on any receiver the text cascade can type (`RunFromValueReceiver -> DoWork`). The emitter now drops a read match whose selector is in FUNCTION position. FEWER capture matches for Go, no other language affected \u2014 go was the only fingerprint of 15 that moved. A method VALUE (`f := h.dep.Work`) is not in function position and is untouched. Prior c9c908f441e3be12fad2448120ed3ea35dc235a12b3f63b0ec532ffdae11d9e9 -> 7bb524a32a2eed57a15b454e3a33480e92a496c683e6856ef02179693c0e02e3.", "_rebaselined_2766_callee_position_marker": "#2766 review fix: a call's callee selector is no longer DROPPED at capture. An earlier commit on this branch dropped it outright, which also deleted the genuine field read on a func-typed struct field (`h.dep.Work()` where `Work func() error`) - callback/hook/mock structs lost their only ACCESSES evidence. The match is now emitted carrying `@reference.callee-position`, and the phantom is suppressed at EMIT by the resolved target's kind instead. Go only: the other 14 languages' fingerprints are byte-identical, which is the check that this is not a cross-language capture change. Prior 7bb524a32a2eed57a15b454e3a33480e92a496c683e6856ef02179693c0e02e3 -> e47302079e17a5e73711bbed5416557b49327cb67e4932008700ec6b8fb468b3; scaling 1.001 < 1.5; fixtures 102 (unchanged), capture_groups_fp 2103.", "_rebaselined_2813_interface_field_dispatch_fixture": "#2813: added test/fixtures/lang-resolution/go-interface-field-dispatch/ (8 Go files) as the committed regression fixture for calls through an interface-typed struct field. Go fixture_count 102 -> 110. FIXTURE-CORPUS GROWTH, NOT A CAPTURE CHANGE: the accompanying fixes are a detection-time method-set change (interface-impls.ts) and a resolution-time fan-out in the shared receiver pass, neither of which emits captures; go/query.ts and go/captures.ts are untouched. Go was the ONLY language whose fingerprint drifted, and every other language matched its baseline on the same run - the same check used for the #2766 fixture growth above. Prior e47302079e17a5e73711bbed5416557b49327cb67e4932008700ec6b8fb468b3 -> cffee41cadbf350855d99bd5aee7c015b1e8b31d1c343d02f113540abe86c765; scaling 1.074 < 1.5, capture_groups_fp 2303.", - "_rebaselined_2837": "#2837: Go struct/interface captures re-anchored from the type_declaration onto the type_spec (@scope.class/@declaration.struct/@declaration.interface in languages/go/query.ts, @definition.struct/@definition.interface in GO_QUERIES). A grouped `type (...)` block used to yield ONE scope and ONE node for every type in it, so each type after the first lost its field typeBindings and every field-receiver call in the file emitted nothing. Capture COUNT is unchanged; only ranges moved, plus the new go-grouped-type-decl fixture. Prior c27fb803598581fa4eb7ddf5ef6f8369b9e3a150082d11362e7aa3ec8faaa832 -> e386598526e502d131e52a17d219635b3a4196d94f1ebdd25922a2582c985d18; scaling 1.054 < 1.5." + "_rebaselined_2837": "#2837: Go struct/interface captures re-anchored from the type_declaration onto the type_spec (@scope.class/@declaration.struct/@declaration.interface in languages/go/query.ts, @definition.struct/@definition.interface in GO_QUERIES). A grouped `type (...)` block used to yield ONE scope and ONE node for every type in it, so each type after the first lost its field typeBindings and every field-receiver call in the file emitted nothing. Capture COUNT is unchanged; only ranges moved, plus the new go-grouped-type-decl fixture. Prior c27fb803598581fa4eb7ddf5ef6f8369b9e3a150082d11362e7aa3ec8faaa832 -> e386598526e502d131e52a17d219635b3a4196d94f1ebdd25922a2582c985d18; scaling 1.054 < 1.5.", + "_rebaselined_2873_undecided_satisfaction_fixtures": "#2873: added test/fixtures/lang-resolution/go-extern-qualified-signatures/ (5 Go files) and go-undecided-satisfaction/ (1 Go file) as the committed regression fixtures for out-of-repo package qualifiers in method signatures and for a satisfaction check that cannot be decided. Go fixture_count 116 -> 122. Prior e386598526e502d131e52a17d219635b3a4196d94f1ebdd25922a2582c985d18 -> 9c554a9d698a2b79fb419852daadca87b8aae88180cceabf9c8d82f3e3300f2e. FIXTURE-CORPUS GROWTH, NOT A CAPTURE CHANGE: the accompanying fix is resolution-time (signatureContextForFile recovers an identity for unresolvable imports) plus a tri-state verdict, neither of which runs during capture; go was the ONLY language whose fingerprint drifted and every other language matched its baseline on the same run." }, "cobol": { "fingerprint": "c8c00b56a7da24e04080eb885714fbbf45e3903324f0cf9df0754f5b5a92e3aa", - "_rebaselined_2813_exact_method_sets": "#2813: Go embedded fields now emit `@reference.embedded-pointer` when spelled `*T` rather than `T`. A CAPTURE-EMISSION CHANGE, not fixture growth: fixture_count is unchanged at 110 and capture_groups_fp moves 2303 -> 2339 (+36), which is the new marker plus the WrongSigRepo/Recount rows added to two existing fixture files. The marker is required for exactness — Go gives `struct{ Base }` and `struct{ *Base }` different method sets, so structural interface satisfaction cannot be correct without knowing which was written (go.dev/ref/spec#Struct_types). Go was the ONLY language of 15 whose fingerprint moved, which is the check that this is a Go capture change and not a cross-language regression. Accompanied by SCHEMA_BUMP 39 -> 43 (skipping 40/41/42, taken by origin/main during review) so a warm cache cannot replay the pre-marker capture set. Prior cffee41cadbf350855d99bd5aee7c015b1e8b31d1c343d02f113540abe86c765 -> c27fb803598581fa4eb7ddf5ef6f8369b9e3a150082d11362e7aa3ec8faaa832; scaling 0.987 < 1.5.", + "_rebaselined_2813_exact_method_sets": "#2813: Go embedded fields now emit `@reference.embedded-pointer` when spelled `*T` rather than `T`. A CAPTURE-EMISSION CHANGE, not fixture growth: fixture_count is unchanged at 110 and capture_groups_fp moves 2303 -> 2339 (+36), which is the new marker plus the WrongSigRepo/Recount rows added to two existing fixture files. The marker is required for exactness \u2014 Go gives `struct{ Base }` and `struct{ *Base }` different method sets, so structural interface satisfaction cannot be correct without knowing which was written (go.dev/ref/spec#Struct_types). Go was the ONLY language of 15 whose fingerprint moved, which is the check that this is a Go capture change and not a cross-language regression. Accompanied by SCHEMA_BUMP 39 -> 43 (skipping 40/41/42, taken by origin/main during review) so a warm cache cannot replay the pre-marker capture set. Prior cffee41cadbf350855d99bd5aee7c015b1e8b31d1c343d02f113540abe86c765 -> c27fb803598581fa4eb7ddf5ef6f8369b9e3a150082d11362e7aa3ec8faaa832; scaling 0.987 < 1.5.", "scaling_budget": 1.5, "_rebaselined_callable_flow_2522_followup": "PR #2522 follow-up: COBOL procedure-pointer callable flow facts; multi-topic extraction now consumes each grouped scope/declaration match once instead of requiring a duplicate declaration-only match. Prior 68ee0e95eb9f86f2d92ca35f730f4c2d4d83abc1b5241ae767ff3437780ec8d1 -> d45bb091b0893d0de4fae2486b31ba21719c9377bf35a0908fd3a36fa1c3bf4e; scaling 0.853 < 1.5.", "_note": "Updated for F17-F23 fixes (P2: TIMES guard, ADD GIVING, SQL AS alias). See PR #1959.", - "_rebaselined_2793_declaratives": "PR #2793: corpus-only re-baseline. `cobol-declaratives` was added to test/fixtures/lang-resolution to reproduce the `Namespace→Record` analyze abort (DECLARATIVES / USE AFTER STANDARD ERROR ON ), and this bench globs `lang-resolution/cobol-*`, so the corpus grew 14 -> 15 files. Verified capture-neutral: with that one fixture moved aside the fingerprint is byte-identical to the prior d45bb091b0893d0de4fae2486b31ba21719c9377bf35a0908fd3a36fa1c3bf4e. No COBOL capture code changed in that PR. Scaling 0.677 < 1.5." + "_rebaselined_2793_declaratives": "PR #2793: corpus-only re-baseline. `cobol-declaratives` was added to test/fixtures/lang-resolution to reproduce the `Namespace\u2192Record` analyze abort (DECLARATIVES / USE AFTER STANDARD ERROR ON ), and this bench globs `lang-resolution/cobol-*`, so the corpus grew 14 -> 15 files. Verified capture-neutral: with that one fixture moved aside the fingerprint is byte-identical to the prior d45bb091b0893d0de4fae2486b31ba21719c9377bf35a0908fd3a36fa1c3bf4e. No COBOL capture code changed in that PR. Scaling 0.677 < 1.5." }, "c": { "fingerprint": "3418cded9f7072152f68992f0a426f43ae7d9d553579a47075fc0cab185848a5", @@ -29,13 +30,15 @@ "_rebaselined_callable_flow_2522_review": "PR #2522 review hardening: callable operands retain expression/qualified identity and formals retain signature metadata. Prior 57fee292147ae6d2db7062da1e07d17122cf355207c8967fa85fd2ec9ca398a4 -> 3418cded9f7072152f68992f0a426f43ae7d9d553579a47075fc0cab185848a5; scaling 1.073 < 1.5.", "_rebaselined_callable_flow_2522_followup": "PR #2522 follow-up: C function-pointer signatures plus direct-callee argument metadata and invocation-result suppression. Prior 75bcdbbf006bf9bd263c0f5857461b118f39b164e9f821cb0651ad0ec46ef6ae -> 57fee292147ae6d2db7062da1e07d17122cf355207c8967fa85fd2ec9ca398a4; scaling 1.035 < 1.5.", "_rebaselined_callable_flow": "Callable-value-flow facts for C function pointers, copies, pointer-to-pointer cells, arguments, and indirect invokes. Prior 12a196b2d6249c8d86a931b12ecebc2a0cdf8d6f47683acdd0d8e9d8bc7657f5 -> 75bcdbbf006bf9bd263c0f5857461b118f39b164e9f821cb0651ad0ec46ef6ae; measured scaling ratio 0.980 < 1.5.", - "_added": "#1956: c added to the scope-capture bench (was UNBENCHED). C has no inheritance — flat scale source. Adding it exposed + fixed a pre-existing O(n^2) findNodeAtRange root-walk in c/captures.ts (threaded c.node, byte-identical over c-* fixtures); scaling 3.475 -> 0.96.", - "_note": "#1983: + c-static-linkage-worker fixture (caller.c/lib.c/lib.h/local.c — worker-path static-linkage side-channel test). Pure fixture-corpus drift: no c/captures.ts or query change branch-vs-main, existing fixtures' captures byte-identical (c-captures.test.ts 45/45), scaling stays linear (~0.97). The baseline was missed when the fixture landed; regenerated here. fingerprint 0de009b->39f3a83.", + "_added": "#1956: c added to the scope-capture bench (was UNBENCHED). C has no inheritance \u2014 flat scale source. Adding it exposed + fixed a pre-existing O(n^2) findNodeAtRange root-walk in c/captures.ts (threaded c.node, byte-identical over c-* fixtures); scaling 3.475 -> 0.96.", + "_note": "#1983: + c-static-linkage-worker fixture (caller.c/lib.c/lib.h/local.c \u2014 worker-path static-linkage side-channel test). Pure fixture-corpus drift: no c/captures.ts or query change branch-vs-main, existing fixtures' captures byte-identical (c-captures.test.ts 45/45), scaling stays linear (~0.97). The baseline was missed when the fixture landed; regenerated here. fingerprint 0de009b->39f3a83.", "_rebaselined": "#1919 open-language coverage: new lang-resolution fixtures + intended capture additions (F5/F9 c-cpp, F26/F28/F29 dart, F47/F48/F49/F51/F52 kotlin, F75/F79 swift). Fingerprint-only drift; scaling_ratio ~1.0 (linear, no perf regression)." }, "cpp": { - "fingerprint": "856d02f3f9d22cb973877211100aee8e052d4bc545922f78704b1a21ce49ddcc", + "fingerprint": "bf3587674267be1759e7c45abef143c3b81fe8629cfd17da5f8af40e83cc39ec", "scaling_budget": 1.5, + "_rebaselined_2833_qualified_member_fields": "#2833 follow-up: the six per-qualifier-depth `field_declaration` type-binding rules for a QUALIFIED generic member are replaced by three depth-agnostic ones that match the outer `qualified_identifier` itself, with the qualifier reduced to its top-level tail in `interpret.ts` (`cppQualifiedTail`). This is a CAPTURE-LOGIC change and it moves the fingerprint in two places at once. (1) A qualified NON-generic member (`ns::Address addr;`, `std::string name;`) was captured by nothing at all and now binds \u2014 that is the whole +24 on the fixture corpus, every one of them a `std::string` member. (2) Qualifier depth is no longer enumerated, so `a::b::c::Repo` (depth 3+) is captured where the old rules stopped at 2. Capture-name histogram, cpp-* corpus (278 files): `@type-binding.field` 8 -> 32, `@type-binding.name` and `@type-binding.type` 401 -> 425; synthetic DAO-20: `@type-binding.field` 40 -> 60, `@type-binding.name` and `@type-binding.type` 61 -> 81 (= 20 entities x the one `std::string name;` member the DAO unit already declared). NO OTHER TAG MOVED in either set \u2014 not one `@declaration.*`, `@scope.*` or `@reference.*` count \u2014 which is the property that says three rules replaced six without widening what a field_declaration matches. Measured over the 13 cpp-* fixture repos whose sources gained a binding, the distinct CALLS edge set is byte-identical before and after (32 edges): a reduced tail that names no workspace class binds nothing. Prior bd47c82d09a83cbf0ac857f41876fa31d22304043735582e913bccde06cf2c1a -> db1156d81b3e3341faf5e938a4a34417f4fd246588b6150b4686481823262529; scaling 1.04 < 1.5.", + "_rebaselined_2833_generic_member_fields": "#2833 review follow-up: the cpp DAO generator's unit gains two GENERIC member fields \u2014 `Repo repo;` (bare template_type) and `std::vector items;` (qualified_identifier wrapping a template_type) \u2014 plus the header declaring `template class Repo`. CORPUS CHANGE, NOT A CAPTURE-LOGIC CHANGE: no extractor edit accompanies it. It exists because the corpus had ZERO template-typed member fields and, across 279 cpp-* fixtures, not one qualified generic member either, so BOTH rounds of new `field_declaration` type-binding rules landed with a byte-identical cpp fingerprint \u2014 the gate was structurally blind to the exact thing being changed. Measured under the new corpus, the three states now differ: pre-#2833 query 0e7cbda71360b7ff35dd76091c77f288d6af6a5cfa9185ad85a372aae8c85191 (4521 groups) -> the three template_type field rules de07d8b5300ed867b460918e16b4d80259c7eb6efc1034d32bebe9ff7cab126d (4541) -> the six qualified rules bd47c82d09a83cbf0ac857f41876fa31d22304043735582e913bccde06cf2c1a (4561); under the OLD corpus all three were 856d02f3f9d22cb973877211100aee8e052d4bc545922f78704b1a21ce49ddcc. Capture-name histogram over the synthetic DAO-20: `@type-binding.field` 0 -> 40, `@declaration.field` 40 -> 80, `@type-binding.type`/`@type-binding.name` 20 -> 61, `@declaration.name` 104 -> 147 \u2014 40 = 20 entities x 2 fields, with the residual +1/+2/+3 attributable to the one-off header declaration; every `@reference.*` count is unchanged. Prior 856d02f3f9d22cb973877211100aee8e052d4bc545922f78704b1a21ce49ddcc -> bd47c82d09a83cbf0ac857f41876fa31d22304043735582e913bccde06cf2c1a; scaling 1.058 < 1.5. `c` is unaffected (3418cded..., unchanged).", "_rebaselined_callable_flow_2522_review": "PR #2522 review hardening: callable operands retain expression/qualified identity and formals retain signature/cv metadata. Prior dde874d2c30bda9f634f9799281a66de800cad9f76cf65e7c31839e2ae9da9ff -> 57860dd2a8d4b06c6d2dd0d854c08b781faee3da8f2b6c42ba0c68a9f70e5ccb; scaling 1.090 < 1.5.", "_rebaselined_callable_flow_2522_followup": "PR #2522 follow-up: C++ overload-aware function/reference/member-pointer flow facts with invocation/constructor-result suppression. Prior 3a503a1513e7eede3f7a223dcce0896c06d15bdfa920445224c9025848c0d710 -> dde874d2c30bda9f634f9799281a66de800cad9f76cf65e7c31839e2ae9da9ff; scaling 1.034 < 1.5.", "_rebaselined_callable_flow": "Callable-value-flow facts for C++ function pointers/references, reference aliases, contextual arity, arguments, and member-pointer syntax. Prior 6ab657c8f9bfe988a3759098c2cffdcc0443def75ff263f1282b82c21d96e931 -> 3a503a1513e7eede3f7a223dcce0896c06d15bdfa920445224c9025848c0d710; measured scaling ratio 1.069 < 1.5.", @@ -43,37 +46,50 @@ "_note_1899_followup": "#1899 follow-up: braced-init metadata now carries element count, intentionally changing C++ capture output; CI benchmark scaling remains linear (1.129 < 1.5).", "_added": "#1956: cpp added to the scope-capture bench (was UNBENCHED). Heritage-bearing scale source (: public Base, public Mixin) drives emitCppInheritanceCaptures at scale. Adding it exposed + fixed a pre-existing O(n^2) findNodeAtRange root-walk in cpp/captures.ts (~12 sites, threaded c.node, byte-identical over 263 cpp-* fixtures); scaling 2.30 -> 1.12.", "_rebaselined": "#1919 open-language coverage: new lang-resolution fixtures + intended capture additions (F5/F9 c-cpp, F26/F28/F29 dart, F47/F48/F49/F51/F52 kotlin, F75/F79 swift). Fingerprint-only drift; scaling_ratio ~1.0 (linear, no perf regression). #2094: deleted C++ declarations retain @declaration.is-deleted metadata; deleted operator and pointer-return shapes plus the expanded deleted-overload fixture are included. Intended capture drift; scaling remains linear (1.139 < 1.5).", - "_note": "#1975: + cpp-out-of-line-class fixture, fixture_count 263->265. #1990: + cpp-adl-ns-plus-hidden-friend-same-name fixture (ADL hidden-friend + namespace-callable merge parity test). Pure fixture-corpus drift — no scope-extractor change; existing fixtures' captures byte-identical. fixture_count 265->267. #1995: + cpp-union-nested-tail-collision and cpp-anon-ns-tail-collision fixtures — pure fixture-corpus drift; fixture_count 270->272, fingerprint 538e8be->d63ded6. #1993: + cpp-cross-namespace-same-tail fixture — pure fixture-corpus drift; fixture_count 272->273, fingerprint d63ded6->6d6207ae. #2077 review follow-up: cpp-member-lattice adds cross-file, qualified-base, nested-template, inherited-using, this-receiver, and non-virtual-override regressions; fixture_count 274->275. Capture scaling remains linear (1.134 < 1.5). #1899: braced-init call arguments emit a conservative parameter-type capture; fixture_count 277, scaling remains linear (1.141 < 1.5).", + "_note": "#1975: + cpp-out-of-line-class fixture, fixture_count 263->265. #1990: + cpp-adl-ns-plus-hidden-friend-same-name fixture (ADL hidden-friend + namespace-callable merge parity test). Pure fixture-corpus drift \u2014 no scope-extractor change; existing fixtures' captures byte-identical. fixture_count 265->267. #1995: + cpp-union-nested-tail-collision and cpp-anon-ns-tail-collision fixtures \u2014 pure fixture-corpus drift; fixture_count 270->272, fingerprint 538e8be->d63ded6. #1993: + cpp-cross-namespace-same-tail fixture \u2014 pure fixture-corpus drift; fixture_count 272->273, fingerprint d63ded6->6d6207ae. #2077 review follow-up: cpp-member-lattice adds cross-file, qualified-base, nested-template, inherited-using, this-receiver, and non-virtual-override regressions; fixture_count 274->275. Capture scaling remains linear (1.134 < 1.5). #1899: braced-init call arguments emit a conservative parameter-type capture; fixture_count 277, scaling remains linear (1.141 < 1.5).", "_rebaselined_2522_review_fixes": "PR #2522 review fixes: outermost-chain passing modes; ->* ERROR-recovery role order; member-store visibility. Prior 57860dd2a8d4b06c6d2dd0d854c08b781faee3da8f2b6c42ba0c68a9f70e5ccb -> f29bc3f7b1622954d6f6b7647bc9cf6c7a2629ffcc0fe00ac7918e4925876b65; scaling ratio re-verified within budget.", - "_rebaselined_2522_prototype_value_cells": "Plain function/method prototypes no longer index as callable value cells (only pointer/parenthesized variable declarators do) — removes the spurious indirect-invoke facts that leaked phantom CALLS past two-phase suppression. Prior f29bc3f7b1622954d6f6b7647bc9cf6c7a2629ffcc0fe00ac7918e4925876b65 -> a70625bb0a9ef74e760d9d79cc5557485d0f0d3fb935e8a22a0c9556c65b5bb1; scaling re-verified within budget.", + "_rebaselined_2522_prototype_value_cells": "Plain function/method prototypes no longer index as callable value cells (only pointer/parenthesized variable declarators do) \u2014 removes the spurious indirect-invoke facts that leaked phantom CALLS past two-phase suppression. Prior f29bc3f7b1622954d6f6b7647bc9cf6c7a2629ffcc0fe00ac7918e4925876b65 -> a70625bb0a9ef74e760d9d79cc5557485d0f0d3fb935e8a22a0c9556c65b5bb1; scaling re-verified within budget.", "_rebaselined_receiver_chain_2747": "#2747: additionally adds the `cpp-receiver-chain-arrow` fixture, the behavioural proof for a `->` BASE receiver (`svc->getUser()->save()`) that the rollout fixed and that `cpp-chain-call/` could never catch because it uses the value `.` form. Prior a70625bb0a9ef74e760d9d79cc5557485d0f0d3fb935e8a22a0c9556c65b5bb1 -> 7e27aea46f3e17f33c41babbe0ddd982d1ab5920f143864763e0a1c6aef882a5.", - "_rebaselined_2766_receiver_chain_wire_v2": "#2766: receiver-chain wire format v1 -> v2 (name-free `await` / `index` step kinds). The VERSION prefix is part of every emitted `@reference.receiver-chain` capture, so every chain-minting language's capture text changed. WIRE-FORMAT CHANGE, NOT A CAPTURE-SET CHANGE: the same chains are minted for the same sites, spelled `2|…` instead of `1|…`. Exactly the 12 chain-minting languages drifted; c, cobol and dart did not, which is the check that this is the prefix and not a capture regression. Accompanied by SCHEMA_BUMP 34 -> 37 and INCREMENTAL_SCHEMA_VERSION 28 -> 31 so a stale index is rejected rather than replaying chains a v2 decoder refuses. Prior 7e27aea46f3e17f33c41babbe0ddd982d1ab5920f143864763e0a1c6aef882a5 -> 856d02f3f9d22cb973877211100aee8e052d4bc545922f78704b1a21ce49ddcc." + "_rebaselined_2766_receiver_chain_wire_v2": "#2766: receiver-chain wire format v1 -> v2 (name-free `await` / `index` step kinds). The VERSION prefix is part of every emitted `@reference.receiver-chain` capture, so every chain-minting language's capture text changed. WIRE-FORMAT CHANGE, NOT A CAPTURE-SET CHANGE: the same chains are minted for the same sites, spelled `2|\u2026` instead of `1|\u2026`. Exactly the 12 chain-minting languages drifted; c, cobol and dart did not, which is the check that this is the prefix and not a capture regression. Accompanied by SCHEMA_BUMP 34 -> 37 and INCREMENTAL_SCHEMA_VERSION 28 -> 31 so a stale index is rejected rather than replaying chains a v2 decoder refuses. Prior 7e27aea46f3e17f33c41babbe0ddd982d1ab5920f143864763e0a1c6aef882a5 -> 856d02f3f9d22cb973877211100aee8e052d4bc545922f78704b1a21ce49ddcc.", + "capture_groups_small": 5021, + "capture_groups_large": 16021, + "capture_groups_fp": 4605, + "fixture_count": 279 }, "csharp": { "_rebaselined": "#1956 synth-widening: + csharp-qualified-base fixture; the synth now walks record_declaration + struct_declaration base_lists and handles alias_qualified_name (matching the #1940 legacy leg), so record/struct heritage now emits. csharp-record-base gains a record inherits capture. (record->record SAME-namespace EXTENDS is a separate registry resolution gap, tracked as follow-up.) Linear (~1.00). (Earlier #1956: heritage-bearing scale source.) | #942: scope-resolution-only cleanup reworded fixture comments; capture byte-positions shift, capture LOGIC unchanged. | #1924 F16: record primary-constructor base bindings now exclude constructor arguments; capture fingerprint changes, scaling remains linear. | #2036 review follow-up: csharp-record-base now exercises primary-constructor base dispatch end to end; +2 capture groups, scaling remains linear.", - "fingerprint": "476d98a7cc659951c315d63319c8077bbcf0e5f3ec12d32ed773992a1f3a2adc", + "fingerprint": "2930ef49fdce984a4c051409880bddfe8445e30e1c6bf802bd90a0a0f8f6b094", "scaling_budget": 1.5, "_rebaselined_callable_flow_2522_review": "PR #2522 review hardening: callable operands retain expression/qualified identity and formals retain signature metadata. Prior f31544530924748f9aa37d11cec570bc10c3ddf9d9b237e6df7a17623fd2bb3a -> 75cf380209fa7d1a8a3ec873be1a9424b4e5173be0b08234c2291e8521a9b3c1; scaling 1.061 < 1.5.", "_rebaselined_callable_flow_2522_followup": "PR #2522 follow-up: C# method-group/delegate callable flow facts with invocation-result suppression. Prior 2bb5bc8c19cb8eb08c9590545ad8a1968a7152951f7e12746e2d7901d542fed9 -> f31544530924748f9aa37d11cec570bc10c3ddf9d9b237e6df7a17623fd2bb3a; scaling 1.115 < 1.5.", "_note": "#2046: F35 qualified-constructor captures now emit @reference.qualified-name + a simple-name @reference.name on `new Ns.Foo()`/`new A.B.Foo()`; namespace_declaration/file_scoped_namespace_declaration now emit @declaration.namespace name captures (feeding the non-destructive namespacePrefix sidecar for `new B.Foo()` same-tail disambiguation). + csharp-interface-only-base and csharp-namespace-qualified-ctor fixtures. Pure capture-additive + fixture-corpus drift; scaling stays linear (~1.11).", "_rebaselined_2563_instance_ownership": "#2563: csharp-using-static adds same-file ownership, local-function, overload, partial-class, and cross-namespace same-name coverage. Prior 75cf380209fa7d1a8a3ec873be1a9424b4e5173be0b08234c2291e8521a9b3c1 -> e05dc27456bde8175948586c9e7689033a378fa40e9ca4ce78cce41fbea0f2f8; scaling 1.058 < 1.5.", - "_rebaselined_receiver_chain_2747": "#2747 receiver-chain rollout: call matches whose receiver is itself an expression now carry `@reference.receiver-chain`, a compact encoding of the receiver's structure, so resolution types it by folding instead of re-parsing receiver source text. Capture GROUP counts are unchanged — the tag is added to existing call matches, never a new match — so this is digest drift only. Prior 05a85bae70cf9c94f42459c843cfc36e3e81c872e5dcc7d77bc42fbc390f4bfe -> 8a282254b93b3ef2ff34c2fdba819ebc95c53c4fcb09942cbad99f96d3687855.", - "_rebaselined_2766_receiver_chain_wire_v2": "#2766: receiver-chain wire format v1 -> v2 (name-free `await` / `index` step kinds). The VERSION prefix is part of every emitted `@reference.receiver-chain` capture, so every chain-minting language's capture text changed. WIRE-FORMAT CHANGE, NOT A CAPTURE-SET CHANGE: the same chains are minted for the same sites, spelled `2|…` instead of `1|…`. Exactly the 12 chain-minting languages drifted; c, cobol and dart did not, which is the check that this is the prefix and not a capture regression. Accompanied by SCHEMA_BUMP 34 -> 37 and INCREMENTAL_SCHEMA_VERSION 28 -> 31 so a stale index is rejected rather than replaying chains a v2 decoder refuses. Prior 8a282254b93b3ef2ff34c2fdba819ebc95c53c4fcb09942cbad99f96d3687855 -> 476d98a7cc659951c315d63319c8077bbcf0e5f3ec12d32ed773992a1f3a2adc." + "_rebaselined_receiver_chain_2747": "#2747 receiver-chain rollout: call matches whose receiver is itself an expression now carry `@reference.receiver-chain`, a compact encoding of the receiver's structure, so resolution types it by folding instead of re-parsing receiver source text. Capture GROUP counts are unchanged \u2014 the tag is added to existing call matches, never a new match \u2014 so this is digest drift only. Prior 05a85bae70cf9c94f42459c843cfc36e3e81c872e5dcc7d77bc42fbc390f4bfe -> 8a282254b93b3ef2ff34c2fdba819ebc95c53c4fcb09942cbad99f96d3687855.", + "_rebaselined_2766_receiver_chain_wire_v2": "#2766: receiver-chain wire format v1 -> v2 (name-free `await` / `index` step kinds). The VERSION prefix is part of every emitted `@reference.receiver-chain` capture, so every chain-minting language's capture text changed. WIRE-FORMAT CHANGE, NOT A CAPTURE-SET CHANGE: the same chains are minted for the same sites, spelled `2|\u2026` instead of `1|\u2026`. Exactly the 12 chain-minting languages drifted; c, cobol and dart did not, which is the check that this is the prefix and not a capture regression. Accompanied by SCHEMA_BUMP 34 -> 37 and INCREMENTAL_SCHEMA_VERSION 28 -> 31 so a stale index is rejected rather than replaying chains a v2 decoder refuses. Prior 8a282254b93b3ef2ff34c2fdba819ebc95c53c4fcb09942cbad99f96d3687855 -> 476d98a7cc659951c315d63319c8077bbcf0e5f3ec12d32ed773992a1f3a2adc.", + "capture_groups_small": 4259, + "capture_groups_large": 13609, + "capture_groups_fp": 2657, + "fixture_count": 178 }, "rust": { - "fingerprint": "6174889b8c98e0af430fa54c268dc781989ca9a8172d690eebae37a95f77e809", + "fingerprint": "e61653008ff2de506cfd47f905fa9eb22d82fbbfe94d2a1d8190c358211b57b7", "scaling_budget": 1.5, - "_rebaselined_mod_node_identity_2745_review": "#2745 review: added rust-2742-mod-members, rust-2742-nested-mods and rust-2742-type-vs-module under lang-resolution for the container/owner-edge fix, nested inline modules, and the imported-type-vs-module precedence. emitRustScopeCaptures is unchanged — verified by removing ONLY those three fixture dirs and re-running, which reproduces the prior fingerprint exactly, so the shift is purely corpus growth (fixture_count 196 -> 202, capture_groups_fp 3432 -> 3556). Prior 90fda086a4e13aa069a5981f63ed58ab1c71f1ed3da5e1480a080e1992b0d3e5 -> 05acbaca48427e0d9e0793bcd0ce4057712d3716b5e7868189c12e05ef8dd300; scaling 1.022 local / 1.057 CI < 1.5. NOTE for the next fixture author: a new rust-* fixture drifts BOTH this bench baseline and the rust-captures-golden snapshot. Updating only the golden is how this reached CI red.", + "_rebaselined_generic_instantiation_2912": "#2912: RUST_SCOPE_QUERY tags trait-impl heritage with the instantiation the impl was written with (`impl Validator for V`), so interface dispatch can prune implementors of an instantiation the receiver cannot hold. Additive capture text on existing impl matches \u2014 the same matches are minted, carrying one more field \u2014 so this is digest drift, not a capture-set change: capture_groups_fp (3556) and fixture_count (202) are both unchanged, which is the check that no match appeared or vanished. Prior 116a971fee0004f340477aff69fa110a1d92bd8ba882d7c926483c6b1e8ca2b9 -> e61653008ff2de506cfd47f905fa9eb22d82fbbfe94d2a1d8190c358211b57b7; scaling 1.018 < 1.5. Only rust and dart move; the other 13 languages are byte-identical.", + "_rebaselined_mod_node_identity_2745_review": "#2745 review: added rust-2742-mod-members, rust-2742-nested-mods and rust-2742-type-vs-module under lang-resolution for the container/owner-edge fix, nested inline modules, and the imported-type-vs-module precedence. emitRustScopeCaptures is unchanged \u2014 verified by removing ONLY those three fixture dirs and re-running, which reproduces the prior fingerprint exactly, so the shift is purely corpus growth (fixture_count 196 -> 202, capture_groups_fp 3432 -> 3556). Prior 90fda086a4e13aa069a5981f63ed58ab1c71f1ed3da5e1480a080e1992b0d3e5 -> 05acbaca48427e0d9e0793bcd0ce4057712d3716b5e7868189c12e05ef8dd300; scaling 1.022 local / 1.057 CI < 1.5. NOTE for the next fixture author: a new rust-* fixture drifts BOTH this bench baseline and the rust-captures-golden snapshot. Updating only the golden is how this reached CI red.", "_rebaselined_dyn_trait_object_2604": "#2604: RUST_SCOPE_QUERY now captures function_signature_item (abstract trait methods, no body) as a scope + declaration, so a &dyn Trait receiver can dispatch a CALLS edge to the trait's own method. Additive capture shift across every bench fixture with a required trait method. Prior df369c5a5f8de7753fc8bab8b4108ef5081750974ea5085ba9a867675ac9eb29 -> f7742f65f14d7d6590df7f16303fc3cc9dc0c233cd80bf90c98b084933cd3846; scaling 1.033 < 1.5.", "_rebaselined_callable_flow_2522_review": "PR #2522 review hardening: callable operands retain expression/qualified identity and formals retain signature metadata. Prior 65e5bca66bb1ca117949409e8fb5c80ee69d6f1b5318908eaaecf08da0482e5c -> df369c5a5f8de7753fc8bab8b4108ef5081750974ea5085ba9a867675ac9eb29; scaling 1.065 < 1.5.", "_rebaselined_callable_flow_2522_followup": "PR #2522 follow-up: Rust fn-value callable flow facts with invocation/constructor-result suppression. Prior ac610bbe97666bf285923479dd7b43a2fe4c5354aae8df1bcbafdc04fb220f82 -> 65e5bca66bb1ca117949409e8fb5c80ee69d6f1b5318908eaaecf08da0482e5c; scaling 1.024 < 1.5.", - "_rebaselined": "#1956 tri-review U1: rust-qualified-trait fixture (scoped + generic-of-scoped impl trait paths); bareTypeIdentifier now resolves scoped_type_identifier bases by their name: tail (additive, no existing-fixture drift); linear (~1.04). #1975: + rust-scoped-impl fixture (impl a::Inner / b::Inner inherent scoped impls) — legacy @definition.impl scoped arm + findEnclosingClassInfo inherent-impl scoped target; rust scope-extractor captures byte-identical. | #942: scope-resolution-only cleanup reworded fixture comments; capture byte-positions shift, capture LOGIC unchanged.", - "_note": "PR #1934: F66/F68 let-binding pattern narrowing; F71 union (Struct-labeled, now materialized via legacy @definition.struct + resolvable); F72 macro FULLY WIRED — @declaration.macro/@reference.macro + MacroRegistry → USES edges to Macro nodes (never a same-named fn). + rust-macro / rust-union fixtures and merged with origin/main #1975 rust-scoped-impl; fingerprint re-baselined (scaling ~0.99, fixture_count 126). #1992: + rust-nested-tail-collision-generic and rust-generic-impl-same-method-name (F3) fixtures — pure fixture-corpus drift, no scope-extractor change; fixture_count 127->129, fingerprint 56ffc1c0->b00aea0f.", + "_rebaselined": "#1956 tri-review U1: rust-qualified-trait fixture (scoped + generic-of-scoped impl trait paths); bareTypeIdentifier now resolves scoped_type_identifier bases by their name: tail (additive, no existing-fixture drift); linear (~1.04). #1975: + rust-scoped-impl fixture (impl a::Inner / b::Inner inherent scoped impls) \u2014 legacy @definition.impl scoped arm + findEnclosingClassInfo inherent-impl scoped target; rust scope-extractor captures byte-identical. | #942: scope-resolution-only cleanup reworded fixture comments; capture byte-positions shift, capture LOGIC unchanged.", + "_note": "PR #1934: F66/F68 let-binding pattern narrowing; F71 union (Struct-labeled, now materialized via legacy @definition.struct + resolvable); F72 macro FULLY WIRED \u2014 @declaration.macro/@reference.macro + MacroRegistry \u2192 USES edges to Macro nodes (never a same-named fn). + rust-macro / rust-union fixtures and merged with origin/main #1975 rust-scoped-impl; fingerprint re-baselined (scaling ~0.99, fixture_count 126). #1992: + rust-nested-tail-collision-generic and rust-generic-impl-same-method-name (F3) fixtures \u2014 pure fixture-corpus drift, no scope-extractor change; fixture_count 127->129, fingerprint 56ffc1c0->b00aea0f.", "_rebaselined_import_disambiguation_2514": "#2514: added rust-import-* and rust-dup-* fixtures under lang-resolution for the range-binding ambiguity latch + import-disambiguated resolution (for-loops / struct destructuring across explicit/aliased/glob use imports). emitRustScopeCaptures is unchanged; the corpus fingerprint shifts purely because the fixture set grew (130 -> 174). Prior f7742f65f14d7d6590df7f16303fc3cc9dc0c233cd80bf90c98b084933cd3846 -> 655aed01cf1b6b84fa0c64d48dfb2526ecb67f47d90f0a91edabacd269a212db; scaling 1.06 < 1.5.", - "_rebaselined_self_type_binding_2714": "#2714: a Rust `Self` type binding now records the enclosing impl's type instead of the literal 'Self'. `let fresh = Self { .. }` inside `impl User` binds `fresh: User`; recorded verbatim it bound `fresh: Self`, which resolves to nothing. The type-env channel already substituted this (type-extractors/rust.ts findEnclosingImplType); the scope-resolution channel did not, so the two disagreed. The gap was invisible while lookupCore Step 1 still walked the lexical chain for NAMED receivers — the impl scope binds the method by name, so fresh.validate() resolved by accident — and became a lost CALLS edge when #2714 stopped that walk. Only the rust fingerprint moves; the other 14 languages are byte-identical.", + "_rebaselined_self_type_binding_2714": "#2714: a Rust `Self` type binding now records the enclosing impl's type instead of the literal 'Self'. `let fresh = Self { .. }` inside `impl User` binds `fresh: User`; recorded verbatim it bound `fresh: Self`, which resolves to nothing. The type-env channel already substituted this (type-extractors/rust.ts findEnclosingImplType); the scope-resolution channel did not, so the two disagreed. The gap was invisible while lookupCore Step 1 still walked the lexical chain for NAMED receivers \u2014 the impl scope binds the method by name, so fresh.validate() resolved by accident \u2014 and became a lost CALLS edge when #2714 stopped that walk. Only the rust fingerprint moves; the other 14 languages are byte-identical.", "_rebaselined_module_tree_2730": "#2730 + #2741 review: RUST_SCOPE_QUERY captures mod_item as @declaration.namespace (a Rust module is an item, mirroring the C++ namespace_definition capture) and tags scoped call sites with @reference.qualified-name so the written path survives to resolution. Both are additive captures: every bench fixture holding a mod block or a Foo::bar() call gains groups, and the corpus also grew by the rust-2730-* fixtures added for the fix and its review (workspace-crates, type-qualified, gaps, samename-wrapper, crate-layout). Prior 7f1240b38457468f06b7931e0c2c578f218f922774d0dc7e2ee6ef3b08d4d689 -> 90fda086a4e13aa069a5981f63ed58ab1c71f1ed3da5e1480a080e1992b0d3e5; scaling 1.061 < 1.5; fixture_count 196. Only the rust fingerprint moves; the other 14 languages are byte-identical. The earlier revision of this note cited 655aed01... as the prior value, which was two rebaselines stale (it predates #2604 and #2714); the CI gate compares live fingerprints, not this prose, so nothing caught it.", - "_rebaselined_receiver_chain_2747": "#2747 receiver-chain rollout: call matches whose receiver is itself an expression now carry `@reference.receiver-chain`, a compact encoding of the receiver's structure, so resolution types it by folding instead of re-parsing receiver source text. Capture GROUP counts are unchanged — the tag is added to existing call matches, never a new match — so this is digest drift only. Prior 05acbaca48427e0d9e0793bcd0ce4057712d3716b5e7868189c12e05ef8dd300 -> 83812d82f0e2c3eb552f3246381ca3dd5ccd6783d63aba3325f1343e7772280c.", - "_rebaselined_2766_receiver_chain_wire_v2": "#2766: receiver-chain wire format v1 -> v2 (name-free `await` / `index` step kinds). The VERSION prefix is part of every emitted `@reference.receiver-chain` capture, so every chain-minting language's capture text changed. WIRE-FORMAT CHANGE, NOT A CAPTURE-SET CHANGE: the same chains are minted for the same sites, spelled `2|…` instead of `1|…`. Exactly the 12 chain-minting languages drifted; c, cobol and dart did not, which is the check that this is the prefix and not a capture regression. Accompanied by SCHEMA_BUMP 34 -> 37 and INCREMENTAL_SCHEMA_VERSION 28 -> 31 so a stale index is rejected rather than replaying chains a v2 decoder refuses. Prior 83812d82f0e2c3eb552f3246381ca3dd5ccd6783d63aba3325f1343e7772280c -> 6174889b8c98e0af430fa54c268dc781989ca9a8172d690eebae37a95f77e809." + "_rebaselined_receiver_chain_2747": "#2747 receiver-chain rollout: call matches whose receiver is itself an expression now carry `@reference.receiver-chain`, a compact encoding of the receiver's structure, so resolution types it by folding instead of re-parsing receiver source text. Capture GROUP counts are unchanged \u2014 the tag is added to existing call matches, never a new match \u2014 so this is digest drift only. Prior 05acbaca48427e0d9e0793bcd0ce4057712d3716b5e7868189c12e05ef8dd300 -> 83812d82f0e2c3eb552f3246381ca3dd5ccd6783d63aba3325f1343e7772280c.", + "_rebaselined_2766_receiver_chain_wire_v2": "#2766: receiver-chain wire format v1 -> v2 (name-free `await` / `index` step kinds). The VERSION prefix is part of every emitted `@reference.receiver-chain` capture, so every chain-minting language's capture text changed. WIRE-FORMAT CHANGE, NOT A CAPTURE-SET CHANGE: the same chains are minted for the same sites, spelled `2|\u2026` instead of `1|\u2026`. Exactly the 12 chain-minting languages drifted; c, cobol and dart did not, which is the check that this is the prefix and not a capture regression. Accompanied by SCHEMA_BUMP 34 -> 37 and INCREMENTAL_SCHEMA_VERSION 28 -> 31 so a stale index is rejected rather than replaying chains a v2 decoder refuses. Prior 83812d82f0e2c3eb552f3246381ca3dd5ccd6783d63aba3325f1343e7772280c -> 6174889b8c98e0af430fa54c268dc781989ca9a8172d690eebae37a95f77e809.", + "capture_groups_small": 5507, + "capture_groups_large": 17607, + "capture_groups_fp": 3556, + "fixture_count": 202 }, "php": { "fingerprint": "b213a872342da2d866b04681dede988770e4d3dfdc0d6e9f62212ec5b59cdc2c", @@ -81,9 +97,9 @@ "_rebaselined_callable_flow_2522_review": "PR #2522 review hardening: callable operands retain expression/qualified identity and formals retain signature metadata. Prior df7b1565f9115d66b1ae32e4a408d651afb2521b14e5ca615f3be426c29af618 -> 4a688fa5a7016546f7f3c6d44de023608ae80c5b0e3670c16f6e61b3632608fd; scaling 1.078 < 1.5.", "_rebaselined_callable_flow_2522_followup": "PR #2522 follow-up: PHP first-class callable and variable-invocation flow facts with invocation-result suppression. Prior 31c9e3f3cb7094a2bf9021cf9db859036e002f8b44605cd993b470fc600e97cb -> df7b1565f9115d66b1ae32e4a408d651afb2521b14e5ca615f3be426c29af618; scaling 1.074 < 1.5.", "_rebaselined": "#1956: heritage-bearing scale source (class extends Base + use trait); both forms gated at scale; linear (~1.04). | #2481/#2482: PHP imports carry a symbol-kind capture so function/constant imports resolve by declaring file; capture shape changes, scaling remains linear (~1.04).", - "_note": "PR #1931: F53 import multi-clause, F54 enum_case, F55 anonymous_class — fixture count 138→140, fingerprint drift expected.", - "_rebaselined_receiver_chain_2747": "#2747 receiver-chain rollout: call matches whose receiver is itself an expression now carry `@reference.receiver-chain`, a compact encoding of the receiver's structure, so resolution types it by folding instead of re-parsing receiver source text. Capture GROUP counts are unchanged — the tag is added to existing call matches, never a new match — so this is digest drift only. Prior 4a688fa5a7016546f7f3c6d44de023608ae80c5b0e3670c16f6e61b3632608fd -> 3745662053c76b6ae0a84a29aad319626ed5ccb88f7b9376c2680d3dc6502e28.", - "_rebaselined_2766_receiver_chain_wire_v2": "#2766: receiver-chain wire format v1 -> v2 (name-free `await` / `index` step kinds). The VERSION prefix is part of every emitted `@reference.receiver-chain` capture, so every chain-minting language's capture text changed. WIRE-FORMAT CHANGE, NOT A CAPTURE-SET CHANGE: the same chains are minted for the same sites, spelled `2|…` instead of `1|…`. Exactly the 12 chain-minting languages drifted; c, cobol and dart did not, which is the check that this is the prefix and not a capture regression. Accompanied by SCHEMA_BUMP 34 -> 37 and INCREMENTAL_SCHEMA_VERSION 28 -> 31 so a stale index is rejected rather than replaying chains a v2 decoder refuses. Prior 3745662053c76b6ae0a84a29aad319626ed5ccb88f7b9376c2680d3dc6502e28 -> b213a872342da2d866b04681dede988770e4d3dfdc0d6e9f62212ec5b59cdc2c." + "_note": "PR #1931: F53 import multi-clause, F54 enum_case, F55 anonymous_class \u2014 fixture count 138\u2192140, fingerprint drift expected.", + "_rebaselined_receiver_chain_2747": "#2747 receiver-chain rollout: call matches whose receiver is itself an expression now carry `@reference.receiver-chain`, a compact encoding of the receiver's structure, so resolution types it by folding instead of re-parsing receiver source text. Capture GROUP counts are unchanged \u2014 the tag is added to existing call matches, never a new match \u2014 so this is digest drift only. Prior 4a688fa5a7016546f7f3c6d44de023608ae80c5b0e3670c16f6e61b3632608fd -> 3745662053c76b6ae0a84a29aad319626ed5ccb88f7b9376c2680d3dc6502e28.", + "_rebaselined_2766_receiver_chain_wire_v2": "#2766: receiver-chain wire format v1 -> v2 (name-free `await` / `index` step kinds). The VERSION prefix is part of every emitted `@reference.receiver-chain` capture, so every chain-minting language's capture text changed. WIRE-FORMAT CHANGE, NOT A CAPTURE-SET CHANGE: the same chains are minted for the same sites, spelled `2|\u2026` instead of `1|\u2026`. Exactly the 12 chain-minting languages drifted; c, cobol and dart did not, which is the check that this is the prefix and not a capture regression. Accompanied by SCHEMA_BUMP 34 -> 37 and INCREMENTAL_SCHEMA_VERSION 28 -> 31 so a stale index is rejected rather than replaying chains a v2 decoder refuses. Prior 3745662053c76b6ae0a84a29aad319626ed5ccb88f7b9376c2680d3dc6502e28 -> b213a872342da2d866b04681dede988770e4d3dfdc0d6e9f62212ec5b59cdc2c." }, "ruby": { "fingerprint": "1c8c9c4b54036fa24c2a81e39ea530e938645c856d369075e5f437da78218c57", @@ -91,10 +107,10 @@ "_rebaselined_callable_flow_2522_review": "PR #2522 review hardening: callable operands retain expression/qualified identity and formals retain signature metadata. Prior cff273ae6cb7232c977d9241581834a2a2fa8bcf6369f7bd8f2471cd4419a6ef -> bf50ec6a53c8c91680dc6feac63a8956e78b1059249232dc25a0cfed25f31236; scaling 1.103 < 1.5.", "_rebaselined_callable_flow_2522_followup": "PR #2522 follow-up: Ruby Method/Proc callable flow facts with invocation/constructor-result suppression. Prior b5ea93bb3d0469c3821a8c70f5d5991c6f326e41097c119ad691154301dcc753 -> cff273ae6cb7232c977d9241581834a2a2fa8bcf6369f7bd8f2471cd4419a6ef; scaling 1.086 < 1.5.", "_rebaselined": "#1956 synth-widening: + ruby-qualified-base fixture; synth now reduces a scope_resolution superclass (class C < Mod::Super) to its trailing constant (matching the #1940 legacy leg), at parity. Linear (~1.03). (Earlier #1956: heritage-bearing scale source.) | #942: scope-resolution-only cleanup reworded fixture comments; capture byte-positions shift, capture LOGIC unchanged.", - "_note": "F62: + scope_resolution class/module declaration captures — fixture count 78→81, fingerprint drift expected. #1975: + ruby-tail-collision fixture (Foo::Bar vs Baz::Bar stay distinct nodes) — pure fixture-corpus drift, scope-extractor captures unchanged; 81→82. #1991: + ruby-nested-mixin-tail-collision fixture (85→86). Recomputed on the #942 merge (fixture-comment rewording shifts capture byte-positions, capture LOGIC unchanged): bf6b13a -> b5ea93bb.", + "_note": "F62: + scope_resolution class/module declaration captures \u2014 fixture count 78\u219281, fingerprint drift expected. #1975: + ruby-tail-collision fixture (Foo::Bar vs Baz::Bar stay distinct nodes) \u2014 pure fixture-corpus drift, scope-extractor captures unchanged; 81\u219282. #1991: + ruby-nested-mixin-tail-collision fixture (85\u219286). Recomputed on the #942 merge (fixture-comment rewording shifts capture byte-positions, capture LOGIC unchanged): bf6b13a -> b5ea93bb.", "_rebaselined_2522_review_fixes": "PR #2522 review fixes: bare identifiers are calls, not callable references (bareNamesAreCalls). Prior bf50ec6a53c8c91680dc6feac63a8956e78b1059249232dc25a0cfed25f31236 -> 070e4e11502442998ddf4048c2981cf1b2b735a87362ff854c5d14d71f98f4e2; scaling ratio re-verified within budget.", - "_rebaselined_receiver_chain_2747": "#2747 receiver-chain rollout: call matches whose receiver is itself an expression now carry `@reference.receiver-chain`, a compact encoding of the receiver's structure, so resolution types it by folding instead of re-parsing receiver source text. Capture GROUP counts are unchanged — the tag is added to existing call matches, never a new match — so this is digest drift only. Prior fea3edf82f521995147874b7f6c5f9e2eb88efdebf6365668f3260e913f0b558 -> fc81941b0a921074fa80dc448284de9a23bd07358ddc84d4894797cc08c3fe83.", - "_rebaselined_2766_receiver_chain_wire_v2": "#2766: receiver-chain wire format v1 -> v2 (name-free `await` / `index` step kinds). The VERSION prefix is part of every emitted `@reference.receiver-chain` capture, so every chain-minting language's capture text changed. WIRE-FORMAT CHANGE, NOT A CAPTURE-SET CHANGE: the same chains are minted for the same sites, spelled `2|…` instead of `1|…`. Exactly the 12 chain-minting languages drifted; c, cobol and dart did not, which is the check that this is the prefix and not a capture regression. Accompanied by SCHEMA_BUMP 34 -> 37 and INCREMENTAL_SCHEMA_VERSION 28 -> 31 so a stale index is rejected rather than replaying chains a v2 decoder refuses. Prior fc81941b0a921074fa80dc448284de9a23bd07358ddc84d4894797cc08c3fe83 -> 1c8c9c4b54036fa24c2a81e39ea530e938645c856d369075e5f437da78218c57." + "_rebaselined_receiver_chain_2747": "#2747 receiver-chain rollout: call matches whose receiver is itself an expression now carry `@reference.receiver-chain`, a compact encoding of the receiver's structure, so resolution types it by folding instead of re-parsing receiver source text. Capture GROUP counts are unchanged \u2014 the tag is added to existing call matches, never a new match \u2014 so this is digest drift only. Prior fea3edf82f521995147874b7f6c5f9e2eb88efdebf6365668f3260e913f0b558 -> fc81941b0a921074fa80dc448284de9a23bd07358ddc84d4894797cc08c3fe83.", + "_rebaselined_2766_receiver_chain_wire_v2": "#2766: receiver-chain wire format v1 -> v2 (name-free `await` / `index` step kinds). The VERSION prefix is part of every emitted `@reference.receiver-chain` capture, so every chain-minting language's capture text changed. WIRE-FORMAT CHANGE, NOT A CAPTURE-SET CHANGE: the same chains are minted for the same sites, spelled `2|\u2026` instead of `1|\u2026`. Exactly the 12 chain-minting languages drifted; c, cobol and dart did not, which is the check that this is the prefix and not a capture regression. Accompanied by SCHEMA_BUMP 34 -> 37 and INCREMENTAL_SCHEMA_VERSION 28 -> 31 so a stale index is rejected rather than replaying chains a v2 decoder refuses. Prior fc81941b0a921074fa80dc448284de9a23bd07358ddc84d4894797cc08c3fe83 -> 1c8c9c4b54036fa24c2a81e39ea530e938645c856d369075e5f437da78218c57." }, "swift": { "fingerprint": "adef9284feaecd39cb490aebce83876e15b9150c7a04b00a396feb78b7e1e0a9", @@ -103,13 +119,14 @@ "_rebaselined_callable_flow_2522_followup": "PR #2522 follow-up: Swift function-value callable flow facts with invocation-result suppression. Prior 180ac68e780bdf6f9089d53f51cbb9a66aed3e7774631cc3fcbaae5020213998 -> 5f923c6604d825d12b249f31c155b0f4d13a8379d532e5dde64a0f9b15cf4725; scaling 1.043 < 1.5.", "_rebaselined": "#1919 open-language coverage: new lang-resolution fixtures + intended capture additions (F5/F9 c-cpp, F26/F28/F29 dart, F47/F48/F49/F51/F52 kotlin, F75/F79 swift). Fingerprint-only drift; scaling_ratio ~1.0 (linear, no perf regression).", "_rebaselined_2522_review_fixes": "PR #2522 review fixes: assignment target:/result: fields join the shared fallback. Prior 7687ee2466e16020a12440a03fbda53e63aa05f94b4481f6133c09867a0d560d -> 115c5da807e36bb12fdeba28e44f2b6484ef322ff26c19fa0f191febaf774248; scaling ratio re-verified within budget.", - "_rebaselined_receiver_chain_2747": "#2747 receiver-chain rollout: call matches whose receiver is itself an expression now carry `@reference.receiver-chain`, a compact encoding of the receiver's structure, so resolution types it by folding instead of re-parsing receiver source text. Capture GROUP counts are unchanged — the tag is added to existing call matches, never a new match — so this is digest drift only. Prior 115c5da807e36bb12fdeba28e44f2b6484ef322ff26c19fa0f191febaf774248 -> a6fca5f052ae5ec635b56051e28a168c864a988b2221a3279ddd69807378ba0b.", - "_rebaselined_2766_receiver_chain_wire_v2": "#2766: receiver-chain wire format v1 -> v2 (name-free `await` / `index` step kinds). The VERSION prefix is part of every emitted `@reference.receiver-chain` capture, so every chain-minting language's capture text changed. WIRE-FORMAT CHANGE, NOT A CAPTURE-SET CHANGE: the same chains are minted for the same sites, spelled `2|…` instead of `1|…`. Exactly the 12 chain-minting languages drifted; c, cobol and dart did not, which is the check that this is the prefix and not a capture regression. Accompanied by SCHEMA_BUMP 34 -> 37 and INCREMENTAL_SCHEMA_VERSION 28 -> 31 so a stale index is rejected rather than replaying chains a v2 decoder refuses. Prior a6fca5f052ae5ec635b56051e28a168c864a988b2221a3279ddd69807378ba0b -> 2f04ae960123cf50138a49fabdc5a146c2963170cecf5755c552b23c9055a9e7.", + "_rebaselined_receiver_chain_2747": "#2747 receiver-chain rollout: call matches whose receiver is itself an expression now carry `@reference.receiver-chain`, a compact encoding of the receiver's structure, so resolution types it by folding instead of re-parsing receiver source text. Capture GROUP counts are unchanged \u2014 the tag is added to existing call matches, never a new match \u2014 so this is digest drift only. Prior 115c5da807e36bb12fdeba28e44f2b6484ef322ff26c19fa0f191febaf774248 -> a6fca5f052ae5ec635b56051e28a168c864a988b2221a3279ddd69807378ba0b.", + "_rebaselined_2766_receiver_chain_wire_v2": "#2766: receiver-chain wire format v1 -> v2 (name-free `await` / `index` step kinds). The VERSION prefix is part of every emitted `@reference.receiver-chain` capture, so every chain-minting language's capture text changed. WIRE-FORMAT CHANGE, NOT A CAPTURE-SET CHANGE: the same chains are minted for the same sites, spelled `2|\u2026` instead of `1|\u2026`. Exactly the 12 chain-minting languages drifted; c, cobol and dart did not, which is the check that this is the prefix and not a capture regression. Accompanied by SCHEMA_BUMP 34 -> 37 and INCREMENTAL_SCHEMA_VERSION 28 -> 31 so a stale index is rejected rather than replaying chains a v2 decoder refuses. Prior a6fca5f052ae5ec635b56051e28a168c864a988b2221a3279ddd69807378ba0b -> 2f04ae960123cf50138a49fabdc5a146c2963170cecf5755c552b23c9055a9e7.", "_rebaselined_inferred_field_receiver_2807": "#2807: optional property annotations (`var a: Outer?`) now emit a type binding. The prior pattern required the `user_type` to be a DIRECT child of the annotation, so an `optional_type` wrapper meant an optional field was never typed at all and its receiver could not resolve. ADDS @type-binding.annotation captures on the optional form only; no capture is removed. Prior 2f04ae960123cf50138a49fabdc5a146c2963170cecf5755c552b23c9055a9e7 -> adef9284feaecd39cb490aebce83876e15b9150c7a04b00a396feb78b7e1e0a9; scaling 1.023 < 1.5." }, "dart": { - "fingerprint": "ba93c90dcd341259e8e088816bc8c76ad27882419f665e35c056dc22fa54cf73", + "fingerprint": "3a8ddabbeb1cba47a4757451d4f79d726ca230fd15e860772b11526fbb1c6687", "scaling_budget": 1.5, + "_rebaselined_generic_instantiation_2912": "#2912: the Dart heritage marker carries a fourth field \u2014 the type arguments the clause was written with (`implements Validator`) \u2014 so interface dispatch can prune implementors of a mismatched instantiation. Additive marker text on existing heritage matches rather than a new match, so this is digest drift only; a marker from a pre-#2912 cache simply has no fourth field and reads as unknown. Prior ba93c90dcd341259e8e088816bc8c76ad27882419f665e35c056dc22fa54cf73 -> 3a8ddabbeb1cba47a4757451d4f79d726ca230fd15e860772b11526fbb1c6687; scaling 1.027 < 1.5.", "_rebaselined_2538": "#2538: Dart extension type headers are preprocessed into normal extension declarations before scope capture, so extension type symbols and their methods are now emitted. Intentional Dart-only capture fingerprint drift; CI measured scaling 1.042 < 1.5.", "_rebaselined_2538_implements": "#2538 tri-review follow-up: Dart extension type implements clauses now emit heritage markers and fixture coverage asserts IMPLEMENTS edges, including multi-arg generic interfaces. Prior committed baseline 66a46d5ff09f3d11b2771db0f48596fe7057e95c5bc8f56241fdb911137298c3 -> ba93c90dcd341259e8e088816bc8c76ad27882419f665e35c056dc22fa54cf73; scaling 0.945 < 1.5.", "_rebaselined_callable_flow_2522_review": "PR #2522 review hardening: callable operands retain expression/qualified identity and formals retain signature metadata. Prior 29ce2bfe70b246b1c9d5e99c0ec11e850c22e9672737592207242b7f4cc824b8 -> 66a46d5ff09f3d11b2771db0f48596fe7057e95c5bc8f56241fdb911137298c3; scaling 1.054 < 1.5.", @@ -118,11 +135,14 @@ "_rebaselined": "#1919 review CF3 fix: extended kotlin-local-property-owner (init/accessor destructuring) + new dart-accessor-owner fixture (getter/setter ownership). Fingerprint-only corpus drift; scaling ~1.0." }, "java": { - "fingerprint": "a9943355e945e03ddb87c800f4cc1f62b3d04feefb3ec64c258d8e0bb3b3fcd9", + "fingerprint": "2bf47cc19b595a9889ac21ec0154c6ce6786271d68551f21d1bc14c626bcd4ff", "scaling_budget": 1.5, + "_rebaselined_2935_synthetic_declarations": "PR #2935 review follow-up: synthesized Java anonymous classes and bodied enum constants now carry the presence-only @declaration.is-synthetic sidecar used to preserve source-written dispatch targets at the fanout cap. DIGEST DRIFT ONLY, NOT A CAPTURE-SET CHANGE: the tag is attached to existing synthetic declaration matches; capture groups and fixture count remain 5755/18405, 3512, and 206. Prior 36d689c58526c4482fbd701d1d9ca156623a3970734ead145717858712271ab5 -> 2e2150b4f4d64519e3f4c6d7a2c12259178d3117872203c904fab8cba96a694a; CI scaling 0.971 < 1.5.", + "_rebaselined_2917_record_component_accessors": "#2917: every implicit Java record-component accessor now emits a component-bounded @scope.function plus @declaration.method/name/zero-arity/return-type metadata. The scope boundary prevents subsequent record-body references from being attributed to the accessor. Java was the only general language fingerprint to move; capture groups scale by exactly two per generated record component (small 5755 -> 6255, large 18405 -> 20005). Prior 36d689c58526c4482fbd701d1d9ca156623a3970734ead145717858712271ab5 -> 901a66c7dc0f071eeef9e4864b2519e5b58a1a141a1f9a7817ea42f7ff70eafb; scaling 0.961 < 1.5. Re-measured after merging origin/main, which carries #2935's is-synthetic sidecar on top of the same corpus: 2e2150b4f4d64519e3f4c6d7a2c12259178d3117872203c904fab8cba96a694a -> 79dafc369eaeb7183ee8cc1149b1a6c21ad672c7e5b806fe8b0060e5a952c79a; scaling 1.085 < 1.5, capture groups 6255/20005, capture_groups_fp 3560, fixture_count 206 (unchanged by the merge).", + "_rebaselined_2900_record_heritage": "#2900 review follow-up: the Java scale unit now includes a record implementing Marker, so the record-declaration @reference.inherits path is fingerprinted and exercised at scale. Prior b29e263524f55151dcb7cfc4c929d3d1d7bb360355cee4e832158f927857f663 -> 36d689c58526c4482fbd701d1d9ca156623a3970734ead145717858712271ab5; scaling 1.042 < 1.5.", "_rebaselined_callable_flow_2522_review": "PR #2522 review hardening: callable operands retain expression/qualified identity and formals retain signature metadata; same-name lexical regions use an O(ancestor-depth) ID-set lookup. Prior d5c59d7dc9e206637515d5aea1163f7c1cdd76410c38c5fe6143d13d19677d6a -> 004a3592998dca1193bd1429a8284513725de7764f2a3eceedaaa984cfd763b4; scaling 0.992 < 1.5.", "_rebaselined_callable_flow_2522_followup": "PR #2522 follow-up: Java method-reference/SAM callable flow facts with invocation-result suppression. Prior 062d754764aaa8a6772fb90875c710502a63e3e7a300e633942381ed914faada -> d5c59d7dc9e206637515d5aea1163f7c1cdd76410c38c5fe6143d13d19677d6a; scaling 1.074 < 1.5.", - "_rebaselined": "#2357 (supersedes #2353): + java-cast-receiver, java-this-field-chain, java-this-dispatch fixtures (cast-wrapped receivers, this.field chains incl. initializer contexts, bare-this dispatch pinning). Drift is purely fixture-additive: with the three new dirs parked, the fingerprint reproduces the prior baseline byte-identically — no emit/capture change. #1956 synth-widening: + java-iface-extends fixture; synthesizeJavaInheritanceReferences now ALSO walks interface_declaration extends_interfaces (interface IA extends IB, IC), matching the #1940 legacy leg. (Earlier U2+review: java-qualified-base fixture covers 2- AND 3-segment qualified bases guarding the legacy end-anchor; synth tail-resolves scoped bases.) Linear (~1.03). (Earliest: java added to bench, exposed+fixed the O(n^2) findNodeAtRange root-walk; 3.09 -> ~0.99.) | #942: scope-resolution-only cleanup reworded fixture comments; capture byte-positions shift, capture LOGIC unchanged.", + "_rebaselined": "#2357 (supersedes #2353): + java-cast-receiver, java-this-field-chain, java-this-dispatch fixtures (cast-wrapped receivers, this.field chains incl. initializer contexts, bare-this dispatch pinning). Drift is purely fixture-additive: with the three new dirs parked, the fingerprint reproduces the prior baseline byte-identically \u2014 no emit/capture change. #1956 synth-widening: + java-iface-extends fixture; synthesizeJavaInheritanceReferences now ALSO walks interface_declaration extends_interfaces (interface IA extends IB, IC), matching the #1940 legacy leg. (Earlier U2+review: java-qualified-base fixture covers 2- AND 3-segment qualified bases guarding the legacy end-anchor; synth tail-resolves scoped bases.) Linear (~1.03). (Earliest: java added to bench, exposed+fixed the O(n^2) findNodeAtRange root-walk; 3.09 -> ~0.99.) | #942: scope-resolution-only cleanup reworded fixture comments; capture byte-positions shift, capture LOGIC unchanged.", "_note": "#1928 / #2045: F35 adds qualified + qualified-generic constructor query captures (`new pkg.Foo()`, `new a.b.Foo()`, `new pkg.Box()`); F38 synthesizes `@reference.call.constructor` on `super(...)`/`this(...)` explicit_constructor_invocation nodes; F41 generic-aware stripQualifier in interpret (type-binding normalization). + java-qualified-constructor and java-explicit-constructor fixtures. Pure capture-additive + fixture-corpus drift; scaling stays linear (~1.06).", "_rebaselined_2522_review_fixes": "PR #2522 review fixes: get/test dropped from callableProtocolMethods. Prior 004a3592998dca1193bd1429a8284513725de7764f2a3eceedaaa984cfd763b4 -> f3b4f4b6610e07c3ac90deb1c53d3572b6ad55a36e5d7134984876d30031ff67; scaling ratio re-verified within budget.", "_rebaselined_2550_instance_model": "PR #2549 (#2550): anonymous class bodies emit synthesized @declaration.class/@declaration.name (Worker$N), an @reference.inherits to the constructed type, and receiver @type-binding.* captures; six new java-* fixtures joined the corpus. Prior f3b4f4b6610e07c3ac90deb1c53d3572b6ad55a36e5d7134984876d30031ff67 -> d79c3b92acfc866094981499b977388ca14f90839bca0c040342ab1cec00aa90; scaling 1.058 < 1.5.", @@ -130,34 +150,51 @@ "_rebaselined_2564_record_capture": "PR for #2564: JAVA_QUERIES gained a (record_declaration name: (identifier) @name) @definition.record capture, previously entirely missing (record_declaration had no structure-phase capture at all, unlike class/interface/enum) - a record's methods existed as ownerless Method nodes with no HAS_METHOD edge. Two new java-* fixtures (java-record-methods, java-new-expr-chain-call) joined the corpus. Prior 975b68aaac6d06094260fb0c67f9b1bc03692ba7220669d192aca9dccd5fc0ca -> 85fc7af9c3c1bceac76cb4f27214410b04967682a2eaa7e468e26efd1f4e2537; scaling 1.059 < 1.5.", "_rebaselined_2561_enum_constant_receiver": "PR for #2561: synthesizeJavaAnonymousClassDeclarations now emits a class-scope @type-binding.annotation/name/type per enum constant (constant simple name -> its E$N synthesized class when bodied, else the host enum) so E.CONST.method() resolves through the existing compound-receiver chain walk. Two drivers of the drift, both in the java-enum-constant-body fixture (this bench's corpus IS test/fixtures/lang-resolution): (1) one extra type-binding match per enum_constant from the capture change; (2) review follow-up added a body-less Plain.java enum + EnumConst.dispatchToConstant/dispatchInherited methods (bodied-override, inherited-via-MRO, and body-less dispatch call sites). The review's fail-safe hardening (bodied constant binds ONLY to E$N, never the host enum, when name synthesis fails on a malformed tree) is output-neutral on this well-formed corpus (verified: fingerprint identical with and without it). Prior 85fc7af9c3c1bceac76cb4f27214410b04967682a2eaa7e468e26efd1f4e2537 -> d04298a91beec76d0fa7099b3d71265723be60c1df688969aa954f135dd49686; scaling < 1.5.", "_rebaselined_2562_local_classes": "#2562: Java block-local classes, enums, records, and interfaces use source-type-relative JLS 13.1 Host$NLocal identities with javac-compatible per-(host, simple-name) numbering; anonymous numbering remains separate. Lexical aliases begin at each declaration and end with its immediate block. Expanded java-local-class-naming fixtures cover declaration order, disjoint blocks, initializers, lambdas, local type kinds, and recursive local/member/anonymous host chains. Prior d04298a91beec76d0fa7099b3d71265723be60c1df688969aa954f135dd49686 -> 6dd5913a58400a191ff54abf9b852b03d5add657d16c11e60a7c4608ba186197; scaling 1.204 < 1.5.", - "_rebaselined_receiver_chain_2747": "#2747 receiver-chain rollout: call matches whose receiver is itself an expression now carry `@reference.receiver-chain`, a compact encoding of the receiver's structure, so resolution types it by folding instead of re-parsing receiver source text. Capture GROUP counts are unchanged — the tag is added to existing call matches, never a new match — so this is digest drift only. Prior 6dd5913a58400a191ff54abf9b852b03d5add657d16c11e60a7c4608ba186197 -> 310adbc2e0827b5ac749acaa981cd12d256fc5b7cbc5592c5bee219e92abf9ee.", - "_rebaselined_2766_receiver_chain_wire_v2": "#2766: receiver-chain wire format v1 -> v2 (name-free `await` / `index` step kinds). The VERSION prefix is part of every emitted `@reference.receiver-chain` capture, so every chain-minting language's capture text changed. WIRE-FORMAT CHANGE, NOT A CAPTURE-SET CHANGE: the same chains are minted for the same sites, spelled `2|…` instead of `1|…`. Exactly the 12 chain-minting languages drifted; c, cobol and dart did not, which is the check that this is the prefix and not a capture regression. Accompanied by SCHEMA_BUMP 34 -> 37 and INCREMENTAL_SCHEMA_VERSION 28 -> 31 so a stale index is rejected rather than replaying chains a v2 decoder refuses. Prior 310adbc2e0827b5ac749acaa981cd12d256fc5b7cbc5592c5bee219e92abf9ee -> a9943355e945e03ddb87c800f4cc1f62b3d04feefb3ec64c258d8e0bb3b3fcd9." + "_rebaselined_receiver_chain_2747": "#2747 receiver-chain rollout: call matches whose receiver is itself an expression now carry `@reference.receiver-chain`, a compact encoding of the receiver's structure, so resolution types it by folding instead of re-parsing receiver source text. Capture GROUP counts are unchanged \u2014 the tag is added to existing call matches, never a new match \u2014 so this is digest drift only. Prior 6dd5913a58400a191ff54abf9b852b03d5add657d16c11e60a7c4608ba186197 -> 310adbc2e0827b5ac749acaa981cd12d256fc5b7cbc5592c5bee219e92abf9ee.", + "_rebaselined_2766_receiver_chain_wire_v2": "#2766: receiver-chain wire format v1 -> v2 (name-free `await` / `index` step kinds). The VERSION prefix is part of every emitted `@reference.receiver-chain` capture, so every chain-minting language's capture text changed. WIRE-FORMAT CHANGE, NOT A CAPTURE-SET CHANGE: the same chains are minted for the same sites, spelled `2|\u2026` instead of `1|\u2026`. Exactly the 12 chain-minting languages drifted; c, cobol and dart did not, which is the check that this is the prefix and not a capture regression. Accompanied by SCHEMA_BUMP 34 -> 37 and INCREMENTAL_SCHEMA_VERSION 28 -> 31 so a stale index is rejected rather than replaying chains a v2 decoder refuses. Prior 310adbc2e0827b5ac749acaa981cd12d256fc5b7cbc5592c5bee219e92abf9ee -> a9943355e945e03ddb87c800f4cc1f62b3d04feefb3ec64c258d8e0bb3b3fcd9.", + "capture_groups_small": 6255, + "capture_groups_large": 20005, + "capture_groups_fp": 3586, + "fixture_count": 209, + "_rebaselined_2910_declared_package_fixtures": "#2910 adds three Java resolver fixture files covering an external JDK lookalike, a path/package mismatch, and wildcard package membership. Fixture-corpus growth only: Java query rules and synthetic scaling sources are unchanged; capture_groups_small/large remain 6255/20005. capture_groups_fp 3560 -> 3586 and fixture_count 206 -> 209." }, "java-local-types": { - "fingerprint": "8c50bbc83dff4f7f5abd06078aa6abc6b64af05fddb17ee826b5f3df3d346633", + "fingerprint": "bdde823fa725e636e257940efb4c8655aa23124c1727cbaa8856d1ad8f71729e", "scaling_budget": 1.5, + "_rebaselined_2935_synthetic_declarations": "PR #2935 review follow-up: the local-type stress corpus includes synthesized anonymous declarations, which now carry the presence-only @declaration.is-synthetic sidecar. DIGEST DRIFT ONLY, NOT A CAPTURE-SET CHANGE. Prior 8c50bbc83dff4f7f5abd06078aa6abc6b64af05fddb17ee826b5f3df3d346633 -> 560734cd053fb4f4b23aa04bc7870c22089a8deedb0217fa9c1b4db689e02a97; CI scaling 1.002 < 1.5.", + "_rebaselined_2917_record_component_accessors": "#2917: the focused local-type fixture corpus contains local records, so their implicit component accessors add the same bounded scope/declaration captures as the general Java corpus. No local-type naming logic changed. Prior 8c50bbc83dff4f7f5abd06078aa6abc6b64af05fddb17ee826b5f3df3d346633 -> 3e22f368a4ee139be7cb91ff4fb77ddadf60c55efe8d66955ec81f366a46e460; scaling 1.032 < 1.5, capture_groups_fp 680. Re-measured on top of #2935's is-synthetic sidecar after merging origin/main: 560734cd053fb4f4b23aa04bc7870c22089a8deedb0217fa9c1b4db689e02a97 -> bdde823fa725e636e257940efb4c8655aa23124c1727cbaa8856d1ad8f71729e; scaling 0.997 < 1.5, capture_groups_fp 680.", "_added": "#2562 performance follow-up: co-scales same-host, same-name local classes and anonymous classes to gate JLS binary-name ordinal allocation. Precomputed per-sequence ordinals reduce the focused 100->800 workload from 176->6655ms to 141->752ms; normalized 250->800 scaling is 1.054.", - "_rebaselined_receiver_chain_2747": "#2747 receiver-chain rollout: call matches whose receiver is itself an expression now carry `@reference.receiver-chain`, a compact encoding of the receiver's structure, so resolution types it by folding instead of re-parsing receiver source text. Capture GROUP counts are unchanged — the tag is added to existing call matches, never a new match — so this is digest drift only. Prior a9ad88de21ca6747a923260dbdf677fb74a004abbf9d57781f745e3a9027530b -> 3ca67847ea2b9a71b0a41e09f943767e5a2d3a113d3e203499ee364e37f40236.", - "_rebaselined_2766_receiver_chain_wire_v2": "#2766: receiver-chain wire format v1 -> v2 (name-free `await` / `index` step kinds). The VERSION prefix is part of every emitted `@reference.receiver-chain` capture, so every chain-minting language's capture text changed. WIRE-FORMAT CHANGE, NOT A CAPTURE-SET CHANGE: the same chains are minted for the same sites, spelled `2|…` instead of `1|…`. Exactly the 12 chain-minting languages drifted; c, cobol and dart did not, which is the check that this is the prefix and not a capture regression. Accompanied by SCHEMA_BUMP 34 -> 37 and INCREMENTAL_SCHEMA_VERSION 28 -> 31 so a stale index is rejected rather than replaying chains a v2 decoder refuses. Prior 3ca67847ea2b9a71b0a41e09f943767e5a2d3a113d3e203499ee364e37f40236 -> 8c50bbc83dff4f7f5abd06078aa6abc6b64af05fddb17ee826b5f3df3d346633." + "_rebaselined_receiver_chain_2747": "#2747 receiver-chain rollout: call matches whose receiver is itself an expression now carry `@reference.receiver-chain`, a compact encoding of the receiver's structure, so resolution types it by folding instead of re-parsing receiver source text. Capture GROUP counts are unchanged \u2014 the tag is added to existing call matches, never a new match \u2014 so this is digest drift only. Prior a9ad88de21ca6747a923260dbdf677fb74a004abbf9d57781f745e3a9027530b -> 3ca67847ea2b9a71b0a41e09f943767e5a2d3a113d3e203499ee364e37f40236.", + "_rebaselined_2766_receiver_chain_wire_v2": "#2766: receiver-chain wire format v1 -> v2 (name-free `await` / `index` step kinds). The VERSION prefix is part of every emitted `@reference.receiver-chain` capture, so every chain-minting language's capture text changed. WIRE-FORMAT CHANGE, NOT A CAPTURE-SET CHANGE: the same chains are minted for the same sites, spelled `2|\u2026` instead of `1|\u2026`. Exactly the 12 chain-minting languages drifted; c, cobol and dart did not, which is the check that this is the prefix and not a capture regression. Accompanied by SCHEMA_BUMP 34 -> 37 and INCREMENTAL_SCHEMA_VERSION 28 -> 31 so a stale index is rejected rather than replaying chains a v2 decoder refuses. Prior 3ca67847ea2b9a71b0a41e09f943767e5a2d3a113d3e203499ee364e37f40236 -> 8c50bbc83dff4f7f5abd06078aa6abc6b64af05fddb17ee826b5f3df3d346633.", + "capture_groups_fp": 680 }, "typescript": { - "fingerprint": "7a960908031331360ce582f5b55b7681e1cd7f8a2eabfd73c00982cb17f2a949", + "fingerprint": "05d1dadd6c9ef35c74079fa50f341b1b36e4fb02c9a89dd1b59f32b7cfd5e633", "scaling_budget": 1.5, + "_rebaselined_2934_import_type_only": "#2934: `import-decomposer.ts` attaches a presence-only `@import.type-only` synthetic capture to specifiers `tsc` erases, so `check --cycles` can stop counting type-only edges as initialization cycles. DIGEST DRIFT ONLY, NOT A CAPTURE-SET CHANGE \u2014 the tag is added to import matches that already existed, never a new match, the same shape as the #2747 receiver-chain rebaseline. Every count is unchanged: capture_groups_fp 2414, fixture_count 155, capture_groups_small/large 4503/14403 (those measure the SYNTHETIC scaling source, which has no imports at all). The fingerprint moves because `canonicalizeMatch` in measure.mjs hashes every TAG on every match, synthetics included, so one extra presence-only tag on an existing match rewrites that match's canonical string. Attribution is exact, not inferred: neutralizing ONLY the `m['@import.type-only'] = \u2026` assignment in import-decomposer.ts and re-running returns the fingerprint to c2fbf8a89e5686dd\u2026 byte-for-byte, so nothing else in the TypeScript capture stream moved. All 14 other languages report ok. Scaling 0.997 < 1.5. NOTE ON THE CONTROL: javascript did not move (2026993b\u2026, 43 fixtures), but it is a WEAK control here \u2014 `import type` is TypeScript-only syntax, so a JS corpus cannot express the construct and could not have drifted either way. It evidences no collateral damage, not the correctness of the TS change; the exact-attribution check above is what does that. Prior c2fbf8a89e5686dd1ff3659b20d41d8b05ebcc9790356e3653ee0c8ca5d365c8 -> f719163eb03a447c9e40ca316a905dd76cee82192a75a403df478ebbdc13e98f.", "_rebaselined_callable_flow_2522_review": "PR #2522 review hardening: callable operands retain expression/qualified identity and formals retain signature metadata. Prior 27f937bfb47d4bded316ea3c785ff659c8cd88a5761d928f113477a08c802c78 -> e05446620c5b80b7aae291cfdf32f693580fada2ae687124769b04a0c03bfe63; scaling 0.983 < 1.5.", "_rebaselined_callable_flow_2522_followup": "PR #2522 follow-up: lexical callable bindings, direct-callee argument metadata, and invocation-result suppression. Prior db5933cc6760234ed7d495123410feba6de243646d583f20d43032b9459f81fd -> 27f937bfb47d4bded316ea3c785ff659c8cd88a5761d928f113477a08c802c78; scaling 0.975 < 1.5.", "_rebaselined_callable_flow": "Callable assignment/copy/formal/argument/invoke facts (also consumed by Vue script blocks). Prior 25de86fd3377132c4e35d3d98f4f94a58e0cfeb7c22948a8ea3be4e793be74fd -> db5933cc6760234ed7d495123410feba6de243646d583f20d43032b9459f81fd; measured scaling ratio 0.951 < 1.5.", - "_rebaselined": "#1962: F44 (class scope@), F85 (enum member declarations), F87 (optional_parameter type annotations) add new captures — fingerprint drift expected.", - "_note": "#1968: F44, F85, F87 — fingerprint drift expected.", + "_rebaselined": "#1962: F44 (class scope@), F85 (enum member declarations), F87 (optional_parameter type annotations) add new captures \u2014 fingerprint drift expected.", + "_note": "#1968: F44, F85, F87 \u2014 fingerprint drift expected.", "_rebaselined_2522": "#2522 intentional @reference.value-ref/property-key capture additions. GitHub Actions run 29553361660 job 87800394279: prior 3f44a4a6892698df2d145c8ff2812c3b318807648983c88aca28fbd694f172f9 -> 25de86fd3377132c4e35d3d98f4f94a58e0cfeb7c22948a8ea3be4e793be74fd; scaling ratio 0.987 < 1.5.", "_rebaselined_2550_instance_model": "PR #2549 (#2545/#2551): object literals emit @scope.object (was unscoped, then @scope.block during development). Prior e05446620c5b80b7aae291cfdf32f693580fada2ae687124769b04a0c03bfe63 -> 3280b13d3f9378ab23eee31c2edc779b5a9ae1e7bb510c23a24855b44406d2f4; scaling 0.981 < 1.5.", - "_rebaselined_receiver_owner_2701": "#2701: every non-arrow function form now carries a `@receiver-owner.this` marker on the same node as `@scope.function`, so a scope that BINDS its own `this` can stop the receiver walk (`Scope.ownsReceivers`). Verified before re-baselining by diffing the capture-name histogram over this same fixture corpus against 1d3088173f6f93827641b476d614d5d15cd4f3ea: the ONLY delta is @receiver-owner.this (typescript +143, javascript +32) — every other capture count is byte-identical, so no existing capture moved. Prior 3280b13d3f9378ab23eee31c2edc779b5a9ae1e7bb510c23a24855b44406d2f4 -> 281e95484203b481094729ca249ef0423c41273eac35e424cdfd032a0dac7699.", - "_rebaselined_receiver_chain_2747": "#2747 receiver-chain rollout: call matches whose receiver is itself an expression now carry `@reference.receiver-chain`, a compact encoding of the receiver's structure, so resolution types it by folding instead of re-parsing receiver source text. Capture GROUP counts are unchanged — the tag is added to existing call matches, never a new match — so this is digest drift only. Prior cad25be9f81d6e021ebae8dcb166bc0af3a1ba8021f1506f6ca93fd4c2649000 -> 9e112415f1169f08576826c12ea1d137d1994e34b44c45986c9ffee83b8b4edc.", - "_rebaselined_2766_receiver_chain_wire_v2": "#2766: receiver-chain wire format v1 -> v2 (name-free `await` / `index` step kinds). The VERSION prefix is part of every emitted `@reference.receiver-chain` capture, so every chain-minting language's capture text changed. WIRE-FORMAT CHANGE, NOT A CAPTURE-SET CHANGE: the same chains are minted for the same sites, spelled `2|…` instead of `1|…`. Exactly the 12 chain-minting languages drifted; c, cobol and dart did not, which is the check that this is the prefix and not a capture regression. Accompanied by SCHEMA_BUMP 34 -> 37 and INCREMENTAL_SCHEMA_VERSION 28 -> 31 so a stale index is rejected rather than replaying chains a v2 decoder refuses. Prior 9e112415f1169f08576826c12ea1d137d1994e34b44c45986c9ffee83b8b4edc -> cdefe88d3c275f31953216c676ef32c7bf5727d56b9c3840b81ee6bf85749dff.", - "_rebaselined_inferred_field_receiver_2807": "#2807: inference-typed class fields now emit a type binding — `public_field_definition` with a `new_expression` value, and `this. = new ...` carrying a @type-binding.this-field marker. ADDS @type-binding.constructor captures only; no capture is removed, and the annotated form is unchanged because annotation outranks constructor-inferred in typeBindingStrength. Prior cdefe88d3c275f31953216c676ef32c7bf5727d56b9c3840b81ee6bf85749dff -> 248b56f0d7a0a6fc7a949dc7afb8611e135ed642bccc2631b96ebb9d686bb965; scaling 0.994 < 1.5.", - "_rebaselined_ts_heritage_2842": "#2842 review: TypeScript heritage capture now emits `@reference.inherits` for `interface_declaration` (bases on `extends_type_clause`) and `abstract_class_declaration` (bases on `class_heritage`), which were both silently skipped — so `interface B extends A` and `abstract class X implements I` produced no edge and every interface-dispatch walk dead-ended on a bodiless declaration. Verified before re-baselining by diffing the capture-name histogram over this same fixture corpus (145 files) with and without the change: the ONLY deltas are @reference.inherits 17 -> 20 (+3) and its paired @reference.name 245 -> 248 (+3), emitted together by emitTsInheritanceBase. Every other capture count is byte-identical, so no existing capture moved. The +3 is the three `interface X extends BasePayload` declarations in typescript-generic-calls/src/{auth,admin,guest}.ts. javascript is unchanged (no interfaces in the language). Prior 248b56f0d7a0a6fc7a949dc7afb8611e135ed642bccc2631b96ebb9d686bb965 -> 7a960908031331360ce582f5b55b7681e1cd7f8a2eabfd73c00982cb17f2a949." + "_rebaselined_receiver_owner_2701": "#2701: every non-arrow function form now carries a `@receiver-owner.this` marker on the same node as `@scope.function`, so a scope that BINDS its own `this` can stop the receiver walk (`Scope.ownsReceivers`). Verified before re-baselining by diffing the capture-name histogram over this same fixture corpus against 1d3088173f6f93827641b476d614d5d15cd4f3ea: the ONLY delta is @receiver-owner.this (typescript +143, javascript +32) \u2014 every other capture count is byte-identical, so no existing capture moved. Prior 3280b13d3f9378ab23eee31c2edc779b5a9ae1e7bb510c23a24855b44406d2f4 -> 281e95484203b481094729ca249ef0423c41273eac35e424cdfd032a0dac7699.", + "_rebaselined_receiver_chain_2747": "#2747 receiver-chain rollout: call matches whose receiver is itself an expression now carry `@reference.receiver-chain`, a compact encoding of the receiver's structure, so resolution types it by folding instead of re-parsing receiver source text. Capture GROUP counts are unchanged \u2014 the tag is added to existing call matches, never a new match \u2014 so this is digest drift only. Prior cad25be9f81d6e021ebae8dcb166bc0af3a1ba8021f1506f6ca93fd4c2649000 -> 9e112415f1169f08576826c12ea1d137d1994e34b44c45986c9ffee83b8b4edc.", + "_rebaselined_2766_receiver_chain_wire_v2": "#2766: receiver-chain wire format v1 -> v2 (name-free `await` / `index` step kinds). The VERSION prefix is part of every emitted `@reference.receiver-chain` capture, so every chain-minting language's capture text changed. WIRE-FORMAT CHANGE, NOT A CAPTURE-SET CHANGE: the same chains are minted for the same sites, spelled `2|\u2026` instead of `1|\u2026`. Exactly the 12 chain-minting languages drifted; c, cobol and dart did not, which is the check that this is the prefix and not a capture regression. Accompanied by SCHEMA_BUMP 34 -> 37 and INCREMENTAL_SCHEMA_VERSION 28 -> 31 so a stale index is rejected rather than replaying chains a v2 decoder refuses. Prior 9e112415f1169f08576826c12ea1d137d1994e34b44c45986c9ffee83b8b4edc -> cdefe88d3c275f31953216c676ef32c7bf5727d56b9c3840b81ee6bf85749dff.", + "_rebaselined_inferred_field_receiver_2807": "#2807: inference-typed class fields now emit a type binding \u2014 `public_field_definition` with a `new_expression` value, and `this. = new ...` carrying a @type-binding.this-field marker. ADDS @type-binding.constructor captures only; no capture is removed, and the annotated form is unchanged because annotation outranks constructor-inferred in typeBindingStrength. Prior cdefe88d3c275f31953216c676ef32c7bf5727d56b9c3840b81ee6bf85749dff -> 248b56f0d7a0a6fc7a949dc7afb8611e135ed642bccc2631b96ebb9d686bb965; scaling 0.994 < 1.5.", + "_rebaselined_ts_heritage_2842": "#2842 review: TypeScript heritage capture now emits `@reference.inherits` for `interface_declaration` (bases on `extends_type_clause`) and `abstract_class_declaration` (bases on `class_heritage`), which were both silently skipped \u2014 so `interface B extends A` and `abstract class X implements I` produced no edge and every interface-dispatch walk dead-ended on a bodiless declaration. Verified before re-baselining by diffing the capture-name histogram over this same fixture corpus (145 files) with and without the change: the ONLY deltas are @reference.inherits 17 -> 20 (+3) and its paired @reference.name 245 -> 248 (+3), emitted together by emitTsInheritanceBase. Every other capture count is byte-identical, so no existing capture moved. The +3 is the three `interface X extends BasePayload` declarations in typescript-generic-calls/src/{auth,admin,guest}.ts. javascript is unchanged (no interfaces in the language). Prior 248b56f0d7a0a6fc7a949dc7afb8611e135ed642bccc2631b96ebb9d686bb965 -> 7a960908031331360ce582f5b55b7681e1cd7f8a2eabfd73c00982cb17f2a949.", + "capture_groups_small": 4503, + "capture_groups_large": 14403, + "capture_groups_fp": 2465, + "fixture_count": 167, + "_rebaselined_blind_spots_2856": "#2856 blind-spots series: the JS/TS SCOPE queries gained capture rules, so fingerprint drift is expected and additive. Verified before re-baselining by diffing the capture-name sets in both scope queries against origin/main: TypeScript gained exactly @reference.read.identifier (A2 bare-identifier reads in value positions) and @reference.type (R2-2 type references, so a declared contract stops reporting incoming:{}); JavaScript gained exactly @reference.read.identifier, @reference.read.destructured (R2-1c) and @reference.write.property-key (R2-1b record-construction writes). NOTHING was removed on either side \u2014 the delta is a pure superset, which is the check that no existing capture moved. capture_groups_small/large are unchanged (4503/14403) because those measure the SYNTHETIC scaling source, which this branch does not touch; only the fixture-corpus count moves. capture_groups_fp 2097 -> 2338 and fixture_count 146 -> 151 from 21 new lang-resolution fixtures. Scaling stayed linear and inside budget: typescript 1.116 < 1.5, javascript 1.010 < 1.5. Prior typescript ed92588e0fc7b28b3a0174339ac378b4dd85965fe007db1208dea97a65ce0571 -> f66a3e6f1e096431e7046505129a627deaa00ca0de5bc846b080591b397248f7; prior javascript 806f70ad3cce5fc849f6d06a08ace8a95f92a1ea84a2418fddabb1eef5846594 -> 2026993b81b873839dd2ef8797d9c14d9c48516b2b57b05ac17d8d43f2f4eba3.", + "_rebaselined_type_parameter_shadowing_w2_8": "W2-8: `@declaration.type-parameters` is now captured on generic FUNCTIONS, generator functions and type ALIASES, not only on class/interface declarations. NO NEW CAPTURE NAME \u2014 verified by diffing the capture-name sets against the wave-1 branch, which returns empty; the tag already existed and simply fires on more declarations. That is the whole delta: capture_groups_fp 2338 -> 2371 (+33 occurrences of an existing tag) and fixture_count 151 -> 152 (one new fixture, typescript-type-parameters). capture_groups_small/large unchanged at 4503/14403, since those measure the synthetic scaling source this does not touch. Scaling 1.06 < 1.5. JavaScript is untouched \u2014 it has no type parameters \u2014 and its fingerprint does not move, which is the check that this is the TS declaration rules and not something broader. Prior f66a3e6f1e096431e7046505129a627deaa00ca0de5bc846b080591b397248f7 -> 62c7f1bfbe568eed927fb78f00061ed5e49d12511fd8260648b876df386f3b4c.", + "_rebaselined_2899_review_type_parameter_scope_fixtures": "PR #2899 review follow-up: FIXTURE-CORPUS GROWTH ONLY \u2014 no query rule changed and no capture name was added or removed. `typescript/query.ts` is byte-identical to the previous baseline; the type-parameter shadowing defect was fixed on the RESOLUTION side (`walkers.ts` gains a `declarationOpenedScope` gate so a declaration's `typeParameters` bind only inside the scope that declaration opened, and the `USES` guard moved from `graph-bridge/references-to-edges.ts` to `resolve-references.ts` where the spelled `site.name` is in hand). The fingerprint moves because measure.mjs fingerprints the whole `lang-resolution/typescript-*` fixture corpus and the regression tests add three files to `typescript-type-parameters/src/` (values.ts, aliased.ts, namespaced.ts) plus two scope-less generic aliases in shapes.ts. Per-file accounting sums exactly to the delta: shapes.ts 33->35 (+2), values.ts +11, aliased.ts +10, namespaced.ts +20 = +43. capture_groups_fp 2371 -> 2414; fixture_count 152 -> 155. capture_groups_small/large unchanged at 4503/14403 (they measure the SYNTHETIC scaling source, untouched). JAVASCRIPT IS THE CONTROL AND DID NOT MOVE (fingerprint 2026993b..., 43 fixtures) \u2014 which is the check that this is corpus growth and not a capture regression; all 14 other languages report `ok`. Scaling 0.976 < 1.5. Prior 62c7f1bfbe568eed927fb78f00061ed5e49d12511fd8260648b876df386f3b4c -> c2fbf8a89e5686dd1ff3659b20d41d8b05ebcc9790356e3653ee0c8ca5d365c8.", + "_rebaselined_2953_workspace_fixture": "#2953 adds test/fixtures/lang-resolution/typescript-pnpm-workspace-imports, a pnpm monorepo of 12 .ts files, and the TypeScript capture corpus is collected from test/fixtures. CORPUS GROWTH ONLY, NOT A CAPTURE CHANGE: fixture_count 155 -> 167 and capture_groups_fp 2414 -> 2465 are the 12 new files' own matches; capture_groups_small/large are unchanged at 4503/14403 because those measure the SYNTHETIC scaling source, which the fixture corpus does not feed. Attribution is exact rather than inferred: moving that one fixture directory aside and re-running returns typescript to f719163eb03a447c9e40ca316a905dd76cee82192a75a403df478ebbdc13e98f byte-for-byte with fixture_count back at 155, and [scope-capture --check] PASSES for all 15 languages - so nothing in the TypeScript capture stream moved. #2953 changes import RESOLUTION, which runs after capture and feeds no capture tag. Prior f719163eb03a447c9e40ca316a905dd76cee82192a75a403df478ebbdc13e98f -> 05d1dadd6c9ef35c74079fa50f341b1b36e4fb02c9a89dd1b59f32b7cfd5e633." }, "javascript": { - "fingerprint": "806f70ad3cce5fc849f6d06a08ace8a95f92a1ea84a2418fddabb1eef5846594", + "fingerprint": "2026993b81b873839dd2ef8797d9c14d9c48516b2b57b05ac17d8d43f2f4eba3", "scaling_budget": 1.5, "_rebaselined_callable_flow_2522_review": "PR #2522 review hardening: callable operands retain expression/qualified identity and formals retain signature metadata. Prior b59fe8135b6a31a12bc3f872b224054b16592588153ae3661d03958d787c76f3 -> 479927409bbdd9852a36172c8260aa56df260e99129a7a9c20a0d1903dd5538b; scaling 1.050 < 1.5.", "_rebaselined_callable_flow_2522_followup": "PR #2522 follow-up: lexical callable bindings, direct-callee argument metadata, and invocation-result suppression. Prior 917a9cd975ba035bdad71fdb70cd72eeddec58c25797e5a1addfa6172808a55c -> b59fe8135b6a31a12bc3f872b224054b16592588153ae3661d03958d787c76f3; scaling 1.093 < 1.5.", @@ -166,23 +203,28 @@ "_rebaselined": "#1956 synth-widening: + javascript-qualified-base fixture; synthesizeJsInheritanceReferences now handles a member_expression base (class S extends ns.Base -> Base), matching the #1940 legacy leg + the TS terminalTsTypeNameNode property_identifier case, at parity. Linear (~1.05). | #942: scope-resolution-only cleanup reworded fixture comments; capture byte-positions shift, capture LOGIC unchanged.", "_rebaselined_2522": "#2522 intentional @reference.value-ref/property-key capture additions. GitHub Actions run 29553361660 job 87800394279: prior d72f03c6c502235d2d4b74d66baa5c7d361f040d7a1b72e84acad61210d05ae8 -> 5567dd47e7ba29821a518c4a9852adc3b774e25ef3e7a6e2b3ecb7b59ddab73c; scaling ratio 1.031 < 1.5.", "_rebaselined_2550_instance_model": "PR #2549 (#2545/#2551): object literals emit @scope.object. Prior 479927409bbdd9852a36172c8260aa56df260e99129a7a9c20a0d1903dd5538b -> f1ccf42a36895c8e34dcb724286f247d469835f2dcbb23ad3347190adc7fde1c; scaling 1.096 < 1.5.", - "_rebaselined_receiver_owner_2701": "#2701: every non-arrow function form now carries a `@receiver-owner.this` marker on the same node as `@scope.function`, so a scope that BINDS its own `this` can stop the receiver walk (`Scope.ownsReceivers`). Verified before re-baselining by diffing the capture-name histogram over this same fixture corpus against 1d3088173f6f93827641b476d614d5d15cd4f3ea: the ONLY delta is @receiver-owner.this (typescript +143, javascript +32) — every other capture count is byte-identical, so no existing capture moved. Prior f1ccf42a36895c8e34dcb724286f247d469835f2dcbb23ad3347190adc7fde1c -> 90601494695b834d3a9af7ac4844eac603f4f432809a05554cc59de0674a4354.", - "_rebaselined_receiver_chain_2747": "#2747 receiver-chain rollout: call matches whose receiver is itself an expression now carry `@reference.receiver-chain`, a compact encoding of the receiver's structure, so resolution types it by folding instead of re-parsing receiver source text. Capture GROUP counts are unchanged — the tag is added to existing call matches, never a new match — so this is digest drift only. Prior 1c71ef628eb75a3b111afa8c2a7c351c16a7f5aab9fac2f098f82b2866312aa8 -> 83344b7cba093702f4528eeee44e438809c229d43b12e69ed288812ce7ffc7bc.", - "_rebaselined_2766_receiver_chain_wire_v2": "#2766: receiver-chain wire format v1 -> v2 (name-free `await` / `index` step kinds). The VERSION prefix is part of every emitted `@reference.receiver-chain` capture, so every chain-minting language's capture text changed. WIRE-FORMAT CHANGE, NOT A CAPTURE-SET CHANGE: the same chains are minted for the same sites, spelled `2|…` instead of `1|…`. Exactly the 12 chain-minting languages drifted; c, cobol and dart did not, which is the check that this is the prefix and not a capture regression. Accompanied by SCHEMA_BUMP 34 -> 37 and INCREMENTAL_SCHEMA_VERSION 28 -> 31 so a stale index is rejected rather than replaying chains a v2 decoder refuses. Prior 83344b7cba093702f4528eeee44e438809c229d43b12e69ed288812ce7ffc7bc -> 806f70ad3cce5fc849f6d06a08ace8a95f92a1ea84a2418fddabb1eef5846594." + "_rebaselined_receiver_owner_2701": "#2701: every non-arrow function form now carries a `@receiver-owner.this` marker on the same node as `@scope.function`, so a scope that BINDS its own `this` can stop the receiver walk (`Scope.ownsReceivers`). Verified before re-baselining by diffing the capture-name histogram over this same fixture corpus against 1d3088173f6f93827641b476d614d5d15cd4f3ea: the ONLY delta is @receiver-owner.this (typescript +143, javascript +32) \u2014 every other capture count is byte-identical, so no existing capture moved. Prior f1ccf42a36895c8e34dcb724286f247d469835f2dcbb23ad3347190adc7fde1c -> 90601494695b834d3a9af7ac4844eac603f4f432809a05554cc59de0674a4354.", + "_rebaselined_receiver_chain_2747": "#2747 receiver-chain rollout: call matches whose receiver is itself an expression now carry `@reference.receiver-chain`, a compact encoding of the receiver's structure, so resolution types it by folding instead of re-parsing receiver source text. Capture GROUP counts are unchanged \u2014 the tag is added to existing call matches, never a new match \u2014 so this is digest drift only. Prior 1c71ef628eb75a3b111afa8c2a7c351c16a7f5aab9fac2f098f82b2866312aa8 -> 83344b7cba093702f4528eeee44e438809c229d43b12e69ed288812ce7ffc7bc.", + "_rebaselined_2766_receiver_chain_wire_v2": "#2766: receiver-chain wire format v1 -> v2 (name-free `await` / `index` step kinds). The VERSION prefix is part of every emitted `@reference.receiver-chain` capture, so every chain-minting language's capture text changed. WIRE-FORMAT CHANGE, NOT A CAPTURE-SET CHANGE: the same chains are minted for the same sites, spelled `2|\u2026` instead of `1|\u2026`. Exactly the 12 chain-minting languages drifted; c, cobol and dart did not, which is the check that this is the prefix and not a capture regression. Accompanied by SCHEMA_BUMP 34 -> 37 and INCREMENTAL_SCHEMA_VERSION 28 -> 31 so a stale index is rejected rather than replaying chains a v2 decoder refuses. Prior 83344b7cba093702f4528eeee44e438809c229d43b12e69ed288812ce7ffc7bc -> 806f70ad3cce5fc849f6d06a08ace8a95f92a1ea84a2418fddabb1eef5846594.", + "_rebaselined_blind_spots_2856": "#2856 blind-spots series: the JS/TS SCOPE queries gained capture rules, so fingerprint drift is expected and additive. Verified before re-baselining by diffing the capture-name sets in both scope queries against origin/main: TypeScript gained exactly @reference.read.identifier (A2 bare-identifier reads in value positions) and @reference.type (R2-2 type references, so a declared contract stops reporting incoming:{}); JavaScript gained exactly @reference.read.identifier, @reference.read.destructured (R2-1c) and @reference.write.property-key (R2-1b record-construction writes). NOTHING was removed on either side \u2014 the delta is a pure superset, which is the check that no existing capture moved. capture_groups_small/large are unchanged (4503/14403) because those measure the SYNTHETIC scaling source, which this branch does not touch; only the fixture-corpus count moves. capture_groups_fp 2097 -> 2338 and fixture_count 146 -> 151 from 21 new lang-resolution fixtures. Scaling stayed linear and inside budget: typescript 1.116 < 1.5, javascript 1.010 < 1.5. Prior typescript ed92588e0fc7b28b3a0174339ac378b4dd85965fe007db1208dea97a65ce0571 -> f66a3e6f1e096431e7046505129a627deaa00ca0de5bc846b080591b397248f7; prior javascript 806f70ad3cce5fc849f6d06a08ace8a95f92a1ea84a2418fddabb1eef5846594 -> 2026993b81b873839dd2ef8797d9c14d9c48516b2b57b05ac17d8d43f2f4eba3." }, "kotlin": { - "fingerprint": "efd5dbf80ffcd3bab2834d1010f6fe2b239dcc5d58229938dea9cff8d0f380f2", + "fingerprint": "a184f8ff0ae40d246db855b63f7ff26bda3afac03e5f4c76e4593c7e2cefce54", "scaling_budget": 1.5, "_rebaselined_callable_flow_2522_review": "PR #2522 review hardening: callable operands retain expression/qualified identity and formals retain signature metadata. Prior bddba25d5a88152bbbee8d70e82c944b5302accb4b625df782adb1d4f7a7ac12 -> e856951c2a779163d555dadc8e1bf59304a86caed78ac1f450d9caa2b50f63d1; scaling 1.090 < 1.5.", "_rebaselined_callable_flow_2522_followup": "PR #2522 follow-up: Kotlin callable-reference flow facts with invocation-result suppression. Prior 4900431791f2b9280009deb2b82659c26ead8aa6fb8731190a7c505dec5a9041 -> bddba25d5a88152bbbee8d70e82c944b5302accb4b625df782adb1d4f7a7ac12; scaling 0.880 < 1.5.", "_added": "#1951: bench coverage added (was ungated); scale source heritage-bearing (: Base()); js/kotlin O(n^2) findNodeAtRange-per-match fixed to threaded captured node, now linear.", "_rebaselined": "#1919 review CF3 fix: extended kotlin-local-property-owner (init/accessor destructuring) + new dart-accessor-owner fixture (getter/setter ownership). Fingerprint-only corpus drift; scaling ~1.0.", - "_rebaselined_2271": "PR #2271: re-vendored tree-sitter-kotlin 0.3.8 -> unreleased fwcd main c8ac3d26 for `fun interface` support + new kotlin-fun-interface fixture in the corpus. Drift is both corpus-additive (the fixture) and grammar-driven (the new grammar parses `fun interface` as a class_declaration, not an ERROR node). Baselined to the NEW grammar's fingerprint, so this --check passes only once the regenerated prebuilds land — until then CI loads the committed 0.3.8 binary and the bench is red, same as the kotlin fun-interface integration tests. scaling ~0.83 (linear).", + "_rebaselined_2271": "PR #2271: re-vendored tree-sitter-kotlin 0.3.8 -> unreleased fwcd main c8ac3d26 for `fun interface` support + new kotlin-fun-interface fixture in the corpus. Drift is both corpus-additive (the fixture) and grammar-driven (the new grammar parses `fun interface` as a class_declaration, not an ERROR node). Baselined to the NEW grammar's fingerprint, so this --check passes only once the regenerated prebuilds land \u2014 until then CI loads the committed 0.3.8 binary and the bench is red, same as the kotlin fun-interface integration tests. scaling ~0.83 (linear).", "_rebaselined_2522_review_fixes": "PR #2522 review fixes: fieldless assignment nodes decomposed positionally. Prior e856951c2a779163d555dadc8e1bf59304a86caed78ac1f450d9caa2b50f63d1 -> 4b31f46cfb004ba769a96feeb06ae4ef109c77410f54e7aaab4a688df599b112; scaling ratio re-verified within budget.", "_rebaselined_2550_instance_model": "PR #2549 (#2545): anonymous object expressions (object_literal) emit @scope.class, and the kotlin-object-literal-scope fixture joined the corpus. Prior 4b31f46cfb004ba769a96feeb06ae4ef109c77410f54e7aaab4a688df599b112 -> a6fce0dff00e88d41d85023eaf3f35016b5217c7e5225f24a598e4c70bb63091; scaling 0.951 < 1.5.", "_rebaselined_2563_instance_ownership": "#2563: kotlin-instance-ownership adds unrelated, inherited, outer-instance, and anonymous-object coverage. Prior a6fce0dff00e88d41d85023eaf3f35016b5217c7e5225f24a598e4c70bb63091 -> 9f159f8810d342ef1c821f466efd6920dad9a190f06000056e6cd2815861b195; scaling 1.257 < 1.5.", - "_rebaselined_receiver_chain_2747": "#2747 receiver-chain rollout: call matches whose receiver is itself an expression now carry `@reference.receiver-chain`, a compact encoding of the receiver's structure, so resolution types it by folding instead of re-parsing receiver source text. Capture GROUP counts are unchanged — the tag is added to existing call matches, never a new match — so this is digest drift only. Prior 9f159f8810d342ef1c821f466efd6920dad9a190f06000056e6cd2815861b195 -> d3c4d2fa0d82d248a2299cfc888b067187ad1faf2c87a97f93c6ed835eefc3f1.", - "_rebaselined_2766_receiver_chain_wire_v2": "#2766: receiver-chain wire format v1 -> v2 (name-free `await` / `index` step kinds). The VERSION prefix is part of every emitted `@reference.receiver-chain` capture, so every chain-minting language's capture text changed. WIRE-FORMAT CHANGE, NOT A CAPTURE-SET CHANGE: the same chains are minted for the same sites, spelled `2|…` instead of `1|…`. Exactly the 12 chain-minting languages drifted; c, cobol and dart did not, which is the check that this is the prefix and not a capture regression. Accompanied by SCHEMA_BUMP 34 -> 37 and INCREMENTAL_SCHEMA_VERSION 28 -> 31 so a stale index is rejected rather than replaying chains a v2 decoder refuses. Prior d3c4d2fa0d82d248a2299cfc888b067187ad1faf2c87a97f93c6ed835eefc3f1 -> c1f0cc9058ab11b7cd6fc8b440deb6db2b2f530f2eb21178923e68a3d0796c4b.", - "_rebaselined_2766_await_subscript_emission": "#2766: extractMixedChain now walks THROUGH await and subscript nodes and peels transparent wrappers at loop entry, so sites whose receiver is `repos[0]` or `(await f())` mint a receiver chain where they previously minted none. EMISSION CHANGE: more sites carry `@reference.receiver-chain`; no existing chain changed shape. Only go and kotlin drifted of 15 — the two whose fixture corpora contain such receivers. Prior c1f0cc9058ab11b7cd6fc8b440deb6db2b2f530f2eb21178923e68a3d0796c4b -> efd5dbf80ffcd3bab2834d1010f6fe2b239dcc5d58229938dea9cff8d0f380f2." + "_rebaselined_receiver_chain_2747": "#2747 receiver-chain rollout: call matches whose receiver is itself an expression now carry `@reference.receiver-chain`, a compact encoding of the receiver's structure, so resolution types it by folding instead of re-parsing receiver source text. Capture GROUP counts are unchanged \u2014 the tag is added to existing call matches, never a new match \u2014 so this is digest drift only. Prior 9f159f8810d342ef1c821f466efd6920dad9a190f06000056e6cd2815861b195 -> d3c4d2fa0d82d248a2299cfc888b067187ad1faf2c87a97f93c6ed835eefc3f1.", + "_rebaselined_2766_receiver_chain_wire_v2": "#2766: receiver-chain wire format v1 -> v2 (name-free `await` / `index` step kinds). The VERSION prefix is part of every emitted `@reference.receiver-chain` capture, so every chain-minting language's capture text changed. WIRE-FORMAT CHANGE, NOT A CAPTURE-SET CHANGE: the same chains are minted for the same sites, spelled `2|\u2026` instead of `1|\u2026`. Exactly the 12 chain-minting languages drifted; c, cobol and dart did not, which is the check that this is the prefix and not a capture regression. Accompanied by SCHEMA_BUMP 34 -> 37 and INCREMENTAL_SCHEMA_VERSION 28 -> 31 so a stale index is rejected rather than replaying chains a v2 decoder refuses. Prior d3c4d2fa0d82d248a2299cfc888b067187ad1faf2c87a97f93c6ed835eefc3f1 -> c1f0cc9058ab11b7cd6fc8b440deb6db2b2f530f2eb21178923e68a3d0796c4b.", + "_rebaselined_2766_await_subscript_emission": "#2766: extractMixedChain now walks THROUGH await and subscript nodes and peels transparent wrappers at loop entry, so sites whose receiver is `repos[0]` or `(await f())` mint a receiver chain where they previously minted none. EMISSION CHANGE: more sites carry `@reference.receiver-chain`; no existing chain changed shape. Only go and kotlin drifted of 15 \u2014 the two whose fixture corpora contain such receivers. Prior c1f0cc9058ab11b7cd6fc8b440deb6db2b2f530f2eb21178923e68a3d0796c4b -> efd5dbf80ffcd3bab2834d1010f6fe2b239dcc5d58229938dea9cff8d0f380f2.", + "capture_groups_small": 4753, + "capture_groups_large": 15203, + "capture_groups_fp": 2334, + "fixture_count": 137 } } diff --git a/gitnexus/bench/scope-capture/measure.mjs b/gitnexus/bench/scope-capture/measure.mjs index 56aa2592d..7a4e4057b 100644 --- a/gitnexus/bench/scope-capture/measure.mjs +++ b/gitnexus/bench/scope-capture/measure.mjs @@ -209,10 +209,22 @@ const LANGS = [ // Heritage-bearing: `: public Base, public Mixin` (single + multiple // inheritance) drives emitCppInheritanceCaptures (#1951) at scale. Added // (was unbenched); adding it exposed + fixed the same O(n²) root-walk (#1956). + // + // Also GENERIC-MEMBER-bearing (#2833): `Repo repo;` is a member + // whose declared type is a bare `template_type`, and + // `std::vector items;` is the far commoner spelling where a + // `qualified_identifier` WRAPS that template_type. Both were absent, and + // their absence is why two successive rounds of `field_declaration` + // type-binding rules landed with a byte-identical cpp fingerprint: the gate + // could not see a member field it had no instance of. With them present, + // reverting either round of rules drifts the fingerprint, which is the + // property that makes the gate worth running. header: - '#include \n\nclass Base {\n public:\n long baseId() const { return 0; }\n};\n\nclass Mixin {\n public:\n void mix() {}\n};\n\n', + '#include \n#include \n\ntemplate \nclass Repo {\n public:\n void save(T v) {}\n};\n\nclass Base {\n public:\n long baseId() const { return 0; }\n};\n\nclass Mixin {\n public:\n void mix() {}\n};\n\n', unit: (n) => `class Entity${n} : public Base, public Mixin {\n public:\n long id;\n std::string name;\n` + + ` Repo repo;\n` + + ` std::vector items;\n` + ` long getId() const { return id; }\n` + ` void setName(std::string v) { name = v; }\n};\n\n`, }, @@ -255,14 +267,15 @@ const LANGS = [ fixturePrefix: 'java', exts: ['.java'], file: 'bench.java', - // Java was previously unbenched. Heritage-bearing: extends Base + implements - // Marker (both forms) so the @reference.inherits synth (#1951) is driven at scale. + // Java was previously unbenched. Class and record heritage both implement + // Marker so the @reference.inherits synth (#1951, #2900) is driven at scale. header: 'package generated;\n\nclass Base {}\n\ninterface Marker {}\n\n', unit: (n) => `class Entity${n} extends Base implements Marker {\n` + ` long id = 0L;\n String name = "";\n` + ` public long getId() { return this.id; }\n` + - ` public void setName(String v) { this.name = v; }\n}\n\n`, + ` public void setName(String v) { this.name = v; }\n}\n\n` + + `record RecordEntity${n}(long id) implements Marker {}\n\n`, }, { name: 'java-local-types', diff --git a/gitnexus/hooks/antigravity/gitnexus-antigravity-hook.cjs b/gitnexus/hooks/antigravity/gitnexus-antigravity-hook.cjs index 3331ae3fc..630195087 100755 --- a/gitnexus/hooks/antigravity/gitnexus-antigravity-hook.cjs +++ b/gitnexus/hooks/antigravity/gitnexus-antigravity-hook.cjs @@ -503,7 +503,7 @@ function buildStaleIndexHint(gitNexusDir, cwd) { if (currentHead === lastCommit) return ''; - const analyzeCmd = formatAnalyzeCommand({ embeddings: hadEmbeddings }); + const analyzeCmd = formatAnalyzeCommand({ embeddings: hadEmbeddings, indexOnly: true }); return ( `[GitNexus] index is stale (last indexed: ${lastCommit ? lastCommit.slice(0, 7) : 'never'}). ` + `Run \`${analyzeCmd}\` to refresh the knowledge graph.` diff --git a/gitnexus/hooks/claude/gitnexus-hook.cjs b/gitnexus/hooks/claude/gitnexus-hook.cjs index 18be614f4..1b75ed17d 100755 --- a/gitnexus/hooks/claude/gitnexus-hook.cjs +++ b/gitnexus/hooks/claude/gitnexus-hook.cjs @@ -523,7 +523,7 @@ function handlePostToolUse(input) { // If HEAD matches last indexed commit, no reindex needed if (currentHead && currentHead === lastCommit) return; - const analyzeCmd = formatAnalyzeCommand({ embeddings: hadEmbeddings }); + const analyzeCmd = formatAnalyzeCommand({ embeddings: hadEmbeddings, indexOnly: true }); sendHookResponse( 'PostToolUse', `GitNexus index is stale (last indexed: ${lastCommit ? lastCommit.slice(0, 7) : 'never'}). ` + diff --git a/gitnexus/hooks/claude/resolve-analyze-cmd.cjs b/gitnexus/hooks/claude/resolve-analyze-cmd.cjs index 56f5235fb..c74f03f5d 100644 --- a/gitnexus/hooks/claude/resolve-analyze-cmd.cjs +++ b/gitnexus/hooks/claude/resolve-analyze-cmd.cjs @@ -276,7 +276,13 @@ function formatBunxCommand(gitnexusArgs) { } function formatAnalyzeCommand(options = {}, deps = {}) { - const suffix = options.embeddings ? ' --embeddings' : ''; + // `--index-only` is what a routine "your index is stale" nudge wants: it + // reindexes without rewriting AGENTS.md / CLAUDE.md / skills, so an agent + // following the nudge on every commit cannot churn the tracked agent guides + // (#2907). Callers that actually want the docs refreshed omit it. + const suffix = `${options.indexOnly ? ' --index-only' : ''}${ + options.embeddings ? ' --embeddings' : '' + }`; // Keep the stale-index hook budget tight by querying each tool at most once. // The memoized `probe` is a spawn-free PATH scan (resolveOnPath) shared with // resolveInvocationMode, so `gitnexus` is scanned only once and no subprocess diff --git a/gitnexus/package-lock.json b/gitnexus/package-lock.json index 2cad2dc9d..23e85358b 100644 --- a/gitnexus/package-lock.json +++ b/gitnexus/package-lock.json @@ -10,7 +10,7 @@ "hasInstallScript": true, "license": "PolyForm-Noncommercial-1.0.0", "dependencies": { - "@ladybugdb/core": "^0.18.3", + "@ladybugdb/core": "^0.19.0", "@modelcontextprotocol/sdk": "^1.0.0", "@scarf/scarf": "^1.4.0", "busboy": "^1.6.0", @@ -79,7 +79,7 @@ "version": "1.0.0", "dev": true, "devDependencies": { - "typescript": "^6.0.3" + "typescript": "^7.0.2" } }, "node_modules/@babel/code-frame": { @@ -1254,9 +1254,9 @@ } }, "node_modules/@ladybugdb/core": { - "version": "0.18.3", - "resolved": "https://registry.npmjs.org/@ladybugdb/core/-/core-0.18.3.tgz", - "integrity": "sha512-XjpPKW4MrL28D2gYGTZuIjiEcPx12L21lx58QggrdrItw8o/e9Lmg/Ejoo4Kz08lZj+rIcC1Fu9thzIYOTUlJw==", + "version": "0.19.1", + "resolved": "https://registry.npmjs.org/@ladybugdb/core/-/core-0.19.1.tgz", + "integrity": "sha512-8W2g6xUi4jm96fs4EayyMcsvEEtIb8vboZhw9/YG98881cIcmZjmqAN91XGUp4vb8NqoFr3Wp7wcu3dqJk0b7w==", "hasInstallScript": true, "license": "MIT", "dependencies": { @@ -1265,17 +1265,17 @@ "node-addon-api": "^6.0.0" }, "optionalDependencies": { - "@ladybugdb/core-darwin-arm64": "0.18.3", - "@ladybugdb/core-darwin-x64": "0.18.3", - "@ladybugdb/core-linux-arm64": "0.18.3", - "@ladybugdb/core-linux-x64": "0.18.3", - "@ladybugdb/core-win32-x64": "0.18.3" + "@ladybugdb/core-darwin-arm64": "0.19.1", + "@ladybugdb/core-darwin-x64": "0.19.1", + "@ladybugdb/core-linux-arm64": "0.19.1", + "@ladybugdb/core-linux-x64": "0.19.1", + "@ladybugdb/core-win32-x64": "0.19.1" } }, "node_modules/@ladybugdb/core-darwin-arm64": { - "version": "0.18.3", - "resolved": "https://registry.npmjs.org/@ladybugdb/core-darwin-arm64/-/core-darwin-arm64-0.18.3.tgz", - "integrity": "sha512-DGZTOlvSS4esEb1vTekY5IDoAvZAeYzR5cXVkECtQj9BVkk05zsvCAdTPo1Rz1BuI0qvqUVF+2WlIerI67iA2g==", + "version": "0.19.1", + "resolved": "https://registry.npmjs.org/@ladybugdb/core-darwin-arm64/-/core-darwin-arm64-0.19.1.tgz", + "integrity": "sha512-VGQs1NThAygMsoOlxud05pqKA9xfUptl55iYkwvW45As5MSI7+M86WN0Pp0VdPEfw8vNQJehrlHR5LvVAuWc2Q==", "cpu": [ "arm64" ], @@ -1286,9 +1286,9 @@ ] }, "node_modules/@ladybugdb/core-darwin-x64": { - "version": "0.18.3", - "resolved": "https://registry.npmjs.org/@ladybugdb/core-darwin-x64/-/core-darwin-x64-0.18.3.tgz", - "integrity": "sha512-Qp6j0CM/orBlK6KD0p/s4ofkIhNUwi1hdCgMw+fj81UHugWHkVLiYV4grRBdHhyplw+snchZpTxvfpxFbkG1Cw==", + "version": "0.19.1", + "resolved": "https://registry.npmjs.org/@ladybugdb/core-darwin-x64/-/core-darwin-x64-0.19.1.tgz", + "integrity": "sha512-CGfM6ostxDS5jztxwjkXXtxrjMDgsFMRoyr5HZDCFw1+iXC1rIzmK/Y7RIw+KbQ49aPzSmkhBC447mFviBJxoA==", "cpu": [ "x64" ], @@ -1299,9 +1299,9 @@ ] }, "node_modules/@ladybugdb/core-linux-arm64": { - "version": "0.18.3", - "resolved": "https://registry.npmjs.org/@ladybugdb/core-linux-arm64/-/core-linux-arm64-0.18.3.tgz", - "integrity": "sha512-F9miYjBuS43I7uNG199FNMqwdHJ98WA6dU3v2SZCeLXmXCdRzmYcuHQWlbNr2Tba9CX58w2XvBZoUaXZKJ/yKQ==", + "version": "0.19.1", + "resolved": "https://registry.npmjs.org/@ladybugdb/core-linux-arm64/-/core-linux-arm64-0.19.1.tgz", + "integrity": "sha512-BZUQwlkvNXENc5GVyXdfRF0Dv9JX8XMlcdMMiB5GKrEhTCpajQ3D58woHPVvn0JEjw7Ms3tHo6kXUAMZKYXIVg==", "cpu": [ "arm64" ], @@ -1312,9 +1312,9 @@ ] }, "node_modules/@ladybugdb/core-linux-x64": { - "version": "0.18.3", - "resolved": "https://registry.npmjs.org/@ladybugdb/core-linux-x64/-/core-linux-x64-0.18.3.tgz", - "integrity": "sha512-AfG5RDp/f/IDctDMpTAT5+2MYNtlWT191xiQNjSaWD4X85DhY3Dzps8Qu5VteIAPih5d6mmoaKGs8q0XIjfkFA==", + "version": "0.19.1", + "resolved": "https://registry.npmjs.org/@ladybugdb/core-linux-x64/-/core-linux-x64-0.19.1.tgz", + "integrity": "sha512-LDx+E1UHlmNXSb3F9QmvdBgZGfB3wI/DcrHzfOwXgT3BP8C4ScB2tZdpiYQiuPp8MiSZ9kuuGqos8A4tQKQu8Q==", "cpu": [ "x64" ], @@ -1325,9 +1325,9 @@ ] }, "node_modules/@ladybugdb/core-win32-x64": { - "version": "0.18.3", - "resolved": "https://registry.npmjs.org/@ladybugdb/core-win32-x64/-/core-win32-x64-0.18.3.tgz", - "integrity": "sha512-bHuFk0m9cnq0WGd9I4D8or8g6cC/BS58iatMtilqM3JpDPIQIFk6MQl6exL7P4xyWbkLwQgsrv2ToDnyoQNKvg==", + "version": "0.19.1", + "resolved": "https://registry.npmjs.org/@ladybugdb/core-win32-x64/-/core-win32-x64-0.19.1.tgz", + "integrity": "sha512-2spst1g+Z050Fz/5z7Pc6Fuc5dVXLzekOuWW4lP+mEGCL3tkv3QWYxk37DiFy+O9fDFxiWK2f3aauab58/f9kQ==", "cpu": [ "x64" ], @@ -1939,9 +1939,9 @@ "license": "MIT" }, "node_modules/@types/node": { - "version": "26.1.2", - "resolved": "https://registry.npmjs.org/@types/node/-/node-26.1.2.tgz", - "integrity": "sha512-Vu4a5UFA9rIIFJ7rB/Vaafh9lrCQszopTCx6KjFboXTGQbPNasehVR5TEiithSDGyd1DEiUByggTZsg8jukeIg==", + "version": "26.2.0", + "resolved": "https://registry.npmjs.org/@types/node/-/node-26.2.0.tgz", + "integrity": "sha512-5IviulTZeRNp2vAJ514cc/HUlY5nZ9fCbq9DMyC52BrhFZACo3nI0R7qBxhQmo/d27NFe96ur/b7Wwxklda+kg==", "devOptional": true, "license": "MIT", "dependencies": { @@ -3018,9 +3018,9 @@ } }, "node_modules/express-rate-limit": { - "version": "8.6.1", - "resolved": "https://registry.npmjs.org/express-rate-limit/-/express-rate-limit-8.6.1.tgz", - "integrity": "sha512-0D493aP61w0TJ2A0wy27riRsO7FMQ7FK+KUHOKCSfPvYo0R55aiC6emCVgFUeShH0fq0ICPVzNcgoS+BsbXQCA==", + "version": "8.6.2", + "resolved": "https://registry.npmjs.org/express-rate-limit/-/express-rate-limit-8.6.2.tgz", + "integrity": "sha512-YH4ru+eOJxQABscKFfRCy9R7x9QFGdezclVMwwgFFndzS2Xnm0uo6B0ABZsLhcpeptGv2qvuJVWlQr9gQZoC3A==", "license": "MIT", "dependencies": { "debug": "^4.4.3", @@ -5430,9 +5430,9 @@ "license": "0BSD" }, "node_modules/tsx": { - "version": "4.23.4", - "resolved": "https://registry.npmjs.org/tsx/-/tsx-4.23.4.tgz", - "integrity": "sha512-ZiUQ8oT/KzN51mJUWPqARYqwFLFJZtGZipRkw1ynHMr9vy3eU77m5yfF3Gzm6meEg/beW+lUu3fHYgskTN2oVQ==", + "version": "4.23.12", + "resolved": "https://registry.npmjs.org/tsx/-/tsx-4.23.12.tgz", + "integrity": "sha512-FDf4L4sYzKtzWYhU/Xm0AQFdTjdIxNo9ElTf2mxXM6k8YMHXzYUe4yODVaXP4V9uMFbVg8c0qyBccK2OOxb45Q==", "dev": true, "license": "MIT", "dependencies": { diff --git a/gitnexus/package.json b/gitnexus/package.json index 480b1e919..5315226f1 100644 --- a/gitnexus/package.json +++ b/gitnexus/package.json @@ -56,7 +56,7 @@ "version": "node scripts/sync-plugin-manifests.mjs" }, "dependencies": { - "@ladybugdb/core": "^0.18.3", + "@ladybugdb/core": "^0.19.0", "@modelcontextprotocol/sdk": "^1.0.0", "@scarf/scarf": "^1.4.0", "busboy": "^1.6.0", diff --git a/gitnexus/scripts/cross-platform-shard.ts b/gitnexus/scripts/cross-platform-shard.ts index bf5100436..1a026a5e0 100644 --- a/gitnexus/scripts/cross-platform-shard.ts +++ b/gitnexus/scripts/cross-platform-shard.ts @@ -47,6 +47,13 @@ export const WINDOWS_WEIGHTS_SEC: Readonly> = { 'test/integration/cli-e2e.test.ts': 361, 'test/integration/worker-pool.test.ts': 222, 'test/unit/incremental-vector-extension-ordering.test.ts': 87, + // ESTIMATE, not a measurement (#2841): this suite drives more full + // `runFullAnalysis` cycles than the VECTOR sibling above, so the 8 s + // PER_FILE_OVERHEAD floor would badly under-charge it and skew the Windows + // split — the failure mode that produced the job timeouts this table exists + // to prevent. Scaled from the sibling's measured 87 s by analyze-run count. + // Replace with a real figure after the first green Windows matrix run. + 'test/unit/incremental-index-extension-dml-gate.test.ts': 180, 'test/integration/cli-limit-e2e.test.ts': 75, 'test/unit/hooks.test.ts': 26, 'test/integration/analyze-heap-oom-e2e.test.ts': 23, diff --git a/gitnexus/scripts/cross-platform-tests.ts b/gitnexus/scripts/cross-platform-tests.ts index 325699953..1994abf6b 100644 --- a/gitnexus/scripts/cross-platform-tests.ts +++ b/gitnexus/scripts/cross-platform-tests.ts @@ -36,6 +36,16 @@ const PLATFORM_LOGIC = [ // must exercise the Windows backslash branch, so run it on the OS matrix (#2394). 'test/unit/cli-entry.test.ts', 'test/unit/platform-capabilities.test.ts', + // The gitnexus-plan safe writer resolves every name through a per-platform + // backend: Linux anchors through /proc/self/fd, macOS resolves lexically and + // verifies each step against descriptors it holds open. Publication is link(2) + // on both. #2905 shipped the Darwin backend after the suite had silently + // skipped on every non-Linux runner, so this file must run on the OS matrix or + // the macOS half is unverified by construction — and the flag, trailing- + // separator and hard-link fixtures assert kernel behaviour that only a real + // Darwin kernel can confirm. Windows is refused by the capability gate; the + // suite asserts that refusal rather than skipping it. + 'test/unit/evidence-provenance-helper.test.ts', // Windows drive-letter case variance in the analyzer runner-identity path // fields (#2668): normalizeAnalyzerRootPath is a POSIX no-op, so the // "identity path fields are normalizer-stable" fixpoint guard only bites on @@ -154,6 +164,18 @@ const LBUG_NATIVE = [ // proven on the windows-latest native addon, not just Ubuntu. Budget: ~25s // on Linux → expect ~2min on the slowest Windows shard. 'test/unit/incremental-vector-extension-ordering.test.ts', + // #2841: the FTS half of that same gate, plus the both-extensions-blocked + // case — and it needs this matrix for two reasons the VECTOR sibling above + // does not cover. The reported failure environment is a machine where the + // extension stopped LOADING, which is the #2374 class and Windows-reported + // (the same reason fts-extension-e2e.test.ts is registered below), so the + // FTS-unavailable branch has to run on a real Windows/macOS runner rather + // than only on Ubuntu where FTS always loads. And its both-blocked case is + // gated on GITNEXUS_REQUIRE_VECTOR=1, which ci-tests.yml sets ONLY on this + // job — everywhere else an unavailable VECTOR extension skips instead of + // failing. Budget: four real analyze runs, so expect it to sit alongside the + // VECTOR sibling's ~87s Windows measurement. + 'test/unit/incremental-index-extension-dml-gate.test.ts', ]; // Process spawning and CLI tests — exercise child_process with real diff --git a/gitnexus/skills/gitnexus-cli.md b/gitnexus/skills/gitnexus-cli.md index 342e8b08f..853d44860 100644 --- a/gitnexus/skills/gitnexus-cli.md +++ b/gitnexus/skills/gitnexus-cli.md @@ -60,7 +60,7 @@ Generates repository documentation from the knowledge graph using an LLM. Requir | Flag | Effect | | ------------------- | ----------------------------------------- | | `--force` | Force full regeneration | -| `--model ` | LLM model (default: minimax/minimax-m2.5) | +| `--model ` | LLM model (default: MiniMax-M3) | | `--base-url ` | LLM API base URL | | `--api-key ` | LLM API key | | `--concurrency ` | Parallel LLM calls (default: 3) | diff --git a/gitnexus/skills/gitnexus-impact-analysis.md b/gitnexus/skills/gitnexus-impact-analysis.md index 0b81795de..ee1cd3496 100644 --- a/gitnexus/skills/gitnexus-impact-analysis.md +++ b/gitnexus/skills/gitnexus-impact-analysis.md @@ -53,6 +53,14 @@ description: "Use when the user wants to know what will break if they change som | 5-15 symbols, 2-5 processes | MEDIUM | | >15 symbols or many processes | HIGH | | Critical path (auth, payments) | CRITICAL | +| **Zero callers found** | **UNKNOWN** | + +`UNKNOWN` is not a low rung on this scale — it means the walk could not answer. +An empty caller set is equally consistent with "genuinely unused" and "the +callers are not resolvable by the index" (plain-object property access, dynamic +dispatch, cross-language calls), so few-callers ⇒ LOW does **not** apply. The +result carries a `riskNote` saying so. Confirm with a text search before +treating the symbol as safe to change or delete. ## Tools @@ -84,6 +92,11 @@ detect_changes({scope: "all"}) → Risk: MEDIUM ``` +`partial: true` (a graph query failed) or `truncated: true` (the changed-symbol +listing was capped) means the result is short of the truth, and reads like +`UNKNOWN` above: a zero there means unseen, not unaffected. Re-run it rather +than tick the pre-commit check. + ## Example: "What breaks if I change validateUser?" ``` diff --git a/gitnexus/skills/gitnexus-plan/README.md b/gitnexus/skills/gitnexus-plan/README.md index f7fe58ab9..153374bb7 100644 --- a/gitnexus/skills/gitnexus-plan/README.md +++ b/gitnexus/skills/gitnexus-plan/README.md @@ -124,12 +124,17 @@ phase that needs them. statement-level claims (never reconstructs fake edges). - No GitNexus at all → fallback mode: targeted grep/read exploration, findings labelled **source-derived**, with a recommendation to index. -- Reading or publishing a plan requires Linux `/proc/self/fd`, `O_DIRECTORY`, - and `O_NOFOLLOW`; publication also requires a validated absolute Python 3 - PATH candidate with libc `renameat2(RENAME_NOREPLACE)` support, a - writable target repository, and a shared filesystem for the plan and - Git-admin vault. The writer fails closed when those guarantees are - unavailable; it never redirects the plan elsewhere. +- Reading or publishing a plan requires `O_DIRECTORY` and `O_NOFOLLOW`, plus + `/proc/self/fd` on Linux; every other platform is refused. No interpreter is + spawned and no native code is loaded. Publication is `link(2)`, which fails + rather than replaces when the destination name is taken. Linux resolves every + name against a held descriptor, so a parent swapped mid-write cannot redirect + the operation; macOS has no equivalent path and instead pins each directory + with an open descriptor and re-proves the chain either side of every step, + which detects such a swap and aborts. Publishing also needs a writable target + repository and a shared filesystem for the plan and Git-admin vault. The + writer fails closed when those guarantees are unavailable; it never redirects + the plan elsewhere. ## Limitations diff --git a/gitnexus/skills/gitnexus-plan/references/evidence-provenance.md b/gitnexus/skills/gitnexus-plan/references/evidence-provenance.md index c686599da..3df5a046d 100644 --- a/gitnexus/skills/gitnexus-plan/references/evidence-provenance.md +++ b/gitnexus/skills/gitnexus-plan/references/evidence-provenance.md @@ -98,8 +98,11 @@ excluded. ## Safe existing-plan read contract -`read-plan` fails closed unless Linux `/proc/self/fd`, `O_DIRECTORY`, and -`O_NOFOLLOW` are available. It resolves the exact Git top-level, opens the +`read-plan` fails closed unless the host platform can resolve names against a +held directory descriptor: Linux `/proc/self/fd` with `O_DIRECTORY` and +`O_NOFOLLOW`, or macOS `O_DIRECTORY`/`O_NOFOLLOW`. Every other platform is +refused outright — an unverified read is not a degraded read, it is a different, +racy operation. It resolves the exact Git top-level, opens the repository root and every plan parent as held no-follow directory descriptors, rejects missing, symlink, non-directory, and escaping parents, and opens the leaf with `O_NOFOLLOW`. It reads at most 16 MiB from that held file descriptor, @@ -109,13 +112,17 @@ Neither Deepen nor work may parse bytes obtained before or outside this receipt. ## Safe generated-plan write contract -The writer fails closed unless Linux `/proc/self/fd`, `O_DIRECTORY`, -`O_NOFOLLOW`, and Python 3 with libc `renameat2(RENAME_NOREPLACE)` support are -available. Python may live in `/usr/local`, a Nix profile, or another absolute -PATH directory, but the helper accepts only a resolved executable and -containing directory owned by root or the current user and not writable by -group/other. The resolved executable is opened without following links and -invoked through that held descriptor. Relative PATH entries are ignored. The plan parent and the +The writer fails closed unless the host platform offers `O_DIRECTORY` and +`O_NOFOLLOW`, plus `/proc/self/fd` on Linux. It spawns no interpreter and loads +no native code: publication is `link(2)`, which is atomic, fails `EEXIST` when +the destination name is taken, and refuses a symlinked destination without +following it — the same no-replace guarantee `renameat2(RENAME_NOREPLACE)` and +`renameatx_np(RENAME_EXCL)` provide, available through `fs.linkSync` on every +supported platform. The temporary name is unlinked once the link succeeds; the +published file is the same inode the writer created and verified, so every +identity check downstream holds by construction. A link that succeeds followed +by an unlink that fails leaves the plan published and is reported as success, +because it is one. The plan parent and the repository's Git-admin directory must also share a filesystem. It resolves the target repository's exact Git top-level, opens that root and every destination parent as held no-follow directory descriptors, creates missing @@ -128,15 +135,45 @@ The writer creates a random exclusive temporary file relative to the held final parent descriptor and keeps its no-follow descriptor open. It writes and flushes the bytes, binds the temporary name to the opened inode, and hashes the open file before publication. Immediately before publication it revalidates -the parent and the temporary path, inode, size, and digest. Publication uses an -atomic no-replace move relative to the held directory descriptor. Initial mode -therefore cannot overwrite a destination that appears after the absent check. +the parent and the temporary path, inode, size, and digest. Publication links +the temporary name to the destination relative to the held directory +descriptor, which fails rather than replaces if the destination is taken. +Initial mode therefore cannot overwrite a destination that appears after the +absent check. The writer then flushes the directory and revalidates the committed path by opening it with `O_NOFOLLOW`, hashing both the original temporary fd and the path-bound fd, and performing a second descriptor-anchored path identity check after hashing. A detected mutation or replacement aborts instead of accepting mixed-era output. +### Linux anchors, macOS verifies + +The two platforms reach the same destination by different proofs, and the +difference is real enough to state rather than smooth over. + +On Linux every name resolves through `/proc/self/fd//`, a magic link +the kernel resolves against the inode the descriptor already holds. The names +above it are never re-walked, so an attacker who renames a parent between the +check and the use cannot redirect the operation. The race is impossible, not +merely detected. + +macOS has no such path. `/dev/fd/` is a devfs node, not a magic link: it can +be opened, but nothing can be resolved through it. `open("/dev/fd//child")` +returns `ENOENT`, and `realpath` of it returns `/dev/fd/` rather than the +directory's path — measured on macOS 26, not inferred. Node exposes no `openat`, +no `dir_fd` parameter, and no FFI, so on macOS the writer resolves names +lexically with `O_NOFOLLOW` at every component, holds an open descriptor on +every directory in the chain for the whole operation, and proves before *and* +after each step that the chain still names exactly the inodes it is holding. +Holding the descriptors is what makes the recorded inode numbers trustworthy: +an open descriptor pins its inode, so a freed number cannot be recycled beneath +the walk. + +What that buys is detection rather than prevention. A parent swapped inside the +window between a check and its use is caught by the check that follows, and the +operation aborts having written nothing — but on Linux it could not have +happened at all. No published byte escapes verification on either platform. + `--replace` accepts only a pre-existing regular file and is reserved for Deepen; without it, accidental overwrite is rejected. It also requires the exact canonical `generated_plan_path` and `plan_digest` from the same session's diff --git a/gitnexus/skills/gitnexus-plan/scripts/evidence-provenance.mjs b/gitnexus/skills/gitnexus-plan/scripts/evidence-provenance.mjs index 181d2120b..793fe4cd8 100644 --- a/gitnexus/skills/gitnexus-plan/scripts/evidence-provenance.mjs +++ b/gitnexus/skills/gitnexus-plan/scripts/evidence-provenance.mjs @@ -479,11 +479,11 @@ function resolveOwnGitTopLevel(absolute) { if (result.status !== 0) return null; let topLevel; try { - topLevel = fs.realpathSync(decodeUtf8(result.stdout, 'nested repository root').trim()); + topLevel = fs.realpathSync.native(decodeUtf8(result.stdout, 'nested repository root').trim()); } catch { return null; } - return topLevel === fs.realpathSync(absolute) ? topLevel : null; + return topLevel === fs.realpathSync.native(absolute) ? topLevel : null; } function readOwnGitlinkHead(absolute) { @@ -616,17 +616,30 @@ function filesystemObject(absolute, expectedKind, mutationGuards, testHooks) { throw new Error(`Unsupported filesystem object at ${absolute}`); } -function guardPathParents(repo, repoPath, mutationGuards) { +// Every dirty path re-walks its own parents, and dirty paths overwhelmingly +// share them — the repository root is re-stat'ed once per path. `guarded` is +// per-snapshot and remembers which absolute directories already carry a guard, +// so each distinct directory is stat'ed and guarded exactly once. +// +// Keeping the first-seen identity is the conservative choice: verifyGuards +// re-checks every guard against the filesystem at the end, so a directory that +// changes after it was guarded still fails there. Skipping a re-stat cannot hide +// a change; it only avoids recording the same directory twice. +function guardPathParents(repo, repoPath, mutationGuards, guarded) { const components = repoPath.split('/'); let current = repo; - const rootStat = fs.lstatSync(repo, { bigint: true }); - mutationGuards.push({ - type: 'directory', - absolute: repo, - identity: stableDirectoryIdentity(rootStat), - }); + if (!guarded.has(repo)) { + guarded.add(repo); + mutationGuards.push({ + type: 'directory', + absolute: repo, + identity: stableDirectoryIdentity(fs.lstatSync(repo, { bigint: true })), + }); + } for (const component of components.slice(0, -1)) { current = path.join(current, component); + // Already proved a real directory and already guarded on an earlier path. + if (guarded.has(current)) continue; let stat; try { stat = fs.lstatSync(current, { bigint: true }); @@ -638,6 +651,7 @@ function guardPathParents(repo, repoPath, mutationGuards) { throw new Error(`Refusing to traverse symlink parent for ${repoPath}`); } if (!stat.isDirectory()) return; + guarded.add(current); mutationGuards.push({ type: 'directory', absolute: current, @@ -646,81 +660,153 @@ function guardPathParents(repo, repoPath, mutationGuards) { } } -function recordAnchoredAbsence(repo, repoPath, mutationGuards) { - requireDescriptorAnchoring(); - const flags = - fs.constants.O_RDONLY | - fs.constants.O_DIRECTORY | - fs.constants.O_NOFOLLOW | - (fs.constants.O_CLOEXEC ?? 0); - const descriptors = []; - let retainedFd; - try { - let currentFd = fs.openSync(repo, flags); - descriptors.push(currentFd); - const components = repoPath.split('/'); - for (let index = 0; index < components.length; index += 1) { - const component = components[index]; - const child = descriptorPath(currentFd, component); - let childStat; - try { - childStat = fs.lstatSync(child, { bigint: true }); - } catch (error) { - if (error?.code !== 'ENOENT' && error?.code !== 'ENOTDIR') throw error; - const parentStat = fs.fstatSync(currentFd, { bigint: true }); - if (!parentStat.isDirectory()) { - throw new Error(`Absence parent is no longer a directory for ${repoPath}`); - } - retainedFd = currentFd; - mutationGuards.push({ - type: 'absence', - fd: retainedFd, - childName: component, - repoPath, - parentIdentity: stableDirectoryIdentity(parentStat), - parentMutationIdentity: statIdentity(parentStat), - }); - for (const fd of descriptors) { - if (fd !== retainedFd) fs.closeSync(fd); - } - return; - } - if (index === components.length - 1) { - throw new Error(`${repoPath} appeared while its absence was being anchored`); - } - if (childStat.isSymbolicLink() || !childStat.isDirectory()) { - throw new Error(`Refusing a non-directory parent while anchoring absence for ${repoPath}`); - } - const nextFd = fs.openSync(child, flags); - descriptors.push(nextFd); - currentFd = nextFd; - } - throw new Error(`Could not anchor absence for ${repoPath}`); - } catch (error) { - for (const fd of descriptors) { - if (fd === retainedFd) continue; - try { - fs.closeSync(fd); - } catch { - // Preserve the primary absence-anchoring error. - } - } - throw error; +// A bound, not a bug: the absence cache deduplicates correctly and leaks nothing, +// but citedPaths is caller-supplied and unbounded, so a pathological snapshot +// could hold more descriptors than the process is allowed (macOS +// kern.maxfilesperproc is 24576). The peak precedes a `git` spawn, so exhaustion +// would surface as a git failure misreported as evidence instability. +// +// Refuse rather than evict: closing a cached descriptor would silently break the +// pinned chain of an absence guard that was already recorded against it, which is +// exactly the inode-recycling hole the pins exist to close. +const ABSENCE_ANCHOR_LIMITS = Object.freeze({ maxPinnedDirectories: 4096 }); + +// Every no-follow read and every exclusive create in this file uses one of these +// two, so a change lands in one place rather than in seven. +const VERIFIED_READ_FLAGS = + fs.constants.O_RDONLY | fs.constants.O_NOFOLLOW | (fs.constants.O_CLOEXEC ?? 0); +const VERIFIED_CREATE_FLAGS = + fs.constants.O_RDWR | + fs.constants.O_CREAT | + fs.constants.O_EXCL | + fs.constants.O_NOFOLLOW | + (fs.constants.O_CLOEXEC ?? 0); + +function requireAbsenceAnchorCapacity(cache) { + if (cache.size >= ABSENCE_ANCHOR_LIMITS.maxPinnedDirectories) { + throw new Error( + `Absence anchoring exceeds ${ABSENCE_ANCHOR_LIMITS.maxPinnedDirectories} pinned directories`, + ); } } -function materializeRecord(repo, statusRecord, layers, mutationGuards, testHooks) { +const ANCHORED_DIRECTORY_FLAGS = + fs.constants.O_RDONLY | + fs.constants.O_DIRECTORY | + fs.constants.O_NOFOLLOW | + (fs.constants.O_CLOEXEC ?? 0); + +// Every absence receipt is verified long after its walk returns, so the chain +// that produced it has to stay pinned until the snapshot ends — an unpinned inode +// number can be recycled by a replacement directory that then reproduces the +// recorded identity exactly. Absent cited paths overwhelmingly share prefixes, so +// the walked directories are cached per snapshot and keyed by repo-relative +// prefix: one open descriptor and one anchored walk per distinct directory rather +// than per path. snapshotEvidence owns every descriptor in this cache and closes +// each exactly once; guards only borrow them for verification. +function anchoredAbsenceRoot(repo, cache) { + const cached = cache.get(''); + if (cached) return cached; + requireAbsenceAnchorCapacity(cache); + const fd = openVerifiedDirectory(repo, ANCHORED_DIRECTORY_FLAGS); + const handle = { + fd, + expectedPath: repo, + chain: [ + { expectedPath: repo, identity: stableDirectoryIdentity(fs.fstatSync(fd, { bigint: true })) }, + ], + descriptors: [fd], + }; + cache.set('', handle); + return handle; +} + +function recordAnchoredAbsence(repo, repoPath, mutationGuards, cache) { + requireDescriptorAnchoring(); + const components = repoPath.split('/'); + let handle = anchoredAbsenceRoot(repo, cache); + let prefix = ''; + for (let index = 0; index < components.length; index += 1) { + const component = components[index]; + const isFinal = index === components.length - 1; + prefix = prefix === '' ? component : `${prefix}/${component}`; + // The final component is always re-checked against the filesystem: it is the + // one whose absence is being recorded, and a cached answer would be a stale + // one. Only the prefix directories are reused. + const cached = isFinal ? undefined : cache.get(prefix); + if (cached) { + handle = cached; + continue; + } + const child = anchoredChild(handle, component); + let childStat; + try { + childStat = lstatChild(child); + } catch (error) { + if (error?.code !== 'ENOENT' && error?.code !== 'ENOTDIR') throw error; + const parentStat = fs.fstatSync(handle.fd, { bigint: true }); + if (!parentStat.isDirectory()) { + throw new Error(`Absence parent is no longer a directory for ${repoPath}`); + } + mutationGuards.push({ + type: 'absence', + // The handle is the holder the guard verifies against, and `ref` is the + // child path already built through the anchoredChild chokepoint — the + // guard must never re-derive that name itself. + handle, + ref: child, + fd: handle.fd, + repoPath, + parentMutationIdentity: statIdentity(parentStat), + }); + return; + } + if (isFinal) { + throw new Error(`${repoPath} appeared while its absence was being anchored`); + } + if (childStat.isSymbolicLink() || !childStat.isDirectory()) { + throw new Error(`Refusing a non-directory parent while anchoring absence for ${repoPath}`); + } + requireAbsenceAnchorCapacity(cache); + const childFd = openVerifiedDirectory(child.path, ANCHORED_DIRECTORY_FLAGS); + const expectedPath = path.join(handle.expectedPath, component); + let next; + try { + if (!anchoringBackend().descriptorMatchesChild(childFd, expectedPath, childStat)) { + throw new Error( + `Absence parent descriptor does not match its verified inode for ${repoPath}`, + ); + } + next = { + fd: childFd, + expectedPath, + chain: [...handle.chain, { expectedPath, identity: stableDirectoryIdentity(childStat) }], + descriptors: [...handle.descriptors, childFd], + }; + } catch (error) { + fs.closeSync(childFd); + throw error; + } + cache.set(prefix, next); + handle = next; + } + throw new Error(`Could not anchor absence for ${repoPath}`); +} + +function materializeRecord(repo, statusRecord, layers, mutationGuards, testHooks, walkState) { const head = layers.head(statusRecord.path); const index = layers.index(statusRecord.path); const expectedKind = index.kind === 'gitlink' || head.kind === 'gitlink' ? 'gitlink' : null; - guardPathParents(repo, statusRecord.path, mutationGuards); + guardPathParents(repo, statusRecord.path, mutationGuards, walkState.guardedDirectories); const filesystem = filesystemObject( path.join(repo, ...statusRecord.path.split('/')), expectedKind, mutationGuards, testHooks, ); - if (filesystem.kind === ABSENT) recordAnchoredAbsence(repo, statusRecord.path, mutationGuards); + if (filesystem.kind === ABSENT) { + recordAnchoredAbsence(repo, statusRecord.path, mutationGuards, walkState.absenceCache); + } if (statusRecord.directory_hint && filesystem.kind !== 'directory') { throw new Error( `Git reported an embedded directory but found ${filesystem.kind}: ${statusRecord.path}`, @@ -789,9 +875,15 @@ export function serializeDirtyRecords(entries) { } function assertRepository(repoInput) { - const repo = fs.realpathSync(requireString(repoInput, 'repo')); + // realpathSync.native, not realpathSync: the JS resolver preserves a Windows + // 8.3 short component (C:\Users\RUNNER~1\...) while git always reports the long + // form, so the two would never compare equal and every caller would be told the + // worktree root is not the worktree root it just named. + const repo = fs.realpathSync.native(requireString(repoInput, 'repo')); const topLevelResult = git(repo, ['rev-parse', '--show-toplevel']); - const topLevel = fs.realpathSync(decodeUtf8(topLevelResult.stdout, 'repository root').trim()); + const topLevel = fs.realpathSync.native( + decodeUtf8(topLevelResult.stdout, 'repository root').trim(), + ); if (topLevel !== repo) throw new Error(`--repo must be the Git worktree root (${topLevel})`); return repo; } @@ -882,17 +974,48 @@ function stableFileIdentity(stat) { return [stat.dev, stat.ino, stat.mode, stat.size].map(String).join(':'); } +// The two backends below differ in one decisive way, and it is worth stating +// plainly because the security properties are not the same. +// +// Linux ANCHORS. A name is resolved through /proc/self/fd//, which +// starts the walk at the inode the descriptor holds, so a parent that is renamed +// away cannot be traversed at all: the descriptor keeps pointing at the original +// directory and the impostor planted at the same name is simply never reached. +// +// macOS VERIFIES. Node cannot resolve a name relative to a descriptor there — +// /dev/fd/ is not a magic link (it stats as the directory but every attempt +// to traverse a child through it returns ENOENT), and fcntl F_GETPATH is a +// name-cache snapshot rather than a live anchor. So the Darwin backend resolves +// lexically, holds an open descriptor on every element of the chain, and proves +// before and after each operation that the path chain still names exactly the +// inodes it is holding. That DETECTS a swapped parent and aborts the write; it +// does not make the swap impossible the way the Linux path does. A swap landing +// inside the window between a check and the call it guards is caught by the +// following check, after the fact, rather than being unreachable. +// +// Every other platform gets neither and is refused outright. function requireDescriptorAnchoring() { - if ( - process.platform !== 'linux' || - fs.constants.O_DIRECTORY === undefined || - fs.constants.O_NOFOLLOW === undefined || - !fs.existsSync('/proc/self/fd') - ) { - throw new Error( - 'Safe generated-plan writes require Linux /proc/self/fd and O_DIRECTORY/O_NOFOLLOW; refusing an unanchored write', - ); + const directoryFlagsAvailable = + fs.constants.O_DIRECTORY !== undefined && fs.constants.O_NOFOLLOW !== undefined; + if (process.platform === 'linux') { + if (!directoryFlagsAvailable || !fs.existsSync('/proc/self/fd')) { + throw new Error( + 'Safe generated-plan writes require Linux /proc/self/fd and O_DIRECTORY/O_NOFOLLOW; refusing an unanchored write', + ); + } + return; } + if (process.platform === 'darwin') { + if (!directoryFlagsAvailable) { + throw new Error( + 'Safe generated-plan writes require macOS O_DIRECTORY/O_NOFOLLOW; refusing an unverified write', + ); + } + return; + } + throw new Error( + `Safe generated-plan writes require Linux /proc/self/fd or macOS O_DIRECTORY/O_NOFOLLOW; ${process.platform} offers neither, so refusing an unanchored write`, + ); } function descriptorPath(fd, childName) { @@ -900,157 +1023,352 @@ function descriptorPath(fd, childName) { return childName === undefined ? base : path.join(base, childName); } -function externalDescriptorPath(fd, childName) { - const base = `/proc/${process.pid}/fd/${fd}`; - return childName === undefined ? base : path.join(base, childName); +// Directory opens are plain O_RDONLY|O_DIRECTORY|O_NOFOLLOW|O_CLOEXEC on both +// platforms, and deliberately nothing else. +// +// O_NOFOLLOW_ANY (macOS 11+) used to be ORed in here on the theory that XNU +// ignores unrecognized open flag bits, so it would be inert where unsupported. +// That was wrong: combined with O_DIRECTORY macOS rejects it outright with +// EINVAL, and every directory open on Darwin failed. It is gone and is not +// coming back behind a probe or a degrade-on-EINVAL path — the per-component +// O_NOFOLLOW walk is what delivers the guarantee. Rust's cap-std, the closest +// reference implementation of this problem, has not adopted O_NOFOLLOW_ANY +// either (their issue #179 is still open). +function openVerifiedDirectory(absolute, flags) { + return fs.openSync(absolute, flags); } -const RENAME_NOREPLACE_SCRIPT = String.raw` -import ctypes -import errno -import os -import sys - -libc = ctypes.CDLL(None, use_errno=True) -try: - renameat2 = libc.renameat2 -except AttributeError: - print("libc does not expose renameat2", file=sys.stderr) - raise SystemExit(125) - -renameat2.argtypes = [ctypes.c_int, ctypes.c_char_p, ctypes.c_int, ctypes.c_char_p, ctypes.c_uint] -renameat2.restype = ctypes.c_int -result = renameat2(-100, os.fsencode(sys.argv[1]), -100, os.fsencode(sys.argv[2]), 1) -if result != 0: - error_number = ctypes.get_errno() - error_name = errno.errorcode.get(error_number, "UNKNOWN") - print(f"renameat2 RENAME_NOREPLACE failed: {error_name}: {os.strerror(error_number)}", file=sys.stderr) - raise SystemExit(17 if error_number == errno.EEXIST else 126) -`; - -let atomicMoverPath; - -function spawnHeldExecutable(executable, args, options) { - const before = fs.fstatSync(executable.fd, { bigint: true }); - if (!before.isFile() || statIdentity(before) !== executable.identity) { - throw new Error('Validated Python executable changed before invocation'); - } - const result = spawnSync('/proc/self/fd/3', args, { - ...options, - stdio: ['ignore', 'pipe', 'pipe', executable.fd], - }); - const after = fs.fstatSync(executable.fd, { bigint: true }); - assertStableIdentity(before, after, 'validated Python executable'); - return result; +// File opens additionally get O_NONBLOCK, which directory opens do not need: +// it stops a FIFO swapped in at the target name from wedging the process on +// open. The identity comparison that follows rejects the FIFO anyway, but only +// if we ever get as far as running it. +function openVerifiedFile(absolute, flags, mode) { + const nonBlocking = flags | (fs.constants.O_NONBLOCK ?? 0); + return mode === undefined + ? fs.openSync(absolute, nonBlocking) + : fs.openSync(absolute, nonBlocking, mode); } -function validatedPathExecutable(candidate) { - if (!path.isAbsolute(candidate)) return null; - const candidateDirectory = path.dirname(candidate); - let resolvedDirectory; - let resolved; - let directoryStats; - let executableStat; +// The publish primitive, identical on both platforms. +// +// link() is the portable no-replace publish: it fails with EEXIST if the +// destination name is taken — by a regular file, by a directory, or by a symlink, +// live or dangling — and it never follows that symlink to clobber its target. +// It also works where renameat2(RENAME_NOREPLACE) does not, notably v9fs, which +// is why the WSL2 9p case that used to fail every time now works. +// +// The published file is the same inode as the temporary, so every identity +// comparison the callers already make still holds, and validateCommittedPlan +// becomes strictly stronger: it compares the destination against the exact inode +// whose bytes were fsynced. +// +// On Linux both paths are /proc/self/fd//, so the publish is anchored +// to the held parent descriptors exactly like every other operation. +// link(2) BUGS: "On NFS filesystems, the return code may be wrong in case the NFS +// server performs the link creation and dies before it can say so. Use stat(2) to +// find out if the link got created." open(2) NOTES gives the remedy this +// implements: on a reported failure, stat the source and see whether its link +// count reached 2. A false positive would need someone to have hardlinked a +// 16-random-byte name inside a directory we hold open — and validateCommittedPlan +// still proves the destination is the exact temporary inode afterwards. +function linkCreatedDespiteError(sourcePath) { try { - resolvedDirectory = fs.realpathSync(candidateDirectory); - resolved = fs.realpathSync(candidate); - const resolvedExecutableDirectory = fs.realpathSync(path.dirname(resolved)); - directoryStats = [...new Set([resolvedDirectory, resolvedExecutableDirectory])].map( - (directory) => fs.statSync(directory), - ); - executableStat = fs.lstatSync(resolved); - fs.accessSync(resolved, fs.constants.X_OK); + return fs.statSync(sourcePath, { bigint: true }).nlink === 2n; } catch { - return null; + return false; } - if ( - directoryStats.some((stat) => !stat.isDirectory()) || - !executableStat.isFile() || - executableStat.isSymbolicLink() - ) { - return null; - } - const uid = typeof process.getuid === 'function' ? process.getuid() : null; - const trustedOwner = (stat) => uid === null || stat.uid === 0 || stat.uid === uid; - if ( - directoryStats.some((stat) => !trustedOwner(stat) || (stat.mode & 0o022) !== 0) || - !trustedOwner(executableStat) || - (executableStat.mode & 0o022) !== 0 - ) { - return null; - } - return resolved; } -function resolveAtomicMover() { - if (atomicMoverPath) return atomicMoverPath; - const candidates = new Set(); - for (const entry of (process.env.PATH ?? '').split(path.delimiter)) { - if (entry && path.isAbsolute(entry)) candidates.add(path.join(entry, 'python3')); - } - for (const entry of ['/usr/local/bin/python3', '/usr/bin/python3', '/bin/python3']) { - candidates.add(entry); - } - for (const candidate of candidates) { - const resolved = validatedPathExecutable(candidate); - if (!resolved) continue; - let fd; - try { - fd = fs.openSync( - resolved, - fs.constants.O_RDONLY | fs.constants.O_NOFOLLOW | (fs.constants.O_CLOEXEC ?? 0), - ); - } catch { - continue; +function linkNoReplace(sourcePath, destinationPath) { + try { + fs.linkSync(sourcePath, destinationPath); + } catch (error) { + // Callers treat "destination taken" as a distinct outcome, not a failure. + if (error?.code === 'EEXIST') return false; + if (!linkCreatedDespiteError(sourcePath)) { + // FAT, Coda, and some SMB/FUSE/virtiofs mounts have no hardlinks at all. + // Git falls back to rename here, but git can afford to lose collision + // detection because its objects are content-addressed; a plan destination + // is a plain name, so a replacing rename would silently clobber whatever + // is already there. Refuse loudly instead. + if (error?.code === 'EPERM' || error?.code === 'ENOTSUP' || error?.code === 'EMLINK') { + throw new Error( + `Generated-plan publication requires hard links, which this filesystem refused (${error.code}); refusing to fall back to a replacing rename`, + ); + } + throw error; } - const opened = fs.fstatSync(fd, { bigint: true }); - const executable = { fd, identity: statIdentity(opened), resolved }; - const version = spawnHeldExecutable( - executable, - ['-I', '-S', '-c', 'import sys; print(sys.version_info[0])'], - { - encoding: 'utf8', - env: { ...process.env, LANG: 'C', LC_ALL: 'C' }, - timeout: 10_000, - windowsHide: true, - }, - ); - if (version.status === 0 && version.stdout.trim() === '3') { - atomicMoverPath = executable; - return executable; - } - fs.closeSync(fd); } - throw new Error( - 'Safe generated-plan publication requires a trusted absolute Python 3 PATH candidate with libc renameat2 support', - ); -} - -function atomicMoveNoReplace(source, destination) { - const mover = resolveAtomicMover(); - const result = spawnHeldExecutable( - mover, - ['-I', '-S', '-c', RENAME_NOREPLACE_SCRIPT, source, destination], - { - encoding: 'utf8', - env: { ...process.env, LANG: 'C', LC_ALL: 'C' }, - timeout: 10_000, - windowsHide: true, - }, - ); - if (result.error) throw result.error; - if (result.status === 17) return false; - if (result.status !== 0) { - throw new Error( - `Atomic no-replace move failed (${result.status}): ${(result.stderr ?? '').trim()}`, - ); + try { + fs.unlinkSync(sourcePath); + } catch { + // The link succeeded, so the plan IS published. A temporary name left behind + // is a stray file, not an unpublished plan: reporting it as a failure would + // be a lie, and rolling back would unpublish a plan that is already live. } return true; } -function lstatOptional(absolute) { +// A directory holder is anything that owns a verified chain: a plan-parent +// handle, a ref's parent directory, or an absence guard. Two arrays describe it, +// both root-first and the same length — `chain` records each element's expected +// path and dev/ino/mode, and `descriptors` holds an open descriptor on each. +// +// Holding those descriptors is load-bearing rather than decorative. dev/ino/mode +// is unique only among *live* inodes: an inode number freed by an rmdir is handed +// straight back to the next mkdir, so a replacement directory can reproduce a +// recorded identity exactly. An open descriptor pins the inode, so the number +// cannot be recycled for as long as the holder exists. +function verifyPinnedDescriptors(holder) { + const { chain, descriptors } = holder; + if (!Array.isArray(descriptors) || descriptors.length !== chain.length) { + throw new Error('Generated-plan parent chain is missing the descriptors that pin it'); + } + chain.forEach((item, index) => { + const pinned = fs.fstatSync(descriptors[index], { bigint: true }); + if (!pinned.isDirectory() || stableDirectoryIdentity(pinned) !== item.identity) { + throw new Error('Generated-plan parent descriptor changed during the write'); + } + }); +} + +function verifyLexicalChain(holder) { + for (const item of holder.chain) { + let lexical; + try { + lexical = fs.lstatSync(item.expectedPath, { bigint: true }); + } catch (error) { + if (error?.code !== 'ENOENT' && error?.code !== 'ENOTDIR') throw error; + // A parent renamed out from under us is a mismatch, not a missing file: + // reporting the raw ENOENT would leak an unrelated-looking error out of a + // check whose whole job is to say the chain no longer holds. + throw new Error('Generated-plan lexical parent no longer matches its directory descriptor'); + } + if ( + lexical.isSymbolicLink() || + !lexical.isDirectory() || + stableDirectoryIdentity(lexical) !== item.identity + ) { + throw new Error('Generated-plan lexical parent no longer matches its directory descriptor'); + } + } +} + +// The whole platform seam, in five methods. Everything else an operation does is +// identical on both platforms and lives in the shared functions below. +// +// Only two things actually differ: how a name becomes a path, and what guard +// wraps the operation that uses it. +// +// Linux ANCHORS. /proc/self/fd// starts the walk at the inode the +// descriptor holds, so a parent renamed away cannot be traversed at all and the +// guard is a no-op — there is nothing left to verify. +// +// macOS VERIFIES. It resolves lexically, so before and after every operation it +// proves that each element of the path chain still names the exact inode being +// held for it. That DETECTS a swapped parent and aborts; it does not make the +// swap impossible. A swap landing inside the window is caught by the trailing +// check, after the fact, rather than being unreachable. The check runs after a +// failure too, because a verdict observed through a chain that has since changed +// is not a verdict. +const LINUX_ANCHORING = { + childPath(dirHandle, childName) { + return descriptorPath(dirHandle.fd, childName); + }, + verified(holders, run) { + return run(); + }, + descriptorMatchesChild(fd, expectedPath) { + return fs.realpathSync.native(descriptorPath(fd)) === expectedPath; + }, + parentStillResolves(parentHandle) { + return fs.realpathSync.native(descriptorPath(parentHandle.fd)) === parentHandle.expectedPath; + }, + verifyAbsentChild(guard) { + if (absentChildIsPresent(guard.ref)) { + throw new Error(`${guard.repoPath} appeared before evidence materialization completed`); + } + }, +}; + +const DARWIN_ANCHORING = { + childPath(dirHandle, childName) { + return path.join(dirHandle.expectedPath, childName); + }, + verified(holders, run) { + const list = Array.isArray(holders) ? holders : [holders]; + const proveChain = () => { + for (const holder of list) { + verifyPinnedDescriptors(holder); + verifyLexicalChain(holder); + } + }; + proveChain(); + let value; + try { + value = run(); + } catch (error) { + proveChain(); + throw error; + } + proveChain(); + return value; + }, + descriptorMatchesChild(fd, _expectedPath, childStat) { + // There is no live fd-to-path oracle on macOS (F_GETPATH is a name-cache + // snapshot, not an anchor), so escape is decided the other way round: the + // name was just resolved under a verified chain, and the descriptor opened + // from it counts only if it is that same inode. + const opened = fs.fstatSync(fd, { bigint: true }); + return ( + opened.isDirectory() && stableDirectoryIdentity(opened) === stableDirectoryIdentity(childStat) + ); + }, + parentStillResolves(parentHandle) { + // Both halves are needed: a directory renamed away keeps its inode, so the + // descriptors alone still match and only the lexical half notices it moved. + try { + verifyPinnedDescriptors(parentHandle); + verifyLexicalChain(parentHandle); + } catch { + return false; + } + return true; + }, + verifyAbsentChild(guard) { + let present; + try { + present = DARWIN_ANCHORING.verified(guard.handle, () => absentChildIsPresent(guard.ref)); + } catch (error) { + // A chain that no longer holds makes the absence verdict meaningless, and + // the caller reports that as the anchor changing rather than as a stray + // parent-descriptor error. Linux cannot reach this: its guard is a no-op. + throw new Error( + `Absence anchor changed for ${guard.repoPath}: ${ + error instanceof Error ? error.message : String(error) + }`, + ); + } + if (present) { + throw new Error(`${guard.repoPath} appeared before evidence materialization completed`); + } + }, +}; + +const ANCHORING_BACKENDS = new Map([ + ['linux', LINUX_ANCHORING], + ['darwin', DARWIN_ANCHORING], +]); + +function anchoringBackend() { + const backend = ANCHORING_BACKENDS.get(process.platform); + if (!backend) { + // requireDescriptorAnchoring normally refuses first; this is the same answer + // from the other side, so an unsupported platform can never fall through to + // whichever backend happened to be the ternary's default. + throw new Error( + `No generated-plan anchoring backend for ${process.platform}; refusing an unanchored write`, + ); + } + return backend; +} + +// Open, fstat, compare, close on mismatch. The descriptor never escapes this +// function unless it refers to the inode the caller already verified by name, so +// a lexical open that landed anywhere else cannot be used by accident. On Linux +// the comparison passes trivially — the /proc walk already resolved from the +// held parent — and costs one fstat to keep the guarantee structural rather than +// dependent on which backend is in play. +function adoptVerifiedFile(ref, expectedStat, flags) { + const fd = openVerifiedFile(ref.path, flags); + let opened; try { - return fs.lstatSync(absolute, { bigint: true }); + opened = fs.fstatSync(fd, { bigint: true }); + } catch (error) { + fs.closeSync(fd); + throw error; + } + if (stableFileIdentity(opened) !== stableFileIdentity(expectedStat)) { + fs.closeSync(fd); + return null; + } + return fd; +} + +function absentChildIsPresent(ref) { + try { + fs.lstatSync(ref.path, { bigint: true }); + } catch (error) { + if (error?.code === 'ENOENT') return false; + throw error; + } + return true; +} + +// The operations. Each is the same on both platforms; only the guard differs. +function lstatChild(ref) { + return anchoringBackend().verified(ref.dir, () => fs.lstatSync(ref.path, { bigint: true })); +} + +function openChildRead(ref, flags, expectedStat) { + return anchoringBackend().verified(ref.dir, () => { + const fd = adoptVerifiedFile(ref, expectedStat, flags); + if (fd === null) { + throw new Error(`${ref.name} was replaced between its verified stat and its no-follow open`); + } + return fd; + }); +} + +function createChild(ref, flags, mode) { + // O_CREAT|O_EXCL|O_NOFOLLOW is atomic at the leaf, so the only thing the guard + // has to cover is which directory the leaf landed in. + return anchoringBackend().verified(ref.dir, () => openVerifiedFile(ref.path, flags, mode)); +} + +function mkdirChild(ref, mode) { + anchoringBackend().verified(ref.dir, () => fs.mkdirSync(ref.path, { mode })); +} + +function publishNoReplace(sourceRef, destinationRef) { + return anchoringBackend().verified([sourceRef.dir, destinationRef.dir], () => + linkNoReplace(sourceRef.path, destinationRef.path), + ); +} + +// The single place a name becomes a path, and therefore the right place to +// enforce that a name is one ordinary component. +// +// A trailing separator is the sharp edge here, not a tidiness concern: +// open(path, O_NOFOLLOW) FOLLOWS a symlink when path ends in "/" — the trap +// behind CVE-2026-39822 / golang/go#79005, which let os.Root escape its own +// root. path.join preserves that trailing slash, so a component carrying one +// would turn every no-follow open in this file into a following one. +// normalizeRepoPath already rejects such components upstream; this is the +// chokepoint that makes it true for every caller, including the generated +// temporary and vault names that never pass through it. +function anchoredChild(dirHandle, childName) { + if ( + typeof childName !== 'string' || + childName === '' || + childName === '.' || + childName === '..' || + childName.includes('/') || + childName.includes('\\') || + childName.includes('\0') + ) { + throw new Error(`Refusing to resolve ${JSON.stringify(childName)} as a single path component`); + } + return { + dir: dirHandle, + name: childName, + path: anchoringBackend().childPath(dirHandle, childName), + }; +} + +function lstatAnchoredOptional(ref) { + try { + return lstatChild(ref); } catch (error) { if (error?.code === 'ENOENT' || error?.code === 'ENOTDIR') return null; throw error; @@ -1063,39 +1381,37 @@ function openPlanParent( { createMissing = true, purpose = 'Generated-plan' } = {}, ) { requireDescriptorAnchoring(); - const flags = - fs.constants.O_RDONLY | - fs.constants.O_DIRECTORY | - fs.constants.O_NOFOLLOW | - (fs.constants.O_CLOEXEC ?? 0); + // Root-first and index-aligned with `chain`: verifyPinnedDescriptors relies on + // that, and the descriptors are what pin each recorded inode against reuse. const descriptors = []; try { - let currentFd = fs.openSync(repo, flags); + let currentFd = openVerifiedDirectory(repo, ANCHORED_DIRECTORY_FLAGS); descriptors.push(currentFd); const rootStat = fs.fstatSync(currentFd, { bigint: true }); const chain = [{ expectedPath: repo, identity: stableDirectoryIdentity(rootStat) }]; + let currentHandle = { fd: currentFd, expectedPath: repo, chain, descriptors }; const traversed = []; for (const component of parentComponents) { traversed.push(component); - const anchoredChild = descriptorPath(currentFd, component); + const child = anchoredChild(currentHandle, component); let childStat; let created = false; try { - childStat = fs.lstatSync(anchoredChild, { bigint: true }); + childStat = lstatChild(child); } catch (error) { if (error?.code !== 'ENOENT' && error?.code !== 'ENOTDIR') throw error; if (!createMissing) { throw new Error(`${purpose} parent does not exist: ${traversed.join('/')}`); } - fs.mkdirSync(anchoredChild, { mode: 0o755 }); - childStat = fs.lstatSync(anchoredChild, { bigint: true }); + mkdirChild(child, 0o755); + childStat = lstatChild(child); created = true; } if (childStat.isSymbolicLink() || !childStat.isDirectory()) { throw new Error(`${purpose} parent is not a real directory: ${traversed.join('/')}`); } const parentFd = currentFd; - const childFd = fs.openSync(anchoredChild, flags); + const childFd = openVerifiedDirectory(child.path, ANCHORED_DIRECTORY_FLAGS); descriptors.push(childFd); currentFd = childFd; if (created) { @@ -1103,18 +1419,16 @@ function openPlanParent( fs.fsyncSync(parentFd); } const expected = path.join(repo, ...traversed); - const actual = fs.realpathSync(descriptorPath(currentFd)); - if (actual !== expected) { + if (!anchoringBackend().descriptorMatchesChild(currentFd, expected, childStat)) { throw new Error(`${purpose} parent escaped the repository: ${traversed.join('/')}`); } const openedStat = fs.fstatSync(currentFd, { bigint: true }); chain.push({ expectedPath: expected, identity: stableDirectoryIdentity(openedStat) }); + currentHandle = { fd: currentFd, expectedPath: expected, chain, descriptors }; } - const stat = fs.fstatSync(currentFd, { bigint: true }); return { descriptors, fd: currentFd, - identity: stableDirectoryIdentity(stat), expectedPath: path.join(repo, ...parentComponents), chain, }; @@ -1134,9 +1448,16 @@ function closeDescriptors(descriptors) { } } +// A handle's identity IS its chain leaf's identity. Storing it twice meant two +// fstats a line apart and a re-stamp helper to keep them agreeing; deriving it +// removes both. +function handleIdentity(handle) { + return handle.chain[handle.chain.length - 1].identity; +} + function resolveGitDirectory(repo) { const result = git(repo, ['rev-parse', '--absolute-git-dir']); - return fs.realpathSync(decodeUtf8(result.stdout, 'Git administrative directory').trim()); + return fs.realpathSync.native(decodeUtf8(result.stdout, 'Git administrative directory').trim()); } function openBackupVault(repo, { createMissing = true } = {}) { @@ -1147,9 +1468,12 @@ function openBackupVault(repo, { createMissing = true } = {}) { }); fs.fchmodSync(handle.fd, 0o700); fs.fsyncSync(handle.fd); - const stat = fs.fstatSync(handle.fd, { bigint: true }); - handle.identity = stableDirectoryIdentity(stat); - handle.chain[handle.chain.length - 1].identity = handle.identity; + // mode is part of every directory identity, so hardening the vault changes the + // identity the chain recorded for it; without this the next verification would + // reject the directory it just hardened. + handle.chain[handle.chain.length - 1].identity = stableDirectoryIdentity( + fs.fstatSync(handle.fd, { bigint: true }), + ); return { ...handle, gitDirectory }; } @@ -1157,33 +1481,28 @@ function validatePlanParent(parentHandle) { const descriptorStat = fs.fstatSync(parentHandle.fd, { bigint: true }); if ( !descriptorStat.isDirectory() || - stableDirectoryIdentity(descriptorStat) !== parentHandle.identity + stableDirectoryIdentity(descriptorStat) !== handleIdentity(parentHandle) ) { throw new Error('Generated-plan parent descriptor changed during the write'); } - const descriptorRealPath = fs.realpathSync(descriptorPath(parentHandle.fd)); - if (descriptorRealPath !== parentHandle.expectedPath) { + if (!anchoringBackend().parentStillResolves(parentHandle)) { throw new Error('Generated-plan parent moved or was replaced during the write'); } - for (const item of parentHandle.chain) { - const lexicalStat = fs.lstatSync(item.expectedPath, { bigint: true }); - if ( - lexicalStat.isSymbolicLink() || - !lexicalStat.isDirectory() || - stableDirectoryIdentity(lexicalStat) !== item.identity - ) { - throw new Error('Generated-plan lexical parent no longer matches its directory descriptor'); - } - } + // Both halves come from the shared helpers rather than being restated here: an + // earlier hand-copy of the lexical loop lost verifyLexicalChain's ENOENT/ENOTDIR + // translation, so a renamed parent could surface a raw errno from a function + // with a dozen call sites. + verifyPinnedDescriptors(parentHandle); + verifyLexicalChain(parentHandle); } function inspectPlanDestination( - finalPath, + finalRef, { replace, expectedIdentity, mustBeAbsent = false } = {}, ) { let stat; try { - stat = fs.lstatSync(finalPath, { bigint: true }); + stat = lstatChild(finalRef); } catch (error) { if (error?.code === 'ENOENT') { if (expectedIdentity) throw new Error('Generated plan disappeared during the write'); @@ -1201,19 +1520,17 @@ function inspectPlanDestination( if (expectedIdentity && identity !== expectedIdentity) { throw new Error('Generated plan changed during the write'); } - return identity; + return stat; } -function openExistingPlanDestination(finalPath, replace) { - const identity = inspectPlanDestination(finalPath, { replace }); - if (identity === null) { +function openExistingPlanDestination(finalRef, replace) { + const stat = inspectPlanDestination(finalRef, { replace }); + if (stat === null) { if (replace) throw new Error('Deepen mode requires an existing generated plan to replace'); return { fd: undefined, identity: null, stableIdentity: null }; } - const fd = fs.openSync( - finalPath, - fs.constants.O_RDONLY | fs.constants.O_NOFOLLOW | (fs.constants.O_CLOEXEC ?? 0), - ); + const identity = statIdentity(stat); + const fd = openChildRead(finalRef, VERIFIED_READ_FLAGS, stat); try { const opened = fs.fstatSync(fd, { bigint: true }); if (!opened.isFile() || statIdentity(opened) !== identity) { @@ -1264,8 +1581,8 @@ function hashOpenFile(fd, label) { }; } -function validateCommittedPlan(finalPath, tempFd, expectedTemp, testHooks) { - const before = fs.lstatSync(finalPath, { bigint: true }); +function validateCommittedPlan(finalRef, tempFd, expectedTemp, testHooks) { + const before = lstatChild(finalRef); if ( before.isSymbolicLink() || !before.isFile() || @@ -1273,19 +1590,16 @@ function validateCommittedPlan(finalPath, tempFd, expectedTemp, testHooks) { ) { throw new Error('Generated-plan destination failed its first post-write identity check'); } - const finalFd = fs.openSync( - finalPath, - fs.constants.O_RDONLY | fs.constants.O_NOFOLLOW | (fs.constants.O_CLOEXEC ?? 0), - ); + const finalFd = openChildRead(finalRef, VERIFIED_READ_FLAGS, before); try { const opened = fs.fstatSync(finalFd, { bigint: true }); if (!opened.isFile() || stableFileIdentity(opened) !== expectedTemp.identity) { throw new Error('Generated-plan destination changed while its no-follow descriptor opened'); } - testHooks?.afterFinalOpen?.({ fd: finalFd, finalPath }); + testHooks?.afterFinalOpen?.({ fd: finalFd, finalPath: finalRef.path }); const committedViaTemp = hashOpenFile(tempFd, 'generated-plan committed file'); const committedViaPath = hashOpenFile(finalFd, 'generated-plan destination descriptor'); - const after = fs.lstatSync(finalPath, { bigint: true }); + const after = lstatChild(finalRef); const openedAfter = fs.fstatSync(finalFd, { bigint: true }); if ( after.isSymbolicLink() || @@ -1320,22 +1634,19 @@ function copyOpenFile(sourceFd, destinationFd, label) { return after; } -function openVerifiedPathFile(absolute, label) { - const before = fs.lstatSync(absolute, { bigint: true }); +function openVerifiedAnchoredFile(ref, label, knownStat) { + const before = knownStat ?? lstatChild(ref); if (before.isSymbolicLink() || !before.isFile()) { throw new Error(`${label} is not a regular no-follow file`); } - const fd = fs.openSync( - absolute, - fs.constants.O_RDONLY | fs.constants.O_NOFOLLOW | (fs.constants.O_CLOEXEC ?? 0), - ); + const fd = openChildRead(ref, VERIFIED_READ_FLAGS, before); try { const opened = fs.fstatSync(fd, { bigint: true }); if (!opened.isFile() || stableFileIdentity(opened) !== stableFileIdentity(before)) { throw new Error(`${label} changed while its descriptor opened`); } const layer = hashOpenFile(fd, label); - const after = fs.lstatSync(absolute, { bigint: true }); + const after = lstatChild(ref); if (after.isSymbolicLink() || !after.isFile() || stableFileIdentity(after) !== layer.identity) { throw new Error(`${label} changed after verification`); } @@ -1358,10 +1669,10 @@ export function readPlanSafely({ repo: repoInput, generatedPlanPath, testHooks } let fd; try { validatePlanParent(parentHandle); - const finalPath = descriptorPath(parentHandle.fd, finalName); + const finalRef = anchoredChild(parentHandle, finalName); let before; try { - before = fs.lstatSync(finalPath, { bigint: true }); + before = lstatChild(finalRef); } catch (error) { if (error?.code === 'ENOENT' || error?.code === 'ENOTDIR') { throw new Error(`Loaded plan does not exist: ${generatedPlan}`); @@ -1371,15 +1682,12 @@ export function readPlanSafely({ repo: repoInput, generatedPlanPath, testHooks } if (before.isSymbolicLink() || !before.isFile()) { throw new Error('Loaded plan must be a regular file, never a symlink'); } - fd = fs.openSync( - finalPath, - fs.constants.O_RDONLY | fs.constants.O_NOFOLLOW | (fs.constants.O_CLOEXEC ?? 0), - ); + fd = openChildRead(finalRef, VERIFIED_READ_FLAGS, before); const opened = fs.fstatSync(fd, { bigint: true }); if (!opened.isFile() || statIdentity(opened) !== statIdentity(before)) { throw new Error('Loaded plan changed while its no-follow descriptor opened'); } - testHooks?.afterPlanOpen?.({ fd, finalPath }); + testHooks?.afterPlanOpen?.({ fd, finalPath: finalRef.path }); const chunks = []; let total = 0; const buffer = Buffer.allocUnsafe(64 * 1024); @@ -1394,7 +1702,7 @@ export function readPlanSafely({ repo: repoInput, generatedPlanPath, testHooks } decodeUtf8(contents, 'loaded plan'); const after = fs.fstatSync(fd, { bigint: true }); assertStableIdentity(opened, after, 'loaded plan'); - const pathAfter = fs.lstatSync(finalPath, { bigint: true }); + const pathAfter = lstatChild(finalRef); if ( pathAfter.isSymbolicLink() || !pathAfter.isFile() || @@ -1419,24 +1727,22 @@ function artifactGitPath(name) { return `gitnexus-plan-backups/${name}`; } -function verifyVaultArtifactFromFreshRoot(repo, gitPath, expectedLayer) { - const components = gitPath.split('/'); - if (components.length !== 2 || components[0] !== 'gitnexus-plan-backups') { - throw new Error(`Invalid Git-admin artifact path: ${gitPath}`); - } +function verifyVaultArtifactFromFreshRoot(repo, name, expectedLayer) { const freshVault = openBackupVault(repo, { createMissing: false }); try { validatePlanParent(freshVault); - const opened = openVerifiedPathFile( - descriptorPath(freshVault.fd, components[1]), - `Git-admin artifact ${gitPath}`, + const opened = openVerifiedAnchoredFile( + anchoredChild(freshVault, name), + `Git-admin artifact ${artifactGitPath(name)}`, ); try { if ( opened.layer.identity !== expectedLayer.identity || opened.layer.digest !== expectedLayer.digest ) { - throw new Error(`Git-admin artifact changed before fresh-root verification: ${gitPath}`); + throw new Error( + `Git-admin artifact changed before fresh-root verification: ${artifactGitPath(name)}`, + ); } } finally { fs.closeSync(opened.fd); @@ -1449,16 +1755,8 @@ function verifyVaultArtifactFromFreshRoot(repo, gitPath, expectedLayer) { function createVaultCopyFromFd(repo, vault, sourceFd, role) { validatePlanParent(vault); const name = `.gitnexus-plan-${role}-${process.pid}-${randomBytes(16).toString('hex')}.bak`; - const absolute = descriptorPath(vault.fd, name); - const destinationFd = fs.openSync( - absolute, - fs.constants.O_RDWR | - fs.constants.O_CREAT | - fs.constants.O_EXCL | - fs.constants.O_NOFOLLOW | - (fs.constants.O_CLOEXEC ?? 0), - 0o600, - ); + const artifact = anchoredChild(vault, name); + const destinationFd = createChild(artifact, VERIFIED_CREATE_FLAGS, 0o600); let destination; try { const sourceStat = copyOpenFile(sourceFd, destinationFd, role); @@ -1469,7 +1767,7 @@ function createVaultCopyFromFd(repo, vault, sourceFd, role) { if (source.size !== destination.size || source.digest !== destination.digest) { throw new Error(`${role} vault copy does not match its held source descriptor`); } - const pathStat = fs.lstatSync(absolute, { bigint: true }); + const pathStat = lstatChild(artifact); if ( pathStat.isSymbolicLink() || !pathStat.isFile() || @@ -1481,24 +1779,15 @@ function createVaultCopyFromFd(repo, vault, sourceFd, role) { } finally { fs.closeSync(destinationFd); } - const gitPath = artifactGitPath(name); - verifyVaultArtifactFromFreshRoot(repo, gitPath, destination); - return { role, gitPath, layer: destination }; + verifyVaultArtifactFromFreshRoot(repo, name, destination); + return { role, gitPath: artifactGitPath(name), layer: destination }; } function createVaultCopyFromBytes(repo, vault, contents, role) { validatePlanParent(vault); const name = `.gitnexus-plan-${role}-${process.pid}-${randomBytes(16).toString('hex')}.bak`; - const absolute = descriptorPath(vault.fd, name); - const fd = fs.openSync( - absolute, - fs.constants.O_RDWR | - fs.constants.O_CREAT | - fs.constants.O_EXCL | - fs.constants.O_NOFOLLOW | - (fs.constants.O_CLOEXEC ?? 0), - 0o600, - ); + const artifact = anchoredChild(vault, name); + const fd = createChild(artifact, VERIFIED_CREATE_FLAGS, 0o600); let layer; try { writeAll(fd, contents); @@ -1508,7 +1797,7 @@ function createVaultCopyFromBytes(repo, vault, contents, role) { if (layer.size !== BigInt(contents.length) || layer.digest !== sha256(contents)) { throw new Error(`${role} vault copy does not match the intended plan bytes`); } - const pathStat = fs.lstatSync(absolute, { bigint: true }); + const pathStat = lstatChild(artifact); if ( pathStat.isSymbolicLink() || !pathStat.isFile() || @@ -1520,32 +1809,31 @@ function createVaultCopyFromBytes(repo, vault, contents, role) { } finally { fs.closeSync(fd); } - const gitPath = artifactGitPath(name); - verifyVaultArtifactFromFreshRoot(repo, gitPath, layer); - return { role, gitPath, layer }; + verifyVaultArtifactFromFreshRoot(repo, name, layer); + return { role, gitPath: artifactGitPath(name), layer }; } function movePathToVault(repo, sourceHandle, sourceName, vault, role) { - const source = descriptorPath(sourceHandle.fd, sourceName); - if (!lstatOptional(source)) return null; + const source = anchoredChild(sourceHandle, sourceName); + if (!lstatAnchoredOptional(source)) return null; const name = `.gitnexus-plan-${role}-${process.pid}-${randomBytes(16).toString('hex')}.bak`; - const destination = descriptorPath(vault.fd, name); - const moved = atomicMoveNoReplace( - externalDescriptorPath(sourceHandle.fd, sourceName), - externalDescriptorPath(vault.fd, name), - ); + const destination = anchoredChild(vault, name); + const moved = publishNoReplace(source, destination); if (!moved) throw new Error(`${role} preservation destination unexpectedly exists`); fs.fsyncSync(sourceHandle.fd); if (vault.fd !== sourceHandle.fd) fs.fsyncSync(vault.fd); - const sourceAfter = lstatOptional(source); - const destinationAfter = lstatOptional(destination); + const sourceAfter = lstatAnchoredOptional(source); + const destinationAfter = lstatAnchoredOptional(destination); if (sourceAfter || !destinationAfter) { throw new Error(`${role} could not be atomically moved into the Git-admin vault`); } - const opened = openVerifiedPathFile(destination, `${role} Git-admin artifact`); - const gitPath = artifactGitPath(name); - verifyVaultArtifactFromFreshRoot(repo, gitPath, opened.layer); - return { role, gitPath, layer: opened.layer, fd: opened.fd }; + const opened = openVerifiedAnchoredFile( + destination, + `${role} Git-admin artifact`, + destinationAfter, + ); + verifyVaultArtifactFromFreshRoot(repo, name, opened.layer); + return { role, gitPath: artifactGitPath(name), layer: opened.layer, fd: opened.fd }; } function formatPreservedArtifacts(artifacts) { @@ -1600,10 +1888,10 @@ export function writePlanSafely({ const finalName = components.pop(); let parentHandle; let vaultHandle; - let tempPath; + let tempRef; let tempName; let tempFd; - let finalPath; + let finalRef; let expectedTemp; let originalDestination; let priorBackup; @@ -1611,7 +1899,6 @@ export function writePlanSafely({ try { parentHandle = openPlanParent(repo, components); vaultHandle = openBackupVault(repo); - resolveAtomicMover(); const parentDevice = fs.fstatSync(parentHandle.fd, { bigint: true }).dev; const vaultDevice = fs.fstatSync(vaultHandle.fd, { bigint: true }).dev; if (parentDevice !== vaultDevice) { @@ -1622,19 +1909,11 @@ export function writePlanSafely({ testHooks?.afterParentOpen?.({ fd: parentHandle.fd, path: parentHandle.expectedPath }); validatePlanParent(parentHandle); validatePlanParent(vaultHandle); - finalPath = descriptorPath(parentHandle.fd, finalName); - originalDestination = openExistingPlanDestination(finalPath, shouldReplace); + finalRef = anchoredChild(parentHandle, finalName); + originalDestination = openExistingPlanDestination(finalRef, shouldReplace); tempName = `.gitnexus-plan-${process.pid}-${randomBytes(16).toString('hex')}.tmp`; - tempPath = descriptorPath(parentHandle.fd, tempName); - tempFd = fs.openSync( - tempPath, - fs.constants.O_RDWR | - fs.constants.O_CREAT | - fs.constants.O_EXCL | - fs.constants.O_NOFOLLOW | - (fs.constants.O_CLOEXEC ?? 0), - 0o600, - ); + tempRef = anchoredChild(parentHandle, tempName); + tempFd = createChild(tempRef, VERIFIED_CREATE_FLAGS, 0o600); writeAll(tempFd, contents); fs.fchmodSync(tempFd, 0o644); fs.fsyncSync(tempFd); @@ -1646,12 +1925,12 @@ export function writePlanSafely({ testHooks?.beforeRename?.({ fd: parentHandle.fd, path: parentHandle.expectedPath, - tempPath, + tempPath: tempRef.path, }); validatePlanParent(parentHandle); validatePlanParent(vaultHandle); validateOpenPlanDestination(originalDestination); - const tempPathStat = fs.lstatSync(tempPath, { bigint: true }); + const tempPathStat = lstatChild(tempRef); const currentTemp = hashOpenFile(tempFd, 'generated-plan temporary file'); if ( tempPathStat.isSymbolicLink() || @@ -1664,7 +1943,7 @@ export function writePlanSafely({ } if (shouldReplace) { - testHooks?.beforeBackupMove?.({ fd: parentHandle.fd, finalPath }); + testHooks?.beforeBackupMove?.({ fd: parentHandle.fd, finalPath: finalRef.path }); const originalLayer = hashOpenFile(originalDestination.fd, 'prior generated plan'); if (originalLayer.digest !== expectedDigest) { throw new Error( @@ -1673,7 +1952,7 @@ export function writePlanSafely({ } validatePlanParent(parentHandle); validateOpenPlanDestination(originalDestination); - inspectPlanDestination(finalPath, { + inspectPlanDestination(finalRef, { replace: true, expectedIdentity: originalDestination.identity, }); @@ -1691,20 +1970,20 @@ export function writePlanSafely({ ); throw new Error('Destination raced while the prior plan was moved into preservation'); } - if (lstatOptional(finalPath)) { + if (lstatAnchoredOptional(finalRef)) { throw new Error('Destination reappeared after the prior plan was preserved'); } } testHooks?.beforePublication?.({ fd: parentHandle.fd, - finalPath, - tempPath, + finalPath: finalRef.path, + tempPath: tempRef.path, replace: shouldReplace, }); validatePlanParent(parentHandle); validatePlanParent(vaultHandle); - const finalTempPathStat = fs.lstatSync(tempPath, { bigint: true }); + const finalTempPathStat = lstatChild(tempRef); const finalTemp = hashOpenFile(tempFd, 'generated-plan temporary file'); if ( finalTempPathStat.isSymbolicLink() || @@ -1715,19 +1994,25 @@ export function writePlanSafely({ ) { throw new Error('Generated-plan temporary path or content changed at publication'); } - atomicMoveNoReplace( - externalDescriptorPath(parentHandle.fd, tempName), - externalDescriptorPath(parentHandle.fd, finalName), - ); - if (lstatOptional(tempPath) || !lstatOptional(finalPath)) { + // link() reports the race itself; re-deriving that verdict from a later pair + // of stats would be both slower and weaker. + if (!publishNoReplace(tempRef, finalRef)) { throw new Error('Generated-plan publication was refused because the destination raced'); } + // link() creates a directory entry, so it needs the parent fsync that rename + // needed: the file's own bytes were fsynced through tempFd before this point, + // and this makes the name that now reaches them durable too. Skipping it is + // the step write-file-atomic omits and maildir, git and atomicwrites all + // mandate. + // + // Honest limitation: on macOS fsync is not a write barrier — the durable + // primitive there is fcntl(F_FULLFSYNC), which Node does not expose. A + // macOS plan write is therefore as durable as fsync makes it and no more. fs.fsyncSync(parentHandle.fd); - testHooks?.afterPublication?.({ fd: parentHandle.fd, finalPath }); - testHooks?.afterRename?.({ fd: parentHandle.fd, finalPath }); + testHooks?.afterPublication?.({ fd: parentHandle.fd, finalPath: finalRef.path }); validatePlanParent(parentHandle); validatePlanParent(vaultHandle); - validateCommittedPlan(finalPath, tempFd, expectedTemp, testHooks); + validateCommittedPlan(finalRef, tempFd, expectedTemp, testHooks); const receipt = { generated_plan_path: generatedPlan, bytes_written: contents.length }; if (priorBackup) receipt.prior_plan_backup_git_path = priorBackup.gitPath; return receipt; @@ -1848,6 +2133,11 @@ export function snapshotEvidence({ const headGuards = captureHeadGuards(repo); const dirty = initialDirty.records; const mutationGuards = []; + // Per-snapshot walk state: `absenceCache` owns every descriptor an absence + // anchor holds, deduplicated by repo-relative prefix and closed exactly once + // below; `guardedDirectories` keeps parent guarding to one stat per directory. + const absenceCache = new Map(); + const walkState = { absenceCache, guardedDirectories: new Set() }; try { testHooks?.afterAnchorCapture?.({ headCommit: head }); @@ -1862,7 +2152,9 @@ export function snapshotEvidence({ testHooks?.afterGitLayerLoad?.({ headCommit: head }); const globalEntries = [...dirty.values()] .filter((record) => record.path !== generatedPlan) - .map((record) => materializeRecord(repo, record, layers, mutationGuards, testHooks)); + .map((record) => + materializeRecord(repo, record, layers, mutationGuards, testHooks, walkState), + ); const citedEntries = [...normalizedCitations].sort(compareUtf8).map((repoPath) => { const status = dirty.get(repoPath) ?? { path: repoPath, @@ -1871,7 +2163,7 @@ export function snapshotEvidence({ rename_to: null, has_untracked: false, }; - const entry = materializeRecord(repo, status, layers, mutationGuards, testHooks); + const entry = materializeRecord(repo, status, layers, mutationGuards, testHooks, walkState); const present = Object.values(entry.object_kind).some((kind) => kind !== ABSENT); if (!present) entry.state = ABSENT; else if (entry.state === 'clean' && entry.object_kind.untracked !== ABSENT) { @@ -1906,21 +2198,13 @@ export function snapshotEvidence({ throw new Error(`${guard.absolute} changed before evidence materialization completed`); } } else if (guard.type === 'absence') { + // statIdentity is a strict superset of stableDirectoryIdentity on the + // same stat, so comparing both could only ever fire together. const parent = fs.fstatSync(guard.fd, { bigint: true }); - if ( - !parent.isDirectory() || - stableDirectoryIdentity(parent) !== guard.parentIdentity || - statIdentity(parent) !== guard.parentMutationIdentity - ) { + if (!parent.isDirectory() || statIdentity(parent) !== guard.parentMutationIdentity) { throw new Error(`Absence anchor changed for ${guard.repoPath}`); } - try { - fs.lstatSync(descriptorPath(guard.fd, guard.childName), { bigint: true }); - } catch (error) { - if (error?.code === 'ENOENT') continue; - throw error; - } - throw new Error(`${guard.repoPath} appeared before evidence materialization completed`); + anchoringBackend().verifyAbsentChild(guard); } } for (const guard of headGuards) verifyControlFile(guard); @@ -1955,12 +2239,10 @@ export function snapshotEvidence({ cited_path_manifest: citedEntries, }; } finally { - const closed = new Set(); - for (const guard of mutationGuards) { - if (guard.type !== 'absence' || closed.has(guard.fd)) continue; - closed.add(guard.fd); + // One entry per distinct anchored directory, so one close per descriptor. + for (const handle of absenceCache.values()) { try { - fs.closeSync(guard.fd); + fs.closeSync(handle.fd); } catch { // Preserve the primary snapshot result/error. } diff --git a/gitnexus/skills/gitnexus-refactoring.md b/gitnexus/skills/gitnexus-refactoring.md index 2dbb71ca0..4f10bbc6a 100644 --- a/gitnexus/skills/gitnexus-refactoring.md +++ b/gitnexus/skills/gitnexus-refactoring.md @@ -87,6 +87,11 @@ detect_changes({scope: "all"}) → Risk: MEDIUM ``` +`partial: true` (a graph query failed) or `truncated: true` (the changed-symbol +listing was capped) means the result is short of the truth: a short or empty +list is not proof that only the expected files changed. Re-run it rather than +treat the refactor as verified. + **cypher** — custom reference queries: ```cypher diff --git a/gitnexus/skills/gitnexus-work/SKILL.md b/gitnexus/skills/gitnexus-work/SKILL.md index 4f7856ea7..f9baab16a 100644 --- a/gitnexus/skills/gitnexus-work/SKILL.md +++ b/gitnexus/skills/gitnexus-work/SKILL.md @@ -216,7 +216,10 @@ Work through plan §7 step by step, in order. For each step: `detect_changes` → commit as one unbroken sequence from the repository root — interleaving other work between the gate and the commit is how the gate gets skipped. Unexpected - affected flows → investigate before committing, not after. + affected flows → investigate before committing, not after. A result + flagged `partial` (a graph query failed) or `truncated` (the symbol + listing was capped) blocks the commit the same way: the gate did not + see every changed symbol, so re-run it rather than read it as clean. A relationship-affecting implementation edit or commit invalidates the procedure's prior proof. The next step must perform the required inter-step diff --git a/gitnexus/skills/gitnexus-work/references/evidence-provenance.md b/gitnexus/skills/gitnexus-work/references/evidence-provenance.md index c686599da..3df5a046d 100644 --- a/gitnexus/skills/gitnexus-work/references/evidence-provenance.md +++ b/gitnexus/skills/gitnexus-work/references/evidence-provenance.md @@ -98,8 +98,11 @@ excluded. ## Safe existing-plan read contract -`read-plan` fails closed unless Linux `/proc/self/fd`, `O_DIRECTORY`, and -`O_NOFOLLOW` are available. It resolves the exact Git top-level, opens the +`read-plan` fails closed unless the host platform can resolve names against a +held directory descriptor: Linux `/proc/self/fd` with `O_DIRECTORY` and +`O_NOFOLLOW`, or macOS `O_DIRECTORY`/`O_NOFOLLOW`. Every other platform is +refused outright — an unverified read is not a degraded read, it is a different, +racy operation. It resolves the exact Git top-level, opens the repository root and every plan parent as held no-follow directory descriptors, rejects missing, symlink, non-directory, and escaping parents, and opens the leaf with `O_NOFOLLOW`. It reads at most 16 MiB from that held file descriptor, @@ -109,13 +112,17 @@ Neither Deepen nor work may parse bytes obtained before or outside this receipt. ## Safe generated-plan write contract -The writer fails closed unless Linux `/proc/self/fd`, `O_DIRECTORY`, -`O_NOFOLLOW`, and Python 3 with libc `renameat2(RENAME_NOREPLACE)` support are -available. Python may live in `/usr/local`, a Nix profile, or another absolute -PATH directory, but the helper accepts only a resolved executable and -containing directory owned by root or the current user and not writable by -group/other. The resolved executable is opened without following links and -invoked through that held descriptor. Relative PATH entries are ignored. The plan parent and the +The writer fails closed unless the host platform offers `O_DIRECTORY` and +`O_NOFOLLOW`, plus `/proc/self/fd` on Linux. It spawns no interpreter and loads +no native code: publication is `link(2)`, which is atomic, fails `EEXIST` when +the destination name is taken, and refuses a symlinked destination without +following it — the same no-replace guarantee `renameat2(RENAME_NOREPLACE)` and +`renameatx_np(RENAME_EXCL)` provide, available through `fs.linkSync` on every +supported platform. The temporary name is unlinked once the link succeeds; the +published file is the same inode the writer created and verified, so every +identity check downstream holds by construction. A link that succeeds followed +by an unlink that fails leaves the plan published and is reported as success, +because it is one. The plan parent and the repository's Git-admin directory must also share a filesystem. It resolves the target repository's exact Git top-level, opens that root and every destination parent as held no-follow directory descriptors, creates missing @@ -128,15 +135,45 @@ The writer creates a random exclusive temporary file relative to the held final parent descriptor and keeps its no-follow descriptor open. It writes and flushes the bytes, binds the temporary name to the opened inode, and hashes the open file before publication. Immediately before publication it revalidates -the parent and the temporary path, inode, size, and digest. Publication uses an -atomic no-replace move relative to the held directory descriptor. Initial mode -therefore cannot overwrite a destination that appears after the absent check. +the parent and the temporary path, inode, size, and digest. Publication links +the temporary name to the destination relative to the held directory +descriptor, which fails rather than replaces if the destination is taken. +Initial mode therefore cannot overwrite a destination that appears after the +absent check. The writer then flushes the directory and revalidates the committed path by opening it with `O_NOFOLLOW`, hashing both the original temporary fd and the path-bound fd, and performing a second descriptor-anchored path identity check after hashing. A detected mutation or replacement aborts instead of accepting mixed-era output. +### Linux anchors, macOS verifies + +The two platforms reach the same destination by different proofs, and the +difference is real enough to state rather than smooth over. + +On Linux every name resolves through `/proc/self/fd//`, a magic link +the kernel resolves against the inode the descriptor already holds. The names +above it are never re-walked, so an attacker who renames a parent between the +check and the use cannot redirect the operation. The race is impossible, not +merely detected. + +macOS has no such path. `/dev/fd/` is a devfs node, not a magic link: it can +be opened, but nothing can be resolved through it. `open("/dev/fd//child")` +returns `ENOENT`, and `realpath` of it returns `/dev/fd/` rather than the +directory's path — measured on macOS 26, not inferred. Node exposes no `openat`, +no `dir_fd` parameter, and no FFI, so on macOS the writer resolves names +lexically with `O_NOFOLLOW` at every component, holds an open descriptor on +every directory in the chain for the whole operation, and proves before *and* +after each step that the chain still names exactly the inodes it is holding. +Holding the descriptors is what makes the recorded inode numbers trustworthy: +an open descriptor pins its inode, so a freed number cannot be recycled beneath +the walk. + +What that buys is detection rather than prevention. A parent swapped inside the +window between a check and its use is caught by the check that follows, and the +operation aborts having written nothing — but on Linux it could not have +happened at all. No published byte escapes verification on either platform. + `--replace` accepts only a pre-existing regular file and is reserved for Deepen; without it, accidental overwrite is rejected. It also requires the exact canonical `generated_plan_path` and `plan_digest` from the same session's diff --git a/gitnexus/skills/gitnexus-work/scripts/evidence-provenance.mjs b/gitnexus/skills/gitnexus-work/scripts/evidence-provenance.mjs index 181d2120b..793fe4cd8 100644 --- a/gitnexus/skills/gitnexus-work/scripts/evidence-provenance.mjs +++ b/gitnexus/skills/gitnexus-work/scripts/evidence-provenance.mjs @@ -479,11 +479,11 @@ function resolveOwnGitTopLevel(absolute) { if (result.status !== 0) return null; let topLevel; try { - topLevel = fs.realpathSync(decodeUtf8(result.stdout, 'nested repository root').trim()); + topLevel = fs.realpathSync.native(decodeUtf8(result.stdout, 'nested repository root').trim()); } catch { return null; } - return topLevel === fs.realpathSync(absolute) ? topLevel : null; + return topLevel === fs.realpathSync.native(absolute) ? topLevel : null; } function readOwnGitlinkHead(absolute) { @@ -616,17 +616,30 @@ function filesystemObject(absolute, expectedKind, mutationGuards, testHooks) { throw new Error(`Unsupported filesystem object at ${absolute}`); } -function guardPathParents(repo, repoPath, mutationGuards) { +// Every dirty path re-walks its own parents, and dirty paths overwhelmingly +// share them — the repository root is re-stat'ed once per path. `guarded` is +// per-snapshot and remembers which absolute directories already carry a guard, +// so each distinct directory is stat'ed and guarded exactly once. +// +// Keeping the first-seen identity is the conservative choice: verifyGuards +// re-checks every guard against the filesystem at the end, so a directory that +// changes after it was guarded still fails there. Skipping a re-stat cannot hide +// a change; it only avoids recording the same directory twice. +function guardPathParents(repo, repoPath, mutationGuards, guarded) { const components = repoPath.split('/'); let current = repo; - const rootStat = fs.lstatSync(repo, { bigint: true }); - mutationGuards.push({ - type: 'directory', - absolute: repo, - identity: stableDirectoryIdentity(rootStat), - }); + if (!guarded.has(repo)) { + guarded.add(repo); + mutationGuards.push({ + type: 'directory', + absolute: repo, + identity: stableDirectoryIdentity(fs.lstatSync(repo, { bigint: true })), + }); + } for (const component of components.slice(0, -1)) { current = path.join(current, component); + // Already proved a real directory and already guarded on an earlier path. + if (guarded.has(current)) continue; let stat; try { stat = fs.lstatSync(current, { bigint: true }); @@ -638,6 +651,7 @@ function guardPathParents(repo, repoPath, mutationGuards) { throw new Error(`Refusing to traverse symlink parent for ${repoPath}`); } if (!stat.isDirectory()) return; + guarded.add(current); mutationGuards.push({ type: 'directory', absolute: current, @@ -646,81 +660,153 @@ function guardPathParents(repo, repoPath, mutationGuards) { } } -function recordAnchoredAbsence(repo, repoPath, mutationGuards) { - requireDescriptorAnchoring(); - const flags = - fs.constants.O_RDONLY | - fs.constants.O_DIRECTORY | - fs.constants.O_NOFOLLOW | - (fs.constants.O_CLOEXEC ?? 0); - const descriptors = []; - let retainedFd; - try { - let currentFd = fs.openSync(repo, flags); - descriptors.push(currentFd); - const components = repoPath.split('/'); - for (let index = 0; index < components.length; index += 1) { - const component = components[index]; - const child = descriptorPath(currentFd, component); - let childStat; - try { - childStat = fs.lstatSync(child, { bigint: true }); - } catch (error) { - if (error?.code !== 'ENOENT' && error?.code !== 'ENOTDIR') throw error; - const parentStat = fs.fstatSync(currentFd, { bigint: true }); - if (!parentStat.isDirectory()) { - throw new Error(`Absence parent is no longer a directory for ${repoPath}`); - } - retainedFd = currentFd; - mutationGuards.push({ - type: 'absence', - fd: retainedFd, - childName: component, - repoPath, - parentIdentity: stableDirectoryIdentity(parentStat), - parentMutationIdentity: statIdentity(parentStat), - }); - for (const fd of descriptors) { - if (fd !== retainedFd) fs.closeSync(fd); - } - return; - } - if (index === components.length - 1) { - throw new Error(`${repoPath} appeared while its absence was being anchored`); - } - if (childStat.isSymbolicLink() || !childStat.isDirectory()) { - throw new Error(`Refusing a non-directory parent while anchoring absence for ${repoPath}`); - } - const nextFd = fs.openSync(child, flags); - descriptors.push(nextFd); - currentFd = nextFd; - } - throw new Error(`Could not anchor absence for ${repoPath}`); - } catch (error) { - for (const fd of descriptors) { - if (fd === retainedFd) continue; - try { - fs.closeSync(fd); - } catch { - // Preserve the primary absence-anchoring error. - } - } - throw error; +// A bound, not a bug: the absence cache deduplicates correctly and leaks nothing, +// but citedPaths is caller-supplied and unbounded, so a pathological snapshot +// could hold more descriptors than the process is allowed (macOS +// kern.maxfilesperproc is 24576). The peak precedes a `git` spawn, so exhaustion +// would surface as a git failure misreported as evidence instability. +// +// Refuse rather than evict: closing a cached descriptor would silently break the +// pinned chain of an absence guard that was already recorded against it, which is +// exactly the inode-recycling hole the pins exist to close. +const ABSENCE_ANCHOR_LIMITS = Object.freeze({ maxPinnedDirectories: 4096 }); + +// Every no-follow read and every exclusive create in this file uses one of these +// two, so a change lands in one place rather than in seven. +const VERIFIED_READ_FLAGS = + fs.constants.O_RDONLY | fs.constants.O_NOFOLLOW | (fs.constants.O_CLOEXEC ?? 0); +const VERIFIED_CREATE_FLAGS = + fs.constants.O_RDWR | + fs.constants.O_CREAT | + fs.constants.O_EXCL | + fs.constants.O_NOFOLLOW | + (fs.constants.O_CLOEXEC ?? 0); + +function requireAbsenceAnchorCapacity(cache) { + if (cache.size >= ABSENCE_ANCHOR_LIMITS.maxPinnedDirectories) { + throw new Error( + `Absence anchoring exceeds ${ABSENCE_ANCHOR_LIMITS.maxPinnedDirectories} pinned directories`, + ); } } -function materializeRecord(repo, statusRecord, layers, mutationGuards, testHooks) { +const ANCHORED_DIRECTORY_FLAGS = + fs.constants.O_RDONLY | + fs.constants.O_DIRECTORY | + fs.constants.O_NOFOLLOW | + (fs.constants.O_CLOEXEC ?? 0); + +// Every absence receipt is verified long after its walk returns, so the chain +// that produced it has to stay pinned until the snapshot ends — an unpinned inode +// number can be recycled by a replacement directory that then reproduces the +// recorded identity exactly. Absent cited paths overwhelmingly share prefixes, so +// the walked directories are cached per snapshot and keyed by repo-relative +// prefix: one open descriptor and one anchored walk per distinct directory rather +// than per path. snapshotEvidence owns every descriptor in this cache and closes +// each exactly once; guards only borrow them for verification. +function anchoredAbsenceRoot(repo, cache) { + const cached = cache.get(''); + if (cached) return cached; + requireAbsenceAnchorCapacity(cache); + const fd = openVerifiedDirectory(repo, ANCHORED_DIRECTORY_FLAGS); + const handle = { + fd, + expectedPath: repo, + chain: [ + { expectedPath: repo, identity: stableDirectoryIdentity(fs.fstatSync(fd, { bigint: true })) }, + ], + descriptors: [fd], + }; + cache.set('', handle); + return handle; +} + +function recordAnchoredAbsence(repo, repoPath, mutationGuards, cache) { + requireDescriptorAnchoring(); + const components = repoPath.split('/'); + let handle = anchoredAbsenceRoot(repo, cache); + let prefix = ''; + for (let index = 0; index < components.length; index += 1) { + const component = components[index]; + const isFinal = index === components.length - 1; + prefix = prefix === '' ? component : `${prefix}/${component}`; + // The final component is always re-checked against the filesystem: it is the + // one whose absence is being recorded, and a cached answer would be a stale + // one. Only the prefix directories are reused. + const cached = isFinal ? undefined : cache.get(prefix); + if (cached) { + handle = cached; + continue; + } + const child = anchoredChild(handle, component); + let childStat; + try { + childStat = lstatChild(child); + } catch (error) { + if (error?.code !== 'ENOENT' && error?.code !== 'ENOTDIR') throw error; + const parentStat = fs.fstatSync(handle.fd, { bigint: true }); + if (!parentStat.isDirectory()) { + throw new Error(`Absence parent is no longer a directory for ${repoPath}`); + } + mutationGuards.push({ + type: 'absence', + // The handle is the holder the guard verifies against, and `ref` is the + // child path already built through the anchoredChild chokepoint — the + // guard must never re-derive that name itself. + handle, + ref: child, + fd: handle.fd, + repoPath, + parentMutationIdentity: statIdentity(parentStat), + }); + return; + } + if (isFinal) { + throw new Error(`${repoPath} appeared while its absence was being anchored`); + } + if (childStat.isSymbolicLink() || !childStat.isDirectory()) { + throw new Error(`Refusing a non-directory parent while anchoring absence for ${repoPath}`); + } + requireAbsenceAnchorCapacity(cache); + const childFd = openVerifiedDirectory(child.path, ANCHORED_DIRECTORY_FLAGS); + const expectedPath = path.join(handle.expectedPath, component); + let next; + try { + if (!anchoringBackend().descriptorMatchesChild(childFd, expectedPath, childStat)) { + throw new Error( + `Absence parent descriptor does not match its verified inode for ${repoPath}`, + ); + } + next = { + fd: childFd, + expectedPath, + chain: [...handle.chain, { expectedPath, identity: stableDirectoryIdentity(childStat) }], + descriptors: [...handle.descriptors, childFd], + }; + } catch (error) { + fs.closeSync(childFd); + throw error; + } + cache.set(prefix, next); + handle = next; + } + throw new Error(`Could not anchor absence for ${repoPath}`); +} + +function materializeRecord(repo, statusRecord, layers, mutationGuards, testHooks, walkState) { const head = layers.head(statusRecord.path); const index = layers.index(statusRecord.path); const expectedKind = index.kind === 'gitlink' || head.kind === 'gitlink' ? 'gitlink' : null; - guardPathParents(repo, statusRecord.path, mutationGuards); + guardPathParents(repo, statusRecord.path, mutationGuards, walkState.guardedDirectories); const filesystem = filesystemObject( path.join(repo, ...statusRecord.path.split('/')), expectedKind, mutationGuards, testHooks, ); - if (filesystem.kind === ABSENT) recordAnchoredAbsence(repo, statusRecord.path, mutationGuards); + if (filesystem.kind === ABSENT) { + recordAnchoredAbsence(repo, statusRecord.path, mutationGuards, walkState.absenceCache); + } if (statusRecord.directory_hint && filesystem.kind !== 'directory') { throw new Error( `Git reported an embedded directory but found ${filesystem.kind}: ${statusRecord.path}`, @@ -789,9 +875,15 @@ export function serializeDirtyRecords(entries) { } function assertRepository(repoInput) { - const repo = fs.realpathSync(requireString(repoInput, 'repo')); + // realpathSync.native, not realpathSync: the JS resolver preserves a Windows + // 8.3 short component (C:\Users\RUNNER~1\...) while git always reports the long + // form, so the two would never compare equal and every caller would be told the + // worktree root is not the worktree root it just named. + const repo = fs.realpathSync.native(requireString(repoInput, 'repo')); const topLevelResult = git(repo, ['rev-parse', '--show-toplevel']); - const topLevel = fs.realpathSync(decodeUtf8(topLevelResult.stdout, 'repository root').trim()); + const topLevel = fs.realpathSync.native( + decodeUtf8(topLevelResult.stdout, 'repository root').trim(), + ); if (topLevel !== repo) throw new Error(`--repo must be the Git worktree root (${topLevel})`); return repo; } @@ -882,17 +974,48 @@ function stableFileIdentity(stat) { return [stat.dev, stat.ino, stat.mode, stat.size].map(String).join(':'); } +// The two backends below differ in one decisive way, and it is worth stating +// plainly because the security properties are not the same. +// +// Linux ANCHORS. A name is resolved through /proc/self/fd//, which +// starts the walk at the inode the descriptor holds, so a parent that is renamed +// away cannot be traversed at all: the descriptor keeps pointing at the original +// directory and the impostor planted at the same name is simply never reached. +// +// macOS VERIFIES. Node cannot resolve a name relative to a descriptor there — +// /dev/fd/ is not a magic link (it stats as the directory but every attempt +// to traverse a child through it returns ENOENT), and fcntl F_GETPATH is a +// name-cache snapshot rather than a live anchor. So the Darwin backend resolves +// lexically, holds an open descriptor on every element of the chain, and proves +// before and after each operation that the path chain still names exactly the +// inodes it is holding. That DETECTS a swapped parent and aborts the write; it +// does not make the swap impossible the way the Linux path does. A swap landing +// inside the window between a check and the call it guards is caught by the +// following check, after the fact, rather than being unreachable. +// +// Every other platform gets neither and is refused outright. function requireDescriptorAnchoring() { - if ( - process.platform !== 'linux' || - fs.constants.O_DIRECTORY === undefined || - fs.constants.O_NOFOLLOW === undefined || - !fs.existsSync('/proc/self/fd') - ) { - throw new Error( - 'Safe generated-plan writes require Linux /proc/self/fd and O_DIRECTORY/O_NOFOLLOW; refusing an unanchored write', - ); + const directoryFlagsAvailable = + fs.constants.O_DIRECTORY !== undefined && fs.constants.O_NOFOLLOW !== undefined; + if (process.platform === 'linux') { + if (!directoryFlagsAvailable || !fs.existsSync('/proc/self/fd')) { + throw new Error( + 'Safe generated-plan writes require Linux /proc/self/fd and O_DIRECTORY/O_NOFOLLOW; refusing an unanchored write', + ); + } + return; } + if (process.platform === 'darwin') { + if (!directoryFlagsAvailable) { + throw new Error( + 'Safe generated-plan writes require macOS O_DIRECTORY/O_NOFOLLOW; refusing an unverified write', + ); + } + return; + } + throw new Error( + `Safe generated-plan writes require Linux /proc/self/fd or macOS O_DIRECTORY/O_NOFOLLOW; ${process.platform} offers neither, so refusing an unanchored write`, + ); } function descriptorPath(fd, childName) { @@ -900,157 +1023,352 @@ function descriptorPath(fd, childName) { return childName === undefined ? base : path.join(base, childName); } -function externalDescriptorPath(fd, childName) { - const base = `/proc/${process.pid}/fd/${fd}`; - return childName === undefined ? base : path.join(base, childName); +// Directory opens are plain O_RDONLY|O_DIRECTORY|O_NOFOLLOW|O_CLOEXEC on both +// platforms, and deliberately nothing else. +// +// O_NOFOLLOW_ANY (macOS 11+) used to be ORed in here on the theory that XNU +// ignores unrecognized open flag bits, so it would be inert where unsupported. +// That was wrong: combined with O_DIRECTORY macOS rejects it outright with +// EINVAL, and every directory open on Darwin failed. It is gone and is not +// coming back behind a probe or a degrade-on-EINVAL path — the per-component +// O_NOFOLLOW walk is what delivers the guarantee. Rust's cap-std, the closest +// reference implementation of this problem, has not adopted O_NOFOLLOW_ANY +// either (their issue #179 is still open). +function openVerifiedDirectory(absolute, flags) { + return fs.openSync(absolute, flags); } -const RENAME_NOREPLACE_SCRIPT = String.raw` -import ctypes -import errno -import os -import sys - -libc = ctypes.CDLL(None, use_errno=True) -try: - renameat2 = libc.renameat2 -except AttributeError: - print("libc does not expose renameat2", file=sys.stderr) - raise SystemExit(125) - -renameat2.argtypes = [ctypes.c_int, ctypes.c_char_p, ctypes.c_int, ctypes.c_char_p, ctypes.c_uint] -renameat2.restype = ctypes.c_int -result = renameat2(-100, os.fsencode(sys.argv[1]), -100, os.fsencode(sys.argv[2]), 1) -if result != 0: - error_number = ctypes.get_errno() - error_name = errno.errorcode.get(error_number, "UNKNOWN") - print(f"renameat2 RENAME_NOREPLACE failed: {error_name}: {os.strerror(error_number)}", file=sys.stderr) - raise SystemExit(17 if error_number == errno.EEXIST else 126) -`; - -let atomicMoverPath; - -function spawnHeldExecutable(executable, args, options) { - const before = fs.fstatSync(executable.fd, { bigint: true }); - if (!before.isFile() || statIdentity(before) !== executable.identity) { - throw new Error('Validated Python executable changed before invocation'); - } - const result = spawnSync('/proc/self/fd/3', args, { - ...options, - stdio: ['ignore', 'pipe', 'pipe', executable.fd], - }); - const after = fs.fstatSync(executable.fd, { bigint: true }); - assertStableIdentity(before, after, 'validated Python executable'); - return result; +// File opens additionally get O_NONBLOCK, which directory opens do not need: +// it stops a FIFO swapped in at the target name from wedging the process on +// open. The identity comparison that follows rejects the FIFO anyway, but only +// if we ever get as far as running it. +function openVerifiedFile(absolute, flags, mode) { + const nonBlocking = flags | (fs.constants.O_NONBLOCK ?? 0); + return mode === undefined + ? fs.openSync(absolute, nonBlocking) + : fs.openSync(absolute, nonBlocking, mode); } -function validatedPathExecutable(candidate) { - if (!path.isAbsolute(candidate)) return null; - const candidateDirectory = path.dirname(candidate); - let resolvedDirectory; - let resolved; - let directoryStats; - let executableStat; +// The publish primitive, identical on both platforms. +// +// link() is the portable no-replace publish: it fails with EEXIST if the +// destination name is taken — by a regular file, by a directory, or by a symlink, +// live or dangling — and it never follows that symlink to clobber its target. +// It also works where renameat2(RENAME_NOREPLACE) does not, notably v9fs, which +// is why the WSL2 9p case that used to fail every time now works. +// +// The published file is the same inode as the temporary, so every identity +// comparison the callers already make still holds, and validateCommittedPlan +// becomes strictly stronger: it compares the destination against the exact inode +// whose bytes were fsynced. +// +// On Linux both paths are /proc/self/fd//, so the publish is anchored +// to the held parent descriptors exactly like every other operation. +// link(2) BUGS: "On NFS filesystems, the return code may be wrong in case the NFS +// server performs the link creation and dies before it can say so. Use stat(2) to +// find out if the link got created." open(2) NOTES gives the remedy this +// implements: on a reported failure, stat the source and see whether its link +// count reached 2. A false positive would need someone to have hardlinked a +// 16-random-byte name inside a directory we hold open — and validateCommittedPlan +// still proves the destination is the exact temporary inode afterwards. +function linkCreatedDespiteError(sourcePath) { try { - resolvedDirectory = fs.realpathSync(candidateDirectory); - resolved = fs.realpathSync(candidate); - const resolvedExecutableDirectory = fs.realpathSync(path.dirname(resolved)); - directoryStats = [...new Set([resolvedDirectory, resolvedExecutableDirectory])].map( - (directory) => fs.statSync(directory), - ); - executableStat = fs.lstatSync(resolved); - fs.accessSync(resolved, fs.constants.X_OK); + return fs.statSync(sourcePath, { bigint: true }).nlink === 2n; } catch { - return null; + return false; } - if ( - directoryStats.some((stat) => !stat.isDirectory()) || - !executableStat.isFile() || - executableStat.isSymbolicLink() - ) { - return null; - } - const uid = typeof process.getuid === 'function' ? process.getuid() : null; - const trustedOwner = (stat) => uid === null || stat.uid === 0 || stat.uid === uid; - if ( - directoryStats.some((stat) => !trustedOwner(stat) || (stat.mode & 0o022) !== 0) || - !trustedOwner(executableStat) || - (executableStat.mode & 0o022) !== 0 - ) { - return null; - } - return resolved; } -function resolveAtomicMover() { - if (atomicMoverPath) return atomicMoverPath; - const candidates = new Set(); - for (const entry of (process.env.PATH ?? '').split(path.delimiter)) { - if (entry && path.isAbsolute(entry)) candidates.add(path.join(entry, 'python3')); - } - for (const entry of ['/usr/local/bin/python3', '/usr/bin/python3', '/bin/python3']) { - candidates.add(entry); - } - for (const candidate of candidates) { - const resolved = validatedPathExecutable(candidate); - if (!resolved) continue; - let fd; - try { - fd = fs.openSync( - resolved, - fs.constants.O_RDONLY | fs.constants.O_NOFOLLOW | (fs.constants.O_CLOEXEC ?? 0), - ); - } catch { - continue; +function linkNoReplace(sourcePath, destinationPath) { + try { + fs.linkSync(sourcePath, destinationPath); + } catch (error) { + // Callers treat "destination taken" as a distinct outcome, not a failure. + if (error?.code === 'EEXIST') return false; + if (!linkCreatedDespiteError(sourcePath)) { + // FAT, Coda, and some SMB/FUSE/virtiofs mounts have no hardlinks at all. + // Git falls back to rename here, but git can afford to lose collision + // detection because its objects are content-addressed; a plan destination + // is a plain name, so a replacing rename would silently clobber whatever + // is already there. Refuse loudly instead. + if (error?.code === 'EPERM' || error?.code === 'ENOTSUP' || error?.code === 'EMLINK') { + throw new Error( + `Generated-plan publication requires hard links, which this filesystem refused (${error.code}); refusing to fall back to a replacing rename`, + ); + } + throw error; } - const opened = fs.fstatSync(fd, { bigint: true }); - const executable = { fd, identity: statIdentity(opened), resolved }; - const version = spawnHeldExecutable( - executable, - ['-I', '-S', '-c', 'import sys; print(sys.version_info[0])'], - { - encoding: 'utf8', - env: { ...process.env, LANG: 'C', LC_ALL: 'C' }, - timeout: 10_000, - windowsHide: true, - }, - ); - if (version.status === 0 && version.stdout.trim() === '3') { - atomicMoverPath = executable; - return executable; - } - fs.closeSync(fd); } - throw new Error( - 'Safe generated-plan publication requires a trusted absolute Python 3 PATH candidate with libc renameat2 support', - ); -} - -function atomicMoveNoReplace(source, destination) { - const mover = resolveAtomicMover(); - const result = spawnHeldExecutable( - mover, - ['-I', '-S', '-c', RENAME_NOREPLACE_SCRIPT, source, destination], - { - encoding: 'utf8', - env: { ...process.env, LANG: 'C', LC_ALL: 'C' }, - timeout: 10_000, - windowsHide: true, - }, - ); - if (result.error) throw result.error; - if (result.status === 17) return false; - if (result.status !== 0) { - throw new Error( - `Atomic no-replace move failed (${result.status}): ${(result.stderr ?? '').trim()}`, - ); + try { + fs.unlinkSync(sourcePath); + } catch { + // The link succeeded, so the plan IS published. A temporary name left behind + // is a stray file, not an unpublished plan: reporting it as a failure would + // be a lie, and rolling back would unpublish a plan that is already live. } return true; } -function lstatOptional(absolute) { +// A directory holder is anything that owns a verified chain: a plan-parent +// handle, a ref's parent directory, or an absence guard. Two arrays describe it, +// both root-first and the same length — `chain` records each element's expected +// path and dev/ino/mode, and `descriptors` holds an open descriptor on each. +// +// Holding those descriptors is load-bearing rather than decorative. dev/ino/mode +// is unique only among *live* inodes: an inode number freed by an rmdir is handed +// straight back to the next mkdir, so a replacement directory can reproduce a +// recorded identity exactly. An open descriptor pins the inode, so the number +// cannot be recycled for as long as the holder exists. +function verifyPinnedDescriptors(holder) { + const { chain, descriptors } = holder; + if (!Array.isArray(descriptors) || descriptors.length !== chain.length) { + throw new Error('Generated-plan parent chain is missing the descriptors that pin it'); + } + chain.forEach((item, index) => { + const pinned = fs.fstatSync(descriptors[index], { bigint: true }); + if (!pinned.isDirectory() || stableDirectoryIdentity(pinned) !== item.identity) { + throw new Error('Generated-plan parent descriptor changed during the write'); + } + }); +} + +function verifyLexicalChain(holder) { + for (const item of holder.chain) { + let lexical; + try { + lexical = fs.lstatSync(item.expectedPath, { bigint: true }); + } catch (error) { + if (error?.code !== 'ENOENT' && error?.code !== 'ENOTDIR') throw error; + // A parent renamed out from under us is a mismatch, not a missing file: + // reporting the raw ENOENT would leak an unrelated-looking error out of a + // check whose whole job is to say the chain no longer holds. + throw new Error('Generated-plan lexical parent no longer matches its directory descriptor'); + } + if ( + lexical.isSymbolicLink() || + !lexical.isDirectory() || + stableDirectoryIdentity(lexical) !== item.identity + ) { + throw new Error('Generated-plan lexical parent no longer matches its directory descriptor'); + } + } +} + +// The whole platform seam, in five methods. Everything else an operation does is +// identical on both platforms and lives in the shared functions below. +// +// Only two things actually differ: how a name becomes a path, and what guard +// wraps the operation that uses it. +// +// Linux ANCHORS. /proc/self/fd// starts the walk at the inode the +// descriptor holds, so a parent renamed away cannot be traversed at all and the +// guard is a no-op — there is nothing left to verify. +// +// macOS VERIFIES. It resolves lexically, so before and after every operation it +// proves that each element of the path chain still names the exact inode being +// held for it. That DETECTS a swapped parent and aborts; it does not make the +// swap impossible. A swap landing inside the window is caught by the trailing +// check, after the fact, rather than being unreachable. The check runs after a +// failure too, because a verdict observed through a chain that has since changed +// is not a verdict. +const LINUX_ANCHORING = { + childPath(dirHandle, childName) { + return descriptorPath(dirHandle.fd, childName); + }, + verified(holders, run) { + return run(); + }, + descriptorMatchesChild(fd, expectedPath) { + return fs.realpathSync.native(descriptorPath(fd)) === expectedPath; + }, + parentStillResolves(parentHandle) { + return fs.realpathSync.native(descriptorPath(parentHandle.fd)) === parentHandle.expectedPath; + }, + verifyAbsentChild(guard) { + if (absentChildIsPresent(guard.ref)) { + throw new Error(`${guard.repoPath} appeared before evidence materialization completed`); + } + }, +}; + +const DARWIN_ANCHORING = { + childPath(dirHandle, childName) { + return path.join(dirHandle.expectedPath, childName); + }, + verified(holders, run) { + const list = Array.isArray(holders) ? holders : [holders]; + const proveChain = () => { + for (const holder of list) { + verifyPinnedDescriptors(holder); + verifyLexicalChain(holder); + } + }; + proveChain(); + let value; + try { + value = run(); + } catch (error) { + proveChain(); + throw error; + } + proveChain(); + return value; + }, + descriptorMatchesChild(fd, _expectedPath, childStat) { + // There is no live fd-to-path oracle on macOS (F_GETPATH is a name-cache + // snapshot, not an anchor), so escape is decided the other way round: the + // name was just resolved under a verified chain, and the descriptor opened + // from it counts only if it is that same inode. + const opened = fs.fstatSync(fd, { bigint: true }); + return ( + opened.isDirectory() && stableDirectoryIdentity(opened) === stableDirectoryIdentity(childStat) + ); + }, + parentStillResolves(parentHandle) { + // Both halves are needed: a directory renamed away keeps its inode, so the + // descriptors alone still match and only the lexical half notices it moved. + try { + verifyPinnedDescriptors(parentHandle); + verifyLexicalChain(parentHandle); + } catch { + return false; + } + return true; + }, + verifyAbsentChild(guard) { + let present; + try { + present = DARWIN_ANCHORING.verified(guard.handle, () => absentChildIsPresent(guard.ref)); + } catch (error) { + // A chain that no longer holds makes the absence verdict meaningless, and + // the caller reports that as the anchor changing rather than as a stray + // parent-descriptor error. Linux cannot reach this: its guard is a no-op. + throw new Error( + `Absence anchor changed for ${guard.repoPath}: ${ + error instanceof Error ? error.message : String(error) + }`, + ); + } + if (present) { + throw new Error(`${guard.repoPath} appeared before evidence materialization completed`); + } + }, +}; + +const ANCHORING_BACKENDS = new Map([ + ['linux', LINUX_ANCHORING], + ['darwin', DARWIN_ANCHORING], +]); + +function anchoringBackend() { + const backend = ANCHORING_BACKENDS.get(process.platform); + if (!backend) { + // requireDescriptorAnchoring normally refuses first; this is the same answer + // from the other side, so an unsupported platform can never fall through to + // whichever backend happened to be the ternary's default. + throw new Error( + `No generated-plan anchoring backend for ${process.platform}; refusing an unanchored write`, + ); + } + return backend; +} + +// Open, fstat, compare, close on mismatch. The descriptor never escapes this +// function unless it refers to the inode the caller already verified by name, so +// a lexical open that landed anywhere else cannot be used by accident. On Linux +// the comparison passes trivially — the /proc walk already resolved from the +// held parent — and costs one fstat to keep the guarantee structural rather than +// dependent on which backend is in play. +function adoptVerifiedFile(ref, expectedStat, flags) { + const fd = openVerifiedFile(ref.path, flags); + let opened; try { - return fs.lstatSync(absolute, { bigint: true }); + opened = fs.fstatSync(fd, { bigint: true }); + } catch (error) { + fs.closeSync(fd); + throw error; + } + if (stableFileIdentity(opened) !== stableFileIdentity(expectedStat)) { + fs.closeSync(fd); + return null; + } + return fd; +} + +function absentChildIsPresent(ref) { + try { + fs.lstatSync(ref.path, { bigint: true }); + } catch (error) { + if (error?.code === 'ENOENT') return false; + throw error; + } + return true; +} + +// The operations. Each is the same on both platforms; only the guard differs. +function lstatChild(ref) { + return anchoringBackend().verified(ref.dir, () => fs.lstatSync(ref.path, { bigint: true })); +} + +function openChildRead(ref, flags, expectedStat) { + return anchoringBackend().verified(ref.dir, () => { + const fd = adoptVerifiedFile(ref, expectedStat, flags); + if (fd === null) { + throw new Error(`${ref.name} was replaced between its verified stat and its no-follow open`); + } + return fd; + }); +} + +function createChild(ref, flags, mode) { + // O_CREAT|O_EXCL|O_NOFOLLOW is atomic at the leaf, so the only thing the guard + // has to cover is which directory the leaf landed in. + return anchoringBackend().verified(ref.dir, () => openVerifiedFile(ref.path, flags, mode)); +} + +function mkdirChild(ref, mode) { + anchoringBackend().verified(ref.dir, () => fs.mkdirSync(ref.path, { mode })); +} + +function publishNoReplace(sourceRef, destinationRef) { + return anchoringBackend().verified([sourceRef.dir, destinationRef.dir], () => + linkNoReplace(sourceRef.path, destinationRef.path), + ); +} + +// The single place a name becomes a path, and therefore the right place to +// enforce that a name is one ordinary component. +// +// A trailing separator is the sharp edge here, not a tidiness concern: +// open(path, O_NOFOLLOW) FOLLOWS a symlink when path ends in "/" — the trap +// behind CVE-2026-39822 / golang/go#79005, which let os.Root escape its own +// root. path.join preserves that trailing slash, so a component carrying one +// would turn every no-follow open in this file into a following one. +// normalizeRepoPath already rejects such components upstream; this is the +// chokepoint that makes it true for every caller, including the generated +// temporary and vault names that never pass through it. +function anchoredChild(dirHandle, childName) { + if ( + typeof childName !== 'string' || + childName === '' || + childName === '.' || + childName === '..' || + childName.includes('/') || + childName.includes('\\') || + childName.includes('\0') + ) { + throw new Error(`Refusing to resolve ${JSON.stringify(childName)} as a single path component`); + } + return { + dir: dirHandle, + name: childName, + path: anchoringBackend().childPath(dirHandle, childName), + }; +} + +function lstatAnchoredOptional(ref) { + try { + return lstatChild(ref); } catch (error) { if (error?.code === 'ENOENT' || error?.code === 'ENOTDIR') return null; throw error; @@ -1063,39 +1381,37 @@ function openPlanParent( { createMissing = true, purpose = 'Generated-plan' } = {}, ) { requireDescriptorAnchoring(); - const flags = - fs.constants.O_RDONLY | - fs.constants.O_DIRECTORY | - fs.constants.O_NOFOLLOW | - (fs.constants.O_CLOEXEC ?? 0); + // Root-first and index-aligned with `chain`: verifyPinnedDescriptors relies on + // that, and the descriptors are what pin each recorded inode against reuse. const descriptors = []; try { - let currentFd = fs.openSync(repo, flags); + let currentFd = openVerifiedDirectory(repo, ANCHORED_DIRECTORY_FLAGS); descriptors.push(currentFd); const rootStat = fs.fstatSync(currentFd, { bigint: true }); const chain = [{ expectedPath: repo, identity: stableDirectoryIdentity(rootStat) }]; + let currentHandle = { fd: currentFd, expectedPath: repo, chain, descriptors }; const traversed = []; for (const component of parentComponents) { traversed.push(component); - const anchoredChild = descriptorPath(currentFd, component); + const child = anchoredChild(currentHandle, component); let childStat; let created = false; try { - childStat = fs.lstatSync(anchoredChild, { bigint: true }); + childStat = lstatChild(child); } catch (error) { if (error?.code !== 'ENOENT' && error?.code !== 'ENOTDIR') throw error; if (!createMissing) { throw new Error(`${purpose} parent does not exist: ${traversed.join('/')}`); } - fs.mkdirSync(anchoredChild, { mode: 0o755 }); - childStat = fs.lstatSync(anchoredChild, { bigint: true }); + mkdirChild(child, 0o755); + childStat = lstatChild(child); created = true; } if (childStat.isSymbolicLink() || !childStat.isDirectory()) { throw new Error(`${purpose} parent is not a real directory: ${traversed.join('/')}`); } const parentFd = currentFd; - const childFd = fs.openSync(anchoredChild, flags); + const childFd = openVerifiedDirectory(child.path, ANCHORED_DIRECTORY_FLAGS); descriptors.push(childFd); currentFd = childFd; if (created) { @@ -1103,18 +1419,16 @@ function openPlanParent( fs.fsyncSync(parentFd); } const expected = path.join(repo, ...traversed); - const actual = fs.realpathSync(descriptorPath(currentFd)); - if (actual !== expected) { + if (!anchoringBackend().descriptorMatchesChild(currentFd, expected, childStat)) { throw new Error(`${purpose} parent escaped the repository: ${traversed.join('/')}`); } const openedStat = fs.fstatSync(currentFd, { bigint: true }); chain.push({ expectedPath: expected, identity: stableDirectoryIdentity(openedStat) }); + currentHandle = { fd: currentFd, expectedPath: expected, chain, descriptors }; } - const stat = fs.fstatSync(currentFd, { bigint: true }); return { descriptors, fd: currentFd, - identity: stableDirectoryIdentity(stat), expectedPath: path.join(repo, ...parentComponents), chain, }; @@ -1134,9 +1448,16 @@ function closeDescriptors(descriptors) { } } +// A handle's identity IS its chain leaf's identity. Storing it twice meant two +// fstats a line apart and a re-stamp helper to keep them agreeing; deriving it +// removes both. +function handleIdentity(handle) { + return handle.chain[handle.chain.length - 1].identity; +} + function resolveGitDirectory(repo) { const result = git(repo, ['rev-parse', '--absolute-git-dir']); - return fs.realpathSync(decodeUtf8(result.stdout, 'Git administrative directory').trim()); + return fs.realpathSync.native(decodeUtf8(result.stdout, 'Git administrative directory').trim()); } function openBackupVault(repo, { createMissing = true } = {}) { @@ -1147,9 +1468,12 @@ function openBackupVault(repo, { createMissing = true } = {}) { }); fs.fchmodSync(handle.fd, 0o700); fs.fsyncSync(handle.fd); - const stat = fs.fstatSync(handle.fd, { bigint: true }); - handle.identity = stableDirectoryIdentity(stat); - handle.chain[handle.chain.length - 1].identity = handle.identity; + // mode is part of every directory identity, so hardening the vault changes the + // identity the chain recorded for it; without this the next verification would + // reject the directory it just hardened. + handle.chain[handle.chain.length - 1].identity = stableDirectoryIdentity( + fs.fstatSync(handle.fd, { bigint: true }), + ); return { ...handle, gitDirectory }; } @@ -1157,33 +1481,28 @@ function validatePlanParent(parentHandle) { const descriptorStat = fs.fstatSync(parentHandle.fd, { bigint: true }); if ( !descriptorStat.isDirectory() || - stableDirectoryIdentity(descriptorStat) !== parentHandle.identity + stableDirectoryIdentity(descriptorStat) !== handleIdentity(parentHandle) ) { throw new Error('Generated-plan parent descriptor changed during the write'); } - const descriptorRealPath = fs.realpathSync(descriptorPath(parentHandle.fd)); - if (descriptorRealPath !== parentHandle.expectedPath) { + if (!anchoringBackend().parentStillResolves(parentHandle)) { throw new Error('Generated-plan parent moved or was replaced during the write'); } - for (const item of parentHandle.chain) { - const lexicalStat = fs.lstatSync(item.expectedPath, { bigint: true }); - if ( - lexicalStat.isSymbolicLink() || - !lexicalStat.isDirectory() || - stableDirectoryIdentity(lexicalStat) !== item.identity - ) { - throw new Error('Generated-plan lexical parent no longer matches its directory descriptor'); - } - } + // Both halves come from the shared helpers rather than being restated here: an + // earlier hand-copy of the lexical loop lost verifyLexicalChain's ENOENT/ENOTDIR + // translation, so a renamed parent could surface a raw errno from a function + // with a dozen call sites. + verifyPinnedDescriptors(parentHandle); + verifyLexicalChain(parentHandle); } function inspectPlanDestination( - finalPath, + finalRef, { replace, expectedIdentity, mustBeAbsent = false } = {}, ) { let stat; try { - stat = fs.lstatSync(finalPath, { bigint: true }); + stat = lstatChild(finalRef); } catch (error) { if (error?.code === 'ENOENT') { if (expectedIdentity) throw new Error('Generated plan disappeared during the write'); @@ -1201,19 +1520,17 @@ function inspectPlanDestination( if (expectedIdentity && identity !== expectedIdentity) { throw new Error('Generated plan changed during the write'); } - return identity; + return stat; } -function openExistingPlanDestination(finalPath, replace) { - const identity = inspectPlanDestination(finalPath, { replace }); - if (identity === null) { +function openExistingPlanDestination(finalRef, replace) { + const stat = inspectPlanDestination(finalRef, { replace }); + if (stat === null) { if (replace) throw new Error('Deepen mode requires an existing generated plan to replace'); return { fd: undefined, identity: null, stableIdentity: null }; } - const fd = fs.openSync( - finalPath, - fs.constants.O_RDONLY | fs.constants.O_NOFOLLOW | (fs.constants.O_CLOEXEC ?? 0), - ); + const identity = statIdentity(stat); + const fd = openChildRead(finalRef, VERIFIED_READ_FLAGS, stat); try { const opened = fs.fstatSync(fd, { bigint: true }); if (!opened.isFile() || statIdentity(opened) !== identity) { @@ -1264,8 +1581,8 @@ function hashOpenFile(fd, label) { }; } -function validateCommittedPlan(finalPath, tempFd, expectedTemp, testHooks) { - const before = fs.lstatSync(finalPath, { bigint: true }); +function validateCommittedPlan(finalRef, tempFd, expectedTemp, testHooks) { + const before = lstatChild(finalRef); if ( before.isSymbolicLink() || !before.isFile() || @@ -1273,19 +1590,16 @@ function validateCommittedPlan(finalPath, tempFd, expectedTemp, testHooks) { ) { throw new Error('Generated-plan destination failed its first post-write identity check'); } - const finalFd = fs.openSync( - finalPath, - fs.constants.O_RDONLY | fs.constants.O_NOFOLLOW | (fs.constants.O_CLOEXEC ?? 0), - ); + const finalFd = openChildRead(finalRef, VERIFIED_READ_FLAGS, before); try { const opened = fs.fstatSync(finalFd, { bigint: true }); if (!opened.isFile() || stableFileIdentity(opened) !== expectedTemp.identity) { throw new Error('Generated-plan destination changed while its no-follow descriptor opened'); } - testHooks?.afterFinalOpen?.({ fd: finalFd, finalPath }); + testHooks?.afterFinalOpen?.({ fd: finalFd, finalPath: finalRef.path }); const committedViaTemp = hashOpenFile(tempFd, 'generated-plan committed file'); const committedViaPath = hashOpenFile(finalFd, 'generated-plan destination descriptor'); - const after = fs.lstatSync(finalPath, { bigint: true }); + const after = lstatChild(finalRef); const openedAfter = fs.fstatSync(finalFd, { bigint: true }); if ( after.isSymbolicLink() || @@ -1320,22 +1634,19 @@ function copyOpenFile(sourceFd, destinationFd, label) { return after; } -function openVerifiedPathFile(absolute, label) { - const before = fs.lstatSync(absolute, { bigint: true }); +function openVerifiedAnchoredFile(ref, label, knownStat) { + const before = knownStat ?? lstatChild(ref); if (before.isSymbolicLink() || !before.isFile()) { throw new Error(`${label} is not a regular no-follow file`); } - const fd = fs.openSync( - absolute, - fs.constants.O_RDONLY | fs.constants.O_NOFOLLOW | (fs.constants.O_CLOEXEC ?? 0), - ); + const fd = openChildRead(ref, VERIFIED_READ_FLAGS, before); try { const opened = fs.fstatSync(fd, { bigint: true }); if (!opened.isFile() || stableFileIdentity(opened) !== stableFileIdentity(before)) { throw new Error(`${label} changed while its descriptor opened`); } const layer = hashOpenFile(fd, label); - const after = fs.lstatSync(absolute, { bigint: true }); + const after = lstatChild(ref); if (after.isSymbolicLink() || !after.isFile() || stableFileIdentity(after) !== layer.identity) { throw new Error(`${label} changed after verification`); } @@ -1358,10 +1669,10 @@ export function readPlanSafely({ repo: repoInput, generatedPlanPath, testHooks } let fd; try { validatePlanParent(parentHandle); - const finalPath = descriptorPath(parentHandle.fd, finalName); + const finalRef = anchoredChild(parentHandle, finalName); let before; try { - before = fs.lstatSync(finalPath, { bigint: true }); + before = lstatChild(finalRef); } catch (error) { if (error?.code === 'ENOENT' || error?.code === 'ENOTDIR') { throw new Error(`Loaded plan does not exist: ${generatedPlan}`); @@ -1371,15 +1682,12 @@ export function readPlanSafely({ repo: repoInput, generatedPlanPath, testHooks } if (before.isSymbolicLink() || !before.isFile()) { throw new Error('Loaded plan must be a regular file, never a symlink'); } - fd = fs.openSync( - finalPath, - fs.constants.O_RDONLY | fs.constants.O_NOFOLLOW | (fs.constants.O_CLOEXEC ?? 0), - ); + fd = openChildRead(finalRef, VERIFIED_READ_FLAGS, before); const opened = fs.fstatSync(fd, { bigint: true }); if (!opened.isFile() || statIdentity(opened) !== statIdentity(before)) { throw new Error('Loaded plan changed while its no-follow descriptor opened'); } - testHooks?.afterPlanOpen?.({ fd, finalPath }); + testHooks?.afterPlanOpen?.({ fd, finalPath: finalRef.path }); const chunks = []; let total = 0; const buffer = Buffer.allocUnsafe(64 * 1024); @@ -1394,7 +1702,7 @@ export function readPlanSafely({ repo: repoInput, generatedPlanPath, testHooks } decodeUtf8(contents, 'loaded plan'); const after = fs.fstatSync(fd, { bigint: true }); assertStableIdentity(opened, after, 'loaded plan'); - const pathAfter = fs.lstatSync(finalPath, { bigint: true }); + const pathAfter = lstatChild(finalRef); if ( pathAfter.isSymbolicLink() || !pathAfter.isFile() || @@ -1419,24 +1727,22 @@ function artifactGitPath(name) { return `gitnexus-plan-backups/${name}`; } -function verifyVaultArtifactFromFreshRoot(repo, gitPath, expectedLayer) { - const components = gitPath.split('/'); - if (components.length !== 2 || components[0] !== 'gitnexus-plan-backups') { - throw new Error(`Invalid Git-admin artifact path: ${gitPath}`); - } +function verifyVaultArtifactFromFreshRoot(repo, name, expectedLayer) { const freshVault = openBackupVault(repo, { createMissing: false }); try { validatePlanParent(freshVault); - const opened = openVerifiedPathFile( - descriptorPath(freshVault.fd, components[1]), - `Git-admin artifact ${gitPath}`, + const opened = openVerifiedAnchoredFile( + anchoredChild(freshVault, name), + `Git-admin artifact ${artifactGitPath(name)}`, ); try { if ( opened.layer.identity !== expectedLayer.identity || opened.layer.digest !== expectedLayer.digest ) { - throw new Error(`Git-admin artifact changed before fresh-root verification: ${gitPath}`); + throw new Error( + `Git-admin artifact changed before fresh-root verification: ${artifactGitPath(name)}`, + ); } } finally { fs.closeSync(opened.fd); @@ -1449,16 +1755,8 @@ function verifyVaultArtifactFromFreshRoot(repo, gitPath, expectedLayer) { function createVaultCopyFromFd(repo, vault, sourceFd, role) { validatePlanParent(vault); const name = `.gitnexus-plan-${role}-${process.pid}-${randomBytes(16).toString('hex')}.bak`; - const absolute = descriptorPath(vault.fd, name); - const destinationFd = fs.openSync( - absolute, - fs.constants.O_RDWR | - fs.constants.O_CREAT | - fs.constants.O_EXCL | - fs.constants.O_NOFOLLOW | - (fs.constants.O_CLOEXEC ?? 0), - 0o600, - ); + const artifact = anchoredChild(vault, name); + const destinationFd = createChild(artifact, VERIFIED_CREATE_FLAGS, 0o600); let destination; try { const sourceStat = copyOpenFile(sourceFd, destinationFd, role); @@ -1469,7 +1767,7 @@ function createVaultCopyFromFd(repo, vault, sourceFd, role) { if (source.size !== destination.size || source.digest !== destination.digest) { throw new Error(`${role} vault copy does not match its held source descriptor`); } - const pathStat = fs.lstatSync(absolute, { bigint: true }); + const pathStat = lstatChild(artifact); if ( pathStat.isSymbolicLink() || !pathStat.isFile() || @@ -1481,24 +1779,15 @@ function createVaultCopyFromFd(repo, vault, sourceFd, role) { } finally { fs.closeSync(destinationFd); } - const gitPath = artifactGitPath(name); - verifyVaultArtifactFromFreshRoot(repo, gitPath, destination); - return { role, gitPath, layer: destination }; + verifyVaultArtifactFromFreshRoot(repo, name, destination); + return { role, gitPath: artifactGitPath(name), layer: destination }; } function createVaultCopyFromBytes(repo, vault, contents, role) { validatePlanParent(vault); const name = `.gitnexus-plan-${role}-${process.pid}-${randomBytes(16).toString('hex')}.bak`; - const absolute = descriptorPath(vault.fd, name); - const fd = fs.openSync( - absolute, - fs.constants.O_RDWR | - fs.constants.O_CREAT | - fs.constants.O_EXCL | - fs.constants.O_NOFOLLOW | - (fs.constants.O_CLOEXEC ?? 0), - 0o600, - ); + const artifact = anchoredChild(vault, name); + const fd = createChild(artifact, VERIFIED_CREATE_FLAGS, 0o600); let layer; try { writeAll(fd, contents); @@ -1508,7 +1797,7 @@ function createVaultCopyFromBytes(repo, vault, contents, role) { if (layer.size !== BigInt(contents.length) || layer.digest !== sha256(contents)) { throw new Error(`${role} vault copy does not match the intended plan bytes`); } - const pathStat = fs.lstatSync(absolute, { bigint: true }); + const pathStat = lstatChild(artifact); if ( pathStat.isSymbolicLink() || !pathStat.isFile() || @@ -1520,32 +1809,31 @@ function createVaultCopyFromBytes(repo, vault, contents, role) { } finally { fs.closeSync(fd); } - const gitPath = artifactGitPath(name); - verifyVaultArtifactFromFreshRoot(repo, gitPath, layer); - return { role, gitPath, layer }; + verifyVaultArtifactFromFreshRoot(repo, name, layer); + return { role, gitPath: artifactGitPath(name), layer }; } function movePathToVault(repo, sourceHandle, sourceName, vault, role) { - const source = descriptorPath(sourceHandle.fd, sourceName); - if (!lstatOptional(source)) return null; + const source = anchoredChild(sourceHandle, sourceName); + if (!lstatAnchoredOptional(source)) return null; const name = `.gitnexus-plan-${role}-${process.pid}-${randomBytes(16).toString('hex')}.bak`; - const destination = descriptorPath(vault.fd, name); - const moved = atomicMoveNoReplace( - externalDescriptorPath(sourceHandle.fd, sourceName), - externalDescriptorPath(vault.fd, name), - ); + const destination = anchoredChild(vault, name); + const moved = publishNoReplace(source, destination); if (!moved) throw new Error(`${role} preservation destination unexpectedly exists`); fs.fsyncSync(sourceHandle.fd); if (vault.fd !== sourceHandle.fd) fs.fsyncSync(vault.fd); - const sourceAfter = lstatOptional(source); - const destinationAfter = lstatOptional(destination); + const sourceAfter = lstatAnchoredOptional(source); + const destinationAfter = lstatAnchoredOptional(destination); if (sourceAfter || !destinationAfter) { throw new Error(`${role} could not be atomically moved into the Git-admin vault`); } - const opened = openVerifiedPathFile(destination, `${role} Git-admin artifact`); - const gitPath = artifactGitPath(name); - verifyVaultArtifactFromFreshRoot(repo, gitPath, opened.layer); - return { role, gitPath, layer: opened.layer, fd: opened.fd }; + const opened = openVerifiedAnchoredFile( + destination, + `${role} Git-admin artifact`, + destinationAfter, + ); + verifyVaultArtifactFromFreshRoot(repo, name, opened.layer); + return { role, gitPath: artifactGitPath(name), layer: opened.layer, fd: opened.fd }; } function formatPreservedArtifacts(artifacts) { @@ -1600,10 +1888,10 @@ export function writePlanSafely({ const finalName = components.pop(); let parentHandle; let vaultHandle; - let tempPath; + let tempRef; let tempName; let tempFd; - let finalPath; + let finalRef; let expectedTemp; let originalDestination; let priorBackup; @@ -1611,7 +1899,6 @@ export function writePlanSafely({ try { parentHandle = openPlanParent(repo, components); vaultHandle = openBackupVault(repo); - resolveAtomicMover(); const parentDevice = fs.fstatSync(parentHandle.fd, { bigint: true }).dev; const vaultDevice = fs.fstatSync(vaultHandle.fd, { bigint: true }).dev; if (parentDevice !== vaultDevice) { @@ -1622,19 +1909,11 @@ export function writePlanSafely({ testHooks?.afterParentOpen?.({ fd: parentHandle.fd, path: parentHandle.expectedPath }); validatePlanParent(parentHandle); validatePlanParent(vaultHandle); - finalPath = descriptorPath(parentHandle.fd, finalName); - originalDestination = openExistingPlanDestination(finalPath, shouldReplace); + finalRef = anchoredChild(parentHandle, finalName); + originalDestination = openExistingPlanDestination(finalRef, shouldReplace); tempName = `.gitnexus-plan-${process.pid}-${randomBytes(16).toString('hex')}.tmp`; - tempPath = descriptorPath(parentHandle.fd, tempName); - tempFd = fs.openSync( - tempPath, - fs.constants.O_RDWR | - fs.constants.O_CREAT | - fs.constants.O_EXCL | - fs.constants.O_NOFOLLOW | - (fs.constants.O_CLOEXEC ?? 0), - 0o600, - ); + tempRef = anchoredChild(parentHandle, tempName); + tempFd = createChild(tempRef, VERIFIED_CREATE_FLAGS, 0o600); writeAll(tempFd, contents); fs.fchmodSync(tempFd, 0o644); fs.fsyncSync(tempFd); @@ -1646,12 +1925,12 @@ export function writePlanSafely({ testHooks?.beforeRename?.({ fd: parentHandle.fd, path: parentHandle.expectedPath, - tempPath, + tempPath: tempRef.path, }); validatePlanParent(parentHandle); validatePlanParent(vaultHandle); validateOpenPlanDestination(originalDestination); - const tempPathStat = fs.lstatSync(tempPath, { bigint: true }); + const tempPathStat = lstatChild(tempRef); const currentTemp = hashOpenFile(tempFd, 'generated-plan temporary file'); if ( tempPathStat.isSymbolicLink() || @@ -1664,7 +1943,7 @@ export function writePlanSafely({ } if (shouldReplace) { - testHooks?.beforeBackupMove?.({ fd: parentHandle.fd, finalPath }); + testHooks?.beforeBackupMove?.({ fd: parentHandle.fd, finalPath: finalRef.path }); const originalLayer = hashOpenFile(originalDestination.fd, 'prior generated plan'); if (originalLayer.digest !== expectedDigest) { throw new Error( @@ -1673,7 +1952,7 @@ export function writePlanSafely({ } validatePlanParent(parentHandle); validateOpenPlanDestination(originalDestination); - inspectPlanDestination(finalPath, { + inspectPlanDestination(finalRef, { replace: true, expectedIdentity: originalDestination.identity, }); @@ -1691,20 +1970,20 @@ export function writePlanSafely({ ); throw new Error('Destination raced while the prior plan was moved into preservation'); } - if (lstatOptional(finalPath)) { + if (lstatAnchoredOptional(finalRef)) { throw new Error('Destination reappeared after the prior plan was preserved'); } } testHooks?.beforePublication?.({ fd: parentHandle.fd, - finalPath, - tempPath, + finalPath: finalRef.path, + tempPath: tempRef.path, replace: shouldReplace, }); validatePlanParent(parentHandle); validatePlanParent(vaultHandle); - const finalTempPathStat = fs.lstatSync(tempPath, { bigint: true }); + const finalTempPathStat = lstatChild(tempRef); const finalTemp = hashOpenFile(tempFd, 'generated-plan temporary file'); if ( finalTempPathStat.isSymbolicLink() || @@ -1715,19 +1994,25 @@ export function writePlanSafely({ ) { throw new Error('Generated-plan temporary path or content changed at publication'); } - atomicMoveNoReplace( - externalDescriptorPath(parentHandle.fd, tempName), - externalDescriptorPath(parentHandle.fd, finalName), - ); - if (lstatOptional(tempPath) || !lstatOptional(finalPath)) { + // link() reports the race itself; re-deriving that verdict from a later pair + // of stats would be both slower and weaker. + if (!publishNoReplace(tempRef, finalRef)) { throw new Error('Generated-plan publication was refused because the destination raced'); } + // link() creates a directory entry, so it needs the parent fsync that rename + // needed: the file's own bytes were fsynced through tempFd before this point, + // and this makes the name that now reaches them durable too. Skipping it is + // the step write-file-atomic omits and maildir, git and atomicwrites all + // mandate. + // + // Honest limitation: on macOS fsync is not a write barrier — the durable + // primitive there is fcntl(F_FULLFSYNC), which Node does not expose. A + // macOS plan write is therefore as durable as fsync makes it and no more. fs.fsyncSync(parentHandle.fd); - testHooks?.afterPublication?.({ fd: parentHandle.fd, finalPath }); - testHooks?.afterRename?.({ fd: parentHandle.fd, finalPath }); + testHooks?.afterPublication?.({ fd: parentHandle.fd, finalPath: finalRef.path }); validatePlanParent(parentHandle); validatePlanParent(vaultHandle); - validateCommittedPlan(finalPath, tempFd, expectedTemp, testHooks); + validateCommittedPlan(finalRef, tempFd, expectedTemp, testHooks); const receipt = { generated_plan_path: generatedPlan, bytes_written: contents.length }; if (priorBackup) receipt.prior_plan_backup_git_path = priorBackup.gitPath; return receipt; @@ -1848,6 +2133,11 @@ export function snapshotEvidence({ const headGuards = captureHeadGuards(repo); const dirty = initialDirty.records; const mutationGuards = []; + // Per-snapshot walk state: `absenceCache` owns every descriptor an absence + // anchor holds, deduplicated by repo-relative prefix and closed exactly once + // below; `guardedDirectories` keeps parent guarding to one stat per directory. + const absenceCache = new Map(); + const walkState = { absenceCache, guardedDirectories: new Set() }; try { testHooks?.afterAnchorCapture?.({ headCommit: head }); @@ -1862,7 +2152,9 @@ export function snapshotEvidence({ testHooks?.afterGitLayerLoad?.({ headCommit: head }); const globalEntries = [...dirty.values()] .filter((record) => record.path !== generatedPlan) - .map((record) => materializeRecord(repo, record, layers, mutationGuards, testHooks)); + .map((record) => + materializeRecord(repo, record, layers, mutationGuards, testHooks, walkState), + ); const citedEntries = [...normalizedCitations].sort(compareUtf8).map((repoPath) => { const status = dirty.get(repoPath) ?? { path: repoPath, @@ -1871,7 +2163,7 @@ export function snapshotEvidence({ rename_to: null, has_untracked: false, }; - const entry = materializeRecord(repo, status, layers, mutationGuards, testHooks); + const entry = materializeRecord(repo, status, layers, mutationGuards, testHooks, walkState); const present = Object.values(entry.object_kind).some((kind) => kind !== ABSENT); if (!present) entry.state = ABSENT; else if (entry.state === 'clean' && entry.object_kind.untracked !== ABSENT) { @@ -1906,21 +2198,13 @@ export function snapshotEvidence({ throw new Error(`${guard.absolute} changed before evidence materialization completed`); } } else if (guard.type === 'absence') { + // statIdentity is a strict superset of stableDirectoryIdentity on the + // same stat, so comparing both could only ever fire together. const parent = fs.fstatSync(guard.fd, { bigint: true }); - if ( - !parent.isDirectory() || - stableDirectoryIdentity(parent) !== guard.parentIdentity || - statIdentity(parent) !== guard.parentMutationIdentity - ) { + if (!parent.isDirectory() || statIdentity(parent) !== guard.parentMutationIdentity) { throw new Error(`Absence anchor changed for ${guard.repoPath}`); } - try { - fs.lstatSync(descriptorPath(guard.fd, guard.childName), { bigint: true }); - } catch (error) { - if (error?.code === 'ENOENT') continue; - throw error; - } - throw new Error(`${guard.repoPath} appeared before evidence materialization completed`); + anchoringBackend().verifyAbsentChild(guard); } } for (const guard of headGuards) verifyControlFile(guard); @@ -1955,12 +2239,10 @@ export function snapshotEvidence({ cited_path_manifest: citedEntries, }; } finally { - const closed = new Set(); - for (const guard of mutationGuards) { - if (guard.type !== 'absence' || closed.has(guard.fd)) continue; - closed.add(guard.fd); + // One entry per distinct anchored directory, so one close per descriptor. + for (const handle of absenceCache.values()) { try { - fs.closeSync(guard.fd); + fs.closeSync(handle.fd); } catch { // Preserve the primary snapshot result/error. } diff --git a/gitnexus/src/cli/ai-context.ts b/gitnexus/src/cli/ai-context.ts index f3ab38cf1..258aeb46c 100644 --- a/gitnexus/src/cli/ai-context.ts +++ b/gitnexus/src/cli/ai-context.ts @@ -9,7 +9,7 @@ import fs from 'fs/promises'; import path from 'path'; import { fileURLToPath } from 'url'; -import { type GeneratedSkillInfo } from './skill-gen.js'; +import { type GeneratedSkillInfo } from './generated-skill.js'; import { STANDARD_SKILL_CATALOG } from './standard-skills.js'; import { logger } from '../core/logger.js'; @@ -157,7 +157,10 @@ export function generateGitNexusContent( ? generatedSkills .map( (s) => - `| Work in the ${s.label} area (${s.symbolCount} symbols) | \`.claude/skills/${s.name}/SKILL.md\` |`, + // The per-cluster count is as volatile as the header parenthetical, + // so --no-stats drops it too (#2907) — otherwise the flag that + // promises "omit volatile symbol counts" left a churning one behind. + `| Work in the ${s.label} area${noStats ? '' : ` (${s.symbolCount} symbols)`} | \`.claude/skills/${s.name}/SKILL.md\` |`, ) .join('\n') : ''; @@ -195,12 +198,23 @@ ${tableBody}` `No \`${runnerPath}\` yet? Bootstrap with \`npx\`, \`bunx\`, or \`pnpm dlx\` — ` + 'e.g. `bunx gitnexus@latest analyze` (npm 11 npx crash; #1939).'; + // This block is injected into every user's repo and its total size is capped + // by test (ai-context.test.ts, #856) — a new bullet or clause has to be paid + // for by trimming an existing one. + // + // The detect_changes bullet carries the degraded-result rule (#2915): a run + // that sets `partial` (a graph query failed) or `truncated` (the changed-symbol + // listing was capped) is not the pre-commit gate passing, and `partial` pairs + // routinely with changed_count:0 — the exact shape that printed "No changes + // detected." and exited 0 on a broken analysis. Same reasoning as the + // `risk: UNKNOWN` bullet below: the tool could not answer, so its zero is not + // an all-clear. return `${GITNEXUS_START_MARKER} # GitNexus — Code Intelligence -This project is indexed by GitNexus as **${projectName}**${noStats ? '' : ` (${stats.nodes || 0} symbols, ${stats.edges || 0} relationships, ${stats.processes || 0} execution flows)`}. Use GitNexus graph tools to understand code, assess impact, and navigate safely. +This project is indexed by GitNexus as **${projectName}**${noStats ? '' : ` (${stats.nodes || 0} symbols, ${stats.edges || 0} relationships, ${stats.processes || 0} execution flows)`}. -> Index stale? Run \`${runner} analyze\` from the project root — it auto-selects an available runner. ${bootstrapNote} +> Index stale? Run \`${runner} analyze --index-only\` from the project root — it auto-selects an available runner. ${bootstrapNote} ## Always Do @@ -209,8 +223,9 @@ This project is indexed by GitNexus as **${projectName}**${noStats ? '' : ` (${s ? ` For unified PDG impact, add \`mode: "pdg"\` with optional \`line: \` — it returns statement-level \`affectedStatements\` over CDG + REACHING_DEF and inter-procedural symbols in \`interproceduralByDepth\`/\`byDepth\`; no-layer/degraded PDG results are UNKNOWN-risk notes (\`--pdg\` layer). CLI equivalent: \`${runner} impact "symbolName" --direction upstream --mode pdg --line --repo .\`.` : '' } -- **MUST analyze graph changes before committing.** Use \`detect_changes({scope: "all"})\` (MCP) or \`${runner} detect-changes --scope all --repo .\` (CLI fallback). For regression review: \`detect_changes({scope: "compare", base_ref: ${JSON.stringify(markdownSafeBranch(defaultBranch))}})\` or \`${runner} detect-changes --scope compare --base-ref ${JSON.stringify(markdownSafeBranch(defaultBranch))} --repo .\`. +- **MUST analyze graph changes before committing.** Use \`detect_changes({scope: "all"})\` (MCP) or \`${runner} detect-changes --scope all --repo .\` (CLI fallback). \`partial: true\` or \`truncated: true\` is not a clean check — a zero means unseen, not unaffected; re-run it. For regression review: \`detect_changes({scope: "compare", base_ref: ${JSON.stringify(markdownSafeBranch(defaultBranch))}})\` or \`${runner} detect-changes --scope compare --base-ref ${JSON.stringify(markdownSafeBranch(defaultBranch))} --repo .\`. - **MUST warn the user** if impact analysis returns HIGH or CRITICAL risk before proceeding with edits. +- **MUST treat \`risk: UNKNOWN\` as unresolved, not as low.** An empty caller set is not evidence the symbol is unused — it can also mean the callers are not resolvable by the index (plain-object property access, dynamic dispatch, cross-language calls). \`impact\` pairs \`UNKNOWN\` with a \`riskNote\` saying so. Confirm with a text search before treating the symbol as safe to change or delete; do not proceed on the strength of a zero. - When exploring unfamiliar code, use \`query({search_query: "concept"})\` to find execution flows instead of grepping. It returns process-grouped results ranked by relevance. - When you need full context on a specific symbol — callers, callees, which execution flows it participates in — use \`context({name: "symbolName"})\`. - For security review, \`explain({target: "fileOrSymbol"})\` lists taint findings (source→sink flows; needs \`analyze --pdg\`).${ @@ -222,7 +237,7 @@ This project is indexed by GitNexus as **${projectName}**${noStats ? '' : ` (${s ## Never Do - NEVER edit a function, class, or method before MCP/CLI impact analysis. -- NEVER ignore HIGH or CRITICAL risk warnings from impact analysis. +- NEVER ignore HIGH or CRITICAL risk warnings from impact analysis, and never read \`UNKNOWN\` as an all-clear — it means the walk could not answer, which is the one verdict that requires confirming by other means. - NEVER rename symbols with find-and-replace — use \`rename\` which understands the call graph. - NEVER commit before MCP/CLI graph change analysis. @@ -266,11 +281,33 @@ async function fileExists(filePath: string): Promise { } } +/** + * Replace the block's volatile counts — the header parenthetical and the + * per-cluster symbol counts in the skills table — with fixed placeholders, so + * two renderings that differ only in those numbers compare equal. + * + * Placeholders rather than deletions: `--no-stats` REMOVES the parenthetical, + * which must still be written through. Deleting instead of substituting would + * make a with-counts block and a without-counts block compare equal, and the + * flag would silently stop taking effect on an already-injected file. + */ +function stripVolatileCounts(section: string): string { + return section + .replace(/ \(\d+ symbols, \d+ relationships, \d+ execution flows\)/g, ' ()') + .replace(/ \(\d+ symbols\)/g, ' ()'); +} + /** * Create or update GitNexus section in a file * - If file doesn't exist: create with GitNexus content * - If file exists without GitNexus section: append - * - If file exists with GitNexus section: replace that section + * - If file exists with GitNexus section: replace that section, UNLESS the only + * delta is the volatile counts (#2907). AGENTS.md and CLAUDE.md are the agent + * guides teams commit, and the counts move with any code change, so a + * count-only rewrite dirties a tracked file on every reindex for no reader + * benefit. Live counts stay available from `gitnexus status` and + * `gitnexus://repo/{name}/context`; the committed block keeps whichever + * numbers it was last materially updated with. */ async function upsertGitNexusSection( filePath: string, @@ -282,7 +319,10 @@ async function upsertGitNexusSection( const exists = await fileExists(filePath); if (!exists) { - await fs.writeFile(filePath, content, 'utf-8'); + // Same `.trim() + '\n'` shape the update paths write. Creating without the + // trailing newline made the NEXT analyze dirty a freshly committed file + // even at unchanged counts, purely to append it (#2907). + await fs.writeFile(filePath, content.trim() + '\n', 'utf-8'); return 'created'; } @@ -343,6 +383,11 @@ async function upsertGitNexusSection( if (statsPattern.test(existingSection)) { const updatedSection = existingSection.replace(statsPattern, statsLine); + // Count-only delta — leave the committed lean block alone (#2907). A + // project rename, or --no-stats dropping the parenthetical, still writes. + if (stripVolatileCounts(updatedSection) === stripVolatileCounts(existingSection)) { + return 'preserved'; + } const before = existingContent.substring(0, startIdx); const after = existingContent.substring(endIdx + GITNEXUS_END_MARKER.length); await fs.writeFile(filePath, (before + updatedSection + after).trim() + '\n', 'utf-8'); @@ -354,7 +399,11 @@ async function upsertGitNexusSection( return 'preserved'; } - // No keep marker — replace existing section with full verbose content + // No keep marker — replace existing section with full verbose content, + // unless the counts are the only thing that moved (#2907). + if (stripVolatileCounts(existingSection) === stripVolatileCounts(content)) { + return 'preserved'; + } const before = existingContent.substring(0, startIdx); const after = existingContent.substring(endIdx + GITNEXUS_END_MARKER.length); const newContent = before + content + after; diff --git a/gitnexus/src/cli/analyze-config.ts b/gitnexus/src/cli/analyze-config.ts index 048ecc52f..6e1afc7bb 100644 --- a/gitnexus/src/cli/analyze-config.ts +++ b/gitnexus/src/cli/analyze-config.ts @@ -30,7 +30,7 @@ import fs from 'node:fs'; import path from 'node:path'; -import type { AnalyzeOptions } from './analyze.js'; +import type { AnalyzeOptions } from './analyze-options.js'; export const GITNEXUS_RC_FILENAME = '.gitnexusrc'; diff --git a/gitnexus/src/cli/analyze-options.ts b/gitnexus/src/cli/analyze-options.ts new file mode 100644 index 000000000..c3d1b3e8f --- /dev/null +++ b/gitnexus/src/cli/analyze-options.ts @@ -0,0 +1,131 @@ +/** + * CLI-facing `analyze` option shape. + * + * This is the *flag* shape: it mirrors what Commander parses off the command + * line and what `.gitnexusrc` may set, before `analyze` translates it into the + * core orchestrator's own `AnalyzeOptions` (`core/run-analyze.ts`) — a + * different, deliberately separate interface (`stats` here vs `noStats` + * there, `embeddings?: boolean | string` here vs a resolved + * `embeddingsNodeLimit` there). + * + * It lives in this leaf module because both `analyze.ts` (which consumes the + * flags) and `analyze-config.ts` (which maps `.gitnexusrc` keys onto them) + * need it, and `analyze.ts` already imports the config loader — a type import + * back the other way put the two files, plus `core/run-analyze.ts`, in an + * import cycle. `analyze.ts` re-exports the type for existing importers. + */ +export interface AnalyzeOptions { + force?: boolean; + repairFts?: boolean; + /** + * Embedding generation toggle. Commander parses `--embeddings [limit]` as: + * - `undefined` when the flag is omitted + * - `true` when passed without an argument (use default 50K node cap) + * - a string when passed with an argument (`--embeddings 0` disables the + * cap, `--embeddings ` uses `` as the cap) + */ + embeddings?: boolean | string; + /** + * Explicitly drop existing embeddings on rebuild instead of preserving + * them. Without this flag, a routine `analyze` keeps any embeddings + * already present in the index even when `--embeddings` is omitted. + */ + dropEmbeddings?: boolean; + skills?: boolean; + verbose?: boolean; + /** Skip AGENTS.md and CLAUDE.md gitnexus block updates. */ + skipAgentsMd?: boolean; + /** + * Build the control-flow-graph / PDG substrate (#2081 M1). Opt-in; off by + * default. Threaded to both the worker (CFG build) and scope-resolution + * (BasicBlock/CFG emit). + */ + pdg?: boolean; + /** + * Stats inclusion in AGENTS.md and CLAUDE.md. + * + * Commander.js represents `--no-stats` as `stats: boolean` (default + * `true`; `false` when the user passes `--no-stats`), NOT as + * `noStats: boolean`. Reading the negated form would always be + * `undefined` and the flag would silently no-op (#1477). Consumers + * that want "did the user request --no-stats?" should compare with + * `=== false` to distinguish the explicit-off case from the + * default-on case. + */ + stats?: boolean; + /** + * Opt-in auto-commit of any AGENTS.md/CLAUDE.md changes this `analyze` run + * makes. Scoped to only those two files (never `git add -A`); no-ops + * silently if neither exists, neither changed, or the commit step itself + * fails (e.g. no git identity configured). See #2639. + */ + selfCommit?: boolean; + /** Skip installing standard GitNexus skill files directly under .claude/skills/. */ + skipSkills?: boolean; + /** + * Default branch for the generated regression-compare example (#243). From + * `--default-branch`; may also be supplied via `.gitnexusrc`. Resolved to a + * concrete branch (CLI > `.gitnexusrc` > auto-detected origin/HEAD > "main") + * before being threaded into the generated AGENTS.md / CLAUDE.md content. + */ + defaultBranch?: string; + /** + * Index-branch selector (#2106). From `--branch`. Distinct from + * `defaultBranch` (cosmetic base_ref): this routes the index to a per-branch + * slot. NOT sourced from `.gitnexusrc` — the `.gitnexusrc` `branch` key is an + * alias for `defaultBranch` and must not change index placement. Defaults to + * the checked-out branch inside `runFullAnalysis` when omitted. + */ + branch?: string; + /** Pure index mode: skip all file injection (AGENTS.md, CLAUDE.md, skills). */ + indexOnly?: boolean; + /** Index the folder even when no .git directory is present. */ + skipGit?: boolean; + /** + * Override the default basename-derived registry `name` with a + * user-supplied alias (#829). Disambiguates repos whose paths share a + * basename. Persisted — subsequent re-analyses of the same path without + * `--name` preserve the alias. + */ + name?: string; + /** + * Allow registration even when another path already uses the same + * `--name` alias (#829). Intentionally a distinct flag from `--force` + * because the user may want to coexist under the same name WITHOUT + * paying the cost of a pipeline re-index. Maps to registerRepo's + * `allowDuplicateName` option end-to-end. + */ + allowDuplicateName?: boolean; + /** + * Override the walker's large-file skip threshold (#991). Value in KB; + * clamped downstream to the tree-sitter 32 MB ceiling. Sets + * `GITNEXUS_MAX_FILE_SIZE` for the rest of the pipeline. + */ + maxFileSize?: string; + /** Override worker sub-batch idle timeout in seconds. */ + workerTimeout?: string; + /** Control LadybugDB WAL auto-checkpoint threshold during analyze. */ + walCheckpointThreshold?: string; + /** Parse worker pool size (>=1); 0 is rejected (no sequential mode). */ + workers?: string; + embeddingThreads?: string; + embeddingBatchSize?: string; + embeddingSubBatchSize?: string; + embeddingDevice?: string; + /** + * Extra fetch-wrapper function names to treat as HTTP consumers (#1589/#1852 + * residual). Supplied via `.gitnexusrc` `fetchWrappers: [...]`. Threaded into + * the routes phase, where the cross-file consumer scan unions them with the + * auto-detected `fetch()` wrappers so a custom/axios-based wrapper named + * outside the built-in convention still produces `route_map` consumers. + */ + fetchWrappers?: string[]; + /** OpenAI-compatible embeddings base URL (incl. /v1). Overrides GITNEXUS_EMBEDDING_URL. */ + embeddingBaseUrl?: string; + /** Embedding model name. Overrides GITNEXUS_EMBEDDING_MODEL. */ + embeddingModel?: string; + /** Bearer token for the embeddings endpoint. Overrides GITNEXUS_EMBEDDING_API_KEY. Never logged. */ + embeddingAuthToken?: string; + /** Embedding vector dimensions (positive integer string). Overrides GITNEXUS_EMBEDDING_DIMS. */ + embeddingDims?: string; +} diff --git a/gitnexus/src/cli/analyze.ts b/gitnexus/src/cli/analyze.ts index 6851c2478..e2208a3c8 100644 --- a/gitnexus/src/cli/analyze.ts +++ b/gitnexus/src/cli/analyze.ts @@ -50,6 +50,7 @@ import { validateBranchName, GitNexusRcError, } from './analyze-config.js'; +import type { AnalyzeOptions } from './analyze-options.js'; import { runFullAnalysis } from '../core/run-analyze.js'; import { getRuntimeFingerprint } from '../core/platform/capabilities.js'; import { getMaxFileSizeBannerMessage } from '../core/ingestion/utils/max-file-size.js'; @@ -661,121 +662,14 @@ const restoreAnalyzeEnv = (snap: AnalyzeEnvSnapshot): void => { } }; -export interface AnalyzeOptions { - force?: boolean; - repairFts?: boolean; - /** - * Embedding generation toggle. Commander parses `--embeddings [limit]` as: - * - `undefined` when the flag is omitted - * - `true` when passed without an argument (use default 50K node cap) - * - a string when passed with an argument (`--embeddings 0` disables the - * cap, `--embeddings ` uses `` as the cap) - */ - embeddings?: boolean | string; - /** - * Explicitly drop existing embeddings on rebuild instead of preserving - * them. Without this flag, a routine `analyze` keeps any embeddings - * already present in the index even when `--embeddings` is omitted. - */ - dropEmbeddings?: boolean; - skills?: boolean; - verbose?: boolean; - /** Skip AGENTS.md and CLAUDE.md gitnexus block updates. */ - skipAgentsMd?: boolean; - /** - * Build the control-flow-graph / PDG substrate (#2081 M1). Opt-in; off by - * default. Threaded to both the worker (CFG build) and scope-resolution - * (BasicBlock/CFG emit). - */ - pdg?: boolean; - /** - * Stats inclusion in AGENTS.md and CLAUDE.md. - * - * Commander.js represents `--no-stats` as `stats: boolean` (default - * `true`; `false` when the user passes `--no-stats`), NOT as - * `noStats: boolean`. Reading the negated form would always be - * `undefined` and the flag would silently no-op (#1477). Consumers - * that want "did the user request --no-stats?" should compare with - * `=== false` to distinguish the explicit-off case from the - * default-on case. - */ - stats?: boolean; - /** - * Opt-in auto-commit of any AGENTS.md/CLAUDE.md changes this `analyze` run - * makes. Scoped to only those two files (never `git add -A`); no-ops - * silently if neither exists, neither changed, or the commit step itself - * fails (e.g. no git identity configured). See #2639. - */ - selfCommit?: boolean; - /** Skip installing standard GitNexus skill files directly under .claude/skills/. */ - skipSkills?: boolean; - /** - * Default branch for the generated regression-compare example (#243). From - * `--default-branch`; may also be supplied via `.gitnexusrc`. Resolved to a - * concrete branch (CLI > `.gitnexusrc` > auto-detected origin/HEAD > "main") - * before being threaded into the generated AGENTS.md / CLAUDE.md content. - */ - defaultBranch?: string; - /** - * Index-branch selector (#2106). From `--branch`. Distinct from - * `defaultBranch` (cosmetic base_ref): this routes the index to a per-branch - * slot. NOT sourced from `.gitnexusrc` — the `.gitnexusrc` `branch` key is an - * alias for `defaultBranch` and must not change index placement. Defaults to - * the checked-out branch inside `runFullAnalysis` when omitted. - */ - branch?: string; - /** Pure index mode: skip all file injection (AGENTS.md, CLAUDE.md, skills). */ - indexOnly?: boolean; - /** Index the folder even when no .git directory is present. */ - skipGit?: boolean; - /** - * Override the default basename-derived registry `name` with a - * user-supplied alias (#829). Disambiguates repos whose paths share a - * basename. Persisted — subsequent re-analyses of the same path without - * `--name` preserve the alias. - */ - name?: string; - /** - * Allow registration even when another path already uses the same - * `--name` alias (#829). Intentionally a distinct flag from `--force` - * because the user may want to coexist under the same name WITHOUT - * paying the cost of a pipeline re-index. Maps to registerRepo's - * `allowDuplicateName` option end-to-end. - */ - allowDuplicateName?: boolean; - /** - * Override the walker's large-file skip threshold (#991). Value in KB; - * clamped downstream to the tree-sitter 32 MB ceiling. Sets - * `GITNEXUS_MAX_FILE_SIZE` for the rest of the pipeline. - */ - maxFileSize?: string; - /** Override worker sub-batch idle timeout in seconds. */ - workerTimeout?: string; - /** Control LadybugDB WAL auto-checkpoint threshold during analyze. */ - walCheckpointThreshold?: string; - /** Parse worker pool size (>=1); 0 is rejected (no sequential mode). */ - workers?: string; - embeddingThreads?: string; - embeddingBatchSize?: string; - embeddingSubBatchSize?: string; - embeddingDevice?: string; - /** - * Extra fetch-wrapper function names to treat as HTTP consumers (#1589/#1852 - * residual). Supplied via `.gitnexusrc` `fetchWrappers: [...]`. Threaded into - * the routes phase, where the cross-file consumer scan unions them with the - * auto-detected `fetch()` wrappers so a custom/axios-based wrapper named - * outside the built-in convention still produces `route_map` consumers. - */ - fetchWrappers?: string[]; - /** OpenAI-compatible embeddings base URL (incl. /v1). Overrides GITNEXUS_EMBEDDING_URL. */ - embeddingBaseUrl?: string; - /** Embedding model name. Overrides GITNEXUS_EMBEDDING_MODEL. */ - embeddingModel?: string; - /** Bearer token for the embeddings endpoint. Overrides GITNEXUS_EMBEDDING_API_KEY. Never logged. */ - embeddingAuthToken?: string; - /** Embedding vector dimensions (positive integer string). Overrides GITNEXUS_EMBEDDING_DIMS. */ - embeddingDims?: string; -} +/** + * CLI `analyze` flag shape. Defined in `./analyze-options.js` so + * `analyze-config.ts` can reference it without importing this module back — + * that type import closed a cycle over `analyze` → `analyze-config` and + * `analyze` → `run-analyze` → `analyze-config`. Re-exported here because this + * is where callers have always imported it from. + */ +export type { AnalyzeOptions }; /** * Whether the post-index skill step should run. @@ -1624,6 +1518,27 @@ const analyzeCommandImpl = async ( // ── Summary ──────────────────────────────────────────────────── const s = result.stats; + // A collapsed graph write is NOT a successful index. The other incomplete + // reasons (`incremental-in-progress`, `embedding-checkpoint-pending`) + // describe a run that did what it said and left work for next time; this + // one means most of your edges are gone, so every query answers a confident + // empty and the exit code is the only thing automation reads. Printing + // "indexed successfully" and exiting 0 here would be the same class of + // false certainty the check itself was written to remove. + if (result.graphWriteCollapsed) { + const { expected, persisted } = result.graphWriteCollapsed; + console.log(`\n Repository indexed INCOMPLETELY (${totalTime}s)\n`); + console.log( + ` Graph write collapsed: the pipeline produced ${expected.toLocaleString()} relationships\n` + + ` but only ${persisted.toLocaleString()} are readable from the index. Queries will answer\n` + + ` with missing edges rather than an error.\n\n` + + ` The index is recorded INCOMPLETE (graph-write-collapsed). Re-run\n` + + ` \`gitnexus analyze --force\`; if it recurs, check disk space and run \`gitnexus doctor\`.`, + ); + console.log(` ${repoPath}`); + process.exitCode = 1; + return; + } console.log(`\n Repository indexed successfully (${totalTime}s)\n`); console.log( ` ${(s.nodes ?? 0).toLocaleString()} nodes | ${(s.edges ?? 0).toLocaleString()} edges | ${s.communities ?? 0} clusters | ${s.processes ?? 0} flows`, @@ -1644,9 +1559,14 @@ const analyzeCommandImpl = async ( ); } else { console.log( + // NOT "then rerun" (#2841 §5.C): this run stamped `lastCommit`, so a + // plain rerun on an unchanged tree takes the up-to-date fast path and + // returns before Phase 3 could rebuild anything — the advice would be + // ineffective exactly when the user follows it. `--repair-fts` is the + // verb that rebuilds the search indexes without re-parsing the repo. `\n Warning: full-text/BM25 search is disabled — the LadybugDB FTS extension was unavailable.\n` + - ` Install it once with network access (GITNEXUS_LBUG_EXTENSION_INSTALL=auto) then rerun, or\n` + - ` run \`gitnexus analyze --repair-fts\` when connected. Run \`gitnexus doctor\` for details.`, + ` Install it once with network access (GITNEXUS_LBUG_EXTENSION_INSTALL=auto), then run\n` + + ` \`gitnexus analyze --repair-fts\` to build the search indexes. Run \`gitnexus doctor\` for details.`, ); } } diff --git a/gitnexus/src/cli/detect-changes-format.ts b/gitnexus/src/cli/detect-changes-format.ts index 7077334ef..98ecffa3e 100644 --- a/gitnexus/src/cli/detect-changes-format.ts +++ b/gitnexus/src/cli/detect-changes-format.ts @@ -1,4 +1,5 @@ import { t } from './i18n/index.js'; +import { formatSymbolLine } from './format-symbol.js'; type DetectChangesSummary = { changed_files?: number; @@ -25,6 +26,8 @@ type AffectedProcess = { type DetectChangesResult = { error?: unknown; + partial?: boolean; + truncated?: boolean; summary?: DetectChangesSummary; changed_symbols?: ChangedSymbol[]; affected_processes?: AffectedProcess[]; @@ -35,11 +38,28 @@ export function formatDetectChangesResult(result: unknown): string { if (payload.error) return t('common.error', { message: String(payload.error) }); const summary = payload.summary ?? {}; + // A swallowed query failure sets `partial` and leaves the counts at zero + // (#2283). Printing only "No changes detected." turns a degraded run into a + // clean bill of health for the pre-commit gate, so say so either way. + // `truncated` is its sibling flag: the backend caps the changed_symbols + // LISTING (never the counts), so a short list is not proof of a short diff. + // Both lead the output — a caveat printed after the summary is read too late. + const notes: string[] = []; + if (payload.partial) notes.push(t('tool.detectChanges.partial')); + // The plain truncation note reassures that the counts are whole. That is only + // true when the run did NOT also degrade — `changed_count` sums the batches + // that succeeded — so the two flags together get a different sentence. + if (payload.truncated) + notes.push( + t(payload.partial ? 'tool.detectChanges.truncatedDegraded' : 'tool.detectChanges.truncated'), + ); + if ((summary.changed_count ?? 0) === 0) { - return t('tool.detectChanges.noChanges'); + return [...notes, t('tool.detectChanges.noChanges')].join('\n'); } const lines: string[] = []; + if (notes.length > 0) lines.push(...notes, ''); lines.push( t('tool.detectChanges.changesSummary', { files: summary.changed_files ?? 0, @@ -59,7 +79,7 @@ export function formatDetectChangesResult(result: unknown): string { lines.push(t('tool.detectChanges.changedSymbols')); const shown = changed.slice(0, 15); for (const symbol of shown) { - lines.push(` ${symbol.type ?? 'Symbol'} ${symbol.name ?? '?'} → ${symbol.filePath ?? '?'}`); + lines.push(formatSymbolLine(symbol.type, symbol.name, symbol.filePath)); } // Overflow is measured against the TRUE total (summary.changed_count), not // the array length — the array may already be `--limit`-sliced, so using its diff --git a/gitnexus/src/cli/eval-server.ts b/gitnexus/src/cli/eval-server.ts index 31289efb2..caf19bc03 100644 --- a/gitnexus/src/cli/eval-server.ts +++ b/gitnexus/src/cli/eval-server.ts @@ -45,6 +45,7 @@ import { import { logger } from '../core/logger.js'; import { cliInfo, cliWarn, cliError } from './cli-message.js'; import { formatDetectChangesResult } from './detect-changes-format.js'; +import { formatSymbolLine } from './format-symbol.js'; export { formatDetectChangesResult } from './detect-changes-format.js'; @@ -209,7 +210,7 @@ export function formatQueryResult(result: any): string { if (defs.length > 0) { lines.push(`Standalone definitions:`); for (const d of defs.slice(0, 8)) { - lines.push(` ${d.type || 'Symbol'} ${d.name} → ${d.filePath || '?'}`); + lines.push(formatSymbolLine(d.type, d.name, d.filePath)); } if (defs.length > 8) lines.push(` ... and ${defs.length - 8} more`); } diff --git a/gitnexus/src/cli/format-symbol.ts b/gitnexus/src/cli/format-symbol.ts new file mode 100644 index 000000000..20a0fc52a --- /dev/null +++ b/gitnexus/src/cli/format-symbol.ts @@ -0,0 +1,22 @@ +/** + * Symbol listing line — the one rendering of `Type name → path` shared by every + * formatter that lists symbols. Kept in its own tool-neutral module so a new + * consumer does not have to import it from another tool's formatter. + */ + +/** + * One indented `Type name → path` listing line for a symbol. Shared by the + * `detect_changes` CLI formatter and the eval-server `query` formatter so the + * two renderings cannot drift apart. + * + * `||`, not `??`: a node whose label came back as an EMPTY STRING (several node + * types do — see enrichCandidateLabels) still needs the placeholder, and `??` + * would print the empty string instead. + */ +export function formatSymbolLine( + type: string | undefined, + name: string | undefined, + filePath: string | undefined, +): string { + return ` ${type || 'Symbol'} ${name || '?'} → ${filePath || '?'}`; +} diff --git a/gitnexus/src/cli/generated-skill.ts b/gitnexus/src/cli/generated-skill.ts new file mode 100644 index 000000000..ab0377f3e --- /dev/null +++ b/gitnexus/src/cli/generated-skill.ts @@ -0,0 +1,17 @@ +/** + * Metadata for one repo-specific skill file generated from a detected + * community. + * + * Produced by `skill-gen`'s `generateSkillFiles` and consumed by `ai-context` + * when it lists the generated skills in AGENTS.md / CLAUDE.md. It lives in this + * leaf module rather than in either of those so the consumer does not have to + * import the producer for a type — `ai-context` already supplies the + * `.agents/` mirror check that `skill-gen` calls, and the two directions + * together made an import cycle. + */ +export interface GeneratedSkillInfo { + name: string; + label: string; + symbolCount: number; + fileCount: number; +} diff --git a/gitnexus/src/cli/i18n/en.ts b/gitnexus/src/cli/i18n/en.ts index ce34011ad..a852be41f 100644 --- a/gitnexus/src/cli/i18n/en.ts +++ b/gitnexus/src/cli/i18n/en.ts @@ -65,6 +65,14 @@ export const en = { 'tool.warn.unknownKind': "--kind '{{kind}}' is not a known symbol kind (e.g. Function, Class, Method); it will not narrow the result.", 'tool.detectChanges.noChanges': 'No changes detected.', + 'tool.detectChanges.partial': + 'PARTIAL RESULT: a graph query failed, so changed symbols may be missing. Do not read this as a clean pre-commit check.', + 'tool.detectChanges.truncated': + 'LISTING CAPPED: the changed-symbol list was capped, so it does not name every changed symbol. The counts and risk level still cover all of them.', + // The reassurance above is only true on its own. When the run also degraded, + // `changed_count` was summed from the batches that SUCCEEDED, so it is a floor. + 'tool.detectChanges.truncatedDegraded': + 'LISTING CAPPED: the changed-symbol list was capped. The run also degraded, so the counts are a lower bound, not a total.', 'tool.detectChanges.changesSummary': 'Changes: {{files}} files, {{symbols}} symbols', 'tool.detectChanges.affectedProcesses': 'Affected processes: {{count}}', 'tool.detectChanges.riskLevel': 'Risk level: {{risk}}', @@ -226,16 +234,15 @@ export const en = { 'Clean parked LadybugDB recovery sidecars (missing-shadow WAL quarantines and dirty-recovery parks)', 'help.option.wiki.force': 'Force full regeneration even if up to date', 'help.option.wiki.provider': - 'LLM provider: openai, openrouter, atlascloud, azure, custom, cursor, claude, codex, or opencode (default: openai)', - 'help.option.wiki.model': 'LLM model or Azure deployment name (default: minimax/minimax-m2.5)', + 'LLM provider: minimax, openai, openrouter, atlascloud, azure, custom, cursor, claude, codex, or opencode (default: minimax)', + 'help.option.wiki.model': 'LLM model or deployment name (default: MiniMax-M3)', 'help.option.wiki.baseUrl': 'LLM API base URL. Azure v1: https://{resource}.openai.azure.com/openai/v1', 'help.option.wiki.apiKey': 'LLM API key or Azure api-key (saved to ~/.gitnexus/config.json)', 'help.option.wiki.apiVersion': 'Azure api-version query param, e.g. 2024-10-21 (legacy Azure API only)', - 'help.option.wiki.reasoningModel': - 'Mark deployment as reasoning model (o1/o3/o4-mini) — strips temperature, uses max_completion_tokens', - 'help.option.wiki.noReasoningModel': 'Disable reasoning model mode (overrides saved config)', + 'help.option.wiki.reasoningModel': 'Enable reasoning mode; MiniMax-M3 uses adaptive thinking', + 'help.option.wiki.noReasoningModel': 'Disable reasoning mode; MiniMax-M3 disables thinking', 'help.option.wiki.concurrency': 'Parallel LLM calls (default: 3)', 'help.option.wiki.timeout': 'LLM request timeout in seconds (default: disabled)', 'help.option.wiki.retries': 'Max LLM retry attempts per request (default: 3)', diff --git a/gitnexus/src/cli/i18n/zh-CN.ts b/gitnexus/src/cli/i18n/zh-CN.ts index 9a381fdb4..8ba5119b0 100644 --- a/gitnexus/src/cli/i18n/zh-CN.ts +++ b/gitnexus/src/cli/i18n/zh-CN.ts @@ -69,6 +69,12 @@ export const zhCN = { 'tool.warn.unknownKind': "--kind '{{kind}}' 不是已知的符号类型(如 Function、Class、Method),不会用于缩小结果范围。", 'tool.detectChanges.noChanges': '未检测到变更。', + 'tool.detectChanges.partial': + '结果不完整:图查询失败,可能遗漏已变更符号。请勿将其视为通过的提交前检查。', + 'tool.detectChanges.truncated': + '列表已截断:已变更符号列表被截断,未列出全部变更符号。计数与风险等级仍涵盖全部符号。', + 'tool.detectChanges.truncatedDegraded': + '列表已截断:已变更符号列表被截断。本次运行同时不完整,因此计数为下限而非总数。', 'tool.detectChanges.changesSummary': '变更:{{files}} 个文件,{{symbols}} 个符号', 'tool.detectChanges.affectedProcesses': '受影响流程:{{count}}', 'tool.detectChanges.riskLevel': '风险等级:{{risk}}', @@ -214,15 +220,14 @@ export const zhCN = { '清理已暂存的 LadybugDB 恢复 sidecar(missing-shadow WAL 隔离文件与 dirty-recovery 暂存文件)', 'help.option.wiki.force': '即使已是最新也强制完整重新生成', 'help.option.wiki.provider': - 'LLM 提供商:openai、openrouter、atlascloud、azure、custom、cursor、claude、codex 或 opencode(默认:openai)', - 'help.option.wiki.model': 'LLM 模型或 Azure deployment 名称(默认:minimax/minimax-m2.5)', + 'LLM 提供商:minimax、openai、openrouter、atlascloud、azure、custom、cursor、claude、codex 或 opencode(默认:minimax)', + 'help.option.wiki.model': 'LLM 模型或 deployment 名称(默认:MiniMax-M3)', 'help.option.wiki.baseUrl': 'LLM API base URL。Azure v1:https://{resource}.openai.azure.com/openai/v1', 'help.option.wiki.apiKey': 'LLM API key 或 Azure api-key(保存到 ~/.gitnexus/config.json)', 'help.option.wiki.apiVersion': 'Azure api-version 查询参数,例如 2024-10-21(仅旧版 Azure API)', - 'help.option.wiki.reasoningModel': - '标记 deployment 为 reasoning model(o1/o3/o4-mini)— 去除 temperature,使用 max_completion_tokens', - 'help.option.wiki.noReasoningModel': '禁用 reasoning model 模式(覆盖已保存配置)', + 'help.option.wiki.reasoningModel': '启用 reasoning 模式;MiniMax-M3 使用自适应 thinking', + 'help.option.wiki.noReasoningModel': '禁用 reasoning 模式;MiniMax-M3 关闭 thinking', 'help.option.wiki.concurrency': '并行 LLM 调用数(默认:3)', 'help.option.wiki.timeout': 'LLM 请求超时时间(秒,默认:禁用)', 'help.option.wiki.retries': '每个请求的最大 LLM 重试次数(默认:3)', diff --git a/gitnexus/src/cli/index.ts b/gitnexus/src/cli/index.ts index b59cb9f64..17d24d6ae 100644 --- a/gitnexus/src/cli/index.ts +++ b/gitnexus/src/cli/index.ts @@ -303,9 +303,9 @@ program .option('-f, --force', 'Force full regeneration even if up to date') .option( '--provider ', - 'LLM provider: openai, openrouter, atlascloud, azure, custom, cursor, claude, codex, or opencode (default: openai)', + 'LLM provider: minimax, openai, openrouter, atlascloud, azure, custom, cursor, claude, codex, or opencode (default: minimax)', ) - .option('--model ', 'LLM model or Azure deployment name (default: minimax/minimax-m2.5)') + .option('--model ', 'LLM model or deployment name (default: MiniMax-M3)') .option( '--base-url ', 'LLM API base URL. Azure v1: https://{resource}.openai.azure.com/openai/v1', @@ -315,11 +315,8 @@ program '--api-version ', 'Azure api-version query param, e.g. 2024-10-21 (legacy Azure API only)', ) - .option( - '--reasoning-model', - 'Mark deployment as reasoning model (o1/o3/o4-mini) — strips temperature, uses max_completion_tokens', - ) - .option('--no-reasoning-model', 'Disable reasoning model mode (overrides saved config)') + .option('--reasoning-model', 'Enable reasoning mode; MiniMax-M3 uses adaptive thinking') + .option('--no-reasoning-model', 'Disable reasoning mode; MiniMax-M3 disables thinking') .option('--concurrency ', 'Parallel LLM calls (default: 3)', '3') .option('--timeout ', 'LLM request timeout in seconds (default: disabled)') .option('--retries ', 'Max LLM retry attempts per request (default: 3)') diff --git a/gitnexus/src/cli/skill-gen.ts b/gitnexus/src/cli/skill-gen.ts index 9e46b1e5c..f1dd45c3b 100644 --- a/gitnexus/src/cli/skill-gen.ts +++ b/gitnexus/src/cli/skill-gen.ts @@ -14,6 +14,7 @@ import { CommunityNode, CommunityMembership } from '../core/ingestion/community- import { ProcessNode } from '../core/ingestion/process-processor.js'; import { KnowledgeGraph } from '../core/graph/types.js'; import { shouldMirrorSkillsToAgents } from './ai-context.js'; +import type { GeneratedSkillInfo } from './generated-skill.js'; const GENERATED_SKILL_PREFIX = 'gitnexus-area-'; const MAX_SKILL_NAME_LENGTH = 64; @@ -23,13 +24,6 @@ const MAX_COMMUNITY_NAME_LENGTH = MAX_SKILL_NAME_LENGTH - GENERATED_SKILL_PREFIX // TYPES // ============================================================================ -export interface GeneratedSkillInfo { - name: string; - label: string; - symbolCount: number; - fileCount: number; -} - interface AggregatedCommunity { label: string; rawIds: string[]; diff --git a/gitnexus/src/cli/tool.ts b/gitnexus/src/cli/tool.ts index 36a45651a..35f973d91 100644 --- a/gitnexus/src/cli/tool.ts +++ b/gitnexus/src/cli/tool.ts @@ -42,9 +42,18 @@ async function getBackend(): Promise { * and write directly to the real stdout fd (#324). * * Falls back to stderr if the fd write fails (e.g., broken pipe). + * + * `render` is for the commands that print prose instead of JSON: they hand over + * the STRUCTURED result and a formatter, so the payload stays visible to the + * exit-code test below — pre-formatting it into a string would hide the very + * fields that test reads. */ -function output(data: any): void { - const text = typeof data === 'string' ? data : JSON.stringify(data, null, 2); +function output(data: T, render?: (data: T) => string): void { + const text = render + ? render(data) + : typeof data === 'string' + ? data + : JSON.stringify(data, null, 2); try { writeSync(1, text + '\n'); } catch (err: any) { @@ -56,18 +65,34 @@ function output(data: any): void { // Fallback: stderr (previous behavior, works on all platforms) process.stderr.write(text + '\n'); } - // Backend failures come back as `{ error }` payloads rather than throws - // (#2469). Every tool command routes its result through here, so this is - // the one place that keeps scripted callers honest: print the payload, - // then exit non-zero. - if ( - data && - typeof data === 'object' && - 'error' in data && - typeof data.error === 'string' && - data.error.trim().length > 0 - ) { - process.exitCode = 1; + // Every tool command routes its result through here, so this is the one place + // that keeps scripted callers honest — `gitnexus impact … && ` and + // `gitnexus detect-changes && git commit` must not proceed on a result that + // did not complete. Two shapes say so, and both exit non-zero: + // + // • `error` — a backend failure, returned as a payload rather than thrown + // (#2469). + // • `partial` — a step failed and was SWALLOWED (#2915), so the counts and + // risk level are lower bounds a caller would otherwise read as clean. It + // is cross-tool vocabulary, not detect_changes' private flag: `query` + // raises it for degraded enrichment or a partial FTS failure, and `impact` + // for an interrupted traversal or capped per-symbol enrichment — a short + // caller set and an under-ranked risk, on the tool AGENTS.md makes a MUST + // gate before every edit. + // + // One code for both, because `&&` cannot tell two apart and a "softer" code + // for `partial` would invite exempting it again. + // + // NOT here: `truncated`, where only the LISTING is capped while the counts and + // risk are computed over the full set — the verdict is sound, so failing on it + // would fire on every large-but-healthy diff. Nor `partialProbe`, a narrower + // per-candidate flag on ambiguous impact targets. + if (data && typeof data === 'object') { + const payload = data as { error?: unknown; partial?: unknown }; + const failed = + (typeof payload.error === 'string' && payload.error.trim().length > 0) || + payload.partial === true; + if (failed) process.exitCode = 1; } } @@ -337,7 +362,9 @@ export async function detectChangesCommand(options?: { if (Array.isArray(result.affected_processes)) result.affected_processes = result.affected_processes.slice(0, limit); } - output(formatDetectChangesResult(result)); + // Hand over the structured result plus its formatter, not the formatted text: + // `output()` reads `error` / `partial` off the payload to set the exit code. + output(result, formatDetectChangesResult); } export async function checkCommand(options?: { @@ -359,21 +386,44 @@ export async function checkCommand(options?: { repo: options.repo, branch: options.branch, }); + // A rendering guard, not an exit-code decision — `output()` owns that. An + // error payload carries no `cycles` array, so the prose branch below would + // throw on it; print the structured payload and stop. if (result?.error) { output(result); - process.exitCode = 1; return; } if (options.json) { output(result); - } else if (result.cycleCount === 0) { + } else if (result.status === 'clean') { output('No circular imports found.'); } else { output( result.cycles.map((cycle: { files: string[] }) => cycle.files.join(' -> ')).join('\n'), ); + // Past the enumeration cap the tool reports one representative cycle per + // component instead of every elementary cycle. Say so, or the short list + // reads as the whole truth on exactly the repositories where it is not. + if (result.enumeration === 'component-representatives') { + // Phrased to need no plural: `checkCommand` predates the `t()` i18n + // layer and none of its output goes through it, so inventing a plural + // here by hand would be the only one in the file. + output( + `\n(showing one representative cycle per circular component — ` + + `${result.componentCount} in total; the full enumeration exceeded the safety limit.)`, + ); + } } - if (result.cycleCount > 0) process.exitCode = 1; + // Policy, not degradation: a clean run that FOUND cycles is `check` failing + // its own check, so `output()` — which fails closed on `error` and `partial` + // — deliberately knows nothing about it. + // + // Keyed on `status`, NOT on `cycleCount`. Past the enumeration cap the + // report carries `cycleCount: null` on purpose, because a partial count must + // not read as a real one — and `null > 0` is false, so counting here would + // exit 0 on precisely the repositories with the most cycles. `status` + // answers "were any found" in both enumeration modes. + if (result.status === 'cycles_found') process.exitCode = 1; } catch (error) { output({ error: error instanceof Error ? error.message : String(error) }); process.exitCode = 1; diff --git a/gitnexus/src/cli/wiki.ts b/gitnexus/src/cli/wiki.ts index 3a653e5f6..2f202f655 100644 --- a/gitnexus/src/cli/wiki.ts +++ b/gitnexus/src/cli/wiki.ts @@ -21,6 +21,8 @@ import { ATLAS_CLOUD_BASE_URL, ATLAS_CLOUD_DEFAULT_MODEL, getProviderEnvApiKey, + MINIMAX_MODEL_IDS, + MINIMAX_OPENAI_BASE_URLS, parseLLMAllowedInsecureHttpHosts, resolveLLMConfig, type LLMProvider, @@ -220,6 +222,14 @@ const wikiCommandImpl = async (inputPath?: string, options?: WikiCommandOptions) ) { const existing = await loadCLIConfig(); const updates: Partial = {}; + const providerChanged = !!options.provider && options.provider !== existing.provider; + if (providerChanged) { + updates.apiKey = undefined; + updates.baseUrl = undefined; + updates.model = undefined; + updates.apiVersion = undefined; + updates.isReasoningModel = undefined; + } if (options.apiKey) updates.apiKey = options.apiKey; if (options.baseUrl) updates.baseUrl = options.baseUrl; if (options.provider) updates.provider = options.provider; @@ -230,6 +240,17 @@ const wikiCommandImpl = async (inputPath?: string, options?: WikiCommandOptions) } if (options.apiVersion) updates.apiVersion = options.apiVersion; if (options.reasoningModel !== undefined) updates.isReasoningModel = options.reasoningModel; + if (options.provider === 'minimax') { + if (providerChanged && options.reasoningModel === undefined) { + updates.isReasoningModel = undefined; + } + if (!options.baseUrl && (providerChanged || !existing.baseUrl)) { + updates.baseUrl = MINIMAX_OPENAI_BASE_URLS.global_en; + } + if (!options.model && (providerChanged || !existing.model)) { + updates.model = MINIMAX_MODEL_IDS[0]; + } + } // Save model to appropriate field based on provider. if (options.model) { const targetProvider = options.provider ?? existing.provider; @@ -246,7 +267,7 @@ const wikiCommandImpl = async (inputPath?: string, options?: WikiCommandOptions) const savedConfig = await loadCLIConfig(); const hasSavedConfig = !!( isLocalProvider(savedConfig.provider) || - (savedConfig.apiKey && savedConfig.baseUrl) + (savedConfig.apiKey && (savedConfig.baseUrl || savedConfig.provider === 'minimax')) ); const hasCLIOverrides = !!( options?.apiKey || @@ -279,7 +300,7 @@ const wikiCommandImpl = async (inputPath?: string, options?: WikiCommandOptions) if (!llmConfig.apiKey && !isLocalProvider(llmConfig.provider)) { console.log(' Error: No LLM API key found.'); console.log( - ' Set ATLASCLOUD_API_KEY, OPENAI_API_KEY, or GITNEXUS_API_KEY environment variable,', + ' Set MINIMAX_API_KEY, ATLASCLOUD_API_KEY, GITNEXUS_API_KEY, or OPENAI_API_KEY,', ); console.log(' or pass --api-key , or use --provider cursor|claude|codex|opencode.\n'); process.exitCode = 1; @@ -289,7 +310,7 @@ const wikiCommandImpl = async (inputPath?: string, options?: WikiCommandOptions) } else { console.log(" No LLM configured. Let's set it up.\n"); console.log( - ' Supports OpenAI, OpenRouter, Atlas Cloud, Azure, any OpenAI-compatible API, Cursor CLI, Claude CLI, Codex CLI, or OpenCode CLI.\n', + ' Supports MiniMax, OpenAI, OpenRouter, Atlas Cloud, Azure, custom OpenAI-compatible APIs, and local agent CLIs.\n', ); // Check if local agent CLIs are available. @@ -308,7 +329,9 @@ const wikiCommandImpl = async (inputPath?: string, options?: WikiCommandOptions) console.log(' [3] Azure OpenAI'); console.log(' [4] Atlas Cloud (api.atlascloud.ai)'); console.log(' [5] Custom endpoint'); - let nextChoice = 6; + console.log(' [6] MiniMax Global (api.minimax.io)'); + console.log(' [7] MiniMax China (api.minimaxi.com)'); + let nextChoice = 8; if (hasCursor) { const choice = String(nextChoice++); localChoices.push({ @@ -429,10 +452,10 @@ const wikiCommandImpl = async (inputPath?: string, options?: WikiCommandOptions) provider: 'azure', }; } else { - // OpenAI-compatible provider (OpenAI, OpenRouter, Atlas Cloud, Custom) + // OpenAI-compatible provider setup if (choice === '2') { baseUrl = 'https://openrouter.ai/api/v1'; - defaultModel = 'minimax/minimax-m2.5'; + defaultModel = ''; provider = 'openrouter'; } else if (choice === '4') { baseUrl = ATLAS_CLOUD_BASE_URL; @@ -447,6 +470,11 @@ const wikiCommandImpl = async (inputPath?: string, options?: WikiCommandOptions) } defaultModel = 'gpt-4o-mini'; provider = 'custom'; + } else if (choice === '6' || choice === '7') { + baseUrl = + choice === '7' ? MINIMAX_OPENAI_BASE_URLS.cn_zh : MINIMAX_OPENAI_BASE_URLS.global_en; + defaultModel = MINIMAX_MODEL_IDS[0]; + provider = 'minimax'; } else { baseUrl = 'https://api.openai.com/v1'; defaultModel = 'gpt-4o-mini'; @@ -454,8 +482,15 @@ const wikiCommandImpl = async (inputPath?: string, options?: WikiCommandOptions) } // Model - const modelInput = await prompt(` Model (default: ${defaultModel}): `); + const modelInput = await prompt( + defaultModel ? ` Model (default: ${defaultModel}): ` : ' Model: ', + ); const model = modelInput || defaultModel; + if (!model) { + console.log('\n No model provided. Aborting.\n'); + process.exitCode = 1; + return; + } // API key — pre-fill hint if env var exists const envKey = getProviderEnvApiKey(provider); @@ -478,7 +513,15 @@ const wikiCommandImpl = async (inputPath?: string, options?: WikiCommandOptions) } // Save - await saveCLIConfig({ apiKey: key, baseUrl, model, provider }); + await saveCLIConfig({ + ...savedConfig, + apiKey: key, + baseUrl, + model, + provider, + apiVersion: undefined, + isReasoningModel: undefined, + }); console.log(' Config saved to ~/.gitnexus/config.json\n'); llmConfig = { ...llmConfig, apiKey: key, baseUrl, model, provider }; diff --git a/gitnexus/src/core/embeddings/http-client.ts b/gitnexus/src/core/embeddings/http-client.ts index 82cce622a..7919b2376 100644 --- a/gitnexus/src/core/embeddings/http-client.ts +++ b/gitnexus/src/core/embeddings/http-client.ts @@ -11,6 +11,7 @@ * via `AbortSignal.timeout` on the underlying fetch. */ +import { chunk } from '../../lib/utils.js'; import { CircuitOpenError, ResilientFetchExhaustedError, @@ -566,9 +567,7 @@ export const httpEmbed = async ( const url = `${config.baseUrl}/embeddings`; const allVectors: Float32Array[] = []; - for (let i = 0; i < texts.length; i += HTTP_BATCH_SIZE) { - const batch = texts.slice(i, i + HTTP_BATCH_SIZE); - const batchIndex = Math.floor(i / HTTP_BATCH_SIZE); + for (const [batchIndex, batch] of chunk(texts, HTTP_BATCH_SIZE).entries()) { const items = await httpEmbedBatch( url, batch, diff --git a/gitnexus/src/core/graph/import-cycles.ts b/gitnexus/src/core/graph/import-cycles.ts index d5dc9eacd..a9bef2a8b 100644 --- a/gitnexus/src/core/graph/import-cycles.ts +++ b/gitnexus/src/core/graph/import-cycles.ts @@ -1,11 +1,568 @@ +/** + * Elementary import-cycle enumeration. + * + * ## What is reported + * + * Every *elementary* cycle of the file-import graph — a closed walk that visits + * no file twice — is reported exactly once. Self-imports (`a -> a`) and + * two-file cycles count. Cycles that are nested inside, or that overlap with, + * other cycles are each reported separately: a strongly connected component + * with three mutually-importing files contributes five cycles, not one. + * + * This replaces an earlier implementation that returned ONE representative + * cycle per cyclic strongly connected component. That made the reported count a + * count of tangles, not of cycles, and it hid every cycle in a component but + * the first — including cycles that a reader would have to break separately. + * The tangle count is still available, as `componentCount`, under a name that + * says what it is. + * + * The scale of what the old shape hid, measured on GitNexus itself (2,079 + * files, 5,320 initialization-forcing import edges): it reported 11 cycles. + * There are 27,939, spread across those same 11 components. It showed 11 of + * them and 27,928 were invisible. + * + * ## Algorithm + * + * Donald B. Johnson, "Finding all the elementary circuits of a directed graph", + * SIAM J. Comput. 4(1), 1975 — SCC decomposition plus a backtracking search + * guarded by the `blocked` flag and the `B` sets, which together guarantee that + * no fruitless path is explored twice between two circuit outputs. That is what + * buys the O((n + e)(c + 1)) bound for `c` circuits: the cost is proportional + * to the answer, not to the size of the search space. + * + * SCCs come from an iterative Tarjan pass rather than the Kosaraju pass this + * module used before. Johnson recomputes SCCs on each induced subgraph as the + * root advances, and Tarjan needs only the forward adjacency, so nothing has to + * rebuild a reverse graph once per root. + * + * ## How this differs from madge + * + * madge's `circular()` walks depth-first from every node carrying its ancestor + * path and records `ancestors.slice(indexOf(dep))` whenever it reaches an + * ancestor. It also marks nodes visited *globally* and skips them on later + * walks, so once a node has been traversed, cycles reachable only by entering + * it from a different predecessor are never seen. madge therefore reports many + * cycles but not all of them, and which ones it misses depends on iteration + * order. Johnson's is strictly stronger: it is complete. + * + * The practical consequence is that GitNexus reports MORE cycles than madge on + * the same graph, and the two counts should not be expected to agree. Anyone + * reconciling the two is not looking at a bug here. + * + * ## Determinism + * + * Adjacency lists and the node order are sorted (default string order, matching + * `Array.prototype.sort`), the search visits neighbours in that order, and the + * finished list is sorted element-wise. Same input, same output, byte for byte. + * + * ## Rotation normalization + * + * `[a, b, c, a]` and `[b, c, a, b]` are the same cycle and must be emitted + * once. That is structural here rather than a post-hoc dedup pass: Johnson's + * search for circuits rooted at `s` runs on the subgraph induced by the nodes + * that sort at or after `s`, so every node of an emitted circuit sorts at or + * after its root. Each cycle is therefore emitted exactly once, rooted at — and + * closed back onto — its own lexicographically smallest node. No other rotation + * of it can ever be produced. + * + * ## Bounds + * + * The number of elementary cycles is exponential in the worst case, so the + * search is bounded twice: by the number of cycles (`IMPORT_CYCLE_LIMIT`) and + * by the work spent finding them (`IMPORT_CYCLE_WORK_LIMIT`). The second is not + * redundant — Johnson's is output-sensitive, so a graph that yields few cycles + * per root can burn unbounded time while staying far under the cycle cap. + * + * Exceeding either bound abandons the enumeration. What a partial run had + * accumulated is discarded rather than returned, because a partial list of + * elementary cycles is indistinguishable from a complete one at the call site + * and would be read as "these are all of them". What is returned instead is a + * different KIND of list — one representative cycle per cyclic component, the + * old pre-enumeration answer — under `enumeration: 'component-representatives'` + * so the difference is machine-readable and not merely documented. Only a run + * that dies inside the decomposition itself reports nothing at all. + */ + +import { compareCodeUnits } from '../../lib/utils.js'; + interface ImportEdge { source: string; target: string; } -function findCyclePath(component: string[], adjacency: Map): string[] { +/** + * Elementary cycles reported before the search fails closed. + * + * The binding constraint is response size, not time. THE MEASUREMENT THAT SETS + * THIS NUMBER, on GitNexus itself — 2,079 files, 5,320 initialization-forcing + * import edges: complete enumeration finds 27,939 elementary cycles across 11 + * components in 241ms. Fast. But those cycles average 13 files each, so + * serializing them is 400,877 path entries — a 21.8 MB JSON response for a tool + * whose result is read by an agent. Time was never going to stop that, and + * neither was the work budget (the same run spends 6.3M of its 10M). + * + * Keep that measurement next to this constant. Without it the cap looks like an + * arbitrary round number and gets raised or deleted by someone who has only + * ever seen it not fire. + * + * So the cap is set where the answer stops being consumable rather than where + * the machine stops coping. Past 10,000 cycles the ten-thousandth path tells a + * reader nothing the first hundred did not, and what a reader acts on is + * `componentCount` plus one cycle per component — which is exactly what a + * report over the cap degrades to, rather than to nothing. + */ +export const IMPORT_CYCLE_LIMIT = 10_000; + +/** + * Units of search effort allowed before the search fails closed — edges + * examined, nodes scanned per root, and emitted cycle nodes at + * `EMITTED_NODE_COST` each — counted across the SCC passes and the circuit + * search alike. + * + * The cycle cap alone does NOT bound this. Johnson's is output-sensitive at + * O((n + e)(c + 1)), so producing `c` cycles still scales with the graph: a + * component of mutually-importing neighbours yields one or two cycles per root, + * so an SCC pass runs per node and the total is quadratic while the cycle count + * stays low. `check` admits import graphs up to 100k edges, so that shape is + * reachable, and there it is minutes of work under a cycle cap that never + * trips. The reverse gap is just as real: a single 50k-file component produces + * cycles 50k files long, and 10k of those exhaust the heap. One bound cannot + * see both, which is why there are two. + * + * Measured on this implementation against the mutual-import chain — the shape + * that spends the whole budget, where every unit buys a fresh SCC pass over a + * barely-smaller component — the rate is 2.9-5.2M units/second (5.2M at 10k + * nodes, 2.9M at 50k; it falls as the component grows). So 10M buys roughly + * 1.9-3.5s of enumeration on this hardware. That is the one shape where a user + * waits, and it is the number to re-measure if this constant is ever moved. + * + * It sits far above what real import graphs cost: a 100k-file acyclic graph + * spends 220k units, and 20k independent three-file tangles spend 576k. Only a + * component both large and densely tangled reaches the cap, and that + * component's honest answer is "too tangled to enumerate", not a + * silently-shortened list. + */ +export const IMPORT_CYCLE_WORK_LIMIT = 10_000_000; + +/** + * Work charged per node of an emitted cycle, relative to one edge examination. + * + * Emitted nodes are retained for the lifetime of the call and then sorted and + * serialized, so they are the term that decides peak memory, while examined + * edges cost nothing but time. Without a weight here, a graph whose cycles are + * tens of thousands of files long exhausts the heap while both bounds still + * read as comfortably unspent. + */ +const EMITTED_NODE_COST = 10; + +/** Which bound stopped the search. */ +export type ImportCycleLimit = 'cycles' | 'work'; + +/** + * The result of an enumeration. + * + * `enumeration` is the union's discriminant rather than a sibling flag, + * deliberately: a caller cannot reach `cycles` without first narrowing on what + * kind of list it is holding. A partial enumeration and a complete one are + * indistinguishable by inspection — both are arrays of real cycles — so the + * difference has to be carried in the type, not in a comment or a count that + * happens to look small. + */ +export type ImportCycleReport = + | { + readonly enumeration: 'complete'; + /** + * Every elementary cycle, each as `[n0, n1, ..., nk, n0]` — the first + * node repeated at the end so the closing edge is explicit. Sorted. + */ + readonly cycles: readonly string[][]; + /** + * Number of cyclic strongly connected components — the count of + * independent tangles. This is what the previous implementation called + * the cycle count; it is NOT the number of cycles. + */ + readonly componentCount: number; + } + | { + /** + * A bound was hit, so the enumeration is abandoned — but the SCC + * decomposition had already finished, so every tangle is known and each + * one gets a representative. This is strictly more useful than an error: + * a CI job can act on "these 11 components are cyclic, here is one cycle + * through each", and cannot act on nothing at all. + * + * What is NOT carried is any count of cycles. `componentCount` is exact; + * the number of elementary cycles is unknown and stays unknown. + */ + readonly enumeration: 'component-representatives'; + /** One cycle per component, same shape and ordering as the complete list. */ + readonly cycles: readonly string[][]; + readonly componentCount: number; + readonly reason: ImportCycleLimit; + readonly limit: number; + } + | { + /** + * A bound was hit inside the decomposition itself, so not even the tangle + * count is known. There is genuinely nothing to report. + */ + readonly enumeration: 'none'; + readonly reason: ImportCycleLimit; + readonly limit: number; + }; + +/** Sorted forward adjacency plus the set of nodes that import themselves. */ +interface ImportGraph { + readonly adjacency: ReadonlyMap; + readonly nodes: readonly string[]; + readonly selfLoops: ReadonlySet; +} + +function buildGraph(edges: readonly ImportEdge[]): ImportGraph { + const targetsBySource = new Map>(); + for (const { source, target } of edges) { + if (!source || !target) continue; + const targets = targetsBySource.get(source) ?? new Set(); + targets.add(target); + targetsBySource.set(source, targets); + if (!targetsBySource.has(target)) targetsBySource.set(target, new Set()); + } + + const adjacency = new Map(); + const selfLoops = new Set(); + for (const [source, targets] of targetsBySource) { + adjacency.set(source, [...targets].sort()); + if (targets.has(source)) selfLoops.add(source); + } + return { adjacency, nodes: [...adjacency.keys()].sort(), selfLoops }; +} + +interface CircuitSearch { + readonly cycles: string[][]; + readonly cycleLimit: number; + readonly workLimit: number; + /** Search effort so far, across the SCC passes and the circuit search alike. */ + work: number; + /** Non-null once a bound is hit; every loop unwinds on it. */ + exceeded: ImportCycleLimit | null; +} + +/** + * Charge `amount` units of search effort. Returns true once the budget is + * spent, which every caller must honour — a bulk charge that is not checked + * would let the search run on past the bound it just crossed. + */ +function overBudget(search: CircuitSearch, amount = 1): boolean { + search.work += amount; + if (search.work <= search.workLimit) return false; + search.exceeded = 'work'; + return true; +} + +/** + * Strongly connected components of the subgraph induced by `allowed`, via an + * iterative Tarjan. Iterative because import graphs reach 10^5 files and a + * recursive walk would blow the stack long before that. + * + * `roots` fixes the order the outer loop starts from, which is what makes the + * component set — and so Johnson's choice of root — deterministic. + * + * Abandons the pass and returns a partial list if the work budget runs out, so + * every caller must check `search.exceeded` before using the result. + */ +function stronglyConnectedComponents( + roots: readonly string[], + adjacency: ReadonlyMap, + allowed: ReadonlySet, + search: CircuitSearch, +): string[][] { + const index = new Map(); + const lowLink = new Map(); + const onStack = new Set(); + const pending: string[] = []; + const components: string[][] = []; + let counter = 0; + // One pass over the roots happens even for a component with no edges left. + if (overBudget(search, roots.length)) return components; + + for (const root of roots) { + if (index.has(root)) continue; + index.set(root, counter); + lowLink.set(root, counter); + counter += 1; + pending.push(root); + onStack.add(root); + const frames = [{ node: root, nextIndex: 0 }]; + + while (frames.length > 0) { + const frame = frames[frames.length - 1]; + const neighbors = adjacency.get(frame.node) ?? []; + if (frame.nextIndex < neighbors.length) { + const next = neighbors[frame.nextIndex]; + frame.nextIndex += 1; + if (overBudget(search)) return components; + if (!allowed.has(next)) continue; + if (!index.has(next)) { + index.set(next, counter); + lowLink.set(next, counter); + counter += 1; + pending.push(next); + onStack.add(next); + frames.push({ node: next, nextIndex: 0 }); + } else if (onStack.has(next)) { + lowLink.set(frame.node, Math.min(lowLink.get(frame.node)!, index.get(next)!)); + } + continue; + } + + frames.pop(); + const node = frame.node; + if (lowLink.get(node)! === index.get(node)!) { + const component: string[] = []; + for (;;) { + const member = pending.pop()!; + onStack.delete(member); + component.push(member); + if (member === node) break; + } + components.push(component); + } + if (frames.length > 0) { + const parent = frames[frames.length - 1].node; + lowLink.set(parent, Math.min(lowLink.get(parent)!, lowLink.get(node)!)); + } + } + } + + return components; +} + +/** A component that contains at least one cycle: two-plus members, or a self-import. */ +function isCyclic(component: readonly string[], selfLoops: ReadonlySet): boolean { + return component.length > 1 || selfLoops.has(component[0]); +} + +function leastNode(nodes: readonly string[]): string { + let least = nodes[0]; + for (const node of nodes) if (node < least) least = node; + return least; +} + +/** Order components by their least node. Components are disjoint, so this is total. */ +function byLeastNode(left: readonly string[], right: readonly string[]): number { + const leftLeast = leastNode(left); + const rightLeast = leastNode(right); + return compareCodeUnits(leftLeast, rightLeast); +} + +/** + * Johnson's `UNBLOCK`, iterative. Lifts `node` and everything transitively + * waiting on it out of `blocked`, so a path that was abandoned as fruitless + * becomes explorable again once the reason it was fruitless is gone. + */ +function unblock(node: string, blocked: Set, blockedBy: Map>): void { + const stack = [node]; + while (stack.length > 0) { + const current = stack.pop()!; + blocked.delete(current); + const waiting = blockedBy.get(current); + if (waiting === undefined || waiting.size === 0) continue; + for (const dependent of waiting) { + if (blocked.has(dependent)) stack.push(dependent); + } + waiting.clear(); + } +} + +/** + * Johnson's `CIRCUIT`, iterative — enumerate the elementary circuits rooted at + * `root` inside `allowed`. + * + * Every circuit found here starts and ends at `root`, and `root` is the least + * node of `allowed` by construction, which is where the rotation guarantee in + * the module docblock comes from. + */ +function enumerateCircuitsFrom( + root: string, + adjacency: ReadonlyMap, + allowed: ReadonlySet, + search: CircuitSearch, +): void { + const blocked = new Set([root]); + const blockedBy = new Map>(); + const path = [root]; + // `neighbors` rides the frame: the list is fixed for a node, while this loop + // re-enters per DFS STEP — ~2.6M iterations against 395k pushes on this + // repository, so looking it up per iteration re-hashes the path each time. + const frames = [ + { node: root, nextIndex: 0, foundCircuit: false, neighbors: adjacency.get(root) ?? [] }, + ]; + + while (frames.length > 0) { + // Budget spent: return rather than unwind. Everything this function owns is + // local and it returns void, so draining the stack would run the full + // `blockedBy` bookkeeping (or an `unblock` walk) per frame, to no effect, + // on exactly the graphs already judged too expensive. + if (search.exceeded !== null) return; + const frame = frames[frames.length - 1]; + const neighbors = frame.neighbors; + + if (frame.nextIndex < neighbors.length) { + const next = neighbors[frame.nextIndex]; + frame.nextIndex += 1; + if (overBudget(search)) continue; + if (!allowed.has(next)) continue; + if (next === root) { + // `path` is the elementary path root -> ... -> frame.node; closing it + // back onto the root yields the cycle in the documented shape. A + // self-import lands here on the first step with path === [root]. + search.cycles.push([...path, root]); + frame.foundCircuit = true; + // One-past, matching the edge-limit guard in `check`: `cycleLimit` + // cycles is an acceptable answer, and it takes finding one MORE to + // prove the graph overflowed. Stopping at `>= cycleLimit` would fail a + // graph that has exactly that many cycles and could have been reported + // in full. + // + // Tested before the emission charge so that a graph over both bounds + // reports the cycle cap, which is the one a reader can act on, rather + // than whichever happened to trip first. + if (search.cycles.length > search.cycleLimit) { + search.exceeded = 'cycles'; + continue; + } + // A found cycle is not merely traversed: it is copied, retained until + // the call returns, sorted, and serialized into an MCP response. So it + // is charged at EMITTED_NODE_COST per node, not 1. This is what bounds + // MEMORY as well as time — 10,000 cycles is a modest cap when cycles + // are four files long and a heap-exhausting one when a single strongly + // connected component is 50,000 files around. + overBudget(search, (path.length + 1) * EMITTED_NODE_COST); + continue; + } + if (!blocked.has(next)) { + blocked.add(next); + path.push(next); + frames.push({ + node: next, + nextIndex: 0, + foundCircuit: false, + neighbors: adjacency.get(next) ?? [], + }); + } + continue; + } + + // Leaving `frame.node`. If it reached the root, it may lie on further + // circuits, so it and its waiters go back in play. If it did not, it is + // recorded as a dead end on each of its successors: it stays blocked until + // one of them is unblocked, which is the pruning that makes Johnson's + // output-sensitive rather than exponential in the graph size. + frames.pop(); + path.pop(); + if (frame.foundCircuit) { + unblock(frame.node, blocked, blockedBy); + } else { + for (const next of neighbors) { + if (!allowed.has(next)) continue; + // `set` only when the entry is created: re-setting an existing key + // re-hashes the path string for no effect, and this runs once per + // out-edge of every unwound frame — measured at 1.06M redundant + // `Map.set` calls on this repository's own import graph. + let waiting = blockedBy.get(next); + if (waiting === undefined) { + waiting = new Set(); + blockedBy.set(next, waiting); + } + waiting.add(frame.node); + } + } + if (frames.length > 0 && frame.foundCircuit) { + frames[frames.length - 1].foundCircuit = true; + } + } +} + +/** + * Johnson's outer loop over one cyclic component: search the circuits rooted at + * the component's least node, drop that node, and repeat on whatever cyclic + * components the remainder falls into. + * + * Dropping the root is the whole rotation guarantee. Every cycle left after the + * drop consists of nodes greater than every root taken so far, so when the + * component holding it finally has that cycle's own minimum as its least node, + * the cycle is emitted once, rooted there. No other rotation is reachable, + * because the other rotations' starting nodes have already been excluded or are + * not the component's least. + * + * Re-decomposing the REMAINDER rather than the original node range also keeps + * each pass proportional to what is left: a tangle that falls apart when its + * busiest file is removed stops costing anything immediately. + */ +function enumerateComponentCycles( + component: readonly string[], + graph: ImportGraph, + search: CircuitSearch, +): void { + // Components still to search. Pushed so that they pop in increasing order of + // least node — see the sort below. + const stack: string[][] = [[...component]]; + + while (stack.length > 0 && search.exceeded === null) { + const current = stack.pop()!; + const root = leastNode(current); + // Scanning for the root and materializing the allowed set both cost one + // pass over the component, and both happen once per root, so they are the + // O(n^2) term on a component that never splits. Charged, or the budget + // would not see the work it exists to bound. + if (overBudget(search, current.length)) return; + enumerateCircuitsFrom(root, graph.adjacency, new Set(current), search); + if (search.exceeded !== null) return; + + const remaining = current.filter((node) => node !== root); + if (remaining.length === 0) continue; + // Deliberately NOT re-sorted: the SCC set is independent of the order its + // roots are visited in, `leastNode` picks Johnson's root regardless, and + // the finished cycle list is sorted at the end. Sorting here would add an + // O(n log n) term to every root for no observable difference. + const decomposed = stronglyConnectedComponents( + remaining, + graph.adjacency, + new Set(remaining), + search, + ); + // Same rule as above the call: once the budget is spent the `while` will + // refuse to pop whatever we push, so the filter/decorate/sort is waste. + if (search.exceeded !== null) return; + const subComponents = decomposed + .filter((subComponent) => isCyclic(subComponent, graph.selfLoops)) + .map((subComponent) => ({ least: leastNode(subComponent), nodes: subComponent })) + // Descending, so the stack pops them in increasing order of least node — + // Johnson's root order, and what makes a budget-stopped run stop at a + // deterministic point rather than wherever iteration happened to be. + .sort((a, b) => -compareCodeUnits(a.least, b.least)); + for (const subComponent of subComponents) stack.push(subComponent.nodes); + } +} + +/** + * The shortest cycle through a component's least node, by breadth-first search + * across the component. + * + * This is the fallback when a bound stops the full enumeration: one concrete, + * checkable cycle naming each tangle. It is also exactly what this module + * returned for every component before elementary enumeration existed, so the + * degraded answer is no worse than the old complete answer. + * + * Linear in the component, and it runs only after the decomposition has already + * succeeded, so it cannot fail the way the enumeration did. The budget is + */ +function representativeCycle( + component: readonly string[], + adjacency: ReadonlyMap, +): string[] { const allowed = new Set(component); - const start = component[0]; + const start = leastNode(component); const parents = new Map([[start, null]]); const queue = [start]; @@ -29,82 +586,73 @@ function findCyclePath(component: string[], adjacency: Map): s } } - throw new Error('Invariant violation: no cycle found through SCC root.'); + // Unreachable: every component reaching here is cyclic, and BFS from its + // least node inside the component must close. Thrown rather than returned + // empty so a future change that breaks the invariant is loud. + throw new Error('Invariant violation: no cycle found through cyclic component root.'); +} + +/** Element-wise lexicographic order, so the finished list is byte-stable. */ +function compareCycles(left: readonly string[], right: readonly string[]): number { + const shared = Math.min(left.length, right.length); + for (let index = 0; index < shared; index += 1) { + const order = compareCodeUnits(left[index], right[index]); + if (order !== 0) return order; + } + return left.length - right.length; } /** - * Return one deterministic concrete cycle for every cyclic strongly connected - * component in the file import graph. + * Enumerate every elementary cycle in the file import graph. + * + * The result is discriminated on `enumeration`; see `ImportCycleReport` for + * what each variant carries. Past either bound the enumeration is discarded + * rather than truncated — see the module docblock for the algorithm, the + * rotation rule, and why a partial cycle list is not a safe thing to return. */ -export function findImportCycles(edges: ImportEdge[]): string[][] { - const adjacency = new Map>(); - for (const { source, target } of edges) { - if (!source || !target) continue; - const targets = adjacency.get(source) ?? new Set(); - targets.add(target); - adjacency.set(source, targets); - if (!adjacency.has(target)) adjacency.set(target, new Set()); +export function findImportCycles( + edges: readonly ImportEdge[], + cycleLimit: number = IMPORT_CYCLE_LIMIT, + workLimit: number = IMPORT_CYCLE_WORK_LIMIT, +): ImportCycleReport { + const graph = buildGraph(edges); + const allNodes = new Set(graph.nodes); + const search: CircuitSearch = { cycles: [], cycleLimit, workLimit, work: 0, exceeded: null }; + + const decomposition = stronglyConnectedComponents(graph.nodes, graph.adjacency, allNodes, search); + // Only a decomposition that ran to completion has a trustworthy count; one + // abandoned mid-pass would undercount silently. + const decompositionComplete = search.exceeded === null; + const cyclicComponents = decomposition + .filter((component) => isCyclic(component, graph.selfLoops)) + .sort(byLeastNode); + + for (const component of cyclicComponents) { + if (search.exceeded !== null) break; + enumerateComponentCycles(component, graph, search); } - const sortedAdjacency = new Map( - [...adjacency].map(([node, targets]) => [node, [...targets].sort()] as const), - ); - const reverseAdjacency = new Map(); - for (const node of sortedAdjacency.keys()) reverseAdjacency.set(node, []); - for (const [source, targets] of sortedAdjacency) { - for (const target of targets) reverseAdjacency.get(target)!.push(source); + if (search.exceeded !== null) { + const reason = search.exceeded; + const limit = reason === 'cycles' ? cycleLimit : workLimit; + if (!decompositionComplete) return { enumeration: 'none', reason, limit }; + // Whatever the abandoned enumeration accumulated is discarded — it is a + // partial list of elementary cycles and would read as a complete one. + // Representatives are a different KIND of list, one per component, and the + // report says so in the type. + return { + enumeration: 'component-representatives', + cycles: cyclicComponents + .map((component) => representativeCycle(component, graph.adjacency)) + .sort(compareCycles), + componentCount: cyclicComponents.length, + reason, + limit, + }; } - for (const sources of reverseAdjacency.values()) sources.sort(); - - const visited = new Set(); - const finishOrder: string[] = []; - const components: string[][] = []; - - for (const start of [...sortedAdjacency.keys()].sort()) { - if (visited.has(start)) continue; - visited.add(start); - const stack = [{ node: start, nextIndex: 0 }]; - while (stack.length > 0) { - const frame = stack[stack.length - 1]; - const neighbors = sortedAdjacency.get(frame.node) ?? []; - if (frame.nextIndex < neighbors.length) { - const next = neighbors[frame.nextIndex++]; - if (!visited.has(next)) { - visited.add(next); - stack.push({ node: next, nextIndex: 0 }); - } - } else { - finishOrder.push(frame.node); - stack.pop(); - } - } - } - - visited.clear(); - for (let index = finishOrder.length - 1; index >= 0; index -= 1) { - const start = finishOrder[index]; - if (visited.has(start)) continue; - const component: string[] = []; - const stack = [start]; - visited.add(start); - while (stack.length > 0) { - const node = stack.pop()!; - component.push(node); - for (const next of reverseAdjacency.get(node) ?? []) { - if (visited.has(next)) continue; - visited.add(next); - stack.push(next); - } - } - component.sort(); - components.push(component); - } - - return components - .filter( - (component) => - component.length > 1 || (sortedAdjacency.get(component[0]) ?? []).includes(component[0]), - ) - .sort((a, b) => (a[0] < b[0] ? -1 : a[0] > b[0] ? 1 : 0)) - .map((component) => findCyclePath(component, sortedAdjacency)); + return { + enumeration: 'complete', + cycles: search.cycles.sort(compareCycles), + componentCount: cyclicComponents.length, + }; } diff --git a/gitnexus/src/core/group/bridge-db.ts b/gitnexus/src/core/group/bridge-db.ts index 15bb1fb15..5fbe711aa 100644 --- a/gitnexus/src/core/group/bridge-db.ts +++ b/gitnexus/src/core/group/bridge-db.ts @@ -1,6 +1,6 @@ import fsp from 'node:fs/promises'; import path from 'node:path'; -import { createHash, randomBytes } from 'node:crypto'; +import { createHash } from 'node:crypto'; import lbug from '@ladybugdb/core'; import type { LbugValue } from '@ladybugdb/core'; import type { BridgeHandle, BridgeMeta, StoredContract, CrossLink, RepoSnapshot } from './types.js'; @@ -12,7 +12,7 @@ import { } from '../lbug/lbug-config.js'; import { dedupeContracts, dedupeCrossLinks } from './normalization.js'; import { createLogger } from '../logger.js'; -import { retryRename } from '../../storage/fs-atomic.js'; +import { retryRename, writeFileAtomic } from '../../storage/fs-atomic.js'; const bridgeLogger = createLogger('bridge-db', { debugEnvVar: 'GITNEXUS_DEBUG_BRIDGE', @@ -647,30 +647,7 @@ export async function closeBridgeDb(handle: BridgeHandle): Promise { /* ------------------------------------------------------------------ */ export async function writeBridgeMeta(groupDir: string, meta: BridgeMeta): Promise { - const target = path.join(groupDir, 'meta.json'); - // Unpredictable suffix + O_EXCL via `'wx'` flag closes the symlink/ - // pre-create attack window. The third argument `0o600` is the - // user-only mode mask — CodeQL's `js/insecure-temporary-file` query - // sources its verdict from the `mode` argument, NOT from `flags`: - // its `isSecureMode(mode)` predicate requires the low 6 bits to be - // zero (no group/world bits). Without an explicit mode the file is - // created with the process umask (typically 0o644 = group/world - // readable), which the query treats as the actual vulnerability. - // Both `'wx'` (runtime O_EXCL) AND `0o600` (CodeQL-credited mode) - // are needed: one closes the symlink race, the other closes the - // permissions exposure. - const tmp = `${target}.tmp.${randomBytes(8).toString('hex')}`; - const handle = await fsp.open(tmp, 'wx', 0o600); - try { - await handle.writeFile(JSON.stringify(meta, null, 2), 'utf-8'); - } finally { - await handle.close(); - } - // Use retryRename for consistency with writeBridge's atomic swap — on - // Windows a concurrent reader can cause EBUSY/EPERM even on a tiny - // meta.json, and we don't want meta write to be less robust than the - // bridge.lbug swap it accompanies. - await retryRename(tmp, target); + await writeFileAtomic(path.join(groupDir, 'meta.json'), JSON.stringify(meta, null, 2)); } export async function readBridgeMeta(groupDir: string): Promise { diff --git a/gitnexus/src/core/group/extractors/http-patterns/java.ts b/gitnexus/src/core/group/extractors/http-patterns/java.ts index 0c83eaab4..4eba0c1f1 100644 --- a/gitnexus/src/core/group/extractors/http-patterns/java.ts +++ b/gitnexus/src/core/group/extractors/http-patterns/java.ts @@ -7,7 +7,8 @@ import { type LanguagePatterns, } from '../tree-sitter-scanner.js'; import { - METHOD_ANNOTATION_TO_HTTP, + springAnnotationHttpMethods, + intersectSpringHttpMethods, isRouteMemberKey, findEnclosingClass, joinPath, @@ -44,7 +45,7 @@ import type { /** * Java HTTP plugin. Handles: - * - Spring `@RequestMapping` class prefixes + `@(Get|Post|...)Mapping` method annotations + * - Spring `@RequestMapping` class prefixes + shortcut/`@RequestMapping` method annotations * - Spring `RestTemplate.getForObject/...`, `exchange(...)` * - Spring `WebClient.method(HttpMethod.X, ...)`, `WebClient.get().uri(...)` * - OkHttp `new Request.Builder().url("...")` @@ -408,6 +409,35 @@ function simpleName(text: string): string { return text.split('.').pop() ?? text; } +function declarationAnnotations(node: Parser.SyntaxNode): Parser.SyntaxNode[] { + const modifiers = node.namedChildren.find((child) => child.type === 'modifiers'); + if (!modifiers) return []; + return modifiers.namedChildren.filter( + (child) => child.type === 'annotation' || child.type === 'marker_annotation', + ); +} + +function annotationHasRouteMember(annotation: Parser.SyntaxNode): boolean { + const args = annotation.childForFieldName('arguments'); + if (!args) return false; + for (const child of args.namedChildren) { + if (child.type !== 'element_value_pair') return true; + const key = child.childForFieldName('key'); + if (isRouteMemberKey(key ?? undefined)) return true; + } + return false; +} + +function typeRequestMethods(typeNode: Parser.SyntaxNode): readonly string[] { + const mappings = declarationAnnotations(typeNode).filter( + (annotation) => + simpleName(annotation.childForFieldName('name')?.text ?? '') === 'RequestMapping', + ); + if (mappings.length === 0) return ['*']; + if (mappings.length !== 1) return []; + return springAnnotationHttpMethods('RequestMapping', mappings[0].text); +} + function hasAnnotation(node: Parser.SyntaxNode, names: string | readonly string[]): boolean { const modifiers = node.namedChildren.find((child) => child.type === 'modifiers'); if (!modifiers) return false; @@ -437,6 +467,8 @@ interface MethodRouteAnnotation { methodName: string | null; httpMethod: string; rawPath: string; + /** OpenFeign's single effective verb; null means its contract is invalid/ambiguous. */ + feignHttpMethod?: string | null; } interface RequestLineAnnotation { @@ -452,7 +484,7 @@ interface RouteAnnotationScan { feignPrefixByInterfaceId: Map; /** Spring HTTP Interface `@HttpExchange(url|value)` type-level prefixes per class/interface node id. */ httpExchangePrefixByTypeId: Map; - /** One entry per resolved Spring `@(Get|...)Mapping` route — a method with N mappings yields N entries. */ + /** Resolved Spring shortcut/`@RequestMapping` routes — paths × verbs yield one entry each. */ methodRoutes: MethodRouteAnnotation[]; /** One entry per OpenFeign `@RequestLine` whose value parses to a verb + path. */ requestLines: RequestLineAnnotation[]; @@ -484,6 +516,7 @@ function scanRouteAnnotations(tree: Parser.Tree): RouteAnnotationScan { const methodRoutes: MethodRouteAnnotation[] = []; const requestLines: RequestLineAnnotation[] = []; const exchangeRoutes: MethodRouteAnnotation[] = []; + const httpMethodsByAnnotationId = new Map(); // Interface `@RequestMapping` prefixes rank below `@FeignClient(path)`; // collect them and apply only after the FeignClient pass below. const interfaceRequestMappingPrefixes: Array<{ id: number; prefix: string }> = []; @@ -505,18 +538,29 @@ function scanRouteAnnotations(tree: Parser.Tree): RouteAnnotationScan { const keyNode = captures.key; // undefined for the positional shape if (node.type === 'method_declaration') { - // Method-level: a Spring `@(Get|...)Mapping` route, or native `@RequestLine`. - const httpMethod = METHOD_ANNOTATION_TO_HTTP[ann]; - if (httpMethod) { + // Method-level: a Spring shortcut/`@RequestMapping` route, or native `@RequestLine`. + const annotationNode = annNode.parent; + if (!annotationNode) continue; + let httpMethods = httpMethodsByAnnotationId.get(annotationNode.id); + if (!httpMethods) { + httpMethods = springAnnotationHttpMethods(ann, annotationNode.text); + httpMethodsByAnnotationId.set(annotationNode.id, httpMethods); + } + if (httpMethods.length > 0) { + const feignHttpMethod = + httpMethods.length === 1 ? (httpMethods[0] === '*' ? 'GET' : httpMethods[0]) : null; if (!isRouteMemberKey(keyNode)) continue; const rawPath = unquoteLiteral(valueNode.text); if (rawPath !== null) { - methodRoutes.push({ - methodNode: node, - methodName: captures.member?.text ?? null, - httpMethod, - rawPath, - }); + for (const httpMethod of httpMethods) { + methodRoutes.push({ + methodNode: node, + methodName: captures.member?.text ?? null, + httpMethod, + rawPath, + feignHttpMethod, + }); + } } } else if (ann === 'RequestLine') { // Feign packs verb + path in one literal; its only named argument is `value`. @@ -573,6 +617,43 @@ function scanRouteAnnotations(tree: Parser.Tree): RouteAnnotationScan { } } + const classHttpMethodsByTypeId = new Map(); + for (const match of runCompiledPatterns(SPRING_TYPE_DECLARATION_PATTERNS, tree)) { + const typeNode = match.captures.type; + if (!typeNode) continue; + const classMethods = typeRequestMethods(typeNode); + classHttpMethodsByTypeId.set(typeNode.id, classMethods); + for (const methodNode of collectDirectMethods(typeNode)) { + for (const annotationNode of declarationAnnotations(methodNode)) { + if (annotationHasRouteMember(annotationNode)) continue; + const ann = simpleName(annotationNode.childForFieldName('name')?.text ?? ''); + const httpMethods = springAnnotationHttpMethods(ann, annotationNode.text); + if (httpMethods.length === 0) continue; + const feignHttpMethod = + httpMethods.length === 1 ? (httpMethods[0] === '*' ? 'GET' : httpMethods[0]) : null; + for (const httpMethod of httpMethods) { + methodRoutes.push({ + methodNode, + methodName: getNodeName(methodNode), + httpMethod, + rawPath: '', + feignHttpMethod, + }); + } + } + } + } + + const constrainedMethodRoutes = methodRoutes.flatMap((route) => { + const typeNode = + findEnclosingInterface(route.methodNode) ?? findEnclosingClass(route.methodNode); + const classMethods = typeNode ? (classHttpMethodsByTypeId.get(typeNode.id) ?? ['*']) : ['*']; + return intersectSpringHttpMethods(classMethods, [route.httpMethod]).map((httpMethod) => ({ + ...route, + httpMethod, + })); + }); + // `@RequestMapping` on a Feign interface is the fallback prefix, but only when // the interface has no `@FeignClient(path)` of its own (path wins). for (const { id, prefix } of interfaceRequestMappingPrefixes) { @@ -583,7 +664,7 @@ function scanRouteAnnotations(tree: Parser.Tree): RouteAnnotationScan { prefixByTypeId, feignPrefixByInterfaceId, httpExchangePrefixByTypeId, - methodRoutes, + methodRoutes: constrainedMethodRoutes, requestLines, exchangeRoutes, }; @@ -705,7 +786,7 @@ export const JAVA_HTTP_PLUGIN: HttpLanguagePlugin = { // ─── Spring providers + OpenFeign consumers (one query pass) ──── // `scanRouteAnnotations` resolves every route-defining annotation — - // class/interface prefixes, method `@(Get|...)Mapping`s and native + // class/interface prefixes, method shortcut/`@RequestMapping`s and native // `@RequestLine`s — from a single `matches()` pass over the tree. const { prefixByTypeId, @@ -724,12 +805,13 @@ export const JAVA_HTTP_PLUGIN: HttpLanguagePlugin = { for (const route of methodRoutes) { const enclosingInterface = findEnclosingInterface(route.methodNode); if (enclosingInterface && hasAnnotation(enclosingInterface, 'FeignClient')) { + if (!route.feignHttpMethod) continue; const prefixes = feignPrefixByInterfaceId.get(enclosingInterface.id) ?? ['']; for (const prefix of prefixes) { out.push({ role: 'consumer', framework: OPENFEIGN_FRAMEWORK, - method: route.httpMethod, + method: route.feignHttpMethod, path: joinPath(prefix, route.rawPath), name: route.methodName, line: route.methodNode.startPosition.row + 1, diff --git a/gitnexus/src/core/group/extractors/http-route-extractor.ts b/gitnexus/src/core/group/extractors/http-route-extractor.ts index a6e0b046f..cd0494441 100644 --- a/gitnexus/src/core/group/extractors/http-route-extractor.ts +++ b/gitnexus/src/core/group/extractors/http-route-extractor.ts @@ -6,6 +6,7 @@ import type { ContractExtractor, CypherExecutor } from '../contract-extractor.js import type { ExtractedContract, RepoHandle } from '../types.js'; import { readSafe } from './fs-utils.js'; import { parseSourceSafe } from '../../tree-sitter/safe-parse.js'; +import { toZeroBasedLine } from '../../ingestion/utils/line-base.js'; import { logger } from '../../logger.js'; import { getPluginForFile, @@ -179,13 +180,17 @@ function resolveContainingSymbol( line: number, ): ResolvedSymbol | null { const norm = (x: unknown): string => String(x ?? ''); - // Detection lines are 1-based; symbol spans are stored 0-based for the - // languages indexed today (parse-worker records `startPosition.row`). So the - // base-correct probe is `line - 1`. Pick the INNERMOST (smallest-span) symbol - // whose span contains the probe. Only if nothing contains `line - 1` do we - // retry with the raw `line` — a defensive fallback for any future language - // that stores 1-based spans. Probing `line - 1` first (rather than OR-ing both) - // avoids the +1 slack mis-picking a one-line sibling that sits on `line`. + // Detection lines are 1-based (`HttpDetection.line`); symbol spans are stored + // 0-based for the languages indexed today (parse-worker records + // `startPosition.row`). So the base-correct probe is `toZeroBasedLine(line)` — + // the same named 1-based→graph-space conversion the ingestion emitters use + // (#2377), rather than a bare literal. Pick the INNERMOST (smallest-span) + // symbol whose span contains the probe. Only if nothing contains the 0-based + // probe do we retry with the raw `line` — a defensive fallback for any future + // language that stores 1-based spans. Probing 0-based first (rather than + // OR-ing both) avoids the +1 slack mis-picking a one-line sibling that sits on + // `line`. The helper's `Math.max(0, …)` clamp is inert here: every plugin sets + // `line` from `startPosition.row + 1`, so it is always >= 1. const pick = (probe: number): ResolvedSymbol | null => { let best: ResolvedSymbol | null = null; let bestSpan = Number.POSITIVE_INFINITY; @@ -208,7 +213,7 @@ function resolveContainingSymbol( } return best && best.uid ? best : null; }; - return pick(line - 1) ?? pick(line); + return pick(toZeroBasedLine(line)) ?? pick(line); } /** A Function/Method in the file matching `name` exactly (for named handlers). */ diff --git a/gitnexus/src/core/group/storage.ts b/gitnexus/src/core/group/storage.ts index b23f48d68..6e568cd6e 100644 --- a/gitnexus/src/core/group/storage.ts +++ b/gitnexus/src/core/group/storage.ts @@ -2,18 +2,8 @@ import * as fs from 'node:fs'; import * as fsp from 'node:fs/promises'; import * as path from 'node:path'; import * as os from 'node:os'; -import { randomBytes } from 'node:crypto'; import type { ContractRegistry } from './types.js'; -import { retryRename } from '../../storage/fs-atomic.js'; - -/** - * Build an unpredictable suffix for atomic-write tmp files. Replaces the - * previous `Date.now()` pattern which CodeQL flagged as - * js/insecure-temporary-file: a guessable suffix in a writable directory - * lets a co-located attacker pre-create or symlink the tmp path before the - * write lands. - */ -const tmpSuffix = (): string => randomBytes(8).toString('hex'); +import { writeFileAtomic } from '../../storage/fs-atomic.js'; const CONTRACTS_FILE = 'contracts.json'; @@ -44,29 +34,7 @@ export async function writeContractRegistry( groupDir: string, registry: ContractRegistry, ): Promise { - const targetPath = path.join(groupDir, CONTRACTS_FILE); - const tmpPath = `${targetPath}.tmp.${tmpSuffix()}`; - - // O_EXCL via `'wx'` flag + explicit `0o600` mode — closes both halves - // of the CodeQL js/insecure-temporary-file finding: `'wx'` rejects a - // pre-planted symlink at the path, and `0o600` (user-only) prevents - // the file from being created group/world readable while it briefly - // contains contract data en route to the rename. The query's - // `isSecureMode` predicate inspects ONLY the mode argument, not the - // flags, so the explicit mode is what credits the fix. - const handle = await fsp.open(tmpPath, 'wx', 0o600); - try { - await handle.writeFile(JSON.stringify(registry, null, 2), 'utf-8'); - } finally { - await handle.close(); - } - // retryRename absorbs the documented Windows EPERM/EBUSY/EACCES race that - // fires when AV scanners or another concurrent rename briefly hold the - // destination handle between rename calls. Same helper bridge-db.ts uses - // (lines 304, 583, 587, 595, 605, 677) for the bridge.lbug atomic swap — - // single source of truth for the Windows-rename pattern across the group - // package. - await retryRename(tmpPath, targetPath); + await writeFileAtomic(path.join(groupDir, CONTRACTS_FILE), JSON.stringify(registry, null, 2)); } export async function readContractRegistry(groupDir: string): Promise { diff --git a/gitnexus/src/core/index-freshness.ts b/gitnexus/src/core/index-freshness.ts index 26c107f38..ea577f834 100644 --- a/gitnexus/src/core/index-freshness.ts +++ b/gitnexus/src/core/index-freshness.ts @@ -5,16 +5,143 @@ export const INDEX_INCOMPLETE_REASONS = [ 'incremental-in-progress', 'embedding-checkpoint-pending', 'embedding-count-unverified', + 'graph-write-collapsed', ] as const; export type IndexIncompleteReason = (typeof INDEX_INCOMPLETE_REASONS)[number]; +/** + * Fraction of the pipeline's relationship count that must survive into the DB + * before the write counts as collapsed. Deliberately generous: this detects + * "most of the graph did not persist" (the reported case lost ~91%), not a + * per-edge reconciliation. + */ +export const GRAPH_WRITE_COLLAPSE_RATIO = 0.5; + +/** + * Below this many relationships the ratio is meaningless — a handful of edges + * lost to legitimate filtering would trip it — so small repos are exempt. + */ +export const GRAPH_WRITE_COLLAPSE_MIN_EDGES = 100; + +/** Why {@link detectGraphWriteCollapse} could reach no verdict at all. */ +export type GraphWriteCollapseUnmeasurableReason = + /** The pipeline's own total was not a usable number (or was zero). */ + | 'expected-unavailable' + /** The DB-side count could not be READ — a query that threw, no connection. */ + | 'persisted-unreadable' + /** Set by the CALLER: an incremental write persists only the changed + * subgraph, so whole-scope counts are not comparable to it. */ + | 'incremental-write'; + +/** + * The three outcomes of the collapse check, kept APART because two of them used + * to share `undefined` and the conflation erased a stamp recording real, + * unrepaired edge loss. + * + * `'healthy'` is a POSITIVE all-clear — the counts were both taken and enough + * rows persisted — and is the only outcome that licenses clearing a previous + * `graph-write-collapsed` stamp. `'unmeasurable'` says the comparison never + * happened; the previous stamp must survive it, because nothing has repaired + * whatever it recorded. + */ +export type GraphWriteCollapseVerdict = + | { verdict: 'collapsed'; expected: number; persisted: number } + | { verdict: 'healthy' } + | { verdict: 'unmeasurable'; reason: GraphWriteCollapseUnmeasurableReason }; + +/** + * Decide whether a finished write collapsed, comparing what the pipeline + * produced against what the DB hands back. + * + * A RATIO, not equality: some relationship types do not round-trip one-for-one + * and `--pdg` writes MORE rows into the same table, so demanding equality would + * fire on healthy runs. Only a collapse is a defect. + * + * FAIL-SAFE at `expected === 0`: an implementation that offloads relationships + * out of memory may not be able to report a total, and a false "your index is + * broken" is worse than a missed one. That case is `'unmeasurable'`, NOT + * `'healthy'` — nothing was compared, so nothing was cleared. + * + * Returns a THREE-WAY verdict rather than `{...} | undefined`. The absent value + * meant both "measured, fine" and "could not measure", and the caller — which + * decides whether to keep or erase the persisted `graph-write-collapsed` stamp — + * cannot tell those apart from a shared `undefined`. It guessed by write mode + * instead, so a full run whose structural count threw took the + * "no collapse ⇒ clear it" branch and deleted a stamp recording real loss. + */ +export function detectGraphWriteCollapse( + expected: number, + /** + * Relationships readable from the DB, or `undefined` when the count could + * not be READ at all (no connection, a query that threw). + * + * The distinction is load-bearing and was got wrong once: `getLbugStats` + * reports `edges: 0` for "no connection", "query threw" AND "empty table" + * alike, so passing it straight in made every run without a readable DB look + * like a total collapse. An unmeasurable count is not a measured zero — + * accepting `undefined` here is what keeps this check from committing the + * same confident-zero error it exists to catch. + */ + persisted: number | undefined, +): GraphWriteCollapseVerdict { + // Both sides must be REAL NUMBERS before any comparison. A non-numeric + // `expected` (a graph implementation that reports no total, a lightweight + // pipeline result) does not merely skip the guards — it INVERTS them: + // `undefined < 100` is false, so the min-edges exemption never fires, and + // `0 >= undefined * 0.5` is `0 >= NaN`, also false, so the ratio check + // "passes" too and a healthy run is reported as a total collapse. Comparing + // against a non-number is the one way this check can manufacture the exact + // false certainty it was written to prevent. + if (!Number.isFinite(expected)) { + return { verdict: 'unmeasurable', reason: 'expected-unavailable' }; + } + if (typeof persisted !== 'number' || !Number.isFinite(persisted)) { + return { verdict: 'unmeasurable', reason: 'persisted-unreadable' }; + } + const expectedCount = expected; + const persistedCount = persisted; + // FAIL-SAFE, and `'unmeasurable'` rather than `'healthy'`: a zero expectation + // is the documented "could not report a total" case, not evidence the write + // went well. Reporting it as an all-clear would let a run that measured + // nothing erase a stamp recording a previous run's real loss. + if (expectedCount === 0) { + return { verdict: 'unmeasurable', reason: 'expected-unavailable' }; + } + // A TOTAL loss is never small enough to excuse. The min-edges exemption + // exists for "a handful of edges lost to legitimate filtering", which its own + // docstring says — it does not describe a persisted count of zero. Evaluated + // before the exemption because the exemption looked only at `expected`: + // `expected = 99, persisted = 0` lost every single edge and still returned + // no verdict, leaving the metadata fresh and the CLI reporting success. + if (expectedCount > 0 && persistedCount === 0) { + return { verdict: 'collapsed', expected: expectedCount, persisted: persistedCount }; + } + // The small-repo exemption and the cleared ratio are both `'healthy'`, not + // `'unmeasurable'`: both counts WERE taken, and the comparison ran. Calling + // the exemption a non-verdict would make a stamp unclearable on any repo that + // shrank below the threshold — a permanent forced-rebuild wedge, which is the + // failure this taxonomy exists to avoid rather than to relocate. + if (expectedCount < GRAPH_WRITE_COLLAPSE_MIN_EDGES) return { verdict: 'healthy' }; + if (persistedCount >= expectedCount * GRAPH_WRITE_COLLAPSE_RATIO) return { verdict: 'healthy' }; + return { verdict: 'collapsed', expected: expectedCount, persisted: persistedCount }; +} + /** Stable machine-readable reasons an index cannot be certified complete. */ export function getIndexIncompleteReasons( - meta: Pick | null | undefined, + meta: + | Pick + | null + | undefined, ): IndexIncompleteReason[] { const reasons: IndexIncompleteReason[] = []; if (meta?.incrementalInProgress) reasons.push('incremental-in-progress'); + // The run finished and wrote metadata, but far fewer edges reached the DB + // than the pipeline produced — the "refresh reported success, the index is + // unusable" failure. Without this the index reads as fresh and every tool + // answers from a graph missing most of its edges, which is indistinguishable + // from a codebase that genuinely has no such relationships. + if (meta?.graphWriteCollapsed) reasons.push('graph-write-collapsed'); if (meta?.embeddingCheckpoint) { // The three checkpoint kinds are not one operator-facing state. GUARDRAILS // and the runbook document `embedding-checkpoint-pending` as "N node(s) diff --git a/gitnexus/src/core/ingestion/cluster-enricher.ts b/gitnexus/src/core/ingestion/cluster-enricher.ts index 06cd4d0cd..32bfe8b3a 100644 --- a/gitnexus/src/core/ingestion/cluster-enricher.ts +++ b/gitnexus/src/core/ingestion/cluster-enricher.ts @@ -7,6 +7,7 @@ import { CommunityNode } from './community-processor.js'; +import { chunk } from '../../lib/utils.js'; import { logger } from '../logger.js'; // ============================================================================ // TYPES @@ -160,11 +161,13 @@ export const enrichClustersBatch = async ( let tokensUsed = 0; // Process in batches - for (let i = 0; i < communities.length; i += batchSize) { - // Report progress - onProgress?.(Math.min(i + batchSize, communities.length), communities.length); - - const batch = communities.slice(i, i + batchSize); + let reported = 0; + for (const batch of chunk(communities, batchSize)) { + // Report progress. `reported` after each whole batch equals the old + // `Math.min(i + batchSize, communities.length)` — the last batch is short + // exactly when that clamp used to bite. + reported += batch.length; + onProgress?.(reported, communities.length); const batchPrompt = batch .map((community, idx) => { diff --git a/gitnexus/src/core/ingestion/di-extractors/index.ts b/gitnexus/src/core/ingestion/di-extractors/index.ts index 5b679feb1..66a42775d 100644 --- a/gitnexus/src/core/ingestion/di-extractors/index.ts +++ b/gitnexus/src/core/ingestion/di-extractors/index.ts @@ -15,56 +15,17 @@ */ import { SupportedLanguages } from 'gitnexus-shared'; -import type { GraphNode } from 'gitnexus-shared'; +import type { DiResolver } from './types.js'; import { springDiResolver } from './spring.js'; -/** A successful injection-site match, produced by a per-language resolver. */ -export interface DiInjectionMatch { - /** The requested dependency type name. */ - targetTypeName: string; - /** A collection receives every matching provider; a single site may need - * framework-specific named/preferred-provider disambiguation. */ - cardinality: 'single' | 'collection'; - /** Statically known provider name requested at the injection site. The - * resolver owns the human-readable explanation of that selection. */ - namedSelection?: { - name: string; - reason: string; - /** Name-first frameworks may fall back to type only for implicit/default - * names. Explicit names remain strict. */ - fallbackToType?: boolean; - }; - /** Most injection edges originate at the owning Class. Factory-method - * parameters preserve the Method as the semantic source. */ - edgeSource?: 'owner-class' | 'site'; - /** Human-readable edge reason. Framework specifics (names, idioms, - * collection wrapper, gating annotation) live in this payload so the - * shared `di` phase stays framework-neutral. */ - reason: string; -} - -/** Provider metadata used by the shared resolver without naming a framework. */ -export interface DiProviderMatch { - /** Provider names and aliases that can satisfy a named injection. */ - names: readonly string[]; - /** Optional type directly provided by a declaration node, such as a - * framework factory method whose node is not itself a Class. */ - providedTypeName?: string; - /** Graph node that declares this provider. The shared phase excludes a - * provider from injection into its own declaration site without knowing the - * framework-specific declaration model. */ - declaredByNodeId?: string; - /** Present when the framework marks this as its preferred candidate. The - * value is appended to the emitted edge reason when it disambiguates. */ - preferenceReason?: string; -} - -/** Per-language DI behavior. Matchers receive whole nodes so the shared phase - * remains ignorant of language/framework-specific property shapes. */ -export interface DiResolver { - matchInjectionSites(node: GraphNode): readonly DiInjectionMatch[]; - matchProvider(node: GraphNode): DiProviderMatch | null; -} +/** The resolver contract lives in the leaf `./types.js` so an implementation + * can depend on it without depending on this registry (which imports every + * implementation). The two match shapes are re-exported here because consumers + * of the registry read them off its results — `pipeline-phases/di.ts` and the + * Spring metadata modules import them from this module alongside + * `DI_RESOLVERS`. `DiResolver` itself is NOT re-exported: only implementations + * need it, and they import it from `./types.js` directly. */ +export type { DiInjectionMatch, DiProviderMatch } from './types.js'; /** All `SupportedLanguages` string values, for narrowing raw graph strings. */ const SUPPORTED_LANGUAGE_VALUES: ReadonlySet = new Set(Object.values(SupportedLanguages)); diff --git a/gitnexus/src/core/ingestion/di-extractors/spring.ts b/gitnexus/src/core/ingestion/di-extractors/spring.ts index ea5c8e727..1b0b35f6e 100644 --- a/gitnexus/src/core/ingestion/di-extractors/spring.ts +++ b/gitnexus/src/core/ingestion/di-extractors/spring.ts @@ -59,7 +59,7 @@ */ import type { GraphNode } from 'gitnexus-shared'; -import type { DiInjectionMatch, DiProviderMatch, DiResolver } from './index.js'; +import type { DiInjectionMatch, DiProviderMatch, DiResolver } from './types.js'; import { isDev } from '../utils/env.js'; import { logger } from '../../logger.js'; diff --git a/gitnexus/src/core/ingestion/di-extractors/types.ts b/gitnexus/src/core/ingestion/di-extractors/types.ts new file mode 100644 index 000000000..5a0621263 --- /dev/null +++ b/gitnexus/src/core/ingestion/di-extractors/types.ts @@ -0,0 +1,64 @@ +/** + * The DI resolver contract — the types a per-language/per-framework resolver + * implements and the shared `di` pipeline phase consumes. + * + * A leaf module by design: it imports nothing from this directory, so the + * barrel (`./index.ts`, which aggregates the resolver *implementations*) and + * each implementation (`./spring.ts`) can both depend on the contract without + * depending on each other. The barrel re-exports the two MATCH types, because + * consumers of the registry read them off its results; `DiResolver` is not + * re-exported, since only implementations need it and they import it from here + * directly. + * + * Mirrors the `import-resolvers/types.ts` split of contract from registry. + */ + +import type { GraphNode } from 'gitnexus-shared'; + +/** A successful injection-site match, produced by a per-language resolver. */ +export interface DiInjectionMatch { + /** The requested dependency type name. */ + targetTypeName: string; + /** A collection receives every matching provider; a single site may need + * framework-specific named/preferred-provider disambiguation. */ + cardinality: 'single' | 'collection'; + /** Statically known provider name requested at the injection site. The + * resolver owns the human-readable explanation of that selection. */ + namedSelection?: { + name: string; + reason: string; + /** Name-first frameworks may fall back to type only for implicit/default + * names. Explicit names remain strict. */ + fallbackToType?: boolean; + }; + /** Most injection edges originate at the owning Class. Factory-method + * parameters preserve the Method as the semantic source. */ + edgeSource?: 'owner-class' | 'site'; + /** Human-readable edge reason. Framework specifics (names, idioms, + * collection wrapper, gating annotation) live in this payload so the + * shared `di` phase stays framework-neutral. */ + reason: string; +} + +/** Provider metadata used by the shared resolver without naming a framework. */ +export interface DiProviderMatch { + /** Provider names and aliases that can satisfy a named injection. */ + names: readonly string[]; + /** Optional type directly provided by a declaration node, such as a + * framework factory method whose node is not itself a Class. */ + providedTypeName?: string; + /** Graph node that declares this provider. The shared phase excludes a + * provider from injection into its own declaration site without knowing the + * framework-specific declaration model. */ + declaredByNodeId?: string; + /** Present when the framework marks this as its preferred candidate. The + * value is appended to the emitted edge reason when it disambiguates. */ + preferenceReason?: string; +} + +/** Per-language DI behavior. Matchers receive whole nodes so the shared phase + * remains ignorant of language/framework-specific property shapes. */ +export interface DiResolver { + matchInjectionSites(node: GraphNode): readonly DiInjectionMatch[]; + matchProvider(node: GraphNode): DiProviderMatch | null; +} diff --git a/gitnexus/src/core/ingestion/filesystem-walker.ts b/gitnexus/src/core/ingestion/filesystem-walker.ts index 823b3670f..fc65cda09 100644 --- a/gitnexus/src/core/ingestion/filesystem-walker.ts +++ b/gitnexus/src/core/ingestion/filesystem-walker.ts @@ -4,6 +4,7 @@ import fs from 'fs/promises'; import path from 'path'; import { glob } from 'glob'; import { createIgnoreFilter } from '../../config/ignore-service.js'; +import { mapConcurrent } from '../../lib/utils.js'; import { logger } from '../logger.js'; @@ -135,21 +136,21 @@ export const readFileContents = async ( ): Promise> => { const contents = new Map(); - for (let start = 0; start < relativePaths.length; start += READ_CONCURRENCY) { - const batch = relativePaths.slice(start, start + READ_CONCURRENCY); - const results = await Promise.allSettled( - batch.map(async (relativePath) => { - const fullPath = path.join(repoPath, relativePath); - const content = await fs.readFile(fullPath, 'utf-8'); - return { path: relativePath, content }; - }), - ); + const results = await mapConcurrent( + relativePaths, + async (relativePath) => { + const fullPath = path.join(repoPath, relativePath); + const content = await fs.readFile(fullPath, 'utf-8'); + return { path: relativePath, content }; + }, + { concurrency: READ_CONCURRENCY }, + ); - for (const result of results) { - if (result.status === 'fulfilled') { - contents.set(result.value.path, result.value.content); - } - } + // An unreadable file yields `undefined` (mapConcurrent's per-item degrade) and + // is skipped, exactly as the previous allSettled/`status === 'fulfilled'` shape + // did — no `onError`, so the skip stays silent per this function's contract. + for (const result of results) { + if (result) contents.set(result.path, result.content); } return contents; diff --git a/gitnexus/src/core/ingestion/frameworks/spring/analysis-features.ts b/gitnexus/src/core/ingestion/frameworks/spring/analysis-features.ts index 37b5f2b14..ca79ed745 100644 --- a/gitnexus/src/core/ingestion/frameworks/spring/analysis-features.ts +++ b/gitnexus/src/core/ingestion/frameworks/spring/analysis-features.ts @@ -43,3 +43,10 @@ export const SPRING_AOP_FEATURE: AnalysisFeatureDescriptor = { version: 1, appliesTo: (filePaths) => filePaths.some(isJvmSourceFile), }; + +/** Durable completeness contract for scheduled, event, messaging, and job entry points (#2417). */ +export const SPRING_NON_HTTP_HANDLERS_FEATURE: AnalysisFeatureDescriptor = { + id: 'spring.non-http-handlers', + version: 1, + appliesTo: (filePaths) => filePaths.some(isJvmSourceFile), +}; diff --git a/gitnexus/src/core/ingestion/frameworks/spring/non-http-handlers.ts b/gitnexus/src/core/ingestion/frameworks/spring/non-http-handlers.ts new file mode 100644 index 000000000..8f3144d01 --- /dev/null +++ b/gitnexus/src/core/ingestion/frameworks/spring/non-http-handlers.ts @@ -0,0 +1,219 @@ +import type { GraphNode, ParsedFile, Range, ScopeId } from 'gitnexus-shared'; +import type { KnowledgeGraph } from '../../../graph/types.js'; +import type { ScopeResolutionIndexes } from '../../model/scope-resolution-indexes.js'; +import { resolveCallerGraphId } from '../../scope-resolution/graph-bridge/ids.js'; +import type { GraphNodeLookup } from '../../scope-resolution/graph-bridge/node-lookup.js'; +import { createSpringAnnotationNameResolver } from './bean-candidates.js'; +import { SPRING_BEAN_ANNOTATION } from './bean-factories.js'; + +export const SPRING_NON_HTTP_HANDLER_ENTRY_POINT_MULTIPLIER = 3.0; + +export type SpringNonHttpHandlerKind = 'scheduled' | 'event' | 'message' | 'xxl-job'; + +export interface SpringNonHttpHandlerAnnotationFact { + readonly name: string; + /** Kotlin use-site targets describe generated/property elements, not the callable. */ + readonly useSiteTarget?: string; +} + +export interface SpringNonHttpHandlerFact< + Annotation extends SpringNonHttpHandlerAnnotationFact = SpringNonHttpHandlerAnnotationFact, +> { + readonly ownerScopeId: ScopeId; + readonly ownerFilePath?: string; + /** Exact syntax range used only as a fail-closed bridge for collapsed language scopes. */ + readonly ownerRange?: Range; + readonly annotations: readonly Annotation[]; +} + +export interface SpringNonHttpHandlerAdapter< + Annotation extends SpringNonHttpHandlerAnnotationFact, +> { + getFacts(filePath: string): readonly SpringNonHttpHandlerFact[]; + isPackageVisibilityIncomplete(filePath: string): boolean; +} + +const SPRING_SERVICE_ACTIVATOR_ANNOTATION = + 'org.springframework.integration.annotation.ServiceActivator'; + +const HANDLER_ANNOTATIONS = new Map([ + ['org.springframework.scheduling.annotation.Scheduled', 'scheduled'], + ['org.springframework.scheduling.annotation.Schedules', 'scheduled'], + ['org.springframework.context.event.EventListener', 'event'], + ['org.springframework.transaction.event.TransactionalEventListener', 'event'], + ['org.springframework.modulith.events.ApplicationModuleListener', 'event'], + ['org.springframework.kafka.annotation.KafkaListener', 'message'], + ['org.springframework.kafka.annotation.KafkaListeners', 'message'], + ['org.springframework.amqp.rabbit.annotation.RabbitListener', 'message'], + ['org.springframework.amqp.rabbit.annotation.RabbitListeners', 'message'], + ['org.springframework.jms.annotation.JmsListener', 'message'], + ['org.springframework.jms.annotation.JmsListeners', 'message'], + ['org.springframework.pulsar.annotation.PulsarListener', 'message'], + ['org.springframework.pulsar.annotation.PulsarListeners', 'message'], + ['io.awspring.cloud.sqs.annotation.SqsListener', 'message'], + ['io.awspring.cloud.messaging.listener.annotation.SqsListener', 'message'], + ['org.springframework.cloud.aws.messaging.listener.annotation.SqsListener', 'message'], + ['org.springframework.cloud.stream.annotation.StreamListener', 'message'], + [SPRING_SERVICE_ACTIVATOR_ANNOTATION, 'message'], + ['org.springframework.messaging.handler.annotation.MessageMapping', 'message'], + ['org.springframework.messaging.simp.annotation.SubscribeMapping', 'message'], + ['com.xxl.job.core.handler.annotation.XxlJob', 'xxl-job'], +]); + +const RECOGNIZED_HANDLER_ANNOTATIONS = new Set(HANDLER_ANNOTATIONS.keys()); +const RESOLVABLE_NON_HTTP_ANNOTATIONS = new Set([ + ...RECOGNIZED_HANDLER_ANNOTATIONS, + SPRING_BEAN_ANNOTATION, +]); + +function simpleName(name: string): string { + const separator = name.lastIndexOf('.'); + return separator === -1 ? name : name.slice(separator + 1); +} + +const CAPTURE_RELEVANT_SIMPLE_NAMES = new Set([...RECOGNIZED_HANDLER_ANNOTATIONS].map(simpleName)); + +export function hasSpringNonHttpHandlerRelevantAnnotation( + annotations: readonly Pick[], +): boolean { + return annotations.some((annotation) => + CAPTURE_RELEVANT_SIMPLE_NAMES.has(simpleName(annotation.name)), + ); +} + +function exactCallableOwnersByRange(graph: KnowledgeGraph): ReadonlyMap { + const owners = new Map(); + for (const node of graph.iterNodes()) { + if ( + (node.label !== 'Method' && node.label !== 'Function') || + typeof node.properties.filePath !== 'string' + ) { + continue; + } + const key = `${node.properties.filePath}\0${node.properties.startLine}\0${node.properties.endLine}`; + owners.set(key, owners.has(key) ? null : node); + } + return owners; +} + +function ownerGraphNode( + fact: SpringNonHttpHandlerFact, + indexes: ScopeResolutionIndexes, + nodeLookup: GraphNodeLookup, + graph: KnowledgeGraph, + getExactOwnerByRange: () => ReadonlyMap, +): GraphNode | undefined { + const ownerId = resolveCallerGraphId(fact.ownerScopeId, indexes, nodeLookup); + if (ownerId !== undefined) { + const owner = graph.getNode(ownerId); + if (owner?.label === 'Method' || owner?.label === 'Function') return owner; + } + if (fact.ownerFilePath !== undefined && fact.ownerRange !== undefined) { + const fallback = getExactOwnerByRange().get( + `${fact.ownerFilePath}\0${fact.ownerRange.startLine - 1}\0${fact.ownerRange.endLine - 1}`, + ); + if (fallback !== null && fallback !== undefined) return fallback; + } + return undefined; +} + +function handlerReason(kinds: ReadonlySet): string { + if (kinds.size !== 1) { + return kinds.has('xxl-job') ? 'managed-non-http-handler' : 'spring-non-http-handler'; + } + const kind = kinds.values().next().value; + if (kind === 'xxl-job') return 'xxl-job-handler'; + return `spring-${kind}-handler`; +} + +/** + * Resolve callable annotations after imports and package visibility finalize, + * then promote confirmed framework-managed handlers into process entry points. + */ +export function createSpringNonHttpHandlerMetadataAttacher< + Annotation extends SpringNonHttpHandlerAnnotationFact, +>(adapter: SpringNonHttpHandlerAdapter) { + return ( + graph: KnowledgeGraph, + parsedFiles: readonly ParsedFile[], + nodeLookup: GraphNodeLookup, + indexes: ScopeResolutionIndexes, + ): void => { + const factsByFile = new Map[]>(); + for (const parsed of parsedFiles) { + const facts = adapter.getFacts(parsed.filePath); + if (facts.length > 0) factsByFile.set(parsed.filePath, facts); + } + if (factsByFile.size === 0) return; + + const resolveAnnotation = createSpringAnnotationNameResolver(indexes); + let exactOwnerByRange: ReadonlyMap | undefined; + const getExactOwnerByRange = (): ReadonlyMap => + (exactOwnerByRange ??= exactCallableOwnersByRange(graph)); + let classIdByMethod: ReadonlyMap | undefined; + const ownerClassLabel = (methodId: string): GraphNode['label'] | undefined => { + if (classIdByMethod === undefined) { + const owners = new Map(); + for (const relationship of graph.iterRelationshipsByType('HAS_METHOD')) { + owners.set(relationship.targetId, relationship.sourceId); + } + classIdByMethod = owners; + } + const classId = classIdByMethod.get(methodId); + return classId === undefined ? undefined : graph.getNode(classId)?.label; + }; + + for (const parsed of parsedFiles) { + const facts = factsByFile.get(parsed.filePath); + if (facts === undefined) continue; + const incomplete = adapter.isPackageVisibilityIncomplete(parsed.filePath); + const resolvedAnnotations = new Map(); + for (const fact of facts) { + const ownerScope = indexes.scopeTree.getScope(fact.ownerScopeId); + const resolvedFactAnnotations = new Set(); + for (const annotation of fact.annotations) { + if (annotation.useSiteTarget !== undefined) continue; + const enclosingScope = ownerScope?.parent ?? null; + const cacheKey = `${enclosingScope ?? ''}\0${annotation.name}`; + let resolved = resolvedAnnotations.get(cacheKey); + if (!resolvedAnnotations.has(cacheKey)) { + resolved = resolveAnnotation( + annotation.name, + parsed, + enclosingScope, + RESOLVABLE_NON_HTTP_ANNOTATIONS, + incomplete, + ); + resolvedAnnotations.set(cacheKey, resolved); + } + if (resolved !== undefined) resolvedFactAnnotations.add(resolved); + } + + const beanFactoryMethod = resolvedFactAnnotations.has(SPRING_BEAN_ANNOTATION); + const kinds = new Set(); + for (const resolved of resolvedFactAnnotations) { + if (beanFactoryMethod && resolved === SPRING_SERVICE_ACTIVATOR_ANNOTATION) continue; + const kind = HANDLER_ANNOTATIONS.get(resolved); + if (kind !== undefined) kinds.add(kind); + } + if (kinds.size === 0) continue; + + const owner = ownerGraphNode(fact, indexes, nodeLookup, graph, getExactOwnerByRange); + if (owner === undefined || ownerClassLabel(owner.id) === 'Interface') continue; + + const currentMultiplier = owner.properties.astFrameworkMultiplier ?? 1.0; + owner.properties.astFrameworkMultiplier = Math.max( + currentMultiplier, + SPRING_NON_HTTP_HANDLER_ENTRY_POINT_MULTIPLIER, + ); + if ( + currentMultiplier < SPRING_NON_HTTP_HANDLER_ENTRY_POINT_MULTIPLIER || + (currentMultiplier === SPRING_NON_HTTP_HANDLER_ENTRY_POINT_MULTIPLIER && + owner.properties.astFrameworkReason === undefined) + ) { + owner.properties.astFrameworkReason = handlerReason(kinds); + } + } + } + }; +} diff --git a/gitnexus/src/core/ingestion/import-resolvers/configs/csharp.ts b/gitnexus/src/core/ingestion/import-resolvers/configs/csharp.ts index d99dbd35b..24e9e6cfa 100644 --- a/gitnexus/src/core/ingestion/import-resolvers/configs/csharp.ts +++ b/gitnexus/src/core/ingestion/import-resolvers/configs/csharp.ts @@ -31,8 +31,10 @@ export const csharpNamespaceStrategy: ImportResolverStrategy = (rawImportPath, _ const resolvedFiles = resolveCSharpImportInternal( rawImportPath, csharpConfigs, - ctx.normalizedFileList, - ctx.allFileList, + // The Set, not `ctx.normalizedFileList`/`ctx.allFileList`: the resolver + // derives both from it through the same per-pass memo the ctx's own arrays + // come from, so this is the identical pair by a shorter route. + ctx.allFilePaths, ctx.index, evidence, ); diff --git a/gitnexus/src/core/ingestion/import-resolvers/configs/swift.ts b/gitnexus/src/core/ingestion/import-resolvers/configs/swift.ts index 8235edf73..05e4b13be 100644 --- a/gitnexus/src/core/ingestion/import-resolvers/configs/swift.ts +++ b/gitnexus/src/core/ingestion/import-resolvers/configs/swift.ts @@ -39,6 +39,19 @@ interface SwiftTargetIndex { * stable reference and the index is built once — not once per import. A * fresh run produces a fresh array → a fresh index, so cross-run staleness * is impossible. + * + * DELIBERATELY NOT ON `import-resolvers/per-file-set.ts` (#2909 sweep): this is + * a TWO-input memo keyed on ONE of them. The index is a function of both + * `ctx` (`allFileList` + the index-aligned `normalizedFileList`) and `targets`, + * but the key is only `ctx.allFileList`, and `perFileSet`'s `build: (key) => T` + * hands the builder nothing but the key. It is sound here only because of an + * invariant OUTSIDE the memo — `targets` is `ctx.configs.swiftPackageConfig + * .targets`, so it shares `ctx`'s lifetime and cannot vary while + * `ctx.allFileList` is fixed — and `perFileSet` has no way to express "and this + * other input is pinned by the same lifetime". Re-keying on `ctx` to make + * `targets` derivable from the key would change what the cache is keyed on and + * force an unreachable null-config arm into the builder, so it is a behaviour + * change rather than a consolidation. Leave it hand-rolled. */ const SWIFT_TARGET_INDEX_CACHE = new WeakMap(); diff --git a/gitnexus/src/core/ingestion/import-resolvers/csharp.ts b/gitnexus/src/core/ingestion/import-resolvers/csharp.ts index 2ce21274f..e8d6fdf7a 100644 --- a/gitnexus/src/core/ingestion/import-resolvers/csharp.ts +++ b/gitnexus/src/core/ingestion/import-resolvers/csharp.ts @@ -5,11 +5,260 @@ * This file contains shared helpers for namespace-based resolution. */ +import { perFileSet } from './per-file-set.js'; +import { getWorkspaceFileIndex } from './workspace-file-index.js'; import type { SuffixIndex } from './utils.js'; import { suffixResolve } from './utils.js'; import type { CSharpProjectConfig, CSharpNamespaceEvidence } from '../language-config.js'; import { csharpSuffixFallbackAllowed } from '../csharp-namespace-gate.js'; +/** + * Directory index backing the namespace-directory fallback below (step 3). + * + * That fallback used to be a full `normalizedFileList` pass per import, per + * matching csproj config — Θ(files), measured at ~1.08 ms per import over + * 50 000 `.cs` files (#2902). #2878 removed the per-import array REBUILD but + * not the scan itself. + * + * The scan's predicate depends only on the file's DIRECTORY, so it can be + * answered from an index built once per file list. Writing `D` for the + * normalized directory of a `.cs` file and `dirPrefix` for the query: + * + * let H = D + '/', P = dirPrefix + '/' + * match ⟺ H.endsWith(P) + * + * Derivation: + * - the scan keeps a file only when nothing after the matched occurrence holds + * a slash, so the occurrence's trailing '/' must be the file's LAST slash — + * i.e. `H` ends with `P`; + * - it used `indexOf`, the FIRST occurrence, so `a/Models/b/Models/x.cs` did + * NOT answer `Models`. That half was removed in #2881: it was an artifact of + * how the pre-index scan was written, not a rule about C# namespaces, and it + * dropped every repository that nests a directory name inside itself. The + * same removal landed in `package-dir-index.ts` and in step 2 below, which + * have to move together — see the note at step 3. + * - the needle ends with '/', so every occurrence of it lies wholly inside + * `D + '/'` and never reaches into the file name — which is what lets the + * whole test be evaluated on `D` alone; + * - and then the '/' cancels. `(D + '/').endsWith(P + '/')` IS `D.endsWith(P)`: + * the appended character only ever matches itself, so it decides nothing and + * the comparison of everything before it is unchanged. The predicate the code + * actually runs is therefore + * + * match ⟺ D.endsWith(dirPrefix) + * + * with no concatenation on either side. Verified rather than argued: over + * every ordered pair of strings up to length 5 over `{a, b, '/'}` including + * the empty string — 132 496 pairs — the two forms disagreed 0 times. + * + * NOT the same query as `package-dir-index.ts`, and the difference is exactly + * one character on each side: that module tests `'/'+D+'/'` against + * `'/'+pkgPath+'/'`, whose leading slash anchors the match to a segment + * boundary. This scan has no leading slash, so `dirPrefix = 'Models'` also + * matches `src/SubModels/` and `dirPrefix = 'src/Models'` also matches + * `vendor/mysrc/Models/`. Those hits are reachable (step 2 below answers only + * the segment-aligned ones, and step 3 runs precisely when step 2 found + * nothing), so the looser predicate is preserved verbatim rather than + * "cleaned up" into a reuse of `filesDirectlyInPkgDir` — see + * `test/unit/import-resolvers/csharp-csproj-parity.test.ts`. + * + * That one character is also why the cancellation above empties this predicate + * out but not that one: the decoration is one term per side here (`D + '/'`) and + * two there (`'/' + D + '/'`), and only the TRAILING '/' cancels. Here nothing + * is left to concatenate; there the leading segment anchor has to stay. + * + * Candidates are narrowed by the directory's LAST segment, the same + * O(directories) bucket `package-dir-index.ts` uses instead of an + * O(files × depth) suffix map (#2649). + */ +interface CsharpNamespaceDirIndex { + /** Last path segment of a directory → every `.cs` directory ending in it. */ + readonly dirsByLastSegment: ReadonlyMap; + /** + * Directory → positions in `WorkspaceFileIndex.normalized` of the `.cs` files + * directly inside it, ascending. + * + * Positions rather than paths: the emitted value is the RAW path, and the two + * arrays are parallel by construction — `normalized` is `all.map(slash)` — so + * a position is the one key that reads correctly in either. Both arrays come + * from the same `getWorkspaceFileIndex(allFilePaths)` object as this index + * itself, so the pairing cannot drift; it used to be a precondition on the + * caller, who passed the two arrays independently. + */ + readonly positionsByDir: ReadonlyMap; + /** + * Directories with no slash of their own — the entire answer to an empty + * `dirPrefix`, which is the one query no last-segment bucket expresses. + */ + readonly singleSegmentDirs: readonly string[]; +} + +/** + * Memoized on the file SET's identity, the same key every other per-file-set + * index in this pipeline uses: the orchestrator builds one Set per pass and + * threads it through every import, so this build runs once. + * + * It used to key on the `normalizedFileList` ARRAY, which was a second key + * shape and — more to the point — one no guard could instrument. Copying an + * array mints a fresh `WeakMap` key while traversing the SET zero extra times, + * so a `[...normalized]` copy at the adapter boundary rebuilt this index once + * per `using` while every scan-counting guard stayed green and only the timing + * bench noticed (#2911 review). Taking the array from + * `getWorkspaceFileIndex(allFilePaths)` inside the builder retires that shape: + * the only way to defeat the memo now is to copy the Set, which is exactly what + * `CountingSet` counts. + * + * It also retires a precondition. The cached positions index `normalized` while + * the emitted value is read from `all`; both now come from the same + * `getWorkspaceFileIndex` object, so the caller can no longer pair a position + * list against a differently-ordered array. + */ +const getCsharpNamespaceDirIndex = perFileSet( + (allFilePaths: ReadonlySet): CsharpNamespaceDirIndex => { + const { normalized: normalizedFileList } = getWorkspaceFileIndex(allFilePaths); + const dirsByLastSegment = new Map(); + const positionsByDir = new Map(); + const singleSegmentDirs: string[] = []; + + for (let i = 0; i < normalizedFileList.length; i++) { + const normalized = normalizedFileList[i]; + if (!normalized.endsWith('.cs')) continue; + const lastSlash = normalized.lastIndexOf('/'); + // A file with no directory can never match: the needle always ends with + // '/', so `indexOf` on a slash-free path is always -1. + if (lastSlash < 0) continue; + + const dir = normalized.slice(0, lastSlash); + let positions = positionsByDir.get(dir); + if (positions === undefined) { + positions = []; + positionsByDir.set(dir, positions); + const lastSegment = dir.slice(dir.lastIndexOf('/') + 1); + if (lastSegment === dir) singleSegmentDirs.push(dir); + let dirs = dirsByLastSegment.get(lastSegment); + if (dirs === undefined) { + dirs = []; + dirsByLastSegment.set(lastSegment, dirs); + } + dirs.push(dir); + } + positions.push(i); + } + + return { dirsByLastSegment, positionsByDir, singleSegmentDirs }; + }, +); + +/** + * Every directory that could satisfy `dirPrefix`, as a superset — the exact + * test runs in `matchingDirPositions`. + * + * When `dirPrefix` contains a '/', its own slash forces a segment boundary in + * any matching directory: `H` ending with `…//` means `D` ends with + * `/`, so `D`'s last segment IS `lastSeg` and the exact bucket is + * complete. Without a '/', `D`'s last segment only has to END with `dirPrefix` + * (`SubModels` for `Models`), which no single bucket holds, so the last-segment + * KEYS are swept. That is the one term here that is not O(matches), and it is + * O(distinct last segments), not O(directories): C# repos reuse `Models`, + * `Services`, `Controllers` under every project, so the sweep collapses on the + * layouts that actually occur. Measured at 200 000 `.cs` files, 25 000 + * directories: 456 µs per import when every directory name is unique, 7.9 µs + * on a `SrcN/Models` layout. Closing the unique-name case needs a character- + * suffix map over the segments, which is the O(files × depth) memory shape + * `package-dir-index.ts` cites #2649 to avoid — a design change, not a tune. + * + * An empty `dirPrefix` would sweep every key and keep every directory, so it is + * answered from `singleSegmentDirs` instead: its needle is a bare '/', which + * only a slash-free directory can carry as its LAST slash. + */ +function* candidateDirs(index: CsharpNamespaceDirIndex, dirPrefix: string): Generator { + if (dirPrefix === '') { + yield* index.singleSegmentDirs; + return; + } + const lastSlash = dirPrefix.lastIndexOf('/'); + if (lastSlash >= 0) { + const bucket = index.dirsByLastSegment.get(dirPrefix.slice(lastSlash + 1)); + if (bucket !== undefined) yield* bucket; + return; + } + for (const [lastSegment, dirs] of index.dirsByLastSegment) { + if (!lastSegment.endsWith(dirPrefix)) continue; + yield* dirs; + } +} + +/** Positions of the `.cs` files in each directory matching `dirPrefix`. */ +function* matchingDirPositions( + index: CsharpNamespaceDirIndex, + dirPrefix: string, +): Generator { + for (const dir of candidateDirs(index, dirPrefix)) { + // `(dir + '/').endsWith(dirPrefix + '/')` IS `dir.endsWith(dirPrefix)` — the + // appended '/' only ever matches itself, so it decides nothing and BOTH + // concatenations go. Exhaustively verified, not assumed: 0 disagreements + // over every ordered pair of strings up to length 5 over `{a, b, '/'}` + // including '' (132 496 pairs). Measured 64.9 ns -> 18.4 ns per candidate + // (Node 22.18); the `dir + '/'` was paid once per candidate, on every sweep + // of the last-segment keys. + // + // Still deliberately UNANCHORED (no leading '/'), so `src/SubModels` keeps + // answering `Models` — see the derivation above. That is also exactly why + // the reduction empties this predicate out while `package-dir-index.ts` + // keeps its concatenations: one decorating term per side here, two there, + // and only the trailing one cancels. + // + // `endsWith` subsumes the length guard the `indexOf` form needed: a shorter + // `dir` is simply false, where `indexOf` returned -1 and + // `haystack.length - needle.length` could also be -1 and report a bogus + // match. + // + // Do NOT "finish the job" with the two-argument overload. `endsWith(search, + // endPosition)` measured 8.8-11.8 ns against 9.5-14.9 ns for the + // one-argument form across seven call-site shapes (Node 22.18) — a wash — + // and `dir.endsWith(dirPrefix, dir.length)` is character-for-character this + // same test anyway. There is nothing left here to win. + if (!dir.endsWith(dirPrefix)) continue; + const positions = index.positionsByDir.get(dir); + if (positions !== undefined) yield positions; + } +} + +/** + * Append every `.cs` file directly inside a directory matching `dirPrefix`, in + * `normalizedFileList` order — the order the single-pass scan emitted, which + * this function's callers return as the whole edge target list. + */ +function pushFilesDirectlyInNamespaceDir( + index: CsharpNamespaceDirIndex, + dirPrefix: string, + allFileList: readonly string[], + results: string[], +): void { + // One matching directory is the overwhelmingly common case, and its positions + // are already ascending, so the first bucket is held by reference. A second + // one promotes it to a real accumulator that is appended to from then on — + // never re-spread per directory, which would cost O(files × dirs²) copies in + // a monorepo carrying the same namespace directory under many projects. + let first: readonly number[] | null = null; + let merged: number[] | null = null; + for (const positions of matchingDirPositions(index, dirPrefix)) { + if (first === null) { + first = positions; + continue; + } + if (merged === null) merged = [...first]; + for (const position of positions) merged.push(position); + } + if (first === null) return; + if (merged === null) { + for (const position of first) results.push(allFileList[position]); + return; + } + merged.sort((a, b) => a - b); + for (const position of merged) results.push(allFileList[position]); +} + /** * Resolve a C# using-directive import path to matching .cs files (low-level helper). * Tries single-file match first, then directory match for namespace imports. @@ -17,15 +266,23 @@ import { csharpSuffixFallbackAllowed } from '../csharp-namespace-gate.js'; * The final unanchored suffix fallback is gated on `evidence` so BCL usings * (e.g. `System.Threading.Tasks`) can't match a coincidentally-named local * file (#1881). When `evidence` is omitted the fallback stays permissive. + * + * Takes the file SET, not the two materialized lists it used to take: both are + * derived here from the per-pass `getWorkspaceFileIndex` memo, which is where + * every caller already got them. That leaves one key shape for the indexes + * below and makes the `normalized`/`all` pairing structural rather than a + * contract the caller has to honour. `index` stays a parameter — the parity + * harness drives this resolver with and without one, and the no-index legs are + * a tested dimension, not a degenerate case. */ export function resolveCSharpImportInternal( importPath: string, csharpConfigs: CSharpProjectConfig[], - normalizedFileList: string[], - allFileList: string[], + allFilePaths: ReadonlySet, index?: SuffixIndex, evidence?: CSharpNamespaceEvidence, ): string[] { + const { normalized: normalizedFileList, all: allFileList } = getWorkspaceFileIndex(allFilePaths); const namespacePath = importPath.replace(/\./g, '/'); const results: string[] = []; @@ -62,34 +319,76 @@ export function resolveCSharpImportInternal( // 2. Try as directory: all .cs files directly inside (namespace import) if (index) { const dirFiles = index.getFilesInDir(dirPrefix, '.cs'); + // `getFilesInDir` already answers "directly inside a directory `D` where + // `D === dirPrefix || D.endsWith('/' + dirPrefix)`" — its keys ARE + // segment-aligned directory suffixes. So for a non-empty `dirPrefix` the + // direct-child re-check this loop used to run cannot reject anything, and + // measurement agrees: zero rejections over 12 008 (prefix, candidate) + // pairs. It rejected before #2881 only because it asked `indexOf` for the + // FIRST `//`, which is the rule that issue removed. + // + // That widening does not stay inside step 2's own bucket. This step + // returns as soon as it pushes anything, so a query it used to answer with + // nothing now also SUPPRESSES step 3, whose unanchored match set is a + // strict superset: over `SubModels/Models/F1.cs` + `SubModels/F3.cs`, + // `using App.Models` answered both through step 3 and now answers only the + // first through step 2. The new answer is the more precise one — a + // directory literally named `Models` beating a character-suffix hit on + // `SubModels` — and it is what this module's step-2-before-step-3 layering + // asks for, so it is kept rather than worked around. Pinned absolutely by + // the parity test, which is differentially blind to it (its frozen legacy + // copy moved in lockstep with this line). + // + // The empty prefix is the exception and keeps a real filter. `getDirMap` + // keys a file under every suffix of its DIRECTORY, so it emits the EMPTY + // one exactly when that directory's last component is empty: a leading '/' + // on a root-level file, or a doubled slash immediately before the file + // name. Probed against `getDirMap`'s own key emission: + // + // src/X.cs -> ['src:.cs'] no empty key + // /X.cs -> [':.cs'] empty key + // a//X.cs -> [':.cs', 'a/:.cs'] empty key + // /a/b/X.cs -> ['b:.cs', 'a/b:.cs', '/a/b:.cs'] no empty key + // + // So the `''` bucket is not "one directory deep" on its own — `a//X.cs` + // sits in it two components down — while step 3 answers that same query + // from `singleSegmentDirs`, which is. Filtering on `D` holding no slash is + // what rejects `a//X.cs` and keeps steps 2 and 3 in agreement. for (const f of dirFiles) { - const normalized = f.replace(/\\/g, '/'); - // Check it's a direct child by finding the dirPrefix and ensuring no deeper slashes - const prefixIdx = normalized.indexOf(dirPrefix + '/'); - if (prefixIdx < 0) continue; - const afterDir = normalized.substring(prefixIdx + dirPrefix.length + 1); - if (!afterDir.includes('/')) { - results.push(f); + if (dirPrefix === '') { + const normalized = f.replace(/\\/g, '/'); + const lastSlash = normalized.lastIndexOf('/'); + if (lastSlash < 0 || normalized.slice(0, lastSlash).includes('/')) continue; } + results.push(f); } if (results.length > 0) return results; } - // 3. Linear scan fallback for directory matching - if (results.length === 0) { - const dirTrail = dirPrefix + '/'; - for (let i = 0; i < normalizedFileList.length; i++) { - const normalized = normalizedFileList[i]; - if (!normalized.endsWith('.cs')) continue; - const prefixIdx = normalized.indexOf(dirTrail); - if (prefixIdx < 0) continue; - const afterDir = normalized.substring(prefixIdx + dirTrail.length); - if (!afterDir.includes('/')) { - results.push(allFileList[i]); - } - } - if (results.length > 0) return results; - } + // 3. Directory matching, UNANCHORED. + // + // Not redundant with step 2, and not skippable when `index` is present: + // `getFilesInDir` is keyed on SEGMENT suffixes of a directory, while this + // leg's predicate is an unanchored ends-with one, so it additionally + // answers `Models` with `src/SubModels/` and `src/Models` with + // `vendor/mysrc/Models/`. It is also the only leg that answers an empty + // `dirPrefix` — the `relative = ''` branch above (the import IS the root + // namespace) with no `projectDir` to stand in for it — because + // `buildSuffixIndex` emits an empty directory suffix only for a path that + // BEGINS with '/', so over repo-relative paths `getFilesInDir('', '.cs')` + // is always empty. See `CsharpNamespaceDirIndex` above for the index that + // replaced the per-import Θ(files) scan this used to be (#2902). + // + // `results` is provably empty here: step 2 returns as soon as it pushes + // anything, and so does this leg, so every iteration of the config loop + // starts empty. + pushFilesDirectlyInNamespaceDir( + getCsharpNamespaceDirIndex(allFilePaths), + dirPrefix, + allFileList, + results, + ); + if (results.length > 0) return results; } // Fallback: suffix matching without namespace stripping (single file). diff --git a/gitnexus/src/core/ingestion/import-resolvers/go.ts b/gitnexus/src/core/ingestion/import-resolvers/go.ts index c33c46422..eb87b0997 100644 --- a/gitnexus/src/core/ingestion/import-resolvers/go.ts +++ b/gitnexus/src/core/ingestion/import-resolvers/go.ts @@ -3,10 +3,23 @@ * * Strategy lives in configs/go.ts. * This file contains the shared helpers used by the strategy. + * + * **Reachability, as of #2929:** nothing in production calls either export + * today. The only path in is `configs/go.ts` → `createImportResolver` → + * the `importResolver` field on Go's `LanguageProvider`, and that field is + * read at exactly two lines — `import-target-adapter.ts:74-75` — whose two + * exports (`buildImportTargetWorkspace`, + * `resolveImportTargetAcrossLanguages`) have no importer anywhere but their + * own unit test. So this is a live-looking but currently unwired leg; the + * tests in `test/unit/import-resolvers/go-package-resolve.test.ts` are the + * only thing watching it. */ import type { GoModuleConfig } from '../language-config.js'; +/** `'/'`, for the parent-directory boundary check in `resolveGoPackage`. */ +const SLASH_CODE = 47; + /** * Extract the package directory suffix from a Go import path. * Returns the suffix string (e.g., "/internal/auth/") or null if invalid. @@ -25,32 +38,42 @@ export function resolveGoPackageDir(importPath: string, goModule: GoModuleConfig export function resolveGoPackage( importPath: string, goModule: GoModuleConfig, - normalizedFileList: string[], - allFileList: string[], + normalizedFileList: readonly string[], + allFileList: readonly string[], ): string[] { - if (!importPath.startsWith(goModule.modulePath)) return []; + // Identical to the six lines this used to re-derive; `resolveGoPackageDir` + // returns the '/'-wrapped form and the scan wants the bare path, so unwrap. + const pkgDir = resolveGoPackageDir(importPath, goModule); + if (pkgDir === null) return []; + const relativePkg = pkgDir.slice(1, -1); // "/internal/auth/" → "internal/auth" - // Strip module path to get relative package path - const relativePkg = importPath.slice(goModule.modulePath.length + 1); // e.g., "internal/auth" - if (!relativePkg) return []; - - const pkgSuffix = '/' + relativePkg + '/'; + const pkgLen = relativePkg.length; // >= 1: `resolveGoPackageDir` rejects empty const matches: string[] = []; for (let i = 0; i < normalizedFileList.length; i++) { - // Prepend '/' so paths like "internal/auth/service.go" match suffix "/internal/auth/" - const normalized = '/' + normalizedFileList[i]; - // File must be directly in the package directory (not a subdirectory) - if ( - normalized.includes(pkgSuffix) && - normalized.endsWith('.go') && - !normalized.endsWith('_test.go') - ) { - const afterPkg = normalized.substring(normalized.indexOf(pkgSuffix) + pkgSuffix.length); - if (!afterPkg.includes('/')) { - matches.push(allFileList[i]); - } - } + const normalized = normalizedFileList[i]; + if (!normalized.endsWith('.go') || normalized.endsWith('_test.go')) continue; + // The file's PARENT directory ends with the package path — the same + // predicate `package-dir-index.ts` states. This used to ask `indexOf` for + // the FIRST `//` and then check that nothing after it held a slash, + // which made `a/pkg/b/pkg/x.go` not a member of `pkg` (#2881). + // + // Expressed as "`relativePkg` sits immediately before the last slash, on a + // segment boundary". The boundary is either the start of the path (an + // import matching from index 0, `internal/auth/x.go`) or a `/` — which is + // what the old `'/' + path` cons bought, at the price of a per-file + // concatenation the first `endsWith` forced V8 to flatten (#2929). + // + // Rewriting this as `endsWith(relativePkg, lastSlash)` buys nothing: the + // two-argument overload measured a wash against `startsWith(needle, pos)` + // here (10.28 ns vs 9.82 ns), so it trades the clarity of an explicit start + // index for no gain. A "the 2-arg overload leaves V8's fast path, 20x" + // claim from review did not reproduce on Node 22.18 — its baseline was a + // one-argument call that early-exited on the length precheck. + const start = normalized.lastIndexOf('/') - pkgLen; // < 0 when there is no parent dir + if (start < 0 || !normalized.startsWith(relativePkg, start)) continue; + if (start > 0 && normalized.charCodeAt(start - 1) !== SLASH_CODE) continue; + matches.push(allFileList[i]); } return matches; diff --git a/gitnexus/src/core/ingestion/import-resolvers/jvm.ts b/gitnexus/src/core/ingestion/import-resolvers/jvm.ts index 194cfdac8..b6723b8e2 100644 --- a/gitnexus/src/core/ingestion/import-resolvers/jvm.ts +++ b/gitnexus/src/core/ingestion/import-resolvers/jvm.ts @@ -31,8 +31,8 @@ export const appendKotlinWildcard = (importPath: string, importNode: SyntaxNode) */ export function resolveJvmWildcard( importPath: string, - normalizedFileList: string[], - allFileList: string[], + normalizedFileList: readonly string[], + allFileList: readonly string[], extensions: readonly string[], index?: SuffixIndex, ): string[] { @@ -90,8 +90,8 @@ export function resolveJvmWildcard( */ export function resolveJvmMemberImport( importPath: string, - normalizedFileList: string[], - allFileList: string[], + normalizedFileList: readonly string[], + allFileList: readonly string[], extensions: readonly string[], index?: SuffixIndex, ): string | null { diff --git a/gitnexus/src/core/ingestion/import-resolvers/node-workspace-packages.ts b/gitnexus/src/core/ingestion/import-resolvers/node-workspace-packages.ts new file mode 100644 index 000000000..712359a41 --- /dev/null +++ b/gitnexus/src/core/ingestion/import-resolvers/node-workspace-packages.ts @@ -0,0 +1,527 @@ +/** + * In-repo `package.json` manifests, as module-resolution input (#2953). + * + * A bare specifier (`@acme/telemetry/nest`, `@repo/utils`, `lodash/fp`) names a + * PACKAGE, not a path, and the manifest is the only thing that says which + * packages exist and where their entry points are. Without it a resolver can do + * nothing but guess — which is what the old suffix matcher did, landing + * `@acme/telemetry/nest` on the repo's only path ending in `nest/index.ts` + * while `@repo/utils`, a real first-party package, resolved to nothing because + * its name appears in no file path at all. + * + * Both directions come from the same missing input, so both are fixed by + * reading it: every in-repo `package.json` contributes its `name`, its `exports` + * map (including subpath patterns), its legacy entry fields, and its `imports` + * map for `#`-prefixed specifiers. + */ + +import fs from 'fs/promises'; +import path from 'path'; +import { createRequire } from 'node:module'; + +import { isHardcodedIgnoredDirectory } from '../../../config/ignore-service.js'; +import { logger } from '../../logger.js'; +import { resolveFile } from '../languages/typescript/file-candidates.js'; + +// `js-yaml` is CJS; the rest of this repository reaches it the same way +// (`core/group/config-parser.ts`, `cli/group.ts`). +const _require = createRequire(import.meta.url); +const yaml = _require('js-yaml') as typeof import('js-yaml'); + +/** One in-repo package. */ +export interface NodeWorkspacePackage { + /** Repo-relative directory holding the `package.json` (`''` for the root). */ + readonly dir: string; + /** + * Repo-relative entry stems for the package root (`import '@repo/utils'`), + * best first: declared `exports["."]`, then `module` / `main` / `types`, then + * the conventional `src/index` and `index`. + * + * A published `dist/...` entry simply fails to match an indexed source file + * (build output is not indexed) and the next candidate is tried, which is why + * the conventional fallbacks stay at the end rather than being a guess: they + * are what the package resolves to when it is consumed from source, which in + * a workspace it always is. + */ + readonly entries: readonly string[]; + /** + * Declared `exports` subpaths, specifier suffix -> repo-relative stems. + * Keys are as written minus the leading `./`, so `"./nest"` is stored `nest`; + * a pattern key keeps its `*` (`"./features/*"` -> `features/*`). + */ + readonly subpathExports: ReadonlyMap; + /** Declared `imports` map, `#name` -> repo-relative stems. */ + readonly subpathImports: ReadonlyMap; +} + +export interface NodeWorkspacePackages { + /** Package name (`@repo/utils`, `utils`) -> that package. */ + readonly byName: ReadonlyMap; +} + +const SCAN_MAX_DIRS = 20_000; +const SCAN_MAX_DEPTH = 24; + +/** + * The package name a bare specifier addresses, or `null` when the specifier + * names a path rather than a package. + * + * `@acme/telemetry/nest` -> `@acme/telemetry`, `lodash/fp` -> `lodash`. + */ +export function nodePackageNameOf(specifier: string): string | null { + if (specifier === '' || specifier.startsWith('.') || specifier.startsWith('/')) return null; + if (specifier.startsWith('#')) return null; + if (specifier.startsWith('@')) { + const parts = specifier.split('/'); + return parts.length >= 2 && parts[0].length > 1 && parts[1] !== '' + ? `${parts[0]}/${parts[1]}` + : null; + } + return specifier.split('/')[0] || null; +} + +/** The in-repo package whose directory most closely contains `filePath`. */ +export function owningPackage( + filePath: string, + packages: NodeWorkspacePackages | null | undefined, +): NodeWorkspacePackage | null { + if (!packages) return null; + let best: NodeWorkspacePackage | null = null; + for (const pkg of packages.byName.values()) { + const inside = pkg.dir === '' || filePath.startsWith(`${pkg.dir}/`); + if (inside && (best === null || pkg.dir.length > best.dir.length)) best = pkg; + } + return best; +} + +/** + * Resolve a bare specifier that names an in-repo package. + * + * `null` means the specifier names no in-repo package — an external dependency, + * whose correct in-repo resolution is nothing — or names one that does not + * export the requested subpath. + */ +export function resolveNodeWorkspaceImport( + specifier: string, + packages: NodeWorkspacePackages | null | undefined, + allFiles: ReadonlySet, +): string | null { + if (!packages) return null; + const packageName = nodePackageNameOf(specifier); + if (packageName === null) return null; + const pkg = packages.byName.get(packageName); + if (pkg === undefined) return null; + + const subpath = specifier.slice(packageName.length).replace(/^\//, ''); + for (const stem of entryStemsFor(pkg, subpath)) { + const hit = resolveFile(stem, allFiles); + if (hit !== null) return hit; + } + return null; +} + +/** + * Look a specifier up in a subpath map — `exports` or `imports`, which share + * Node's matching rule exactly: an exact key wins, otherwise the pattern with + * the longest literal prefix does, and its `*` takes whatever the specifier put + * there. + * + * Shared because they diverged once: the `imports` side did an exact lookup + * only, so a declared `"#internal/*"` could never match `#internal/foo`. + */ +export function matchSubpathMap( + map: ReadonlyMap, + specifier: string, +): readonly string[] | null { + const exact = map.get(specifier); + if (exact !== undefined) return exact; + + const patterns = [...map.entries()] + .filter(([key]) => key.includes('*')) + .map(([key, stems]) => { + const star = key.indexOf('*'); + return { prefix: key.slice(0, star), suffix: key.slice(star + 1), stems }; + }) + .filter( + ({ prefix, suffix }) => + specifier.startsWith(prefix) && + specifier.endsWith(suffix) && + specifier.length >= prefix.length + suffix.length, + ) + .sort((a, b) => b.prefix.length - a.prefix.length); + + for (const { prefix, suffix, stems } of patterns) { + const stem = specifier.slice(prefix.length, specifier.length - suffix.length); + return stems.map((target) => substituteStar(target, stem)); + } + return null; +} + +/** + * Substitute a subpath pattern's single `*`. + * + * Node's subpath patterns and TypeScript's `paths` both allow AT MOST one `*`, + * so replacing the first occurrence is the specified behaviour rather than a + * partial one — but `String.replace` with a string needle says that only by + * accident, and reads as a bug to anyone (CodeQL included) who has met the + * replace-all footgun. Slicing at the known index states the rule instead. + */ +export function substituteStar(target: string, stem: string): string { + const star = target.indexOf('*'); + return star === -1 ? target : target.slice(0, star) + stem + target.slice(star + 1); +} + +/** Candidate stems for one specifier into `pkg`, best first. */ +function entryStemsFor(pkg: NodeWorkspacePackage, subpath: string): readonly string[] { + if (subpath === '') return pkg.entries; + + const declared = matchSubpathMap(pkg.subpathExports, subpath); + if (declared !== null) return declared; + + // A package with NO `exports` map is not restricted: Node resolves any + // subpath against the package DIRECTORY, and only against it. A package WITH + // one exposes only what it lists, so an unlisted subpath resolves to nothing. + // + // Both restrictions are real, and neither is softened here. An earlier draft + // also tried `/src/`, on the theory that a workspace package is + // consumed from source — but nothing declares that mapping, so it is the same + // kind of guess this module exists to remove: it would resolve + // `@repo/utils/deep/thing` to `packages/utils/src/deep/thing.ts` for a + // package whose manifest never said `deep/thing` lives under `src/`, and the + // import would be broken in the real project too. + if (pkg.subpathExports.size > 0) return []; + return [joinRepoPath(pkg.dir, subpath)]; +} + +/** + * The directories the workspace ADMITS as packages. + * + * `null` means the repository declares no workspace at all, in which case the + * only package is the one at the root — a nested `package.json` somewhere in + * `examples/` or `test/fixtures/` is not a member of anything and its name is + * not addressable by an import. + * + * This gate is the difference between reading manifests and trusting them. + * Without it, finding a `package.json` anywhere in the tree was enough to + * register its name, which recreates the false-positive half of #2953 from a + * different source: an app importing registry package `foo` would bind to an + * excluded fixture that happens to declare `name: "foo"`. THIS repository is + * the example — `test/fixtures/**` alone declares `@repo/utils` (added by this + * very change) among others. + */ +interface WorkspaceScope { + /** Positive patterns, repo-relative, as declared. */ + readonly include: readonly string[]; + /** `!`-prefixed patterns, with the `!` stripped. */ + readonly exclude: readonly string[]; +} + +/** Whether `dir` (repo-relative, `''` for the root) is an admitted package. */ +function admits(scope: WorkspaceScope | null, dir: string): boolean { + // The root package is always itself, workspace or not. + if (dir === '') return true; + if (scope === null) return false; + if (scope.exclude.some((pattern) => globToRegExp(pattern).test(dir))) return false; + return scope.include.some((pattern) => globToRegExp(pattern).test(dir)); +} + +/** + * Match one workspace glob. + * + * The subset npm, pnpm, yarn and lerna actually use in `workspaces` / + * `packages`: `*` within a segment, `**` across segments, `?`, and a leading + * `!` for exclusion (handled by the caller). Deliberately not a general glob + * engine — the patterns are a documented, narrow dialect, and `minimatch` is + * only present here transitively through `glob`. + */ +function globToRegExp(pattern: string): RegExp { + const normalized = pattern.replace(/^\.\//, '').replace(/\/$/, ''); + let out = ''; + for (let i = 0; i < normalized.length; i++) { + const ch = normalized[i]; + if (ch === '*') { + if (normalized[i + 1] === '*') { + // `**/` may match nothing at all, so `packages/**/x` also matches + // `packages/x`; a trailing `**` matches any depth below. + if (normalized[i + 2] === '/') { + out += '(?:.*/)?'; + i += 2; + } else { + out += '.*'; + i += 1; + } + } else { + out += '[^/]*'; + } + continue; + } + if (ch === '?') { + out += '[^/]'; + continue; + } + out += ch.replace(/[.+^${}()|[\]\\]/g, '\\$&'); + } + return new RegExp(`^${out}$`); +} + +/** + * Read the repository's workspace declaration. + * + * All three spellings are read and merged, because a repo may carry more than + * one (a pnpm workspace whose root `package.json` also lists `workspaces` for + * tooling that does not read pnpm's file). + */ +async function loadWorkspaceScope(repoRoot: string): Promise { + const patterns: string[] = []; + + const rootManifest = await readJsonFile(path.join(repoRoot, 'package.json')); + const workspaces = rootManifest?.workspaces; + if (Array.isArray(workspaces)) { + patterns.push(...workspaces.filter((w): w is string => typeof w === 'string')); + } else if (workspaces !== null && typeof workspaces === 'object') { + // Yarn's object form: `{ "packages": [...], "nohoist": [...] }`. + const nested = (workspaces as { packages?: unknown }).packages; + if (Array.isArray(nested)) { + patterns.push(...nested.filter((w): w is string => typeof w === 'string')); + } + } + + patterns.push(...(await readYamlPackages(path.join(repoRoot, 'pnpm-workspace.yaml')))); + patterns.push(...(await readYamlPackages(path.join(repoRoot, 'pnpm-workspace.yml')))); + + const lerna = await readJsonFile(path.join(repoRoot, 'lerna.json')); + if (Array.isArray(lerna?.packages)) { + patterns.push(...lerna.packages.filter((w): w is string => typeof w === 'string')); + } + + if (patterns.length === 0) return null; + return { + include: patterns.filter((p) => !p.startsWith('!')), + exclude: patterns.filter((p) => p.startsWith('!')).map((p) => p.slice(1)), + }; +} + +async function readJsonFile(filePath: string): Promise | null> { + try { + return JSON.parse(await fs.readFile(filePath, 'utf-8')) as Record; + } catch { + return null; + } +} + +async function readYamlPackages(filePath: string): Promise { + let raw: string; + try { + raw = await fs.readFile(filePath, 'utf-8'); + } catch { + return []; + } + try { + const parsed = yaml.load(raw) as { packages?: unknown } | null; + const packages = parsed?.packages; + return Array.isArray(packages) + ? packages.filter((p): p is string => typeof p === 'string') + : []; + } catch { + return []; + } +} + +/** + * Collect the `package.json` of every ADMITTED workspace package. + * + * Directory-only BFS: the sole files opened are manifests and the workspace + * declaration, so this is far cheaper than the C# namespace scan next door, + * which reads every `.cs` file. + */ +export async function loadNodeWorkspacePackages( + repoRoot: string, +): Promise { + const scope = await loadWorkspaceScope(repoRoot); + const byName = new Map(); + const queue: { dir: string; depth: number }[] = [{ dir: repoRoot, depth: 0 }]; + let dirsScanned = 0; + + while (queue.length > 0) { + if (dirsScanned >= SCAN_MAX_DIRS) { + logger.warn( + `[node] package.json scan of ${repoRoot} hit the ${SCAN_MAX_DIRS}-directory cap; workspace packages below it will not resolve`, + ); + break; + } + const { dir, depth } = queue.shift()!; + dirsScanned++; + + let entries: import('fs').Dirent[]; + try { + entries = await fs.readdir(dir, { withFileTypes: true }); + } catch { + continue; + } + + for (const entry of entries) { + if (entry.isDirectory()) { + if (isHardcodedIgnoredDirectory(entry.name)) continue; + if (depth < SCAN_MAX_DEPTH) { + queue.push({ dir: path.join(dir, entry.name), depth: depth + 1 }); + } + continue; + } + if (!entry.isFile() || entry.name !== 'package.json') continue; + + const relDir = repoRelativeDir(repoRoot, dir); + // Found is not the same as admitted. A manifest outside the declared + // workspace belongs to something this repository does not build — a + // fixture, an example, a vendored copy — and its name is not addressable. + if (!admits(scope, relDir)) continue; + + const pkg = await readManifest(path.join(dir, entry.name), repoRoot, dir); + // First declaration wins: BFS visits shallower directories first, so a + // top-level package outranks a nested one that reuses the name. + if (pkg !== null && !byName.has(pkg.name)) byName.set(pkg.name, pkg.package); + } + } + + return byName.size === 0 ? null : { byName }; +} + +async function readManifest( + manifestPath: string, + repoRoot: string, + dir: string, +): Promise<{ name: string; package: NodeWorkspacePackage } | null> { + let parsed: Record; + try { + parsed = JSON.parse(await fs.readFile(manifestPath, 'utf-8')) as Record; + } catch { + return null; + } + const name = typeof parsed.name === 'string' ? parsed.name : ''; + if (name === '') return null; + + const packageDir = repoRelativeDir(repoRoot, dir); + const rebase = (raw: string): string => joinRepoPath(packageDir, stripEntryPrefixes(raw)); + + const subpathExports = new Map(); + const rootExports: string[] = []; + collectExports(parsed.exports, subpathExports, rootExports, rebase); + + // `exports`, when present, is the package's ENTIRE public interface: Node + // ignores `main` outright and refuses any subpath the map does not list. This + // resolver already honoured that restriction for subpaths (`entryStemsFor`) + // and not for the ROOT, which is the same rule — so a manifest exporting only + // `"./feature"` still answered a bare `@repo/pkg` with `src/index`, an edge + // for an import that does not resolve in the real project. + const declaresExports = parsed.exports !== undefined && parsed.exports !== null; + const entries: string[] = [...rootExports]; + if (!declaresExports) { + for (const field of ['module', 'main', 'types', 'typings']) { + const value = parsed[field]; + if (typeof value === 'string') push(entries, rebase(value)); + } + for (const conventional of ['src/index', 'index', 'lib/index']) { + push(entries, joinRepoPath(packageDir, conventional)); + } + } + + const subpathImports = new Map(); + collectImports(parsed.imports, subpathImports, rebase); + + return { name, package: { dir: packageDir, entries, subpathExports, subpathImports } }; +} + +/** + * Walk an `exports` value into the root-entry list and the subpath map. + * + * `exports` nests three ways at once — a bare string, a subpath map, and + * condition maps (`import` / `require` / `types` / `default`) at any depth — so + * this collects string leaves per subpath rather than assuming a shape. + */ +function collectExports( + node: unknown, + subpaths: Map, + rootStems: string[], + rebase: (raw: string) => string, + currentSubpath: string | null = '', +): void { + if (typeof node === 'string') { + if (currentSubpath === null) return; + if (currentSubpath === '') { + push(rootStems, rebase(node)); + return; + } + subpaths.set(currentSubpath, [...(subpaths.get(currentSubpath) ?? []), rebase(node)]); + return; + } + // An array is an ordered FALLBACK LIST, not an opaque value: Node tries each + // entry in turn. `{"./feature": ["./dist/feature.js", "./src/feature.ts"]}` is + // the shape a workspace package publishes to say "built output, or source" — + // and the source arm is the one that matters here, because `dist/` is build + // output and is not indexed. Skipping arrays dropped the declaration entirely + // and left the package looking as though it declared no subpath exports. + if (Array.isArray(node)) { + for (const element of node) + collectExports(element, subpaths, rootStems, rebase, currentSubpath); + return; + } + if (node === null || typeof node !== 'object') return; + + for (const [key, value] of Object.entries(node as Record)) { + if (key.startsWith('.')) { + // A subpath key: `"."` is the package root, `"./nest"` the subpath `nest`. + collectExports( + value, + subpaths, + rootStems, + rebase, + key === '.' ? '' : key.replace(/^\.\//, ''), + ); + } else { + // A condition key — stays on whatever subpath we were already resolving. + collectExports(value, subpaths, rootStems, rebase, currentSubpath); + } + } +} + +/** Walk an `imports` map (`"#env": "./src/env.node.ts"`) into stems. */ +function collectImports( + node: unknown, + out: Map, + rebase: (raw: string) => string, + currentKey: string | null = null, +): void { + if (typeof node === 'string') { + if (currentKey === null) return; + out.set(currentKey, [...(out.get(currentKey) ?? []), rebase(node)]); + return; + } + // Same ordered-fallback rule as `exports` — see `collectExports`. + if (Array.isArray(node)) { + for (const element of node) collectImports(element, out, rebase, currentKey); + return; + } + if (node === null || typeof node !== 'object') return; + for (const [key, value] of Object.entries(node as Record)) { + collectImports(value, out, rebase, key.startsWith('#') ? key : currentKey); + } +} + +/** `"./src/index.ts"` -> `"src/index"`; leaves an extension-less path alone. */ +function stripEntryPrefixes(entry: string): string { + const withoutDot = entry.replace(/^\.\//, '').replace(/^\//, ''); + return withoutDot.replace(/\.(ts|tsx|mts|cts|js|jsx|mjs|cjs|vue)$/, ''); +} + +function push(list: string[], value: string): void { + if (value !== '' && !list.includes(value)) list.push(value); +} + +/** `/repo/packages/utils` -> `packages/utils`; the root -> `''`. */ +function repoRelativeDir(repoRoot: string, dir: string): string { + const rel = path.relative(repoRoot, dir).split(path.sep).join('/'); + return rel === '.' ? '' : rel; +} + +function joinRepoPath(dir: string, rest: string): string { + return dir === '' ? rest : `${dir}/${rest}`; +} diff --git a/gitnexus/src/core/ingestion/import-resolvers/package-dir-index.ts b/gitnexus/src/core/ingestion/import-resolvers/package-dir-index.ts new file mode 100644 index 000000000..cb7e77ed0 --- /dev/null +++ b/gitnexus/src/core/ingestion/import-resolvers/package-dir-index.ts @@ -0,0 +1,255 @@ +/** + * "Which files live DIRECTLY inside a directory whose path ends with + * ``?" — the query Go's package resolution and C#'s namespace-directory + * fallback both answered with a full `allFilePaths` scan per import. + * + * Both scans ran the same predicate: normalize to forward slashes, apply the + * language's extension filter, find the FIRST `'/' + pkgPath + '/'` occurrence, + * and keep the file only if nothing after that occurrence contains a slash. + * + * That predicate depends only on the file's DIRECTORY, so it can be answered + * from an index built once per file set: + * + * let D = '/' + + '/' + * let P = '/' + pkgPath + '/' + * match ⟺ D.endsWith(P) + * + * It used to say one more thing, and #2881 removed it: + * + * match ⟺ D.length >= P.length && D.indexOf(P) === D.length - P.length + * + * — i.e. `D` ends with `P` AND that trailing occurrence is the FIRST one, so + * `a/pkg/b/pkg/x.go` did NOT answer `pkg`. The second half was never a rule + * anyone chose. It is what the pre-index per-import scan happened to compute + * (it called `indexOf`, then checked that nothing after the match contained a + * slash), and the index was built to reproduce that scan byte for byte. It + * dropped exactly the repositories that nest a directory name inside itself: + * `internal/…/internal`, `Models/…/Models`, and the reported shape + * `data/src/main/kotlin/com/example/data/Repo.kt`, where `import data.helper` + * resolved to null. Kotlin was fixed first, in its own `dirChildren` + * (`languages/kotlin/import-target.ts`); this index, the C# csproj index and + * the legacy `go.ts` scan followed. + * + * The strongest evidence that the rule was accidental is that a sixth + * implementation of the same question never had it. `import-resolvers/jvm.ts` + * answers "files directly inside a directory ending with " for + * Java and Kotlin wildcard imports, and has used `lastIndexOf` since #488. + * + * That is evidence about how the predicate was WRITTEN, not about live + * behaviour, and the distinction matters enough to spell out. `jvm.ts` is + * reached only through `provider.importResolver`, which `languages/java.ts` and + * `languages/kotlin.ts` do wire — but that field currently has no production + * READER. Its only reader anywhere is `import-target-adapter.ts`, whose own + * docblock says it is "threaded through `finalizeScopeModel`"; nothing threads + * it, and neither that module nor its two exports + * (`buildImportTargetWorkspace`, `resolveImportTargetAcrossLanguages`) is + * referenced outside its own unit test. So `jvm.ts`'s `resolveJvmWildcard` and + * `import-resolvers/go.ts`'s `resolveGoPackage` are dormant, while THIS index, + * `csharp.ts`'s `resolveCSharpImportInternal` and Kotlin's `dirChildren` are + * the ones that run. Whether those two dormant resolvers should be deleted or + * actually wired up is an open question and wants its own issue; it is not + * settled here. + * + * The argument survives that correction intact, because it never needed the + * resolvers to be live: an independent implementation of the same question, + * written without reference to the pre-index scan, reached for `lastIndexOf`. + * The extra clause was never a rule anyone chose. All six spellings now agree. + * + * The length guard the `indexOf` form needed is gone with it: `endsWith` is + * false for a shorter `D` instead of comparing -1 to -1. + * + * Candidates are narrowed by the directory's LAST segment rather than by + * indexing every directory suffix: a suffix map costs O(files × depth) entries, + * which is exactly the memory this codebase runs out of at kernel scale + * (#2649), while the last-segment bucket is O(directories) and is a superset of + * the matches (`D` ends with `P` ⟹ the dir's last segment is `pkgPath`'s last + * segment). + * + * Results keep Set-iteration order via the recorded `ord`, because the callers' + * scans emitted in that order and Go returns the whole list as the import + * target (one `ImportEdge` per file). + * + * Each language owns its own `WeakMap` memo and `accept` predicate, so the + * STORED index holds only that language's files — the build pass itself still + * walks every path it is handed once per language. That is not a polyglot tax + * in practice: `scope-resolution/pipeline/run.ts:673` rebuilds `allFilePaths` + * from the provider's own `parsedFiles`, so the set already contains only that + * language's files. + */ + +interface IndexedFile { + readonly raw: string; + /** + * Position in `allFilePaths` iteration order. Still load-bearing: + * `filesDirectlyInPkgDir` sorts on it to interleave several directories back + * into the order the original single-pass scan emitted. + */ + readonly ord: number; +} + +/** + * Deeply read-only on purpose. The memo hoist turned what used to be per-call + * scratch into state shared by every import in a run, and `readonly` on the + * PROPERTY still lets a caller do `idx.rootFiles.sort()` in place. Typing the + * containers as read-only makes the copy-before-mutating rule compile-enforced + * instead of comment-enforced — but `readonly` is erased at runtime and is not + * hard to widen back (the sibling Kotlin index documents `Array.isArray`'s + * `arg is any[]` predicate doing exactly that), so the one container callers + * read directly is handed out through `sortedRootFiles` rather than raw. + */ +export interface PackageDirIndex { + /** Last path segment of a directory → every normalized directory ending in it. */ + readonly dirsByLastSegment: ReadonlyMap; + /** Normalized directory → the accepted files directly inside it, in Set order. */ + readonly filesByDir: ReadonlyMap; + /** Accepted files with no directory at all, in Set order. */ + readonly rootFiles: readonly string[]; +} + +/** + * @param accept Runs on the normalized (forward-slash) path; return `false` to + * leave the file out of the index entirely. + */ +export function buildPackageDirIndex( + allFilePaths: ReadonlySet, + accept: (normalized: string) => boolean, +): PackageDirIndex { + const dirsByLastSegment = new Map(); + const filesByDir = new Map(); + const rootFiles: string[] = []; + + let ord = 0; + for (const raw of allFilePaths) { + const ownOrd = ord++; + const normalized = raw.replace(/\\/g, '/'); + if (!accept(normalized)) continue; + + const lastSlash = normalized.lastIndexOf('/'); + if (lastSlash < 0) { + // No directory: `'/x.go'.indexOf('/pkg/')` can never hit, so a root file + // answers no `pkgPath` query. Kept separately for Go's root-package leg. + rootFiles.push(raw); + continue; + } + + const dir = normalized.slice(0, lastSlash); + let files = filesByDir.get(dir); + if (files === undefined) { + files = []; + filesByDir.set(dir, files); + const lastSegment = dir.slice(dir.lastIndexOf('/') + 1); + let dirs = dirsByLastSegment.get(lastSegment); + if (dirs === undefined) { + dirs = []; + dirsByLastSegment.set(lastSegment, dirs); + } + dirs.push(dir); + } + files.push({ raw, ord: ownOrd }); + } + + return { dirsByLastSegment, filesByDir, rootFiles }; +} + +/** Every indexed directory matching `pkgPath`, in first-seen order. */ +function* matchingDirs(index: PackageDirIndex, pkgPath: string): Generator { + const lastSegment = pkgPath.slice(pkgPath.lastIndexOf('/') + 1); + const dirs = index.dirsByLastSegment.get(lastSegment); + if (dirs === undefined) return; + // `('/' + D + '/').endsWith('/' + P + '/')` ⟺ `D === P || D.endsWith('/' + P)`, + // which is the same predicate without the two strings per candidate the + // wrapped form built: 32.58 ns → 8.22 ns per candidate, 3.96x, over 2001 + // directories of which 668 match (Node 22.18.0, best-of-80 after 300 warmup + // passes). Verified exhaustively rather than argued, over every pair of + // strings up to length 5 over `{a, b, /}` including the empty string — + // 132 496 pairs, 911 of them matching: 0 divergences. The match count is + // reported beside the timing on purpose: two predicates that agree on `false` + // everywhere also show 0 divergences. + // + // Worth the care because this loop is genuinely hot: Go's GOPATH fallback + // calls `matchingDirs` once per import-path segment over the bucket holding + // EVERY directory that shares the queried last segment (every service's + // `internal`). At 1000 services × 100 000 unresolved imports × 4 segments + // that is ~34.6 s against ~14.0 s, and ~0.5 GB of transient garbage not + // allocated. + // + // There is nothing further to win by reaching for the two-argument + // `endsWith(search, endPosition)` or for `startsWith(needle, pos)`: on the + // same data all three land together — 8.15 ns one-argument, 8.55 ns + // two-argument, 8.54 ns `startsWith` — and against the unwrapped string the + // two-argument form is character-for-character the same test. A "the 2-arg + // overload leaves V8's fast path, 20x" claim was measured during review and + // did NOT reproduce here or in two independent re-runs; its 1.87 ns baseline + // was a one-argument call whose needle failed the length/last-char precheck + // and early-exited without comparing. Recorded because the retraction is the + // useful part: compare forms that do the same work and report the hit count. + // + // The equality arm also carries the length guard the `indexOf` form needed: a + // shorter `dir` is simply false, where `indexOf` returned -1 and + // `haystack.length - needle.length` could also be -1 and report a bogus match. + const suffix = `/${pkgPath}`; + for (const dir of dirs) { + if (dir !== pkgPath && !dir.endsWith(suffix)) continue; + const files = index.filesByDir.get(dir); + if (files !== undefined) yield files; + } +} + +/** + * Every accepted file directly inside a directory ending with `pkgPath`, in + * `allFilePaths` iteration order. + */ +export function filesDirectlyInPkgDir(index: PackageDirIndex, pkgPath: string): string[] { + // The first bucket is held by reference, not copied into an accumulator: one + // matching directory is the overwhelmingly common case (every unique-leaf + // call, and any query whose package path has more than one segment), and it + // then reaches the `map` with zero intermediate copies. + // + // A second directory promotes that reference to a real accumulator, which is + // appended to once per file from then on — never re-spread per directory, + // because that costs O(files × dirs²) copies, which a monorepo carrying the + // same package directory under many services (`svcN/internal/models`, queried + // by Go's two-segment GOPATH tail) would pay on every import. + let first: readonly IndexedFile[] | null = null; + let merged: IndexedFile[] | null = null; + for (const files of matchingDirs(index, pkgPath)) { + if (first === null) { + first = files; + continue; + } + if (merged === null) merged = [...first]; + for (const f of files) merged.push(f); + } + if (first === null) return []; + // One directory is already in Set order; several interleave and need merging + // back onto the order the original single-pass scan emitted. + if (merged === null) return first.map((f) => f.raw); + merged.sort((a, b) => a.ord - b.ord); + return merged.map((f) => f.raw); +} + +/** Root-package files in sorted order. Copies: the index array is shared by + * every import in the run and the result leaves as an edge target list. */ +export function sortedRootFiles(index: PackageDirIndex): string[] { + return [...index.rootFiles].sort(); +} + +/** + * The FIRST accepted file (in `allFilePaths` iteration order) directly inside a + * directory ending with `pkgPath`, or `null`. + */ +export function firstFileDirectlyInPkgDir(index: PackageDirIndex, pkgPath: string): string | null { + // Returning the FIRST match is already the minimum-`ord` answer, and it is + // the build loop that makes it so: `buildPackageDirIndex` appends a directory + // to its last-segment bucket at the moment it accepts that directory's first + // file, so bucket order IS ascending first-file-`ord` order. Comparing `ord` + // across the remaining directories can never improve on the first hit + // (differentially verified: 0 divergences). Change that append point — buffer + // the directories, sort them, populate `filesByDir` before `dirsByLastSegment` + // — and this early return silently starts answering with the wrong file. + for (const files of matchingDirs(index, pkgPath)) { + const first = files[0]; + if (first !== undefined) return first.raw; + } + return null; +} diff --git a/gitnexus/src/core/ingestion/import-resolvers/pass-cache.ts b/gitnexus/src/core/ingestion/import-resolvers/pass-cache.ts new file mode 100644 index 000000000..4c307cfd0 --- /dev/null +++ b/gitnexus/src/core/ingestion/import-resolvers/pass-cache.ts @@ -0,0 +1,79 @@ +import { buildSuffixIndex, type SuffixIndex } from './utils.js'; + +/** + * Everything the standard `resolveTsTarget` path derives from one workspace + * file set: the file list, the lower-cased file list, the suffix index and the + * per-pass `resolveCache`. + * + * Without this memoization the resolver re-derived `allFileList` and + * `normalizedFileList` (both O(N_files)), rebuilt the index and threw away the + * `resolveCache` on every import — O(N_files × N_imports) total work for what + * should be O(N_files + N_imports). + */ +export interface ImportPassCache { + readonly allFilePaths: Set; + readonly allFileList: readonly string[]; + readonly normalizedFileList: readonly string[]; + readonly index: SuffixIndex; + readonly resolveCache: Map; +} + +/** + * Build that state. Shared by every adapter whose resolution runs through + * `resolveTsTarget`. + * + * Not a dedup of identical copies, and the difference is the point. At + * 49c5b7d81 each of those adapters carried this record inline and they did NOT + * agree: `languages/typescript/scope-resolver.ts` and + * `languages/vue/import-target.ts` held six byte-identical fields built around + * `index: buildSuffixIndex(normalizedFileList, allFileList)`, while + * `languages/javascript/import-target.ts` held five and never called + * `buildSuffixIndex` at all. That one missing field IS the O(imports × files) + * defect PR #2911 fixed — `resolveTsTarget` fell back to `suffixResolve`'s + * linear scan for every JavaScript import — and the header of + * `languages/javascript/import-target.ts` carries the measurements. Hoisting + * the builder is what makes a fourth adapter unable to omit it again: `index` + * is not optional on `ImportPassCache`. + * + * The BUILDER is shared; the MEMO deliberately is not. Each adapter wraps this + * in its own `perFileSet(...)`, so each gets its own `WeakMap`, its own index + * instance and — the one that would be a behaviour change — its own + * `resolveCache`. The languages disagree about what a specifier resolves to + * (`tsconfigPaths` is read from config for TypeScript and Vue, pinned to `null` + * for JavaScript, and the tried extension list differs), so one shared resolve + * cache across them would hand a language another language's answers. + * + * Sharing the builder is a code dedup and nothing more: it buys no runtime + * reuse, because there is none to buy. Each provider pass builds its own + * `allFilePaths` Set (`scope-resolution/pipeline/run.ts`, per provider), so + * TypeScript's set and JavaScript's set are different objects and therefore + * different `WeakMap` keys even where the two memos are the same code. + */ +export function buildImportPassCache(allFilePaths: ReadonlySet): ImportPassCache { + const allFileList = Array.from(allFilePaths); + // LOWERCASED, not slash-normalized — unlike every other caller of + // `buildSuffixIndex`. That is what `alreadyLowercased` below records. + const normalizedFileList = allFileList.map((f) => f.toLowerCase()); + return { + // Copied ONCE per file set, not once per import: `TsResolveContext` wants a + // mutable `Set` and the orchestrator hands us a `ReadonlySet`. The copy is + // not the #1918 hazard because the cache KEY is the caller's original Set. + allFilePaths: new Set(allFilePaths), + allFileList, + normalizedFileList, + // Every suffix of an all-lowercase path is itself lowercase, so the index's + // case-folded map came out a byte-for-byte copy of its exact map — same + // keys, same values, same insertion order — one per `ImportPassCache`, so + // once per adapter per pass. Measured 14.00 MiB at 32 000 paths, 29.8% of + // the retained `ImportPassCache`. The flag drops the copy; it does not change + // what `getInsensitive` answers, because the copy was the identity (see + // `SuffixIndexOptions`). Checked, not assumed: over 474 524 probes on four + // mixed-case corpora — Vue PascalCase plus alias specifiers, case-colliding + // twins, a 600-file deep monorepo, and Unicode paths carrying final sigma, + // dotted-I and sharp-S — the two maps came out byte-identical, the exact + // map was the sole answerer 0 times, and `get(s) || getInsensitive(s)` + // returned the same file 474 524 times out of 474 524. + index: buildSuffixIndex(normalizedFileList, allFileList, { alreadyLowercased: true }), + resolveCache: new Map(), + }; +} diff --git a/gitnexus/src/core/ingestion/import-resolvers/per-file-set.ts b/gitnexus/src/core/ingestion/import-resolvers/per-file-set.ts new file mode 100644 index 000000000..4de0ee361 --- /dev/null +++ b/gitnexus/src/core/ingestion/import-resolvers/per-file-set.ts @@ -0,0 +1,83 @@ +/** + * The one memo every per-file-set index in this pipeline is built on. + * + * The scope-resolution orchestrator builds ONE file-set object per provider + * pass and threads that same object through every `resolveImportTarget` call in + * the pass, so anything derived from it — a suffix index, a package-directory + * map, a basename bucket — can be built once and read by every import instead + * of rebuilt per import. Keying on the object's IDENTITY is what makes that + * work, and it is equally the contract callers must keep: the set is passed + * THROUGH, never copied. A defensive `new Set(allFilePaths)` at an adapter + * boundary hands a fresh key per import and silently restores + * O(imports × files) — the bug PR #1918 shipped and had to fix in review (P1). + * The guards are `test/integration/-import-index-reuse.test.ts` and, for + * every registered language at once, + * `test/unit/scope-resolution/import-target-index-reuse.contract.test.ts`, + * whose inventory arm fails when an entry of `SCOPE_RESOLVERS` has no fixture. + * That arm is why no language is named here: the registry is the census, and a + * hand-copied list of languages goes stale the release after it is written. + * + * A `WeakMap` rather than a `Map`: the entry is reclaimed with the file set it + * was derived from, so a pass can never read a previous pass's index and memory + * does not grow across runs. There is no invalidation rule to get wrong because + * there is nothing to invalidate — a new file set is a new key. + * + * The KEY TYPE is constrained rather than described, because which object is + * the key decides whether the guards above can see the memo fail, and a prose + * list of call sites is the thing this file elsewhere tells you not to write. + * `K` admits exactly the two shapes the orchestrator keeps stable for a pass: + * + * - `ReadonlySet`, the pass's file set — every index derived from it, + * including the derived header-closure sets that `languages/{c,cpp}/ + * scope-resolver.ts` memoize inside an outer per-file-set memo. Defeating + * one of these means copying the SET, which re-traverses it, which the + * `CountingSet` instrument (`test/helpers/counting-file-set.ts`) reads as a + * scan count rising with the import count. + * - `readonly ParsedFile[]`, the pass's parsed-file array. Not derived from + * the file set at all, so the file-set guards do not reach them; these key + * on the array the orchestrator already threads through the pass, and their + * contract is that same pass-through discipline. The instrument that CAN see + * them counts element reads on that array — `countedParsedFiles`, beside + * `CountingSet`, driven by the contract test's `minimumParsedFileReads`. + * + * A THIRD shape — an array materialized from the file set — is what the type + * exists to reject. `import-resolvers/csharp.ts` used one until #2911, and it + * is worth a compile error rather than a rule: copying an array mints a fresh + * `WeakMap` key while traversing the Set zero extra times, so every + * scan-counting guard stays green at its correct value while the index rebuilds + * once per import. That failure is invisible to the whole instrument family + * above and was caught only by a timing ratio in `bench/import-target/`. Derive + * the array inside the builder from `getWorkspaceFileIndex(allFilePaths)` + * instead. `string[]` is not assignable to `K`, so the shape cannot come back + * silently — `configs/swift.ts` keeps the one hand-rolled `WeakMap` on + * `ctx.allFileList` in the tree, deliberately and with its reasons written + * down, and it is deliberately NOT on this primitive. + * + * `T extends object` is deliberate, chosen over probing `has` before `get`. + * `WeakMap.get` returning `undefined` cannot distinguish "not built yet" from + * "built, and the value is `undefined`"; constraining the value to an object + * makes the second case unrepresentable rather than paying a second lookup on + * every import, and it needs no cast to type-check. Every index memoized here + * is a record, `Map` or `Set`, so the constraint costs nothing today — and a + * later caller wanting to memoize a `string | null` gets a compile error + * pointing at this line instead of a memo that silently rebuilds on every miss. + * + * A `build` that THROWS stores nothing, so the next call for that key runs it + * again: failures are not memoized, and a half-filled index is never published. + * Inert for the builders here — each is a pure, total pass over the file set — + * and the safer of the two behaviours if that ever stops being true. + */ +import type { ParsedFile } from 'gitnexus-shared'; + +export function perFileSet | readonly ParsedFile[], T extends object>( + build: (key: K) => T, +): (key: K) => T { + const cache = new WeakMap(); + return (key) => { + const cached = cache.get(key); + if (cached !== undefined) return cached; + const built = build(key); + cache.set(key, built); + return built; + }; +} diff --git a/gitnexus/src/core/ingestion/import-resolvers/php.ts b/gitnexus/src/core/ingestion/import-resolvers/php.ts index 303bf5546..6652ecbfa 100644 --- a/gitnexus/src/core/ingestion/import-resolvers/php.ts +++ b/gitnexus/src/core/ingestion/import-resolvers/php.ts @@ -37,8 +37,8 @@ export function resolvePhpImportInternal( importPath: string, composerConfig: ComposerConfig | null, allFiles: Set, - normalizedFileList: string[], - allFileList: string[], + normalizedFileList: readonly string[], + allFileList: readonly string[], index?: SuffixIndex, ): string | null { // Normalize: replace backslashes with forward slashes @@ -67,21 +67,38 @@ export function resolvePhpImportInternal( const lastSlash = remainder.lastIndexOf('/'); const nsDir = lastSlash >= 0 ? dirPrefix + '/' + remainder.slice(0, lastSlash) : dirPrefix; - // Prefer SuffixIndex directory lookup (O(log n + matches)) over linear scan + // Prefer SuffixIndex directory lookup (O(log n + matches)) over linear scan. + // + // An EMPTY bucket is a final answer, not a miss to retry with the scan + // below — which is what the `else` restores, and what this comment + // always claimed. Re-scanning on empty was the last per-import + // workspace traversal left in PHP resolution after #2901: any `use` + // matching a PSR-4 prefix whose directory holds no direct `.php` child + // (`App\Legacy\Ghost`) paid a full pass, measured at 201 traversals for + // 200 imports. + // + // The bucket is a superset of what the scan can find, for BOTH index + // shapes that reach here. A root-anchored direct child `nsDir/.php` + // has its directory exactly equal to `nsDir`, and `nsDir` is always one + // of that directory's own suffixes — so the shared `dirMap` (keyed on + // every directory suffix) necessarily contains it, as does the + // root-anchored parity index `languages/php/import-target.ts` builds. + // Empty superset therefore implies empty scan, and control falls + // through to the next PSR-4 prefix exactly as before. if (index) { const candidates = index.getFilesInDir(nsDir, '.php'); if (candidates.length > 0) return candidates[0]; - } - - // Fallback: linear scan (only when SuffixIndex unavailable) - const nsDirPrefix = nsDir.endsWith('/') ? nsDir : nsDir + '/'; - for (const f of allFiles) { - if ( - f.startsWith(nsDirPrefix) && - f.endsWith('.php') && - !f.slice(nsDirPrefix.length).includes('/') - ) { - return f; + } else { + // Linear scan, only when a SuffixIndex is genuinely unavailable. + const nsDirPrefix = nsDir.endsWith('/') ? nsDir : nsDir + '/'; + for (const f of allFiles) { + if ( + f.startsWith(nsDirPrefix) && + f.endsWith('.php') && + !f.slice(nsDirPrefix.length).includes('/') + ) { + return f; + } } } } diff --git a/gitnexus/src/core/ingestion/import-resolvers/python-file-index.ts b/gitnexus/src/core/ingestion/import-resolvers/python-file-index.ts new file mode 100644 index 000000000..d624bb45a --- /dev/null +++ b/gitnexus/src/core/ingestion/import-resolvers/python-file-index.ts @@ -0,0 +1,371 @@ +/** + * The one per-file-set index behind Python import resolution, plus the two + * importer-chain memos that ride inside it. + * + * ## Why this is its own module + * + * Everything here is derived from `allFilePaths` and nothing here is specific + * to either CALLER, and there are two of them on opposite sides of a layer + * boundary: `import-resolvers/python.ts` resolves the single-segment bare tier + * and `languages/python/import-target.ts` resolves the dotted tiers. The second + * imports the first, so the index could not live in either without the other + * reaching back through a cycle — it used to live in `import-target.ts`, which + * is why the bare tier had no O(1) proof of absence and probed the whole + * ancestor chain for every `import os`. + * + * The shape is the one `workspace-file-index.ts` and `package-dir-index.ts` + * already use in this directory: an interface, one `perFileSet` builder, and + * query functions taking the index. + */ + +import { perFileSet } from './per-file-set.js'; + +/** + * The importer's ancestor directories, CLOSEST FIRST and excluding the + * workspace root — `["backend/routers", "backend"]` for `backend/routers/x.py` + * — memoized per importer DIRECTORY for the lifetime of the pass. + * + * This is the #2913 fix. Both consumers used to rebuild the chain inline, one + * `dirParts.slice(0, i).join('/')` per component, on EVERY import: a per-import + * cost proportional to the importer's path depth, and quadratic in characters, + * on a file index that is itself depth-free. Real Python layouts are deep + * (`src/pkg/sub/feature/impl/mod.py` is ordinary), so the resolver was 6.8x + * slower on a deep corpus than on a shallow one holding the file count fixed, + * where every other language sat between 1.0x and 3.4x. + * + * A directory's ancestors are a pure function of the directory, and a pass + * resolves many imports per file, so one entry serves every import issued from + * anywhere in that directory. + * + * ## Lifetime and memory + * + * The Map lives INSIDE the per-file-set index, so it is reclaimed with the file + * set it was reached through (`perFileSet` is a `WeakMap`): it cannot leak + * across passes or repos, and there is no invalidation rule to get wrong. It is + * filled lazily, so it holds one entry per directory that actually ISSUES a + * Python import, never one per file and never one per directory in the repo — + * the bound #2649 (kernel-scale OOM) asks for. Each entry's strings are + * `slice`s of the longest one, so a chain costs pointers rather than a copy of + * the path per component. + * + * The derived key is the importer's directory exactly as the old inline code + * computed it — `norm.split('/').slice(0, -1).join('/')`, which for a path + * without a separator is `''` (a root-level importer, whose chain is empty). + */ +/** + * The importer's own directory, normalized — the key BOTH per-directory memos + * below are stored under. + * + * One exported derivation rather than one per accessor: the two memos live in + * the same index and must agree on what "the importer's directory" is, and a + * caller that already holds the directory (the bare-import tier computes it for + * its own proximity check) should not pay for it twice. It was three copies of + * `replace / lastIndexOf / slice` across two modules before, byte-identical by + * inspection and by nothing else. + */ +export function importerDirOf(fromFile: string): string { + const norm = fromFile.replace(/\\/g, '/'); + const lastSlash = norm.lastIndexOf('/'); + return lastSlash === -1 ? '' : norm.slice(0, lastSlash); +} + +export function importerAncestors(index: PythonFileIndex, importerDir: string): readonly string[] { + const memoized = index.ancestorsByDir.get(importerDir); + if (memoized !== undefined) return memoized; + const built = buildImporterAncestors(importerDir); + index.ancestorsByDir.set(importerDir, built); + return built; +} + +/** + * `["a/b/c", "a/b", "a"]` for `a/b/c`. Empty components are dropped first, so + * an absolute `/a/b` yields `["a/b", "a"]` — matching the `filter(Boolean)` the + * two inline walks did, and with it the absolute-path gating pinned by + * `python-import-target-parity.test.ts` (PR #1918 review P3a). + */ +function buildImporterAncestors(importerDir: string): readonly string[] { + const chain: string[] = []; + const parts = importerDir.split('/').filter(Boolean); + if (parts.length === 0) return chain; + chain.push(parts.join('/')); + for (let i = 1; i < parts.length; i++) { + const child = chain[i - 1]; + chain.push(child.slice(0, child.lastIndexOf('/'))); + } + return chain; +} + +/** + * Per-file-set index for Python import resolution, memoized on the + * `allFilePaths` Set object (the same Set is passed for every import in a run, + * so the index is built once and reused). Replaces the per-import O(files) + * scans in `resolveAbsoluteFromFiles` (suffix match) and `hasRepoCandidate` + * (package-existence gate) with O(1)/O(bucket) lookups. + * + * - `normSet`: every file path, normalized to forward slashes (for the exact + * `f === rootFile|initFile` membership checks). It IS derivable from the two + * buckets below — both probes could be a `.some(c => c.norm === …)` over + * `byBasename.get(rootFile)` / `byInitParent.get(initFile)` — and it is kept + * anyway, deliberately. `byBasename` is keyed on the BASENAME, so its bucket + * for a common Python file name is not small and grows with the repo: on a + * 9 000-file service tree, `utils.py`, `models.py` and `views.py` hold 1 000 + * entries each. `import utils` would then scan every `utils.py` in the + * workspace on every import — a per-import cost proportional to corpus size, + * which is the exact defect class #2901/#2902/#2908 removed. The Set trades + * ~1.6 MB at 32 000 files, against a 6.4 MB reading, to keep both probes + * O(1). Do not "simplify" it away without re-measuring that bucket. + * - `byBasename`: last path component (e.g. `models.py`, `__init__.py`) -> + * all `{ raw, norm }` candidates, so suffix matches can be gathered from the + * relevant bucket and the exact tie-break applied across ALL of them. + * - `byInitParent`: `__init__.py` files keyed by their last TWO components + * (`/__init__.py`). The package suffix lookup (`pkg.sub` -> + * `…/sub/__init__.py`) targets only same-named package dirs via this map + * instead of scanning every `__init__.py` in the repo — the common + * multi-segment import path no longer scales with package count + * (PR #1918 review P2b). `__init__.py` files stay in `byBasename` too, for + * the rarer explicit `pkg.__init__` import that resolves via the module + * (`….py`) lookup. + * - `dirPrefixes`: every directory prefix of a `.py` file, trailing-slashed + * (`a/b/c.py` -> `a/`, `a/b/`), for "is there a .py file under `/`". + * - `nestedDirNames`: the NAME of every such directory that has a non-empty + * parent (`a/b/c.py` -> `b`, not `a`), which is exactly the set of segments + * `hasRepoCandidate`'s ancestor walk can ever match — so a segment absent + * from it settles the walk in one lookup (#2913). + * - `ancestorsByDir`: the per-importer-directory ancestor-chain memo behind + * `importerAncestors`. The one structure here that is NOT derived from the + * file set: it is filled lazily, from the importer paths the pass actually + * resolves against, and lives here so it dies with the pass. + * - `bareImportPrefixesByDir`: the same idea for the OTHER chain — the + * sys.path-style prefixes `resolvePythonImportInternal`'s single-segment + * walk probes. A different sequence, not a different spelling: see + * `importerBarePrefixes`. Two memos in one index rather than two indexes, + * because they are keyed on the same thing and must die together. + * + * Exported for `test/unit/scope-resolution/python/python-importer-ancestors.test.ts` + * and `test/unit/import-resolvers/python-importer-prefixes.test.ts`, which read + * the two memos after driving the production adapters. No counter ships for + * either — the Map IS the memo, and its SIZE is the assertion: one entry per + * importer directory, however many imports were resolved. Everything else about + * the index stays internal. + */ +export interface PythonFileIndex { + readonly normSet: Set; + readonly byBasename: Map; + readonly byInitParent: Map; + readonly dirPrefixes: Set; + readonly nestedDirNames: Set; + readonly ancestorsByDir: Map; + readonly bareImportPrefixesByDir: Map; +} + +export const getPythonFileIndex = perFileSet( + (allFilePaths: ReadonlySet): PythonFileIndex => { + // Runs on a cache miss only. That it happens once per run and not once per + // import is asserted by counting traversals of the Set itself, in + // `test/integration/python-import-index-reuse.test.ts` — the PR #1918 review + // P1 guard (#2909). + + const normSet = new Set(); + const byBasename = new Map(); + const byInitParent = new Map(); + const dirPrefixes = new Set(); + const nestedDirNames = new Set(); + + for (const raw of allFilePaths) { + const norm = raw.replace(/\\/g, '/'); + // Python import resolution only ever queries `.py` paths: module `.py` + // and package `/__init__.py` membership (normSet), `.py` / + // `__init__.py` basename buckets (byBasename), and `.py` directory prefixes + // (dirPrefixes). Non-`.py` files can never match any of those, so skip them + // — they were dead weight in every structure on polyglot monorepos + // (PR #1918 review P3b; dirPrefixes was already `.py`-gated). + if (!norm.endsWith('.py')) continue; + normSet.add(norm); + + // ONE entry object per file, shared by both buckets below: a package file + // lands in `byBasename` and `byInitParent`, and two literals for the same + // `(raw, norm)` pair cost ~40 B each on every `__init__.py`. + const entry = { raw, norm }; + + const lastSlash = norm.lastIndexOf('/'); + const base = lastSlash >= 0 ? norm.slice(lastSlash + 1) : norm; + // `set(base, [entry])` rather than `set(base, [])` then `push`: an empty + // array literal that is immediately pushed to makes V8 grow the backing + // store to its 16-slot minimum, so every bucket holding ONE file retains + // 15 empty pointer slots — 128 B — for the whole pass. `byBasename` has + // roughly one bucket per file, which made that the dominant term in this + // index: measured 5.50 MiB against 1.60 MiB for the one-element form at + // 32 000 `.py` paths, byte-identical contents. Same shape as + // `languages/php/import-target.ts`'s directory buckets. + const bucket = byBasename.get(base); + if (bucket === undefined) byBasename.set(base, [entry]); + else bucket.push(entry); + + // Package files also get a parent-keyed bucket so a `pkg.sub` lookup hits + // only `…/sub/__init__.py` candidates, not every `__init__.py` (P2b). + if (base === '__init__.py' && lastSlash >= 0) { + const dir = norm.slice(0, lastSlash); + const parentSlash = dir.lastIndexOf('/'); + const parentName = parentSlash >= 0 ? dir.slice(parentSlash + 1) : dir; + if (parentName) { + const initKey = `${parentName}/__init__.py`; + const ib = byInitParent.get(initKey); + if (ib === undefined) byInitParent.set(initKey, [entry]); + else ib.push(entry); + } + } + + // Directory prefixes: every slash-terminated prefix of the path (every + // index just past a '/', up to and including the file's own directory). + // Scanning the FULL normalized path — including any leading '/' for + // absolute paths — makes `dirPrefixes.has(X)` match exactly when the old + // gate's `f.startsWith(X)` (X always ends in '/') matched. The previous + // split+`filter(Boolean)` dropped the leading empty component, so an + // absolute file `/repo/svc/x.py` yielded `repo/svc/` (no leading slash) and + // gate-passed where `"/repo/svc/x.py".startsWith("repo/svc/")` is false + // (PR #1918 review P3a). For relative paths the set is identical. + // + // The walk runs from the DEEPEST prefix outward and stops at the first + // one already recorded. Every prefix is added together with all of its + // own ancestors, so a hit proves the rest of the chain is already there — + // which makes the second and later files of a directory cost ONE lookup + // instead of one insert per path component. This build was the last part + // of Python's resolution that still scaled with path depth (#2913): the + // same 400-file corpus moved sixteen directories down went from 800 + // inserts to 7200, for the same ~120 distinct prefixes. + // + // `nestedDirNames` rides the same walk. A directory prefix has the shape + // `//` — the only shape `hasRepoCandidate`'s check (3) + // probes — exactly when another slash precedes it at index > 0. Index 0 + // is excluded on purpose: `a/` and `/` name a directory whose parent is + // empty, which check (2) already answers and which the ancestor walk + // (non-empty ancestors only) never probes. + for (let i = lastSlash; i >= 0; i--) { + if (norm[i] !== '/') continue; + const dirPrefix = norm.slice(0, i + 1); + if (dirPrefixes.has(dirPrefix)) break; + dirPrefixes.add(dirPrefix); + const parentSlash = i > 0 ? norm.lastIndexOf('/', i - 1) : -1; + if (parentSlash > 0) nestedDirNames.add(norm.slice(parentSlash + 1, i)); + } + } + + return { + normSet, + byBasename, + byInitParent, + dirPrefixes, + nestedDirNames, + ancestorsByDir: new Map(), + bareImportPrefixesByDir: new Map(), + }; + }, +); + +/** + * The sys.path-style prefixes `resolvePythonImportInternal`'s single-segment + * bare-import walk probes, in order, for an importer sitting in `importerDir` — + * memoized per DIRECTORY for the lifetime of the pass, in the same index and + * for the same reasons as `importerAncestors`. + * + * ## Why this is not `ancestorsByDir` + * + * A DIFFERENT SEQUENCE, not a different spelling. For `backend/routers/cron.py`: + * + * importerAncestors ["backend/routers", "backend"] + * importerBarePrefixes ["backend/", ""] + * + * Three differences, each load-bearing: + * + * 1. `importerAncestors` opens with the importer's OWN directory; this walk + * does not, because its proximity check has already probed that directory. + * 2. This walk ENDS at the workspace root (`""`, which probes `.py` + * unprefixed); `importerAncestors` stops short of it, because + * `resolveAbsoluteFromFiles` probes the root before its walk instead. + * 3. `importerAncestors` drops empty components (`filter(Boolean)`); this walk + * keeps them, and the difference decides real resolutions — for + * `/abs/a/b/mod.py` this walk probes `/abs/a/`, `/abs/`, `""`, `""` where a + * filtered chain would probe `abs/a/b/`, `abs/a/`, `abs/`, none of which is + * a prefix of any file in an absolute-path workspace. + * + * So the two cannot share one chain without changing which files resolve. They + * do share the index, the key and the lifetime, which is what actually matters + * for #2649: both are filled lazily, hold one entry per directory that ISSUES + * an import, and die with the pass because the index does. + */ +export function importerBarePrefixes( + index: PythonFileIndex, + importerDir: string, +): readonly string[] { + const memoized = index.bareImportPrefixesByDir.get(importerDir); + if (memoized !== undefined) return memoized; + const built = buildImporterBarePrefixes(importerDir); + index.bareImportPrefixesByDir.set(importerDir, built); + return built; +} + +/** + * `["a/b/", "a/", ""]` for `a/b/c` — every proper ancestor of `importerDir`, + * closest first, slash-terminated, ending at the workspace root. + * + * Cutting the string at each `lastIndexOf('/')` walks the same ancestors the + * pre-#2913-followup `dirParts.slice(0, i).join('/')` produced, INCLUDING the + * empty components a `filter(Boolean)` would have dropped: `/abs/a/b` yields + * `["/abs/a/", "/abs/", "", ""]`, the second `""` being the `i === 0` step that + * followed the leading empty component. Byte-identical sequences, duplicates + * kept, so the probes this feeds are unchanged in content, order and count. + */ +function buildImporterBarePrefixes(importerDir: string): readonly string[] { + const prefixes: string[] = []; + let dir = importerDir; + let slash = dir.lastIndexOf('/'); + while (slash !== -1) { + dir = dir.slice(0, slash); + prefixes.push(dir === '' ? '' : `${dir}/`); + slash = dir.lastIndexOf('/'); + } + prefixes.push(''); + return prefixes; +} + +/** + * "No file anywhere in the workspace can be `/.py` or + * `//__init__.py`, for ANY prefix ``" — in two Map lookups. + * + * This is a PROOF OF ABSENCE, not a heuristic filter, and it is what lets the + * single-segment bare walk skip itself entirely. Both shapes it rules out are + * the only two shapes that walk probes: a probe `${prefix}${segment}.py` that + * is a member of the file set is a path with no backslash (the prefix comes + * from a normalized importer and the guard below rejects a segment carrying + * one), so it equals its own normalized form and its basename is exactly + * `${segment}.py` — which puts it in `byBasename`. A probe + * `${prefix}${segment}/__init__.py` that is a member likewise has parent + * directory name exactly `segment`, non-empty, which puts it in `byInitParent` + * whether or not `prefix` is empty. So a miss in both buckets means every probe + * the walk would issue is guaranteed to miss. + * + * Two inputs cannot be proven absent and get `false` — walk as before: + * + * - the EMPTY segment (a target spelled with a trailing dot). + * `byInitParent` skips `__init__.py` files whose parent directory name is + * empty, so its absence proves nothing. Same carve-out + * `resolveAbsoluteFromFiles` makes for `lastSeg === ''`. + * - a segment containing a BACKSLASH. The buckets are keyed on normalized + * paths, so a raw `a\b.py` is filed under basename `b.py`; a probe for the + * segment `a\b` would look up `a\b.py`, miss, and wrongly conclude absence + * while `allFilePaths.has('a\\b.py')` is true. Not reachable from a Python + * import statement, but this function is a proof and a proof has no + * unstated preconditions. + * + * The dotted tier in `languages/python/import-target.ts` asks the same question + * of the same two buckets and is deliberately NOT routed through here: it needs + * the candidate ARRAYS for its suffix fallback, so it does the two `get`s it + * already needs and derives the answer, rather than paying two extra `has` + * lookups per import to share four lines. + */ +export function pythonSegmentAbsent(index: PythonFileIndex, segment: string): boolean { + if (segment === '' || segment.includes('\\')) return false; + if (index.byBasename.has(`${segment}.py`)) return false; + if (index.byInitParent.has(`${segment}/__init__.py`)) return false; + return true; +} diff --git a/gitnexus/src/core/ingestion/import-resolvers/python.ts b/gitnexus/src/core/ingestion/import-resolvers/python.ts index 2de11cc55..9914a7613 100644 --- a/gitnexus/src/core/ingestion/import-resolvers/python.ts +++ b/gitnexus/src/core/ingestion/import-resolvers/python.ts @@ -6,6 +6,12 @@ * This file contains the shared internal helper used by the strategy and tests. */ +import { + getPythonFileIndex, + importerBarePrefixes, + importerDirOf, + pythonSegmentAbsent, +} from './python-file-index.js'; import { tryResolveWithExtensions } from './utils.js'; /** @@ -51,8 +57,24 @@ export function resolvePythonImportInternal( const pathLike = importPath.replace(/\./g, '/'); if (pathLike.includes('/')) return null; - // Normalize for Windows backslashes - const importerDir = currentFile.replace(/\\/g, '/').split('/').slice(0, -1).join('/'); + // O(1) proof of absence, before any probing. Every probe below — the two + // proximity probes and the two per ancestor step — has the shape + // `/.py` or `//__init__.py`, and + // `pythonSegmentAbsent` answers "no file in the workspace has EITHER shape, + // for any prefix" in two Map lookups on the index the dotted tiers already + // build. That is `true` for `os`, `sys`, `django` and every other + // distribution the repo does not vendor — i.e. for most imports in most + // Python repos — and it retires the whole walk for them instead of running + // it to the workspace root. It is exact, not a filter: a miss here means + // every probe the walk would have issued was guaranteed to miss. + const index = getPythonFileIndex(allFiles); + if (pythonSegmentAbsent(index, pathLike)) return null; + + // One derivation, shared with the index's other per-directory memo — see + // `importerDirOf`. It replaced `split('/').slice(0, -1).join('/')`: identical + // for every input (a path with no separator has no directory, which is `''` + // both ways) without the per-import array of one element per path component. + const importerDir = importerDirOf(currentFile); // Proximity check — only applies when the importer lives in a subdirectory. // Root-level importers (importerDir === '') skip straight to the ancestor @@ -68,10 +90,12 @@ export function resolvePythonImportInternal( // importer's directory to find the module in an ancestor, preferring the closest match. // This prevents cross-language misresolution (e.g., Python `from middleware import X` // resolving to a TypeScript middleware.ts via suffix matching). Issue #417. - const dirParts = importerDir.split('/'); - for (let i = dirParts.length - 1; i >= 0; i--) { - const ancestorDir = dirParts.slice(0, i).join('/'); - const prefix = ancestorDir ? `${ancestorDir}/` : ''; + // + // The prefixes come from `importerBarePrefixes`, built ONCE per importer + // directory per pass and stored in the same index consulted above. Rebuilding + // them here — `dirParts.slice(0, i).join('/')`, one array and one string per + // path component — was the last per-import ancestor walk left after #2913. + for (const prefix of importerBarePrefixes(index, importerDir)) { if (allFiles.has(`${prefix}${pathLike}/__init__.py`)) return `${prefix}${pathLike}/__init__.py`; if (allFiles.has(`${prefix}${pathLike}.py`)) return `${prefix}${pathLike}.py`; } diff --git a/gitnexus/src/core/ingestion/import-resolvers/ruby.ts b/gitnexus/src/core/ingestion/import-resolvers/ruby.ts index 4bf47d31f..b19a6b3c3 100644 --- a/gitnexus/src/core/ingestion/import-resolvers/ruby.ts +++ b/gitnexus/src/core/ingestion/import-resolvers/ruby.ts @@ -16,8 +16,8 @@ import { suffixResolve } from './utils.js'; */ export function resolveRubyImportInternal( importPath: string, - normalizedFileList: string[], - allFileList: string[], + normalizedFileList: readonly string[], + allFileList: readonly string[], index?: SuffixIndex, ): string | null { const pathParts = importPath.replace(/^\.\//, '').split('/').filter(Boolean); diff --git a/gitnexus/src/core/ingestion/import-resolvers/standard.ts b/gitnexus/src/core/ingestion/import-resolvers/standard.ts index 888e80208..4cc4c1c60 100644 --- a/gitnexus/src/core/ingestion/import-resolvers/standard.ts +++ b/gitnexus/src/core/ingestion/import-resolvers/standard.ts @@ -29,8 +29,8 @@ export const resolveImportPath = ( currentFile: string, importPath: string, allFiles: Set, - allFileList: string[], - normalizedFileList: string[], + allFileList: readonly string[], + normalizedFileList: readonly string[], resolveCache: Map, language: SupportedLanguages, tsconfigPaths: TsconfigPaths | null, diff --git a/gitnexus/src/core/ingestion/import-resolvers/types.ts b/gitnexus/src/core/ingestion/import-resolvers/types.ts index 864206fc3..081972b77 100644 --- a/gitnexus/src/core/ingestion/import-resolvers/types.ts +++ b/gitnexus/src/core/ingestion/import-resolvers/types.ts @@ -4,14 +4,7 @@ * Extracted from import-resolution.ts to co-locate types with their consumers. */ -import type { - TsconfigPaths, - GoModuleConfig, - CSharpProjectConfig, - CSharpNamespaceEvidence, - ComposerConfig, -} from '../language-config.js'; -import type { SwiftPackageConfig } from '../language-config.js'; +import type { ImportConfigs } from '../language-config.js'; import type { SuffixIndex } from './utils.js'; import type { SupportedLanguages } from 'gitnexus-shared'; @@ -26,17 +19,6 @@ export type ImportResult = | { kind: 'package'; files: string[]; dirSuffix: string } | null; -/** Bundled language-specific configs loaded once per ingestion run. */ -export interface ImportConfigs { - tsconfigPaths: TsconfigPaths | null; - goModule: GoModuleConfig | null; - composerConfig: ComposerConfig | null; - swiftPackageConfig: SwiftPackageConfig | null; - csharpConfigs: CSharpProjectConfig[]; - /** In-repo namespace evidence gating C# suffix-fallback resolution (#1881). */ - csharpNamespaces?: CSharpNamespaceEvidence; -} - /** Pre-built lookup structures for import resolution. Build once, reuse across chunks. */ export interface ImportResolutionContext { allFilePaths: Set; diff --git a/gitnexus/src/core/ingestion/import-resolvers/utils.ts b/gitnexus/src/core/ingestion/import-resolvers/utils.ts index 6a033c1ee..5dec720bc 100644 --- a/gitnexus/src/core/ingestion/import-resolvers/utils.ts +++ b/gitnexus/src/core/ingestion/import-resolvers/utils.ts @@ -79,66 +79,304 @@ export function tryResolveWithExtensions( * etc. */ export interface SuffixIndex { - /** Exact suffix lookup (case-sensitive) */ + /** + * Exact suffix lookup (case-sensitive). + * + * The map behind this is built on the FIRST call and memoized — see + * `buildSuffixIndex`. All three maps are deferred; a consumer pays only for + * the questions it actually asks. + */ get(suffix: string): string | undefined; - /** Case-insensitive suffix lookup */ + /** + * Case-insensitive suffix lookup. + * + * Deferred like `get`, and — when `get` was asked first — DERIVED from that + * map rather than traversed for a second time. See `buildSuffixIndex`. + */ getInsensitive(suffix: string): string | undefined; - /** Get all files in a directory suffix */ - getFilesInDir(dirSuffix: string, extension: string): string[]; + /** + * Get all files in a directory suffix. + * + * `dirSuffix` is matched as a SEGMENT-aligned directory suffix — every + * returned file is a direct child of a directory `D` with + * `D === dirSuffix || D.endsWith('/' + dirSuffix)`. Callers may rely on this + * and skip a direct-child re-check; `import-resolvers/csharp.ts` step 2 does + * exactly that. It bounds what may be RETURNED, not what must be found: an + * implementation is free to answer with fewer files, and the root-anchored + * index in `languages/php/import-target.ts` answers only the `D === dirSuffix` + * arm. + * + * `readonly` is the CONTRACT, and it is the contract for every implementation + * of this interface, not a description of any one of them: an implementation + * is free to return its own bucket by reference, so callers must treat the + * result as shared and never `sort`/`splice` it in place. The compiler now + * refuses that at the call site. Whether a given implementation shares or + * copies is its own business and documented where it is built — + * `buildSuffixIndex` shares, the root-anchored parity index in + * `languages/php/import-target.ts` returns a filtered copy. + * + * Implementations that memoize should note the directory map behind this may + * be built on the FIRST call rather than up front, so a caller that never + * asks a directory question never pays for it — see `buildSuffixIndex`. + */ + getFilesInDir(dirSuffix: string, extension: string): readonly string[]; } -export function buildSuffixIndex(normalizedFileList: string[], allFileList: string[]): SuffixIndex { - // Map: normalized suffix -> original file path - const exactMap = new Map(); - // Map: lowercase suffix -> original file path - const lowerMap = new Map(); - // Map: directory suffix -> list of file paths in that directory - const dirMap = new Map(); +export interface SuffixIndexOptions { + /** + * Promise from the caller that `normalizedFileList[i] === normalizedFileList[i].toLowerCase()` + * for every `i` — i.e. the "normalized" list is a LOWERCASED file list, not + * merely a slash-normalized one. + * + * `import-resolvers/pass-cache.ts` is the one caller that can make it: it + * builds `normalizedFileList` as `allFileList.map((f) => f.toLowerCase())`. + * Every suffix of an all-lowercase path is itself lowercase, so + * `suffix.toLowerCase() === suffix` and the case-folded map came out a + * byte-identical copy of the exact one — same keys, same values, same + * insertion order. Measured 14.00 MiB at 32 000 paths, 29.8% of the retained + * `ImportPassCache` — and one `ImportPassCache` is built per ts-family + * adapter per pass, so the waste was carried once for each of them. + * + * With this set, `getInsensitive` reads the exact map directly instead. It is + * the same map the derivation below would have produced, so this is a skipped + * copy and not a second lookup rule — see `getLowerMap`. + * + * Setting it over a list that is NOT all-lowercase is a behaviour change, not + * an optimization: `getInsensitive` would then answer case-sensitively. + */ + readonly alreadyLowercased?: boolean; +} - for (let i = 0; i < normalizedFileList.length; i++) { - const normalized = normalizedFileList[i]; - const original = allFileList[i]; - const parts = normalized.split('/'); +export function buildSuffixIndex( + normalizedFileList: readonly string[], + allFileList: readonly string[], + options?: SuffixIndexOptions, +): SuffixIndex { + const alreadyLowercased = options?.alreadyLowercased === true; - // Index all suffixes: "a/b/c.java" -> ["c.java", "b/c.java", "a/b/c.java"] - for (let j = parts.length - 1; j >= 0; j--) { - const suffix = parts.slice(j).join('/'); - // Only store first match (longest path wins for ambiguous suffixes) - if (!exactMap.has(suffix)) { - exactMap.set(suffix, original); + /** + * Map: normalized suffix -> original file path. + * + * DEFERRED, like `dirMap` below and for the same reason (#2903 extended to + * the two suffix maps). Several consumers on the ScopeResolver path ask only + * ONE of the two suffix questions and were paying for both: + * + * - `languages/java/import-target.ts` and the no-csproj leg of + * `languages/csharp/import-target.ts` call `get` and never + * `getInsensitive` — measured 49.98 MiB dead of a 100.82 MiB Java index + * at 32 000 paths (49.6%), against a gated ceiling of 146.9 MiB; + * - `languages/php/import-target.ts` calls `getInsensitive` and never `get` + * — 34.49 MiB of 69.85 MiB (49.4%). + * + * Ruby, the csproj leg of C#, `group/extractors/include-extractor.ts` and + * `suffixResolve` below read both, and all four read `get` FIRST (they are + * written `get(s) || getInsensitive(s)`), which is what makes the derivation + * in `getLowerMap` the cheap order rather than the expensive one. + */ + let exactMap: Map | null = null; + + const getExactMap = (): Map => { + if (exactMap !== null) return exactMap; + const built = new Map(); + for (let i = 0; i < normalizedFileList.length; i++) { + const normalized = normalizedFileList[i]; + const original = allFileList[i]; + + // Index all suffixes: "a/b/c.java" -> ["c.java", "b/c.java", "a/b/c.java"]. + // + // Walked as slash offsets into `normalized` rather than as + // `normalized.split('/')` + `parts.slice(j).join('/')`: the slice of the + // ORIGINAL string is byte-identical to the re-joined parts (no separator + // is invented or dropped — verified over 361 865 suffix strings including + // leading, doubled and trailing slashes), and it allocates one string + // instead of a parts array, a slice array and a joined string per suffix. + // Measured 357.4 ms -> 264.5 ms at 32 000 paths. + let slash = normalized.lastIndexOf('/'); + while (slash >= 0) { + const suffix = normalized.slice(slash + 1); + // Only store first match (longest path wins for ambiguous suffixes) + if (!built.has(suffix)) built.set(suffix, original); + // A path may begin with '/', whose suffix is the whole string below. + if (slash === 0) break; + slash = normalized.lastIndexOf('/', slash - 1); } - const lower = suffix.toLowerCase(); - if (!lowerMap.has(lower)) { - lowerMap.set(lower, original); + // j = 0 — the whole path, which the slash walk cannot emit. + if (!built.has(normalized)) built.set(normalized, original); + } + exactMap = built; + return built; + }; + + /** + * Map: lowercase suffix -> original file path. + * + * Deferred, and when the exact map already exists DERIVED from it instead of + * traversed for: one pass over that map's DISTINCT keys rather than a second + * pass over every (file × depth) suffix. Measured 330.3 ms total (200.6 build + * + 129.7 derive) against 388.8 ms for the single fused traversal that built + * both eagerly — so the two-map consumers get cheaper too, which per-map + * laziness on its own does not (407.1 ms, a second full traversal). + * + * The derivation is EQUAL, not approximate, and the argument is short. Let + * the fused loop's global order be the pairs (suffix, file) it visited. For a + * lowercase key L, let p be the first position whose suffix lowercases to L — + * the entry today's `lowerMap` keeps. Nothing before p carries that suffix + * spelled ANY way, so p is also the first occurrence of its exact spelling + * and is therefore in the exact map, holding that same file. Exact-map + * insertion order is by first-occurrence position, so among the exact keys + * folding to L, p's is reached first and first-wins keeps it. Insertion order + * of the derived map is the order of those p's, which is the order today's + * `lowerMap` inserts L. Verified rather than only argued: byte-equal keys, + * values and order over 968 418 entries across bench-shaped, PascalCase, + * case-colliding, deep-monorepo, Unicode-adversarial and 400 seeded-fuzz + * corpora. + * + * When `getInsensitive` is asked FIRST (PHP), there is nothing to derive + * from, so it is built straight — one traversal, one map, which is the point. + * Asking `get` afterwards would then cost the second traversal; no consumer + * does, and the fallback stays correct if one ever starts. + */ + let lowerMap: Map | null = null; + + const getLowerMap = (): Map => { + // Over an already-lowercased file list the derivation is the identity, so + // the exact map IS the case-folded map. Skip the copy. + if (alreadyLowercased) return getExactMap(); + if (lowerMap !== null) return lowerMap; + + const built = new Map(); + if (exactMap !== null) { + for (const [suffix, original] of exactMap) { + const lower = suffix.toLowerCase(); + if (!built.has(lower)) built.set(lower, original); } + lowerMap = built; + return built; } - // Index directory membership - const lastSlash = normalized.lastIndexOf('/'); - if (lastSlash >= 0) { - // Build all directory suffixes - const dirParts = parts.slice(0, -1); - const fileName = parts[parts.length - 1]; - const ext = fileName.substring(fileName.lastIndexOf('.')); + for (let i = 0; i < normalizedFileList.length; i++) { + const normalized = normalizedFileList[i]; + const original = allFileList[i]; + let slash = normalized.lastIndexOf('/'); + while (slash >= 0) { + const lower = normalized.slice(slash + 1).toLowerCase(); + if (!built.has(lower)) built.set(lower, original); + if (slash === 0) break; + slash = normalized.lastIndexOf('/', slash - 1); + } + const whole = normalized.toLowerCase(); + if (!built.has(whole)) built.set(whole, original); + } + lowerMap = built; + return built; + }; - for (let j = dirParts.length - 1; j >= 0; j--) { - const dirSuffix = dirParts.slice(j).join('/'); - const key = `${dirSuffix}:${ext}`; - let list = dirMap.get(key); + /** + * Map: `${directory suffix}:${extension}` -> file paths in that directory. + * + * DEFERRED, not dropped (#2903). This is the array-valued map of the three + * and by far the most expensive: one entry — and one array push — per file + * per directory component, so O(files × depth) in entries AND in array + * churn. Measured on the 32k-path arms of `bench/import-target/`, it is + * ~15% of the retained C# index and ~19% of the retained Ruby one. + * + * Only `getFilesInDir` reads it, and only four call sites reach that: + * `import-resolvers/{php,csharp,jvm}.ts` and `import-resolvers/configs/ + * python.ts`. Every other consumer of this index — `workspace-file-index.ts` + * serving Ruby, `languages/typescript/scope-resolver.ts`, + * `languages/vue/import-target.ts`, `group/extractors/include-extractor.ts` + * — asks only suffix questions and was paying the whole footprint for a map + * it never touched. Since these indexes are now retained for a whole + * resolution pass rather than rebuilt per import (#2877-#2880), that is + * retained memory against the #2649 kernel-scale OOM constraint. + * + * `null` until the first `getFilesInDir`; the MAP is memoized, not the + * decision to build it, so a repeated miss cannot rebuild it. Building it + * later is behaviour-identical because it is a pure function of + * `normalizedFileList` / `allFileList`, and it retains nothing new: every + * production caller already holds both arrays alive alongside the index + * (`WorkspaceFileIndex.normalized`/`.all`, the TS and Vue `PassCache`s, + * `IncludeExtractor.extract`'s locals). + */ + let dirMap: Map | null = null; + + const getDirMap = (): Map => { + if (dirMap !== null) return dirMap; + const built = new Map(); + for (let i = 0; i < normalizedFileList.length; i++) { + const normalized = normalizedFileList[i]; + const original = allFileList[i]; + const lastSlash = normalized.lastIndexOf('/'); + // A file at the repo root is in no directory suffix. + if (lastSlash < 0) continue; + + // The file name from its last '.', or the WHOLE file name when it carries + // none — `substring(-1)` clamps to 0, which is what the `parts` form + // (`fileName.substring(fileName.lastIndexOf('.'))`) spelled. A '.' in a + // DIRECTORY is not an extension, hence `dot > lastSlash` rather than + // `dot >= 0`. + const dot = normalized.lastIndexOf('.'); + const ext = dot > lastSlash ? normalized.slice(dot) : normalized.slice(lastSlash + 1); + + // Every directory suffix of `normalized.slice(0, lastSlash)`, shortest + // first — the order `for (j = dirParts.length - 1; j >= 0; j--)` emitted, + // and load-bearing: `php.ts` returns `candidates[0]` of a bucket, so a + // reordered bucket is a behaviour change, not a wash. + // + // Walked as slash offsets into `normalized`, the same rewrite `getExactMap` + // above documents and for the same reason — a slice of the ORIGINAL string + // is byte-identical to the re-joined parts, and it allocates one string per + // suffix instead of a parts array, a slice array and a joined string per + // suffix. This is the map where it pays most: one entry, one array push AND + // one key per file per directory component, the "by far the most expensive" + // of the three. Measured 226.9 ms -> 173.1 ms at 32 000 paths averaging + // ~10 directory components (min of 9, both loops alternating in one + // process). Verified rather than argued, over a 32 000-path corpus + // carrying absolute paths, doubled separators (`a//b`), backslash paths, + // root-level and extensionless files, dotted directories and trailing + // separators: 272 956 keys and 329 361 bucket entries came out with + // identical key sets in identical INSERTION order and identical buckets + // element-for-element, and 767 732 probes of the built index — every + // emitted (directory, extension) pair plus a wrong-extension and a + // one-level-deeper miss for each — answered exactly as the `parts` form's + // map did. 0 differences. + // + // `slash < 0` is the whole directory, which no slash search can emit and + // the only suffix a one-component directory has. + let start = lastSlash; + while (start >= 0) { + const slash = start > 0 ? normalized.lastIndexOf('/', start - 1) : -1; + const key = `${normalized.slice(slash + 1, lastSlash)}:${ext}`; + let list = built.get(key); if (!list) { list = []; - dirMap.set(key, list); + built.set(key, list); } list.push(original); + start = slash; } } - } + dirMap = built; + return built; + }; return { - get: (suffix: string) => exactMap.get(suffix), - getInsensitive: (suffix: string) => lowerMap.get(suffix.toLowerCase()), + get: (suffix: string) => getExactMap().get(suffix), + getInsensitive: (suffix: string) => getLowerMap().get(suffix.toLowerCase()), + // THIS implementation shares: it hands back `dirMap`'s own bucket rather + // than a copy. The map is built on first query and then held for the whole + // pass, so the window in which a mutating caller could corrupt later + // imports is the whole pass — which is why the interface makes the result + // `readonly` and the compiler refuses the mutation at the call site. + // + // Sharing beats copying because no caller keeps the array: two only measure + // it and two build a fresh array from it, so a defensive copy would + // allocate a whole bucket per import on the path this index exists to keep + // flat. `package-dir-index.ts` reached the same conclusion the same way — + // read-only containers, plus one copy where a bucket genuinely LEAVES + // (`sortedRootFiles`), which is the case `configs/swift.ts` is in. getFilesInDir: (dirSuffix: string, extension: string) => { - return dirMap.get(`${dirSuffix}:${extension}`) || []; + return getDirMap().get(`${dirSuffix}:${extension}`) || []; }, }; } @@ -148,8 +386,8 @@ export function buildSuffixIndex(normalizedFileList: string[], allFileList: stri */ export function suffixResolve( pathParts: string[], - normalizedFileList: string[], - allFileList: string[], + normalizedFileList: readonly string[], + allFileList: readonly string[], index?: SuffixIndex, ): string | null { if (index) { diff --git a/gitnexus/src/core/ingestion/import-resolvers/workspace-file-index.ts b/gitnexus/src/core/ingestion/import-resolvers/workspace-file-index.ts new file mode 100644 index 000000000..ad5c882ec --- /dev/null +++ b/gitnexus/src/core/ingestion/import-resolvers/workspace-file-index.ts @@ -0,0 +1,94 @@ +/** + * Per-file-set workspace index for the import-target resolvers that need the + * shared `SuffixIndex` (C#, Java, PHP, Ruby). + * + * The scope-resolution orchestrator passes the SAME `allFilePaths` Set object to + * every `resolveImportTarget` call in a pass (`pipeline/run.ts` builds it once), + * so memoizing on the Set's identity in a `WeakMap` turns the per-import + * "materialize two arrays + build a suffix index" cost into a one-time build. + * + * IMPORTANT for callers: the Set must be passed THROUGH, never copied. A + * defensive `new Set(allFilePaths)` in an adapter hands a fresh `WeakMap` key + * per call and silently restores the O(imports × files) behaviour — the exact + * bug PR #1918 shipped and had to fix in review (P1). + * + * Three layers guard that, and they guard different things: + * - ADAPTER BOUNDARY, where the defensive-copy hazard actually lives: + * `test/integration/*-import-index-reuse.test.ts` resolves through + * `ScopeResolver.resolveImportTarget` — the orchestrator adapter — and + * pins the EXACT number of times a run traverses the file set, one file per + * covered language over that language's own corpus. (The expected count is + * per language and legitimately differs: it is however many times the + * adapter derives something from the Set — two indexes, or an index plus the + * mutable copy the ts-family context wants.) All of them count traversals + * of a `CountingSet` (`test/helpers/counting-file-set.ts`): one instrument, + * no production surface, and it catches both the per-import rebuild and a + * scan reintroduced beside a reused index (#2909). + * - EVERY REGISTERED LANGUAGE, at the same boundary but as one property rather + * than one corpus per language: `test/unit/scope-resolution/import-target-index-reuse.contract.test.ts` + * drives each entry of `SCOPE_RESOLVERS` and asserts the traversal count for + * many imports equals the count for two. A new language cannot skip it, and + * the enforcement is a test rather than a roster anyone maintains: that + * file's inventory arm compares `SCOPE_RESOLVERS`' keys against its own + * fixture table and fails on a registered resolver that has neither a + * fixture nor an exemption, and its next arm pins the exemption map empty. + * - RESOLVER LEVEL: `test/unit/scope-resolution/import-target-index-parity.test.ts` + * calls the resolvers directly, so it never crosses the adapter boundary and + * a copy there leaves it green. What it catches is a rescan reintroduced + * INSIDE a resolver, by counting how many times the Set is iterated. + */ + +import { perFileSet } from './per-file-set.js'; +import { buildSuffixIndex, type SuffixIndex } from './utils.js'; + +/** + * `normalized` and `all` are `readonly string[]`, and — like + * `SuffixIndex.getFilesInDir` — that is the CONTRACT rather than a description + * of the arrays: they are built once and then held for the whole pass, so an + * in-place `sort`/`splice`/`reverse` would corrupt every later import in that + * pass, and these two are the largest shared arrays here (one element per file, + * read by C#, Java, PHP and Ruby). `readonly` on the field is what makes the + * compiler refuse the mutation at the call site instead of leaving it to a + * comment. `ImportPassCache` (`pass-cache.ts`) states the same contract the + * same way for the ts-family lists. + * + * The positional pairing is load-bearing too and depends on it: `csharp.ts` + * caches POSITIONS into `normalized` and reads the answer out of `all`, so a + * reordering of either array alone silently re-points every cached position. + */ +export interface WorkspaceFileIndex { + /** Every path, backslashes normalized to `/`. Parallel to `all`. */ + readonly normalized: readonly string[]; + /** Every path, exactly as it appears in the Set. Parallel to `normalized`. */ + readonly all: readonly string[]; + /** Segment-suffix → first file (in Set iteration order) carrying that suffix. */ + readonly index: SuffixIndex; + /** + * Normalized path → first raw path that normalizes to it. Answers "is there a + * file whose WHOLE path is X", which `index.get(X)` cannot: the suffix map + * conflates a whole-path hit with a `…/X` suffix hit, and C#'s + * `resolveDirectMatch` lets a whole-path match win over an earlier suffix + * match. + */ + readonly normToRaw: Map; +} + +export const getWorkspaceFileIndex = perFileSet( + (allFilePaths: ReadonlySet): WorkspaceFileIndex => { + const all = [...allFilePaths]; + const normalized = all.map((f) => f.replace(/\\/g, '/')); + const normToRaw = new Map(); + for (let i = 0; i < normalized.length; i++) { + // First wins, mirroring the `for (const raw of allFilePaths)` scans this + // replaces: they returned on the first match in iteration order. + if (!normToRaw.has(normalized[i])) normToRaw.set(normalized[i], all[i]); + } + + return { + normalized, + all, + index: buildSuffixIndex(normalized, all), + normToRaw, + }; + }, +); diff --git a/gitnexus/src/core/ingestion/language-config.ts b/gitnexus/src/core/ingestion/language-config.ts index 16ad25e43..15558f689 100644 --- a/gitnexus/src/core/ingestion/language-config.ts +++ b/gitnexus/src/core/ingestion/language-config.ts @@ -2,11 +2,11 @@ import fs from 'fs/promises'; import { createReadStream } from 'fs'; import { createInterface } from 'readline'; import path from 'path'; -import type { ImportConfigs } from './import-resolvers/types.js'; import type { CsharpStructureLineScanner } from './languages/csharp/namespace-siblings.js'; import { isDev } from './utils/env.js'; +import { mapConcurrent } from '../../lib/utils.js'; import { logger } from '../logger.js'; // ============================================================================ // LANGUAGE-SPECIFIC CONFIG TYPES @@ -276,33 +276,32 @@ export async function scanCSharpProject(repoRoot: string): Promise readCsprojConfig(path.join(dir, name), name, repoRoot, dir)), - ); - for (const r of settled) { - const config = r.status === 'fulfilled' ? r.value : null; - if (config) { - configs.push(config); - rootNamespaces.add(config.rootNamespace); - } + // `mapConcurrent` runs the same bounded waves and degrades per item + // (a rejection becomes `undefined`), so entry order is still preserved. + const csprojResults = await mapConcurrent( + csprojNames, + (name) => readCsprojConfig(path.join(dir, name), name, repoRoot, dir), + { concurrency: CSHARP_SCAN_READ_CONCURRENCY }, + ); + for (const config of csprojResults) { + if (config) { + configs.push(config); + rootNamespaces.add(config.rootNamespace); } } - for (let i = 0; i < csNames.length; i += CSHARP_SCAN_READ_CONCURRENCY) { - const batch = csNames.slice(i, i + CSHARP_SCAN_READ_CONCURRENCY); - const settled = await Promise.allSettled( - batch.map((name) => - collectDeclaredNamespaces(path.join(dir, name), declaredNamespaces, rootNamespaces), - ), - ); - // A `.cs` that was unreadable (or whose read/scan unexpectedly rejected) - // leaves its namespaces uncollected → mark truncated to fail the #1881 - // gate OPEN rather than wrongly suppress an import. The scan streams each - // file, so file size no longer trips truncation. - for (const r of settled) { - if (r.status !== 'fulfilled' || r.value === 'truncated') truncated = true; - } + const csResults = await mapConcurrent( + csNames, + (name) => collectDeclaredNamespaces(path.join(dir, name), declaredNamespaces, rootNamespaces), + { concurrency: CSHARP_SCAN_READ_CONCURRENCY }, + ); + // A `.cs` that was unreadable (or whose read/scan unexpectedly rejected) + // leaves its namespaces uncollected → mark truncated to fail the #1881 + // gate OPEN rather than wrongly suppress an import. The scan streams each + // file, so file size no longer trips truncation. A rejected read arrives + // here as `undefined`, which is `!== 'ok'` just like the old + // `r.status !== 'fulfilled'` arm. + for (const r of csResults) { + if (r !== 'ok') truncated = true; } } @@ -470,6 +469,27 @@ export async function loadSwiftPackageConfig(repoRoot: string): Promise { const csharpScan = await scanCSharpProject(repoRoot); diff --git a/gitnexus/src/core/ingestion/language-provider.ts b/gitnexus/src/core/ingestion/language-provider.ts index 329faf42f..ec9006450 100644 --- a/gitnexus/src/core/ingestion/language-provider.ts +++ b/gitnexus/src/core/ingestion/language-provider.ts @@ -214,6 +214,20 @@ interface LanguageProviderConfig { * Default: undefined (standard label assignment). */ readonly labelOverride?: (functionNode: SyntaxNode, defaultLabel: NodeLabel) => NodeLabel | null; + /** + * Suppress a definition query match after its default label is known. + * Languages use this for syntax that represents an implicit declaration + * unless an explicit declaration with the same semantics is present. + * + * `defaultLabel` is supplied so an implementation can scope itself to one + * kind of definition; implementations whose capture map alone decides the + * question may ignore it. + */ + readonly shouldSkipDefinitionCapture?: ( + captureMap: CaptureMap, + defaultLabel: NodeLabel, + ) => boolean; + // ── MRO ─────────────────────────────────────────────────────────── /** MRO strategy for multiple inheritance resolution. * Default: 'first-wins'. */ @@ -282,14 +296,22 @@ interface LanguageProviderConfig { ) => ExtractedRoute[]; /** - * Extract decorator-style route annotations from a parsed file. + * Extract routes that a parsed file declares in its own AST. * * When defined, the parse worker calls this after per-file capture processing - * to extract framework route definitions that require AST-level analysis beyond + * to extract route definitions that require AST-level analysis beyond * generic `@decorator` captures (e.g., Java Spring class-level prefix joining, * multi-class handling). The returned routes are appended to `decoratorRoutes`. * - * Default: undefined (no language-specific decorator route extraction). + * Decorators are the common case and the reason for the name, but not the only + * shape: JS/TS uses this hook for hand-rolled dispatch guards + * (`route-extractors/dispatch-guard.ts`), where a raw `node:http` server + * declares a route by comparing the request path to a literal. Anything that + * yields a `(path, verb, handler)` triple from one file's AST belongs here — + * set `ExtractedDecoratorRoute.source` when the provenance is not a decorator, + * so the `HANDLES_ROUTE` edge does not claim one. + * + * Default: undefined (no language-specific route extraction). */ readonly extractDecoratorRoutes?: ( tree: Parser.Tree, @@ -435,6 +457,94 @@ interface LanguageProviderConfig { */ readonly interpretImport?: (captures: CaptureMatch) => ParsedImport | null; + /** + * Do this language's imports EXECUTE at the point in the program where they + * are written? + * + * The scope extractor marks an import `runsOnlyWhenCalled` when the statement + * sits inside a `Function` scope (Pass 3): where imports are executed + * statements, one written in a function body runs only when that function is + * called — Python's `def f(): from x import Y`, Ruby's + * `def f; require 'x'; end`, a CommonJS `require()` in a body (which + * `javascript/captures.ts` does capture, via its own AST walk). That rule is + * about EXECUTION. It says nothing true about a language whose "import" is + * not an executed statement at all, and in such a language moving one into a + * function body defers exactly nothing. + * + * "Where written" is about execution time, not textual placement: a + * `#include` is spliced precisely where it is written and still answers + * `false`, because splicing is not running. + * + * **The failure directions are not symmetric, which is why the default is + * what it is.** Answering `false` for a language that really does execute + * its imports un-defers a deliberately lazy one, and `check --cycles` + * reports a cycle its author broke on purpose — wrong, but visible on screen + * and arguable by whoever reads it. Answering `true` for a language that + * does not SUPPRESSES a cycle that is entirely real: nobody sees it, so + * nobody can argue with it. Getting this wrong in the `true` direction hides + * a true cycle, and that is the failure that matters. + * + * Absent — the default — reads as `true`, so every provider that does not + * name this keeps today's behaviour exactly. The flag never ADDS deferral; + * declaring `false` only WITHHOLDS it. + * + * Declared `false` by, and only by: + * + * - **C and C++** — `#include` is a preprocessor directive. The header's + * text is spliced in before a line of the program runs, wherever the + * directive sits, and C permits one inside a function body. C++'s only + * other Pass-3 import form, `using ns::name` / `using namespace ns`, is + * compile-time name lookup, is legal in a function body too, and defers + * no more than an `#include` does. + * - **Rust** — `use` is a compile-time path alias, not a statement that + * runs. `fn f() { use crate::m::X; }` is legal, and putting the `use` + * there changes only where the name is VISIBLE, never when anything + * happens; Rust has no module-initialization order in the JS/Python + * sense and permits intra-crate module cycles outright. It is the + * structural twin of C++'s `using ns::name`. `rust/query.ts` captures + * `(use_declaration)` and nothing else, so this covers the whole + * surface. (The claim here is the narrow one: POSITION does not defer a + * Rust import. Whether a Rust `use` can create an initialization + * dependency *at all* is a larger and separate question, and this flag + * deliberately does not answer it.) + * - **COBOL** — `COPY` is a pure textual splice performed by the copybook + * preprocessor, the `#include` case exactly. Latent today: the COBOL + * `@scope.function` capture covers a single line (`cobol/captures.ts` + * ranges sections and paragraphs `line → line`), so a `COPY` on any + * later line never resolves inside one and Pass 3 has nothing to mark. + * Declared anyway, so that giving those anchors their true multi-line + * ranges cannot silently start suppressing real copybook cycles. + * + * Per-provider rather than per-import, and that is sufficient — the question + * this answers is narrower than "do this language's imports execute". It is + * only ever asked of an import that resolved INSIDE A FUNCTION SCOPE, so the + * real domain is: *can a function-local import in this language be + * non-executing?* No supported language has two forms that are both + * function-local and disagree. C++ has two forms, `#include` and + * `using ns::name`; both can appear in a body and both are compile-time. + * + * PHP is the case that looks like a counterexample and is not. It does mix — + * `use Foo\Bar;` aliases at compile time while `require` executes — but + * `use` cannot appear in a function body at all (`php/query.ts` records this + * twice: "`namespace_use_declaration` is an import only at top level / inside + * namespace scope"), so it never reaches this flag. Absent is therefore + * PERMANENTLY correct for PHP, including the day `require` is captured: a + * `require` in a body will correctly defer, and a `use` still cannot get + * here. Do not read PHP as a reason to build a per-`ParsedImport` + * classification hook — it would cost a provider call per import on the + * extractor's hot path, and `ParsedImport.kind` does not discriminate the + * thing being asked anyway. + * + * A capability on the provider rather than a language check in + * `scope-extractor.ts`: shared `core/ingestion/` pipeline code must not name + * languages (AGENTS.md), and "imports here are not executed statements" is a + * property of the language, not of the walk. + * + * Default: undefined, read as `true` (imports execute where they are + * written; position defers them). + */ + readonly importsExecuteWhereWritten?: boolean; + /** * What is the implicit receiver on a Function scope? For instance methods * this is `self`/`this`; for standalone functions it is `null`. Consulted @@ -519,7 +629,7 @@ interface LanguageProviderConfig { readonly resolveImportTarget?: ( parsedImport: ParsedImport, workspaceIndex: WorkspaceIndex, - ) => string | null; + ) => string | readonly string[] | null; /** * Enumerate the exported names of a file — used by the finalize algorithm diff --git a/gitnexus/src/core/ingestion/languages/c-cpp.ts b/gitnexus/src/core/ingestion/languages/c-cpp.ts index 5378c2260..88961bcf9 100644 --- a/gitnexus/src/core/ingestion/languages/c-cpp.ts +++ b/gitnexus/src/core/ingestion/languages/c-cpp.ts @@ -420,6 +420,13 @@ export const cProvider = defineLanguage({ collectCaptureSideChannel: (filePath) => assertCloneable(collectCStaticLinkageSideChannel(filePath)), interpretImport: interpretCImport, + // `#include` is a preprocessor directive, not a statement that runs. The + // header text is spliced in before the program starts, wherever the directive + // sits — and C allows it inside a function body. Without this the central + // Pass-3 position rule would mark such an include `runsOnlyWhenCalled` and + // `check --cycles` would silently drop an include cycle that is entirely + // real. See `LanguageProvider.importsExecuteWhereWritten`. + importsExecuteWhereWritten: false, interpretTypeBinding: interpretCTypeBinding, bindingScopeFor: cBindingScopeFor, importOwningScope: cImportOwningScope, @@ -500,6 +507,12 @@ export const cppProvider = defineLanguage({ // re-parse (#1983). See `cpp/capture-side-channel.ts`. collectCaptureSideChannel: (filePath) => assertCloneable(collectCppCaptureSideChannel(filePath)), interpretImport: interpretCppImport, + // Same as C — `cpp/query.ts` emits `@import.statement` for `preproc_include` + // too. It holds for C++'s whole import surface: the only other form Pass 3 + // sees is `@import.using-decl` (`using ns::name` / `using namespace ns`), + // which is a compile-time name-lookup declaration and executes no more than + // an `#include` does. See the note on `cProvider`. + importsExecuteWhereWritten: false, interpretTypeBinding: interpretCppTypeBinding, bindingScopeFor: cppBindingScopeFor, importOwningScope: cppImportOwningScope, diff --git a/gitnexus/src/core/ingestion/languages/c/import-target.ts b/gitnexus/src/core/ingestion/languages/c/import-target.ts index 495846030..5590bb2e6 100644 --- a/gitnexus/src/core/ingestion/languages/c/import-target.ts +++ b/gitnexus/src/core/ingestion/languages/c/import-target.ts @@ -1,4 +1,5 @@ import { dirname, join } from 'path'; +import { perFileSet } from '../../import-resolvers/per-file-set.js'; /** * A workspace file path pre-decomposed for the suffix-match fallback: @@ -28,26 +29,20 @@ interface CSuffixCandidate { * `WeakMap`-keyed so it is reclaimed with the pass (no cross-pass staleness). * Shared by C and C++ (`resolveCppImportTarget` delegates here). */ -const suffixIndexByPaths = new WeakMap, Map>(); - -function suffixIndex(allFilePaths: ReadonlySet): Map { - let index = suffixIndexByPaths.get(allFilePaths); - if (index === undefined) { - index = new Map(); - for (const original of allFilePaths) { - const normalized = original.replace(/\\/g, '/'); - const basename = normalized.slice(normalized.lastIndexOf('/') + 1); - let bucket = index.get(basename); - if (bucket === undefined) { - bucket = []; - index.set(basename, bucket); - } - bucket.push({ original, normalized, depth: normalized.split('/').length }); +const suffixIndex = perFileSet((allFilePaths: ReadonlySet) => { + const index = new Map(); + for (const original of allFilePaths) { + const normalized = original.replace(/\\/g, '/'); + const basename = normalized.slice(normalized.lastIndexOf('/') + 1); + let bucket = index.get(basename); + if (bucket === undefined) { + bucket = []; + index.set(basename, bucket); } - suffixIndexByPaths.set(allFilePaths, index); + bucket.push({ original, normalized, depth: normalized.split('/').length }); } return index; -} +}); /** * Resolve a C #include path to a file in the workspace. diff --git a/gitnexus/src/core/ingestion/languages/c/scope-resolver.ts b/gitnexus/src/core/ingestion/languages/c/scope-resolver.ts index af31ec572..c3f9fb36f 100644 --- a/gitnexus/src/core/ingestion/languages/c/scope-resolver.ts +++ b/gitnexus/src/core/ingestion/languages/c/scope-resolver.ts @@ -8,6 +8,7 @@ import { cArityCompatibility, cMergeBindings, resolveCImportTarget } from './ind import { scanHeaderFiles } from './header-scan.js'; import { expandCWildcardNames, isStaticName, clearStaticNames } from './static-linkage.js'; import { applyCStaticLinkageSideChannel } from './capture-side-channel.js'; +import { perFileSet } from '../../import-resolvers/per-file-set.js'; /** * Per-pass memo of the augmented `#include`-resolution file set @@ -19,31 +20,26 @@ import { applyCStaticLinkageSideChannel } from './capture-side-channel.js'; * handing it a new set identity each time. Both `allFilePaths` (built once in * scope-resolution `run.ts`) and the header set (`loadResolutionConfig` * result) are stable per pass, so the union is built once and reused. - * `WeakMap`-keyed → reclaimed with the pass (no cross-pass staleness). + * Reclaimed with the pass (no cross-pass staleness). + * + * Two inputs, so two levels of `perFileSet` composed rather than a second + * primitive: the outer memo's value is the inner memo, and a function is an + * object, which is all `T extends object` asks for. + * + * The MEMO stays private to this file even though the C++ resolver's twin is + * byte-identical. The augmented set's IDENTITY is load-bearing downstream — + * C++ delegates to `resolveCImportTarget`, whose `suffixIndex` memo is keyed on + * exactly this set — so one memo shared across the two languages would hand + * each the other's index. Same builder-shared/memo-separate rule as + * `import-resolvers/pass-cache.ts`. */ -const augmentedPathsByPass = new WeakMap< - ReadonlySet, - WeakMap, ReadonlySet> ->(); - -function augmentedFilePaths( - allFilePaths: ReadonlySet, - headerPaths: ReadonlySet, -): ReadonlySet { - let byHeaders = augmentedPathsByPass.get(allFilePaths); - if (byHeaders === undefined) { - byHeaders = new WeakMap(); - augmentedPathsByPass.set(allFilePaths, byHeaders); - } - let augmented = byHeaders.get(headerPaths); - if (augmented === undefined) { +const augmentedFilePathsFor = perFileSet((allFilePaths: ReadonlySet) => + perFileSet((headerPaths: ReadonlySet): ReadonlySet => { const set = new Set(allFilePaths); for (const h of headerPaths) set.add(h); - augmented = set; - byHeaders.set(headerPaths, augmented); - } - return augmented; -} + return set; + }), +); /** * C `ScopeResolver` registered in `SCOPE_RESOLVERS` and consumed by @@ -94,7 +90,7 @@ export const cScopeResolver: ScopeResolver = { return resolveCImportTarget( targetRaw, fromFile, - augmentedFilePaths(allFilePaths, headerPaths), + augmentedFilePathsFor(allFilePaths)(headerPaths), ); } return resolveCImportTarget(targetRaw, fromFile, allFilePaths); diff --git a/gitnexus/src/core/ingestion/languages/c/static-linkage.ts b/gitnexus/src/core/ingestion/languages/c/static-linkage.ts index 2cc195205..a81398354 100644 --- a/gitnexus/src/core/ingestion/languages/c/static-linkage.ts +++ b/gitnexus/src/core/ingestion/languages/c/static-linkage.ts @@ -1,4 +1,5 @@ import type { ParsedFile, ScopeId, SymbolDefinition } from 'gitnexus-shared'; +import { perFileSet } from '../../import-resolvers/per-file-set.js'; /** * Per-file set of function names declared with `static` storage class. @@ -59,27 +60,23 @@ export function clearStaticNames(): void { * thousands of resolved includes) that is ~10^10+ comparisons on a single * thread — the dominant term in the scope-resolution finalize grind. * - * Building the lookup once collapses it to O(R_include + F). `WeakMap`-keyed - * on the array so the index is reclaimed with the pass — no cross-pass + * Building the lookup once collapses it to O(R_include + F). `perFileSet` keys + * on the array identity so the index is reclaimed with the pass — no cross-pass * staleness (mirrors the {@link clearStaticNames} discipline for server-mode * / multi-repo reuse), and a fresh array transparently rebuilds. */ -const moduleScopeIndexByPass = new WeakMap>(); - -function moduleScopeIndex(parsedFiles: readonly ParsedFile[]): Map { - let index = moduleScopeIndexByPass.get(parsedFiles); - if (index === undefined) { - index = new Map(); +const moduleScopeIndex = perFileSet( + (parsedFiles: readonly ParsedFile[]): Map => { + const index = new Map(); // First-wins to preserve `Array.find` semantics (returns the first match). // `moduleScope` is unique per file in practice, so collisions are absent; // the guard only formalises identical behaviour to the prior `.find`. for (const p of parsedFiles) { if (!index.has(p.moduleScope)) index.set(p.moduleScope, p); } - moduleScopeIndexByPass.set(parsedFiles, index); - } - return index; -} + return index; + }, +); /** * Return the names visible through a C wildcard import (`#include`). diff --git a/gitnexus/src/core/ingestion/languages/cobol.ts b/gitnexus/src/core/ingestion/languages/cobol.ts index 44891cef4..4a8ba2e2e 100644 --- a/gitnexus/src/core/ingestion/languages/cobol.ts +++ b/gitnexus/src/core/ingestion/languages/cobol.ts @@ -42,6 +42,19 @@ export const cobolProvider = defineLanguage({ // ── Scope-resolution hooks ─────────────────────────────────────── emitScopeCaptures: emitCobolScopeCaptures, interpretImport: interpretCobolImport, + // `COPY` is a pure textual splice by the copybook preprocessor — the + // `#include` case exactly, and COBOL's only import form. It is spliced before + // anything runs, so a copybook cycle built from `COPY` statements is real and + // must not be tagged `runsOnlyWhenCalled` by the central Pass-3 position rule. + // + // LATENT today, declared anyway. `cobol/captures.ts` ranges every + // `@scope.function` (PROCEDURE DIVISION sections and paragraphs) over a + // SINGLE line — `rangeOf(line, 0, line, endCol)` — so a `COPY` on any later + // line never resolves inside a Function scope and Pass 3 has nothing to mark. + // The flag is here so that giving those anchors their true multi-line ranges + // is a scope-resolution fix and not, silently, a cycle-suppression bug. + // See `LanguageProvider.importsExecuteWhereWritten`. + importsExecuteWhereWritten: false, importOwningScope: cobolImportOwningScope, receiverBinding: cobolReceiverBinding, }); diff --git a/gitnexus/src/core/ingestion/languages/cobol/scope-resolver.ts b/gitnexus/src/core/ingestion/languages/cobol/scope-resolver.ts index 9e528ce16..cfea8f39e 100644 --- a/gitnexus/src/core/ingestion/languages/cobol/scope-resolver.ts +++ b/gitnexus/src/core/ingestion/languages/cobol/scope-resolver.ts @@ -12,12 +12,71 @@ import path from 'node:path'; import type { ParsedFile } from 'gitnexus-shared'; import { SupportedLanguages } from 'gitnexus-shared'; +import { perFileSet } from '../../import-resolvers/per-file-set.js'; import { populateClassOwnedMembers } from '../../scope-resolution/scope/walkers.js'; import type { ScopeResolver } from '../../scope-resolution/contract/scope-resolver.js'; import { cobolProvider } from '../cobol.js'; // Copybook file extensions for COPY name resolution const COPYBOOK_EXTENSIONS = new Set(['.cpy', '.copybook']); +// COBOL source files, searched only after every copybook has missed. +const COBOL_SOURCE_EXTENSIONS = new Set(['.cbl', '.cob', '.cobol']); + +/** + * Uppercased-basename → first file carrying it, one map PER TIER, memoized on + * the `allFilePaths` Set identity (#2908). + * + * `resolveImportTarget` used to run two full workspace scans per `COPY` — one + * for the copybook tier, one for the source tier — each calling `path.extname` + * + `path.basename` + `toUpperCase` on every entry. A `COPY` of a member that + * lives outside the repo (the common case: vendor and system copybooks) missed + * in both, so both scans always ran to completion, making resolution + * O(copies × files). The orchestrator passes the SAME Set to every import in a + * pass (`pipeline/run.ts` builds it once), so a `WeakMap` keyed on that Set + * turns the scans into one build per run. + * + * Two tiers rather than one map is the tie-break, not a stylistic choice: a + * `.cpy`/`.copybook` hit beats a `.cbl`/`.cob`/`.cobol` hit even when the source + * file comes FIRST in Set-iteration order, which is exactly what collapsing the + * tiers into a single first-wins map would silently discard. Within a tier the + * first file in Set-iteration order wins, mirroring the `return` on first match + * in the scans this replaces. + * + * The per-file key is derived with the same `path.extname(fp).toLowerCase()` → + * `path.basename(fp, ext)` → `toUpperCase()` sequence the scans used, including + * its quirk: `path.basename` strips the suffix only on an exact, case-sensitive + * match, so `Foo.CPY` indexes under `FOO.CPY` rather than `FOO`. Node's `path` + * stays in the loop for the same reason — on POSIX it does not treat `\` as a + * separator, and hand-rolled slicing on `/` would start resolving backslash + * paths the scans never resolved. + */ +interface CobolCopyIndex { + /** `.cpy` / `.copybook` files — tier 1. */ + readonly copybooks: ReadonlyMap; + /** `.cbl` / `.cob` / `.cobol` files — tier 2. */ + readonly sources: ReadonlyMap; +} + +const getCobolCopyIndex = perFileSet((allFilePaths: ReadonlySet): CobolCopyIndex => { + const copybooks = new Map(); + const sources = new Map(); + // One pass builds both tiers: the two scans walked the same files and + // classified each by the same extension test. + for (const fp of allFilePaths) { + const ext = path.extname(fp).toLowerCase(); + const tier = COPYBOOK_EXTENSIONS.has(ext) + ? copybooks + : COBOL_SOURCE_EXTENSIONS.has(ext) + ? sources + : undefined; + if (tier === undefined) continue; + const basename = path.basename(fp, ext).toUpperCase(); + // First in Set-iteration order wins, as the scans' first-match `return` did. + if (!tier.has(basename)) tier.set(basename, fp); + } + + return { copybooks, sources }; +}); const cobolScopeResolver: ScopeResolver = { language: SupportedLanguages.Cobol, @@ -27,22 +86,9 @@ const cobolScopeResolver: ScopeResolver = { // ── Resolve COPY bookname to file path ───────────────────────────── resolveImportTarget: (targetRaw, _fromFile, allFilePaths) => { const upper = targetRaw.toUpperCase(); - // Check copybook files first - for (const fp of allFilePaths) { - const ext = path.extname(fp).toLowerCase(); - if (!COPYBOOK_EXTENSIONS.has(ext)) continue; - const basename = path.basename(fp, ext).toUpperCase(); - if (basename === upper) return fp; - } - // Also search COBOL source files (.cbl, .cob, .cobol) - const COBOL_SOURCE_EXTS = new Set(['.cbl', '.cob', '.cobol']); - for (const fp of allFilePaths) { - const ext = path.extname(fp).toLowerCase(); - if (!COBOL_SOURCE_EXTS.has(ext)) continue; - const basename = path.basename(fp, ext).toUpperCase(); - if (basename === upper) return fp; - } - return null; + const index = getCobolCopyIndex(allFilePaths); + // Copybooks first, then COBOL sources — the tier order IS the tie-break. + return index.copybooks.get(upper) ?? index.sources.get(upper) ?? null; }, // COBOL has no binding-merge rules beyond the default (local-first-then-imports). diff --git a/gitnexus/src/core/ingestion/languages/cpp/file-local-linkage.ts b/gitnexus/src/core/ingestion/languages/cpp/file-local-linkage.ts index c6b383bf0..e578b20d4 100644 --- a/gitnexus/src/core/ingestion/languages/cpp/file-local-linkage.ts +++ b/gitnexus/src/core/ingestion/languages/cpp/file-local-linkage.ts @@ -1,4 +1,5 @@ import type { ParsedFile, Scope, ScopeId, SymbolDefinition } from 'gitnexus-shared'; +import { perFileSet } from '../../import-resolvers/per-file-set.js'; import { isCppInlineNamespaceScope } from './inline-namespaces.js'; /** @@ -283,24 +284,20 @@ export function isCppDefGloballyVisible(filePath: string, nodeId: string): boole * `parsedFiles` reference; the old `parsedFiles.find(...)` was therefore O(F) * per edge → O(R·F) overall (at kernel scale the ~25–30k `.h` headers are * classified C++, so this fires hard — the C twin in `c/static-linkage.ts`). - * Building the lookup once collapses it to O(R+F). `WeakMap`-keyed so it is - * reclaimed with the pass (no cross-pass staleness; mirrors - * {@link clearFileLocalNames}). + * Building the lookup once collapses it to O(R+F). `perFileSet` keys on the + * array identity so it is reclaimed with the pass (no cross-pass staleness; + * mirrors {@link clearFileLocalNames}). */ -const moduleScopeIndexByPass = new WeakMap>(); - -function moduleScopeIndex(parsedFiles: readonly ParsedFile[]): Map { - let index = moduleScopeIndexByPass.get(parsedFiles); - if (index === undefined) { - index = new Map(); +const moduleScopeIndex = perFileSet( + (parsedFiles: readonly ParsedFile[]): Map => { + const index = new Map(); // First-wins to preserve `Array.find` semantics (returns the first match). for (const p of parsedFiles) { if (!index.has(p.moduleScope)) index.set(p.moduleScope, p); } - moduleScopeIndexByPass.set(parsedFiles, index); - } - return index; -} + return index; + }, +); export function expandCppWildcardNames( targetModuleScope: ScopeId, diff --git a/gitnexus/src/core/ingestion/languages/cpp/interpret.ts b/gitnexus/src/core/ingestion/languages/cpp/interpret.ts index 1627232c5..e5a3635ca 100644 --- a/gitnexus/src/core/ingestion/languages/cpp/interpret.ts +++ b/gitnexus/src/core/ingestion/languages/cpp/interpret.ts @@ -77,10 +77,71 @@ export function interpretCppTypeBinding(captures: CaptureMatch): ParsedTypeBindi source = 'annotation'; } - const declaredSpelling = cppPointerSpelling(captures, type, name); + // A member field's type is captured AS WRITTEN, qualifier and all + // (`ns::Repo`), because the query matches the outer + // `qualified_identifier` — one depth-agnostic pattern per declarator shape + // instead of one per qualifier depth. The qualifier is dropped HERE; see the + // "Field type, QUALIFIED" block in query.ts for why the qualified spelling + // resolves to nothing and the tail resolves like the bare one. + // + // FIELDS ONLY. `@type-binding.parameter` and `@type-binding.assignment` also + // capture qualified spellings (their patterns use `type: (_)`), and reducing + // THOSE would newly bind every qualified local and parameter in the workspace + // — a far wider change than the member-field miss this closes, and not one + // anything here has measured. + const effectiveType = + captures['@type-binding.field'] === undefined ? type : cppQualifiedTail(type); + // The reduced spelling is also the AS-WRITTEN one, and saying so is load + // bearing. `collectTypeBindings` derives `TypeRef.declaredSpelling` from + // `@type-binding.type` whenever that text differs from `rawTypeName`, and it + // now does for every qualified member. `declaredSpelling` exists to keep a + // CONTAINER distinguishable from a class of the same name after capture + // reduced it; a qualifier is not a container — `ns::Address` and `Address` + // have the identical member set — so recording one here would answer + // "container, as written" for a plain member and hand `elementTypeOf` a + // spelling it never sees for the bare form. + const declaredSpelling = + cppPointerSpelling(captures, effectiveType, name) ?? + (effectiveType === type ? undefined : effectiveType); return declaredSpelling === undefined - ? { boundName: name, rawTypeName: normalizeCppTypeName(type), source } - : { boundName: name, rawTypeName: normalizeCppTypeName(type), declaredSpelling, source }; + ? { boundName: name, rawTypeName: normalizeCppTypeName(effectiveType), source } + : { + boundName: name, + rawTypeName: normalizeCppTypeName(effectiveType), + declaredSpelling, + source, + }; +} + +/** + * The tail of a `::`-qualified type spelling — `a::b::Repo` → `Repo`, + * `ns::Address` → `Address`, an unqualified spelling unchanged. + * + * Only TOP-LEVEL separators count, so a qualified TYPE ARGUMENT survives: + * `std::vector` reduces to `vector`, not to `string`. + * That is the same string the old per-depth rules produced by capturing the + * inner node, so the reduction is textual where it used to be structural and + * the result is identical for every depth they covered. + */ +function cppQualifiedTail(text: string): string { + let angleDepth = 0; + let lastSeparator = -1; + for (let i = 0; i < text.length; i++) { + const ch = text[i]; + if (ch === '<') angleDepth++; + else if (ch === '>') { + if (angleDepth > 0) angleDepth--; + } else if (angleDepth === 0 && ch === ':' && text[i + 1] === ':') { + lastSeparator = i; + i++; + } + } + if (lastSeparator === -1) return text; + const tail = text.slice(lastSeparator + 2).trim(); + // A spelling that ends in `::` has no tail to reduce to. Cannot arise from a + // parsed `qualified_identifier`, but returning an empty type name would make + // the binding claim a type of `""`, so the written spelling is kept instead. + return tail.length === 0 ? text : tail; } /** Anchors whose capture spans a whole declaration, so the declarator — and diff --git a/gitnexus/src/core/ingestion/languages/cpp/query.ts b/gitnexus/src/core/ingestion/languages/cpp/query.ts index 52fcc0785..7b708b60b 100644 --- a/gitnexus/src/core/ingestion/languages/cpp/query.ts +++ b/gitnexus/src/core/ingestion/languages/cpp/query.ts @@ -55,12 +55,26 @@ const CPP_SCOPE_QUERY = ` declarator: (type_identifier) @declaration.name) @declaration.struct ;; ─── Declarations — class / struct inside template_declaration ─────── +;; \`parameters:\` is the DECLARED parameter list (\`template \`), which +;; lives on the template_declaration and not on the specifier — the opposite +;; nesting from \`@declaration.template-arguments\` above, which is part of the +;; specifier's own NAME. A partial specialization carries both, and that pairing +;; is the only thing separating it from a full specialization written against +;; the identical arguments. +;; +;; These four patterns are TWINS of the four standalone specifier patterns +;; above: a templated struct matches both, minting two defs with one id, and +;; only this half can see the parameter list. The duplicate-declaration backfill +;; in scope-extractor.ts is what stops match order from deciding which twin +;; keeps the parameters. (template_declaration + parameters: (template_parameter_list) @declaration.type-parameters (class_specifier name: (type_identifier) @declaration.name body: (field_declaration_list)) @declaration.class) (template_declaration + parameters: (template_parameter_list) @declaration.type-parameters (class_specifier name: (template_type (type_identifier) @declaration.name @@ -68,11 +82,13 @@ const CPP_SCOPE_QUERY = ` body: (field_declaration_list)) @declaration.class) (template_declaration + parameters: (template_parameter_list) @declaration.type-parameters (struct_specifier name: (type_identifier) @declaration.name body: (field_declaration_list)) @declaration.struct) (template_declaration + parameters: (template_parameter_list) @declaration.type-parameters (struct_specifier name: (template_type (type_identifier) @declaration.name @@ -537,6 +553,86 @@ const CPP_SCOPE_QUERY = ` declarator: (reference_declarator (field_identifier) @type-binding.name)) @type-binding.field +;; Generic field type: Repo repo; (#2833) +;; The three rules above all require type: (type_identifier), so a member whose +;; type carries template arguments is a template_type and matched NONE of them — +;; the field got no type binding at all, and every call through it lost its edge +;; in BOTH spellings (repo.save() and this->repo.save()), while the same type in +;; a LOCAL resolved fine because the local declaration rules gained their +;; template_type variant long ago (see "Covers: List users;" above). +;; These three mirror the three above, one per declarator shape. Written as +;; separate patterns rather than one alternation: a node-type alternation in a +;; field position is the tree-sitter 0.21 hazard this repo has been bitten by +;; before. +(field_declaration + type: (template_type) @type-binding.type + declarator: (field_identifier) @type-binding.name) @type-binding.field + +;; Generic field, pointer: Repo* repo; +(field_declaration + type: (template_type) @type-binding.type + declarator: (pointer_declarator + declarator: (field_identifier) @type-binding.name)) @type-binding.field + +;; Generic field, reference: Repo& repo; +(field_declaration + type: (template_type) @type-binding.type + declarator: (reference_declarator + (field_identifier) @type-binding.name)) @type-binding.field + +;; ─── Field type, QUALIFIED: ns::Address addr; std::vector items; ─ +;; The six rules above require the type node to BE a type_identifier or a +;; template_type, and a qualified member type is NEITHER: tree-sitter-cpp parses +;; ns::Address as a qualified_identifier WRAPPING the type_identifier, and +;; std::vector as one wrapping the template_type. So every qualified +;; member — generic or not — matched none of the six and bound nothing, which +;; covers the commonest member spellings in real C++ (std::string, std::mutex, +;; std::vector, ns::Config). +;; +;; ONE PATTERN PER DECLARATOR SHAPE, MATCHING THE OUTER qualified_identifier, +;; and that is the whole design. A tree-sitter query cannot match a node at +;; arbitrary nesting depth, and a::b::c::Repo nests one +;; qualified_identifier per qualifier — so enumerating the inner node instead +;; costs 3 patterns per depth per genericity and STILL ends at whatever depth +;; the last author enumerated (that boundary was real: depth 3 was uncaptured). +;; Matching the outer node is depth-agnostic and genericity-agnostic, and it is +;; a single node type in the field position, not an alternation — the +;; tree-sitter 0.21 hazard this repo has been bitten by before. +;; +;; The QUALIFIER IS THEN DROPPED, by cppQualifiedTail in interpret.ts, not +;; here — and dropping it was measured rather than assumed. Recording +;; ns::Repo resolves to NOTHING: findClassBindingInScope's dotted-tail +;; fallback splits on "." and C++ writes "::", and resolveClassBindingForName's +;; generic branch then looks up the base ns::Repo, which is not a key either +;; because C++ emits no @declaration.qualified_name and indexes ns::Repo under +;; Repo. Reducing to the tail lands on exactly the path the BARE spelling +;; already takes — one class-like match or decline — so a qualified member field +;; behaves like the bare one instead of like nothing. A tail that names no +;; workspace class (std::string with no "class string" in the repo) binds +;; nothing and emits nothing, which is why this is a miss-closing change rather +;; than an edge-fabricating one. +;; +;; Like the six above, each requires the declarator to reach the field_identifier +;; DIRECTLY, so a method whose return type is qualified (ns::Thing method();) +;; still captures no field — a function_declarator sits in between and none of +;; these match it. Same for a function-pointer member, a using/typedef alias, a +;; friend declaration and an operator declaration. +(field_declaration + type: (qualified_identifier) @type-binding.type + declarator: (field_identifier) @type-binding.name) @type-binding.field + +;; Qualified field, pointer: ns::Address* addr; std::unique_ptr* repo; +(field_declaration + type: (qualified_identifier) @type-binding.type + declarator: (pointer_declarator + declarator: (field_identifier) @type-binding.name)) @type-binding.field + +;; Qualified field, reference: ns::Address& addr; std::vector& items; +(field_declaration + type: (qualified_identifier) @type-binding.type + declarator: (reference_declarator + (field_identifier) @type-binding.name)) @type-binding.field + ;; ─── References — constructor calls (new Foo()) ───────────────────── (new_expression type: (type_identifier) @reference.name) @reference.call.constructor diff --git a/gitnexus/src/core/ingestion/languages/cpp/scope-resolver.ts b/gitnexus/src/core/ingestion/languages/cpp/scope-resolver.ts index 13f21183e..9421975ea 100644 --- a/gitnexus/src/core/ingestion/languages/cpp/scope-resolver.ts +++ b/gitnexus/src/core/ingestion/languages/cpp/scope-resolver.ts @@ -48,6 +48,7 @@ import { resolveCppReceiverMember, } from './member-lookup.js'; import { stripCppSpecifiers } from './interpret.js'; +import { perFileSet } from '../../import-resolvers/per-file-set.js'; /** A pointee worth binding: a bare identifier, not `T**`, `T[]`, `A::B` or a * template spelling. Hoisted — a literal here would mint a fresh RegExp on @@ -61,32 +62,25 @@ const CPP_SIMPLE_POINTEE_RE = /^[A-Za-z_]\w*$/; * a fresh ~F-entry `Set` on every call AND defeated the shared * `resolveCImportTarget` suffix-index memo (in `c/import-target.ts`) by handing * it a new set identity each time. Both inputs are stable per pass, so the - * union is built once and reused. `WeakMap`-keyed → reclaimed with the pass. - * (Twin of the C resolver's `augmentedFilePaths`.) + * union is built once and reused. Reclaimed with the pass. + * + * Two inputs, so two levels of `perFileSet` composed rather than a second + * primitive: the outer memo's value is the inner memo, and a function is an + * object, which is all `T extends object` asks for. + * + * (Twin of the C resolver's `augmentedFilePathsFor`.) The two memos stay + * SEPARATE deliberately. C++ delegates to `resolveCImportTarget`, whose + * `suffixIndex` memo is keyed on the augmented set, so a single memo shared + * with C would hand each language the other's index — same + * builder-shared/memo-separate rule as `import-resolvers/pass-cache.ts`. */ -const augmentedPathsByPass = new WeakMap< - ReadonlySet, - WeakMap, ReadonlySet> ->(); - -function augmentedFilePaths( - allFilePaths: ReadonlySet, - headerPaths: ReadonlySet, -): ReadonlySet { - let byHeaders = augmentedPathsByPass.get(allFilePaths); - if (byHeaders === undefined) { - byHeaders = new WeakMap(); - augmentedPathsByPass.set(allFilePaths, byHeaders); - } - let augmented = byHeaders.get(headerPaths); - if (augmented === undefined) { +const augmentedFilePathsFor = perFileSet((allFilePaths: ReadonlySet) => + perFileSet((headerPaths: ReadonlySet): ReadonlySet => { const set = new Set(allFilePaths); for (const h of headerPaths) set.add(h); - augmented = set; - byHeaders.set(headerPaths, augmented); - } - return augmented; -} + return set; + }), +); /** * C++ `ScopeResolver` registered in `SCOPE_RESOLVERS` and consumed by @@ -128,7 +122,7 @@ export const cppScopeResolver: ScopeResolver = { return resolveCppImportTarget( targetRaw, fromFile, - augmentedFilePaths(allFilePaths, headerPaths), + augmentedFilePathsFor(allFilePaths)(headerPaths), ); } return resolveCppImportTarget(targetRaw, fromFile, allFilePaths); diff --git a/gitnexus/src/core/ingestion/languages/csharp/import-target.ts b/gitnexus/src/core/ingestion/languages/csharp/import-target.ts index 3745e16de..68f1e484d 100644 --- a/gitnexus/src/core/ingestion/languages/csharp/import-target.ts +++ b/gitnexus/src/core/ingestion/languages/csharp/import-target.ts @@ -22,7 +22,16 @@ import type { ParsedImport, WorkspaceIndex } from 'gitnexus-shared'; import type { CSharpProjectConfig, CSharpNamespaceEvidence } from '../../language-config.js'; import { resolveCSharpImportInternal } from '../../import-resolvers/csharp.js'; -import { buildSuffixIndex, type SuffixIndex } from '../../import-resolvers/utils.js'; +import { + getWorkspaceFileIndex, + type WorkspaceFileIndex, +} from '../../import-resolvers/workspace-file-index.js'; +import { + buildPackageDirIndex, + firstFileDirectlyInPkgDir, + type PackageDirIndex, +} from '../../import-resolvers/package-dir-index.js'; +import { perFileSet } from '../../import-resolvers/per-file-set.js'; import { csharpSuffixFallbackAllowed } from '../../csharp-namespace-gate.js'; export interface CsharpResolveContext { @@ -32,27 +41,16 @@ export interface CsharpResolveContext { readonly namespaces?: CSharpNamespaceEvidence; } -/** Normalized file list + suffix index, built once per workspace `allFilePaths`. */ -interface WorkspaceFileIndex { - readonly normalized: string[]; - readonly all: string[]; - readonly index: SuffixIndex; -} - -// Memoize on Set identity: the orchestrator passes the SAME `allFilePaths` -// Set through every `resolveImportTarget` call in a pass, so this rebuilds -// the normalized list + suffix index once instead of once per import (#1881 #2). -const workspaceFileIndexCache = new WeakMap, WorkspaceFileIndex>(); - -function getWorkspaceFileIndex(allFilePaths: ReadonlySet): WorkspaceFileIndex { - const cached = workspaceFileIndexCache.get(allFilePaths); - if (cached) return cached; - const all = [...allFilePaths]; - const normalized = all.map((f) => f.replace(/\\/g, '/')); - const built: WorkspaceFileIndex = { normalized, all, index: buildSuffixIndex(normalized, all) }; - workspaceFileIndexCache.set(allFilePaths, built); - return built; -} +/** + * Namespace-directory index over the `.cs` files, memoized on the Set's + * identity. Feeds `firstFileDirectlyInPkgDir` (in + * `import-resolvers/package-dir-index.ts`), which the no-csproj path calls once + * for the direct match and then up to once per stripped namespace prefix. + */ +const getCsharpDirIndex = perFileSet( + (allFilePaths: ReadonlySet): PackageDirIndex => + buildPackageDirIndex(allFilePaths, (normalized) => normalized.endsWith('.cs')), +); export function resolveCsharpImportTarget( parsedImport: ParsedImport, @@ -67,12 +65,11 @@ export function resolveCsharpImportTarget( const csharpConfigs = ctx.csharpConfigs ?? []; if (csharpConfigs.length > 0) { - const { normalized, all, index } = getWorkspaceFileIndex(ctx.allFilePaths); + const { index } = getWorkspaceFileIndex(ctx.allFilePaths); const fromCsproj = resolveCSharpImportInternal( targetRaw, [...csharpConfigs], - normalized, - all, + ctx.allFilePaths, index, evidence, ); @@ -101,12 +98,19 @@ export function resolveCsharpImportTarget( } // Exact file / nested-suffix / namespace-dir direct-child match. - const direct = resolveDirectMatch(ctx.allFilePaths, pathLike); + // + // The no-csproj path used to take the raw Set and re-scan it — up to eight + // full workspace passes for a four-segment `using` — past the memoized index + // sitting right there for the csproj branch (#2878). Both legs now read the + // same per-run indexes. + const ws = getWorkspaceFileIndex(ctx.allFilePaths); + const dirs = getCsharpDirIndex(ctx.allFilePaths); + const direct = resolveDirectMatch(ws, dirs, pathLike); if (direct !== null) return direct; // Progressive prefix stripping — mirrors csproj's root-namespace mapping // without the csproj. - return resolveByProgressiveStripping(ctx.allFilePaths, pathLike); + return resolveByProgressiveStripping(ws, dirs, pathLike); } /** @@ -131,40 +135,28 @@ function narrowContext(workspaceIndex: WorkspaceIndex): CsharpResolveContext | n * exact whole-path file > nested suffix file > first `.cs` directly inside * the namespace directory. */ -function resolveDirectMatch(allFilePaths: ReadonlySet, pathLike: string): string | null { +function resolveDirectMatch( + ws: WorkspaceFileIndex, + dirs: PackageDirIndex, + pathLike: string, +): string | null { const exactName = `${pathLike}.cs`; - const nestedSuffix = `/${exactName}`; - let suffixFile: string | null = null; - for (const raw of allFilePaths) { - const f = raw.replace(/\\/g, '/'); - if (!f.endsWith('.cs')) continue; - if (f === exactName) return raw; // exact whole-path match wins - if (suffixFile === null && f.endsWith(nestedSuffix)) suffixFile = raw; - } - if (suffixFile !== null) return suffixFile; - return findDirectChild(allFilePaths, pathLike); -} - -/** - * First `.cs` file that lives directly inside the namespace directory - * `dirSegment` (at repo root or nested under a project prefix), not deeper. - * The legacy resolver emits all of them; the scope-resolver contract is - * single-target so we take one. - */ -function findDirectChild(allFilePaths: ReadonlySet, dirSegment: string): string | null { - const dirPrefix = `${dirSegment}/`; - const nestedDirPrefix = `/${dirPrefix}`; - for (const raw of allFilePaths) { - const f = raw.replace(/\\/g, '/'); - if (!f.endsWith('.cs')) continue; - const atRoot = f.startsWith(dirPrefix); - const atNested = f.includes(nestedDirPrefix); - if (!atRoot && !atNested) continue; - const idx = atRoot ? 0 : f.indexOf(nestedDirPrefix) + 1; - const after = f.slice(idx + dirPrefix.length); - if (after.length > 0 && !after.includes('/')) return raw; - } - return null; + // An exact whole-path match wins even when a `…/` suffix match + // appeared EARLIER in iteration order, so the two lookups stay separate: + // `index.get` conflates them and would return the earlier suffix hit. + const exact = ws.normToRaw.get(exactName); + if (exact !== undefined) return exact; + // No whole-path file exists, so every segment-suffix hit is a `/` + // match and `index.get` yields the first one in iteration order — exactly the + // `suffixFile` the scan kept. Only a `.cs` file can carry a `.cs` suffix key, + // so the old `endsWith('.cs')` filter is implied. + const suffixFile = ws.index.get(exactName); + if (suffixFile !== undefined) return suffixFile; + // First `.cs` file living directly inside the namespace directory `pathLike` + // (at repo root or nested under a project prefix), not deeper. The legacy + // resolver emits all of them; the scope-resolver contract is single-target so + // we take one. + return firstFileDirectlyInPkgDir(dirs, pathLike); } /** @@ -174,26 +166,20 @@ function findDirectChild(allFilePaths: ReadonlySet, dirSegment: string): * prefix (the scope-resolver layer has no csproj to consult). */ function resolveByProgressiveStripping( - allFilePaths: ReadonlySet, + ws: WorkspaceFileIndex, + dirs: PackageDirIndex, pathLike: string, ): string | null { const segments = pathLike.split('/').filter(Boolean); for (let skip = 1; skip < segments.length; skip++) { const tail = segments.slice(skip).join('/'); if (tail === '') continue; - const tailFile = `${tail}.cs`; - const tailSuffix = `/${tailFile}`; - let tailFileMatch: string | null = null; - for (const raw of allFilePaths) { - const f = raw.replace(/\\/g, '/'); - if (!f.endsWith('.cs')) continue; - if (f === tailFile || f.endsWith(tailSuffix)) { - tailFileMatch = raw; - break; - } - } - if (tailFileMatch !== null) return tailFileMatch; - const child = findDirectChild(allFilePaths, tail); + // `f === tailFile || f.endsWith('/' + tailFile)`, first in iteration order — + // no exact-wins rule here, unlike `resolveDirectMatch`, so the conflated + // suffix lookup is the right one. + const tailFileMatch = ws.index.get(`${tail}.cs`); + if (tailFileMatch !== undefined) return tailFileMatch; + const child = firstFileDirectlyInPkgDir(dirs, tail); if (child !== null) return child; } return null; diff --git a/gitnexus/src/core/ingestion/languages/csharp/query.ts b/gitnexus/src/core/ingestion/languages/csharp/query.ts index 0a0bb65ca..18c37ba2b 100644 --- a/gitnexus/src/core/ingestion/languages/csharp/query.ts +++ b/gitnexus/src/core/ingestion/languages/csharp/query.ts @@ -63,24 +63,44 @@ const CSHARP_SCOPE_QUERY = ` ;; Anonymous methods / lambdas are not scoped — out of scope per plan. ;; Declarations — types +;; The parameter list is matched as an UNNAMED optional child, not through a +;; \`type_parameters:\` field: the C# grammar gives \`interface_declaration\` that +;; field but \`class_declaration\` / \`struct_declaration\` / \`record_declaration\` +;; only a bare \`type_parameter_list\` child, so the field form would silently +;; capture nothing on exactly the three most common declarations. The unnamed +;; form matches all four. +;; +;; A \`where T : IRepo\` constraint is a SEPARATE sibling clause +;; (\`type_parameter_constraints_clause\`) and is deliberately not read here — the +;; bound stays absent for C#, which reads as "unknown", the safe direction. (class_declaration - name: (identifier) @declaration.name) @declaration.class + name: (identifier) @declaration.name + (type_parameter_list)? @declaration.type-parameters) @declaration.class (interface_declaration - name: (identifier) @declaration.name) @declaration.interface + name: (identifier) @declaration.name + (type_parameter_list)? @declaration.type-parameters) @declaration.interface (struct_declaration - name: (identifier) @declaration.name) @declaration.struct + name: (identifier) @declaration.name + (type_parameter_list)? @declaration.type-parameters) @declaration.struct (record_declaration - name: (identifier) @declaration.name) @declaration.record + name: (identifier) @declaration.name + (type_parameter_list)? @declaration.type-parameters) @declaration.record (enum_declaration name: (identifier) @declaration.name) @declaration.enum ;; Declarations — methods / constructors / properties +;; +;; A generic METHOD's parameters are read for the same reason a generic type's +;; are (#2912 review): \`void Run(IValidator v)\` writes a receiver whose +;; argument is a type VARIABLE, and a pass that cannot tell that from a concrete +;; type prunes every implementor of \`IValidator\` from the call's fan-out. (method_declaration - name: (identifier) @declaration.name) @declaration.method + name: (identifier) @declaration.name + (type_parameter_list)? @declaration.type-parameters) @declaration.method (constructor_declaration name: (identifier) @declaration.name) @declaration.constructor diff --git a/gitnexus/src/core/ingestion/languages/csharp/scope-resolver.ts b/gitnexus/src/core/ingestion/languages/csharp/scope-resolver.ts index eb1d3b10d..4b50efc67 100644 --- a/gitnexus/src/core/ingestion/languages/csharp/scope-resolver.ts +++ b/gitnexus/src/core/ingestion/languages/csharp/scope-resolver.ts @@ -102,6 +102,82 @@ const csharpScopeResolver: ScopeResolver = { // files. The compound-receiver walker needs to walk up from the // class scope to find them; see the contract field for rationale. hoistTypeBindingsToModule: true, + + // `IValidator` and `IValidator` are one instantiation, so the + // dispatch fan-out must not read them as two (#2912). See the alias table. + normalizeTypeArgument: normalizeCsharpTypeArgument, }; +/** + * C# predefined type aliases — the 15 keywords the language defines as exact + * synonyms for `System` types (`string` ≡ `System.String`), plus `nint`/`nuint`. + * A codebase mixing the spellings is common enough that StyleCop ships a rule + * about it (SA1121), so the two forms genuinely meet across files. + * + * Keyword → BCL simple name; anything else is returned unchanged, including the + * BCL names themselves (already canonical) and any qualified spelling, which is + * compared as written. + * + * A workspace may legally declare its OWN type named `String`, which shadows the + * BCL simple name; this table then reads `IValidator` as the `string` + * instantiation and KEEPS that implementor in the fan-out. Deliberate, and the + * safe direction: the alternative is pruning on the belief that two spellings + * differ, which is the missing-edge failure `generic-instantiation.ts` is built + * to avoid. Resolving instead of normalizing cannot settle it either — the + * identity comparison needs a `definitionId` from BOTH sides, and a built-in + * name has none, so "built-in versus workspace-declared" would be a new prune + * with no positive evidence behind it. The result is one surplus edge in a + * shape that is rare on its own terms, i.e. exactly the pre-#2912 fan-out for + * that pair and no worse. + */ +const CSHARP_PREDEFINED_TYPE_ALIASES: ReadonlyMap = new Map([ + ['bool', 'Boolean'], + ['byte', 'Byte'], + ['sbyte', 'SByte'], + ['char', 'Char'], + ['decimal', 'Decimal'], + ['double', 'Double'], + ['float', 'Single'], + ['int', 'Int32'], + ['uint', 'UInt32'], + ['long', 'Int64'], + ['ulong', 'UInt64'], + ['short', 'Int16'], + ['ushort', 'UInt16'], + ['nint', 'IntPtr'], + ['nuint', 'UIntPtr'], + ['object', 'Object'], + ['string', 'String'], +]); + +/** The BCL simple names the keywords alias. A spelling that reduces to one of + * these IS the predefined type; anything else that merely happens to sit in + * `System` is an ordinary type and keeps its qualifier. */ +const CSHARP_PREDEFINED_TYPE_NAMES: ReadonlySet = new Set( + CSHARP_PREDEFINED_TYPE_ALIASES.values(), +); + +const CSHARP_SYSTEM_QUALIFIER = /^(?:global::)?System\./; + +function normalizeCsharpTypeArgument(name: string): string { + const named = name.trim(); + // A keyword answers immediately: `string` → `String`. + const aliased = CSHARP_PREDEFINED_TYPE_ALIASES.get(named); + if (aliased !== undefined) return aliased; + // Otherwise the `System.` qualifier is dropped so the fully-qualified + // spelling of a predefined type meets that keyword: `System.String` → + // `String` ≡ `string` → `String`. The optional `global::` alias qualifier goes + // with it — `import-decomposer` already unwraps that spelling elsewhere, and + // leaving it on would make `global::System.String` unequal to `string` and + // prune a live implementor. + // + // ONLY when what remains is a predefined type. `System.Custom` is an ordinary + // type that happens to live in `System`, and answering `Custom` for it would + // equate it with an unrelated `Custom` elsewhere in the workspace. Returned as + // written instead, which sends it to the identity comparison — the step that + // can actually tell two declarations apart. + const bare = named.replace(CSHARP_SYSTEM_QUALIFIER, ''); + return bare !== named && CSHARP_PREDEFINED_TYPE_NAMES.has(bare) ? bare : named; +} + export { csharpScopeResolver }; diff --git a/gitnexus/src/core/ingestion/languages/dart/captures.ts b/gitnexus/src/core/ingestion/languages/dart/captures.ts index 351eaee7d..a6c5ef773 100644 --- a/gitnexus/src/core/ingestion/languages/dart/captures.ts +++ b/gitnexus/src/core/ingestion/languages/dart/captures.ts @@ -1069,9 +1069,15 @@ function emitHeritage(classNode: SyntaxNode, out: CaptureMatch[]): void { for (let i = 0; i < superclass.namedChildCount; i++) { const c = superclass.namedChild(i); if (c !== null && c.type === 'type_identifier') { + // `extends Base` spells the arguments in a SIBLING node, so the + // anchor's own text cannot carry them; the sub-tag does (#2912). + const args = typeArgumentsAfter(superclass, i); out.push({ '@reference.inherits': nodeToCapture('@reference.inherits', c), '@reference.name': nodeToCapture('@reference.name', c), + ...(args === null + ? {} + : { '@reference.type-arguments': nodeToCapture('@reference.type-arguments', args) }), }); break; } @@ -1144,7 +1150,26 @@ function emitHeritageMarkers( for (let i = 0; i < container.namedChildCount; i++) { const c = container.namedChild(i); if (c === null || c.type !== 'type_identifier') continue; - const payload = encodeMarker('heritage', [kind, c.text, className]); + // `implements Validator` / `with M`: the arguments ride the + // marker payload, because this heritage never becomes a reference SITE — + // `emitDartHeritageEdges` reads the marker and emits the edge (#2912). + // Dropped rather than encoded when the spelling contains the marker's own + // ':' delimiter, which `encodeMarker` rejects outright; absence is the + // fail-open value everywhere this is read. + const args = typeArgumentsAfter(container, i)?.text; + const fields = + args === undefined || args.includes(':') + ? [kind, c.text, className] + : [kind, c.text, className, args]; + const payload = encodeMarker('heritage', fields); out.push({ '@import.heritage': syntheticCapture('@import.heritage', c, payload) }); } } + +/** The `type_arguments` node written immediately after `container`'s named + * child at `index` — the arguments of the type that child names — or `null` + * when that type was written without any. */ +function typeArgumentsAfter(container: SyntaxNode, index: number): SyntaxNode | null { + const next = container.namedChild(index + 1); + return next !== null && next.type === 'type_arguments' ? next : null; +} diff --git a/gitnexus/src/core/ingestion/languages/dart/import-target.ts b/gitnexus/src/core/ingestion/languages/dart/import-target.ts index 443edfcbb..fd6c5224a 100644 --- a/gitnexus/src/core/ingestion/languages/dart/import-target.ts +++ b/gitnexus/src/core/ingestion/languages/dart/import-target.ts @@ -13,8 +13,57 @@ * `targetRaw` arrives already quote-stripped from `interpretDartImport`. */ +import { perFileSet } from '../../import-resolvers/per-file-set.js'; import { DART_HERITAGE_PREFIX } from './interpret.js'; +/** + * Basename → files carrying it, in `allFilePaths` iteration order, memoized on + * the Set's identity (#2879). + * + * Both resolution legs answered `fp === candidate || fp.endsWith('/' + candidate)` + * with a full workspace scan, and the `package:` leg ran one scan PER candidate + * — for an external package both candidates miss, so both scans always ran to + * completion. The orchestrator passes the same Set to every import in a pass, + * so the index is built once per run. + * + * Bucketing by basename is exact rather than a heuristic: a path satisfying + * either arm of the match ends with `candidate`, so its last `/`-delimited + * segment is `candidate`'s. Paths are indexed RAW, without slash normalization, + * because the scans this replaces compared raw paths too — normalizing here + * would start resolving backslash paths that previously returned null. + */ +interface DartFileIndex { + readonly byBasename: Map; +} + +const getDartFileIndex = perFileSet((allFilePaths: ReadonlySet): DartFileIndex => { + const byBasename = new Map(); + for (const fp of allFilePaths) { + const base = fp.slice(fp.lastIndexOf('/') + 1); + let bucket = byBasename.get(base); + if (bucket === undefined) { + bucket = []; + byBasename.set(base, bucket); + } + bucket.push(fp); + } + return { byBasename }; +}); + +/** First file (in Set-iteration order) that IS `candidate` or ends with + * `/` — the exact predicate of the scans this replaces. */ +function findByPathSuffix(allFilePaths: ReadonlySet, candidate: string): string | null { + const bucket = getDartFileIndex(allFilePaths).byBasename.get( + candidate.slice(candidate.lastIndexOf('/') + 1), + ); + if (bucket === undefined) return null; + const suffix = '/' + candidate; + for (const fp of bucket) { + if (fp === candidate || fp.endsWith(suffix)) return fp; + } + return null; +} + /** Resolve a relative path against the importer's directory, normalizing * `.`/`..` segments, then confirm it exists in the workspace file set. */ function resolveRelative( @@ -33,10 +82,7 @@ function resolveRelative( const target = parts.join('/'); if (allFilePaths.has(target)) return target; // Suffix fallback for absolute/rooted workspace paths. - for (const fp of allFilePaths) { - if (fp === target || fp.endsWith('/' + target)) return fp; - } - return null; + return findByPathSuffix(allFilePaths, target); } export function resolveDartImportTarget( @@ -56,10 +102,10 @@ export function resolveDartImportTarget( const slash = targetRaw.indexOf('/'); if (slash === -1) return null; const relPath = targetRaw.slice(slash + 1); + // Candidate priority is load-bearing: `lib/` before bare ``. for (const candidate of [`lib/${relPath}`, relPath]) { - for (const fp of allFilePaths) { - if (fp === candidate || fp.endsWith('/' + candidate)) return fp; - } + const hit = findByPathSuffix(allFilePaths, candidate); + if (hit !== null) return hit; } return null; // external package } diff --git a/gitnexus/src/core/ingestion/languages/dart/query.ts b/gitnexus/src/core/ingestion/languages/dart/query.ts index 38496f2ec..fb93f5fb8 100644 --- a/gitnexus/src/core/ingestion/languages/dart/query.ts +++ b/gitnexus/src/core/ingestion/languages/dart/query.ts @@ -42,7 +42,15 @@ const DART_SCOPE_QUERY = ` (enum_declaration) @scope.class ; ── Declarations — types ───────────────────────────────────────────────────── -(class_definition name: (identifier) @declaration.name) @declaration.class +; The type-parameter list is matched as an UNNAMED optional child: the Dart +; grammar hangs \`type_parameters\` off \`class_definition\` without a field name. +; Recording it is what lets instantiation-aware interface dispatch tell a type +; VARIABLE (\`class Box implements Validator\`) from a concrete argument +; (\`class V implements Validator\`) — see #2912; absent parameters are +; indistinguishable from a language that captures none, and read as unknown. +(class_definition + name: (identifier) @declaration.name + (type_parameters)? @declaration.type-parameters) @declaration.class (mixin_declaration (identifier) @declaration.name) @declaration.trait (extension_declaration name: (identifier) @declaration.name) @declaration.class (enum_declaration name: (identifier) @declaration.name) @declaration.enum diff --git a/gitnexus/src/core/ingestion/languages/dart/scope-resolver.ts b/gitnexus/src/core/ingestion/languages/dart/scope-resolver.ts index 22e1171d1..76bec5d74 100644 --- a/gitnexus/src/core/ingestion/languages/dart/scope-resolver.ts +++ b/gitnexus/src/core/ingestion/languages/dart/scope-resolver.ts @@ -38,6 +38,8 @@ import { generateId } from '../../../../lib/utils.js'; import { dartProvider } from '../dart.js'; import { dartArityCompatibility, dartMergeBindings, resolveDartImportTarget } from './index.js'; import { decodeMarker } from '../../utils/heritage-marker.js'; +import { typeApplicationArguments } from '../../utils/template-arguments.js'; +import type { HeritageTypeArgumentSink } from '../../scope-resolution/utils/generic-instantiation.js'; import { expandDartWildcardNames } from './expand-wildcards.js'; interface ClassDefRef { @@ -77,6 +79,7 @@ function emitDartHeritageEdges( graph: KnowledgeGraph, parsedFiles: readonly ParsedFile[], nodeLookup: GraphNodeLookup, + recordTypeArguments?: HeritageTypeArgumentSink, ): void { const defsByName = new Map(); for (const parsed of parsedFiles) { @@ -110,10 +113,19 @@ function emitDartHeritageEdges( if (decoded?.kind !== 'heritage') continue; const parts = decoded.fields; if (parts.length < 3) continue; - const [kind, baseName, childName] = parts; + const [kind, baseName, childName, rawTypeArguments] = parts; const childId = pickClassByName(childName!, parsed.filePath, defsByName); const baseId = pickClassByName(baseName!, parsed.filePath, defsByName); if (childId === undefined || baseId === undefined || childId === baseId) continue; + // The instantiation this clause was written with — `implements + // Validator` (#2912). Recorded before the dedup below, since the + // FIRST writer wins on both sides and an edge deduped here still needs + // its arguments. A marker from a pre-#2912 cache has no fourth field, + // which reads as unknown. + if (rawTypeArguments !== undefined) { + const typeArguments = typeApplicationArguments(rawTypeArguments); + if (typeArguments !== undefined) recordTypeArguments?.(childId, baseId, typeArguments); + } const key = `${childId}->${baseId}:${kind}`; if (emitted.has(key)) continue; emitted.add(key); @@ -211,8 +223,8 @@ export const dartScopeResolver: ScopeResolver = { // `implements` / `with` IMPLEMENTS edges (extends rides the generic // inherits pre-pass; these need an explicit, kind-independent edge type). - emitHeritageEdges: (graph, parsedFiles, nodeLookup) => - emitDartHeritageEdges(graph, parsedFiles, nodeLookup), + emitHeritageEdges: (graph, parsedFiles, nodeLookup, _scopes, recordTypeArguments) => + emitDartHeritageEdges(graph, parsedFiles, nodeLookup, recordTypeArguments), // Dart is statically typed — the field-fallback heuristic over-connects. fieldFallbackOnMethodLookup: false, diff --git a/gitnexus/src/core/ingestion/languages/go/generic-type-parameters.ts b/gitnexus/src/core/ingestion/languages/go/generic-type-parameters.ts new file mode 100644 index 000000000..d03e7b4e2 --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/go/generic-type-parameters.ts @@ -0,0 +1,157 @@ +import type { ParsedFile, Range, SymbolDefinition } from 'gitnexus-shared'; + +/** + * A generic Go interface's type-parameter names, in DECLARATION ORDER, stamped + * onto its `Interface` def as a Go-private sidecar. + * + * Same mechanism and lifecycle as `goReceiverKind` (method-owners.ts): an extra + * property on a def the Go resolver owns, written on the main thread and read by + * `interface-impls.ts`. It is deliberately NOT a shared `SymbolDefinition` field + * and deliberately NOT a capture — see {@link stampGoInterfaceTypeParameters}. + * + * ORDER IS THE POINT. Substitution is positional (`Repo[User]` binds the FIRST + * type parameter), so a set or a name→constraint map would lose exactly the + * information this exists to carry. + */ +type GoGenericInterfaceDefinition = SymbolDefinition & { + readonly goTypeParameters?: readonly string[]; +}; + +/** + * Stamp every generic interface in `parsedFiles` with its type-parameter names, + * read out of the declaration's own source text. + * + * WHY SOURCE TEXT AND NOT A CAPTURE. The tree has the list right there + * (`type_spec` carries a `type_parameters` field), and capturing it would be two + * lines. But captures run inside the PARSE WORKER, whose script is resolved from + * the compiled `dist/` build, and their output is additionally memoized by the + * parse cache and the durable ParsedFile store — so a capture-side change is + * invisible until a rebuild AND a cache-version bump, and silently wrong in + * between. Everything here runs on the main thread from data the pipeline + * already materialized, so it is correct on the first run and needs neither. + * + * The scan is exact rather than a grep over the file: an interface declaration + * owns a `Class` scope whose range spans exactly its `type_spec` + * (`Repo[T any] interface{ … }`), so the text is sliced by that range and the + * type parameters are, by grammar, whatever sits between the brackets that + * IMMEDIATELY follow the name. Comments and strings elsewhere in the file cannot + * reach it. + */ +export function stampGoInterfaceTypeParameters( + parsedFiles: readonly ParsedFile[], + fileContents: ReadonlyMap, +): void { + for (const parsed of parsedFiles) { + // Deferred so a file with no interface declaration never indexes its lines. + let lines: { readonly source: string; readonly starts: readonly number[] } | undefined; + for (const scope of parsed.scopes) { + if (scope.kind !== 'Class') continue; + const iface = scope.ownedDefs.find((def) => def.type === 'Interface'); + if (iface?.qualifiedName === undefined) continue; + if (lines === undefined) { + const source = fileContents.get(parsed.filePath); + if (source === undefined) break; + lines = { source, starts: buildLineStarts(source) }; + } + const declaration = sliceRange(lines.source, lines.starts, scope.range); + if (declaration === undefined) continue; + const names = goTypeParameterNames(declaration, simpleGoName(iface.qualifiedName)); + if (names === undefined) continue; + (iface as { goTypeParameters?: readonly string[] }).goTypeParameters = names; + } + } +} + +/** Read back a stamp, rejecting anything whose shape does not match — the + * sidecar is optional and a hand-built fixture def carries none. */ +export function readGoTypeParameters(def: SymbolDefinition): readonly string[] | undefined { + const names = (def as GoGenericInterfaceDefinition).goTypeParameters; + if (!Array.isArray(names) || names.length === 0) return undefined; + return names.every((name): name is string => typeof name === 'string') ? names : undefined; +} + +/** + * The declared type-parameter names of `Name[…] interface{…}`, in source order, + * or `undefined` when the declaration is not generic. + * + * Go spec, Type parameter declarations: the list is comma-separated and one + * entry may declare SEVERAL names sharing one constraint — `[K, V any]` declares + * `K` and `V`, and `[S ~[]E, E any]` declares `S` and `E`. Each entry therefore + * contributes exactly its FIRST token as a name; anything after it is the + * constraint, which is not needed here (satisfaction of a constraint is a + * separate question from implementation of an interface, and constraints are + * never harvested as instantiations — see `interface-impls.ts`). + */ +function goTypeParameterNames(declaration: string, interfaceName: string): string[] | undefined { + if (!declaration.startsWith(interfaceName)) return undefined; + if (declaration[interfaceName.length] !== '[') return undefined; + const close = matchingGoDelimiter(declaration, interfaceName.length); + if (close === -1) return undefined; + const names: string[] = []; + for (const entry of splitTopLevelGoList(declaration.slice(interfaceName.length + 1, close))) { + const name = /^[A-Za-z_][A-Za-z0-9_]*/.exec(entry)?.[0]; + if (name === undefined) return undefined; + names.push(name); + } + return names.length === 0 ? undefined : names; +} + +/** Index of the delimiter closing the one at `open`, or -1 when unbalanced. + * Tracks `[]`, `{}` and `()` together so an `interface{ M(a, b int) }` + * constraint cannot end the list early. */ +export function matchingGoDelimiter(text: string, open: number): number { + let depth = 0; + for (let i = open; i < text.length; i += 1) { + const ch = text[i]; + if (ch === '[' || ch === '{' || ch === '(') depth += 1; + else if (ch === ']' || ch === '}' || ch === ')') { + depth -= 1; + if (depth === 0) return i; + } + } + return -1; +} + +/** Split on commas that are not nested inside brackets, braces or parens. */ +export function splitTopLevelGoList(text: string): string[] { + const parts: string[] = []; + let depth = 0; + let start = 0; + for (let i = 0; i < text.length; i += 1) { + const ch = text[i]; + if (ch === '[' || ch === '{' || ch === '(') depth += 1; + else if (ch === ']' || ch === '}' || ch === ')') depth -= 1; + else if (ch === ',' && depth === 0) { + parts.push(text.slice(start, i)); + start = i + 1; + } + } + parts.push(text.slice(start)); + return parts.map((part) => part.trim()).filter((part) => part.length > 0); +} + +function simpleGoName(qualifiedName: string): string { + const dot = qualifiedName.lastIndexOf('.'); + return dot === -1 ? qualifiedName : qualifiedName.slice(dot + 1); +} + +/** Offsets at which each 1-based line begins. */ +function buildLineStarts(source: string): number[] { + const starts = [0, 0]; + for (let i = 0; i < source.length; i += 1) { + if (source[i] === '\n') starts.push(i + 1); + } + return starts; +} + +/** `Range` is 1-based on lines and 0-based on columns (`syntheticCapture`). */ +function sliceRange( + source: string, + lineStarts: readonly number[], + range: Range, +): string | undefined { + const start = lineStarts[range.startLine]; + const end = lineStarts[range.endLine]; + if (start === undefined || end === undefined) return undefined; + return source.slice(start + range.startCol, end + range.endCol); +} diff --git a/gitnexus/src/core/ingestion/languages/go/import-target.ts b/gitnexus/src/core/ingestion/languages/go/import-target.ts index 28e8113fd..847af0c09 100644 --- a/gitnexus/src/core/ingestion/languages/go/import-target.ts +++ b/gitnexus/src/core/ingestion/languages/go/import-target.ts @@ -1,4 +1,11 @@ import type { GoModuleConfig } from '../../language-config.js'; +import { + buildPackageDirIndex, + filesDirectlyInPkgDir, + sortedRootFiles, + type PackageDirIndex, +} from '../../import-resolvers/package-dir-index.js'; +import { perFileSet } from '../../import-resolvers/per-file-set.js'; /** * Resolve a Go import path to ALL .go files in the matching package directory. @@ -50,34 +57,35 @@ export function resolveGoImportTarget( return null; } +/** Go packages exclude `_test.go` files: they are a separate package. */ +function isGoPackageFile(normalized: string): boolean { + return normalized.endsWith('.go') && !normalized.endsWith('_test.go'); +} + +/** + * Package index over the file set, memoized on the Set's identity (#2877). + * + * Every leg above used to walk all of `allFilePaths`, and the GOPATH fallback + * walks once per path segment — so a single unresolved import (which is most of + * them: stdlib and third-party module paths run the whole cascade to completion + * before returning null) cost several full workspace scans, making resolution + * O(imports × files). + * + * The orchestrator hands the same Set to every import in a pass, so the index + * is built once per run. `resolveGoImportTarget` must therefore never copy the + * Set before this point — see `import-resolvers/workspace-file-index.ts`. + */ +const getGoPackageIndex = perFileSet( + (allFilePaths: ReadonlySet): PackageDirIndex => + buildPackageDirIndex(allFilePaths, isGoPackageFile), +); + function findRootPackageFiles(allFilePaths: ReadonlySet): string[] { - const result: string[] = []; - for (const raw of allFilePaths) { - const normalized = raw.replace(/\\/g, '/'); - if (normalized.includes('/')) continue; - if (!normalized.endsWith('.go') || normalized.endsWith('_test.go')) continue; - result.push(raw); - } - return result.sort(); + return sortedRootFiles(getGoPackageIndex(allFilePaths)); } function findAllFilesInPkgDir(allFilePaths: ReadonlySet, pkgPath: string): string[] { - const pkgDir = '/' + pkgPath + '/'; - const result: string[] = []; - for (const raw of allFilePaths) { - const normalized = '/' + raw.replace(/\\/g, '/'); - if (!normalized.includes(pkgDir)) continue; - if (!normalized.endsWith('.go') || normalized.endsWith('_test.go')) continue; - // Ensure file is directly in the package directory (not a subdirectory) - const afterPkg = normalized.substring(normalized.indexOf(pkgDir) + pkgDir.length); - if (!afterPkg.includes('/')) result.push(raw); - } - return result; -} - -/** Preserved for backward compat. */ -export interface GoResolveContext { - readonly fromFile: string; - readonly allFilePaths: ReadonlySet; - readonly goModule?: GoModuleConfig; + // Deliberately UNSORTED, unlike the root leg: the previous single-pass scan + // emitted in Set-iteration order and `filesDirectlyInPkgDir` reproduces it. + return filesDirectlyInPkgDir(getGoPackageIndex(allFilePaths), pkgPath); } diff --git a/gitnexus/src/core/ingestion/languages/go/index.ts b/gitnexus/src/core/ingestion/languages/go/index.ts index a71cba2fc..c9d5271c8 100644 --- a/gitnexus/src/core/ingestion/languages/go/index.ts +++ b/gitnexus/src/core/ingestion/languages/go/index.ts @@ -10,7 +10,7 @@ export { synthesizeGoTypeBindings } from './type-binding.js'; export { goArityCompatibility } from './arity.js'; export { goMergeBindings } from './merge-bindings.js'; export { goBindingScopeFor, goImportOwningScope, goReceiverBinding } from './simple-hooks.js'; -export { resolveGoImportTarget, type GoResolveContext } from './import-target.js'; +export { resolveGoImportTarget } from './import-target.js'; export { populateGoPackageSiblings } from './package-siblings.js'; export { populateGoRangeBindings } from './range-binding.js'; export { detectGoInterfaceImplementations } from './interface-impls.js'; diff --git a/gitnexus/src/core/ingestion/languages/go/interface-impls.ts b/gitnexus/src/core/ingestion/languages/go/interface-impls.ts index fad2ee8f1..913ff37f1 100644 --- a/gitnexus/src/core/ingestion/languages/go/interface-impls.ts +++ b/gitnexus/src/core/ingestion/languages/go/interface-impls.ts @@ -1,9 +1,18 @@ import type { ParsedFile, ReferenceSite, SymbolDefinition } from 'gitnexus-shared'; +import type { + StructuralImplementationResult, + UndecidedSatisfaction, +} from '../../scope-resolution/contract/scope-resolver.js'; import type { SemanticModel } from '../../model/semantic-model.js'; import type { ScopeResolutionIndexes } from '../../model/scope-resolution-indexes.js'; import { simpleQualifiedName } from '../../scope-resolution/graph-bridge/ids.js'; import { resolveInheritanceBaseInScope } from '../../scope-resolution/scope/walkers.js'; import { goPackageDir } from './package-clause.js'; +import { + matchingGoDelimiter, + readGoTypeParameters, + splitTopLevelGoList, +} from './generic-type-parameters.js'; type MethodSet = ReadonlyMap; type MutableMethodSet = Map; @@ -24,6 +33,23 @@ type EmbeddedParent = { readonly structId: string; readonly asPointer: boolean } type DualMethodSet = { readonly value: MutableMethodSet; readonly pointer: MutableMethodSet }; /** Which method set satisfied an interface. `value` implies pointer too. */ export type GoReceiverForm = 'value' | 'pointer'; +/** + * Whether a type satisfies an interface — or whether we could not tell. + * + * `undecided` is the state #2873 was missing. It means a required signature + * named something we could not give an identity to (a package qualifier with no + * recoverable import path), so the comparison was never actually performed. + * Folding it into `unsatisfied` is what let `impact()` answer a confident zero + * for a method that in fact had callers. + * + * It stays distinct from `unsatisfied` in exactly one direction: an undecided + * pair mints NO edge (a speculative one would fan out into fabricated CALLS), + * but it IS reported, so the answer downstream is a lower bound instead of a + * fact. Compare `go/types`, which folds the same case the other way — its + * `hasAllMethods` returns true for an invalid type — because a type checker's + * job is to avoid cascading errors, not to bound a blast radius. + */ +type Verdict = 'satisfied' | 'unsatisfied' | 'undecided'; /** One structural implementor plus the form in which it implements. */ export type GoStructuralImplementor = { readonly structDefId: string; @@ -31,20 +57,44 @@ export type GoStructuralImplementor = { }; type SignatureContext = { readonly packageQualifier: string | undefined; - readonly importQualifiers: ReadonlyMap; + /** Every token this file may write before a `.`, mapped to the package it + * names. Keyed on the token the SOURCE uses, which is the import's local name + * except where that had to be recovered from the path (`…/bar/v2` -> `bar`). + * An `undefined` value is a name that is claimed but has no agreeable + * qualifier; it reads the same as an absent key at the one consumer. */ + readonly importQualifiers: ReadonlyMap; }; type DetectionIndexes = { readonly interfaces: readonly SymbolDefinition[]; readonly structsById: ReadonlyMap; readonly methodsByOwner: ReadonlyMap; readonly effectiveMethodsByStructId: ReadonlyMap; - readonly interfaceById: ReadonlyMap; + /** Every interface in the program keyed by `qualifiedName`, `null` where more + * than one declares that name — the single probe behind + * {@link uniqueInterfaceNamed}. */ + readonly interfacesByQualifiedName: ReadonlyMap; readonly interfaceOwnMethodsById: ReadonlyMap; readonly embeddedSitesByInterfaceId: ReadonlyMap; readonly parentStructIdsByStructId: ReadonlyMap; readonly valueMethodsByStructId: ReadonlyMap; readonly structIdsByMethodName: ReadonlyMap>; readonly signatureContextByDefId: ReadonlyMap; + /** Type-parameter names, in declaration order, for every GENERIC interface. + * Absence means "not generic" and is the gate on the whole instantiation + * path — no entry, nothing below runs. */ + readonly typeParametersByInterfaceId: ReadonlyMap; + /** + * Every distinct instantiation of each generic interface observed anywhere in + * the program: interface id → the instantiation's normalized type ARGUMENTS, + * keyed by that list joined — which is what deduplicates it. + * + * An instantiation is nothing but that list. `Repo[User]` reduces to the type + * arguments ALREADY normalized in the signature context of the file that wrote + * them, so a cross-package `repo.Repo[model.User]` and the implementor's own + * `model.User` compare as the same type without either side re-qualifying the + * other's spelling. + */ + readonly instantiationsByInterfaceId: ReadonlyMap>; readonly scopeIndexes: ScopeResolutionIndexes; }; @@ -52,7 +102,7 @@ export function detectGoInterfaceImplementations( parsedFiles: readonly ParsedFile[], _indexes: ScopeResolutionIndexes, _model: SemanticModel, -): Map { +): StructuralImplementationResult { return detectGoInterfaceImplementationsFromIndexes(buildDetectionIndexes(parsedFiles, _indexes)); } @@ -73,14 +123,21 @@ function buildDetectionIndexes( const signatureContextByDefId = new Map(); const interfaceIdByScopeId = new Map(); const structIdByScopeId = new Map(); + const typeParametersByInterfaceId = new Map(); + const signatureContextByFilePath = new Map(); for (const parsed of parsedFiles) { const signatureContext = signatureContextForFile(parsed, indexes); + signatureContextByFilePath.set(parsed.filePath, signatureContext); for (const def of parsed.localDefs) { signatureContextByDefId.set(def.nodeId, signatureContext); if (def.type === 'Interface') { interfaces.push(def); interfaceById.set(def.nodeId, def); + const typeParameters = readGoTypeParameters(def); + if (typeParameters !== undefined) { + typeParametersByInterfaceId.set(def.nodeId, typeParameters); + } continue; } if (def.type === 'Struct') { @@ -130,6 +187,16 @@ function buildDetectionIndexes( } } + // Built from the nodeId-keyed map, so a def that appears in two ParsedFiles is + // one interface here just as it is one there — not a name collision with + // itself. + const interfacesByQualifiedName = new Map(); + for (const iface of interfaceById.values()) { + const name = iface.qualifiedName; + if (name === undefined || name.length === 0) continue; + interfacesByQualifiedName.set(name, interfacesByQualifiedName.has(name) ? null : iface); + } + for (const parsed of parsedFiles) { for (const scope of parsed.scopes) { const iface = scope.ownedDefs.find((def) => def.type === 'Interface'); @@ -216,49 +283,444 @@ function buildDetectionIndexes( structsById, methodsByOwner, effectiveMethodsByStructId, - interfaceById, + interfacesByQualifiedName, interfaceOwnMethodsById, embeddedSitesByInterfaceId, parentStructIdsByStructId, structIdsByMethodName, valueMethodsByStructId, signatureContextByDefId, + typeParametersByInterfaceId, + // Gated on the repo declaring at least one generic interface. A Go codebase + // with none — the overwhelming majority — never runs the harvest at all, + // which matters because the spellings it would scan (`[]byte`, + // `map[string]X`) are among the commonest types in the language. + instantiationsByInterfaceId: + typeParametersByInterfaceId.size === 0 + ? new Map() + : collectGoInstantiations( + parsedFiles, + signatureContextByFilePath, + typeParametersByInterfaceId, + interfacesByQualifiedName, + indexes, + ), scopeIndexes: indexes, }; } +/** + * Every distinct instantiation of a generic interface written anywhere in the + * program, resolved and deduplicated in one pass. + * + * Go records no instantiation anywhere on the DECLARATION — `Repo[User]` exists + * only where it is written — so the sites are the field/parameter/variable type + * spellings the capture layer already preserved: `TypeRef.declaredSpelling` + * (which keeps the arguments `rawName` drops), and the def-side `declaredType` / + * `parameterTypes` / `returnType`. + * + * A spelling is scanned rather than parsed as a whole, so decorated and nested + * forms yield their inner instantiations too: `[]Repo[User]`, `*Repo[User]` and + * `map[string]Repo[User]` all yield `Repo[User]`, and `Outer[Repo[User]]` yields + * both — each of which really is an instantiation present in the program. False + * bases (`map[` scans as base `map`) resolve to no interface and drop out. + */ +function collectGoInstantiations( + parsedFiles: readonly ParsedFile[], + signatureContextByFilePath: ReadonlyMap, + typeParametersByInterfaceId: ReadonlyMap, + interfacesByQualifiedName: ReadonlyMap, + indexes: ScopeResolutionIndexes, +): ReadonlyMap> { + // One map where there were two on the same key: the inner map's KEY is the + // joined argument list, so holding it is the deduplication. + const argsByInterfaceId = new Map>(); + // A base name resolves once per scope. The bracket gate below cannot filter + // Go's commonest types — `map[string]string` scans as base `map` — so the + // FALSE bases dominate this pass, and each one otherwise re-walks the whole + // scope chain for a name that will never bind. + const basesInScope = new Map(); + const resolveBase = (baseName: string, inScope: string): SymbolDefinition | undefined => { + // NUL-joined for the same reason `methodSetKey` is: it cannot occur in Go + // source, so no two (scope, name) pairs can collide on one key. + const key = `${inScope}\u0000${baseName}`; + const memo = basesInScope.get(key); + if (memo !== undefined) return memo ?? undefined; + const iface = resolveGoInstantiationBase(baseName, inScope, interfacesByQualifiedName, indexes); + basesInScope.set(key, iface ?? null); + return iface; + }; + const record = ( + spelling: string | undefined, + inScope: string, + context: SignatureContext, + ): void => { + // Cheap gate first: most Go type spellings have no bracket at all, and the + // scan below is the only per-spelling cost this pass adds. + if (spelling === undefined || !spelling.includes('[')) return; + for (const { baseName, rawArgs } of parseGoInstantiationSpellings(spelling)) { + const iface = resolveBase(baseName, inScope); + if (iface === undefined) continue; + const typeParameters = typeParametersByInterfaceId.get(iface.nodeId); + // A partial or over-long argument list is not a valid instantiation + // ("For a generic type, all type arguments must always be provided + // explicitly" — go.dev/ref/spec#Instantiations), so there is nothing to + // substitute and the site is dropped. + if (typeParameters === undefined || typeParameters.length !== rawArgs.length) continue; + const normalizedArgs = normalizeGoTypeArguments(rawArgs, context); + if (normalizedArgs === undefined) continue; + let byArgs = argsByInterfaceId.get(iface.nodeId); + if (byArgs === undefined) { + byArgs = new Map(); + argsByInterfaceId.set(iface.nodeId, byArgs); + } + const key = normalizedArgs.join(','); + if (!byArgs.has(key)) byArgs.set(key, normalizedArgs); + } + }; + + for (const parsed of parsedFiles) { + const context = signatureContextByFilePath.get(parsed.filePath); + if (context === undefined) continue; + for (const scope of parsed.scopes) { + for (const binding of scope.typeBindings.values()) { + record(binding.declaredSpelling ?? binding.rawName, scope.id, context); + } + for (const def of scope.ownedDefs) { + record(def.declaredType, scope.id, context); + record(def.returnType, scope.id, context); + for (const parameterType of def.parameterTypes ?? []) { + record(parameterType, scope.id, context); + } + } + } + } + return argsByInterfaceId; +} + +/** Every `Ident[…]` / `pkg.Ident[…]` application in a type spelling, with its + * top-level (comma-separated, delimiter-balanced) arguments. */ +function parseGoInstantiationSpellings( + spelling: string, +): Array<{ readonly baseName: string; readonly rawArgs: readonly string[] }> { + const out: Array<{ baseName: string; rawArgs: string[] }> = []; + const namePattern = /[A-Za-z_][A-Za-z0-9_.]*(?=\[)/g; + let match: RegExpExecArray | null; + while ((match = namePattern.exec(spelling)) !== null) { + const open = match.index + match[0].length; + const close = matchingGoDelimiter(spelling, open); + if (close === -1) continue; + const rawArgs = splitTopLevelGoList(spelling.slice(open + 1, close)); + if (rawArgs.length === 0) continue; + out.push({ baseName: match[0], rawArgs }); + } + return out; +} + +/** Normalize each type argument in the context of the file that WROTE it, or + * `undefined` when any of them carries an unresolvable import qualifier — a + * half-normalized argument list would compare against nothing meaningful. */ +function normalizeGoTypeArguments( + rawArgs: readonly string[], + context: SignatureContext, +): string[] | undefined { + const normalizedArgs: string[] = []; + for (const rawArg of rawArgs) { + const normalized = normalizeSignatureType(rawArg, context); + if (normalized === undefined) return undefined; + normalizedArgs.push(normalized); + } + return normalizedArgs; +} + +/** + * Bind an instantiation's base name to the generic interface it names. + * + * Goes through `resolveInheritanceBaseInScope` first — the same real scope + * resolution the embedded-interface path uses — and falls back to a globally + * UNIQUE name match. + */ +function resolveGoInstantiationBase( + baseName: string, + inScope: string, + interfacesByQualifiedName: ReadonlyMap, + indexes: ScopeResolutionIndexes, +): SymbolDefinition | undefined { + const bound = resolveInheritanceBaseInScope(inScope, simpleTypeName(baseName), indexes); + if (bound !== undefined) return bound.type === 'Interface' ? bound : undefined; + return uniqueInterfaceNamed(baseName, interfacesByQualifiedName); +} + +/** + * The one interface a written name denotes by NAME ALONE — the fallback both + * name-match routes here share, once the real scope resolution above them has + * declined. + * + * Ambiguity drops the site rather than guessing: two same-named interfaces in + * different packages would otherwise cross-pollinate each other's + * instantiations, and a dropped site only costs fan-out that does not exist + * today anyway. A qualified spelling is tried under both its own name and its + * simple tail, and a hit under EACH is two matches, so it declines as well. + * + * One probe rather than a scan of every interface in the program, which is what + * made this quadratic: the bracket gate in `collectGoInstantiations` cannot + * filter `map[…]` or `[]T`, so a Go program pays this once per bracketed + * spelling it writes. + */ +function uniqueInterfaceNamed( + name: string, + interfacesByQualifiedName: ReadonlyMap, +): SymbolDefinition | undefined { + const exact = interfacesByQualifiedName.get(name); + if (exact === null) return undefined; + const simpleName = simpleTypeName(name); + if (simpleName === name) return exact; + const simple = interfacesByQualifiedName.get(simpleName); + if (simple === null) return undefined; + if (exact !== undefined && simple !== undefined) return undefined; + return exact ?? simple; +} + function detectGoInterfaceImplementationsFromIndexes( indexes: DetectionIndexes, -): Map { +): StructuralImplementationResult { const implementations = new Map(); + const undecided: UndecidedSatisfaction[] = []; const methodSetCache = new Map(); for (const iface of indexes.interfaces) { const required = collectInterfaceMethodSet(iface, indexes, new Set(), methodSetCache); if (required === undefined || required.size === 0) continue; if (!methodSetHasVerifiableSignatures(required)) continue; - const implementors: GoStructuralImplementor[] = []; - for (const structId of candidateStructIdsFor(required, indexes)) { - const pointerSet = indexes.effectiveMethodsByStructId.get(structId); - if (pointerSet === undefined) continue; - // MS(*T) is the superset: if it does not satisfy, neither does MS(T). - if (!methodSetSatisfies(pointerSet, required, indexes.signatureContextByDefId)) continue; - // Then ask the narrower question separately — does the VALUE type satisfy? - // This is the distinction `var x I = T{}` turns on, and it is a fact about - // the program, not a heuristic. - const valueSet = indexes.valueMethodsByStructId.get(structId); - const satisfiesByValue = - valueSet !== undefined && - methodSetSatisfies(valueSet, required, indexes.signatureContextByDefId); - implementors.push({ - structDefId: structId, - receiverForm: satisfiesByValue ? 'value' : 'pointer', + // Hoisted, not recomputed per set: `substituteMethodSet` rewrites + // SIGNATURES and returns the identical key set, and `candidateStructIdsFor` + // keys off nothing but those method names — so every set below has exactly + // these candidates. Materialized because it is iterated once per + // instantiation and one of the branches behind it yields a live iterator. + const candidateStructIds = [...candidateStructIdsFor(required, indexes)]; + // The declaration's own method set, then one per observed instantiation. + // The declaration set runs FIRST and unconditionally, so this is strictly + // additive: every implementor found before #2855 is still found, in the + // same order, and instantiation only ever appends. + const formByStructId = new Map(); + const undecidedStructIds = new Set(); + for (const candidateSet of [required, ...instantiatedMethodSetsFor(iface, required, indexes)]) { + for (const structId of candidateStructIds) { + if (formByStructId.get(structId) === 'value') continue; + const pointerSet = indexes.effectiveMethodsByStructId.get(structId); + if (pointerSet === undefined) continue; + // MS(*T) is the superset: if it does not satisfy, neither does MS(T). + const verdict = methodSetSatisfies( + pointerSet, + candidateSet, + indexes.signatureContextByDefId, + ); + if (verdict !== 'satisfied') { + // `undecided` mints no edge — a speculative IMPLEMENTS would fan out + // into fabricated CALLS through `emitReceiverBoundCalls`. It is + // recorded instead, so `impact` can report a lower bound rather than + // a confident zero (#2873). A decided `unsatisfied` records nothing: + // that answer is trustworthy. + if (verdict === 'undecided') undecidedStructIds.add(structId); + continue; + } + // Then ask the narrower question separately — does the VALUE type satisfy? + // This is the distinction `var x I = T{}` turns on, and it is a fact about + // the program, not a heuristic. + const valueSet = indexes.valueMethodsByStructId.get(structId); + const satisfiesByValue = + valueSet !== undefined && + methodSetSatisfies(valueSet, candidateSet, indexes.signatureContextByDefId) === + 'satisfied'; + formByStructId.set(structId, satisfiesByValue ? 'value' : 'pointer'); + undecidedStructIds.delete(structId); + } + } + const implementors: GoStructuralImplementor[] = [...formByStructId].map( + ([structDefId, receiverForm]) => ({ structDefId, receiverForm }), + ); + if (implementors.length > 0) implementations.set(iface.nodeId, implementors); + if (undecidedStructIds.size > 0) { + const candidateNames: string[] = []; + for (const structId of undecidedStructIds) { + const name = indexes.structsById.get(structId)?.qualifiedName; + if (name !== undefined) candidateNames.push(name); + } + undecided.push({ + interfaceDefId: iface.nodeId, + interfaceName: iface.qualifiedName, + filePath: iface.filePath, + undecidedCandidates: undecidedStructIds.size, + candidateNames, }); } - if (implementors.length > 0) implementations.set(iface.nodeId, implementors); } - return implementations; + return { implementations, undecided }; +} + +/** + * The method set of each observed INSTANTIATION of a generic interface. + * + * Go spec, Instantiations: "A generic function or type is instantiated by + * substituting type arguments for the type parameters. … Each type argument is + * substituted for its corresponding type parameter in the generic declaration. … + * Instantiating a type results in a new non-generic named type." Combined with + * Type definitions ("Generic types must be instantiated when they are used") the + * consequence is that `Repo` is not a type at all and `Repo[User]` is — with + * method set `{ Save(x User) }` after substitution. Implementing an interface + * then asks whether a type "is an element of the type set of I", and Basic + * interfaces defines that type set as "the set of types which implement all of + * those methods". `UserRepo`, whose method set contains `Save(x User)`, is an + * element of `Repo[User]`'s type set — so it implements `Repo[User]`, and a call + * through a `Repo[User]`-typed field really can land on `UserRepo.Save`. Before + * this, it could not: the required parameter type stayed the type PARAMETER `T`, + * matched no implementor's `User`, and the interface got no IMPLEMENTS edge at + * all. That is the same false-silence shape as #2813/#2829, one abstraction up. + * + * SUBSTITUTION, NOT ERASURE. `Repo[Order]` instantiates to `Save(x Order)` and + * is NOT satisfied by a `Save(x User)` implementor. Treating `T` as a wildcard + * would satisfy both and mint an edge Go does not have; the whole point of + * #2829 was that an exact model beats an approximate one. + * + * WHAT THIS DELIBERATELY DOES NOT MODEL. GitNexus holds one node per generic + * DECLARATION, not one per instantiation, so an interface instantiated at two + * different arguments in the same program unions their implementors onto the one + * `Repo` node — `Repo[User]` and `Repo[Order]` in the same repo both fan out to + * every type satisfying either. That is the same one-node-per-declaration + * over-approximation every nominal language in the graph already carries (a + * Kotlin `class UserRepo : Repo` yields `UserRepo IMPLEMENTS Repo`, argument + * discarded), and it is bounded by the arguments the program actually writes — + * strictly narrower than erasure, which admits arguments that appear nowhere. + * + * Constraints are out of reach by construction and that is correct: a generic + * interface used as a CONSTRAINT (`func F[T Repo[X]](…)`) is written in a type + * parameter list, which produces no type binding and no declared type, so no + * such site is ever harvested. Non-basic interfaces — the union/type-set kind + * that "may only be used as type constraints" (General interfaces) — declare no + * methods and are already dropped by the empty-method-set guard above. + */ +function instantiatedMethodSetsFor( + iface: SymbolDefinition, + required: MethodSet, + indexes: DetectionIndexes, +): MethodSet[] { + const typeParameters = indexes.typeParametersByInterfaceId.get(iface.nodeId); + if (typeParameters === undefined) return []; + const instantiations = indexes.instantiationsByInterfaceId.get(iface.nodeId); + if (instantiations === undefined || instantiations.size === 0) return []; + const indexByName = new Map(typeParameters.map((name, index) => [name, index])); + const sets: MethodSet[] = []; + for (const normalizedArgs of instantiations.values()) { + const substituted = substituteMethodSet( + required, + indexByName, + normalizedArgs, + indexes.signatureContextByDefId, + ); + if (substituted !== undefined) sets.push(substituted); + } + return sets; +} + +/** + * Rewrite a required method set under one instantiation, or `undefined` when any + * signature in it cannot be normalized (an unresolved import qualifier) — a + * partially substituted set would compare a mix of instantiated and + * uninstantiated types, so the instantiation is dropped whole. + * + * The substituted defs carry a synthetic node id that is deliberately absent + * from `signatureContextByDefId`. Their parameter/return types come out of here + * ALREADY normalized — the type arguments in the context that WROTE them, the + * rest in the interface's own — and `normalizeSignatureType` with no context is + * the identity beyond whitespace, so the comparison in `signaturesCompatible` + * cannot re-qualify a spelling that is already fully qualified. + */ +function substituteMethodSet( + required: MethodSet, + indexByName: ReadonlyMap, + normalizedArgs: readonly string[], + signatureContextByDefId: ReadonlyMap, +): MutableMethodSet | undefined { + const out = new Map(); + for (const [name, overloads] of required) { + const substitutedOverloads: SymbolDefinition[] = []; + for (const def of overloads) { + const context = signatureContextByDefId.get(def.nodeId); + const parameterTypes: string[] = []; + for (const parameterType of def.parameterTypes ?? []) { + const substituted = substituteSignatureType( + parameterType, + indexByName, + normalizedArgs, + context, + ); + if (substituted === undefined) return undefined; + parameterTypes.push(substituted); + } + let returnType: string | undefined; + if (def.returnType !== undefined) { + returnType = substituteSignatureType(def.returnType, indexByName, normalizedArgs, context); + if (returnType === undefined) return undefined; + } + substitutedOverloads.push({ + ...def, + nodeId: `${def.nodeId}\u0000instantiated`, + ...(def.parameterTypes !== undefined ? { parameterTypes } : {}), + ...(returnType !== undefined ? { returnType } : {}), + }); + } + out.set(name, substitutedOverloads); + } + return out; +} + +/** + * Placeholder for the type argument at position `i` while its enclosing type is + * normalized. NUL-delimited — the same separator `methodSetKey` already uses, + * and for the same reason: it cannot occur in Go source. + * + * Both halves of that choice are load-bearing. `qualifyGoSignatureTypes` rewrites + * only tokens matching `[A-Za-z_][A-Za-z0-9_]*`, which can start with neither NUL + * nor a digit, so the placeholder survives normalization untouched; and + * `normalizeSignatureType` strips `\s+` FIRST, so a whitespace-delimited + * placeholder would lose its delimiters and become indistinguishable from an + * array length (`[5]int`). + */ +const TYPE_PARAMETER_PLACEHOLDER = /\u0000(\d+)\u0000/g; + +/** + * Substitute type arguments into one signature type, preserving Go's type + * identity rules for everything around them. + * + * Substitution happens BEFORE normalization and reinstatement AFTER, so the + * argument's own spelling is never re-qualified by the interface's package while + * the rest of the type still is: `[]T` in package `repo` with argument + * `internal/model.User` yields `[]internal/model.User`, not + * `[]repo.internal/model.User`. Pointer, slice, map and variadic shape survive + * because only the identifier token is replaced (`*T` -> `*model.User`), which is + * what makes `Save(x T)` and `Save(x *T)` stay different methods. + */ +function substituteSignatureType( + typeName: string, + indexByName: ReadonlyMap, + normalizedArgs: readonly string[], + context: SignatureContext | undefined, +): string | undefined { + const placeheld = typeName.replace( + /[A-Za-z_][A-Za-z0-9_]*/g, + (token, offset: number, source: string) => { + // `pkg.T` names `T` in package `pkg`, never the type parameter `T`. + if (hasPackageQualifierDot(source, offset)) return token; + const index = indexByName.get(token); + return index === undefined ? token : `\u0000${index}\u0000`; + }, + ); + const normalized = normalizeSignatureType(placeheld, context); + if (normalized === undefined) return undefined; + return normalized.replace(TYPE_PARAMETER_PLACEHOLDER, (_match, digits: string) => { + return normalizedArgs[Number(digits)] ?? _match; + }); } /** @@ -529,15 +991,7 @@ function resolveEmbeddedInterface( ): SymbolDefinition | undefined { const bound = resolveInheritanceBaseInScope(site.inScope, site.name, indexes.scopeIndexes); if (bound !== undefined) return bound.type === 'Interface' ? bound : undefined; - - const simpleName = simpleTypeName(site.name); - const matches: SymbolDefinition[] = []; - for (const iface of indexes.interfaceById.values()) { - if (iface.qualifiedName === site.name || iface.qualifiedName === simpleName) { - matches.push(iface); - } - } - return matches.length === 1 ? matches[0] : undefined; + return uniqueInterfaceNamed(site.name, indexes.interfacesByQualifiedName); } function simpleTypeName(name: string): string { @@ -606,36 +1060,54 @@ function methodSetSatisfies( actual: MethodSet, required: MethodSet, signatureContextByDefId: ReadonlyMap, -): boolean { +): Verdict { + let undecided = false; for (const [name, requiredOverloads] of required) { const actualOverloads = actual.get(name); - if (actualOverloads === undefined) return false; + if (actualOverloads === undefined) return 'unsatisfied'; for (const requiredMethod of requiredOverloads) { // Fast arity pre-filter: if the required method has a known parameter // count, reject immediately when no actual overload matches it. This // avoids the expensive signature normalization loop for obvious mismatches. if (requiredMethod.parameterCount !== undefined) { if (!actualOverloads.some((a) => a.parameterCount === requiredMethod.parameterCount)) { - return false; + return 'unsatisfied'; } } - if (!hasCompatibleMethod(actualOverloads, requiredMethod, signatureContextByDefId)) { - return false; - } + const verdict = compatibleMethodVerdict( + actualOverloads, + requiredMethod, + signatureContextByDefId, + ); + // A decided mismatch anywhere ends it — a type that provably lacks ONE + // required method does not implement the interface, however many other + // methods we could not read. Undecided keeps scanning for exactly that + // reason: a hard no may still be waiting, and it is the better answer. + if (verdict === 'unsatisfied') return 'unsatisfied'; + if (verdict === 'undecided') undecided = true; } } - return true; + return undecided ? 'undecided' : 'satisfied'; } -function hasCompatibleMethod( +function compatibleMethodVerdict( actualOverloads: readonly SymbolDefinition[], requiredMethod: SymbolDefinition, signatureContextByDefId: ReadonlyMap, -): boolean { - if (!hasVerifiableSignature(requiredMethod)) return false; - return actualOverloads.some((actualMethod) => - signaturesCompatible(actualMethod, requiredMethod, signatureContextByDefId), - ); +): Verdict { + // Nothing in the interface's own method to compare against: this is missing + // information, not a difference. It was a `false` before #2873. + if (!hasVerifiableSignature(requiredMethod)) return 'undecided'; + let undecided = false; + for (const actualMethod of actualOverloads) { + const verdict = signaturesCompatible(actualMethod, requiredMethod, signatureContextByDefId); + // One overload that provably matches settles the method — the unknowns on + // the others cannot unsettle it. (Pyright does the same: a resolvable path + // suppresses the partially-unknown diagnostic from the others.) + if (verdict === 'satisfied') return 'satisfied'; + if (verdict === 'undecided') undecided = true; + } + return undecided ? 'undecided' : 'unsatisfied'; } function methodSetHasVerifiableSignatures(methods: MethodSet): boolean { @@ -658,61 +1130,115 @@ function signaturesCompatible( actual: SymbolDefinition, required: SymbolDefinition, signatureContextByDefId: ReadonlyMap, -): boolean { +): Verdict { const actualContext = signatureContextByDefId.get(actual.nodeId); const requiredContext = signatureContextByDefId.get(required.nodeId); - return ( - countsCompatible(actual.parameterCount, required.parameterCount) && - countsCompatible(actual.requiredParameterCount, required.requiredParameterCount) && - parameterTypesCompatible( - actual.parameterTypes, - required.parameterTypes, - actualContext, - requiredContext, - ) && - returnTypesCompatible(actual.returnType, required.returnType, actualContext, requiredContext) + if ( + !countsCompatible(actual.parameterCount, required.parameterCount) || + !countsCompatible(actual.requiredParameterCount, required.requiredParameterCount) + ) { + return 'unsatisfied'; + } + // A decided mismatch beats an unknown — it is the answer we can stand behind — + // so the parameter verdict only short-circuits when it is `unsatisfied`. + const parameters = parameterTypesVerdict(actual, required, actualContext, requiredContext); + if (parameters === 'unsatisfied') return 'unsatisfied'; + const returns = returnTypeVerdict( + actual.returnType, + required.returnType, + actualContext, + requiredContext, ); + if (returns === 'unsatisfied') return 'unsatisfied'; + return parameters === 'undecided' || returns === 'undecided' ? 'undecided' : 'satisfied'; } function countsCompatible(actual: number | undefined, required: number | undefined): boolean { return actual === undefined || required === undefined || actual === required; } -function parameterTypesCompatible( - actual: readonly string[] | undefined, - required: readonly string[] | undefined, +function parameterTypesVerdict( + actualDef: SymbolDefinition, + requiredDef: SymbolDefinition, actualContext: SignatureContext | undefined, requiredContext: SignatureContext | undefined, -): boolean { - if (actual === undefined || required === undefined) return true; - if (actual.length !== required.length) return false; - return actual.every((type, index) => { - const actualType = normalizeSignatureType(type, actualContext); - const requiredType = normalizeSignatureType(required[index]!, requiredContext); - return actualType !== undefined && requiredType !== undefined && actualType === requiredType; - }); +): Verdict { + const actual = actualDef.parameterTypes; + const required = requiredDef.parameterTypes; + if (actual === undefined || required === undefined) { + // A method that takes nothing has no list to carry — that is a decided + // agreement, not a gap, and it is the shape of every `Close() error`. + if (actualDef.parameterCount === 0 && requiredDef.parameterCount === 0) return 'satisfied'; + // Otherwise the types really are unread. This is where the old code assumed + // `true` and called two signatures compatible without comparing them. + return 'undecided'; + } + if (actual.length !== required.length) return 'unsatisfied'; + // Indexed loop, not `entries()`: this is the innermost comparison in the + // detection pass and runs once per parameter per candidate pair. + let undecided = false; + for (let index = 0; index < actual.length; index++) { + const verdict = typeVerdict(actual[index]!, required[index]!, actualContext, requiredContext); + if (verdict === 'unsatisfied') return 'unsatisfied'; + if (verdict === 'undecided') undecided = true; + } + return undecided ? 'undecided' : 'satisfied'; } -function returnTypesCompatible( +function returnTypeVerdict( actual: string | undefined, required: string | undefined, actualContext: SignatureContext | undefined, requiredContext: SignatureContext | undefined, -): boolean { - if (required === undefined) return actual === undefined; - if (actual === undefined) return false; +): Verdict { + if (required === undefined) return actual === undefined ? 'satisfied' : 'unsatisfied'; + if (actual === undefined) return 'unsatisfied'; + return typeVerdict(actual, required, actualContext, requiredContext); +} + +/** The one place a type spelling decides anything, and the only mint site of + * `undecided` below the method level: `normalizeSignatureType` returns + * `undefined` when a package qualifier has no identity we could recover, and + * two spellings we could not normalize are not thereby different. */ +function typeVerdict( + actual: string, + required: string, + actualContext: SignatureContext | undefined, + requiredContext: SignatureContext | undefined, +): Verdict { const actualType = normalizeSignatureType(actual, actualContext); const requiredType = normalizeSignatureType(required, requiredContext); - return actualType !== undefined && requiredType !== undefined && actualType === requiredType; + if (actualType === undefined || requiredType === undefined) return 'undecided'; + return actualType === requiredType ? 'satisfied' : 'unsatisfied'; } +/** Normalized form of every type spelling seen in a file, keyed by its context. + * + * Normalization is a pure function of (spelling, context), and a context is + * immutable once built — so this is a cache, not state. It earns its keep + * because #2873 removed the early bail: a parameter list that named an + * out-of-repo type used to normalize to `undefined` and stop the comparison at + * parameter 0, and now every pair of (interface method, candidate struct) walks + * its whole signature, re-normalizing both sides once per candidate. */ +const normalizedTypesByContext = new WeakMap>(); + function normalizeSignatureType(typeName: string, context?: SignatureContext): string | undefined { // Go type identity includes pointer/slice/map/variadic shape and package // qualifiers. Only erase whitespace and qualify bare local type names; stripping // `*`, `[]`, `...`, or `pkg.` would make non-identical signatures compare equal. const compact = typeName.replace(/\s+/g, ''); if (context === undefined) return compact; - return qualifyGoSignatureTypes(compact, context); + let normalized = normalizedTypesByContext.get(context); + if (normalized === undefined) { + normalized = new Map(); + normalizedTypesByContext.set(context, normalized); + } + // `has`, not a truthiness check: `undefined` — "no agreeable identity" — is + // itself a result worth caching, and it is the one this file mints most. + if (normalized.has(compact)) return normalized.get(compact); + const qualified = qualifyGoSignatureTypes(compact, context); + normalized.set(compact, qualified); + return qualified; } function qualifyGoSignatureTypes(typeName: string, context: SignatureContext): string | undefined { @@ -740,12 +1266,30 @@ function signatureContextForFile( parsed: ParsedFile, indexes: ScopeResolutionIndexes, ): SignatureContext { - const importQualifiers = new Map(); + // A key present with an `undefined` value means "in-repo, but no directory to + // name it by" — a repo-ROOT package, whose own file spells its types bare, so + // no qualifier either side can agree on exists. It still has to occupy the + // name, or the fallback below would label a repo package external. + const importQualifiers = new Map(); const importEdges = indexes.imports?.get(parsed.moduleScope) ?? []; for (const edge of importEdges) { if (edge.kind !== 'namespace' || edge.targetFile === null) continue; - const qualifier = packageQualifierForFile(edge.targetFile); - if (qualifier !== undefined) importQualifiers.set(edge.localName, qualifier); + importQualifiers.set(edge.localName, packageQualifierForFile(edge.targetFile)); + } + // An import that resolves to no file in the repository — every stdlib and + // third-party package — still has an identity: its import path, which the + // parsed directive kept even though the finalized `ImportEdge` did not (#2873). + // Without this fallback `ctx context.Context` normalized to `undefined`, and + // `undefined` reads as "signatures differ" on both sides at once, so two + // textually identical methods compared unequal and Go interface satisfaction + // only ever succeeded for builtin-only signatures. + for (const directive of parsed.parsedImports) { + if (directive.kind !== 'namespace') continue; + const token = goImportToken(directive.localName, directive.targetRaw); + // The edges ran first, so a token an in-repo import already claimed keeps + // its package directory — the fallback fills gaps, it does not compete. + if (importQualifiers.has(token)) continue; + importQualifiers.set(token, externalPackageQualifier(directive.targetRaw)); } return { packageQualifier: packageQualifierForFile(parsed.filePath), @@ -753,6 +1297,39 @@ function signatureContextForFile( }; } +/** Identity for a package that lives outside the repository. + * + * The import path is the exact identity — `net/http` and `example.com/x/http` + * are different packages that both spell their qualifier `http`, so keying on + * the local name would make them compare equal. The prefix keeps the result in + * a namespace no in-repo qualifier can reach: no package directory can begin + * with the literal `extern:`. */ +function externalPackageQualifier(importPath: string): string { + return `extern:${importPath}`; +} + +/** The token Go source writes before the `.` for this import. + * + * An alias names its own token, so it is returned as-is. An unaliased import + * arrives here spelled as the last path segment, which is right until a module + * carries a major version: `github.com/foo/bar/v2` is written `bar` and + * `gopkg.in/yaml.v3` is written `yaml`, per the rule the go tool applies. + * + * Deriving "aliased" from the path rather than from `importedName` is + * deliberate — the Go extractor sets both names to the alias when there is one + * (`import-decomposer.ts`), so the two fields never disagree. + * + * A package whose name diverges from its path for any OTHER reason cannot be + * recovered without reading the dependency's own source, which is by definition + * outside the repository. Those stay unresolved, which is the safe direction. */ +function goImportToken(localName: string, importPath: string): string { + const segments = importPath.split('/').filter((segment) => segment.length > 0); + const leaf = segments.pop() ?? importPath; + if (localName !== leaf) return localName; + const name = /^v\d+$/.test(leaf) ? (segments.pop() ?? leaf) : leaf; + return name.replace(/\.v\d+$/, ''); +} + /** The package directory, or `undefined` for a repo-root file. * * Shares `goPackageDir` with the package-clause resolver rather than repeating diff --git a/gitnexus/src/core/ingestion/languages/go/method-owners.ts b/gitnexus/src/core/ingestion/languages/go/method-owners.ts index 329c94447..d32e4d79b 100644 --- a/gitnexus/src/core/ingestion/languages/go/method-owners.ts +++ b/gitnexus/src/core/ingestion/languages/go/method-owners.ts @@ -3,6 +3,7 @@ import { logger } from '../../../logger.js'; import { isClassLike, populateClassOwnedMembers } from '../../scope-resolution/scope/walkers.js'; import { goPackageDir, inferGoPackageName } from './package-clause.js'; +import { stampGoInterfaceTypeParameters } from './generic-type-parameters.js'; /** Bound on the sample of no-package-clause paths named in the warning. */ const SKIPPED_SAMPLE_CAP = 5; @@ -32,6 +33,13 @@ export function populateGoWorkspaceOwners( parsedFiles: readonly ParsedFile[], ctx: { readonly fileContents: ReadonlyMap }, ): void { + // Generic interfaces get their type-parameter list stamped here rather than at + // capture time, because this is the first main-thread hook that sees BOTH the + // parsed scopes and the file text (it already reads `fileContents` for the + // package clause). `detectGoInterfaceImplementations` runs later in the same + // pass and is the only reader. See `stampGoInterfaceTypeParameters`. + stampGoInterfaceTypeParameters(parsedFiles, ctx.fileContents); + const filesByPackage = new Map(); // A file with no resolvable package clause is dropped from ownership // resolution entirely — its methods never attach to a struct declared in a diff --git a/gitnexus/src/core/ingestion/languages/java.ts b/gitnexus/src/core/ingestion/languages/java.ts index 047a87791..317a853ac 100644 --- a/gitnexus/src/core/ingestion/languages/java.ts +++ b/gitnexus/src/core/ingestion/languages/java.ts @@ -23,14 +23,16 @@ import { createCallExtractor } from '../call-extractors/generic.js'; import { javaCallConfig } from '../call-extractors/configs/jvm.js'; import { createFieldExtractor } from '../field-extractors/generic.js'; import { javaConfig } from '../field-extractors/configs/jvm.js'; -import { createMethodExtractor } from '../method-extractors/generic.js'; -import { javaMethodConfig } from '../method-extractors/configs/jvm.js'; import { createVariableExtractor } from '../variable-extractors/generic.js'; import { javaVariableConfig } from '../variable-extractors/configs/jvm.js'; import { createJavaCfgVisitor } from '../cfg/visitors/java.js'; import { assertCloneable } from '../workers/clone-safety.js'; import { collectJavaCaptureSideChannel } from './java/capture-side-channel.js'; import type { SymbolDefinition } from 'gitnexus-shared'; +import { + javaRecordMethodExtractor, + shouldSkipJavaRecordComponentDefinition, +} from './java/record-components.js'; import { emitJavaScopeCaptures, interpretJavaImport, @@ -186,7 +188,8 @@ export const javaProvider = defineLanguage({ mroStrategy: 'implements-split', callExtractor: createCallExtractor(javaCallConfig), fieldExtractor: createFieldExtractor(javaConfig), - methodExtractor: createMethodExtractor(javaMethodConfig), + methodExtractor: javaRecordMethodExtractor, + shouldSkipDefinitionCapture: shouldSkipJavaRecordComponentDefinition, variableExtractor: createVariableExtractor(javaVariableConfig), classExtractor: createClassExtractor(javaClassConfig), diff --git a/gitnexus/src/core/ingestion/languages/java/analysis-features.ts b/gitnexus/src/core/ingestion/languages/java/analysis-features.ts index b85616602..73969670d 100644 --- a/gitnexus/src/core/ingestion/languages/java/analysis-features.ts +++ b/gitnexus/src/core/ingestion/languages/java/analysis-features.ts @@ -17,3 +17,17 @@ export const SPRING_CONFIG_BINDINGS_FEATURE: AnalysisFeatureDescriptor = { (filePath) => filePath.toLowerCase().endsWith('.java') || isSpringApplicationConfig(filePath), ), }; + +/** Durable completeness contract for implicit Java record-component accessors. */ +export const JAVA_RECORD_COMPONENT_ACCESSORS_FEATURE: AnalysisFeatureDescriptor = { + id: 'java.record-component-accessors', + version: 1, + appliesTo: (filePaths) => filePaths.some((filePath) => filePath.toLowerCase().endsWith('.java')), +}; + +/** Durable completeness contract for Java heritage captures. */ +export const JAVA_ENUM_INTERFACE_HERITAGE_FEATURE: AnalysisFeatureDescriptor = { + id: 'java.heritage-captures', + version: 1, + appliesTo: (filePaths) => filePaths.some((filePath) => filePath.toLowerCase().endsWith('.java')), +}; diff --git a/gitnexus/src/core/ingestion/languages/java/capture-side-channel.ts b/gitnexus/src/core/ingestion/languages/java/capture-side-channel.ts index 7a7f82685..91a910fa5 100644 --- a/gitnexus/src/core/ingestion/languages/java/capture-side-channel.ts +++ b/gitnexus/src/core/ingestion/languages/java/capture-side-channel.ts @@ -13,6 +13,7 @@ import type { JavaSpringConfigConsumerFact } from './spring-config-bindings.js'; import type { JavaSpringAopFact } from './spring-aop.js'; import type { JavaSpringConditionalFact } from './spring-conditionals.js'; import type { JavaSpringDiClassFact } from './spring-di.js'; +import type { JavaSpringNonHttpHandlerFact } from './spring-non-http-handlers.js'; export type JavaClassAnnotationFact = ClassAnnotationFact; @@ -24,6 +25,7 @@ export interface JavaCaptureSideChannel { readonly springConfigConsumers?: readonly JavaSpringConfigConsumerFact[]; readonly springConditionalFacts?: readonly JavaSpringConditionalFact[]; readonly springDiFacts?: readonly JavaSpringDiClassFact[]; + readonly springNonHttpHandlerFacts?: readonly JavaSpringNonHttpHandlerFact[]; } const classAnnotations = createClassAnnotationFactStore(); @@ -31,6 +33,7 @@ const springAopFacts = new Map(); const springConfigConsumers = new Map(); const springConditionalFacts = new Map(); const springDiFacts = new Map(); +const springNonHttpHandlerFacts = new Map(); /** Clear facts retained by a prior workspace pass in a long-lived process. */ export function clearJavaClassAnnotationFacts(): void { @@ -39,6 +42,7 @@ export function clearJavaClassAnnotationFacts(): void { springConfigConsumers.clear(); springConditionalFacts.clear(); springDiFacts.clear(); + springNonHttpHandlerFacts.clear(); } export function setJavaSpringAopFacts(filePath: string, facts: readonly JavaSpringAopFact[]): void { @@ -98,6 +102,20 @@ export function getJavaSpringDiFacts(filePath: string): readonly JavaSpringDiCla return springDiFacts.get(filePath) ?? []; } +export function setJavaSpringNonHttpHandlerFacts( + filePath: string, + facts: readonly JavaSpringNonHttpHandlerFact[], +): void { + if (facts.length === 0) springNonHttpHandlerFacts.delete(filePath); + else springNonHttpHandlerFacts.set(filePath, facts); +} + +export function getJavaSpringNonHttpHandlerFacts( + filePath: string, +): readonly JavaSpringNonHttpHandlerFact[] { + return springNonHttpHandlerFacts.get(filePath) ?? []; +} + /** Snapshot worker-local Java annotation facts for ParsedFile serialization. */ export function collectJavaCaptureSideChannel( filePath: string, @@ -107,6 +125,7 @@ export function collectJavaCaptureSideChannel( const configConsumers = springConfigConsumers.get(filePath) ?? []; const conditionFacts = springConditionalFacts.get(filePath) ?? []; const diFacts = springDiFacts.get(filePath) ?? []; + const nonHttpHandlerFacts = springNonHttpHandlerFacts.get(filePath) ?? []; const packageFact = getJavaPackageFact(filePath); if ( facts.length === 0 && @@ -114,6 +133,7 @@ export function collectJavaCaptureSideChannel( configConsumers.length === 0 && conditionFacts.length === 0 && diFacts.length === 0 && + nonHttpHandlerFacts.length === 0 && packageFact === undefined ) { return undefined; @@ -126,6 +146,7 @@ export function collectJavaCaptureSideChannel( ...(configConsumers.length > 0 ? { springConfigConsumers: configConsumers } : {}), ...(conditionFacts.length > 0 ? { springConditionalFacts: conditionFacts } : {}), ...(diFacts.length > 0 ? { springDiFacts: diFacts } : {}), + ...(nonHttpHandlerFacts.length > 0 ? { springNonHttpHandlerFacts: nonHttpHandlerFacts } : {}), }; } @@ -148,6 +169,7 @@ export function applyJavaCaptureSideChannel(parsed: ParsedFile): void { setJavaSpringConfigConsumerFacts(parsed.filePath, []); setJavaSpringConditionalFacts(parsed.filePath, []); setJavaSpringDiFacts(parsed.filePath, []); + setJavaSpringNonHttpHandlerFacts(parsed.filePath, []); setJavaPackageFact(parsed.filePath, UNKNOWN_JVM_PACKAGE_FACT); return; } @@ -168,6 +190,10 @@ export function applyJavaCaptureSideChannel(parsed: ParsedFile): void { parsed.filePath, Array.isArray(data.springDiFacts) ? data.springDiFacts : [], ); + setJavaSpringNonHttpHandlerFacts( + parsed.filePath, + Array.isArray(data.springNonHttpHandlerFacts) ? data.springNonHttpHandlerFacts : [], + ); setJavaPackageFact( parsed.filePath, isJvmPackageFact(data.packageFact) ? data.packageFact : UNKNOWN_JVM_PACKAGE_FACT, diff --git a/gitnexus/src/core/ingestion/languages/java/captures.ts b/gitnexus/src/core/ingestion/languages/java/captures.ts index 60d315df5..ce5ed4b93 100644 --- a/gitnexus/src/core/ingestion/languages/java/captures.ts +++ b/gitnexus/src/core/ingestion/languages/java/captures.ts @@ -39,6 +39,7 @@ import { setJavaSpringConfigConsumerFacts, setJavaSpringConditionalFacts, setJavaSpringDiFacts, + setJavaSpringNonHttpHandlerFacts, } from './capture-side-channel.js'; import { captureJavaPackageFact } from './package-facts.js'; import { synthesizeCallableFlowCaptures } from '../../utils/callable-flow-captures.js'; @@ -50,6 +51,11 @@ import { captureJavaSpringConditionalFacts, type JavaSpringConditionalFact, } from './spring-conditionals.js'; +import { + captureJavaSpringNonHttpHandlerFacts, + type JavaSpringNonHttpHandlerFact, +} from './spring-non-http-handlers.js'; +import { synthesizeJavaRecordComponentAccessorCaptures } from './record-components.js'; /** Declaration anchors that carry function-like arity metadata. */ const FUNCTION_DECL_TAGS = ['@declaration.method', '@declaration.constructor'] as const; @@ -138,6 +144,7 @@ export function emitJavaScopeCaptures( const springAopTypeNodeIds = new Set(); const springConditionalFacts: JavaSpringConditionalFact[] = []; const springDiFacts: JavaSpringDiClassFact[] = []; + const springNonHttpHandlerFacts: JavaSpringNonHttpHandlerFact[] = []; const springDiClassNodeIds = new Set(); for (const m of rawMatches) { @@ -173,6 +180,9 @@ export function emitJavaScopeCaptures( springConditionalFacts.push( ...captureJavaSpringConditionalFacts(springDiClassNode, filePath), ); + springNonHttpHandlerFacts.push( + ...captureJavaSpringNonHttpHandlerFacts(springDiClassNode, filePath), + ); const fact = captureJavaSpringDiClassFact(springDiClassNode, filePath); if (fact !== null) springDiFacts.push(fact); } @@ -391,12 +401,14 @@ export function emitJavaScopeCaptures( setJavaSpringAopFacts(filePath, springAopFacts); setJavaSpringConditionalFacts(filePath, springConditionalFacts); setJavaSpringDiFacts(filePath, springDiFacts); + setJavaSpringNonHttpHandlerFacts(filePath, springNonHttpHandlerFacts); return [ ...resolveVarTypeBindings(out), ...synthesizeJavaInheritanceReferences(tree.rootNode), ...synthesizeJavaExplicitConstructorReferences(tree.rootNode), ...synthesizeJavaAnonymousClassDeclarations(tree.rootNode), + ...synthesizeJavaRecordComponentAccessorCaptures(tree.rootNode), ...synthesizeCallableFlowCaptures(tree.rootNode, JAVA_CALLABLE_CAPTURE_OPTIONS), ]; } @@ -424,6 +436,7 @@ function synthesizeJavaAnonymousClassDeclarations(rootNode: SyntaxNode): Capture out.push({ '@declaration.class': nodeToCapture('@declaration.class', body), '@declaration.name': syntheticCapture('@declaration.name', body, identity.name), + '@declaration.is-synthetic': syntheticCapture('@declaration.is-synthetic', body, 'true'), }); // Inheritance: the anonymous class extends/implements its constructed @@ -485,6 +498,11 @@ function synthesizeJavaAnonymousClassDeclarations(rootNode: SyntaxNode): Capture out.push({ '@declaration.class': nodeToCapture('@declaration.class', bodyNode), '@declaration.name': syntheticCapture('@declaration.name', bodyNode, bodiedIdentity.name), + '@declaration.is-synthetic': syntheticCapture( + '@declaration.is-synthetic', + bodyNode, + 'true', + ), }); if (hostEnum !== undefined) { out.push({ @@ -631,40 +649,29 @@ function findEnclosingTypeDeclaration(node: SyntaxNode): SyntaxNode | null { } /** - * Synthesize `@reference.inherits` captures from Java class heritage so the - * registry-primary scope-resolution path emits EXTENDS / IMPLEMENTS edges - * (mirrors C++ `emitCppInheritanceCaptures`). Without this, Java inheritance - * edges came only from the legacy heritage-capture leg (removed in #942), which - * is dropped for registry-primary languages in the worker pipeline (issue #1951). + * Synthesize `@reference.inherits` captures from Java type heritage for the + * authoritative registry-primary EXTENDS / IMPLEMENTS pre-pass (mirrors C++ + * `emitCppInheritanceCaptures`). * * Scope covers `class_declaration` (`superclass` extends + `interfaces` - * implements clauses) AND `interface_declaration` (`extends_interfaces` → - * interface-to-interface EXTENDS), matching the legacy Java heritage query - * (tree-sitter-queries.ts), which has a dedicated `interface_declaration - * (extends_interfaces (type_list …))` arm. Without the interface arm the - * registry-primary synth silently dropped every `interface IA extends IB` - * edge while the legacy leg emitted it — the exact =0/=N parity break #1951 - * targets. Enum/record heritage stays unemitted (no legacy arm). Generic - * bases (`extends Box`, `implements IFoo`) ARE emitted here: the legacy - * heritage query was widened to capture the inner `type_identifier` of a - * `generic_type` (tree-sitter-queries.ts), so both paths now agree on SIMPLE - * (unqualified) generic bases — the more-correct behavior, consistent with - * C#/Rust (#1951). Qualified bases (`a.b.Base`, `a.b.Box`, `a.b.IFoo`) are - * ALSO now at parity (#1956 tri-review U2): the synth resolves them by their - * `scoped_type_identifier` tail, and the legacy heritage query was widened - * with matching `scoped_type_identifier` arms (plain + generic-wrapped). The + * implements clauses), `record_declaration` and `enum_declaration` + * (`interfaces` implements clauses), and `interface_declaration` + * (`extends_interfaces` clauses). Interface + * inheritance was restored for registry-primary resolution in #1951. Record + * graph nodes became canonical link targets in #2801 / PR #2871, so their + * `implements` clauses must participate for interface dispatch (#2900). + * Enums use the same tree-sitter `interfaces` field and participate as + * class-like `Enum` graph nodes (#2918). + * + * Generic bases (`extends Box`, `implements IFoo`) and qualified bases + * (`a.b.Base`, `a.b.Box`, `a.b.IFoo`) are normalized to their simple + * lookup-name tails, consistent with C#/Rust and the V1 binding contract. The * EXTENDS-vs-IMPLEMENTS split is decided downstream from the resolved target's * symbol kind (`preEmitInheritanceEdges`): a superclass resolves to a class * (EXTENDS), an implemented interface resolves to an interface (IMPLEMENTS). * An `interface IA extends IB` base resolves to an Interface too, so it is - * emitted as IMPLEMENTS — matching the legacy `interface_declaration` arm, - * which tagged the bases as implements (`kind: 'implements'`) and likewise - * resolves them as interfaces. The synth therefore does not need to know the - * declaration's own kind; it only emits inherits sites and lets the resolved - * target decide the edge type. - * Base names are normalized to their bare simple identifier (`Box` → `Box`, - * `java.io.Serializable` → `Serializable`) to match the V1 simple-name - * `findClassBindingInScope` contract. + * emitted as IMPLEMENTS. The synth only emits inheritance sites and lets the + * resolved target decide the edge type. */ function synthesizeJavaInheritanceReferences(root: SyntaxNode): CaptureMatch[] { const out: CaptureMatch[] = []; @@ -676,6 +683,14 @@ function synthesizeJavaInheritanceReferences(root: SyntaxNode): CaptureMatch[] { if (superclass !== null) { for (const base of superclass.namedChildren) emitJavaInheritanceBase(out, base); } + } + if ( + node.type === 'class_declaration' || + node.type === 'record_declaration' || + node.type === 'enum_declaration' + ) { + // Records and enums cannot declare a superclass; all three declarations + // expose implemented interfaces through the same tree-sitter field. const interfaces = node.childForFieldName('interfaces'); if (interfaces !== null) { for (const typeList of interfaces.namedChildren) { @@ -733,15 +748,22 @@ function javaBaseSimpleNameOf(typeNode: SyntaxNode): string | undefined { function javaBaseLookupNameNode(node: SyntaxNode): SyntaxNode | null { switch (node.type) { case 'type_identifier': - return node; - case 'scoped_type_identifier': + return node.isMissing || node.text.length === 0 ? null : node; + case 'scoped_type_identifier': { // `java.io.Serializable` → trailing `type_identifier` (`Serializable`). - return node.lastNamedChild; + const tail = node.lastNamedChild; + return tail === null ? null : javaBaseLookupNameNode(tail); + } case 'generic_type': { // `Box` → recurse into the base type (`Box`). const first = node.firstNamedChild; return first === null ? null : javaBaseLookupNameNode(first); } + case 'annotated_type': { + // The final named child is the base type; preceding children are annotations. + const type = node.lastNamedChild; + return type === null ? null : javaBaseLookupNameNode(type); + } default: return null; } diff --git a/gitnexus/src/core/ingestion/languages/java/import-target.ts b/gitnexus/src/core/ingestion/languages/java/import-target.ts index b78b6369b..4843e854d 100644 --- a/gitnexus/src/core/ingestion/languages/java/import-target.ts +++ b/gitnexus/src/core/ingestion/languages/java/import-target.ts @@ -1,26 +1,104 @@ /** - * Adapter from `(ParsedImport, WorkspaceIndex)` → concrete file path. + * Adapter from `(ParsedImport, WorkspaceIndex)` → the file(s) an import names. * - * Converts Java package paths (dots → slashes) and tries: - * 1. Exact file match: `com/example/User.java` - * 2. Suffix match for nested layouts - * 3. Directory match (wildcard imports) - * 4. Progressive prefix stripping for non-standard layouts + * Delegates to `module-resolution.ts`, which resolves a Java import the way + * Java defines it: a fully-qualified type name looked up against the packages + * the workspace's files DECLARE. * - * Returns `null` for unresolvable / JDK imports. + * ## What #2953 replaced, and why path shape could not work + * + * This resolver used to turn dots into slashes and hunt for a file whose path + * ended that way — exact whole path, then any segment-suffix, then the first + * `.java` directly inside a matching directory — retrying the whole cascade + * with each leading segment stripped. Four legs, all describing where a file + * SITS rather than what it DECLARES. + * + * Path shape is a convention, so it mostly worked, and failed hardest on the + * case that matters: it had no way to tell an import of something outside the + * repository from one inside it. `java.util.List` became `util/List`, then + * `List`, and bound to any `List.java` anywhere in the tree — a fabricated + * IMPORTS edge at full confidence, for an import naming a JDK class. Every JDK + * and third-party import in a repository was a candidate. + * + * Two secondary defects went with it, both consequences of resolving by shape: + * + * - a wildcard `import com.example.*;` answered with ONE arbitrary file — the + * first `.java` in the package directory in `allFilePaths` iteration order, + * which the previous header documented at length as being decided by "a + * property of the file list, not of the import". It now answers with every + * file declaring that package, which is what the import actually names. + * - a file's location and its package were assumed to agree. They need not: + * `weird/path/User.java` declaring `package com.example;` is importable as + * `com.example.User`, and `com/example/User.java` declaring nothing is in + * the default package and importable as nothing at all. Both now resolve + * correctly, because the declaration is what is read. + * + * The package declaration was already being extracted during the parse pass and + * has been reachable here through `getJavaPackageFact` the whole time; nothing + * read it. So this costs no new I/O — no `pom.xml`, no `build.gradle`, no + * source-root inference. The workspace describes itself. */ -import type { ParsedImport, WorkspaceIndex } from 'gitnexus-shared'; +import type { ParsedFile, ParsedImport, WorkspaceIndex } from 'gitnexus-shared'; +import { perFileSet } from '../../import-resolvers/per-file-set.js'; +import { getJavaPackageFact } from './package-facts.js'; +import { + buildJavaPackageIndex, + resolveJavaModule, + type JavaPackageIndex, +} from './module-resolution.js'; export interface JavaResolveContext { readonly fromFile: string; readonly allFilePaths: ReadonlySet; + /** + * The pass's parsed Java files — the only input this resolver needs, because + * the package index is built from their declarations. + * + * Absent means "no workspace was supplied", not "the workspace declares + * nothing": the index would be empty and every import would answer `null`. + * The orchestrator always supplies it (`scope-resolution/pipeline/run.ts` + * threads `context.parsedFiles`), and it must be passed THROUGH rather than + * copied — the memo below keys on the array's identity. + */ + readonly parsedFiles?: readonly ParsedFile[]; } +/** + * The package index, built once per pass and read by every import. + * + * Keyed on the `parsedFiles` array the orchestrator already threads through the + * pass, like PHP's `filesByDirectory` and Python's `parsedFileByPath`. The + * instrument that can see this memo fail counts element reads on that array — + * `countedParsedFiles` in `test/helpers/counting-file-set.ts`, asserted for + * every language by `import-target-index-reuse.contract.test.ts`. + */ +const getJavaPackageIndex = perFileSet( + (parsedFiles: readonly ParsedFile[]): JavaPackageIndex => + buildJavaPackageIndex(parsedFiles, getJavaPackageFact), +); + export function resolveJavaImportTarget( parsedImport: ParsedImport, workspaceIndex: WorkspaceIndex, -): string | null { +): string | readonly string[] | null { + const ctx = narrowContext(workspaceIndex); + if (ctx === null) return null; + if (parsedImport.kind === 'dynamic-unresolved') return null; + if (parsedImport.targetRaw === null || parsedImport.targetRaw === '') return null; + + const parsedFiles = ctx.parsedFiles; + if (parsedFiles === undefined || parsedFiles.length === 0) return null; + + return resolveJavaModule(parsedImport.targetRaw, getJavaPackageIndex(parsedFiles)); +} + +/** + * `WorkspaceIndex` is an opaque `unknown` placeholder in the shared contract; + * the orchestrator hands us a `JavaResolveContext`-shaped object. Narrow + * structurally rather than via a cast chain so unexpected shapes fail cleanly. + */ +function narrowContext(workspaceIndex: WorkspaceIndex): JavaResolveContext | null { const ctx = workspaceIndex as JavaResolveContext | undefined; if ( ctx === undefined || @@ -29,80 +107,5 @@ export function resolveJavaImportTarget( ) { return null; } - if (parsedImport.kind === 'dynamic-unresolved') return null; - if (parsedImport.targetRaw === null || parsedImport.targetRaw === '') return null; - - // Strip trailing `.*` for wildcard imports: `com.example.*` → `com.example` - let target = parsedImport.targetRaw; - if (target.endsWith('.*')) { - target = target.slice(0, -2); - } - - // Package path: `com.example.User` → `com/example/User` - const pathLike = target.replace(/\./g, '/'); - const suffix = `/${pathLike}`; - - let exactFile: string | null = null; - let suffixFile: string | null = null; - let directoryChild: string | null = null; - const dirPrefix = `${pathLike}/`; - const suffixDirPrefix = `/${dirPrefix}`; - - for (const raw of ctx.allFilePaths) { - const f = raw.replace(/\\/g, '/'); - if (!f.endsWith('.java')) continue; - if (f === `${pathLike}.java`) { - exactFile = raw; - break; - } - if (suffixFile === null && f.endsWith(`${suffix}.java`)) { - suffixFile = raw; - } - if (directoryChild === null) { - const atRoot = f.startsWith(dirPrefix); - const atNested = f.includes(suffixDirPrefix); - if (atRoot || atNested) { - const idx = atRoot ? 0 : f.indexOf(suffixDirPrefix) + 1; - const after = f.slice(idx + dirPrefix.length); - if (after.length > 0 && !after.includes('/')) { - directoryChild = raw; - } - } - } - } - - if (exactFile !== null) return exactFile; - if (suffixFile !== null) return suffixFile; - if (directoryChild !== null) return directoryChild; - - // Progressive prefix stripping — handles `import com.example.User;` - // in a repo laid out `User.java` (no `com/example/` prefix). - const segments = pathLike.split('/').filter(Boolean); - for (let skip = 1; skip < segments.length; skip++) { - const tail = segments.slice(skip).join('/'); - if (tail === '') continue; - const tailFile = `${tail}.java`; - const tailSuffix = `/${tailFile}`; - const tailDir = `${tail}/`; - const tailSuffixDir = `/${tailDir}`; - let tailDirectChild: string | null = null; - for (const raw of ctx.allFilePaths) { - const f = raw.replace(/\\/g, '/'); - if (!f.endsWith('.java')) continue; - if (f === tailFile) return raw; - if (f.endsWith(tailSuffix)) return raw; - if (tailDirectChild === null) { - const atRoot = f.startsWith(tailDir); - const atNested = f.includes(tailSuffixDir); - if (atRoot || atNested) { - const idx = atRoot ? 0 : f.indexOf(tailSuffixDir) + 1; - const after = f.slice(idx + tailDir.length); - if (after.length > 0 && !after.includes('/')) tailDirectChild = raw; - } - } - } - if (tailDirectChild !== null) return tailDirectChild; - } - - return null; + return ctx; } diff --git a/gitnexus/src/core/ingestion/languages/java/interpret.ts b/gitnexus/src/core/ingestion/languages/java/interpret.ts index 38d6128d4..721edbff8 100644 --- a/gitnexus/src/core/ingestion/languages/java/interpret.ts +++ b/gitnexus/src/core/ingestion/languages/java/interpret.ts @@ -57,10 +57,11 @@ export function interpretJavaImport(captures: CaptureMatch): ParsedImport | null // `import static com.example.Utils.*;` // The source is the class path (e.g. `com.example.Utils`). // Resolution should target the class file, not a wildcard directory - // scan — `Utils.java` is the file that contains the static members. + // scan — keeping the type path unstarred also distinguishes it from a + // package wildcard when a same-named package exists. return { kind: 'wildcard', - targetRaw: sourceCap.text + '.*', + targetRaw: sourceCap.text, }; } default: diff --git a/gitnexus/src/core/ingestion/languages/java/module-resolution.ts b/gitnexus/src/core/ingestion/languages/java/module-resolution.ts new file mode 100644 index 000000000..9eea50d74 --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/java/module-resolution.ts @@ -0,0 +1,167 @@ +/** + * Java import resolution against DECLARED packages (#2953). + * + * A Java import is a fully-qualified type name, not a path. `com.example.model.User` + * names the type `User` in the package `com.example.model`, and what places a + * file in that package is its own `package` declaration — not where it sits on + * disk. A file at `weird/path/User.java` declaring `package com.example.model;` + * IS `com.example.model.User`; a file at `com/example/model/User.java` declaring + * nothing is in the DEFAULT package and cannot be imported at all. + * + * The previous resolver worked the other way round: it turned dots into slashes + * and looked for a file whose path ended that way, retrying with each leading + * segment stripped. Path shape is a convention, so that mostly worked — and + * failed in the one case that matters most, because it could not tell an import + * of something outside the repository from one inside it. `java.util.List` + * became `util/List`, then `List`, and bound to any `List.java` in the tree. + * Every JDK and third-party import in a repo was a candidate for a fabricated + * IMPORTS edge at full confidence. + * + * The fix needs no new I/O. Every Java file's `package` declaration is already + * extracted during the parse pass and available here through + * `getJavaPackageFact` — the resolver simply never read it. So resolution + * becomes a lookup in an index the workspace already knows how to describe: + * + * `com.example.model.User` -> package `com.example.model` declares `User` + * `java.util.List` -> no file declares package `java.util` -> null + * + * `null` for the second is the complete and correct answer: the JDK is not in + * this repository, so there is no in-repo file the import could name. + */ + +import type { ParsedFile } from 'gitnexus-shared'; +import type { JvmPackageFact } from '../jvm/package-facts.js'; + +export interface JavaPackageIndex { + /** Declared package -> importable type name -> the file declaring it. */ + readonly typesByPackage: ReadonlyMap>; + /** Declared package -> every file declaring it, for wildcard imports. */ + readonly filesByPackage: ReadonlyMap; + /** + * Files whose `package` header could not be read (a malformed header — see + * `extractJvmPackageFact`). They are in no package, so nothing can import + * them; counted so the gap is observable rather than silent. + */ + readonly unreadablePackageFiles: number; +} + +const EMPTY_INDEX: JavaPackageIndex = { + typesByPackage: new Map(), + filesByPackage: new Map(), + unreadablePackageFiles: 0, +}; + +/** + * Index the workspace by what each file DECLARES. + * + * The importable type name is the file's base name, which is not a convention + * being relied on but the rule the language enforces: a type importable from + * another package must be `public`, and a public type must live in a file named + * after it. Additional package-private top-level types in the same file are + * deliberately not indexed — they are unimportable from elsewhere, so an import + * naming one is not a resolution this should find. + */ +export function buildJavaPackageIndex( + parsedFiles: readonly ParsedFile[], + packageOf: (filePath: string) => JvmPackageFact | undefined, +): JavaPackageIndex { + if (parsedFiles.length === 0) return EMPTY_INDEX; + + const typesByPackage = new Map>(); + const filesByPackage = new Map(); + let unreadablePackageFiles = 0; + + for (const parsed of parsedFiles) { + const filePath = parsed.filePath; + const fact = packageOf(filePath); + if (fact === undefined) continue; + if (fact.status !== 'known') { + unreadablePackageFiles++; + continue; + } + // The default package (`''`) is indexed like any other so a workspace of + // package-less files still answers its own wildcards, but Java forbids + // importing FROM it, which `resolveJavaModule` enforces rather than + // pretending here that the entry does not exist. + const packageName = fact.packageName; + + const typeName = baseTypeName(filePath); + if (typeName !== null) { + let types = typesByPackage.get(packageName); + if (types === undefined) { + types = new Map(); + typesByPackage.set(packageName, types); + } + // First declaration wins. Two files claiming the same package+type is not + // legal Java; picking either is as correct as the input allows. + if (!types.has(typeName)) types.set(typeName, filePath); + } + + const files = filesByPackage.get(packageName); + if (files === undefined) filesByPackage.set(packageName, [filePath]); + else files.push(filePath); + } + + return { typesByPackage, filesByPackage, unreadablePackageFiles }; +} + +/** + * Resolve one import specifier to the file(s) it names, or `null`. + * + * A wildcard answers with every file in the package; a type import answers with + * one file. Anything the workspace does not declare answers `null`. + */ +export function resolveJavaModule( + targetRaw: string, + index: JavaPackageIndex, +): string | readonly string[] | null { + if (targetRaw === '') return null; + + if (targetRaw.endsWith('.*')) { + const stem = targetRaw.slice(0, -2); + const inPackage = index.filesByPackage.get(stem); + // Only package wildcards retain `.*`. Static wildcards are interpreted with + // the owning type path so a same-named package cannot capture the import. + return inPackage !== undefined && stem !== '' ? inPackage : null; + } + + return resolveTypeName(targetRaw, index); +} + +/** + * Split a qualified name into the longest DECLARED package prefix and the type + * that follows it. + * + * Longest-first is what makes both of these land correctly without a rule about + * capitalization, which Java does not actually enforce: + * + * `com.example.model.User` -> package `com.example.model`, type `User` + * `com.example.Utils.method` -> package `com.example`, type `Utils` + * + * The second is a static member import; its trailing segments name members + * inside the type, and the file the import binds to is the type's. + */ +function resolveTypeName(qualified: string, index: JavaPackageIndex): string | null { + const parts = qualified.split('.').filter((part) => part !== ''); + // A single bare segment names a type in the default package, which Java + // forbids importing. Nothing to resolve, and nothing to guess at. + if (parts.length < 2) return null; + + for (let split = parts.length - 1; split >= 1; split--) { + const packageName = parts.slice(0, split).join('.'); + const types = index.typesByPackage.get(packageName); + if (types === undefined) continue; + const file = types.get(parts[split]); + if (file !== undefined) return file; + } + return null; +} + +/** `src/main/java/com/example/User.java` -> `User`. */ +function baseTypeName(filePath: string): string | null { + const slash = filePath.replace(/\\/g, '/').lastIndexOf('/'); + const base = slash === -1 ? filePath : filePath.slice(slash + 1); + if (!base.endsWith('.java')) return null; + const name = base.slice(0, -'.java'.length); + return name === '' ? null : name; +} diff --git a/gitnexus/src/core/ingestion/languages/java/query.ts b/gitnexus/src/core/ingestion/languages/java/query.ts index b51daa4e9..507d3ce79 100644 --- a/gitnexus/src/core/ingestion/languages/java/query.ts +++ b/gitnexus/src/core/ingestion/languages/java/query.ts @@ -58,17 +58,23 @@ const JAVA_SCOPE_QUERY = ` (compact_constructor_declaration) @scope.function ;; Declarations — types +;; Optional-quantifier capture rather than a second pattern: a separate rule +;; would make every GENERIC declaration match twice under one def id, leaving +;; match order to decide which twin kept the parameters. (class_declaration - name: (identifier) @declaration.name) @declaration.class + name: (identifier) @declaration.name + type_parameters: (type_parameters)? @declaration.type-parameters) @declaration.class (interface_declaration - name: (identifier) @declaration.name) @declaration.interface + name: (identifier) @declaration.name + type_parameters: (type_parameters)? @declaration.type-parameters) @declaration.interface (enum_declaration name: (identifier) @declaration.name) @declaration.enum (record_declaration - name: (identifier) @declaration.name) @declaration.record + name: (identifier) @declaration.name + type_parameters: (type_parameters)? @declaration.type-parameters) @declaration.record (annotation_type_declaration name: (identifier) @declaration.name) @declaration.class @@ -83,7 +89,13 @@ const JAVA_SCOPE_QUERY = ` ])) @class-annotation.class ;; Declarations — methods / constructors +;; +;; A generic METHOD's parameters are read for the same reason a generic type's +;; are (#2912 review): \` boolean runAny(Validator v)\` writes a receiver +;; whose argument is a type VARIABLE, and a pass that cannot tell that from a +;; concrete type prunes every implementor from the call's dispatch fan-out. (method_declaration + type_parameters: (type_parameters)? @declaration.type-parameters name: (identifier) @declaration.name) @declaration.method (constructor_declaration diff --git a/gitnexus/src/core/ingestion/languages/java/record-components.ts b/gitnexus/src/core/ingestion/languages/java/record-components.ts new file mode 100644 index 000000000..51053510c --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/java/record-components.ts @@ -0,0 +1,233 @@ +import { SupportedLanguages, type CaptureMatch } from 'gitnexus-shared'; +import type { CaptureMap } from '../../language-provider.js'; +import { createMethodExtractor } from '../../method-extractors/generic.js'; +import { javaMethodConfig } from '../../method-extractors/configs/jvm.js'; +import { extractAnnotations } from '../../field-extractors/configs/helpers.js'; +import type { + ExtractedMethods, + MethodExtractor, + MethodExtractorContext, + MethodInfo, +} from '../../method-types.js'; +import { nodeToCapture, syntheticCapture, type SyntaxNode } from '../../utils/ast-helpers.js'; + +const javaExplicitMethodExtractor = createMethodExtractor(javaMethodConfig); + +function recordComponents(recordNode: SyntaxNode): SyntaxNode[] { + const parameters = recordNode.childForFieldName('parameters'); + if (parameters === null) return []; + return parameters.namedChildren.filter( + (node): node is SyntaxNode => + node !== null && (node.type === 'formal_parameter' || node.type === 'spread_parameter'), + ); +} + +/** + * A record component is named by a real `identifier` and nothing else. + * + * Two node shapes reach this that are not one, and both would mint a graph node + * for source that does not compile: + * + * - `record M(int x, y) {}` — a dropped type. tree-sitter recovers by + * synthesizing `name: (MISSING identifier)`, a zero-width node whose text is + * `''`. It still satisfies the query's `name: (identifier)`, so testing the + * node TYPE alone does not reject it. + * - `record R(int _) {}` — the grammar declares both `formal_parameter.name` + * and `variable_declarator.name` as `identifier | underscore_pattern`, and + * `_` parses with no error at all. `_` is illegal as a component name, and + * admitting it here while the query rejects it is what let the structure and + * scope paths disagree. + * + * Same degenerate-node shape as `javaBaseLookupNameNode` in captures.ts (#2935). + */ +function isRecordComponentName(node: SyntaxNode | null | undefined): node is SyntaxNode { + return ( + node !== null && + node !== undefined && + node.type === 'identifier' && + !node.isMissing && + node.text.length > 0 + ); +} + +function recordComponentNameNode(component: SyntaxNode): SyntaxNode | null { + const name = + component.type === 'formal_parameter' + ? component.childForFieldName('name') + : (component.namedChildren + .find((node) => node?.type === 'variable_declarator') + ?.childForFieldName('name') ?? null); + return isRecordComponentName(name) ? name : null; +} + +/** + * Memoised per record node. `shouldSkipJavaRecordComponentDefinition` is called + * once per component capture, so recomputing this would rescan the whole record + * body per component — O(components x body members) for a single record. The + * scope-capture path hoists the call out of its own loop instead; this cache is + * what gives the structure path the same cost. Keyed weakly on the AST node, so + * it drops with the tree at the end of the file's parse. + */ +const explicitZeroArgAccessorNamesCache = new WeakMap>(); + +function explicitZeroArgAccessorNames(recordNode: SyntaxNode): Set { + const memoized = explicitZeroArgAccessorNamesCache.get(recordNode); + if (memoized !== undefined) return memoized; + const names = computeExplicitZeroArgAccessorNames(recordNode); + explicitZeroArgAccessorNamesCache.set(recordNode, names); + return names; +} + +function computeExplicitZeroArgAccessorNames(recordNode: SyntaxNode): Set { + const names = new Set(); + const body = recordNode.childForFieldName('body'); + if (body === null) return names; + + for (const node of body.namedChildren) { + if (node === null || node.type !== 'method_declaration') continue; + const name = node.childForFieldName('name')?.text; + const parameters = node.childForFieldName('parameters'); + const parameterCount = + parameters?.namedChildren.filter( + (parameter) => + parameter !== null && + (parameter.type === 'formal_parameter' || parameter.type === 'spread_parameter'), + ).length ?? 0; + if (name !== undefined && parameterCount === 0) names.add(name); + } + return names; +} + +function recordComponentReturnType(component: SyntaxNode): string | null { + const typeNode = + component.childForFieldName('type') ?? + (component.type === 'spread_parameter' + ? component.namedChildren.find( + (node) => node?.type !== 'modifiers' && node?.type !== 'variable_declarator', + ) + : undefined); + const type = typeNode?.text; + if (type === undefined) return null; + return component.type === 'spread_parameter' ? `${type}[]` : type; +} + +function implicitAccessorInfo( + component: SyntaxNode, + context: MethodExtractorContext, +): MethodInfo | null { + const name = recordComponentNameNode(component)?.text; + if (name === undefined) return null; + + return { + name, + receiverType: null, + returnType: recordComponentReturnType(component), + parameters: [], + visibility: 'public', + isStatic: false, + isAbstract: false, + isFinal: false, + // JLS 8.10.3 / 9.7.4: a component annotation reaches the generated accessor + // when its @Target admits METHOD (or TYPE_USE, in the return-type position). + // ponytail: over-approximate — we propagate every component annotation, + // because @Target lives in another file and parsing is per-file, so the + // target set is not knowable here. Nothing reads Method annotations today: + // `annotations` is not a column in METHOD_SCHEMA/FUNCTION_SCHEMA + // (src/core/lbug/schema.ts), so it lives only in the in-memory graph for one + // analyze run, and the sole in-memory reader (springDiFieldMatcher) is gated + // to `Property` nodes. If that column is ever added, revisit this: the set + // would then become an agent-visible claim that may over-state the target. + annotations: extractAnnotations(component, 'modifiers'), + sourceFile: context.filePath, + line: component.startPosition.row + 1, + column: component.startPosition.column, + }; +} + +/** Java records synthesize one public, zero-argument accessor per component. */ +export const javaRecordMethodExtractor: MethodExtractor = { + ...javaExplicitMethodExtractor, + language: SupportedLanguages.Java, + extract(node: SyntaxNode, context: MethodExtractorContext): ExtractedMethods | null { + const extracted = javaExplicitMethodExtractor.extract(node, context); + if (extracted === null || node.type !== 'record_declaration') return extracted; + + const explicitAccessors = explicitZeroArgAccessorNames(node); + const implicitAccessors = recordComponents(node) + .filter((component) => { + const name = recordComponentNameNode(component)?.text; + return name !== undefined && !explicitAccessors.has(name); + }) + .map((component) => implicitAccessorInfo(component, context)) + .filter((method): method is MethodInfo => method !== null); + + return { ...extracted, methods: [...extracted.methods, ...implicitAccessors] }; + }, +}; + +/** Scope declarations matching the structure-phase synthetic accessor nodes. */ +export function synthesizeJavaRecordComponentAccessorCaptures( + rootNode: SyntaxNode, +): CaptureMatch[] { + const captures: CaptureMatch[] = []; + for (const recordNode of rootNode.descendantsOfType('record_declaration')) { + const explicitAccessors = explicitZeroArgAccessorNames(recordNode); + for (const component of recordComponents(recordNode)) { + const nameNode = recordComponentNameNode(component); + const returnType = recordComponentReturnType(component); + if (nameNode === null || returnType === null || explicitAccessors.has(nameNode.text)) + continue; + + captures.push({ + '@scope.function': nodeToCapture('@scope.function', component), + }); + captures.push({ + '@declaration.method': nodeToCapture('@declaration.method', component), + '@declaration.name': nodeToCapture('@declaration.name', nameNode), + '@declaration.parameter-count': syntheticCapture( + '@declaration.parameter-count', + component, + '0', + ), + '@declaration.required-parameter-count': syntheticCapture( + '@declaration.required-parameter-count', + component, + '0', + ), + '@declaration.return-type': syntheticCapture( + '@declaration.return-type', + component, + returnType, + ), + }); + } + } + return captures; +} + +/** + * The structure query sees every record component. Suppress that synthetic + * definition when the record body provides the canonical zero-argument + * accessor explicitly, leaving the explicit method as the single authority. + */ +export function shouldSkipJavaRecordComponentDefinition(captureMap: CaptureMap): boolean { + const component = captureMap['definition.method']; + if (component?.type !== 'formal_parameter' && component?.type !== 'spread_parameter') { + return false; + } + + const parameters = component.parent; + const recordNode = parameters?.parent; + if (parameters?.type !== 'formal_parameters' || recordNode?.type !== 'record_declaration') { + return false; + } + + // Same predicate the scope path applies, so the two can never disagree about + // which components have an accessor. The query's `name: (identifier)` is + // satisfied by tree-sitter's zero-width MISSING recovery token, so the + // structure path has to re-check what the query cannot express. + const nameNode = captureMap['name']; + if (!isRecordComponentName(nameNode)) return true; + + return explicitZeroArgAccessorNames(recordNode).has(nameNode.text); +} diff --git a/gitnexus/src/core/ingestion/languages/java/scope-resolver.ts b/gitnexus/src/core/ingestion/languages/java/scope-resolver.ts index 86c94bccb..0410baa6c 100644 --- a/gitnexus/src/core/ingestion/languages/java/scope-resolver.ts +++ b/gitnexus/src/core/ingestion/languages/java/scope-resolver.ts @@ -34,6 +34,7 @@ import { attachJavaSpringAopMetadata } from './spring-aop.js'; import { attachJavaSpringConfigBindings } from './spring-config-bindings.js'; import { attachJavaSpringConditionalMetadata } from './spring-conditionals.js'; import { attachJavaSpringDiMetadata } from './spring-di.js'; +import { attachJavaSpringNonHttpHandlerMetadata } from './spring-non-http-handlers.js'; import { applyJavaCaptureSideChannel, clearJavaClassAnnotationFacts, @@ -54,8 +55,10 @@ const javaScopeResolver: ScopeResolver = { return undefined; }, - resolveImportTarget: (targetRaw, fromFile, allFilePaths) => { - const ws: JavaResolveContext = { fromFile, allFilePaths }; + resolveImportTarget: (targetRaw, fromFile, allFilePaths, _resolutionConfig, context) => { + // `context.parsedFiles` is the whole input now: a Java import names a type + // in a DECLARED package, and the declarations live on those files (#2953). + const ws: JavaResolveContext = { fromFile, allFilePaths, parsedFiles: context?.parsedFiles }; return resolveJavaImportTarget( { kind: 'named', localName: '_', importedName: '_', targetRaw }, ws, @@ -92,6 +95,7 @@ const javaScopeResolver: ScopeResolver = { attachJavaSpringAopMetadata(graph, parsedFiles, nodeLookup, indexes); attachJavaSpringConditionalMetadata(graph, parsedFiles, nodeLookup, indexes); attachJavaSpringDiMetadata(graph, parsedFiles, nodeLookup, indexes); + attachJavaSpringNonHttpHandlerMetadata(graph, parsedFiles, nodeLookup, indexes); attachJavaSpringConfigBindings(graph, parsedFiles, nodeLookup, indexes, ctx); }, }; diff --git a/gitnexus/src/core/ingestion/languages/java/spring-non-http-handlers.ts b/gitnexus/src/core/ingestion/languages/java/spring-non-http-handlers.ts new file mode 100644 index 000000000..259dace3a --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/java/spring-non-http-handlers.ts @@ -0,0 +1,40 @@ +import { makeScopeId } from 'gitnexus-shared'; +import { + createSpringNonHttpHandlerMetadataAttacher, + hasSpringNonHttpHandlerRelevantAnnotation, + type SpringNonHttpHandlerFact, +} from '../../frameworks/spring/non-http-handlers.js'; +import { nodeToCapture, type SyntaxNode } from '../../utils/ast-helpers.js'; +import { getJavaSpringNonHttpHandlerFacts } from './capture-side-channel.js'; +import { isJavaPackageSiblingVisibilityIncomplete } from './package-siblings.js'; +import { javaSpringAnnotationFacts, type JavaAnnotationSyntaxFact } from './spring-di.js'; + +export type JavaSpringNonHttpHandlerFact = SpringNonHttpHandlerFact; + +/** Capture callable syntax while the Java class AST is already in hand. */ +export function captureJavaSpringNonHttpHandlerFacts( + classNode: SyntaxNode, + filePath: string, +): JavaSpringNonHttpHandlerFact[] { + const facts: JavaSpringNonHttpHandlerFact[] = []; + const body = classNode.childForFieldName('body'); + if (body === null) return facts; + for (const member of body.namedChildren) { + if (member.type !== 'method_declaration') continue; + const annotations = javaSpringAnnotationFacts(member); + if (!hasSpringNonHttpHandlerRelevantAnnotation(annotations)) continue; + const ownerRange = nodeToCapture('@spring-non-http-handler.owner', member).range; + facts.push({ + ownerScopeId: makeScopeId({ filePath, range: ownerRange, kind: 'Function' }), + ownerFilePath: filePath, + ownerRange, + annotations, + }); + } + return facts; +} + +export const attachJavaSpringNonHttpHandlerMetadata = createSpringNonHttpHandlerMetadataAttacher({ + getFacts: getJavaSpringNonHttpHandlerFacts, + isPackageVisibilityIncomplete: isJavaPackageSiblingVisibilityIncomplete, +}); diff --git a/gitnexus/src/core/ingestion/languages/javascript/captures.ts b/gitnexus/src/core/ingestion/languages/javascript/captures.ts index a5cf3059a..d3ba94791 100644 --- a/gitnexus/src/core/ingestion/languages/javascript/captures.ts +++ b/gitnexus/src/core/ingestion/languages/javascript/captures.ts @@ -18,7 +18,9 @@ * inferred from leading JSDoc comments. A lightweight regex scanner * (`parseJsDocParams` / `parseJsDocReturn`) extracts `@param {T} n` * and `@returns {T}` tags and emits synthetic captures positioned on - * the annotated function node. + * the annotated function node. `@type {T}` on a class FIELD is the same + * story one level down — it is the only way JavaScript can declare a + * field's type at all — and emits `@type-binding.class-field` (#2833). * * 4. **Shared synthesis passes** — destructuring, for-of map-tuple, and * instanceof narrowing passes are duplicated from `typescript/captures.ts` @@ -40,6 +42,7 @@ import { computeTsArityMetadata } from '../typescript/arity-metadata.js'; import { synthesizeTsReceiverBinding } from '../typescript/receiver-binding.js'; import { isArrayMethodCallbackArrow } from '../typescript/array-callback.js'; import { isStaticClassFieldBinding } from '../typescript/captures.js'; +import { reducesToContainedType } from '../typescript/interpret.js'; /** JavaScript's spelling of a class-field declaration — the TypeScript grammar * calls the same construct `public_field_definition`. Named here, not in the @@ -372,9 +375,140 @@ function parseJsDocType(text: string): string | null { return m ? m[1].trim() : null; } +/** + * A type REFERENCE, possibly qualified, generic, or unioned: + * `Repo`, `Repo`, `models.Repo`, `Handler`, `Repo|null`, + * `Repo | null`. + * + * Applied only to a string already capped by {@link JSDOC_TYPE_MAX_LENGTH}: + * the union and generic groups both nest quantifiers, so an unbounded + * non-matching input is a backtracking hazard, and a docblock's `{…}` payload + * is attacker-shaped text (it is whatever the file says). + * + * JSDoc's `{…}` payload is free text and carries shapes that are not + * references at all — record types (`{{a: number}}`), function types + * (`{function(string): void}`), the any-type `{*}`, parenthesized unions + * (`{(Repo|Other)}`). None of those name a class, so a field annotated with + * one is DECLINED rather than bound to whatever substring survives + * normalization. (`parseJsDocType`'s `[^}]+` also truncates a record type at + * its first `}`, which this rejects too.) + */ +/** Longest `@type {…}` payload considered. A type REFERENCE that names a class + * is far shorter; past this the string is a structural type or generated + * noise, which this pass declines anyway, and the cap is what keeps + * {@link JSDOC_TYPE_REFERENCE_RE}'s nested quantifiers off an unbounded + * input. */ +const JSDOC_TYPE_MAX_LENGTH = 200; + +const JSDOC_TYPE_REFERENCE_RE = + /^[A-Za-z_$][\w$]*(?:\.[A-Za-z_$][\w$]*)*(?:\s*<[\w$.,<>\s]*>)?(?:\s*\|\s*[A-Za-z_$][\w$]*(?:\.[A-Za-z_$][\w$]*)*(?:\s*<[\w$.,<>\s]*>)?)*$/; + +/** + * The spelling a JSDoc `@type` should bind a class FIELD to, or `null` to + * decline. + * + * The as-written spelling is returned, NOT a reduced one: `interpretJsTypeBinding` + * carries `Repo` through to `TypeRef.rawName` untouched (user generics are + * not on `stripGeneric`'s wrapper list), and `resolveClassBindingForName` erases + * the arguments to `Repo` at lookup time. That is the same erasure every other + * language in #2833 relies on, so generics need no code here — verified, not + * assumed, by the capture probe in that issue. + * + * Two declines: + * - `reducesToContainedType` — the container spellings whose interpretation + * would yield the ELEMENT (`Repo[]`, `Array`, `Promise`). See + * that predicate for why a field must not take its element's type. + * - anything that is not a type reference (see JSDOC_TYPE_REFERENCE_RE). + * + * The leading `?` / `!` nullability sigils are JSDoc-specific decoration with no + * bearing on which class is named, so they are peeled first — `{?Repo}` binds + * `Repo` exactly as `{Repo|null}` does. + */ +function jsDocFieldTypeSpelling(rawType: string): string | null { + const spelling = rawType + .trim() + .replace(/^[?!]+/, '') + .trim(); + if (spelling === '' || spelling.length > JSDOC_TYPE_MAX_LENGTH) return null; + if (reducesToContainedType(spelling)) return null; + if (!JSDOC_TYPE_REFERENCE_RE.test(spelling)) return null; + return spelling; +} + +/** + * The identifier a JSDoc `@type` may bind a `field_definition` to, or `null` if + * this field takes no docblock binding at all (#2833). + * + * Two refusals, and both are cheaper to answer than the docblock search they + * gate, which is why they run before it: + * + * - `static` fields are dropped, exactly as the query-driven annotation path + * drops them in `emitJsScopeCaptures` — a static member belongs to the class + * object and would silently RETYPE an instance field of the same name. The + * full cost of that trade, measured, is in `isStaticClassFieldBinding` + * (#2807). Re-checked here because the synthesis pass runs outside the + * match loop that applies it. + * - a name that is not a plain identifier (a computed key, a string key) + * names nothing `this.x` could look up. + * + * The JavaScript grammar names a field's name `property:`, not `name:`. `#priv` + * arrives as `private_property_identifier`; TypeScript binds those under their + * `#`-prefixed spelling, which is how `this.#priv` looks it up. + */ +function jsDocBindableFieldName(node: SyntaxNode): SyntaxNode | null { + if (isStaticClassFieldBinding(node, JS_CLASS_FIELD_DEFINITION_TYPES)) return null; + const nameNode = node.childForFieldName('property'); + if ( + nameNode === null || + (nameNode.type !== 'property_identifier' && nameNode.type !== 'private_property_identifier') + ) { + return null; + } + return nameNode; +} + +/** + * Emit the class-FIELD type binding a JSDoc `@type {T}` block declares (#2833). + * + * JavaScript has no type annotations, so a docblock is the only way a field + * can declare one — and measured before this branch, `/** @type {Repo} *​/ + * repo;` bound NOTHING, taking down the non-generic control (`{Plain}`) with + * it. TypeScript's equivalent `repo: Repo` has always bound, via the + * `@type-binding.annotation` rule on `public_field_definition`; this reaches the + * same DESTINATION from the docblock — an annotation-strength binding on the + * enclosing Class scope, which is the only place `typeOfMemberOnClass` reads a + * field's type — so the compound-receiver resolver finds it the way it always + * has. No resolution-side change. See the tag note on the emit below for why + * the marker is `class-field` rather than `annotation`. + */ +function emitJsDocFieldBinding( + docComment: string, + nameNode: SyntaxNode, + out: CaptureMatch[], +): void { + const rawType = parseJsDocType(docComment); + const spelling = rawType === null ? null : jsDocFieldTypeSpelling(rawType); + if (spelling === null) return; + out.push({ + '@type-binding.name': syntheticCapture('@type-binding.name', nameNode, nameNode.text), + '@type-binding.type': syntheticCapture('@type-binding.type', nameNode, spelling), + // `class-field`, not `annotation`: this is the JS provider's own + // marker for a binding that must be HOISTED to the enclosing Class + // scope, which is where `typeOfMemberOnClass` reads a field's type. + // `jsBindingScopeFor` does that walk; `interpretJsTypeBinding` then + // remaps the tag to `annotation` so the source strength is the same + // as TypeScript's `repo: Repo`. Measured: with `annotation` + // the binding lands on the innermost scope and the field never + // types — the same shape `synthesizeConstructorFieldBindings` needs + // for `this.p = new Outer()`. + '@type-binding.class-field': syntheticCapture('@type-binding.class-field', nameNode, '1'), + }); +} + /** * Walk the AST and synthesize `@type-binding.*` captures from JSDoc - * comments immediately preceding function declarations / expressions. + * comments immediately preceding function declarations / expressions and class + * field definitions. * * Only `/** … *​/` block comments are scanned. Line comments (`//`) are * intentionally excluded — JSDoc lives in block comments. @@ -385,10 +519,23 @@ function parseJsDocType(text: string): string | null { * - `@type-binding.annotation` for `@type {T}` on `let`/`const`/`var` * declarations — covers the common `/** @type {User} *​/ const u = …` * pattern (ECMA-262 §14.3.1/§14.3.2 variable declarations). + * - `@type-binding.class-field` for `@type {T}` on a `field_definition` + * (#2833) — see {@link emitJsDocFieldBinding}. * * The binding is anchored on the function node so `tsBindingScopeFor` * can hoist method return-type bindings to Module scope (matching the * TypeScript path where `hoistTypeBindingsToModule: true`). + * + * `field_definition` is a node kind of THIS walk rather than a pass of its own, + * even though a field's anchor and name are the field itself while every other + * branch keys off a function-like anchor. A separate pass would be a ninth + * full-tree traversal of `emitJsScopeCaptures`, and measured on + * `dist/core/ingestion/workers/parse-worker.js` (2.4k lines, 17.3k nodes) one + * `namedChildren` walk costs 14.3 ms against 7.5 ms to PARSE the whole file — + * `node.namedChildren` materializes a fresh array of node wrappers across the + * N-API boundary at every node. The two node kinds share this walk's preceding- + * comment search and nothing else, so the branch below returns as soon as it + * has emitted. */ function synthesizeJsDocBindings(root: SyntaxNode, out: CaptureMatch[]): void { const stack: SyntaxNode[] = [root]; @@ -404,8 +551,15 @@ function synthesizeJsDocBindings(root: SyntaxNode, out: CaptureMatch[]): void { const isMethodDef = node.type === 'method_definition'; // Also check lexical_declaration containing an arrow/fn-expression const isLexDecl = node.type === 'lexical_declaration' || node.type === 'variable_declaration'; + const isFieldDef = node.type === 'field_definition'; - if (!isFnDecl && !isMethodDef && !isLexDecl) continue; + if (!isFnDecl && !isMethodDef && !isLexDecl && !isFieldDef) continue; + + // Non-null exactly for a field that can carry a binding, so it doubles as + // the branch selector inside the comment search below. Answered before that + // search because an unbindable field has no reason to look for a docblock. + const fieldNameNode = isFieldDef ? jsDocBindableFieldName(node) : null; + if (isFieldDef && fieldNameNode === null) continue; // For `export function foo() { ... }`, the JSDoc comment precedes the // wrapping export_statement, not the inner function_declaration. @@ -418,6 +572,14 @@ function synthesizeJsDocBindings(root: SyntaxNode, out: CaptureMatch[]): void { while (sibling !== null && sibling.type === 'comment') { const text = sibling.text; if (text.startsWith('/**')) { + // A field's docblock declares its own type and nothing else — `@param` / + // `@returns` on a field name no callable — so this branch does not fall + // through to the function-like tags below. + if (fieldNameNode !== null) { + emitJsDocFieldBinding(text, fieldNameNode, out); + break; + } + // Found a JSDoc block. const params = parseJsDocParams(text); const retType = parseJsDocReturn(text); diff --git a/gitnexus/src/core/ingestion/languages/javascript/import-target.ts b/gitnexus/src/core/ingestion/languages/javascript/import-target.ts index bfdfe9951..47b8e9fa8 100644 --- a/gitnexus/src/core/ingestion/languages/javascript/import-target.ts +++ b/gitnexus/src/core/ingestion/languages/javascript/import-target.ts @@ -1,72 +1,50 @@ /** * Import-target resolver for JavaScript. * - * Delegates to the TypeScript `resolveTsTarget` standard-strategy resolver - * with `language: SupportedLanguages.JavaScript` so the resolver tries - * `.js` / `.jsx` extensions in addition to (or instead of) `.ts` / `.tsx`. + * Delegates to the TypeScript resolver, which is correct rather than merely + * convenient: `jsconfig.json` is a tsconfig by another name, `package.json` + * governs both languages identically, and Node's algorithm does not branch on + * which of the two wrote the file. The extension list already carries the JS + * family, so a `.js`/`.jsx`/`.mjs`/`.cjs` source resolves the same way. * - * The `TsResolveContext.language` flag already exists in `import-target.ts` - * and the resolver (`resolveImportPath`) already branches on it — this - * adapter just wires the right value in. + * CJS `require()` calls reference the same module-path strings as ESM `import` + * statements, so the resolver handles them uniformly with no CJS-specific + * logic here. * - * CJS `require()` calls reference the same module-path strings as ESM - * `import` statements, so the resolver handles them uniformly without any - * CJS-specific logic here. + * ## What #2953 removed * - * No `tsconfig.json` path-alias support (JavaScript projects don't use - * `tsconfig.json` compilerOptions.paths in general). Projects that DO use - * tsconfig-based aliases alongside JavaScript can still resolve via the - * standard extension-suffix fallback; the alias branch is a no-op when - * `tsconfigPaths` is null. + * This adapter used to reach `resolveImportPath`, whose last step was + * `suffixResolve` — a search for any repo file whose path ends in the + * specifier, retried with each leading segment dropped. The header this + * replaces recorded the symptom without naming it a defect: `import 'app/main'` + * resolving to `node_modules/dep/lib/main.js`, "the first `/main.js` in file + * order". A bare specifier now resolves only through a declared tsconfig + * mapping or a package manifest, and otherwise not at all. */ -import { SupportedLanguages } from 'gitnexus-shared'; -import { resolveTsTarget, type TsResolveContext } from '../typescript/import-target.js'; +import type { NodeWorkspacePackages } from '../../import-resolvers/node-workspace-packages.js'; +import { resolveTsTarget } from '../typescript/import-target.js'; +import type { TsconfigIndex } from '../typescript/tsconfig.js'; -export type JsResolveContext = TsResolveContext; +interface JsResolutionConfig { + readonly tsconfigs?: TsconfigIndex | null; + readonly nodeWorkspacePackages?: NodeWorkspacePackages | null; +} -type PassCache = { - readonly key: ReadonlySet; - readonly allFilePaths: Set; - readonly allFileList: readonly string[]; - readonly normalizedFileList: readonly string[]; - readonly resolveCache: Map; -}; - -/** - * Build a memoized `resolveImportTarget` adapter for JavaScript. - * Caches the derived arrays and per-pass resolve cache across - * `resolveImportTarget` calls within a single workspace pass. - */ +/** Build the JavaScript `resolveImportTarget` adapter. */ export function makeJsResolveImportTarget(): ( targetRaw: string, fromFile: string, allFilePaths: ReadonlySet, resolutionConfig?: unknown, ) => string | readonly string[] | null { - let cached: PassCache | null = null; - - return (targetRaw, fromFile, allFilePaths) => { - if (cached === null || cached.key !== allFilePaths) { - const allFileList = Array.from(allFilePaths); - cached = { - key: allFilePaths, - allFilePaths: new Set(allFilePaths), - allFileList, - normalizedFileList: allFileList.map((f) => f.toLowerCase()), - resolveCache: new Map(), - }; - } - - const ws: JsResolveContext = { + return (targetRaw, fromFile, allFilePaths, resolutionConfig) => { + const cfg = resolutionConfig as JsResolutionConfig | undefined; + return resolveTsTarget(targetRaw, { fromFile, - language: SupportedLanguages.JavaScript, - allFilePaths: cached.allFilePaths, - allFileList: cached.allFileList, - normalizedFileList: cached.normalizedFileList, - resolveCache: cached.resolveCache, - tsconfigPaths: null, - }; - return resolveTsTarget(targetRaw, ws); + allFilePaths, + tsconfigs: cfg?.tsconfigs ?? null, + nodeWorkspacePackages: cfg?.nodeWorkspacePackages ?? null, + }); }; } diff --git a/gitnexus/src/core/ingestion/languages/javascript/query.ts b/gitnexus/src/core/ingestion/languages/javascript/query.ts index ec2d6a3f2..168f866d7 100644 --- a/gitnexus/src/core/ingestion/languages/javascript/query.ts +++ b/gitnexus/src/core/ingestion/languages/javascript/query.ts @@ -107,6 +107,15 @@ export const JAVASCRIPT_SCOPE_QUERY = ` (field_definition property: (property_identifier) @declaration.name) @declaration.property +;; Object-literal keys of a NAMED object (A1/A5) — the scope-resolution half of +;; the same rule in TYPESCRIPT/JAVASCRIPT_QUERIES. The parse query mints the +;; Property NODE; this mints the DEF the resolver can point a read/write at. +(variable_declarator + name: (identifier) + value: (object + (pair + key: (property_identifier) @declaration.name) @declaration.property)) + ;; Declarations — free functions (function_declaration name: (identifier) @declaration.name) @declaration.function @@ -589,6 +598,99 @@ export const JAVASCRIPT_SCOPE_QUERY = ` (object (shorthand_property_identifier) @reference.name @reference.property-key @reference.value-ref) + +;; Bare-identifier reads (A2). VALUE POSITIONS ONLY — a blanket +;; \`(identifier)\` rule would mint a site for every token in the file. +(arguments + (identifier) @reference.name @reference.read.identifier) + +(assignment_pattern + right: (identifier) @reference.name @reference.read.identifier) + +(return_statement + (identifier) @reference.name @reference.read.identifier) + +;; \`const next = LIMIT\` and \`n > LIMIT\` — both plainly value reads, and both +;; named in review as gaps between what A2 claimed and what it matched. +(variable_declarator + value: (identifier) @reference.name @reference.read.identifier) + +(binary_expression + left: (identifier) @reference.name @reference.read.identifier) + +(binary_expression + right: (identifier) @reference.name @reference.read.identifier) + +;; Destructured PARAMETER keys (R2-1c). \`function exit({ exitMinAtrMult = 0 })\` +;; reads that property off whatever the caller passes, exactly as +;; \`cfg.exitMinAtrMult\` would — the field just never appears in a +;; member_expression, so the read had no site at all and the function that +;; implements the behaviour was missing from "who reads this setting?". +;; +;; A distinct anchor rather than @reference.read.member: that tag is filtered +;; emit-side to matches with a member_expression ancestor (calls and writes +;; share its shape), and a destructuring pattern has none, so it would be +;; dropped. The \`read.\` head is what maps this to a read kind, so the new tag +;; needs no mapping change. +;; +;; The object_pattern is the receiver. It is anonymous — there is no name to +;; type — which is precisely the untyped-receiver case the name-narrowing pass +;; exists to serve. +;; +;; Scoped to formal_parameters deliberately. A destructuring binding elsewhere +;; (\`const { x } = require('m')\`) is often an import rather than a field read, +;; and minting a property read for it would attribute module bindings to +;; unrelated same-named keys. +(formal_parameters + (object_pattern + (shorthand_property_identifier_pattern) @reference.name + @reference.read.destructured) @reference.receiver) + +(formal_parameters + (object_pattern + (object_assignment_pattern + left: (shorthand_property_identifier_pattern) @reference.name + @reference.read.destructured)) @reference.receiver) + +(formal_parameters + (object_pattern + (pair_pattern + key: (property_identifier) @reference.name + @reference.read.destructured)) @reference.receiver) + +;; Object-literal keys in RECORD CONSTRUCTION position (R2-1b). Building +;; \`{ exitContract: { exitMinAtrMult: settings.x } }\` SETS that field, so this +;; is the write counterpart to the destructured read above — without it +;; "who reads this setting?" answers well and "who SETS it?" misses the code +;; that stamps the value. +;; +;; A WRITE REFERENCE, deliberately not a definition. The round-1 rule already +;; mints Property nodes for literals bound to a variable; minting more for +;; anonymous records would add same-named competitors to the very name-narrowing +;; that makes these reads resolvable — measured at 26 competing definitions for +;; one field on the reporting repo. A construction site is a USE of a field, not +;; another declaration of it. +;; +;; Two positions only: nested under a key, and returned. Both are records with a +;; name attached (the key, or the function). An inline call argument +;; (\`doThing({ id: 1 })\`) stays excluded for the same reason round 1 excluded +;; it from definitions — it is call-site data, not a named surface. +;; +;; The enclosing literal is the receiver, and it is anonymous, which routes +;; these through the same narrowing and the same refusal-to-guess as every other +;; untyped receiver. +(pair + value: (object + (pair + key: (property_identifier) @reference.name + @reference.write.property-key) @_r2b.nested) @reference.receiver) + +(return_statement + (object + (pair + key: (property_identifier) @reference.name + @reference.write.property-key) @_r2b.returned) @reference.receiver) + `; /** JSX-only suffix — appended when compiling against the JSX grammar for .jsx files. */ diff --git a/gitnexus/src/core/ingestion/languages/javascript/scope-resolver.ts b/gitnexus/src/core/ingestion/languages/javascript/scope-resolver.ts index 94967384e..8d74798ee 100644 --- a/gitnexus/src/core/ingestion/languages/javascript/scope-resolver.ts +++ b/gitnexus/src/core/ingestion/languages/javascript/scope-resolver.ts @@ -42,6 +42,8 @@ import { javascriptProvider } from '../typescript.js'; import { jsMergeBindings } from './merge-bindings.js'; import { jsArityCompatibility } from './arity.js'; import { makeJsResolveImportTarget } from './import-target.js'; +import { loadTsconfigIndex } from '../typescript/tsconfig.js'; +import { loadNodeWorkspacePackages } from '../../import-resolvers/node-workspace-packages.js'; const javascriptScopeResolver: ScopeResolver = { // Construction is keyword-prefixed: `new Service(db).doWork()` (#2708). @@ -52,6 +54,15 @@ const javascriptScopeResolver: ScopeResolver = { resolveImportTarget: makeJsResolveImportTarget(), + // JavaScript resolution reads the same declared inputs TypeScript does — + // `jsconfig.json` is a tsconfig by another name, and `package.json` is shared + // outright. Without them a bare specifier used to fall through to suffix + // matching (#2953); now it simply does not resolve. + loadResolutionConfig: async (repoPath: string) => ({ + tsconfigs: await loadTsconfigIndex(repoPath), + nodeWorkspacePackages: await loadNodeWorkspacePackages(repoPath), + }), + // JavaScript LEGB — same tier ordering as TypeScript; no declaration- // merging across type/value/namespace spaces. mergeBindings: (existing, incoming) => [...jsMergeBindings([...existing, ...incoming])], diff --git a/gitnexus/src/core/ingestion/languages/kotlin/capture-side-channel.ts b/gitnexus/src/core/ingestion/languages/kotlin/capture-side-channel.ts index 3c211bcc0..6ea9480a8 100644 --- a/gitnexus/src/core/ingestion/languages/kotlin/capture-side-channel.ts +++ b/gitnexus/src/core/ingestion/languages/kotlin/capture-side-channel.ts @@ -52,11 +52,13 @@ import { getKotlinPackageFact, setKotlinPackageFact } from './package-facts.js'; import type { KotlinSpringAopFact } from './spring-aop.js'; import type { KotlinSpringConditionalFact } from './spring-conditionals.js'; import type { KotlinSpringDiClassFact } from './spring-di.js'; +import type { KotlinSpringNonHttpHandlerFact } from './spring-non-http-handlers.js'; const classAnnotations = createClassAnnotationFactStore(); const springAopFacts = new Map(); const springConditionalFacts = new Map(); const springDiFacts = new Map(); +const springNonHttpHandlerFacts = new Map(); /** * Plain JSON-serializable snapshot of the per-file Kotlin capture-time @@ -78,6 +80,8 @@ export interface KotlinCaptureSideChannel { readonly springConditionalFacts?: readonly KotlinSpringConditionalFact[]; /** Constructor, property, and method injection syntax captured per class. */ readonly springDiFacts?: readonly KotlinSpringDiClassFact[]; + /** Scheduled, event, messaging, and managed-job handler syntax captured per callable. */ + readonly springNonHttpHandlerFacts?: readonly KotlinSpringNonHttpHandlerFact[]; } export function clearKotlinClassAnnotationFacts(): void { @@ -85,6 +89,7 @@ export function clearKotlinClassAnnotationFacts(): void { springAopFacts.clear(); springConditionalFacts.clear(); springDiFacts.clear(); + springNonHttpHandlerFacts.clear(); } export function setKotlinSpringAopFacts( @@ -136,6 +141,20 @@ export function getKotlinSpringDiFacts(filePath: string): readonly KotlinSpringD return springDiFacts.get(filePath) ?? []; } +export function setKotlinSpringNonHttpHandlerFacts( + filePath: string, + facts: readonly KotlinSpringNonHttpHandlerFact[], +): void { + if (facts.length === 0) springNonHttpHandlerFacts.delete(filePath); + else springNonHttpHandlerFacts.set(filePath, facts); +} + +export function getKotlinSpringNonHttpHandlerFacts( + filePath: string, +): readonly KotlinSpringNonHttpHandlerFact[] { + return springNonHttpHandlerFacts.get(filePath) ?? []; +} + /** * `LanguageProvider.collectCaptureSideChannel` implementation for Kotlin. * Returns `undefined` when this file recorded no side-channel state at all, so @@ -149,6 +168,7 @@ export function collectKotlinCaptureSideChannel( const aopFacts = springAopFacts.get(filePath) ?? []; const conditionFacts = springConditionalFacts.get(filePath) ?? []; const diFacts = springDiFacts.get(filePath) ?? []; + const nonHttpHandlerFacts = springNonHttpHandlerFacts.get(filePath) ?? []; const packageFact = getKotlinPackageFact(filePath); if ( companionScopes.length === 0 && @@ -156,6 +176,7 @@ export function collectKotlinCaptureSideChannel( aopFacts.length === 0 && conditionFacts.length === 0 && diFacts.length === 0 && + nonHttpHandlerFacts.length === 0 && packageFact === undefined ) { return undefined; @@ -168,6 +189,7 @@ export function collectKotlinCaptureSideChannel( ...(aopFacts.length > 0 ? { springAopFacts: aopFacts } : {}), ...(conditionFacts.length > 0 ? { springConditionalFacts: conditionFacts } : {}), ...(diFacts.length > 0 ? { springDiFacts: diFacts } : {}), + ...(nonHttpHandlerFacts.length > 0 ? { springNonHttpHandlerFacts: nonHttpHandlerFacts } : {}), }; } @@ -193,6 +215,7 @@ export function applyKotlinCaptureSideChannel(parsed: ParsedFile): void { setKotlinSpringAopFacts(parsed.filePath, []); setKotlinSpringConditionalFacts(parsed.filePath, []); setKotlinSpringDiFacts(parsed.filePath, []); + setKotlinSpringNonHttpHandlerFacts(parsed.filePath, []); setKotlinPackageFact(parsed.filePath, UNKNOWN_JVM_PACKAGE_FACT); return; } @@ -212,6 +235,10 @@ export function applyKotlinCaptureSideChannel(parsed: ParsedFile): void { parsed.filePath, Array.isArray(data.springDiFacts) ? data.springDiFacts : [], ); + setKotlinSpringNonHttpHandlerFacts( + parsed.filePath, + Array.isArray(data.springNonHttpHandlerFacts) ? data.springNonHttpHandlerFacts : [], + ); setKotlinPackageFact( parsed.filePath, isJvmPackageFact(data.packageFact) ? data.packageFact : UNKNOWN_JVM_PACKAGE_FACT, diff --git a/gitnexus/src/core/ingestion/languages/kotlin/captures.ts b/gitnexus/src/core/ingestion/languages/kotlin/captures.ts index 4e063c3cf..84ec8dc4d 100644 --- a/gitnexus/src/core/ingestion/languages/kotlin/captures.ts +++ b/gitnexus/src/core/ingestion/languages/kotlin/captures.ts @@ -23,6 +23,7 @@ import { setKotlinSpringAopFacts, setKotlinSpringConditionalFacts, setKotlinSpringDiFacts, + setKotlinSpringNonHttpHandlerFacts, } from './capture-side-channel.js'; import { captureKotlinPackageFact } from './package-facts.js'; import { synthesizeCallableFlowCaptures } from '../../utils/callable-flow-captures.js'; @@ -33,6 +34,10 @@ import { captureKotlinSpringConditionalFacts, type KotlinSpringConditionalFact, } from './spring-conditionals.js'; +import { + captureKotlinSpringNonHttpHandlerFacts, + type KotlinSpringNonHttpHandlerFact, +} from './spring-non-http-handlers.js'; const FUNCTION_DECL_TAGS = ['@declaration.function'] as const; @@ -99,6 +104,8 @@ export function emitKotlinScopeCaptures( const springAopTypeNodeIds = new Set(); const springConditionalFacts: KotlinSpringConditionalFact[] = []; const springDiFacts: KotlinSpringDiClassFact[] = []; + const springNonHttpHandlerFacts: KotlinSpringNonHttpHandlerFact[] = []; + const springNonHttpHandlerTypeNodeIds = new Set(); const springDiClassNodeIds = new Set(); const returnTypes = collectKotlinReturnTypeTexts(tree.rootNode); out.push(...synthesizeKotlinLocalAssignmentBindings(tree.rootNode, returnTypes)); @@ -130,9 +137,17 @@ export function emitKotlinScopeCaptures( nodeIfType(groupedNodes['@scope.class'], 'object_declaration'), nodeIfType(groupedNodes['@scope.class'], 'companion_object'), ].find((node): node is SyntaxNode => node !== null); - if (springAopTypeNode !== undefined && !springAopTypeNodeIds.has(springAopTypeNode.id)) { - springAopTypeNodeIds.add(springAopTypeNode.id); - springAopFacts.push(...captureKotlinSpringAopFacts(springAopTypeNode, filePath)); + if (springAopTypeNode !== undefined) { + if (!springAopTypeNodeIds.has(springAopTypeNode.id)) { + springAopTypeNodeIds.add(springAopTypeNode.id); + springAopFacts.push(...captureKotlinSpringAopFacts(springAopTypeNode, filePath)); + } + if (!springNonHttpHandlerTypeNodeIds.has(springAopTypeNode.id)) { + springNonHttpHandlerTypeNodeIds.add(springAopTypeNode.id); + springNonHttpHandlerFacts.push( + ...captureKotlinSpringNonHttpHandlerFacts(springAopTypeNode, filePath), + ); + } } const springDiClassNode = nodeIfType(groupedNodes['@scope.class'], 'class_declaration'); @@ -342,6 +357,7 @@ export function emitKotlinScopeCaptures( setKotlinSpringAopFacts(filePath, springAopFacts); setKotlinSpringConditionalFacts(filePath, springConditionalFacts); setKotlinSpringDiFacts(filePath, springDiFacts); + setKotlinSpringNonHttpHandlerFacts(filePath, springNonHttpHandlerFacts); out.push(...synthesizeCallableFlowCaptures(tree.rootNode, KOTLIN_CALLABLE_CAPTURE_OPTIONS)); return out; } diff --git a/gitnexus/src/core/ingestion/languages/kotlin/import-target.ts b/gitnexus/src/core/ingestion/languages/kotlin/import-target.ts index 71d6a6b53..33de337ee 100644 --- a/gitnexus/src/core/ingestion/languages/kotlin/import-target.ts +++ b/gitnexus/src/core/ingestion/languages/kotlin/import-target.ts @@ -1,4 +1,6 @@ import type { ParsedImport, WorkspaceIndex } from 'gitnexus-shared'; +import { KOTLIN_EXTENSIONS } from '../../import-resolvers/jvm.js'; +import { perFileSet } from '../../import-resolvers/per-file-set.js'; export interface KotlinResolveContext { readonly fromFile: string; @@ -38,20 +40,27 @@ export function resolveKotlinImportTarget( // export the imported name (#1759). // 4. Progressive prefix strip for deeper namespace aliases that // don't map 1:1 to directories. - const stripped = pathLike.split('/').slice(0, -1).join('/'); + const index = getKotlinFileIndex(ctx.allFilePaths); + const direct = findKotlinFile(index, pathLike); + if (direct !== null) return direct; + + // Only tiers 2 and 3 need the stripped path, and tier 1 answers most + // imports, so it is computed here rather than above. `lastIndexOf`/`slice` + // rather than `split`/`slice`/`join`: same result for every input, two + // allocations fewer per import. The `li < 0` guard is load-bearing — + // `'a'.slice(0, -1)` is `''`, which is what the split form yields for a + // single-segment path, but only by accident of `[].join('/')`. + const li = pathLike.lastIndexOf('/'); + const stripped = li < 0 ? '' : pathLike.slice(0, li); return ( - findKotlinFile(ctx.allFilePaths, pathLike) ?? - findKotlinExactOrSuffix(ctx.allFilePaths, stripped) ?? - findKotlinPackageFiles(ctx.allFilePaths, stripped) ?? - findByProgressivePrefixStrip(ctx.allFilePaths, pathLike) + findKotlinExactOrSuffix(index, stripped) ?? + findKotlinPackageFiles(index, stripped) ?? + findByProgressivePrefixStrip(index, pathLike) ); } -function findKotlinFile(allFilePaths: ReadonlySet, pathLike: string): string | null { - return ( - findKotlinExactOrSuffix(allFilePaths, pathLike) ?? - findKotlinDirectoryChild(allFilePaths, pathLike) - ); +function findKotlinFile(index: KotlinFileIndex, pathLike: string): string | null { + return findKotlinExactOrSuffix(index, pathLike) ?? findKotlinDirectoryChild(index, pathLike); } /** Exact (`file === pathLike+ext`) or suffix (`file ends with /pathLike+ext`) @@ -59,26 +68,15 @@ function findKotlinFile(allFilePaths: ReadonlySet, pathLike: string): st * `pathLike/` directory. Used by the stripped-path tier in * `resolveKotlinImportTarget` so a package import like `models.getRepo` * delegates to `findKotlinPackageFiles` (multi-file fan-out) instead of - * silently committing to the first directory child. */ -function findKotlinExactOrSuffix( - allFilePaths: ReadonlySet, - pathLike: string, -): string | null { + * silently committing to the first directory child. + * + * An exact match anywhere in the workspace beats a suffix match anywhere, + * which is why the two are separate maps rather than one lookup: the old scan + * returned on the first exact hit but only remembered the first suffix hit, + * so an exact match found late still won. */ +function findKotlinExactOrSuffix(index: KotlinFileIndex, pathLike: string): string | null { if (pathLike === '') return null; - const extensions = ['.kt', '.kts']; - const suffix = `/${pathLike}`; - let suffixFile: string | null = null; - - for (const raw of allFilePaths) { - const file = raw.replace(/\\/g, '/'); - if (!extensions.some((ext) => file.endsWith(ext))) continue; - for (const ext of extensions) { - if (file === `${pathLike}${ext}`) return raw; - if (suffixFile === null && file.endsWith(`${suffix}${ext}`)) suffixFile = raw; - } - } - - return suffixFile; + return index.exactByStem.get(pathLike) ?? index.suffixByStem.get(pathLike) ?? null; } /** First directory child of `pathLike/` — preserves the legacy single- @@ -86,27 +84,15 @@ function findKotlinExactOrSuffix( * package reference (rare in real Kotlin code; some fixtures rely on * it). Multi-file package fan-out goes through * `findKotlinPackageFiles` instead. */ -function findKotlinDirectoryChild( - allFilePaths: ReadonlySet, - pathLike: string, -): string | null { +function findKotlinDirectoryChild(index: KotlinFileIndex, pathLike: string): string | null { if (pathLike === '') return null; - const extensions = ['.kt', '.kts']; - const dirPrefix = `${pathLike}/`; - const suffixDirPrefix = `/${dirPrefix}`; - - for (const raw of allFilePaths) { - const file = raw.replace(/\\/g, '/'); - if (!extensions.some((ext) => file.endsWith(ext))) continue; - const atRoot = file.startsWith(dirPrefix); - const atNested = file.includes(suffixDirPrefix); - if (!atRoot && !atNested) continue; - const idx = atRoot ? 0 : file.indexOf(suffixDirPrefix) + 1; - const after = file.slice(idx + dirPrefix.length); - if (after.length > 0 && !after.includes('/')) return raw; - } - - return null; + const children = index.dirChildren.get(pathLike); + // "First" is first in `allFilePaths` iteration order, which the index + // preserves by appending as it walks the set. Since #2881 that can be an + // EARLIER file than the pre-index scan returned, never a later one: the + // guards that fell take members away from no bucket, so a bucket only ever + // gains, and a gained member lands wherever set iteration puts it. + return children === undefined ? null : (children[0] ?? null); } /** @@ -118,41 +104,270 @@ function findKotlinDirectoryChild( * candidate and picks the one whose `localDefs` actually export the * imported name (#1759). */ -function findKotlinPackageFiles( - allFilePaths: ReadonlySet, - dirPath: string, -): readonly string[] | null { +function findKotlinPackageFiles(index: KotlinFileIndex, dirPath: string): readonly string[] | null { if (dirPath === '') return null; - const extensions = ['.kt', '.kts']; - const dirPrefix = `${dirPath}/`; - const suffixDirPrefix = `/${dirPrefix}`; - const out: string[] = []; - - for (const raw of allFilePaths) { - const file = raw.replace(/\\/g, '/'); - if (!extensions.some((ext) => file.endsWith(ext))) continue; - const atRoot = file.startsWith(dirPrefix); - const atNested = file.includes(suffixDirPrefix); - if (!atRoot && !atNested) continue; - const idx = atRoot ? 0 : file.indexOf(suffixDirPrefix) + 1; - const after = file.slice(idx + dirPrefix.length); - // Direct children only — `models/sub/Util.kt` is a different package - // (`models.sub`) and must not be merged with `models`. - if (after.length === 0 || after.includes('/')) continue; - out.push(raw); - } - - return out.length === 0 ? null : out; + return index.dirChildren.get(dirPath) ?? null; } -function findByProgressivePrefixStrip( - allFilePaths: ReadonlySet, - pathLike: string, -): string | null { +function findByProgressivePrefixStrip(index: KotlinFileIndex, pathLike: string): string | null { const segments = pathLike.split('/').filter(Boolean); for (let skip = 1; skip < segments.length; skip++) { - const found = findKotlinFile(allFilePaths, segments.slice(skip).join('/')); + const found = findKotlinFile(index, segments.slice(skip).join('/')); if (found !== null) return found; } return null; } + +/** + * Per-file-set lookup tables for Kotlin import resolution, memoized on the + * `allFilePaths` Set object (the same Set is passed for every import in a run, + * so the index is built once and reused). + * + * WHY: every tier of `resolveKotlinImportTarget` used to walk the whole + * workspace — `for (const raw of allFilePaths)` with a `replace(/\\/g, '/')` + * and several string scans per entry — and the tiers are tried in cascade, so a + * single unresolved import cost two to four full passes. Across a repository + * with tens of thousands of Kotlin files that is `O(imports × files)` — on the + * order of 10^10 string operations on one thread, which presents as `analyze` + * sitting at exactly 1.00 core with a flat heap and no output for hours (every + * allocation is a short-lived string, so nothing accumulates to hint at + * progress). Small repositories hide it completely: at a few hundred files each + * pass is free. + * + * The maps below make each tier O(1), so resolution cost becomes O(files) once + * plus O(1) per import. + * + * - `exactByStem`: path minus its `.kt`/`.kts` extension -> raw path, for the + * `file === pathLike+ext` tier. + * - `suffixByStem`: every component-suffix of that stem -> raw path, for the + * `file ends with /pathLike+ext` tier. Keyed per suffix rather than per + * basename so a multi-segment import (`util/OneArg`) hits one bucket instead + * of filtering a basename bucket. The basename-bucket form Python uses was + * built and measured against this one during review: byte-identical output, + * ~66% less memory, and 7.3x slower per query on a repeated-basename corpus + * — enough to fail this resolver's own scaling budget at ~2.0. The memory + * the per-suffix keying costs is small in absolute terms (~60 MiB at 100k + * Kotlin files at depth 8), so it is not a trade worth revisiting. + * - `dirChildren`: package directory -> its direct `.kt`/`.kts` children, in + * set-iteration order, serving both the fan-out tier and the + * first-child fallback. + * + * Both stem maps keep the FIRST path inserted for a key, because the scans they + * replace returned the first match in set-iteration order. + * + * The shared `buildSuffixIndex` (`import-resolvers/utils.ts`, used by C#, Ruby, + * Vue and TypeScript) is deliberately NOT reused — the same call Python + * documents at `python/import-target.ts`. Run side by side against this + * resolver, three probes out of five diverge: + * + * - `['deep/util/User.kt', 'util/User.kt']` for `util.User` — it conflates + * exact and proper-suffix matches in one map, so the deep path wins where + * the scan returned the exact one; + * - `['deep/util/User.kt', 'util/User.kts']` for `util.User` — its keys carry + * the extension, so a `.kt` SUFFIX beats a `.kts` EXACT; + * - `['models/A.kts', 'models/B.kt']` for `models.getThing` — it splits the + * package into `:kt` and `:kts` buckets instead of returning both in set + * order. + * + * A fourth probe — `['data/src/…/data/Repo.kt']` for `data.getRepo`, where the + * shared index fanned out and this one returned null — stopped diverging in + * #2881, which removed the first-occurrence rule that caused it. The remaining + * three are still edges that would move in every Kotlin repository, so + * consolidating the two is a behaviour change, not a cleanup. + * + * `dirChildren` is likewise NOT the shared `import-resolvers/package-dir-index.ts` + * — the consolidation a maintainer will actually propose, since Java routes + * through exactly it (`languages/java/import-target.ts`). Measured, that swap is + * output-identical for 26.2% less retained memory, and costs 8114x on + * `import data.*` at 200 matching directories. `bench/kotlin-import-target` + * CANNOT see that regression — its corpus gives every module a unique package + * leaf — and the arm that can is `bench/import-target`'s kotlin `collide`. See + * `_blind_spot` in `bench/kotlin-import-target/baselines.json`. + * + * `dirChildren`'s bucket rule, and what #2881 removed + * --------------------------------------------------- + * A file is a child of its own directory and of every component-suffix of that + * directory, unguarded. The suffix half used to carry two guards inherited from + * the pre-index per-import scan rather than from anything Kotlin requires (the + * same generation of code put the `indexOf` half into + * `import-resolvers/package-dir-index.ts` and `import-resolvers/csharp.ts`, + * where it was removed under the same issue): `startsWith(s + '/')` skipped the + * bucket outright, and an `indexOf` equality demanded that the parent be the + * FIRST `/s/` in the path. Between them they dropped the bucket whenever the + * package name repeated higher up the tree, so + * `data/src/main/kotlin/com/example/data/Repo.kt` was not a child of `data` + * (leading segment, `startsWith`) and neither was `top/data/mid/data/Repo.kt` + * (mid-path, `indexOf`). `import data.helper` resolved to null in both. Only + * the fan-out tier looked affected — `data.Repo` answers from `suffixByStem`, + * which never had such a guard — which is why the shape looked narrow enough to + * preserve. + * + * Nothing downstream narrows a widened bucket back at the FILE level. The + * `localDefs` filter of #1759 constrains `targetDefId`/`BindingRef` ONLY: the + * finalize pass mints one draft PER CANDIDATE, each keeping its own + * `targetFile` (`gitnexus-shared/src/scope-resolution/finalize-algorithm.ts`), + * and the one File→File filter downstream + * (`scope-resolution/graph-bridge/imports-to-edges.ts`) tests `targetFile` + * against `null` and against the source file, never reads `linkStatus`, and + * emits `IMPORTS` at confidence 1.0. So every extra bucket member becomes an + * unconditional File→File `IMPORTS` edge: one `import data.load` on an + * Android-style layout measured 5 → 6 edges, the added one `unresolved`. That + * is a real cost, paid deliberately — a MISSING bucket is unrecoverable, and + * there is no version of the bucket that is right for one consumer and wrong + * for the other. Narrowing the File→File side, if it is ever wanted, is a + * downstream filter and a separate change. + * + * What moved, over the corpus: of the 235 records the published census counted, + * 149 are a different first child, 32 are a wider fan-out array, and 54 are + * null → resolved — none of which turns a bound answer into an unbound one. + * That taxonomy has no bucket for a fourth outcome class this change + * introduces, and did not count it. Tier 3 + * (`findKotlinPackageFiles`) precedes tier 4 (`findByProgressivePrefixStrip`), + * so a bucket the guards used to leave empty returned null and let tier 4 run; + * a now-populated bucket stops tier 4 from running at all, which turns a + * resolved answer into an unresolved one and a `string` into an array: + * + * ['data/src/main/kotlin/com/example/data/Repo.kt', 'common/helper.kt'] + * with `import data.helper` + * before → 'common/helper.kt' (bound) + * after → ['data/src/main/kotlin/com/example/data/Repo.kt'] (no `helper`) + * + * Over the census corpus that class is ZERO records — and the zero is the + * point, not a reprieve. The shape above is real and reproduces by hand in + * both iteration orders; running `bench/kotlin-import-target`'s own generator + * at 10x (4000 repositories, ~198 600 distinct records) hits it 4-12 times per + * seed, i.e. an expectation of about ONE over this corpus's 19 968. So the + * fingerprint does not gate this class: it is the same blindness the go arm had + * before #2881 widened its corpus — a gate cannot catch a shape its corpus + * cannot express. Adding a case is a deliberate fingerprint move and belongs in + * its own change, with the re-baseline that implies. + */ +interface KotlinFileIndex { + readonly exactByStem: Map; + readonly suffixByStem: Map; + /** Buckets are frozen once the build loop finishes — see `getKotlinFileIndex`. */ + readonly dirChildren: Map; +} + +const getKotlinFileIndex = perFileSet((allFilePaths: ReadonlySet): KotlinFileIndex => { + // Runs on a cache miss only. That it happens once per run and not once per + // import is asserted by counting traversals of the Set itself, in + // `test/integration/kotlin-import-index-reuse.test.ts` (#2909). + + const exactByStem = new Map(); + const suffixByStem = new Map(); + const dirChildren = new Map(); + /** + * BUILD-LOCAL: `dir` -> every `dirChildren` key a file in that directory + * contributes to. That list is a pure function of `dir`, and a package + * directory holds many files, so without this the walk below cuts one `slice` + * per component of the SAME directory once per FILE — and every slice after + * the first file's is a freshly allocated string that hashes to a key the map + * already holds and is then dropped. Interning them once per DIRECTORY + * instead of once per FILE is ~21% of the build at 32 000 files. + * + * It cannot move an answer. The array is filled on the first file of a + * directory, in the order the per-file walk produced, and every later file in + * that directory finds those keys already present — so the key set, the Map's + * key insertion order and every bucket's order are what the per-file form + * produced. `kotlin-index-internals.test.ts` pins the part of that a consumer + * can observe, and does it through the resolver's own surface rather than over + * the built maps: bucket CONTENTS and ORDER (from the fan-out tier, which + * hands out the bucket array itself), bucket IDENTITY across calls, and that + * the array handed out is FROZEN. Be precise about the limits, because the + * mutation matrix in that file's header measured them: a MIS-KEYED memo is + * caught, a DELETED one is not — the memo is output-identical by construction, + * so nothing observable can prove it ran. Likewise `Object.isFrozen` catches a + * missing freeze and a compacted-but-never-stored copy, but NOT a deleted + * `slice()`: a JS array's backing-store capacity has no reflective surface, so + * the compaction's only instrument is `heap_ceiling_bytes.kotlin` in + * `bench/import-target/baselines.json` — a CEILING, because compaction + * reclaims, so losing it makes the retained reading grow (+12.57% measured). + * Map key insertion ORDER is + * unasserted BY DESIGN: + * `dirChildren` is only ever read by `.get(key)`, so key order has no + * consumer and pinning it would assert an implementation detail nothing + * depends on. Nothing else watches it either — the correctness fingerprint + * sees this index only through the four tiers, so no fingerprint could catch + * a key-order move. + * + * Dropped with this frame, so it costs nothing retained. + */ + const dirKeys = new Map(); + + for (const raw of allFilePaths) { + const norm = raw.replace(/\\/g, '/'); + const ext = KOTLIN_EXTENSIONS.find((e) => norm.endsWith(e)); + // Kotlin resolution only ever queries `.kt`/`.kts` paths, exactly as the + // scans did before skipping everything else first. + if (ext === undefined) continue; + + const stem = norm.slice(0, norm.length - ext.length); + if (!exactByStem.has(stem)) exactByStem.set(stem, raw); + // Component-suffixes of the stem: one per '/' in it. `a/b/User` yields + // `b/User` and `User`, matching `norm.endsWith('/' + key + ext)`. + for (let i = 0; i < stem.length; i++) { + if (stem[i] !== '/') continue; + const suffix = stem.slice(i + 1); + if (!suffixByStem.has(suffix)) suffixByStem.set(suffix, raw); + } + + // From `stem`, not `norm`: an extension carries no '/', so the last '/' of + // the two is the same character at the same index, and `stem.slice(0, + // lastSlash)` IS the string `norm.slice(0, norm.lastIndexOf('/'))` was. One + // backwards scan instead of two, over the string this loop already walked. + const lastSlash = stem.lastIndexOf('/'); + if (lastSlash < 0) continue; // repo-root file has no package directory + const dir = stem.slice(0, lastSlash); + + // The keys this file's directory contributes to, unguarded: `dir` itself, + // plus every component-suffix of it. A suffix `s` starts just after a '/', + // so `dir` ends with `/s` by construction and the file IS a direct child of + // a directory named `s`. The absence of a narrowing guard is deliberate — + // see the `dirChildren` section on `KotlinFileIndex` for the two guards + // #2881 dropped and for what the resulting width costs downstream. + let keys = dirKeys.get(dir); + if (keys === undefined) { + keys = [dir]; + for (let i = 0; i < lastSlash; i++) { + if (dir[i] === '/') keys.push(dir.slice(i + 1)); + } + dirKeys.set(dir, keys); + } + for (const key of keys) { + const bucket = dirChildren.get(key); + if (bucket === undefined) dirChildren.set(key, [raw]); + else bucket.push(raw); + } + } + + // Buckets are mutable only while this function runs; the index type hands + // them out `readonly` and they are frozen here, before it is cached. + // `findKotlinPackageFiles` hands a bucket straight out of the index — the + // same array `findKotlinDirectoryChild` reads `children[0]` from — so a + // downstream sort would permanently reorder the cached bucket and flip the + // FIRST-child tier's answer for every later import in the run. The finalize + // pass normalizes with `Array.isArray(t) ? t : [t]` and `isArray`'s + // `arg is any[]` predicate widens the true branch; that one call site now + // carries an explicit `readonly string[]` annotation, but the annotation is + // one deletion away and covers only that site. Freezing makes the contract + // true at runtime, so a future mutation is a loud TypeError, not a silent + // edge move. + // + // COMPACTED as they are frozen: buckets grow by `push`, so V8's growth + // overshoot stays retained for the life of the index. `length === 1` never + // grew and is skipped — slicing it saves zero bytes and costs 31% of the + // build on a corpus of single-file packages. The byte accounting lives once, + // in `bench/import-target/baselines.json`. + for (const [key, bucket] of dirChildren) { + if (bucket.length === 1) { + Object.freeze(bucket); + continue; + } + const compacted = bucket.slice(); + Object.freeze(compacted); + dirChildren.set(key, compacted); + } + + return { exactByStem, suffixByStem, dirChildren }; +}); diff --git a/gitnexus/src/core/ingestion/languages/kotlin/query.ts b/gitnexus/src/core/ingestion/languages/kotlin/query.ts index 7ec7cecc8..94dadc59e 100644 --- a/gitnexus/src/core/ingestion/languages/kotlin/query.ts +++ b/gitnexus/src/core/ingestion/languages/kotlin/query.ts @@ -81,13 +81,22 @@ const KOTLIN_SCOPE_QUERY = ` (lambda_literal) @scope.block ;; Declarations — types +;; The Kotlin grammar puts NO named fields on \`class_declaration\`, so the +;; parameter list is matched positionally as an optional unnamed child, exactly +;; as the name already is. +;; +;; Only the INLINE bound (\`\`) is read. A \`where T : Repo\` clause is a +;; separate \`type_constraints\` sibling and is left alone, so its bound reads as +;; absent — "unknown", not "unbounded". (class_declaration "interface" - (type_identifier) @declaration.name) @declaration.interface + (type_identifier) @declaration.name + (type_parameters)? @declaration.type-parameters) @declaration.interface (class_declaration "class" - (type_identifier) @declaration.name) @declaration.class + (type_identifier) @declaration.name + (type_parameters)? @declaration.type-parameters) @declaration.class (object_declaration (type_identifier) @declaration.name) @declaration.class @@ -112,7 +121,13 @@ const KOTLIN_SCOPE_QUERY = ` ])) @class-annotation.class ;; Declarations — functions / methods / properties +;; +;; A generic FUNCTION's parameters are read for the same reason a generic type's +;; are (#2912 review): \`fun runAny(v: Validator)\` writes a receiver whose +;; argument is a type VARIABLE, and a pass that cannot tell that from a concrete +;; type prunes every implementor from the call's dispatch fan-out. (function_declaration + (type_parameters)? @declaration.type-parameters (simple_identifier) @declaration.name) @declaration.function ;; Lambda bound to a val/var: val handler = { x: Int -> target(x) } diff --git a/gitnexus/src/core/ingestion/languages/kotlin/scope-resolver.ts b/gitnexus/src/core/ingestion/languages/kotlin/scope-resolver.ts index af101af3c..f5bd9bb29 100644 --- a/gitnexus/src/core/ingestion/languages/kotlin/scope-resolver.ts +++ b/gitnexus/src/core/ingestion/languages/kotlin/scope-resolver.ts @@ -26,6 +26,7 @@ import { attachKotlinSpringAopMetadata } from './spring-aop.js'; import { clearKotlinPackageFacts } from './package-facts.js'; import { attachKotlinSpringDiMetadata } from './spring-di.js'; import { attachKotlinSpringConditionalMetadata } from './spring-conditionals.js'; +import { attachKotlinSpringNonHttpHandlerMetadata } from './spring-non-http-handlers.js'; /** * Kotlin scope resolver for RFC #909 Ring 3. @@ -142,6 +143,7 @@ export const kotlinScopeResolver: ScopeResolver = { attachKotlinSpringAopMetadata(graph, parsedFiles, nodeLookup, indexes); attachKotlinSpringConditionalMetadata(graph, parsedFiles, nodeLookup, indexes); attachKotlinSpringDiMetadata(graph, parsedFiles, nodeLookup, indexes); + attachKotlinSpringNonHttpHandlerMetadata(graph, parsedFiles, nodeLookup, indexes); }, }; diff --git a/gitnexus/src/core/ingestion/languages/kotlin/spring-non-http-handlers.ts b/gitnexus/src/core/ingestion/languages/kotlin/spring-non-http-handlers.ts new file mode 100644 index 000000000..b46132073 --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/kotlin/spring-non-http-handlers.ts @@ -0,0 +1,52 @@ +import { makeScopeId } from 'gitnexus-shared'; +import { + createSpringNonHttpHandlerMetadataAttacher, + type SpringNonHttpHandlerAnnotationFact, + type SpringNonHttpHandlerFact, +} from '../../frameworks/spring/non-http-handlers.js'; +import { nodeToCapture, type SyntaxNode } from '../../utils/ast-helpers.js'; +import { getKotlinSpringNonHttpHandlerFacts } from './capture-side-channel.js'; +import { isKotlinPackageSiblingVisibilityIncomplete } from './package-siblings.js'; +import { kotlinSpringAnnotationFacts } from './spring-di.js'; + +export type KotlinSpringNonHttpHandlerFact = + SpringNonHttpHandlerFact; + +/** + * Capture annotated callables conservatively. A simple-name prefilter would + * discard Kotlin aliases (for example, `EventListener as SpringEvent`) before + * the post-import resolver can map the local name back to its annotation FQN. + */ +export function captureKotlinSpringNonHttpHandlerFacts( + classNode: SyntaxNode, + filePath: string, +): KotlinSpringNonHttpHandlerFact[] { + const facts: KotlinSpringNonHttpHandlerFact[] = []; + const body = classNode.namedChildren.find( + (child) => child.type === 'class_body' || child.type === 'enum_class_body', + ); + if (body === undefined) return facts; + for (const member of body.namedChildren) { + if (member.type !== 'function_declaration') continue; + const annotations = kotlinSpringAnnotationFacts(member); + if (annotations.length === 0) continue; + const ownerRange = nodeToCapture('@spring-non-http-handler.owner', member).range; + facts.push({ + ownerScopeId: makeScopeId({ filePath, range: ownerRange, kind: 'Function' }), + ownerFilePath: filePath, + ownerRange, + annotations: annotations.map((annotation) => ({ + name: annotation.name, + ...(annotation.useSiteTarget === undefined + ? {} + : { useSiteTarget: annotation.useSiteTarget }), + })), + }); + } + return facts; +} + +export const attachKotlinSpringNonHttpHandlerMetadata = createSpringNonHttpHandlerMetadataAttacher({ + getFacts: getKotlinSpringNonHttpHandlerFacts, + isPackageVisibilityIncomplete: isKotlinPackageSiblingVisibilityIncomplete, +}); diff --git a/gitnexus/src/core/ingestion/languages/php/captures.ts b/gitnexus/src/core/ingestion/languages/php/captures.ts index de7e0a8c7..7d0ba28c2 100644 --- a/gitnexus/src/core/ingestion/languages/php/captures.ts +++ b/gitnexus/src/core/ingestion/languages/php/captures.ts @@ -28,6 +28,11 @@ * a `@type-binding.alias` match binding the loop variable to the * element type of the iterable (resolved from PHPDoc or scopeEnv). * + * 6. **PHPDoc `@var` property synthesis** — a docblock on an UNTYPED + * property emits the `@type-binding.annotation` + `@declaration.property` + * pair the native typed-property rules emit, which is the only way PHP + * can declare a generic field type (#2833). + * * Pure given the input source text. No I/O, no globals consulted. */ @@ -136,6 +141,17 @@ export function emitPhpScopeCaptures( } } + // The one full-tree walk: class/trait heritage, and PHPDoc `@var` on an + // untyped property. Run BEFORE the match loop rather than appended after it, + // because the property declarations the `@var` half claims must join + // `typedPropertyAnchorIds`: it emits the same `@declaration.property` the + // typed rule does, so without this the loose `@declaration.variable` + // catch-all would declare the very same node a second time under its + // `$`-sigilled name — exactly the duplicate the set above exists to suppress. + // Its matches are still appended in the original order after the loop. + const walked = synthesizePhpTreeWalkCaptures(tree.rootNode); + for (const id of walked.docPropertyAnchorIds) typedPropertyAnchorIds.add(id); + for (const m of rawMatches) { // Group captures by their tag name. Tree-sitter strips the leading // `@`; we put it back so the central extractor's prefix lookups work. @@ -360,21 +376,55 @@ export function emitPhpScopeCaptures( out.push(grouped); } - out.push(...synthesizePhpInheritanceReferences(tree.rootNode)); + out.push(...walked.inheritance); + out.push(...walked.docProperties); out.push(...synthesizeCallableFlowCaptures(tree.rootNode, PHP_CALLABLE_CAPTURE_OPTIONS)); return out; } -// ─── PHP inheritance synthesis ─────────────────────────────────────────────── +// ─── PHP whole-tree synthesis ──────────────────────────────────────────────── /** - * Synthesize `@reference.inherits` captures from PHP class/trait heritage so - * the registry-primary scope-resolution path emits EXTENDS / IMPLEMENTS edges - * (mirrors C# `synthesizeCsharpInheritanceReferences` / C++ - * `emitCppInheritanceCaptures`). Without this, PHP inheritance edges came only - * from the legacy heritage-capture leg (removed in #942), which the worker - * pipeline drops for registry-primary languages (issue #1951). + * The single `walkNamedTree` pass of `emitPhpScopeCaptures`, dispatching every + * synthesis that needs to see the whole tree. + * + * ONE walk, not one per synthesis. A tree-sitter node walk is not cheap next to + * the work it feeds: measured on a 1.2k-line PHP source (9.6k nodes), a single + * `walkNamedTree` pass costs 7.4 ms against 2.1 ms to PARSE the file, because + * every step materializes node wrappers across the N-API boundary. So a new + * node kind is a branch here rather than a pass of its own — the two below emit + * into separate arrays, and `emitPhpScopeCaptures` appends them in the order + * they were appended when they were two passes. + * + * The `@reference.inherits` half exists so the registry-primary + * scope-resolution path emits EXTENDS / IMPLEMENTS edges (mirrors C# + * `synthesizeCsharpInheritanceReferences` / C++ `emitCppInheritanceCaptures`). + * Without it, PHP inheritance edges came only from the legacy heritage-capture + * leg (removed in #942), which the worker pipeline drops for registry-primary + * languages (issue #1951). See {@link emitPhpDocPropertyBinding} for the other. + */ +function synthesizePhpTreeWalkCaptures(root: SyntaxNode): { + readonly inheritance: readonly CaptureMatch[]; + readonly docProperties: readonly CaptureMatch[]; + readonly docPropertyAnchorIds: ReadonlySet; +} { + const inheritance: CaptureMatch[] = []; + const docProperties: CaptureMatch[] = []; + const docPropertyAnchorIds = new Set(); + walkNamedTree(root, (node) => { + if (node.type === 'class_declaration' || node.type === 'trait_declaration') { + emitPhpHeritageReferences(node, inheritance); + } else if (node.type === 'property_declaration') { + emitPhpDocPropertyBinding(node, docProperties, docPropertyAnchorIds); + } + }); + return { inheritance, docProperties, docPropertyAnchorIds }; +} + +/** + * Emit `@reference.inherits` for the heritage of one `class_declaration` or + * `trait_declaration`. * * Scope matches the legacy PHP heritage query (tree-sitter-queries.ts * PHP_QUERIES extends / implements / trait-use captures): @@ -396,24 +446,18 @@ export function emitPhpScopeCaptures( * || type === 'Trait' ? 'IMPLEMENTS' : 'EXTENDS'`), so `use Trait` resolves to * IMPLEMENTS on both the legacy and registry-primary paths. */ -function synthesizePhpInheritanceReferences(root: SyntaxNode): CaptureMatch[] { - const out: CaptureMatch[] = []; - walkNamedTree(root, (node) => { - if (node.type === 'class_declaration') { - // extends: single base_clause child carrying one base name. - const baseClause = findNamedChild(node, 'base_clause'); - if (baseClause !== null) emitPhpBaseNames(baseClause, out); - // implements: class_interface_clause may list several interfaces. - const ifaceClause = findNamedChild(node, 'class_interface_clause'); - if (ifaceClause !== null) emitPhpBaseNames(ifaceClause, out); - // trait use: `use TraitName;` inside the class body. - emitPhpTraitUses(node, out); - } else if (node.type === 'trait_declaration') { - // trait-uses-trait: `use OtherTrait;` inside a trait body. - emitPhpTraitUses(node, out); - } - }); - return out; +function emitPhpHeritageReferences(node: SyntaxNode, out: CaptureMatch[]): void { + if (node.type === 'class_declaration') { + // extends: single base_clause child carrying one base name. + const baseClause = findNamedChild(node, 'base_clause'); + if (baseClause !== null) emitPhpBaseNames(baseClause, out); + // implements: class_interface_clause may list several interfaces. + const ifaceClause = findNamedChild(node, 'class_interface_clause'); + if (ifaceClause !== null) emitPhpBaseNames(ifaceClause, out); + } + // trait use: `use TraitName;` inside the class body, and trait-uses-trait: + // `use OtherTrait;` inside a trait body. + emitPhpTraitUses(node, out); } /** @@ -648,21 +692,55 @@ const PHP_PRIMITIVES = new Set([ ]); /** - * Collect comment text from siblings immediately before `fnNode`. - * Skips PHP 8+ attribute_list nodes. + * The comment siblings immediately preceding `node`, in SOURCE order (the + * nearest comment last), stopping at the first named sibling that is not a + * comment or a PHP 8+ attribute. + * + * The single implementation of that chain walk. Every PHPDoc reader in this + * file wants the same siblings under the same stop rule — `@param`/`@return` on + * a method, `@var` for a foreach element type, `@var` for a field type — and + * three hand-copied walks meant a fix to the stop rule (attributes between the + * docblock and the declaration, say) could land on one reader and not the + * others, which shows up as a field typed differently from its own foreach + * element type. */ -function collectPrecedingComments(fnNode: SyntaxNode): string { - const texts: string[] = []; - let sibling = fnNode.previousSibling; +function precedingCommentSiblings(node: SyntaxNode): SyntaxNode[] { + const comments: SyntaxNode[] = []; + let sibling = node.previousSibling; while (sibling !== null) { if (sibling.type === 'comment') { - texts.unshift(sibling.text); + comments.unshift(sibling); } else if (sibling.isNamed && !SKIP_SIBLING_TYPES.has(sibling.type)) { break; } sibling = sibling.previousSibling; } - return texts.join('\n'); + return comments; +} + +/** + * First match of `re` over {@link precedingCommentSiblings}, searched from the + * NEAREST comment outward — a docblock written directly above the declaration + * wins over one further up, and an earlier comment is still reached when the + * nearest one carries no such tag. + */ +function nearestPrecedingCommentMatch(node: SyntaxNode, re: RegExp): RegExpExecArray | null { + const comments = precedingCommentSiblings(node); + for (let i = comments.length - 1; i >= 0; i--) { + const m = re.exec(comments[i].text); + if (m !== null) return m; + } + return null; +} + +/** + * Collect comment text from siblings immediately before `fnNode`. + * Skips PHP 8+ attribute_list nodes. + */ +function collectPrecedingComments(fnNode: SyntaxNode): string { + return precedingCommentSiblings(fnNode) + .map((comment) => comment.text) + .join('\n'); } /** @@ -962,8 +1040,20 @@ function findClassPropertyElementType( return null; } -/** Regex for PHPDoc @var: `@var Type` */ -const PHPDOC_VAR_RE = /@var\s+(\S+)/; +/** + * PHPDoc `@var`, with the optional variable name PHPStan/Psalm allow + * (`@var Repo $repo`). `\S+` for the type deliberately: a docblock type is + * untyped text and everything past the first space is prose. + * + * ONE regex for both readings of the tag. The FIELD type + * ({@link synthesizePhpDocPropertyBindings}) needs group 2 to tell `@var Repo + * $other` from `@var Repo`; the foreach ELEMENT type + * ({@link extractPropertyElementType}) ignores it — and since the trailing group + * is optional it can never change what group 1 captures, so a second, narrower + * copy bought nothing but the chance of the two readings of one annotation + * drifting apart. + */ +const PHPDOC_VAR_RE = /@var\s+(\S+)(?:\s+\$(\w+))?/; /** * Extract element type from a property_declaration node: @@ -971,17 +1061,11 @@ const PHPDOC_VAR_RE = /@var\s+(\S+)/; * 2. PHP 7.4+ native type field (non-array) */ function extractPropertyElementType(propDecl: SyntaxNode): string | null { - // Strategy 1: PHPDoc @var on a preceding comment sibling - let sibling = propDecl.previousSibling; - while (sibling !== null) { - if (sibling.type === 'comment') { - const m = PHPDOC_VAR_RE.exec(sibling.text); - if (m !== null) return normalizePhpDocType(m[1]); - } else if (sibling.isNamed && !SKIP_SIBLING_TYPES.has(sibling.type)) { - break; - } - sibling = sibling.previousSibling; - } + // Strategy 1: PHPDoc @var on a preceding comment sibling. The `$name` group + // is not consulted: an element type is asked for by the ONE foreach that + // already named this property, so a mismatched name cannot mis-attribute it. + const varTag = nearestPrecedingCommentMatch(propDecl, PHPDOC_VAR_RE); + if (varTag !== null) return normalizePhpDocType(varTag[1]); // Strategy 2: native type field — skip generic 'array' const typeNode = propDecl.childForFieldName('type'); if (typeNode === null) return null; @@ -989,3 +1073,184 @@ function extractPropertyElementType(propDecl: SyntaxNode): string | null { if (typeName === 'array' || typeName === '') return null; return normalizePhpDocType(typeName); } + +// ─── PHPDoc @var property synthesis ────────────────────────────────────────── + +/** + * Container spellings that base-name erasure would turn into a PHANTOM class. + * + * Erasing `list` to `list` names nothing — PHP has no `list` type — so the + * binding could only ever bind a user class that happens to be called `list`, + * i.e. exactly the wrong-edge direction. Every OTHER PHPDoc container erases to + * a name `normalizePhpType` already rejects as a primitive (`array` → + * `array`, `iterable` → `iterable`) or to a real class whose methods are + * what the field's receiver actually calls (`Collection` → `Collection`, + * `Generator` → `Generator`), so this set holds one entry, not a + * catalogue. + * + * Compared CASE-FOLDED, not by listing spellings: a deny-set that must be kept + * in sync by vigilance drifts (#2833, the same lesson python/interpret.ts + * records for its own reduction). + */ +const PHPDOC_PHANTOM_CONTAINER_BASES: ReadonlySet = new Set(['list']); + +/** + * Erase type ARGUMENTS from a docblock type, leaving the base name: + * `Repo` → `Repo`, `Repo>` → `Repo`, `Repo|null` → + * `Repo|null`. Bracket-counting rather than a regex so a nested or + * multi-argument spelling reduces in one pass; an unbalanced `<` simply + * swallows the tail, which is the declining direction. + * + * NOT the shared `stripTemplateArguments`, and the difference is the UNION: + * that one truncates at the first `<`, so `Repo|null` becomes `Repo` and + * the nullability is lost with the arguments. A docblock type is the one place + * a union survives to the binding — `interpretPhpTypeBinding` runs + * `normalizePhpType` over what this returns, and that is what strips `|null` + * exactly as it does for a native `Repo|null` property. So a PHP docblock needs + * the arguments gone and the rest of the spelling intact, which is a different + * operation and not a candidate for a seventh caller of the shared one. + */ +function erasePhpDocTypeArguments(text: string): string { + let out = ''; + let depth = 0; + for (const ch of text) { + if (ch === '<') depth++; + else if (ch === '>') { + if (depth > 0) depth--; + } else if (depth === 0) out += ch; + } + return out; +} + +/** + * The type name a property's PHPDoc `@var` should bind the FIELD to, or `null` + * to decline. + * + * Two normalizations happen here and nowhere else, and each is forced: + * + * 1. TYPE-ARGUMENT ERASURE (`Repo` → `Repo`). Every sibling language in + * #2833 lets the as-written spelling reach `TypeRef.rawName` and leaves the + * erasure to `resolveClassBindingForName`. PHP cannot: `normalizePhpType` + * reduces `X` to `Y` — the CONTAINER-ELEMENT convention, pinned by + * `test/integration/resolvers/php.test.ts` ("normalizePhpType + * ('Collection') must yield 'User', not 'Collection'") because the + * foreach path depends on it. Measured: passing `Repo` through binds + * the field to `User` and `$this->repo->save()` emits `User::save` — a + * WRONG edge, not a missing one. So a field's type arguments are erased + * HERE, before that rule can read them, and the element convention is left + * exactly as it was for `@param` / `@return` / foreach. + * + * 2. ARRAY DECLINE (`Repo[]` → nothing). A field annotated `Repo[]` holds an + * ARRAY; typing it `Repo` is a wrong field type, and the collision is real + * rather than theoretical — a repository class with a `find` / `filter` / + * `map` method would claim `$this->repos->find(…)`. The element type is + * already extracted separately for the one construct that wants it: + * `extractPropertyElementType` reads the same `@var` for `foreach + * ($this->repos as $r)`. Declining here keeps the two readings of one + * annotation from colliding. + * + * Everything else is delegated: `interpretPhpTypeBinding` applies the SAME + * `normalizePhpType` the native typed property (`private Repo $repo;`) goes + * through, so nullable (`?Repo`), null-union (`Repo|null`), intersection, + * fully-qualified (`\App\Models\Repo`, kept qualified on purpose — see that + * function) and every primitive / `mixed` / `self` / `static` rejection behave + * identically for the two spellings by construction, not by duplication. + */ +function phpDocPropertyFieldType(rawType: string): string | null { + const erased = erasePhpDocTypeArguments(rawType).trim(); + if (erased === '') return null; + // Array-of: declined (see 2 above). Checked AFTER erasure so `Repo[]` + // is recognised as an array too. + if (erased.endsWith('[]')) return null; + if (PHPDOC_PHANTOM_CONTAINER_BASES.has(erased.toLowerCase())) return null; + return erased; +} + +/** + * Emit the field type-binding a PHPDoc `@var` block declares on one UNTYPED + * property declaration (`/** @var Repo *​/ private $repo;`), and record its + * anchor id in `anchorIds`. + * + * PHP's own type story leans on docblocks for everything its native syntax + * cannot spell — and generics are exactly that, since `private Repo + * $repo;` is a parse error. The native TYPED property already binds via the + * `@type-binding.annotation` rule in `query.ts`; measured before this pass, the + * docblock form bound NOTHING, so `$this->repo->save()` lost its edge for both + * the generic spelling and its non-generic control (#2833). + * + * The emitted match is byte-identical in SHAPE to what that query rule emits — + * `@type-binding.annotation` anchored on the `property_declaration`, with + * `@type-binding.name` carrying the `$`-sigilled variable name. That is the + * whole design: `interpretPhpTypeBinding` strips the sigil for source + * `'annotation'`, `phpBindingScopeFor` places it on the same scope, and the + * compound-receiver resolver finds it in `typeBindings` the way it always has. + * No resolution-side code changes. + * + * Declines, each because the annotation cannot be ATTRIBUTED rather than + * because the type is unusable: + * - a property that already has a native `type:` — the query rule owns it, + * and a docblock repeating it must not emit a second, competing binding; + * - `private $a, $b;` — one `@var` cannot say which element it types; + * - `@var Repo $other` naming a DIFFERENT property than the one it precedes. + */ +function emitPhpDocPropertyBinding( + node: SyntaxNode, + matches: CaptureMatch[], + anchorIds: Set, +): void { + // A native type hint already produces the binding via query.ts. + if (node.childForFieldName('type') !== null) return; + + const elements = node.namedChildren.filter( + (c): c is SyntaxNode => c !== null && c.type === 'property_element', + ); + if (elements.length !== 1) return; + const varNameNode = elements[0].childForFieldName('name') ?? elements[0].firstNamedChild; + if (varNameNode === null || varNameNode.type !== 'variable_name') return; + + const raw = findPhpDocVarTag(node); + if (raw === null) return; + // `@var Repo $other` on `private $repo;` types neither — decline. + if (raw.varName !== undefined && '$' + raw.varName !== varNameNode.text) return; + + const typeName = phpDocPropertyFieldType(raw.type); + if (typeName === null) return; + + anchorIds.add(node.id); + matches.push({ + '@type-binding.annotation': nodeToCapture('@type-binding.annotation', node), + '@type-binding.name': syntheticCapture('@type-binding.name', varNameNode, varNameNode.text), + '@type-binding.type': syntheticCapture('@type-binding.type', varNameNode, typeName), + }); + // …and the FIELD declaration, which the native rule emits as its own + // separate match. Without it the property stays a `@declaration.variable` + // named `$repo` — a Variable, not a class-owned member — and the type + // binding alone is not enough: measured, `$this->repo->save()` resolved + // while `save` was unique to one class and went UNRESOLVED as soon as a + // second class declared a `save`, because narrowing a same-named method + // needs the receiver's member to be owned. The native typed property + // resolved the identical file. The `$` is stripped for the same reason it + // is on the native path: PHP stores field names unsigilled so `$obj->repo` + // looks up `repo`. + matches.push({ + '@declaration.property': nodeToCapture('@declaration.property', node), + '@declaration.name': syntheticCapture( + '@declaration.name', + varNameNode, + varNameNode.text.replace(/^\$/, ''), + ), + }); +} + +/** + * The `@var` tag on the comment siblings immediately preceding `propDecl` — + * the same chain, the same regex and the same nearest-first order + * `extractPropertyElementType` reads the tag through, so the two readings of + * one annotation cannot disagree about WHICH annotation they read. + */ +function findPhpDocVarTag( + propDecl: SyntaxNode, +): { readonly type: string; readonly varName?: string } | null { + const m = nearestPrecedingCommentMatch(propDecl, PHPDOC_VAR_RE); + return m === null ? null : { type: m[1], varName: m[2] }; +} diff --git a/gitnexus/src/core/ingestion/languages/php/import-target.ts b/gitnexus/src/core/ingestion/languages/php/import-target.ts index 523c2b1c7..96711ea02 100644 --- a/gitnexus/src/core/ingestion/languages/php/import-target.ts +++ b/gitnexus/src/core/ingestion/languages/php/import-target.ts @@ -18,6 +18,9 @@ import type { ParsedFile, ParsedImport, WorkspaceIndex } from 'gitnexus-shared'; import type { ImportResolutionContext } from '../../scope-resolution/contract/scope-resolver.js'; import { resolvePhpImportInternal } from '../../import-resolvers/php.js'; +import type { SuffixIndex } from '../../import-resolvers/utils.js'; +import { perFileSet } from '../../import-resolvers/per-file-set.js'; +import { getWorkspaceFileIndex } from '../../import-resolvers/workspace-file-index.js'; import type { ComposerConfig } from '../../language-config.js'; import { readFileSync } from 'node:fs'; import { join } from 'node:path'; @@ -72,12 +75,6 @@ function namespaceDirectories( return [...directories]; } -// A scope-resolution pass shares one stable parsedFiles array across imports. -const phpDirectoryIndexCache = new WeakMap< - readonly ParsedFile[], - ReadonlyMap ->(); - function parentDirectory(filePath: string): string { const normalizedPath = normalizePhpPath(filePath); const separator = normalizedPath.lastIndexOf('/'); @@ -98,24 +95,210 @@ function directoryAliases(filePath: string): string[] { return [...aliases]; } -function filesByDirectory( - parsedFiles: readonly ParsedFile[], -): ReadonlyMap { - const cached = phpDirectoryIndexCache.get(parsedFiles); - if (cached) return cached; - - const mutable = new Map(); - for (const parsed of parsedFiles) { - for (const directory of directoryAliases(parsed.filePath)) { - const files = mutable.get(directory) ?? []; - files.push(parsed); - mutable.set(directory, files); +/** + * Directory alias → the files under it, built once per pass. + * + * A scope-resolution pass shares one stable `parsedFiles` array across imports, + * so the array identity is the memo key — see `perFileSet`. + */ +const filesByDirectory = perFileSet( + (parsedFiles: readonly ParsedFile[]): ReadonlyMap => { + const mutable = new Map(); + for (const parsed of parsedFiles) { + for (const directory of directoryAliases(parsed.filePath)) { + const files = mutable.get(directory) ?? []; + files.push(parsed); + mutable.set(directory, files); + } } - } - phpDirectoryIndexCache.set(parsedFiles, mutable); - return mutable; + return mutable; + }, +); + +// ─── workspace index (#2901) ─────────────────────────────────────────────── + +/** + * PHP's view of the shared per-file-set workspace index. + * + * Both adapters below used to materialize `[...allFilePaths]` twice per import + * and then hand `resolvePhpImportInternal` an `index` of `undefined`, which + * dropped it onto `suffixResolve`'s linear `findIndex` — one full pass over + * every file per path-part × per extension (≈50 extensions). That is the 98 ms + * per import measured at 20k files, and the arrays were the small half of it. + * + * PASSING THE SHARED `SuffixIndex` STRAIGHT THROUGH IS NOT A HOIST — IT MOVES + * IMPORTS EDGES. `resolvePhpImportInternal` reads the index at three sites, and + * all three answer a DIFFERENT question than the scan they short-circuit + * (measured, one example each): + * + * 1. `index.getInsensitive(filePath)` on the PSR-4 class-style leg has no + * no-index counterpart at all — that leg is `allFiles.has(filePath)`, an + * exact whole-path test. The index turns it into a case-insensitive SUFFIX + * probe, so `App\Models\User` under `psr-4: {"App\\": "src"}` would start + * matching `vendor/x/src/models/user.php`. + * 2. `index.getFilesInDir(nsDir, '.php')` is keyed on every directory SUFFIX, + * while the scan it replaces is anchored at the repo root + * (`f.startsWith(nsDir + '/')`). With `app/Models/Aaa.php` and + * `vendor/pkg/app/Models/Zed.php` present, `use function App\Models\getUser` + * resolves to the former today and to the latter with the raw index. + * 3. `suffixResolve` with an index probes `index.get(S) || index.getInsensitive(S)`, + * which matches WHOLE paths too (`buildSuffixIndex` indexes the `j = 0` + * suffix); the scan compares `endsWith('/' + S)` and so can only match a + * PROPER suffix. Root-level `Foo.php` is unresolvable for `use Foo;` today + * and resolvable with the raw index; and where both match, + * `App/Models/User.php` (whole path, later in iteration order) would beat + * `vendor/x/Models/User.php` (proper suffix, earlier), which is the file the + * scan returns. + * + * So this builds a PARITY view instead: the same memoized arrays, and a + * `SuffixIndex` whose three methods reproduce the no-index answers exactly. + * - `getInsensitive` returns `undefined` unconditionally, which makes site 1 a + * no-op and falls through exactly as `index === undefined` did. It is safe to + * hollow out because `suffixResolve` reads it only as + * `get(S) || getInsensitive(S)`, so `get` can carry both halves — see below. + * - `getFilesInDir` answers from a root-anchored raw-path directory bucket, so + * site 2 returns what the scan returned, in the same order. + * - `get` answers site 3, defined as "first file in Set order whose normalized + * path has `S` as a proper segment suffix, compared case-insensitively". + * That single rule IS the scan: its predicate is + * `endsWith(p) || toLowerCase().endsWith(p.toLowerCase())`, whose first + * disjunct is subsumed by the second, so a case-sensitive hit never outranks + * an earlier case-insensitive one the way `get() || getInsensitive()` does. + * + * `get` is built on the shared `index.getInsensitive`, which is that same rule + * plus the whole-path (`j = 0`) entries. The correction needs one extra map, and + * only O(files) of it: the shared lookup can only over-match when `S` IS some + * file's whole normalized path, so `firstProperSuffixMatch` is keyed on exactly + * those strings. (Whole-string vs per-segment lowercasing agree here: no case + * mapping in Unicode produces or consumes `/`, so `lower(p).split('/')` and + * `p.split('/').map(lower)` are the same list.) + * + * `index.getInsensitive` is the ONLY shared-index method this file calls — it + * never asks the case-sensitive question — which is why `buildSuffixIndex` + * defers its two suffix maps rather than fusing them: PHP builds and retains + * one of the pair instead of both (34.49 MiB of 69.85 MiB at 32 000 paths). + * + * The two maps built HERE are deferred for the same reason and are each cheap + * only in ENTRIES, not in the walk that fills them — see the notes on + * `getFirstProperSuffixMatch` (O(paths × depth) to fill, typically zero entries) + * and `getFilesByRawDirectory` (unreachable without a `composer.json`). + */ +interface PhpWorkspaceIndex { + /** Every path, backslashes normalized to `/`. Parallel to `all`. */ + readonly normalized: readonly string[]; + /** Every path, exactly as it appears in the Set. Parallel to `normalized`. */ + readonly all: readonly string[]; + /** Scan-equivalent `SuffixIndex` for `resolvePhpImportInternal`. */ + readonly suffixIndex: SuffixIndex; } +/** Memoized on the `allFilePaths` Set identity, like `getWorkspaceFileIndex`. */ +const getPhpWorkspaceIndex = perFileSet((allFilePaths: ReadonlySet): PhpWorkspaceIndex => { + // The Set is passed THROUGH to the shared cache, never copied — a defensive + // `new Set(...)` here or in `scope-resolver.ts` would hand both WeakMaps a + // fresh key per import and silently restore O(imports × files) (#1918 P1). + const { normalized, all, index } = getWorkspaceFileIndex(allFilePaths); + + /** + * Whole-path-lowercase → the first PROPER-suffix match, the correction `get` + * applies to a whole-path hit from the shared index. + * + * DEFERRED, and deferred all the way to the branch that reads it rather than + * to the first `get`. The builder walks every slash of every path and + * lowercases a slice at each, so it is O(paths × depth) in both time and + * allocation — measured 46.0 ms at 32 000 paths on the PHP arm of + * `bench/import-target/`, filling a map that held ZERO entries, because it + * can only hold one when some file's whole path is also a proper suffix of + * another's. Most repos never produce that, and the ones that do reach this + * branch only for the imports that actually hit a whole path. Pure function + * of `normalized`/`all`, both of which the returned object already retains, + * so building it late is behaviour-identical and retains nothing new. + * + * `wholePathLower` is a scratch set of the builder, not state: nothing reads + * it afterwards, so deferring the map defers it too. + */ + let firstProperSuffixMatch: Map | null = null; + const getFirstProperSuffixMatch = (): Map => { + if (firstProperSuffixMatch !== null) return firstProperSuffixMatch; + const wholePathLower = new Set(); + for (const path of normalized) wholePathLower.add(path.toLowerCase()); + + // Only the suffixes that a whole path can shadow are worth storing; see the + // header. Built from `normalized`, so it costs no traversal of the Set. + const built = new Map(); + for (let i = 0; i < normalized.length; i++) { + const lower = normalized[i].toLowerCase(); + for (let slash = lower.indexOf('/'); slash >= 0; slash = lower.indexOf('/', slash + 1)) { + const suffix = lower.slice(slash + 1); + if (!wholePathLower.has(suffix)) continue; + if (!built.has(suffix)) built.set(suffix, all[i]); + } + } + firstProperSuffixMatch = built; + return built; + }; + + /** + * Raw directory → the files directly in it, for `getFilesInDir`. + * + * DEFERRED for the same reason as the shared `dirMap` (#2903), and here the + * case is stronger: `getFilesInDir` has exactly one caller, + * `import-resolvers/php.ts`'s PSR-4 function/constant fallback, and that + * caller sits inside `if (composerConfig) { … }`. `resolvePhpImportTarget` + * hard-codes `composerConfig: null`, so on the LanguageProvider path the map + * is statically unreachable; on the ScopeResolver path it is reachable only + * in a repo that has a parseable `composer.json` with `autoload.psr-4`. + * Measured 6.8 ms / 3.56 MiB at 32 000 paths, paid by every PHP repo without + * one. Pure function of `all`, which the returned object retains. + */ + let filesByRawDirectory: Map | null = null; + const getFilesByRawDirectory = (): Map => { + if (filesByRawDirectory !== null) return filesByRawDirectory; + // Raw paths, not normalized: the scan this replaces tests `f.startsWith(...)` + // against the Set's own strings, so a backslash path is a miss there and must + // stay a miss here. Insertion order is Set order, so `[0]` is the file the + // scan would have returned first. + const built = new Map(); + for (const raw of all) { + const separator = raw.lastIndexOf('/'); + if (separator < 0) continue; + const directory = raw.slice(0, separator); + const bucket = built.get(directory); + if (bucket === undefined) built.set(directory, [raw]); + else bucket.push(raw); + } + filesByRawDirectory = built; + return built; + }; + + const suffixIndex: SuffixIndex = { + get: (suffix: string): string | undefined => { + const hit = index.getInsensitive(suffix); + if (hit === undefined) return undefined; + const lower = suffix.toLowerCase(); + // A proper-suffix hit is already the scan's answer: the shared map holds + // the first file matching EITHER way, so nothing earlier matched at all. + if (hit.replace(/\\/g, '/').toLowerCase() !== lower) return hit; + // Whole-path hit — invisible to `endsWith('/' + S)`. The scan keeps going. + // The only branch that needs the correction map, hence the only one that + // builds it. + return getFirstProperSuffixMatch().get(lower); + }, + // Site 1 must stay a no-op, and `suffixResolve` folds this into `get`. + getInsensitive: (): undefined => undefined, + getFilesInDir: (dirSuffix: string, extension: string): string[] => { + // `nsDirPrefix` is `nsDir` when it already ends in `/`, else `nsDir + '/'` + // — either way the directory is `nsDir` minus one trailing slash. + const directory = dirSuffix.endsWith('/') ? dirSuffix.slice(0, -1) : dirSuffix; + const bucket = getFilesByRawDirectory().get(directory); + if (bucket === undefined) return []; + return bucket.filter((file) => file.endsWith(extension)); + }, + }; + + return { normalized, all, suffixIndex }; +}); + // ─── loadResolutionConfig ────────────────────────────────────────────────── /** @@ -181,17 +364,17 @@ export function resolvePhpImportTarget( if (parsedImport.kind === 'dynamic-unresolved') return null; if (parsedImport.targetRaw === null || parsedImport.targetRaw === '') return null; + // Cast, not copy: `getPhpWorkspaceIndex` memoizes on this exact Set object. const allFiles = ctx.allFilePaths as Set; - const normalizedFileList = [...allFiles].map((f) => f.replace(/\\/g, '/')); - const allFileList = [...allFiles]; + const { normalized, all, suffixIndex } = getPhpWorkspaceIndex(allFiles); return resolvePhpImportInternal( parsedImport.targetRaw, null, // composerConfig not available through LanguageProvider path allFiles, - normalizedFileList, - allFileList, - undefined, + normalized, + all, + suffixIndex, ); } @@ -216,17 +399,17 @@ export function resolvePhpImportTargetInternal( ? (resolutionConfig as ComposerConfig) : null; + // Cast, not copy: `getPhpWorkspaceIndex` memoizes on this exact Set object. const allFiles = allFilePaths as Set; - const normalizedFileList = [...allFiles].map((f) => f.replace(/\\/g, '/')); - const allFileList = [...allFiles]; + const { normalized, all, suffixIndex } = getPhpWorkspaceIndex(allFiles); const resolved = resolvePhpImportInternal( targetRaw, composerConfig, allFiles, - normalizedFileList, - allFileList, - undefined, + normalized, + all, + suffixIndex, ); const parsedImport = context?.parsedImport; diff --git a/gitnexus/src/core/ingestion/languages/python/import-decomposer.ts b/gitnexus/src/core/ingestion/languages/python/import-decomposer.ts index 7cc855796..5a7d8cba2 100644 --- a/gitnexus/src/core/ingestion/languages/python/import-decomposer.ts +++ b/gitnexus/src/core/ingestion/languages/python/import-decomposer.ts @@ -13,6 +13,7 @@ import type { Capture, CaptureMatch } from 'gitnexus-shared'; import { + findAncestorBeforeBoundary, findChild, nodeToCapture, syntheticCapture, @@ -23,12 +24,30 @@ import { * `interpretPythonImport`. */ type ImportKind = 'plain' | 'aliased' | 'from' | 'from-alias' | 'wildcard' | 'dynamic'; +/** + * The only two constructs that stop a module-level `from m import x` from + * publishing `x` as `.x`. Python has no block scope, so an import + * under `if` / `try` / `for` / `with` still publishes when its branch runs — + * verified against CPython 3.11; only `def` and `class` bodies suppress it. + */ +const PUBLICATION_SUPPRESSING_ANCESTORS: ReadonlySet = new Set([ + 'function_definition', + 'class_definition', +]); +const NO_BOUNDARY: ReadonlySet = new Set(); + interface ImportSpec { readonly kind: ImportKind; readonly source: string; readonly name: string; readonly alias?: string; readonly atNode: SyntaxNode; + /** + * Statement sits at module level, so the bound name joins the module + * namespace and is importable from this module. Read by + * `interpretPythonImport` to set `ParsedImport.reexportsName`. + */ + readonly publishesToModule?: boolean; } export function splitImportStatement(stmtNode: SyntaxNode): CaptureMatch[] { @@ -76,6 +95,9 @@ function splitImportFromStmt(stmtNode: SyntaxNode): CaptureMatch[] { const out: CaptureMatch[] = []; const moduleField = stmtNode.childForFieldName('module_name'); const moduleText = moduleField?.text ?? ''; + // Once per statement, not once per name. + const publishesToModule = + findAncestorBeforeBoundary(stmtNode, PUBLICATION_SUPPRESSING_ANCESTORS, NO_BOUNDARY) === null; // Wildcard? tree-sitter-python represents `*` as a `wildcard_import` // child and emits no name children. @@ -105,6 +127,7 @@ function splitImportFromStmt(stmtNode: SyntaxNode): CaptureMatch[] { source: moduleText, name: child.text, atNode: child, + publishesToModule, }), ); } else if (child.type === 'aliased_import') { @@ -118,6 +141,7 @@ function splitImportFromStmt(stmtNode: SyntaxNode): CaptureMatch[] { name: dotted.text, alias: alias.text, atNode: child, + publishesToModule, }), ); } @@ -137,5 +161,11 @@ function buildImportMatch(stmtNode: SyntaxNode, spec: ImportSpec): CaptureMatch if (spec.alias !== undefined) { m['@import.alias'] = syntheticCapture('@import.alias', spec.atNode, spec.alias); } + // Anchored at `spec.atNode`, never `stmtNode`: `anchorCaptureFor` picks the + // broadest span with a strict `>`, so a statement-wide span here would tie + // with `@import.statement` and let key order decide the anchor. + if (spec.publishesToModule === true) { + m['@import.publishes'] = syntheticCapture('@import.publishes', spec.atNode, 'module'); + } return m; } diff --git a/gitnexus/src/core/ingestion/languages/python/import-target.ts b/gitnexus/src/core/ingestion/languages/python/import-target.ts index d06912cf5..e31f038b0 100644 --- a/gitnexus/src/core/ingestion/languages/python/import-target.ts +++ b/gitnexus/src/core/ingestion/languages/python/import-target.ts @@ -11,8 +11,13 @@ */ import type { ParsedFile, ParsedImport, WorkspaceIndex } from 'gitnexus-shared'; +import { perFileSet } from '../../import-resolvers/per-file-set.js'; +import { + getPythonFileIndex, + importerAncestors, + importerDirOf, +} from '../../import-resolvers/python-file-index.js'; import { resolvePythonImportInternal } from '../../import-resolvers/python.js'; -import { recordPythonFileIndexBuild } from './index-stats.js'; export interface PythonResolveContext { readonly fromFile: string; @@ -82,7 +87,35 @@ export function resolvePythonImportTarget( workspaceIndex, ); if (submodule !== null) return submodule; - if (packageTarget !== null) return packageTarget; + + // `return packageTarget`, not `if (packageTarget !== null) return …` — + // falling through when it is null RE-RAN THE ENTIRE TAIL BELOW, a second + // time, with byte-identical arguments. + // + // `packageTarget` IS this function's tail for this import. The recursion + // above differs from the outer frame in exactly one field, + // `targetIncludesImportedName`, whose only effect is to make + // `pythonImportedSubmoduleTarget` return null and so skip this branch: the + // spread preserves `kind` (still `named`/`alias`, so the + // `dynamic-unresolved` guard cannot fire) and `targetRaw` (which already + // passed the null/empty guard), and `workspaceIndex` is the same object, so + // `ctx.fromFile`, `ctx.allFilePaths` and `ctx.parsedFiles` are the same + // references. The recursion therefore ran `resolvePythonImportInternal` → + // relative gate → `hasRepoCandidate` → `resolveAbsoluteFromFiles` on + // exactly the inputs the fallthrough would use. + // + // That tail is a pure function of (`fromFile`, `targetRaw`, + // `allFilePaths`): it only reads the Set and indexes memoized on the Set, + // and the `submodule` probe in between is equally read-only, so nothing can + // have changed the answer. Reaching this line means the tail already + // returned null; running it again returns null again, after another + // proximity probe and another full ancestor walk to the workspace root. + // + // Measured before this change, `from x import y` at four directory + // components: 24 `allFilePaths.has` probes per import, of which probes + // 12-23 were byte-identical repeats of 0-11. `python-import-probe-count + // .test.ts` is the gate. + return packageTarget; } // PEP-328 relative + single-segment proximity bare imports. @@ -122,13 +155,43 @@ export function resolvePythonImportTarget( return resolveAbsoluteFromFiles(pathLike, ctx.allFilePaths, ctx.fromFile); } +/** + * Answers "does this package expose `importedName` as an attribute?" from + * `localDefs` alone — so it says no for a name the package only re-exports. + * + * KNOWN DIVERGENCE from `buildReexportClosures`, which since #2864 does carry + * re-exported names (`ParsedImport.reexportsName`). With + * `pkg/__init__.py: from .impl import log`, `pkg/impl.py: def log`, and a + * same-named `pkg/log.py`, this returns false, the caller falls through to the + * submodule probe, and `from pkg import log` targets `pkg/log.py` — where + * `log` is not a local def either, so the edge ends unresolved and the closure + * is never consulted, for exactly the case it was built for. CPython binds + * `pkg.log` to the function. + * + * NOT fixed by reusing the flag here, which is the obvious three-line change + * and is wrong: `reexportsName` is also set for `pkg/__init__.py: from . + * import log`, where CPython binds `pkg.log` to the **module** `pkg/log.py` + * (verified on 3.11) and returning true here would kill the correct namespace + * edge. Separating the two needs the re-export's own resolved target, i.e. + * re-entering `resolvePythonImportTarget` from a different `fromFile` — and + * that classification is what open issue #2882 is about, so it belongs with + * that fix rather than bolted on here. Not a regression: both halves behave + * exactly as they did before #2864. + * + * The `parsedFiles.find` this used to open with was the same O(imports x files) + * shape #2913 removes on the path Set, keyed on the other collection the + * orchestrator threads: every import whose package probe resolves scanned the + * whole parsed workspace, and on a repo where `from pkg import X` usually + * resolves that is most imports. `parsedFileByPath` replaces it with one pass + * per pass. + */ function pythonFileExportsName( targetFile: string, importedName: string, parsedFiles: readonly ParsedFile[] | undefined, ): boolean { if (parsedFiles === undefined) return false; - const parsed = parsedFiles.find((file) => file.filePath === targetFile); + const parsed = parsedFileByPath(parsedFiles).get(targetFile); if (parsed === undefined) return false; return parsed.localDefs.some((def) => { const qualifiedName = def.qualifiedName; @@ -138,6 +201,27 @@ function pythonFileExportsName( }); } +/** + * `filePath -> ParsedFile`, memoized on the identity of the pass's + * `parsedFiles` array — the second stable object the orchestrator threads + * through `resolveImportTarget`, beside the path Set. + * + * FIRST WINS on a duplicated path, which is what `Array.prototype.find` + * returned, so the answer is unchanged for a workspace that somehow parsed one + * path twice. Values are references to the array's own elements: the Map costs + * one pointer per parsed file and, living in a `WeakMap` keyed on the array, + * is reclaimed with the pass rather than accumulating across runs (#2649). + */ +const parsedFileByPath = perFileSet( + (parsedFiles: readonly ParsedFile[]): Map => { + const byPath = new Map(); + for (const file of parsedFiles) { + if (!byPath.has(file.filePath)) byPath.set(file.filePath, file); + } + return byPath; + }, +); + /** * Resolve `package/sub/module` style paths (already dot-flattened) to a * concrete file in `allFilePaths`. Tries the exact path first, then walks @@ -173,19 +257,44 @@ function resolveAbsoluteFromFiles( if (allFilePaths.has(directFile)) return directFile; if (allFilePaths.has(directPkg)) return directPkg; + // Both remaining tiers — the ancestor walk and the suffix fallback — can only + // ever land on a file whose basename is `.py`, or on an `__init__.py` + // whose parent directory is named ``. The two buckets the suffix + // fallback already needs therefore also decide, in O(1) and before the walk, + // whether the walk can hit at all: neither bucket present means no tier below + // can match, and one bucket absent removes that tier's probe from EVERY step + // of the walk. On the deep corpus that is half the walk's probes (#2913). + // + // `pythonSegmentAbsent` states this same rule for the single-segment bare + // tier. It is deliberately not called here: that tier needs only the answer, + // this one needs the candidate ARRAYS for the suffix fallback below, so + // sharing would mean two extra `has` lookups per import to save four lines. + const index = getPythonFileIndex(allFilePaths); + const lastSeg = pathLike.slice(pathLike.lastIndexOf('/') + 1); + const moduleCandidates = index.byBasename.get(`${lastSeg}.py`); + const packageCandidates = index.byInitParent.get(`${lastSeg}/__init__.py`); + const mayBeModule = moduleCandidates !== undefined; + // `byInitParent` skips `__init__.py` files whose parent directory name is + // empty (a doubled separator), so an empty `` — a target spelled + // with a trailing dot — cannot use the bucket as proof of absence and keeps + // probing exactly as before. + const mayBePackage = packageCandidates !== undefined || lastSeg === ''; + if (!mayBeModule && !mayBePackage) return null; + // Ancestor walk — match the single-segment resolver's behavior at - // multi-segment granularity. Closest match wins. Stop at `i > 0` because - // `i === 0` would re-check the workspace-root candidates already covered - // by the direct check above. - const importerDir = fromFile.replace(/\\/g, '/').split('/').slice(0, -1).join('/'); - if (importerDir) { - const dirParts = importerDir.split('/').filter(Boolean); - for (let i = dirParts.length; i > 0; i--) { - const ancestor = dirParts.slice(0, i).join('/'); - const prefix = `${ancestor}/`; - const candidateFile = `${prefix}${directFile}`; - const candidatePkg = `${prefix}${directPkg}`; + // multi-segment granularity. Closest match wins. The chain stops short of the + // workspace root because the root candidates are the direct check above. + // + // The chain comes from `importerAncestors`, which builds it ONCE per importer + // directory per pass. Rebuilding it here — one `slice(0, i).join('/')` per + // path component, on every import — was half of the depth quadratic in #2913. + for (const ancestor of importerAncestors(index, importerDirOf(fromFile))) { + if (mayBeModule) { + const candidateFile = `${ancestor}/${directFile}`; if (allFilePaths.has(candidateFile)) return candidateFile; + } + if (mayBePackage) { + const candidatePkg = `${ancestor}/${directPkg}`; if (allFilePaths.has(candidatePkg)) return candidatePkg; } } @@ -214,17 +323,15 @@ function resolveAbsoluteFromFiles( // shared buildSuffixIndex is deliberately NOT used: it keeps only one // path per suffix (longest wins) and so cannot reproduce this exact // fewest-segments-then-lexicographic tie-break across all candidates. - const index = getPythonFileIndex(allFilePaths); - const lastSeg = pathLike.slice(pathLike.lastIndexOf('/') + 1); const matches: { raw: string; norm: string }[] = []; - for (const cand of index.byBasename.get(`${lastSeg}.py`) ?? []) { + for (const cand of moduleCandidates ?? []) { if (cand.norm.endsWith(suffixFile)) matches.push(cand); } // Package form: only `__init__.py` files whose parent dir is named `` // can match `…//__init__.py` — look them up by parent key (P2b) and // confirm the full suffix. Same final candidate set as the old `__init__.py` // scan, just without iterating unrelated packages. - for (const cand of index.byInitParent.get(`${lastSeg}/__init__.py`) ?? []) { + for (const cand of packageCandidates ?? []) { if (cand.norm.endsWith(suffixPkg)) matches.push(cand); } if (matches.length === 0) return null; @@ -270,131 +377,33 @@ function hasRepoCandidate( const rootFile = `${leadingSegment}.py`; const initFile = `${leadingSegment}/__init__.py`; - // Build importer-ancestor prefixes: for `backend/routers/cron.py`, - // produces `["backend/routers/services/", "backend/services/"]` for - // segment `services` (closest first, root excluded — covered above). - const importerDir = fromFile.replace(/\\/g, '/').split('/').slice(0, -1).join('/'); - const dirParts = importerDir ? importerDir.split('/').filter(Boolean) : []; - const ancestorPrefixes: string[] = []; - for (let i = dirParts.length; i > 0; i--) { - ancestorPrefixes.push(`${dirParts.slice(0, i).join('/')}/${leadingSegment}/`); - } - // Indexed equivalents of the old O(files) scan: // (1) `f === rootFile || f === initFile` -> normalized-path membership. // (2) `f.startsWith(`${seg}/`) && f.endsWith('.py')` -> some .py file lives // under directory `${seg}/`, i.e. `${seg}/` is a known .py dir prefix. // (3) ancestor namespace case -> `${ancestor}/${seg}/` is a known .py dir - // prefix. + // prefix, for some ancestor of the importer's directory. const index = getPythonFileIndex(allFilePaths); if (index.normSet.has(rootFile) || index.normSet.has(initFile)) return true; if (index.dirPrefixes.has(prefix)) return true; - for (const ap of ancestorPrefixes) { - if (index.dirPrefixes.has(ap)) return true; + // (3) used to MATERIALIZE one `${ancestor}/${seg}/` string per component of + // the importer's directory, eagerly, before checks (1) and (2) had even run — + // O(depth^2) characters on every import, and the other half of #2913. Two + // things replace that: `nestedDirNames` answers "is `seg` the name of any + // directory sitting under a non-empty parent?" in O(1), which is `false` for + // every external import (`os`, `django`, an unknown distribution) and skips + // the walk outright; and what remains walks the per-directory ancestor chain, + // built once per pass, closest first, so the common in-repo hit exits after a + // step or two. `nestedDirNames` is exact, not a filter: `${A}/${seg}/` can + // only be a directory prefix if `seg` names a directory under the non-empty + // parent `A`, so a miss here means the old loop would have missed too. + if (!index.nestedDirNames.has(leadingSegment)) return false; + for (const ancestor of importerAncestors(index, importerDirOf(fromFile))) { + if (index.dirPrefixes.has(`${ancestor}/${prefix}`)) return true; } return false; } -/** - * Per-file-set index for Python import resolution, memoized on the - * `allFilePaths` Set object (the same Set is passed for every import in a run, - * so the index is built once and reused). Replaces the per-import O(files) - * scans in `resolveAbsoluteFromFiles` (suffix match) and `hasRepoCandidate` - * (package-existence gate) with O(1)/O(bucket) lookups. - * - * - `normSet`: every file path, normalized to forward slashes (for the exact - * `f === rootFile|initFile` membership checks). - * - `byBasename`: last path component (e.g. `models.py`, `__init__.py`) -> - * all `{ raw, norm }` candidates, so suffix matches can be gathered from the - * relevant bucket and the exact tie-break applied across ALL of them. - * - `byInitParent`: `__init__.py` files keyed by their last TWO components - * (`/__init__.py`). The package suffix lookup (`pkg.sub` -> - * `…/sub/__init__.py`) targets only same-named package dirs via this map - * instead of scanning every `__init__.py` in the repo — the common - * multi-segment import path no longer scales with package count - * (PR #1918 review P2b). `__init__.py` files stay in `byBasename` too, for - * the rarer explicit `pkg.__init__` import that resolves via the module - * (`….py`) lookup. - * - `dirPrefixes`: every directory prefix of a `.py` file, trailing-slashed - * (`a/b/c.py` -> `a/`, `a/b/`), for "is there a .py file under `/`". - */ -interface PythonFileIndex { - readonly normSet: Set; - readonly byBasename: Map; - readonly byInitParent: Map; - readonly dirPrefixes: Set; -} - -const PYTHON_FILE_INDEX_CACHE = new WeakMap, PythonFileIndex>(); - -function getPythonFileIndex(allFilePaths: ReadonlySet): PythonFileIndex { - const cached = PYTHON_FILE_INDEX_CACHE.get(allFilePaths); - if (cached !== undefined) return cached; - // Cache miss: materialize a fresh index. Counted so a test can assert this - // happens once per run, not once per import (PR #1918 review P1 guard). - recordPythonFileIndexBuild(); - - const normSet = new Set(); - const byBasename = new Map(); - const byInitParent = new Map(); - const dirPrefixes = new Set(); - - for (const raw of allFilePaths) { - const norm = raw.replace(/\\/g, '/'); - // Python import resolution only ever queries `.py` paths: module `.py` - // and package `/__init__.py` membership (normSet), `.py` / - // `__init__.py` basename buckets (byBasename), and `.py` directory prefixes - // (dirPrefixes). Non-`.py` files can never match any of those, so skip them - // — they were dead weight in every structure on polyglot monorepos - // (PR #1918 review P3b; dirPrefixes was already `.py`-gated). - if (!norm.endsWith('.py')) continue; - normSet.add(norm); - - const lastSlash = norm.lastIndexOf('/'); - const base = lastSlash >= 0 ? norm.slice(lastSlash + 1) : norm; - let bucket = byBasename.get(base); - if (bucket === undefined) { - bucket = []; - byBasename.set(base, bucket); - } - bucket.push({ raw, norm }); - - // Package files also get a parent-keyed bucket so a `pkg.sub` lookup hits - // only `…/sub/__init__.py` candidates, not every `__init__.py` (P2b). - if (base === '__init__.py' && lastSlash >= 0) { - const dir = norm.slice(0, lastSlash); - const parentSlash = dir.lastIndexOf('/'); - const parentName = parentSlash >= 0 ? dir.slice(parentSlash + 1) : dir; - if (parentName) { - const initKey = `${parentName}/__init__.py`; - let ib = byInitParent.get(initKey); - if (ib === undefined) { - ib = []; - byInitParent.set(initKey, ib); - } - ib.push({ raw, norm }); - } - } - - // Directory prefixes: every slash-terminated prefix of the path (every - // index just past a '/', up to and including the file's own directory). - // Scanning the FULL normalized path — including any leading '/' for - // absolute paths — makes `dirPrefixes.has(X)` match exactly when the old - // gate's `f.startsWith(X)` (X always ends in '/') matched. The previous - // split+`filter(Boolean)` dropped the leading empty component, so an - // absolute file `/repo/svc/x.py` yielded `repo/svc/` (no leading slash) and - // gate-passed where `"/repo/svc/x.py".startsWith("repo/svc/")` is false - // (PR #1918 review P3a). For relative paths the set is identical. - for (let i = 0; i <= lastSlash; i++) { - if (norm[i] === '/') dirPrefixes.add(norm.slice(0, i + 1)); - } - } - - const index: PythonFileIndex = { normSet, byBasename, byInitParent, dirPrefixes }; - PYTHON_FILE_INDEX_CACHE.set(allFilePaths, index); - return index; -} - function pythonImportedSubmoduleTarget(parsedImport: ParsedImport): string | null { if (parsedImport.kind !== 'named' && parsedImport.kind !== 'alias') return null; if (parsedImport.targetIncludesImportedName === true) return null; diff --git a/gitnexus/src/core/ingestion/languages/python/index-stats.ts b/gitnexus/src/core/ingestion/languages/python/index-stats.ts deleted file mode 100644 index 2e3d2ae82..000000000 --- a/gitnexus/src/core/ingestion/languages/python/index-stats.ts +++ /dev/null @@ -1,29 +0,0 @@ -/** - * Build counter for the per-file-set Python import-resolution index - * (`getPythonFileIndex` in `import-target.ts`). - * - * A "build" is a `WeakMap` cache MISS that materializes a fresh - * `PythonFileIndex` (O(files)). Unlike `cache-stats.ts` (which gates its - * counters behind `PROF_SCOPE_RESOLUTION` because they sit on the per-capture - * hot path), this counter is always live: an index build happens at most once - * per resolution run, so the single increment is negligible and an unconditional - * counter avoids env-var load-order fragility in tests. - * - * Used by `test/integration/python-import-index-reuse.test.ts` to assert the - * index is reused across imports (built once per run) rather than rebuilt per - * import — the regression guard for PR #1918 review finding P1. - */ - -let INDEX_BUILDS = 0; - -export function recordPythonFileIndexBuild(): void { - INDEX_BUILDS++; -} - -export function getPythonFileIndexBuildCount(): number { - return INDEX_BUILDS; -} - -export function resetPythonFileIndexBuildCount(): void { - INDEX_BUILDS = 0; -} diff --git a/gitnexus/src/core/ingestion/languages/python/interpret.ts b/gitnexus/src/core/ingestion/languages/python/interpret.ts index 1144a7671..8c36f5c41 100644 --- a/gitnexus/src/core/ingestion/languages/python/interpret.ts +++ b/gitnexus/src/core/ingestion/languages/python/interpret.ts @@ -21,11 +21,19 @@ export function interpretPythonImport(captures: CaptureMatch): ParsedImport | nu // `@import.name` : the imported symbol name (or module name for plain imports) // `@import.alias` : the local alias name (for `as` forms) // `@import.source`: the module path (always present except for `dynamic`) + // `@import.publishes`: present iff the statement is at module level const kindCap = captures['@import.kind']; const nameCap = captures['@import.name']; const aliasCap = captures['@import.alias']; const sourceCap = captures['@import.source']; + // Python has no dedicated re-export form: a module-level `from m import x` + // binds `x` AND publishes it as `.x`. See `reexportsName` on + // `ParsedImport` for the contract, and `import-decomposer.ts` for why the + // marker — not this function — decides whether the statement is at module + // level. + const republishes = captures['@import.publishes'] !== undefined; + const kind = kindCap?.text; if (kind === undefined) return null; @@ -58,10 +66,14 @@ export function interpretPythonImport(captures: CaptureMatch): ParsedImport | nu localName: nameCap.text, importedName: nameCap.text, targetRaw: sourceCap.text, + ...(republishes ? { reexportsName: true } : {}), }; } case 'from-alias': { - // `from m import x as y` + // `from m import x as y` — republished under the alias (`.y`). + // PEP 484 treats `import x as x` as an explicit re-export; Python's + // runtime namespace republishes every module-level form, so the flag + // follows module level rather than the redundant-alias case. if (sourceCap === undefined || nameCap === undefined || aliasCap === undefined) return null; return { kind: 'alias', @@ -69,6 +81,7 @@ export function interpretPythonImport(captures: CaptureMatch): ParsedImport | nu importedName: nameCap.text, alias: aliasCap.text, targetRaw: sourceCap.text, + ...(republishes ? { reexportsName: true } : {}), }; } case 'wildcard': { @@ -147,6 +160,50 @@ function stripForwardRefQuotes(text: string): string { return text; } +/** + * Container bases whose SINGLE type argument is the element type. + * + * The single source of truth for both the matcher below and the property test + * that asserts every one of them is also declined as a user generic — the two + * lists drifting apart is the defect this arrangement exists to make + * impossible. Order is significant only in that it is the regex alternation + * order; keep additions grouped with their family. + */ +export const SINGLE_ARG_CONTAINERS: readonly string[] = [ + 'list', + 'List', + 'set', + 'Set', + 'tuple', + 'Tuple', + 'Iterable', + 'Iterator', + 'Sequence', + 'Generator', + 'AsyncIterable', + 'AsyncIterator', +]; + +/** Container bases whose SECOND type argument is the value type. See {@link SINGLE_ARG_CONTAINERS}. */ +export const MAPPING_CONTAINERS: readonly string[] = [ + 'dict', + 'Dict', + 'Mapping', + 'MutableMapping', + 'OrderedDict', + 'DefaultDict', +]; + +const QUALIFIER = '(?:[A-Za-z_][A-Za-z0-9_]*\\.)?'; + +const SINGLE_ARG_CONTAINER_RE = new RegExp( + `^${QUALIFIER}(?:${SINGLE_ARG_CONTAINERS.join('|')})\\[([^,\\]]+)\\]$`, +); + +const MAPPING_CONTAINER_RE = new RegExp( + `^${QUALIFIER}(?:${MAPPING_CONTAINERS.join('|')})\\[[^,\\]]+,\\s*([^\\]]+)\\]$`, +); + /** * Unwrap a single-arg generic collection wrapper — `list[User]`, * `set[User]`, `Iterable[User]`, `Sequence[User]`, `Iterator[User]`, @@ -159,9 +216,7 @@ function stripForwardRefQuotes(text: string): string { * resolution time. */ function stripGeneric(text: string): string { - const single = text.match( - /^(?:[A-Za-z_][A-Za-z0-9_]*\.)?(?:list|List|set|Set|tuple|Tuple|Iterable|Iterator|Sequence|Generator|AsyncIterable|AsyncIterator)\[([^,\]]+)\]$/, - ); + const single = text.match(SINGLE_ARG_CONTAINER_RE); if (single !== null) return single[1].trim(); // dict[K, V] / Dict[K, V] / Mapping[K, V] — strip to value type V. // For-loop destructuring of `for k, v in d.items()` binds `v` to @@ -170,13 +225,205 @@ function stripGeneric(text: string): string { // only shape worth handling. Match a top-level K up to the first // comma and a V to the closing bracket; nested generics in V (e.g. // `dict[str, list[User]]`) are left for a downstream strip pass. - const dict = text.match( - /^(?:[A-Za-z_][A-Za-z0-9_]*\.)?(?:dict|Dict|Mapping|MutableMapping|OrderedDict|DefaultDict)\[[^,\]]+,\s*([^\]]+)\]$/, - ); + const dict = text.match(MAPPING_CONTAINER_RE); if (dict !== null) return dict[1].trim(); + + // A subscripted type the two allow-lists above did NOT claim is a + // user-defined GENERIC, not a container: `Repo[User]`, `Handler[Req, Res]`. + // Its base names one declaration — `Repo[User]` and `Repo[Order]` are the + // same `class Repo(Generic[T])` — so reduce to that base, exactly as Java's + // and Swift's interpreters already do for their `<…>` spelling (#2833). + // + // Guarded by a DENY set rather than reached by fallthrough, because "the two + // rules above did not match" is NOT the same as "not a container". Two + // measured counterexamples, both of which this branch got wrong before the + // guard existed: + // - `dict[str, list[User]]` — the dict rule's value group cannot span a + // nested `]`, so it declines and the shape falls through. Reducing it to + // `dict` destroys the value type the dict rule explicitly leaves "for a + // downstream strip pass"; the annotation must survive intact instead. + // - `Callable[[int], User]`, `Literal["a"]`, `Annotated[int, F()]`, + // `Union[A, B]`, `tuple[int, ...]` — typing SPECIAL FORMS, not classes. + // Reducing them yields a bare `Callable`/`Literal`/`Union`, which binds + // to a workspace class of that name if one exists — a fabricated edge, + // and those names are ordinary enough for a real codebase to declare. + // Anything named here keeps its as-written text and resolves as it did + // before #2833. + // + // Only reached for genuine annotations: every Python `@type-binding.type` + // capture is a `(type)`, `(identifier)`, `(attribute)` or `(dotted_name)` + // node, so a subscripted VALUE expression (`arr[0]`) never arrives here. + // + // The as-written spelling is not lost — `scope-extractor` keeps it on + // `TypeRef.declaredSpelling` whenever it differs from the reduced name, + // which is what the receiver fold's index step reads. + const userGeneric = text.match(/^((?:[A-Za-z_][A-Za-z0-9_]*\.)*[A-Za-z_][A-Za-z0-9_]*)\[.+\]$/s); + if (userGeneric !== null) { + const qualified = userGeneric[1].trim(); + const base = qualified.slice(qualified.lastIndexOf('.') + 1); + if (!isNotAUserGenericBase(base)) return qualified; + } return text; } +/** + * Whether a subscripted annotation's base names a Python type-system construct + * rather than a workspace class — see {@link NOT_A_USER_GENERIC_SPELLINGS}. + * + * CASE-FOLDED, and that is the load-bearing part. PEP 585 gave nearly every + * container two spellings — the builtin/`collections` one and the `typing` + * alias (`deque` / `typing.Deque`, `frozenset` / `typing.FrozenSet`, + * `defaultdict` / `typing.DefaultDict`) — which differ ONLY in case. Matching + * exactly meant each pair had to be listed twice and any half-pair was a silent + * escape: `deque` was listed, `Deque` was not, so `self.dq: Deque[User]` + * reduced to `Deque` and bound to a workspace `class Deque` (#2855). Folding + * case closes that axis by construction instead of by vigilance. + * + * The cost is that a workspace class whose name is a case VARIANT of a stdlib + * construct (`class deque(Generic[T])`) stops reducing. PEP 8 makes such a + * class vanishingly rare, and the loss is a missing edge — recoverable — where + * the gain is not minting a confident wrong one. + */ +function isNotAUserGenericBase(base: string): boolean { + return NOT_A_USER_GENERIC.has(base.toLowerCase()); +} + +/** + * Bases a subscripted annotation may carry that are NOT user-defined generics. + * + * SCOPE — the standard library, and deliberately nothing else. The names below + * are the documented Python type-system surface (`typing`'s deprecated PEP 585 + * aliases and its special forms, plus the stdlib classes those aliases point + * at); that universe is CLOSED and versioned by CPython, so the list is + * auditable against + * . + * + * Third-party generics (`Mapped[int]`, `QuerySet[User]`) are NOT listed. That + * universe is open, so enumerating it only ever chases the last escape, and + * denying an ordinary name like `Model` would cost real edges in the many + * projects that legitimately declare one. Those spellings still reduce to their + * base, and the base is now admitted only on the grounds `resolveErasedBaseName` + * applies at resolution time — the file's scope chain binds it, the declaration + * is in this very file, the index proves the name is a template family, or the + * file has no cross-file class channel to be absent from. A `Mapped[User]` whose + * base the file cannot see therefore binds nothing, which is the structural + * answer this parse-time pass cannot give and no longer has to. + * + * Two distinct reasons to decline, both always-correct at this layer: + * - CONTAINERS, including ones the two rules above do not own. Reducing + * `deque[User]` to `deque` types a receiver as the container and retargets + * every call in a for-loop chain, and reducing `dict[str, list[User]]` to + * `dict` destroys the value type the dict rule leaves for a downstream pass. + * - `typing` SPECIAL FORMS, which are not classes at all. `Callable`, + * `Literal`, `Union` reduce to a bare name that binds to a workspace class + * of that name if one exists — a fabricated edge. + * + * Members are listed ONCE per case-insensitive concept: {@link + * isNotAUserGenericBase} folds case, so the builtin spelling covers its PEP 585 + * `typing` twin (`deque` covers `Deque`, `frozenset` covers `FrozenSet`). + * Non-generic ABCs (`Hashable`, `Sized`) are omitted — they cannot be written + * subscripted, so they never reach this branch. + * + * Exported for the property test that asserts the case-fold closure holds + * behaviourally; nothing else should read it. + */ +export const NOT_A_USER_GENERIC_SPELLINGS: readonly string[] = [ + // ── builtins subscriptable since PEP 585 ────────────────────────────────── + 'list', + 'set', + 'frozenset', + 'tuple', + 'dict', + 'type', + // ── `collections` ───────────────────────────────────────────────────────── + 'defaultdict', + 'OrderedDict', + 'ChainMap', + 'Counter', + 'deque', + // ── `collections.abc`, the subscriptable members ────────────────────────── + 'Mapping', + 'MutableMapping', + 'Sequence', + 'MutableSequence', + 'AbstractSet', + 'MutableSet', + 'Collection', + 'Container', + 'Reversible', + 'Iterable', + 'Iterator', + 'Generator', + 'AsyncIterable', + 'AsyncIterator', + 'AsyncGenerator', + 'Awaitable', + 'Coroutine', + 'KeysView', + 'ValuesView', + 'ItemsView', + 'MappingView', + // ── `contextlib`, and the `typing` aliases to it ────────────────────────── + 'ContextManager', + 'AsyncContextManager', + 'AbstractContextManager', + 'AbstractAsyncContextManager', + // ── `re`, and the `typing` aliases to it ────────────────────────────────── + // `Pattern` and `Match` ARE classes, so reducing them is not wrong the way + // reducing `Callable` is; they are declined because in Python annotations + // these spellings are overwhelmingly the `re` types, while a workspace class + // of the same name is a parser's own `Pattern`/`Match` and would be bound + // with no import evidence whatsoever. Same policy as the receiver-chain + // resolver's: a missing edge is recoverable, a confident wrong one is not. + 'Pattern', + 'Match', + // ── I/O streams (`typing.IO` and its two subclasses) ────────────────────── + 'IO', + 'TextIO', + 'BinaryIO', + // ── stdlib generic classes with ordinary names ──────────────────────────── + // Same policy call as `Pattern`/`Match` above, and the sharpest instance of + // it: `asyncio.Task[Result]` reduces to `asyncio.Task`, whose dotted-tail + // fallback then single-matches an unrelated workspace `class Task`. + 'Queue', + 'Task', + 'Future', + 'PathLike', + // ── `typing` special forms — not classes ────────────────────────────────── + 'Callable', + 'Literal', + 'Annotated', + 'Union', + 'Optional', + 'Final', + 'ClassVar', + // `typing.Type` is the PEP 585 alias for the builtin `type` listed above, and + // the case fold already covers it — see the one-entry-per-concept rule. + 'TypeGuard', + 'TypeIs', + 'Unpack', + 'Required', + 'NotRequired', + 'ReadOnly', + 'Concatenate', + 'LiteralString', + // ── generic machinery: bases and type-parameter declarations ────────────── + // `Generic[T]`/`Protocol[T]` are written subscripted for real. The three + // declaration forms are not subscriptable in valid Python, but this + // interpreter checks no grammar — it reduces whatever text the annotation + // capture carried — so they are declined defensively. + 'Generic', + 'Protocol', + 'TypeVar', + 'ParamSpec', + 'TypeVarTuple', +]; + +/** Case-folded lookup index over {@link NOT_A_USER_GENERIC_SPELLINGS}. */ +const NOT_A_USER_GENERIC: ReadonlySet = new Set( + NOT_A_USER_GENERIC_SPELLINGS.map((name) => name.toLowerCase()), +); + /** * Unwrap nullable type annotations so downstream resolution treats * `User | None`, `None | User`, and `Optional[User]` identically to diff --git a/gitnexus/src/core/ingestion/languages/ruby/import-target.ts b/gitnexus/src/core/ingestion/languages/ruby/import-target.ts index a31f75feb..62d6fca3d 100644 --- a/gitnexus/src/core/ingestion/languages/ruby/import-target.ts +++ b/gitnexus/src/core/ingestion/languages/ruby/import-target.ts @@ -8,7 +8,7 @@ */ import { resolveRubyImportInternal } from '../../import-resolvers/ruby.js'; -import { buildSuffixIndex } from '../../import-resolvers/utils.js'; +import { getWorkspaceFileIndex } from '../../import-resolvers/workspace-file-index.js'; import { isHeritageMarker } from '../../utils/heritage-marker.js'; export interface RubyResolveContext { @@ -100,9 +100,10 @@ function resolveRelative( * via suffix matching using the existing Ruby import resolver. */ function resolveBare(targetRaw: string, allFilePaths: ReadonlySet): string | null { - const normalizedFileList = [...allFilePaths].map((f) => f.replace(/\\/g, '/')); - const allFileList = [...allFilePaths]; - const index = buildSuffixIndex(normalizedFileList, allFileList); - - return resolveRubyImportInternal(targetRaw, normalizedFileList, allFileList, index); + // Was: two array materializations plus a full `buildSuffixIndex` per require, + // thrown away on return — every require paid to index every file in the repo + // (#2880). `buildSuffixIndex` is a pure function of the file set, so this is a + // hoist, not a behaviour change. + const { normalized, all, index } = getWorkspaceFileIndex(allFilePaths); + return resolveRubyImportInternal(targetRaw, normalized, all, index); } diff --git a/gitnexus/src/core/ingestion/languages/rust.ts b/gitnexus/src/core/ingestion/languages/rust.ts index 842da9e07..0659e3f29 100644 --- a/gitnexus/src/core/ingestion/languages/rust.ts +++ b/gitnexus/src/core/ingestion/languages/rust.ts @@ -188,6 +188,23 @@ export const rustProvider = defineLanguage({ emitScopeCaptures: emitRustScopeCaptures, cfgVisitor: createRustCfgVisitor(), interpretImport: interpretRustImport, + // `use` is a compile-time path alias, not a statement that runs. Writing one + // inside a function body — `fn f() { use crate::m::X; }`, which is legal — + // narrows where the NAME is visible and defers nothing: Rust has no + // module-initialization order in the JS/Python sense and permits intra-crate + // module cycles outright. `rust/query.ts` captures `(use_declaration)` and + // nothing else, so this covers every import form the pipeline sees; the + // structural twin is C++'s `using ns::name`, exempt under the same + // capability. Without this the central Pass-3 position rule would tag an + // fn-local `use` `runsOnlyWhenCalled` and `check --cycles` would drop a + // cycle it is part of. + // + // Deliberately the NARROW claim — position does not defer a Rust import. It + // is not a claim that no Rust import can create an initialization + // dependency; that is a bigger semantic question (statics, `OnceLock`, + // `lazy_static`) which this capability does not reach and should not be read + // as settling. See `LanguageProvider.importsExecuteWhereWritten`. + importsExecuteWhereWritten: false, interpretTypeBinding: interpretRustTypeBinding, bindingScopeFor: rustBindingScopeFor, importOwningScope: rustImportOwningScope, diff --git a/gitnexus/src/core/ingestion/languages/rust/captures.ts b/gitnexus/src/core/ingestion/languages/rust/captures.ts index ae15c99e2..d6685dde3 100644 --- a/gitnexus/src/core/ingestion/languages/rust/captures.ts +++ b/gitnexus/src/core/ingestion/languages/rust/captures.ts @@ -1,5 +1,6 @@ import type { Capture, CaptureMatch } from 'gitnexus-shared'; import { + findChild, nodeIfType, nodeToCapture, syntheticCapture, @@ -252,10 +253,22 @@ function synthesizeRustInheritanceReferences(root: SyntaxNode): CaptureMatch[] { const traitName = bareTypeIdentifier(traitField); const structName = bareTypeIdentifier(typeField); if (traitName === null || structName === null) return; + // The trait's generic ARGUMENTS (`impl Validator for V`), so + // interface dispatch can tell one instantiation of a trait from another + // (#2912). Emitted as a sub-tag rather than by widening the anchor: the + // anchor is the bare `type_identifier` inside the `generic_type`, and its + // range is part of the inheritance edge's id. + const traitArguments = + traitField.type === 'generic_type' ? findChild(traitField, 'type_arguments') : null; out.push({ '@reference.inherits': nodeToCapture('@reference.inherits', traitName), '@reference.name': nodeToCapture('@reference.name', traitName), '@reference.receiver': syntheticCapture('@reference.receiver', structName, structName.text), + ...(traitArguments === null + ? {} + : { + '@reference.type-arguments': nodeToCapture('@reference.type-arguments', traitArguments), + }), }); }); return out; diff --git a/gitnexus/src/core/ingestion/languages/rust/qualified-call.ts b/gitnexus/src/core/ingestion/languages/rust/qualified-call.ts index 65d1dd5f4..54047c6ff 100644 --- a/gitnexus/src/core/ingestion/languages/rust/qualified-call.ts +++ b/gitnexus/src/core/ingestion/languages/rust/qualified-call.ts @@ -33,6 +33,7 @@ */ import type { ParsedFile, Scope, ScopeId, SymbolDefinition } from 'gitnexus-shared'; +import { perFileSet } from '../../import-resolvers/per-file-set.js'; import { isOverloadableCallable } from '../../utils/callable-labels.js'; import { lookupBindingsAt } from '../../scope-resolution/scope/walkers.js'; import type { ScopeResolutionIndexes } from '../../model/scope-resolution-indexes.js'; @@ -53,16 +54,9 @@ import { * The hook is invoked per call site; rebuilding the index each time would make * qualified-call resolution O(sites x files). */ -const MODULE_INDEX_CACHE = new WeakMap, RustModuleIndex>(); - -function moduleIndexFor(allFilePaths: ReadonlySet): RustModuleIndex { - let index = MODULE_INDEX_CACHE.get(allFilePaths); - if (index === undefined) { - index = buildRustModuleIndex(allFilePaths); - MODULE_INDEX_CACHE.set(allFilePaths, index); - } - return index; -} +const moduleIndexFor = perFileSet( + (allFilePaths: ReadonlySet): RustModuleIndex => buildRustModuleIndex(allFilePaths), +); export function resolveRustQualifiedFreeCall( site: { readonly name: string; readonly rawQualifiedName?: string; readonly inScope: ScopeId }, @@ -488,6 +482,15 @@ interface PassModuleIndex { readonly inlineModuleKeys: ReadonlySet; } +/** + * DELIBERATELY NOT ON `import-resolvers/per-file-set.ts` (#2909 sweep), unlike + * {@link moduleIndexFor} above. {@link passIndexFor} takes THREE inputs — + * `workspaceIndex`, `index` and `scopes` — and keys on the first alone; the + * builder reads `scopes.defs.byId` and `index`, neither of which is derivable + * from the key, and `perFileSet`'s `build: (key) => T` hands the builder + * nothing but the key. Sound here only because all three share the resolution + * pass's lifetime, which is an invariant the primitive cannot express. + */ const MODULE_SCOPE_CACHE = new WeakMap(); function moduleKey(module: RustModule): string { diff --git a/gitnexus/src/core/ingestion/languages/rust/query.ts b/gitnexus/src/core/ingestion/languages/rust/query.ts index 762efb336..f2ef4f4a0 100644 --- a/gitnexus/src/core/ingestion/languages/rust/query.ts +++ b/gitnexus/src/core/ingestion/languages/rust/query.ts @@ -22,15 +22,18 @@ const RUST_SCOPE_QUERY = ` ;; Declarations — struct (struct_item - name: (type_identifier) @declaration.name) @declaration.struct + name: (type_identifier) @declaration.name + type_parameters: (type_parameters)? @declaration.type-parameters) @declaration.struct ;; Declarations — trait (trait_item - name: (type_identifier) @declaration.name) @declaration.trait + name: (type_identifier) @declaration.name + type_parameters: (type_parameters)? @declaration.type-parameters) @declaration.trait ;; Declarations — enum (enum_item - name: (type_identifier) @declaration.name) @declaration.enum + name: (type_identifier) @declaration.name + type_parameters: (type_parameters)? @declaration.type-parameters) @declaration.enum ;; Declarations — union ;; Deliberately tagged @declaration.struct (→ Struct label), NOT a @@ -42,7 +45,8 @@ const RUST_SCOPE_QUERY = ` ;; constructor, so Struct is both the resolvable and the semantically ;; honest label here. #1934 F71. (union_item - name: (type_identifier) @declaration.name) @declaration.struct + name: (type_identifier) @declaration.name + type_parameters: (type_parameters)? @declaration.type-parameters) @declaration.struct ;; Declarations — module (mod foo { ... } / mod foo;) ;; A Rust mod is an ITEM, not just a lexical region: rustc resolves the first diff --git a/gitnexus/src/core/ingestion/languages/rust/scope-resolver.ts b/gitnexus/src/core/ingestion/languages/rust/scope-resolver.ts index 5fd0f1570..bea53d9fd 100644 --- a/gitnexus/src/core/ingestion/languages/rust/scope-resolver.ts +++ b/gitnexus/src/core/ingestion/languages/rust/scope-resolver.ts @@ -16,6 +16,7 @@ import { import type { ScopeResolutionIndexes } from '../../model/scope-resolution-indexes.js'; import { resolveDefGraphId } from '../../scope-resolution/graph-bridge/ids.js'; import type { GraphNodeLookup } from '../../scope-resolution/graph-bridge/node-lookup.js'; +import type { HeritageTypeArgumentSink } from '../../scope-resolution/utils/generic-instantiation.js'; import type { KnowledgeGraph } from '../../../graph/types.js'; import { generateId } from '../../../../lib/utils.js'; @@ -54,6 +55,7 @@ function emitRustTraitImplEdges( parsedFiles: readonly ParsedFile[], nodeLookup: GraphNodeLookup, scopes: ScopeResolutionIndexes | undefined, + recordTypeArguments?: HeritageTypeArgumentSink, ): void { if (scopes === undefined) return; @@ -83,6 +85,14 @@ function emitRustTraitImplEdges( const traitGraphId = resolveDefGraphId(traitDef.filePath, traitDef, nodeLookup); if (structGraphId === undefined || traitGraphId === undefined) continue; + // The instantiation the impl was written with — `impl Validator + // for V` (#2912). Recorded against THIS edge's ids, not the pre-pass's: + // the pre-pass sources its edge from the enclosing def, and interface + // dispatch crosses the corrected one emitted here. + if (site.typeArguments !== undefined) { + recordTypeArguments?.(structGraphId, traitGraphId, site.typeArguments); + } + const edgeKey = `${structGraphId}->${traitGraphId}`; if (emitted.has(edgeKey)) continue; emitted.add(edgeKey); @@ -159,8 +169,8 @@ export const rustScopeResolver: ScopeResolver = { buildMro: (graph, parsedFiles, nodeLookup) => buildRustMro(graph, parsedFiles, nodeLookup), - emitHeritageEdges: (graph, parsedFiles, nodeLookup, scopes) => - emitRustTraitImplEdges(graph, parsedFiles, nodeLookup, scopes), + emitHeritageEdges: (graph, parsedFiles, nodeLookup, scopes, recordTypeArguments) => + emitRustTraitImplEdges(graph, parsedFiles, nodeLookup, scopes, recordTypeArguments), populateOwners: (parsed: ParsedFile) => populateRustOwners(parsed), diff --git a/gitnexus/src/core/ingestion/languages/swift/import-target.ts b/gitnexus/src/core/ingestion/languages/swift/import-target.ts index 5c0c2f662..e0d5e8219 100644 --- a/gitnexus/src/core/ingestion/languages/swift/import-target.ts +++ b/gitnexus/src/core/ingestion/languages/swift/import-target.ts @@ -25,6 +25,7 @@ */ import type { ParsedImport, WorkspaceIndex } from 'gitnexus-shared'; +import { perFileSet } from '../../import-resolvers/per-file-set.js'; export interface SwiftResolveContext { readonly fromFile: string; @@ -39,12 +40,7 @@ interface SwiftModuleIndex { readonly byModule: Map; } -const SWIFT_MODULE_INDEX_CACHE = new WeakMap, SwiftModuleIndex>(); - -function getSwiftModuleIndex(allFilePaths: ReadonlySet): SwiftModuleIndex { - const cached = SWIFT_MODULE_INDEX_CACHE.get(allFilePaths); - if (cached !== undefined) return cached; - +const getSwiftModuleIndex = perFileSet((allFilePaths: ReadonlySet): SwiftModuleIndex => { const byModule = new Map(); for (const raw of allFilePaths) { const norm = raw.replace(/\\/g, '/'); @@ -66,10 +62,8 @@ function getSwiftModuleIndex(allFilePaths: ReadonlySet): SwiftModuleInde } } - const index: SwiftModuleIndex = { byModule }; - SWIFT_MODULE_INDEX_CACHE.set(allFilePaths, index); - return index; -} + return { byModule }; +}); export function resolveSwiftImportTarget( parsedImport: ParsedImport, diff --git a/gitnexus/src/core/ingestion/languages/typescript.ts b/gitnexus/src/core/ingestion/languages/typescript.ts index 2ccfbb577..c24ab6ae8 100644 --- a/gitnexus/src/core/ingestion/languages/typescript.ts +++ b/gitnexus/src/core/ingestion/languages/typescript.ts @@ -124,6 +124,7 @@ import { jsMergeBindings, jsArityCompatibility, } from './javascript/index.js'; +import { extractDispatchGuardRoutes } from '../route-extractors/dispatch-guard.js'; /** * TypeScript/JavaScript: arrow_function and function_expression are @@ -454,6 +455,10 @@ export const typescriptProvider = defineLanguage({ receiverBinding: tsReceiverBinding, arityCompatibility: typescriptArityCompatibility, resolveImportTarget: resolveTsImportTarget, + // A raw `node:http` server declares its routes by comparing the request path + // to a literal; nothing else in this pipeline can see that shape. TS and JS + // share the grammar, so they share the extractor. + extractDecoratorRoutes: extractDispatchGuardRoutes, }); export const javascriptProvider = defineLanguage({ @@ -526,4 +531,6 @@ export const javascriptProvider = defineLanguage({ mergeBindings: (_scope, bindings) => jsMergeBindings(bindings), receiverBinding: jsReceiverBinding, arityCompatibility: jsArityCompatibility, + // See the TypeScript provider above. + extractDecoratorRoutes: extractDispatchGuardRoutes, }); diff --git a/gitnexus/src/core/ingestion/languages/typescript/captures.ts b/gitnexus/src/core/ingestion/languages/typescript/captures.ts index b3d9315ee..2e3a76290 100644 --- a/gitnexus/src/core/ingestion/languages/typescript/captures.ts +++ b/gitnexus/src/core/ingestion/languages/typescript/captures.ts @@ -6,7 +6,7 @@ * synthesized streams on top: * * 1. **Import decomposition** — each `import_statement` / re-export is - * re-emitted with `@import.kind/source/name/alias/typeOnly` markers so + * re-emitted with `@import.kind/source/name/alias/type-only` markers so * `interpretTsImport` can recover the `ParsedImport` shape without * re-parsing raw text (see `import-decomposer.ts`). Unit 2 adds this; * until then, raw `@import.statement` matches flow through as-is. diff --git a/gitnexus/src/core/ingestion/languages/typescript/file-candidates.ts b/gitnexus/src/core/ingestion/languages/typescript/file-candidates.ts new file mode 100644 index 000000000..421959788 --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/typescript/file-candidates.ts @@ -0,0 +1,70 @@ +/** + * Turning a resolved stem into a real file, the way TypeScript does (#2953). + * + * Shared by `module-resolution.ts` and the package-manifest resolver so both + * try the same three shapes — exact path, extension, directory index — and, as + * importantly, the same NARROW extension list. The repo-wide `EXTENSIONS` in + * `import-resolvers/utils.ts` carries ~39 entries spanning every language the + * indexer supports; a TypeScript import cannot resolve to a `.py` or `.rb` + * file, and letting it try was part of how the old suffix matcher found files + * that had nothing to do with the import. + */ + +/** Extension candidates, in the order TypeScript tries them. */ +export const TS_EXTENSIONS = [ + '.ts', + '.tsx', + '.d.ts', + '.mts', + '.cts', + '.js', + '.jsx', + '.mjs', + '.cjs', + '.vue', + '.json', +] as const; + +/** + * JS-family extensions a specifier may carry for a TypeScript source file. + * + * TypeScript ESM requires the specifier to name the EMITTED file (`./m.js`) + * while the file on disk is `./m.ts`, so a resolver that only tried the literal + * extension would miss every ESM-style relative import in a modern codebase. + */ +export const JS_TO_TS: ReadonlyMap = new Map([ + ['.js', ['.ts', '.tsx', '.d.ts']], + ['.jsx', ['.tsx']], + ['.mjs', ['.mts']], + ['.cjs', ['.cts']], +]); + +/** + * A repo-relative stem resolved to a real indexed file, or `null`. + * + * Exact match, then the ESM `.js` → `.ts` rewrite, then each extension, then + * the directory-index form. Nothing here searches: every candidate is derived + * from the stem the caller already resolved from a declared source. + */ +export function resolveFile(stem: string, allFiles: ReadonlySet): string | null { + if (stem === '') return null; + if (allFiles.has(stem)) return stem; + + const dot = stem.lastIndexOf('.'); + const ext = dot === -1 ? '' : stem.slice(dot); + const tsEquivalents = JS_TO_TS.get(ext); + if (tsEquivalents !== undefined) { + const stripped = stem.slice(0, -ext.length); + for (const candidate of tsEquivalents) { + if (allFiles.has(stripped + candidate)) return stripped + candidate; + } + } + + for (const candidate of TS_EXTENSIONS) { + if (allFiles.has(stem + candidate)) return stem + candidate; + } + for (const candidate of TS_EXTENSIONS) { + if (allFiles.has(`${stem}/index${candidate}`)) return `${stem}/index${candidate}`; + } + return null; +} diff --git a/gitnexus/src/core/ingestion/languages/typescript/import-decomposer.ts b/gitnexus/src/core/ingestion/languages/typescript/import-decomposer.ts index babcb45da..c79046026 100644 --- a/gitnexus/src/core/ingestion/languages/typescript/import-decomposer.ts +++ b/gitnexus/src/core/ingestion/languages/typescript/import-decomposer.ts @@ -29,7 +29,34 @@ * Type-only constructs (`import type { X }`, `import { type X }`, * `export type { X }`) emit the same kinds as runtime forms — at the * TypeScript scope-resolution layer, types and values share the same - * lookup; runtime-emission is a downstream concern. + * lookup, so the KIND is unchanged. They additionally carry an + * `@import.type-only` marker, because `tsc` deletes them: no `require` / + * `import` for the source module survives in the emitted JavaScript, so + * the pair cannot force a module-initialization order. `check --cycles` + * is the consumer — see `graph-bridge/imports-to-edges.ts`. + * + * Both spellings put the `type` keyword in a different place, so both are + * read (see `hasTypeKeyword`): + * + * import type { X, Y } from './m' — anonymous `type` token on the + * `import_statement`, covering EVERY + * specifier it decomposes to + * import { type X, Y } from './m' — anonymous `type` token on the + * `import_specifier`, covering only X + * + * The marker is therefore per-specifier, which is what makes the mixed + * statement come out right: `X` is erased, `Y` is not, and the pair + * `./m` is a real initialization dependency because of `Y`. Emission + * dedupes per `(sourceFile, targetFile)` and lets any non-erased edge win, + * so a statement counts as type-only exactly when every specifier it + * decomposes to is — without this file having to aggregate anything. + * + * Known gap: TypeScript 5.0's `export type * from './m'` / `export type * + * as ns from './m'`. The vendored grammar does not parse them — the bare + * `type` token lands in an `ERROR` node beside the `*`, not as a statement + * child — so they emit no marker and are treated as value imports. That is + * the fail-safe direction (`check --cycles` over-reports rather than + * hiding a real cycle), and neither form appears in this repository. * * Side-effect imports (`import './polyfill'`) produce a single match * with `kind: 'side-effect'`. The shared finalize algorithm resolves @@ -73,6 +100,66 @@ interface ImportSpec { /** Set on `dynamic` kind imports when the argument is a string literal — * enables `interpretTsImport` to emit `dynamic-resolved`. */ readonly literalSource?: boolean; + /** This specifier is erased by `tsc` (`import type` / `{ type X }`) — + * enables `interpretTsImport` to set `ParsedImport.typeOnly`. */ + readonly typeOnly?: boolean; +} + +/** + * Cheap prefilter for {@link hasTypeKeyword}. + * + * The keyword's text is exactly `type`, and every specifier lies inside its + * statement's text, so a statement whose text holds no `type` substring + * anywhere cannot carry the token at either level. Sound in one direction only, + * which is the direction that matters: it can admit a statement that turns out + * to have no keyword (`import { getType }`), never reject one that has it. + * + * Worth the extra test because the two are not the same order of cost. + * `node.text` is one slice; {@link hasTypeKeyword} crosses the N-API boundary + * and allocates a node wrapper once per direct child, and it runs per statement + * AND per specifier — so a statement of N specifiers pays N+1 walks. + * + * It pays most where there is nothing to find, and that case is not rare: + * `javascript/captures.ts` shares this decomposer, JavaScript has no + * `import type` at all, and no `.js` import can contain the token — so every + * JavaScript file was paying the full walk, never early-exiting, for an answer + * that is structurally always `false`. + */ +function mayHaveTypeKeyword(stmtNode: SyntaxNode): boolean { + return stmtNode.text.includes('type'); +} + +/** + * Does this node carry the `type` keyword that erases the import? + * + * The keyword is an ANONYMOUS token, so `findChild` (named children only) + * cannot see it and the direct child list has to be walked. Two nodes are + * ever asked: + * + * - `import_statement` / `export_statement` — `import type { X } from './m'` + * - `import_specifier` / `export_specifier` — `import { type X } from './m'` + * + * Only DIRECT children are considered. A nested `type` token means something + * else entirely — `export type Foo = Bar` puts one inside the child + * `type_alias_declaration` — and a subtree scan would read those as erasure. + * No NAMED node in this grammar is called `type`, so matching the type name + * alone identifies the keyword without asking about `isNamed`, whose spelling + * differs between tree-sitter bindings. + * + * `field-extractors/configs/helpers.ts`'s `hasKeyword` walks the same direct + * children and must NOT be reused here, for a sharper reason than the walk: it + * matches on `child.text.trim()`, not on the node type. `import type from './m'` + * is a DEFAULT import binding the name `type`, and its `import_clause`'s whole + * text is `type` — so `hasKeyword` reports erasure for an import that really + * runs, and the pair would be dropped from cycle reporting. Matching the token's + * TYPE is what separates the keyword from an identifier that happens to spell + * it. + */ +function hasTypeKeyword(node: SyntaxNode): boolean { + for (let i = 0; i < node.childCount; i++) { + if (node.child(i)?.type === 'type') return true; + } + return false; } /** @@ -117,6 +204,13 @@ function splitImport(stmtNode: SyntaxNode): CaptureMatch[] { ]; } + // `import type ...` erases every specifier in the statement. `import + // { type X, Y }` erases only the marked ones, which is read per specifier + // in `decomposeNamedSpecifier`. Default and namespace forms have no + // per-specifier spelling, so the statement keyword is all there is. + const mayHaveType = mayHaveTypeKeyword(stmtNode); + const statementTypeOnly = mayHaveType && hasTypeKeyword(stmtNode); + const out: CaptureMatch[] = []; // An import_clause can have any combination of: // - leading identifier (default import) @@ -135,6 +229,7 @@ function splitImport(stmtNode: SyntaxNode): CaptureMatch[] { name: 'default', alias: child.text, atNode: child, + typeOnly: statementTypeOnly, }), ); continue; @@ -151,6 +246,7 @@ function splitImport(stmtNode: SyntaxNode): CaptureMatch[] { name: source, alias: aliasId.text, atNode: child, + typeOnly: statementTypeOnly, }), ); } @@ -161,14 +257,20 @@ function splitImport(stmtNode: SyntaxNode): CaptureMatch[] { for (let j = 0; j < child.namedChildCount; j++) { const spec = child.namedChild(j); if (spec === null || spec.type !== 'import_specifier') continue; - const decomposed = decomposeNamedSpecifier(spec, source, stmtNode); + const decomposed = decomposeNamedSpecifier( + spec, + source, + stmtNode, + statementTypeOnly, + mayHaveType, + ); if (decomposed !== null) out.push(decomposed); } continue; } - // Other children (e.g. `type` keyword token for `import type { ... }`) - // are ignored — they carry no per-specifier info; we fold type-only - // semantics into the same emitted kinds. + // No other named children exist on an `import_clause`. The `type` + // keyword of `import type { ... }` is an ANONYMOUS token on the + // statement, not a clause child, and is read by `hasTypeKeyword` above. } return out; @@ -179,13 +281,20 @@ function splitImport(stmtNode: SyntaxNode): CaptureMatch[] { * * - `{ X }` → named * - `{ X as Y }` → named-alias - * - `{ type X }` → named (type-only; same shape) - * - `{ type X as Y }` → named-alias (type-only) + * - `{ type X }` → named (+ `@import.type-only`) + * - `{ type X as Y }` → named-alias (+ `@import.type-only`) + * + * `statementTypeOnly` is the `import type { … }` form, which erases this + * specifier regardless of what the specifier itself spells; the two are + * ORed rather than one overriding the other, because `import type { type X }` + * is legal-ish input and both spellings mean the same erasure. */ function decomposeNamedSpecifier( spec: SyntaxNode, source: string, stmtNode: SyntaxNode, + statementTypeOnly: boolean, + mayHaveType: boolean, ): CaptureMatch | null { // `import_specifier` layout: // name: identifier @@ -199,6 +308,7 @@ function decomposeNamedSpecifier( const aliasNode = spec.childForFieldName('alias'); if (nameNode === null) return null; const name = nameNode.text; + const typeOnly = statementTypeOnly || (mayHaveType && hasTypeKeyword(spec)); if (aliasNode !== null && aliasNode.startIndex !== nameNode.startIndex) { return buildImportMatch(stmtNode, { @@ -207,6 +317,7 @@ function decomposeNamedSpecifier( name, alias: aliasNode.text, atNode: spec, + typeOnly, }); } return buildImportMatch(stmtNode, { @@ -214,6 +325,7 @@ function decomposeNamedSpecifier( source, name, atNode: spec, + typeOnly, }); } @@ -234,13 +346,24 @@ function splitReexport(stmtNode: SyntaxNode): CaptureMatch[] { const source = extractSource(stmtNode); if (source === null) return []; + // `export type { X } from './m'`. Its `export type *` sibling is NOT + // detectable — see the known gap in the module header. + const mayHaveType = mayHaveTypeKeyword(stmtNode); + const statementTypeOnly = mayHaveType && hasTypeKeyword(stmtNode); + const exportClause = findChild(stmtNode, 'export_clause'); if (exportClause !== null) { const out: CaptureMatch[] = []; for (let i = 0; i < exportClause.namedChildCount; i++) { const spec = exportClause.namedChild(i); if (spec === null || spec.type !== 'export_specifier') continue; - const decomposed = decomposeReexportSpecifier(spec, source, stmtNode); + const decomposed = decomposeReexportSpecifier( + spec, + source, + stmtNode, + statementTypeOnly, + mayHaveType, + ); if (decomposed !== null) out.push(decomposed); } return out; @@ -273,6 +396,7 @@ function splitReexport(stmtNode: SyntaxNode): CaptureMatch[] { name: source, alias: aliasId.text, atNode: namespaceExport, + typeOnly: statementTypeOnly, }), buildNamespaceDeclarationMatch(namespaceExport, aliasId), ]; @@ -292,15 +416,20 @@ function splitReexport(stmtNode: SyntaxNode): CaptureMatch[] { ]; } +/** Mirror of {@link decomposeNamedSpecifier} for `export { … } from './m'`, + * including the per-specifier `export { type X } from './m'` spelling. */ function decomposeReexportSpecifier( spec: SyntaxNode, source: string, stmtNode: SyntaxNode, + statementTypeOnly: boolean, + mayHaveType: boolean, ): CaptureMatch | null { const nameNode = spec.childForFieldName('name'); const aliasNode = spec.childForFieldName('alias'); if (nameNode === null) return null; const name = nameNode.text; + const typeOnly = statementTypeOnly || (mayHaveType && hasTypeKeyword(spec)); if (aliasNode !== null && aliasNode.startIndex !== nameNode.startIndex) { return buildImportMatch(stmtNode, { @@ -309,6 +438,7 @@ function decomposeReexportSpecifier( name, alias: aliasNode.text, atNode: spec, + typeOnly, }); } return buildImportMatch(stmtNode, { @@ -316,6 +446,7 @@ function decomposeReexportSpecifier( source, name, atNode: spec, + typeOnly, }); } @@ -420,6 +551,12 @@ function buildImportMatch(stmtNode: SyntaxNode, spec: ImportSpec): CaptureMatch if (spec.literalSource === true) { m['@import.literal'] = syntheticCapture('@import.literal', spec.atNode, ''); } + // Presence-only, like `@import.literal`: absent means "not erased", so the + // marker is added rather than spelled `'false'`, and every non-TypeScript + // provider's matches keep the shape they already have. + if (spec.typeOnly === true) { + m['@import.type-only'] = syntheticCapture('@import.type-only', spec.atNode, ''); + } return m; } diff --git a/gitnexus/src/core/ingestion/languages/typescript/import-target.ts b/gitnexus/src/core/ingestion/languages/typescript/import-target.ts index 7d39e1f64..2100a7717 100644 --- a/gitnexus/src/core/ingestion/languages/typescript/import-target.ts +++ b/gitnexus/src/core/ingestion/languages/typescript/import-target.ts @@ -1,44 +1,35 @@ /** * Adapter from `(ParsedImport, WorkspaceIndex)` → concrete file path. * - * Delegates to the existing standard-strategy resolver - * (`resolveImportPath`) so tsconfig path aliases (`@/`, `~/`, …) and - * suffix-based resolution follow the same rules as the legacy path. + * Delegates to `module-resolution.ts`, which runs the algorithm `tsc` and Node + * actually run. It used to delegate to the shared `resolveImportPath`, whose + * final step was `suffixResolve` — a repo-wide search for any file path ending + * in the specifier. That is what #2953 removed: this path now resolves only + * against declared inputs (real paths, tsconfig `paths`/`baseUrl`, package + * manifests) and answers `null` for everything else. * - * The `WorkspaceIndex` is opaque at the shared contract layer; we - * narrow it to a TypeScript-shaped context that carries `fromFile` + - * the full `allFilePaths` set + the optional `tsconfigPaths` the - * resolver reads. + * The `WorkspaceIndex` is opaque at the shared contract layer; we narrow it to + * a TypeScript-shaped context carrying `fromFile`, the workspace file set, and + * the two config indexes the algorithm reads. * * Returning `null` lets the finalize algorithm mark the edge as - * `linkStatus: 'unresolved'`. + * `linkStatus: 'unresolved'` — which for an external package is the correct + * and complete answer. */ import type { ParsedImport, WorkspaceIndex } from 'gitnexus-shared'; -import { SupportedLanguages } from 'gitnexus-shared'; -import { resolveImportPath } from '../../import-resolvers/standard.js'; -import type { SuffixIndex } from '../../import-resolvers/utils.js'; -import type { TsconfigPaths } from '../../language-config.js'; +import type { NodeWorkspacePackages } from '../../import-resolvers/node-workspace-packages.js'; +import { resolveTsModule } from './module-resolution.js'; +import type { TsconfigIndex } from './tsconfig.js'; export interface TsResolveContext { readonly fromFile: string; - /** Mutable `Set` because the standard resolver consumes `Set`. - * Callers holding a `ReadonlySet` should copy via `new Set(...)`. */ - readonly allFilePaths: Set; - /** Repo file list, normalized (lowercased) for suffix matching. May - * be supplied by the orchestrator; if absent we derive it on the - * fly from `allFilePaths`. */ - readonly allFileList?: readonly string[]; - readonly normalizedFileList?: readonly string[]; - /** Per-call resolution cache to dedupe repeated lookups. */ - readonly resolveCache?: Map; - /** Prebuilt suffix index for O(1)-style package/absolute import matching. */ - readonly index?: SuffixIndex; - /** Parsed tsconfig path-aliases. `null` = no aliases configured. */ - readonly tsconfigPaths?: TsconfigPaths | null; - /** JavaScript vs TypeScript switch — affects the extensions the - * resolver tries. Defaults to TypeScript. */ - readonly language?: SupportedLanguages.TypeScript | SupportedLanguages.JavaScript; + /** The workspace file set. */ + readonly allFilePaths: ReadonlySet; + /** Every tsconfig in the repo; `null` when the repo declares none. */ + readonly tsconfigs?: TsconfigIndex | null; + /** Every in-repo `package.json`; `null` when the repo declares none. */ + readonly nodeWorkspacePackages?: NodeWorkspacePackages | null; } export function resolveTsImportTarget( @@ -59,36 +50,21 @@ export function resolveTsImportTarget( } /** - * Resolve a raw module-path string to a workspace file path using the - * same standard-strategy resolver as the legacy DAG. Operates directly on - * the source string without requiring a `ParsedImport`, so the - * `ScopeResolver.resolveImportTarget` adapter doesn't need to construct - * a fake `ParsedImport` to reach the resolver. + * Resolve a raw module-path string to a workspace file path. Operates directly + * on the source string without requiring a `ParsedImport`, so the + * `ScopeResolver.resolveImportTarget` adapter doesn't need to construct a fake + * one to reach the resolver. * - * Returns `null` when: - * - the context is malformed (missing `fromFile` / `allFilePaths`) - * - `targetRaw` is empty - * - the resolver finds no matching file + * Returns `null` when `targetRaw` is empty, names an external package, or names + * something no declared config maps into the repo. */ export function resolveTsTarget(targetRaw: string, ctx: TsResolveContext): string | null { - if (targetRaw === '') return null; - - const language = ctx.language ?? SupportedLanguages.TypeScript; - const allFileList = ctx.allFileList ?? Array.from(ctx.allFilePaths); - const normalizedFileList = ctx.normalizedFileList ?? allFileList.map((f) => f.toLowerCase()); - const resolveCache = ctx.resolveCache ?? new Map(); - - return resolveImportPath( - ctx.fromFile, - targetRaw, - ctx.allFilePaths, - allFileList as string[], - normalizedFileList as string[], - resolveCache, - language, - ctx.tsconfigPaths ?? null, - ctx.index, - ); + return resolveTsModule(targetRaw, { + fromFile: ctx.fromFile, + allFilePaths: ctx.allFilePaths, + tsconfigs: ctx.tsconfigs ?? null, + workspacePackages: ctx.nodeWorkspacePackages ?? null, + }); } function narrowTsContext(workspaceIndex: WorkspaceIndex): TsResolveContext | null { diff --git a/gitnexus/src/core/ingestion/languages/typescript/interpret.ts b/gitnexus/src/core/ingestion/languages/typescript/interpret.ts index acf511527..30236f95f 100644 --- a/gitnexus/src/core/ingestion/languages/typescript/interpret.ts +++ b/gitnexus/src/core/ingestion/languages/typescript/interpret.ts @@ -8,7 +8,8 @@ * * The import matches arrive pre-decomposed by `emitTsScopeCaptures` * (one imported name per match, with synthesized - * `@import.kind/source/name/alias` markers — see `import-decomposer.ts`). + * `@import.kind/source/name/alias/type-only` markers — see + * `import-decomposer.ts`). * The type-binding matches arrive straight from the raw query captures — * each `@type-binding.*` anchor carries `@type-binding.name` + * `@type-binding.type`. @@ -16,14 +17,18 @@ import type { CaptureMatch, ParsedImport, ParsedTypeBinding, TypeRef } from 'gitnexus-shared'; +/** Shared empty result for the non-type-only path — see `typeOnly` below. */ +const NO_TYPE_ONLY: { typeOnly?: true } = Object.freeze({}); + // ─── interpretImport ────────────────────────────────────────────────────── export function interpretTsImport(captures: CaptureMatch): ParsedImport | null { // Markers attached by `splitImportStatement` (import-decomposer.ts): - // @import.kind : one of the kinds documented there - // @import.name : imported name from the source module - // @import.alias : local alias name (for default / aliased / namespace forms) - // @import.source : module path (always present except dynamic-unresolved) + // @import.kind : one of the kinds documented there + // @import.name : imported name from the source module + // @import.alias : local alias name (for default / aliased / namespace forms) + // @import.source : module path (always present except dynamic-unresolved) + // @import.type-only : presence-only — this specifier is erased by `tsc` const kindCap = captures['@import.kind']; const nameCap = captures['@import.name']; const aliasCap = captures['@import.alias']; @@ -32,6 +37,15 @@ export function interpretTsImport(captures: CaptureMatch): ParsedImport | null { const kind = kindCap?.text; if (kind === undefined) return null; + // Spread rather than `typeOnly: ` so a value import keeps the exact + // object shape it had before this marker existed — every `ParsedImport` + // equality assertion in the suite compares whole objects. + // `NO_TYPE_ONLY` is shared rather than a fresh `{}` per import: the spread + // reads it and never retains it, and the ~99% of imports that are not + // type-only would otherwise each allocate an object to contribute nothing. + const typeOnly: { typeOnly?: true } = + captures['@import.type-only'] !== undefined ? { typeOnly: true } : NO_TYPE_ONLY; + switch (kind) { case 'default': { // `import D from './m'` — semantically "alias for the module's @@ -45,6 +59,7 @@ export function interpretTsImport(captures: CaptureMatch): ParsedImport | null { importedName: 'default', alias: aliasCap.text, targetRaw: sourceCap.text, + ...typeOnly, }; } case 'named': { @@ -55,6 +70,7 @@ export function interpretTsImport(captures: CaptureMatch): ParsedImport | null { localName: nameCap.text, importedName: nameCap.text, targetRaw: sourceCap.text, + ...typeOnly, }; } case 'named-alias': { @@ -68,6 +84,7 @@ export function interpretTsImport(captures: CaptureMatch): ParsedImport | null { importedName: nameCap.text, alias: aliasCap.text, targetRaw: sourceCap.text, + ...typeOnly, }; } case 'namespace': { @@ -78,6 +95,7 @@ export function interpretTsImport(captures: CaptureMatch): ParsedImport | null { localName: aliasCap.text, importedName: sourceCap.text, targetRaw: sourceCap.text, + ...typeOnly, }; } case 'reexport': { @@ -88,6 +106,7 @@ export function interpretTsImport(captures: CaptureMatch): ParsedImport | null { localName: nameCap.text, importedName: nameCap.text, targetRaw: sourceCap.text, + ...typeOnly, }; } case 'reexport-alias': { @@ -101,6 +120,7 @@ export function interpretTsImport(captures: CaptureMatch): ParsedImport | null { importedName: nameCap.text, alias: aliasCap.text, targetRaw: sourceCap.text, + ...typeOnly, }; } case 'reexport-wildcard': { @@ -119,6 +139,7 @@ export function interpretTsImport(captures: CaptureMatch): ParsedImport | null { localName: aliasCap.text, importedName: sourceCap.text, targetRaw: sourceCap.text, + ...typeOnly, }; } case 'dynamic': { @@ -310,3 +331,26 @@ function stripQualifier(text: string): string { if (lastDot === -1) return text; return text.slice(lastDot + 1); } + +/** + * Would this interpreter reduce `text` to the type it CONTAINS rather than to + * the type it names? True for the array suffix (`Repo[]`) and for every + * transparent wrapper on {@link stripGeneric}'s list (`Array`, + * `Promise`, `Set`, …). + * + * Exported for the ONE caller that must decline exactly what this returns true + * for: the JavaScript provider's JSDoc `@type` FIELD binding (#2833). Element + * reduction is right where it was built — a chain step, a `for…of` variable, an + * awaited value — and wrong for a field, whose declared type IS the container: + * a field annotated `{Repo[]}` reduced to `Repo` makes `this.repos.find(…)`, + * an Array method call, resolve to a repository class's own `find`. A wrong + * edge, which #2833 treats as strictly worse than a missing one. + * + * A predicate rather than a copied name list on purpose: the list lives in + * `stripGeneric` and a second copy would drift out of sync silently, exactly + * the failure mode `python/interpret.ts` records for its own reduction. + */ +export function reducesToContainedType(text: string): boolean { + const trimmed = stripReadonly(text.trim()); + return stripArraySuffix(trimmed) !== trimmed || stripGeneric(trimmed) !== trimmed; +} diff --git a/gitnexus/src/core/ingestion/languages/typescript/module-resolution.ts b/gitnexus/src/core/ingestion/languages/typescript/module-resolution.ts new file mode 100644 index 000000000..ea8d2d637 --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/typescript/module-resolution.ts @@ -0,0 +1,199 @@ +/** + * TypeScript / JavaScript module resolution (#2953). + * + * This is the algorithm `tsc` and Node actually run, in the order they run it. + * It replaces `import-resolvers/utils.ts:suffixResolve` on the TS/JS/Vue path, + * which answered a different question — "does any file in this repo have a path + * ending in this specifier?" — and answered it by dropping leading segments + * until something matched. That is why `@acme/telemetry/nest`, a registry + * dependency, landed on the repo's only path ending in `nest/index.ts`. + * + * Every rule below resolves against something DECLARED: a real path, a + * `tsconfig` mapping, or a `package.json` manifest. A specifier that matches + * none of them is external, and external resolves to nothing. There is + * deliberately no fallback: a guess is what this module exists to remove, and + * an edge nobody declared is worse than a missing one precisely because it + * cannot be told apart from a real one downstream. + * + * ## The order, and why it is this order + * + * 1. relative / absolute — a path is a path; nothing else can claim it. + * 2. `#`-prefixed — package.json `imports`, which is scoped to the importing + * package and shadows everything else by design. + * 3. tsconfig `paths` — explicit mappings win over `baseUrl`, and the LONGEST + * matching pattern wins among them (tsc's rule, not first-declared). + * 4. tsconfig `baseUrl` — the rule that makes `import 'src/utils/foo'` legal. + * Note it applies only when a config actually declares one; without it, + * TypeScript treats a non-relative specifier as a package lookup, and so + * does this module. + * 5. workspace package — the manifest map, resolved through that package's + * own `exports` / `main` / `module` / `types`. + * 6. anything else — external. `null`. + */ + +import type { NodeWorkspacePackages } from '../../import-resolvers/node-workspace-packages.js'; +import { + matchSubpathMap, + nodePackageNameOf, + owningPackage, + resolveNodeWorkspaceImport, + substituteStar, +} from '../../import-resolvers/node-workspace-packages.js'; +import { resolveFile } from './file-candidates.js'; +import { tsconfigFor, type TsconfigIndex, type TsPathMapping } from './tsconfig.js'; + +export interface TsModuleResolutionContext { + readonly fromFile: string; + readonly allFilePaths: ReadonlySet; + readonly tsconfigs: TsconfigIndex | null; + readonly workspacePackages: NodeWorkspacePackages | null; +} + +/** + * Resolve one specifier to a repo file, or `null` when nothing in the repo + * declares it. + */ +export function resolveTsModule(specifier: string, ctx: TsModuleResolutionContext): string | null { + if (specifier === '') return null; + + // 1. A path specifier. + if (specifier.startsWith('.')) { + const joined = joinFrom(ctx.fromFile, specifier); + return joined === null ? null : resolveFile(joined, ctx.allFilePaths); + } + if (specifier.startsWith('/')) { + return resolveFile(specifier.slice(1), ctx.allFilePaths); + } + + // 2. Package-internal `#imports`. Scoped to the importing package, so it is + // looked up there and nowhere else — a `#` specifier that the package does + // not declare is an error in Node, not a repo-wide search. + if (specifier.startsWith('#')) { + return resolveSubpathImport(specifier, ctx); + } + + const config = tsconfigFor(ctx.tsconfigs, ctx.fromFile); + + // 3. `paths`, longest matching pattern first. + if (config !== null && config.paths.length > 0) { + const viaPaths = resolveViaPaths(specifier, config.paths, ctx.allFilePaths); + if (viaPaths !== null) return viaPaths; + } + + // 4. `baseUrl`. + if (config !== null && config.baseUrl !== null) { + const viaBaseUrl = resolveFile(joinRepo(config.baseUrl, specifier), ctx.allFilePaths); + if (viaBaseUrl !== null) return viaBaseUrl; + } + + // 5. A package that lives in this repo. + const viaWorkspace = resolveNodeWorkspaceImport( + specifier, + ctx.workspacePackages, + ctx.allFilePaths, + ); + if (viaWorkspace !== null) return viaWorkspace; + + // 6. External. Nothing in the repo declared it, so it resolves to nothing — + // which for a registry dependency is the correct and complete answer. + return null; +} + +/** + * Apply `paths` the way tsc does: the pattern with the longest literal prefix + * before `*` wins, and its targets are tried in declaration order. + * + * The old loader kept `targets[0]` and treated the pattern as a plain prefix, + * which silently mis-resolves the common `"@/*": ["./src/*", "./generated/*"]` + * shape — the second target is where half of a generated-code monorepo lives. + */ +function resolveViaPaths( + specifier: string, + paths: readonly TsPathMapping[], + allFiles: ReadonlySet, +): string | null { + const matches: { mapping: TsPathMapping; stem: string | null; prefixLength: number }[] = []; + + for (const mapping of paths) { + const star = mapping.pattern.indexOf('*'); + if (star === -1) { + if (mapping.pattern === specifier) { + matches.push({ mapping, stem: null, prefixLength: mapping.pattern.length }); + } + continue; + } + const prefix = mapping.pattern.slice(0, star); + const suffix = mapping.pattern.slice(star + 1); + if (!specifier.startsWith(prefix) || !specifier.endsWith(suffix)) continue; + if (specifier.length < prefix.length + suffix.length) continue; + matches.push({ + mapping, + stem: specifier.slice(prefix.length, specifier.length - suffix.length), + prefixLength: prefix.length, + }); + } + + // An exact (starless) pattern outranks any wildcard, THEN longer prefix wins. + // Sorting on prefix length alone left that first rule to luck: `a` and `a*` + // both match `a` with prefix length 1, so whichever was declared first won. + matches.sort( + (a, b) => Number(a.stem !== null) - Number(b.stem !== null) || b.prefixLength - a.prefixLength, + ); + + for (const match of matches) { + for (const target of match.mapping.targets) { + const candidate = match.stem === null ? target : substituteStar(target, match.stem); + const resolved = resolveFile(candidate, allFiles); + if (resolved !== null) return resolved; + } + } + return null; +} + +/** Resolve `#name` against the importing file's own package manifest. */ +function resolveSubpathImport(specifier: string, ctx: TsModuleResolutionContext): string | null { + const packages = ctx.workspacePackages; + if (packages === null) return null; + const owner = owningPackage(ctx.fromFile, packages); + if (owner === null) return null; + // `imports` takes pattern keys (`"#internal/*"`) exactly like `exports`, so + // it gets the same matcher rather than an exact lookup. + for (const stem of matchSubpathMap(owner.subpathImports, specifier) ?? []) { + const resolved = resolveFile(stem, ctx.allFilePaths); + if (resolved !== null) return resolved; + } + return null; +} + +/** + * Resolve a relative specifier against the importing file's directory, or + * `null` when it climbs out of the repository. + * + * Popping an empty segment list would silently CLAMP at the root, so + * `../../../secret` from `src/main.ts` became `secret` and could resolve a + * repo-root file the specifier never named. Outside the repo there is nothing + * indexed to resolve to, so the honest answer is nothing. + */ +function joinFrom(fromFile: string, specifier: string): string | null { + const segments = fromFile.split('/').slice(0, -1); + for (const part of specifier.split('/')) { + if (part === '.' || part === '') continue; + if (part === '..') { + if (segments.length === 0) return null; + segments.pop(); + } else { + segments.push(part); + } + } + return segments.join('/'); +} + +function joinRepo(dir: string, rest: string): string { + return dir === '' ? rest : `${dir}/${rest}`; +} + +/** Whether a specifier names a package rather than a path — used by callers + * that want to report an unresolved import as external rather than missing. */ +export function isPackageSpecifier(specifier: string): boolean { + return nodePackageNameOf(specifier) !== null; +} diff --git a/gitnexus/src/core/ingestion/languages/typescript/query.ts b/gitnexus/src/core/ingestion/languages/typescript/query.ts index c67f8f777..babdb273b 100644 --- a/gitnexus/src/core/ingestion/languages/typescript/query.ts +++ b/gitnexus/src/core/ingestion/languages/typescript/query.ts @@ -150,30 +150,45 @@ export const TYPESCRIPT_SCOPE_QUERY = ` value: (object_type)) @scope.class ;; Declarations — types +;; The type-parameter list is captured with \`?\` rather than as a second +;; pattern: a separate rule would make a GENERIC declaration match twice, and +;; both matches mint the same def id (filePath+range+type+name), so which one +;; survived — the one carrying the parameters or the one without — would be +;; decided by match order. An optional child keeps it at one match either way. (class_declaration - name: (type_identifier) @declaration.name) @declaration.class + name: (type_identifier) @declaration.name + type_parameters: (type_parameters)? @declaration.type-parameters) @declaration.class (abstract_class_declaration - name: (type_identifier) @declaration.name) @declaration.class + name: (type_identifier) @declaration.name + type_parameters: (type_parameters)? @declaration.type-parameters) @declaration.class (interface_declaration - name: (type_identifier) @declaration.name) @declaration.interface + name: (type_identifier) @declaration.name + type_parameters: (type_parameters)? @declaration.type-parameters) @declaration.interface (enum_declaration name: (identifier) @declaration.name) @declaration.enum +;; Tagged @declaration.type_alias, NOT @declaration.type: normalizeNodeLabel +;; accepts typealias / type_alias and has no "type" case, so the old tag mapped +;; to no label and TypeScript aliases produced NO scope-resolution def at all. +;; Kotlin and Dart already spell it this way. (type_alias_declaration - name: (type_identifier) @declaration.name) @declaration.type + name: (type_identifier) @declaration.name + type_parameters: (type_parameters)? @declaration.type-parameters) @declaration.type_alias (internal_module name: (identifier) @declaration.name) @declaration.namespace ;; Declarations — methods / functions / constructors (function_declaration - name: (identifier) @declaration.name) @declaration.function + name: (identifier) @declaration.name + type_parameters: (type_parameters)? @declaration.type-parameters) @declaration.function (generator_function_declaration - name: (identifier) @declaration.name) @declaration.function + name: (identifier) @declaration.name + type_parameters: (type_parameters)? @declaration.type-parameters) @declaration.function ;; Function overload signatures (declaration-only; body in a separate ;; function_declaration). Extractors dedup by (name, parameterTypes). @@ -498,6 +513,48 @@ export const TYPESCRIPT_SCOPE_QUERY = ` (method_signature name: (property_identifier) @declaration.name) @declaration.method +;; Members of a declared SHAPE — interface bodies and object-type aliases both +;; spell them as property_signature (A4). The sibling method_signature rule +;; above declared interface METHODS, so only properties were missing: a typed +;; receiver resolved to the shape's scope and then found no member there, and +;; the field's consumers were unreachable. TypeScript sets +;; fieldFallbackOnMethodLookup:false, so there is no name-based safety net +;; here — the precise path is the only one, and it needs the declaration. +;; ANCHORED to declared shapes — see the matching rule in TYPESCRIPT_QUERIES +;; for why. Unanchored this matched inline parameter and return types and +;; nested object types, whose members then collided onto the enclosing +;; class/interface/alias. +(interface_body + (property_signature + name: (property_identifier) @declaration.name) @declaration.property) + +;; Object-literal keys of a NAMED object — the scope-resolution half of the +;; matching rule in TYPESCRIPT_QUERIES. The parse query mints the Property NODE; +;; this mints the DEF a precise read can resolve to. +(variable_declarator + name: (identifier) + value: (object + (pair + key: (property_identifier) @declaration.name) @declaration.property)) + +(variable_declarator + name: (identifier) + value: (call_expression + function: (member_expression + object: (identifier) @_ts.identity.obj + property: (property_identifier) @_ts.identity.fn) + arguments: (arguments + (object + (pair + key: (property_identifier) @declaration.name) @declaration.property))) + (#eq? @_ts.identity.obj "Object") + (#match? @_ts.identity.fn "^(freeze|seal|preventExtensions)$")) + +(type_alias_declaration + value: (object_type + (property_signature + name: (property_identifier) @declaration.name) @declaration.property)) + ;; Declarations — class fields (public_field_definition name: (property_identifier) @declaration.name) @declaration.property @@ -1204,6 +1261,65 @@ export const TYPESCRIPT_SCOPE_QUERY = ` (object (shorthand_property_identifier) @reference.name @reference.property-key @reference.value-ref) + +;; Bare-identifier reads (A2), VALUE POSITIONS ONLY — a blanket \`(identifier)\` +;; rule would mint a site for every token in the file. +;; +;; These existed only in the JavaScript query, so A2 did not work for +;; TypeScript AT ALL: a \`.ts\` module reading its own \`const\` by bare name +;; produced no reference site, and "who uses this constant?" answered a +;; confident zero for an entire language. Found by writing the namespace +;; fixture below and watching it fail for the wrong reason. +(arguments + (identifier) @reference.name @reference.read.identifier) + +(assignment_pattern + right: (identifier) @reference.name @reference.read.identifier) + +(return_statement + (identifier) @reference.name @reference.read.identifier) + +;; \`const next = LIMIT\` and \`n > LIMIT\` — both plainly value reads, and both +;; named in review as gaps between what A2 claimed and what it matched. +(variable_declarator + value: (identifier) @reference.name @reference.read.identifier) + +(binary_expression + left: (identifier) @reference.name @reference.read.identifier) + +(binary_expression + right: (identifier) @reference.name @reference.read.identifier) + +;; References — TYPE POSITION (R2-2). An annotation naming a declared type is +;; the only thing that makes that type's declaration reachable from the code +;; that depends on it, and TypeScript captured none: only cpp and csharp emitted +;; type references at all. So an exported API-contract type owned its members +;; (round 1) but had \`incoming: {}\`, and "what breaks if I remove this field?" +;; — the question a contract type exists to answer — had no edge to walk. +;; +;; The resolution path was already complete on the other side: +;; \`type-reference\` routes to the ClassRegistry, whose CLASS_KINDS already +;; lists TypeAlias, Interface and Enum, and \`edges.ts\` already maps the kind to +;; USES. Only the capture was missing. +;; +;; Anchored to the CONTEXTS a type is used in — annotations, type arguments, +;; and heritage \`implements\` — never a bare \`(type_identifier)\`. A blanket rule +;; would also match the identifier in \`type X = …\` and \`interface X\`, making +;; every declaration a consumer of itself. +(type_annotation + (type_identifier) @reference.name @reference.type_reference) + +(type_annotation + (generic_type + name: (type_identifier) @reference.name @reference.type_reference)) + +(type_arguments + (type_identifier) @reference.name @reference.type_reference) + +;; \`x as SomeType\` / \`satisfies SomeType\` — an assertion is a claim ABOUT a +;; declared type, so the code making it depends on that declaration. +(as_expression + (type_identifier) @reference.name @reference.type_reference) `; /** diff --git a/gitnexus/src/core/ingestion/languages/typescript/scope-resolver.ts b/gitnexus/src/core/ingestion/languages/typescript/scope-resolver.ts index fe7a30da4..bc2af4bcb 100644 --- a/gitnexus/src/core/ingestion/languages/typescript/scope-resolver.ts +++ b/gitnexus/src/core/ingestion/languages/typescript/scope-resolver.ts @@ -21,15 +21,13 @@ import type { ScopeResolver } from '../../scope-resolution/contract/scope-resolv import { simpleKey } from '../../scope-resolution/graph-bridge/node-lookup.js'; import type { ScopeResolutionIndexes } from '../../model/scope-resolution-indexes.js'; import { typescriptProvider } from '../typescript.js'; -import { loadTsconfigPaths, type TsconfigPaths } from '../../language-config.js'; -import { buildSuffixIndex, type SuffixIndex } from '../../import-resolvers/utils.js'; -import { indexOnlyElementType } from '../../type-extractors/shared.js'; +import { loadTsconfigIndex, type TsconfigIndex } from './tsconfig.js'; import { - typescriptArityCompatibility, - typescriptMergeBindings, - resolveTsTarget, - type TsResolveContext, -} from './index.js'; + loadNodeWorkspacePackages, + type NodeWorkspacePackages, +} from '../../import-resolvers/node-workspace-packages.js'; +import { indexOnlyElementType } from '../../type-extractors/shared.js'; +import { typescriptArityCompatibility, typescriptMergeBindings, resolveTsTarget } from './index.js'; import { getNuxtAutoImportEntry, hasNuxtAutoImports, @@ -39,7 +37,10 @@ import { /** Shape the orchestrator threads in via `RunScopeResolutionInput.resolutionConfig`. */ interface TypescriptResolutionConfig { - readonly tsconfigPaths: TsconfigPaths | null; + /** Every tsconfig in the repo, `extends` resolved (#2953). */ + readonly tsconfigs: TsconfigIndex | null; + /** Every in-repo `package.json`, for workspace-package resolution (#2953). */ + readonly nodeWorkspacePackages: NodeWorkspacePackages | null; /** Nuxt/Nitro auto-import map. Null for non-Nuxt projects. */ readonly nuxtAutoImports: NuxtAutoImportConfig | null; } @@ -55,54 +56,22 @@ const TYPESCRIPT_TYPE_ONLY_BINDING_TYPES = new Set([ ]); /** - * Build a `resolveImportTarget` adapter that memoizes the workspace - * file list, the lower-cased file list, and the per-pass `resolveCache` - * across every import lookup in a single workspace pass. The - * orchestrator passes the same `ReadonlySet` reference for every call - * within a pass — we use that identity to detect when the workspace - * changes and recompute the derived state lazily. + * Build the `resolveImportTarget` adapter. * - * Without this memoization, `resolveTsTarget` re-derived - * `allFileList` and `normalizedFileList` (both O(N_files)) and threw - * away the `resolveCache` on every import — O(N_files × N_imports) - * total work for what should be O(N_files + N_imports). + * No per-file-set memo any more: the suffix index it existed to amortize is + * gone with #2953. Real resolution derives nothing from the file list — every + * candidate comes from a config the repo declares, and checking one is a + * `Set.has` — so there is nothing left to cache per pass. */ function makeTsResolveImportTarget(): ScopeResolver['resolveImportTarget'] { - interface PassCache { - readonly key: ReadonlySet; - readonly allFilePaths: Set; - readonly allFileList: readonly string[]; - readonly normalizedFileList: readonly string[]; - readonly index: SuffixIndex; - readonly resolveCache: Map; - } - let cached: PassCache | null = null; - return (targetRaw, fromFile, allFilePaths, resolutionConfig) => { - if (cached === null || cached.key !== allFilePaths) { - const allFileList = Array.from(allFilePaths); - const normalizedFileList = allFileList.map((f) => f.toLowerCase()); - cached = { - key: allFilePaths, - allFilePaths: new Set(allFilePaths), - allFileList, - normalizedFileList, - index: buildSuffixIndex(normalizedFileList, allFileList), - resolveCache: new Map(), - }; - } - const cfg = resolutionConfig as TypescriptResolutionConfig | undefined; - const ws: TsResolveContext = { + return resolveTsTarget(targetRaw, { fromFile, - allFilePaths: cached.allFilePaths, - allFileList: cached.allFileList, - normalizedFileList: cached.normalizedFileList, - index: cached.index, - resolveCache: cached.resolveCache, - tsconfigPaths: cfg?.tsconfigPaths ?? null, - }; - return resolveTsTarget(targetRaw, ws); + allFilePaths, + tsconfigs: cfg?.tsconfigs ?? null, + nodeWorkspacePackages: cfg?.nodeWorkspacePackages ?? null, + }); }; } @@ -128,7 +97,8 @@ const typescriptScopeResolver: ScopeResolver = { // `nuxtAutoImports` is null for non-Nuxt projects (no .nuxt/imports.d.ts), // so this adds zero overhead to ordinary TypeScript repos. loadResolutionConfig: async (repoPath: string) => ({ - tsconfigPaths: await loadTsconfigPaths(repoPath), + tsconfigs: await loadTsconfigIndex(repoPath), + nodeWorkspacePackages: await loadNodeWorkspacePackages(repoPath), nuxtAutoImports: await loadNuxtAutoImports(repoPath), }), diff --git a/gitnexus/src/core/ingestion/languages/typescript/tsconfig.ts b/gitnexus/src/core/ingestion/languages/typescript/tsconfig.ts new file mode 100644 index 000000000..a111f5752 --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/typescript/tsconfig.ts @@ -0,0 +1,338 @@ +/** + * Real `tsconfig.json` loading for module resolution (#2953). + * + * The previous loader (`language-config.ts:loadTsconfigPaths`) was built to feed + * a heuristic, and it shows: it reads three filenames at the repo ROOT only, + * gives up unless `compilerOptions.paths` exists, keeps only `targets[0]` of + * each mapping, and treats a pattern as a plain prefix. That is enough to make + * a guess look plausible and not enough to resolve anything correctly: + * + * - a monorepo has one tsconfig PER PACKAGE, and `apps/web/tsconfig.json` is + * what governs `apps/web/src/main.ts` — the root config governs nothing; + * - `extends` is how essentially every real config is written, and the + * `baseUrl` / `paths` almost always live in the extended base; + * - `baseUrl` alone (no `paths`) is a complete resolution rule on its own, and + * it is exactly the rule that makes `import 'src/utils/foo'` legal — the + * case the old suffix matcher was really standing in for; + * - `paths` maps a pattern to an ORDERED LIST of targets, tried in order. + * + * So this module answers the question TypeScript actually asks: for THIS file, + * what are `baseUrl` and `paths`? + */ + +import fs from 'fs/promises'; +import path from 'path'; + +import { isHardcodedIgnoredDirectory } from '../../../../config/ignore-service.js'; +import { logger } from '../../../logger.js'; + +/** One `paths` entry, pattern and targets kept in declaration order. */ +export interface TsPathMapping { + /** The pattern as written, e.g. `@/*`, `@app/*`, `exact`. */ + readonly pattern: string; + /** Targets as written, relative to `baseUrl`. Tried in order. */ + readonly targets: readonly string[]; +} + +/** The resolution-relevant part of one resolved tsconfig. */ +export interface TsconfigScope { + /** Repo-relative directory the config governs (the tsconfig's own directory). */ + readonly dir: string; + /** + * Repo-relative `baseUrl`, or `null` when the config declares none. + * + * `null` is not the same as `'.'`: without `baseUrl`, TypeScript does NOT + * resolve non-relative specifiers against the project at all (they are + * package lookups), and `paths` targets are resolved against the tsconfig's + * own directory instead. + */ + readonly baseUrl: string | null; + readonly paths: readonly TsPathMapping[]; +} + +/** Every tsconfig in the repo, indexed so the nearest one to a file wins. */ +export interface TsconfigIndex { + /** Deepest-first, so the first `dir` that prefixes a file path governs it. */ + readonly scopes: readonly TsconfigScope[]; +} + +const SCAN_MAX_DIRS = 20_000; +const SCAN_MAX_DEPTH = 24; +/** Guard against an `extends` cycle or a pathological chain. */ +const MAX_EXTENDS_DEPTH = 16; + +/** + * The config governing `filePath` — the nearest tsconfig at or above it. + * + * TypeScript resolves a file against the project that includes it; the nearest + * enclosing tsconfig is the faithful approximation of that without evaluating + * `include`/`exclude` globs, and it is what makes a monorepo's per-package + * `baseUrl` apply to that package's files instead of the root's. + */ +export function tsconfigFor(index: TsconfigIndex | null, filePath: string): TsconfigScope | null { + if (index === null) return null; + for (const scope of index.scopes) { + if (scope.dir === '') return scope; + if (filePath.startsWith(`${scope.dir}/`)) return scope; + } + return null; +} + +/** Load every tsconfig in the repo, resolving `extends` chains. */ +export async function loadTsconfigIndex(repoRoot: string): Promise { + const files = await findTsconfigFiles(repoRoot); + if (files.length === 0) return null; + + const ranked: { scope: TsconfigScope; rank: number }[] = []; + for (const absPath of files) { + const options = await readCompilerOptions(absPath, 0); + if (options === null) continue; + // `readCompilerOptions` resolves both to ABSOLUTE paths against whichever + // config in the `extends` chain declared them, which is the only way the + // chain stays unambiguous. Rebasing to repo-relative happens once, here. + const baseUrl = options.baseUrl === undefined ? null : repoRelative(repoRoot, options.baseUrl); + const paths = (options.paths ?? []).map((mapping) => ({ + pattern: mapping.pattern, + targets: mapping.targets.map((t) => rebaseTarget(repoRoot, t)), + })); + // A config declaring NEITHER is kept, not skipped. Dropping it let + // `tsconfigFor` fall through to an enclosing config, so a package whose own + // tsconfig declares no `baseUrl` — meaning its non-relative specifiers are + // package lookups — silently inherited the repo root's aliases instead. + // An empty scope is the accurate answer for such a file, and only a scope + // can express it. + ranked.push({ + scope: { dir: repoRelative(repoRoot, path.dirname(absPath)), baseUrl, paths }, + rank: configRank(path.basename(absPath)), + }); + } + if (ranked.length === 0) return null; + + // Deepest first, because `tsconfigFor` takes the first match and it must be + // the most specific config rather than whichever the walk reached first. + // + // Then by filename rank WITHIN a directory, which is the half that is easy to + // miss: `tsconfig.json` and `tsconfig.base.json` routinely sit side by side, + // and the base exists to be extended, not to govern. Reading whichever the + // directory listing returned first made a config's own `paths` invisible + // whenever its base happened to be listed earlier. + ranked.sort((a, b) => b.scope.dir.length - a.scope.dir.length || a.rank - b.rank); + return { scopes: ranked.map((entry) => entry.scope) }; +} + +/** + * Precedence among configs sharing a directory: the project config governs, and + * everything else is a base or a variant that exists to be extended. + */ +function configRank(fileName: string): number { + if (fileName === 'tsconfig.json') return 0; + if (fileName === 'jsconfig.json') return 1; + return 2; +} + +/** Resolved compiler options, rebased to repo-relative paths. */ +interface ResolvedOptions { + baseUrl?: string; + paths?: TsPathMapping[]; +} + +/** + * Read one tsconfig and merge in whatever it `extends`. + * + * Rebasing happens per FILE, before merging, because `extends` does not rebase + * `baseUrl`: a base config at `configs/tsconfig.base.json` declaring + * `"baseUrl": "."` means `configs/`, even when extended from `apps/web`. Doing + * the rebase at read time is what keeps that true through the chain. + */ +async function readCompilerOptions( + absPath: string, + depth: number, + repoRootHint?: string, +): Promise { + if (depth > MAX_EXTENDS_DEPTH) { + logger.warn(`[typescript] tsconfig extends chain too deep at ${absPath}; ignoring the rest`); + return null; + } + + let parsed: Record; + try { + parsed = parseJsonc(await fs.readFile(absPath, 'utf-8')); + } catch { + return null; + } + + const dir = path.dirname(absPath); + // Read what this config extends FIRST: `paths` targets resolve against the + // EFFECTIVE `baseUrl`, which a config declaring `paths` alone inherits from + // its base. Resolving them against this config's own directory instead would + // load the right alias pattern and point every target at the wrong place. + const inherited = await readExtended(parsed.extends, dir, depth, repoRootHint); + + const own: ResolvedOptions = {}; + const compilerOptions = parsed.compilerOptions; + if (compilerOptions !== null && typeof compilerOptions === 'object') { + const opts = compilerOptions as Record; + if (typeof opts.baseUrl === 'string') { + own.baseUrl = path.resolve(dir, opts.baseUrl); + } + if (opts.paths !== null && typeof opts.paths === 'object' && !Array.isArray(opts.paths)) { + // tsc resolves `paths` targets against the effective `baseUrl` — this + // config's own if it declares one, otherwise the inherited one — and + // against the config's own directory only when neither exists. Doing it + // here, per file, is what keeps an `extends` chain unambiguous: by the + // time these merge, every target is already absolute. + const pathsBase = own.baseUrl ?? inherited?.baseUrl ?? dir; + own.paths = []; + for (const [pattern, targets] of Object.entries(opts.paths as Record)) { + if (!Array.isArray(targets)) continue; + const asStrings = targets + .filter((t): t is string => typeof t === 'string') + .map((t) => path.resolve(pathsBase, t)); + if (asStrings.length > 0) own.paths.push({ pattern, targets: asStrings }); + } + } + } + + // Own options win over inherited ones — that is what `extends` means. `paths` + // is replaced wholesale rather than merged, matching tsc. + return { + ...(inherited ?? {}), + ...own, + }; +} + +/** Follow `extends`, which may be a string or (TS 5+) an array, base-first. */ +async function readExtended( + value: unknown, + fromDir: string, + depth: number, + repoRootHint?: string, +): Promise { + const specs = typeof value === 'string' ? [value] : Array.isArray(value) ? value : []; + let merged: ResolvedOptions | null = null; + for (const spec of specs) { + if (typeof spec !== 'string') continue; + const resolved = await resolveExtendsTarget(spec, fromDir); + if (resolved === null) continue; + const options = await readCompilerOptions(resolved, depth + 1, repoRootHint); + if (options === null) continue; + // Later entries win over earlier ones, per tsc's array semantics. + merged = { ...(merged ?? {}), ...options }; + } + return merged; +} + +/** + * An `extends` value is either a path or a package name. + * + * The package form (`"extends": "@tsconfig/node20/tsconfig.json"`, + * `"@acme/tsconfig"`) lives in `node_modules`, which this tool deliberately + * does NOT index — it is dependency code, not the repository's own. But not + * indexing it is different from not READING it, and the distinction matters + * here: a shared internal base config is exactly where a monorepo puts the + * `paths` its packages import through, so refusing to open it loses aliases + * that the repository genuinely declares. + * + * So the file is read from disk when it is there, walking `node_modules` up + * from the extending config the way Node does. When it is absent — an + * un-installed checkout, which is a shape a static analyser must expect and a + * compiler may refuse — the answer is `null`, and the caller keeps whatever the + * extending config declared itself. That degrades to fewer resolutions, never + * to invented ones. + */ +async function resolveExtendsTarget(spec: string, fromDir: string): Promise { + if (spec.startsWith('.') || path.isAbsolute(spec)) { + return firstReadableConfig(path.resolve(fromDir, spec)); + } + for (const modulesDir of nodeModulesChain(fromDir)) { + const found = await firstReadableConfig(path.join(modulesDir, spec)); + if (found !== null) return found; + } + return null; +} + +/** `/node_modules`, then each ancestor's, the way Node resolves. */ +function* nodeModulesChain(fromDir: string): Generator { + let dir = fromDir; + for (;;) { + if (path.basename(dir) !== 'node_modules') yield path.join(dir, 'node_modules'); + const parent = path.dirname(dir); + if (parent === dir) return; + dir = parent; + } +} + +/** The first spelling of `base` that is a readable file. */ +async function firstReadableConfig(base: string): Promise { + for (const candidate of [base, `${base}.json`, path.join(base, 'tsconfig.json')]) { + try { + const stat = await fs.stat(candidate); + if (stat.isFile()) return candidate; + } catch { + // try the next spelling + } + } + return null; +} + +async function findTsconfigFiles(repoRoot: string): Promise { + const found: string[] = []; + const queue: { dir: string; depth: number }[] = [{ dir: repoRoot, depth: 0 }]; + let dirsScanned = 0; + + while (queue.length > 0 && dirsScanned < SCAN_MAX_DIRS) { + const { dir, depth } = queue.shift()!; + dirsScanned++; + let entries: import('fs').Dirent[]; + try { + entries = await fs.readdir(dir, { withFileTypes: true }); + } catch { + continue; + } + for (const entry of entries) { + if (entry.isDirectory()) { + if (isHardcodedIgnoredDirectory(entry.name)) continue; + if (depth < SCAN_MAX_DEPTH) + queue.push({ dir: path.join(dir, entry.name), depth: depth + 1 }); + continue; + } + if (!entry.isFile()) continue; + // `tsconfig.json`, `tsconfig.app.json`, `jsconfig.json`, … — any of them + // can carry the `baseUrl`/`paths` that governs its directory. + if (/^(ts|js)config(\..+)?\.json$/.test(entry.name)) { + found.push(path.join(dir, entry.name)); + } + } + } + return found; +} + +/** Strip comments and trailing commas — tsconfig is JSONC, not JSON. */ +function parseJsonc(raw: string): Record { + const withoutComments = raw + .replace(/\\"|"(?:\\"|[^"])*"|(\/\/.*$)|(\/\*[\s\S]*?\*\/)/gm, (match, line, block) => + line !== undefined || block !== undefined ? '' : match, + ) + .replace(/,(\s*[}\]])/g, '$1'); + return JSON.parse(withoutComments) as Record; +} + +function repoRelative(repoRoot: string, absDir: string): string { + const rel = path.relative(repoRoot, absDir).split(path.sep).join('/'); + return rel === '.' || rel === '' ? '' : rel; +} + +/** + * A `paths` target rebased to repo-relative, keeping any trailing `*`. + * + * `path.resolve` swallows the wildcard into a path segment, so it is stripped + * before resolving and re-appended after — the `*` is a substitution marker, + * not a directory named `*`. + */ +function rebaseTarget(repoRoot: string, absTarget: string): string { + // `/repo/src/*` must come back as `src/*`, not `src*`: stripping only the + // star leaves a trailing slash that `path.relative` then eats. + const suffix = absTarget.endsWith('/*') ? '/*' : absTarget.endsWith('*') ? '*' : ''; + const base = suffix === '' ? absTarget : absTarget.slice(0, -suffix.length); + return `${repoRelative(repoRoot, base)}${suffix}`; +} diff --git a/gitnexus/src/core/ingestion/languages/vue/import-target.ts b/gitnexus/src/core/ingestion/languages/vue/import-target.ts index a16877459..886580b0e 100644 --- a/gitnexus/src/core/ingestion/languages/vue/import-target.ts +++ b/gitnexus/src/core/ingestion/languages/vue/import-target.ts @@ -1,81 +1,36 @@ /** * Import-target resolver for Vue SFCs (RFC #909 Ring 3, issue #940). * - * Vue `