mirror of
https://github.com/abhigyanpatwari/GitNexus.git
synced 2026-10-04 02:31:36 +00:00
Merge branch 'main' into fix/skill-evolution-gate
This commit is contained in:
commit
609c28859b
38 changed files with 3536 additions and 414 deletions
13
.github/workflows/ci-tests.yml
vendored
13
.github/workflows/ci-tests.yml
vendored
|
|
@ -500,6 +500,14 @@ jobs:
|
|||
run: node --import tsx bench/callable-value-flow/measure.mjs --check
|
||||
working-directory: gitnexus
|
||||
|
||||
- name: C++ qualified-namespace resolution guards (#2788)
|
||||
# Build-free: asserts resolveCppQualifiedNamespaceMember resolves an
|
||||
# unchanged symbol set (fingerprint) and that per-call-site cost stays
|
||||
# independent of corpus size. Rationale and history: see the header of
|
||||
# bench/cpp-qualified-ns/measure.mjs.
|
||||
run: node --import tsx bench/cpp-qualified-ns/measure.mjs --check
|
||||
working-directory: gitnexus
|
||||
|
||||
- name: Receiver-resolution drop guards
|
||||
# NOT build-free: this one runs the real pipeline, so it needs dist/
|
||||
# (the setup action above builds). ~2m15s.
|
||||
|
|
@ -566,12 +574,17 @@ jobs:
|
|||
working-directory: gitnexus
|
||||
|
||||
- name: Cross-language pipeline benchmarks (GITNEXUS_BENCH, serial)
|
||||
# cpp-adl-benchmark.test.ts is not a `*-pipeline-benchmark.test.ts` but
|
||||
# belongs here for the same reason: it is skipIf-gated on GITNEXUS_BENCH,
|
||||
# so it had never run in CI and the PR #1990 ADL emit-scaling guard it
|
||||
# holds was dead. ~45s of test time.
|
||||
env:
|
||||
GITNEXUS_BENCH: '1'
|
||||
run: >-
|
||||
npx vitest run --no-file-parallelism
|
||||
test/integration/cobol-pipeline-benchmark.test.ts
|
||||
test/integration/csharp-pipeline-benchmark.test.ts
|
||||
test/integration/cpp-adl-benchmark.test.ts
|
||||
test/integration/instance-ownership-pipeline-benchmark.test.ts
|
||||
test/integration/spring-bean-resource-benchmark.test.ts
|
||||
test/integration/rust-pipeline-benchmark.test.ts
|
||||
|
|
|
|||
7
gitnexus/bench/cpp-qualified-ns/baselines.json
Normal file
7
gitnexus/bench/cpp-qualified-ns/baselines.json
Normal file
|
|
@ -0,0 +1,7 @@
|
|||
{
|
||||
"_comment": "Baselines for bench/cpp-qualified-ns/measure.mjs --check (#2788). `fingerprint` is a sha256 over every `receiver::member(arity|argumentTypes) -> outcome` the synthetic corpus resolves at the LARGE scale (hit nodeId, `<ambiguous>` per #1564, or `<none>`); it is a CORRECTNESS gate, so drift means C++ qualified `ns::member()` lookup started resolving a different symbol set and must be explained, never re-baselined to make CI green. THAT RULE IS UNCHANGED and applies to every future edit of inline-namespaces.ts. `scaling_budget` is a timing gate and carries deliberate headroom for shared CI runners.",
|
||||
"_rebaseline_2788_review": "This fingerprint was moved ONCE, deliberately, during review of #2788 — because the bench CORPUS was expanded, not because a check failed. Do not read it as precedent. What changed: (1) receivers now mirror production — ~1 in 5 name a declared namespace, ~4 in 5 are plain identifiers naming none (`obj0`, `Widget3`, `buf12`). The previous corpus drew every receiver from `ns_${…}`, so the receiver lookup NEVER missed, while Case 1.5 in scope-resolution/passes/receiver-bound-calls.ts is reached by every plain-identifier receiver call and misses on the overwhelming majority. (2) A namespace reopened across two files (C++ namespaces are open — the cross-file merge property). (3) A same-name inline nest `namespace ns { inline namespace ns { … } }`, which is the only shape that observes `gatherQualifiedNsMember`'s `visited` dedup. (4) A member declared at BOTH the namespace level and in an inline child, selected apart by argument type, pinning both collection sources by nodeId. (5) Call sites carrying a real `Callsite`, without which narrowOverloadCandidates / cppConversionRank / isOverloadAmbiguousAfterNormalization were outside the fingerprinted surface entirely. Measured effect, same patched resolver, old bench vs new: removing the `visited` dedup — old PASS with a byte-identical fingerprint, new FAIL (fingerprint 1e6c51b9… != aba39c34…); resetting `visited` per root instead of across roots — old PASS, new FAIL on the same fingerprint. A PURE reorder of a candidate list still passes both, and correctly so: the resolver's return contract is order-blind by construction (see QualifiedNsMemberIndex's doc comment), so there is no behaviour there to gate.",
|
||||
"fingerprint": "aba39c342ce536006bebded8b32260dc7807487be91f5c7ee9548f9e9283f9c9",
|
||||
"scaling_budget": 1.8,
|
||||
"_scaling_note": "(t_large/t_small)/(1600/400). ~1.0 is linear. OBSERVED BAND: 1.28-1.45 over ten unloaded runs on a 24-core dev box. The band this file previously claimed — 0.93-1.21 — did not reproduce and was an artifact: the small arm then measured ~1.7 ms, small enough that timer granularity and JIT warm-up, not scaling, set the number (the same ten-run sweep of that bench spanned 1.11-1.40). CALLS_PER_FILE is now sized so the small arm lands at ~14 ms; that halves the unloaded spread (0.30 -> 0.16) and costs ~2.0 s of wall time for the whole bench. The residual above 1.0 is real and not a defect: at LARGE the index and corpus are 4x the working set, so per-call-site locality is worse (~85 ns/site vs ~63 ns) while the algorithm stays linear. TRIAGE: a scaling failure is a TIMING signal — RE-RUN IT on an idle machine before investigating. Runner contention dominates everything above: pinned to 2 CPUs against 2 spinners the identical binary produced 1.16-2.18, i.e. a spurious FAIL, and the sibling bench/callable-value-flow drifts out of its own documented band the same way. The fingerprint arm is the opposite — it is deterministic; a re-run never changes it and must never be used to wish it away. Floor check: a per-call-site workspace rescan reintroduced ONLY on the receiver-bucket-absent path (the most plausible way #2788 returns) measures 4.538 at these same 400/1600 file scales — 812x slower on the small arm, 2850x on the large — while leaving the fingerprint byte-identical. The old always-hits corpus scored that same patch 1.279 and printed PASS. Resolution is timed alone; the fingerprint's outcome strings are built in a separate untimed pass because their allocation cost grows with the corpus and would otherwise show up as scaling."
|
||||
}
|
||||
496
gitnexus/bench/cpp-qualified-ns/measure.mjs
Normal file
496
gitnexus/bench/cpp-qualified-ns/measure.mjs
Normal file
|
|
@ -0,0 +1,496 @@
|
|||
/**
|
||||
* Build-free scaling + identity bench for `resolveCppQualifiedNamespaceMember`,
|
||||
* the C++ qualified `ns::member()` receiver resolver (issue #2788).
|
||||
*
|
||||
* Before #2788 this function re-scanned EVERY parsed file — rebuilding a
|
||||
* per-file `scopesById` map each time — once per qualified call site, so the
|
||||
* scope-resolution emit phase cost O(callsites × scopes). On a 1,473-file C++
|
||||
* repo that was 25.3 min of a 33-min analyze, with 75% of total self-time in
|
||||
* this one function. It is the same bug #1990 had already fixed in the sibling
|
||||
* ADL path (`pickCppAdlCandidates` → `AdlCandidateIndex`). #1990 DID ship a
|
||||
* scaling gate for that path — `test/integration/cpp-adl-benchmark.test.ts`,
|
||||
* which asserts `emitRatio < fileRatio^1.5` — but it could never have caught
|
||||
* this one, for two independent reasons: its corpus asserts
|
||||
* `callsResolved === 0`, i.e. it generates only UNRESOLVED ADL sites, so it
|
||||
* never drives the qualified-receiver path at all; and it is
|
||||
* `describe.skipIf(!BENCH_ENABLED)` while the single CI step that sets
|
||||
* `GITNEXUS_BENCH=1` names its test files explicitly and, until this PR wired
|
||||
* it in, listed neither C++ bench — so it had never executed in CI. Even now
|
||||
* that it runs, the `callsResolved === 0` half stands: it still cannot reach
|
||||
* this path. Hence this bench, in an always-on step: a per-call-site workspace
|
||||
* scan must not be reintroduced silently.
|
||||
*
|
||||
* For a synthetic corpus at two scales it reports:
|
||||
* - `elapsed_ms` per scale (fastest of REPS, see `fastest`) for resolving
|
||||
* every call site once, INCLUDING the one-time index build — that build is
|
||||
* the work the per-site scan was traded for, so hiding it would let an
|
||||
* index that is itself quadratic pass;
|
||||
* - a scaling ratio `(t_large/t_small)/(LARGE/SMALL)`: ~1.0 linear,
|
||||
* ~4.x quadratic at this scale gap;
|
||||
* - a sha256 fingerprint over every `receiver::member(callsite) → outcome`
|
||||
* the corpus resolves, as the correctness gate. A fingerprint change means
|
||||
* qualified lookup started resolving different symbols — a behaviour
|
||||
* change, never a performance one.
|
||||
*
|
||||
* A gate only covers the code path its corpus drives. Two properties below are
|
||||
* therefore load-bearing and must not be "simplified" away:
|
||||
*
|
||||
* 1. **The receiver mix is production-shaped: ~1 in 5 receivers names a
|
||||
* namespace, the other ~4 name nothing.** Case 1.5 in
|
||||
* `scope-resolution/passes/receiver-bound-calls.ts` is reached by EVERY
|
||||
* plain-identifier receiver call — `obj.size()`, `Widget::make()`,
|
||||
* `buf.data()` — so in real source the overwhelming majority of calls into
|
||||
* this resolver are receiver MISSES, not member misses inside a resolved
|
||||
* receiver. An earlier revision of this bench drew every receiver from
|
||||
* `ns_${…}`, i.e. always a namespace the corpus declared, so the receiver
|
||||
* lookup never missed. Measured consequence: a "defensive full rescan when
|
||||
* the receiver bucket is absent" regression — the single most plausible
|
||||
* way this bug returns — scored 1.332 against the 1.8 budget and printed
|
||||
* PASS, while costing 507× on a production-shaped corpus.
|
||||
* 2. **The corpus contains every structural shape whose loss the fingerprint
|
||||
* is supposed to catch** (see `buildCorpus`), including a batch of sites
|
||||
* that pass a real `Callsite`. Without those, `narrowOverloadCandidates` /
|
||||
* `cppConversionRank` / `isOverloadAmbiguousAfterNormalization` are not in
|
||||
* the fingerprinted surface at all, and behaviour-only regressions there
|
||||
* re-fingerprint byte-identically.
|
||||
*
|
||||
* Build-free: imports the `.ts` hotpath through tsx
|
||||
* (`node --import tsx bench/cpp-qualified-ns/measure.mjs`).
|
||||
*
|
||||
* Without args: prints the JSON report.
|
||||
* With `--check`: asserts the fingerprint == the committed baseline AND the
|
||||
* scaling ratio is within budget; exits non-zero on drift/regression.
|
||||
*/
|
||||
import fs from 'node:fs';
|
||||
import path from 'node:path';
|
||||
import crypto from 'node:crypto';
|
||||
import { fileURLToPath } from 'node:url';
|
||||
|
||||
import {
|
||||
clearCppInlineNamespaces,
|
||||
markCppInlineNamespaceRange,
|
||||
populateCppInlineNamespaceScopes,
|
||||
resolveCppQualifiedNamespaceMember,
|
||||
} from '../../src/core/ingestion/languages/cpp/inline-namespaces.ts';
|
||||
|
||||
const __dirname = path.dirname(fileURLToPath(import.meta.url));
|
||||
const BASELINE_PATH = path.resolve(__dirname, 'baselines.json');
|
||||
|
||||
const SMALL = 400;
|
||||
const LARGE = 1600;
|
||||
/** Sized so the SMALL arm measures in the tens of ms rather than ~1.7 ms.
|
||||
* Sub-2 ms samples are dominated by timer granularity and scheduler noise on
|
||||
* a shared runner, which is what made the ratio drift out of its documented
|
||||
* band under load; see `_scaling_note` in baselines.json. */
|
||||
const CALLS_PER_FILE = 480;
|
||||
const REPS = 7;
|
||||
const WARMUP = 3;
|
||||
|
||||
/** 1 in N receivers names a declared namespace; the rest name nothing. Header
|
||||
* property 1 is why this ratio, and not an always-hits corpus. */
|
||||
const NS_RECEIVER_IN = 5;
|
||||
|
||||
const NO_SCOPES = {};
|
||||
|
||||
/**
|
||||
* Deterministic 32-bit avalanche (murmur3 finalizer). Stands in for
|
||||
* `Math.random()` — the corpus, the receiver mix and therefore the fingerprint
|
||||
* must be byte-reproducible across machines and Node versions.
|
||||
*/
|
||||
function mix(n) {
|
||||
let x = n >>> 0;
|
||||
x = Math.imul(x ^ (x >>> 16), 0x85ebca6b) >>> 0;
|
||||
x = Math.imul(x ^ (x >>> 13), 0xc2b2ae35) >>> 0;
|
||||
return (x ^ (x >>> 16)) >>> 0;
|
||||
}
|
||||
|
||||
/**
|
||||
* Deterministic synthetic corpus — no randomness, so the fingerprint is stable.
|
||||
*
|
||||
* Per file `f`, three top-level namespaces. Every shape here exists because
|
||||
* some behaviour of `resolveCppQualifiedNamespaceMember` is unobservable
|
||||
* without it; dropping one silently un-gates that behaviour.
|
||||
*
|
||||
* namespace ns_f { // ABI-versioning idiom (std::__1)
|
||||
* void own0(); void own1(); // direct members → hit
|
||||
* void both(int); // ALSO declared in v1 below
|
||||
* void over(int); // overload set spanning levels
|
||||
* inline namespace v1 { // transitively visible
|
||||
* void inl0(); // → hit
|
||||
* void dup();
|
||||
* void both(double); // the inline-child twin of `both`
|
||||
* void over(int, int); void over(double);
|
||||
* void same(int);
|
||||
* }
|
||||
* inline namespace v2 {
|
||||
* void dup(); // two inline children → ambiguous
|
||||
* void same(int); // identical signature → ambiguous
|
||||
* }
|
||||
* namespace detail { void hidden0(); } // NOT inline → invisible → miss
|
||||
* }
|
||||
*
|
||||
* namespace twin_f { inline namespace twin_f { void twinned(); } }
|
||||
* namespace shared_{f>>1} { void part{f&1}(); }
|
||||
*
|
||||
* What each shape gates:
|
||||
* - `own0` / `inl0` / `hidden0` / `nosuch`: the three outcome classes (hit
|
||||
* from the namespace's own defs, hit through an inline child, miss), each
|
||||
* a different exit from the resolver.
|
||||
* - `dup` across v1 and v2: `'ambiguous'` (#1564).
|
||||
* - `twin_f`: the same-name inline nest — the only shape that observes
|
||||
* `gatherQualifiedNsMember`'s `visited` dedup, without which the one
|
||||
* `twinned` is collected twice and a resolved def flips to `'ambiguous'`
|
||||
* (that function's comment explains why both scopes land on one receiver).
|
||||
* - `shared_g` declared by files 2g and 2g+1: C++ namespaces are open, so one
|
||||
* receiver's members are spread over however many files reopen it. The
|
||||
* legacy per-call-site scan got this for free; the index has to merge
|
||||
* across the whole `parsedFiles` array. `part0` and `part1` are declared in
|
||||
* DIFFERENT files and both must resolve.
|
||||
* - `both` at the namespace level and in the inline child: pins BOTH
|
||||
* collection sources by nodeId, via the two `both` call sites that select
|
||||
* between them on argument type. Drop own-def collection and the `int`
|
||||
* probe moves; drop inline-child descent and the `double` probe moves. A
|
||||
* pure REORDER of the two stays invisible, and correctly so: the return
|
||||
* contract is order-blind — see `QualifiedNsMemberIndex.rootsByReceiver`.
|
||||
* - `over` / `same` with a real `Callsite`: see `NS_PROBES`.
|
||||
*/
|
||||
function buildCorpus(fileCount) {
|
||||
const parsedFiles = [];
|
||||
for (let f = 0; f < fileCount; f++) {
|
||||
const filePath = `src/file${f}.cpp`;
|
||||
const scopes = [];
|
||||
const inlineRanges = [];
|
||||
let line = 1;
|
||||
/** Push one Namespace scope with a range unique within this file, so
|
||||
* `populateCppInlineNamespaceScopes` marks exactly the intended scopes. */
|
||||
const scope = (id, parent, ownedDefs, isInline = false) => {
|
||||
const range = { startLine: line, startCol: 0, endLine: line + 1, endCol: 0 };
|
||||
line += 2;
|
||||
scopes.push({ id, kind: 'Namespace', parent, ownedDefs, range });
|
||||
if (isInline) inlineRanges.push(range);
|
||||
return id;
|
||||
};
|
||||
const ns = (qualifiedName) => ({
|
||||
nodeId: `def:${filePath}#${qualifiedName}`,
|
||||
type: 'Namespace',
|
||||
qualifiedName,
|
||||
});
|
||||
/** A callable def. `parameterTypes` are what makes overloads distinguishable
|
||||
* both to `narrowOverloadCandidates` and — via the nodeId, exactly as the
|
||||
* real C++ node keys do it — to the fingerprint. */
|
||||
const fn = (qualifiedName, parameterTypes) =>
|
||||
parameterTypes === undefined
|
||||
? { nodeId: `def:${filePath}#${qualifiedName}`, type: 'Function', qualifiedName }
|
||||
: {
|
||||
nodeId: `def:${filePath}#${qualifiedName}(${parameterTypes.join(',')})`,
|
||||
type: 'Function',
|
||||
qualifiedName,
|
||||
parameterTypes,
|
||||
parameterCount: parameterTypes.length,
|
||||
requiredParameterCount: parameterTypes.length,
|
||||
};
|
||||
|
||||
const nsId = scope(`sc:${f}:ns`, null, [
|
||||
ns(`ns_${f}`),
|
||||
fn(`ns_${f}.own0`),
|
||||
fn(`ns_${f}.own1`),
|
||||
fn(`ns_${f}.both`, ['int']),
|
||||
fn(`ns_${f}.over`, ['int']),
|
||||
]);
|
||||
scope(
|
||||
`sc:${f}:v1`,
|
||||
nsId,
|
||||
[
|
||||
ns(`ns_${f}.v1`),
|
||||
fn(`ns_${f}.v1.inl0`),
|
||||
fn(`ns_${f}.v1.dup`),
|
||||
fn(`ns_${f}.v1.both`, ['double']),
|
||||
fn(`ns_${f}.v1.over`, ['int', 'int']),
|
||||
fn(`ns_${f}.v1.over`, ['double']),
|
||||
fn(`ns_${f}.v1.same`, ['int']),
|
||||
],
|
||||
true,
|
||||
);
|
||||
scope(
|
||||
`sc:${f}:v2`,
|
||||
nsId,
|
||||
[ns(`ns_${f}.v2`), fn(`ns_${f}.v2.dup`), fn(`ns_${f}.v2.same`, ['int'])],
|
||||
true,
|
||||
);
|
||||
scope(`sc:${f}:detail`, nsId, [ns(`ns_${f}.detail`), fn(`ns_${f}.detail.hidden0`)]);
|
||||
|
||||
const twinId = scope(`sc:${f}:twin`, null, [ns(`twin_${f}`)]);
|
||||
scope(
|
||||
`sc:${f}:twin@inner`,
|
||||
twinId,
|
||||
[ns(`twin_${f}.twin_${f}`), fn(`twin_${f}.twin_${f}.twinned`)],
|
||||
true,
|
||||
);
|
||||
|
||||
const group = f >> 1;
|
||||
scope(`sc:${f}:shared`, null, [ns(`shared_${group}`), fn(`shared_${group}.part${f & 1}`)]);
|
||||
|
||||
parsedFiles.push({ filePath, scopes, inlineRanges });
|
||||
}
|
||||
return parsedFiles;
|
||||
}
|
||||
|
||||
/** Capture-time inline marking + `populateOwners`-time scope-id resolution, in
|
||||
* the same order the pipeline runs them. Must re-run after every
|
||||
* `clearCppInlineNamespaces`, which drops both the marks and the index. */
|
||||
function populateInlineState(parsedFiles) {
|
||||
clearCppInlineNamespaces();
|
||||
for (const parsed of parsedFiles) {
|
||||
for (const range of parsed.inlineRanges) markCppInlineNamespaceRange(parsed.filePath, range);
|
||||
populateCppInlineNamespaceScopes(parsed);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Namespace-receiver probes: `[family, member, callsite]`. Drawn for the ~1 in
|
||||
* `NS_RECEIVER_IN` call sites whose receiver actually names a namespace.
|
||||
*
|
||||
* The tail entries pass a real `Callsite`, which is the only way any of
|
||||
* `narrowOverloadCandidates`, `cppConversionRank` or
|
||||
* `isOverloadAmbiguousAfterNormalization` is reached — the resolver threads
|
||||
* `callsite?.arity` / `callsite?.argumentTypes` into narrowing, and with no
|
||||
* callsite those filters are pass-throughs. Each one is chosen to land on a
|
||||
* DIFFERENT exit, so the fingerprint pins the whole narrowing ladder:
|
||||
* - `over(int)` → exact-type filter, unique survivor (ns level)
|
||||
* - `over(int,int)` → arity filter, unique survivor (inline child)
|
||||
* - `over(double)` → exact-type filter, unique survivor (inline child)
|
||||
* - `over(char)` → no exact match, `cppConversionRank` dominance
|
||||
* picks `over(int)` (promotion 1) over
|
||||
* `over(double)` (standard conversion 2)
|
||||
* - `over(braced-init)` → conversion ranking rejects every candidate and
|
||||
* `CPP_CONVERSION_ONLY_ARG_TYPE_PREFIXES` turns
|
||||
* that into an empty set → `undefined`
|
||||
* - `over` with arity 9 → arity filter empties an all-known-bounds set,
|
||||
* the authoritative-empty branch → `undefined`
|
||||
* - `same(int)` → two identical signatures survive narrowing →
|
||||
* `isOverloadAmbiguousAfterNormalization` → `'ambiguous'`
|
||||
* - `both(int)`/`both(double)` → select the namespace-level def and the
|
||||
* inline-child def respectively, pinning both
|
||||
* collection sources by nodeId
|
||||
*/
|
||||
const NS_PROBES = [
|
||||
['ns', 'own0', undefined],
|
||||
['ns', 'own1', undefined],
|
||||
['ns', 'inl0', undefined],
|
||||
['ns', 'dup', undefined],
|
||||
['ns', 'hidden0', undefined],
|
||||
['ns', 'nosuch', undefined],
|
||||
['ns', 'both', undefined],
|
||||
['twin', 'twinned', undefined],
|
||||
['twin', 'nosuch', undefined],
|
||||
['shared', 'part0', undefined],
|
||||
['shared', 'part1', undefined],
|
||||
['ns', 'over', { arity: 1, argumentTypes: ['int'] }],
|
||||
['ns', 'over', { arity: 2, argumentTypes: ['int', 'int'] }],
|
||||
['ns', 'over', { arity: 1, argumentTypes: ['double'] }],
|
||||
['ns', 'over', { arity: 1, argumentTypes: ['char'] }],
|
||||
['ns', 'over', { arity: 1, argumentTypes: ['braced-init:int:3'] }],
|
||||
['ns', 'over', { arity: 9, argumentTypes: [] }],
|
||||
['ns', 'same', { arity: 1, argumentTypes: ['int'] }],
|
||||
['ns', 'both', { arity: 1, argumentTypes: ['int'] }],
|
||||
['ns', 'both', { arity: 1, argumentTypes: ['double'] }],
|
||||
];
|
||||
|
||||
/** Members asked of the non-namespace receivers. Real-source member names, so
|
||||
* the miss is a receiver miss and not a member miss. */
|
||||
const MISS_MEMBERS = ['size', 'begin', 'data', 'reset', 'own0', 'dup'];
|
||||
|
||||
/** A plain identifier naming NO namespace in the corpus — a local, a type, a
|
||||
* buffer. The ~4-in-5 majority of header property 1. */
|
||||
function missReceiver(key) {
|
||||
const shape = key % 3;
|
||||
if (shape === 0) return `obj${key % 97}`;
|
||||
if (shape === 1) return `Widget${key % 31}`;
|
||||
return `buf${key % 197}`;
|
||||
}
|
||||
|
||||
/**
|
||||
* The call sites: `[receiver, member, callsite]`, deterministic, with the
|
||||
* production receiver mix (~1 in `NS_RECEIVER_IN` names a namespace).
|
||||
*
|
||||
* Two independently mixed keys per site so the receiver class (`a`) and the
|
||||
* probe choice (`b`) do not correlate — deriving both from one linear key made
|
||||
* `key % NS_RECEIVER_IN === 0` imply `key % NS_PROBES.length ∈ {0, 5}`, which
|
||||
* silently reduced the probe set to two entries.
|
||||
*/
|
||||
function callSites(fileCount) {
|
||||
const sharedGroups = Math.ceil(fileCount / 2);
|
||||
const sites = [];
|
||||
let nsReceiverSites = 0;
|
||||
/** The declared-namespace receiver a probe family asks for. */
|
||||
const nsReceiver = (family, key) => {
|
||||
if (family === 'twin') return `twin_${key % fileCount}`;
|
||||
if (family === 'shared') return `shared_${key % sharedGroups}`;
|
||||
return `ns_${key % fileCount}`;
|
||||
};
|
||||
const pushNs = (probe, key) => {
|
||||
sites.push([nsReceiver(probe[0], key), probe[1], probe[2]]);
|
||||
nsReceiverSites++;
|
||||
};
|
||||
// Coverage prelude: every probe at least once at BOTH scales, so the
|
||||
// fingerprinted outcome set never depends on how the mixer happens to spread.
|
||||
for (let p = 0; p < NS_PROBES.length; p++) pushNs(NS_PROBES[p], p);
|
||||
for (let f = 0; f < fileCount; f++) {
|
||||
for (let c = 0; c < CALLS_PER_FILE; c++) {
|
||||
const a = mix(f * 65599 + c);
|
||||
const b = mix(a ^ 0x9e3779b9);
|
||||
if (a % NS_RECEIVER_IN === 0) pushNs(NS_PROBES[b % NS_PROBES.length], b);
|
||||
else sites.push([missReceiver(b), MISS_MEMBERS[a % MISS_MEMBERS.length], undefined]);
|
||||
}
|
||||
}
|
||||
return { sites, nsReceiverSites };
|
||||
}
|
||||
|
||||
/** The timed loop: resolution only. The outcome strings the fingerprint needs
|
||||
* are built in a separate untimed pass (`outcomesOf`), so their allocation
|
||||
* cost — which grows with the corpus and would inflate the scaling ratio on
|
||||
* its own — never lands in the measurement. `sink` keeps the calls live. */
|
||||
function resolveAll(parsedFiles, sites) {
|
||||
let sink = 0;
|
||||
for (const [receiver, member, callsite] of sites) {
|
||||
const hit = resolveCppQualifiedNamespaceMember(
|
||||
receiver,
|
||||
member,
|
||||
parsedFiles,
|
||||
NO_SCOPES,
|
||||
callsite,
|
||||
);
|
||||
if (hit !== undefined) sink++;
|
||||
}
|
||||
return sink;
|
||||
}
|
||||
|
||||
/** Fingerprint key for one site. The callsite is part of the key: `ns::over`
|
||||
* resolves to a different def per arity/argument-type, and collapsing those
|
||||
* onto one key would drop the whole narrowing ladder from the gate. */
|
||||
function siteKey(receiver, member, callsite) {
|
||||
const args =
|
||||
callsite === undefined ? '' : `${callsite.arity}|${callsite.argumentTypes.join(',')}`;
|
||||
return `${receiver}::${member}(${args})`;
|
||||
}
|
||||
|
||||
/** Untimed identity pass, one resolve per DISTINCT `siteKey`. On a fixed corpus
|
||||
* the resolver is a pure function of `(receiver, member, callsite)`, so a
|
||||
* repeated site can only re-derive what the first occurrence already put in
|
||||
* the Set — the same argument that makes collecting into a Set correct makes
|
||||
* skipping the repeat correct. That is nearly the whole pass: the 192k/768k
|
||||
* sites carry only 2,330/3,470 distinct outcomes. */
|
||||
function outcomesOf(parsedFiles, sites) {
|
||||
const outcomes = new Set();
|
||||
const seen = new Set();
|
||||
for (const [receiver, member, callsite] of sites) {
|
||||
const key = siteKey(receiver, member, callsite);
|
||||
if (seen.has(key)) continue;
|
||||
seen.add(key);
|
||||
const hit = resolveCppQualifiedNamespaceMember(
|
||||
receiver,
|
||||
member,
|
||||
parsedFiles,
|
||||
NO_SCOPES,
|
||||
callsite,
|
||||
);
|
||||
outcomes.add(
|
||||
`${key}\u0000${hit === undefined ? '<none>' : hit === 'ambiguous' ? '<ambiguous>' : hit.nodeId}`,
|
||||
);
|
||||
}
|
||||
return outcomes;
|
||||
}
|
||||
|
||||
/**
|
||||
* MIN, not median — same rationale as bench/callable-value-flow: both scales
|
||||
* are timed in one process and every error source (scheduler preemption, GC, a
|
||||
* noisy neighbour on a shared CI runner) is additive, so the fastest observed
|
||||
* run is the closest estimate of the uncontended cost and keeps the derived
|
||||
* ratio comparable across machines.
|
||||
*/
|
||||
function fastest(values) {
|
||||
return Math.min(...values);
|
||||
}
|
||||
|
||||
/** Time one full pass: index build (lazy, on the first call) + every call
|
||||
* site. The corpus state is reset OUTSIDE the timer so the reset's own
|
||||
* O(files) cost never lands in the measurement. */
|
||||
function timeResolution(parsedFiles, sites) {
|
||||
for (let w = 0; w < WARMUP; w++) {
|
||||
populateInlineState(parsedFiles);
|
||||
resolveAll(parsedFiles, sites);
|
||||
}
|
||||
const samples = [];
|
||||
for (let r = 0; r < REPS; r++) {
|
||||
populateInlineState(parsedFiles);
|
||||
const t0 = performance.now();
|
||||
resolveAll(parsedFiles, sites);
|
||||
samples.push(performance.now() - t0);
|
||||
}
|
||||
return { ms: fastest(samples), outcomes: outcomesOf(parsedFiles, sites) };
|
||||
}
|
||||
|
||||
function fingerprint(outcomes) {
|
||||
return crypto
|
||||
.createHash('sha256')
|
||||
.update([...outcomes].sort().join('\n'))
|
||||
.digest('hex');
|
||||
}
|
||||
|
||||
const scales = {};
|
||||
for (const [name, fileCount] of [
|
||||
['small', SMALL],
|
||||
['large', LARGE],
|
||||
]) {
|
||||
const parsedFiles = buildCorpus(fileCount);
|
||||
const { sites, nsReceiverSites } = callSites(fileCount);
|
||||
const { ms, outcomes } = timeResolution(parsedFiles, sites);
|
||||
scales[name] = {
|
||||
files: fileCount,
|
||||
call_sites: sites.length,
|
||||
// Reported, not asserted: `ns_receiver_sites` evidences header property 1's
|
||||
// mix and `distinct_outcomes` the fingerprinted surface's size — a corpus
|
||||
// edit collapsing either still yields a "valid" fingerprint over far less.
|
||||
ns_receiver_sites: nsReceiverSites,
|
||||
distinct_outcomes: outcomes.size,
|
||||
ms: Number(ms.toFixed(3)),
|
||||
fingerprint: fingerprint(outcomes),
|
||||
};
|
||||
}
|
||||
|
||||
const scalingRatio = scales.large.ms / scales.small.ms / (LARGE / SMALL);
|
||||
|
||||
const report = {
|
||||
small: scales.small,
|
||||
large: scales.large,
|
||||
scaling_ratio: Number(scalingRatio.toFixed(3)),
|
||||
fingerprint: scales.large.fingerprint,
|
||||
};
|
||||
|
||||
if (!process.argv.includes('--check')) {
|
||||
console.log(JSON.stringify(report, null, 2));
|
||||
process.exit(0);
|
||||
}
|
||||
|
||||
const baseline = JSON.parse(fs.readFileSync(BASELINE_PATH, 'utf-8'));
|
||||
const failures = [];
|
||||
if (report.fingerprint !== baseline.fingerprint) {
|
||||
failures.push(
|
||||
`fingerprint drift: ${report.fingerprint} != ${baseline.fingerprint} — qualified ` +
|
||||
`namespace lookup resolved a DIFFERENT symbol set. This is a behaviour change, not a perf one.`,
|
||||
);
|
||||
}
|
||||
if (report.scaling_ratio > baseline.scaling_budget) {
|
||||
failures.push(
|
||||
`scaling ${report.scaling_ratio} > budget ${baseline.scaling_budget} — per-call-site cost ` +
|
||||
`now grows with corpus size again (#2788). Timing arm: re-run on an idle machine before ` +
|
||||
`investigating (see _scaling_note in baselines.json); the fingerprint arm never warrants a re-run.`,
|
||||
);
|
||||
}
|
||||
|
||||
console.log(JSON.stringify(report, null, 2));
|
||||
if (failures.length > 0) {
|
||||
console.error(`[cpp-qualified-ns --check] FAIL\n - ${failures.join('\n - ')}`);
|
||||
process.exit(1);
|
||||
}
|
||||
console.log('[cpp-qualified-ns --check] PASS');
|
||||
100
gitnexus/bench/schema-pairs/README.md
Normal file
100
gitnexus/bench/schema-pairs/README.md
Normal file
|
|
@ -0,0 +1,100 @@
|
|||
# Schema pair-set bench (#2793)
|
||||
|
||||
What a bigger `CodeRelation` FROM/TO pair set costs at query time, measured
|
||||
against a real `@ladybugdb/core` database.
|
||||
|
||||
```bash
|
||||
# from gitnexus/
|
||||
node --import tsx bench/schema-pairs/measure.mjs # print one JSON line per size + a summary
|
||||
node --import tsx bench/schema-pairs/measure.mjs --check # gate vs baselines.json
|
||||
```
|
||||
|
||||
## Why it exists
|
||||
|
||||
`src/core/lbug/schema.ts` generates its relation pairs from two cross products,
|
||||
and declines to add a third one **on the strength of a number** — roughly 1.04×
|
||||
at 450 declared pairs, 1.6× at 786, 2.1× at 1024. That measurement used to live
|
||||
in a scratch directory, so nobody proposing a third rule could re-run it. This
|
||||
harness is that measurement, committed — and it reproduces those figures.
|
||||
|
||||
Run it before widening a rule, and quote the new ratio in the review.
|
||||
|
||||
Observed on the reference box, **four runs** (ratios vs the 332-pair list):
|
||||
|
||||
| pairs | untyped | typed (floor) |
|
||||
| ----- | ---------- | ------------- |
|
||||
| 332 | 1.00× | 1.00× |
|
||||
| 450 | 0.93–1.05× | 0.98–1.17× |
|
||||
| 641 | 1.22–1.43× | 1.11–1.23× |
|
||||
| 786 | 1.52–1.75× | 1.19–1.31× |
|
||||
| 1024 | 2.03–2.34× | 1.31–1.57× |
|
||||
|
||||
Production's 450 came out _faster_ than 332 on three of the four runs, so at this
|
||||
size the pair count is inside run-to-run noise. Everything past ~640 is not.
|
||||
**Quote the range, not a single run** — one run is not evidence here.
|
||||
|
||||
## What it measures
|
||||
|
||||
For each pair-set size it builds a fresh database with all 32 node tables, a
|
||||
`CodeRelation` table declaring exactly that many FROM/TO pairs, and **identical
|
||||
data**, then times two query shapes over 40 anchors × 15 reps (median):
|
||||
|
||||
- **`untyped_ms_<size>`** — `MATCH (a {id: $id})-[r:CodeRelation]->(b)`. Neither
|
||||
endpoint is labelled, so LadybugDB must treat every declared pair as a
|
||||
candidate. This is the shape `impact`, `context` and `detect_changes` issue
|
||||
when they walk out from one node id, and the only one whose plan depends on
|
||||
how many pairs the table declares.
|
||||
- **`typed_ms_<size>`** — `MATCH (a:Function {…})-[r]->(b:Function)`, the lower
|
||||
bound. Both endpoints labelled prunes the plan to a single pair, so this was
|
||||
expected to be flat in the pair count. **It is not** — up to 1.17× at 450 and
|
||||
1.57× at 1024 — so a declared-but-unused pair costs something even when the
|
||||
planner never considers it. `typed_ratio_*` is therefore the floor, not a noise
|
||||
control; the real cost of widening sits between it and `ratio_*`. A run where
|
||||
`typed_ratio` moves _more_ than `ratio` is noise-dominated and should be
|
||||
rerun.
|
||||
- **`ratio_<size>`** — `untyped_ms_<size> / untyped_ms_332`. `ratio_450` is the
|
||||
figure `schema.ts` quotes.
|
||||
|
||||
### Sizes
|
||||
|
||||
The pair set is a prefix of a fixed 32×32 (`NODE_TABLES`²) enumeration, so each
|
||||
size is a strict superset of the smaller ones. The four pairs the synthetic data
|
||||
uses are pinned to the front, so **the same rows are reachable by the same query
|
||||
at every size** — the only variable is how many unused pairs are declared. The
|
||||
harness fails if the row counts ever differ across sizes.
|
||||
|
||||
| size | what it is |
|
||||
| ---- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| 332 | the pre-#2792 hand-written list — the reference for every ratio |
|
||||
| 450 | production today (two cross products + 72 hand-declared pairs) |
|
||||
| 641 | the third cross product `schema.ts` defers (`DEFINITION_ANCHOR_LABELS × {CodeElement, Section, Typedef, Union, Namespace, Impl, TypeAlias, Static, Template}`), which would leave ~29 hand-declared lines |
|
||||
| 786 | the size an earlier revision of that comment attributed to the third rule — it is 641; kept as a measured waypoint |
|
||||
| 1024 | the full cross product, the ceiling |
|
||||
|
||||
## Correctness gate
|
||||
|
||||
Before timing anything, the harness round-trips the **real** `SCHEMA_QUERIES`
|
||||
through a real database and asserts that `CALL SHOW_CONNECTION('CodeRelation')`
|
||||
reports exactly the pairs `parseRelationSchemaPairs` finds in `RELATION_SCHEMA`.
|
||||
|
||||
No magic number is baked in: the invariant is that the DDL LadybugDB _accepted_
|
||||
carries the pair set our own parser believes it declares. The absolute count is
|
||||
reported as `declared_pairs`. A pair declared twice would not reach this check at
|
||||
all — LadybugDB rejects the `CREATE REL TABLE` outright, which is why a duplicate
|
||||
kills every `analyze` rather than one repository's.
|
||||
|
||||
## What it does NOT measure
|
||||
|
||||
- **Ingest / `COPY` cost.** Pair-set size also multiplies the number of per-pair
|
||||
CSVs the emitter routes to (`src/core/lbug/rel-pair-routing.ts`); that cost is
|
||||
covered by `bench/emit-persistence`.
|
||||
- **At-scale absolute numbers.** Row counts here are small and deliberately
|
||||
constant. The ratios are the signal; the milliseconds are box-specific.
|
||||
|
||||
## Regenerating the baseline
|
||||
|
||||
`baselines.json` holds one budget, `ratio_450_budget` — the ceiling on what
|
||||
production's own pair count may cost relative to the 332-pair hand-list it
|
||||
replaced. Re-run without `--check` **several times** and copy the top of the
|
||||
observed `ratio_450` range plus headroom — the spread between runs on this box
|
||||
is wider than the effect being measured at 450, so a single run cannot set it.
|
||||
4
gitnexus/bench/schema-pairs/baselines.json
Normal file
4
gitnexus/bench/schema-pairs/baselines.json
Normal file
|
|
@ -0,0 +1,4 @@
|
|||
{
|
||||
"_comment": "ratio_450_budget — ceiling on what production's 450-pair set may cost on untyped-endpoint anchored queries, relative to the 332-pair hand-list it replaced. Observed 0.94x and 1.05x across two runs on the reference box (i.e. inside run-to-run noise; it came out faster than 332 once). The budget carries headroom for that spread — compare typed_ratio_450 (1.10-1.17x) for this box's floor. Raise it only with a measured range, never a single run.",
|
||||
"ratio_450_budget": 1.3
|
||||
}
|
||||
357
gitnexus/bench/schema-pairs/measure.mjs
Normal file
357
gitnexus/bench/schema-pairs/measure.mjs
Normal file
|
|
@ -0,0 +1,357 @@
|
|||
/**
|
||||
* What a bigger `CodeRelation` FROM/TO pair set costs at query time (#2793).
|
||||
*
|
||||
* `src/core/lbug/schema.ts` declares its relation pairs from two cross products
|
||||
* plus a small hand-written remainder, and it justifies NOT adding a third cross
|
||||
* product with a number: anchored queries cost ~1.04× at 450 declared pairs but
|
||||
* 1.6× at 786 and 2.1× at 1024. That measurement previously lived in a scratch
|
||||
* directory, so the claim could not be re-checked when someone proposed
|
||||
* widening a rule. This is it, committed.
|
||||
*
|
||||
* WHAT IT MEASURES. Against a real `@ladybugdb/core` database, with byte-identical
|
||||
* DATA at every size, it times the query shape whose plan actually depends on the
|
||||
* declared pair set:
|
||||
*
|
||||
* MATCH (a {id: $id})-[r:CodeRelation]->(b) RETURN b.id
|
||||
*
|
||||
* Neither endpoint is labelled, so LadybugDB must consider every declared
|
||||
* FROM/TO pair as a candidate — this is the shape `impact`, `context` and
|
||||
* `detect_changes` all issue when they walk out from one node id.
|
||||
*
|
||||
* A LABEL-typed query (`MATCH (a:Function)-[r]->(b:Function)`) is measured
|
||||
* alongside it as the LOWER BOUND. Its plan prunes to a single pair, so it was
|
||||
* expected to be flat in the pair count — it is NOT. Measured here it reaches
|
||||
* 1.17× at 450 and 1.57× at 1024 against the same 332-pair reference, i.e. a
|
||||
* declared-but-unused pair costs something even when the planner never
|
||||
* considers it (per-pair catalog/storage overhead the query pays regardless).
|
||||
* So `typed_ratio_*` is not a noise control: it is the floor, and the true cost
|
||||
* of a wider pair set lies between it and `ratio_*`. Treat any run where
|
||||
* `typed_ratio` moves MORE than `ratio` as noise-dominated.
|
||||
*
|
||||
* SIZES. The pair set is a prefix of a fixed 32×32 (`NODE_TABLES`²) enumeration
|
||||
* so every size is a strict SUPERSET of the smaller ones, and the four pairs the
|
||||
* data actually uses are pinned first — so the same rows are reachable by the
|
||||
* same query at every size, and the only variable is how many UNUSED pairs the
|
||||
* table declares:
|
||||
* - 332 — the pre-#2792 hand-written list (the historical baseline);
|
||||
* - 450 — production today (two cross products + 72 hand-declared);
|
||||
* - 641 — the third cross product schema.ts defers
|
||||
* (`DEFINITION_ANCHOR_LABELS × {CodeElement, Section, Typedef, Union,
|
||||
* Namespace, Impl, TypeAlias, Static, Template}`), which would leave
|
||||
* only ~29 hand-declared lines;
|
||||
* - 786 — the size an earlier revision of that comment attributed to the
|
||||
* third rule (it is 641; 786 is kept as a measured waypoint);
|
||||
* - 1024 — the full cross product, the ceiling.
|
||||
*
|
||||
* Ratios are reported against 332, the smallest size — `ratio_450` is the
|
||||
* number schema.ts quotes.
|
||||
*
|
||||
* CORRECTNESS GATE. Before timing anything it round-trips the REAL
|
||||
* `SCHEMA_QUERIES` through a real database and asserts that
|
||||
* `CALL SHOW_CONNECTION('CodeRelation')` reports exactly the pairs
|
||||
* `parseRelationSchemaPairs` finds in `RELATION_SCHEMA`. That is the invariant
|
||||
* that matters and it needs no magic number: it proves the DDL LadybugDB
|
||||
* ACCEPTED carries the pair set our own parser believes it declares. (A
|
||||
* duplicated FROM/TO would not even get this far — LadybugDB rejects the
|
||||
* `CREATE REL TABLE` outright, which is why that failure kills every `analyze`.)
|
||||
* The absolute count is reported as `declared_pairs` for the record.
|
||||
*
|
||||
* Build-free: imports the `.ts` sources through tsx.
|
||||
*
|
||||
* node --import tsx bench/schema-pairs/measure.mjs # print JSON lines
|
||||
* node --import tsx bench/schema-pairs/measure.mjs --check # gate vs baselines.json
|
||||
*
|
||||
* `--check` fails if the correctness gate breaks, or if `ratio_450` exceeds its
|
||||
* budget — i.e. if production's own pair count starts costing materially more
|
||||
* than the hand-written list it replaced.
|
||||
*/
|
||||
import fs from 'node:fs';
|
||||
import os from 'node:os';
|
||||
import path from 'node:path';
|
||||
import { fileURLToPath } from 'node:url';
|
||||
|
||||
import { NODE_TABLES } from 'gitnexus-shared';
|
||||
import {
|
||||
NODE_SCHEMA_QUERIES,
|
||||
RELATION_SCHEMA,
|
||||
REL_TABLE_NAME,
|
||||
} from '../../src/core/lbug/schema.ts';
|
||||
import { parseRelationSchemaPairs } from '../../src/core/lbug/rel-pair-routing.ts';
|
||||
|
||||
const __dirname = path.dirname(fileURLToPath(import.meta.url));
|
||||
const BASELINE_PATH = path.resolve(__dirname, 'baselines.json');
|
||||
|
||||
const lbug = (await import('@ladybugdb/core')).default;
|
||||
|
||||
// ---- sizes + the pair enumeration every size is a prefix of ----
|
||||
|
||||
const SIZES = [332, 450, 641, 786, 1024];
|
||||
const REFERENCE_SIZE = 332; // ratios are relative to this
|
||||
const PRODUCTION_SIZE = 450; // the size schema.ts ships
|
||||
|
||||
// The four pairs the synthetic data uses. Pinned to the FRONT of the
|
||||
// enumeration so they are declared at every size — otherwise a smaller pair set
|
||||
// would simply carry fewer rows and the comparison would measure data volume,
|
||||
// not pair-set size.
|
||||
const DATA_PAIRS = [
|
||||
['File', 'Function'],
|
||||
['Function', 'Function'],
|
||||
['Function', 'Class'],
|
||||
['Class', 'Method'],
|
||||
];
|
||||
|
||||
const pairKey = ([from, to]) => `${from}|${to}`;
|
||||
|
||||
// NODE_TABLES² in declaration order, data pairs first, deduped. 32² = 1024.
|
||||
const PAIR_UNIVERSE = (() => {
|
||||
const seen = new Set(DATA_PAIRS.map(pairKey));
|
||||
const all = [...DATA_PAIRS];
|
||||
for (const from of NODE_TABLES) {
|
||||
for (const to of NODE_TABLES) {
|
||||
const key = `${from}|${to}`;
|
||||
if (seen.has(key)) continue;
|
||||
seen.add(key);
|
||||
all.push([from, to]);
|
||||
}
|
||||
}
|
||||
return all;
|
||||
})();
|
||||
|
||||
if (PAIR_UNIVERSE.length !== NODE_TABLES.length ** 2) {
|
||||
throw new Error(
|
||||
`bench: pair universe is ${PAIR_UNIVERSE.length}, expected ${NODE_TABLES.length ** 2} ` +
|
||||
`(NODE_TABLES changed — update SIZES, the 1024 ceiling is no longer the ceiling)`,
|
||||
);
|
||||
}
|
||||
for (const size of SIZES) {
|
||||
if (size > PAIR_UNIVERSE.length) {
|
||||
throw new Error(`bench: size ${size} exceeds the ${PAIR_UNIVERSE.length}-pair universe`);
|
||||
}
|
||||
}
|
||||
|
||||
const relTableDdlFor = (size) => {
|
||||
const pairs = PAIR_UNIVERSE.slice(0, size).map(([from, to]) => ` FROM \`${from}\` TO \`${to}\``);
|
||||
return `CREATE REL TABLE ${REL_TABLE_NAME} (\n${pairs.join(',\n')},\n type STRING,\n confidence DOUBLE,\n reason STRING,\n step INT32\n)`;
|
||||
};
|
||||
|
||||
// ---- synthetic data (identical at every size) ----
|
||||
|
||||
const FILES = 20;
|
||||
const FNS_PER_FILE = 8;
|
||||
const CLASSES = 40;
|
||||
const METHODS_PER_CLASS = 4;
|
||||
const CALLS_PER_FN = 3;
|
||||
const REPS = 15; // median over reps
|
||||
const ANCHORS = 40; // distinct anchor ids queried per rep
|
||||
|
||||
// Batched with UNWIND rather than one statement per row: per-statement overhead
|
||||
// dwarfs the insert itself here, and load time is not what this bench measures.
|
||||
function dataStatements() {
|
||||
const stmts = [];
|
||||
const fnIds = [];
|
||||
const classIds = [];
|
||||
const methodIds = [];
|
||||
const fileIds = [];
|
||||
for (let f = 0; f < FILES; f++) fileIds.push(`file-${f}`);
|
||||
for (let f = 0; f < FILES; f++) {
|
||||
for (let i = 0; i < FNS_PER_FILE; i++) fnIds.push(`fn-${f}-${i}`);
|
||||
}
|
||||
for (let c = 0; c < CLASSES; c++) {
|
||||
classIds.push(`cls-${c}`);
|
||||
for (let m = 0; m < METHODS_PER_CLASS; m++) methodIds.push(`m-${c}-${m}`);
|
||||
}
|
||||
|
||||
const nodeBatch = (label, ids) =>
|
||||
`UNWIND [${ids.map((id) => `{id: '${id}'}`).join(', ')}] AS r ` +
|
||||
`CREATE (:\`${label}\` {id: r.id, name: r.id, filePath: 'bench.ts'})`;
|
||||
stmts.push(nodeBatch('File', fileIds));
|
||||
stmts.push(nodeBatch('Function', fnIds));
|
||||
stmts.push(nodeBatch('Class', classIds));
|
||||
stmts.push(nodeBatch('Method', methodIds));
|
||||
|
||||
const relBatch = (fromLabel, toLabel, type, edges) =>
|
||||
`UNWIND [${edges.map(([f, t]) => `{f: '${f}', t: '${t}'}`).join(', ')}] AS e ` +
|
||||
`MATCH (a:\`${fromLabel}\` {id: e.f}), (b:\`${toLabel}\` {id: e.t}) ` +
|
||||
`CREATE (a)-[:${REL_TABLE_NAME} {type: '${type}', confidence: 1.0, reason: 'bench', step: 0}]->(b)`;
|
||||
|
||||
const contains = [];
|
||||
for (let f = 0; f < FILES; f++) {
|
||||
for (let i = 0; i < FNS_PER_FILE; i++) contains.push([`file-${f}`, `fn-${f}-${i}`]);
|
||||
}
|
||||
stmts.push(relBatch('File', 'Function', 'CONTAINS', contains));
|
||||
|
||||
// Function→Function calls: each fn calls the next CALLS_PER_FN, wrapping.
|
||||
const calls = [];
|
||||
for (let i = 0; i < fnIds.length; i++) {
|
||||
for (let k = 1; k <= CALLS_PER_FN; k++) calls.push([fnIds[i], fnIds[(i + k) % fnIds.length]]);
|
||||
}
|
||||
stmts.push(relBatch('Function', 'Function', 'CALLS', calls));
|
||||
|
||||
const uses = fnIds.map((id, i) => [id, classIds[i % classIds.length]]);
|
||||
stmts.push(relBatch('Function', 'Class', 'USES', uses));
|
||||
|
||||
const hasMethod = [];
|
||||
for (let c = 0; c < CLASSES; c++) {
|
||||
for (let m = 0; m < METHODS_PER_CLASS; m++) hasMethod.push([`cls-${c}`, `m-${c}-${m}`]);
|
||||
}
|
||||
stmts.push(relBatch('Class', 'Method', 'HAS_METHOD', hasMethod));
|
||||
|
||||
// Anchors: functions, which have out-edges on two distinct declared pairs.
|
||||
return { stmts, anchors: fnIds.slice(0, ANCHORS) };
|
||||
}
|
||||
|
||||
const { stmts: DATA_STATEMENTS, anchors: ANCHOR_IDS } = dataStatements();
|
||||
|
||||
// ---- timing ----
|
||||
|
||||
const median = (xs) => {
|
||||
const s = [...xs].sort((a, b) => a - b);
|
||||
const m = Math.floor(s.length / 2);
|
||||
return s.length % 2 ? s[m] : (s[m - 1] + s[m]) / 2;
|
||||
};
|
||||
|
||||
const withDb = async (fn) => {
|
||||
const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'gitnexus-bench-pairs-'));
|
||||
const db = new lbug.Database(path.join(dir, 'db'));
|
||||
const conn = new lbug.Connection(db);
|
||||
try {
|
||||
return await fn(conn);
|
||||
} finally {
|
||||
await conn.close().catch(() => {});
|
||||
await db.close?.().catch?.(() => {});
|
||||
fs.rmSync(dir, { recursive: true, force: true });
|
||||
}
|
||||
};
|
||||
|
||||
async function runAll(conn, statements) {
|
||||
for (const s of statements) await conn.query(s);
|
||||
}
|
||||
|
||||
// The measured shape: BOTH endpoints untyped, anchored by id. LadybugDB must
|
||||
// consider every declared FROM/TO pair as a candidate.
|
||||
const UNTYPED_QUERY = (id) =>
|
||||
`MATCH (a {id: '${id}'})-[r:${REL_TABLE_NAME}]->(b) RETURN b.id AS id, r.type AS type`;
|
||||
// The lower bound: both endpoints labelled, so the planner prunes to one pair.
|
||||
// Still not flat in the pair count (see the header) — an unused declared pair
|
||||
// costs something even when the plan never touches it.
|
||||
const TYPED_QUERY = (id) =>
|
||||
`MATCH (a:Function {id: '${id}'})-[r:${REL_TABLE_NAME}]->(b:Function) RETURN b.id AS id`;
|
||||
|
||||
async function timeQueries(conn, build) {
|
||||
// Warm: run the whole anchor sweep once uncounted (plan cache + page cache).
|
||||
for (const id of ANCHOR_IDS) await (await conn.query(build(id))).getAll();
|
||||
const samples = [];
|
||||
let rows = 0;
|
||||
for (let rep = 0; rep < REPS; rep++) {
|
||||
const start = process.hrtime.bigint();
|
||||
let n = 0;
|
||||
for (const id of ANCHOR_IDS) n += (await (await conn.query(build(id))).getAll()).length;
|
||||
samples.push(Number(process.hrtime.bigint() - start) / 1e6);
|
||||
rows = n;
|
||||
}
|
||||
return { ms: median(samples), rows };
|
||||
}
|
||||
|
||||
async function measureSize(size) {
|
||||
return withDb(async (conn) => {
|
||||
for (const q of NODE_SCHEMA_QUERIES) await conn.query(q);
|
||||
await conn.query(relTableDdlFor(size));
|
||||
await runAll(conn, DATA_STATEMENTS);
|
||||
const untyped = await timeQueries(conn, UNTYPED_QUERY);
|
||||
const typed = await timeQueries(conn, TYPED_QUERY);
|
||||
return {
|
||||
pairs: size,
|
||||
untyped_ms: Number(untyped.ms.toFixed(3)),
|
||||
untyped_rows: untyped.rows,
|
||||
typed_ms: Number(typed.ms.toFixed(3)),
|
||||
typed_rows: typed.rows,
|
||||
};
|
||||
});
|
||||
}
|
||||
|
||||
// ---- correctness gate: the REAL schema, round-tripped ----
|
||||
|
||||
async function verifyRealSchema() {
|
||||
return withDb(async (conn) => {
|
||||
for (const q of NODE_SCHEMA_QUERIES) await conn.query(q);
|
||||
// If RELATION_SCHEMA declared a pair twice, LadybugDB rejects this outright
|
||||
// — the failure mode that kills every `analyze`, not just one repo's.
|
||||
await conn.query(RELATION_SCHEMA);
|
||||
const res = await conn.query(`CALL SHOW_CONNECTION('${REL_TABLE_NAME}') RETURN *`);
|
||||
const rows = await res.getAll();
|
||||
const actual = new Set(
|
||||
rows.map(
|
||||
(r) =>
|
||||
`${r['source table name'] ?? r.source}|${r['destination table name'] ?? r.destination}`,
|
||||
),
|
||||
);
|
||||
const expected = parseRelationSchemaPairs(RELATION_SCHEMA);
|
||||
const missing = [...expected].filter((p) => !actual.has(p)).sort();
|
||||
const extra = [...actual].filter((p) => !expected.has(p)).sort();
|
||||
return { declared_pairs: expected.size, db_pairs: actual.size, missing, extra };
|
||||
});
|
||||
}
|
||||
|
||||
// ---- run ----
|
||||
|
||||
const CHECK = process.argv.includes('--check');
|
||||
const failures = [];
|
||||
|
||||
const verified = await verifyRealSchema();
|
||||
if (verified.missing.length > 0 || verified.extra.length > 0) {
|
||||
failures.push(
|
||||
`RELATION_SCHEMA round-trip mismatch: ${verified.missing.length} pair(s) parsed but absent ` +
|
||||
`from SHOW_CONNECTION (${verified.missing.slice(0, 5).join(', ')}), ${verified.extra.length} ` +
|
||||
`present in the DB but unparsed (${verified.extra.slice(0, 5).join(', ')})`,
|
||||
);
|
||||
}
|
||||
|
||||
const results = [];
|
||||
for (const size of SIZES) results.push(await measureSize(size));
|
||||
|
||||
const reference = results.find((r) => r.pairs === REFERENCE_SIZE);
|
||||
const summary = {
|
||||
...verified,
|
||||
missing: undefined,
|
||||
extra: undefined,
|
||||
reference_pairs: REFERENCE_SIZE,
|
||||
};
|
||||
for (const r of results) {
|
||||
summary[`untyped_ms_${r.pairs}`] = r.untyped_ms;
|
||||
summary[`typed_ms_${r.pairs}`] = r.typed_ms;
|
||||
summary[`ratio_${r.pairs}`] = Number((r.untyped_ms / reference.untyped_ms).toFixed(3));
|
||||
summary[`typed_ratio_${r.pairs}`] = Number((r.typed_ms / reference.typed_ms).toFixed(3));
|
||||
}
|
||||
|
||||
// Row counts must be identical at every size — otherwise the sizes are not
|
||||
// carrying the same data and the ratios mean nothing.
|
||||
const rowShapes = new Set(results.map((r) => `${r.untyped_rows}/${r.typed_rows}`));
|
||||
if (rowShapes.size !== 1) {
|
||||
failures.push(
|
||||
`row counts differ across pair-set sizes (${[...rowShapes].join(' vs ')}) — the data pins ` +
|
||||
`in DATA_PAIRS are not holding, so the ratios compare different graphs`,
|
||||
);
|
||||
}
|
||||
|
||||
if (!CHECK) {
|
||||
for (const r of results) process.stdout.write(JSON.stringify(r) + '\n');
|
||||
process.stdout.write(JSON.stringify(summary) + '\n');
|
||||
} else {
|
||||
const baselines = JSON.parse(fs.readFileSync(BASELINE_PATH, 'utf8'));
|
||||
const budget = baselines[`ratio_${PRODUCTION_SIZE}_budget`];
|
||||
if (budget !== undefined && summary[`ratio_${PRODUCTION_SIZE}`] >= budget) {
|
||||
failures.push(
|
||||
`production pair set (${PRODUCTION_SIZE}) costs ${summary[`ratio_${PRODUCTION_SIZE}`]}× vs ` +
|
||||
`${REFERENCE_SIZE} pairs, >= budget ${budget} (untyped ${reference.untyped_ms}ms -> ` +
|
||||
`${summary[`untyped_ms_${PRODUCTION_SIZE}`]}ms; typed control ` +
|
||||
`${summary[`typed_ratio_${PRODUCTION_SIZE}`]}×)`,
|
||||
);
|
||||
}
|
||||
process.stdout.write(JSON.stringify(summary) + '\n');
|
||||
}
|
||||
|
||||
if (failures.length > 0) {
|
||||
for (const f of failures) process.stderr.write(`[schema-pairs] FAIL: ${f}\n`);
|
||||
process.exit(1);
|
||||
}
|
||||
if (CHECK) process.stderr.write(`[schema-pairs --check] PASS (${results.length} sizes)\n`);
|
||||
|
|
@ -14,10 +14,11 @@
|
|||
"_rebaselined_2766_callee_position_marker": "#2766 review fix: a call's callee selector is no longer DROPPED at capture. An earlier commit on this branch dropped it outright, which also deleted the genuine field read on a func-typed struct field (`h.dep.Work()` where `Work func() error`) - callback/hook/mock structs lost their only ACCESSES evidence. The match is now emitted carrying `@reference.callee-position`, and the phantom is suppressed at EMIT by the resolved target's kind instead. Go only: the other 14 languages' fingerprints are byte-identical, which is the check that this is not a cross-language capture change. Prior 7bb524a32a2eed57a15b454e3a33480e92a496c683e6856ef02179693c0e02e3 -> e47302079e17a5e73711bbed5416557b49327cb67e4932008700ec6b8fb468b3; scaling 1.001 < 1.5; fixtures 102 (unchanged), capture_groups_fp 2103."
|
||||
},
|
||||
"cobol": {
|
||||
"fingerprint": "d45bb091b0893d0de4fae2486b31ba21719c9377bf35a0908fd3a36fa1c3bf4e",
|
||||
"fingerprint": "c8c00b56a7da24e04080eb885714fbbf45e3903324f0cf9df0754f5b5a92e3aa",
|
||||
"scaling_budget": 1.5,
|
||||
"_rebaselined_callable_flow_2522_followup": "PR #2522 follow-up: COBOL procedure-pointer callable flow facts; multi-topic extraction now consumes each grouped scope/declaration match once instead of requiring a duplicate declaration-only match. Prior 68ee0e95eb9f86f2d92ca35f730f4c2d4d83abc1b5241ae767ff3437780ec8d1 -> d45bb091b0893d0de4fae2486b31ba21719c9377bf35a0908fd3a36fa1c3bf4e; scaling 0.853 < 1.5.",
|
||||
"_note": "Updated for F17-F23 fixes (P2: TIMES guard, ADD GIVING, SQL AS alias). See PR #1959."
|
||||
"_note": "Updated for F17-F23 fixes (P2: TIMES guard, ADD GIVING, SQL AS alias). See PR #1959.",
|
||||
"_rebaselined_2793_declaratives": "PR #2793: corpus-only re-baseline. `cobol-declaratives` was added to test/fixtures/lang-resolution to reproduce the `Namespace\u2192Record` analyze abort (DECLARATIVES / USE AFTER STANDARD ERROR ON <file>), and this bench globs `lang-resolution/cobol-*`, so the corpus grew 14 -> 15 files. Verified capture-neutral: with that one fixture moved aside the fingerprint is byte-identical to the prior d45bb091b0893d0de4fae2486b31ba21719c9377bf35a0908fd3a36fa1c3bf4e. No COBOL capture code changed in that PR. Scaling 0.677 < 1.5."
|
||||
},
|
||||
"c": {
|
||||
"fingerprint": "3418cded9f7072152f68992f0a426f43ae7d9d553579a47075fc0cab185848a5",
|
||||
|
|
|
|||
|
|
@ -15,6 +15,8 @@ import v8 from 'v8';
|
|||
import cliProgress from 'cli-progress';
|
||||
import { isLbugReady, LbugWipeError } from '../core/lbug/lbug-adapter.js';
|
||||
import { boundedCheckpointBeforeExit } from '../core/lbug/shutdown-helpers.js';
|
||||
import { findUndeclaredRelationPairError } from '../core/lbug/rel-pair-routing.js';
|
||||
import { causeChain } from '../lib/utils.js';
|
||||
import {
|
||||
getOsPageSize,
|
||||
isLbugCheckpointIoError,
|
||||
|
|
@ -101,15 +103,13 @@ const writeFatalToStderr = (label: string, err: unknown): void => {
|
|||
// #2068) is only reachable via `.cause`. Without this the user sees the
|
||||
// wrapper's main-thread stack and never the real frame. `cause.stack` already
|
||||
// begins with the cause's message, so we print the stack alone (not message +
|
||||
// stack) to avoid repeating it. Depth-bounded so a cyclic `cause` can't loop
|
||||
// (the phase runner wraps one level; the bound leaves headroom for future
|
||||
// nesting); uses realStderrWrite so the redirected console.error's ANSI
|
||||
// clear-line wrapping can't erase it (#1169).
|
||||
const MAX_CAUSE_DEPTH = 5;
|
||||
let cause: unknown = isErr ? (err as { cause?: unknown }).cause : undefined;
|
||||
for (let depth = 0; depth < MAX_CAUSE_DEPTH && cause instanceof Error; depth++) {
|
||||
// stack) to avoid repeating it. `causeChain` owns the traversal and the depth
|
||||
// bound that stops a cyclic `cause` looping — this used to be one of four
|
||||
// hand-rolled copies that had already drifted apart on both. Uses
|
||||
// realStderrWrite so the redirected console.error's ANSI clear-line wrapping
|
||||
// can't erase it (#1169). The head is skipped: it was just printed above.
|
||||
for (const cause of causeChain(isErr ? (err as { cause?: unknown }).cause : undefined)) {
|
||||
realStderrWrite(`\n Caused by: ${cause.stack ?? cause.message}\n`);
|
||||
cause = (cause as { cause?: unknown }).cause;
|
||||
}
|
||||
};
|
||||
|
||||
|
|
@ -1717,6 +1717,36 @@ const analyzeCommandImpl = async (
|
|||
return;
|
||||
}
|
||||
|
||||
// An extracted edge whose FROM→TO label pair is missing from GitNexus's own
|
||||
// relation DDL (#2789). `assertDeclaredPair` aborts the run rather than let
|
||||
// the bulk COPY fail late and silently drop the edge, so the user sees a
|
||||
// mid-run crash inside GitNexus internals with nothing to act on. Name the
|
||||
// pair, the relationship and the file that produced it, and say plainly that
|
||||
// a re-run cannot help — this is deterministic for the same input.
|
||||
// Checked by TYPE (repo norm, #2385) BEFORE the message-text heuristics
|
||||
// below, and through the `cause` chain because the ingestion phase runner
|
||||
// rewraps every phase failure as `Phase 'X' failed: …`.
|
||||
const undeclaredPair = findUndeclaredRelationPairError(err);
|
||||
if (undeclaredPair !== undefined) {
|
||||
// Render the error's OWN message indented — same idiom as the
|
||||
// `LbugWipeError` and page-size branches below. `UndeclaredRelationPairError`
|
||||
// builds a fully self-contained message (pair, relationship type, both node
|
||||
// ids, source file, issue URL, `.gitnexusignore` workaround) precisely
|
||||
// because `gitnexus serve` forwards only `err.message` over worker IPC.
|
||||
// Re-rendering those fields here would be a second copy of one string, free
|
||||
// to drift from the first — and the actionable half would reach CLI users
|
||||
// only. `undeclaredPair.message`, not the outer `msg`: the real error may be
|
||||
// several `cause` levels below the phase wrapper `msg` came from.
|
||||
cliError(` ${undeclaredPair.message.replace(/\n/g, '\n ')}\n`, {
|
||||
recoveryHint: 'undeclared-relation-pair',
|
||||
labelPair: undeclaredPair.pairKey,
|
||||
relationType: undeclaredPair.relationType,
|
||||
sourceFile: undeclaredPair.sourceFile,
|
||||
});
|
||||
process.exitCode = 1;
|
||||
return;
|
||||
}
|
||||
|
||||
// WAL corruption — the index file is unreadable. Give a clear recovery
|
||||
// path without a confusing stack trace (the native error message alone
|
||||
// is enough signal).
|
||||
|
|
|
|||
|
|
@ -60,7 +60,8 @@ export type RecoveryHint =
|
|||
| 'module-not-found'
|
||||
| 'gitnexusrc-invalid'
|
||||
| 'default-branch-invalid'
|
||||
| 'index-lock-timeout';
|
||||
| 'index-lock-timeout'
|
||||
| 'undeclared-relation-pair';
|
||||
|
||||
/**
|
||||
* Common shape for the optional structured-field bag passed to
|
||||
|
|
|
|||
|
|
@ -58,8 +58,8 @@
|
|||
* - `noAdlSites` — call sites with parenthesized function (capture-time)
|
||||
* - `classToNamespaceQualifiedName` — class def → its enclosing namespace
|
||||
* qualified name (`populateCppAssociatedNamespaces` time)
|
||||
* - `adlIndex` / `adlIndexSource` — the lazily-built candidate index and the
|
||||
* `parsedFiles` reference it was built from (first-`pickCppAdlCandidates`
|
||||
* - `adlIndexByPass` — the lazily-built candidate index, memoized weakly on
|
||||
* the `parsedFiles` array it was built from (first-`pickCppAdlCandidates`
|
||||
* time; see `ensureAdlIndex`)
|
||||
*
|
||||
* The class→namespace map uses qualified names (not scope IDs) because
|
||||
|
|
@ -176,8 +176,21 @@ export interface AdlCandidateIndex {
|
|||
readonly seqByNodeId: Map<string, number>;
|
||||
}
|
||||
|
||||
let adlIndex: AdlCandidateIndex | undefined;
|
||||
let adlIndexSource: readonly ParsedFile[] | undefined;
|
||||
/**
|
||||
* Per-pass memo: `parsedFiles` array identity → the built index. `WeakMap`-keyed
|
||||
* so the entry lives only as long as the caller's `parsedFiles` array does, and
|
||||
* is reclaimed with the pass — mirrors `qualifiedNsIndexByPass` in
|
||||
* `inline-namespaces.ts` and `moduleScopeIndexByPass` in `file-local-linkage.ts`.
|
||||
*
|
||||
* Weak keying is load-bearing, not stylistic: the index holds
|
||||
* `SymbolDefinition` references reaching into every ParsedFile's scopes, so the
|
||||
* previous module-level strong `let` pair (index + source array) kept the whole
|
||||
* C/C++ ParsedFile set alive past the point `scope-resolution/pipeline/phase.ts`
|
||||
* evicts `files`/`contents`/`preExtractedByPath` and calls `forceGc()` to
|
||||
* reclaim it — C++ is 7th of 16 `SCOPE_RESOLVERS` entries, so the retention
|
||||
* survived nine later language passes plus emit.
|
||||
*/
|
||||
let adlIndexByPass = new WeakMap<readonly ParsedFile[], AdlCandidateIndex>();
|
||||
|
||||
function siteKey(filePath: string, line: number, col: number): string {
|
||||
return `${filePath}:${line}:${col}`;
|
||||
|
|
@ -373,25 +386,39 @@ export function validateAdlSeqCoverage(idx: AdlCandidateIndex): string[] {
|
|||
}
|
||||
|
||||
/** Build the ADL index on first use of a given `parsedFiles` set; reuse it for
|
||||
* all subsequent call sites in the same pipeline run. Reset by
|
||||
* every subsequent call site in the same pipeline run. Reset by
|
||||
* `clearCppAdlState`.
|
||||
*
|
||||
* Lifetime: the memoized entry is reachable only while the caller still holds
|
||||
* the `parsedFiles` array. Once the pipeline drops it, the entry — and the
|
||||
* `SymbolDefinition` references it holds into those files' scopes — becomes
|
||||
* collectable; nothing here pins it (see {@link adlIndexByPass}).
|
||||
*
|
||||
* Staleness is keyed on `parsedFiles` reference identity ONLY, but the index
|
||||
* is a function of THREE inputs: `parsedFiles` (namespace/friend candidates),
|
||||
* `scopes` (`classDefsBySimple`, read from `scopes.defs.byId`), and the
|
||||
* module-level `classToNamespaceQualifiedName` (friend-candidate keys). This
|
||||
* is sound for the current pipeline because all three are built together once
|
||||
* per `runScopeResolution` pass and `clearCppAdlState` runs in
|
||||
* `loadResolutionConfig` at the start of every pass. Callers MUST call
|
||||
* `clearCppAdlState` between any two passes that change `scopes` or
|
||||
* `classToNamespaceQualifiedName` while reusing the same `parsedFiles` array
|
||||
* reference — otherwise a stale index would be served. (No such caller exists
|
||||
* today; widening the guard to also key on `scopes` is deferred until one
|
||||
* does.) */
|
||||
function ensureAdlIndex(scopes: ScopeResolutionIndexes, parsedFiles: readonly ParsedFile[]): void {
|
||||
if (adlIndex !== undefined && adlIndexSource === parsedFiles) return;
|
||||
adlIndex = buildAdlIndex(scopes, parsedFiles);
|
||||
adlIndexSource = parsedFiles;
|
||||
* module-level `classToNamespaceQualifiedName` (friend-candidate keys). The
|
||||
* WeakMap key covers only the first, so the explicit clear is still REQUIRED
|
||||
* for the other two: a later pass may hand back the same `parsedFiles`
|
||||
* reference with different `scopes` / class→namespace state, and identity
|
||||
* alone would serve the stale index. This is sound for the current pipeline
|
||||
* because all three are built together once per `runScopeResolution` pass and
|
||||
* `clearCppAdlState` runs in `loadResolutionConfig` at the start of every
|
||||
* pass. Callers MUST call `clearCppAdlState` between any two passes that
|
||||
* change `scopes` or `classToNamespaceQualifiedName` while reusing the same
|
||||
* `parsedFiles` array reference — otherwise a stale index would be served.
|
||||
* (No such caller exists today; widening the key to also cover `scopes` is
|
||||
* deferred until one does.) */
|
||||
function ensureAdlIndex(
|
||||
scopes: ScopeResolutionIndexes,
|
||||
parsedFiles: readonly ParsedFile[],
|
||||
): AdlCandidateIndex {
|
||||
let index = adlIndexByPass.get(parsedFiles);
|
||||
if (index === undefined) {
|
||||
index = buildAdlIndex(scopes, parsedFiles);
|
||||
adlIndexByPass.set(parsedFiles, index);
|
||||
}
|
||||
return index;
|
||||
}
|
||||
|
||||
/** Record per-call-site argument info. Called once per call site from
|
||||
|
|
@ -504,8 +531,14 @@ export function clearCppAdlState(): void {
|
|||
argInfoSiteKeysByFile.clear();
|
||||
noAdlSiteKeysByFile.clear();
|
||||
classToNamespaceQualifiedName.clear();
|
||||
adlIndex = undefined;
|
||||
adlIndexSource = undefined;
|
||||
// The candidate index is dropped by REASSIGNING a fresh `WeakMap`: `WeakMap`
|
||||
// has no `.clear()`, and swapping the instance discards every memoized entry
|
||||
// at once — the exact invalidation the previous `adlIndex = undefined` pair
|
||||
// provided. Required even though the map is keyed on `parsedFiles` identity,
|
||||
// because the index also depends on `scopes` and
|
||||
// `classToNamespaceQualifiedName`, which the key cannot observe (see
|
||||
// {@link ensureAdlIndex}).
|
||||
adlIndexByPass = new WeakMap();
|
||||
}
|
||||
|
||||
/**
|
||||
|
|
@ -572,15 +605,20 @@ export function pickCppAdlCandidates(
|
|||
if (args === undefined || args.length === 0) return undefined;
|
||||
|
||||
// Build the workspace-wide ADL candidate index once; reuse for every site.
|
||||
ensureAdlIndex(scopes, parsedFiles);
|
||||
const idx = ensureAdlIndex(scopes, parsedFiles);
|
||||
|
||||
// Collect associated namespace QNames from every participating class-typed arg
|
||||
// and from function-reference args.
|
||||
const associatedNamespaces = new Set<string>();
|
||||
for (const arg of args) {
|
||||
collectAssociatedNamespacesForAdlArg(arg, scopes, associatedNamespaces);
|
||||
collectAssociatedNamespacesForAdlArg(arg, scopes, idx, associatedNamespaces);
|
||||
if (arg.functionRefText !== undefined) {
|
||||
collectFunctionTypeAssociatedNamespaces(arg.functionRefText, scopes, associatedNamespaces);
|
||||
collectFunctionTypeAssociatedNamespaces(
|
||||
arg.functionRefText,
|
||||
scopes,
|
||||
idx,
|
||||
associatedNamespaces,
|
||||
);
|
||||
}
|
||||
}
|
||||
if (associatedNamespaces.size === 0) return undefined;
|
||||
|
|
@ -593,8 +631,6 @@ export function pickCppAdlCandidates(
|
|||
// (`friendCandidates`, ISO C++ `[basic.lookup.argdep]` §2).
|
||||
// Dedup by nodeId and sort by visitation sequence so the candidate list is
|
||||
// byte-for-byte identical to the legacy file-major scan order.
|
||||
const idx = adlIndex;
|
||||
if (idx === undefined) return undefined;
|
||||
const bySeq = new Map<number, SymbolDefinition>();
|
||||
const seenKey = new Set<string>();
|
||||
const collectFrom = (buckets: Map<string, Map<string, SymbolDefinition[]>>): void => {
|
||||
|
|
@ -621,12 +657,13 @@ export function pickCppAdlCandidates(
|
|||
function collectAssociatedNamespacesForAdlArg(
|
||||
arg: CppAdlArgInfo,
|
||||
scopes: ScopeResolutionIndexes,
|
||||
idx: AdlCandidateIndex,
|
||||
associatedNamespaces: Set<string>,
|
||||
): void {
|
||||
// For template args this may be the template name itself (e.g. `vector`);
|
||||
// simple-name lookup can match project classes with the same name (known
|
||||
// V1/V2 simplification).
|
||||
addAssociatedNamespaceForClassName(arg.simpleClassName, scopes, associatedNamespaces);
|
||||
addAssociatedNamespaceForClassName(arg.simpleClassName, scopes, idx, associatedNamespaces);
|
||||
|
||||
// Includes template-owner namespaces (e.g. `std` in std::vector<T>). If
|
||||
// that surfaces extra candidates, merged-candidate overload narrowing in
|
||||
|
|
@ -637,17 +674,18 @@ function collectAssociatedNamespacesForAdlArg(
|
|||
if (ns.length > 0) associatedNamespaces.add(ns);
|
||||
}
|
||||
for (const className of arg.templateArgClassNames) {
|
||||
addAssociatedNamespaceForClassName(className, scopes, associatedNamespaces);
|
||||
addAssociatedNamespaceForClassName(className, scopes, idx, associatedNamespaces);
|
||||
}
|
||||
}
|
||||
|
||||
function addAssociatedNamespaceForClassName(
|
||||
simpleClassName: string,
|
||||
scopes: ScopeResolutionIndexes,
|
||||
idx: AdlCandidateIndex,
|
||||
associatedNamespaces: Set<string>,
|
||||
): void {
|
||||
if (simpleClassName.length === 0) return;
|
||||
const classLookup = findCppClassDefBySimpleName(simpleClassName);
|
||||
const classLookup = findCppClassDefBySimpleName(idx, simpleClassName);
|
||||
if (classLookup === undefined) return;
|
||||
const { classDef, ambiguous } = classLookup;
|
||||
const nsQName = classToNamespaceQualifiedName.get(classDef.nodeId);
|
||||
|
|
@ -762,11 +800,12 @@ function findNamespaceDefInScope(scope: {
|
|||
* ISO C++ `[basic.lookup.argdep]` §2: enumerations contribute their
|
||||
* enclosing namespace to the associated set, just like class types. */
|
||||
function findCppClassDefBySimpleName(
|
||||
idx: AdlCandidateIndex,
|
||||
simpleName: string,
|
||||
): { classDef: SymbolDefinition; ambiguous: boolean } | undefined {
|
||||
// `classDefsBySimple` preserves `scopes.defs.byId` order, so `[0]` is the
|
||||
// legacy first-match and `length > 1` is the legacy `ambiguous` flag.
|
||||
const matches = adlIndex?.classDefsBySimple.get(simpleName);
|
||||
const matches = idx.classDefsBySimple.get(simpleName);
|
||||
if (matches === undefined) return undefined;
|
||||
const first = matches[0];
|
||||
if (first === undefined) return undefined;
|
||||
|
|
@ -780,10 +819,9 @@ function findCppClassDefBySimpleName(
|
|||
function collectFunctionTypeAssociatedNamespaces(
|
||||
refText: string,
|
||||
scopes: ScopeResolutionIndexes,
|
||||
idx: AdlCandidateIndex,
|
||||
out: Set<string>,
|
||||
): void {
|
||||
const idx = adlIndex;
|
||||
if (idx === undefined) return;
|
||||
const colonIdx = refText.lastIndexOf('::');
|
||||
if (colonIdx !== -1) {
|
||||
// Qualified ref: extract namespace prefix and normalise :: → dot notation.
|
||||
|
|
@ -796,7 +834,7 @@ function collectFunctionTypeAssociatedNamespaces(
|
|||
// contributing `a` to the associated set (false-positive CALLS edge).
|
||||
const matches = idx.nsFunctionsByQName.get(nsText)?.get(simpleName);
|
||||
if (matches !== undefined) {
|
||||
for (const def of matches) collectAssociatedNamespacesForFunctionDef(def, scopes, out);
|
||||
for (const def of matches) collectAssociatedNamespacesForFunctionDef(def, scopes, idx, out);
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
|
@ -807,32 +845,34 @@ function collectFunctionTypeAssociatedNamespaces(
|
|||
// the function's own enclosing namespace.
|
||||
const matches = idx.nsFunctionsBySimple.get(refText);
|
||||
if (matches !== undefined) {
|
||||
for (const def of matches) collectAssociatedNamespacesForFunctionDef(def, scopes, out);
|
||||
for (const def of matches) collectAssociatedNamespacesForFunctionDef(def, scopes, idx, out);
|
||||
}
|
||||
}
|
||||
|
||||
function collectAssociatedNamespacesForFunctionDef(
|
||||
def: SymbolDefinition,
|
||||
scopes: ScopeResolutionIndexes,
|
||||
idx: AdlCandidateIndex,
|
||||
out: Set<string>,
|
||||
): void {
|
||||
const parameterTypes = def.parameterTypeClasses?.map((typeClass) => typeClass.base);
|
||||
for (const paramType of parameterTypes ?? def.parameterTypes ?? []) {
|
||||
collectAssociatedNamespacesForFunctionTypeText(paramType, scopes, out);
|
||||
collectAssociatedNamespacesForFunctionTypeText(paramType, scopes, idx, out);
|
||||
}
|
||||
if (def.returnType !== undefined) {
|
||||
collectAssociatedNamespacesForFunctionTypeText(def.returnType, scopes, out);
|
||||
collectAssociatedNamespacesForFunctionTypeText(def.returnType, scopes, idx, out);
|
||||
}
|
||||
}
|
||||
|
||||
function collectAssociatedNamespacesForFunctionTypeText(
|
||||
typeText: string,
|
||||
scopes: ScopeResolutionIndexes,
|
||||
idx: AdlCandidateIndex,
|
||||
out: Set<string>,
|
||||
): void {
|
||||
for (const token of extractCppTypeNameTokens(typeText)) {
|
||||
if (isIgnoredCppAdlNamespace(token.namespaceName)) continue;
|
||||
addAssociatedNamespaceForClassName(token.simpleName, scopes, out);
|
||||
addAssociatedNamespaceForClassName(token.simpleName, scopes, idx, out);
|
||||
if (token.namespaceName !== '') out.add(token.namespaceName);
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -14,12 +14,14 @@
|
|||
* 2. **Transitive qualified visibility.** `outer::foo()` resolves to
|
||||
* `outer::v1::foo()` when `v1` is inline. The qualified-namespace
|
||||
* receiver resolver (`resolveCppQualifiedNamespaceMember`) walks
|
||||
* inline-namespace children transitively when collecting candidates.
|
||||
* inline-namespace children transitively when collecting candidates —
|
||||
* once per pipeline run, into {@link QualifiedNsMemberIndex} (#2788).
|
||||
*
|
||||
* State lifecycle: capture-time `markCppInlineNamespaceRange` records each
|
||||
* inline namespace's source range; `populateCppInlineNamespaceScopes`
|
||||
* resolves ranges to `ScopeId`s during `populateOwners`. Cleared via
|
||||
* `clearCppInlineNamespaces`, called from `clearFileLocalNames`.
|
||||
* `clearCppInlineNamespaces`, called from
|
||||
* `cppScopeResolver.loadResolutionConfig` at the start of every pass.
|
||||
*
|
||||
* STL idiom this enables: `std::__1::vector` (libc++) and `std::__cxx11`
|
||||
* (libstdc++) are inline namespaces of `std`. With this support,
|
||||
|
|
@ -32,7 +34,10 @@ import type { ScopeResolutionIndexes } from '../../model/scope-resolution-indexe
|
|||
import {
|
||||
isOverloadAmbiguousAfterNormalization,
|
||||
narrowOverloadCandidates,
|
||||
type OverloadNarrowingHookCtx,
|
||||
} from '../../scope-resolution/passes/overload-narrowing.js';
|
||||
import { isOverloadableCallable } from '../../utils/callable-labels.js';
|
||||
import { isSemanticModelValidatorEnabled } from '../../utils/env.js';
|
||||
import { CPP_CONVERSION_ONLY_ARG_TYPE_PREFIXES, cppConversionRank } from './conversion-rank.js';
|
||||
|
||||
interface RangeKey {
|
||||
|
|
@ -44,6 +49,11 @@ interface RangeKey {
|
|||
|
||||
const inlineNamespaceRangesByFile = new Map<string, Set<string>>();
|
||||
const inlineNamespaceScopeIds = new Set<ScopeId>();
|
||||
/** Bumped by every writer of {@link inlineNamespaceScopeIds}. The qualified-ns
|
||||
* index is a function of both `parsedFiles` and that Set; its WeakMap key sees
|
||||
* only the first, so the memo stamps this epoch and a missed
|
||||
* {@link clearCppInlineNamespaces} degrades to a rebuild, not a stale answer. */
|
||||
let inlineNamespaceEpoch = 0;
|
||||
|
||||
function rangeKey(r: RangeKey): string {
|
||||
return `${r.startLine}:${r.startCol}:${r.endLine}:${r.endCol}`;
|
||||
|
|
@ -85,10 +95,19 @@ export function applyCppInlineNamespaceSideChannel(
|
|||
for (const r of ranges) set.add(r);
|
||||
}
|
||||
|
||||
/** Clear all inline-namespace state. Called from `clearFileLocalNames`. */
|
||||
/** Clear all inline-namespace state. Called from
|
||||
* `cppScopeResolver.loadResolutionConfig` at the start of every pass.
|
||||
*
|
||||
* The qualified-namespace index is dropped by REASSIGNING a fresh `WeakMap`
|
||||
* (`WeakMap` has no `.clear()`) so its entries — and the `SymbolDefinition`
|
||||
* references they hold — are released here rather than waiting on the key.
|
||||
* Correctness does not rest on that: {@link inlineNamespaceEpoch} invalidates
|
||||
* any surviving entry. */
|
||||
export function clearCppInlineNamespaces(): void {
|
||||
inlineNamespaceRangesByFile.clear();
|
||||
inlineNamespaceScopeIds.clear();
|
||||
inlineNamespaceEpoch++;
|
||||
qualifiedNsIndexByPass = new WeakMap();
|
||||
}
|
||||
|
||||
/** Resolve captured ranges to actual ScopeIds by matching scope ranges
|
||||
|
|
@ -98,6 +117,7 @@ export function clearCppInlineNamespaces(): void {
|
|||
export function populateCppInlineNamespaceScopes(parsed: ParsedFile): void {
|
||||
const ranges = inlineNamespaceRangesByFile.get(parsed.filePath);
|
||||
if (ranges === undefined || ranges.size === 0) return;
|
||||
inlineNamespaceEpoch++;
|
||||
for (const scope of parsed.scopes) {
|
||||
if (scope.kind !== 'Namespace') continue;
|
||||
if (ranges.has(rangeKey(scope.range))) {
|
||||
|
|
@ -115,17 +135,316 @@ export function isCppInlineNamespaceScope(scopeId: ScopeId): boolean {
|
|||
}
|
||||
|
||||
/**
|
||||
* Walk every parsed file looking for a Namespace scope whose qualified
|
||||
* name matches `receiverName`, collect its callable ownedDefs matching
|
||||
* `memberName`, transitively descending into any inline-namespace
|
||||
* children (since they're members of the enclosing namespace under ISO
|
||||
* C++).
|
||||
* Qualified-namespace member index — built **once** per pipeline run from
|
||||
* `parsedFiles` and reused by every qualified call site.
|
||||
*
|
||||
* The legacy lookup re-scanned every parsed file (rebuilding a per-file
|
||||
* `scopesById` map each time) once **per qualified call site**, making the
|
||||
* scope-resolution emit phase O(callsites × scopes): 25.3 min of a 33-min
|
||||
* analyze on a 1,473-file C++ repo, 75% of total self-time in this one
|
||||
* function (#2788). Mirrors the same fix #1990 applied to ADL
|
||||
* (`pickCppAdlCandidates` → {@link AdlCandidateIndex}); per-site cost drops
|
||||
* to a Map lookup once a `(receiver, member)` pair has been resolved.
|
||||
*
|
||||
* **Why the index is a graph and not a flattened member table.** Every
|
||||
* `Namespace` scope is a legal qualified receiver on its own — in
|
||||
* `outer { inline v1 { inline v2 { … } } }`, `v2::foo()` is valid C++ — so an
|
||||
* eager table has to record, for every namespace, every member of every inline
|
||||
* descendant. On a chain of depth D that is D + (D−1) + … + 1 entries: quadratic
|
||||
* in **both** time and memory, and materializing it recursively also recursed D
|
||||
* deep. Neither bound was theoretical — a generated chain 6,000 deep with one
|
||||
* function per level exhausted a 4 GB heap, and a member-less one 8,000 deep
|
||||
* threw `RangeError: Maximum call stack size exceeded`. Nothing on the analyze
|
||||
* path catches either (`run.ts` only wraps the CFG/PDG emit block; `phase.ts`'s
|
||||
* `try` has a `finally` and no `catch`), so one pathological generated file
|
||||
* aborted the whole `analyze` — the failure class of #2769.
|
||||
*
|
||||
* Storing the *shape* instead — each namespace's own members plus links to its
|
||||
* inline children — makes the build linear in scopes+defs, and a call site pays
|
||||
* one iterative pre-order walk of the receiver's inline subtree, memoized per
|
||||
* `(receiver, member)`. Real inline nesting is 1–2 deep (`std::__1`,
|
||||
* `std::__cxx11`), so that walk is 2–3 nodes.
|
||||
*/
|
||||
interface QualifiedNsMemberIndex {
|
||||
/**
|
||||
* Namespace simple name → the scopes declaring it, in the order the legacy
|
||||
* linear scan visited them (file-major, then `parsed.scopes` declaration
|
||||
* order). Nothing downstream observes that order — every multi-candidate
|
||||
* path in {@link resolveCppQualifiedNamespaceMember} either narrows to a
|
||||
* unique survivor or returns `'ambiguous'` — but it is free to preserve, and
|
||||
* a future tie-break should inherit the legacy order rather than Map
|
||||
* insertion happenstance.
|
||||
*/
|
||||
readonly rootsByReceiver: ReadonlyMap<string, readonly QualifiedNsNode[]>;
|
||||
/**
|
||||
* Memo of resolved candidate lists, receiver → member → defs. Filled on
|
||||
* demand by {@link qualifiedNsMembers}; only pairs a call site actually asks
|
||||
* for are ever materialized, which is what keeps the deep-chain case linear.
|
||||
*/
|
||||
readonly membersByReceiver: Map<string, Map<string, readonly SymbolDefinition[]>>;
|
||||
}
|
||||
|
||||
/** One `Namespace` scope as the qualified-lookup walk sees it. */
|
||||
interface QualifiedNsNode {
|
||||
/** This scope's OWN callable defs bucketed by member simple name, in
|
||||
* `ownedDefs` order. Computed once at build time, so a namespace with many
|
||||
* members costs one pass no matter how many distinct members are queried. */
|
||||
readonly ownMembers: ReadonlyMap<string, readonly SymbolDefinition[]>;
|
||||
/** Inline-namespace children in `parsed.scopes` declaration order. Direct
|
||||
* object links rather than `ScopeId` lookups, so the walk needs no scope
|
||||
* table and ids that repeat across files can never splice one file's
|
||||
* children under another file's namespace. */
|
||||
readonly inlineChildren: readonly QualifiedNsNode[];
|
||||
}
|
||||
|
||||
/** Build-time view of {@link QualifiedNsNode}: `inlineChildren` is appended to
|
||||
* as the build's second pass links each child, then only ever read. */
|
||||
interface MutableNsNode extends QualifiedNsNode {
|
||||
readonly inlineChildren: QualifiedNsNode[];
|
||||
}
|
||||
|
||||
type NsScope = ParsedFile['scopes'][number];
|
||||
|
||||
/**
|
||||
* Dev/test-only mutation tripwire on the arrays this module shares across call
|
||||
* sites ({@link NO_DEFS} and every `membersByReceiver` memo entry): an in-place
|
||||
* `.sort()`/`.splice()` ever added to `overload-narrowing.ts` would corrupt
|
||||
* every LATER resolution of the same pair, not just its own call, and freezing
|
||||
* makes that a loud `TypeError` instead. DEV-ONLY because the types already
|
||||
* reject it — the memo and `narrowOverloadCandidates`' parameter are both
|
||||
* `readonly SymbolDefinition[]`, so only a cast gets past — while the freeze is
|
||||
* not free: V8 moves a frozen array to `PACKED_FROZEN_ELEMENTS`, off the
|
||||
* builtin fast path for the `.filter`/`.map`/`.some` narrowing runs over it at
|
||||
* every multi-candidate call site (measured 4.6×; large bench arm 100.3 → 70.6ms
|
||||
* with it off). Gated on `isSemanticModelValidatorEnabled()` — the OPT-IN form,
|
||||
* not `adl.ts`'s opt-out `NODE_ENV !== 'production'`: `NODE_ENV` is unset in a
|
||||
* CLI `analyze`, so the opt-out form would keep paying the freeze exactly where
|
||||
* the cost lands. Read once at load because this one is per-call-site.
|
||||
*/
|
||||
const FREEZE_SHARED_CANDIDATES = isSemanticModelValidatorEnabled();
|
||||
|
||||
/** Shared empties: most namespace scopes declare no callables, and most
|
||||
* qualified receivers name no namespace at all. `NO_DEFS` is process-wide, so
|
||||
* it gets the same dev-only freeze the memoized arrays do. */
|
||||
const NO_MEMBERS: ReadonlyMap<string, readonly SymbolDefinition[]> = new Map();
|
||||
const NO_DEFS: readonly SymbolDefinition[] = FREEZE_SHARED_CANDIDATES ? Object.freeze([]) : [];
|
||||
|
||||
/** Simple (last-segment) name of a def, matching the legacy scan exactly —
|
||||
* including the empty-string fallback for a def with no `qualifiedName`, and
|
||||
* the empty last segment of a trailing-dot name. `lastIndexOf` + `slice`
|
||||
* rather than `split('.').pop()`: same result, no intermediate array, and this
|
||||
* runs per callable def and per namespace scope at build. */
|
||||
function simpleNameOf(def: SymbolDefinition): string {
|
||||
const qualified = def.qualifiedName;
|
||||
if (qualified == null) return '';
|
||||
const lastDot = qualified.lastIndexOf('.');
|
||||
return lastDot === -1 ? qualified : qualified.slice(lastDot + 1);
|
||||
}
|
||||
|
||||
/**
|
||||
* Per-pass memo: `parsedFiles` array identity → the built index. `WeakMap`-keyed
|
||||
* so the entry lives only as long as the caller's `parsedFiles` array does, and
|
||||
* is reclaimed with the pass — mirrors `moduleScopeIndexByPass` in
|
||||
* `file-local-linkage.ts`.
|
||||
*
|
||||
* Weak keying is load-bearing, not stylistic: the index holds
|
||||
* `SymbolDefinition` references reaching into every ParsedFile's scopes, so a
|
||||
* module-level strong `let` pair (index + source array) kept the whole C/C++
|
||||
* ParsedFile set alive past the point `scope-resolution/pipeline/phase.ts`
|
||||
* evicts `files`/`contents`/`preExtractedByPath` and calls `forceGc()` to
|
||||
* reclaim it — C++ is 7th of 16 `SCOPE_RESOLVERS` entries, so the retention
|
||||
* survived nine later language passes plus emit (104.2 MB measured).
|
||||
*/
|
||||
let qualifiedNsIndexByPass = new WeakMap<readonly ParsedFile[], MemoizedIndex>();
|
||||
|
||||
/** Memo cell: the index plus the {@link inlineNamespaceEpoch} it was built under. */
|
||||
interface MemoizedIndex {
|
||||
readonly epoch: number;
|
||||
readonly index: QualifiedNsMemberIndex;
|
||||
}
|
||||
|
||||
/** Build the index in two linear passes per file: one to make a node per
|
||||
* `Namespace` scope, one to link inline children and register receivers.
|
||||
* Linking needs both endpoints to exist and `parsed.scopes` does not promise
|
||||
* parents precede children, hence two passes rather than one — but only the
|
||||
* first pass filters `parsed.scopes`; it hands the second the `[scope, node]`
|
||||
* pairs it already found. */
|
||||
function buildQualifiedNsMemberIndex(parsedFiles: readonly ParsedFile[]): QualifiedNsMemberIndex {
|
||||
const rootsByReceiver = new Map<string, QualifiedNsNode[]>();
|
||||
|
||||
for (const parsed of parsedFiles) {
|
||||
const nodesByScope = new Map<ScopeId, MutableNsNode>();
|
||||
const namespaces: [NsScope, MutableNsNode][] = [];
|
||||
for (const sc of parsed.scopes) {
|
||||
if (sc.kind !== 'Namespace') continue;
|
||||
const node: MutableNsNode = { ownMembers: bucketOwnMembers(sc), inlineChildren: [] };
|
||||
nodesByScope.set(sc.id, node);
|
||||
namespaces.push([sc, node]);
|
||||
}
|
||||
|
||||
for (const [sc, node] of namespaces) {
|
||||
// A non-`Namespace` parent (a Module scope, say) has no node, so the
|
||||
// inline child links to nothing — hence the `?.` below.
|
||||
if (sc.parent !== null && inlineNamespaceScopeIds.has(sc.id)) {
|
||||
nodesByScope.get(sc.parent)?.inlineChildren.push(node);
|
||||
}
|
||||
const nsDef = findNamespaceDefInScope(sc);
|
||||
if (nsDef === undefined) continue;
|
||||
const nsName = simpleNameOf(nsDef);
|
||||
let roots = rootsByReceiver.get(nsName);
|
||||
if (roots === undefined) {
|
||||
roots = [];
|
||||
rootsByReceiver.set(nsName, roots);
|
||||
}
|
||||
roots.push(node);
|
||||
}
|
||||
}
|
||||
return { rootsByReceiver, membersByReceiver: new Map() };
|
||||
}
|
||||
|
||||
/** Bucket a namespace scope's own callable `ownedDefs` by member simple name.
|
||||
* No descent: inline children are separate nodes reached by the walk. */
|
||||
function bucketOwnMembers(scope: NsScope): ReadonlyMap<string, readonly SymbolDefinition[]> {
|
||||
let members: Map<string, SymbolDefinition[]> | undefined;
|
||||
for (const def of scope.ownedDefs) {
|
||||
if (!isOverloadableCallable(def.type)) continue;
|
||||
const simple = simpleNameOf(def);
|
||||
members ??= new Map();
|
||||
let arr = members.get(simple);
|
||||
if (arr === undefined) {
|
||||
arr = [];
|
||||
members.set(simple, arr);
|
||||
}
|
||||
arr.push(def);
|
||||
}
|
||||
return members ?? NO_MEMBERS;
|
||||
}
|
||||
|
||||
/** Candidates for `receiver::member`, memoized on the index per pair so a
|
||||
* repeated call site costs two Map lookups. */
|
||||
function qualifiedNsMembers(
|
||||
index: QualifiedNsMemberIndex,
|
||||
receiverName: string,
|
||||
memberName: string,
|
||||
): readonly SymbolDefinition[] {
|
||||
const roots = index.rootsByReceiver.get(receiverName);
|
||||
// No namespace by that name — the most common outcome in real source. Left
|
||||
// unmemoized deliberately: it is already O(1), and memoizing would grow the
|
||||
// index by an entry per unresolved receiver name in the workspace.
|
||||
if (roots === undefined) return NO_DEFS;
|
||||
let byMember = index.membersByReceiver.get(receiverName);
|
||||
if (byMember === undefined) {
|
||||
byMember = new Map();
|
||||
index.membersByReceiver.set(receiverName, byMember);
|
||||
}
|
||||
let hits = byMember.get(memberName);
|
||||
if (hits === undefined) {
|
||||
// Memoization makes this array shared by every call site in the pass rather
|
||||
// than rebuilt per call, so it carries the dev-only mutation tripwire (see
|
||||
// {@link FREEZE_SHARED_CANDIDATES}).
|
||||
hits = gatherQualifiedNsMember(roots, memberName);
|
||||
if (FREEZE_SHARED_CANDIDATES) hits = Object.freeze(hits);
|
||||
byMember.set(memberName, hits);
|
||||
}
|
||||
return hits;
|
||||
}
|
||||
|
||||
/** Pre-order depth-first walk of every scope named `receiverName` and its
|
||||
* transitive inline-namespace children, collecting callables named
|
||||
* `memberName` — the index twin of the legacy
|
||||
* `findMemberInNamespaceTransitive`.
|
||||
*
|
||||
* Iterative, not recursive: nesting depth is whatever the input file says, and
|
||||
* a recursive descent threw `RangeError` past ~8k levels (see
|
||||
* {@link QualifiedNsMemberIndex}). */
|
||||
function gatherQualifiedNsMember(
|
||||
roots: readonly QualifiedNsNode[],
|
||||
memberName: string,
|
||||
): readonly SymbolDefinition[] {
|
||||
const hits: SymbolDefinition[] = [];
|
||||
const stack: QualifiedNsNode[] = [];
|
||||
// `visited` spans ALL roots, reproducing the legacy per-receiver `seenNodeId`
|
||||
// dedup: `namespace ns { inline namespace ns { … } }` registers both scopes
|
||||
// under receiver `ns`, and the outer walk already collected the inner's defs.
|
||||
// Node identity is equivalent to the legacy nodeId key because a node always
|
||||
// contributes the same defs. It doubles as the guard that a malformed parent
|
||||
// cycle terminates instead of spinning — the one regression an explicit stack
|
||||
// could otherwise introduce over recursion's `RangeError`.
|
||||
const visited = new Set<QualifiedNsNode>();
|
||||
for (const root of roots) {
|
||||
stack.push(root);
|
||||
while (stack.length > 0) {
|
||||
const node = stack.pop();
|
||||
if (visited.has(node)) continue;
|
||||
visited.add(node);
|
||||
const own = node.ownMembers.get(memberName);
|
||||
if (own !== undefined) for (const def of own) hits.push(def);
|
||||
// Reverse push so children pop in declaration order — the legacy walk's
|
||||
// pre-order depth-first candidate order (see
|
||||
// {@link QualifiedNsMemberIndex.rootsByReceiver}).
|
||||
const kids = node.inlineChildren;
|
||||
for (let i = kids.length - 1; i >= 0; i--) stack.push(kids[i]);
|
||||
}
|
||||
}
|
||||
return hits.length === 0 ? NO_DEFS : hits;
|
||||
}
|
||||
|
||||
/** Build the index on first use of a given `parsedFiles` set; reuse it for
|
||||
* every subsequent call site that passes the same array reference AND the same
|
||||
* {@link inlineNamespaceEpoch}.
|
||||
*
|
||||
* Both keys are needed because the index is a function of TWO inputs:
|
||||
* `parsedFiles` and the module-level `inlineNamespaceScopeIds` (which inline
|
||||
* children get descended into). A later pass may hand back the same
|
||||
* `parsedFiles` reference with different inline-namespace state, so identity
|
||||
* alone would serve a stale index — the epoch is what makes that a checked
|
||||
* rebuild instead of a documented obligation on the clear site.
|
||||
*
|
||||
* Lifetime: the entry is reachable only while the caller still holds the
|
||||
* `parsedFiles` array; nothing here pins it (see {@link qualifiedNsIndexByPass}).
|
||||
*
|
||||
* Same STALENESS contract as `ensureAdlIndex` in `adl.ts`, minus that
|
||||
* sibling's dev-gated `validateAdlSeqCoverage` guard: this index has no `?? 0`
|
||||
* defaulting read to silently collide on, so there is nothing for such a guard
|
||||
* to catch. */
|
||||
function qualifiedNsMemberIndex(parsedFiles: readonly ParsedFile[]): QualifiedNsMemberIndex {
|
||||
const memo = qualifiedNsIndexByPass.get(parsedFiles);
|
||||
if (memo !== undefined && memo.epoch === inlineNamespaceEpoch) return memo.index;
|
||||
const index = buildQualifiedNsMemberIndex(parsedFiles);
|
||||
qualifiedNsIndexByPass.set(parsedFiles, { epoch: inlineNamespaceEpoch, index });
|
||||
return index;
|
||||
}
|
||||
|
||||
/** Constant narrowing hooks for the C++ qualified-receiver path — hoisted so
|
||||
* the literal is not reallocated at every multi-candidate call site. */
|
||||
const CPP_NARROWING_HOOKS: OverloadNarrowingHookCtx = {
|
||||
conversionRankFn: cppConversionRank,
|
||||
conversionOnlyArgTypePrefixes: CPP_CONVERSION_ONLY_ARG_TYPE_PREFIXES,
|
||||
};
|
||||
|
||||
/**
|
||||
* Find the Namespace scopes whose simple name matches `receiverName` and
|
||||
* return their callable members matching `memberName`, transitively
|
||||
* including inline-namespace children (since they're members of the
|
||||
* enclosing namespace under ISO C++). Served from a per-pipeline index
|
||||
* ({@link QualifiedNsMemberIndex}), not a per-call-site workspace scan.
|
||||
*
|
||||
* Returns the most specific (innermost) match — for `outer::foo()`
|
||||
* where `inline namespace v1` declares `foo`, returns `v1::foo`. When
|
||||
* multiple inline-namespace children declare the same name, ISO C++
|
||||
* leaves the call ambiguous; returns `'ambiguous'` so the caller
|
||||
* suppresses edge emission rather than picking arbitrarily (#1564).
|
||||
*
|
||||
* Two production call sites, both in `scope-resolver.ts`:
|
||||
* - `resolveQualifiedReceiverMember` — the `outer::foo()` qualified-receiver
|
||||
* hook. Passes `callsite`, so a multi-candidate set can be narrowed by
|
||||
* arity and argument types.
|
||||
* - `resolveAdlCandidates` — resolving a `using ns::name;` named import back
|
||||
* to its namespace member, for template-class method bodies where the
|
||||
* lexical walk misses that visibility. Passes NO `callsite`, which makes
|
||||
* the narrowing filters below pass-throughs: on that path any receiver
|
||||
* whose member has more than one candidate returns `'ambiguous'` (and the
|
||||
* caller then skips it) rather than being disambiguated.
|
||||
*/
|
||||
export function resolveCppQualifiedNamespaceMember(
|
||||
receiverName: string,
|
||||
|
|
@ -134,27 +453,7 @@ export function resolveCppQualifiedNamespaceMember(
|
|||
_scopes: ScopeResolutionIndexes,
|
||||
callsite?: Callsite,
|
||||
): SymbolDefinition | 'ambiguous' | undefined {
|
||||
const allHits: SymbolDefinition[] = [];
|
||||
const seenNodeId = new Set<string>();
|
||||
for (const parsed of parsedFiles) {
|
||||
const scopesById = new Map<ScopeId, (typeof parsed.scopes)[number]>();
|
||||
for (const sc of parsed.scopes) scopesById.set(sc.id, sc);
|
||||
for (const scope of parsed.scopes) {
|
||||
if (scope.kind !== 'Namespace') continue;
|
||||
const nsDef = findNamespaceDefInScope(scope);
|
||||
if (nsDef === undefined) continue;
|
||||
const nsName = nsDef.qualifiedName?.split('.').pop() ?? nsDef.qualifiedName ?? '';
|
||||
if (nsName !== receiverName) continue;
|
||||
// Found a matching namespace scope in this file. Collect ALL
|
||||
// members transitively through any inline-namespace children.
|
||||
const hits = findMemberInNamespaceTransitive(scope, scopesById, memberName);
|
||||
for (const hit of hits) {
|
||||
if (seenNodeId.has(hit.nodeId)) continue;
|
||||
seenNodeId.add(hit.nodeId);
|
||||
allHits.push(hit);
|
||||
}
|
||||
}
|
||||
}
|
||||
const allHits = qualifiedNsMembers(qualifiedNsMemberIndex(parsedFiles), receiverName, memberName);
|
||||
if (allHits.length === 0) return undefined;
|
||||
if (allHits.length === 1) return allHits[0];
|
||||
|
||||
|
|
@ -167,12 +466,7 @@ export function resolveCppQualifiedNamespaceMember(
|
|||
allHits,
|
||||
callsite?.arity,
|
||||
callsite?.argumentTypes,
|
||||
callsite !== undefined
|
||||
? {
|
||||
conversionRankFn: cppConversionRank,
|
||||
conversionOnlyArgTypePrefixes: CPP_CONVERSION_ONLY_ARG_TYPE_PREFIXES,
|
||||
}
|
||||
: undefined,
|
||||
callsite !== undefined ? CPP_NARROWING_HOOKS : undefined,
|
||||
);
|
||||
if (narrowed.length === 1) return narrowed[0];
|
||||
if (narrowed.length === 0) return undefined;
|
||||
|
|
@ -182,46 +476,6 @@ export function resolveCppQualifiedNamespaceMember(
|
|||
return 'ambiguous';
|
||||
}
|
||||
|
||||
/** Recursively search a namespace scope and any inline-namespace
|
||||
* descendants for callable defs with the given simple name. Non-inline
|
||||
* nested namespaces are NOT traversed — they require explicit
|
||||
* qualification (`outer::nested::foo`). Returns ALL matches so the
|
||||
* caller can detect same-name ambiguity across inline children (#1564). */
|
||||
function findMemberInNamespaceTransitive(
|
||||
scope: {
|
||||
readonly id: ScopeId;
|
||||
readonly ownedDefs: readonly SymbolDefinition[];
|
||||
readonly parent: ScopeId | null;
|
||||
},
|
||||
scopesById: ReadonlyMap<
|
||||
ScopeId,
|
||||
{
|
||||
readonly id: ScopeId;
|
||||
readonly kind: string;
|
||||
readonly parent: ScopeId | null;
|
||||
readonly ownedDefs: readonly SymbolDefinition[];
|
||||
}
|
||||
>,
|
||||
memberName: string,
|
||||
): SymbolDefinition[] {
|
||||
const results: SymbolDefinition[] = [];
|
||||
// Check this scope's own ownedDefs first.
|
||||
for (const def of scope.ownedDefs) {
|
||||
if (def.type !== 'Function' && def.type !== 'Method' && def.type !== 'Constructor') continue;
|
||||
const simple = def.qualifiedName?.split('.').pop() ?? def.qualifiedName ?? '';
|
||||
if (simple === memberName) results.push(def);
|
||||
}
|
||||
// Descend into inline-namespace children.
|
||||
for (const childScope of scopesById.values()) {
|
||||
if (childScope.parent !== scope.id) continue;
|
||||
if (childScope.kind !== 'Namespace') continue;
|
||||
if (!inlineNamespaceScopeIds.has(childScope.id)) continue;
|
||||
const childHits = findMemberInNamespaceTransitive(childScope, scopesById, memberName);
|
||||
for (const hit of childHits) results.push(hit);
|
||||
}
|
||||
return results;
|
||||
}
|
||||
|
||||
function findNamespaceDefInScope(scope: {
|
||||
readonly ownedDefs: readonly SymbolDefinition[];
|
||||
}): SymbolDefinition | undefined {
|
||||
|
|
|
|||
|
|
@ -54,17 +54,19 @@ import { definitionIdPosition } from '../utils/definition-id.js';
|
|||
* restricted to function/class-likes, those calls correctly fall
|
||||
* through to the File-node fallback at the bottom of the walk.
|
||||
*/
|
||||
export const CALLER_ANCHOR_LABELS: ReadonlySet<NodeLabel> = new Set<NodeLabel>([
|
||||
'Function',
|
||||
'Method',
|
||||
'Constructor',
|
||||
'Module',
|
||||
'Class',
|
||||
'Interface',
|
||||
'Struct',
|
||||
'Enum',
|
||||
]);
|
||||
|
||||
function isCallerAnchorLabel(label: NodeLabel): boolean {
|
||||
return (
|
||||
label === 'Function' ||
|
||||
label === 'Method' ||
|
||||
label === 'Constructor' ||
|
||||
label === 'Module' ||
|
||||
label === 'Class' ||
|
||||
label === 'Interface' ||
|
||||
label === 'Struct' ||
|
||||
label === 'Enum'
|
||||
);
|
||||
return CALLER_ANCHOR_LABELS.has(label);
|
||||
}
|
||||
|
||||
function rangeContainsPoint(
|
||||
|
|
|
|||
|
|
@ -244,38 +244,53 @@ export function buildGraphNodeLookup(graph: KnowledgeGraph): GraphNodeLookup {
|
|||
return lookup;
|
||||
}
|
||||
|
||||
/**
|
||||
* Every label {@link buildGraphNodeLookup} registers — and therefore the ONLY
|
||||
* labels `resolveDefGraphId` can ever return an id for. Both endpoints of every
|
||||
* scope-resolution edge come from that lookup (the one exception is the File
|
||||
* fallback in `resolveCallerGraphId`), so this set defines the whole FROM/TO
|
||||
* surface those edges can produce.
|
||||
*
|
||||
* That makes it load-bearing for the LadybugDB relation DDL: a label added here
|
||||
* without the matching `FROM x TO y` pairs in `RELATION_SCHEMA` crashes
|
||||
* `analyze` at `assertDeclaredPair` on whichever codebase first emits the pair
|
||||
* (#2792). `test/unit/schema-pair-coverage.test.ts` derives the required pairs
|
||||
* from this set and fails in CI instead.
|
||||
*/
|
||||
export const LINKABLE_LABELS: ReadonlySet<NodeLabel> = new Set<NodeLabel>([
|
||||
'Function',
|
||||
'Method',
|
||||
'Constructor',
|
||||
// Program-like module declarations are provider-gated callable-value
|
||||
// targets and need the same def→graph bridge.
|
||||
'Module',
|
||||
'Class',
|
||||
'Interface',
|
||||
'Struct',
|
||||
'Enum',
|
||||
// Trait nodes are linkable so MRO builders can bridge PHP/Rust trait
|
||||
// defs between scope-resolution DefIds and the graph's node ids.
|
||||
// IMPLEMENTS edges from classes to traits are otherwise invisible to
|
||||
// the scope-resolution MRO pass.
|
||||
'Trait',
|
||||
// Variable / Property are linkable too — receiver-bound write/read
|
||||
// ACCESSES edges target field nodes (e.g. `user.name = "x"` →
|
||||
// ACCESSES edge to User's `name` Variable/Property node).
|
||||
'Variable',
|
||||
'Property',
|
||||
// Const is linkable so the value-receiver-owner bridge in
|
||||
// `receiver-bound-calls.ts` Case 5 can translate the scope-resolution
|
||||
// `Variable` def for `export const fooService = {...}` to the canonical
|
||||
// `Const:filePath:name` graph node id, against which object-literal
|
||||
// method symbols register their `ownerId` (PR #1718 / issue #1358).
|
||||
'Const',
|
||||
// Macro nodes are linkable so a macro invocation (`log!(…)`) resolved
|
||||
// via `MacroRegistry` can bridge its scope-resolution `Macro` def to
|
||||
// the legacy `@definition.macro` graph node and emit the `USES` edge
|
||||
// (Rust #1934 F72; also covers C/C++ `#define` macro defs).
|
||||
'Macro',
|
||||
]);
|
||||
|
||||
export function isLinkableLabel(label: NodeLabel): boolean {
|
||||
return (
|
||||
label === 'Function' ||
|
||||
label === 'Method' ||
|
||||
label === 'Constructor' ||
|
||||
// Program-like module declarations are provider-gated callable-value
|
||||
// targets and need the same def→graph bridge.
|
||||
label === 'Module' ||
|
||||
label === 'Class' ||
|
||||
label === 'Interface' ||
|
||||
label === 'Struct' ||
|
||||
label === 'Enum' ||
|
||||
// Trait nodes are linkable so MRO builders can bridge PHP/Rust trait
|
||||
// defs between scope-resolution DefIds and the graph's node ids.
|
||||
// IMPLEMENTS edges from classes to traits are otherwise invisible to
|
||||
// the scope-resolution MRO pass.
|
||||
label === 'Trait' ||
|
||||
// Variable / Property are linkable too — receiver-bound write/read
|
||||
// ACCESSES edges target field nodes (e.g. `user.name = "x"` →
|
||||
// ACCESSES edge to User's `name` Variable/Property node).
|
||||
label === 'Variable' ||
|
||||
label === 'Property' ||
|
||||
// Const is linkable so the value-receiver-owner bridge in
|
||||
// `receiver-bound-calls.ts` Case 5 can translate the scope-resolution
|
||||
// `Variable` def for `export const fooService = {...}` to the canonical
|
||||
// `Const:filePath:name` graph node id, against which object-literal
|
||||
// method symbols register their `ownerId` (PR #1718 / issue #1358).
|
||||
label === 'Const' ||
|
||||
// Macro nodes are linkable so a macro invocation (`log!(…)`) resolved
|
||||
// via `MacroRegistry` can bridge its scope-resolution `Macro` def to
|
||||
// the legacy `@definition.macro` graph node and emit the `USES` edge
|
||||
// (Rust #1934 F72; also covers C/C++ `#define` macro defs).
|
||||
label === 'Macro'
|
||||
);
|
||||
return LINKABLE_LABELS.has(label);
|
||||
}
|
||||
|
|
|
|||
|
|
@ -17,8 +17,8 @@ import { createWriteStream, WriteStream } from 'fs';
|
|||
import path from 'path';
|
||||
import type { GraphNode, GraphRelationship } from 'gitnexus-shared';
|
||||
import { KnowledgeGraph } from '../graph/types.js';
|
||||
import { NodeTableName, NODE_TABLES, RELATION_SCHEMA } from './schema.js';
|
||||
import { parseRelationSchemaPairs, RelPairRouter } from './rel-pair-routing.js';
|
||||
import { NodeTableName, RELATION_SCHEMA } from './schema.js';
|
||||
import { VALID_NODE_TABLES, parseRelationSchemaPairs, RelPairRouter } from './rel-pair-routing.js';
|
||||
import { parseTruthyEnv } from '../ingestion/utils/env.js';
|
||||
import { SYMBOL_NODE_LABELS } from '../ingestion/utils/symbol-labels.js';
|
||||
import { applyCjkSegmentationIfEnabled } from '../search/cjk-segmentation.js';
|
||||
|
|
@ -793,13 +793,13 @@ export const streamAllCSVsToDisk = async (
|
|||
const relRouter = new RelPairRouter(
|
||||
csvDir,
|
||||
REL_CSV_HEADER,
|
||||
new Set<string>(NODE_TABLES),
|
||||
VALID_NODE_TABLES,
|
||||
DECLARED_RELATION_PAIRS,
|
||||
);
|
||||
try {
|
||||
let emitted = 0;
|
||||
for (const rel of orderedRelationships(graph, sortOutput)) {
|
||||
const pending = relRouter.route(rel.sourceId, rel.targetId, buildRelRow(rel));
|
||||
const pending = relRouter.route(rel.sourceId, rel.targetId, buildRelRow(rel), rel.type);
|
||||
if (pending) await pending;
|
||||
// Periodically hand the event loop back so the overlapped node COPY and
|
||||
// write-stream drains run instead of starving behind this synchronous
|
||||
|
|
|
|||
|
|
@ -85,9 +85,10 @@
|
|||
* ## Correctness contract
|
||||
*
|
||||
* Structural sibling of {@link PdgEmitSink}, and reuses its row builder
|
||||
* (`buildRelRow`), header (`REL_CSV_HEADER`), label derivation (`getNodeLabel`)
|
||||
* and `RelPairRouter` validity check, so the streamed row SET equals the
|
||||
* whole-graph emit's and the bulk COPY loads the same rows. Set-level, not
|
||||
* (`buildRelRow`), header (`REL_CSV_HEADER`) and pair classification
|
||||
* (`relPairKeyFor`, which is also what `RelPairRouter` routes and skips by), so
|
||||
* the streamed row SET equals the whole-graph emit's and the bulk COPY loads
|
||||
* the same rows. Set-level, not
|
||||
* byte-level: rows stream in emit order and are not re-sorted under
|
||||
* `GITNEXUS_SORT_GRAPH_OUTPUT`.
|
||||
*/
|
||||
|
|
@ -96,8 +97,12 @@ import path from 'path';
|
|||
import type { GraphNode, GraphRelationship, RelationshipType } from 'gitnexus-shared';
|
||||
import type { KnowledgeGraph } from '../graph/types.js';
|
||||
import { DECLARED_RELATION_PAIRS, REL_CSV_HEADER, buildRelRow } from './csv-generator.js';
|
||||
import { assertDeclaredPair, getNodeLabel } from './rel-pair-routing.js';
|
||||
import { NODE_TABLES } from './schema.js';
|
||||
import {
|
||||
VALID_NODE_TABLES,
|
||||
assertDeclaredPair,
|
||||
relPairKeyFor,
|
||||
splitRelPairKey,
|
||||
} from './rel-pair-routing.js';
|
||||
import { DEFAULT_EMIT_CHUNK_ROWS, SyncCsvWriter } from './sync-csv-writer.js';
|
||||
|
||||
/**
|
||||
|
|
@ -229,7 +234,6 @@ export class StreamedRelationshipRemovalError extends Error {
|
|||
* {@link finalize} once after the pipeline, before `loadGraphToLbug`.
|
||||
*/
|
||||
export class GraphEmitSink implements KnowledgeGraph, GraphEmitControl {
|
||||
private readonly validTables: Set<string>;
|
||||
private readonly relWriters = new Map<string, SyncCsvWriter>();
|
||||
/**
|
||||
* Ids of relationships already streamed. `KnowledgeGraph.addRelationship`
|
||||
|
|
@ -303,7 +307,6 @@ export class GraphEmitSink implements KnowledgeGraph, GraphEmitControl {
|
|||
private readonly csvDir: string,
|
||||
private readonly chunkRows: number = DEFAULT_EMIT_CHUNK_ROWS,
|
||||
) {
|
||||
this.validTables = new Set<string>(NODE_TABLES as readonly string[]);
|
||||
// Own directory, distinct from the PDG sink's: PdgEmitSink wipes and
|
||||
// recreates its dir on construction and opens with O_EXCL, so a shared dir
|
||||
// would destroy the other sink's manifest on a combined --pdg run.
|
||||
|
|
@ -407,16 +410,24 @@ export class GraphEmitSink implements KnowledgeGraph, GraphEmitControl {
|
|||
}
|
||||
// Mirror KnowledgeGraph.addRelationship's first-writer-wins dedup.
|
||||
|
||||
const fromLabel = getNodeLabel(relationship.sourceId);
|
||||
const toLabel = getNodeLabel(relationship.targetId);
|
||||
// Skip edges whose endpoint labels are not valid node tables — mirrors
|
||||
// `RelPairRouter` exactly so the streamed set matches the whole-graph set.
|
||||
if (!this.validTables.has(fromLabel) || !this.validTables.has(toLabel)) return;
|
||||
// Classify + skip via the SHARED `relPairKeyFor`, not a local copy of its
|
||||
// three lines, so the streamed set cannot drift from the whole-graph set
|
||||
// `RelPairRouter` produces. `undefined` = an endpoint label is not a node
|
||||
// table, so the edge is dropped exactly as the router drops it.
|
||||
const pairKey = relPairKeyFor(relationship.sourceId, relationship.targetId, VALID_NODE_TABLES);
|
||||
if (pairKey === undefined) return;
|
||||
|
||||
const pairKey = `${fromLabel}|${toLabel}`;
|
||||
assertDeclaredPair(pairKey, DECLARED_RELATION_PAIRS);
|
||||
assertDeclaredPair(
|
||||
pairKey,
|
||||
DECLARED_RELATION_PAIRS,
|
||||
relationship.type,
|
||||
relationship.sourceId,
|
||||
relationship.targetId,
|
||||
);
|
||||
let writer = this.relWriters.get(pairKey);
|
||||
if (writer === undefined) {
|
||||
// Cold: once per pair, so decoding the key back into its labels is free.
|
||||
const [fromLabel, toLabel] = splitRelPairKey(pairKey);
|
||||
try {
|
||||
writer = new SyncCsvWriter(
|
||||
path.join(this.csvDir, `rel_${fromLabel}_${toLabel}.csv`),
|
||||
|
|
|
|||
|
|
@ -24,8 +24,8 @@
|
|||
* `storage/parsedfile-store.ts`.
|
||||
*
|
||||
* Byte-identity (issue acceptance): the sink reuses the SAME shared row
|
||||
* builders (`buildBasicBlockRow`, `buildRelRow`) and label derivation
|
||||
* (`getNodeLabel`) as `streamAllCSVsToDisk`, so the streamed CSV line SET is
|
||||
* builders (`buildBasicBlockRow`, `buildRelRow`) and pair classification
|
||||
* (`relPairKeyFor`) as `streamAllCSVsToDisk`, so the streamed CSV line SET is
|
||||
* identical to the whole-graph emit's, and the bulk COPY loads the same rows →
|
||||
* the persisted graph is SET-identical and DB-identical. The guarantee is
|
||||
* set-level, not byte-level on the CSV file: the sink streams rows in emit
|
||||
|
|
@ -54,9 +54,14 @@ import {
|
|||
buildBasicBlockRow,
|
||||
buildRelRow,
|
||||
} from './csv-generator.js';
|
||||
import { assertDeclaredPair, getNodeLabel } from './rel-pair-routing.js';
|
||||
import {
|
||||
VALID_NODE_TABLES,
|
||||
assertDeclaredPair,
|
||||
relPairKeyFor,
|
||||
splitRelPairKey,
|
||||
} from './rel-pair-routing.js';
|
||||
import { DEFAULT_EMIT_CHUNK_ROWS, SyncCsvWriter } from './sync-csv-writer.js';
|
||||
import { NODE_TABLES, type NodeTableName } from './schema.js';
|
||||
import { type NodeTableName } from './schema.js';
|
||||
|
||||
/**
|
||||
* PDG edge types streamed per-file (all intra-block BasicBlock→BasicBlock).
|
||||
|
|
@ -98,7 +103,6 @@ export interface PdgEmitManifest {
|
|||
* `--pdg` emit, then {@link finalize} once after the last language.
|
||||
*/
|
||||
export class PdgEmitSink implements KnowledgeGraph {
|
||||
private readonly validTables: Set<string>;
|
||||
private bbWriter: SyncCsvWriter | undefined;
|
||||
/** pairKey (`From|To`) → writer. PDG edges are all `BasicBlock|BasicBlock`,
|
||||
* but the map keeps the sink general and the manifest pair-keyed. */
|
||||
|
|
@ -129,7 +133,6 @@ export class PdgEmitSink implements KnowledgeGraph {
|
|||
private readonly pdgCsvDir: string,
|
||||
private readonly chunkRows: number = DEFAULT_PDG_EMIT_CHUNK_ROWS,
|
||||
) {
|
||||
this.validTables = new Set<string>(NODE_TABLES as readonly string[]);
|
||||
// Clear any streamed CSVs left by a previous (possibly crashed) run so a
|
||||
// later COPY never picks up stale rows.
|
||||
fs.rmSync(pdgCsvDir, { recursive: true, force: true });
|
||||
|
|
@ -160,15 +163,27 @@ export class PdgEmitSink implements KnowledgeGraph {
|
|||
|
||||
addRelationship(relationship: GraphRelationship): void {
|
||||
if (PDG_EDGE_TYPES.has(relationship.type)) {
|
||||
const fromLabel = getNodeLabel(relationship.sourceId);
|
||||
const toLabel = getNodeLabel(relationship.targetId);
|
||||
// Skip edges whose endpoint labels are not valid node tables — mirrors
|
||||
// `RelPairRouter` exactly so the streamed set matches the whole-graph set.
|
||||
if (!this.validTables.has(fromLabel) || !this.validTables.has(toLabel)) return;
|
||||
const pairKey = `${fromLabel}|${toLabel}`;
|
||||
assertDeclaredPair(pairKey, DECLARED_RELATION_PAIRS);
|
||||
// Classify + skip via the SHARED `relPairKeyFor`, not a local copy of its
|
||||
// three lines, so the streamed set cannot drift from the whole-graph set
|
||||
// `RelPairRouter` produces. `undefined` = an endpoint label is not a node
|
||||
// table, so the edge is dropped exactly as the router drops it.
|
||||
const pairKey = relPairKeyFor(
|
||||
relationship.sourceId,
|
||||
relationship.targetId,
|
||||
VALID_NODE_TABLES,
|
||||
);
|
||||
if (pairKey === undefined) return;
|
||||
assertDeclaredPair(
|
||||
pairKey,
|
||||
DECLARED_RELATION_PAIRS,
|
||||
relationship.type,
|
||||
relationship.sourceId,
|
||||
relationship.targetId,
|
||||
);
|
||||
let writer = this.relWriters.get(pairKey);
|
||||
if (writer === undefined) {
|
||||
// Cold: once per pair, so decoding the key back into labels is free.
|
||||
const [fromLabel, toLabel] = splitRelPairKey(pairKey);
|
||||
try {
|
||||
writer = new SyncCsvWriter(
|
||||
path.join(this.pdgCsvDir, `rel_${fromLabel}_${toLabel}.csv`),
|
||||
|
|
|
|||
|
|
@ -31,10 +31,29 @@ import path from 'path';
|
|||
import { createWriteStream, type WriteStream } from 'fs';
|
||||
import { once } from 'events';
|
||||
import { finished } from 'stream/promises';
|
||||
import { NODE_TABLES } from 'gitnexus-shared';
|
||||
import { findInCauseChain } from '../../lib/utils.js';
|
||||
|
||||
/** Injectable for tests (backpressure/error simulation), mirroring split. */
|
||||
export type WriteStreamFactory = (filePath: string) => WriteStream;
|
||||
|
||||
/**
|
||||
* Every label LadybugDB has a node table for — the filter that decides whether
|
||||
* an edge is routable at all.
|
||||
*
|
||||
* ONE shared instance, deliberately. `RelPairRouter`, `GraphEmitSink` and
|
||||
* `PdgEmitSink` each used to build their own `new Set(NODE_TABLES)`; three
|
||||
* copies of the same immutable set are three chances to seed one of them from
|
||||
* a different source. Typed `ReadonlySet` because that — not `Object.freeze`,
|
||||
* which does not touch a Set's internal slots — is what actually stops a
|
||||
* consumer mutating the shared instance.
|
||||
*
|
||||
* Imported straight from `gitnexus-shared` rather than `./schema.js`: schema.ts
|
||||
* imports `parseRelationSchemaPairs` from this module, so the reverse import
|
||||
* would close a cycle.
|
||||
*/
|
||||
export const VALID_NODE_TABLES: ReadonlySet<string> = new Set<string>(NODE_TABLES);
|
||||
|
||||
/**
|
||||
* Derive a node's table label from its graph id. Matches the legacy
|
||||
* `getNodeLabel` that lived inline in `loadGraphToLbug`:
|
||||
|
|
@ -48,20 +67,92 @@ export const getNodeLabel = (nodeId: string): string => {
|
|||
return nodeId.split(':')[0];
|
||||
};
|
||||
|
||||
/**
|
||||
* Classify one edge into its `From|To` pair key, or `undefined` when the edge
|
||||
* must be SKIPPED because an endpoint's label is not a real node table.
|
||||
*
|
||||
* THE single definition of "which pair does this edge belong to, and is it
|
||||
* routable at all". `RelPairRouter.route`, `GraphEmitSink.addRelationship`,
|
||||
* `PdgEmitSink.addRelationship` and the `structural-pair-coverage` corpus guard
|
||||
* each used to inline the same three lines (label both ends → drop if either
|
||||
* label is not a node table → join with `|`). The corpus guard's docblock said
|
||||
* it "mirrors `RelPairRouter.route`" — a mirror is a drift marker: change the
|
||||
* skip rule here and the guard would keep classifying by the old one, report
|
||||
* green, and let `analyze` abort on a pair it had already declared covered.
|
||||
*
|
||||
* HOT PATH — called once per edge (~1M on a large repo). Returns the key
|
||||
* string (which every caller needs anyway for its own Map lookup) rather than
|
||||
* a `{ pairKey, fromLabel, toLabel }` object or a tuple, so the success path
|
||||
* allocates nothing beyond what `getNodeLabel` already did. Callers that need
|
||||
* the two labels back — only when opening a new pair's CSV, once per pair —
|
||||
* decode the key with {@link splitRelPairKey}.
|
||||
*/
|
||||
export const relPairKeyFor = (
|
||||
fromId: string,
|
||||
toId: string,
|
||||
validTables: ReadonlySet<string>,
|
||||
): string | undefined => {
|
||||
const fromLabel = getNodeLabel(fromId);
|
||||
const toLabel = getNodeLabel(toId);
|
||||
if (!validTables.has(fromLabel) || !validTables.has(toLabel)) return undefined;
|
||||
return `${fromLabel}|${toLabel}`;
|
||||
};
|
||||
|
||||
/**
|
||||
* Decode a `From|To` pair key back into its two labels.
|
||||
*
|
||||
* Safe because `|` cannot occur inside a node label: every label is a
|
||||
* `NODE_TABLES` identifier (`[A-Za-z][A-Za-z0-9_]*`), so the FIRST `|` is
|
||||
* always the separator. That invariant was documented in one comment and
|
||||
* enforced nowhere while every consumer re-derived it with a bare
|
||||
* `key.split('|')`.
|
||||
*
|
||||
* DECODE ONLY — there is deliberately no matching `encode` helper. The key is
|
||||
* built once per edge inside {@link relPairKeyFor} (~1M edges on a large
|
||||
* repo), where a function call is a real regression risk; every decode site is
|
||||
* cold by construction (once per pair when its CSV is opened, or on the
|
||||
* throw path of {@link assertDeclaredPair}).
|
||||
*/
|
||||
export const splitRelPairKey = (key: string): readonly [from: string, to: string] => {
|
||||
const sep = key.indexOf('|');
|
||||
return sep < 0 ? [key, ''] : [key.slice(0, sep), key.slice(sep + 1)];
|
||||
};
|
||||
|
||||
/**
|
||||
* Build a fresh matcher for the `FROM <label> TO <label>` clauses of a
|
||||
* relationship DDL. Capture group 1 is the FROM label, group 2 the TO label;
|
||||
* backticks quote schema labels and are not part of the graph label.
|
||||
*
|
||||
* THE SINGLE SOURCE OF TRUTH for that pattern. `parseRelationSchemaPairs`
|
||||
* below builds its pair set from it, and `test/unit/schema-pair-coverage.test.ts`
|
||||
* counts raw `FROM…TO` occurrences with it to catch a pair DUPLICATED in the
|
||||
* DDL (a duplicate makes LadybugDB reject `CREATE REL TABLE`, killing every
|
||||
* `analyze` — strictly worse than one missing pair). That guard used to inline
|
||||
* its own copy of the regex: the two matched identically, so it worked, but any
|
||||
* widening here (dotted identifiers, `IF NOT EXISTS`, a multi-target
|
||||
* `FROM x TO y, z` form) would have silently degraded it to the tautology
|
||||
* `declared.size === declared.size`. Consume this factory instead of
|
||||
* re-inlining a copy.
|
||||
*
|
||||
* A FACTORY, not a shared `RegExp`: a module-level `/g` regex carries
|
||||
* `lastIndex` between calls, so one consumer's `exec`/`test` would corrupt
|
||||
* everyone else's next match. Each call returns a private instance.
|
||||
*/
|
||||
export const createRelationPairMatcher = (): RegExp =>
|
||||
/\bFROM\s+`?([A-Za-z][A-Za-z0-9_]*)`?\s+TO\s+`?([A-Za-z][A-Za-z0-9_]*)`?/g;
|
||||
|
||||
/**
|
||||
* Extract the FROM→TO pairs accepted by a relationship DDL.
|
||||
*
|
||||
* This belongs at the routing boundary: schema.ts owns the DDL, while the CSV
|
||||
* router owns the fail-fast check that prevents writing a pair LadybugDB cannot
|
||||
* COPY. Backticks quote schema labels and are not part of the graph label.
|
||||
* COPY.
|
||||
*/
|
||||
export const parseRelationSchemaPairs = (relationSchema: string): ReadonlySet<string> =>
|
||||
new Set(
|
||||
[
|
||||
...relationSchema.matchAll(
|
||||
/\bFROM\s+`?([A-Za-z][A-Za-z0-9_]*)`?\s+TO\s+`?([A-Za-z][A-Za-z0-9_]*)`?/g,
|
||||
),
|
||||
].map((match) => `${match[1]}|${match[2]}`),
|
||||
[...relationSchema.matchAll(createRelationPairMatcher())].map(
|
||||
(match) => `${match[1]}|${match[2]}`,
|
||||
),
|
||||
);
|
||||
|
||||
export interface RelPairMeta {
|
||||
|
|
@ -69,6 +160,108 @@ export interface RelPairMeta {
|
|||
rows: number;
|
||||
}
|
||||
|
||||
/**
|
||||
* Best-effort source file for a graph node id.
|
||||
*
|
||||
* Node ids are `<Label>:<repo-relative path>:…` (`comm_*` / `proc_*` synthetic
|
||||
* ids carry no file). Repo-relative paths are POSIX-normalized and never carry
|
||||
* a drive letter, so the segment after the label is the file path. Returns
|
||||
* `undefined` rather than guessing when the id has no path segment — a wrong
|
||||
* file in a crash message is worse than none.
|
||||
*/
|
||||
const deriveNodeFilePath = (nodeId: string): string | undefined => {
|
||||
if (nodeId.startsWith('comm_') || nodeId.startsWith('proc_')) return undefined;
|
||||
const filePath = nodeId.split(':')[1];
|
||||
return filePath === undefined || filePath === '' ? undefined : filePath;
|
||||
};
|
||||
|
||||
/** Where a user files the missing pair. Part of the message — see below. */
|
||||
const UNDECLARED_PAIR_ISSUE_URL = 'https://github.com/abhigyanpatwari/GitNexus/issues/new';
|
||||
|
||||
/**
|
||||
* An edge whose endpoint-label pair is absent from the relationship DDL.
|
||||
*
|
||||
* Carries the context the emit call site already has — relationship type, both
|
||||
* node ids, and the source file derived from them — so a user whose `analyze`
|
||||
* just died mid-run can see WHICH of their files produced the edge and file a
|
||||
* bug report that names the missing pair. The abstract label pair alone is
|
||||
* unactionable outside GitNexus's own source (#2789).
|
||||
*
|
||||
* Classify by TYPE (`err instanceof UndeclaredRelationPairError`, or
|
||||
* {@link findUndeclaredRelationPairError} when the error may be wrapped in a
|
||||
* phase `cause` chain) — the repo norm from #2385 — never by message text.
|
||||
*
|
||||
* THE MESSAGE IS THE ONLY RENDERING. It carries the five context fields AND
|
||||
* the two actionable next steps (report the pair; `.gitnexusignore` the file to
|
||||
* finish the rest of the index), because `gitnexus serve` forwards nothing but
|
||||
* `err.message` over worker IPC — anything a consumer re-renders from the
|
||||
* structured fields instead is invisible to a serve-hosted user. The CLI
|
||||
* branch in `cli/analyze.ts` therefore prints this message indented and adds
|
||||
* only the machine-readable `cliError` fields, the same idiom `LbugWipeError`
|
||||
* uses there. It used to re-render the five fields with its own wording; the
|
||||
* two copies had already drifted on the pair separator, the no-file text and
|
||||
* the closing sentence within a single PR, and each had its own pinning test.
|
||||
*/
|
||||
export class UndeclaredRelationPairError extends Error {
|
||||
/** `From|To` label pair, exactly as keyed against the declared-pair set. */
|
||||
readonly pairKey: string;
|
||||
/** Relationship type of the edge that could not be routed (e.g. `CALLS`). */
|
||||
readonly relationType: string;
|
||||
readonly fromId: string;
|
||||
readonly toId: string;
|
||||
/** Source file derived from the node ids; `undefined` for synthetic ids. */
|
||||
readonly sourceFile: string | undefined;
|
||||
|
||||
constructor(pairKey: string, relationType: string, fromId: string, toId: string) {
|
||||
const sourceFile = deriveNodeFilePath(fromId) ?? deriveNodeFilePath(toId);
|
||||
const [fromLabel, toLabel] = splitRelPairKey(pairKey);
|
||||
super(
|
||||
`GitNexus extracted a relationship its own database schema cannot store.\n` +
|
||||
`Relationship label pair ${fromLabel} → ${toLabel} is not declared in the ` +
|
||||
`LadybugDB relation schema.\n` +
|
||||
` relationship type: ${relationType}\n` +
|
||||
` from node: ${fromId}\n` +
|
||||
` to node: ${toId}\n` +
|
||||
` source file: ${sourceFile ?? '(none — synthetic node id)'}\n` +
|
||||
`This is a gap in GitNexus's own relation schema, not a problem with the ` +
|
||||
`analyzed code, and re-running the analysis will fail in exactly the same place.\n` +
|
||||
`Suggestions:\n` +
|
||||
` 1. Report the missing pair so it can be declared:\n` +
|
||||
` ${UNDECLARED_PAIR_ISSUE_URL}\n` +
|
||||
` Include the label pair, the relationship type, and the source file above.\n` +
|
||||
` 2. To finish indexing the rest meanwhile, add that file (or its directory)\n` +
|
||||
` to .gitnexusignore and re-run.`,
|
||||
);
|
||||
this.name = 'UndeclaredRelationPairError';
|
||||
this.pairKey = pairKey;
|
||||
this.relationType = relationType;
|
||||
this.fromId = fromId;
|
||||
this.toId = toId;
|
||||
this.sourceFile = sourceFile;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Find an {@link UndeclaredRelationPairError} in `err` or its `cause` chain.
|
||||
*
|
||||
* The guard throws deep inside an ingestion phase, and the phase runner rewraps
|
||||
* every phase failure as `new Error("Phase 'X' failed: …", { cause })` — so a
|
||||
* bare `instanceof` at the CLI boundary would miss it and fall through to the
|
||||
* generic stack dump.
|
||||
*
|
||||
* The traversal and its depth bound come from `lib/utils.ts` rather than being
|
||||
* re-rolled here: this was the fourth hand-written copy in the repo and the
|
||||
* only one that used `depth <= MAX` (six levels) while claiming to mirror
|
||||
* `cli/analyze.ts`'s `depth < 5`.
|
||||
*/
|
||||
export const findUndeclaredRelationPairError = (
|
||||
err: unknown,
|
||||
): UndeclaredRelationPairError | undefined =>
|
||||
findInCauseChain(
|
||||
err,
|
||||
(e): e is UndeclaredRelationPairError => e instanceof UndeclaredRelationPairError,
|
||||
);
|
||||
|
||||
/**
|
||||
* Fail fast on an endpoint-label pair absent from the relationship DDL, the
|
||||
* same guard `RelPairRouter.route` applies to the whole-graph emit. Exported
|
||||
|
|
@ -77,18 +270,27 @@ export interface RelPairMeta {
|
|||
* the bulk insert, and is silently dropped by the per-edge fallback instead
|
||||
* of failing loudly like the non-streaming path does.
|
||||
*
|
||||
* Takes the already-built `From|To` pairKey rather than the two labels — every
|
||||
* caller needs that same key immediately after for its own Map/stream lookup,
|
||||
* and this is on the per-edge hot path, so building it twice would be a
|
||||
* needless allocation per edge. `|` cannot appear inside a label (node labels
|
||||
* are `NODE_TABLES` identifiers), so splitting it back apart for the error
|
||||
* message is safe.
|
||||
* Takes the already-built `From|To` pairKey (from {@link relPairKeyFor})
|
||||
* rather than the two labels — every caller needs that same key immediately
|
||||
* after for its own Map/stream lookup, and this is on the per-edge hot path,
|
||||
* so building it twice would be a needless allocation per edge. The error
|
||||
* splits it back apart with {@link splitRelPairKey}, which only the throw path
|
||||
* reaches. The edge context is passed POSITIONALLY for the same reason: a
|
||||
* `{ relationType, fromId, toId }` context object would allocate on every
|
||||
* edge, including the ~1M that never fail.
|
||||
*
|
||||
* The success path must stay allocation-free: no object literal, no template
|
||||
* string, no closure, no `Error` constructed before the failure branch.
|
||||
*/
|
||||
export const assertDeclaredPair = (pairKey: string, declaredPairs: ReadonlySet<string>): void => {
|
||||
export const assertDeclaredPair = (
|
||||
pairKey: string,
|
||||
declaredPairs: ReadonlySet<string>,
|
||||
relationType: string,
|
||||
fromId: string,
|
||||
toId: string,
|
||||
): void => {
|
||||
if (!declaredPairs.has(pairKey)) {
|
||||
throw new Error(
|
||||
`Relationship label pair ${pairKey.replaceAll('|', '→')} is not declared in the LadybugDB relation schema`,
|
||||
);
|
||||
throw new UndeclaredRelationPairError(pairKey, relationType, fromId, toId);
|
||||
}
|
||||
};
|
||||
|
||||
|
|
@ -110,7 +312,7 @@ export class RelPairRouter {
|
|||
constructor(
|
||||
private readonly csvDir: string,
|
||||
private readonly header: string,
|
||||
private readonly validTables: Set<string>,
|
||||
private readonly validTables: ReadonlySet<string>,
|
||||
private readonly declaredPairs: ReadonlySet<string>,
|
||||
private readonly wsFactory: WriteStreamFactory = (p) => createWriteStream(p, 'utf-8'),
|
||||
) {}
|
||||
|
|
@ -135,23 +337,25 @@ export class RelPairRouter {
|
|||
* Returns `void` on the synchronous hot path; a `Promise<void>` only when a
|
||||
* stream signals backpressure (or a new pair's header does) — the caller
|
||||
* awaits the promise before routing the next edge.
|
||||
*
|
||||
* `relType` is not used for routing — it is carried purely so an undeclared
|
||||
* pair can name the offending relationship in its error (the row is already
|
||||
* CSV-escaped by then, so the type is not recoverable from it).
|
||||
*/
|
||||
route(fromId: string, toId: string, row: string): void | Promise<void> {
|
||||
route(fromId: string, toId: string, row: string, relType: string): void | Promise<void> {
|
||||
if (this.streamError) throw this.streamError;
|
||||
|
||||
const fromLabel = getNodeLabel(fromId);
|
||||
const toLabel = getNodeLabel(toId);
|
||||
if (!this.validTables.has(fromLabel) || !this.validTables.has(toLabel)) {
|
||||
const pairKey = relPairKeyFor(fromId, toId, this.validTables);
|
||||
if (pairKey === undefined) {
|
||||
this.skipped++;
|
||||
return;
|
||||
}
|
||||
|
||||
const pairKey = `${fromLabel}|${toLabel}`;
|
||||
assertDeclaredPair(pairKey, this.declaredPairs);
|
||||
assertDeclaredPair(pairKey, this.declaredPairs, relType, fromId, toId);
|
||||
const ws = this.streams.get(pairKey);
|
||||
if (ws === undefined) {
|
||||
// First edge for this pair: open the stream, write header + row.
|
||||
return this.openAndWrite(pairKey, fromLabel, toLabel, row);
|
||||
return this.openAndWrite(pairKey, row);
|
||||
}
|
||||
|
||||
this.byPair.get(pairKey)!.rows++;
|
||||
|
|
@ -161,12 +365,9 @@ export class RelPairRouter {
|
|||
}
|
||||
}
|
||||
|
||||
private async openAndWrite(
|
||||
pairKey: string,
|
||||
fromLabel: string,
|
||||
toLabel: string,
|
||||
row: string,
|
||||
): Promise<void> {
|
||||
/** Cold: runs once per pair, so decoding the key back is free here. */
|
||||
private async openAndWrite(pairKey: string, row: string): Promise<void> {
|
||||
const [fromLabel, toLabel] = splitRelPairKey(pairKey);
|
||||
const csvPath = path.join(this.csvDir, `rel_${fromLabel}_${toLabel}.csv`);
|
||||
const ws = this.wsFactory(csvPath);
|
||||
ws.on('error', this.markError);
|
||||
|
|
|
|||
|
|
@ -11,6 +11,10 @@
|
|||
|
||||
// Import from shared package (single source of truth) — used in DDL templates below
|
||||
import { NODE_TABLES, REL_TABLE_NAME, REL_TYPES, EMBEDDING_TABLE_NAME } from 'gitnexus-shared';
|
||||
import type { NodeLabel, NodeTableName } from 'gitnexus-shared';
|
||||
import { parseRelationSchemaPairs } from './rel-pair-routing.js';
|
||||
import { LINKABLE_LABELS } from '../ingestion/scope-resolution/graph-bridge/node-lookup.js';
|
||||
import { CALL_TARGET_TYPES } from '../ingestion/model/symbol-table.js';
|
||||
// Re-export so downstream consumers keep the same import path
|
||||
export { NODE_TABLES, REL_TABLE_NAME, REL_TYPES, EMBEDDING_TABLE_NAME };
|
||||
export type { NodeTableName, RelType } from 'gitnexus-shared';
|
||||
|
|
@ -248,92 +252,214 @@ CREATE NODE TABLE BasicBlock (
|
|||
// Single table with 'type' property - connects all node tables
|
||||
// ============================================================================
|
||||
|
||||
export const RELATION_SCHEMA = `
|
||||
CREATE REL TABLE ${REL_TABLE_NAME} (
|
||||
FROM File TO File,
|
||||
FROM File TO Folder,
|
||||
FROM File TO Function,
|
||||
FROM File TO Class,
|
||||
FROM File TO Interface,
|
||||
FROM File TO Method,
|
||||
/**
|
||||
* Labels the scope-resolution graph bridge can put on the SOURCE side of a
|
||||
* CALLS / ACCESSES / USES / EXTENDS edge: everything `buildGraphNodeLookup`
|
||||
* registers (`LINKABLE_LABELS`), plus the `File` node `resolveCallerGraphId`
|
||||
* falls back to for a module-level call site.
|
||||
*
|
||||
* Imported from the ingestion layer, NOT re-listed here: a hand-copied twin is
|
||||
* pure drift risk, since a label added to it (or dropped from the original) is
|
||||
* invisible to every guard. `csv-generator.ts` and `lbug-adapter.ts`, both
|
||||
* siblings in this directory, already import from `../ingestion/`.
|
||||
*
|
||||
* The cost is that importing this module pulls five ingestion modules into the
|
||||
* runtime closure. `gitnexus-web` does not depend on this package at all (only
|
||||
* on `gitnexus-shared`), so nothing here reaches a browser bundle. The MCP
|
||||
* server still pays it, though: `local-backend.ts` no longer imports this
|
||||
* module directly (its two embedding constants come from `gitnexus-shared`),
|
||||
* but pool-adapter -> lbug-adapter -> csv-generator reaches it anyway.
|
||||
*/
|
||||
const SCOPE_BRIDGE_SOURCE_LABELS: readonly NodeLabel[] = ['File', ...LINKABLE_LABELS];
|
||||
|
||||
/**
|
||||
* Labels the bridge can put on the TARGET side: `LINKABLE_LABELS` again (every
|
||||
* `resolveDefGraphId` hit), plus `CALL_TARGET_TYPES` —
|
||||
* `tryEmitEdgeWithExplicitTargetId` bypasses the lookup and emits such a def's
|
||||
* own node id, and C# `Delegate` is in that set without being linkable.
|
||||
*
|
||||
* Both sets are `NodeLabel`-typed rather than `NodeTableName`-typed because
|
||||
* that is what the originals carry, and `NodeLabel` is the wider union — it
|
||||
* admits five labels with no node table (`Project`, `Package`, `Decorator`,
|
||||
* `Import`, `Type`). A label from that gap would emit DDL naming a table that
|
||||
* does not exist, so `test/unit/schema-pair-coverage.test.ts` asserts every
|
||||
* declared endpoint against `NODE_TABLES`. The hand-written sets below take the
|
||||
* narrower `NodeTableName` constraint, where a typo is the actual risk.
|
||||
*/
|
||||
const SCOPE_BRIDGE_TARGET_LABELS: ReadonlySet<NodeLabel> = new Set<NodeLabel>([
|
||||
...LINKABLE_LABELS,
|
||||
...CALL_TARGET_TYPES,
|
||||
]);
|
||||
|
||||
/**
|
||||
* Node tables that are NOT definitions.
|
||||
*
|
||||
* - `Community` / `Process` are analysis overlays synthesized after ingestion;
|
||||
* nothing is ever attached to one, they are only attached TO.
|
||||
* - `Route` / `Tool` are framework overlays. They do source exactly two edges
|
||||
* — `ENTRY_POINT_OF` to a `Process` (`pipeline-phases/processes.ts`) — but
|
||||
* that emitter hard-codes both labels as literals in one file rather than
|
||||
* resolving an anchor through a lookup, so those two pairs stay in
|
||||
* {@link STRUCTURAL_PAIR_DDL}. Admitting them as anchors would mint twelve
|
||||
* further pairs (`Route→Annotation`, `Tool→Record`, …) no emitter can reach.
|
||||
* - `Folder` is a filesystem container (`Folder→Folder` / `Folder→File` only).
|
||||
* - `BasicBlock` is the PDG substrate (`BasicBlock→BasicBlock` only; measured
|
||||
* over 300k PDG edges, no other pair is emitted).
|
||||
*/
|
||||
const NON_DEFINITION_LABELS: readonly NodeTableName[] = [
|
||||
'Community',
|
||||
'Process',
|
||||
'Route',
|
||||
'Tool',
|
||||
'Folder',
|
||||
'BasicBlock',
|
||||
];
|
||||
|
||||
/**
|
||||
* Every label a DEFINITION node can carry — derived from `NODE_TABLES` by
|
||||
* subtraction so a new node table joins this set automatically and only an
|
||||
* explicit entry above can keep it out.
|
||||
*/
|
||||
const DEFINITION_ANCHOR_LABELS: readonly NodeTableName[] = NODE_TABLES.filter(
|
||||
(label) => !NON_DEFINITION_LABELS.includes(label),
|
||||
);
|
||||
|
||||
/**
|
||||
* Labels whose nodes are minted OUTSIDE the scope-resolution bridge, by a
|
||||
* phase or framework emitter, and then hung off whichever definition node that
|
||||
* emitter happened to resolve. For most of them the anchor is a LOOKUP RESULT,
|
||||
* so its label is not constrained by the emitter — which is exactly why
|
||||
* hand-listing these pairs has crashed `analyze` four separate times:
|
||||
*
|
||||
* | target | emitter | anchor comes from |
|
||||
* |--------------|------------------------------------------------|---------------------------------------|
|
||||
* | `Annotation` | `frameworks/spring/conditionals.ts` CONDITIONAL_ON | `resolveDefGraphId` / `resolveCallerGraphId` |
|
||||
* | `Community` | `pipeline-phases/communities.ts` MEMBER_OF | Leiden membership, `isCommunitySymbol`-gated |
|
||||
* | `Process` | `pipeline-phases/processes.ts` STEP_IN_PROCESS | trace step node |
|
||||
* | `Route` | `pipeline-phases/routes.ts` HANDLES_ROUTE | `generateId('File', handlerPath)` — a literal |
|
||||
* | `Tool` | `pipeline-phases/tools.ts` HANDLES_TOOL | `handlerNodeId` — whatever definition the decorator sat on |
|
||||
* | `File` | `languages/vue/scope-resolver.ts` BINDS_EVENT_HANDLER | handler node |
|
||||
* | `Record` | `cobol-processor.ts` × 8 external-resource sites | `scopedCallerLookup` |
|
||||
*
|
||||
* The four reproduced hard-aborts are one cell of this table each:
|
||||
* `Method→Annotation` (Spring `@Bean` + `@ConditionalOnMissingBean`),
|
||||
* `Method→File` (Vue Options-API handler), `Namespace→Record` (COBOL
|
||||
* `DECLARATIVES`), `Class→Tool` (`@mcp.tool()` on a class). Declaring
|
||||
* {@link DEFINITION_ANCHOR_LABELS} × this set covers all four plus every
|
||||
* sibling the same emitters can reach.
|
||||
*
|
||||
* TWO TARGETS ARE LABEL-GATED TODAY, and the cross product over-declares for
|
||||
* them ON PURPOSE (~47 of the 182 attachment pairs are unreachable right now):
|
||||
* - `Community` — `isCommunitySymbol` (`community-processor.ts`) admits only
|
||||
* `Function` / `Class` / `Method` / `Interface` as members, so the other 22
|
||||
* anchors cannot source a MEMBER_OF edge until that predicate widens.
|
||||
* - `Route` — HANDLES_ROUTE sources `generateId('File', handlerPath)`, a
|
||||
* literal `File`, so every non-`File` anchor is headroom.
|
||||
*
|
||||
* Those pairs stay declared because the two sides of the error are not
|
||||
* symmetric: an UNDECLARED pair makes LadybugDB reject the edge and aborts
|
||||
* `analyze` outright on a user's repo, while an unused DECLARED pair costs
|
||||
* almost nothing — `bench/schema-pairs` measures the whole 332→450 growth
|
||||
* (118 pairs, of which these ~47 are a part) at 0.93–1.05×, i.e. inside
|
||||
* run-to-run noise.
|
||||
* Every one of the four aborts above came from re-narrowing a set to what one
|
||||
* predicate looked like it allowed — so a reading of `isCommunitySymbol` is not
|
||||
* grounds to shrink this. Widening either predicate is then a no-op here.
|
||||
*
|
||||
* `Route` / `Tool` being excluded as ANCHORS (see {@link NON_DEFINITION_LABELS})
|
||||
* is likewise a SIZE choice, not something derived from a rule: they do source
|
||||
* `ENTRY_POINT_OF`, and admitting them would mint twelve further pairs no
|
||||
* emitter can currently reach.
|
||||
*
|
||||
* Sized deliberately: this rule brings the DDL to 450 pairs. `bench/schema-pairs`
|
||||
* measures it against real `@ladybugdb/core` with identical data — untyped-endpoint
|
||||
* anchored queries (`MATCH (a {id: $id})-[r:CodeRelation]->(b)`, the shape
|
||||
* `impact` / `context` / `detect_changes` issue), relative to the 332-pair
|
||||
* hand-list this replaced. Four runs on the same box:
|
||||
*
|
||||
* 450 → 0.93–1.05× 641 → 1.22–1.43× 786 → 1.52–1.75× 1024 → 2.03–2.34×
|
||||
*
|
||||
* 450 is inside run-to-run noise (it came out FASTER than 332 on three of the
|
||||
* four runs); everything past ~640 is not. The knee sits just above 450, so the containment half below
|
||||
* stays hand-declared rather than being folded into a third cross product.
|
||||
* Re-run that bench and quote the range — not one run — before proposing one.
|
||||
*/
|
||||
const ATTACHMENT_TARGET_LABELS: readonly NodeTableName[] = [
|
||||
'Annotation',
|
||||
'Community',
|
||||
'Process',
|
||||
'Route',
|
||||
'Tool',
|
||||
'File',
|
||||
'Record',
|
||||
];
|
||||
|
||||
/**
|
||||
* The 72 pairs NEITHER rule above generates — everything left after the two
|
||||
* cross products are subtracted. Carried by CONTAINMENT, inheritance, imports
|
||||
* and DI: a container label crossed with a contained label. No predicate
|
||||
* describes that surface (any container can hold any definition).
|
||||
*
|
||||
* What survives here is characteristic, not arbitrary. Almost all of it is a
|
||||
* TARGET no rule reaches — `CodeElement`, `Impl`, `Namespace`, `Template`,
|
||||
* `TypeAlias`, `Typedef`, `Union`, `Static`, `Section`, `Folder` are in neither
|
||||
* `SCOPE_BRIDGE_TARGET_LABELS` nor {@link ATTACHMENT_TARGET_LABELS} — plus the
|
||||
* `Impl|*` and `Template|*` member rows (Rust `impl`/`trait` bodies, C++
|
||||
* templates), the two `Route|Process` / `Tool|Process` entry points whose
|
||||
* emitter names both labels as literals, and `BasicBlock|BasicBlock`, the PDG
|
||||
* substrate.
|
||||
*
|
||||
* NOTHING A RULE ALREADY COVERS BELONGS HERE. `generatedRelationPairs` skips
|
||||
* any pair present in this block, so a redundant line does not merely duplicate
|
||||
* — it SUPPRESSES generation, and later narrowing a rule would silently keep
|
||||
* that pair alive with no test failing. 161 such lines were deleted from this
|
||||
* block (the DDL's pair set is unchanged: they moved into the generated half);
|
||||
* `test/unit/schema-pair-coverage.test.ts` now fails if one comes back.
|
||||
*
|
||||
* Folding this remainder into a third cross product
|
||||
* (`DEFINITION_ANCHOR_LABELS × {CodeElement, Section, Typedef, Union,
|
||||
* Namespace, Impl, TypeAlias, Static, Template}`) would take the table to 641
|
||||
* pairs and leave only ~29 lines here. `bench/schema-pairs` measures 641 at
|
||||
* 1.22–1.43× on anchored queries, where production's 450 is inside noise — so
|
||||
* that trade buys ~43 fewer hand-written lines for a real ~22–43% on the query
|
||||
* shape `impact` uses, which is why it is deferred rather than taken.
|
||||
*
|
||||
* Exported so `test/unit/schema-pair-coverage.test.ts` can subtract it and
|
||||
* assert the GENERATED region of the DDL for exact equality against the two
|
||||
* rules, rather than one-directional containment.
|
||||
* `test/integration/structural-pair-coverage.test.ts` guards this half from a
|
||||
* corpus — that is the guard, not this comment.
|
||||
*/
|
||||
export const STRUCTURAL_PAIR_DDL = ` FROM File TO Folder,
|
||||
FROM File TO CodeElement,
|
||||
FROM File TO \`Struct\`,
|
||||
FROM File TO \`Enum\`,
|
||||
FROM File TO \`Macro\`,
|
||||
FROM File TO \`Typedef\`,
|
||||
FROM File TO \`Union\`,
|
||||
FROM File TO \`Namespace\`,
|
||||
FROM File TO \`Trait\`,
|
||||
FROM File TO \`Impl\`,
|
||||
FROM File TO \`TypeAlias\`,
|
||||
FROM File TO \`Const\`,
|
||||
FROM File TO \`Static\`,
|
||||
FROM File TO \`Variable\`,
|
||||
FROM File TO \`Property\`,
|
||||
FROM File TO \`Record\`,
|
||||
FROM File TO \`Delegate\`,
|
||||
FROM File TO \`Annotation\`,
|
||||
FROM File TO \`Constructor\`,
|
||||
FROM File TO \`Template\`,
|
||||
FROM File TO \`Module\`,
|
||||
FROM File TO Section,
|
||||
FROM Folder TO Folder,
|
||||
FROM Folder TO File,
|
||||
FROM Function TO Function,
|
||||
FROM Function TO Method,
|
||||
FROM Function TO Class,
|
||||
FROM Function TO Community,
|
||||
FROM Function TO \`Macro\`,
|
||||
FROM Function TO \`Struct\`,
|
||||
FROM Function TO \`Template\`,
|
||||
FROM Function TO \`Enum\`,
|
||||
FROM Function TO \`Namespace\`,
|
||||
FROM Function TO \`TypeAlias\`,
|
||||
FROM Function TO \`Module\`,
|
||||
FROM Function TO \`Impl\`,
|
||||
FROM Function TO Interface,
|
||||
FROM Function TO \`Constructor\`,
|
||||
FROM Function TO \`Const\`,
|
||||
FROM Function TO \`Typedef\`,
|
||||
FROM Function TO \`Union\`,
|
||||
FROM Function TO \`Property\`,
|
||||
FROM Function TO CodeElement,
|
||||
FROM Class TO Method,
|
||||
FROM Class TO Function,
|
||||
FROM Class TO Class,
|
||||
FROM Class TO Interface,
|
||||
FROM Class TO Community,
|
||||
FROM Class TO \`Template\`,
|
||||
FROM Class TO \`TypeAlias\`,
|
||||
FROM Class TO \`Struct\`,
|
||||
FROM Class TO \`Enum\`,
|
||||
FROM Class TO \`Annotation\`,
|
||||
FROM Class TO \`Constructor\`,
|
||||
FROM Class TO \`Trait\`,
|
||||
FROM Class TO \`Macro\`,
|
||||
FROM Class TO \`Impl\`,
|
||||
FROM Class TO \`Union\`,
|
||||
FROM Class TO \`Namespace\`,
|
||||
FROM Class TO \`Typedef\`,
|
||||
FROM Class TO \`Property\`,
|
||||
FROM Class TO CodeElement,
|
||||
FROM Method TO Function,
|
||||
FROM Method TO Method,
|
||||
FROM Method TO Class,
|
||||
FROM Method TO Community,
|
||||
FROM Method TO \`Template\`,
|
||||
FROM Method TO \`Struct\`,
|
||||
FROM Method TO \`TypeAlias\`,
|
||||
FROM Method TO \`Enum\`,
|
||||
FROM Method TO \`Macro\`,
|
||||
FROM Method TO \`Namespace\`,
|
||||
FROM Method TO \`Module\`,
|
||||
FROM Method TO \`Impl\`,
|
||||
FROM Method TO Interface,
|
||||
FROM Method TO \`Constructor\`,
|
||||
FROM Method TO \`Property\`,
|
||||
FROM Method TO \`Variable\`,
|
||||
FROM Method TO \`Const\`,
|
||||
FROM Method TO CodeElement,
|
||||
FROM \`Template\` TO \`Template\`,
|
||||
FROM \`Template\` TO Function,
|
||||
|
|
@ -345,134 +471,90 @@ CREATE REL TABLE ${REL_TABLE_NAME} (
|
|||
FROM \`Template\` TO \`Macro\`,
|
||||
FROM \`Template\` TO Interface,
|
||||
FROM \`Template\` TO \`Constructor\`,
|
||||
FROM \`Module\` TO \`Module\`,
|
||||
FROM \`Module\` TO CodeElement,
|
||||
FROM \`Module\` TO \`Namespace\`,
|
||||
FROM \`Namespace\` TO Function,
|
||||
FROM CodeElement TO CodeElement,
|
||||
FROM CodeElement TO \`Module\`,
|
||||
FROM CodeElement TO \`Property\`,
|
||||
FROM Section TO Section,
|
||||
FROM Section TO File,
|
||||
FROM File TO Route,
|
||||
FROM Function TO Route,
|
||||
FROM Method TO Route,
|
||||
FROM File TO Tool,
|
||||
FROM Function TO Tool,
|
||||
FROM Method TO Tool,
|
||||
FROM CodeElement TO Community,
|
||||
FROM Interface TO Community,
|
||||
FROM Interface TO Function,
|
||||
FROM Interface TO Method,
|
||||
FROM Interface TO Class,
|
||||
FROM Interface TO Interface,
|
||||
FROM Interface TO CodeElement,
|
||||
FROM Interface TO \`TypeAlias\`,
|
||||
FROM Interface TO \`Struct\`,
|
||||
FROM Interface TO \`Constructor\`,
|
||||
FROM Interface TO \`Property\`,
|
||||
FROM \`Struct\` TO Community,
|
||||
FROM \`Struct\` TO \`Trait\`,
|
||||
FROM \`Struct\` TO \`Struct\`,
|
||||
FROM \`Struct\` TO Class,
|
||||
FROM \`Struct\` TO \`Enum\`,
|
||||
FROM \`Struct\` TO Function,
|
||||
FROM \`Struct\` TO Method,
|
||||
FROM \`Struct\` TO Interface,
|
||||
FROM \`Struct\` TO \`Constructor\`,
|
||||
FROM \`Struct\` TO \`Property\`,
|
||||
FROM \`Enum\` TO \`Enum\`,
|
||||
FROM \`Enum\` TO Community,
|
||||
FROM \`Enum\` TO Class,
|
||||
FROM \`Enum\` TO Interface,
|
||||
FROM \`Enum\` TO Function,
|
||||
FROM \`Enum\` TO Method,
|
||||
FROM \`Enum\` TO \`Struct\`,
|
||||
FROM \`Enum\` TO \`Constructor\`,
|
||||
FROM \`Enum\` TO \`Property\`,
|
||||
FROM \`Enum\` TO \`TypeAlias\`,
|
||||
FROM \`Macro\` TO Community,
|
||||
FROM \`Macro\` TO Function,
|
||||
FROM \`Macro\` TO Method,
|
||||
FROM \`Module\` TO Function,
|
||||
FROM \`Module\` TO Method,
|
||||
FROM \`Typedef\` TO Community,
|
||||
FROM \`Union\` TO Community,
|
||||
FROM \`Namespace\` TO Community,
|
||||
FROM \`Namespace\` TO \`Struct\`,
|
||||
FROM \`Trait\` TO Method,
|
||||
FROM \`Trait\` TO Function,
|
||||
FROM \`Trait\` TO \`Constructor\`,
|
||||
FROM \`Trait\` TO \`Property\`,
|
||||
FROM \`Trait\` TO Community,
|
||||
FROM \`Impl\` TO Method,
|
||||
FROM \`Impl\` TO Function,
|
||||
FROM \`Impl\` TO \`Constructor\`,
|
||||
FROM \`Impl\` TO \`Property\`,
|
||||
FROM \`Impl\` TO Community,
|
||||
FROM \`Impl\` TO \`Trait\`,
|
||||
FROM \`Impl\` TO \`Struct\`,
|
||||
FROM \`Impl\` TO \`Impl\`,
|
||||
FROM \`TypeAlias\` TO Community,
|
||||
FROM \`TypeAlias\` TO \`Trait\`,
|
||||
FROM \`TypeAlias\` TO Class,
|
||||
FROM \`Const\` TO Community,
|
||||
FROM \`Const\` TO Method,
|
||||
FROM \`Static\` TO Community,
|
||||
FROM \`Variable\` TO Community,
|
||||
FROM \`Variable\` TO Method,
|
||||
FROM \`Property\` TO Community,
|
||||
FROM \`Property\` TO \`Property\`,
|
||||
FROM \`Property\` TO Class,
|
||||
FROM \`Property\` TO \`Enum\`,
|
||||
FROM \`Property\` TO Function,
|
||||
FROM \`Property\` TO \`Struct\`,
|
||||
FROM \`Record\` TO Method,
|
||||
FROM \`Record\` TO \`Constructor\`,
|
||||
FROM \`Record\` TO \`Property\`,
|
||||
FROM \`Record\` TO Community,
|
||||
FROM \`Delegate\` TO Community,
|
||||
FROM \`Annotation\` TO Community,
|
||||
FROM \`Constructor\` TO Community,
|
||||
FROM \`Constructor\` TO Interface,
|
||||
FROM \`Constructor\` TO Class,
|
||||
FROM \`Constructor\` TO Method,
|
||||
FROM \`Constructor\` TO Function,
|
||||
FROM \`Constructor\` TO \`Constructor\`,
|
||||
FROM \`Constructor\` TO \`Struct\`,
|
||||
FROM \`Constructor\` TO \`Macro\`,
|
||||
FROM \`Constructor\` TO \`Template\`,
|
||||
FROM \`Constructor\` TO \`TypeAlias\`,
|
||||
FROM \`Constructor\` TO \`Enum\`,
|
||||
FROM \`Constructor\` TO \`Annotation\`,
|
||||
FROM \`Constructor\` TO \`Impl\`,
|
||||
FROM \`Constructor\` TO \`Namespace\`,
|
||||
FROM \`Constructor\` TO \`Module\`,
|
||||
FROM \`Constructor\` TO \`Property\`,
|
||||
FROM \`Constructor\` TO \`Typedef\`,
|
||||
FROM \`Template\` TO Community,
|
||||
FROM \`Module\` TO Community,
|
||||
FROM Function TO Process,
|
||||
FROM Method TO Process,
|
||||
FROM Class TO Process,
|
||||
FROM Interface TO Process,
|
||||
FROM \`Struct\` TO Process,
|
||||
FROM \`Constructor\` TO Process,
|
||||
FROM \`Module\` TO Process,
|
||||
FROM \`Macro\` TO Process,
|
||||
FROM \`Impl\` TO Process,
|
||||
FROM \`Typedef\` TO Process,
|
||||
FROM \`TypeAlias\` TO Process,
|
||||
FROM \`Enum\` TO Process,
|
||||
FROM \`Union\` TO Process,
|
||||
FROM \`Namespace\` TO Process,
|
||||
FROM \`Trait\` TO Process,
|
||||
FROM \`Const\` TO Process,
|
||||
FROM \`Static\` TO Process,
|
||||
FROM \`Variable\` TO Process,
|
||||
FROM \`Property\` TO Process,
|
||||
FROM \`Record\` TO Process,
|
||||
FROM \`Delegate\` TO Process,
|
||||
FROM \`Annotation\` TO Process,
|
||||
FROM \`Template\` TO Process,
|
||||
FROM CodeElement TO Process,
|
||||
FROM Route TO Process,
|
||||
FROM Tool TO Process,
|
||||
FROM BasicBlock TO BasicBlock,
|
||||
FROM BasicBlock TO BasicBlock`;
|
||||
|
||||
/**
|
||||
* The generated half of the DDL — one ` FROM \`x\` TO \`y\`` line per pair of
|
||||
* the two cross products below.
|
||||
*
|
||||
* 1. SCOPE BRIDGE — `SCOPE_BRIDGE_SOURCE_LABELS × SCOPE_BRIDGE_TARGET_LABELS`.
|
||||
* Those sets ARE the bridge's emit surface: `buildGraphNodeLookup` holds
|
||||
* only `LINKABLE_LABELS`, so every id `resolveDefGraphId` returns wears one
|
||||
* of those labels (#2792).
|
||||
* 2. ATTACHMENT — `DEFINITION_ANCHOR_LABELS × ATTACHMENT_TARGET_LABELS`, the
|
||||
* phase/framework overlays hung off a resolved anchor (#2793).
|
||||
*
|
||||
* Generated rather than hand-listed because in both families the endpoint
|
||||
* labels are LOOKUP RESULTS, not literals at the emit site — so any pair drawn
|
||||
* from the sets can reach `assertDeclaredPair`, and an undeclared one aborts
|
||||
* `analyze` outright on whichever codebase happens to produce it. Every
|
||||
* hand-listed fix so far declared only the pair in the stack trace and left the
|
||||
* rest of its family missing: `Const→Method` (#2781), `Class→Variable` (#2792),
|
||||
* `Interface→CodeElement` (#2416), then `Method→Annotation` / `Method→File` /
|
||||
* `Namespace→Record` / `Class→Tool` (#2793) — four more from three different
|
||||
* emitters, all live at once.
|
||||
*
|
||||
* A `Set` guards the emit against a DUPLICATED pair — the asymmetric failure
|
||||
* mode `rel-pair-routing.ts` documents at length: a duplicate makes LadybugDB
|
||||
* reject `CREATE REL TABLE` and kills EVERY `analyze`, where a missing pair only
|
||||
* kills the codebases that emit it. The two target sets are disjoint TODAY, so
|
||||
* nothing is deduped in practice; the Set is here so that moving one label
|
||||
* between the rules can never cause it. Pairs already in
|
||||
* {@link STRUCTURAL_PAIR_DDL} are skipped for the same reason.
|
||||
*/
|
||||
const generatedPairDdl = (): string => {
|
||||
const structural = parseRelationSchemaPairs(STRUCTURAL_PAIR_DDL);
|
||||
const seen = new Set<string>();
|
||||
const lines: string[] = [];
|
||||
const add = (from: NodeLabel, to: NodeLabel): void => {
|
||||
const pairKey = `${from}|${to}`;
|
||||
if (structural.has(pairKey) || seen.has(pairKey)) return;
|
||||
seen.add(pairKey);
|
||||
lines.push(` FROM \`${from}\` TO \`${to}\``);
|
||||
};
|
||||
for (const from of SCOPE_BRIDGE_SOURCE_LABELS) {
|
||||
for (const to of SCOPE_BRIDGE_TARGET_LABELS) add(from, to);
|
||||
}
|
||||
for (const from of DEFINITION_ANCHOR_LABELS) {
|
||||
for (const to of ATTACHMENT_TARGET_LABELS) add(from, to);
|
||||
}
|
||||
return lines.join(',\n');
|
||||
};
|
||||
|
||||
export const RELATION_SCHEMA = `
|
||||
CREATE REL TABLE ${REL_TABLE_NAME} (
|
||||
${STRUCTURAL_PAIR_DDL},
|
||||
${generatedPairDdl()},
|
||||
type STRING,
|
||||
confidence DOUBLE,
|
||||
reason STRING,
|
||||
|
|
|
|||
|
|
@ -893,10 +893,6 @@ async function runFullAnalysisInner(
|
|||
const progress = (phase: string, percent: number, message: string) =>
|
||||
callbacks.onProgress(phase, percent, message);
|
||||
|
||||
// Streamed structural emit (#2680), resolved once so the pipeline flag and the
|
||||
// CSV-dir resolution below cannot disagree.
|
||||
const streamGraphEmitActive = resolveStreamGraphEmit(options);
|
||||
|
||||
// FTS-config validation and the degraded-parse counter reset happen in the
|
||||
// `runFullAnalysis` wrapper (before the lock is taken).
|
||||
|
||||
|
|
@ -1504,6 +1500,22 @@ async function runFullAnalysisInner(
|
|||
// in-place (cache hits leave entries unchanged; misses add new ones).
|
||||
const parseCache = await loadParseCache(storagePath);
|
||||
|
||||
// Streamed structural emit (#2680). Resolved ONCE, so the pipeline flag and
|
||||
// the CSV-dir resolution below cannot disagree — and resolved HERE, not at
|
||||
// function entry, because the POSITION is load-bearing: the gate is
|
||||
// `options.force`, and every freshness guard above REBINDS `options` with
|
||||
// `force: true` (embedding-checkpoint drop, dirty-flag recovery, pdg-mode
|
||||
// flip, schema-version bump, analysis-feature drift, runner-identity change,
|
||||
// CJK-mode change). Resolving before them froze the answer at `false` for
|
||||
// every rebuild they trigger — including the whole-fleet rebuild an
|
||||
// INCREMENTAL_SCHEMA_VERSION bump forces on every existing index at once,
|
||||
// which is exactly when the #2649 memory relief matters most. So this MUST
|
||||
// stay below the last guard that can set `force` and above its first use.
|
||||
// (The post-pipeline analysis-feature re-check can also set `force`, but the
|
||||
// pipeline has already run by then; that run emits non-streamed, precisely as
|
||||
// `resolveStreamPdgEmit` — read fresh at the same point — behaves.)
|
||||
const streamGraphEmitActive = resolveStreamGraphEmit(options);
|
||||
|
||||
// ── Phase 1: Full Pipeline (0–60%) ────────────────────────────────
|
||||
const pipelineResult = await runPipelineFromRepo(
|
||||
repoPath,
|
||||
|
|
|
|||
|
|
@ -2,6 +2,60 @@ export const generateId = (label: string, name: string): string => {
|
|||
return `${label}:${name}`;
|
||||
};
|
||||
|
||||
/**
|
||||
* How many links of an `Error.cause` chain any walker below visits, counting
|
||||
* the head. THE bound — hand-rolled copies had drifted to three different
|
||||
* numbers with three different loop conditions (`depth < 5` in two places,
|
||||
* `depth <= 5` in a third, i.e. six levels).
|
||||
*
|
||||
* The bound exists purely so a cyclic chain (`a.cause = b; b.cause = a`)
|
||||
* cannot loop forever. Real chains are one or two links deep — the ingestion
|
||||
* phase runner wraps a phase failure once as
|
||||
* `new Error("Phase 'X' failed: …", { cause })` — so five leaves ample
|
||||
* headroom for future nesting.
|
||||
*/
|
||||
export const CAUSE_CHAIN_MAX_DEPTH = 5;
|
||||
|
||||
/**
|
||||
* Walk `err` and its `cause` chain, head first, yielding each `Error` link.
|
||||
*
|
||||
* Stops at the first non-`Error` link (a `cause` may legally be any value, and
|
||||
* a non-Error carries no further `cause` worth following) and after
|
||||
* `maxDepth` links. Non-`Error` input yields nothing.
|
||||
*
|
||||
* THE single cause-chain traversal. Consume this (or {@link findInCauseChain})
|
||||
* rather than re-rolling the `for (let depth = 0; …; current = current.cause)`
|
||||
* loop: every hand-rolled copy has to re-decide the bound and the loop
|
||||
* condition, and they did not agree.
|
||||
*/
|
||||
export function* causeChain(err: unknown, maxDepth = CAUSE_CHAIN_MAX_DEPTH): Generator<Error> {
|
||||
let current: unknown = err;
|
||||
for (let depth = 0; depth < maxDepth && current instanceof Error; depth++) {
|
||||
yield current;
|
||||
current = (current as { cause?: unknown }).cause;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Find the first link of `err`'s cause chain that `match` accepts.
|
||||
*
|
||||
* Load-bearing for CLI error classification: the ingestion phase runner
|
||||
* rewraps every phase failure as `new Error("Phase 'X' failed: …", { cause })`,
|
||||
* so a bare `instanceof` at the CLI boundary misses the real error entirely
|
||||
* and falls through to a generic stack dump. Classify by TYPE through this
|
||||
* helper (the repo norm from #2385), never by message text.
|
||||
*/
|
||||
export function findInCauseChain<T>(
|
||||
err: unknown,
|
||||
match: (e: unknown) => e is T,
|
||||
maxDepth: number = CAUSE_CHAIN_MAX_DEPTH,
|
||||
): T | undefined {
|
||||
for (const link of causeChain(err, maxDepth)) {
|
||||
if (match(link)) return link;
|
||||
}
|
||||
return undefined;
|
||||
}
|
||||
|
||||
/**
|
||||
* Drop a Windows extended-length (`\\?\`) prefix from a path (#2667).
|
||||
*
|
||||
|
|
|
|||
|
|
@ -61,6 +61,13 @@ import {
|
|||
rankExactEmbeddingRows,
|
||||
type ExactEmbeddingRow,
|
||||
} from '../../core/embeddings/exact-search.js';
|
||||
// These are two bare string constants, but `schema.ts` builds its DDL from
|
||||
// `LINKABLE_LABELS` / `CALL_TARGET_TYPES` and so pulls five ingestion modules
|
||||
// into the runtime closure (~70ms on a cold MCP start). Re-homing them in
|
||||
// `gitnexus-shared` does NOT fix that on its own: `pool-adapter.js` below
|
||||
// reaches `schema.ts` anyway via pool-adapter -> lbug-adapter -> csv-generator,
|
||||
// all value imports. Cutting `csv-generator` (analyze-only code the MCP server
|
||||
// never runs) out of the adapter chain is the change that would make it real.
|
||||
import { EMBEDDING_TABLE_NAME, EMBEDDING_INDEX_NAME } from '../../core/lbug/schema.js';
|
||||
import { getExactScanLimit } from '../../core/platform/capabilities.js';
|
||||
import { PhaseTimer } from '../../core/search/phase-timer.js';
|
||||
|
|
|
|||
|
|
@ -687,8 +687,28 @@ export interface RepoMeta {
|
|||
* Numbered 34, not 33: `main` took 33 for Spring AOP (#2416) mid-flight, landing
|
||||
* on exactly this branch's number — the seventh collision in this series and the
|
||||
* first exact clash. Re-check against origin/main before merge.
|
||||
*
|
||||
* v35: the relation DDL is GENERATED from two closed-form rules instead of the
|
||||
* pairs someone happened to hit — 223 → 450 declared pairs (#2792, #2793).
|
||||
* Rule 1, the scope-resolution bridge: `LINKABLE_LABELS` + the `File` caller
|
||||
* fallback, crossed with `LINKABLE_LABELS` + `CALL_TARGET_TYPES`. Rule 2, the
|
||||
* phase/framework overlays: every definition label (`NODE_TABLES` minus
|
||||
* Community/Process/Route/Tool/Folder/BasicBlock) crossed with the labels those
|
||||
* emitters mint and hang off a resolved anchor — Annotation, Community,
|
||||
* Process, Route, Tool, File, Record. In both families the endpoint labels are
|
||||
* LOOKUP RESULTS, not literals at the emit site, so hand-listing could only
|
||||
* ever declare the pair in the latest stack trace: v32, v33 and #2781 were each
|
||||
* that same piecemeal fix, and `analyze` kept aborting at `assertDeclaredPair`
|
||||
* on the next codebase with a different edge shape (`Class→Variable` on Java
|
||||
* initializers, then `Method→Annotation` on Spring `@Bean`, `Method→File` on a
|
||||
* Vue Options-API handler, `Namespace→Record` on COBOL `DECLARATIVES`, and
|
||||
* `Class→Tool` on `@mcp.tool()` applied to a class — four at once, from three
|
||||
* different emitters). What remains hand-declared is only the containment /
|
||||
* inheritance / import surface, which no label predicate describes and which a
|
||||
* corpus test guards instead. A pre-v35 database physically lacks all of these
|
||||
* from-to pairs, so force a full re-analyze.
|
||||
*/
|
||||
export const INCREMENTAL_SCHEMA_VERSION = 34;
|
||||
export const INCREMENTAL_SCHEMA_VERSION = 35;
|
||||
|
||||
export interface IndexedRepo {
|
||||
repoPath: string;
|
||||
|
|
|
|||
48
gitnexus/test/fixtures/lang-resolution/cobol-declaratives/ERRDEMO.cbl
vendored
Normal file
48
gitnexus/test/fixtures/lang-resolution/cobol-declaratives/ERRDEMO.cbl
vendored
Normal file
|
|
@ -0,0 +1,48 @@
|
|||
*****************************************************************
|
||||
* DECLARATIVES + USE AFTER STANDARD ERROR.
|
||||
*
|
||||
* `cobol-processor.ts` turns each DECLARATIVES handler into an
|
||||
* ACCESSES edge from the handler SECTION's `Namespace` node to a
|
||||
* synthesized `Record` node for the file the USE clause names --
|
||||
* the `Namespace -> Record` FROM/TO pair. Both endpoints are
|
||||
* structural (neither label is in `LINKABLE_LABELS`), so no
|
||||
* scope-bridge cross product reaches it.
|
||||
*****************************************************************
|
||||
IDENTIFICATION DIVISION.
|
||||
PROGRAM-ID. ERRDEMO.
|
||||
ENVIRONMENT DIVISION.
|
||||
INPUT-OUTPUT SECTION.
|
||||
FILE-CONTROL.
|
||||
SELECT CUSTOMER-FILE ASSIGN TO "CUST.DAT"
|
||||
ORGANIZATION IS SEQUENTIAL.
|
||||
SELECT AUDIT-FILE ASSIGN TO "AUDIT.DAT"
|
||||
ORGANIZATION IS SEQUENTIAL.
|
||||
DATA DIVISION.
|
||||
FILE SECTION.
|
||||
FD CUSTOMER-FILE.
|
||||
01 CUSTOMER-REC.
|
||||
05 CUST-ID PIC X(10).
|
||||
05 CUST-BALANCE PIC 9(7)V99.
|
||||
FD AUDIT-FILE.
|
||||
01 AUDIT-REC.
|
||||
05 AUDIT-TEXT PIC X(60).
|
||||
WORKING-STORAGE SECTION.
|
||||
01 WS-EOF PIC X VALUE "N".
|
||||
PROCEDURE DIVISION.
|
||||
DECLARATIVES.
|
||||
CUSTOMER-ERR-HANDLER SECTION.
|
||||
USE AFTER STANDARD ERROR ON CUSTOMER-FILE.
|
||||
CUSTOMER-ERR-PARA.
|
||||
DISPLAY "CUSTOMER IO ERROR".
|
||||
AUDIT-ERR-HANDLER SECTION.
|
||||
USE AFTER STANDARD ERROR ON AUDIT-FILE.
|
||||
AUDIT-ERR-PARA.
|
||||
DISPLAY "AUDIT IO ERROR".
|
||||
END DECLARATIVES.
|
||||
MAIN-SECTION SECTION.
|
||||
MAIN-PARA.
|
||||
OPEN INPUT CUSTOMER-FILE
|
||||
OPEN OUTPUT AUDIT-FILE
|
||||
CLOSE CUSTOMER-FILE
|
||||
CLOSE AUDIT-FILE
|
||||
STOP RUN.
|
||||
22
gitnexus/test/fixtures/lang-resolution/mcp-tool-class/server.py
vendored
Normal file
22
gitnexus/test/fixtures/lang-resolution/mcp-tool-class/server.py
vendored
Normal file
|
|
@ -0,0 +1,22 @@
|
|||
"""`@mcp.tool()` applied to a CLASS rather than a function.
|
||||
|
||||
`pipeline-phases/tools.ts` hangs the HANDLES_TOOL edge off whatever node the
|
||||
tool decorator sat on (`handlerNodeId`), so a class-decorated tool produces a
|
||||
`Class -> Tool` edge. Every other tool fixture decorates a function or falls
|
||||
back to the file, so `Function -> Tool` / `File -> Tool` are the only pairs
|
||||
they exercise.
|
||||
"""
|
||||
|
||||
from mcp import tool
|
||||
|
||||
|
||||
def _render(payload: dict) -> str:
|
||||
return str(payload)
|
||||
|
||||
|
||||
@mcp.tool()
|
||||
class WeatherTool:
|
||||
"""Class-based MCP tool."""
|
||||
|
||||
def run(self, payload: dict) -> str:
|
||||
return _render(payload)
|
||||
30
gitnexus/test/fixtures/lang-resolution/spring-conditional-app/java/AppAutoConfiguration.java
vendored
Normal file
30
gitnexus/test/fixtures/lang-resolution/spring-conditional-app/java/AppAutoConfiguration.java
vendored
Normal file
|
|
@ -0,0 +1,30 @@
|
|||
package com.example;
|
||||
|
||||
import org.springframework.boot.autoconfigure.condition.ConditionalOnMissingBean;
|
||||
import org.springframework.boot.autoconfigure.condition.ConditionalOnProperty;
|
||||
import org.springframework.context.annotation.Bean;
|
||||
import org.springframework.context.annotation.Configuration;
|
||||
|
||||
/**
|
||||
* `frameworks/spring/conditionals.ts` mints one `Annotation` node per Spring
|
||||
* condition and links its OWNER to it with a CONDITIONAL_ON edge. When the
|
||||
* owner is an `@Bean` factory METHOD (rather than the `@Configuration` class),
|
||||
* the edge is `Method -> Annotation` — the source label comes from the
|
||||
* scope-resolution bridge, the target is a structural `Annotation` node that
|
||||
* is in neither scope-bridge label set.
|
||||
*/
|
||||
@Configuration
|
||||
public class AppAutoConfiguration {
|
||||
|
||||
@Bean
|
||||
@ConditionalOnMissingBean
|
||||
public PaymentService paymentService() {
|
||||
return new PaymentService();
|
||||
}
|
||||
|
||||
@Bean
|
||||
@ConditionalOnProperty(prefix = "billing", name = "enabled", havingValue = "true")
|
||||
public PaymentService fallbackPaymentService() {
|
||||
return new PaymentService();
|
||||
}
|
||||
}
|
||||
5
gitnexus/test/fixtures/lang-resolution/spring-conditional-app/java/PaymentService.java
vendored
Normal file
5
gitnexus/test/fixtures/lang-resolution/spring-conditional-app/java/PaymentService.java
vendored
Normal file
|
|
@ -0,0 +1,5 @@
|
|||
package com.example;
|
||||
|
||||
public class PaymentService {
|
||||
public void pay() {}
|
||||
}
|
||||
20
gitnexus/test/fixtures/lang-resolution/spring-conditional-app/kotlin/KotlinAutoConfiguration.kt
vendored
Normal file
20
gitnexus/test/fixtures/lang-resolution/spring-conditional-app/kotlin/KotlinAutoConfiguration.kt
vendored
Normal file
|
|
@ -0,0 +1,20 @@
|
|||
package com.example.kotlin
|
||||
|
||||
import org.springframework.boot.autoconfigure.condition.ConditionalOnMissingBean
|
||||
import org.springframework.context.annotation.Bean
|
||||
import org.springframework.context.annotation.Configuration
|
||||
|
||||
class ReportService
|
||||
|
||||
/**
|
||||
* Kotlin twin of the Java `@Bean` + `@ConditionalOnMissingBean` case — the same
|
||||
* `Method -> Annotation` pair, reached through the Kotlin conditional metadata
|
||||
* adapter instead of the Java one.
|
||||
*/
|
||||
@Configuration
|
||||
class KotlinAutoConfiguration {
|
||||
|
||||
@Bean
|
||||
@ConditionalOnMissingBean
|
||||
fun reportService(): ReportService = ReportService()
|
||||
}
|
||||
29
gitnexus/test/fixtures/lang-resolution/vue-basic/OptionsHost.vue
vendored
Normal file
29
gitnexus/test/fixtures/lang-resolution/vue-basic/OptionsHost.vue
vendored
Normal file
|
|
@ -0,0 +1,29 @@
|
|||
<template>
|
||||
<div class="options-host">
|
||||
<Button variant="secondary" @click="onOptionsClick">
|
||||
Options click
|
||||
</Button>
|
||||
</div>
|
||||
</template>
|
||||
|
||||
<script lang="ts">
|
||||
import { defineComponent } from 'vue';
|
||||
import Button from './components/Button.vue';
|
||||
|
||||
/**
|
||||
* Options-API host: the `@click` handler lives in `methods:`, so the graph node
|
||||
* for it is a `Method`, not a `Function`. The BINDS_EVENT_HANDLER edge Vue's
|
||||
* scope resolver emits for a component-element binding therefore has a `Method`
|
||||
* source and a `File` target — the `Method→File` pair. The `<script setup>`
|
||||
* sibling (App.vue) only ever produces `Function→File`.
|
||||
*/
|
||||
export default defineComponent({
|
||||
name: 'OptionsHost',
|
||||
components: { Button },
|
||||
methods: {
|
||||
onOptionsClick() {
|
||||
return 'options-clicked';
|
||||
},
|
||||
},
|
||||
});
|
||||
</script>
|
||||
180
gitnexus/test/integration/structural-pair-coverage.test.ts
Normal file
180
gitnexus/test/integration/structural-pair-coverage.test.ts
Normal file
|
|
@ -0,0 +1,180 @@
|
|||
/**
|
||||
* Corpus-derived coverage for the HAND-DECLARED half of `RELATION_SCHEMA`.
|
||||
*
|
||||
* `test/unit/schema-pair-coverage.test.ts` derives its requirement from
|
||||
* schema.ts's two rules — the scope-resolution bridge cross product, and
|
||||
* `DEFINITION_ANCHOR_LABELS × ATTACHMENT_TARGET_LABELS` for the framework and
|
||||
* pipeline-phase overlays — so both generated halves are covered there. What
|
||||
* neither rule can reach is `STRUCTURAL_PAIR_DDL`: the containment, inheritance
|
||||
* and import pairs BETWEEN TWO DEFINITION LABELS. Eleven node tables are absent
|
||||
* from every rule's target side (`CodeElement`, `Impl`, `Namespace`,
|
||||
* `Template`, `TypeAlias`, `Typedef`, `Union`, `Static`, `Section`, `Folder`,
|
||||
* and the PDG-only `BasicBlock`), so a pair pointing at one is hand-declared or
|
||||
* it does not exist. No predicate describes that surface — any container can
|
||||
* hold any definition — so this asks the emitters directly: run the real
|
||||
* pipeline and require every FROM/TO pair it produces to be declared.
|
||||
*
|
||||
* Each entry also pins the pair it exists to guard. Without that the suite is
|
||||
* vacuous: `undeclared` is derived from what the pipeline emitted, so a fixture
|
||||
* that stopped emitting — renamed directory, grammar that failed to load,
|
||||
* swallowed parse error — yields an empty set and passes green while guarding
|
||||
* nothing. `sentinels` turns each case from "nothing undeclared" into "this
|
||||
* emitter still fires, and everything it emits is declared".
|
||||
*
|
||||
* Coverage is bounded by `NON_BRIDGE_CORPUS`: a sample, not a proof. A
|
||||
* language whose fixture is absent is unguarded, so a new structural emitter
|
||||
* should land with an entry here.
|
||||
*
|
||||
* Deliberately isolated from the resolver suites that already build three of
|
||||
* these graphs (`resolvers/cobol.test.ts`, `resolvers/vue.test.ts`,
|
||||
* `resolvers/php.test.ts`) — see NOTE below the corpus before re-raising that.
|
||||
*/
|
||||
import { describe, it, expect, vi, beforeAll, afterAll } from 'vitest';
|
||||
import path from 'path';
|
||||
import { NODE_TABLES } from 'gitnexus-shared';
|
||||
import { FIXTURES, runPipelineFromRepo } from './resolvers/helpers.js';
|
||||
import { RELATION_SCHEMA } from '../../src/core/lbug/schema.js';
|
||||
import { parseRelationSchemaPairs, relPairKeyFor } from '../../src/core/lbug/rel-pair-routing.js';
|
||||
import { DIST_WORKER_URL, distWorkerExists } from '../helpers/worker-parse.js';
|
||||
|
||||
vi.setConfig({ testTimeout: 180_000 });
|
||||
|
||||
const describeIfWorkerBuilt = distWorkerExists() ? describe : describe.skip;
|
||||
|
||||
type CorpusEntry = {
|
||||
/** Fixture directory under `test/fixtures/lang-resolution`. */
|
||||
readonly fixture: string;
|
||||
/**
|
||||
* The emitter this fixture exists to exercise, short enough that vitest does
|
||||
* not truncate it out of the case title (~36 chars).
|
||||
*/
|
||||
readonly emitter: string;
|
||||
/**
|
||||
* FROM|TO pairs the fixture MUST still emit. These are the anti-vacuity
|
||||
* check: they fail loudly when the fixture stops reaching the emitter,
|
||||
* which is the failure mode "no undeclared pairs" cannot see.
|
||||
*/
|
||||
readonly sentinels: readonly string[];
|
||||
};
|
||||
|
||||
/**
|
||||
* Fixtures chosen to reach a structural emitter that no other suite drives.
|
||||
*
|
||||
* The sentinels are the anti-vacuity anchor: each names an emitter that must
|
||||
* still fire. Two of them (`CodeElement|Property`, `Module|Namespace`) are also
|
||||
* the only pairs here that no generated rule can reach; the other nine are
|
||||
* rule-derived, and stay because a rule DECLARING a pair says nothing about
|
||||
* whether any emitter still PRODUCES it — which is the failure this corpus
|
||||
* exists to catch.
|
||||
*/
|
||||
const NON_BRIDGE_CORPUS = [
|
||||
{
|
||||
// `cobol-processor.ts`: CONTAINS/CALLS/ACCESSES over Module / Namespace /
|
||||
// Record / Property / CodeElement.
|
||||
fixture: 'cobol-app',
|
||||
emitter: 'cobol-processor containment',
|
||||
sentinels: ['CodeElement|Property', 'Module|Namespace', 'Module|Record', 'Record|Record'],
|
||||
},
|
||||
{
|
||||
// Same processor, DECLARATIVES section: the USE-procedure Namespace
|
||||
// ACCESSES a file Record.
|
||||
fixture: 'cobol-declaratives',
|
||||
emitter: 'cobol-processor DECLARATIVES',
|
||||
sentinels: ['Namespace|Record'],
|
||||
},
|
||||
{
|
||||
// `languages/vue/scope-resolver.ts`: the only edge whose target is a
|
||||
// `File`. Function→File from `<script setup>` (App.vue), Method→File from
|
||||
// the Options-API `methods:` host (OptionsHost.vue).
|
||||
fixture: 'vue-basic',
|
||||
emitter: 'vue BINDS_EVENT_HANDLER',
|
||||
sentinels: ['Function|File', 'Method|File'],
|
||||
},
|
||||
{
|
||||
// The inheritance pass, including trait-to-trait IMPLEMENTS.
|
||||
fixture: 'php-transitive-traits',
|
||||
emitter: 'inheritance-pass IMPLEMENTS',
|
||||
sentinels: ['Class|Trait', 'Trait|Trait'],
|
||||
},
|
||||
{
|
||||
// `frameworks/spring/conditionals.ts`: @ConditionalOn* on a @Bean method,
|
||||
// in both Java and Kotlin.
|
||||
fixture: 'spring-conditional-app',
|
||||
emitter: 'spring CONDITIONAL_ON',
|
||||
sentinels: ['Method|Annotation'],
|
||||
},
|
||||
{
|
||||
// `pipeline-phases/tools.ts`: a class-based MCP tool handler.
|
||||
fixture: 'mcp-tool-class',
|
||||
emitter: 'tools-phase HANDLES_TOOL',
|
||||
sentinels: ['Class|Tool'],
|
||||
},
|
||||
] as const satisfies readonly CorpusEntry[];
|
||||
|
||||
/*
|
||||
* NOTE — why this suite runs its own pipelines instead of reusing the resolver
|
||||
* suites' graphs (measured, not assumed):
|
||||
*
|
||||
* 1. Vitest runs with `pool: 'forks'` and default isolation, so every test
|
||||
* FILE gets its own child process. A fixture-keyed result cache in
|
||||
* `resolvers/helpers.ts` would be per-file module state and would share
|
||||
* nothing across files — zero saving.
|
||||
* 2. The graphs are not interchangeable. `resolvers/cobol.test.ts` builds
|
||||
* cobol-app with `{ skipGraphPhases: true }`, which drops the phase-emitted
|
||||
* edges (`Function|Process`, `Function|Community`, and the whole class of
|
||||
* pair `mcp-tool-class` exists to guard: HANDLES_TOOL is a pipeline phase).
|
||||
* Asserting there would silently cover LESS surface than here.
|
||||
* 3. Half the corpus has no existing home anyway, and the cases run
|
||||
* concurrently, so they overlap into roughly one fixture's wall time: all
|
||||
* 6 measured ~6s of test time against the ~16s this file spends on
|
||||
* transform+import before the first case starts. Hosting only the 3
|
||||
* homeless ones measured ~5s — a ~1s saving that still cannot remove a
|
||||
* file, so it cannot remove that ~16s.
|
||||
*/
|
||||
|
||||
const DECLARED = parseRelationSchemaPairs(RELATION_SCHEMA);
|
||||
const VALID_TABLES = new Set<string>(NODE_TABLES);
|
||||
|
||||
// Cold worker-pool startups otherwise flake against the 5s default ready budget
|
||||
// on a loaded runner, failing for reasons unrelated to the schema (#1741).
|
||||
// Safe under `it.concurrent`: env stubs are process-global, but `pool: 'forks'`
|
||||
// gives this file its own process and every case wants the same value.
|
||||
beforeAll(() => vi.stubEnv('GITNEXUS_WORKER_READY_TIMEOUT_MS', '60000'));
|
||||
afterAll(() => vi.unstubAllEnvs());
|
||||
|
||||
/**
|
||||
* The FROM/TO pairs a fixture emits, deduped.
|
||||
*
|
||||
* Classifies through `relPairKeyFor` — the same call the emit path routes with
|
||||
* — so this guard and the router cannot disagree about which edges are skipped.
|
||||
* It keys off the node IDS, which is what `assertDeclaredPair` actually sees,
|
||||
* not the node's `label` field.
|
||||
*/
|
||||
const pairsEmittedBy = async (fixture: string): Promise<Set<string>> => {
|
||||
const result = await runPipelineFromRepo(path.join(FIXTURES, fixture), () => {}, {
|
||||
workerPoolSize: 1,
|
||||
workerUrlForTest: DIST_WORKER_URL,
|
||||
});
|
||||
const emitted = new Set<string>();
|
||||
for (const rel of result.graph.iterRelationships()) {
|
||||
const pair = relPairKeyFor(rel.sourceId, rel.targetId, VALID_TABLES);
|
||||
if (pair !== undefined) emitted.add(pair);
|
||||
}
|
||||
return emitted;
|
||||
};
|
||||
|
||||
describeIfWorkerBuilt('RELATION_SCHEMA covers the non-bridge emitters', () => {
|
||||
// Concurrent because the cases share nothing but cost ~5s each serially,
|
||||
// almost all of it worker spawn and grammar load, which overlaps well.
|
||||
it.concurrent.each(NON_BRIDGE_CORPUS)(
|
||||
'$fixture emits only declared FROM/TO pairs, and still reaches $emitter',
|
||||
async ({ fixture, sentinels }) => {
|
||||
const emitted = await pairsEmittedBy(fixture);
|
||||
// Sorted so a failure is stable and names the pair to declare.
|
||||
expect({
|
||||
undeclaredPairs: [...emitted].filter((pair) => !DECLARED.has(pair)).sort(),
|
||||
missingSentinelPairs: sentinels.filter((pair) => !emitted.has(pair)),
|
||||
}).toEqual({ undeclaredPairs: [], missingSentinelPairs: [] });
|
||||
},
|
||||
);
|
||||
});
|
||||
169
gitnexus/test/unit/analyze-undeclared-pair-error.test.ts
Normal file
169
gitnexus/test/unit/analyze-undeclared-pair-error.test.ts
Normal file
|
|
@ -0,0 +1,169 @@
|
|||
/**
|
||||
* Tests for the undeclared FROM→TO label-pair failure path in the
|
||||
* `analyzeCommand` CLI (#2789).
|
||||
*
|
||||
* `assertDeclaredPair` aborts the run when an extracted edge's endpoint-label
|
||||
* pair is missing from GitNexus's own relation DDL — deliberately, because the
|
||||
* alternative is a late `COPY` failure that silently drops edges. Before this
|
||||
* branch existed the user got `Analysis failed` plus a stack trace through
|
||||
* GitNexus internals: no file, no relationship, and an implicit "try again"
|
||||
* that can never work. The CLI must instead name the pair, the relationship
|
||||
* type and the offending file, and point at an issue report.
|
||||
*
|
||||
* The CLI does NOT compose that text: it prints `err.message` indented (the
|
||||
* `LbugWipeError` idiom in the same catch block), because the message is
|
||||
* self-contained — `gitnexus serve` forwards only `err.message` over worker
|
||||
* IPC, so anything rendered here instead would be invisible to serve users.
|
||||
* The needles below are therefore the SAME strings
|
||||
* `test/unit/rel-pair-routing.test.ts` pins on the message itself.
|
||||
*
|
||||
* Mirrors analyze-http-endpoint-error.test.ts:
|
||||
* - vi.mock the heavy dependencies so no real DB / git is touched
|
||||
* - drive `analyzeCommand` with a mocked `runFullAnalysis` that rejects
|
||||
* - assert on process.exitCode and the captured logger records
|
||||
*/
|
||||
import { beforeEach, describe, expect, it, vi } from 'vitest';
|
||||
|
||||
const runFullAnalysisMock = vi.fn();
|
||||
|
||||
vi.mock('../../src/core/run-analyze.js', () => ({
|
||||
runFullAnalysis: runFullAnalysisMock,
|
||||
}));
|
||||
|
||||
vi.mock('../../src/core/lbug/lbug-adapter.js', () => ({
|
||||
closeLbug: vi.fn(async () => undefined),
|
||||
closeLbugBeforeExit: vi.fn(async () => undefined),
|
||||
isLbugReady: vi.fn(() => false),
|
||||
LbugWipeError: class LbugWipeError extends Error {},
|
||||
}));
|
||||
|
||||
vi.mock('../../src/storage/repo-manager.js', () => ({
|
||||
getStoragePaths: vi.fn(() => ({ storagePath: '.gitnexus', lbugPath: '.gitnexus/lbug' })),
|
||||
getGlobalRegistryPath: vi.fn(() => 'registry.json'),
|
||||
RegistryNameCollisionError: class RegistryNameCollisionError extends Error {},
|
||||
AnalysisNotFinalizedError: class AnalysisNotFinalizedError extends Error {},
|
||||
assertAnalysisFinalized: vi.fn(async () => undefined),
|
||||
}));
|
||||
|
||||
vi.mock('../../src/storage/git.js', () => ({
|
||||
getGitRoot: vi.fn(() => '/repo'),
|
||||
hasGitDir: vi.fn(() => true),
|
||||
}));
|
||||
|
||||
vi.mock('../../src/core/ingestion/utils/max-file-size.js', () => ({
|
||||
getMaxFileSizeBannerMessage: vi.fn(() => null),
|
||||
}));
|
||||
|
||||
// analyze.ts imports isHfDownloadFailure from hf-env.js — mock it to break the
|
||||
// transitive gitnexus-shared chain (same reason as the sibling suite).
|
||||
vi.mock('../../src/core/embeddings/hf-env.js', () => ({
|
||||
isHfDownloadFailure: vi.fn(() => false),
|
||||
}));
|
||||
|
||||
const PAIR_ERROR_ARGS = [
|
||||
'Method|Annotation',
|
||||
'ANNOTATED_BY',
|
||||
'Method:src/main/java/app/BeanConfig.java:BeanConfig.dataSource#42',
|
||||
'Annotation:src/main/java/app/BeanConfig.java:ConditionalOnMissingBean',
|
||||
] as const;
|
||||
|
||||
describe('analyzeCommand undeclared relation-pair handling (#2789)', () => {
|
||||
beforeEach(() => {
|
||||
vi.resetModules();
|
||||
runFullAnalysisMock.mockReset();
|
||||
process.exitCode = undefined;
|
||||
process.env.NODE_OPTIONS = `${process.env.NODE_OPTIONS ?? ''} --max-old-space-size=8192`.trim();
|
||||
});
|
||||
|
||||
it('renders an actionable schema-gap message naming the pair, relationship, ids and file', async () => {
|
||||
const { UndeclaredRelationPairError } = await import('../../src/core/lbug/rel-pair-routing.js');
|
||||
runFullAnalysisMock.mockRejectedValue(new UndeclaredRelationPairError(...PAIR_ERROR_ARGS));
|
||||
|
||||
const { _captureLogger } = await import('../../src/core/logger.js');
|
||||
const cap = _captureLogger();
|
||||
const { analyzeCommand } = await import('../../src/cli/analyze.js');
|
||||
|
||||
await analyzeCommand(undefined, {});
|
||||
|
||||
expect(process.exitCode).toBe(1);
|
||||
const record = cap.records().find((r) => r.recoveryHint === 'undeclared-relation-pair');
|
||||
cap.restore();
|
||||
|
||||
expect(record).toMatchObject({
|
||||
recoveryHint: 'undeclared-relation-pair',
|
||||
labelPair: 'Method|Annotation',
|
||||
relationType: 'ANNOTATED_BY',
|
||||
sourceFile: 'src/main/java/app/BeanConfig.java',
|
||||
});
|
||||
// The CLI renders `err.message` indented (the `LbugWipeError` idiom) rather
|
||||
// than re-formatting the structured fields, so these needles are the ONE
|
||||
// wording — `test/unit/rel-pair-routing.test.ts` pins the same strings on
|
||||
// the message itself and a reword updates one place, not two.
|
||||
// Filter-to-empty, not an array of booleans: the failure output NAMES the
|
||||
// missing string instead of making you count positions.
|
||||
const text = typeof record?.msg === 'string' ? record.msg : '';
|
||||
const required = [
|
||||
'Method → Annotation',
|
||||
'ANNOTATED_BY',
|
||||
'src/main/java/app/BeanConfig.java',
|
||||
'Method:src/main/java/app/BeanConfig.java:BeanConfig.dataSource#42',
|
||||
'Annotation:src/main/java/app/BeanConfig.java:ConditionalOnMissingBean',
|
||||
// Names it as a GitNexus gap, tells the user a re-run is pointless, and
|
||||
// points at both actionable next steps.
|
||||
"gap in GitNexus's own relation schema",
|
||||
're-running the analysis will fail',
|
||||
'https://github.com/abhigyanpatwari/GitNexus/issues/new',
|
||||
'.gitnexusignore',
|
||||
];
|
||||
expect(required.filter((needle) => !text.includes(needle))).toEqual([]);
|
||||
});
|
||||
|
||||
it('still fires when the ingestion phase runner has rewrapped it as a cause', async () => {
|
||||
// The guard throws inside an emit phase, and the phase runner rewraps every
|
||||
// phase failure as `new Error("Phase 'X' failed: …", { cause })` — a bare
|
||||
// instanceof check at the CLI boundary would miss it entirely.
|
||||
const { UndeclaredRelationPairError } = await import('../../src/core/lbug/rel-pair-routing.js');
|
||||
const original = new UndeclaredRelationPairError(...PAIR_ERROR_ARGS);
|
||||
// The wrapper deliberately does NOT interpolate the cause's message: the
|
||||
// branch must render `undeclaredPair.message`, i.e. the message of the link
|
||||
// it FOUND in the chain, not the outer wrapper's message.
|
||||
runFullAnalysisMock.mockRejectedValue(
|
||||
new Error(`Phase 'graph-emit' failed`, { cause: original }),
|
||||
);
|
||||
|
||||
const { _captureLogger } = await import('../../src/core/logger.js');
|
||||
const cap = _captureLogger();
|
||||
const { analyzeCommand } = await import('../../src/cli/analyze.js');
|
||||
|
||||
await analyzeCommand(undefined, {});
|
||||
|
||||
expect(process.exitCode).toBe(1);
|
||||
const records = cap.records();
|
||||
cap.restore();
|
||||
|
||||
const record = records.find((r) => r.recoveryHint === 'undeclared-relation-pair');
|
||||
expect(record).toMatchObject({ labelPair: 'Method|Annotation' });
|
||||
const text = typeof record?.msg === 'string' ? record.msg : '';
|
||||
expect(
|
||||
['Method → Annotation', '.gitnexusignore'].filter((needle) => !text.includes(needle)),
|
||||
).toEqual([]);
|
||||
// The generic large-repo / module-not-found guidance must not also appear.
|
||||
expect(records.some((r) => r.recoveryHint === 'large-repo')).toBe(false);
|
||||
expect(records.some((r) => r.recoveryHint === 'module-not-found')).toBe(false);
|
||||
});
|
||||
|
||||
it('does not claim an unrelated failure', async () => {
|
||||
runFullAnalysisMock.mockRejectedValue(new Error('LadybugDB write failed'));
|
||||
|
||||
const { _captureLogger } = await import('../../src/core/logger.js');
|
||||
const cap = _captureLogger();
|
||||
const { analyzeCommand } = await import('../../src/cli/analyze.js');
|
||||
|
||||
await analyzeCommand(undefined, {});
|
||||
|
||||
expect(process.exitCode).toBe(1);
|
||||
const records = cap.records();
|
||||
cap.restore();
|
||||
expect(records.some((r) => r.recoveryHint === 'undeclared-relation-pair')).toBe(false);
|
||||
});
|
||||
});
|
||||
|
|
@ -73,12 +73,12 @@ describe('CALL_SUMMARY relation-type exclusion (U-C1)', () => {
|
|||
});
|
||||
|
||||
describe('CALL_SUMMARY incremental reuse gate (U-C5)', () => {
|
||||
it('INCREMENTAL_SCHEMA_VERSION is bumped to 34 (Spring AOP relation pairs #2416, then receiver-chain wire format v2)', () => {
|
||||
it('INCREMENTAL_SCHEMA_VERSION is bumped to 35 (receiver-chain wire format v2, then the full scope-resolution relation cross product #2792)', () => {
|
||||
// Moves with every bump BY DESIGN — that is the point of pinning it. A
|
||||
// change that alters emitted ids or edges without bumping would otherwise
|
||||
// ship silently, and an existing index would keep serving the old graph
|
||||
// through the reuse gate below.
|
||||
expect(INCREMENTAL_SCHEMA_VERSION).toBe(34);
|
||||
expect(INCREMENTAL_SCHEMA_VERSION).toBe(35);
|
||||
});
|
||||
|
||||
it('a pre-current stamp fails the `=== INCREMENTAL_SCHEMA_VERSION` reuse gate → forces full re-analyze', () => {
|
||||
|
|
@ -223,7 +223,11 @@ describe('CALL_SUMMARY incremental reuse gate (U-C5)', () => {
|
|||
// would silently fall back to the text cascade for every chain-carrying
|
||||
// site → must NOT reuse.
|
||||
expect(passesReuseGate(33)).toBe(false);
|
||||
// A pre-v35 (v34) index was created against a relation DDL missing 99 of the
|
||||
// scope-resolution FROM/TO pairs (#2792) — LadybugDB fixes endpoint pairs at
|
||||
// CREATE time, so those edges cannot be written into it at all.
|
||||
expect(passesReuseGate(34)).toBe(false);
|
||||
// The current stamp passes the gate (incremental top-up eligible).
|
||||
expect(passesReuseGate(34)).toBe(true);
|
||||
expect(passesReuseGate(35)).toBe(true);
|
||||
});
|
||||
});
|
||||
|
|
|
|||
|
|
@ -147,7 +147,7 @@ describe('GraphEmitSink routing', () => {
|
|||
reason: 'direct',
|
||||
};
|
||||
expect(() => sink.addRelationship(undeclared)).toThrow(
|
||||
/Relationship label pair Static→Static is not declared/,
|
||||
/Relationship label pair Static → Static is not declared/,
|
||||
);
|
||||
});
|
||||
});
|
||||
|
|
|
|||
|
|
@ -95,7 +95,7 @@ describe('PdgEmitSink — routing', () => {
|
|||
reason: 'seq',
|
||||
};
|
||||
expect(() => sink.addRelationship(undeclared)).toThrow(
|
||||
/Relationship label pair Static→Static is not declared/,
|
||||
/Relationship label pair Static → Static is not declared/,
|
||||
);
|
||||
});
|
||||
|
||||
|
|
|
|||
|
|
@ -5,8 +5,14 @@ import path from 'path';
|
|||
import os from 'os';
|
||||
import {
|
||||
RelPairRouter,
|
||||
UndeclaredRelationPairError,
|
||||
assertDeclaredPair,
|
||||
createRelationPairMatcher,
|
||||
findUndeclaredRelationPairError,
|
||||
getNodeLabel,
|
||||
parseRelationSchemaPairs,
|
||||
relPairKeyFor,
|
||||
splitRelPairKey,
|
||||
} from '../../src/core/lbug/rel-pair-routing.js';
|
||||
|
||||
/**
|
||||
|
|
@ -102,6 +108,47 @@ describe('getNodeLabel', () => {
|
|||
});
|
||||
});
|
||||
|
||||
/**
|
||||
* `relPairKeyFor` is the ONE classifier `RelPairRouter.route`,
|
||||
* `GraphEmitSink.addRelationship`, `PdgEmitSink.addRelationship` and the
|
||||
* `structural-pair-coverage` corpus guard all route through. Each used to
|
||||
* inline the same three lines; the corpus guard's docblock said it "mirrors
|
||||
* RelPairRouter.route", which meant a change to the skip rule here would leave
|
||||
* the guard classifying by the old rule — green while `analyze` aborts.
|
||||
*/
|
||||
describe('relPairKeyFor', () => {
|
||||
const VALID_PAIR_TABLES = new Set(['File', 'Function', 'Community']);
|
||||
|
||||
it('keys an edge whose endpoints are both node tables, and skips one that is not', () => {
|
||||
expect(relPairKeyFor('File:src/a.ts', 'Function:src/a.ts:f:1', VALID_PAIR_TABLES)).toBe(
|
||||
'File|Function',
|
||||
);
|
||||
// Synthetic ids still classify through getNodeLabel's prefix rules.
|
||||
expect(relPairKeyFor('comm_1', 'comm_2', VALID_PAIR_TABLES)).toBe('Community|Community');
|
||||
// `undefined` = SKIP, on either endpoint. Every caller drops the edge.
|
||||
expect(
|
||||
relPairKeyFor('Bogus:src/a.ts', 'Function:src/a.ts:f:1', VALID_PAIR_TABLES),
|
||||
).toBeUndefined();
|
||||
expect(relPairKeyFor('File:src/a.ts', 'Bogus:src/a.ts', VALID_PAIR_TABLES)).toBeUndefined();
|
||||
});
|
||||
|
||||
it('agrees with the labels getNodeLabel derives (no second derivation rule)', () => {
|
||||
const from = 'File:src/a.ts';
|
||||
const to = 'Function:src/a.ts:f:1';
|
||||
expect(relPairKeyFor(from, to, VALID_PAIR_TABLES)).toBe(
|
||||
`${getNodeLabel(from)}|${getNodeLabel(to)}`,
|
||||
);
|
||||
});
|
||||
|
||||
it('round-trips through splitRelPairKey, the only sanctioned decoder', () => {
|
||||
const key = relPairKeyFor('File:src/a.ts', 'Function:src/a.ts:f:1', VALID_PAIR_TABLES);
|
||||
expect(splitRelPairKey(key ?? '')).toEqual(['File', 'Function']);
|
||||
// `|` cannot occur inside a NODE_TABLES identifier, so the FIRST `|` is
|
||||
// always the separator — that invariant is what makes decoding safe.
|
||||
expect(splitRelPairKey('BasicBlock|BasicBlock')).toEqual(['BasicBlock', 'BasicBlock']);
|
||||
});
|
||||
});
|
||||
|
||||
describe('parseRelationSchemaPairs', () => {
|
||||
it('extracts plain and quoted FROM→TO labels for router validation', () => {
|
||||
expect(
|
||||
|
|
@ -116,13 +163,166 @@ describe('parseRelationSchemaPairs', () => {
|
|||
});
|
||||
});
|
||||
|
||||
/**
|
||||
* The `FROM…TO` pattern is exported so `test/unit/schema-pair-coverage.test.ts`
|
||||
* can COUNT raw occurrences with the very regex the parser de-duplicates with
|
||||
* (count > set size ⇒ a pair is duplicated in the DDL ⇒ LadybugDB rejects
|
||||
* `CREATE REL TABLE` and every `analyze` dies). While that guard inlined its own
|
||||
* copy, a widening on either side would have degraded it to
|
||||
* `declared.size === declared.size` with nothing failing. These tests pin both
|
||||
* halves of the coupling: the factory's freshness contract, and the exact pair
|
||||
* set the shared pattern produces for the widening-adjacent DDL shapes.
|
||||
*/
|
||||
describe('createRelationPairMatcher', () => {
|
||||
it('returns a fresh global matcher per call so lastIndex cannot leak between consumers', () => {
|
||||
const first = createRelationPairMatcher();
|
||||
const second = createRelationPairMatcher();
|
||||
expect(first).not.toBe(second);
|
||||
expect([first.global, first.lastIndex, second.lastIndex]).toEqual([true, 0, 0]);
|
||||
|
||||
first.exec('FROM Class TO CodeElement');
|
||||
// The used instance advanced; a newly built one is still at the start.
|
||||
expect([first.lastIndex > 0, createRelationPairMatcher().lastIndex]).toEqual([true, 0]);
|
||||
});
|
||||
|
||||
it('is the pattern parseRelationSchemaPairs itself uses (no re-inlined copy)', () => {
|
||||
// Each shape is either a form the pattern accepts today or a widening the
|
||||
// finding calls out (dotted identifier, multi-target `FROM x TO y, z`).
|
||||
const ddlShapes = [
|
||||
'FROM Class TO CodeElement',
|
||||
'FROM `Enum` TO `TypeAlias`',
|
||||
'CREATE REL TABLE IF NOT EXISTS CodeRelation(FROM A TO B, FROM A TO B, type STRING)',
|
||||
'FROM ns.Class TO Other',
|
||||
'FROM A TO B, C',
|
||||
];
|
||||
const viaMatcher = ddlShapes.map((ddl) =>
|
||||
[...ddl.matchAll(createRelationPairMatcher())].map((m) => `${m[1]}|${m[2]}`),
|
||||
);
|
||||
|
||||
// Pins what the shared pattern matches. Widening the exported matcher
|
||||
// without updating the duplicate-count guard's expectations fails here.
|
||||
expect(viaMatcher).toEqual([
|
||||
['Class|CodeElement'],
|
||||
['Enum|TypeAlias'],
|
||||
['A|B', 'A|B'], // duplicate survives the raw count; the parser dedups it
|
||||
[], // dotted identifiers are NOT matched today
|
||||
['A|B'], // multi-target: only the first target is matched today
|
||||
]);
|
||||
// Re-inlining a DIFFERENT regex inside parseRelationSchemaPairs breaks this.
|
||||
expect(ddlShapes.map((ddl) => [...parseRelationSchemaPairs(ddl)])).toEqual(
|
||||
viaMatcher.map((pairs) => [...new Set(pairs)]),
|
||||
);
|
||||
});
|
||||
});
|
||||
|
||||
describe('assertDeclaredPair', () => {
|
||||
const DECLARED_ONE = new Set<string>(['Function|Function']);
|
||||
|
||||
it('passes a declared pair through and throws a typed error for an undeclared one', () => {
|
||||
expect(
|
||||
assertDeclaredPair(
|
||||
'Function|Function',
|
||||
DECLARED_ONE,
|
||||
'CALLS',
|
||||
'Function:src/a.ts:f:1',
|
||||
'Function:src/a.ts:g:2',
|
||||
),
|
||||
).toBeUndefined();
|
||||
expect(() =>
|
||||
assertDeclaredPair(
|
||||
'Method|Annotation',
|
||||
DECLARED_ONE,
|
||||
'ANNOTATED_BY',
|
||||
'Method:src/app/Config.java:Config.dataSource#12',
|
||||
'Annotation:src/app/Config.java:ConditionalOnMissingBean',
|
||||
),
|
||||
).toThrow(UndeclaredRelationPairError);
|
||||
});
|
||||
|
||||
it('carries the pair, relationship type, both node ids and the source file (#2789)', () => {
|
||||
const thrown = (() => {
|
||||
try {
|
||||
assertDeclaredPair(
|
||||
'Method|Annotation',
|
||||
DECLARED_ONE,
|
||||
'ANNOTATED_BY',
|
||||
'Method:src/app/Config.java:Config.dataSource#12',
|
||||
'Annotation:src/app/Config.java:ConditionalOnMissingBean',
|
||||
);
|
||||
return undefined;
|
||||
} catch (err) {
|
||||
return err;
|
||||
}
|
||||
})();
|
||||
|
||||
expect(thrown).toBeInstanceOf(UndeclaredRelationPairError);
|
||||
expect(thrown).toMatchObject({
|
||||
name: 'UndeclaredRelationPairError',
|
||||
pairKey: 'Method|Annotation',
|
||||
relationType: 'ANNOTATED_BY',
|
||||
fromId: 'Method:src/app/Config.java:Config.dataSource#12',
|
||||
toId: 'Annotation:src/app/Config.java:ConditionalOnMissingBean',
|
||||
sourceFile: 'src/app/Config.java',
|
||||
});
|
||||
// Everything a bug report needs must also survive in the message alone:
|
||||
// `gitnexus serve` forwards nothing but `err.message` over worker IPC, so
|
||||
// this message is the ONLY rendering — `cli/analyze.ts` prints it verbatim
|
||||
// rather than re-formatting the structured fields into a second copy.
|
||||
// Filter-to-empty rather than an array of booleans: the failure output
|
||||
// NAMES the missing string instead of making you count `true`s.
|
||||
const message = (thrown as UndeclaredRelationPairError).message;
|
||||
const required = [
|
||||
'Method → Annotation is not declared in the LadybugDB relation schema',
|
||||
'ANNOTATED_BY',
|
||||
'Method:src/app/Config.java:Config.dataSource#12',
|
||||
'Annotation:src/app/Config.java:ConditionalOnMissingBean',
|
||||
'src/app/Config.java',
|
||||
// The two ACTIONABLE items. They live in the message, not in the CLI
|
||||
// branch, so a `gitnexus serve` user gets them too.
|
||||
'https://github.com/abhigyanpatwari/GitNexus/issues/new',
|
||||
'.gitnexusignore',
|
||||
"gap in GitNexus's own relation schema",
|
||||
're-running the analysis will fail in exactly the same place',
|
||||
];
|
||||
expect(required.filter((needle) => !message.includes(needle))).toEqual([]);
|
||||
});
|
||||
|
||||
it('reports no source file for synthetic community/process ids instead of guessing', () => {
|
||||
const err = new UndeclaredRelationPairError(
|
||||
'Community|Process',
|
||||
'BELONGS_TO',
|
||||
'comm_4',
|
||||
'proc_7',
|
||||
);
|
||||
expect(err.sourceFile).toBeUndefined();
|
||||
expect(['(none — synthetic node id)'].filter((n) => !err.message.includes(n))).toEqual([]);
|
||||
});
|
||||
|
||||
it("is findable through the phase runner's cause chain", () => {
|
||||
const original = new UndeclaredRelationPairError(
|
||||
'Method|Annotation',
|
||||
'ANNOTATED_BY',
|
||||
'Method:src/app/Config.java:Config.dataSource#12',
|
||||
'Annotation:src/app/Config.java:ConditionalOnMissingBean',
|
||||
);
|
||||
const wrapped = new Error("Phase 'graph-emit' failed: …", {
|
||||
cause: new Error('emit failed', { cause: original }),
|
||||
});
|
||||
|
||||
expect(findUndeclaredRelationPairError(wrapped)).toBe(original);
|
||||
expect(findUndeclaredRelationPairError(original)).toBe(original);
|
||||
expect(findUndeclaredRelationPairError(new Error('unrelated'))).toBeUndefined();
|
||||
expect(findUndeclaredRelationPairError('not an error')).toBeUndefined();
|
||||
});
|
||||
});
|
||||
|
||||
describe('RelPairRouter', () => {
|
||||
it('routes valid edges to per-pair files (header first) and skips invalid-label edges', async () => {
|
||||
const streams: MockWriteStream[] = [];
|
||||
const router = new RelPairRouter(tmpDir, HEADER, VALID, DECLARED, mockFactory(streams));
|
||||
|
||||
const route = async (from: string, to: string) => {
|
||||
const p = router.route(from, to, row(from, to));
|
||||
const p = router.route(from, to, row(from, to), 'CALLS');
|
||||
if (p) await p;
|
||||
};
|
||||
await route('File:a', 'Function:a:f:1');
|
||||
|
|
@ -145,9 +345,20 @@ describe('RelPairRouter', () => {
|
|||
const streams: MockWriteStream[] = [];
|
||||
const router = new RelPairRouter(tmpDir, HEADER, VALID, DECLARED, mockFactory(streams));
|
||||
|
||||
expect(() => router.route('File:a', 'Community:1', row('File:a', 'Community:1'))).toThrow(
|
||||
'File→Community is not declared in the LadybugDB relation schema',
|
||||
);
|
||||
const route = () =>
|
||||
router.route(
|
||||
'File:src/a.ts',
|
||||
'Community:1',
|
||||
row('File:src/a.ts', 'Community:1', 'DEFINES'),
|
||||
'DEFINES',
|
||||
);
|
||||
expect(route).toThrow('File → Community is not declared in the LadybugDB relation schema');
|
||||
// The row is already CSV-escaped here, so the router must forward the edge
|
||||
// context itself — otherwise the crash names only the abstract label pair.
|
||||
expect(route).toThrow(UndeclaredRelationPairError);
|
||||
expect(route).toThrow(/DEFINES/);
|
||||
expect(route).toThrow(/File:src\/a\.ts/);
|
||||
expect(route).toThrow(/Community:1/);
|
||||
expect(streams).toHaveLength(0);
|
||||
expect(router.skipped).toBe(0);
|
||||
expect(router.total).toBe(0);
|
||||
|
|
@ -163,7 +374,12 @@ describe('RelPairRouter', () => {
|
|||
mockFactory(streams, { blocked: true }),
|
||||
);
|
||||
|
||||
const pending = router.route('File:a', 'Function:a:f:1', row('File:a', 'Function:a:f:1'));
|
||||
const pending = router.route(
|
||||
'File:a',
|
||||
'Function:a:f:1',
|
||||
row('File:a', 'Function:a:f:1'),
|
||||
'DEFINES',
|
||||
);
|
||||
expect(pending).toBeInstanceOf(Promise); // header write hit backpressure
|
||||
streams[0].unblock();
|
||||
await pending;
|
||||
|
|
@ -177,16 +393,21 @@ describe('RelPairRouter', () => {
|
|||
const streams: MockWriteStream[] = [];
|
||||
const router = new RelPairRouter(tmpDir, HEADER, VALID, DECLARED, mockFactory(streams));
|
||||
|
||||
const first = router.route('File:a', 'Function:a:f:1', row('File:a', 'Function:a:f:1'));
|
||||
const first = router.route(
|
||||
'File:a',
|
||||
'Function:a:f:1',
|
||||
row('File:a', 'Function:a:f:1'),
|
||||
'DEFINES',
|
||||
);
|
||||
if (first) await first;
|
||||
|
||||
const err = new Error('EMFILE: too many open files');
|
||||
streams[0].triggerError(err);
|
||||
|
||||
// The next route surfaces the REAL error, not a generic AbortError.
|
||||
expect(() => router.route('File:a', 'Function:a:g:2', row('File:a', 'Function:a:g:2'))).toThrow(
|
||||
'EMFILE',
|
||||
);
|
||||
expect(() =>
|
||||
router.route('File:a', 'Function:a:g:2', row('File:a', 'Function:a:g:2'), 'DEFINES'),
|
||||
).toThrow('EMFILE');
|
||||
expect(router.lastError).toBe(err);
|
||||
await expect(router.close()).rejects.toThrow('EMFILE');
|
||||
expect(streams[0].destroyed).toBe(true);
|
||||
|
|
@ -196,9 +417,14 @@ describe('RelPairRouter', () => {
|
|||
const streams: MockWriteStream[] = [];
|
||||
const router = new RelPairRouter(tmpDir, HEADER, VALID, DECLARED, mockFactory(streams));
|
||||
|
||||
const a = router.route('File:a', 'Function:a:f:1', row('File:a', 'Function:a:f:1'));
|
||||
const a = router.route('File:a', 'Function:a:f:1', row('File:a', 'Function:a:f:1'), 'DEFINES');
|
||||
if (a) await a;
|
||||
const b = router.route('Community:1', 'Community:2', row('Community:1', 'Community:2'));
|
||||
const b = router.route(
|
||||
'Community:1',
|
||||
'Community:2',
|
||||
row('Community:1', 'Community:2'),
|
||||
'RELATED_TO',
|
||||
);
|
||||
if (b) await b;
|
||||
|
||||
router.destroy();
|
||||
|
|
|
|||
211
gitnexus/test/unit/schema-pair-coverage.test.ts
Normal file
211
gitnexus/test/unit/schema-pair-coverage.test.ts
Normal file
|
|
@ -0,0 +1,211 @@
|
|||
import { describe, it, expect } from 'vitest';
|
||||
import type { NodeLabel, NodeTableName } from 'gitnexus-shared';
|
||||
import { NODE_TABLES } from 'gitnexus-shared';
|
||||
import { RELATION_SCHEMA, STRUCTURAL_PAIR_DDL } from '../../src/core/lbug/schema.js';
|
||||
import {
|
||||
createRelationPairMatcher,
|
||||
parseRelationSchemaPairs,
|
||||
} from '../../src/core/lbug/rel-pair-routing.js';
|
||||
import { LINKABLE_LABELS } from '../../src/core/ingestion/scope-resolution/graph-bridge/node-lookup.js';
|
||||
import { CALLER_ANCHOR_LABELS } from '../../src/core/ingestion/scope-resolution/graph-bridge/ids.js';
|
||||
import { CALL_TARGET_TYPES } from '../../src/core/ingestion/model/symbol-table.js';
|
||||
|
||||
/**
|
||||
* `RELATION_SCHEMA` is generated from two cross products plus one hand-written
|
||||
* block. This file guards the generated half; a pair drawn from either rule and
|
||||
* absent from the DDL does not degrade — `assertDeclaredPair` throws and
|
||||
* `analyze` dies mid-phase on whichever codebase first produces it. Failing
|
||||
* here means fixing the rule, not the assertion.
|
||||
*
|
||||
* WHAT SCHEMA.TS NO LONGER RISKS. The DDL used to carry hand-copied twins of
|
||||
* `LINKABLE_LABELS` / `CALL_TARGET_TYPES`. Those are deleted — schema.ts now
|
||||
* imports the originals — so two whole failure modes are structurally
|
||||
* impossible rather than merely asserted: a label present only in the twin, and
|
||||
* a label removed from the original while the DDL keeps its pairs. Neither was
|
||||
* catchable before (a twin-only `Class|Record` left every assertion green).
|
||||
*
|
||||
* WHAT STILL NEEDS ASSERTING, and is:
|
||||
* - the rules themselves — the constants below are a deliberate PIN of
|
||||
* schema.ts's two rules, recomputed here from `LINKABLE_LABELS`,
|
||||
* `CALL_TARGET_TYPES` and `NODE_TABLES`. Widening a rule in schema.ts alone
|
||||
* fails `generated region matches …`, so the widening has to be stated
|
||||
* twice, on purpose.
|
||||
* - the generated region for EXACT equality, not containment — so a pair
|
||||
* hand-added to the generated half (the reflex that produced #2781, #2792
|
||||
* and #2793) fails just as loudly as a missing one.
|
||||
* - `STRUCTURAL_PAIR_DDL` carrying nothing a rule already generates. Every
|
||||
* other assertion here subtracts `structural` from BOTH sides, so a
|
||||
* redundant hand-declaration was invisible to all of them; `no hand-declared
|
||||
* pair …` below is the one that sees it.
|
||||
*
|
||||
* WHAT THIS FILE CANNOT SEE: a pair hand-added to {@link STRUCTURAL_PAIR_DDL}
|
||||
* that NEITHER rule covers — the ~72 containment/inheritance/import pairs
|
||||
* between two definition labels. That surface has no predicate, so it is
|
||||
* bounded by a corpus instead, in
|
||||
* `test/integration/structural-pair-coverage.test.ts`.
|
||||
*/
|
||||
|
||||
/** Rule 1 — the scope-resolution graph bridge (#2792). */
|
||||
const scopeBridgePairs = (): readonly string[] => {
|
||||
const sources: NodeLabel[] = ['File', ...LINKABLE_LABELS];
|
||||
const targets = new Set<NodeLabel>([...LINKABLE_LABELS, ...CALL_TARGET_TYPES]);
|
||||
return sources.flatMap((from) => [...targets].map((to) => `${from}|${to}`));
|
||||
};
|
||||
|
||||
/**
|
||||
* Pin of `NON_DEFINITION_LABELS` — the node tables schema.ts refuses as an
|
||||
* attachment anchor. `Route` / `Tool` are here despite sourcing `ENTRY_POINT_OF`
|
||||
* to a `Process`: that emitter names both labels as literals, so those two pairs
|
||||
* are hand-declared rather than generated.
|
||||
*/
|
||||
const NON_DEFINITION_LABELS: readonly NodeTableName[] = [
|
||||
'Community',
|
||||
'Process',
|
||||
'Route',
|
||||
'Tool',
|
||||
'Folder',
|
||||
'BasicBlock',
|
||||
];
|
||||
|
||||
/**
|
||||
* Pin of `ATTACHMENT_TARGET_LABELS` — labels minted outside the bridge by a
|
||||
* phase/framework emitter and hung off whichever definition node that emitter
|
||||
* resolved. For most of them the anchor is a lookup result, so its label is
|
||||
* unconstrained. `Community` and `Route` are the exceptions and over-declare on
|
||||
* purpose (their anchors are label-gated today — see schema.ts); keeping them in
|
||||
* the cross product is deliberate headroom, so do NOT narrow this pin to make an
|
||||
* assertion smaller.
|
||||
*/
|
||||
const ATTACHMENT_TARGET_LABELS: readonly NodeTableName[] = [
|
||||
'Annotation',
|
||||
'Community',
|
||||
'Process',
|
||||
'Route',
|
||||
'Tool',
|
||||
'File',
|
||||
'Record',
|
||||
];
|
||||
|
||||
/** Rule 2 — phase/framework overlays hung off a resolved anchor (#2793). */
|
||||
const attachmentPairs = (): readonly string[] => {
|
||||
const anchors = NODE_TABLES.filter((label) => !NON_DEFINITION_LABELS.includes(label));
|
||||
return anchors.flatMap((from) => ATTACHMENT_TARGET_LABELS.map((to) => `${from}|${to}`));
|
||||
};
|
||||
|
||||
describe('RELATION_SCHEMA pair coverage', () => {
|
||||
const declared = parseRelationSchemaPairs(RELATION_SCHEMA);
|
||||
const structural = parseRelationSchemaPairs(STRUCTURAL_PAIR_DDL);
|
||||
|
||||
it('declares every pair the scope-resolution bridge can emit', () => {
|
||||
const missing = scopeBridgePairs()
|
||||
.filter((pair) => !declared.has(pair))
|
||||
.sort();
|
||||
expect(missing).toEqual([]);
|
||||
});
|
||||
|
||||
it('declares every pair a phase/framework overlay can attach', () => {
|
||||
// `Method|Annotation` (Spring @Bean + @ConditionalOnMissingBean),
|
||||
// `Method|File` (Vue Options-API handler), `Namespace|Record` (COBOL
|
||||
// DECLARATIVES) and `Class|Tool` (@mcp.tool() on a class) were all live
|
||||
// aborts at PR #2793's head, from three different emitters.
|
||||
const missing = attachmentPairs()
|
||||
.filter((pair) => !declared.has(pair))
|
||||
.sort();
|
||||
expect(missing).toEqual([]);
|
||||
});
|
||||
|
||||
it('declares no hand-declared pair that a rule already generates', () => {
|
||||
// The guard against the exact failure mode this PR exists to eliminate, and
|
||||
// the ONLY one that can see it. `generatedPairDdl` skips any pair
|
||||
// already in STRUCTURAL_PAIR_DDL, so a redundant hand-declaration is not an
|
||||
// inert duplicate — it SUPPRESSES generation. Narrowing a rule later would
|
||||
// then silently keep that pair alive, and every other assertion in this file
|
||||
// subtracts `structural` from both sides, so none of them would notice.
|
||||
// (The integration corpus only checks emitted ⊆ declared — blind to an
|
||||
// EXCESS declaration by construction.) 161 such lines were deleted when this
|
||||
// guard went in; the DDL's pair set did not change, they moved into the
|
||||
// generated half. Failing here means deleting the line, not widening this.
|
||||
const allRulePairs = new Set([...scopeBridgePairs(), ...attachmentPairs()]);
|
||||
expect([...structural].filter((pair) => allRulePairs.has(pair))).toEqual([]);
|
||||
});
|
||||
|
||||
it('generated region matches the two rules exactly, with nothing extra', () => {
|
||||
// Equality, not containment: a pair hand-added to the generated half would
|
||||
// pass a subset check while quietly re-establishing the hand-list this PR
|
||||
// replaced. Both directions are load-bearing.
|
||||
const generated = [...declared].filter((pair) => !structural.has(pair)).sort();
|
||||
const expected = [...new Set([...scopeBridgePairs(), ...attachmentPairs()])]
|
||||
.filter((pair) => !structural.has(pair))
|
||||
.sort();
|
||||
expect(generated).toEqual(expected);
|
||||
});
|
||||
|
||||
it('keeps caller anchors a subset of linkable labels', () => {
|
||||
// A caller anchor outside the lookup's label set can never resolve to an
|
||||
// id, so `resolveCallerGraphId` would silently climb past it to the File
|
||||
// fallback and attribute the call to the module.
|
||||
const unlinkable = [...CALLER_ANCHOR_LABELS].filter((label) => !LINKABLE_LABELS.has(label));
|
||||
expect(unlinkable).toEqual([]);
|
||||
});
|
||||
|
||||
it('names only real node tables on both endpoints', () => {
|
||||
const tables = new Set<string>(NODE_TABLES);
|
||||
const unknown = [...declared]
|
||||
.flatMap((pair) => pair.split('|'))
|
||||
.filter((label) => !tables.has(label))
|
||||
.sort();
|
||||
expect(unknown).toEqual([]);
|
||||
});
|
||||
|
||||
it('declares each pair exactly once', () => {
|
||||
// LadybugDB rejects a duplicated FROM/TO pair in the DDL, which would take
|
||||
// out every `analyze` rather than one codebase's edge shape. Counts raw
|
||||
// occurrences with the SAME matcher `parseRelationSchemaPairs` dedups
|
||||
// through — re-inlining a copy of that regex would let any widening of it
|
||||
// degrade this into the tautology `declared.size === declared.size`.
|
||||
const occurrences = [...RELATION_SCHEMA.matchAll(createRelationPairMatcher())].length;
|
||||
expect(occurrences).toBe(declared.size);
|
||||
});
|
||||
|
||||
it('declares the pairs from the reported analyze crashes', () => {
|
||||
// Java static/field initializer referencing a Variable (#2792); Vue/JS
|
||||
// `const obj = { method() {} }` receiver (#2781); then the four #2793
|
||||
// aborts, each reproduced on the default `analyze` path against its own
|
||||
// fixture under `test/fixtures/lang-resolution/`.
|
||||
const reported = [
|
||||
'Class|Variable',
|
||||
'Const|Method',
|
||||
'Method|Annotation',
|
||||
'Method|File',
|
||||
'Namespace|Record',
|
||||
'Class|Tool',
|
||||
];
|
||||
expect(reported.filter((pair) => !declared.has(pair))).toEqual([]);
|
||||
});
|
||||
|
||||
it('declares the non-bridge structural pairs (#2789)', () => {
|
||||
// COBOL containment/call/access. Every pair listed here is outside BOTH
|
||||
// rules above, so no derived requirement can reach it and it must stay
|
||||
// hand-declared in STRUCTURAL_PAIR_DDL;
|
||||
// `test/integration/structural-pair-coverage.test.ts` guards them from a
|
||||
// corpus. Pinned here too so deleting a fixture there cannot silently drop
|
||||
// the guard, and so the cheap check does not need a build.
|
||||
//
|
||||
// #2789's original list also named `CodeElement|Record`, `Function|File`,
|
||||
// `Module|Record` and `Record|Record`. Those are rule-2 pairs — `Record`
|
||||
// and `File` are both in ATTACHMENT_TARGET_LABELS — so `declares every pair
|
||||
// a phase/framework overlay can attach` already covers them, and repeating
|
||||
// them here would have asserted the opposite of what the comment claims.
|
||||
const nonBridge = [
|
||||
'CodeElement|CodeElement',
|
||||
'CodeElement|Module',
|
||||
'CodeElement|Property',
|
||||
'Module|CodeElement',
|
||||
'Module|Namespace',
|
||||
'Namespace|Function',
|
||||
];
|
||||
const allRulePairs = new Set([...scopeBridgePairs(), ...attachmentPairs()]);
|
||||
expect(nonBridge.filter((pair) => allRulePairs.has(pair))).toEqual([]);
|
||||
expect(nonBridge.filter((pair) => !structural.has(pair))).toEqual([]);
|
||||
});
|
||||
});
|
||||
|
|
@ -188,22 +188,31 @@ describe('LadybugDB Schema', () => {
|
|||
expect(RELATION_SCHEMA).toContain('step INT32');
|
||||
});
|
||||
|
||||
// These four go through `parseRelationSchemaPairs`, not a raw substring:
|
||||
// each of their pairs is now emitted by a cross product rather than
|
||||
// hand-written, and the generated half backticks EVERY label
|
||||
// (`FROM \`Function\` TO \`Function\``) while the hand-written half
|
||||
// backticks only multi-language names. Asserting the runtime's own parse
|
||||
// keeps them about the pair being declared, which is what LadybugDB
|
||||
// enforces — a cosmetic DDL formatting change cannot fail them.
|
||||
it('connects Function to Function (CALLS)', () => {
|
||||
expect(RELATION_SCHEMA).toContain('FROM Function TO Function');
|
||||
expect(parseRelationSchemaPairs(RELATION_SCHEMA).has('Function|Function')).toBe(true);
|
||||
});
|
||||
|
||||
it('connects File to Function (CONTAINS/DEFINES)', () => {
|
||||
expect(RELATION_SCHEMA).toContain('FROM File TO Function');
|
||||
expect(parseRelationSchemaPairs(RELATION_SCHEMA).has('File|Function')).toBe(true);
|
||||
});
|
||||
|
||||
it('connects symbols to Community (MEMBER_OF)', () => {
|
||||
expect(RELATION_SCHEMA).toContain('FROM Function TO Community');
|
||||
expect(RELATION_SCHEMA).toContain('FROM Class TO Community');
|
||||
const declaredPairs = parseRelationSchemaPairs(RELATION_SCHEMA);
|
||||
expect(declaredPairs.has('Function|Community')).toBe(true);
|
||||
expect(declaredPairs.has('Class|Community')).toBe(true);
|
||||
});
|
||||
|
||||
it('connects symbols to Process (STEP_IN_PROCESS)', () => {
|
||||
expect(RELATION_SCHEMA).toContain('FROM Function TO Process');
|
||||
expect(RELATION_SCHEMA).toContain('FROM Method TO Process');
|
||||
const declaredPairs = parseRelationSchemaPairs(RELATION_SCHEMA);
|
||||
expect(declaredPairs.has('Function|Process')).toBe(true);
|
||||
expect(declaredPairs.has('Method|Process')).toBe(true);
|
||||
});
|
||||
|
||||
it('connects BasicBlock to BasicBlock (taint/PDG substrate edges, #2080)', () => {
|
||||
|
|
|
|||
|
|
@ -0,0 +1,337 @@
|
|||
/**
|
||||
* #2788 — `resolveCppQualifiedNamespaceMember` serves qualified `ns::member()`
|
||||
* lookups from a per-pipeline index instead of rescanning every parsed file per
|
||||
* call site. These tests pin the properties the index must not lose:
|
||||
*
|
||||
* 1. Transitive inline-namespace collection — parent- and child-level member
|
||||
* visibility through one receiver — and same-name ambiguity (#1564): the
|
||||
* semantics the old linear scan provided.
|
||||
* 2. Cross-file accumulation — C++ namespaces are open, so one receiver's
|
||||
* members are spread over however many files reopen it. The legacy scan
|
||||
* re-derived its hits per call site and got this for free; the index has
|
||||
* to merge across the whole `parsedFiles` array to match it.
|
||||
* 3. Cache invalidation — a new `parsedFiles` array, or a
|
||||
* `clearCppInlineNamespaces()` between passes, must not serve stale hits.
|
||||
* This is the failure mode the index introduces; nothing else covers it.
|
||||
*/
|
||||
|
||||
import type {
|
||||
ParsedFile,
|
||||
ScopeId,
|
||||
ScopeResolutionIndexes,
|
||||
SymbolDefinition,
|
||||
} from 'gitnexus-shared';
|
||||
import { beforeEach, describe, expect, it } from 'vitest';
|
||||
import {
|
||||
clearCppInlineNamespaces,
|
||||
markCppInlineNamespaceRange,
|
||||
populateCppInlineNamespaceScopes,
|
||||
resolveCppQualifiedNamespaceMember,
|
||||
} from '../../../../src/core/ingestion/languages/cpp/inline-namespaces.js';
|
||||
|
||||
const NO_SCOPES = {} as unknown as ScopeResolutionIndexes;
|
||||
|
||||
interface ScopeSpec {
|
||||
readonly id: string;
|
||||
readonly parent: string | null;
|
||||
readonly defs: readonly SymbolDefinition[];
|
||||
/** Distinguishes each scope's range so inline marking targets exactly one. */
|
||||
readonly line: number;
|
||||
}
|
||||
|
||||
function def(nodeId: string, type: string, qualifiedName: string): SymbolDefinition {
|
||||
return { nodeId, type, qualifiedName } as unknown as SymbolDefinition;
|
||||
}
|
||||
|
||||
function nsDef(nodeId: string, qualifiedName: string): SymbolDefinition {
|
||||
return def(nodeId, 'Namespace', qualifiedName);
|
||||
}
|
||||
|
||||
function fnDef(nodeId: string, qualifiedName: string): SymbolDefinition {
|
||||
return def(nodeId, 'Function', qualifiedName);
|
||||
}
|
||||
|
||||
function ns(
|
||||
id: string,
|
||||
parent: string | null,
|
||||
defs: readonly SymbolDefinition[],
|
||||
line: number,
|
||||
): ScopeSpec {
|
||||
return { id, parent, defs, line };
|
||||
}
|
||||
|
||||
function range(line: number): {
|
||||
startLine: number;
|
||||
startCol: number;
|
||||
endLine: number;
|
||||
endCol: number;
|
||||
} {
|
||||
return { startLine: line, startCol: 0, endLine: line + 1, endCol: 0 };
|
||||
}
|
||||
|
||||
/** Build one `ParsedFile` from scope specs, marking the scopes named in
|
||||
* `inlineIds` as inline namespaces (capture-time range mark +
|
||||
* `populateOwners`-time scope-id resolution, same order as the pipeline).
|
||||
* Separate from {@link makeParsedFiles} so a test can compose a genuinely
|
||||
* MULTI-file `parsedFiles` array, which is the only way to exercise the
|
||||
* index's cross-file accumulation. */
|
||||
function makeParsedFile(
|
||||
filePath: string,
|
||||
specs: readonly ScopeSpec[],
|
||||
inlineIds: readonly string[],
|
||||
): ParsedFile {
|
||||
const parsed = {
|
||||
filePath,
|
||||
scopes: specs.map((s) => ({
|
||||
id: s.id as unknown as ScopeId,
|
||||
kind: 'Namespace',
|
||||
parent: s.parent as unknown as ScopeId | null,
|
||||
ownedDefs: s.defs,
|
||||
range: range(s.line),
|
||||
})),
|
||||
} as unknown as ParsedFile;
|
||||
markInline(parsed, specs, inlineIds);
|
||||
return parsed;
|
||||
}
|
||||
|
||||
/** Single-file `parsedFiles` array — the shape most of these tests need. */
|
||||
function makeParsedFiles(
|
||||
filePath: string,
|
||||
specs: readonly ScopeSpec[],
|
||||
inlineIds: readonly string[],
|
||||
): readonly ParsedFile[] {
|
||||
return [makeParsedFile(filePath, specs, inlineIds)];
|
||||
}
|
||||
|
||||
/** Capture-time inline marking + `populateOwners`-time scope-id resolution,
|
||||
* in the same order the pipeline runs them. Spec lookup is a Map, not
|
||||
* `Array.find`, so marking a deep chain stays linear in `specs`. */
|
||||
function markInline(
|
||||
parsed: ParsedFile,
|
||||
specs: readonly ScopeSpec[],
|
||||
inlineIds: readonly string[],
|
||||
): void {
|
||||
const byId = new Map(specs.map((s) => [s.id, s]));
|
||||
for (const id of inlineIds) {
|
||||
const spec = byId.get(id);
|
||||
if (spec === undefined) throw new Error(`inline scope ${id} must exist`);
|
||||
markCppInlineNamespaceRange(parsed.filePath, range(spec.line));
|
||||
}
|
||||
populateCppInlineNamespaceScopes(parsed);
|
||||
}
|
||||
|
||||
/** `namespace outer { <ownDefs> inline namespace v1 { <inlineDefs> } }` */
|
||||
function outerWithInlineChild(
|
||||
filePath: string,
|
||||
ownDefs: readonly SymbolDefinition[],
|
||||
inlineDefs: readonly SymbolDefinition[],
|
||||
): readonly ParsedFile[] {
|
||||
return makeParsedFiles(
|
||||
filePath,
|
||||
[
|
||||
ns('sc:outer', null, [nsDef('n:outer', 'outer'), ...ownDefs], 1),
|
||||
ns('sc:v1', 'sc:outer', [nsDef('n:v1', 'outer.v1'), ...inlineDefs], 10),
|
||||
],
|
||||
['sc:v1'],
|
||||
);
|
||||
}
|
||||
|
||||
/** `namespace n0 { inline namespace n1 { … inline namespace n<depth-1> {
|
||||
* void leaf(); } … } }` — one function, declared in the innermost namespace,
|
||||
* reachable by qualified lookup from every level above it. */
|
||||
function inlineChain(filePath: string, depth: number): readonly ParsedFile[] {
|
||||
const specs: ScopeSpec[] = [];
|
||||
for (let d = 0; d < depth; d++) {
|
||||
const self = nsDef(`n:ns${d}`, `n${d}`);
|
||||
specs.push(
|
||||
ns(
|
||||
`sc:n${d}`,
|
||||
d === 0 ? null : `sc:n${d - 1}`,
|
||||
d === depth - 1 ? [self, fnDef('n:leaf', `n${d}.leaf`)] : [self],
|
||||
d * 2 + 1,
|
||||
),
|
||||
);
|
||||
}
|
||||
// Every level below the outermost is `inline`, so the whole chain is one
|
||||
// transitively-visible run of namespaces.
|
||||
return makeParsedFiles(
|
||||
filePath,
|
||||
specs,
|
||||
specs.slice(1).map((s) => s.id),
|
||||
);
|
||||
}
|
||||
|
||||
describe('C++ qualified-namespace member index (#2788)', () => {
|
||||
beforeEach(() => {
|
||||
clearCppInlineNamespaces();
|
||||
});
|
||||
|
||||
it('resolves outer::foo through an inline-namespace child', () => {
|
||||
const files = outerWithInlineChild('a.cpp', [], [fnDef('n:foo@v1', 'outer.v1.foo')]);
|
||||
expect(resolveCppQualifiedNamespaceMember('outer', 'foo', files, NO_SCOPES)).toMatchObject({
|
||||
nodeId: 'n:foo@v1',
|
||||
});
|
||||
});
|
||||
|
||||
it('returns undefined for an unknown namespace or member', () => {
|
||||
const files = outerWithInlineChild('a.cpp', [], [fnDef('n:foo@v1', 'outer.v1.foo')]);
|
||||
expect(resolveCppQualifiedNamespaceMember('nope', 'foo', files, NO_SCOPES)).toBeUndefined();
|
||||
expect(resolveCppQualifiedNamespaceMember('outer', 'nope', files, NO_SCOPES)).toBeUndefined();
|
||||
});
|
||||
|
||||
it('does not descend into a non-inline nested namespace', () => {
|
||||
const files = makeParsedFiles(
|
||||
'a.cpp',
|
||||
[
|
||||
ns('sc:outer', null, [nsDef('n:outer', 'outer')], 1),
|
||||
ns(
|
||||
'sc:nested',
|
||||
'sc:outer',
|
||||
[nsDef('n:nested', 'outer.nested'), fnDef('n:foo@nested', 'outer.nested.foo')],
|
||||
10,
|
||||
),
|
||||
],
|
||||
[],
|
||||
);
|
||||
expect(resolveCppQualifiedNamespaceMember('outer', 'foo', files, NO_SCOPES)).toBeUndefined();
|
||||
expect(resolveCppQualifiedNamespaceMember('nested', 'foo', files, NO_SCOPES)).toMatchObject({
|
||||
nodeId: 'n:foo@nested',
|
||||
});
|
||||
});
|
||||
|
||||
it('reports same-name hits across two inline children as ambiguous (#1564)', () => {
|
||||
const files = makeParsedFiles(
|
||||
'a.cpp',
|
||||
[
|
||||
ns('sc:outer', null, [nsDef('n:outer', 'outer')], 1),
|
||||
ns('sc:v1', 'sc:outer', [nsDef('n:v1', 'outer.v1'), fnDef('n:foo@v1', 'outer.v1.foo')], 10),
|
||||
ns('sc:v2', 'sc:outer', [nsDef('n:v2', 'outer.v2'), fnDef('n:foo@v2', 'outer.v2.foo')], 20),
|
||||
],
|
||||
['sc:v1', 'sc:v2'],
|
||||
);
|
||||
expect(resolveCppQualifiedNamespaceMember('outer', 'foo', files, NO_SCOPES)).toBe('ambiguous');
|
||||
});
|
||||
|
||||
it('resolves through an inline namespace that reuses its parent’s name', () => {
|
||||
// Pins the `visited` dedup: `namespace ns { inline namespace ns { … } }`
|
||||
// registers both scopes under receiver `ns`, and without dedup the one
|
||||
// `foo` is collected twice and degrades to 'ambiguous', dropping the CALLS
|
||||
// edge. Why node identity is the right dedup key: see
|
||||
// `gatherQualifiedNsMember` in `inline-namespaces.ts`.
|
||||
const files = makeParsedFiles(
|
||||
'a.cpp',
|
||||
[
|
||||
ns('sc:ns@outer', null, [nsDef('n:ns@outer', 'ns')], 1),
|
||||
ns(
|
||||
'sc:ns@inner',
|
||||
'sc:ns@outer',
|
||||
[nsDef('n:ns@inner', 'ns.ns'), fnDef('n:foo', 'ns.ns.foo')],
|
||||
10,
|
||||
),
|
||||
],
|
||||
['sc:ns@inner'],
|
||||
);
|
||||
expect(resolveCppQualifiedNamespaceMember('ns', 'foo', files, NO_SCOPES)).toMatchObject({
|
||||
nodeId: 'n:foo',
|
||||
});
|
||||
});
|
||||
|
||||
it('finds both a namespace-owned member and its inline child member', () => {
|
||||
// Both levels are visible through the same receiver: `foo` is owned by
|
||||
// `outer` itself, `bar` only by its inline child `v1`. Distinct member
|
||||
// names per level on purpose — two same-named candidates with no call-site
|
||||
// info collapse to 'ambiguous', which would assert nothing about either.
|
||||
const files = outerWithInlineChild(
|
||||
'a.cpp',
|
||||
[fnDef('n:foo@outer', 'outer.foo')],
|
||||
[fnDef('n:bar@v1', 'outer.v1.bar')],
|
||||
);
|
||||
expect(resolveCppQualifiedNamespaceMember('outer', 'foo', files, NO_SCOPES)).toMatchObject({
|
||||
nodeId: 'n:foo@outer',
|
||||
});
|
||||
expect(resolveCppQualifiedNamespaceMember('outer', 'bar', files, NO_SCOPES)).toMatchObject({
|
||||
nodeId: 'n:bar@v1',
|
||||
});
|
||||
});
|
||||
|
||||
it('resolves through a 20,000-deep inline chain without exhausting the stack', () => {
|
||||
// Pins that a deep chain neither overflows the stack nor blows the heap —
|
||||
// the two failure modes a recursive build and an eager member table had.
|
||||
// Why an uncaught throw here aborts the whole `analyze`: see
|
||||
// `QualifiedNsMemberIndex` in `inline-namespaces.ts`. 20,000 is ~2.5x the
|
||||
// depth that overflowed in this same runner, for margin over per-platform
|
||||
// stack sizes.
|
||||
const files = inlineChain('deep.cpp', 20_000);
|
||||
// From the outermost namespace: the full chain is one transitive walk.
|
||||
expect(resolveCppQualifiedNamespaceMember('n0', 'leaf', files, NO_SCOPES)).toMatchObject({
|
||||
nodeId: 'n:leaf',
|
||||
});
|
||||
// Every inline namespace is a legal qualified receiver of its own, so
|
||||
// mid-chain and innermost receivers must resolve too.
|
||||
expect(resolveCppQualifiedNamespaceMember('n10000', 'leaf', files, NO_SCOPES)).toMatchObject({
|
||||
nodeId: 'n:leaf',
|
||||
});
|
||||
expect(resolveCppQualifiedNamespaceMember('n19999', 'leaf', files, NO_SCOPES)).toMatchObject({
|
||||
nodeId: 'n:leaf',
|
||||
});
|
||||
// A miss down the same chain stays a miss rather than becoming a throw.
|
||||
expect(resolveCppQualifiedNamespaceMember('n0', 'nosuch', files, NO_SCOPES)).toBeUndefined();
|
||||
});
|
||||
|
||||
it('merges one namespace’s members across every file that reopens it', () => {
|
||||
// C++ namespaces are open: `namespace outer { void a(); }` in one file and
|
||||
// `namespace outer { void b(); }` in another are the SAME namespace, and
|
||||
// `outer::a()` / `outer::b()` must both resolve. The legacy scan re-derived
|
||||
// its hits from all of `parsedFiles` per call site; the index accumulates
|
||||
// once, so the per-receiver state has to outlive the file loop. Distinct
|
||||
// scope ids per file, as the real pipeline mints them.
|
||||
const files: readonly ParsedFile[] = [
|
||||
makeParsedFile(
|
||||
'a.cpp',
|
||||
[ns('sc:outer@a', null, [nsDef('n:outer@a', 'outer'), fnDef('n:a', 'outer.a')], 1)],
|
||||
[],
|
||||
),
|
||||
makeParsedFile(
|
||||
'b.cpp',
|
||||
[ns('sc:outer@b', null, [nsDef('n:outer@b', 'outer'), fnDef('n:b', 'outer.b')], 1)],
|
||||
[],
|
||||
),
|
||||
];
|
||||
expect(resolveCppQualifiedNamespaceMember('outer', 'a', files, NO_SCOPES)).toMatchObject({
|
||||
nodeId: 'n:a',
|
||||
});
|
||||
expect(resolveCppQualifiedNamespaceMember('outer', 'b', files, NO_SCOPES)).toMatchObject({
|
||||
nodeId: 'n:b',
|
||||
});
|
||||
});
|
||||
|
||||
it('does not serve one parsedFiles array’s index to another', () => {
|
||||
const first = outerWithInlineChild('a.cpp', [], [fnDef('n:foo@a', 'outer.v1.foo')]);
|
||||
expect(resolveCppQualifiedNamespaceMember('outer', 'foo', first, NO_SCOPES)).toMatchObject({
|
||||
nodeId: 'n:foo@a',
|
||||
});
|
||||
const second = outerWithInlineChild('b.cpp', [], [fnDef('n:foo@b', 'outer.v1.foo')]);
|
||||
expect(resolveCppQualifiedNamespaceMember('outer', 'foo', second, NO_SCOPES)).toMatchObject({
|
||||
nodeId: 'n:foo@b',
|
||||
});
|
||||
});
|
||||
|
||||
it('rebuilds after clearCppInlineNamespaces even when parsedFiles is reused', () => {
|
||||
// Pass 1: `v1` is inline, so `outer::foo` reaches through it.
|
||||
const specs: readonly ScopeSpec[] = [
|
||||
ns('sc:outer', null, [nsDef('n:outer', 'outer')], 1),
|
||||
ns('sc:v1', 'sc:outer', [nsDef('n:v1', 'outer.v1'), fnDef('n:foo@v1', 'outer.v1.foo')], 10),
|
||||
];
|
||||
const files = makeParsedFiles('a.cpp', specs, ['sc:v1']);
|
||||
expect(resolveCppQualifiedNamespaceMember('outer', 'foo', files, NO_SCOPES)).toMatchObject({
|
||||
nodeId: 'n:foo@v1',
|
||||
});
|
||||
|
||||
// Pass 2: SAME `parsedFiles` reference (so identity alone would serve the
|
||||
// cached index), but `v1` is no longer inline. Without the index reset in
|
||||
// `clearCppInlineNamespaces` the stale pass-1 hit survives.
|
||||
clearCppInlineNamespaces();
|
||||
markInline(files[0], specs, []);
|
||||
expect(resolveCppQualifiedNamespaceMember('outer', 'foo', files, NO_SCOPES)).toBeUndefined();
|
||||
});
|
||||
});
|
||||
110
gitnexus/test/unit/stream-graph-emit-force-ordering.test.ts
Normal file
110
gitnexus/test/unit/stream-graph-emit-force-ordering.test.ts
Normal file
|
|
@ -0,0 +1,110 @@
|
|||
/**
|
||||
* Streamed structural emit must be resolved AFTER the guards that force a full
|
||||
* rebuild (#2680 / PR #2793).
|
||||
*
|
||||
* `resolveStreamGraphEmit` gates on `options.force`, which several freshness
|
||||
* guards rebind long after function entry — see the comment at the
|
||||
* `resolveStreamGraphEmit` call in `run-analyze.ts` for the full list.
|
||||
* Resolving at entry froze it `false` for every rebuild they trigger, so the
|
||||
* pipeline took the in-memory emit path exactly when the #2649 memory relief
|
||||
* matters most — and a schema bump makes EVERY existing index take that path on
|
||||
* its next `analyze`.
|
||||
*
|
||||
* The seam: mock `runPipelineFromRepo` so it records the `PipelineOptions` the
|
||||
* orchestrator actually built and then rejects. That asserts the real wiring
|
||||
* (`streamGraphEmit` + `graphEmitCsvDir` as handed to the pipeline) rather than
|
||||
* re-testing the pure resolver, which `stream-graph-emit-config.test.ts`
|
||||
* already covers. Everything after the pipeline call is out of scope, so the
|
||||
* mock's rejection is the intended end of the run.
|
||||
*/
|
||||
import { describe, it, expect, vi, afterEach } from 'vitest';
|
||||
import fsp from 'node:fs/promises';
|
||||
import path from 'node:path';
|
||||
|
||||
import {
|
||||
getStoragePaths,
|
||||
saveMeta,
|
||||
INCREMENTAL_SCHEMA_VERSION,
|
||||
} from '../../src/storage/repo-manager.js';
|
||||
import { createTempDir } from '../helpers/test-db.js';
|
||||
|
||||
type PipelineModule = typeof import('../../src/core/ingestion/pipeline.js');
|
||||
type CapturedPipelineOptions = NonNullable<Parameters<PipelineModule['runPipelineFromRepo']>[2]>;
|
||||
|
||||
/** Sentinel: the pipeline was reached, and the run ends there by design. */
|
||||
const PIPELINE_REACHED = 'stream-graph-emit-ordering: pipeline reached';
|
||||
|
||||
const captured = vi.hoisted(() => ({ options: [] as unknown[] }));
|
||||
|
||||
vi.mock('../../src/core/ingestion/pipeline.js', async (importOriginal) => {
|
||||
const actual = await importOriginal<PipelineModule>();
|
||||
return {
|
||||
...actual,
|
||||
runPipelineFromRepo: (
|
||||
_repoPath: string,
|
||||
_onProgress: unknown,
|
||||
options: unknown,
|
||||
): Promise<never> => {
|
||||
captured.options.push(options);
|
||||
return Promise.reject(new Error(PIPELINE_REACHED));
|
||||
},
|
||||
};
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
captured.options.length = 0;
|
||||
vi.unstubAllEnvs();
|
||||
});
|
||||
|
||||
describe('streamGraphEmit is resolved after the force-mutating freshness guards', () => {
|
||||
it('arms streaming for the rebuild an INCREMENTAL_SCHEMA_VERSION bump forces', async () => {
|
||||
// Pin the escape hatch ON so the assertion cannot be moved by ambient env.
|
||||
// Before the fix this changed nothing: `force` was still unset at the entry
|
||||
// read, and the `force !== true` short-circuit precedes the env lookup.
|
||||
vi.stubEnv('GITNEXUS_STREAM_GRAPH_EMIT', '1');
|
||||
|
||||
const tmpRepo = await createTempDir('gitnexus-stream-order-');
|
||||
const repoPath = tmpRepo.dbPath;
|
||||
try {
|
||||
const { metaPath } = getStoragePaths(repoPath);
|
||||
const metaDir = path.dirname(metaPath);
|
||||
await fsp.mkdir(metaDir, { recursive: true });
|
||||
// An index stamped by the PREVIOUS schema — what every already-indexed
|
||||
// repo looks like on its first analyze after the bump.
|
||||
await saveMeta(metaDir, {
|
||||
repoPath,
|
||||
lastCommit: '',
|
||||
indexedAt: new Date(0).toISOString(),
|
||||
schemaVersion: INCREMENTAL_SCHEMA_VERSION - 1,
|
||||
fileHashes: { 'src/a.ts': 'stale-hash' },
|
||||
});
|
||||
|
||||
const { runFullAnalysis } = await import('../../src/core/run-analyze.js');
|
||||
const logs: string[] = [];
|
||||
|
||||
// NOTE: no `force` from the caller — the rebuild is entirely guard-driven,
|
||||
// which is the whole point.
|
||||
await expect(
|
||||
runFullAnalysis(
|
||||
repoPath,
|
||||
{ skipAgentsMd: true },
|
||||
{ onProgress: () => {}, onLog: (m: string) => logs.push(m) },
|
||||
),
|
||||
).rejects.toThrow(PIPELINE_REACHED);
|
||||
|
||||
// The schema-version guard is what supplied `force` on this run.
|
||||
expect(logs.filter((m) => m.includes('index schema changed'))).toHaveLength(1);
|
||||
|
||||
expect(captured.options).toHaveLength(1);
|
||||
const pipelineOptions = captured.options[0] as CapturedPipelineOptions;
|
||||
// The regression: pre-fix this was `false` / `undefined`, and the run
|
||||
// built the whole relationship set in memory.
|
||||
expect(pipelineOptions).toMatchObject({ streamGraphEmit: true });
|
||||
// The paired CSV dir must be armed with it — the two are resolved from one
|
||||
// value precisely so they cannot disagree.
|
||||
expect(typeof pipelineOptions.graphEmitCsvDir).toBe('string');
|
||||
} finally {
|
||||
await tmpRepo.cleanup();
|
||||
}
|
||||
}, 120_000);
|
||||
});
|
||||
Loading…
Add table
Reference in a new issue