diff --git a/ARCHITECTURE.md b/ARCHITECTURE.md index 03e3730b3..138530d24 100644 --- a/ARCHITECTURE.md +++ b/ARCHITECTURE.md @@ -277,6 +277,12 @@ The solver is flow-insensitive but bounded: dependency-indexed work items rerun Property-key dispatch remains a separate conservative fallback. Its per-key fan-out cap is 32; capped keys synthesize no partial calls and are reported at warning level with language, skipped-key count, dropped key names (bounded), and cap; the count also travels in `RunScopeResolutionStats.propertyDispatchSkippedKeys`. +Interface-dispatch fan-out walks the subtype closure of the receiver's interface and is **generic-instantiation aware** (#2912): a call through `IValidator` must not reach an implementor of `IValidator`, which shares its declaration and therefore its subtype list. Each heritage clause's arguments reach resolution by one of three routes — read off the `@reference.inherits` anchor's own spelling where that anchor spans the whole base (most languages, no query change), through the `@reference.type-arguments` sub-tag where the anchor is the bare name and moving it would renumber inheritance edge ids (Rust `impl T for S`, Dart `extends`), or on a heritage MARKER payload for clauses that never become reference sites (Dart `implements`/`with`). Whichever pass emits the edge records the pair through one sink: `preEmitInheritanceEdges` for heritage clauses, `ScopeResolver.emitHeritageEdges` for the rest. + +The walk then carries a substitution: a subtype's own type parameters bind to the receiver's arguments, so `class Wrapper : IValidator` stays reachable from every instantiation while `class IntValidator : IValidator` is pruned from the `string` one. Receiver arguments come from the declared type (Case 4), a class-level field's declared type (Case 6), or — for a compound receiver such as `this._repo` — the spelling the compound fold typed that position from, reported back through `recordReceiverType` and accepted only when it names the class the fold returned. + +The filter prunes only on positive evidence: an unknown instantiation on either side, an argument list whose arity does not line up, a name that may be a type variable the language's captures never recorded, or an unresolved spelling whose simple name matches all keep the target. A type parameter of the declaration ENCLOSING either side is recognised as such and never compared — `void Run(IValidator v)` writes a receiver with no known instantiation, so it keeps the unfiltered fan-out. That recognition is what generic METHODS now carry `@declaration.type-parameters` for in C#, Java and Kotlin (TypeScript already did): without it an unbounded `T` grounds to nothing and a bounded one grounds to its BOUND, and both compare unequal to an implementor's concrete argument. Languages that capture neither type arguments nor type parameters therefore emit exactly the pre-#2912 fan-out. The fan-out cap (32, `GITNEXUS_MAX_INTERFACE_DISPATCH_FANOUT`) and its skipped-target reporting are unchanged and apply after filtering. Note the fan-out itself still fires only for a receiver whose folded type is an `Interface` symbol, so a Rust `Trait` or a Dart abstract `Class` receiver emits no secondary targets to filter in the first place. + Standalone (regex-based) providers such as COBOL participate via `ScopeResolver.scopeResolutionEdgeMode: 'callable-flow-only'`: `runScopeResolution` runs for them, but every ordinary emission path — heritage, interface implementations, receiver-bound, free-call fallback, reference/import edges, post-resolution hooks — is gated off, so their legacy phase (e.g. `cobolPhase`) remains the sole owner of structural edges and the callable solver's `CALLS` are purely additive. A callable-flow-only provider whose files emitted no callable facts exits early, before finalize, keeping the opt-in proportional to source scanning. ### Receiver chains and the drop census (#2766) diff --git a/gitnexus-shared/src/scope-resolution/reference-site.ts b/gitnexus-shared/src/scope-resolution/reference-site.ts index 6629dacd3..b559d32e3 100644 --- a/gitnexus-shared/src/scope-resolution/reference-site.ts +++ b/gitnexus-shared/src/scope-resolution/reference-site.ts @@ -82,6 +82,28 @@ export interface ReferenceSite { * otherwise, in which case resolution is unchanged. */ readonly rawQualifiedName?: string; + /** + * Top-level generic/template arguments the source wrote ON this reference — + * `class UserValidator : IValidator` yields `['string']` on the + * `inherits` site whose `name` is `IValidator`. + * + * `name` is the BASE name and stays that way: every lookup in resolution is + * keyed by it, and one declaration answers for every instantiation of itself. + * This records what the erasure threw away, so a consumer that needs the + * INSTANTIATION — receiver-bound interface dispatch, which must not fan a + * `IValidator` receiver out to an `IValidator` implementor + * (#2912) — can ask for it without re-parsing the source. + * + * Derived generically from the anchor capture's own text (see + * `collectReferenceSites`), so no language query change is needed: an emitter + * whose `@reference.inherits` anchor spans the whole base gets this for free, + * and one whose anchor is the bare name simply leaves it absent. + * + * ABSENT MEANS UNKNOWN, never "not generic" — the two are indistinguishable + * here, and only the first is safe to act on. Consumers must fail OPEN on + * absence (keep the target), matching `SymbolDefinition.typeParameters`. + */ + readonly typeArguments?: readonly string[]; /** Source-text range of this reference. */ readonly atRange: Range; /** diff --git a/gitnexus/bench/import-target/baselines.json b/gitnexus/bench/import-target/baselines.json index 1a158f5d8..4c8e75f9f 100644 --- a/gitnexus/bench/import-target/baselines.json +++ b/gitnexus/bench/import-target/baselines.json @@ -1,10 +1,11 @@ { - "_what": "Baselines for bench/import-target/measure.mjs — EVERY import-target resolver registered in SCOPE_RESOLVERS, on one shared corpus, plus csharp a second time WITH csproj configs. One entry per registered language and one more for the csproj arm, no registered language ungated — and that is ASSERTED rather than asserted-in-a-comment, which is also why no roster of language names is kept in this prose to go stale: measure.mjs derives its language list from a LANG_REGISTRY table and a --check inventory arm reconciles that table against SCOPE_RESOLVERS in both directions. A C/C++ #include is an import site for this purpose and is gated like every other registered language. csharp and csharp_csproj resolve the IDENTICAL file corpus (buildFiles aliases the two) and differ in exactly one thing: whether csharpConfigs is supplied. Without that second arm the csproj namespace-directory index ships unmeasured, because every C# import in the no-csproj arm returns before reaching it. C and C++ follow that same precedent for a different context — their HEADERS arrive through resolutionConfig rather than through allFilePaths, and augmentedFilePaths unions the two once per pass, so the corpus is split at newPass rather than pre-merged. The first nine were added as their own O(imports x files) scans were indexed away (#2877/#2878/#2879/#2880, #2872, #2901, #2902, #2908) and this is the forward guard on each; the other eight were ungated until now, and PR #2911 — JavaScript reaching suffixResolve with no index at all, 25972 us per import at 8000 files — is what that costs.", - "_fingerprint_note": "Per-language sha256 over every distinct fromFile|target -> resolved target. A change here is a BEHAVIOUR change: the resolver returned a different target set, and IMPORTS/CALLS edges moved. Explain it, never re-baseline to make CI green. For the languages these PRs changed, the pre-change implementations produce these same values on this corpus at both 400 and 1600 files — that is what makes the index hoist a performance change. The tie-break-level proof lives in test/unit/scope-resolution/import-target-index-parity.test.ts (verbatim copies of the pre-change code, diffed) for Kotlin in test/unit/scope-resolution/kotlin/kotlin-import-target-parity.test.ts, and for the four resolvers added there in test/unit/scope-resolution/{php,java,cobol}-import-target-parity.test.ts and test/unit/import-resolvers/csharp-csproj-parity.test.ts, and for JavaScript in test/unit/scope-resolution/javascript-import-target-parity.test.ts (a differential over 211200 old-vs-new pairs, PR #2911). The eight languages added last have no per-language parity harness against a pre-change implementation and do NOT need one: nothing about their resolution changed, so there is no before to diff against. Their fingerprints are pure forward guards, minted from the current implementations, and their adapter-boundary index reuse is covered for every registered language at once by test/unit/scope-resolution/import-target-index-reuse.contract.test.ts. NOTE for csharp_csproj: on this corpus the #2902 indexed leg (step 3 of resolveCSharpImportInternal) is reached by 2221 of the 3200 small-arm imports but answers null for every one of them — the 979 that resolve do so at step 2 — so this fingerprint pins that legs cost and its null answers, while its positive tie-breaks (unanchored substring, iteration order) are pinned by csharp-csproj-parity.test.ts. NOTE for kotlin, go, csharp and java: twenty fingerprints across these four languages were re-baselined in #2881, the one deliberate behaviour change any language in this file has had. It landed in two steps and the second is the reason the first is not a special case: Kotlin first, then the shared package-dir-index (go, java, csharp) and the csproj namespace index once the same rule was found live there. `getKotlinFileIndex` no longer requires a file's package directory to be the FIRST occurrence of that name in its own path, so the unique arm's `d % 7` nested slice (`mod{d}/src/main/kotlin/com/example/pkg{d}/inner/pkg{d}`) now belongs to package `pkg{d}` and its wildcard imports resolve: resolved 1100 -> 1153 small and deep, 4456 -> 4681 large. The collide arm needed a CORPUS edit alongside it, not just a new number — its `d % 7` slice deliberately imported `com.example.vendor{d}`, a package that exists nowhere, purely to mirror the unique arm's nested-slice MISS, so leaving it would have left collide at 1100 against small's 1153 and broken the same-workload invariant the arm is built on (that assertion is what caught it). It now uses the same `com.example.models.*` spelling as the rest of the arm, which is why its distinct_outcomes fell (2775 -> 2744, 11087 -> 10961): one shared target instead of one per d. The record-level evidence for the resolver change — 235 of 19968 records moved, 54 null -> resolved, 0 buckets losing a member — is in bench/kotlin-import-target/baselines.json `_provenance`. The kotlin heap_reading_bytes and heap_ceiling_bytes moved with it, together as `_heap_reading_note` requires: 48073096 -> 48200224 bytes_large (+127128, +0.264%), ceiling still exactly 1.5x. Small, and it is worth saying WHY it is small rather than reading the number as evidence that the change is cheap. `dirChildren` grows by one entry per component-suffix the old rule used to skip, and this arm can only see part of that: the heap corpus is built with HEAP_PAD 8, which prefixes every path with `d0/…/d7/`, so no path can begin with a suffix of its own directory and the leading-segment half of the old rule is structurally invisible here. What moves the reading is the `d % 7` nested slice alone. Read +0.264% as this arm's ceiling on the effect, not as the effect. GO NEEDED A CORPUS EDIT TO BE GATED AT ALL. Its nested slice was `src/pkg{d}/internal/pkg{d}`, repeating only the LAST segment, while a Go query addresses the whole package path `src/pkg{d}` — so the directory never even ended with the query and the first-occurrence rule was never reached. Every go arm sat unchanged through the resolver fix. `uniqueDir`/`collideDir` now repeat the shape at the granularity Go actually queries (`src/pkg{d}/internal/src/pkg{d}`, `svc{d}/internal/sub/svc{d}/internal`), which is what moved go from 979 to 1153 resolved and bumped `languages.go.heap.path_segments` 13 -> 14. The general lesson: a corpus that carries a shape the QUERY cannot express does not gate that shape. CSHARP AND JAVA HIT THE SAME COLLIDE-ARM TRAP AS KOTLIN. Both collide arms sent their `d % 7` slice to a namespace that exists nowhere (`App.Src{d}.Vendor`, `com.svc{d}.vendor`) purely to MIRROR the unique arm's nested-slice miss; once that miss became a hit, collide sat at 979/1100 against small's 1153 and the same-workload assertion failed. Both now use the same spelling as the rest of their arm. HEAP: no reading here moved for the resolver change. An earlier revision of this branch re-recorded `csharp_csproj` 73703384 -> 73116520 as a -0.79% effect of the step-2 filter; review measured base and branch three times each and got the same 73.10e6 on BOTH sides — the recorded 73703384 was simply not reproducible on this box, and re-recording it would have dropped that language's derived floor by 0.8% for no reason belonging to this change. Reverted. Everything else sat within +/-0.03%. Note that `_heap_reading_note`'s claim that these readings 'reproduce to the byte across processes on one box' did NOT hold on the box this was measured on: go, dart, ruby, python, php and cpp all wandered by a few hundred to a few thousand bytes between processes with no code change touching them. Treat sub-0.05% movement as jitter, not signal. HEAP, kotlin, second movement: 48200224 -> 42802456 (-11.20%), re-recorded with its ceiling. `getKotlinFileIndex` now compacts each `dirChildren` bucket as it freezes it. `addChild` mints a bucket as `[raw]` and pushes the rest, and V8 grows a backing store by `old + old/2 + 16`, so the second child takes a 1-slot store to 17: 61144 buckets, 52.9% of their slots empty, 88 B each. Same fix and same accounting as the python `byBasename` sentence above. Note what this means for the gate: a memory WIN of this size passes every arm — it is under the ceiling and over the 0.5x floor — so it is recorded because the convention says a reading and its ceiling move together, not because anything went red. kotlin now reads 40.82 MiB. The prose in measure.mjs calling it '45.85 MiB, the second-largest reading in this file' is corrected with it — and was already wrong on the ranking before this change, since csharp_csproj (69.73) and php (47.28) both read higher; kotlin was third. A measurement written into prose is not re-taken, which is the finding `_heap_bound_note` records about this very file. One further corpus edit, made in review and MEASURED rather than assumed: kotlin's collide layout repeated only the `models` leaf (`…/com/example/models/inner/models`) while a Kotlin query addresses the whole dotted path, so a full revert of the Kotlin guards left both collide fingerprints UNMOVED — the arm was blind to the rule it was re-baselined for. Deepening it to `…/models/inner/com/example/models` makes the revert move both, and those two fingerprints are the only ones that changed for it. The same deepening was applied to the java and kotlin UNIQUE arms and REVERTED: it moved ten more fingerprints, grew java's heap reading 43%, and bought nothing — progressive stripping lands those queries on the same file with or without the rule, so the control still failed only on go.", - "_shape_note": "files/imports/resolved/distinct_outcomes AND the fingerprint are asserted exactly, per scale. A fingerprint alone cannot tell a legitimate resolution change from a corpus quietly shrunk below the size at which the timing arms can see anything; conversely the counts alone cannot see a defect confined to one arm, because the arms differ only in path padding and directory layout and both of those are count-neutral by design. Two cross-arm assertions close the remaining hole: the deep and collide arms must resolve exactly what small resolves (they are the same workload), and each of their fingerprints must DIFFER from small's (they are not the same corpus). Without the second, setting DEEP_PAD to 0 — which deletes the entire depth arm — moves no asserted number and prints PASS; the same is true of a collideDir that forwards to uniqueDir. THE HEAP ARM IS ASSERTED THE SAME WAY, by the same loop, and was not before: files_small, files_large, path_segments and probe decide WHAT it measures, and every one of them was reported and compared to nothing. Swapping HEAP_PROBE_TARGET.csharp_csproj for a target matching no CSPROJ_CONFIGS rootNamespace skips the whole config loop, so the getFilesInDir and getInsensitive legs never run and the arm the header calls the witness that the read pattern IS the footprint quietly becomes a two-map arm — 73703384 -> 59921216 B, ratio 1.017 -> 1.011, ceiling and floor both still passing and --check still exiting 0. Setting HEAP_SMALL equal to HEAP_LARGE is the same hole from the other side: ratio goes to ~1.0 by construction and bytes_large never moves. bytes_small and bytes_large are deliberately NOT asserted for equality — heap_ceiling_bytes and the heap_reading_bytes floor bound them with ~50% either way, because heapUsed accounting moves across platforms and Node majors and an exact byte assertion would be a re-baseline per runner. THE CONTEXT ARM IS ASSERTED THE SAME WAY, by the same loop, and more strictly than either: target, with_context and without_context are exact strings with no tolerance at all, because the arm resolves one import over a three-file corpus and has no measurement noise to tolerate. A separate check requires the last two to DIFFER, for the same reason deep.fingerprint must differ from small.fingerprint — a probe on which both call shapes agree asserts one number twice. Both halves run through resolveOne, so what the arm gates is this bench threading run.ts's fifth argument, not the resolvers' behaviour.", - "_arms_note": "Five timing arms, one memory arm and one deterministic arm elsewhere, because none of them gates alone. scaling_ratio (t_large/t_small)/(1600/400) catches cost growing with FILE COUNT — the #2877-#2880, #2901, #2902 and #2908 regressions themselves; every one of those legs was Theta(files) per import, so a revert scores ~4 here by construction. depth_ratio (t_deep/t_small at a FIXED file count, ~6x the path components) catches cost growing with path DEPTH, which scaling_ratio divides out and structurally cannot see; buildSuffixIndex (C#, Ruby, PHP, Java) and Kotlin suffixByStem emit one entry per component, so they legitimately sit above 1.0 while Go, Dart and COBOL, whose indexes are depth-free, sit at ~1.0. csharp's depth_budget has now been retightened twice for the same reason, and the second time it did lock the win in. It was 5 against a then-measured 3.318; #2903 made buildSuffixIndex's dirMap lazy and it became 3.5 against 2.31, with the file stating plainly that 3.5 did NOT lock that win in because a revert to an eager dirMap scores 3.318 and passes. Extending the laziness to the two SUFFIX maps drops it again, to 1.438 (java likewise 2.214 -> 1.402), because the deep arm has ~6x the path components and an O(files x depth) build of a map the no-csproj leg never reads is exactly the cost that scales with depth. Both are now 2.2, which is this file's 1.5x convention against measurements whose own peak-to-peak over 4 runs is 1.04x and 1.07x — and 2.2 DOES lock it in: an eager rebuild scores 2.3+ and fails. The other fifteen depth budgets sit at 1.37-1.75x measured and are unchanged. collide_scaling_ratio is the same measurement on a SHARED-LEAF layout (svcN/internal, SrcN/Models, com/example/model in every service, a repeated mod0.dart/mod0.rb/Mod0.cpy basename) carrying an identical file, import and resolved count: the small/large/deep arms mint one directory name per index, so every index bucket in them holds exactly ONE entry (measured: max last-segment bucket 1 and max matching directories 1 for go and csharp at 400 and 1600 files; max basename bucket 1 for dart and ruby), and bucket cardinality is the only non-constant term the new indexes have. On the shared-leaf shape go, csharp, dart and java legitimately score 2.1-3.9 because the bucket grows with the file count BY CONSTRUCTION — this is a limit on the SCOPE of the \"independent of corpus size\" claim, not a regression (the indexed code is still faster there than the pre-change full scan); their collide budgets say so honestly instead of pretending 1.8. Ruby, Kotlin, PHP and COBOL answer from keyed maps and are collision-immune, so they keep the linear 1.8 budget and that immunity is the assertion. csharp_csproj is the one arm that runs the other way: its shared leaf collapses dirsByLastSegment to the single key Models, so the slash-free sweep (see CSPROJ_CONFIGS) is CHEAPER on the collide layout than on the unique one and its expensive scale arm is large, not collide_large. Its 1.8 collide budget is therefore the linear one, and the arm that carries its real cost is the unique one. The collide arm is also the only arm that reaches filesDirectlyInPkgDir's dirCount > 1 merge (go: 388 multi-directory calls at 400 files, up to 9 directories; 1517 at 1600 files, up to 34) and the only one that reaches COBOL's copybook-over-source tier tie-break, which needs one bookname to name two files. small_ms_ceiling and collide_ms_ceiling are ABSOLUTE (~4x the measured arm), because a constant-factor regression that grows both scale arms equally passes every ratio. The five arms added here use 4.2x, the middle of the 3.7-4.6x the original five already carry; the two COBOL arms use ~5x, the multiplier dart's sub-1 ms arm has always carried, because a fixed scheduler hiccup is a larger fraction of a smaller number — measured over 8 runs they sat at 0.25-0.37 ms and 0.18-0.30 ms, and the pre-#2908 two-scans-per-COPY implementation costs ~300 ms on the same arm, so 2.0 and 1.5 still separate fixed from broken by two orders of magnitude. NOISE, measured rather than assumed: depth_ratio divides two sub-3 ms numbers (Dart's are sub-1 ms) and is by far the noisiest arm here, so it set N for the whole file. fastest() is a min-of-N estimator, so N is the knob. Over 22 --check runs on an idle box, peak-to-peak: at N=5 go ran 0.757-1.748 (2.31x) and tripped its own 1.6 budget about 1 run in 20; at N=7 (the kotlin-import-target setting) Dart still ran 0.678-2.043 (3.01x) and tripped once; at N=15 (bench/cfg, bench/schema-pairs, bench/callable-value-flow) every language collapsed to a 1.13-1.26x swing with 22/22 passing. The budgets were NOT widened; the estimator was fixed instead, which is why the headroom above is real rather than granted. N IS NOW PER LANGUAGE, and that is a refinement of the same finding rather than a retreat from it. The overshoot of min-of-K against min-of-15 is a function of the CELL's absolute duration, not of the language: replayed against two independent runs' full sample sets, the worst overshoots at K=7 land on swift.small (0.43 ms, 31.8%) and dart.collide (1.5 ms, 37.6%), while every cell at or above 10 ms overshoots by at most 6.3%. So repsFor() keeps 15 while a language's cheapest arm is under 5 ms and otherwise spends ~150 ms per cell, floored at 7 — 15 for go, csharp, dart, kotlin, java, cobol, swift, rust, python, c and cpp (every language the flakiness above was ever about, cheapest arm 0.19-3.2 ms) and 7-8 for csharp_csproj, ruby, php, javascript, typescript and vue (cheapest arm 20-28 ms). Per LANGUAGE, not per cell, so all five arms of a language share one estimator and the four ratios stay comparisons of like with like. The replay passed all 85 cells on all five gates at 0.4-0.7 of budget and saved 12.8 s and 12.4 s of a 46 s run; min-of-7 also reads slightly HIGHER than min-of-15, so the ceilings get marginally more sensitive rather than less. Confirmed on 4 fresh runs with the adaptive estimator live: every small arm inside 1.12x peak-to-peak and every collide arm inside 1.07x, with the six 7-8 rep languages at 1.008-1.071 — no worse than the 11 that kept 15. The chosen N is reported per language as `reps`. heap_ceiling_bytes bounds the retained per-pass import index, the only arm here that can see memory: buildSuffixIndex emits maps at O(files x depth), the profile package-dir-index.ts cites #2649 to avoid for itself, and csharp, ruby, php and java all retained NOTHING across imports at BASE (C#'s no-csproj leg and PHP's and Java's every leg re-scanned the raw Set; Ruby rebuilt and discarded a suffix index per require). It is measured at 8000 and 32000 files at HEAP_PAD depth rather than at the timing arms' sizes, because the finding is an ABSOLUTE footprint at repository scale. THE ARM NOW READS WHAT THE LANGUAGE READS, and that change is the whole reason this file was re-baselined. Four of these arms used to call getWorkspaceFileIndex(set) directly and then read index.all.length, which asks no suffix question at all — harmless only while buildSuffixIndex built both maps eagerly. The moment they went lazy the direct call built NO map, csharp, ruby, php and java each reported 0 B at 32000 files, and 0 B is under every ceiling: --check printed PASS over four gates that had silently become ceilings over nothing, which is precisely the failure this file's own header warns about for rust and cobol. Every arm now resolves a real MISSING import through the real resolver (HEAP_PROBE_TARGET, asserted to miss), so the maps it forces are the maps production forces, and a resolver that starts asking a new question moves the number without anyone editing the bench. That makes the READ PATTERN the dominant term, and the eight numbers say so: java 34958600 B and csharp 29862200 B ask index.get and never getInsensitive; php 37579888 B asks getInsensitive and never get, plus its own first-proper-suffix map; ruby 41025360 B and javascript 26745296 B read get(s) || getInsensitive(s) and pay for both, the second DERIVED from the first; and csharp_csproj 73705944 B additionally asks getFilesInDir. csharp_csproj IS NOW GATED, reversing the earlier decision that it would be 'a ceiling on a duplicate': at +20.8% of the C# index it was one, and at 2.47x of it — same corpus, same getWorkspaceFileIndex, three maps instead of one — it is the witness that the read pattern is the footprint. The old RESIDUAL note is superseded by that number: a dirMap-sized addition is no longer +18%, and a consumer that asks all three questions blows csharp's ceiling by 1.64x rather than sliding under it. A SECOND MEASUREMENT BIAS was removed at the same time and it moved every figure here, so do not read these against the old ones as if only the read pattern changed. buildFiles mints paths with template literals, which V8 keeps as ropes; the first traversal that slices one flattens it, allocating the flat string and dropping the rope's pieces, so a build measured over an unflattened corpus reports the index MINUS that net release — 11% low, uniformly. bytes_small was read over a corpus a discarded warm-up pass had already flattened and bytes_large over a fresh one, so every ratio read ~0.85-0.89 for structures that are exactly linear in the file count. measureHeap now flattens each corpus before measuring it; all eight ratios read 0.998-1.017, and the warm-up pass is gone because with the corpus flat a language's first and second reads agree to within 0.3%. python's figure rises from 7624992 to 10362976 for this reason and not because anything regressed, and then to 10543152 (+1.7%) because #2913's nestedDirNames set is retained for the pass, and then FALLS to 6360936 (-39.7%) for a reason worth knowing: byBasename holds roughly one bucket per file, and building each with `[]` followed by `push` made V8 grow the backing store to its 16-slot minimum, so every single-file bucket retained 15 empty pointer slots. Constructing the one-element buckets directly (`set(base, [entry])`) is byte-identical in contents and 3.9 MiB smaller at 32000 paths — 37% of what this arm used to read was empty array slots — the ancestorsByDir memo itself is NOT in this reading, because python's probe target misses at the nested-name rejection and never reaches the walk, so this arm does not bound that memo; measured separately with a probe that does reach it, a 32000-file corpus with every file in its own 10-deep directory retains ~19 MB, which would clear this ceiling, so repointing python's heap probe at a walking spelling means re-recording the ceiling in the same change, and c is unchanged at 10018816 because its basename map does not slice paths. Its ceiling is 1.5x the measured arm, and the DIFFERENCE FROM THE 4x TIMING CONVENTION IS DELIBERATE — do not harmonise it back. 4x exists because runner contention dominates a wall-clock number; this one has essentially no measurement noise (across 4 runs the widest spread was 0.11% on python, 0.03% on csharp_csproj and 0.00% — identical to the byte — on ruby, php, java, javascript and c, and the same holds across separate processes), so 4x would throw away almost all of the gate's power and sail straight past the regression this arm exists to catch. 1.5x still tolerates ~50% of cross-platform and Node-version drift, far more than a Node major bump plausibly moves heapUsed accounting; it catches a duplicated index (+100%) or a second exactMap-sized suffix map (+~85%). heap_floor_fraction is the arm the 0 B incident proved was missing. A ceiling can only say 'not too big'; nothing said 'still measuring something', which is why four dead arms passed. The floor is 0.5 x each language's RECORDED READING (heap_reading_bytes), which is half the measured size and says so. It used to be 0.33 x the CEILING, described the same way — true only while every ceiling stayed at exactly 1.5x its reading, a convention this file states and nothing enforces, so re-tuning one ceiling upward would have loosened that language's floor by the same factor in the one direction a floor exists to watch. The two forms agree to within 0.8% for all eight today, so this is a correction of derivation, not of strength. It sits ~400x above the readings' own reproducibility and far below any collapse. A genuine 2x memory WIN trips it too, and that is intended: like a fingerprint move, it must be explained and re-baselined rather than absorbed. COBOL is left out for the opposite reason: its index is two Map, O(files) with no depth term, and at 32000 files its retained delta does not clear the noise of the measurement itself. heap_ratio_budget, the linear-growth check across the 4x file-count gap, is the orthogonal arm: it sees per-file and per-depth growth but not a constant factor. ---- THE EIGHT LANGUAGES ADDED LAST (swift, rust, python, javascript, typescript, vue, c, cpp) ---- They carry the SAME five arms and the same gates; what differs is which arm can actually fail for each, because each resolver has a different cost axis, and the budgets below say so instead of copying a number across. Every figure quoted is the MAXIMUM over 5 full runs on an idle box, and the peak-to-peak of every one of these arms stayed inside 1.10x over those runs — tighter than the 1.13-1.26x the original nine record, because none of these arms divides two sub-1 ms numbers the way dart depth_ratio does. depth_budget is ~1.5x measured throughout: swift 2.3 (1.487), rust 2.1 (1.377), javascript 2.1 (1.376), typescript 2.1 (1.381), vue 2.3 (1.563), c 3.0 (1.990), cpp 3.0 (1.999). PYTHON WAS 11 AGAINST 7.389 AND IS NOW 2.6 AGAINST 1.872, because #2913 fixed the resolver rather than the budget. Its INDEX was always depth-free; hasRepoCandidate and resolveAbsoluteFromFiles each rebuilt one ancestor prefix per directory component of the importer on EVERY import, and the index's own dirPrefixes build inserted one entry per component per file, so the resolver was quadratic in path depth where every other language here is linear or flat. The prefixes are a pure function of the importer's DIRECTORY, so they are now memoized per directory inside getPythonFileIndex (ancestorsByDir), the leading segment is rejected up front against a set of nested directory names, the module and package buckets are consulted before the walk rather than inside it, and the dirPrefixes build stops at the first ancestor already stored. All five fingerprints are byte-identical, so it is a hoist. The budget is 2.2, and BOTH numbers behind it were re-measured on a quiet box AFTER the context leg below started being measured, because that change moved the arm: the work it adds is depth-FLAT, so python's absolute cost more than doubled while depth_ratio FELL to 1.405-1.563 over 5 serial runs (peak-to-peak 1.11x). A budget carried over from before that change would have been slack against a smaller ratio. 2.2 is 1.41x the measured maximum, inside the 1.37-1.75x band the other fifteen sit in, and it LOCKS THE WIN IN: reverting the per-directory ancestor memo alone scores 2.524 and reverting the nested-name rejection alone scores 2.553, both measured under the current call shape, so each fails at 2.2 with 13% to spare. Do not read those two figures as the pre-#2913 cost — 7.239 was that, and the gap closed because the bare-import tier stopped walking at all (see below). The other two parts of the fix are not gated by this arm and are not meant to be: reverting the bucket prune or the dirPrefixes early break lands under any budget this arm's noise supports, so they are gated deterministically instead, by the prefix-parity and package-probe arms of test/unit/scope-resolution/python/python-importer-ancestors.test.ts and python-import-target-parity.test.ts, which go red on exactly those two mutations. A timing budget catches what it can measure; the counts catch the rest. THE BARE-IMPORT TIER (`import os`, single segment, no dot) was a separate O(depth) walk in import-resolvers/python.ts that this bench cannot see at all, because every python arm here spells its imports with a dot and returns at the `pathLike.includes('/')` guard before reaching it. It ran TWICE per `from x import y` — the package probe's recursion re-ran the whole tail on identical inputs — and is now one memoized chain plus an O(1) proof-of-absence against the index's basename buckets: 12/24/72 Set probes at depth 1/4/16 became a flat 2, and 11.615 us/import at 18 path components became 0.740. Gated by probe COUNT in test/unit/scope-resolution/python/python-import-probe-count.test.ts, not here. collide_scaling_budget splits three ways. Three languages scan a bucket that grows with the corpus and get their measured value x1.5: swift 4.9 (3.279 — its bucket is the module file list it RETURNS, and its collide arm is four modules instead of dirs of them so that bucket is fileCount/4, i.e. 100 files at 400 and 400 at 1600), c 3.8 (2.535) and cpp 4.0 (2.639, the same basename bucket its suffix fallback walks). Four answer from keyed maps and keep the linear 1.8 — python 1.097, javascript 1.083, typescript 1.053, vue 1.079 — and that immunity IS the assertion, exactly as for ruby, kotlin, php and cobol. RUST IS THE ONE ARM THAT WAS REDESIGNED RATHER THAN BUDGETED. It resolves by probing candidate paths with allFilePaths.has(...) and never searches, so its cost is O(path segments) and provably flat in the file count (1.095 scaling, 1.061 collide scaling): a shared-leaf collide arm for rust would have asserted nothing, which is worse than no arm. Its collide corpus is instead a deep module tree (src/l0/l1/l2/l3/l4/mod{d}) whose targets carry ~2x the :: segments, so the arm exercises the axis that CAN grow, its 1.8 budget asserts the flatness across file counts, and collide_ms_ceiling 19 bounds the absolute cost of the long-path probe. small_ms_ceiling and collide_ms_ceiling are ~4x measured as everywhere else: rust 10/19 (2.609/4.704), python 7/8 (1.76/1.929, retightened from 12/15 against 3.044/3.771 by #2913), javascript 85/89 (21.254/22.145), typescript 85/86 (21.250/21.464), vue 81/93 (20.164/23.227), c 7/11 (1.620/2.850), cpp 7/12 (1.581/3.009). Swift takes ~5x (2 against 0.421 and 4 against 0.821) — the multiplier dart and cobol already carry, because a fixed scheduler hiccup is a larger fraction of a sub-1 ms number. ONE CAVEAT ON THE THREE ts-FAMILY MS NUMBERS, stated because nothing else in this file would reveal it: resolveTsTarget carries a per-pass resolveCache keyed currentFile::importPath, which no other resolver here has, and ~10% of this corpus is repeat pairs. Their us/import is therefore a slight underestimate of a cold resolve. It is left in rather than defeated because it is what the real pipeline does, and it is identical across all three so the arms stay comparable. HEAP for the eight: rust, swift, typescript, vue, cpp and cobol are still NOT gated, all of them measured before being left out. rust builds no index on this hook (16 B at 8000 files, 0 B at 32000); swift holds one pointer per file-times-segment and mints no strings, reading 0.98 MB at 8000 files against 0.29 MB at 32000 — a 4x larger corpus reading 3x SMALLER, which is what a measurement below its own noise floor looks like, and the same reading cobol gives (0.54 MB then 0 B); typescript and vue duplicate javascript through the same builder over the same-shaped corpus, and cpp duplicates c (10021320 against 10016960, 0.04% apart). Those four duplications are the ONLY exclusions that still rest on 'it would be a duplicate', and they are duplicates of a builder AND of a read pattern, which is the pairing csharp_csproj failed once the read pattern started to matter — if any of the four ever diverges in what it ASKS the index, it earns an arm the same way csharp_csproj just did. All eight gated arms are read the same way now (retainedPassBytes, one real import), so unlike before they are directly comparable to one another. WALL CLOCK — ~33-35 s in report mode, down from ~46 s, and ~44-45 s for --check, which is essentially UNCHANGED from ~46 s. Only report mode got faster; do not read the pair as 46 -> 42. The breakdown is worth having before anyone trims it. Timing arms: go 2.02, csharp 1.09, csharp_csproj 3.22, dart 0.41, ruby 2.90, kotlin 0.85, php 3.46, java 1.57, cobol 0.09, swift 0.46, rust 0.85, python 1.22, javascript 3.23, typescript 2.72, vue 2.89, c 0.86, cpp 0.91 (28.7 s, from 39.8 s: repsFor() accounts for all of it, and every second of it comes from the six languages whose cheapest cell is 20-28 ms); heap arms 3.43 s for SEVENTEEN languages, from 2.06 s for eight (every registered language is measured now; the nine added cost 1.37 s, of which kotlin alone is 0.57 s — see _heap_bound_note), and 2.1 s came from 3.0 s for seven when flattening retired the warm-up pass; module load 3.9 s. --check pays one import that report mode does not: the inventory arm loads pipeline/registry.ts, which drags in every registered scope resolver and its providers. Measured in isolation with the bench's own static imports already resident, that import costs 6.3-6.5 s on one box and 9.3-10.0 s on another — i.e. it consumes almost the whole repsFor win, which is why --check did not get faster. It is loaded dynamically at the point of use rather than at the top of the file, so report mode does not pay it and both modes take their measurements in the same module state. IT WAS WEIGHED AND KEPT, on the number that decides it: the benchmarks job is not CI's critical path. On the last green run of main it took 9 m 23 s against 12 m 58 s for the sharded coverage job that gates the merge, so ~4 m 40 s of slack sits above this bench and those seconds buy zero merge latency. Moving the arm to a vitest file would move the registry load ONTO the critical path, and would weaken it as well: this reconciles LANG_REGISTRY's SupportedLanguages values, which are what the five dispatcher branches key off, whereas a test that cannot import measure.mjs can only reconcile this file's arm NAMES plus a hand-written rule for de-aliasing csharp_csproj. The contract test import-target-index-reuse.contract.test.ts already covers the ADAPTER-boundary contract for every registered resolver; this arm covers a different claim, that the BENCH covers the pipeline. The ts family is still the largest single block of the timing phase (8.8 s) — its cost is suffixResolve probing ~39 extensions per path part on a miss, which is the real resolver and cannot be tuned away from the bench side. IF IT HAS TO SHRINK, drop collide and collide_large for typescript and vue and nothing else: -3.9 s, and it is the only cut that removes near-duplicate work rather than coverage, because all three run the same resolveTsTarget over the same buildSuffixIndex and javascript keeps the collide arm that covers their shared collision axis. Do NOT reach for REPS_MAX: it is 15 because depth_ratio tripped its own budget about 1 run in 20 at 5 and once at 7, and lowering it would re-open that for the eleven languages whose cheapest cell is sub-5 ms — which is where every recorded trip happened. The six languages it was safe to lower have already been lowered, per language and from a measurement, by repsFor(). ---- THE FIFTH ARGUMENT (context) AND THE TWO ARMS IT MOVED ---- resolveOne now makes run.ts's five-argument call for the two hooks that declare a fifth parameter, so php and python time the legs behind it. Nothing else moved: the other fifteen arms are handed no context and build no ParsedFile[] at all, and over five runs their five ms numbers and four ratios sit exactly where they did. Both languages' ten fingerprints, resolved counts and distinct_outcomes are IDENTICAL — the leg AGREES with the cascade on this corpus, which is the whole reason the context arm had to be added rather than leaving the fingerprint to notice. PHP: small_ms 27.762 -> 35.125 (+26.5%) and collide_ms 29.407 -> 36.182 (+23.0%), which is filesByDirectory plus, on every import that resolves, a candidate gather over the resolved file's directory and a localDefs filter; the ms ceilings keep PHP's own 4.21x and 4.26x multipliers (117 -> 148, 125 -> 154). depth_ratio 1.144 -> 1.283 and the 1.9 budget is UNCHANGED, which makes it 1.48x measured rather than 1.66x: directoryAliases emits one entry per path segment, so filesByDirectory is O(files x depth) and the depth arm is the only one that can see it — that budget got TIGHTER relative to its measurement, not looser, and 1.48x sits inside the 1.37-1.75x band the other sixteen carry. Its heap reading rises 37576816 -> 49574008 (+31.9%) for the same structure, and the reading is the MEMO rather than the workspace it indexes: newPass allocates the ParsedFile objects before retainedPassBytes takes its baseline sample, so they sit outside the delta. PYTHON, WHOSE FIGURES ARE THE LEAST SETTLED THING IN THIS FILE AND ARE RECORDED IN TWO SNAPSHOTS BECAUSE OF IT. A named import is the only spelling that reads context.parsedFiles, and it costs up to three entries into the resolver per import (package probe, exports check, submodule probe) where the synthetic namespace spelling this arm used to pass costs one. Against the resolver as it stood when the call shape changed that read small_ms 1.76 -> 5.751 and collide_ms 1.929 -> 5.894, ~3.1x. Against the resolver a few commits later — which stopped re-running the whole tail after a null package probe, a double-probe this bench could not previously see because the namespace spelling never entered that branch — the same arms read 4.404 and 4.505. The ceilings are 18 and 19, chosen to clear BOTH: 4.09x and 4.22x of the current numbers, 3.13x and 3.22x of the higher ones, so neither state is red. Retighten toward 4x once that resolver settles. ITS DEPTH ARM WAS DILUTED AND THE BUDGET IS RETIGHTENED TO MATCH, which is the one thing here worth arguing about: the added work is depth-FLAT, so depth_ratio FALLS 1.872 -> 1.478 while the absolute cost more than doubles, and 2.6 against 1.478 would be 1.76x — far looser than the 1.39x #2913 chose deliberately to lock its own fix in. 2.1 restores that multiplier (1.42x). THE TWO MUTATION SCORES #2913 RECORDED (3.123 for reverting the per-directory memo, 2.734 for reverting the nested-name rejection) WERE TAKEN AGAINST THE OLD CALL SHAPE AND HAVE NOT BEEN RE-TAKEN. Modelled forward, with the depth-quadratic term reappearing in every resolver entry so its absolute contribution scales with the entry count, they land near 2.8 and 2.4 — both above 2.1, and the second BELOW 2.6, which is the arithmetic that decided the budget. Re-run the two mutations before trusting the lock-in claim above. python's heap reading is unchanged (10543152 recorded; 10529848-10544616 across eight runs) because its probe misses before the branch that reads parsedFiles — see _blind_spot for why no probe can reach that memo. Every figure in this section is the MAXIMUM over its snapshot's runs (five, then three), with peak-to-peak 1.031-1.058 on php and 1.019-1.081 on python, taken on a box that was NOT idle and with another change landing in python's resolver mid-measurement. Re-take them serially before merging.", - "_triage": "Every ratio and ms ceiling here is a TIMING signal — re-run on an idle machine before investigating; runner contention dominates. depth_ratio is the noisiest of them by a wide margin (it divides two sub-3 ms numbers, and Dart's are sub-1 ms): if exactly one arm fails and it is that one, suspect the machine first. N is 15 for every language whose cheapest arm is under 5 ms, rather than this bench's original 5, specifically to hold that arm's peak-to-peak swing under 1.26x — see _arms_note for the measured distributions and for why the six languages that drop to 7-8 are the ones where cell size makes it safe — so a depth_ratio failure that REPRODUCES is a real signal, not noise. Each language's chosen N is printed as `reps`; read it before blaming the estimator. The fingerprint, shape and heap arms are the opposite: deterministic (over 4 runs the heap arm's widest spread was 0.11% on python and 0.00% on java, javascript and c), a re-run never changes them, and they must never be wished away. TWO heap failures mean the arm STOPPED MEASURING rather than that memory grew, and both are deterministic: a heap floor failure says the probe no longer forces the index it used to (this is how four arms read 0 B when buildSuffixIndex went lazy, and 0 B passes every ceiling), and a `heap probe ... resolved` throw says a probe target that must MISS now hits, so the reading is a materialized answer and the legs past it were never reached. A heap BOUND failure is deterministic in the same way and means one specific thing: a language excluded from the budgeted tier has grown a structure, or started asking its index a question it did not ask when the exclusion was recorded — never a timing signal, never a re-run, and never fixed by raising the bound without saying what grew. The context arm is deterministic too, and a failure there means one specific thing rather than a range of them: run.ts's fifth argument is not reaching that resolver from this bench, or the leg behind it stopped running. Never a timing signal, never a re-run. TIGHTENED IN #2881, because the measurements they bound got faster and a budget left alone while its reading falls is a gate loosening without anyone deciding to. Each new value holds the headroom the old one expressed over the old reading, computed from `_measured` on both sides: kotlin depth 3.4 -> 2.8 (reading 2.219 -> 1.813), go depth 1.6 -> 1.4 (1.169 -> 0.999), csharp depth 2.2 -> 2.0 (1.438 -> 1.279), java depth 2.2 -> 2.1 (1.402 -> 1.354), kotlin collide_scaling 1.8 -> 1.65 (1.179 -> 1.081), go collide_scaling 5.5 -> 5.1 (3.763 -> 3.465). The ABSOLUTE ms ceilings were deliberately NOT tightened by the same reasoning: they carry runner-contention headroom rather than measurement headroom, and a ratio is runner-speed-invariant where a millisecond is not.", - "_floor": "Measured against the pre-change implementations on THIS corpus at 150/600 files: go 3.36, csharp 4.10, dart 3.32, ruby 3.87. The issues report 4.00 / 3.43 / 4.05 on their own corpora; those are DIFFERENT numbers from different repositories and are not reproduced here — what they and these share is that both independently land in the quadratic band, well clear of the ~1.0 a linear result gives. Note also that this floor was taken at 150/600 while the gate runs at 400/1600, so it is a lower bound on what the pre-change code would score today. Kotlin's own bench measured its pre-index floor at 3.737. The four resolvers added later were NOT re-floored on this corpus, and the reason is that they do not need to be: every one of their pre-change legs walked the whole file set per import (PHP one findIndex per path part per extension, Java one scan per stripped prefix, COBOL two full scans per COPY, C# csproj one normalizedFileList pass per import per matching config), so their scaling_ratio is ~4 by construction rather than by measurement. Their per-import costs were measured on their own issue corpora instead: PHP 96.40 ms -> 0.036 ms, Java 8.05 ms -> 0.62 ms, COBOL 3879 us -> 10.5 us, C# csproj 1103 us -> 7.6 us. The 1.8 budget sits well above the linear result and well below every one of those. The eight languages added last were NOT floored either, and for a different reason again: they are not fixes, so there is no pre-change implementation to floor against. Their scaling budgets are the global linear 1.8 and the point of the arms is to hold the current numbers (measured 1.01-1.13) rather than to separate a fix from a break. The one exception is javascript, which IS a fix and does have a floor: 6448.9 us per import at 2000 files and 25972.6 us at 8000 — 4.12x the per-import cost for 4x the files, i.e. O(imports x files) — against 28.5 / 27.4 us with the index PR #2911 gave it, and 25.0 / 27.0 us for TypeScript over the identical corpus.", + "_what": "Baselines for bench/import-target/measure.mjs \u2014 EVERY import-target resolver registered in SCOPE_RESOLVERS, on one shared corpus, plus csharp a second time WITH csproj configs. One entry per registered language and one more for the csproj arm, no registered language ungated \u2014 and that is ASSERTED rather than asserted-in-a-comment, which is also why no roster of language names is kept in this prose to go stale: measure.mjs derives its language list from a LANG_REGISTRY table and a --check inventory arm reconciles that table against SCOPE_RESOLVERS in both directions. A C/C++ #include is an import site for this purpose and is gated like every other registered language. csharp and csharp_csproj resolve the IDENTICAL file corpus (buildFiles aliases the two) and differ in exactly one thing: whether csharpConfigs is supplied. Without that second arm the csproj namespace-directory index ships unmeasured, because every C# import in the no-csproj arm returns before reaching it. C and C++ follow that same precedent for a different context \u2014 their HEADERS arrive through resolutionConfig rather than through allFilePaths, and augmentedFilePaths unions the two once per pass, so the corpus is split at newPass rather than pre-merged. The first nine were added as their own O(imports x files) scans were indexed away (#2877/#2878/#2879/#2880, #2872, #2901, #2902, #2908) and this is the forward guard on each; the other eight were ungated until now, and PR #2911 \u2014 JavaScript reaching suffixResolve with no index at all, 25972 us per import at 8000 files \u2014 is what that costs.", + "_fingerprint_note": "Per-language sha256 over every distinct fromFile|target -> resolved target. A change here is a BEHAVIOUR change: the resolver returned a different target set, and IMPORTS/CALLS edges moved. Explain it, never re-baseline to make CI green. For the languages these PRs changed, the pre-change implementations produce these same values on this corpus at both 400 and 1600 files \u2014 that is what makes the index hoist a performance change. The tie-break-level proof lives in test/unit/scope-resolution/import-target-index-parity.test.ts (verbatim copies of the pre-change code, diffed) for Kotlin in test/unit/scope-resolution/kotlin/kotlin-import-target-parity.test.ts, and for the four resolvers added there in test/unit/scope-resolution/{php,java,cobol}-import-target-parity.test.ts and test/unit/import-resolvers/csharp-csproj-parity.test.ts, and for JavaScript in test/unit/scope-resolution/javascript-import-target-parity.test.ts (a differential over 211200 old-vs-new pairs, PR #2911). The eight languages added last have no per-language parity harness against a pre-change implementation and do NOT need one: nothing about their resolution changed, so there is no before to diff against. Their fingerprints are pure forward guards, minted from the current implementations, and their adapter-boundary index reuse is covered for every registered language at once by test/unit/scope-resolution/import-target-index-reuse.contract.test.ts. NOTE for csharp_csproj: on this corpus the #2902 indexed leg (step 3 of resolveCSharpImportInternal) is reached by 2221 of the 3200 small-arm imports but answers null for every one of them \u2014 the 979 that resolve do so at step 2 \u2014 so this fingerprint pins that legs cost and its null answers, while its positive tie-breaks (unanchored substring, iteration order) are pinned by csharp-csproj-parity.test.ts. NOTE for kotlin, go, csharp and java: twenty fingerprints across these four languages were re-baselined in #2881, the one deliberate behaviour change any language in this file has had. It landed in two steps and the second is the reason the first is not a special case: Kotlin first, then the shared package-dir-index (go, java, csharp) and the csproj namespace index once the same rule was found live there. `getKotlinFileIndex` no longer requires a file's package directory to be the FIRST occurrence of that name in its own path, so the unique arm's `d % 7` nested slice (`mod{d}/src/main/kotlin/com/example/pkg{d}/inner/pkg{d}`) now belongs to package `pkg{d}` and its wildcard imports resolve: resolved 1100 -> 1153 small and deep, 4456 -> 4681 large. The collide arm needed a CORPUS edit alongside it, not just a new number \u2014 its `d % 7` slice deliberately imported `com.example.vendor{d}`, a package that exists nowhere, purely to mirror the unique arm's nested-slice MISS, so leaving it would have left collide at 1100 against small's 1153 and broken the same-workload invariant the arm is built on (that assertion is what caught it). It now uses the same `com.example.models.*` spelling as the rest of the arm, which is why its distinct_outcomes fell (2775 -> 2744, 11087 -> 10961): one shared target instead of one per d. The record-level evidence for the resolver change \u2014 235 of 19968 records moved, 54 null -> resolved, 0 buckets losing a member \u2014 is in bench/kotlin-import-target/baselines.json `_provenance`. The kotlin heap_reading_bytes and heap_ceiling_bytes moved with it, together as `_heap_reading_note` requires: 48073096 -> 48200224 bytes_large (+127128, +0.264%), ceiling still exactly 1.5x. Small, and it is worth saying WHY it is small rather than reading the number as evidence that the change is cheap. `dirChildren` grows by one entry per component-suffix the old rule used to skip, and this arm can only see part of that: the heap corpus is built with HEAP_PAD 8, which prefixes every path with `d0/\u2026/d7/`, so no path can begin with a suffix of its own directory and the leading-segment half of the old rule is structurally invisible here. What moves the reading is the `d % 7` nested slice alone. Read +0.264% as this arm's ceiling on the effect, not as the effect. GO NEEDED A CORPUS EDIT TO BE GATED AT ALL. Its nested slice was `src/pkg{d}/internal/pkg{d}`, repeating only the LAST segment, while a Go query addresses the whole package path `src/pkg{d}` \u2014 so the directory never even ended with the query and the first-occurrence rule was never reached. Every go arm sat unchanged through the resolver fix. `uniqueDir`/`collideDir` now repeat the shape at the granularity Go actually queries (`src/pkg{d}/internal/src/pkg{d}`, `svc{d}/internal/sub/svc{d}/internal`), which is what moved go from 979 to 1153 resolved and bumped `languages.go.heap.path_segments` 13 -> 14. The general lesson: a corpus that carries a shape the QUERY cannot express does not gate that shape. CSHARP AND JAVA HIT THE SAME COLLIDE-ARM TRAP AS KOTLIN. Both collide arms sent their `d % 7` slice to a namespace that exists nowhere (`App.Src{d}.Vendor`, `com.svc{d}.vendor`) purely to MIRROR the unique arm's nested-slice miss; once that miss became a hit, collide sat at 979/1100 against small's 1153 and the same-workload assertion failed. Both now use the same spelling as the rest of their arm. HEAP: no reading here moved for the resolver change. An earlier revision of this branch re-recorded `csharp_csproj` 73703384 -> 73116520 as a -0.79% effect of the step-2 filter; review measured base and branch three times each and got the same 73.10e6 on BOTH sides \u2014 the recorded 73703384 was simply not reproducible on this box, and re-recording it would have dropped that language's derived floor by 0.8% for no reason belonging to this change. Reverted. Everything else sat within +/-0.03%. Note that `_heap_reading_note`'s claim that these readings 'reproduce to the byte across processes on one box' did NOT hold on the box this was measured on: go, dart, ruby, python, php and cpp all wandered by a few hundred to a few thousand bytes between processes with no code change touching them. Treat sub-0.05% movement as jitter, not signal. HEAP, kotlin, second movement: 48200224 -> 42802456 (-11.20%), re-recorded with its ceiling. `getKotlinFileIndex` now compacts each `dirChildren` bucket as it freezes it. `addChild` mints a bucket as `[raw]` and pushes the rest, and V8 grows a backing store by `old + old/2 + 16`, so the second child takes a 1-slot store to 17: 61144 buckets, 52.9% of their slots empty, 88 B each. Same fix and same accounting as the python `byBasename` sentence above. Note what this means for the gate: a memory WIN of this size passes every arm \u2014 it is under the ceiling and over the 0.5x floor \u2014 so it is recorded because the convention says a reading and its ceiling move together, not because anything went red. kotlin now reads 40.82 MiB. The prose in measure.mjs calling it '45.85 MiB, the second-largest reading in this file' is corrected with it \u2014 and was already wrong on the ranking before this change, since csharp_csproj (69.73) and php (47.28) both read higher; kotlin was third. A measurement written into prose is not re-taken, which is the finding `_heap_bound_note` records about this very file. One further corpus edit, made in review and MEASURED rather than assumed: kotlin's collide layout repeated only the `models` leaf (`\u2026/com/example/models/inner/models`) while a Kotlin query addresses the whole dotted path, so a full revert of the Kotlin guards left both collide fingerprints UNMOVED \u2014 the arm was blind to the rule it was re-baselined for. Deepening it to `\u2026/models/inner/com/example/models` makes the revert move both, and those two fingerprints are the only ones that changed for it. The same deepening was applied to the java and kotlin UNIQUE arms and REVERTED: it moved ten more fingerprints, grew java's heap reading 43%, and bought nothing \u2014 progressive stripping lands those queries on the same file with or without the rule, so the control still failed only on go.", + "_shape_note": "files/imports/resolved/distinct_outcomes AND the fingerprint are asserted exactly, per scale. A fingerprint alone cannot tell a legitimate resolution change from a corpus quietly shrunk below the size at which the timing arms can see anything; conversely the counts alone cannot see a defect confined to one arm, because the arms differ only in path padding and directory layout and both of those are count-neutral by design. Two cross-arm assertions close the remaining hole: the deep and collide arms must resolve exactly what small resolves (they are the same workload), and each of their fingerprints must DIFFER from small's (they are not the same corpus). Without the second, setting DEEP_PAD to 0 \u2014 which deletes the entire depth arm \u2014 moves no asserted number and prints PASS; the same is true of a collideDir that forwards to uniqueDir. THE HEAP ARM IS ASSERTED THE SAME WAY, by the same loop, and was not before: files_small, files_large, path_segments and probe decide WHAT it measures, and every one of them was reported and compared to nothing. Swapping HEAP_PROBE_TARGET.csharp_csproj for a target matching no CSPROJ_CONFIGS rootNamespace skips the whole config loop, so the getFilesInDir and getInsensitive legs never run and the arm the header calls the witness that the read pattern IS the footprint quietly becomes a two-map arm \u2014 73703384 -> 59921216 B, ratio 1.017 -> 1.011, ceiling and floor both still passing and --check still exiting 0. Setting HEAP_SMALL equal to HEAP_LARGE is the same hole from the other side: ratio goes to ~1.0 by construction and bytes_large never moves. bytes_small and bytes_large are deliberately NOT asserted for equality \u2014 heap_ceiling_bytes and the heap_reading_bytes floor bound them with ~50% either way, because heapUsed accounting moves across platforms and Node majors and an exact byte assertion would be a re-baseline per runner. THE CONTEXT ARM IS ASSERTED THE SAME WAY, by the same loop, and more strictly than either: target, with_context and without_context are exact strings with no tolerance at all, because the arm resolves one import over a three-file corpus and has no measurement noise to tolerate. A separate check requires the last two to DIFFER, for the same reason deep.fingerprint must differ from small.fingerprint \u2014 a probe on which both call shapes agree asserts one number twice. Both halves run through resolveOne, so what the arm gates is this bench threading run.ts's fifth argument, not the resolvers' behaviour.", + "_arms_note": "Five timing arms, one memory arm and one deterministic arm elsewhere, because none of them gates alone. scaling_ratio (t_large/t_small)/(1600/400) catches cost growing with FILE COUNT \u2014 the #2877-#2880, #2901, #2902 and #2908 regressions themselves; every one of those legs was Theta(files) per import, so a revert scores ~4 here by construction. depth_ratio (t_deep/t_small at a FIXED file count, ~6x the path components) catches cost growing with path DEPTH, which scaling_ratio divides out and structurally cannot see; buildSuffixIndex (C#, Ruby, PHP, Java) and Kotlin suffixByStem emit one entry per component, so they legitimately sit above 1.0 while Go, Dart and COBOL, whose indexes are depth-free, sit at ~1.0. csharp's depth_budget has now been retightened twice for the same reason, and the second time it did lock the win in. It was 5 against a then-measured 3.318; #2903 made buildSuffixIndex's dirMap lazy and it became 3.5 against 2.31, with the file stating plainly that 3.5 did NOT lock that win in because a revert to an eager dirMap scores 3.318 and passes. Extending the laziness to the two SUFFIX maps drops it again, to 1.438 (java likewise 2.214 -> 1.402), because the deep arm has ~6x the path components and an O(files x depth) build of a map the no-csproj leg never reads is exactly the cost that scales with depth. Both are now 2.2, which is this file's 1.5x convention against measurements whose own peak-to-peak over 4 runs is 1.04x and 1.07x \u2014 and 2.2 DOES lock it in: an eager rebuild scores 2.3+ and fails. The other fifteen depth budgets sit at 1.37-1.75x measured and are unchanged. collide_scaling_ratio is the same measurement on a SHARED-LEAF layout (svcN/internal, SrcN/Models, com/example/model in every service, a repeated mod0.dart/mod0.rb/Mod0.cpy basename) carrying an identical file, import and resolved count: the small/large/deep arms mint one directory name per index, so every index bucket in them holds exactly ONE entry (measured: max last-segment bucket 1 and max matching directories 1 for go and csharp at 400 and 1600 files; max basename bucket 1 for dart and ruby), and bucket cardinality is the only non-constant term the new indexes have. On the shared-leaf shape go, csharp, dart and java legitimately score 2.1-3.9 because the bucket grows with the file count BY CONSTRUCTION \u2014 this is a limit on the SCOPE of the \"independent of corpus size\" claim, not a regression (the indexed code is still faster there than the pre-change full scan); their collide budgets say so honestly instead of pretending 1.8. Ruby, Kotlin, PHP and COBOL answer from keyed maps and are collision-immune, so they keep the linear 1.8 budget and that immunity is the assertion. csharp_csproj is the one arm that runs the other way: its shared leaf collapses dirsByLastSegment to the single key Models, so the slash-free sweep (see CSPROJ_CONFIGS) is CHEAPER on the collide layout than on the unique one and its expensive scale arm is large, not collide_large. Its 1.8 collide budget is therefore the linear one, and the arm that carries its real cost is the unique one. The collide arm is also the only arm that reaches filesDirectlyInPkgDir's dirCount > 1 merge (go: 388 multi-directory calls at 400 files, up to 9 directories; 1517 at 1600 files, up to 34) and the only one that reaches COBOL's copybook-over-source tier tie-break, which needs one bookname to name two files. small_ms_ceiling and collide_ms_ceiling are ABSOLUTE (~4x the measured arm), because a constant-factor regression that grows both scale arms equally passes every ratio. The five arms added here use 4.2x, the middle of the 3.7-4.6x the original five already carry; the two COBOL arms use ~5x, the multiplier dart's sub-1 ms arm has always carried, because a fixed scheduler hiccup is a larger fraction of a smaller number \u2014 measured over 8 runs they sat at 0.25-0.37 ms and 0.18-0.30 ms, and the pre-#2908 two-scans-per-COPY implementation costs ~300 ms on the same arm, so 2.0 and 1.5 still separate fixed from broken by two orders of magnitude. NOISE, measured rather than assumed: depth_ratio divides two sub-3 ms numbers (Dart's are sub-1 ms) and is by far the noisiest arm here, so it set N for the whole file. fastest() is a min-of-N estimator, so N is the knob. Over 22 --check runs on an idle box, peak-to-peak: at N=5 go ran 0.757-1.748 (2.31x) and tripped its own 1.6 budget about 1 run in 20; at N=7 (the kotlin-import-target setting) Dart still ran 0.678-2.043 (3.01x) and tripped once; at N=15 (bench/cfg, bench/schema-pairs, bench/callable-value-flow) every language collapsed to a 1.13-1.26x swing with 22/22 passing. The budgets were NOT widened; the estimator was fixed instead, which is why the headroom above is real rather than granted. N IS NOW PER LANGUAGE, and that is a refinement of the same finding rather than a retreat from it. The overshoot of min-of-K against min-of-15 is a function of the CELL's absolute duration, not of the language: replayed against two independent runs' full sample sets, the worst overshoots at K=7 land on swift.small (0.43 ms, 31.8%) and dart.collide (1.5 ms, 37.6%), while every cell at or above 10 ms overshoots by at most 6.3%. So repsFor() keeps 15 while a language's cheapest arm is under 5 ms and otherwise spends ~150 ms per cell, floored at 7 \u2014 15 for go, csharp, dart, kotlin, java, cobol, swift, rust, python, c and cpp (every language the flakiness above was ever about, cheapest arm 0.19-3.2 ms) and 7-8 for csharp_csproj, ruby, php, javascript, typescript and vue (cheapest arm 20-28 ms). Per LANGUAGE, not per cell, so all five arms of a language share one estimator and the four ratios stay comparisons of like with like. The replay passed all 85 cells on all five gates at 0.4-0.7 of budget and saved 12.8 s and 12.4 s of a 46 s run; min-of-7 also reads slightly HIGHER than min-of-15, so the ceilings get marginally more sensitive rather than less. Confirmed on 4 fresh runs with the adaptive estimator live: every small arm inside 1.12x peak-to-peak and every collide arm inside 1.07x, with the six 7-8 rep languages at 1.008-1.071 \u2014 no worse than the 11 that kept 15. The chosen N is reported per language as `reps`. heap_ceiling_bytes bounds the retained per-pass import index, the only arm here that can see memory: buildSuffixIndex emits maps at O(files x depth), the profile package-dir-index.ts cites #2649 to avoid for itself, and csharp, ruby, php and java all retained NOTHING across imports at BASE (C#'s no-csproj leg and PHP's and Java's every leg re-scanned the raw Set; Ruby rebuilt and discarded a suffix index per require). It is measured at 8000 and 32000 files at HEAP_PAD depth rather than at the timing arms' sizes, because the finding is an ABSOLUTE footprint at repository scale. THE ARM NOW READS WHAT THE LANGUAGE READS, and that change is the whole reason this file was re-baselined. Four of these arms used to call getWorkspaceFileIndex(set) directly and then read index.all.length, which asks no suffix question at all \u2014 harmless only while buildSuffixIndex built both maps eagerly. The moment they went lazy the direct call built NO map, csharp, ruby, php and java each reported 0 B at 32000 files, and 0 B is under every ceiling: --check printed PASS over four gates that had silently become ceilings over nothing, which is precisely the failure this file's own header warns about for rust and cobol. Every arm now resolves a real MISSING import through the real resolver (HEAP_PROBE_TARGET, asserted to miss), so the maps it forces are the maps production forces, and a resolver that starts asking a new question moves the number without anyone editing the bench. That makes the READ PATTERN the dominant term, and the eight numbers say so: java 34958600 B and csharp 29862200 B ask index.get and never getInsensitive; php 37579888 B asks getInsensitive and never get, plus its own first-proper-suffix map; ruby 41025360 B and javascript 26745296 B read get(s) || getInsensitive(s) and pay for both, the second DERIVED from the first; and csharp_csproj 73705944 B additionally asks getFilesInDir. csharp_csproj IS NOW GATED, reversing the earlier decision that it would be 'a ceiling on a duplicate': at +20.8% of the C# index it was one, and at 2.47x of it \u2014 same corpus, same getWorkspaceFileIndex, three maps instead of one \u2014 it is the witness that the read pattern is the footprint. The old RESIDUAL note is superseded by that number: a dirMap-sized addition is no longer +18%, and a consumer that asks all three questions blows csharp's ceiling by 1.64x rather than sliding under it. A SECOND MEASUREMENT BIAS was removed at the same time and it moved every figure here, so do not read these against the old ones as if only the read pattern changed. buildFiles mints paths with template literals, which V8 keeps as ropes; the first traversal that slices one flattens it, allocating the flat string and dropping the rope's pieces, so a build measured over an unflattened corpus reports the index MINUS that net release \u2014 11% low, uniformly. bytes_small was read over a corpus a discarded warm-up pass had already flattened and bytes_large over a fresh one, so every ratio read ~0.85-0.89 for structures that are exactly linear in the file count. measureHeap now flattens each corpus before measuring it; all eight ratios read 0.998-1.017, and the warm-up pass is gone because with the corpus flat a language's first and second reads agree to within 0.3%. python's figure rises from 7624992 to 10362976 for this reason and not because anything regressed, and then to 10543152 (+1.7%) because #2913's nestedDirNames set is retained for the pass, and then FALLS to 6360936 (-39.7%) for a reason worth knowing: byBasename holds roughly one bucket per file, and building each with `[]` followed by `push` made V8 grow the backing store to its 16-slot minimum, so every single-file bucket retained 15 empty pointer slots. Constructing the one-element buckets directly (`set(base, [entry])`) is byte-identical in contents and 3.9 MiB smaller at 32000 paths \u2014 37% of what this arm used to read was empty array slots \u2014 the ancestorsByDir memo itself is NOT in this reading, because python's probe target misses at the nested-name rejection and never reaches the walk, so this arm does not bound that memo; measured separately with a probe that does reach it, a 32000-file corpus with every file in its own 10-deep directory retains ~19 MB, which would clear this ceiling, so repointing python's heap probe at a walking spelling means re-recording the ceiling in the same change, and c is unchanged at 10018816 because its basename map does not slice paths. Its ceiling is 1.5x the measured arm, and the DIFFERENCE FROM THE 4x TIMING CONVENTION IS DELIBERATE \u2014 do not harmonise it back. 4x exists because runner contention dominates a wall-clock number; this one has essentially no measurement noise (across 4 runs the widest spread was 0.11% on python, 0.03% on csharp_csproj and 0.00% \u2014 identical to the byte \u2014 on ruby, php, java, javascript and c, and the same holds across separate processes), so 4x would throw away almost all of the gate's power and sail straight past the regression this arm exists to catch. 1.5x still tolerates ~50% of cross-platform and Node-version drift, far more than a Node major bump plausibly moves heapUsed accounting; it catches a duplicated index (+100%) or a second exactMap-sized suffix map (+~85%). heap_floor_fraction is the arm the 0 B incident proved was missing. A ceiling can only say 'not too big'; nothing said 'still measuring something', which is why four dead arms passed. The floor is 0.5 x each language's RECORDED READING (heap_reading_bytes), which is half the measured size and says so. It used to be 0.33 x the CEILING, described the same way \u2014 true only while every ceiling stayed at exactly 1.5x its reading, a convention this file states and nothing enforces, so re-tuning one ceiling upward would have loosened that language's floor by the same factor in the one direction a floor exists to watch. The two forms agree to within 0.8% for all eight today, so this is a correction of derivation, not of strength. It sits ~400x above the readings' own reproducibility and far below any collapse. A genuine 2x memory WIN trips it too, and that is intended: like a fingerprint move, it must be explained and re-baselined rather than absorbed. COBOL is left out for the opposite reason: its index is two Map, O(files) with no depth term, and at 32000 files its retained delta does not clear the noise of the measurement itself. heap_ratio_budget, the linear-growth check across the 4x file-count gap, is the orthogonal arm: it sees per-file and per-depth growth but not a constant factor. ---- THE EIGHT LANGUAGES ADDED LAST (swift, rust, python, javascript, typescript, vue, c, cpp) ---- They carry the SAME five arms and the same gates; what differs is which arm can actually fail for each, because each resolver has a different cost axis, and the budgets below say so instead of copying a number across. Every figure quoted is the MAXIMUM over 5 full runs on an idle box, and the peak-to-peak of every one of these arms stayed inside 1.10x over those runs \u2014 tighter than the 1.13-1.26x the original nine record, because none of these arms divides two sub-1 ms numbers the way dart depth_ratio does. depth_budget is ~1.5x measured throughout: swift 2.3 (1.487), rust 2.1 (1.377), javascript 2.1 (1.376), typescript 2.1 (1.381), vue 2.3 (1.563), c 3.0 (1.990), cpp 3.0 (1.999). PYTHON WAS 11 AGAINST 7.389 AND IS NOW 2.6 AGAINST 1.872, because #2913 fixed the resolver rather than the budget. Its INDEX was always depth-free; hasRepoCandidate and resolveAbsoluteFromFiles each rebuilt one ancestor prefix per directory component of the importer on EVERY import, and the index's own dirPrefixes build inserted one entry per component per file, so the resolver was quadratic in path depth where every other language here is linear or flat. The prefixes are a pure function of the importer's DIRECTORY, so they are now memoized per directory inside getPythonFileIndex (ancestorsByDir), the leading segment is rejected up front against a set of nested directory names, the module and package buckets are consulted before the walk rather than inside it, and the dirPrefixes build stops at the first ancestor already stored. All five fingerprints are byte-identical, so it is a hoist. The budget is 2.2, and BOTH numbers behind it were re-measured on a quiet box AFTER the context leg below started being measured, because that change moved the arm: the work it adds is depth-FLAT, so python's absolute cost more than doubled while depth_ratio FELL to 1.405-1.563 over 5 serial runs (peak-to-peak 1.11x). A budget carried over from before that change would have been slack against a smaller ratio. 2.2 is 1.41x the measured maximum, inside the 1.37-1.75x band the other fifteen sit in, and it LOCKS THE WIN IN: reverting the per-directory ancestor memo alone scores 2.524 and reverting the nested-name rejection alone scores 2.553, both measured under the current call shape, so each fails at 2.2 with 13% to spare. Do not read those two figures as the pre-#2913 cost \u2014 7.239 was that, and the gap closed because the bare-import tier stopped walking at all (see below). The other two parts of the fix are not gated by this arm and are not meant to be: reverting the bucket prune or the dirPrefixes early break lands under any budget this arm's noise supports, so they are gated deterministically instead, by the prefix-parity and package-probe arms of test/unit/scope-resolution/python/python-importer-ancestors.test.ts and python-import-target-parity.test.ts, which go red on exactly those two mutations. A timing budget catches what it can measure; the counts catch the rest. THE BARE-IMPORT TIER (`import os`, single segment, no dot) was a separate O(depth) walk in import-resolvers/python.ts that this bench cannot see at all, because every python arm here spells its imports with a dot and returns at the `pathLike.includes('/')` guard before reaching it. It ran TWICE per `from x import y` \u2014 the package probe's recursion re-ran the whole tail on identical inputs \u2014 and is now one memoized chain plus an O(1) proof-of-absence against the index's basename buckets: 12/24/72 Set probes at depth 1/4/16 became a flat 2, and 11.615 us/import at 18 path components became 0.740. Gated by probe COUNT in test/unit/scope-resolution/python/python-import-probe-count.test.ts, not here. collide_scaling_budget splits three ways. Three languages scan a bucket that grows with the corpus and get their measured value x1.5: swift 4.9 (3.279 \u2014 its bucket is the module file list it RETURNS, and its collide arm is four modules instead of dirs of them so that bucket is fileCount/4, i.e. 100 files at 400 and 400 at 1600), c 3.8 (2.535) and cpp 4.0 (2.639, the same basename bucket its suffix fallback walks). Four answer from keyed maps and keep the linear 1.8 \u2014 python 1.097, javascript 1.083, typescript 1.053, vue 1.079 \u2014 and that immunity IS the assertion, exactly as for ruby, kotlin, php and cobol. RUST IS THE ONE ARM THAT WAS REDESIGNED RATHER THAN BUDGETED. It resolves by probing candidate paths with allFilePaths.has(...) and never searches, so its cost is O(path segments) and provably flat in the file count (1.095 scaling, 1.061 collide scaling): a shared-leaf collide arm for rust would have asserted nothing, which is worse than no arm. Its collide corpus is instead a deep module tree (src/l0/l1/l2/l3/l4/mod{d}) whose targets carry ~2x the :: segments, so the arm exercises the axis that CAN grow, its 1.8 budget asserts the flatness across file counts, and collide_ms_ceiling 19 bounds the absolute cost of the long-path probe. small_ms_ceiling and collide_ms_ceiling are ~4x measured as everywhere else: rust 10/19 (2.609/4.704), python 7/8 (1.76/1.929, retightened from 12/15 against 3.044/3.771 by #2913), javascript 85/89 (21.254/22.145), typescript 85/86 (21.250/21.464), vue 81/93 (20.164/23.227), c 7/11 (1.620/2.850), cpp 7/12 (1.581/3.009). Swift takes ~5x (2 against 0.421 and 4 against 0.821) \u2014 the multiplier dart and cobol already carry, because a fixed scheduler hiccup is a larger fraction of a sub-1 ms number. ONE CAVEAT ON THE THREE ts-FAMILY MS NUMBERS, stated because nothing else in this file would reveal it: resolveTsTarget carries a per-pass resolveCache keyed currentFile::importPath, which no other resolver here has, and ~10% of this corpus is repeat pairs. Their us/import is therefore a slight underestimate of a cold resolve. It is left in rather than defeated because it is what the real pipeline does, and it is identical across all three so the arms stay comparable. HEAP for the eight: rust, swift, typescript, vue, cpp and cobol are still NOT gated, all of them measured before being left out. rust builds no index on this hook (16 B at 8000 files, 0 B at 32000); swift holds one pointer per file-times-segment and mints no strings, reading 0.98 MB at 8000 files against 0.29 MB at 32000 \u2014 a 4x larger corpus reading 3x SMALLER, which is what a measurement below its own noise floor looks like, and the same reading cobol gives (0.54 MB then 0 B); typescript and vue duplicate javascript through the same builder over the same-shaped corpus, and cpp duplicates c (10021320 against 10016960, 0.04% apart). Those four duplications are the ONLY exclusions that still rest on 'it would be a duplicate', and they are duplicates of a builder AND of a read pattern, which is the pairing csharp_csproj failed once the read pattern started to matter \u2014 if any of the four ever diverges in what it ASKS the index, it earns an arm the same way csharp_csproj just did. All eight gated arms are read the same way now (retainedPassBytes, one real import), so unlike before they are directly comparable to one another. WALL CLOCK \u2014 ~33-35 s in report mode, down from ~46 s, and ~44-45 s for --check, which is essentially UNCHANGED from ~46 s. Only report mode got faster; do not read the pair as 46 -> 42. The breakdown is worth having before anyone trims it. Timing arms: go 2.02, csharp 1.09, csharp_csproj 3.22, dart 0.41, ruby 2.90, kotlin 0.85, php 3.46, java 1.57, cobol 0.09, swift 0.46, rust 0.85, python 1.22, javascript 3.23, typescript 2.72, vue 2.89, c 0.86, cpp 0.91 (28.7 s, from 39.8 s: repsFor() accounts for all of it, and every second of it comes from the six languages whose cheapest cell is 20-28 ms); heap arms 3.43 s for SEVENTEEN languages, from 2.06 s for eight (every registered language is measured now; the nine added cost 1.37 s, of which kotlin alone is 0.57 s \u2014 see _heap_bound_note), and 2.1 s came from 3.0 s for seven when flattening retired the warm-up pass; module load 3.9 s. --check pays one import that report mode does not: the inventory arm loads pipeline/registry.ts, which drags in every registered scope resolver and its providers. Measured in isolation with the bench's own static imports already resident, that import costs 6.3-6.5 s on one box and 9.3-10.0 s on another \u2014 i.e. it consumes almost the whole repsFor win, which is why --check did not get faster. It is loaded dynamically at the point of use rather than at the top of the file, so report mode does not pay it and both modes take their measurements in the same module state. IT WAS WEIGHED AND KEPT, on the number that decides it: the benchmarks job is not CI's critical path. On the last green run of main it took 9 m 23 s against 12 m 58 s for the sharded coverage job that gates the merge, so ~4 m 40 s of slack sits above this bench and those seconds buy zero merge latency. Moving the arm to a vitest file would move the registry load ONTO the critical path, and would weaken it as well: this reconciles LANG_REGISTRY's SupportedLanguages values, which are what the five dispatcher branches key off, whereas a test that cannot import measure.mjs can only reconcile this file's arm NAMES plus a hand-written rule for de-aliasing csharp_csproj. The contract test import-target-index-reuse.contract.test.ts already covers the ADAPTER-boundary contract for every registered resolver; this arm covers a different claim, that the BENCH covers the pipeline. The ts family is still the largest single block of the timing phase (8.8 s) \u2014 its cost is suffixResolve probing ~39 extensions per path part on a miss, which is the real resolver and cannot be tuned away from the bench side. IF IT HAS TO SHRINK, drop collide and collide_large for typescript and vue and nothing else: -3.9 s, and it is the only cut that removes near-duplicate work rather than coverage, because all three run the same resolveTsTarget over the same buildSuffixIndex and javascript keeps the collide arm that covers their shared collision axis. Do NOT reach for REPS_MAX: it is 15 because depth_ratio tripped its own budget about 1 run in 20 at 5 and once at 7, and lowering it would re-open that for the eleven languages whose cheapest cell is sub-5 ms \u2014 which is where every recorded trip happened. The six languages it was safe to lower have already been lowered, per language and from a measurement, by repsFor(). ---- THE FIFTH ARGUMENT (context) AND THE TWO ARMS IT MOVED ---- resolveOne now makes run.ts's five-argument call for the two hooks that declare a fifth parameter, so php and python time the legs behind it. Nothing else moved: the other fifteen arms are handed no context and build no ParsedFile[] at all, and over five runs their five ms numbers and four ratios sit exactly where they did. Both languages' ten fingerprints, resolved counts and distinct_outcomes are IDENTICAL \u2014 the leg AGREES with the cascade on this corpus, which is the whole reason the context arm had to be added rather than leaving the fingerprint to notice. PHP: small_ms 27.762 -> 35.125 (+26.5%) and collide_ms 29.407 -> 36.182 (+23.0%), which is filesByDirectory plus, on every import that resolves, a candidate gather over the resolved file's directory and a localDefs filter; the ms ceilings keep PHP's own 4.21x and 4.26x multipliers (117 -> 148, 125 -> 154). depth_ratio 1.144 -> 1.283 and the 1.9 budget is UNCHANGED, which makes it 1.48x measured rather than 1.66x: directoryAliases emits one entry per path segment, so filesByDirectory is O(files x depth) and the depth arm is the only one that can see it \u2014 that budget got TIGHTER relative to its measurement, not looser, and 1.48x sits inside the 1.37-1.75x band the other sixteen carry. Its heap reading rises 37576816 -> 49574008 (+31.9%) for the same structure, and the reading is the MEMO rather than the workspace it indexes: newPass allocates the ParsedFile objects before retainedPassBytes takes its baseline sample, so they sit outside the delta. PYTHON, WHOSE FIGURES ARE THE LEAST SETTLED THING IN THIS FILE AND ARE RECORDED IN TWO SNAPSHOTS BECAUSE OF IT. A named import is the only spelling that reads context.parsedFiles, and it costs up to three entries into the resolver per import (package probe, exports check, submodule probe) where the synthetic namespace spelling this arm used to pass costs one. Against the resolver as it stood when the call shape changed that read small_ms 1.76 -> 5.751 and collide_ms 1.929 -> 5.894, ~3.1x. Against the resolver a few commits later \u2014 which stopped re-running the whole tail after a null package probe, a double-probe this bench could not previously see because the namespace spelling never entered that branch \u2014 the same arms read 4.404 and 4.505. The ceilings are 18 and 19, chosen to clear BOTH: 4.09x and 4.22x of the current numbers, 3.13x and 3.22x of the higher ones, so neither state is red. Retighten toward 4x once that resolver settles. ITS DEPTH ARM WAS DILUTED AND THE BUDGET IS RETIGHTENED TO MATCH, which is the one thing here worth arguing about: the added work is depth-FLAT, so depth_ratio FALLS 1.872 -> 1.478 while the absolute cost more than doubles, and 2.6 against 1.478 would be 1.76x \u2014 far looser than the 1.39x #2913 chose deliberately to lock its own fix in. 2.1 restores that multiplier (1.42x). THE TWO MUTATION SCORES #2913 RECORDED (3.123 for reverting the per-directory memo, 2.734 for reverting the nested-name rejection) WERE TAKEN AGAINST THE OLD CALL SHAPE AND HAVE NOT BEEN RE-TAKEN. Modelled forward, with the depth-quadratic term reappearing in every resolver entry so its absolute contribution scales with the entry count, they land near 2.8 and 2.4 \u2014 both above 2.1, and the second BELOW 2.6, which is the arithmetic that decided the budget. Re-run the two mutations before trusting the lock-in claim above. python's heap reading is unchanged (10543152 recorded; 10529848-10544616 across eight runs) because its probe misses before the branch that reads parsedFiles \u2014 see _blind_spot for why no probe can reach that memo. Every figure in this section is the MAXIMUM over its snapshot's runs (five, then three), with peak-to-peak 1.031-1.058 on php and 1.019-1.081 on python, taken on a box that was NOT idle and with another change landing in python's resolver mid-measurement. Re-take them serially before merging.", + "_triage": "Every ratio and ms ceiling here is a TIMING signal \u2014 re-run on an idle machine before investigating; runner contention dominates. depth_ratio is the noisiest of them by a wide margin (it divides two sub-3 ms numbers, and Dart's are sub-1 ms): if exactly one arm fails and it is that one, suspect the machine first. N is 15 for every language whose cheapest arm is under 5 ms, rather than this bench's original 5, specifically to hold that arm's peak-to-peak swing under 1.26x \u2014 see _arms_note for the measured distributions and for why the six languages that drop to 7-8 are the ones where cell size makes it safe \u2014 so a depth_ratio failure that REPRODUCES is a real signal, not noise. Each language's chosen N is printed as `reps`; read it before blaming the estimator. The fingerprint, shape and heap arms are the opposite: deterministic (over 4 runs the heap arm's widest spread was 0.11% on python and 0.00% on java, javascript and c), a re-run never changes them, and they must never be wished away. TWO heap failures mean the arm STOPPED MEASURING rather than that memory grew, and both are deterministic: a heap floor failure says the probe no longer forces the index it used to (this is how four arms read 0 B when buildSuffixIndex went lazy, and 0 B passes every ceiling), and a `heap probe ... resolved` throw says a probe target that must MISS now hits, so the reading is a materialized answer and the legs past it were never reached. A heap BOUND failure is deterministic in the same way and means one specific thing: a language excluded from the budgeted tier has grown a structure, or started asking its index a question it did not ask when the exclusion was recorded \u2014 never a timing signal, never a re-run, and never fixed by raising the bound without saying what grew. The context arm is deterministic too, and a failure there means one specific thing rather than a range of them: run.ts's fifth argument is not reaching that resolver from this bench, or the leg behind it stopped running. Never a timing signal, never a re-run. TIGHTENED IN #2881, because the measurements they bound got faster and a budget left alone while its reading falls is a gate loosening without anyone deciding to. Each new value holds the headroom the old one expressed over the old reading, computed from `_measured` on both sides: kotlin depth 3.4 -> 2.8 (reading 2.219 -> 1.813), go depth 1.6 -> 1.4 (1.169 -> 0.999), csharp depth 2.2 -> 2.0 (1.438 -> 1.279), java depth 2.2 -> 2.1 (1.402 -> 1.354), kotlin collide_scaling 1.8 -> 1.65 (1.179 -> 1.081), go collide_scaling 5.5 -> 5.1 (3.763 -> 3.465). The ABSOLUTE ms ceilings were deliberately NOT tightened by the same reasoning: they carry runner-contention headroom rather than measurement headroom, and a ratio is runner-speed-invariant where a millisecond is not.", + "_floor": "Measured against the pre-change implementations on THIS corpus at 150/600 files: go 3.36, csharp 4.10, dart 3.32, ruby 3.87. The issues report 4.00 / 3.43 / 4.05 on their own corpora; those are DIFFERENT numbers from different repositories and are not reproduced here \u2014 what they and these share is that both independently land in the quadratic band, well clear of the ~1.0 a linear result gives. Note also that this floor was taken at 150/600 while the gate runs at 400/1600, so it is a lower bound on what the pre-change code would score today. Kotlin's own bench measured its pre-index floor at 3.737. The four resolvers added later were NOT re-floored on this corpus, and the reason is that they do not need to be: every one of their pre-change legs walked the whole file set per import (PHP one findIndex per path part per extension, Java one scan per stripped prefix, COBOL two full scans per COPY, C# csproj one normalizedFileList pass per import per matching config), so their scaling_ratio is ~4 by construction rather than by measurement. Their per-import costs were measured on their own issue corpora instead: PHP 96.40 ms -> 0.036 ms, Java 8.05 ms -> 0.62 ms, COBOL 3879 us -> 10.5 us, C# csproj 1103 us -> 7.6 us. The 1.8 budget sits well above the linear result and well below every one of those. The eight languages added last were NOT floored either, and for a different reason again: they are not fixes, so there is no pre-change implementation to floor against. Their scaling budgets are the global linear 1.8 and the point of the arms is to hold the current numbers (measured 1.01-1.13) rather than to separate a fix from a break. The one exception is javascript, which IS a fix and does have a floor: 6448.9 us per import at 2000 files and 25972.6 us at 8000 \u2014 4.12x the per-import cost for 4x the files, i.e. O(imports x files) \u2014 against 28.5 / 27.4 us with the index PR #2911 gave it, and 25.0 / 27.0 us for TypeScript over the identical corpus.", + "_rebaselined_2910_java_declared_packages": "#2910 replaces Java path-suffix fallback with declared-package resolution. The benchmark now restores package capture side channels, threads parsedFiles through javaScopeResolver, proves the context leg with a positive path/package-mismatch probe, and models the collide arm as one package declared across service paths. External imports now remain unresolved; local exact and wildcard imports preserve the 1153/4681 workload. Java's index is package/type maps rather than suffix maps: bytes_large 34958600 -> 3676984, with its floor and ceiling re-recorded together. Depth and collision scaling budgets tighten to the shared linear 1.8 gate.", "scaling_budget": 1.8, "collide_scaling_budget": { "go": 5.1, @@ -14,7 +15,7 @@ "ruby": 1.8, "kotlin": 1.65, "php": 1.8, - "java": 3.4, + "java": 1.8, "cobol": 1.8, "swift": 4.9, "rust": 1.8, @@ -33,13 +34,13 @@ "ruby": 2.2, "kotlin": 2.8, "php": 1.9, - "java": 2.1, + "java": 1.8, "cobol": 1.6, "swift": 2.3, "rust": 2.1, "python": 2.2, - "javascript": 2.1, - "typescript": 2.1, + "javascript": 2.6, + "typescript": 2.6, "vue": 2.3, "c": 3, "cpp": 3 @@ -83,8 +84,6 @@ "cpp": 12 }, "heap_ceiling_bytes": { - "vue": 43326024, - "typescript": 40117944, "kotlin": 46000000, "go": 4497696, "dart": 11751300, @@ -93,16 +92,13 @@ "csharp_csproj": 110600000, "ruby": 61600000, "php": 74400000, - "java": 52500000, + "java": 5600000, "python": 9541404, - "javascript": 40200000, "c": 15000000 }, - "_heap_reading_note": "The measured bytes_large each heap_ceiling_bytes entry above is 1.5x — every entry except kotlin's, which is 1.0747x for a stated reason (see _heap_compaction_gate). Recorded so the FLOOR can be derived from the reading instead of from the ceiling. It used to be 0.33 x the ceiling, described as 'half the measured size' — which held only while every ceiling stayed at exactly 1.5x its reading, a convention this file states and nothing enforces, so re-tuning one ceiling upward would have loosened that language's floor by the same factor in the one direction a floor exists to watch. 0.5 x the reading is the same effective floor to within 0.8% for all eight and says what it means — and it is what let kotlin's ceiling be tightened to 1.0747x without moving kotlin's floor by a byte, which is exactly the independence this key was introduced for. These are NOT asserted for equality: they reproduce to the byte across processes on one box, but a Node major or a different platform moves heapUsed accounting, and the ceiling/floor pair is what tolerates that (+50%/-50%, and +7.5%/-50% for kotlin). Re-baseline a ceiling and re-baseline the reading with it — they are two views of one measurement.", - "_heap_compaction_gate": "WHY KOTLIN'S CEILING IS TIGHT AND EVERY OTHER ONE IS 1.5x. It is the only ceiling in this file that gates a size REDUCTION being preserved rather than a footprint not growing: #2881 compacts getKotlinFileIndex's dirChildren buckets (`bucket.slice()` before the freeze), and until this entry existed nothing anywhere could see that compaction disappear. MEASURED, not assumed — head against a copy of languages/kotlin/import-target.ts with the slice deleted and Object.freeze kept, one process, the same corpus this arm builds: bytes_large 42805256 -> 48184784 (+12.57%), bytes_small 10676432 -> 12020736 (+12.6%), byte-identical over three runs. NOTHING ELSE MOVES for that mutation. Every arm of bench/kotlin-import-target is output-identical (its fingerprint, cases and non_null cannot see an array's spare capacity); test/unit/scope-resolution/kotlin/kotlin-index-internals.test.ts stays green and says so in its own header, because a JS array's backing-store capacity has no reflective surface; heap ratio is 1.002 either way, since both scales grow together and a ratio divides the growth out; and at the old ceiling of 64203684 the heap arm passed with 25% to spare. DIRECTION MATTERS: compaction RECLAIMS, so losing it makes the reading GROW. The gate is therefore the CEILING. A floor cannot see this mutation in any sizing, and kotlin's floor stays the file-wide 0.5 x reading. WHERE THE 5.4 MB COMES FROM, so the number can be re-derived rather than trusted: the heap corpus is 32000 files over 4000 directories, 8 files each, and a directory contributes one dirChildren key per component-suffix of its path (~15.3 keys at HEAP_PAD 8), so ~61000 buckets of length 8. On this repo's Node a bucket minted as [raw] and pushed to 8 sits in a 19-slot backing store — the capacity steps kotlin-index-internals.test.ts records — leaving 11 slots, 88 B, of retained slack per bucket. 61000 x 88 B is ~5.4 MB, which is the delta. HOW 46000000 WAS CHOSEN: reading 42802456, plus 7.5% is 46012640, rounded down to 46000000 (1.0747x). PROVEN both ways through the real gate, not argued: a full `--check` over a copy of measure.mjs whose only difference is the kotlin import, pointed at a resolver with the slice deleted, reads 48203376 B (45.97 MiB) and fails on THIS ARM ALONE — every fingerprint, every corpus count, every timing ratio and the heap ratio all stay green, which is the claim 'nothing else moves' turned into a run. That is 4.8% clear above the ceiling. Both margins are three orders of magnitude larger than the measurement's own spread (peak-to-peak 1.0001 over three runs of the isolated arm, 1.0004 over the five runs _heap_bound_note records). The 7.5% is also an order of magnitude above the widest cross-run movement any heap arm in this file shows on this box: kotlin itself reads 42802456 B in a full `--check`, byte-identical to the recorded value, and the noisiest reading here — csharp_csproj, the one prior sessions found unreproducible — moves 0.78% between runs. A LOADED RUNNER DOES NOT MOVE THIS NUMBER and the tolerance is not for one: this is a forced-GC heapUsed delta over structures held alive across the window (see HEAP_RETAINED), so scheduler contention has no term in it. What can move it is heapUsed ACCOUNTING — a Node major, a heap above the pointer-compression cage, a 32-bit platform. TRIAGE, and it is what makes the tight ceiling safe to run: that class of change moves EVERY reading in the run, so compare kotlin against the other 13 budgeted readings in the SAME run before touching this key. kotlin alone over its ceiling with the rest of the file at its recorded values is a lost compaction; everything moving together is a runner change and a whole-file re-baseline. WHAT IT DOES NOT CATCH: any regression under 7.5%, and a compaction that still runs while something else in the index grows to fill the headroom.", + "_heap_reading_note": "The measured bytes_large each heap_ceiling_bytes entry above is 1.5x \u2014 every entry except kotlin's, which is 1.0747x for a stated reason (see _heap_compaction_gate). Recorded so the FLOOR can be derived from the reading instead of from the ceiling. It used to be 0.33 x the ceiling, described as 'half the measured size' \u2014 which held only while every ceiling stayed at exactly 1.5x its reading, a convention this file states and nothing enforces, so re-tuning one ceiling upward would have loosened that language's floor by the same factor in the one direction a floor exists to watch. 0.5 x the reading is the same effective floor to within 0.8% for all eight and says what it means \u2014 and it is what let kotlin's ceiling be tightened to 1.0747x without moving kotlin's floor by a byte, which is exactly the independence this key was introduced for. These are NOT asserted for equality: they reproduce to the byte across processes on one box, but a Node major or a different platform moves heapUsed accounting, and the ceiling/floor pair is what tolerates that (+50%/-50%, and +7.5%/-50% for kotlin). Re-baseline a ceiling and re-baseline the reading with it \u2014 they are two views of one measurement.", + "_heap_compaction_gate": "WHY KOTLIN'S CEILING IS TIGHT AND EVERY OTHER ONE IS 1.5x. It is the only ceiling in this file that gates a size REDUCTION being preserved rather than a footprint not growing: #2881 compacts getKotlinFileIndex's dirChildren buckets (`bucket.slice()` before the freeze), and until this entry existed nothing anywhere could see that compaction disappear. MEASURED, not assumed \u2014 head against a copy of languages/kotlin/import-target.ts with the slice deleted and Object.freeze kept, one process, the same corpus this arm builds: bytes_large 42805256 -> 48184784 (+12.57%), bytes_small 10676432 -> 12020736 (+12.6%), byte-identical over three runs. NOTHING ELSE MOVES for that mutation. Every arm of bench/kotlin-import-target is output-identical (its fingerprint, cases and non_null cannot see an array's spare capacity); test/unit/scope-resolution/kotlin/kotlin-index-internals.test.ts stays green and says so in its own header, because a JS array's backing-store capacity has no reflective surface; heap ratio is 1.002 either way, since both scales grow together and a ratio divides the growth out; and at the old ceiling of 64203684 the heap arm passed with 25% to spare. DIRECTION MATTERS: compaction RECLAIMS, so losing it makes the reading GROW. The gate is therefore the CEILING. A floor cannot see this mutation in any sizing, and kotlin's floor stays the file-wide 0.5 x reading. WHERE THE 5.4 MB COMES FROM, so the number can be re-derived rather than trusted: the heap corpus is 32000 files over 4000 directories, 8 files each, and a directory contributes one dirChildren key per component-suffix of its path (~15.3 keys at HEAP_PAD 8), so ~61000 buckets of length 8. On this repo's Node a bucket minted as [raw] and pushed to 8 sits in a 19-slot backing store \u2014 the capacity steps kotlin-index-internals.test.ts records \u2014 leaving 11 slots, 88 B, of retained slack per bucket. 61000 x 88 B is ~5.4 MB, which is the delta. HOW 46000000 WAS CHOSEN: reading 42802456, plus 7.5% is 46012640, rounded down to 46000000 (1.0747x). PROVEN both ways through the real gate, not argued: a full `--check` over a copy of measure.mjs whose only difference is the kotlin import, pointed at a resolver with the slice deleted, reads 48203376 B (45.97 MiB) and fails on THIS ARM ALONE \u2014 every fingerprint, every corpus count, every timing ratio and the heap ratio all stay green, which is the claim 'nothing else moves' turned into a run. That is 4.8% clear above the ceiling. Both margins are three orders of magnitude larger than the measurement's own spread (peak-to-peak 1.0001 over three runs of the isolated arm, 1.0004 over the five runs _heap_bound_note records). The 7.5% is also an order of magnitude above the widest cross-run movement any heap arm in this file shows on this box: kotlin itself reads 42802456 B in a full `--check`, byte-identical to the recorded value, and the noisiest reading here \u2014 csharp_csproj, the one prior sessions found unreproducible \u2014 moves 0.78% between runs. A LOADED RUNNER DOES NOT MOVE THIS NUMBER and the tolerance is not for one: this is a forced-GC heapUsed delta over structures held alive across the window (see HEAP_RETAINED), so scheduler contention has no term in it. What can move it is heapUsed ACCOUNTING \u2014 a Node major, a heap above the pointer-compression cage, a 32-bit platform. TRIAGE, and it is what makes the tight ceiling safe to run: that class of change moves EVERY reading in the run, so compare kotlin against the other 13 budgeted readings in the SAME run before touching this key. kotlin alone over its ceiling with the rest of the file at its recorded values is a lost compaction; everything moving together is a runner change and a whole-file re-baseline. WHAT IT DOES NOT CATCH: any regression under 7.5%, and a compaction that still runs while something else in the index grows to fill the headroom.", "heap_reading_bytes": { - "vue": 28884016, - "typescript": 26745296, "kotlin": 42802456, "go": 2998464, "dart": 7834200, @@ -111,16 +107,18 @@ "csharp_csproj": 73703384, "ruby": 41020808, "php": 49574008, - "java": 34958600, + "java": 3676984, "python": 6360936, - "javascript": 26745296, "c": 10018816 }, - "_heap_bound_note": "THE SECOND HEAP TIER. Every registered language is measured now; heap_bound_bytes gates the nine that are not BUDGETED above, and it gates them with one comparison and no floor. A ceiling says 'this index is not too big'. A bound says something narrower and it is the thing that was missing: 'the exclusion still holds' — this language has not grown an index since it was left out. measure.mjs's MEMORY section states the re-entry condition (if a language ever diverges in what it ASKS its index, it earns a budgeted arm) and until now nothing watched for the divergence; HEAP_LANGS was a hand-maintained list of eight whose two neighbours, LANG_REGISTRY and CONTEXT_LANGS, are both reconciled against a derived predicate in both directions. HEAP_BOUNDED is derived too — it is LANGS minus HEAP_BUDGETED — so the two tiers partition the languages and a new one cannot land outside both. WHAT RE-MEASURING FOUND, five runs each, maximum quoted, peak-to-peak in brackets. go 2998464 B [1.0021], dart 7834200 B [1.0006] and kotlin 42802456 B [1.0004] HAD NO STATED REASON AT ALL: the old prose opened 'SIX of the seventeen are deliberately NOT in HEAP_LANGS' against a list of eight of seventeen, and these three were the three nobody counted. All three retain a real per-pass structure (go's PackageDirIndex, dart's basename buckets, kotlin's suffixByStem cascade) and kotlin's 40.82 MiB is above ruby's 39.12 and java's 33.34, both of which carry a full budget. (It read 45.85 MiB when this was written, described here as 'the second-largest reading in this file' — it was third even then, behind csharp_csproj and php; #2881 later compacted its dirChildren buckets and took 11% off it. Same staleness this paragraph exists to document.) swift 3449216 B [1.0024] and cobol 2320456 B [1.0000] were excluded as 'below the measurement's own noise floor' on readings of 0.29 MB and 0 B at 32000 files; they now read 3.29 MB and 2.21 MB, growing with the corpus (969120 B and 536264 B at 8000). Those old numbers were not wrong when taken — the ARM changed under them, when #2903's follow-up made every probe resolve a real import and when measureHeap began flattening its corpus — which is the whole finding: a measurement written into prose is not re-taken, and this file had already gone stale against itself, quoting javascript at 46208832 B four paragraphs after quoting it at 25.51 MiB. rust is the one exclusion that survived unchanged: 16 B at 8000 files and 16 B at 32000, identical in all five runs. typescript 26745296 B, vue 28884016 B and cpp 10023344 B are duplicates of a builder AND of a read pattern: typescript is byte-identical to javascript's 26745296 in four runs of five, cpp is +0.05% of c's 10018816, vue is +8.0% of javascript. HOW THE BOUNDS WERE CHOSEN. Each takes 1.5x its measured maximum, rounded up to the next 100000 B: cobol 3500000 (1.508x), swift 5200000 (1.508x). (This sentence used to list eight, including go, dart, kotlin, typescript, vue and cpp. Those six were promoted to the budgeted tier and their bounds deleted; the numbers stayed here, unread by any gate, and #2881 dutifully updated kotlin's to 64300000 before anyone noticed heap_bound_bytes holds only cobol, swift and rust. A number nothing asserts is a number that rots — the finding this paragraph is otherwise about.) 1.5x is NOT copied from the ceilings out of habit — it is the same number for a stated reason, and the reason is not noise: measured peak-to-peak on this box is at most 1.0024, so noise alone would justify 1.05x. What a bound has to survive is a RUNNER change, since heapUsed accounting moves across platforms and Node majors, and this file already fixes that allowance at 50% for exactly this measurement on exactly this arm. Using a second allowance for the same uncertainty on the same number would be two conventions, not more rigour. At 1.5x the bound catches what the re-entry condition is about — a language growing an index, which costs +85% for one more suffix map and +100% for a duplicate — and it does NOT catch a duplicate diverging by 8%. That limit is real and is stated rather than hidden: the tight form is a same-process ratio against the arm each duplicate is a duplicate OF, which is the only form immune to the drift the absolute bound has to tolerate. RUST TAKES AN ABSOLUTE BOUND INSTEAD, 1048576 B (1 MiB), because 1.5 x 16 B is 24 B and would fail on the first byte of anything — a multiplier on a reading that is already nothing is a gate that flakes rather than a gate that bites. 1 MiB is ~65000x the reading and still 2.2x below the smallest real index measured here (cobol's 2.32 MB at the same file count), so it separates 'builds nothing' from 'builds something' with room on both sides. NO FLOOR ON ANY OF THE NINE, and the reason differs by language rather than being uniform. For rust a floor would be a floor on noise. For the other eight the readings are stable enough to floor today, and for kotlin and dart — larger than budgeted arms — a floor would be worth having, since a lazily-built map going quiet is exactly how the four budgeted arms once read 0 B. Adding one is a PROMOTION to the budgeted tier, with a ceiling and a recorded reading beside it, not a line here: a floor whose companion ceiling does not exist asserts 'still measuring' against a number nothing else bounds. Recommended next, in order: kotlin, then dart, then go.", + "_heap_bound_note": "THE SECOND HEAP TIER. Every registered language is measured now; heap_bound_bytes gates the nine that are not BUDGETED above, and it gates them with one comparison and no floor. A ceiling says 'this index is not too big'. A bound says something narrower and it is the thing that was missing: 'the exclusion still holds' \u2014 this language has not grown an index since it was left out. measure.mjs's MEMORY section states the re-entry condition (if a language ever diverges in what it ASKS its index, it earns a budgeted arm) and until now nothing watched for the divergence; HEAP_LANGS was a hand-maintained list of eight whose two neighbours, LANG_REGISTRY and CONTEXT_LANGS, are both reconciled against a derived predicate in both directions. HEAP_BOUNDED is derived too \u2014 it is LANGS minus HEAP_BUDGETED \u2014 so the two tiers partition the languages and a new one cannot land outside both. WHAT RE-MEASURING FOUND, five runs each, maximum quoted, peak-to-peak in brackets. go 2998464 B [1.0021], dart 7834200 B [1.0006] and kotlin 42802456 B [1.0004] HAD NO STATED REASON AT ALL: the old prose opened 'SIX of the seventeen are deliberately NOT in HEAP_LANGS' against a list of eight of seventeen, and these three were the three nobody counted. All three retain a real per-pass structure (go's PackageDirIndex, dart's basename buckets, kotlin's suffixByStem cascade) and kotlin's 40.82 MiB is above ruby's 39.12 and java's 33.34, both of which carry a full budget. (It read 45.85 MiB when this was written, described here as 'the second-largest reading in this file' \u2014 it was third even then, behind csharp_csproj and php; #2881 later compacted its dirChildren buckets and took 11% off it. Same staleness this paragraph exists to document.) swift 3449216 B [1.0024] and cobol 2320456 B [1.0000] were excluded as 'below the measurement's own noise floor' on readings of 0.29 MB and 0 B at 32000 files; they now read 3.29 MB and 2.21 MB, growing with the corpus (969120 B and 536264 B at 8000). Those old numbers were not wrong when taken \u2014 the ARM changed under them, when #2903's follow-up made every probe resolve a real import and when measureHeap began flattening its corpus \u2014 which is the whole finding: a measurement written into prose is not re-taken, and this file had already gone stale against itself, quoting javascript at 46208832 B four paragraphs after quoting it at 25.51 MiB. rust is the one exclusion that survived unchanged: 16 B at 8000 files and 16 B at 32000, identical in all five runs. typescript 26745296 B, vue 28884016 B and cpp 10023344 B are duplicates of a builder AND of a read pattern: typescript is byte-identical to javascript's 26745296 in four runs of five, cpp is +0.05% of c's 10018816, vue is +8.0% of javascript. HOW THE BOUNDS WERE CHOSEN. Each takes 1.5x its measured maximum, rounded up to the next 100000 B: cobol 3500000 (1.508x), swift 5200000 (1.508x). (This sentence used to list eight, including go, dart, kotlin, typescript, vue and cpp. Those six were promoted to the budgeted tier and their bounds deleted; the numbers stayed here, unread by any gate, and #2881 dutifully updated kotlin's to 64300000 before anyone noticed heap_bound_bytes holds only cobol, swift and rust. A number nothing asserts is a number that rots \u2014 the finding this paragraph is otherwise about.) 1.5x is NOT copied from the ceilings out of habit \u2014 it is the same number for a stated reason, and the reason is not noise: measured peak-to-peak on this box is at most 1.0024, so noise alone would justify 1.05x. What a bound has to survive is a RUNNER change, since heapUsed accounting moves across platforms and Node majors, and this file already fixes that allowance at 50% for exactly this measurement on exactly this arm. Using a second allowance for the same uncertainty on the same number would be two conventions, not more rigour. At 1.5x the bound catches what the re-entry condition is about \u2014 a language growing an index, which costs +85% for one more suffix map and +100% for a duplicate \u2014 and it does NOT catch a duplicate diverging by 8%. That limit is real and is stated rather than hidden: the tight form is a same-process ratio against the arm each duplicate is a duplicate OF, which is the only form immune to the drift the absolute bound has to tolerate. RUST TAKES AN ABSOLUTE BOUND INSTEAD, 1048576 B (1 MiB), because 1.5 x 16 B is 24 B and would fail on the first byte of anything \u2014 a multiplier on a reading that is already nothing is a gate that flakes rather than a gate that bites. 1 MiB is ~65000x the reading and still 2.2x below the smallest real index measured here (cobol's 2.32 MB at the same file count), so it separates 'builds nothing' from 'builds something' with room on both sides. NO FLOOR ON ANY OF THE NINE, and the reason differs by language rather than being uniform. For rust a floor would be a floor on noise. For the other eight the readings are stable enough to floor today, and for kotlin and dart \u2014 larger than budgeted arms \u2014 a floor would be worth having, since a lazily-built map going quiet is exactly how the four budgeted arms once read 0 B. Adding one is a PROMOTION to the budgeted tier, with a ceiling and a recorded reading beside it, not a line here: a floor whose companion ceiling does not exist asserts 'still measuring' against a number nothing else bounds. Recommended next, in order: kotlin, then dart, then go.", "heap_bound_bytes": { "cobol": 3500000, "swift": 5200000, - "rust": 1048576 + "rust": 1048576, + "javascript": 1048576, + "typescript": 1048576, + "vue": 1048576 }, "heap_floor_fraction": 0.5, "heap_ratio_budget": 1.25, @@ -493,49 +491,54 @@ "imports": 3200, "resolved": 1153, "distinct_outcomes": 2868, - "fingerprint": "c66d780f4b5549e0a9596ed30eb9035dca8e8168e90371124d1cbad968205569" + "fingerprint": "8e347c485c47a1a9f67ae0183b68327b40b1e5d3510160fda2a2fe54c0f8a453" }, "large": { "files": 1600, "imports": 12800, "resolved": 4681, "distinct_outcomes": 11512, - "fingerprint": "354cc4de030ce581982eae15d7f6c95ba8ed14b7a14a2e0e9c3b06f90efb1eee" + "fingerprint": "6773de19833d9936cb098c5897b5a44ba19b07bd8990af5a8b70fc03b309794f" }, "deep": { "files": 400, "imports": 3200, "resolved": 1153, "distinct_outcomes": 2868, - "fingerprint": "de26edd2268593f35120801c8f96799786ff01976dc84f9c342e7426be0afa9a" + "fingerprint": "0ba22e27f87395535533bac3270c481ea50aa4cd7eba4958325ef792f308bfc0" }, "collide": { "files": 400, "imports": 3200, "resolved": 1153, - "distinct_outcomes": 2868, - "fingerprint": "260aa5fa381b853e2cb92548a5bbb684f4afe6ba513d419e2fb2bd178f00c293" + "distinct_outcomes": 2744, + "fingerprint": "2ab215bd5109f13c0f15513c3bf578ca467896dd28d489c9ac4477d2b647c39f" }, "collide_large": { "files": 1600, "imports": 12800, "resolved": 4681, - "distinct_outcomes": 11512, - "fingerprint": "a8bcf5e432dcedf82c7b23c533311b051b00a68cbbd8bb4e816fd1c8ae055f33" + "distinct_outcomes": 10961, + "fingerprint": "31e762ed838528a426e5cc4510956661fc2aef7aa7c743291d75fcec54e46235" }, - "fingerprint": "354cc4de030ce581982eae15d7f6c95ba8ed14b7a14a2e0e9c3b06f90efb1eee", + "fingerprint": "6773de19833d9936cb098c5897b5a44ba19b07bd8990af5a8b70fc03b309794f", "heap": { "files_small": 8000, "files_large": 32000, "path_segments": 18, "probe": "com.google.common.vendor0.Missing" }, + "context": { + "target": "com.example.model.User", + "with_context": "weird/path/User.java", + "without_context": "" + }, "_measured": { - "collide_ms": 5.34, - "collide_scaling_ratio": 2.508, - "depth_ratio": 1.354, - "scaling_ratio": 1.147, - "small_ms": 3.317 + "collide_ms": 5.219, + "collide_scaling_ratio": 1.07, + "depth_ratio": 0.978, + "scaling_ratio": 1.0, + "small_ms": 5.674 } }, "cobol": { @@ -1003,5 +1006,7 @@ } } }, - "_blind_spot": "MEASURED, so nobody has to rediscover it: a full workspace scan reintroduced on 1-in-32 imports passes EVERY arm here — dart scored 1.458 scaling and 1.736 ms against the 1.8 budget and 4 ms ceiling of an earlier revision. At 1-in-8 the scaling arm catches it (2.414). The gate that NARROWS this is not a timing gate at all: test/unit/scope-resolution/import-target-index-parity.test.ts counts iterations of the file-set Set and reads 14 instead of 1 for that same 1-in-32 mutation, deterministically and for all five languages. It does NOT close it. The counter watches the Set, and the resolvers no longer read the Set — they read materialized copies of the same file list: WorkspaceFileIndex.normalized and .all (C#, Ruby), Dart's byBasename buckets, and PackageDirIndex.filesByDir (Go, C#). A 1-in-32 scan over any of those three touches the Set zero extra times, so it passes the parity test AND passes --check. Closing it would take an iteration counter on the materialized arrays themselves. Read the two gates together; tightening these ceilings toward the noise floor to chase that case would only buy flaky CI. CONFIRMED THE HARD WAY by PR #2911: JavaScript resolution was scanning ImportPassCache.normalizedFileList on every import — a materialized array, not the Set — at 25972 us per import at 8000 files, and no instrument on the #2901-#2909 branch could see it. It took a differential parity test over 211200 old-vs-new pairs to find. The arms added here would have caught THAT one on absolute ms (85 ms budget against a 20 ms arm; the unindexed resolver costs ~83000 ms on the same corpus), which is the argument for gating every registered language rather than only the ones a PR happens to touch. THE SECOND BLIND SPOT IS CLOSED, and this records what closing it changed. This harness used to call the inner resolvers with the NO-CONTEXT shape: run.ts calls provider.resolveImportTarget with five arguments, the fifth being { parsedFiles, parsedImport }, and resolveOne supplied three. resolveOne now makes the production call, newPass mints the ParsedFile[] FIRST and derives the path set from it exactly as run.ts does, and both legs behind the argument run on every import of their arms — PHP's named/alias function-or-const leg over filesByDirectory(context.parsedFiles), whose memo defeated measures 197.0 us -> 9976.2 us per import (50.6x), and Python's from-import submodule-precedence branch, the only spelling that reads context.parsedFiles at all. Fifteen of the seventeen arms cannot observe a context (their hooks declare three or four parameters) and are handed none, so their numbers did not move; which two CAN is now reconciled against SCOPE_RESOLVERS' hook arity rather than asserted in prose. NOTHING ELSE IN THIS FILE COULD HAVE GATED IT, which is why the context arm exists: on this corpus the leg AGREES with the cascade for every import, so all ten of PHP's and Python's fingerprints, their resolved counts and their distinct_outcomes are unchanged; a dropped context makes the timing arms FASTER and no arm here has a lower bound on ms; and the heap floor (0.5 x 49573840 = 24.8 MB) still passes the 37576816 B a no-context PHP pass reads. The arm is one import per language resolved through resolveOne twice, with and without the pass's parsedFiles, whose two answers must DIFFER and must both match what is recorded. WHAT REMAINS UNMEASURED, narrowed rather than deleted: Python's parsedFileByPath memo is exercised by the five timing arms and cannot be reached by the heap arm at all, because retainedPassBytes requires a probe that MISSES while every path that builds that memo returns a non-null packageTarget — so no ceiling bounds that Map (one pointer per parsed file, O(files), no depth term) and the contract test's count gate is what holds it to one build per pass. PHP's leg is measured with NO composer.json, so namespaceDirectories only ever returns the directory of an already-resolved file and the PSR-4 mapping branch stays unreached, exactly as csharp cannot reach the csproj leg; closing that is a second PHP arm on the csharp_csproj precedent, not a parameter. And the const tail of PHP's leg is a different ANSWER at the same cost — it runs the identical candidate gather and localDefs filter and diverges in the last two lines — so it is gated by count in test/unit/scope-resolution/import-target-index-reuse.contract.test.ts, which stays the gate to read alongside this file." + "_blind_spot": "MEASURED, so nobody has to rediscover it: a full workspace scan reintroduced on 1-in-32 imports passes EVERY arm here \u2014 dart scored 1.458 scaling and 1.736 ms against the 1.8 budget and 4 ms ceiling of an earlier revision. At 1-in-8 the scaling arm catches it (2.414). The gate that NARROWS this is not a timing gate at all: test/unit/scope-resolution/import-target-index-parity.test.ts counts iterations of the file-set Set and reads 14 instead of 1 for that same 1-in-32 mutation, deterministically and for all five languages. It does NOT close it. The counter watches the Set, and the resolvers no longer read the Set \u2014 they read materialized copies of the same file list: WorkspaceFileIndex.normalized and .all (C#, Ruby), Dart's byBasename buckets, and PackageDirIndex.filesByDir (Go, C#). A 1-in-32 scan over any of those three touches the Set zero extra times, so it passes the parity test AND passes --check. Closing it would take an iteration counter on the materialized arrays themselves. Read the two gates together; tightening these ceilings toward the noise floor to chase that case would only buy flaky CI. CONFIRMED THE HARD WAY by PR #2911: JavaScript resolution was scanning ImportPassCache.normalizedFileList on every import \u2014 a materialized array, not the Set \u2014 at 25972 us per import at 8000 files, and no instrument on the #2901-#2909 branch could see it. It took a differential parity test over 211200 old-vs-new pairs to find. The arms added here would have caught THAT one on absolute ms (85 ms budget against a 20 ms arm; the unindexed resolver costs ~83000 ms on the same corpus), which is the argument for gating every registered language rather than only the ones a PR happens to touch. THE SECOND BLIND SPOT IS CLOSED, and this records what closing it changed. This harness used to call the inner resolvers with the NO-CONTEXT shape: run.ts calls provider.resolveImportTarget with five arguments, the fifth being { parsedFiles, parsedImport }, and resolveOne supplied three. resolveOne now makes the production call, newPass mints the ParsedFile[] FIRST and derives the path set from it exactly as run.ts does, and both legs behind the argument run on every import of their arms \u2014 PHP's named/alias function-or-const leg over filesByDirectory(context.parsedFiles), whose memo defeated measures 197.0 us -> 9976.2 us per import (50.6x), and Python's from-import submodule-precedence branch, the only spelling that reads context.parsedFiles at all. Fifteen of the seventeen arms cannot observe a context (their hooks declare three or four parameters) and are handed none, so their numbers did not move; which two CAN is now reconciled against SCOPE_RESOLVERS' hook arity rather than asserted in prose. NOTHING ELSE IN THIS FILE COULD HAVE GATED IT, which is why the context arm exists: on this corpus the leg AGREES with the cascade for every import, so all ten of PHP's and Python's fingerprints, their resolved counts and their distinct_outcomes are unchanged; a dropped context makes the timing arms FASTER and no arm here has a lower bound on ms; and the heap floor (0.5 x 49573840 = 24.8 MB) still passes the 37576816 B a no-context PHP pass reads. The arm is one import per language resolved through resolveOne twice, with and without the pass's parsedFiles, whose two answers must DIFFER and must both match what is recorded. WHAT REMAINS UNMEASURED, narrowed rather than deleted: Python's parsedFileByPath memo is exercised by the five timing arms and cannot be reached by the heap arm at all, because retainedPassBytes requires a probe that MISSES while every path that builds that memo returns a non-null packageTarget \u2014 so no ceiling bounds that Map (one pointer per parsed file, O(files), no depth term) and the contract test's count gate is what holds it to one build per pass. PHP's leg is measured with NO composer.json, so namespaceDirectories only ever returns the directory of an already-resolved file and the PSR-4 mapping branch stays unreached, exactly as csharp cannot reach the csproj leg; closing that is a second PHP arm on the csharp_csproj precedent, not a parameter. And the const tail of PHP's leg is a different ANSWER at the same cost \u2014 it runs the identical candidate gather and localDefs filter and diverges in the last two lines \u2014 so it is gated by count in test/unit/scope-resolution/import-target-index-reuse.contract.test.ts, which stays the gate to read alongside this file.", + "_depth_budget_note_2953": "javascript/typescript/vue moved from 2.0-2.1 to ~2.2 in #2953 and their budgets were raised to 2.6, which is a real shift with an understood cause rather than a loosened guard. Declared resolution never walks path components, so the deep arm's uniform d0/../d15/ prefix reaches these resolvers as the tsconfig baseUrl (see tsBaseUrlFor in measure.mjs) and every candidate string carries it: resolveFile probes ~11 extensions plus their /index forms, and hashing a 60-character path costs more than hashing a 12-character one. The growth is linear in path LENGTH and independent of file COUNT, which is what the ratio exists to bound - a resolver that started walking the corpus again would move scaling_ratio, not just this. Measured over three runs on a loaded box: js 2.109/2.257/2.240, ts 2.129/2.222/2.467, vue 2.116/2.151/2.102.", + "_heap_bound_note_2953": "javascript, typescript and vue moved from heap_reading_bytes/heap_ceiling_bytes to heap_bound_bytes in #2953. They retained 26745296 B (js, ts) and 28884016 B (vue) at 32000 files for a per-pass SuffixIndex over the whole file list; they now build no per-pass structure at all and read 0-16 B, because declared resolution derives nothing from the file set. That is a real saving rather than an arm that stopped measuring - the distinction this floor exists to make - and the evidence it is real is that the resolver fingerprints did NOT move: the same corpus resolves to the same targets, once the config it always implied is passed explicitly. The 1048576 B bound is rust's, chosen the same way: far above a 16 B reading, far below the index whose return it must catch." } diff --git a/gitnexus/bench/import-target/measure.mjs b/gitnexus/bench/import-target/measure.mjs index c4318d558..61aed7775 100644 --- a/gitnexus/bench/import-target/measure.mjs +++ b/gitnexus/bench/import-target/measure.mjs @@ -82,12 +82,27 @@ * component of the IMPORTER, so per-import cost is quadratic in path depth: * measured depth_ratio 7.39, by far the largest here, and the reason its * depth budget is 11 rather than the ~2 most languages carry. - * - javascript, typescript, vue: one resolver (`resolveTsTarget`) behind + * - javascript, typescript, vue: one resolver (`resolveTsModule`) behind * three adapters, so the three corpora are the same shape and differ only - * in what actually differs — the extension list (`.js` vs `.ts`) and, for - * Vue, the tsconfig alias branch (see `VUE_TSCONFIG`). All three are - * miss-dominated bare specifiers, because a relative import resolves by - * exact `Set.has` and never reaches the leg that had no index. + * in what actually differs — the extension list (`.js` vs `.ts`) and which + * config leg the arm exercises (`tsBaseUrlConfig` vs `vueTsconfig`). All + * three are miss-dominated bare specifiers. + * + * What they measure CHANGED with #2953. The leg used to be `suffixResolve`, + * a repo-wide search for a path ending in the specifier; these three no + * longer have it, and resolve only against a declared tsconfig mapping or a + * package manifest. Two consequences the numbers show: + * + * - the arms need a `resolutionConfig` to resolve anything at all. With + * none they all reported `resolved: 0` — every import correctly + * external — while still printing a clean scaling ratio, which is a + * bench measuring an empty branch and passing exactly like one + * measuring a full one. + * - `depth_ratio` is now structurally flat for them, and that is the + * result rather than a weakened arm: declared resolution never walks + * path components, so the `deep` arm's uniform prefix reaches the + * config (see `tsBaseUrlFor`) and its cost is the same keyed lookup the + * other arms pay. * - c, cpp: `resolveCppImportTarget` delegates to `resolveCImportTarget`, so * the two share a resolver and differ in extension set and in which adapter * builds the augmented set. Cost is a basename bucket walk with a @@ -196,9 +211,10 @@ * - of the eight added later, swift (3.28) and c/cpp (2.54/2.64) are the two * that scan a bucket, and they scan DIFFERENT buckets: swift's is the * module's own file list, which it returns, and C's is the basename bucket - * its suffix fallback walks. python, javascript, typescript and vue answer - * from keyed maps and sit at 1.03-1.10, so they keep the linear budget and - * that immunity is their assertion, exactly as for ruby and kotlin; + * its suffix fallback walks. python answers from keyed maps and sits at + * 1.03-1.10, and javascript, typescript and vue answer from a declared + * config (#2953) — so all four keep the linear budget and that immunity is + * their assertion, exactly as for ruby and kotlin; * - rust's collide arm is the one that is NOT a shared-leaf layout, and the * reason is in the list above: file count is not an axis its cost has, so a * shared-leaf rust arm would have been an arm that cannot fail. Its collide @@ -345,7 +361,7 @@ * inventory arm at the foot of the file reads * `SCOPE_RESOLVERS.get(language).resolveImportTarget.length` and reconciles it * against `CONTEXT_LANGS` in both directions, so a language that grows a - * context leg cannot ship with the leg unmeasured. The other fifteen are handed + * context leg cannot ship with the leg unmeasured. The other fourteen are handed * nothing and build no `ParsedFile[]` at all, so their numbers are unmoved. * * `newPass` mints the `ParsedFile[]` FIRST and derives the path set from it @@ -369,9 +385,9 @@ * `parsedFiles` and once without, whose two answers must DIFFER and must both * equal what baselines.json records. Dropping the fifth argument, dropping * `importedSymbolKind`, or reverting Python to `namespace` collapses the two - * onto one value and fails — and no other arm here would: on the main corpus - * the leg agrees with the cascade, so both languages' fingerprints are - * UNCHANGED by this (measured, all ten). + * onto one value and fails. PHP and Python still agree with their fallback on + * the main corpus; Java deliberately has no context-free fallback, so its + * fingerprints move to the declared-package answers recorded here. * * WHAT IS STILL NOT MEASURED, narrowed rather than deleted: * @@ -446,7 +462,7 @@ import { resolveRubyImportTarget } from '../../src/core/ingestion/languages/ruby import { resolveCsharpImportTarget } from '../../src/core/ingestion/languages/csharp/import-target.ts'; import { resolveKotlinImportTarget } from '../../src/core/ingestion/languages/kotlin/import-target.ts'; import { resolvePhpImportTargetInternal } from '../../src/core/ingestion/languages/php/import-target.ts'; -import { resolveJavaImportTarget } from '../../src/core/ingestion/languages/java/import-target.ts'; +import { javaScopeResolver } from '../../src/core/ingestion/languages/java/scope-resolver.ts'; import { cobolScopeResolver } from '../../src/core/ingestion/languages/cobol/scope-resolver.ts'; import { resolveSwiftImportTarget } from '../../src/core/ingestion/languages/swift/import-target.ts'; import { resolveRustImportTarget } from '../../src/core/ingestion/languages/rust/import-target.ts'; @@ -556,7 +572,6 @@ const HEAP_BUDGETED = [ 'ruby', 'php', 'java', - 'javascript', 'python', 'c', // Promoted once every language was actually measured. Each retains a real @@ -580,24 +595,35 @@ const HEAP_BUDGETED = [ 'kotlin', 'dart', 'go', - 'typescript', - 'vue', 'cpp', ]; +// javascript, typescript and vue were budgeted here until #2953 and are now +// BOUNDED, which is a demotion in gate strength and a promotion in what the +// number means. They retained ~26.7 MiB each because they built a per-pass +// `SuffixIndex` over the whole file list; they no longer build one at all, +// because declared resolution derives nothing from the file set — a candidate +// comes from a tsconfig mapping or a manifest and is checked with one +// `Set.has`. The readings are 0-16 B. +// +// A floor over a reading at or below its own noise gates the noise, which is +// the same reason rust sits in this tier at 16 B — so they take a bound and no +// floor. The bound is what still matters: it catches these three growing an +// index again, which is the re-entry condition for the cost #2911 and #1918 +// were about. /** * The arms handed the fifth `context` argument — `{ parsedFiles, parsedImport }` - * — because their registered hook DECLARES it. Two of seventeen, and the + * — because their registered hook DECLARES it. Three of seventeen, and the * inventory arm at the foot of this file reconciles that claim against * `SCOPE_RESOLVERS` in both directions rather than trusting this line. * * These are also the only arms for which `newPass` builds a `ParsedFile[]` at - * all. Building one for the other fifteen would cost their timed loop an + * all. Building one for the other fourteen would cost their timed loop an * O(files) allocation per pass that no resolver of theirs can even observe — * their hooks declare three or four parameters — so their numbers stay exactly * where they were. */ -const CONTEXT_LANGS = ['php', 'python']; +const CONTEXT_LANGS = ['php', 'java', 'python']; /** * Needs `node --expose-gc` to force collection for a clean delta; without it @@ -649,17 +675,57 @@ const CSPROJ_CONFIGS = [ { rootNamespace: 'Lib', projectDir: '' }, ]; /** - * The `tsconfigPaths` the Vue arm threads as `resolutionConfig`. + * The `resolutionConfig` the ts-family arms thread (#2953). * - * The Vue adapter is `resolveTsTarget` with `language: TypeScript` and nothing - * else, so with a null config its arm would be a byte-for-byte re-run of the - * TypeScript one over a differently-spelled corpus. The alias branch - * (`standard.ts:57-70`) is the one leg of the shared resolver that neither the - * `javascript` arm (which pins `tsconfigPaths: null`) nor the `typescript` arm - * here reaches, so wiring it is what makes this a third measurement rather than - * a third copy — and every local Vue import below is spelled `@/…`. + * These three used to run with `tsconfigPaths: null` for javascript and + * typescript and an alias map for vue, because the leg being measured was + * `suffixResolve` — a repo-wide search for a path ending in the specifier, + * which needs no configuration to answer and answered even when nothing + * declared the import. #2953 deleted that leg for the ts family: a specifier + * now resolves only against a declared tsconfig mapping or a package manifest. + * + * With no config, therefore, all three arms resolve NOTHING — every import is + * correctly external — and the bench measures an empty branch while reporting a + * perfect scaling ratio. A bench that measures nothing passes exactly like one + * that measures something, so each arm is given the config its corpus is + * spelled for, and the two configs cover the two legs the new resolver has: + * + * - `TS_BASE_URL` — `baseUrl` at the repo root, so `src/mod3/file7` resolves + * the way a `baseUrl` project's absolute import does. Used by javascript and + * typescript. + * - `vueTsconfig` — a `paths` PATTERN, which is a different branch: + * longest-prefix selection and `*` substitution, then a candidate probe per + * target. Every local Vue import below is spelled `@/…`, so the vue arm + * stays a third measurement rather than a third copy — the same role it had + * before, now against the branch that replaced the alias rewrite. + * + * Not covered here: the workspace-manifest leg + * (`node-workspace-packages.ts`), which is a `Map.get` on a package name and + * does not scale with the file set. */ -const VUE_TSCONFIG = { tsconfigPaths: { aliases: new Map([['@/', 'src/']]), baseUrl: '.' } }; +const tsBaseUrlConfig = (baseUrl) => ({ + tsconfigs: { scopes: [{ dir: '', baseUrl, paths: [] }] }, + nodeWorkspacePackages: null, +}); +const vueTsconfig = (baseUrl) => ({ + tsconfigs: { + scopes: [ + { dir: '', baseUrl, paths: [{ pattern: '@/*', targets: [joinBase(baseUrl, 'src/*')] }] }, + ], + }, + nodeWorkspacePackages: null, +}); +const joinBase = (baseUrl, rest) => (baseUrl === '' ? rest : `${baseUrl}/${rest}`); +/** + * The `deep` arm prepends a UNIFORM `d0/…/d15/` prefix to every path + * (`buildFiles`), and the import spellings do not change. Under the old suffix + * matcher that was the point: the resolver walked path components, so depth was + * the cost. Declared resolution never walks — the config names an exact base — + * so the prefix has to reach the config or the whole arm resolves nothing and + * measures the miss path at depth instead of the hit path at depth. + */ +const tsBaseUrlFor = (pad) => + pad === 0 ? '' : Array.from({ length: pad }, (_, n) => `d${n}`).join('/'); /** Keyed by LAYOUT name, so there is no `csharp_csproj` row: `buildFiles` * aliases that arm to `csharp` before this table is read. */ const EXTENSION = { @@ -1022,8 +1088,26 @@ const probeFile = (filePath, defs) => ({ referenceSites: [], }); +const javaProbeFile = (filePath, packageName) => ({ + ...probeFile(filePath, []), + captureSideChannel: { + kind: 'java', + packageFact: { status: 'known', packageName }, + classAnnotations: [], + }, +}); + +function javaBenchmarkPackage(filePath) { + const uniquePackage = /\/com\/example\/(pkg\d+)(?:\/|$)/.exec(`/${filePath}`)?.[1]; + if (uniquePackage !== undefined) return `com.example.${uniquePackage}`; + + return /\/svc\d+\/.*\/com\/example\/model(?:\/|$)/.test(`/${filePath}`) + ? 'com.example.model' + : ''; +} + /** - * The `ParsedFile[]` the orchestrator threads beside the path set, for the two + * The `ParsedFile[]` the orchestrator threads beside the path set, for the three * languages whose hook declares a `context` — see `CONTEXT_LANGS`. * * Two defs per file, and both are real shapes rather than padding. PHP keeps @@ -1045,6 +1129,10 @@ const probeFile = (filePath, defs) => ({ function buildParsedFiles(lang, files) { const parsedFiles = []; for (const filePath of files) { + if (lang === 'java') { + parsedFiles.push(javaProbeFile(filePath, javaBenchmarkPackage(filePath))); + continue; + } const slash = filePath.lastIndexOf('/'); const stem = filePath.slice(slash + 1, filePath.lastIndexOf('.')); const parent = slash < 0 ? '' : filePath.slice(0, slash); @@ -1358,17 +1446,12 @@ function collideTarget(lang, { local, r, d, j, dirs }) { : `Vendor${(r >>> 4) % 97}\\Ghost\\Missing`; } if (lang === 'java') { - // `com.svc{d}.model` matches no directory, but its LAST segment is the one - // every directory now ends in, so `firstFileDirectlyInPkgDir` walks the - // whole `model` bucket twice — at the direct match and again after the - // first strip — before the third strip finds `model` on its own. That walk - // is the non-constant term this arm exists to measure. The `d % 7` slice - // used to import `com.svc{d}.vendor`, which buckets to nothing, mirroring - // the unique arm's nested slice — which missed until #2881 and resolves - // now, so the mirror follows it or the same-workload invariant below breaks. + // Every file declares the same package despite living under different + // service paths. Exact and wildcard imports therefore exercise one growing + // declared-package bucket without relying on directory layout. return local ? (r >>> 3) % 3 === 0 - ? `com.svc${d}.model.*` + ? 'com.example.model.*' : `com.example.model.File${j}` : (r >>> 3) % 2 === 0 ? ['java.util.List', 'java.io.IOException', 'java.util.concurrent.ConcurrentHashMap'][ @@ -1541,16 +1624,25 @@ function buildRepo(lang, fileCount, pad = 0, shape = 'unique') { * `perFileSet` memos keyed on this ARRAY's identity, so reusing one array would * hide their build from rep 2 onward and `fastest()` reports the minimum. */ -function newPass(lang, files) { +function newPass(lang, files, pad = 0) { if (HEADER_EXTENSION[lang] !== undefined) { const sources = []; const headers = []; for (const f of files) (f.endsWith(HEADER_EXTENSION[lang]) ? headers : sources).push(f); return { allFilePaths: new Set(sources), config: new Set(headers) }; } - if (lang === 'vue') return { allFilePaths: new Set(files), config: VUE_TSCONFIG }; + if (lang === 'vue') { + return { allFilePaths: new Set(files), config: vueTsconfig(tsBaseUrlFor(pad)) }; + } + // javascript and typescript resolve their `src/mod{d}/file{j}` locals through + // `baseUrl`; without a config every arm would correctly resolve nothing and + // measure an empty branch (#2953 — see `tsBaseUrlConfig`). + if (lang === 'javascript' || lang === 'typescript') { + return { allFilePaths: new Set(files), config: tsBaseUrlConfig(tsBaseUrlFor(pad)) }; + } if (CONTEXT_LANGS.includes(lang)) { const parsedFiles = buildParsedFiles(lang, files); + restoreBenchmarkSideChannels(lang, parsedFiles); return { allFilePaths: new Set(parsedFiles.map((f) => f.filePath)), config: undefined, @@ -1569,12 +1661,20 @@ function newPass(lang, files) { * to prove the arm can tell the two call shapes apart. */ const contextFor = (pass, parsedImport) => - pass.parsedFiles === undefined ? undefined : { parsedFiles: pass.parsedFiles, parsedImport }; + pass.parsedFiles === undefined + ? undefined + : { parsedFiles: pass.parsedFiles, parsedImport, filesSkipped: 0 }; + +function restoreBenchmarkSideChannels(lang, parsedFiles) { + if (lang !== 'java') return; + javaScopeResolver.loadResolutionConfig?.(''); + for (const parsed of parsedFiles) javaScopeResolver.applyCaptureSideChannel?.(parsed); +} /** The timed loop. One `newPass` per pass, so every pass pays exactly one index * build — see `newPass`. */ -function resolveAll(lang, files, imports) { - const pass = newPass(lang, files); +function resolveAll(lang, files, imports, pad = 0) { + const pass = newPass(lang, files, pad); let sink = 0; for (const [from, target] of imports) { const hit = resolveOne(lang, from, target, pass); @@ -1624,9 +1724,18 @@ function resolveOne(lang, from, target, pass) { ); } if (lang === 'java') { - return resolveJavaImportTarget( - { kind: 'named', localName: 'X', importedName: 'X', targetRaw: target }, - { fromFile: from, allFilePaths }, + const parsedImport = { + kind: 'named', + localName: 'X', + importedName: 'X', + targetRaw: target, + }; + return javaScopeResolver.resolveImportTarget( + target, + from, + allFilePaths, + pass.config, + contextFor(pass, parsedImport), ); } // The `ScopeResolver` hook itself — COBOL's copy index has no other export. @@ -1665,7 +1774,9 @@ function resolveOne(lang, from, target, pass) { { fromFile: from, allFilePaths, parsedFiles: pass.parsedFiles }, ); } - if (lang === 'javascript') return jsResolveImportTarget(target, from, allFilePaths); + if (lang === 'javascript') { + return jsResolveImportTarget(target, from, allFilePaths, pass.config); + } if (lang === 'vue') return vueResolveImportTarget(target, from, allFilePaths, pass.config); // TypeScript, C and C++ go through the registered `ScopeResolver` hook rather // than an inner resolver, because for all three the thing under test lives IN @@ -1674,7 +1785,7 @@ function resolveOne(lang, from, target, pass) { // is private to theirs. Calling past it would benchmark a copy of the adapter // instead of the adapter. if (lang === 'typescript') { - return typescriptScopeResolver.resolveImportTarget(target, from, allFilePaths, undefined); + return typescriptScopeResolver.resolveImportTarget(target, from, allFilePaths, pass.config); } if (lang === 'c') { return cScopeResolver.resolveImportTarget(target, from, allFilePaths, pass.config); @@ -1708,8 +1819,8 @@ function resolveOne(lang, from, target, pass) { * Deliberately NOT shared with `resolveAll`, which is the TIMED loop: the memo * that makes this pass cheap is exactly what would hide the cost that loop * exists to measure. */ -function identityPass(lang, files, imports) { - const pass = newPass(lang, files); +function identityPass(lang, files, imports, pad = 0) { + const pass = newPass(lang, files, pad); const outcomes = new Set(); const wasNullByKey = new Map(); let resolved = 0; @@ -1747,12 +1858,12 @@ function fastest(values) { return Math.min(...values); } -function timeResolution(lang, files, imports, reps) { - for (let w = 0; w < WARMUP; w++) resolveAll(lang, files, imports); +function timeResolution(lang, files, imports, reps, pad = 0) { + for (let w = 0; w < WARMUP; w++) resolveAll(lang, files, imports, pad); const samples = []; for (let r = 0; r < reps; r++) { const t0 = performance.now(); - resolveAll(lang, files, imports); + resolveAll(lang, files, imports, pad); samples.push(performance.now() - t0); } return fastest(samples); @@ -1774,10 +1885,10 @@ function timeResolution(lang, files, imports, reps) { * reads several times high, which would push the expensive languages to * `REPS_MIN` for the wrong reason. */ -function probeMs(lang, files, imports) { - for (let w = 0; w < WARMUP; w++) resolveAll(lang, files, imports); +function probeMs(lang, files, imports, pad = 0) { + for (let w = 0; w < WARMUP; w++) resolveAll(lang, files, imports, pad); const t0 = performance.now(); - resolveAll(lang, files, imports); + resolveAll(lang, files, imports, pad); return performance.now() - t0; } @@ -1813,8 +1924,8 @@ function probeMs(lang, files, imports) { * Set, which is part of what they hold; for every language it includes the one * or two resolve-cache entries the probe leaves behind. */ -function retainedPassBytes(lang, files, probeTarget) { - const pass = newPass(lang, files); +function retainedPassBytes(lang, files, probeTarget, pad = 0) { + const pass = newPass(lang, files, pad); // See `HEAP_RETAINED`: nothing built for this language is released until the // next one starts, so no deferred collection can land between the two samples // below and cancel part of the delta. @@ -1985,12 +2096,11 @@ function measureHeap(lang) { * of files carrying ONE import whose answer DIFFERS between the production * five-argument call and the three-argument one this harness used to make. * - * That difference is the whole arm. The main corpus cannot serve as one: there - * the leg AGREES with the cascade for every import (measured — both languages' - * ten fingerprints are unchanged by threading the context), which is the right - * outcome for a corpus built to measure cost, and useless for proving the - * context arrives. Timing cannot prove it either; a dropped context makes the - * arms FASTER, and nothing here has a lower bound on ms. + * That difference is the whole arm. PHP and Python need it because their main + * corpus answers agree with the fallback. Java's main fingerprint also catches + * a dropped context, but this tiny positive probe isolates the adapter contract + * from aggregate corpus changes. Timing cannot prove any of these; a dropped + * context makes the arms faster, and nothing here has a lower bound on ms. * * Both are resolved THROUGH `resolveOne`, not through the resolvers directly, * because what is under test is this file's threading rather than the @@ -2018,6 +2128,15 @@ const CONTEXT_PROBE = { probeFile('src/App/Ns0/Helpers.php', [['Function', 'App\\Ns0\\Dup']]), ], }, + /** A declared package resolves its type only when the parsed workspace arrives. */ + java: { + from: 'app/Main.java', + target: 'com.example.model.User', + parsedFiles: [ + javaProbeFile('app/Main.java', 'app'), + javaProbeFile('weird/path/User.java', 'com.example.model'), + ], + }, /** * `from pkg import X`, with `pkg/__init__.py` exporting `X` AND a same-named * submodule `pkg/X.py` beside it — the precedence CPython documents and the @@ -2051,10 +2170,12 @@ const CONTEXT_PROBE = { function measureContext(lang) { const { from, target, parsedFiles } = CONTEXT_PROBE[lang]; const allFilePaths = new Set(parsedFiles.map((f) => f.filePath)); - const answer = (files) => - renderResolved( + const answer = (files) => { + restoreBenchmarkSideChannels(lang, files ?? []); + return renderResolved( resolveOne(lang, from, target, { allFilePaths, config: undefined, parsedFiles: files }), ); + }; return { target, with_context: answer(parsedFiles), @@ -2169,8 +2290,8 @@ for (const lang of LANGS) { let reps = null; for (const [name, fileCount, pad, shape] of ARMS) { const { files, imports } = buildRepo(lang, fileCount, pad, shape); - const { outcomes, resolved } = identityPass(lang, files, imports); - if (reps === null) reps = repsFor(probeMs(lang, files, imports)); + const { outcomes, resolved } = identityPass(lang, files, imports, pad); + if (reps === null) reps = repsFor(probeMs(lang, files, imports, pad)); scales[name] = { files: files.length, imports: imports.length, @@ -2178,7 +2299,7 @@ for (const lang of LANGS) { // resolved share would still produce a "valid" fingerprint over far less. resolved, distinct_outcomes: outcomes.size, - ms: Number(timeResolution(lang, files, imports, reps).toFixed(3)), + ms: Number(timeResolution(lang, files, imports, reps, pad).toFixed(3)), fingerprint: fingerprint(outcomes), }; } @@ -2217,7 +2338,7 @@ for (const lang of LANGS) { // phase goes 2.06 s -> 3.43 s, of which kotlin alone is 0.57 s. See COST. for (const lang of LANGS) report[lang].heap = measureHeap(lang); -// Deterministic and microseconds — it resolves six imports over two three-file +// Deterministic and microseconds — it resolves six imports over three tiny // corpora — so unlike the heap arm it neither needs nor deserves isolation from // the timing phase. It runs last only because it reads best beside the heap arm // in the report. @@ -2765,7 +2886,7 @@ expectNoOrphanKeys( // against a claim in a comment. `run.ts` passes the fifth argument to every // provider; which ones can OBSERVE it is decided by how many parameters each // hook declares, and that is a number the registry can be asked for. Today -// exactly two answer 5 (php, python) and the other fourteen answer 3 or 4 — +// exactly three answer 5 (php, java, python) and the other fourteen answer 3 or 4 — // which is why fourteen arms could ignore this whole question and their numbers // did not move when it was fixed. // diff --git a/gitnexus/bench/scope-capture/baselines.json b/gitnexus/bench/scope-capture/baselines.json index 817029c5c..9cc89b431 100644 --- a/gitnexus/bench/scope-capture/baselines.json +++ b/gitnexus/bench/scope-capture/baselines.json @@ -6,11 +6,11 @@ "_rebaselined_callable_flow_2522_review": "PR #2522 review hardening: callable operands retain expression/qualified identity and formals retain signature metadata. Prior 3d4e32e7490c830516126e28931827949baa3594cb521f7a3d8dcfed95b6018a -> 57b3c55135af8d2af33b9a7c4bf89796a7bee5b5822b402a2dea91af7232cf4a; scaling 1.058 < 1.5.", "_rebaselined_callable_flow_2522_followup": "PR #2522 follow-up: provider-owned callable assignment/copy/formal/argument/invoke facts with invocation/constructor-result suppression. Prior 09ecd94911b830f52fa8807560abcbd79f163d02a2072870c1a59297e9a326e1 -> 3d4e32e7490c830516126e28931827949baa3594cb521f7a3d8dcfed95b6018a; scaling 1.039 < 1.5.", "_rebaselined": "#1976: F33 generic composite literal constructor inference adds generic_type captures in composite_literal patterns; fingerprint drift expected.", - "_rebaselined_receiver_chain_2747": "#2747 receiver-chain rollout: call matches whose receiver is itself an expression now carry `@reference.receiver-chain`, a compact encoding of the receiver's structure, so resolution types it by folding instead of re-parsing receiver source text. Capture GROUP counts are unchanged — the tag is added to existing call matches, never a new match — so this is digest drift only. Prior 57b3c55135af8d2af33b9a7c4bf89796a7bee5b5822b402a2dea91af7232cf4a -> 5d6c59c2f2c0dd937c53bf5d736e0f8376b2899a381e488a33aec23524823efb.", + "_rebaselined_receiver_chain_2747": "#2747 receiver-chain rollout: call matches whose receiver is itself an expression now carry `@reference.receiver-chain`, a compact encoding of the receiver's structure, so resolution types it by folding instead of re-parsing receiver source text. Capture GROUP counts are unchanged \u2014 the tag is added to existing call matches, never a new match \u2014 so this is digest drift only. Prior 57b3c55135af8d2af33b9a7c4bf89796a7bee5b5822b402a2dea91af7232cf4a -> 5d6c59c2f2c0dd937c53bf5d736e0f8376b2899a381e488a33aec23524823efb.", "_rebaselined_2766_go_pointer_receiver_fixture": "#2766: added test/fixtures/lang-resolution/go-pointer-receiver-field-chain/ (2 Go files) as the committed regression fixture for pointer-receiver base resolution. Go fixture_count 100 -> 102. Prior 5d6c59c2f2c0dd937c53bf5d736e0f8376b2899a381e488a33aec23524823efb -> 8cba537ff211fab3bac5fb4456cd1ffba14d6a2db75c40acae28ab8bf29f3d2e. FIXTURE-CORPUS GROWTH, NOT A CAPTURE CHANGE: the accompanying fix is a resolution-time lookup fallback (stripTypePreservingDecoration) and cannot move capture output; go was the ONLY language whose fingerprint drifted, and every other language matched its baseline on the same run.", - "_rebaselined_2766_receiver_chain_wire_v2": "#2766: receiver-chain wire format v1 -> v2 (name-free `await` / `index` step kinds). The VERSION prefix is part of every emitted `@reference.receiver-chain` capture, so every chain-minting language's capture text changed. WIRE-FORMAT CHANGE, NOT A CAPTURE-SET CHANGE: the same chains are minted for the same sites, spelled `2|…` instead of `1|…`. Exactly the 12 chain-minting languages drifted; c, cobol and dart did not, which is the check that this is the prefix and not a capture regression. Accompanied by SCHEMA_BUMP 34 -> 37 and INCREMENTAL_SCHEMA_VERSION 28 -> 31 so a stale index is rejected rather than replaying chains a v2 decoder refuses. Prior 8cba537ff211fab3bac5fb4456cd1ffba14d6a2db75c40acae28ab8bf29f3d2e -> 8162272bb897b0b89472c406321cf8d88a5ae4ea83ea9e3c45f8e817041bff9f.", - "_rebaselined_2766_await_subscript_emission": "#2766: extractMixedChain now walks THROUGH await and subscript nodes and peels transparent wrappers at loop entry, so sites whose receiver is `repos[0]` or `(await f())` mint a receiver chain where they previously minted none. EMISSION CHANGE: more sites carry `@reference.receiver-chain`; no existing chain changed shape. Only go and kotlin drifted of 15 — the two whose fixture corpora contain such receivers. Prior 8162272bb897b0b89472c406321cf8d88a5ae4ea83ea9e3c45f8e817041bff9f -> c9c908f441e3be12fad2448120ed3ea35dc235a12b3f63b0ec532ffdae11d9e9.", - "_rebaselined_2766_phantom_callee_read_site": "#2766: Go's `@reference.read` pattern matches EVERY selector_expression, so a member call `h.dep.Work()` minted THREE sites — the call, the genuine `h.dep` field read, and a PHANTOM read on the callee `h.dep.Work`. The phantom resolved through findOwnedMember (which prefers methods over fields) and emitted an ACCESSES edge to the METHOD duplicating the CALLS edge at the same position; visible today on any receiver the text cascade can type (`RunFromValueReceiver -> DoWork`). The emitter now drops a read match whose selector is in FUNCTION position. FEWER capture matches for Go, no other language affected — go was the only fingerprint of 15 that moved. A method VALUE (`f := h.dep.Work`) is not in function position and is untouched. Prior c9c908f441e3be12fad2448120ed3ea35dc235a12b3f63b0ec532ffdae11d9e9 -> 7bb524a32a2eed57a15b454e3a33480e92a496c683e6856ef02179693c0e02e3.", + "_rebaselined_2766_receiver_chain_wire_v2": "#2766: receiver-chain wire format v1 -> v2 (name-free `await` / `index` step kinds). The VERSION prefix is part of every emitted `@reference.receiver-chain` capture, so every chain-minting language's capture text changed. WIRE-FORMAT CHANGE, NOT A CAPTURE-SET CHANGE: the same chains are minted for the same sites, spelled `2|\u2026` instead of `1|\u2026`. Exactly the 12 chain-minting languages drifted; c, cobol and dart did not, which is the check that this is the prefix and not a capture regression. Accompanied by SCHEMA_BUMP 34 -> 37 and INCREMENTAL_SCHEMA_VERSION 28 -> 31 so a stale index is rejected rather than replaying chains a v2 decoder refuses. Prior 8cba537ff211fab3bac5fb4456cd1ffba14d6a2db75c40acae28ab8bf29f3d2e -> 8162272bb897b0b89472c406321cf8d88a5ae4ea83ea9e3c45f8e817041bff9f.", + "_rebaselined_2766_await_subscript_emission": "#2766: extractMixedChain now walks THROUGH await and subscript nodes and peels transparent wrappers at loop entry, so sites whose receiver is `repos[0]` or `(await f())` mint a receiver chain where they previously minted none. EMISSION CHANGE: more sites carry `@reference.receiver-chain`; no existing chain changed shape. Only go and kotlin drifted of 15 \u2014 the two whose fixture corpora contain such receivers. Prior 8162272bb897b0b89472c406321cf8d88a5ae4ea83ea9e3c45f8e817041bff9f -> c9c908f441e3be12fad2448120ed3ea35dc235a12b3f63b0ec532ffdae11d9e9.", + "_rebaselined_2766_phantom_callee_read_site": "#2766: Go's `@reference.read` pattern matches EVERY selector_expression, so a member call `h.dep.Work()` minted THREE sites \u2014 the call, the genuine `h.dep` field read, and a PHANTOM read on the callee `h.dep.Work`. The phantom resolved through findOwnedMember (which prefers methods over fields) and emitted an ACCESSES edge to the METHOD duplicating the CALLS edge at the same position; visible today on any receiver the text cascade can type (`RunFromValueReceiver -> DoWork`). The emitter now drops a read match whose selector is in FUNCTION position. FEWER capture matches for Go, no other language affected \u2014 go was the only fingerprint of 15 that moved. A method VALUE (`f := h.dep.Work`) is not in function position and is untouched. Prior c9c908f441e3be12fad2448120ed3ea35dc235a12b3f63b0ec532ffdae11d9e9 -> 7bb524a32a2eed57a15b454e3a33480e92a496c683e6856ef02179693c0e02e3.", "_rebaselined_2766_callee_position_marker": "#2766 review fix: a call's callee selector is no longer DROPPED at capture. An earlier commit on this branch dropped it outright, which also deleted the genuine field read on a func-typed struct field (`h.dep.Work()` where `Work func() error`) - callback/hook/mock structs lost their only ACCESSES evidence. The match is now emitted carrying `@reference.callee-position`, and the phantom is suppressed at EMIT by the resolved target's kind instead. Go only: the other 14 languages' fingerprints are byte-identical, which is the check that this is not a cross-language capture change. Prior 7bb524a32a2eed57a15b454e3a33480e92a496c683e6856ef02179693c0e02e3 -> e47302079e17a5e73711bbed5416557b49327cb67e4932008700ec6b8fb468b3; scaling 1.001 < 1.5; fixtures 102 (unchanged), capture_groups_fp 2103.", "_rebaselined_2813_interface_field_dispatch_fixture": "#2813: added test/fixtures/lang-resolution/go-interface-field-dispatch/ (8 Go files) as the committed regression fixture for calls through an interface-typed struct field. Go fixture_count 102 -> 110. FIXTURE-CORPUS GROWTH, NOT A CAPTURE CHANGE: the accompanying fixes are a detection-time method-set change (interface-impls.ts) and a resolution-time fan-out in the shared receiver pass, neither of which emits captures; go/query.ts and go/captures.ts are untouched. Go was the ONLY language whose fingerprint drifted, and every other language matched its baseline on the same run - the same check used for the #2766 fixture growth above. Prior e47302079e17a5e73711bbed5416557b49327cb67e4932008700ec6b8fb468b3 -> cffee41cadbf350855d99bd5aee7c015b1e8b31d1c343d02f113540abe86c765; scaling 1.074 < 1.5, capture_groups_fp 2303.", "_rebaselined_2837": "#2837: Go struct/interface captures re-anchored from the type_declaration onto the type_spec (@scope.class/@declaration.struct/@declaration.interface in languages/go/query.ts, @definition.struct/@definition.interface in GO_QUERIES). A grouped `type (...)` block used to yield ONE scope and ONE node for every type in it, so each type after the first lost its field typeBindings and every field-receiver call in the file emitted nothing. Capture COUNT is unchanged; only ranges moved, plus the new go-grouped-type-decl fixture. Prior c27fb803598581fa4eb7ddf5ef6f8369b9e3a150082d11362e7aa3ec8faaa832 -> e386598526e502d131e52a17d219635b3a4196d94f1ebdd25922a2582c985d18; scaling 1.054 < 1.5.", @@ -18,11 +18,11 @@ }, "cobol": { "fingerprint": "c8c00b56a7da24e04080eb885714fbbf45e3903324f0cf9df0754f5b5a92e3aa", - "_rebaselined_2813_exact_method_sets": "#2813: Go embedded fields now emit `@reference.embedded-pointer` when spelled `*T` rather than `T`. A CAPTURE-EMISSION CHANGE, not fixture growth: fixture_count is unchanged at 110 and capture_groups_fp moves 2303 -> 2339 (+36), which is the new marker plus the WrongSigRepo/Recount rows added to two existing fixture files. The marker is required for exactness — Go gives `struct{ Base }` and `struct{ *Base }` different method sets, so structural interface satisfaction cannot be correct without knowing which was written (go.dev/ref/spec#Struct_types). Go was the ONLY language of 15 whose fingerprint moved, which is the check that this is a Go capture change and not a cross-language regression. Accompanied by SCHEMA_BUMP 39 -> 43 (skipping 40/41/42, taken by origin/main during review) so a warm cache cannot replay the pre-marker capture set. Prior cffee41cadbf350855d99bd5aee7c015b1e8b31d1c343d02f113540abe86c765 -> c27fb803598581fa4eb7ddf5ef6f8369b9e3a150082d11362e7aa3ec8faaa832; scaling 0.987 < 1.5.", + "_rebaselined_2813_exact_method_sets": "#2813: Go embedded fields now emit `@reference.embedded-pointer` when spelled `*T` rather than `T`. A CAPTURE-EMISSION CHANGE, not fixture growth: fixture_count is unchanged at 110 and capture_groups_fp moves 2303 -> 2339 (+36), which is the new marker plus the WrongSigRepo/Recount rows added to two existing fixture files. The marker is required for exactness \u2014 Go gives `struct{ Base }` and `struct{ *Base }` different method sets, so structural interface satisfaction cannot be correct without knowing which was written (go.dev/ref/spec#Struct_types). Go was the ONLY language of 15 whose fingerprint moved, which is the check that this is a Go capture change and not a cross-language regression. Accompanied by SCHEMA_BUMP 39 -> 43 (skipping 40/41/42, taken by origin/main during review) so a warm cache cannot replay the pre-marker capture set. Prior cffee41cadbf350855d99bd5aee7c015b1e8b31d1c343d02f113540abe86c765 -> c27fb803598581fa4eb7ddf5ef6f8369b9e3a150082d11362e7aa3ec8faaa832; scaling 0.987 < 1.5.", "scaling_budget": 1.5, "_rebaselined_callable_flow_2522_followup": "PR #2522 follow-up: COBOL procedure-pointer callable flow facts; multi-topic extraction now consumes each grouped scope/declaration match once instead of requiring a duplicate declaration-only match. Prior 68ee0e95eb9f86f2d92ca35f730f4c2d4d83abc1b5241ae767ff3437780ec8d1 -> d45bb091b0893d0de4fae2486b31ba21719c9377bf35a0908fd3a36fa1c3bf4e; scaling 0.853 < 1.5.", "_note": "Updated for F17-F23 fixes (P2: TIMES guard, ADD GIVING, SQL AS alias). See PR #1959.", - "_rebaselined_2793_declaratives": "PR #2793: corpus-only re-baseline. `cobol-declaratives` was added to test/fixtures/lang-resolution to reproduce the `Namespace→Record` analyze abort (DECLARATIVES / USE AFTER STANDARD ERROR ON ), and this bench globs `lang-resolution/cobol-*`, so the corpus grew 14 -> 15 files. Verified capture-neutral: with that one fixture moved aside the fingerprint is byte-identical to the prior d45bb091b0893d0de4fae2486b31ba21719c9377bf35a0908fd3a36fa1c3bf4e. No COBOL capture code changed in that PR. Scaling 0.677 < 1.5." + "_rebaselined_2793_declaratives": "PR #2793: corpus-only re-baseline. `cobol-declaratives` was added to test/fixtures/lang-resolution to reproduce the `Namespace\u2192Record` analyze abort (DECLARATIVES / USE AFTER STANDARD ERROR ON ), and this bench globs `lang-resolution/cobol-*`, so the corpus grew 14 -> 15 files. Verified capture-neutral: with that one fixture moved aside the fingerprint is byte-identical to the prior d45bb091b0893d0de4fae2486b31ba21719c9377bf35a0908fd3a36fa1c3bf4e. No COBOL capture code changed in that PR. Scaling 0.677 < 1.5." }, "c": { "fingerprint": "3418cded9f7072152f68992f0a426f43ae7d9d553579a47075fc0cab185848a5", @@ -30,15 +30,15 @@ "_rebaselined_callable_flow_2522_review": "PR #2522 review hardening: callable operands retain expression/qualified identity and formals retain signature metadata. Prior 57fee292147ae6d2db7062da1e07d17122cf355207c8967fa85fd2ec9ca398a4 -> 3418cded9f7072152f68992f0a426f43ae7d9d553579a47075fc0cab185848a5; scaling 1.073 < 1.5.", "_rebaselined_callable_flow_2522_followup": "PR #2522 follow-up: C function-pointer signatures plus direct-callee argument metadata and invocation-result suppression. Prior 75bcdbbf006bf9bd263c0f5857461b118f39b164e9f821cb0651ad0ec46ef6ae -> 57fee292147ae6d2db7062da1e07d17122cf355207c8967fa85fd2ec9ca398a4; scaling 1.035 < 1.5.", "_rebaselined_callable_flow": "Callable-value-flow facts for C function pointers, copies, pointer-to-pointer cells, arguments, and indirect invokes. Prior 12a196b2d6249c8d86a931b12ecebc2a0cdf8d6f47683acdd0d8e9d8bc7657f5 -> 75bcdbbf006bf9bd263c0f5857461b118f39b164e9f821cb0651ad0ec46ef6ae; measured scaling ratio 0.980 < 1.5.", - "_added": "#1956: c added to the scope-capture bench (was UNBENCHED). C has no inheritance — flat scale source. Adding it exposed + fixed a pre-existing O(n^2) findNodeAtRange root-walk in c/captures.ts (threaded c.node, byte-identical over c-* fixtures); scaling 3.475 -> 0.96.", - "_note": "#1983: + c-static-linkage-worker fixture (caller.c/lib.c/lib.h/local.c — worker-path static-linkage side-channel test). Pure fixture-corpus drift: no c/captures.ts or query change branch-vs-main, existing fixtures' captures byte-identical (c-captures.test.ts 45/45), scaling stays linear (~0.97). The baseline was missed when the fixture landed; regenerated here. fingerprint 0de009b->39f3a83.", + "_added": "#1956: c added to the scope-capture bench (was UNBENCHED). C has no inheritance \u2014 flat scale source. Adding it exposed + fixed a pre-existing O(n^2) findNodeAtRange root-walk in c/captures.ts (threaded c.node, byte-identical over c-* fixtures); scaling 3.475 -> 0.96.", + "_note": "#1983: + c-static-linkage-worker fixture (caller.c/lib.c/lib.h/local.c \u2014 worker-path static-linkage side-channel test). Pure fixture-corpus drift: no c/captures.ts or query change branch-vs-main, existing fixtures' captures byte-identical (c-captures.test.ts 45/45), scaling stays linear (~0.97). The baseline was missed when the fixture landed; regenerated here. fingerprint 0de009b->39f3a83.", "_rebaselined": "#1919 open-language coverage: new lang-resolution fixtures + intended capture additions (F5/F9 c-cpp, F26/F28/F29 dart, F47/F48/F49/F51/F52 kotlin, F75/F79 swift). Fingerprint-only drift; scaling_ratio ~1.0 (linear, no perf regression)." }, "cpp": { "fingerprint": "bf3587674267be1759e7c45abef143c3b81fe8629cfd17da5f8af40e83cc39ec", "scaling_budget": 1.5, - "_rebaselined_2833_qualified_member_fields": "#2833 follow-up: the six per-qualifier-depth `field_declaration` type-binding rules for a QUALIFIED generic member are replaced by three depth-agnostic ones that match the outer `qualified_identifier` itself, with the qualifier reduced to its top-level tail in `interpret.ts` (`cppQualifiedTail`). This is a CAPTURE-LOGIC change and it moves the fingerprint in two places at once. (1) A qualified NON-generic member (`ns::Address addr;`, `std::string name;`) was captured by nothing at all and now binds — that is the whole +24 on the fixture corpus, every one of them a `std::string` member. (2) Qualifier depth is no longer enumerated, so `a::b::c::Repo` (depth 3+) is captured where the old rules stopped at 2. Capture-name histogram, cpp-* corpus (278 files): `@type-binding.field` 8 -> 32, `@type-binding.name` and `@type-binding.type` 401 -> 425; synthetic DAO-20: `@type-binding.field` 40 -> 60, `@type-binding.name` and `@type-binding.type` 61 -> 81 (= 20 entities x the one `std::string name;` member the DAO unit already declared). NO OTHER TAG MOVED in either set — not one `@declaration.*`, `@scope.*` or `@reference.*` count — which is the property that says three rules replaced six without widening what a field_declaration matches. Measured over the 13 cpp-* fixture repos whose sources gained a binding, the distinct CALLS edge set is byte-identical before and after (32 edges): a reduced tail that names no workspace class binds nothing. Prior bd47c82d09a83cbf0ac857f41876fa31d22304043735582e913bccde06cf2c1a -> db1156d81b3e3341faf5e938a4a34417f4fd246588b6150b4686481823262529; scaling 1.04 < 1.5.", - "_rebaselined_2833_generic_member_fields": "#2833 review follow-up: the cpp DAO generator's unit gains two GENERIC member fields — `Repo repo;` (bare template_type) and `std::vector items;` (qualified_identifier wrapping a template_type) — plus the header declaring `template class Repo`. CORPUS CHANGE, NOT A CAPTURE-LOGIC CHANGE: no extractor edit accompanies it. It exists because the corpus had ZERO template-typed member fields and, across 279 cpp-* fixtures, not one qualified generic member either, so BOTH rounds of new `field_declaration` type-binding rules landed with a byte-identical cpp fingerprint — the gate was structurally blind to the exact thing being changed. Measured under the new corpus, the three states now differ: pre-#2833 query 0e7cbda71360b7ff35dd76091c77f288d6af6a5cfa9185ad85a372aae8c85191 (4521 groups) -> the three template_type field rules de07d8b5300ed867b460918e16b4d80259c7eb6efc1034d32bebe9ff7cab126d (4541) -> the six qualified rules bd47c82d09a83cbf0ac857f41876fa31d22304043735582e913bccde06cf2c1a (4561); under the OLD corpus all three were 856d02f3f9d22cb973877211100aee8e052d4bc545922f78704b1a21ce49ddcc. Capture-name histogram over the synthetic DAO-20: `@type-binding.field` 0 -> 40, `@declaration.field` 40 -> 80, `@type-binding.type`/`@type-binding.name` 20 -> 61, `@declaration.name` 104 -> 147 — 40 = 20 entities x 2 fields, with the residual +1/+2/+3 attributable to the one-off header declaration; every `@reference.*` count is unchanged. Prior 856d02f3f9d22cb973877211100aee8e052d4bc545922f78704b1a21ce49ddcc -> bd47c82d09a83cbf0ac857f41876fa31d22304043735582e913bccde06cf2c1a; scaling 1.058 < 1.5. `c` is unaffected (3418cded..., unchanged).", + "_rebaselined_2833_qualified_member_fields": "#2833 follow-up: the six per-qualifier-depth `field_declaration` type-binding rules for a QUALIFIED generic member are replaced by three depth-agnostic ones that match the outer `qualified_identifier` itself, with the qualifier reduced to its top-level tail in `interpret.ts` (`cppQualifiedTail`). This is a CAPTURE-LOGIC change and it moves the fingerprint in two places at once. (1) A qualified NON-generic member (`ns::Address addr;`, `std::string name;`) was captured by nothing at all and now binds \u2014 that is the whole +24 on the fixture corpus, every one of them a `std::string` member. (2) Qualifier depth is no longer enumerated, so `a::b::c::Repo` (depth 3+) is captured where the old rules stopped at 2. Capture-name histogram, cpp-* corpus (278 files): `@type-binding.field` 8 -> 32, `@type-binding.name` and `@type-binding.type` 401 -> 425; synthetic DAO-20: `@type-binding.field` 40 -> 60, `@type-binding.name` and `@type-binding.type` 61 -> 81 (= 20 entities x the one `std::string name;` member the DAO unit already declared). NO OTHER TAG MOVED in either set \u2014 not one `@declaration.*`, `@scope.*` or `@reference.*` count \u2014 which is the property that says three rules replaced six without widening what a field_declaration matches. Measured over the 13 cpp-* fixture repos whose sources gained a binding, the distinct CALLS edge set is byte-identical before and after (32 edges): a reduced tail that names no workspace class binds nothing. Prior bd47c82d09a83cbf0ac857f41876fa31d22304043735582e913bccde06cf2c1a -> db1156d81b3e3341faf5e938a4a34417f4fd246588b6150b4686481823262529; scaling 1.04 < 1.5.", + "_rebaselined_2833_generic_member_fields": "#2833 review follow-up: the cpp DAO generator's unit gains two GENERIC member fields \u2014 `Repo repo;` (bare template_type) and `std::vector items;` (qualified_identifier wrapping a template_type) \u2014 plus the header declaring `template class Repo`. CORPUS CHANGE, NOT A CAPTURE-LOGIC CHANGE: no extractor edit accompanies it. It exists because the corpus had ZERO template-typed member fields and, across 279 cpp-* fixtures, not one qualified generic member either, so BOTH rounds of new `field_declaration` type-binding rules landed with a byte-identical cpp fingerprint \u2014 the gate was structurally blind to the exact thing being changed. Measured under the new corpus, the three states now differ: pre-#2833 query 0e7cbda71360b7ff35dd76091c77f288d6af6a5cfa9185ad85a372aae8c85191 (4521 groups) -> the three template_type field rules de07d8b5300ed867b460918e16b4d80259c7eb6efc1034d32bebe9ff7cab126d (4541) -> the six qualified rules bd47c82d09a83cbf0ac857f41876fa31d22304043735582e913bccde06cf2c1a (4561); under the OLD corpus all three were 856d02f3f9d22cb973877211100aee8e052d4bc545922f78704b1a21ce49ddcc. Capture-name histogram over the synthetic DAO-20: `@type-binding.field` 0 -> 40, `@declaration.field` 40 -> 80, `@type-binding.type`/`@type-binding.name` 20 -> 61, `@declaration.name` 104 -> 147 \u2014 40 = 20 entities x 2 fields, with the residual +1/+2/+3 attributable to the one-off header declaration; every `@reference.*` count is unchanged. Prior 856d02f3f9d22cb973877211100aee8e052d4bc545922f78704b1a21ce49ddcc -> bd47c82d09a83cbf0ac857f41876fa31d22304043735582e913bccde06cf2c1a; scaling 1.058 < 1.5. `c` is unaffected (3418cded..., unchanged).", "_rebaselined_callable_flow_2522_review": "PR #2522 review hardening: callable operands retain expression/qualified identity and formals retain signature/cv metadata. Prior dde874d2c30bda9f634f9799281a66de800cad9f76cf65e7c31839e2ae9da9ff -> 57860dd2a8d4b06c6d2dd0d854c08b781faee3da8f2b6c42ba0c68a9f70e5ccb; scaling 1.090 < 1.5.", "_rebaselined_callable_flow_2522_followup": "PR #2522 follow-up: C++ overload-aware function/reference/member-pointer flow facts with invocation/constructor-result suppression. Prior 3a503a1513e7eede3f7a223dcce0896c06d15bdfa920445224c9025848c0d710 -> dde874d2c30bda9f634f9799281a66de800cad9f76cf65e7c31839e2ae9da9ff; scaling 1.034 < 1.5.", "_rebaselined_callable_flow": "Callable-value-flow facts for C++ function pointers/references, reference aliases, contextual arity, arguments, and member-pointer syntax. Prior 6ab657c8f9bfe988a3759098c2cffdcc0443def75ff263f1282b82c21d96e931 -> 3a503a1513e7eede3f7a223dcce0896c06d15bdfa920445224c9025848c0d710; measured scaling ratio 1.069 < 1.5.", @@ -46,11 +46,11 @@ "_note_1899_followup": "#1899 follow-up: braced-init metadata now carries element count, intentionally changing C++ capture output; CI benchmark scaling remains linear (1.129 < 1.5).", "_added": "#1956: cpp added to the scope-capture bench (was UNBENCHED). Heritage-bearing scale source (: public Base, public Mixin) drives emitCppInheritanceCaptures at scale. Adding it exposed + fixed a pre-existing O(n^2) findNodeAtRange root-walk in cpp/captures.ts (~12 sites, threaded c.node, byte-identical over 263 cpp-* fixtures); scaling 2.30 -> 1.12.", "_rebaselined": "#1919 open-language coverage: new lang-resolution fixtures + intended capture additions (F5/F9 c-cpp, F26/F28/F29 dart, F47/F48/F49/F51/F52 kotlin, F75/F79 swift). Fingerprint-only drift; scaling_ratio ~1.0 (linear, no perf regression). #2094: deleted C++ declarations retain @declaration.is-deleted metadata; deleted operator and pointer-return shapes plus the expanded deleted-overload fixture are included. Intended capture drift; scaling remains linear (1.139 < 1.5).", - "_note": "#1975: + cpp-out-of-line-class fixture, fixture_count 263->265. #1990: + cpp-adl-ns-plus-hidden-friend-same-name fixture (ADL hidden-friend + namespace-callable merge parity test). Pure fixture-corpus drift — no scope-extractor change; existing fixtures' captures byte-identical. fixture_count 265->267. #1995: + cpp-union-nested-tail-collision and cpp-anon-ns-tail-collision fixtures — pure fixture-corpus drift; fixture_count 270->272, fingerprint 538e8be->d63ded6. #1993: + cpp-cross-namespace-same-tail fixture — pure fixture-corpus drift; fixture_count 272->273, fingerprint d63ded6->6d6207ae. #2077 review follow-up: cpp-member-lattice adds cross-file, qualified-base, nested-template, inherited-using, this-receiver, and non-virtual-override regressions; fixture_count 274->275. Capture scaling remains linear (1.134 < 1.5). #1899: braced-init call arguments emit a conservative parameter-type capture; fixture_count 277, scaling remains linear (1.141 < 1.5).", + "_note": "#1975: + cpp-out-of-line-class fixture, fixture_count 263->265. #1990: + cpp-adl-ns-plus-hidden-friend-same-name fixture (ADL hidden-friend + namespace-callable merge parity test). Pure fixture-corpus drift \u2014 no scope-extractor change; existing fixtures' captures byte-identical. fixture_count 265->267. #1995: + cpp-union-nested-tail-collision and cpp-anon-ns-tail-collision fixtures \u2014 pure fixture-corpus drift; fixture_count 270->272, fingerprint 538e8be->d63ded6. #1993: + cpp-cross-namespace-same-tail fixture \u2014 pure fixture-corpus drift; fixture_count 272->273, fingerprint d63ded6->6d6207ae. #2077 review follow-up: cpp-member-lattice adds cross-file, qualified-base, nested-template, inherited-using, this-receiver, and non-virtual-override regressions; fixture_count 274->275. Capture scaling remains linear (1.134 < 1.5). #1899: braced-init call arguments emit a conservative parameter-type capture; fixture_count 277, scaling remains linear (1.141 < 1.5).", "_rebaselined_2522_review_fixes": "PR #2522 review fixes: outermost-chain passing modes; ->* ERROR-recovery role order; member-store visibility. Prior 57860dd2a8d4b06c6d2dd0d854c08b781faee3da8f2b6c42ba0c68a9f70e5ccb -> f29bc3f7b1622954d6f6b7647bc9cf6c7a2629ffcc0fe00ac7918e4925876b65; scaling ratio re-verified within budget.", - "_rebaselined_2522_prototype_value_cells": "Plain function/method prototypes no longer index as callable value cells (only pointer/parenthesized variable declarators do) — removes the spurious indirect-invoke facts that leaked phantom CALLS past two-phase suppression. Prior f29bc3f7b1622954d6f6b7647bc9cf6c7a2629ffcc0fe00ac7918e4925876b65 -> a70625bb0a9ef74e760d9d79cc5557485d0f0d3fb935e8a22a0c9556c65b5bb1; scaling re-verified within budget.", + "_rebaselined_2522_prototype_value_cells": "Plain function/method prototypes no longer index as callable value cells (only pointer/parenthesized variable declarators do) \u2014 removes the spurious indirect-invoke facts that leaked phantom CALLS past two-phase suppression. Prior f29bc3f7b1622954d6f6b7647bc9cf6c7a2629ffcc0fe00ac7918e4925876b65 -> a70625bb0a9ef74e760d9d79cc5557485d0f0d3fb935e8a22a0c9556c65b5bb1; scaling re-verified within budget.", "_rebaselined_receiver_chain_2747": "#2747: additionally adds the `cpp-receiver-chain-arrow` fixture, the behavioural proof for a `->` BASE receiver (`svc->getUser()->save()`) that the rollout fixed and that `cpp-chain-call/` could never catch because it uses the value `.` form. Prior a70625bb0a9ef74e760d9d79cc5557485d0f0d3fb935e8a22a0c9556c65b5bb1 -> 7e27aea46f3e17f33c41babbe0ddd982d1ab5920f143864763e0a1c6aef882a5.", - "_rebaselined_2766_receiver_chain_wire_v2": "#2766: receiver-chain wire format v1 -> v2 (name-free `await` / `index` step kinds). The VERSION prefix is part of every emitted `@reference.receiver-chain` capture, so every chain-minting language's capture text changed. WIRE-FORMAT CHANGE, NOT A CAPTURE-SET CHANGE: the same chains are minted for the same sites, spelled `2|…` instead of `1|…`. Exactly the 12 chain-minting languages drifted; c, cobol and dart did not, which is the check that this is the prefix and not a capture regression. Accompanied by SCHEMA_BUMP 34 -> 37 and INCREMENTAL_SCHEMA_VERSION 28 -> 31 so a stale index is rejected rather than replaying chains a v2 decoder refuses. Prior 7e27aea46f3e17f33c41babbe0ddd982d1ab5920f143864763e0a1c6aef882a5 -> 856d02f3f9d22cb973877211100aee8e052d4bc545922f78704b1a21ce49ddcc.", + "_rebaselined_2766_receiver_chain_wire_v2": "#2766: receiver-chain wire format v1 -> v2 (name-free `await` / `index` step kinds). The VERSION prefix is part of every emitted `@reference.receiver-chain` capture, so every chain-minting language's capture text changed. WIRE-FORMAT CHANGE, NOT A CAPTURE-SET CHANGE: the same chains are minted for the same sites, spelled `2|\u2026` instead of `1|\u2026`. Exactly the 12 chain-minting languages drifted; c, cobol and dart did not, which is the check that this is the prefix and not a capture regression. Accompanied by SCHEMA_BUMP 34 -> 37 and INCREMENTAL_SCHEMA_VERSION 28 -> 31 so a stale index is rejected rather than replaying chains a v2 decoder refuses. Prior 7e27aea46f3e17f33c41babbe0ddd982d1ab5920f143864763e0a1c6aef882a5 -> 856d02f3f9d22cb973877211100aee8e052d4bc545922f78704b1a21ce49ddcc.", "capture_groups_small": 5021, "capture_groups_large": 16021, "capture_groups_fp": 4605, @@ -64,27 +64,28 @@ "_rebaselined_callable_flow_2522_followup": "PR #2522 follow-up: C# method-group/delegate callable flow facts with invocation-result suppression. Prior 2bb5bc8c19cb8eb08c9590545ad8a1968a7152951f7e12746e2d7901d542fed9 -> f31544530924748f9aa37d11cec570bc10c3ddf9d9b237e6df7a17623fd2bb3a; scaling 1.115 < 1.5.", "_note": "#2046: F35 qualified-constructor captures now emit @reference.qualified-name + a simple-name @reference.name on `new Ns.Foo()`/`new A.B.Foo()`; namespace_declaration/file_scoped_namespace_declaration now emit @declaration.namespace name captures (feeding the non-destructive namespacePrefix sidecar for `new B.Foo()` same-tail disambiguation). + csharp-interface-only-base and csharp-namespace-qualified-ctor fixtures. Pure capture-additive + fixture-corpus drift; scaling stays linear (~1.11).", "_rebaselined_2563_instance_ownership": "#2563: csharp-using-static adds same-file ownership, local-function, overload, partial-class, and cross-namespace same-name coverage. Prior 75cf380209fa7d1a8a3ec873be1a9424b4e5173be0b08234c2291e8521a9b3c1 -> e05dc27456bde8175948586c9e7689033a378fa40e9ca4ce78cce41fbea0f2f8; scaling 1.058 < 1.5.", - "_rebaselined_receiver_chain_2747": "#2747 receiver-chain rollout: call matches whose receiver is itself an expression now carry `@reference.receiver-chain`, a compact encoding of the receiver's structure, so resolution types it by folding instead of re-parsing receiver source text. Capture GROUP counts are unchanged — the tag is added to existing call matches, never a new match — so this is digest drift only. Prior 05a85bae70cf9c94f42459c843cfc36e3e81c872e5dcc7d77bc42fbc390f4bfe -> 8a282254b93b3ef2ff34c2fdba819ebc95c53c4fcb09942cbad99f96d3687855.", - "_rebaselined_2766_receiver_chain_wire_v2": "#2766: receiver-chain wire format v1 -> v2 (name-free `await` / `index` step kinds). The VERSION prefix is part of every emitted `@reference.receiver-chain` capture, so every chain-minting language's capture text changed. WIRE-FORMAT CHANGE, NOT A CAPTURE-SET CHANGE: the same chains are minted for the same sites, spelled `2|…` instead of `1|…`. Exactly the 12 chain-minting languages drifted; c, cobol and dart did not, which is the check that this is the prefix and not a capture regression. Accompanied by SCHEMA_BUMP 34 -> 37 and INCREMENTAL_SCHEMA_VERSION 28 -> 31 so a stale index is rejected rather than replaying chains a v2 decoder refuses. Prior 8a282254b93b3ef2ff34c2fdba819ebc95c53c4fcb09942cbad99f96d3687855 -> 476d98a7cc659951c315d63319c8077bbcf0e5f3ec12d32ed773992a1f3a2adc.", + "_rebaselined_receiver_chain_2747": "#2747 receiver-chain rollout: call matches whose receiver is itself an expression now carry `@reference.receiver-chain`, a compact encoding of the receiver's structure, so resolution types it by folding instead of re-parsing receiver source text. Capture GROUP counts are unchanged \u2014 the tag is added to existing call matches, never a new match \u2014 so this is digest drift only. Prior 05a85bae70cf9c94f42459c843cfc36e3e81c872e5dcc7d77bc42fbc390f4bfe -> 8a282254b93b3ef2ff34c2fdba819ebc95c53c4fcb09942cbad99f96d3687855.", + "_rebaselined_2766_receiver_chain_wire_v2": "#2766: receiver-chain wire format v1 -> v2 (name-free `await` / `index` step kinds). The VERSION prefix is part of every emitted `@reference.receiver-chain` capture, so every chain-minting language's capture text changed. WIRE-FORMAT CHANGE, NOT A CAPTURE-SET CHANGE: the same chains are minted for the same sites, spelled `2|\u2026` instead of `1|\u2026`. Exactly the 12 chain-minting languages drifted; c, cobol and dart did not, which is the check that this is the prefix and not a capture regression. Accompanied by SCHEMA_BUMP 34 -> 37 and INCREMENTAL_SCHEMA_VERSION 28 -> 31 so a stale index is rejected rather than replaying chains a v2 decoder refuses. Prior 8a282254b93b3ef2ff34c2fdba819ebc95c53c4fcb09942cbad99f96d3687855 -> 476d98a7cc659951c315d63319c8077bbcf0e5f3ec12d32ed773992a1f3a2adc.", "capture_groups_small": 4259, "capture_groups_large": 13609, "capture_groups_fp": 2657, "fixture_count": 178 }, "rust": { - "fingerprint": "116a971fee0004f340477aff69fa110a1d92bd8ba882d7c926483c6b1e8ca2b9", + "fingerprint": "e61653008ff2de506cfd47f905fa9eb22d82fbbfe94d2a1d8190c358211b57b7", "scaling_budget": 1.5, - "_rebaselined_mod_node_identity_2745_review": "#2745 review: added rust-2742-mod-members, rust-2742-nested-mods and rust-2742-type-vs-module under lang-resolution for the container/owner-edge fix, nested inline modules, and the imported-type-vs-module precedence. emitRustScopeCaptures is unchanged — verified by removing ONLY those three fixture dirs and re-running, which reproduces the prior fingerprint exactly, so the shift is purely corpus growth (fixture_count 196 -> 202, capture_groups_fp 3432 -> 3556). Prior 90fda086a4e13aa069a5981f63ed58ab1c71f1ed3da5e1480a080e1992b0d3e5 -> 05acbaca48427e0d9e0793bcd0ce4057712d3716b5e7868189c12e05ef8dd300; scaling 1.022 local / 1.057 CI < 1.5. NOTE for the next fixture author: a new rust-* fixture drifts BOTH this bench baseline and the rust-captures-golden snapshot. Updating only the golden is how this reached CI red.", + "_rebaselined_generic_instantiation_2912": "#2912: RUST_SCOPE_QUERY tags trait-impl heritage with the instantiation the impl was written with (`impl Validator for V`), so interface dispatch can prune implementors of an instantiation the receiver cannot hold. Additive capture text on existing impl matches \u2014 the same matches are minted, carrying one more field \u2014 so this is digest drift, not a capture-set change: capture_groups_fp (3556) and fixture_count (202) are both unchanged, which is the check that no match appeared or vanished. Prior 116a971fee0004f340477aff69fa110a1d92bd8ba882d7c926483c6b1e8ca2b9 -> e61653008ff2de506cfd47f905fa9eb22d82fbbfe94d2a1d8190c358211b57b7; scaling 1.018 < 1.5. Only rust and dart move; the other 13 languages are byte-identical.", + "_rebaselined_mod_node_identity_2745_review": "#2745 review: added rust-2742-mod-members, rust-2742-nested-mods and rust-2742-type-vs-module under lang-resolution for the container/owner-edge fix, nested inline modules, and the imported-type-vs-module precedence. emitRustScopeCaptures is unchanged \u2014 verified by removing ONLY those three fixture dirs and re-running, which reproduces the prior fingerprint exactly, so the shift is purely corpus growth (fixture_count 196 -> 202, capture_groups_fp 3432 -> 3556). Prior 90fda086a4e13aa069a5981f63ed58ab1c71f1ed3da5e1480a080e1992b0d3e5 -> 05acbaca48427e0d9e0793bcd0ce4057712d3716b5e7868189c12e05ef8dd300; scaling 1.022 local / 1.057 CI < 1.5. NOTE for the next fixture author: a new rust-* fixture drifts BOTH this bench baseline and the rust-captures-golden snapshot. Updating only the golden is how this reached CI red.", "_rebaselined_dyn_trait_object_2604": "#2604: RUST_SCOPE_QUERY now captures function_signature_item (abstract trait methods, no body) as a scope + declaration, so a &dyn Trait receiver can dispatch a CALLS edge to the trait's own method. Additive capture shift across every bench fixture with a required trait method. Prior df369c5a5f8de7753fc8bab8b4108ef5081750974ea5085ba9a867675ac9eb29 -> f7742f65f14d7d6590df7f16303fc3cc9dc0c233cd80bf90c98b084933cd3846; scaling 1.033 < 1.5.", "_rebaselined_callable_flow_2522_review": "PR #2522 review hardening: callable operands retain expression/qualified identity and formals retain signature metadata. Prior 65e5bca66bb1ca117949409e8fb5c80ee69d6f1b5318908eaaecf08da0482e5c -> df369c5a5f8de7753fc8bab8b4108ef5081750974ea5085ba9a867675ac9eb29; scaling 1.065 < 1.5.", "_rebaselined_callable_flow_2522_followup": "PR #2522 follow-up: Rust fn-value callable flow facts with invocation/constructor-result suppression. Prior ac610bbe97666bf285923479dd7b43a2fe4c5354aae8df1bcbafdc04fb220f82 -> 65e5bca66bb1ca117949409e8fb5c80ee69d6f1b5318908eaaecf08da0482e5c; scaling 1.024 < 1.5.", - "_rebaselined": "#1956 tri-review U1: rust-qualified-trait fixture (scoped + generic-of-scoped impl trait paths); bareTypeIdentifier now resolves scoped_type_identifier bases by their name: tail (additive, no existing-fixture drift); linear (~1.04). #1975: + rust-scoped-impl fixture (impl a::Inner / b::Inner inherent scoped impls) — legacy @definition.impl scoped arm + findEnclosingClassInfo inherent-impl scoped target; rust scope-extractor captures byte-identical. | #942: scope-resolution-only cleanup reworded fixture comments; capture byte-positions shift, capture LOGIC unchanged.", - "_note": "PR #1934: F66/F68 let-binding pattern narrowing; F71 union (Struct-labeled, now materialized via legacy @definition.struct + resolvable); F72 macro FULLY WIRED — @declaration.macro/@reference.macro + MacroRegistry → USES edges to Macro nodes (never a same-named fn). + rust-macro / rust-union fixtures and merged with origin/main #1975 rust-scoped-impl; fingerprint re-baselined (scaling ~0.99, fixture_count 126). #1992: + rust-nested-tail-collision-generic and rust-generic-impl-same-method-name (F3) fixtures — pure fixture-corpus drift, no scope-extractor change; fixture_count 127->129, fingerprint 56ffc1c0->b00aea0f.", + "_rebaselined": "#1956 tri-review U1: rust-qualified-trait fixture (scoped + generic-of-scoped impl trait paths); bareTypeIdentifier now resolves scoped_type_identifier bases by their name: tail (additive, no existing-fixture drift); linear (~1.04). #1975: + rust-scoped-impl fixture (impl a::Inner / b::Inner inherent scoped impls) \u2014 legacy @definition.impl scoped arm + findEnclosingClassInfo inherent-impl scoped target; rust scope-extractor captures byte-identical. | #942: scope-resolution-only cleanup reworded fixture comments; capture byte-positions shift, capture LOGIC unchanged.", + "_note": "PR #1934: F66/F68 let-binding pattern narrowing; F71 union (Struct-labeled, now materialized via legacy @definition.struct + resolvable); F72 macro FULLY WIRED \u2014 @declaration.macro/@reference.macro + MacroRegistry \u2192 USES edges to Macro nodes (never a same-named fn). + rust-macro / rust-union fixtures and merged with origin/main #1975 rust-scoped-impl; fingerprint re-baselined (scaling ~0.99, fixture_count 126). #1992: + rust-nested-tail-collision-generic and rust-generic-impl-same-method-name (F3) fixtures \u2014 pure fixture-corpus drift, no scope-extractor change; fixture_count 127->129, fingerprint 56ffc1c0->b00aea0f.", "_rebaselined_import_disambiguation_2514": "#2514: added rust-import-* and rust-dup-* fixtures under lang-resolution for the range-binding ambiguity latch + import-disambiguated resolution (for-loops / struct destructuring across explicit/aliased/glob use imports). emitRustScopeCaptures is unchanged; the corpus fingerprint shifts purely because the fixture set grew (130 -> 174). Prior f7742f65f14d7d6590df7f16303fc3cc9dc0c233cd80bf90c98b084933cd3846 -> 655aed01cf1b6b84fa0c64d48dfb2526ecb67f47d90f0a91edabacd269a212db; scaling 1.06 < 1.5.", - "_rebaselined_self_type_binding_2714": "#2714: a Rust `Self` type binding now records the enclosing impl's type instead of the literal 'Self'. `let fresh = Self { .. }` inside `impl User` binds `fresh: User`; recorded verbatim it bound `fresh: Self`, which resolves to nothing. The type-env channel already substituted this (type-extractors/rust.ts findEnclosingImplType); the scope-resolution channel did not, so the two disagreed. The gap was invisible while lookupCore Step 1 still walked the lexical chain for NAMED receivers — the impl scope binds the method by name, so fresh.validate() resolved by accident — and became a lost CALLS edge when #2714 stopped that walk. Only the rust fingerprint moves; the other 14 languages are byte-identical.", + "_rebaselined_self_type_binding_2714": "#2714: a Rust `Self` type binding now records the enclosing impl's type instead of the literal 'Self'. `let fresh = Self { .. }` inside `impl User` binds `fresh: User`; recorded verbatim it bound `fresh: Self`, which resolves to nothing. The type-env channel already substituted this (type-extractors/rust.ts findEnclosingImplType); the scope-resolution channel did not, so the two disagreed. The gap was invisible while lookupCore Step 1 still walked the lexical chain for NAMED receivers \u2014 the impl scope binds the method by name, so fresh.validate() resolved by accident \u2014 and became a lost CALLS edge when #2714 stopped that walk. Only the rust fingerprint moves; the other 14 languages are byte-identical.", "_rebaselined_module_tree_2730": "#2730 + #2741 review: RUST_SCOPE_QUERY captures mod_item as @declaration.namespace (a Rust module is an item, mirroring the C++ namespace_definition capture) and tags scoped call sites with @reference.qualified-name so the written path survives to resolution. Both are additive captures: every bench fixture holding a mod block or a Foo::bar() call gains groups, and the corpus also grew by the rust-2730-* fixtures added for the fix and its review (workspace-crates, type-qualified, gaps, samename-wrapper, crate-layout). Prior 7f1240b38457468f06b7931e0c2c578f218f922774d0dc7e2ee6ef3b08d4d689 -> 90fda086a4e13aa069a5981f63ed58ab1c71f1ed3da5e1480a080e1992b0d3e5; scaling 1.061 < 1.5; fixture_count 196. Only the rust fingerprint moves; the other 14 languages are byte-identical. The earlier revision of this note cited 655aed01... as the prior value, which was two rebaselines stale (it predates #2604 and #2714); the CI gate compares live fingerprints, not this prose, so nothing caught it.", - "_rebaselined_receiver_chain_2747": "#2747 receiver-chain rollout: call matches whose receiver is itself an expression now carry `@reference.receiver-chain`, a compact encoding of the receiver's structure, so resolution types it by folding instead of re-parsing receiver source text. Capture GROUP counts are unchanged — the tag is added to existing call matches, never a new match — so this is digest drift only. Prior 05acbaca48427e0d9e0793bcd0ce4057712d3716b5e7868189c12e05ef8dd300 -> 83812d82f0e2c3eb552f3246381ca3dd5ccd6783d63aba3325f1343e7772280c.", - "_rebaselined_2766_receiver_chain_wire_v2": "#2766: receiver-chain wire format v1 -> v2 (name-free `await` / `index` step kinds). The VERSION prefix is part of every emitted `@reference.receiver-chain` capture, so every chain-minting language's capture text changed. WIRE-FORMAT CHANGE, NOT A CAPTURE-SET CHANGE: the same chains are minted for the same sites, spelled `2|…` instead of `1|…`. Exactly the 12 chain-minting languages drifted; c, cobol and dart did not, which is the check that this is the prefix and not a capture regression. Accompanied by SCHEMA_BUMP 34 -> 37 and INCREMENTAL_SCHEMA_VERSION 28 -> 31 so a stale index is rejected rather than replaying chains a v2 decoder refuses. Prior 83812d82f0e2c3eb552f3246381ca3dd5ccd6783d63aba3325f1343e7772280c -> 6174889b8c98e0af430fa54c268dc781989ca9a8172d690eebae37a95f77e809.", + "_rebaselined_receiver_chain_2747": "#2747 receiver-chain rollout: call matches whose receiver is itself an expression now carry `@reference.receiver-chain`, a compact encoding of the receiver's structure, so resolution types it by folding instead of re-parsing receiver source text. Capture GROUP counts are unchanged \u2014 the tag is added to existing call matches, never a new match \u2014 so this is digest drift only. Prior 05acbaca48427e0d9e0793bcd0ce4057712d3716b5e7868189c12e05ef8dd300 -> 83812d82f0e2c3eb552f3246381ca3dd5ccd6783d63aba3325f1343e7772280c.", + "_rebaselined_2766_receiver_chain_wire_v2": "#2766: receiver-chain wire format v1 -> v2 (name-free `await` / `index` step kinds). The VERSION prefix is part of every emitted `@reference.receiver-chain` capture, so every chain-minting language's capture text changed. WIRE-FORMAT CHANGE, NOT A CAPTURE-SET CHANGE: the same chains are minted for the same sites, spelled `2|\u2026` instead of `1|\u2026`. Exactly the 12 chain-minting languages drifted; c, cobol and dart did not, which is the check that this is the prefix and not a capture regression. Accompanied by SCHEMA_BUMP 34 -> 37 and INCREMENTAL_SCHEMA_VERSION 28 -> 31 so a stale index is rejected rather than replaying chains a v2 decoder refuses. Prior 83812d82f0e2c3eb552f3246381ca3dd5ccd6783d63aba3325f1343e7772280c -> 6174889b8c98e0af430fa54c268dc781989ca9a8172d690eebae37a95f77e809.", "capture_groups_small": 5507, "capture_groups_large": 17607, "capture_groups_fp": 3556, @@ -96,9 +97,9 @@ "_rebaselined_callable_flow_2522_review": "PR #2522 review hardening: callable operands retain expression/qualified identity and formals retain signature metadata. Prior df7b1565f9115d66b1ae32e4a408d651afb2521b14e5ca615f3be426c29af618 -> 4a688fa5a7016546f7f3c6d44de023608ae80c5b0e3670c16f6e61b3632608fd; scaling 1.078 < 1.5.", "_rebaselined_callable_flow_2522_followup": "PR #2522 follow-up: PHP first-class callable and variable-invocation flow facts with invocation-result suppression. Prior 31c9e3f3cb7094a2bf9021cf9db859036e002f8b44605cd993b470fc600e97cb -> df7b1565f9115d66b1ae32e4a408d651afb2521b14e5ca615f3be426c29af618; scaling 1.074 < 1.5.", "_rebaselined": "#1956: heritage-bearing scale source (class extends Base + use trait); both forms gated at scale; linear (~1.04). | #2481/#2482: PHP imports carry a symbol-kind capture so function/constant imports resolve by declaring file; capture shape changes, scaling remains linear (~1.04).", - "_note": "PR #1931: F53 import multi-clause, F54 enum_case, F55 anonymous_class — fixture count 138→140, fingerprint drift expected.", - "_rebaselined_receiver_chain_2747": "#2747 receiver-chain rollout: call matches whose receiver is itself an expression now carry `@reference.receiver-chain`, a compact encoding of the receiver's structure, so resolution types it by folding instead of re-parsing receiver source text. Capture GROUP counts are unchanged — the tag is added to existing call matches, never a new match — so this is digest drift only. Prior 4a688fa5a7016546f7f3c6d44de023608ae80c5b0e3670c16f6e61b3632608fd -> 3745662053c76b6ae0a84a29aad319626ed5ccb88f7b9376c2680d3dc6502e28.", - "_rebaselined_2766_receiver_chain_wire_v2": "#2766: receiver-chain wire format v1 -> v2 (name-free `await` / `index` step kinds). The VERSION prefix is part of every emitted `@reference.receiver-chain` capture, so every chain-minting language's capture text changed. WIRE-FORMAT CHANGE, NOT A CAPTURE-SET CHANGE: the same chains are minted for the same sites, spelled `2|…` instead of `1|…`. Exactly the 12 chain-minting languages drifted; c, cobol and dart did not, which is the check that this is the prefix and not a capture regression. Accompanied by SCHEMA_BUMP 34 -> 37 and INCREMENTAL_SCHEMA_VERSION 28 -> 31 so a stale index is rejected rather than replaying chains a v2 decoder refuses. Prior 3745662053c76b6ae0a84a29aad319626ed5ccb88f7b9376c2680d3dc6502e28 -> b213a872342da2d866b04681dede988770e4d3dfdc0d6e9f62212ec5b59cdc2c." + "_note": "PR #1931: F53 import multi-clause, F54 enum_case, F55 anonymous_class \u2014 fixture count 138\u2192140, fingerprint drift expected.", + "_rebaselined_receiver_chain_2747": "#2747 receiver-chain rollout: call matches whose receiver is itself an expression now carry `@reference.receiver-chain`, a compact encoding of the receiver's structure, so resolution types it by folding instead of re-parsing receiver source text. Capture GROUP counts are unchanged \u2014 the tag is added to existing call matches, never a new match \u2014 so this is digest drift only. Prior 4a688fa5a7016546f7f3c6d44de023608ae80c5b0e3670c16f6e61b3632608fd -> 3745662053c76b6ae0a84a29aad319626ed5ccb88f7b9376c2680d3dc6502e28.", + "_rebaselined_2766_receiver_chain_wire_v2": "#2766: receiver-chain wire format v1 -> v2 (name-free `await` / `index` step kinds). The VERSION prefix is part of every emitted `@reference.receiver-chain` capture, so every chain-minting language's capture text changed. WIRE-FORMAT CHANGE, NOT A CAPTURE-SET CHANGE: the same chains are minted for the same sites, spelled `2|\u2026` instead of `1|\u2026`. Exactly the 12 chain-minting languages drifted; c, cobol and dart did not, which is the check that this is the prefix and not a capture regression. Accompanied by SCHEMA_BUMP 34 -> 37 and INCREMENTAL_SCHEMA_VERSION 28 -> 31 so a stale index is rejected rather than replaying chains a v2 decoder refuses. Prior 3745662053c76b6ae0a84a29aad319626ed5ccb88f7b9376c2680d3dc6502e28 -> b213a872342da2d866b04681dede988770e4d3dfdc0d6e9f62212ec5b59cdc2c." }, "ruby": { "fingerprint": "1c8c9c4b54036fa24c2a81e39ea530e938645c856d369075e5f437da78218c57", @@ -106,10 +107,10 @@ "_rebaselined_callable_flow_2522_review": "PR #2522 review hardening: callable operands retain expression/qualified identity and formals retain signature metadata. Prior cff273ae6cb7232c977d9241581834a2a2fa8bcf6369f7bd8f2471cd4419a6ef -> bf50ec6a53c8c91680dc6feac63a8956e78b1059249232dc25a0cfed25f31236; scaling 1.103 < 1.5.", "_rebaselined_callable_flow_2522_followup": "PR #2522 follow-up: Ruby Method/Proc callable flow facts with invocation/constructor-result suppression. Prior b5ea93bb3d0469c3821a8c70f5d5991c6f326e41097c119ad691154301dcc753 -> cff273ae6cb7232c977d9241581834a2a2fa8bcf6369f7bd8f2471cd4419a6ef; scaling 1.086 < 1.5.", "_rebaselined": "#1956 synth-widening: + ruby-qualified-base fixture; synth now reduces a scope_resolution superclass (class C < Mod::Super) to its trailing constant (matching the #1940 legacy leg), at parity. Linear (~1.03). (Earlier #1956: heritage-bearing scale source.) | #942: scope-resolution-only cleanup reworded fixture comments; capture byte-positions shift, capture LOGIC unchanged.", - "_note": "F62: + scope_resolution class/module declaration captures — fixture count 78→81, fingerprint drift expected. #1975: + ruby-tail-collision fixture (Foo::Bar vs Baz::Bar stay distinct nodes) — pure fixture-corpus drift, scope-extractor captures unchanged; 81→82. #1991: + ruby-nested-mixin-tail-collision fixture (85→86). Recomputed on the #942 merge (fixture-comment rewording shifts capture byte-positions, capture LOGIC unchanged): bf6b13a -> b5ea93bb.", + "_note": "F62: + scope_resolution class/module declaration captures \u2014 fixture count 78\u219281, fingerprint drift expected. #1975: + ruby-tail-collision fixture (Foo::Bar vs Baz::Bar stay distinct nodes) \u2014 pure fixture-corpus drift, scope-extractor captures unchanged; 81\u219282. #1991: + ruby-nested-mixin-tail-collision fixture (85\u219286). Recomputed on the #942 merge (fixture-comment rewording shifts capture byte-positions, capture LOGIC unchanged): bf6b13a -> b5ea93bb.", "_rebaselined_2522_review_fixes": "PR #2522 review fixes: bare identifiers are calls, not callable references (bareNamesAreCalls). Prior bf50ec6a53c8c91680dc6feac63a8956e78b1059249232dc25a0cfed25f31236 -> 070e4e11502442998ddf4048c2981cf1b2b735a87362ff854c5d14d71f98f4e2; scaling ratio re-verified within budget.", - "_rebaselined_receiver_chain_2747": "#2747 receiver-chain rollout: call matches whose receiver is itself an expression now carry `@reference.receiver-chain`, a compact encoding of the receiver's structure, so resolution types it by folding instead of re-parsing receiver source text. Capture GROUP counts are unchanged — the tag is added to existing call matches, never a new match — so this is digest drift only. Prior fea3edf82f521995147874b7f6c5f9e2eb88efdebf6365668f3260e913f0b558 -> fc81941b0a921074fa80dc448284de9a23bd07358ddc84d4894797cc08c3fe83.", - "_rebaselined_2766_receiver_chain_wire_v2": "#2766: receiver-chain wire format v1 -> v2 (name-free `await` / `index` step kinds). The VERSION prefix is part of every emitted `@reference.receiver-chain` capture, so every chain-minting language's capture text changed. WIRE-FORMAT CHANGE, NOT A CAPTURE-SET CHANGE: the same chains are minted for the same sites, spelled `2|…` instead of `1|…`. Exactly the 12 chain-minting languages drifted; c, cobol and dart did not, which is the check that this is the prefix and not a capture regression. Accompanied by SCHEMA_BUMP 34 -> 37 and INCREMENTAL_SCHEMA_VERSION 28 -> 31 so a stale index is rejected rather than replaying chains a v2 decoder refuses. Prior fc81941b0a921074fa80dc448284de9a23bd07358ddc84d4894797cc08c3fe83 -> 1c8c9c4b54036fa24c2a81e39ea530e938645c856d369075e5f437da78218c57." + "_rebaselined_receiver_chain_2747": "#2747 receiver-chain rollout: call matches whose receiver is itself an expression now carry `@reference.receiver-chain`, a compact encoding of the receiver's structure, so resolution types it by folding instead of re-parsing receiver source text. Capture GROUP counts are unchanged \u2014 the tag is added to existing call matches, never a new match \u2014 so this is digest drift only. Prior fea3edf82f521995147874b7f6c5f9e2eb88efdebf6365668f3260e913f0b558 -> fc81941b0a921074fa80dc448284de9a23bd07358ddc84d4894797cc08c3fe83.", + "_rebaselined_2766_receiver_chain_wire_v2": "#2766: receiver-chain wire format v1 -> v2 (name-free `await` / `index` step kinds). The VERSION prefix is part of every emitted `@reference.receiver-chain` capture, so every chain-minting language's capture text changed. WIRE-FORMAT CHANGE, NOT A CAPTURE-SET CHANGE: the same chains are minted for the same sites, spelled `2|\u2026` instead of `1|\u2026`. Exactly the 12 chain-minting languages drifted; c, cobol and dart did not, which is the check that this is the prefix and not a capture regression. Accompanied by SCHEMA_BUMP 34 -> 37 and INCREMENTAL_SCHEMA_VERSION 28 -> 31 so a stale index is rejected rather than replaying chains a v2 decoder refuses. Prior fc81941b0a921074fa80dc448284de9a23bd07358ddc84d4894797cc08c3fe83 -> 1c8c9c4b54036fa24c2a81e39ea530e938645c856d369075e5f437da78218c57." }, "swift": { "fingerprint": "adef9284feaecd39cb490aebce83876e15b9150c7a04b00a396feb78b7e1e0a9", @@ -118,13 +119,14 @@ "_rebaselined_callable_flow_2522_followup": "PR #2522 follow-up: Swift function-value callable flow facts with invocation-result suppression. Prior 180ac68e780bdf6f9089d53f51cbb9a66aed3e7774631cc3fcbaae5020213998 -> 5f923c6604d825d12b249f31c155b0f4d13a8379d532e5dde64a0f9b15cf4725; scaling 1.043 < 1.5.", "_rebaselined": "#1919 open-language coverage: new lang-resolution fixtures + intended capture additions (F5/F9 c-cpp, F26/F28/F29 dart, F47/F48/F49/F51/F52 kotlin, F75/F79 swift). Fingerprint-only drift; scaling_ratio ~1.0 (linear, no perf regression).", "_rebaselined_2522_review_fixes": "PR #2522 review fixes: assignment target:/result: fields join the shared fallback. Prior 7687ee2466e16020a12440a03fbda53e63aa05f94b4481f6133c09867a0d560d -> 115c5da807e36bb12fdeba28e44f2b6484ef322ff26c19fa0f191febaf774248; scaling ratio re-verified within budget.", - "_rebaselined_receiver_chain_2747": "#2747 receiver-chain rollout: call matches whose receiver is itself an expression now carry `@reference.receiver-chain`, a compact encoding of the receiver's structure, so resolution types it by folding instead of re-parsing receiver source text. Capture GROUP counts are unchanged — the tag is added to existing call matches, never a new match — so this is digest drift only. Prior 115c5da807e36bb12fdeba28e44f2b6484ef322ff26c19fa0f191febaf774248 -> a6fca5f052ae5ec635b56051e28a168c864a988b2221a3279ddd69807378ba0b.", - "_rebaselined_2766_receiver_chain_wire_v2": "#2766: receiver-chain wire format v1 -> v2 (name-free `await` / `index` step kinds). The VERSION prefix is part of every emitted `@reference.receiver-chain` capture, so every chain-minting language's capture text changed. WIRE-FORMAT CHANGE, NOT A CAPTURE-SET CHANGE: the same chains are minted for the same sites, spelled `2|…` instead of `1|…`. Exactly the 12 chain-minting languages drifted; c, cobol and dart did not, which is the check that this is the prefix and not a capture regression. Accompanied by SCHEMA_BUMP 34 -> 37 and INCREMENTAL_SCHEMA_VERSION 28 -> 31 so a stale index is rejected rather than replaying chains a v2 decoder refuses. Prior a6fca5f052ae5ec635b56051e28a168c864a988b2221a3279ddd69807378ba0b -> 2f04ae960123cf50138a49fabdc5a146c2963170cecf5755c552b23c9055a9e7.", + "_rebaselined_receiver_chain_2747": "#2747 receiver-chain rollout: call matches whose receiver is itself an expression now carry `@reference.receiver-chain`, a compact encoding of the receiver's structure, so resolution types it by folding instead of re-parsing receiver source text. Capture GROUP counts are unchanged \u2014 the tag is added to existing call matches, never a new match \u2014 so this is digest drift only. Prior 115c5da807e36bb12fdeba28e44f2b6484ef322ff26c19fa0f191febaf774248 -> a6fca5f052ae5ec635b56051e28a168c864a988b2221a3279ddd69807378ba0b.", + "_rebaselined_2766_receiver_chain_wire_v2": "#2766: receiver-chain wire format v1 -> v2 (name-free `await` / `index` step kinds). The VERSION prefix is part of every emitted `@reference.receiver-chain` capture, so every chain-minting language's capture text changed. WIRE-FORMAT CHANGE, NOT A CAPTURE-SET CHANGE: the same chains are minted for the same sites, spelled `2|\u2026` instead of `1|\u2026`. Exactly the 12 chain-minting languages drifted; c, cobol and dart did not, which is the check that this is the prefix and not a capture regression. Accompanied by SCHEMA_BUMP 34 -> 37 and INCREMENTAL_SCHEMA_VERSION 28 -> 31 so a stale index is rejected rather than replaying chains a v2 decoder refuses. Prior a6fca5f052ae5ec635b56051e28a168c864a988b2221a3279ddd69807378ba0b -> 2f04ae960123cf50138a49fabdc5a146c2963170cecf5755c552b23c9055a9e7.", "_rebaselined_inferred_field_receiver_2807": "#2807: optional property annotations (`var a: Outer?`) now emit a type binding. The prior pattern required the `user_type` to be a DIRECT child of the annotation, so an `optional_type` wrapper meant an optional field was never typed at all and its receiver could not resolve. ADDS @type-binding.annotation captures on the optional form only; no capture is removed. Prior 2f04ae960123cf50138a49fabdc5a146c2963170cecf5755c552b23c9055a9e7 -> adef9284feaecd39cb490aebce83876e15b9150c7a04b00a396feb78b7e1e0a9; scaling 1.023 < 1.5." }, "dart": { - "fingerprint": "ba93c90dcd341259e8e088816bc8c76ad27882419f665e35c056dc22fa54cf73", + "fingerprint": "3a8ddabbeb1cba47a4757451d4f79d726ca230fd15e860772b11526fbb1c6687", "scaling_budget": 1.5, + "_rebaselined_generic_instantiation_2912": "#2912: the Dart heritage marker carries a fourth field \u2014 the type arguments the clause was written with (`implements Validator`) \u2014 so interface dispatch can prune implementors of a mismatched instantiation. Additive marker text on existing heritage matches rather than a new match, so this is digest drift only; a marker from a pre-#2912 cache simply has no fourth field and reads as unknown. Prior ba93c90dcd341259e8e088816bc8c76ad27882419f665e35c056dc22fa54cf73 -> 3a8ddabbeb1cba47a4757451d4f79d726ca230fd15e860772b11526fbb1c6687; scaling 1.027 < 1.5.", "_rebaselined_2538": "#2538: Dart extension type headers are preprocessed into normal extension declarations before scope capture, so extension type symbols and their methods are now emitted. Intentional Dart-only capture fingerprint drift; CI measured scaling 1.042 < 1.5.", "_rebaselined_2538_implements": "#2538 tri-review follow-up: Dart extension type implements clauses now emit heritage markers and fixture coverage asserts IMPLEMENTS edges, including multi-arg generic interfaces. Prior committed baseline 66a46d5ff09f3d11b2771db0f48596fe7057e95c5bc8f56241fdb911137298c3 -> ba93c90dcd341259e8e088816bc8c76ad27882419f665e35c056dc22fa54cf73; scaling 0.945 < 1.5.", "_rebaselined_callable_flow_2522_review": "PR #2522 review hardening: callable operands retain expression/qualified identity and formals retain signature metadata. Prior 29ce2bfe70b246b1c9d5e99c0ec11e850c22e9672737592207242b7f4cc824b8 -> 66a46d5ff09f3d11b2771db0f48596fe7057e95c5bc8f56241fdb911137298c3; scaling 1.054 < 1.5.", @@ -133,13 +135,14 @@ "_rebaselined": "#1919 review CF3 fix: extended kotlin-local-property-owner (init/accessor destructuring) + new dart-accessor-owner fixture (getter/setter ownership). Fingerprint-only corpus drift; scaling ~1.0." }, "java": { - "fingerprint": "2e2150b4f4d64519e3f4c6d7a2c12259178d3117872203c904fab8cba96a694a", + "fingerprint": "2bf47cc19b595a9889ac21ec0154c6ce6786271d68551f21d1bc14c626bcd4ff", "scaling_budget": 1.5, "_rebaselined_2935_synthetic_declarations": "PR #2935 review follow-up: synthesized Java anonymous classes and bodied enum constants now carry the presence-only @declaration.is-synthetic sidecar used to preserve source-written dispatch targets at the fanout cap. DIGEST DRIFT ONLY, NOT A CAPTURE-SET CHANGE: the tag is attached to existing synthetic declaration matches; capture groups and fixture count remain 5755/18405, 3512, and 206. Prior 36d689c58526c4482fbd701d1d9ca156623a3970734ead145717858712271ab5 -> 2e2150b4f4d64519e3f4c6d7a2c12259178d3117872203c904fab8cba96a694a; CI scaling 0.971 < 1.5.", + "_rebaselined_2917_record_component_accessors": "#2917: every implicit Java record-component accessor now emits a component-bounded @scope.function plus @declaration.method/name/zero-arity/return-type metadata. The scope boundary prevents subsequent record-body references from being attributed to the accessor. Java was the only general language fingerprint to move; capture groups scale by exactly two per generated record component (small 5755 -> 6255, large 18405 -> 20005). Prior 36d689c58526c4482fbd701d1d9ca156623a3970734ead145717858712271ab5 -> 901a66c7dc0f071eeef9e4864b2519e5b58a1a141a1f9a7817ea42f7ff70eafb; scaling 0.961 < 1.5. Re-measured after merging origin/main, which carries #2935's is-synthetic sidecar on top of the same corpus: 2e2150b4f4d64519e3f4c6d7a2c12259178d3117872203c904fab8cba96a694a -> 79dafc369eaeb7183ee8cc1149b1a6c21ad672c7e5b806fe8b0060e5a952c79a; scaling 1.085 < 1.5, capture groups 6255/20005, capture_groups_fp 3560, fixture_count 206 (unchanged by the merge).", "_rebaselined_2900_record_heritage": "#2900 review follow-up: the Java scale unit now includes a record implementing Marker, so the record-declaration @reference.inherits path is fingerprinted and exercised at scale. Prior b29e263524f55151dcb7cfc4c929d3d1d7bb360355cee4e832158f927857f663 -> 36d689c58526c4482fbd701d1d9ca156623a3970734ead145717858712271ab5; scaling 1.042 < 1.5.", "_rebaselined_callable_flow_2522_review": "PR #2522 review hardening: callable operands retain expression/qualified identity and formals retain signature metadata; same-name lexical regions use an O(ancestor-depth) ID-set lookup. Prior d5c59d7dc9e206637515d5aea1163f7c1cdd76410c38c5fe6143d13d19677d6a -> 004a3592998dca1193bd1429a8284513725de7764f2a3eceedaaa984cfd763b4; scaling 0.992 < 1.5.", "_rebaselined_callable_flow_2522_followup": "PR #2522 follow-up: Java method-reference/SAM callable flow facts with invocation-result suppression. Prior 062d754764aaa8a6772fb90875c710502a63e3e7a300e633942381ed914faada -> d5c59d7dc9e206637515d5aea1163f7c1cdd76410c38c5fe6143d13d19677d6a; scaling 1.074 < 1.5.", - "_rebaselined": "#2357 (supersedes #2353): + java-cast-receiver, java-this-field-chain, java-this-dispatch fixtures (cast-wrapped receivers, this.field chains incl. initializer contexts, bare-this dispatch pinning). Drift is purely fixture-additive: with the three new dirs parked, the fingerprint reproduces the prior baseline byte-identically — no emit/capture change. #1956 synth-widening: + java-iface-extends fixture; synthesizeJavaInheritanceReferences now ALSO walks interface_declaration extends_interfaces (interface IA extends IB, IC), matching the #1940 legacy leg. (Earlier U2+review: java-qualified-base fixture covers 2- AND 3-segment qualified bases guarding the legacy end-anchor; synth tail-resolves scoped bases.) Linear (~1.03). (Earliest: java added to bench, exposed+fixed the O(n^2) findNodeAtRange root-walk; 3.09 -> ~0.99.) | #942: scope-resolution-only cleanup reworded fixture comments; capture byte-positions shift, capture LOGIC unchanged.", + "_rebaselined": "#2357 (supersedes #2353): + java-cast-receiver, java-this-field-chain, java-this-dispatch fixtures (cast-wrapped receivers, this.field chains incl. initializer contexts, bare-this dispatch pinning). Drift is purely fixture-additive: with the three new dirs parked, the fingerprint reproduces the prior baseline byte-identically \u2014 no emit/capture change. #1956 synth-widening: + java-iface-extends fixture; synthesizeJavaInheritanceReferences now ALSO walks interface_declaration extends_interfaces (interface IA extends IB, IC), matching the #1940 legacy leg. (Earlier U2+review: java-qualified-base fixture covers 2- AND 3-segment qualified bases guarding the legacy end-anchor; synth tail-resolves scoped bases.) Linear (~1.03). (Earliest: java added to bench, exposed+fixed the O(n^2) findNodeAtRange root-walk; 3.09 -> ~0.99.) | #942: scope-resolution-only cleanup reworded fixture comments; capture byte-positions shift, capture LOGIC unchanged.", "_note": "#1928 / #2045: F35 adds qualified + qualified-generic constructor query captures (`new pkg.Foo()`, `new a.b.Foo()`, `new pkg.Box()`); F38 synthesizes `@reference.call.constructor` on `super(...)`/`this(...)` explicit_constructor_invocation nodes; F41 generic-aware stripQualifier in interpret (type-binding normalization). + java-qualified-constructor and java-explicit-constructor fixtures. Pure capture-additive + fixture-corpus drift; scaling stays linear (~1.06).", "_rebaselined_2522_review_fixes": "PR #2522 review fixes: get/test dropped from callableProtocolMethods. Prior 004a3592998dca1193bd1429a8284513725de7764f2a3eceedaaa984cfd763b4 -> f3b4f4b6610e07c3ac90deb1c53d3572b6ad55a36e5d7134984876d30031ff67; scaling ratio re-verified within budget.", "_rebaselined_2550_instance_model": "PR #2549 (#2550): anonymous class bodies emit synthesized @declaration.class/@declaration.name (Worker$N), an @reference.inherits to the constructed type, and receiver @type-binding.* captures; six new java-* fixtures joined the corpus. Prior f3b4f4b6610e07c3ac90deb1c53d3572b6ad55a36e5d7134984876d30031ff67 -> d79c3b92acfc866094981499b977388ca14f90839bca0c040342ab1cec00aa90; scaling 1.058 < 1.5.", @@ -147,44 +150,48 @@ "_rebaselined_2564_record_capture": "PR for #2564: JAVA_QUERIES gained a (record_declaration name: (identifier) @name) @definition.record capture, previously entirely missing (record_declaration had no structure-phase capture at all, unlike class/interface/enum) - a record's methods existed as ownerless Method nodes with no HAS_METHOD edge. Two new java-* fixtures (java-record-methods, java-new-expr-chain-call) joined the corpus. Prior 975b68aaac6d06094260fb0c67f9b1bc03692ba7220669d192aca9dccd5fc0ca -> 85fc7af9c3c1bceac76cb4f27214410b04967682a2eaa7e468e26efd1f4e2537; scaling 1.059 < 1.5.", "_rebaselined_2561_enum_constant_receiver": "PR for #2561: synthesizeJavaAnonymousClassDeclarations now emits a class-scope @type-binding.annotation/name/type per enum constant (constant simple name -> its E$N synthesized class when bodied, else the host enum) so E.CONST.method() resolves through the existing compound-receiver chain walk. Two drivers of the drift, both in the java-enum-constant-body fixture (this bench's corpus IS test/fixtures/lang-resolution): (1) one extra type-binding match per enum_constant from the capture change; (2) review follow-up added a body-less Plain.java enum + EnumConst.dispatchToConstant/dispatchInherited methods (bodied-override, inherited-via-MRO, and body-less dispatch call sites). The review's fail-safe hardening (bodied constant binds ONLY to E$N, never the host enum, when name synthesis fails on a malformed tree) is output-neutral on this well-formed corpus (verified: fingerprint identical with and without it). Prior 85fc7af9c3c1bceac76cb4f27214410b04967682a2eaa7e468e26efd1f4e2537 -> d04298a91beec76d0fa7099b3d71265723be60c1df688969aa954f135dd49686; scaling < 1.5.", "_rebaselined_2562_local_classes": "#2562: Java block-local classes, enums, records, and interfaces use source-type-relative JLS 13.1 Host$NLocal identities with javac-compatible per-(host, simple-name) numbering; anonymous numbering remains separate. Lexical aliases begin at each declaration and end with its immediate block. Expanded java-local-class-naming fixtures cover declaration order, disjoint blocks, initializers, lambdas, local type kinds, and recursive local/member/anonymous host chains. Prior d04298a91beec76d0fa7099b3d71265723be60c1df688969aa954f135dd49686 -> 6dd5913a58400a191ff54abf9b852b03d5add657d16c11e60a7c4608ba186197; scaling 1.204 < 1.5.", - "_rebaselined_receiver_chain_2747": "#2747 receiver-chain rollout: call matches whose receiver is itself an expression now carry `@reference.receiver-chain`, a compact encoding of the receiver's structure, so resolution types it by folding instead of re-parsing receiver source text. Capture GROUP counts are unchanged — the tag is added to existing call matches, never a new match — so this is digest drift only. Prior 6dd5913a58400a191ff54abf9b852b03d5add657d16c11e60a7c4608ba186197 -> 310adbc2e0827b5ac749acaa981cd12d256fc5b7cbc5592c5bee219e92abf9ee.", - "_rebaselined_2766_receiver_chain_wire_v2": "#2766: receiver-chain wire format v1 -> v2 (name-free `await` / `index` step kinds). The VERSION prefix is part of every emitted `@reference.receiver-chain` capture, so every chain-minting language's capture text changed. WIRE-FORMAT CHANGE, NOT A CAPTURE-SET CHANGE: the same chains are minted for the same sites, spelled `2|…` instead of `1|…`. Exactly the 12 chain-minting languages drifted; c, cobol and dart did not, which is the check that this is the prefix and not a capture regression. Accompanied by SCHEMA_BUMP 34 -> 37 and INCREMENTAL_SCHEMA_VERSION 28 -> 31 so a stale index is rejected rather than replaying chains a v2 decoder refuses. Prior 310adbc2e0827b5ac749acaa981cd12d256fc5b7cbc5592c5bee219e92abf9ee -> a9943355e945e03ddb87c800f4cc1f62b3d04feefb3ec64c258d8e0bb3b3fcd9.", - "capture_groups_small": 5755, - "capture_groups_large": 18405, - "capture_groups_fp": 3512, - "fixture_count": 206 + "_rebaselined_receiver_chain_2747": "#2747 receiver-chain rollout: call matches whose receiver is itself an expression now carry `@reference.receiver-chain`, a compact encoding of the receiver's structure, so resolution types it by folding instead of re-parsing receiver source text. Capture GROUP counts are unchanged \u2014 the tag is added to existing call matches, never a new match \u2014 so this is digest drift only. Prior 6dd5913a58400a191ff54abf9b852b03d5add657d16c11e60a7c4608ba186197 -> 310adbc2e0827b5ac749acaa981cd12d256fc5b7cbc5592c5bee219e92abf9ee.", + "_rebaselined_2766_receiver_chain_wire_v2": "#2766: receiver-chain wire format v1 -> v2 (name-free `await` / `index` step kinds). The VERSION prefix is part of every emitted `@reference.receiver-chain` capture, so every chain-minting language's capture text changed. WIRE-FORMAT CHANGE, NOT A CAPTURE-SET CHANGE: the same chains are minted for the same sites, spelled `2|\u2026` instead of `1|\u2026`. Exactly the 12 chain-minting languages drifted; c, cobol and dart did not, which is the check that this is the prefix and not a capture regression. Accompanied by SCHEMA_BUMP 34 -> 37 and INCREMENTAL_SCHEMA_VERSION 28 -> 31 so a stale index is rejected rather than replaying chains a v2 decoder refuses. Prior 310adbc2e0827b5ac749acaa981cd12d256fc5b7cbc5592c5bee219e92abf9ee -> a9943355e945e03ddb87c800f4cc1f62b3d04feefb3ec64c258d8e0bb3b3fcd9.", + "capture_groups_small": 6255, + "capture_groups_large": 20005, + "capture_groups_fp": 3586, + "fixture_count": 209, + "_rebaselined_2910_declared_package_fixtures": "#2910 adds three Java resolver fixture files covering an external JDK lookalike, a path/package mismatch, and wildcard package membership. Fixture-corpus growth only: Java query rules and synthetic scaling sources are unchanged; capture_groups_small/large remain 6255/20005. capture_groups_fp 3560 -> 3586 and fixture_count 206 -> 209." }, "java-local-types": { - "fingerprint": "560734cd053fb4f4b23aa04bc7870c22089a8deedb0217fa9c1b4db689e02a97", + "fingerprint": "bdde823fa725e636e257940efb4c8655aa23124c1727cbaa8856d1ad8f71729e", "scaling_budget": 1.5, "_rebaselined_2935_synthetic_declarations": "PR #2935 review follow-up: the local-type stress corpus includes synthesized anonymous declarations, which now carry the presence-only @declaration.is-synthetic sidecar. DIGEST DRIFT ONLY, NOT A CAPTURE-SET CHANGE. Prior 8c50bbc83dff4f7f5abd06078aa6abc6b64af05fddb17ee826b5f3df3d346633 -> 560734cd053fb4f4b23aa04bc7870c22089a8deedb0217fa9c1b4db689e02a97; CI scaling 1.002 < 1.5.", + "_rebaselined_2917_record_component_accessors": "#2917: the focused local-type fixture corpus contains local records, so their implicit component accessors add the same bounded scope/declaration captures as the general Java corpus. No local-type naming logic changed. Prior 8c50bbc83dff4f7f5abd06078aa6abc6b64af05fddb17ee826b5f3df3d346633 -> 3e22f368a4ee139be7cb91ff4fb77ddadf60c55efe8d66955ec81f366a46e460; scaling 1.032 < 1.5, capture_groups_fp 680. Re-measured on top of #2935's is-synthetic sidecar after merging origin/main: 560734cd053fb4f4b23aa04bc7870c22089a8deedb0217fa9c1b4db689e02a97 -> bdde823fa725e636e257940efb4c8655aa23124c1727cbaa8856d1ad8f71729e; scaling 0.997 < 1.5, capture_groups_fp 680.", "_added": "#2562 performance follow-up: co-scales same-host, same-name local classes and anonymous classes to gate JLS binary-name ordinal allocation. Precomputed per-sequence ordinals reduce the focused 100->800 workload from 176->6655ms to 141->752ms; normalized 250->800 scaling is 1.054.", - "_rebaselined_receiver_chain_2747": "#2747 receiver-chain rollout: call matches whose receiver is itself an expression now carry `@reference.receiver-chain`, a compact encoding of the receiver's structure, so resolution types it by folding instead of re-parsing receiver source text. Capture GROUP counts are unchanged — the tag is added to existing call matches, never a new match — so this is digest drift only. Prior a9ad88de21ca6747a923260dbdf677fb74a004abbf9d57781f745e3a9027530b -> 3ca67847ea2b9a71b0a41e09f943767e5a2d3a113d3e203499ee364e37f40236.", - "_rebaselined_2766_receiver_chain_wire_v2": "#2766: receiver-chain wire format v1 -> v2 (name-free `await` / `index` step kinds). The VERSION prefix is part of every emitted `@reference.receiver-chain` capture, so every chain-minting language's capture text changed. WIRE-FORMAT CHANGE, NOT A CAPTURE-SET CHANGE: the same chains are minted for the same sites, spelled `2|…` instead of `1|…`. Exactly the 12 chain-minting languages drifted; c, cobol and dart did not, which is the check that this is the prefix and not a capture regression. Accompanied by SCHEMA_BUMP 34 -> 37 and INCREMENTAL_SCHEMA_VERSION 28 -> 31 so a stale index is rejected rather than replaying chains a v2 decoder refuses. Prior 3ca67847ea2b9a71b0a41e09f943767e5a2d3a113d3e203499ee364e37f40236 -> 8c50bbc83dff4f7f5abd06078aa6abc6b64af05fddb17ee826b5f3df3d346633." + "_rebaselined_receiver_chain_2747": "#2747 receiver-chain rollout: call matches whose receiver is itself an expression now carry `@reference.receiver-chain`, a compact encoding of the receiver's structure, so resolution types it by folding instead of re-parsing receiver source text. Capture GROUP counts are unchanged \u2014 the tag is added to existing call matches, never a new match \u2014 so this is digest drift only. Prior a9ad88de21ca6747a923260dbdf677fb74a004abbf9d57781f745e3a9027530b -> 3ca67847ea2b9a71b0a41e09f943767e5a2d3a113d3e203499ee364e37f40236.", + "_rebaselined_2766_receiver_chain_wire_v2": "#2766: receiver-chain wire format v1 -> v2 (name-free `await` / `index` step kinds). The VERSION prefix is part of every emitted `@reference.receiver-chain` capture, so every chain-minting language's capture text changed. WIRE-FORMAT CHANGE, NOT A CAPTURE-SET CHANGE: the same chains are minted for the same sites, spelled `2|\u2026` instead of `1|\u2026`. Exactly the 12 chain-minting languages drifted; c, cobol and dart did not, which is the check that this is the prefix and not a capture regression. Accompanied by SCHEMA_BUMP 34 -> 37 and INCREMENTAL_SCHEMA_VERSION 28 -> 31 so a stale index is rejected rather than replaying chains a v2 decoder refuses. Prior 3ca67847ea2b9a71b0a41e09f943767e5a2d3a113d3e203499ee364e37f40236 -> 8c50bbc83dff4f7f5abd06078aa6abc6b64af05fddb17ee826b5f3df3d346633.", + "capture_groups_fp": 680 }, "typescript": { - "fingerprint": "f719163eb03a447c9e40ca316a905dd76cee82192a75a403df478ebbdc13e98f", + "fingerprint": "05d1dadd6c9ef35c74079fa50f341b1b36e4fb02c9a89dd1b59f32b7cfd5e633", "scaling_budget": 1.5, - "_rebaselined_2934_import_type_only": "#2934: `import-decomposer.ts` attaches a presence-only `@import.type-only` synthetic capture to specifiers `tsc` erases, so `check --cycles` can stop counting type-only edges as initialization cycles. DIGEST DRIFT ONLY, NOT A CAPTURE-SET CHANGE — the tag is added to import matches that already existed, never a new match, the same shape as the #2747 receiver-chain rebaseline. Every count is unchanged: capture_groups_fp 2414, fixture_count 155, capture_groups_small/large 4503/14403 (those measure the SYNTHETIC scaling source, which has no imports at all). The fingerprint moves because `canonicalizeMatch` in measure.mjs hashes every TAG on every match, synthetics included, so one extra presence-only tag on an existing match rewrites that match's canonical string. Attribution is exact, not inferred: neutralizing ONLY the `m['@import.type-only'] = …` assignment in import-decomposer.ts and re-running returns the fingerprint to c2fbf8a89e5686dd… byte-for-byte, so nothing else in the TypeScript capture stream moved. All 14 other languages report ok. Scaling 0.997 < 1.5. NOTE ON THE CONTROL: javascript did not move (2026993b…, 43 fixtures), but it is a WEAK control here — `import type` is TypeScript-only syntax, so a JS corpus cannot express the construct and could not have drifted either way. It evidences no collateral damage, not the correctness of the TS change; the exact-attribution check above is what does that. Prior c2fbf8a89e5686dd1ff3659b20d41d8b05ebcc9790356e3653ee0c8ca5d365c8 -> f719163eb03a447c9e40ca316a905dd76cee82192a75a403df478ebbdc13e98f.", + "_rebaselined_2934_import_type_only": "#2934: `import-decomposer.ts` attaches a presence-only `@import.type-only` synthetic capture to specifiers `tsc` erases, so `check --cycles` can stop counting type-only edges as initialization cycles. DIGEST DRIFT ONLY, NOT A CAPTURE-SET CHANGE \u2014 the tag is added to import matches that already existed, never a new match, the same shape as the #2747 receiver-chain rebaseline. Every count is unchanged: capture_groups_fp 2414, fixture_count 155, capture_groups_small/large 4503/14403 (those measure the SYNTHETIC scaling source, which has no imports at all). The fingerprint moves because `canonicalizeMatch` in measure.mjs hashes every TAG on every match, synthetics included, so one extra presence-only tag on an existing match rewrites that match's canonical string. Attribution is exact, not inferred: neutralizing ONLY the `m['@import.type-only'] = \u2026` assignment in import-decomposer.ts and re-running returns the fingerprint to c2fbf8a89e5686dd\u2026 byte-for-byte, so nothing else in the TypeScript capture stream moved. All 14 other languages report ok. Scaling 0.997 < 1.5. NOTE ON THE CONTROL: javascript did not move (2026993b\u2026, 43 fixtures), but it is a WEAK control here \u2014 `import type` is TypeScript-only syntax, so a JS corpus cannot express the construct and could not have drifted either way. It evidences no collateral damage, not the correctness of the TS change; the exact-attribution check above is what does that. Prior c2fbf8a89e5686dd1ff3659b20d41d8b05ebcc9790356e3653ee0c8ca5d365c8 -> f719163eb03a447c9e40ca316a905dd76cee82192a75a403df478ebbdc13e98f.", "_rebaselined_callable_flow_2522_review": "PR #2522 review hardening: callable operands retain expression/qualified identity and formals retain signature metadata. Prior 27f937bfb47d4bded316ea3c785ff659c8cd88a5761d928f113477a08c802c78 -> e05446620c5b80b7aae291cfdf32f693580fada2ae687124769b04a0c03bfe63; scaling 0.983 < 1.5.", "_rebaselined_callable_flow_2522_followup": "PR #2522 follow-up: lexical callable bindings, direct-callee argument metadata, and invocation-result suppression. Prior db5933cc6760234ed7d495123410feba6de243646d583f20d43032b9459f81fd -> 27f937bfb47d4bded316ea3c785ff659c8cd88a5761d928f113477a08c802c78; scaling 0.975 < 1.5.", "_rebaselined_callable_flow": "Callable assignment/copy/formal/argument/invoke facts (also consumed by Vue script blocks). Prior 25de86fd3377132c4e35d3d98f4f94a58e0cfeb7c22948a8ea3be4e793be74fd -> db5933cc6760234ed7d495123410feba6de243646d583f20d43032b9459f81fd; measured scaling ratio 0.951 < 1.5.", - "_rebaselined": "#1962: F44 (class scope@), F85 (enum member declarations), F87 (optional_parameter type annotations) add new captures — fingerprint drift expected.", - "_note": "#1968: F44, F85, F87 — fingerprint drift expected.", + "_rebaselined": "#1962: F44 (class scope@), F85 (enum member declarations), F87 (optional_parameter type annotations) add new captures \u2014 fingerprint drift expected.", + "_note": "#1968: F44, F85, F87 \u2014 fingerprint drift expected.", "_rebaselined_2522": "#2522 intentional @reference.value-ref/property-key capture additions. GitHub Actions run 29553361660 job 87800394279: prior 3f44a4a6892698df2d145c8ff2812c3b318807648983c88aca28fbd694f172f9 -> 25de86fd3377132c4e35d3d98f4f94a58e0cfeb7c22948a8ea3be4e793be74fd; scaling ratio 0.987 < 1.5.", "_rebaselined_2550_instance_model": "PR #2549 (#2545/#2551): object literals emit @scope.object (was unscoped, then @scope.block during development). Prior e05446620c5b80b7aae291cfdf32f693580fada2ae687124769b04a0c03bfe63 -> 3280b13d3f9378ab23eee31c2edc779b5a9ae1e7bb510c23a24855b44406d2f4; scaling 0.981 < 1.5.", - "_rebaselined_receiver_owner_2701": "#2701: every non-arrow function form now carries a `@receiver-owner.this` marker on the same node as `@scope.function`, so a scope that BINDS its own `this` can stop the receiver walk (`Scope.ownsReceivers`). Verified before re-baselining by diffing the capture-name histogram over this same fixture corpus against 1d3088173f6f93827641b476d614d5d15cd4f3ea: the ONLY delta is @receiver-owner.this (typescript +143, javascript +32) — every other capture count is byte-identical, so no existing capture moved. Prior 3280b13d3f9378ab23eee31c2edc779b5a9ae1e7bb510c23a24855b44406d2f4 -> 281e95484203b481094729ca249ef0423c41273eac35e424cdfd032a0dac7699.", - "_rebaselined_receiver_chain_2747": "#2747 receiver-chain rollout: call matches whose receiver is itself an expression now carry `@reference.receiver-chain`, a compact encoding of the receiver's structure, so resolution types it by folding instead of re-parsing receiver source text. Capture GROUP counts are unchanged — the tag is added to existing call matches, never a new match — so this is digest drift only. Prior cad25be9f81d6e021ebae8dcb166bc0af3a1ba8021f1506f6ca93fd4c2649000 -> 9e112415f1169f08576826c12ea1d137d1994e34b44c45986c9ffee83b8b4edc.", - "_rebaselined_2766_receiver_chain_wire_v2": "#2766: receiver-chain wire format v1 -> v2 (name-free `await` / `index` step kinds). The VERSION prefix is part of every emitted `@reference.receiver-chain` capture, so every chain-minting language's capture text changed. WIRE-FORMAT CHANGE, NOT A CAPTURE-SET CHANGE: the same chains are minted for the same sites, spelled `2|…` instead of `1|…`. Exactly the 12 chain-minting languages drifted; c, cobol and dart did not, which is the check that this is the prefix and not a capture regression. Accompanied by SCHEMA_BUMP 34 -> 37 and INCREMENTAL_SCHEMA_VERSION 28 -> 31 so a stale index is rejected rather than replaying chains a v2 decoder refuses. Prior 9e112415f1169f08576826c12ea1d137d1994e34b44c45986c9ffee83b8b4edc -> cdefe88d3c275f31953216c676ef32c7bf5727d56b9c3840b81ee6bf85749dff.", - "_rebaselined_inferred_field_receiver_2807": "#2807: inference-typed class fields now emit a type binding — `public_field_definition` with a `new_expression` value, and `this. = new ...` carrying a @type-binding.this-field marker. ADDS @type-binding.constructor captures only; no capture is removed, and the annotated form is unchanged because annotation outranks constructor-inferred in typeBindingStrength. Prior cdefe88d3c275f31953216c676ef32c7bf5727d56b9c3840b81ee6bf85749dff -> 248b56f0d7a0a6fc7a949dc7afb8611e135ed642bccc2631b96ebb9d686bb965; scaling 0.994 < 1.5.", - "_rebaselined_ts_heritage_2842": "#2842 review: TypeScript heritage capture now emits `@reference.inherits` for `interface_declaration` (bases on `extends_type_clause`) and `abstract_class_declaration` (bases on `class_heritage`), which were both silently skipped — so `interface B extends A` and `abstract class X implements I` produced no edge and every interface-dispatch walk dead-ended on a bodiless declaration. Verified before re-baselining by diffing the capture-name histogram over this same fixture corpus (145 files) with and without the change: the ONLY deltas are @reference.inherits 17 -> 20 (+3) and its paired @reference.name 245 -> 248 (+3), emitted together by emitTsInheritanceBase. Every other capture count is byte-identical, so no existing capture moved. The +3 is the three `interface X extends BasePayload` declarations in typescript-generic-calls/src/{auth,admin,guest}.ts. javascript is unchanged (no interfaces in the language). Prior 248b56f0d7a0a6fc7a949dc7afb8611e135ed642bccc2631b96ebb9d686bb965 -> 7a960908031331360ce582f5b55b7681e1cd7f8a2eabfd73c00982cb17f2a949.", + "_rebaselined_receiver_owner_2701": "#2701: every non-arrow function form now carries a `@receiver-owner.this` marker on the same node as `@scope.function`, so a scope that BINDS its own `this` can stop the receiver walk (`Scope.ownsReceivers`). Verified before re-baselining by diffing the capture-name histogram over this same fixture corpus against 1d3088173f6f93827641b476d614d5d15cd4f3ea: the ONLY delta is @receiver-owner.this (typescript +143, javascript +32) \u2014 every other capture count is byte-identical, so no existing capture moved. Prior 3280b13d3f9378ab23eee31c2edc779b5a9ae1e7bb510c23a24855b44406d2f4 -> 281e95484203b481094729ca249ef0423c41273eac35e424cdfd032a0dac7699.", + "_rebaselined_receiver_chain_2747": "#2747 receiver-chain rollout: call matches whose receiver is itself an expression now carry `@reference.receiver-chain`, a compact encoding of the receiver's structure, so resolution types it by folding instead of re-parsing receiver source text. Capture GROUP counts are unchanged \u2014 the tag is added to existing call matches, never a new match \u2014 so this is digest drift only. Prior cad25be9f81d6e021ebae8dcb166bc0af3a1ba8021f1506f6ca93fd4c2649000 -> 9e112415f1169f08576826c12ea1d137d1994e34b44c45986c9ffee83b8b4edc.", + "_rebaselined_2766_receiver_chain_wire_v2": "#2766: receiver-chain wire format v1 -> v2 (name-free `await` / `index` step kinds). The VERSION prefix is part of every emitted `@reference.receiver-chain` capture, so every chain-minting language's capture text changed. WIRE-FORMAT CHANGE, NOT A CAPTURE-SET CHANGE: the same chains are minted for the same sites, spelled `2|\u2026` instead of `1|\u2026`. Exactly the 12 chain-minting languages drifted; c, cobol and dart did not, which is the check that this is the prefix and not a capture regression. Accompanied by SCHEMA_BUMP 34 -> 37 and INCREMENTAL_SCHEMA_VERSION 28 -> 31 so a stale index is rejected rather than replaying chains a v2 decoder refuses. Prior 9e112415f1169f08576826c12ea1d137d1994e34b44c45986c9ffee83b8b4edc -> cdefe88d3c275f31953216c676ef32c7bf5727d56b9c3840b81ee6bf85749dff.", + "_rebaselined_inferred_field_receiver_2807": "#2807: inference-typed class fields now emit a type binding \u2014 `public_field_definition` with a `new_expression` value, and `this. = new ...` carrying a @type-binding.this-field marker. ADDS @type-binding.constructor captures only; no capture is removed, and the annotated form is unchanged because annotation outranks constructor-inferred in typeBindingStrength. Prior cdefe88d3c275f31953216c676ef32c7bf5727d56b9c3840b81ee6bf85749dff -> 248b56f0d7a0a6fc7a949dc7afb8611e135ed642bccc2631b96ebb9d686bb965; scaling 0.994 < 1.5.", + "_rebaselined_ts_heritage_2842": "#2842 review: TypeScript heritage capture now emits `@reference.inherits` for `interface_declaration` (bases on `extends_type_clause`) and `abstract_class_declaration` (bases on `class_heritage`), which were both silently skipped \u2014 so `interface B extends A` and `abstract class X implements I` produced no edge and every interface-dispatch walk dead-ended on a bodiless declaration. Verified before re-baselining by diffing the capture-name histogram over this same fixture corpus (145 files) with and without the change: the ONLY deltas are @reference.inherits 17 -> 20 (+3) and its paired @reference.name 245 -> 248 (+3), emitted together by emitTsInheritanceBase. Every other capture count is byte-identical, so no existing capture moved. The +3 is the three `interface X extends BasePayload` declarations in typescript-generic-calls/src/{auth,admin,guest}.ts. javascript is unchanged (no interfaces in the language). Prior 248b56f0d7a0a6fc7a949dc7afb8611e135ed642bccc2631b96ebb9d686bb965 -> 7a960908031331360ce582f5b55b7681e1cd7f8a2eabfd73c00982cb17f2a949.", "capture_groups_small": 4503, "capture_groups_large": 14403, - "capture_groups_fp": 2414, - "fixture_count": 155, - "_rebaselined_blind_spots_2856": "#2856 blind-spots series: the JS/TS SCOPE queries gained capture rules, so fingerprint drift is expected and additive. Verified before re-baselining by diffing the capture-name sets in both scope queries against origin/main: TypeScript gained exactly @reference.read.identifier (A2 bare-identifier reads in value positions) and @reference.type (R2-2 type references, so a declared contract stops reporting incoming:{}); JavaScript gained exactly @reference.read.identifier, @reference.read.destructured (R2-1c) and @reference.write.property-key (R2-1b record-construction writes). NOTHING was removed on either side — the delta is a pure superset, which is the check that no existing capture moved. capture_groups_small/large are unchanged (4503/14403) because those measure the SYNTHETIC scaling source, which this branch does not touch; only the fixture-corpus count moves. capture_groups_fp 2097 -> 2338 and fixture_count 146 -> 151 from 21 new lang-resolution fixtures. Scaling stayed linear and inside budget: typescript 1.116 < 1.5, javascript 1.010 < 1.5. Prior typescript ed92588e0fc7b28b3a0174339ac378b4dd85965fe007db1208dea97a65ce0571 -> f66a3e6f1e096431e7046505129a627deaa00ca0de5bc846b080591b397248f7; prior javascript 806f70ad3cce5fc849f6d06a08ace8a95f92a1ea84a2418fddabb1eef5846594 -> 2026993b81b873839dd2ef8797d9c14d9c48516b2b57b05ac17d8d43f2f4eba3.", - "_rebaselined_type_parameter_shadowing_w2_8": "W2-8: `@declaration.type-parameters` is now captured on generic FUNCTIONS, generator functions and type ALIASES, not only on class/interface declarations. NO NEW CAPTURE NAME — verified by diffing the capture-name sets against the wave-1 branch, which returns empty; the tag already existed and simply fires on more declarations. That is the whole delta: capture_groups_fp 2338 -> 2371 (+33 occurrences of an existing tag) and fixture_count 151 -> 152 (one new fixture, typescript-type-parameters). capture_groups_small/large unchanged at 4503/14403, since those measure the synthetic scaling source this does not touch. Scaling 1.06 < 1.5. JavaScript is untouched — it has no type parameters — and its fingerprint does not move, which is the check that this is the TS declaration rules and not something broader. Prior f66a3e6f1e096431e7046505129a627deaa00ca0de5bc846b080591b397248f7 -> 62c7f1bfbe568eed927fb78f00061ed5e49d12511fd8260648b876df386f3b4c.", - "_rebaselined_2899_review_type_parameter_scope_fixtures": "PR #2899 review follow-up: FIXTURE-CORPUS GROWTH ONLY — no query rule changed and no capture name was added or removed. `typescript/query.ts` is byte-identical to the previous baseline; the type-parameter shadowing defect was fixed on the RESOLUTION side (`walkers.ts` gains a `declarationOpenedScope` gate so a declaration's `typeParameters` bind only inside the scope that declaration opened, and the `USES` guard moved from `graph-bridge/references-to-edges.ts` to `resolve-references.ts` where the spelled `site.name` is in hand). The fingerprint moves because measure.mjs fingerprints the whole `lang-resolution/typescript-*` fixture corpus and the regression tests add three files to `typescript-type-parameters/src/` (values.ts, aliased.ts, namespaced.ts) plus two scope-less generic aliases in shapes.ts. Per-file accounting sums exactly to the delta: shapes.ts 33->35 (+2), values.ts +11, aliased.ts +10, namespaced.ts +20 = +43. capture_groups_fp 2371 -> 2414; fixture_count 152 -> 155. capture_groups_small/large unchanged at 4503/14403 (they measure the SYNTHETIC scaling source, untouched). JAVASCRIPT IS THE CONTROL AND DID NOT MOVE (fingerprint 2026993b..., 43 fixtures) — which is the check that this is corpus growth and not a capture regression; all 14 other languages report `ok`. Scaling 0.976 < 1.5. Prior 62c7f1bfbe568eed927fb78f00061ed5e49d12511fd8260648b876df386f3b4c -> c2fbf8a89e5686dd1ff3659b20d41d8b05ebcc9790356e3653ee0c8ca5d365c8." + "capture_groups_fp": 2465, + "fixture_count": 167, + "_rebaselined_blind_spots_2856": "#2856 blind-spots series: the JS/TS SCOPE queries gained capture rules, so fingerprint drift is expected and additive. Verified before re-baselining by diffing the capture-name sets in both scope queries against origin/main: TypeScript gained exactly @reference.read.identifier (A2 bare-identifier reads in value positions) and @reference.type (R2-2 type references, so a declared contract stops reporting incoming:{}); JavaScript gained exactly @reference.read.identifier, @reference.read.destructured (R2-1c) and @reference.write.property-key (R2-1b record-construction writes). NOTHING was removed on either side \u2014 the delta is a pure superset, which is the check that no existing capture moved. capture_groups_small/large are unchanged (4503/14403) because those measure the SYNTHETIC scaling source, which this branch does not touch; only the fixture-corpus count moves. capture_groups_fp 2097 -> 2338 and fixture_count 146 -> 151 from 21 new lang-resolution fixtures. Scaling stayed linear and inside budget: typescript 1.116 < 1.5, javascript 1.010 < 1.5. Prior typescript ed92588e0fc7b28b3a0174339ac378b4dd85965fe007db1208dea97a65ce0571 -> f66a3e6f1e096431e7046505129a627deaa00ca0de5bc846b080591b397248f7; prior javascript 806f70ad3cce5fc849f6d06a08ace8a95f92a1ea84a2418fddabb1eef5846594 -> 2026993b81b873839dd2ef8797d9c14d9c48516b2b57b05ac17d8d43f2f4eba3.", + "_rebaselined_type_parameter_shadowing_w2_8": "W2-8: `@declaration.type-parameters` is now captured on generic FUNCTIONS, generator functions and type ALIASES, not only on class/interface declarations. NO NEW CAPTURE NAME \u2014 verified by diffing the capture-name sets against the wave-1 branch, which returns empty; the tag already existed and simply fires on more declarations. That is the whole delta: capture_groups_fp 2338 -> 2371 (+33 occurrences of an existing tag) and fixture_count 151 -> 152 (one new fixture, typescript-type-parameters). capture_groups_small/large unchanged at 4503/14403, since those measure the synthetic scaling source this does not touch. Scaling 1.06 < 1.5. JavaScript is untouched \u2014 it has no type parameters \u2014 and its fingerprint does not move, which is the check that this is the TS declaration rules and not something broader. Prior f66a3e6f1e096431e7046505129a627deaa00ca0de5bc846b080591b397248f7 -> 62c7f1bfbe568eed927fb78f00061ed5e49d12511fd8260648b876df386f3b4c.", + "_rebaselined_2899_review_type_parameter_scope_fixtures": "PR #2899 review follow-up: FIXTURE-CORPUS GROWTH ONLY \u2014 no query rule changed and no capture name was added or removed. `typescript/query.ts` is byte-identical to the previous baseline; the type-parameter shadowing defect was fixed on the RESOLUTION side (`walkers.ts` gains a `declarationOpenedScope` gate so a declaration's `typeParameters` bind only inside the scope that declaration opened, and the `USES` guard moved from `graph-bridge/references-to-edges.ts` to `resolve-references.ts` where the spelled `site.name` is in hand). The fingerprint moves because measure.mjs fingerprints the whole `lang-resolution/typescript-*` fixture corpus and the regression tests add three files to `typescript-type-parameters/src/` (values.ts, aliased.ts, namespaced.ts) plus two scope-less generic aliases in shapes.ts. Per-file accounting sums exactly to the delta: shapes.ts 33->35 (+2), values.ts +11, aliased.ts +10, namespaced.ts +20 = +43. capture_groups_fp 2371 -> 2414; fixture_count 152 -> 155. capture_groups_small/large unchanged at 4503/14403 (they measure the SYNTHETIC scaling source, untouched). JAVASCRIPT IS THE CONTROL AND DID NOT MOVE (fingerprint 2026993b..., 43 fixtures) \u2014 which is the check that this is corpus growth and not a capture regression; all 14 other languages report `ok`. Scaling 0.976 < 1.5. Prior 62c7f1bfbe568eed927fb78f00061ed5e49d12511fd8260648b876df386f3b4c -> c2fbf8a89e5686dd1ff3659b20d41d8b05ebcc9790356e3653ee0c8ca5d365c8.", + "_rebaselined_2953_workspace_fixture": "#2953 adds test/fixtures/lang-resolution/typescript-pnpm-workspace-imports, a pnpm monorepo of 12 .ts files, and the TypeScript capture corpus is collected from test/fixtures. CORPUS GROWTH ONLY, NOT A CAPTURE CHANGE: fixture_count 155 -> 167 and capture_groups_fp 2414 -> 2465 are the 12 new files' own matches; capture_groups_small/large are unchanged at 4503/14403 because those measure the SYNTHETIC scaling source, which the fixture corpus does not feed. Attribution is exact rather than inferred: moving that one fixture directory aside and re-running returns typescript to f719163eb03a447c9e40ca316a905dd76cee82192a75a403df478ebbdc13e98f byte-for-byte with fixture_count back at 155, and [scope-capture --check] PASSES for all 15 languages - so nothing in the TypeScript capture stream moved. #2953 changes import RESOLUTION, which runs after capture and feeds no capture tag. Prior f719163eb03a447c9e40ca316a905dd76cee82192a75a403df478ebbdc13e98f -> 05d1dadd6c9ef35c74079fa50f341b1b36e4fb02c9a89dd1b59f32b7cfd5e633." }, "javascript": { "fingerprint": "2026993b81b873839dd2ef8797d9c14d9c48516b2b57b05ac17d8d43f2f4eba3", @@ -196,10 +203,10 @@ "_rebaselined": "#1956 synth-widening: + javascript-qualified-base fixture; synthesizeJsInheritanceReferences now handles a member_expression base (class S extends ns.Base -> Base), matching the #1940 legacy leg + the TS terminalTsTypeNameNode property_identifier case, at parity. Linear (~1.05). | #942: scope-resolution-only cleanup reworded fixture comments; capture byte-positions shift, capture LOGIC unchanged.", "_rebaselined_2522": "#2522 intentional @reference.value-ref/property-key capture additions. GitHub Actions run 29553361660 job 87800394279: prior d72f03c6c502235d2d4b74d66baa5c7d361f040d7a1b72e84acad61210d05ae8 -> 5567dd47e7ba29821a518c4a9852adc3b774e25ef3e7a6e2b3ecb7b59ddab73c; scaling ratio 1.031 < 1.5.", "_rebaselined_2550_instance_model": "PR #2549 (#2545/#2551): object literals emit @scope.object. Prior 479927409bbdd9852a36172c8260aa56df260e99129a7a9c20a0d1903dd5538b -> f1ccf42a36895c8e34dcb724286f247d469835f2dcbb23ad3347190adc7fde1c; scaling 1.096 < 1.5.", - "_rebaselined_receiver_owner_2701": "#2701: every non-arrow function form now carries a `@receiver-owner.this` marker on the same node as `@scope.function`, so a scope that BINDS its own `this` can stop the receiver walk (`Scope.ownsReceivers`). Verified before re-baselining by diffing the capture-name histogram over this same fixture corpus against 1d3088173f6f93827641b476d614d5d15cd4f3ea: the ONLY delta is @receiver-owner.this (typescript +143, javascript +32) — every other capture count is byte-identical, so no existing capture moved. Prior f1ccf42a36895c8e34dcb724286f247d469835f2dcbb23ad3347190adc7fde1c -> 90601494695b834d3a9af7ac4844eac603f4f432809a05554cc59de0674a4354.", - "_rebaselined_receiver_chain_2747": "#2747 receiver-chain rollout: call matches whose receiver is itself an expression now carry `@reference.receiver-chain`, a compact encoding of the receiver's structure, so resolution types it by folding instead of re-parsing receiver source text. Capture GROUP counts are unchanged — the tag is added to existing call matches, never a new match — so this is digest drift only. Prior 1c71ef628eb75a3b111afa8c2a7c351c16a7f5aab9fac2f098f82b2866312aa8 -> 83344b7cba093702f4528eeee44e438809c229d43b12e69ed288812ce7ffc7bc.", - "_rebaselined_2766_receiver_chain_wire_v2": "#2766: receiver-chain wire format v1 -> v2 (name-free `await` / `index` step kinds). The VERSION prefix is part of every emitted `@reference.receiver-chain` capture, so every chain-minting language's capture text changed. WIRE-FORMAT CHANGE, NOT A CAPTURE-SET CHANGE: the same chains are minted for the same sites, spelled `2|…` instead of `1|…`. Exactly the 12 chain-minting languages drifted; c, cobol and dart did not, which is the check that this is the prefix and not a capture regression. Accompanied by SCHEMA_BUMP 34 -> 37 and INCREMENTAL_SCHEMA_VERSION 28 -> 31 so a stale index is rejected rather than replaying chains a v2 decoder refuses. Prior 83344b7cba093702f4528eeee44e438809c229d43b12e69ed288812ce7ffc7bc -> 806f70ad3cce5fc849f6d06a08ace8a95f92a1ea84a2418fddabb1eef5846594.", - "_rebaselined_blind_spots_2856": "#2856 blind-spots series: the JS/TS SCOPE queries gained capture rules, so fingerprint drift is expected and additive. Verified before re-baselining by diffing the capture-name sets in both scope queries against origin/main: TypeScript gained exactly @reference.read.identifier (A2 bare-identifier reads in value positions) and @reference.type (R2-2 type references, so a declared contract stops reporting incoming:{}); JavaScript gained exactly @reference.read.identifier, @reference.read.destructured (R2-1c) and @reference.write.property-key (R2-1b record-construction writes). NOTHING was removed on either side — the delta is a pure superset, which is the check that no existing capture moved. capture_groups_small/large are unchanged (4503/14403) because those measure the SYNTHETIC scaling source, which this branch does not touch; only the fixture-corpus count moves. capture_groups_fp 2097 -> 2338 and fixture_count 146 -> 151 from 21 new lang-resolution fixtures. Scaling stayed linear and inside budget: typescript 1.116 < 1.5, javascript 1.010 < 1.5. Prior typescript ed92588e0fc7b28b3a0174339ac378b4dd85965fe007db1208dea97a65ce0571 -> f66a3e6f1e096431e7046505129a627deaa00ca0de5bc846b080591b397248f7; prior javascript 806f70ad3cce5fc849f6d06a08ace8a95f92a1ea84a2418fddabb1eef5846594 -> 2026993b81b873839dd2ef8797d9c14d9c48516b2b57b05ac17d8d43f2f4eba3." + "_rebaselined_receiver_owner_2701": "#2701: every non-arrow function form now carries a `@receiver-owner.this` marker on the same node as `@scope.function`, so a scope that BINDS its own `this` can stop the receiver walk (`Scope.ownsReceivers`). Verified before re-baselining by diffing the capture-name histogram over this same fixture corpus against 1d3088173f6f93827641b476d614d5d15cd4f3ea: the ONLY delta is @receiver-owner.this (typescript +143, javascript +32) \u2014 every other capture count is byte-identical, so no existing capture moved. Prior f1ccf42a36895c8e34dcb724286f247d469835f2dcbb23ad3347190adc7fde1c -> 90601494695b834d3a9af7ac4844eac603f4f432809a05554cc59de0674a4354.", + "_rebaselined_receiver_chain_2747": "#2747 receiver-chain rollout: call matches whose receiver is itself an expression now carry `@reference.receiver-chain`, a compact encoding of the receiver's structure, so resolution types it by folding instead of re-parsing receiver source text. Capture GROUP counts are unchanged \u2014 the tag is added to existing call matches, never a new match \u2014 so this is digest drift only. Prior 1c71ef628eb75a3b111afa8c2a7c351c16a7f5aab9fac2f098f82b2866312aa8 -> 83344b7cba093702f4528eeee44e438809c229d43b12e69ed288812ce7ffc7bc.", + "_rebaselined_2766_receiver_chain_wire_v2": "#2766: receiver-chain wire format v1 -> v2 (name-free `await` / `index` step kinds). The VERSION prefix is part of every emitted `@reference.receiver-chain` capture, so every chain-minting language's capture text changed. WIRE-FORMAT CHANGE, NOT A CAPTURE-SET CHANGE: the same chains are minted for the same sites, spelled `2|\u2026` instead of `1|\u2026`. Exactly the 12 chain-minting languages drifted; c, cobol and dart did not, which is the check that this is the prefix and not a capture regression. Accompanied by SCHEMA_BUMP 34 -> 37 and INCREMENTAL_SCHEMA_VERSION 28 -> 31 so a stale index is rejected rather than replaying chains a v2 decoder refuses. Prior 83344b7cba093702f4528eeee44e438809c229d43b12e69ed288812ce7ffc7bc -> 806f70ad3cce5fc849f6d06a08ace8a95f92a1ea84a2418fddabb1eef5846594.", + "_rebaselined_blind_spots_2856": "#2856 blind-spots series: the JS/TS SCOPE queries gained capture rules, so fingerprint drift is expected and additive. Verified before re-baselining by diffing the capture-name sets in both scope queries against origin/main: TypeScript gained exactly @reference.read.identifier (A2 bare-identifier reads in value positions) and @reference.type (R2-2 type references, so a declared contract stops reporting incoming:{}); JavaScript gained exactly @reference.read.identifier, @reference.read.destructured (R2-1c) and @reference.write.property-key (R2-1b record-construction writes). NOTHING was removed on either side \u2014 the delta is a pure superset, which is the check that no existing capture moved. capture_groups_small/large are unchanged (4503/14403) because those measure the SYNTHETIC scaling source, which this branch does not touch; only the fixture-corpus count moves. capture_groups_fp 2097 -> 2338 and fixture_count 146 -> 151 from 21 new lang-resolution fixtures. Scaling stayed linear and inside budget: typescript 1.116 < 1.5, javascript 1.010 < 1.5. Prior typescript ed92588e0fc7b28b3a0174339ac378b4dd85965fe007db1208dea97a65ce0571 -> f66a3e6f1e096431e7046505129a627deaa00ca0de5bc846b080591b397248f7; prior javascript 806f70ad3cce5fc849f6d06a08ace8a95f92a1ea84a2418fddabb1eef5846594 -> 2026993b81b873839dd2ef8797d9c14d9c48516b2b57b05ac17d8d43f2f4eba3." }, "kotlin": { "fingerprint": "a184f8ff0ae40d246db855b63f7ff26bda3afac03e5f4c76e4593c7e2cefce54", @@ -208,13 +215,13 @@ "_rebaselined_callable_flow_2522_followup": "PR #2522 follow-up: Kotlin callable-reference flow facts with invocation-result suppression. Prior 4900431791f2b9280009deb2b82659c26ead8aa6fb8731190a7c505dec5a9041 -> bddba25d5a88152bbbee8d70e82c944b5302accb4b625df782adb1d4f7a7ac12; scaling 0.880 < 1.5.", "_added": "#1951: bench coverage added (was ungated); scale source heritage-bearing (: Base()); js/kotlin O(n^2) findNodeAtRange-per-match fixed to threaded captured node, now linear.", "_rebaselined": "#1919 review CF3 fix: extended kotlin-local-property-owner (init/accessor destructuring) + new dart-accessor-owner fixture (getter/setter ownership). Fingerprint-only corpus drift; scaling ~1.0.", - "_rebaselined_2271": "PR #2271: re-vendored tree-sitter-kotlin 0.3.8 -> unreleased fwcd main c8ac3d26 for `fun interface` support + new kotlin-fun-interface fixture in the corpus. Drift is both corpus-additive (the fixture) and grammar-driven (the new grammar parses `fun interface` as a class_declaration, not an ERROR node). Baselined to the NEW grammar's fingerprint, so this --check passes only once the regenerated prebuilds land — until then CI loads the committed 0.3.8 binary and the bench is red, same as the kotlin fun-interface integration tests. scaling ~0.83 (linear).", + "_rebaselined_2271": "PR #2271: re-vendored tree-sitter-kotlin 0.3.8 -> unreleased fwcd main c8ac3d26 for `fun interface` support + new kotlin-fun-interface fixture in the corpus. Drift is both corpus-additive (the fixture) and grammar-driven (the new grammar parses `fun interface` as a class_declaration, not an ERROR node). Baselined to the NEW grammar's fingerprint, so this --check passes only once the regenerated prebuilds land \u2014 until then CI loads the committed 0.3.8 binary and the bench is red, same as the kotlin fun-interface integration tests. scaling ~0.83 (linear).", "_rebaselined_2522_review_fixes": "PR #2522 review fixes: fieldless assignment nodes decomposed positionally. Prior e856951c2a779163d555dadc8e1bf59304a86caed78ac1f450d9caa2b50f63d1 -> 4b31f46cfb004ba769a96feeb06ae4ef109c77410f54e7aaab4a688df599b112; scaling ratio re-verified within budget.", "_rebaselined_2550_instance_model": "PR #2549 (#2545): anonymous object expressions (object_literal) emit @scope.class, and the kotlin-object-literal-scope fixture joined the corpus. Prior 4b31f46cfb004ba769a96feeb06ae4ef109c77410f54e7aaab4a688df599b112 -> a6fce0dff00e88d41d85023eaf3f35016b5217c7e5225f24a598e4c70bb63091; scaling 0.951 < 1.5.", "_rebaselined_2563_instance_ownership": "#2563: kotlin-instance-ownership adds unrelated, inherited, outer-instance, and anonymous-object coverage. Prior a6fce0dff00e88d41d85023eaf3f35016b5217c7e5225f24a598e4c70bb63091 -> 9f159f8810d342ef1c821f466efd6920dad9a190f06000056e6cd2815861b195; scaling 1.257 < 1.5.", - "_rebaselined_receiver_chain_2747": "#2747 receiver-chain rollout: call matches whose receiver is itself an expression now carry `@reference.receiver-chain`, a compact encoding of the receiver's structure, so resolution types it by folding instead of re-parsing receiver source text. Capture GROUP counts are unchanged — the tag is added to existing call matches, never a new match — so this is digest drift only. Prior 9f159f8810d342ef1c821f466efd6920dad9a190f06000056e6cd2815861b195 -> d3c4d2fa0d82d248a2299cfc888b067187ad1faf2c87a97f93c6ed835eefc3f1.", - "_rebaselined_2766_receiver_chain_wire_v2": "#2766: receiver-chain wire format v1 -> v2 (name-free `await` / `index` step kinds). The VERSION prefix is part of every emitted `@reference.receiver-chain` capture, so every chain-minting language's capture text changed. WIRE-FORMAT CHANGE, NOT A CAPTURE-SET CHANGE: the same chains are minted for the same sites, spelled `2|…` instead of `1|…`. Exactly the 12 chain-minting languages drifted; c, cobol and dart did not, which is the check that this is the prefix and not a capture regression. Accompanied by SCHEMA_BUMP 34 -> 37 and INCREMENTAL_SCHEMA_VERSION 28 -> 31 so a stale index is rejected rather than replaying chains a v2 decoder refuses. Prior d3c4d2fa0d82d248a2299cfc888b067187ad1faf2c87a97f93c6ed835eefc3f1 -> c1f0cc9058ab11b7cd6fc8b440deb6db2b2f530f2eb21178923e68a3d0796c4b.", - "_rebaselined_2766_await_subscript_emission": "#2766: extractMixedChain now walks THROUGH await and subscript nodes and peels transparent wrappers at loop entry, so sites whose receiver is `repos[0]` or `(await f())` mint a receiver chain where they previously minted none. EMISSION CHANGE: more sites carry `@reference.receiver-chain`; no existing chain changed shape. Only go and kotlin drifted of 15 — the two whose fixture corpora contain such receivers. Prior c1f0cc9058ab11b7cd6fc8b440deb6db2b2f530f2eb21178923e68a3d0796c4b -> efd5dbf80ffcd3bab2834d1010f6fe2b239dcc5d58229938dea9cff8d0f380f2.", + "_rebaselined_receiver_chain_2747": "#2747 receiver-chain rollout: call matches whose receiver is itself an expression now carry `@reference.receiver-chain`, a compact encoding of the receiver's structure, so resolution types it by folding instead of re-parsing receiver source text. Capture GROUP counts are unchanged \u2014 the tag is added to existing call matches, never a new match \u2014 so this is digest drift only. Prior 9f159f8810d342ef1c821f466efd6920dad9a190f06000056e6cd2815861b195 -> d3c4d2fa0d82d248a2299cfc888b067187ad1faf2c87a97f93c6ed835eefc3f1.", + "_rebaselined_2766_receiver_chain_wire_v2": "#2766: receiver-chain wire format v1 -> v2 (name-free `await` / `index` step kinds). The VERSION prefix is part of every emitted `@reference.receiver-chain` capture, so every chain-minting language's capture text changed. WIRE-FORMAT CHANGE, NOT A CAPTURE-SET CHANGE: the same chains are minted for the same sites, spelled `2|\u2026` instead of `1|\u2026`. Exactly the 12 chain-minting languages drifted; c, cobol and dart did not, which is the check that this is the prefix and not a capture regression. Accompanied by SCHEMA_BUMP 34 -> 37 and INCREMENTAL_SCHEMA_VERSION 28 -> 31 so a stale index is rejected rather than replaying chains a v2 decoder refuses. Prior d3c4d2fa0d82d248a2299cfc888b067187ad1faf2c87a97f93c6ed835eefc3f1 -> c1f0cc9058ab11b7cd6fc8b440deb6db2b2f530f2eb21178923e68a3d0796c4b.", + "_rebaselined_2766_await_subscript_emission": "#2766: extractMixedChain now walks THROUGH await and subscript nodes and peels transparent wrappers at loop entry, so sites whose receiver is `repos[0]` or `(await f())` mint a receiver chain where they previously minted none. EMISSION CHANGE: more sites carry `@reference.receiver-chain`; no existing chain changed shape. Only go and kotlin drifted of 15 \u2014 the two whose fixture corpora contain such receivers. Prior c1f0cc9058ab11b7cd6fc8b440deb6db2b2f530f2eb21178923e68a3d0796c4b -> efd5dbf80ffcd3bab2834d1010f6fe2b239dcc5d58229938dea9cff8d0f380f2.", "capture_groups_small": 4753, "capture_groups_large": 15203, "capture_groups_fp": 2334, diff --git a/gitnexus/package-lock.json b/gitnexus/package-lock.json index 790dbc010..23e85358b 100644 --- a/gitnexus/package-lock.json +++ b/gitnexus/package-lock.json @@ -79,7 +79,7 @@ "version": "1.0.0", "dev": true, "devDependencies": { - "typescript": "^6.0.3" + "typescript": "^7.0.2" } }, "node_modules/@babel/code-frame": { @@ -1939,9 +1939,9 @@ "license": "MIT" }, "node_modules/@types/node": { - "version": "26.1.2", - "resolved": "https://registry.npmjs.org/@types/node/-/node-26.1.2.tgz", - "integrity": "sha512-Vu4a5UFA9rIIFJ7rB/Vaafh9lrCQszopTCx6KjFboXTGQbPNasehVR5TEiithSDGyd1DEiUByggTZsg8jukeIg==", + "version": "26.2.0", + "resolved": "https://registry.npmjs.org/@types/node/-/node-26.2.0.tgz", + "integrity": "sha512-5IviulTZeRNp2vAJ514cc/HUlY5nZ9fCbq9DMyC52BrhFZACo3nI0R7qBxhQmo/d27NFe96ur/b7Wwxklda+kg==", "devOptional": true, "license": "MIT", "dependencies": { @@ -5430,9 +5430,9 @@ "license": "0BSD" }, "node_modules/tsx": { - "version": "4.23.11", - "resolved": "https://registry.npmjs.org/tsx/-/tsx-4.23.11.tgz", - "integrity": "sha512-Ry2oTEUnhBdeEdWIztY8kf3/nBGnPnjMLVGL0YfdRXMORuPER5NlKmayqxtxRxwB1xBN+RivRaJfe7PM1rtiyw==", + "version": "4.23.12", + "resolved": "https://registry.npmjs.org/tsx/-/tsx-4.23.12.tgz", + "integrity": "sha512-FDf4L4sYzKtzWYhU/Xm0AQFdTjdIxNo9ElTf2mxXM6k8YMHXzYUe4yODVaXP4V9uMFbVg8c0qyBccK2OOxb45Q==", "dev": true, "license": "MIT", "dependencies": { diff --git a/gitnexus/src/core/ingestion/import-resolvers/node-workspace-packages.ts b/gitnexus/src/core/ingestion/import-resolvers/node-workspace-packages.ts new file mode 100644 index 000000000..712359a41 --- /dev/null +++ b/gitnexus/src/core/ingestion/import-resolvers/node-workspace-packages.ts @@ -0,0 +1,527 @@ +/** + * In-repo `package.json` manifests, as module-resolution input (#2953). + * + * A bare specifier (`@acme/telemetry/nest`, `@repo/utils`, `lodash/fp`) names a + * PACKAGE, not a path, and the manifest is the only thing that says which + * packages exist and where their entry points are. Without it a resolver can do + * nothing but guess — which is what the old suffix matcher did, landing + * `@acme/telemetry/nest` on the repo's only path ending in `nest/index.ts` + * while `@repo/utils`, a real first-party package, resolved to nothing because + * its name appears in no file path at all. + * + * Both directions come from the same missing input, so both are fixed by + * reading it: every in-repo `package.json` contributes its `name`, its `exports` + * map (including subpath patterns), its legacy entry fields, and its `imports` + * map for `#`-prefixed specifiers. + */ + +import fs from 'fs/promises'; +import path from 'path'; +import { createRequire } from 'node:module'; + +import { isHardcodedIgnoredDirectory } from '../../../config/ignore-service.js'; +import { logger } from '../../logger.js'; +import { resolveFile } from '../languages/typescript/file-candidates.js'; + +// `js-yaml` is CJS; the rest of this repository reaches it the same way +// (`core/group/config-parser.ts`, `cli/group.ts`). +const _require = createRequire(import.meta.url); +const yaml = _require('js-yaml') as typeof import('js-yaml'); + +/** One in-repo package. */ +export interface NodeWorkspacePackage { + /** Repo-relative directory holding the `package.json` (`''` for the root). */ + readonly dir: string; + /** + * Repo-relative entry stems for the package root (`import '@repo/utils'`), + * best first: declared `exports["."]`, then `module` / `main` / `types`, then + * the conventional `src/index` and `index`. + * + * A published `dist/...` entry simply fails to match an indexed source file + * (build output is not indexed) and the next candidate is tried, which is why + * the conventional fallbacks stay at the end rather than being a guess: they + * are what the package resolves to when it is consumed from source, which in + * a workspace it always is. + */ + readonly entries: readonly string[]; + /** + * Declared `exports` subpaths, specifier suffix -> repo-relative stems. + * Keys are as written minus the leading `./`, so `"./nest"` is stored `nest`; + * a pattern key keeps its `*` (`"./features/*"` -> `features/*`). + */ + readonly subpathExports: ReadonlyMap; + /** Declared `imports` map, `#name` -> repo-relative stems. */ + readonly subpathImports: ReadonlyMap; +} + +export interface NodeWorkspacePackages { + /** Package name (`@repo/utils`, `utils`) -> that package. */ + readonly byName: ReadonlyMap; +} + +const SCAN_MAX_DIRS = 20_000; +const SCAN_MAX_DEPTH = 24; + +/** + * The package name a bare specifier addresses, or `null` when the specifier + * names a path rather than a package. + * + * `@acme/telemetry/nest` -> `@acme/telemetry`, `lodash/fp` -> `lodash`. + */ +export function nodePackageNameOf(specifier: string): string | null { + if (specifier === '' || specifier.startsWith('.') || specifier.startsWith('/')) return null; + if (specifier.startsWith('#')) return null; + if (specifier.startsWith('@')) { + const parts = specifier.split('/'); + return parts.length >= 2 && parts[0].length > 1 && parts[1] !== '' + ? `${parts[0]}/${parts[1]}` + : null; + } + return specifier.split('/')[0] || null; +} + +/** The in-repo package whose directory most closely contains `filePath`. */ +export function owningPackage( + filePath: string, + packages: NodeWorkspacePackages | null | undefined, +): NodeWorkspacePackage | null { + if (!packages) return null; + let best: NodeWorkspacePackage | null = null; + for (const pkg of packages.byName.values()) { + const inside = pkg.dir === '' || filePath.startsWith(`${pkg.dir}/`); + if (inside && (best === null || pkg.dir.length > best.dir.length)) best = pkg; + } + return best; +} + +/** + * Resolve a bare specifier that names an in-repo package. + * + * `null` means the specifier names no in-repo package — an external dependency, + * whose correct in-repo resolution is nothing — or names one that does not + * export the requested subpath. + */ +export function resolveNodeWorkspaceImport( + specifier: string, + packages: NodeWorkspacePackages | null | undefined, + allFiles: ReadonlySet, +): string | null { + if (!packages) return null; + const packageName = nodePackageNameOf(specifier); + if (packageName === null) return null; + const pkg = packages.byName.get(packageName); + if (pkg === undefined) return null; + + const subpath = specifier.slice(packageName.length).replace(/^\//, ''); + for (const stem of entryStemsFor(pkg, subpath)) { + const hit = resolveFile(stem, allFiles); + if (hit !== null) return hit; + } + return null; +} + +/** + * Look a specifier up in a subpath map — `exports` or `imports`, which share + * Node's matching rule exactly: an exact key wins, otherwise the pattern with + * the longest literal prefix does, and its `*` takes whatever the specifier put + * there. + * + * Shared because they diverged once: the `imports` side did an exact lookup + * only, so a declared `"#internal/*"` could never match `#internal/foo`. + */ +export function matchSubpathMap( + map: ReadonlyMap, + specifier: string, +): readonly string[] | null { + const exact = map.get(specifier); + if (exact !== undefined) return exact; + + const patterns = [...map.entries()] + .filter(([key]) => key.includes('*')) + .map(([key, stems]) => { + const star = key.indexOf('*'); + return { prefix: key.slice(0, star), suffix: key.slice(star + 1), stems }; + }) + .filter( + ({ prefix, suffix }) => + specifier.startsWith(prefix) && + specifier.endsWith(suffix) && + specifier.length >= prefix.length + suffix.length, + ) + .sort((a, b) => b.prefix.length - a.prefix.length); + + for (const { prefix, suffix, stems } of patterns) { + const stem = specifier.slice(prefix.length, specifier.length - suffix.length); + return stems.map((target) => substituteStar(target, stem)); + } + return null; +} + +/** + * Substitute a subpath pattern's single `*`. + * + * Node's subpath patterns and TypeScript's `paths` both allow AT MOST one `*`, + * so replacing the first occurrence is the specified behaviour rather than a + * partial one — but `String.replace` with a string needle says that only by + * accident, and reads as a bug to anyone (CodeQL included) who has met the + * replace-all footgun. Slicing at the known index states the rule instead. + */ +export function substituteStar(target: string, stem: string): string { + const star = target.indexOf('*'); + return star === -1 ? target : target.slice(0, star) + stem + target.slice(star + 1); +} + +/** Candidate stems for one specifier into `pkg`, best first. */ +function entryStemsFor(pkg: NodeWorkspacePackage, subpath: string): readonly string[] { + if (subpath === '') return pkg.entries; + + const declared = matchSubpathMap(pkg.subpathExports, subpath); + if (declared !== null) return declared; + + // A package with NO `exports` map is not restricted: Node resolves any + // subpath against the package DIRECTORY, and only against it. A package WITH + // one exposes only what it lists, so an unlisted subpath resolves to nothing. + // + // Both restrictions are real, and neither is softened here. An earlier draft + // also tried `/src/`, on the theory that a workspace package is + // consumed from source — but nothing declares that mapping, so it is the same + // kind of guess this module exists to remove: it would resolve + // `@repo/utils/deep/thing` to `packages/utils/src/deep/thing.ts` for a + // package whose manifest never said `deep/thing` lives under `src/`, and the + // import would be broken in the real project too. + if (pkg.subpathExports.size > 0) return []; + return [joinRepoPath(pkg.dir, subpath)]; +} + +/** + * The directories the workspace ADMITS as packages. + * + * `null` means the repository declares no workspace at all, in which case the + * only package is the one at the root — a nested `package.json` somewhere in + * `examples/` or `test/fixtures/` is not a member of anything and its name is + * not addressable by an import. + * + * This gate is the difference between reading manifests and trusting them. + * Without it, finding a `package.json` anywhere in the tree was enough to + * register its name, which recreates the false-positive half of #2953 from a + * different source: an app importing registry package `foo` would bind to an + * excluded fixture that happens to declare `name: "foo"`. THIS repository is + * the example — `test/fixtures/**` alone declares `@repo/utils` (added by this + * very change) among others. + */ +interface WorkspaceScope { + /** Positive patterns, repo-relative, as declared. */ + readonly include: readonly string[]; + /** `!`-prefixed patterns, with the `!` stripped. */ + readonly exclude: readonly string[]; +} + +/** Whether `dir` (repo-relative, `''` for the root) is an admitted package. */ +function admits(scope: WorkspaceScope | null, dir: string): boolean { + // The root package is always itself, workspace or not. + if (dir === '') return true; + if (scope === null) return false; + if (scope.exclude.some((pattern) => globToRegExp(pattern).test(dir))) return false; + return scope.include.some((pattern) => globToRegExp(pattern).test(dir)); +} + +/** + * Match one workspace glob. + * + * The subset npm, pnpm, yarn and lerna actually use in `workspaces` / + * `packages`: `*` within a segment, `**` across segments, `?`, and a leading + * `!` for exclusion (handled by the caller). Deliberately not a general glob + * engine — the patterns are a documented, narrow dialect, and `minimatch` is + * only present here transitively through `glob`. + */ +function globToRegExp(pattern: string): RegExp { + const normalized = pattern.replace(/^\.\//, '').replace(/\/$/, ''); + let out = ''; + for (let i = 0; i < normalized.length; i++) { + const ch = normalized[i]; + if (ch === '*') { + if (normalized[i + 1] === '*') { + // `**/` may match nothing at all, so `packages/**/x` also matches + // `packages/x`; a trailing `**` matches any depth below. + if (normalized[i + 2] === '/') { + out += '(?:.*/)?'; + i += 2; + } else { + out += '.*'; + i += 1; + } + } else { + out += '[^/]*'; + } + continue; + } + if (ch === '?') { + out += '[^/]'; + continue; + } + out += ch.replace(/[.+^${}()|[\]\\]/g, '\\$&'); + } + return new RegExp(`^${out}$`); +} + +/** + * Read the repository's workspace declaration. + * + * All three spellings are read and merged, because a repo may carry more than + * one (a pnpm workspace whose root `package.json` also lists `workspaces` for + * tooling that does not read pnpm's file). + */ +async function loadWorkspaceScope(repoRoot: string): Promise { + const patterns: string[] = []; + + const rootManifest = await readJsonFile(path.join(repoRoot, 'package.json')); + const workspaces = rootManifest?.workspaces; + if (Array.isArray(workspaces)) { + patterns.push(...workspaces.filter((w): w is string => typeof w === 'string')); + } else if (workspaces !== null && typeof workspaces === 'object') { + // Yarn's object form: `{ "packages": [...], "nohoist": [...] }`. + const nested = (workspaces as { packages?: unknown }).packages; + if (Array.isArray(nested)) { + patterns.push(...nested.filter((w): w is string => typeof w === 'string')); + } + } + + patterns.push(...(await readYamlPackages(path.join(repoRoot, 'pnpm-workspace.yaml')))); + patterns.push(...(await readYamlPackages(path.join(repoRoot, 'pnpm-workspace.yml')))); + + const lerna = await readJsonFile(path.join(repoRoot, 'lerna.json')); + if (Array.isArray(lerna?.packages)) { + patterns.push(...lerna.packages.filter((w): w is string => typeof w === 'string')); + } + + if (patterns.length === 0) return null; + return { + include: patterns.filter((p) => !p.startsWith('!')), + exclude: patterns.filter((p) => p.startsWith('!')).map((p) => p.slice(1)), + }; +} + +async function readJsonFile(filePath: string): Promise | null> { + try { + return JSON.parse(await fs.readFile(filePath, 'utf-8')) as Record; + } catch { + return null; + } +} + +async function readYamlPackages(filePath: string): Promise { + let raw: string; + try { + raw = await fs.readFile(filePath, 'utf-8'); + } catch { + return []; + } + try { + const parsed = yaml.load(raw) as { packages?: unknown } | null; + const packages = parsed?.packages; + return Array.isArray(packages) + ? packages.filter((p): p is string => typeof p === 'string') + : []; + } catch { + return []; + } +} + +/** + * Collect the `package.json` of every ADMITTED workspace package. + * + * Directory-only BFS: the sole files opened are manifests and the workspace + * declaration, so this is far cheaper than the C# namespace scan next door, + * which reads every `.cs` file. + */ +export async function loadNodeWorkspacePackages( + repoRoot: string, +): Promise { + const scope = await loadWorkspaceScope(repoRoot); + const byName = new Map(); + const queue: { dir: string; depth: number }[] = [{ dir: repoRoot, depth: 0 }]; + let dirsScanned = 0; + + while (queue.length > 0) { + if (dirsScanned >= SCAN_MAX_DIRS) { + logger.warn( + `[node] package.json scan of ${repoRoot} hit the ${SCAN_MAX_DIRS}-directory cap; workspace packages below it will not resolve`, + ); + break; + } + const { dir, depth } = queue.shift()!; + dirsScanned++; + + let entries: import('fs').Dirent[]; + try { + entries = await fs.readdir(dir, { withFileTypes: true }); + } catch { + continue; + } + + for (const entry of entries) { + if (entry.isDirectory()) { + if (isHardcodedIgnoredDirectory(entry.name)) continue; + if (depth < SCAN_MAX_DEPTH) { + queue.push({ dir: path.join(dir, entry.name), depth: depth + 1 }); + } + continue; + } + if (!entry.isFile() || entry.name !== 'package.json') continue; + + const relDir = repoRelativeDir(repoRoot, dir); + // Found is not the same as admitted. A manifest outside the declared + // workspace belongs to something this repository does not build — a + // fixture, an example, a vendored copy — and its name is not addressable. + if (!admits(scope, relDir)) continue; + + const pkg = await readManifest(path.join(dir, entry.name), repoRoot, dir); + // First declaration wins: BFS visits shallower directories first, so a + // top-level package outranks a nested one that reuses the name. + if (pkg !== null && !byName.has(pkg.name)) byName.set(pkg.name, pkg.package); + } + } + + return byName.size === 0 ? null : { byName }; +} + +async function readManifest( + manifestPath: string, + repoRoot: string, + dir: string, +): Promise<{ name: string; package: NodeWorkspacePackage } | null> { + let parsed: Record; + try { + parsed = JSON.parse(await fs.readFile(manifestPath, 'utf-8')) as Record; + } catch { + return null; + } + const name = typeof parsed.name === 'string' ? parsed.name : ''; + if (name === '') return null; + + const packageDir = repoRelativeDir(repoRoot, dir); + const rebase = (raw: string): string => joinRepoPath(packageDir, stripEntryPrefixes(raw)); + + const subpathExports = new Map(); + const rootExports: string[] = []; + collectExports(parsed.exports, subpathExports, rootExports, rebase); + + // `exports`, when present, is the package's ENTIRE public interface: Node + // ignores `main` outright and refuses any subpath the map does not list. This + // resolver already honoured that restriction for subpaths (`entryStemsFor`) + // and not for the ROOT, which is the same rule — so a manifest exporting only + // `"./feature"` still answered a bare `@repo/pkg` with `src/index`, an edge + // for an import that does not resolve in the real project. + const declaresExports = parsed.exports !== undefined && parsed.exports !== null; + const entries: string[] = [...rootExports]; + if (!declaresExports) { + for (const field of ['module', 'main', 'types', 'typings']) { + const value = parsed[field]; + if (typeof value === 'string') push(entries, rebase(value)); + } + for (const conventional of ['src/index', 'index', 'lib/index']) { + push(entries, joinRepoPath(packageDir, conventional)); + } + } + + const subpathImports = new Map(); + collectImports(parsed.imports, subpathImports, rebase); + + return { name, package: { dir: packageDir, entries, subpathExports, subpathImports } }; +} + +/** + * Walk an `exports` value into the root-entry list and the subpath map. + * + * `exports` nests three ways at once — a bare string, a subpath map, and + * condition maps (`import` / `require` / `types` / `default`) at any depth — so + * this collects string leaves per subpath rather than assuming a shape. + */ +function collectExports( + node: unknown, + subpaths: Map, + rootStems: string[], + rebase: (raw: string) => string, + currentSubpath: string | null = '', +): void { + if (typeof node === 'string') { + if (currentSubpath === null) return; + if (currentSubpath === '') { + push(rootStems, rebase(node)); + return; + } + subpaths.set(currentSubpath, [...(subpaths.get(currentSubpath) ?? []), rebase(node)]); + return; + } + // An array is an ordered FALLBACK LIST, not an opaque value: Node tries each + // entry in turn. `{"./feature": ["./dist/feature.js", "./src/feature.ts"]}` is + // the shape a workspace package publishes to say "built output, or source" — + // and the source arm is the one that matters here, because `dist/` is build + // output and is not indexed. Skipping arrays dropped the declaration entirely + // and left the package looking as though it declared no subpath exports. + if (Array.isArray(node)) { + for (const element of node) + collectExports(element, subpaths, rootStems, rebase, currentSubpath); + return; + } + if (node === null || typeof node !== 'object') return; + + for (const [key, value] of Object.entries(node as Record)) { + if (key.startsWith('.')) { + // A subpath key: `"."` is the package root, `"./nest"` the subpath `nest`. + collectExports( + value, + subpaths, + rootStems, + rebase, + key === '.' ? '' : key.replace(/^\.\//, ''), + ); + } else { + // A condition key — stays on whatever subpath we were already resolving. + collectExports(value, subpaths, rootStems, rebase, currentSubpath); + } + } +} + +/** Walk an `imports` map (`"#env": "./src/env.node.ts"`) into stems. */ +function collectImports( + node: unknown, + out: Map, + rebase: (raw: string) => string, + currentKey: string | null = null, +): void { + if (typeof node === 'string') { + if (currentKey === null) return; + out.set(currentKey, [...(out.get(currentKey) ?? []), rebase(node)]); + return; + } + // Same ordered-fallback rule as `exports` — see `collectExports`. + if (Array.isArray(node)) { + for (const element of node) collectImports(element, out, rebase, currentKey); + return; + } + if (node === null || typeof node !== 'object') return; + for (const [key, value] of Object.entries(node as Record)) { + collectImports(value, out, rebase, key.startsWith('#') ? key : currentKey); + } +} + +/** `"./src/index.ts"` -> `"src/index"`; leaves an extension-less path alone. */ +function stripEntryPrefixes(entry: string): string { + const withoutDot = entry.replace(/^\.\//, '').replace(/^\//, ''); + return withoutDot.replace(/\.(ts|tsx|mts|cts|js|jsx|mjs|cjs|vue)$/, ''); +} + +function push(list: string[], value: string): void { + if (value !== '' && !list.includes(value)) list.push(value); +} + +/** `/repo/packages/utils` -> `packages/utils`; the root -> `''`. */ +function repoRelativeDir(repoRoot: string, dir: string): string { + const rel = path.relative(repoRoot, dir).split(path.sep).join('/'); + return rel === '.' ? '' : rel; +} + +function joinRepoPath(dir: string, rest: string): string { + return dir === '' ? rest : `${dir}/${rest}`; +} diff --git a/gitnexus/src/core/ingestion/language-provider.ts b/gitnexus/src/core/ingestion/language-provider.ts index fdfc75839..ec9006450 100644 --- a/gitnexus/src/core/ingestion/language-provider.ts +++ b/gitnexus/src/core/ingestion/language-provider.ts @@ -214,6 +214,20 @@ interface LanguageProviderConfig { * Default: undefined (standard label assignment). */ readonly labelOverride?: (functionNode: SyntaxNode, defaultLabel: NodeLabel) => NodeLabel | null; + /** + * Suppress a definition query match after its default label is known. + * Languages use this for syntax that represents an implicit declaration + * unless an explicit declaration with the same semantics is present. + * + * `defaultLabel` is supplied so an implementation can scope itself to one + * kind of definition; implementations whose capture map alone decides the + * question may ignore it. + */ + readonly shouldSkipDefinitionCapture?: ( + captureMap: CaptureMap, + defaultLabel: NodeLabel, + ) => boolean; + // ── MRO ─────────────────────────────────────────────────────────── /** MRO strategy for multiple inheritance resolution. * Default: 'first-wins'. */ @@ -615,7 +629,7 @@ interface LanguageProviderConfig { readonly resolveImportTarget?: ( parsedImport: ParsedImport, workspaceIndex: WorkspaceIndex, - ) => string | null; + ) => string | readonly string[] | null; /** * Enumerate the exported names of a file — used by the finalize algorithm diff --git a/gitnexus/src/core/ingestion/languages/csharp/query.ts b/gitnexus/src/core/ingestion/languages/csharp/query.ts index 615da3c21..18c37ba2b 100644 --- a/gitnexus/src/core/ingestion/languages/csharp/query.ts +++ b/gitnexus/src/core/ingestion/languages/csharp/query.ts @@ -93,8 +93,14 @@ const CSHARP_SCOPE_QUERY = ` name: (identifier) @declaration.name) @declaration.enum ;; Declarations — methods / constructors / properties +;; +;; A generic METHOD's parameters are read for the same reason a generic type's +;; are (#2912 review): \`void Run(IValidator v)\` writes a receiver whose +;; argument is a type VARIABLE, and a pass that cannot tell that from a concrete +;; type prunes every implementor of \`IValidator\` from the call's fan-out. (method_declaration - name: (identifier) @declaration.name) @declaration.method + name: (identifier) @declaration.name + (type_parameter_list)? @declaration.type-parameters) @declaration.method (constructor_declaration name: (identifier) @declaration.name) @declaration.constructor diff --git a/gitnexus/src/core/ingestion/languages/csharp/scope-resolver.ts b/gitnexus/src/core/ingestion/languages/csharp/scope-resolver.ts index eb1d3b10d..4b50efc67 100644 --- a/gitnexus/src/core/ingestion/languages/csharp/scope-resolver.ts +++ b/gitnexus/src/core/ingestion/languages/csharp/scope-resolver.ts @@ -102,6 +102,82 @@ const csharpScopeResolver: ScopeResolver = { // files. The compound-receiver walker needs to walk up from the // class scope to find them; see the contract field for rationale. hoistTypeBindingsToModule: true, + + // `IValidator` and `IValidator` are one instantiation, so the + // dispatch fan-out must not read them as two (#2912). See the alias table. + normalizeTypeArgument: normalizeCsharpTypeArgument, }; +/** + * C# predefined type aliases — the 15 keywords the language defines as exact + * synonyms for `System` types (`string` ≡ `System.String`), plus `nint`/`nuint`. + * A codebase mixing the spellings is common enough that StyleCop ships a rule + * about it (SA1121), so the two forms genuinely meet across files. + * + * Keyword → BCL simple name; anything else is returned unchanged, including the + * BCL names themselves (already canonical) and any qualified spelling, which is + * compared as written. + * + * A workspace may legally declare its OWN type named `String`, which shadows the + * BCL simple name; this table then reads `IValidator` as the `string` + * instantiation and KEEPS that implementor in the fan-out. Deliberate, and the + * safe direction: the alternative is pruning on the belief that two spellings + * differ, which is the missing-edge failure `generic-instantiation.ts` is built + * to avoid. Resolving instead of normalizing cannot settle it either — the + * identity comparison needs a `definitionId` from BOTH sides, and a built-in + * name has none, so "built-in versus workspace-declared" would be a new prune + * with no positive evidence behind it. The result is one surplus edge in a + * shape that is rare on its own terms, i.e. exactly the pre-#2912 fan-out for + * that pair and no worse. + */ +const CSHARP_PREDEFINED_TYPE_ALIASES: ReadonlyMap = new Map([ + ['bool', 'Boolean'], + ['byte', 'Byte'], + ['sbyte', 'SByte'], + ['char', 'Char'], + ['decimal', 'Decimal'], + ['double', 'Double'], + ['float', 'Single'], + ['int', 'Int32'], + ['uint', 'UInt32'], + ['long', 'Int64'], + ['ulong', 'UInt64'], + ['short', 'Int16'], + ['ushort', 'UInt16'], + ['nint', 'IntPtr'], + ['nuint', 'UIntPtr'], + ['object', 'Object'], + ['string', 'String'], +]); + +/** The BCL simple names the keywords alias. A spelling that reduces to one of + * these IS the predefined type; anything else that merely happens to sit in + * `System` is an ordinary type and keeps its qualifier. */ +const CSHARP_PREDEFINED_TYPE_NAMES: ReadonlySet = new Set( + CSHARP_PREDEFINED_TYPE_ALIASES.values(), +); + +const CSHARP_SYSTEM_QUALIFIER = /^(?:global::)?System\./; + +function normalizeCsharpTypeArgument(name: string): string { + const named = name.trim(); + // A keyword answers immediately: `string` → `String`. + const aliased = CSHARP_PREDEFINED_TYPE_ALIASES.get(named); + if (aliased !== undefined) return aliased; + // Otherwise the `System.` qualifier is dropped so the fully-qualified + // spelling of a predefined type meets that keyword: `System.String` → + // `String` ≡ `string` → `String`. The optional `global::` alias qualifier goes + // with it — `import-decomposer` already unwraps that spelling elsewhere, and + // leaving it on would make `global::System.String` unequal to `string` and + // prune a live implementor. + // + // ONLY when what remains is a predefined type. `System.Custom` is an ordinary + // type that happens to live in `System`, and answering `Custom` for it would + // equate it with an unrelated `Custom` elsewhere in the workspace. Returned as + // written instead, which sends it to the identity comparison — the step that + // can actually tell two declarations apart. + const bare = named.replace(CSHARP_SYSTEM_QUALIFIER, ''); + return bare !== named && CSHARP_PREDEFINED_TYPE_NAMES.has(bare) ? bare : named; +} + export { csharpScopeResolver }; diff --git a/gitnexus/src/core/ingestion/languages/dart/captures.ts b/gitnexus/src/core/ingestion/languages/dart/captures.ts index 351eaee7d..a6c5ef773 100644 --- a/gitnexus/src/core/ingestion/languages/dart/captures.ts +++ b/gitnexus/src/core/ingestion/languages/dart/captures.ts @@ -1069,9 +1069,15 @@ function emitHeritage(classNode: SyntaxNode, out: CaptureMatch[]): void { for (let i = 0; i < superclass.namedChildCount; i++) { const c = superclass.namedChild(i); if (c !== null && c.type === 'type_identifier') { + // `extends Base` spells the arguments in a SIBLING node, so the + // anchor's own text cannot carry them; the sub-tag does (#2912). + const args = typeArgumentsAfter(superclass, i); out.push({ '@reference.inherits': nodeToCapture('@reference.inherits', c), '@reference.name': nodeToCapture('@reference.name', c), + ...(args === null + ? {} + : { '@reference.type-arguments': nodeToCapture('@reference.type-arguments', args) }), }); break; } @@ -1144,7 +1150,26 @@ function emitHeritageMarkers( for (let i = 0; i < container.namedChildCount; i++) { const c = container.namedChild(i); if (c === null || c.type !== 'type_identifier') continue; - const payload = encodeMarker('heritage', [kind, c.text, className]); + // `implements Validator` / `with M`: the arguments ride the + // marker payload, because this heritage never becomes a reference SITE — + // `emitDartHeritageEdges` reads the marker and emits the edge (#2912). + // Dropped rather than encoded when the spelling contains the marker's own + // ':' delimiter, which `encodeMarker` rejects outright; absence is the + // fail-open value everywhere this is read. + const args = typeArgumentsAfter(container, i)?.text; + const fields = + args === undefined || args.includes(':') + ? [kind, c.text, className] + : [kind, c.text, className, args]; + const payload = encodeMarker('heritage', fields); out.push({ '@import.heritage': syntheticCapture('@import.heritage', c, payload) }); } } + +/** The `type_arguments` node written immediately after `container`'s named + * child at `index` — the arguments of the type that child names — or `null` + * when that type was written without any. */ +function typeArgumentsAfter(container: SyntaxNode, index: number): SyntaxNode | null { + const next = container.namedChild(index + 1); + return next !== null && next.type === 'type_arguments' ? next : null; +} diff --git a/gitnexus/src/core/ingestion/languages/dart/query.ts b/gitnexus/src/core/ingestion/languages/dart/query.ts index 38496f2ec..fb93f5fb8 100644 --- a/gitnexus/src/core/ingestion/languages/dart/query.ts +++ b/gitnexus/src/core/ingestion/languages/dart/query.ts @@ -42,7 +42,15 @@ const DART_SCOPE_QUERY = ` (enum_declaration) @scope.class ; ── Declarations — types ───────────────────────────────────────────────────── -(class_definition name: (identifier) @declaration.name) @declaration.class +; The type-parameter list is matched as an UNNAMED optional child: the Dart +; grammar hangs \`type_parameters\` off \`class_definition\` without a field name. +; Recording it is what lets instantiation-aware interface dispatch tell a type +; VARIABLE (\`class Box implements Validator\`) from a concrete argument +; (\`class V implements Validator\`) — see #2912; absent parameters are +; indistinguishable from a language that captures none, and read as unknown. +(class_definition + name: (identifier) @declaration.name + (type_parameters)? @declaration.type-parameters) @declaration.class (mixin_declaration (identifier) @declaration.name) @declaration.trait (extension_declaration name: (identifier) @declaration.name) @declaration.class (enum_declaration name: (identifier) @declaration.name) @declaration.enum diff --git a/gitnexus/src/core/ingestion/languages/dart/scope-resolver.ts b/gitnexus/src/core/ingestion/languages/dart/scope-resolver.ts index 22e1171d1..76bec5d74 100644 --- a/gitnexus/src/core/ingestion/languages/dart/scope-resolver.ts +++ b/gitnexus/src/core/ingestion/languages/dart/scope-resolver.ts @@ -38,6 +38,8 @@ import { generateId } from '../../../../lib/utils.js'; import { dartProvider } from '../dart.js'; import { dartArityCompatibility, dartMergeBindings, resolveDartImportTarget } from './index.js'; import { decodeMarker } from '../../utils/heritage-marker.js'; +import { typeApplicationArguments } from '../../utils/template-arguments.js'; +import type { HeritageTypeArgumentSink } from '../../scope-resolution/utils/generic-instantiation.js'; import { expandDartWildcardNames } from './expand-wildcards.js'; interface ClassDefRef { @@ -77,6 +79,7 @@ function emitDartHeritageEdges( graph: KnowledgeGraph, parsedFiles: readonly ParsedFile[], nodeLookup: GraphNodeLookup, + recordTypeArguments?: HeritageTypeArgumentSink, ): void { const defsByName = new Map(); for (const parsed of parsedFiles) { @@ -110,10 +113,19 @@ function emitDartHeritageEdges( if (decoded?.kind !== 'heritage') continue; const parts = decoded.fields; if (parts.length < 3) continue; - const [kind, baseName, childName] = parts; + const [kind, baseName, childName, rawTypeArguments] = parts; const childId = pickClassByName(childName!, parsed.filePath, defsByName); const baseId = pickClassByName(baseName!, parsed.filePath, defsByName); if (childId === undefined || baseId === undefined || childId === baseId) continue; + // The instantiation this clause was written with — `implements + // Validator` (#2912). Recorded before the dedup below, since the + // FIRST writer wins on both sides and an edge deduped here still needs + // its arguments. A marker from a pre-#2912 cache has no fourth field, + // which reads as unknown. + if (rawTypeArguments !== undefined) { + const typeArguments = typeApplicationArguments(rawTypeArguments); + if (typeArguments !== undefined) recordTypeArguments?.(childId, baseId, typeArguments); + } const key = `${childId}->${baseId}:${kind}`; if (emitted.has(key)) continue; emitted.add(key); @@ -211,8 +223,8 @@ export const dartScopeResolver: ScopeResolver = { // `implements` / `with` IMPLEMENTS edges (extends rides the generic // inherits pre-pass; these need an explicit, kind-independent edge type). - emitHeritageEdges: (graph, parsedFiles, nodeLookup) => - emitDartHeritageEdges(graph, parsedFiles, nodeLookup), + emitHeritageEdges: (graph, parsedFiles, nodeLookup, _scopes, recordTypeArguments) => + emitDartHeritageEdges(graph, parsedFiles, nodeLookup, recordTypeArguments), // Dart is statically typed — the field-fallback heuristic over-connects. fieldFallbackOnMethodLookup: false, diff --git a/gitnexus/src/core/ingestion/languages/java.ts b/gitnexus/src/core/ingestion/languages/java.ts index 047a87791..317a853ac 100644 --- a/gitnexus/src/core/ingestion/languages/java.ts +++ b/gitnexus/src/core/ingestion/languages/java.ts @@ -23,14 +23,16 @@ import { createCallExtractor } from '../call-extractors/generic.js'; import { javaCallConfig } from '../call-extractors/configs/jvm.js'; import { createFieldExtractor } from '../field-extractors/generic.js'; import { javaConfig } from '../field-extractors/configs/jvm.js'; -import { createMethodExtractor } from '../method-extractors/generic.js'; -import { javaMethodConfig } from '../method-extractors/configs/jvm.js'; import { createVariableExtractor } from '../variable-extractors/generic.js'; import { javaVariableConfig } from '../variable-extractors/configs/jvm.js'; import { createJavaCfgVisitor } from '../cfg/visitors/java.js'; import { assertCloneable } from '../workers/clone-safety.js'; import { collectJavaCaptureSideChannel } from './java/capture-side-channel.js'; import type { SymbolDefinition } from 'gitnexus-shared'; +import { + javaRecordMethodExtractor, + shouldSkipJavaRecordComponentDefinition, +} from './java/record-components.js'; import { emitJavaScopeCaptures, interpretJavaImport, @@ -186,7 +188,8 @@ export const javaProvider = defineLanguage({ mroStrategy: 'implements-split', callExtractor: createCallExtractor(javaCallConfig), fieldExtractor: createFieldExtractor(javaConfig), - methodExtractor: createMethodExtractor(javaMethodConfig), + methodExtractor: javaRecordMethodExtractor, + shouldSkipDefinitionCapture: shouldSkipJavaRecordComponentDefinition, variableExtractor: createVariableExtractor(javaVariableConfig), classExtractor: createClassExtractor(javaClassConfig), diff --git a/gitnexus/src/core/ingestion/languages/java/analysis-features.ts b/gitnexus/src/core/ingestion/languages/java/analysis-features.ts index 17b153ab1..73969670d 100644 --- a/gitnexus/src/core/ingestion/languages/java/analysis-features.ts +++ b/gitnexus/src/core/ingestion/languages/java/analysis-features.ts @@ -18,6 +18,13 @@ export const SPRING_CONFIG_BINDINGS_FEATURE: AnalysisFeatureDescriptor = { ), }; +/** Durable completeness contract for implicit Java record-component accessors. */ +export const JAVA_RECORD_COMPONENT_ACCESSORS_FEATURE: AnalysisFeatureDescriptor = { + id: 'java.record-component-accessors', + version: 1, + appliesTo: (filePaths) => filePaths.some((filePath) => filePath.toLowerCase().endsWith('.java')), +}; + /** Durable completeness contract for Java heritage captures. */ export const JAVA_ENUM_INTERFACE_HERITAGE_FEATURE: AnalysisFeatureDescriptor = { id: 'java.heritage-captures', diff --git a/gitnexus/src/core/ingestion/languages/java/captures.ts b/gitnexus/src/core/ingestion/languages/java/captures.ts index 3a4c131f2..47c694b37 100644 --- a/gitnexus/src/core/ingestion/languages/java/captures.ts +++ b/gitnexus/src/core/ingestion/languages/java/captures.ts @@ -50,6 +50,7 @@ import { captureJavaSpringConditionalFacts, type JavaSpringConditionalFact, } from './spring-conditionals.js'; +import { synthesizeJavaRecordComponentAccessorCaptures } from './record-components.js'; /** Declaration anchors that carry function-like arity metadata. */ const FUNCTION_DECL_TAGS = ['@declaration.method', '@declaration.constructor'] as const; @@ -397,6 +398,7 @@ export function emitJavaScopeCaptures( ...synthesizeJavaInheritanceReferences(tree.rootNode), ...synthesizeJavaExplicitConstructorReferences(tree.rootNode), ...synthesizeJavaAnonymousClassDeclarations(tree.rootNode), + ...synthesizeJavaRecordComponentAccessorCaptures(tree.rootNode), ...synthesizeCallableFlowCaptures(tree.rootNode, JAVA_CALLABLE_CAPTURE_OPTIONS), ]; } diff --git a/gitnexus/src/core/ingestion/languages/java/import-target.ts b/gitnexus/src/core/ingestion/languages/java/import-target.ts index d3cec9205..4843e854d 100644 --- a/gitnexus/src/core/ingestion/languages/java/import-target.ts +++ b/gitnexus/src/core/ingestion/languages/java/import-target.ts @@ -1,123 +1,96 @@ /** - * Adapter from `(ParsedImport, WorkspaceIndex)` → concrete file path. + * Adapter from `(ParsedImport, WorkspaceIndex)` → the file(s) an import names. * - * Converts Java package paths (dots → slashes) and tries: - * 1. Exact file match: `com/example/User.java` - * 2. Suffix match for nested layouts - * 3. Directory match (wildcard imports) - * 4. Progressive prefix stripping for non-standard layouts + * Delegates to `module-resolution.ts`, which resolves a Java import the way + * Java defines it: a fully-qualified type name looked up against the packages + * the workspace's files DECLARE. * - * Returns `null` for unresolvable / JDK imports. + * ## What #2953 replaced, and why path shape could not work * - * ## Why the scans are gone (#2908) + * This resolver used to turn dots into slashes and hunt for a file whose path + * ended that way — exact whole path, then any segment-suffix, then the first + * `.java` directly inside a matching directory — retrying the whole cascade + * with each leading segment stripped. Four legs, all describing where a file + * SITS rather than what it DECLARES. * - * Every leg above used to be answered by `for (const raw of ctx.allFilePaths)`, - * and the stripping loop ran that scan again per stripped segment — so one - * unresolvable `import a.b.c.D;` (the COMMON case: JDK and third-party imports - * run the whole cascade to completion) cost four full workspace passes. This is - * byte-for-byte the shape C# carried until #2878; both now read the same two - * per-file-set indexes, memoized on the Set's identity: + * Path shape is a convention, so it mostly worked, and failed hardest on the + * case that matters: it had no way to tell an import of something outside the + * repository from one inside it. `java.util.List` became `util/List`, then + * `List`, and bound to any `List.java` anywhere in the tree — a fabricated + * IMPORTS edge at full confidence, for an import naming a JDK class. Every JDK + * and third-party import in a repository was a candidate. * - * - `getWorkspaceFileIndex` — `normToRaw` (whole-path lookup) and `index` - * (segment-suffix lookup); - * - `getJavaDirIndex` — `firstFileDirectlyInPkgDir`'s package-directory index. + * Two secondary defects went with it, both consequences of resolving by shape: * - * ## The tie-breaks the scans encoded, and where they now live + * - a wildcard `import com.example.*;` answered with ONE arbitrary file — the + * first `.java` in the package directory in `allFilePaths` iteration order, + * which the previous header documented at length as being decided by "a + * property of the file list, not of the import". It now answers with every + * file declaring that package, which is what the import actually names. + * - a file's location and its package were assumed to agree. They need not: + * `weird/path/User.java` declaring `package com.example;` is importable as + * `com.example.User`, and `com/example/User.java` declaring nothing is in + * the default package and importable as nothing at all. Both now resolve + * correctly, because the declaration is what is read. * - * 1. The first pass `break`s on an exact whole-path hit but keeps scanning - * otherwise, then returns `exactFile ?? suffixFile ?? directoryChild`. So an - * exact match wins over a suffix or directory-child match found EARLIER in - * iteration order — hence `normToRaw` before `index`, which conflates the - * two (see `resolveDirectMatch`). - * 2. The stripping loop instead `return`s mid-scan on `f === tailFile || - * f.endsWith(tailSuffix)`, i.e. at the first hit of EITHER, and only returns - * its directory child after the scan completes. So file/suffix beats - * directory child within one `skip` level regardless of order, and the - * conflated `index.get` is the CORRECT lookup there (see - * `resolveByProgressiveStripping`). - * 3. Wildcard imports drop their trailing `.*` before resolution, so - * `com.example.*` resolves as the package directory. - * 4. `.java` filter and backslash normalization, with the RAW path returned: - * the indexes normalize for their keys and hand back the raw Set member, and - * only a `.java` file can carry a `…/.java` suffix key, so the - * extension filter is implied on the file/suffix legs and explicit in the - * directory index's `accept`. - * 5. The directory-child leg used to match on the FIRST `'/' + pathLike + '/'` - * occurrence, so `com/example/com/example/Deep.java` did NOT answer - * `com.example`. #2881 removed that: the rule came from how the pre-index - * scan was written, not from Java, and it made a package whose name repeats - * higher in the path unresolvable. `firstFileDirectlyInPkgDir` now answers - * plain "the parent directory ends with `pathLike`" (see the header of - * `import-resolvers/package-dir-index.ts`). This leg commits to ONE file - * with no downstream filter, so widening it can change which file an - * already-resolving import binds to, not only turn a null into a hit. - * WHICH file it binds to is decided by nothing in this resolver: it is - * `allFilePaths` iteration order, i.e. the insertion order of the Set built - * from `parsedFiles` in `scope-resolution/pipeline/run.ts`, which for a full - * scan is the canonical sorted path order `filesystem-walker.ts` imposes on - * its unsorted recursive-`glob` result. So the widened set's winner is a - * property of the file list, not of the import — pinned explicitly, in both - * insertion orders, by "pins WHICH of two competing package directories the - * first-child leg takes" in - * `test/unit/scope-resolution/java-import-target-parity.test.ts` (Kotlin's - * twin, which has the same unfiltered first-child leg, is in - * `test/unit/scope-resolution/kotlin/kotlin-import-target-parity.test.ts`). + * The package declaration was already being extracted during the parse pass and + * has been reachable here through `getJavaPackageFact` the whole time; nothing + * read it. So this costs no new I/O — no `pom.xml`, no `build.gradle`, no + * source-root inference. The workspace describes itself. */ -import type { ParsedImport, WorkspaceIndex } from 'gitnexus-shared'; -import { - getWorkspaceFileIndex, - type WorkspaceFileIndex, -} from '../../import-resolvers/workspace-file-index.js'; -import { - buildPackageDirIndex, - firstFileDirectlyInPkgDir, - type PackageDirIndex, -} from '../../import-resolvers/package-dir-index.js'; +import type { ParsedFile, ParsedImport, WorkspaceIndex } from 'gitnexus-shared'; import { perFileSet } from '../../import-resolvers/per-file-set.js'; +import { getJavaPackageFact } from './package-facts.js'; +import { + buildJavaPackageIndex, + resolveJavaModule, + type JavaPackageIndex, +} from './module-resolution.js'; export interface JavaResolveContext { readonly fromFile: string; readonly allFilePaths: ReadonlySet; + /** + * The pass's parsed Java files — the only input this resolver needs, because + * the package index is built from their declarations. + * + * Absent means "no workspace was supplied", not "the workspace declares + * nothing": the index would be empty and every import would answer `null`. + * The orchestrator always supplies it (`scope-resolution/pipeline/run.ts` + * threads `context.parsedFiles`), and it must be passed THROUGH rather than + * copied — the memo below keys on the array's identity. + */ + readonly parsedFiles?: readonly ParsedFile[]; } /** - * Package-directory index over the `.java` files, memoized on the Set's - * identity. Feeds `firstFileDirectlyInPkgDir`, which is called once for the - * direct match and then up to once per stripped package prefix. + * The package index, built once per pass and read by every import. + * + * Keyed on the `parsedFiles` array the orchestrator already threads through the + * pass, like PHP's `filesByDirectory` and Python's `parsedFileByPath`. The + * instrument that can see this memo fail counts element reads on that array — + * `countedParsedFiles` in `test/helpers/counting-file-set.ts`, asserted for + * every language by `import-target-index-reuse.contract.test.ts`. */ -const getJavaDirIndex = perFileSet( - (allFilePaths: ReadonlySet): PackageDirIndex => - buildPackageDirIndex(allFilePaths, (normalized) => normalized.endsWith('.java')), +const getJavaPackageIndex = perFileSet( + (parsedFiles: readonly ParsedFile[]): JavaPackageIndex => + buildJavaPackageIndex(parsedFiles, getJavaPackageFact), ); export function resolveJavaImportTarget( parsedImport: ParsedImport, workspaceIndex: WorkspaceIndex, -): string | null { +): string | readonly string[] | null { const ctx = narrowContext(workspaceIndex); if (ctx === null) return null; if (parsedImport.kind === 'dynamic-unresolved') return null; if (parsedImport.targetRaw === null || parsedImport.targetRaw === '') return null; - // Strip trailing `.*` for wildcard imports: `com.example.*` → `com.example` - let target = parsedImport.targetRaw; - if (target.endsWith('.*')) { - target = target.slice(0, -2); - } + const parsedFiles = ctx.parsedFiles; + if (parsedFiles === undefined || parsedFiles.length === 0) return null; - // Package path: `com.example.User` → `com/example/User` - const pathLike = target.replace(/\./g, '/'); - - const ws = getWorkspaceFileIndex(ctx.allFilePaths); - const dirs = getJavaDirIndex(ctx.allFilePaths); - - const direct = resolveDirectMatch(ws, dirs, pathLike); - if (direct !== null) return direct; - - // Progressive prefix stripping — handles `import com.example.User;` - // in a repo laid out `User.java` (no `com/example/` prefix). - return resolveByProgressiveStripping(ws, dirs, pathLike); + return resolveJavaModule(parsedImport.targetRaw, getJavaPackageIndex(parsedFiles)); } /** @@ -136,59 +109,3 @@ function narrowContext(workspaceIndex: WorkspaceIndex): JavaResolveContext | nul } return ctx; } - -/** - * First-pass resolution against the full package path: - * exact whole-path file > nested suffix file > first `.java` directly inside - * the package directory. - */ -function resolveDirectMatch( - ws: WorkspaceFileIndex, - dirs: PackageDirIndex, - pathLike: string, -): string | null { - const exactName = `${pathLike}.java`; - // The scan `break`s here, so an exact whole-path match wins even when a - // `…/` suffix match appeared EARLIER in iteration order. The two - // lookups therefore stay separate: `index.get` conflates them and would - // return the earlier suffix hit. - const exact = ws.normToRaw.get(exactName); - if (exact !== undefined) return exact; - // No whole-path file exists, so every segment-suffix hit is a `/` - // match and `index.get` yields the first one in iteration order — exactly the - // `suffixFile` the scan kept. Only a `.java` file can carry a `.java` suffix - // key, so the old `endsWith('.java')` filter is implied. - const suffixFile = ws.index.get(exactName); - if (suffixFile !== undefined) return suffixFile; - // First `.java` file living directly inside the package directory `pathLike` - // (at repo root or nested under a source-root prefix), not deeper — the leg - // wildcard imports land on. - return firstFileDirectlyInPkgDir(dirs, pathLike); -} - -/** - * Try each suffix of the package path against `.java` files and directories, - * stripping leading segments one at a time. Models `import com.example.User;` - * resolving to `User.java` in a repo laid out without the `com/example/` prefix. - */ -function resolveByProgressiveStripping( - ws: WorkspaceFileIndex, - dirs: PackageDirIndex, - pathLike: string, -): string | null { - const segments = pathLike.split('/').filter(Boolean); - for (let skip = 1; skip < segments.length; skip++) { - const tail = segments.slice(skip).join('/'); - if (tail === '') continue; - // `f === tailFile || f.endsWith('/' + tailFile)`, first in iteration order — - // the scan returned at the first hit of EITHER, with no exact-wins rule, - // so here the conflated suffix lookup is the right one. - const tailFileMatch = ws.index.get(`${tail}.java`); - if (tailFileMatch !== undefined) return tailFileMatch; - // Collected mid-scan but returned only after it, so the file/suffix hit - // above beats it even when this one came first in iteration order. - const child = firstFileDirectlyInPkgDir(dirs, tail); - if (child !== null) return child; - } - return null; -} diff --git a/gitnexus/src/core/ingestion/languages/java/interpret.ts b/gitnexus/src/core/ingestion/languages/java/interpret.ts index 38d6128d4..721edbff8 100644 --- a/gitnexus/src/core/ingestion/languages/java/interpret.ts +++ b/gitnexus/src/core/ingestion/languages/java/interpret.ts @@ -57,10 +57,11 @@ export function interpretJavaImport(captures: CaptureMatch): ParsedImport | null // `import static com.example.Utils.*;` // The source is the class path (e.g. `com.example.Utils`). // Resolution should target the class file, not a wildcard directory - // scan — `Utils.java` is the file that contains the static members. + // scan — keeping the type path unstarred also distinguishes it from a + // package wildcard when a same-named package exists. return { kind: 'wildcard', - targetRaw: sourceCap.text + '.*', + targetRaw: sourceCap.text, }; } default: diff --git a/gitnexus/src/core/ingestion/languages/java/module-resolution.ts b/gitnexus/src/core/ingestion/languages/java/module-resolution.ts new file mode 100644 index 000000000..9eea50d74 --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/java/module-resolution.ts @@ -0,0 +1,167 @@ +/** + * Java import resolution against DECLARED packages (#2953). + * + * A Java import is a fully-qualified type name, not a path. `com.example.model.User` + * names the type `User` in the package `com.example.model`, and what places a + * file in that package is its own `package` declaration — not where it sits on + * disk. A file at `weird/path/User.java` declaring `package com.example.model;` + * IS `com.example.model.User`; a file at `com/example/model/User.java` declaring + * nothing is in the DEFAULT package and cannot be imported at all. + * + * The previous resolver worked the other way round: it turned dots into slashes + * and looked for a file whose path ended that way, retrying with each leading + * segment stripped. Path shape is a convention, so that mostly worked — and + * failed in the one case that matters most, because it could not tell an import + * of something outside the repository from one inside it. `java.util.List` + * became `util/List`, then `List`, and bound to any `List.java` in the tree. + * Every JDK and third-party import in a repo was a candidate for a fabricated + * IMPORTS edge at full confidence. + * + * The fix needs no new I/O. Every Java file's `package` declaration is already + * extracted during the parse pass and available here through + * `getJavaPackageFact` — the resolver simply never read it. So resolution + * becomes a lookup in an index the workspace already knows how to describe: + * + * `com.example.model.User` -> package `com.example.model` declares `User` + * `java.util.List` -> no file declares package `java.util` -> null + * + * `null` for the second is the complete and correct answer: the JDK is not in + * this repository, so there is no in-repo file the import could name. + */ + +import type { ParsedFile } from 'gitnexus-shared'; +import type { JvmPackageFact } from '../jvm/package-facts.js'; + +export interface JavaPackageIndex { + /** Declared package -> importable type name -> the file declaring it. */ + readonly typesByPackage: ReadonlyMap>; + /** Declared package -> every file declaring it, for wildcard imports. */ + readonly filesByPackage: ReadonlyMap; + /** + * Files whose `package` header could not be read (a malformed header — see + * `extractJvmPackageFact`). They are in no package, so nothing can import + * them; counted so the gap is observable rather than silent. + */ + readonly unreadablePackageFiles: number; +} + +const EMPTY_INDEX: JavaPackageIndex = { + typesByPackage: new Map(), + filesByPackage: new Map(), + unreadablePackageFiles: 0, +}; + +/** + * Index the workspace by what each file DECLARES. + * + * The importable type name is the file's base name, which is not a convention + * being relied on but the rule the language enforces: a type importable from + * another package must be `public`, and a public type must live in a file named + * after it. Additional package-private top-level types in the same file are + * deliberately not indexed — they are unimportable from elsewhere, so an import + * naming one is not a resolution this should find. + */ +export function buildJavaPackageIndex( + parsedFiles: readonly ParsedFile[], + packageOf: (filePath: string) => JvmPackageFact | undefined, +): JavaPackageIndex { + if (parsedFiles.length === 0) return EMPTY_INDEX; + + const typesByPackage = new Map>(); + const filesByPackage = new Map(); + let unreadablePackageFiles = 0; + + for (const parsed of parsedFiles) { + const filePath = parsed.filePath; + const fact = packageOf(filePath); + if (fact === undefined) continue; + if (fact.status !== 'known') { + unreadablePackageFiles++; + continue; + } + // The default package (`''`) is indexed like any other so a workspace of + // package-less files still answers its own wildcards, but Java forbids + // importing FROM it, which `resolveJavaModule` enforces rather than + // pretending here that the entry does not exist. + const packageName = fact.packageName; + + const typeName = baseTypeName(filePath); + if (typeName !== null) { + let types = typesByPackage.get(packageName); + if (types === undefined) { + types = new Map(); + typesByPackage.set(packageName, types); + } + // First declaration wins. Two files claiming the same package+type is not + // legal Java; picking either is as correct as the input allows. + if (!types.has(typeName)) types.set(typeName, filePath); + } + + const files = filesByPackage.get(packageName); + if (files === undefined) filesByPackage.set(packageName, [filePath]); + else files.push(filePath); + } + + return { typesByPackage, filesByPackage, unreadablePackageFiles }; +} + +/** + * Resolve one import specifier to the file(s) it names, or `null`. + * + * A wildcard answers with every file in the package; a type import answers with + * one file. Anything the workspace does not declare answers `null`. + */ +export function resolveJavaModule( + targetRaw: string, + index: JavaPackageIndex, +): string | readonly string[] | null { + if (targetRaw === '') return null; + + if (targetRaw.endsWith('.*')) { + const stem = targetRaw.slice(0, -2); + const inPackage = index.filesByPackage.get(stem); + // Only package wildcards retain `.*`. Static wildcards are interpreted with + // the owning type path so a same-named package cannot capture the import. + return inPackage !== undefined && stem !== '' ? inPackage : null; + } + + return resolveTypeName(targetRaw, index); +} + +/** + * Split a qualified name into the longest DECLARED package prefix and the type + * that follows it. + * + * Longest-first is what makes both of these land correctly without a rule about + * capitalization, which Java does not actually enforce: + * + * `com.example.model.User` -> package `com.example.model`, type `User` + * `com.example.Utils.method` -> package `com.example`, type `Utils` + * + * The second is a static member import; its trailing segments name members + * inside the type, and the file the import binds to is the type's. + */ +function resolveTypeName(qualified: string, index: JavaPackageIndex): string | null { + const parts = qualified.split('.').filter((part) => part !== ''); + // A single bare segment names a type in the default package, which Java + // forbids importing. Nothing to resolve, and nothing to guess at. + if (parts.length < 2) return null; + + for (let split = parts.length - 1; split >= 1; split--) { + const packageName = parts.slice(0, split).join('.'); + const types = index.typesByPackage.get(packageName); + if (types === undefined) continue; + const file = types.get(parts[split]); + if (file !== undefined) return file; + } + return null; +} + +/** `src/main/java/com/example/User.java` -> `User`. */ +function baseTypeName(filePath: string): string | null { + const slash = filePath.replace(/\\/g, '/').lastIndexOf('/'); + const base = slash === -1 ? filePath : filePath.slice(slash + 1); + if (!base.endsWith('.java')) return null; + const name = base.slice(0, -'.java'.length); + return name === '' ? null : name; +} diff --git a/gitnexus/src/core/ingestion/languages/java/query.ts b/gitnexus/src/core/ingestion/languages/java/query.ts index 99c72dc09..507d3ce79 100644 --- a/gitnexus/src/core/ingestion/languages/java/query.ts +++ b/gitnexus/src/core/ingestion/languages/java/query.ts @@ -89,7 +89,13 @@ const JAVA_SCOPE_QUERY = ` ])) @class-annotation.class ;; Declarations — methods / constructors +;; +;; A generic METHOD's parameters are read for the same reason a generic type's +;; are (#2912 review): \` boolean runAny(Validator v)\` writes a receiver +;; whose argument is a type VARIABLE, and a pass that cannot tell that from a +;; concrete type prunes every implementor from the call's dispatch fan-out. (method_declaration + type_parameters: (type_parameters)? @declaration.type-parameters name: (identifier) @declaration.name) @declaration.method (constructor_declaration diff --git a/gitnexus/src/core/ingestion/languages/java/record-components.ts b/gitnexus/src/core/ingestion/languages/java/record-components.ts new file mode 100644 index 000000000..51053510c --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/java/record-components.ts @@ -0,0 +1,233 @@ +import { SupportedLanguages, type CaptureMatch } from 'gitnexus-shared'; +import type { CaptureMap } from '../../language-provider.js'; +import { createMethodExtractor } from '../../method-extractors/generic.js'; +import { javaMethodConfig } from '../../method-extractors/configs/jvm.js'; +import { extractAnnotations } from '../../field-extractors/configs/helpers.js'; +import type { + ExtractedMethods, + MethodExtractor, + MethodExtractorContext, + MethodInfo, +} from '../../method-types.js'; +import { nodeToCapture, syntheticCapture, type SyntaxNode } from '../../utils/ast-helpers.js'; + +const javaExplicitMethodExtractor = createMethodExtractor(javaMethodConfig); + +function recordComponents(recordNode: SyntaxNode): SyntaxNode[] { + const parameters = recordNode.childForFieldName('parameters'); + if (parameters === null) return []; + return parameters.namedChildren.filter( + (node): node is SyntaxNode => + node !== null && (node.type === 'formal_parameter' || node.type === 'spread_parameter'), + ); +} + +/** + * A record component is named by a real `identifier` and nothing else. + * + * Two node shapes reach this that are not one, and both would mint a graph node + * for source that does not compile: + * + * - `record M(int x, y) {}` — a dropped type. tree-sitter recovers by + * synthesizing `name: (MISSING identifier)`, a zero-width node whose text is + * `''`. It still satisfies the query's `name: (identifier)`, so testing the + * node TYPE alone does not reject it. + * - `record R(int _) {}` — the grammar declares both `formal_parameter.name` + * and `variable_declarator.name` as `identifier | underscore_pattern`, and + * `_` parses with no error at all. `_` is illegal as a component name, and + * admitting it here while the query rejects it is what let the structure and + * scope paths disagree. + * + * Same degenerate-node shape as `javaBaseLookupNameNode` in captures.ts (#2935). + */ +function isRecordComponentName(node: SyntaxNode | null | undefined): node is SyntaxNode { + return ( + node !== null && + node !== undefined && + node.type === 'identifier' && + !node.isMissing && + node.text.length > 0 + ); +} + +function recordComponentNameNode(component: SyntaxNode): SyntaxNode | null { + const name = + component.type === 'formal_parameter' + ? component.childForFieldName('name') + : (component.namedChildren + .find((node) => node?.type === 'variable_declarator') + ?.childForFieldName('name') ?? null); + return isRecordComponentName(name) ? name : null; +} + +/** + * Memoised per record node. `shouldSkipJavaRecordComponentDefinition` is called + * once per component capture, so recomputing this would rescan the whole record + * body per component — O(components x body members) for a single record. The + * scope-capture path hoists the call out of its own loop instead; this cache is + * what gives the structure path the same cost. Keyed weakly on the AST node, so + * it drops with the tree at the end of the file's parse. + */ +const explicitZeroArgAccessorNamesCache = new WeakMap>(); + +function explicitZeroArgAccessorNames(recordNode: SyntaxNode): Set { + const memoized = explicitZeroArgAccessorNamesCache.get(recordNode); + if (memoized !== undefined) return memoized; + const names = computeExplicitZeroArgAccessorNames(recordNode); + explicitZeroArgAccessorNamesCache.set(recordNode, names); + return names; +} + +function computeExplicitZeroArgAccessorNames(recordNode: SyntaxNode): Set { + const names = new Set(); + const body = recordNode.childForFieldName('body'); + if (body === null) return names; + + for (const node of body.namedChildren) { + if (node === null || node.type !== 'method_declaration') continue; + const name = node.childForFieldName('name')?.text; + const parameters = node.childForFieldName('parameters'); + const parameterCount = + parameters?.namedChildren.filter( + (parameter) => + parameter !== null && + (parameter.type === 'formal_parameter' || parameter.type === 'spread_parameter'), + ).length ?? 0; + if (name !== undefined && parameterCount === 0) names.add(name); + } + return names; +} + +function recordComponentReturnType(component: SyntaxNode): string | null { + const typeNode = + component.childForFieldName('type') ?? + (component.type === 'spread_parameter' + ? component.namedChildren.find( + (node) => node?.type !== 'modifiers' && node?.type !== 'variable_declarator', + ) + : undefined); + const type = typeNode?.text; + if (type === undefined) return null; + return component.type === 'spread_parameter' ? `${type}[]` : type; +} + +function implicitAccessorInfo( + component: SyntaxNode, + context: MethodExtractorContext, +): MethodInfo | null { + const name = recordComponentNameNode(component)?.text; + if (name === undefined) return null; + + return { + name, + receiverType: null, + returnType: recordComponentReturnType(component), + parameters: [], + visibility: 'public', + isStatic: false, + isAbstract: false, + isFinal: false, + // JLS 8.10.3 / 9.7.4: a component annotation reaches the generated accessor + // when its @Target admits METHOD (or TYPE_USE, in the return-type position). + // ponytail: over-approximate — we propagate every component annotation, + // because @Target lives in another file and parsing is per-file, so the + // target set is not knowable here. Nothing reads Method annotations today: + // `annotations` is not a column in METHOD_SCHEMA/FUNCTION_SCHEMA + // (src/core/lbug/schema.ts), so it lives only in the in-memory graph for one + // analyze run, and the sole in-memory reader (springDiFieldMatcher) is gated + // to `Property` nodes. If that column is ever added, revisit this: the set + // would then become an agent-visible claim that may over-state the target. + annotations: extractAnnotations(component, 'modifiers'), + sourceFile: context.filePath, + line: component.startPosition.row + 1, + column: component.startPosition.column, + }; +} + +/** Java records synthesize one public, zero-argument accessor per component. */ +export const javaRecordMethodExtractor: MethodExtractor = { + ...javaExplicitMethodExtractor, + language: SupportedLanguages.Java, + extract(node: SyntaxNode, context: MethodExtractorContext): ExtractedMethods | null { + const extracted = javaExplicitMethodExtractor.extract(node, context); + if (extracted === null || node.type !== 'record_declaration') return extracted; + + const explicitAccessors = explicitZeroArgAccessorNames(node); + const implicitAccessors = recordComponents(node) + .filter((component) => { + const name = recordComponentNameNode(component)?.text; + return name !== undefined && !explicitAccessors.has(name); + }) + .map((component) => implicitAccessorInfo(component, context)) + .filter((method): method is MethodInfo => method !== null); + + return { ...extracted, methods: [...extracted.methods, ...implicitAccessors] }; + }, +}; + +/** Scope declarations matching the structure-phase synthetic accessor nodes. */ +export function synthesizeJavaRecordComponentAccessorCaptures( + rootNode: SyntaxNode, +): CaptureMatch[] { + const captures: CaptureMatch[] = []; + for (const recordNode of rootNode.descendantsOfType('record_declaration')) { + const explicitAccessors = explicitZeroArgAccessorNames(recordNode); + for (const component of recordComponents(recordNode)) { + const nameNode = recordComponentNameNode(component); + const returnType = recordComponentReturnType(component); + if (nameNode === null || returnType === null || explicitAccessors.has(nameNode.text)) + continue; + + captures.push({ + '@scope.function': nodeToCapture('@scope.function', component), + }); + captures.push({ + '@declaration.method': nodeToCapture('@declaration.method', component), + '@declaration.name': nodeToCapture('@declaration.name', nameNode), + '@declaration.parameter-count': syntheticCapture( + '@declaration.parameter-count', + component, + '0', + ), + '@declaration.required-parameter-count': syntheticCapture( + '@declaration.required-parameter-count', + component, + '0', + ), + '@declaration.return-type': syntheticCapture( + '@declaration.return-type', + component, + returnType, + ), + }); + } + } + return captures; +} + +/** + * The structure query sees every record component. Suppress that synthetic + * definition when the record body provides the canonical zero-argument + * accessor explicitly, leaving the explicit method as the single authority. + */ +export function shouldSkipJavaRecordComponentDefinition(captureMap: CaptureMap): boolean { + const component = captureMap['definition.method']; + if (component?.type !== 'formal_parameter' && component?.type !== 'spread_parameter') { + return false; + } + + const parameters = component.parent; + const recordNode = parameters?.parent; + if (parameters?.type !== 'formal_parameters' || recordNode?.type !== 'record_declaration') { + return false; + } + + // Same predicate the scope path applies, so the two can never disagree about + // which components have an accessor. The query's `name: (identifier)` is + // satisfied by tree-sitter's zero-width MISSING recovery token, so the + // structure path has to re-check what the query cannot express. + const nameNode = captureMap['name']; + if (!isRecordComponentName(nameNode)) return true; + + return explicitZeroArgAccessorNames(recordNode).has(nameNode.text); +} diff --git a/gitnexus/src/core/ingestion/languages/java/scope-resolver.ts b/gitnexus/src/core/ingestion/languages/java/scope-resolver.ts index 86c94bccb..1e26e2194 100644 --- a/gitnexus/src/core/ingestion/languages/java/scope-resolver.ts +++ b/gitnexus/src/core/ingestion/languages/java/scope-resolver.ts @@ -54,8 +54,10 @@ const javaScopeResolver: ScopeResolver = { return undefined; }, - resolveImportTarget: (targetRaw, fromFile, allFilePaths) => { - const ws: JavaResolveContext = { fromFile, allFilePaths }; + resolveImportTarget: (targetRaw, fromFile, allFilePaths, _resolutionConfig, context) => { + // `context.parsedFiles` is the whole input now: a Java import names a type + // in a DECLARED package, and the declarations live on those files (#2953). + const ws: JavaResolveContext = { fromFile, allFilePaths, parsedFiles: context?.parsedFiles }; return resolveJavaImportTarget( { kind: 'named', localName: '_', importedName: '_', targetRaw }, ws, diff --git a/gitnexus/src/core/ingestion/languages/javascript/import-target.ts b/gitnexus/src/core/ingestion/languages/javascript/import-target.ts index aa1914522..47b8e9fa8 100644 --- a/gitnexus/src/core/ingestion/languages/javascript/import-target.ts +++ b/gitnexus/src/core/ingestion/languages/javascript/import-target.ts @@ -1,118 +1,50 @@ /** * Import-target resolver for JavaScript. * - * Delegates to the TypeScript `resolveTsTarget` standard-strategy resolver - * with `language: SupportedLanguages.JavaScript` so the resolver tries - * `.js` / `.jsx` extensions in addition to (or instead of) `.ts` / `.tsx`. + * Delegates to the TypeScript resolver, which is correct rather than merely + * convenient: `jsconfig.json` is a tsconfig by another name, `package.json` + * governs both languages identically, and Node's algorithm does not branch on + * which of the two wrote the file. The extension list already carries the JS + * family, so a `.js`/`.jsx`/`.mjs`/`.cjs` source resolves the same way. * - * The `TsResolveContext.language` flag already exists in `import-target.ts` - * and the resolver (`resolveImportPath`) already branches on it — this - * adapter just wires the right value in. + * CJS `require()` calls reference the same module-path strings as ESM `import` + * statements, so the resolver handles them uniformly with no CJS-specific + * logic here. * - * CJS `require()` calls reference the same module-path strings as ESM - * `import` statements, so the resolver handles them uniformly without any - * CJS-specific logic here. + * ## What #2953 removed * - * No `tsconfig.json` path-alias support (JavaScript projects don't use - * `tsconfig.json` compilerOptions.paths in general). Projects that DO use - * tsconfig-based aliases alongside JavaScript can still resolve via the - * standard extension-suffix fallback; the alias branch is a no-op when - * `tsconfigPaths` is null. - * - * ## The suffix index changes bare-specifier answers (PR #2911) - * - * Supplying `index` is not only a speed-up: `suffixResolve` answers a different - * question with one than without. Without an index it tests - * `filePath.endsWith('/' + suffix)`, so only a PROPER suffix can match; with - * one it reads `buildSuffixIndex`, which indexes `j = 0` and therefore matches - * WHOLE paths too. Two classes of answer move, both only on the bare/absolute - * specifier leg (relative imports resolve by exact `Set.has` and never reach - * it), and both toward what TypeScript and Vue have always answered: - * - * 1. a repo-root file becomes reachable at all — `require('config')` now - * finds `config.js`, where before no proper suffix existed and the answer - * was null; - * 2. a whole-path candidate outranks a proper-suffix candidate found at a - * SHORTER path suffix or a later extension — `import 'app/main'` resolved - * to `node_modules/dep/lib/main.js` (the first `/main.js` in file order) - * and now resolves to `app/main.js`. - * - * Measured over 211 200 old-vs-new pairs there is no third class: the index - * never loses a match the scan found, and its answer is never matched at a less - * specific (path-part, extension) position. `test/unit/scope-resolution/ - * javascript-import-target-parity.test.ts` is that differential, and pins both - * classes by witness. + * This adapter used to reach `resolveImportPath`, whose last step was + * `suffixResolve` — a search for any repo file whose path ends in the + * specifier, retried with each leading segment dropped. The header this + * replaces recorded the symptom without naming it a defect: `import 'app/main'` + * resolving to `node_modules/dep/lib/main.js`, "the first `/main.js` in file + * order". A bare specifier now resolves only through a declared tsconfig + * mapping or a package manifest, and otherwise not at all. */ -import { SupportedLanguages } from 'gitnexus-shared'; -import { resolveTsTarget, type TsResolveContext } from '../typescript/import-target.js'; -import { buildImportPassCache } from '../../import-resolvers/pass-cache.js'; -import { perFileSet } from '../../import-resolvers/per-file-set.js'; +import type { NodeWorkspacePackages } from '../../import-resolvers/node-workspace-packages.js'; +import { resolveTsTarget } from '../typescript/import-target.js'; +import type { TsconfigIndex } from '../typescript/tsconfig.js'; -export type JsResolveContext = TsResolveContext; +interface JsResolutionConfig { + readonly tsconfigs?: TsconfigIndex | null; + readonly nodeWorkspacePackages?: NodeWorkspacePackages | null; +} -/** - * Everything `resolveTsTarget` derives from one workspace file set, built once - * per set rather than once per import. - * - * `index` is not optional, and its absence was the defect (PR #2911). The - * TypeScript adapter has carried a `SuffixIndex` since #1918; this one did not, - * so every JavaScript import reached `suffixResolve` with `index === undefined` - * and took its linear-`findIndex` fallback — one pass over `normalizedFileList` - * per path part per extension, and `EXTENSIONS` has ~39 entries. Measured on - * mostly-missing bare specifiers (imports scaling with files, as in - * `bench/import-target/`): 6448.9 µs per import at 2000 files and 25972.6 µs at - * 8000 — 4.12x the per-import cost for 4x the files, which is O(imports × - * files) — against 25.0 / 27.0 µs for TypeScript over the identical corpus. - * With the index it is 28.5 / 27.4 µs and the scaling factor is 1.09x. - * - * No instrument on the #2901-#2909 branch could see it: `CountingSet` counts - * traversals of the SET, and this scan walks the materialized array behind it. - * See `test/integration/javascript-import-index-reuse.test.ts` for the guard - * that can. - * - * Memoized on the `allFilePaths` Set identity, like every other language's - * import index (`import-resolvers/workspace-file-index.ts` and friends). - * - * A single-slot `let cached` keyed on `cached.key !== allFilePaths` — what this - * adapter used before — is correct for one file set and degenerate for two: - * alternating calls across two sets rebuild everything every time. Measured on - * the TypeScript adapter at 4000 files × 400 imports: 12.0 ms for one set, - * 1438.2 ms alternating between two (120x). A `WeakMap` has no such state to - * thrash, which is also what lets this adapter carry the standard - * `expectDistinctFileSetsGetOwnIndex` guard every other indexed adapter - * carries. - * - * The Set must be passed THROUGH by the caller, never copied: a defensive - * `new Set(allFilePaths)` at the adapter boundary hands a fresh key per import - * and restores the per-import rebuild (PR #1918 review P1). - */ -const passCacheFor = perFileSet(buildImportPassCache); - -/** - * Build a memoized `resolveImportTarget` adapter for JavaScript. - * Caches the derived arrays, the suffix index and the per-pass resolve cache - * across `resolveImportTarget` calls over one workspace file set. - */ +/** Build the JavaScript `resolveImportTarget` adapter. */ export function makeJsResolveImportTarget(): ( targetRaw: string, fromFile: string, allFilePaths: ReadonlySet, resolutionConfig?: unknown, ) => string | readonly string[] | null { - return (targetRaw, fromFile, allFilePaths) => { - const cached = passCacheFor(allFilePaths); - - const ws: JsResolveContext = { + return (targetRaw, fromFile, allFilePaths, resolutionConfig) => { + const cfg = resolutionConfig as JsResolutionConfig | undefined; + return resolveTsTarget(targetRaw, { fromFile, - language: SupportedLanguages.JavaScript, - allFilePaths: cached.allFilePaths, - allFileList: cached.allFileList, - normalizedFileList: cached.normalizedFileList, - index: cached.index, - resolveCache: cached.resolveCache, - tsconfigPaths: null, - }; - return resolveTsTarget(targetRaw, ws); + allFilePaths, + tsconfigs: cfg?.tsconfigs ?? null, + nodeWorkspacePackages: cfg?.nodeWorkspacePackages ?? null, + }); }; } diff --git a/gitnexus/src/core/ingestion/languages/javascript/scope-resolver.ts b/gitnexus/src/core/ingestion/languages/javascript/scope-resolver.ts index 94967384e..8d74798ee 100644 --- a/gitnexus/src/core/ingestion/languages/javascript/scope-resolver.ts +++ b/gitnexus/src/core/ingestion/languages/javascript/scope-resolver.ts @@ -42,6 +42,8 @@ import { javascriptProvider } from '../typescript.js'; import { jsMergeBindings } from './merge-bindings.js'; import { jsArityCompatibility } from './arity.js'; import { makeJsResolveImportTarget } from './import-target.js'; +import { loadTsconfigIndex } from '../typescript/tsconfig.js'; +import { loadNodeWorkspacePackages } from '../../import-resolvers/node-workspace-packages.js'; const javascriptScopeResolver: ScopeResolver = { // Construction is keyword-prefixed: `new Service(db).doWork()` (#2708). @@ -52,6 +54,15 @@ const javascriptScopeResolver: ScopeResolver = { resolveImportTarget: makeJsResolveImportTarget(), + // JavaScript resolution reads the same declared inputs TypeScript does — + // `jsconfig.json` is a tsconfig by another name, and `package.json` is shared + // outright. Without them a bare specifier used to fall through to suffix + // matching (#2953); now it simply does not resolve. + loadResolutionConfig: async (repoPath: string) => ({ + tsconfigs: await loadTsconfigIndex(repoPath), + nodeWorkspacePackages: await loadNodeWorkspacePackages(repoPath), + }), + // JavaScript LEGB — same tier ordering as TypeScript; no declaration- // merging across type/value/namespace spaces. mergeBindings: (existing, incoming) => [...jsMergeBindings([...existing, ...incoming])], diff --git a/gitnexus/src/core/ingestion/languages/kotlin/query.ts b/gitnexus/src/core/ingestion/languages/kotlin/query.ts index f442a2b37..94dadc59e 100644 --- a/gitnexus/src/core/ingestion/languages/kotlin/query.ts +++ b/gitnexus/src/core/ingestion/languages/kotlin/query.ts @@ -121,7 +121,13 @@ const KOTLIN_SCOPE_QUERY = ` ])) @class-annotation.class ;; Declarations — functions / methods / properties +;; +;; A generic FUNCTION's parameters are read for the same reason a generic type's +;; are (#2912 review): \`fun runAny(v: Validator)\` writes a receiver whose +;; argument is a type VARIABLE, and a pass that cannot tell that from a concrete +;; type prunes every implementor from the call's dispatch fan-out. (function_declaration + (type_parameters)? @declaration.type-parameters (simple_identifier) @declaration.name) @declaration.function ;; Lambda bound to a val/var: val handler = { x: Int -> target(x) } diff --git a/gitnexus/src/core/ingestion/languages/rust/captures.ts b/gitnexus/src/core/ingestion/languages/rust/captures.ts index ae15c99e2..d6685dde3 100644 --- a/gitnexus/src/core/ingestion/languages/rust/captures.ts +++ b/gitnexus/src/core/ingestion/languages/rust/captures.ts @@ -1,5 +1,6 @@ import type { Capture, CaptureMatch } from 'gitnexus-shared'; import { + findChild, nodeIfType, nodeToCapture, syntheticCapture, @@ -252,10 +253,22 @@ function synthesizeRustInheritanceReferences(root: SyntaxNode): CaptureMatch[] { const traitName = bareTypeIdentifier(traitField); const structName = bareTypeIdentifier(typeField); if (traitName === null || structName === null) return; + // The trait's generic ARGUMENTS (`impl Validator for V`), so + // interface dispatch can tell one instantiation of a trait from another + // (#2912). Emitted as a sub-tag rather than by widening the anchor: the + // anchor is the bare `type_identifier` inside the `generic_type`, and its + // range is part of the inheritance edge's id. + const traitArguments = + traitField.type === 'generic_type' ? findChild(traitField, 'type_arguments') : null; out.push({ '@reference.inherits': nodeToCapture('@reference.inherits', traitName), '@reference.name': nodeToCapture('@reference.name', traitName), '@reference.receiver': syntheticCapture('@reference.receiver', structName, structName.text), + ...(traitArguments === null + ? {} + : { + '@reference.type-arguments': nodeToCapture('@reference.type-arguments', traitArguments), + }), }); }); return out; diff --git a/gitnexus/src/core/ingestion/languages/rust/scope-resolver.ts b/gitnexus/src/core/ingestion/languages/rust/scope-resolver.ts index 5fd0f1570..bea53d9fd 100644 --- a/gitnexus/src/core/ingestion/languages/rust/scope-resolver.ts +++ b/gitnexus/src/core/ingestion/languages/rust/scope-resolver.ts @@ -16,6 +16,7 @@ import { import type { ScopeResolutionIndexes } from '../../model/scope-resolution-indexes.js'; import { resolveDefGraphId } from '../../scope-resolution/graph-bridge/ids.js'; import type { GraphNodeLookup } from '../../scope-resolution/graph-bridge/node-lookup.js'; +import type { HeritageTypeArgumentSink } from '../../scope-resolution/utils/generic-instantiation.js'; import type { KnowledgeGraph } from '../../../graph/types.js'; import { generateId } from '../../../../lib/utils.js'; @@ -54,6 +55,7 @@ function emitRustTraitImplEdges( parsedFiles: readonly ParsedFile[], nodeLookup: GraphNodeLookup, scopes: ScopeResolutionIndexes | undefined, + recordTypeArguments?: HeritageTypeArgumentSink, ): void { if (scopes === undefined) return; @@ -83,6 +85,14 @@ function emitRustTraitImplEdges( const traitGraphId = resolveDefGraphId(traitDef.filePath, traitDef, nodeLookup); if (structGraphId === undefined || traitGraphId === undefined) continue; + // The instantiation the impl was written with — `impl Validator + // for V` (#2912). Recorded against THIS edge's ids, not the pre-pass's: + // the pre-pass sources its edge from the enclosing def, and interface + // dispatch crosses the corrected one emitted here. + if (site.typeArguments !== undefined) { + recordTypeArguments?.(structGraphId, traitGraphId, site.typeArguments); + } + const edgeKey = `${structGraphId}->${traitGraphId}`; if (emitted.has(edgeKey)) continue; emitted.add(edgeKey); @@ -159,8 +169,8 @@ export const rustScopeResolver: ScopeResolver = { buildMro: (graph, parsedFiles, nodeLookup) => buildRustMro(graph, parsedFiles, nodeLookup), - emitHeritageEdges: (graph, parsedFiles, nodeLookup, scopes) => - emitRustTraitImplEdges(graph, parsedFiles, nodeLookup, scopes), + emitHeritageEdges: (graph, parsedFiles, nodeLookup, scopes, recordTypeArguments) => + emitRustTraitImplEdges(graph, parsedFiles, nodeLookup, scopes, recordTypeArguments), populateOwners: (parsed: ParsedFile) => populateRustOwners(parsed), diff --git a/gitnexus/src/core/ingestion/languages/typescript/file-candidates.ts b/gitnexus/src/core/ingestion/languages/typescript/file-candidates.ts new file mode 100644 index 000000000..421959788 --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/typescript/file-candidates.ts @@ -0,0 +1,70 @@ +/** + * Turning a resolved stem into a real file, the way TypeScript does (#2953). + * + * Shared by `module-resolution.ts` and the package-manifest resolver so both + * try the same three shapes — exact path, extension, directory index — and, as + * importantly, the same NARROW extension list. The repo-wide `EXTENSIONS` in + * `import-resolvers/utils.ts` carries ~39 entries spanning every language the + * indexer supports; a TypeScript import cannot resolve to a `.py` or `.rb` + * file, and letting it try was part of how the old suffix matcher found files + * that had nothing to do with the import. + */ + +/** Extension candidates, in the order TypeScript tries them. */ +export const TS_EXTENSIONS = [ + '.ts', + '.tsx', + '.d.ts', + '.mts', + '.cts', + '.js', + '.jsx', + '.mjs', + '.cjs', + '.vue', + '.json', +] as const; + +/** + * JS-family extensions a specifier may carry for a TypeScript source file. + * + * TypeScript ESM requires the specifier to name the EMITTED file (`./m.js`) + * while the file on disk is `./m.ts`, so a resolver that only tried the literal + * extension would miss every ESM-style relative import in a modern codebase. + */ +export const JS_TO_TS: ReadonlyMap = new Map([ + ['.js', ['.ts', '.tsx', '.d.ts']], + ['.jsx', ['.tsx']], + ['.mjs', ['.mts']], + ['.cjs', ['.cts']], +]); + +/** + * A repo-relative stem resolved to a real indexed file, or `null`. + * + * Exact match, then the ESM `.js` → `.ts` rewrite, then each extension, then + * the directory-index form. Nothing here searches: every candidate is derived + * from the stem the caller already resolved from a declared source. + */ +export function resolveFile(stem: string, allFiles: ReadonlySet): string | null { + if (stem === '') return null; + if (allFiles.has(stem)) return stem; + + const dot = stem.lastIndexOf('.'); + const ext = dot === -1 ? '' : stem.slice(dot); + const tsEquivalents = JS_TO_TS.get(ext); + if (tsEquivalents !== undefined) { + const stripped = stem.slice(0, -ext.length); + for (const candidate of tsEquivalents) { + if (allFiles.has(stripped + candidate)) return stripped + candidate; + } + } + + for (const candidate of TS_EXTENSIONS) { + if (allFiles.has(stem + candidate)) return stem + candidate; + } + for (const candidate of TS_EXTENSIONS) { + if (allFiles.has(`${stem}/index${candidate}`)) return `${stem}/index${candidate}`; + } + return null; +} diff --git a/gitnexus/src/core/ingestion/languages/typescript/import-target.ts b/gitnexus/src/core/ingestion/languages/typescript/import-target.ts index 782dc9cbf..2100a7717 100644 --- a/gitnexus/src/core/ingestion/languages/typescript/import-target.ts +++ b/gitnexus/src/core/ingestion/languages/typescript/import-target.ts @@ -1,44 +1,35 @@ /** * Adapter from `(ParsedImport, WorkspaceIndex)` → concrete file path. * - * Delegates to the existing standard-strategy resolver - * (`resolveImportPath`) so tsconfig path aliases (`@/`, `~/`, …) and - * suffix-based resolution follow the same rules as the legacy path. + * Delegates to `module-resolution.ts`, which runs the algorithm `tsc` and Node + * actually run. It used to delegate to the shared `resolveImportPath`, whose + * final step was `suffixResolve` — a repo-wide search for any file path ending + * in the specifier. That is what #2953 removed: this path now resolves only + * against declared inputs (real paths, tsconfig `paths`/`baseUrl`, package + * manifests) and answers `null` for everything else. * - * The `WorkspaceIndex` is opaque at the shared contract layer; we - * narrow it to a TypeScript-shaped context that carries `fromFile` + - * the full `allFilePaths` set + the optional `tsconfigPaths` the - * resolver reads. + * The `WorkspaceIndex` is opaque at the shared contract layer; we narrow it to + * a TypeScript-shaped context carrying `fromFile`, the workspace file set, and + * the two config indexes the algorithm reads. * * Returning `null` lets the finalize algorithm mark the edge as - * `linkStatus: 'unresolved'`. + * `linkStatus: 'unresolved'` — which for an external package is the correct + * and complete answer. */ import type { ParsedImport, WorkspaceIndex } from 'gitnexus-shared'; -import { SupportedLanguages } from 'gitnexus-shared'; -import { resolveImportPath } from '../../import-resolvers/standard.js'; -import type { SuffixIndex } from '../../import-resolvers/utils.js'; -import type { TsconfigPaths } from '../../language-config.js'; +import type { NodeWorkspacePackages } from '../../import-resolvers/node-workspace-packages.js'; +import { resolveTsModule } from './module-resolution.js'; +import type { TsconfigIndex } from './tsconfig.js'; export interface TsResolveContext { readonly fromFile: string; - /** Mutable `Set` because the standard resolver consumes `Set`. - * Callers holding a `ReadonlySet` should copy via `new Set(...)`. */ - readonly allFilePaths: Set; - /** Repo file list, normalized (lowercased) for suffix matching. May - * be supplied by the orchestrator; if absent we derive it on the - * fly from `allFilePaths`. */ - readonly allFileList?: readonly string[]; - readonly normalizedFileList?: readonly string[]; - /** Per-call resolution cache to dedupe repeated lookups. */ - readonly resolveCache?: Map; - /** Prebuilt suffix index for O(1)-style package/absolute import matching. */ - readonly index?: SuffixIndex; - /** Parsed tsconfig path-aliases. `null` = no aliases configured. */ - readonly tsconfigPaths?: TsconfigPaths | null; - /** JavaScript vs TypeScript switch — affects the extensions the - * resolver tries. Defaults to TypeScript. */ - readonly language?: SupportedLanguages.TypeScript | SupportedLanguages.JavaScript; + /** The workspace file set. */ + readonly allFilePaths: ReadonlySet; + /** Every tsconfig in the repo; `null` when the repo declares none. */ + readonly tsconfigs?: TsconfigIndex | null; + /** Every in-repo `package.json`; `null` when the repo declares none. */ + readonly nodeWorkspacePackages?: NodeWorkspacePackages | null; } export function resolveTsImportTarget( @@ -59,36 +50,21 @@ export function resolveTsImportTarget( } /** - * Resolve a raw module-path string to a workspace file path using the - * same standard-strategy resolver as the legacy DAG. Operates directly on - * the source string without requiring a `ParsedImport`, so the - * `ScopeResolver.resolveImportTarget` adapter doesn't need to construct - * a fake `ParsedImport` to reach the resolver. + * Resolve a raw module-path string to a workspace file path. Operates directly + * on the source string without requiring a `ParsedImport`, so the + * `ScopeResolver.resolveImportTarget` adapter doesn't need to construct a fake + * one to reach the resolver. * - * Returns `null` when: - * - the context is malformed (missing `fromFile` / `allFilePaths`) - * - `targetRaw` is empty - * - the resolver finds no matching file + * Returns `null` when `targetRaw` is empty, names an external package, or names + * something no declared config maps into the repo. */ export function resolveTsTarget(targetRaw: string, ctx: TsResolveContext): string | null { - if (targetRaw === '') return null; - - const language = ctx.language ?? SupportedLanguages.TypeScript; - const allFileList = ctx.allFileList ?? Array.from(ctx.allFilePaths); - const normalizedFileList = ctx.normalizedFileList ?? allFileList.map((f) => f.toLowerCase()); - const resolveCache = ctx.resolveCache ?? new Map(); - - return resolveImportPath( - ctx.fromFile, - targetRaw, - ctx.allFilePaths, - allFileList, - normalizedFileList, - resolveCache, - language, - ctx.tsconfigPaths ?? null, - ctx.index, - ); + return resolveTsModule(targetRaw, { + fromFile: ctx.fromFile, + allFilePaths: ctx.allFilePaths, + tsconfigs: ctx.tsconfigs ?? null, + workspacePackages: ctx.nodeWorkspacePackages ?? null, + }); } function narrowTsContext(workspaceIndex: WorkspaceIndex): TsResolveContext | null { diff --git a/gitnexus/src/core/ingestion/languages/typescript/module-resolution.ts b/gitnexus/src/core/ingestion/languages/typescript/module-resolution.ts new file mode 100644 index 000000000..ea8d2d637 --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/typescript/module-resolution.ts @@ -0,0 +1,199 @@ +/** + * TypeScript / JavaScript module resolution (#2953). + * + * This is the algorithm `tsc` and Node actually run, in the order they run it. + * It replaces `import-resolvers/utils.ts:suffixResolve` on the TS/JS/Vue path, + * which answered a different question — "does any file in this repo have a path + * ending in this specifier?" — and answered it by dropping leading segments + * until something matched. That is why `@acme/telemetry/nest`, a registry + * dependency, landed on the repo's only path ending in `nest/index.ts`. + * + * Every rule below resolves against something DECLARED: a real path, a + * `tsconfig` mapping, or a `package.json` manifest. A specifier that matches + * none of them is external, and external resolves to nothing. There is + * deliberately no fallback: a guess is what this module exists to remove, and + * an edge nobody declared is worse than a missing one precisely because it + * cannot be told apart from a real one downstream. + * + * ## The order, and why it is this order + * + * 1. relative / absolute — a path is a path; nothing else can claim it. + * 2. `#`-prefixed — package.json `imports`, which is scoped to the importing + * package and shadows everything else by design. + * 3. tsconfig `paths` — explicit mappings win over `baseUrl`, and the LONGEST + * matching pattern wins among them (tsc's rule, not first-declared). + * 4. tsconfig `baseUrl` — the rule that makes `import 'src/utils/foo'` legal. + * Note it applies only when a config actually declares one; without it, + * TypeScript treats a non-relative specifier as a package lookup, and so + * does this module. + * 5. workspace package — the manifest map, resolved through that package's + * own `exports` / `main` / `module` / `types`. + * 6. anything else — external. `null`. + */ + +import type { NodeWorkspacePackages } from '../../import-resolvers/node-workspace-packages.js'; +import { + matchSubpathMap, + nodePackageNameOf, + owningPackage, + resolveNodeWorkspaceImport, + substituteStar, +} from '../../import-resolvers/node-workspace-packages.js'; +import { resolveFile } from './file-candidates.js'; +import { tsconfigFor, type TsconfigIndex, type TsPathMapping } from './tsconfig.js'; + +export interface TsModuleResolutionContext { + readonly fromFile: string; + readonly allFilePaths: ReadonlySet; + readonly tsconfigs: TsconfigIndex | null; + readonly workspacePackages: NodeWorkspacePackages | null; +} + +/** + * Resolve one specifier to a repo file, or `null` when nothing in the repo + * declares it. + */ +export function resolveTsModule(specifier: string, ctx: TsModuleResolutionContext): string | null { + if (specifier === '') return null; + + // 1. A path specifier. + if (specifier.startsWith('.')) { + const joined = joinFrom(ctx.fromFile, specifier); + return joined === null ? null : resolveFile(joined, ctx.allFilePaths); + } + if (specifier.startsWith('/')) { + return resolveFile(specifier.slice(1), ctx.allFilePaths); + } + + // 2. Package-internal `#imports`. Scoped to the importing package, so it is + // looked up there and nowhere else — a `#` specifier that the package does + // not declare is an error in Node, not a repo-wide search. + if (specifier.startsWith('#')) { + return resolveSubpathImport(specifier, ctx); + } + + const config = tsconfigFor(ctx.tsconfigs, ctx.fromFile); + + // 3. `paths`, longest matching pattern first. + if (config !== null && config.paths.length > 0) { + const viaPaths = resolveViaPaths(specifier, config.paths, ctx.allFilePaths); + if (viaPaths !== null) return viaPaths; + } + + // 4. `baseUrl`. + if (config !== null && config.baseUrl !== null) { + const viaBaseUrl = resolveFile(joinRepo(config.baseUrl, specifier), ctx.allFilePaths); + if (viaBaseUrl !== null) return viaBaseUrl; + } + + // 5. A package that lives in this repo. + const viaWorkspace = resolveNodeWorkspaceImport( + specifier, + ctx.workspacePackages, + ctx.allFilePaths, + ); + if (viaWorkspace !== null) return viaWorkspace; + + // 6. External. Nothing in the repo declared it, so it resolves to nothing — + // which for a registry dependency is the correct and complete answer. + return null; +} + +/** + * Apply `paths` the way tsc does: the pattern with the longest literal prefix + * before `*` wins, and its targets are tried in declaration order. + * + * The old loader kept `targets[0]` and treated the pattern as a plain prefix, + * which silently mis-resolves the common `"@/*": ["./src/*", "./generated/*"]` + * shape — the second target is where half of a generated-code monorepo lives. + */ +function resolveViaPaths( + specifier: string, + paths: readonly TsPathMapping[], + allFiles: ReadonlySet, +): string | null { + const matches: { mapping: TsPathMapping; stem: string | null; prefixLength: number }[] = []; + + for (const mapping of paths) { + const star = mapping.pattern.indexOf('*'); + if (star === -1) { + if (mapping.pattern === specifier) { + matches.push({ mapping, stem: null, prefixLength: mapping.pattern.length }); + } + continue; + } + const prefix = mapping.pattern.slice(0, star); + const suffix = mapping.pattern.slice(star + 1); + if (!specifier.startsWith(prefix) || !specifier.endsWith(suffix)) continue; + if (specifier.length < prefix.length + suffix.length) continue; + matches.push({ + mapping, + stem: specifier.slice(prefix.length, specifier.length - suffix.length), + prefixLength: prefix.length, + }); + } + + // An exact (starless) pattern outranks any wildcard, THEN longer prefix wins. + // Sorting on prefix length alone left that first rule to luck: `a` and `a*` + // both match `a` with prefix length 1, so whichever was declared first won. + matches.sort( + (a, b) => Number(a.stem !== null) - Number(b.stem !== null) || b.prefixLength - a.prefixLength, + ); + + for (const match of matches) { + for (const target of match.mapping.targets) { + const candidate = match.stem === null ? target : substituteStar(target, match.stem); + const resolved = resolveFile(candidate, allFiles); + if (resolved !== null) return resolved; + } + } + return null; +} + +/** Resolve `#name` against the importing file's own package manifest. */ +function resolveSubpathImport(specifier: string, ctx: TsModuleResolutionContext): string | null { + const packages = ctx.workspacePackages; + if (packages === null) return null; + const owner = owningPackage(ctx.fromFile, packages); + if (owner === null) return null; + // `imports` takes pattern keys (`"#internal/*"`) exactly like `exports`, so + // it gets the same matcher rather than an exact lookup. + for (const stem of matchSubpathMap(owner.subpathImports, specifier) ?? []) { + const resolved = resolveFile(stem, ctx.allFilePaths); + if (resolved !== null) return resolved; + } + return null; +} + +/** + * Resolve a relative specifier against the importing file's directory, or + * `null` when it climbs out of the repository. + * + * Popping an empty segment list would silently CLAMP at the root, so + * `../../../secret` from `src/main.ts` became `secret` and could resolve a + * repo-root file the specifier never named. Outside the repo there is nothing + * indexed to resolve to, so the honest answer is nothing. + */ +function joinFrom(fromFile: string, specifier: string): string | null { + const segments = fromFile.split('/').slice(0, -1); + for (const part of specifier.split('/')) { + if (part === '.' || part === '') continue; + if (part === '..') { + if (segments.length === 0) return null; + segments.pop(); + } else { + segments.push(part); + } + } + return segments.join('/'); +} + +function joinRepo(dir: string, rest: string): string { + return dir === '' ? rest : `${dir}/${rest}`; +} + +/** Whether a specifier names a package rather than a path — used by callers + * that want to report an unresolved import as external rather than missing. */ +export function isPackageSpecifier(specifier: string): boolean { + return nodePackageNameOf(specifier) !== null; +} diff --git a/gitnexus/src/core/ingestion/languages/typescript/scope-resolver.ts b/gitnexus/src/core/ingestion/languages/typescript/scope-resolver.ts index bcaeffa51..bc2af4bcb 100644 --- a/gitnexus/src/core/ingestion/languages/typescript/scope-resolver.ts +++ b/gitnexus/src/core/ingestion/languages/typescript/scope-resolver.ts @@ -21,16 +21,13 @@ import type { ScopeResolver } from '../../scope-resolution/contract/scope-resolv import { simpleKey } from '../../scope-resolution/graph-bridge/node-lookup.js'; import type { ScopeResolutionIndexes } from '../../model/scope-resolution-indexes.js'; import { typescriptProvider } from '../typescript.js'; -import { loadTsconfigPaths, type TsconfigPaths } from '../../language-config.js'; -import { buildImportPassCache } from '../../import-resolvers/pass-cache.js'; -import { perFileSet } from '../../import-resolvers/per-file-set.js'; -import { indexOnlyElementType } from '../../type-extractors/shared.js'; +import { loadTsconfigIndex, type TsconfigIndex } from './tsconfig.js'; import { - typescriptArityCompatibility, - typescriptMergeBindings, - resolveTsTarget, - type TsResolveContext, -} from './index.js'; + loadNodeWorkspacePackages, + type NodeWorkspacePackages, +} from '../../import-resolvers/node-workspace-packages.js'; +import { indexOnlyElementType } from '../../type-extractors/shared.js'; +import { typescriptArityCompatibility, typescriptMergeBindings, resolveTsTarget } from './index.js'; import { getNuxtAutoImportEntry, hasNuxtAutoImports, @@ -40,7 +37,10 @@ import { /** Shape the orchestrator threads in via `RunScopeResolutionInput.resolutionConfig`. */ interface TypescriptResolutionConfig { - readonly tsconfigPaths: TsconfigPaths | null; + /** Every tsconfig in the repo, `extends` resolved (#2953). */ + readonly tsconfigs: TsconfigIndex | null; + /** Every in-repo `package.json`, for workspace-package resolution (#2953). */ + readonly nodeWorkspacePackages: NodeWorkspacePackages | null; /** Nuxt/Nitro auto-import map. Null for non-Nuxt projects. */ readonly nuxtAutoImports: NuxtAutoImportConfig | null; } @@ -56,43 +56,22 @@ const TYPESCRIPT_TYPE_ONLY_BINDING_TYPES = new Set([ ]); /** - * Memoized on the `allFilePaths` Set identity, like every other language's - * import index (`import-resolvers/workspace-file-index.ts` and friends). + * Build the `resolveImportTarget` adapter. * - * This used to be a single-slot `let cached` invalidated by - * `cached.key !== allFilePaths` — correct for one file set and degenerate for - * two: alternating calls across two sets rebuilt everything every time. - * Measured here at 4000 files × 400 imports: 12.0 ms for one set, 1438.2 ms - * alternating between two (120x). A `WeakMap` has no such state to thrash, and - * it is what lets this adapter carry the standard - * `expectDistinctFileSetsGetOwnIndex` guard the other languages carry - * (`test/integration/typescript-import-index-reuse.test.ts`). - * - * The Set must be passed THROUGH by the caller, never copied: a defensive - * `new Set(allFilePaths)` at the adapter boundary hands a fresh key per import - * and restores the per-import rebuild (PR #1918 review P1). - */ -const tsPassCacheFor = perFileSet(buildImportPassCache); - -/** - * Build a `resolveImportTarget` adapter that reads the memoized per-file-set - * state above rather than re-deriving it on every import lookup. + * No per-file-set memo any more: the suffix index it existed to amortize is + * gone with #2953. Real resolution derives nothing from the file list — every + * candidate comes from a config the repo declares, and checking one is a + * `Set.has` — so there is nothing left to cache per pass. */ function makeTsResolveImportTarget(): ScopeResolver['resolveImportTarget'] { return (targetRaw, fromFile, allFilePaths, resolutionConfig) => { - const cached = tsPassCacheFor(allFilePaths); - const cfg = resolutionConfig as TypescriptResolutionConfig | undefined; - const ws: TsResolveContext = { + return resolveTsTarget(targetRaw, { fromFile, - allFilePaths: cached.allFilePaths, - allFileList: cached.allFileList, - normalizedFileList: cached.normalizedFileList, - index: cached.index, - resolveCache: cached.resolveCache, - tsconfigPaths: cfg?.tsconfigPaths ?? null, - }; - return resolveTsTarget(targetRaw, ws); + allFilePaths, + tsconfigs: cfg?.tsconfigs ?? null, + nodeWorkspacePackages: cfg?.nodeWorkspacePackages ?? null, + }); }; } @@ -118,7 +97,8 @@ const typescriptScopeResolver: ScopeResolver = { // `nuxtAutoImports` is null for non-Nuxt projects (no .nuxt/imports.d.ts), // so this adds zero overhead to ordinary TypeScript repos. loadResolutionConfig: async (repoPath: string) => ({ - tsconfigPaths: await loadTsconfigPaths(repoPath), + tsconfigs: await loadTsconfigIndex(repoPath), + nodeWorkspacePackages: await loadNodeWorkspacePackages(repoPath), nuxtAutoImports: await loadNuxtAutoImports(repoPath), }), diff --git a/gitnexus/src/core/ingestion/languages/typescript/tsconfig.ts b/gitnexus/src/core/ingestion/languages/typescript/tsconfig.ts new file mode 100644 index 000000000..a111f5752 --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/typescript/tsconfig.ts @@ -0,0 +1,338 @@ +/** + * Real `tsconfig.json` loading for module resolution (#2953). + * + * The previous loader (`language-config.ts:loadTsconfigPaths`) was built to feed + * a heuristic, and it shows: it reads three filenames at the repo ROOT only, + * gives up unless `compilerOptions.paths` exists, keeps only `targets[0]` of + * each mapping, and treats a pattern as a plain prefix. That is enough to make + * a guess look plausible and not enough to resolve anything correctly: + * + * - a monorepo has one tsconfig PER PACKAGE, and `apps/web/tsconfig.json` is + * what governs `apps/web/src/main.ts` — the root config governs nothing; + * - `extends` is how essentially every real config is written, and the + * `baseUrl` / `paths` almost always live in the extended base; + * - `baseUrl` alone (no `paths`) is a complete resolution rule on its own, and + * it is exactly the rule that makes `import 'src/utils/foo'` legal — the + * case the old suffix matcher was really standing in for; + * - `paths` maps a pattern to an ORDERED LIST of targets, tried in order. + * + * So this module answers the question TypeScript actually asks: for THIS file, + * what are `baseUrl` and `paths`? + */ + +import fs from 'fs/promises'; +import path from 'path'; + +import { isHardcodedIgnoredDirectory } from '../../../../config/ignore-service.js'; +import { logger } from '../../../logger.js'; + +/** One `paths` entry, pattern and targets kept in declaration order. */ +export interface TsPathMapping { + /** The pattern as written, e.g. `@/*`, `@app/*`, `exact`. */ + readonly pattern: string; + /** Targets as written, relative to `baseUrl`. Tried in order. */ + readonly targets: readonly string[]; +} + +/** The resolution-relevant part of one resolved tsconfig. */ +export interface TsconfigScope { + /** Repo-relative directory the config governs (the tsconfig's own directory). */ + readonly dir: string; + /** + * Repo-relative `baseUrl`, or `null` when the config declares none. + * + * `null` is not the same as `'.'`: without `baseUrl`, TypeScript does NOT + * resolve non-relative specifiers against the project at all (they are + * package lookups), and `paths` targets are resolved against the tsconfig's + * own directory instead. + */ + readonly baseUrl: string | null; + readonly paths: readonly TsPathMapping[]; +} + +/** Every tsconfig in the repo, indexed so the nearest one to a file wins. */ +export interface TsconfigIndex { + /** Deepest-first, so the first `dir` that prefixes a file path governs it. */ + readonly scopes: readonly TsconfigScope[]; +} + +const SCAN_MAX_DIRS = 20_000; +const SCAN_MAX_DEPTH = 24; +/** Guard against an `extends` cycle or a pathological chain. */ +const MAX_EXTENDS_DEPTH = 16; + +/** + * The config governing `filePath` — the nearest tsconfig at or above it. + * + * TypeScript resolves a file against the project that includes it; the nearest + * enclosing tsconfig is the faithful approximation of that without evaluating + * `include`/`exclude` globs, and it is what makes a monorepo's per-package + * `baseUrl` apply to that package's files instead of the root's. + */ +export function tsconfigFor(index: TsconfigIndex | null, filePath: string): TsconfigScope | null { + if (index === null) return null; + for (const scope of index.scopes) { + if (scope.dir === '') return scope; + if (filePath.startsWith(`${scope.dir}/`)) return scope; + } + return null; +} + +/** Load every tsconfig in the repo, resolving `extends` chains. */ +export async function loadTsconfigIndex(repoRoot: string): Promise { + const files = await findTsconfigFiles(repoRoot); + if (files.length === 0) return null; + + const ranked: { scope: TsconfigScope; rank: number }[] = []; + for (const absPath of files) { + const options = await readCompilerOptions(absPath, 0); + if (options === null) continue; + // `readCompilerOptions` resolves both to ABSOLUTE paths against whichever + // config in the `extends` chain declared them, which is the only way the + // chain stays unambiguous. Rebasing to repo-relative happens once, here. + const baseUrl = options.baseUrl === undefined ? null : repoRelative(repoRoot, options.baseUrl); + const paths = (options.paths ?? []).map((mapping) => ({ + pattern: mapping.pattern, + targets: mapping.targets.map((t) => rebaseTarget(repoRoot, t)), + })); + // A config declaring NEITHER is kept, not skipped. Dropping it let + // `tsconfigFor` fall through to an enclosing config, so a package whose own + // tsconfig declares no `baseUrl` — meaning its non-relative specifiers are + // package lookups — silently inherited the repo root's aliases instead. + // An empty scope is the accurate answer for such a file, and only a scope + // can express it. + ranked.push({ + scope: { dir: repoRelative(repoRoot, path.dirname(absPath)), baseUrl, paths }, + rank: configRank(path.basename(absPath)), + }); + } + if (ranked.length === 0) return null; + + // Deepest first, because `tsconfigFor` takes the first match and it must be + // the most specific config rather than whichever the walk reached first. + // + // Then by filename rank WITHIN a directory, which is the half that is easy to + // miss: `tsconfig.json` and `tsconfig.base.json` routinely sit side by side, + // and the base exists to be extended, not to govern. Reading whichever the + // directory listing returned first made a config's own `paths` invisible + // whenever its base happened to be listed earlier. + ranked.sort((a, b) => b.scope.dir.length - a.scope.dir.length || a.rank - b.rank); + return { scopes: ranked.map((entry) => entry.scope) }; +} + +/** + * Precedence among configs sharing a directory: the project config governs, and + * everything else is a base or a variant that exists to be extended. + */ +function configRank(fileName: string): number { + if (fileName === 'tsconfig.json') return 0; + if (fileName === 'jsconfig.json') return 1; + return 2; +} + +/** Resolved compiler options, rebased to repo-relative paths. */ +interface ResolvedOptions { + baseUrl?: string; + paths?: TsPathMapping[]; +} + +/** + * Read one tsconfig and merge in whatever it `extends`. + * + * Rebasing happens per FILE, before merging, because `extends` does not rebase + * `baseUrl`: a base config at `configs/tsconfig.base.json` declaring + * `"baseUrl": "."` means `configs/`, even when extended from `apps/web`. Doing + * the rebase at read time is what keeps that true through the chain. + */ +async function readCompilerOptions( + absPath: string, + depth: number, + repoRootHint?: string, +): Promise { + if (depth > MAX_EXTENDS_DEPTH) { + logger.warn(`[typescript] tsconfig extends chain too deep at ${absPath}; ignoring the rest`); + return null; + } + + let parsed: Record; + try { + parsed = parseJsonc(await fs.readFile(absPath, 'utf-8')); + } catch { + return null; + } + + const dir = path.dirname(absPath); + // Read what this config extends FIRST: `paths` targets resolve against the + // EFFECTIVE `baseUrl`, which a config declaring `paths` alone inherits from + // its base. Resolving them against this config's own directory instead would + // load the right alias pattern and point every target at the wrong place. + const inherited = await readExtended(parsed.extends, dir, depth, repoRootHint); + + const own: ResolvedOptions = {}; + const compilerOptions = parsed.compilerOptions; + if (compilerOptions !== null && typeof compilerOptions === 'object') { + const opts = compilerOptions as Record; + if (typeof opts.baseUrl === 'string') { + own.baseUrl = path.resolve(dir, opts.baseUrl); + } + if (opts.paths !== null && typeof opts.paths === 'object' && !Array.isArray(opts.paths)) { + // tsc resolves `paths` targets against the effective `baseUrl` — this + // config's own if it declares one, otherwise the inherited one — and + // against the config's own directory only when neither exists. Doing it + // here, per file, is what keeps an `extends` chain unambiguous: by the + // time these merge, every target is already absolute. + const pathsBase = own.baseUrl ?? inherited?.baseUrl ?? dir; + own.paths = []; + for (const [pattern, targets] of Object.entries(opts.paths as Record)) { + if (!Array.isArray(targets)) continue; + const asStrings = targets + .filter((t): t is string => typeof t === 'string') + .map((t) => path.resolve(pathsBase, t)); + if (asStrings.length > 0) own.paths.push({ pattern, targets: asStrings }); + } + } + } + + // Own options win over inherited ones — that is what `extends` means. `paths` + // is replaced wholesale rather than merged, matching tsc. + return { + ...(inherited ?? {}), + ...own, + }; +} + +/** Follow `extends`, which may be a string or (TS 5+) an array, base-first. */ +async function readExtended( + value: unknown, + fromDir: string, + depth: number, + repoRootHint?: string, +): Promise { + const specs = typeof value === 'string' ? [value] : Array.isArray(value) ? value : []; + let merged: ResolvedOptions | null = null; + for (const spec of specs) { + if (typeof spec !== 'string') continue; + const resolved = await resolveExtendsTarget(spec, fromDir); + if (resolved === null) continue; + const options = await readCompilerOptions(resolved, depth + 1, repoRootHint); + if (options === null) continue; + // Later entries win over earlier ones, per tsc's array semantics. + merged = { ...(merged ?? {}), ...options }; + } + return merged; +} + +/** + * An `extends` value is either a path or a package name. + * + * The package form (`"extends": "@tsconfig/node20/tsconfig.json"`, + * `"@acme/tsconfig"`) lives in `node_modules`, which this tool deliberately + * does NOT index — it is dependency code, not the repository's own. But not + * indexing it is different from not READING it, and the distinction matters + * here: a shared internal base config is exactly where a monorepo puts the + * `paths` its packages import through, so refusing to open it loses aliases + * that the repository genuinely declares. + * + * So the file is read from disk when it is there, walking `node_modules` up + * from the extending config the way Node does. When it is absent — an + * un-installed checkout, which is a shape a static analyser must expect and a + * compiler may refuse — the answer is `null`, and the caller keeps whatever the + * extending config declared itself. That degrades to fewer resolutions, never + * to invented ones. + */ +async function resolveExtendsTarget(spec: string, fromDir: string): Promise { + if (spec.startsWith('.') || path.isAbsolute(spec)) { + return firstReadableConfig(path.resolve(fromDir, spec)); + } + for (const modulesDir of nodeModulesChain(fromDir)) { + const found = await firstReadableConfig(path.join(modulesDir, spec)); + if (found !== null) return found; + } + return null; +} + +/** `/node_modules`, then each ancestor's, the way Node resolves. */ +function* nodeModulesChain(fromDir: string): Generator { + let dir = fromDir; + for (;;) { + if (path.basename(dir) !== 'node_modules') yield path.join(dir, 'node_modules'); + const parent = path.dirname(dir); + if (parent === dir) return; + dir = parent; + } +} + +/** The first spelling of `base` that is a readable file. */ +async function firstReadableConfig(base: string): Promise { + for (const candidate of [base, `${base}.json`, path.join(base, 'tsconfig.json')]) { + try { + const stat = await fs.stat(candidate); + if (stat.isFile()) return candidate; + } catch { + // try the next spelling + } + } + return null; +} + +async function findTsconfigFiles(repoRoot: string): Promise { + const found: string[] = []; + const queue: { dir: string; depth: number }[] = [{ dir: repoRoot, depth: 0 }]; + let dirsScanned = 0; + + while (queue.length > 0 && dirsScanned < SCAN_MAX_DIRS) { + const { dir, depth } = queue.shift()!; + dirsScanned++; + let entries: import('fs').Dirent[]; + try { + entries = await fs.readdir(dir, { withFileTypes: true }); + } catch { + continue; + } + for (const entry of entries) { + if (entry.isDirectory()) { + if (isHardcodedIgnoredDirectory(entry.name)) continue; + if (depth < SCAN_MAX_DEPTH) + queue.push({ dir: path.join(dir, entry.name), depth: depth + 1 }); + continue; + } + if (!entry.isFile()) continue; + // `tsconfig.json`, `tsconfig.app.json`, `jsconfig.json`, … — any of them + // can carry the `baseUrl`/`paths` that governs its directory. + if (/^(ts|js)config(\..+)?\.json$/.test(entry.name)) { + found.push(path.join(dir, entry.name)); + } + } + } + return found; +} + +/** Strip comments and trailing commas — tsconfig is JSONC, not JSON. */ +function parseJsonc(raw: string): Record { + const withoutComments = raw + .replace(/\\"|"(?:\\"|[^"])*"|(\/\/.*$)|(\/\*[\s\S]*?\*\/)/gm, (match, line, block) => + line !== undefined || block !== undefined ? '' : match, + ) + .replace(/,(\s*[}\]])/g, '$1'); + return JSON.parse(withoutComments) as Record; +} + +function repoRelative(repoRoot: string, absDir: string): string { + const rel = path.relative(repoRoot, absDir).split(path.sep).join('/'); + return rel === '.' || rel === '' ? '' : rel; +} + +/** + * A `paths` target rebased to repo-relative, keeping any trailing `*`. + * + * `path.resolve` swallows the wildcard into a path segment, so it is stripped + * before resolving and re-appended after — the `*` is a substitution marker, + * not a directory named `*`. + */ +function rebaseTarget(repoRoot: string, absTarget: string): string { + // `/repo/src/*` must come back as `src/*`, not `src*`: stripping only the + // star leaves a trailing slash that `path.relative` then eats. + const suffix = absTarget.endsWith('/*') ? '/*' : absTarget.endsWith('*') ? '*' : ''; + const base = suffix === '' ? absTarget : absTarget.slice(0, -suffix.length); + return `${repoRelative(repoRoot, base)}${suffix}`; +} diff --git a/gitnexus/src/core/ingestion/languages/vue/import-target.ts b/gitnexus/src/core/ingestion/languages/vue/import-target.ts index a50c4aa54..886580b0e 100644 --- a/gitnexus/src/core/ingestion/languages/vue/import-target.ts +++ b/gitnexus/src/core/ingestion/languages/vue/import-target.ts @@ -1,60 +1,23 @@ /** * Import-target resolver for Vue SFCs (RFC #909 Ring 3, issue #940). * - * Vue `