Merge branch 'main' into fix/issue-1518-docker-local-path

This commit is contained in:
Gergő Magyar 2026-06-09 07:36:19 +01:00 • committed by GitHub
commit bb0d3c96d6
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
259 changed files with 13740 additions and 4405 deletions

View file

@ -11,7 +11,7 @@
"plugins": [
{
"name": "gitnexus",
"version": "1.3.3",
"version": "1.6.6",
"source": "./gitnexus-claude-plugin",
"description": "Code intelligence powered by a knowledge graph. Provides execution flow tracing, blast radius analysis, and augmented search across your codebase."
}

View file

@ -16,10 +16,10 @@ description: "Use when the user is debugging a bug, tracing an error, or asking
## Workflow
```
1. gitnexus_query({query: "<error or symptom>"}) → Find related execution flows
2. gitnexus_context({name: "<suspect>"}) → See callers/callees/processes
1. query({query: "<error or symptom>"}) → Find related execution flows
2. context({name: "<suspect>"}) → See callers/callees/processes
3. READ gitnexus://repo/{name}/process/{name} → Trace execution flow
4. gitnexus_cypher({query: "MATCH path..."}) → Custom traces if needed
4. cypher({query: "MATCH path..."}) → Custom traces if needed
```
> If "Index is stale" → run `node .gitnexus/run.cjs analyze` in terminal.
@ -28,11 +28,11 @@ description: "Use when the user is debugging a bug, tracing an error, or asking
```
- [ ] Understand the symptom (error message, unexpected behavior)
- [ ] gitnexus_query for error text or related code
- [ ] query for error text or related code
- [ ] Identify the suspect function from returned processes
- [ ] gitnexus_context to see callers and callees
- [ ] context to see callers and callees
- [ ] Trace execution flow via process resource if applicable
- [ ] gitnexus_cypher for custom call chain traces if needed
- [ ] cypher for custom call chain traces if needed
- [ ] Read source files to confirm root cause
```
@ -40,7 +40,7 @@ description: "Use when the user is debugging a bug, tracing an error, or asking
| Symptom | GitNexus Approach |
| -------------------- | ---------------------------------------------------------- |
| Error message | `gitnexus_query` for error text → `context` on throw sites |
| Error message | `query` for error text → `context` on throw sites |
| Wrong return value | `context` on the function → trace callees for data flow |
| Intermittent failure | `context` → look for external calls, async deps |
| Performance issue | `context` → find symbols with many callers (hot paths) |
@ -48,24 +48,24 @@ description: "Use when the user is debugging a bug, tracing an error, or asking
## Tools
**gitnexus_query** — find code related to error:
**query** — find code related to error:
```
gitnexus_query({query: "payment validation error"})
query({query: "payment validation error"})
→ Processes: CheckoutFlow, ErrorHandling
→ Symbols: validatePayment, handlePaymentError, PaymentException
```
**gitnexus_context** — full context for a suspect:
**context** — full context for a suspect:
```
gitnexus_context({name: "validatePayment"})
context({name: "validatePayment"})
→ Incoming calls: processCheckout, webhookHandler
→ Outgoing calls: verifyCard, fetchRates (external API!)
→ Processes: CheckoutFlow (step 3/7)
```
**gitnexus_cypher** — custom call chain traces:
**cypher** — custom call chain traces:
```cypher
MATCH path = (a)-[:CodeRelation {type: 'CALLS'}*1..2]->(b:Function {name: "validatePayment"})
@ -75,11 +75,11 @@ RETURN [n IN nodes(path) | n.name] AS chain
## Example: "Payment endpoint returns 500 intermittently"
```
1. gitnexus_query({query: "payment error handling"})
1. query({query: "payment error handling"})
→ Processes: CheckoutFlow, ErrorHandling
→ Symbols: validatePayment, handlePaymentError
2. gitnexus_context({name: "validatePayment"})
2. context({name: "validatePayment"})
→ Outgoing calls: verifyCard, fetchRates (external API!)
3. READ gitnexus://repo/my-app/process/CheckoutFlow

View file

@ -18,8 +18,8 @@ description: "Use when the user asks how code works, wants to understand archite
```
1. READ gitnexus://repos → Discover indexed repos
2. READ gitnexus://repo/{name}/context → Codebase overview, check staleness
3. gitnexus_query({query: "<what you want to understand>"}) → Find related execution flows
4. gitnexus_context({name: "<symbol>"}) → Deep dive on specific symbol
3. query({query: "<what you want to understand>"}) → Find related execution flows
4. context({name: "<symbol>"}) → Deep dive on specific symbol
5. READ gitnexus://repo/{name}/process/{name} → Trace full execution flow
```
@ -29,9 +29,9 @@ description: "Use when the user asks how code works, wants to understand archite
```
- [ ] READ gitnexus://repo/{name}/context
- [ ] gitnexus_query for the concept you want to understand
- [ ] query for the concept you want to understand
- [ ] Review returned processes (execution flows)
- [ ] gitnexus_context on key symbols for callers/callees
- [ ] context on key symbols for callers/callees
- [ ] READ process resource for full execution traces
- [ ] Read source files for implementation details
```
@ -47,18 +47,18 @@ description: "Use when the user asks how code works, wants to understand archite
## Tools
**gitnexus_query** — find execution flows related to a concept:
**query** — find execution flows related to a concept:
```
gitnexus_query({query: "payment processing"})
query({query: "payment processing"})
→ Processes: CheckoutFlow, RefundFlow, WebhookHandler
→ Symbols grouped by flow with file locations
```
**gitnexus_context** — 360-degree view of a symbol:
**context** — 360-degree view of a symbol:
```
gitnexus_context({name: "validateUser"})
context({name: "validateUser"})
→ Incoming calls: loginHandler, apiMiddleware
→ Outgoing calls: checkToken, getUserById
→ Processes: LoginFlow (step 2/5), TokenRefresh (step 1/3)
@ -68,10 +68,10 @@ gitnexus_context({name: "validateUser"})
```
1. READ gitnexus://repo/my-app/context → 918 symbols, 45 processes
2. gitnexus_query({query: "payment processing"})
2. query({query: "payment processing"})
→ CheckoutFlow: processPayment → validateCard → chargeStripe
→ RefundFlow: initiateRefund → calculateRefund → processRefund
3. gitnexus_context({name: "processPayment"})
3. context({name: "processPayment"})
→ Incoming: checkoutHandler, webhookHandler
→ Outgoing: validateCard, chargeStripe, saveTransaction
4. Read src/payments/processor.ts for implementation details

View file

@ -17,9 +17,9 @@ description: "Use when the user wants to know what will break if they change som
## Workflow
```
1. gitnexus_impact({target: "X", direction: "upstream"}) → What depends on this
1. impact({target: "X", direction: "upstream"}) → What depends on this
2. READ gitnexus://repo/{name}/processes → Check affected execution flows
3. gitnexus_detect_changes() → Map current git changes to affected flows
3. detect_changes() → Map current git changes to affected flows
4. Assess risk and report to user
```
@ -28,11 +28,11 @@ description: "Use when the user wants to know what will break if they change som
## Checklist
```
- [ ] gitnexus_impact({target, direction: "upstream"}) to find dependents
- [ ] impact({target, direction: "upstream"}) to find dependents
- [ ] Review d=1 items first (these WILL BREAK)
- [ ] Check high-confidence (>0.8) dependencies
- [ ] READ processes to check affected execution flows
- [ ] gitnexus_detect_changes() for pre-commit check
- [ ] detect_changes() for pre-commit check
- [ ] Assess risk level and report to user
```
@ -55,10 +55,10 @@ description: "Use when the user wants to know what will break if they change som
## Tools
**gitnexus_impact** — the primary tool for symbol blast radius:
**impact** — the primary tool for symbol blast radius:
```
gitnexus_impact({
impact({
target: "validateUser",
direction: "upstream",
minConfidence: 0.8,
@ -73,10 +73,10 @@ gitnexus_impact({
- authRouter (src/routes/auth.ts:22) [CALLS, 95%]
```
**gitnexus_detect_changes** — git-diff based impact analysis:
**detect_changes** — git-diff based impact analysis:
```
gitnexus_detect_changes({scope: "staged"})
detect_changes({scope: "staged"})
→ Changed: 5 symbols in 3 files
→ Affected: LoginFlow, TokenRefresh, APIMiddlewarePipeline
@ -86,7 +86,7 @@ gitnexus_detect_changes({scope: "staged"})
## Example: "What breaks if I change validateUser?"
```
1. gitnexus_impact({target: "validateUser", direction: "upstream"})
1. impact({target: "validateUser", direction: "upstream"})
→ d=1: loginHandler, apiMiddleware (WILL BREAK)
→ d=2: authRouter, sessionManager (LIKELY AFFECTED)

View file

@ -18,10 +18,10 @@ description: "Use when the user wants to review a pull request, understand what
```
1. gh pr diff <number> → Get the raw diff
2. gitnexus_detect_changes({scope: "compare", base_ref: "main"}) → Map diff to affected flows
2. detect_changes({scope: "compare", base_ref: "main"}) → Map diff to affected flows
3. For each changed symbol:
gitnexus_impact({target: "<symbol>", direction: "upstream"}) → Blast radius per change
4. gitnexus_context({name: "<key symbol>"}) → Understand callers/callees
impact({target: "<symbol>", direction: "upstream"}) → Blast radius per change
4. context({name: "<key symbol>"}) → Understand callers/callees
5. READ gitnexus://repo/{name}/processes → Check affected execution flows
6. Summarize findings with risk assessment
```
@ -32,10 +32,10 @@ description: "Use when the user wants to review a pull request, understand what
```
- [ ] Fetch PR diff (gh pr diff or git diff base...head)
- [ ] gitnexus_detect_changes to map changes to affected execution flows
- [ ] gitnexus_impact on each non-trivial changed symbol
- [ ] detect_changes to map changes to affected execution flows
- [ ] impact on each non-trivial changed symbol
- [ ] Review d=1 items (WILL BREAK) — are callers updated?
- [ ] gitnexus_context on key changed symbols to understand full picture
- [ ] context on key changed symbols to understand full picture
- [ ] Check if affected processes have test coverage
- [ ] Assess overall risk level
- [ ] Write review summary with findings
@ -63,20 +63,20 @@ description: "Use when the user wants to review a pull request, understand what
## Tools
**gitnexus_detect_changes** — map PR diff to affected execution flows:
**detect_changes** — map PR diff to affected execution flows:
```
gitnexus_detect_changes({scope: "compare", base_ref: "main"})
detect_changes({scope: "compare", base_ref: "main"})
→ Changed: 8 symbols in 4 files
→ Affected processes: CheckoutFlow, RefundFlow, WebhookHandler
→ Risk: MEDIUM
```
**gitnexus_impact** — blast radius per changed symbol:
**impact** — blast radius per changed symbol:
```
gitnexus_impact({target: "validatePayment", direction: "upstream"})
impact({target: "validatePayment", direction: "upstream"})
→ d=1 (WILL BREAK):
- processCheckout (src/checkout.ts:42) [CALLS, 100%]
@ -86,20 +86,20 @@ gitnexus_impact({target: "validatePayment", direction: "upstream"})
- checkoutRouter (src/routes/checkout.ts:22) [CALLS, 95%]
```
**gitnexus_impact with tests** — check test coverage:
**impact with tests** — check test coverage:
```
gitnexus_impact({target: "validatePayment", direction: "upstream", includeTests: true})
impact({target: "validatePayment", direction: "upstream", includeTests: true})
→ Tests that cover this symbol:
- validatePayment.test.ts [direct]
- checkout.integration.test.ts [via processCheckout]
```
**gitnexus_context** — understand a changed symbol's role:
**context** — understand a changed symbol's role:
```
gitnexus_context({name: "validatePayment"})
context({name: "validatePayment"})
→ Incoming calls: processCheckout, webhookHandler
→ Outgoing calls: verifyCard, fetchRates
@ -112,20 +112,20 @@ gitnexus_context({name: "validatePayment"})
1. gh pr diff 42 > /tmp/pr42.diff
→ 4 files changed: payments.ts, checkout.ts, types.ts, utils.ts
2. gitnexus_detect_changes({scope: "compare", base_ref: "main"})
2. detect_changes({scope: "compare", base_ref: "main"})
→ Changed symbols: validatePayment, PaymentInput, formatAmount
→ Affected processes: CheckoutFlow, RefundFlow
→ Risk: MEDIUM
3. gitnexus_impact({target: "validatePayment", direction: "upstream"})
3. impact({target: "validatePayment", direction: "upstream"})
→ d=1: processCheckout, webhookHandler (WILL BREAK)
→ webhookHandler is NOT in the PR diff — potential breakage!
4. gitnexus_impact({target: "PaymentInput", direction: "upstream"})
4. impact({target: "PaymentInput", direction: "upstream"})
→ d=1: validatePayment (in PR), createPayment (NOT in PR)
→ createPayment uses the old PaymentInput shape — breaking change!
5. gitnexus_context({name: "formatAmount"})
5. context({name: "formatAmount"})
→ Called by 12 functions — but change is backwards-compatible (added optional param)
6. Review summary:

View file

@ -16,9 +16,9 @@ description: "Use when the user wants to rename, extract, split, move, or restru
## Workflow
```
1. gitnexus_impact({target: "X", direction: "upstream"}) → Map all dependents
2. gitnexus_query({query: "X"}) → Find execution flows involving X
3. gitnexus_context({name: "X"}) → See all incoming/outgoing refs
1. impact({target: "X", direction: "upstream"}) → Map all dependents
2. query({query: "X"}) → Find execution flows involving X
3. context({name: "X"}) → See all incoming/outgoing refs
4. Plan update order: interfaces → implementations → callers → tests
```
@ -29,65 +29,65 @@ description: "Use when the user wants to rename, extract, split, move, or restru
### Rename Symbol
```
- [ ] gitnexus_rename({symbol_name: "oldName", new_name: "newName", dry_run: true}) — preview all edits
- [ ] rename({symbol_name: "oldName", new_name: "newName", dry_run: true}) — preview all edits
- [ ] Review graph edits (high confidence) and ast_search edits (review carefully)
- [ ] If satisfied: gitnexus_rename({..., dry_run: false}) — apply edits
- [ ] gitnexus_detect_changes() — verify only expected files changed
- [ ] If satisfied: rename({..., dry_run: false}) — apply edits
- [ ] detect_changes() — verify only expected files changed
- [ ] Run tests for affected processes
```
### Extract Module
```
- [ ] gitnexus_context({name: target}) — see all incoming/outgoing refs
- [ ] gitnexus_impact({target, direction: "upstream"}) — find all external callers
- [ ] context({name: target}) — see all incoming/outgoing refs
- [ ] impact({target, direction: "upstream"}) — find all external callers
- [ ] Define new module interface
- [ ] Extract code, update imports
- [ ] gitnexus_detect_changes() — verify affected scope
- [ ] detect_changes() — verify affected scope
- [ ] Run tests for affected processes
```
### Split Function/Service
```
- [ ] gitnexus_context({name: target}) — understand all callees
- [ ] context({name: target}) — understand all callees
- [ ] Group callees by responsibility
- [ ] gitnexus_impact({target, direction: "upstream"}) — map callers to update
- [ ] impact({target, direction: "upstream"}) — map callers to update
- [ ] Create new functions/services
- [ ] Update callers
- [ ] gitnexus_detect_changes() — verify affected scope
- [ ] detect_changes() — verify affected scope
- [ ] Run tests for affected processes
```
## Tools
**gitnexus_rename** — automated multi-file rename:
**rename** — automated multi-file rename:
```
gitnexus_rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: true})
rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: true})
→ 12 edits across 8 files
→ 10 graph edits (high confidence), 2 ast_search edits (review)
→ Changes: [{file_path, edits: [{line, old_text, new_text, confidence}]}]
```
**gitnexus_impact** — map all dependents first:
**impact** — map all dependents first:
```
gitnexus_impact({target: "validateUser", direction: "upstream"})
impact({target: "validateUser", direction: "upstream"})
→ d=1: loginHandler, apiMiddleware, testUtils
→ Affected Processes: LoginFlow, TokenRefresh
```
**gitnexus_detect_changes** — verify your changes after refactoring:
**detect_changes** — verify your changes after refactoring:
```
gitnexus_detect_changes({scope: "all"})
detect_changes({scope: "all"})
→ Changed: 8 files, 12 symbols
→ Affected processes: LoginFlow, TokenRefresh
→ Risk: MEDIUM
```
**gitnexus_cypher** — custom reference queries:
**cypher** — custom reference queries:
```cypher
MATCH (caller)-[:CodeRelation {type: 'CALLS'}]->(f:Function {name: "validateUser"})
@ -98,24 +98,24 @@ RETURN caller.name, caller.filePath ORDER BY caller.filePath
| Risk Factor | Mitigation |
| ------------------- | ----------------------------------------- |
| Many callers (>5) | Use gitnexus_rename for automated updates |
| Many callers (>5) | Use rename for automated updates |
| Cross-area refs | Use detect_changes after to verify scope |
| String/dynamic refs | gitnexus_query to find them |
| String/dynamic refs | query to find them |
| External/public API | Version and deprecate properly |
## Example: Rename `validateUser` to `authenticateUser`
```
1. gitnexus_rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: true})
1. rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: true})
→ 12 edits: 10 graph (safe), 2 ast_search (review)
→ Files: validator.ts, login.ts, middleware.ts, config.json...
2. Review ast_search edits (config.json: dynamic reference!)
3. gitnexus_rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: false})
3. rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: false})
→ Applied 12 edits across 8 files
4. gitnexus_detect_changes({scope: "all"})
4. detect_changes({scope: "all"})
→ Affected: LoginFlow, TokenRefresh
→ Risk: MEDIUM — run tests for these flows
```

View file

@ -80,18 +80,18 @@ This project is indexed by GitNexus as **GitNexus** (26675 symbols, 35395 relati
## Always Do
- **MUST run impact analysis before editing any symbol.** Before modifying a function, class, or method, run `gitnexus_impact({target: "symbolName", direction: "upstream"})` and report the blast radius (direct callers, affected processes, risk level) to the user.
- **MUST run `gitnexus_detect_changes()` before committing** to verify your changes only affect expected symbols and execution flows.
- **MUST run impact analysis before editing any symbol.** Before modifying a function, class, or method, run `impact({target: "symbolName", direction: "upstream"})` and report the blast radius (direct callers, affected processes, risk level) to the user.
- **MUST run `detect_changes()` before committing** to verify your changes only affect expected symbols and execution flows.
- **MUST warn the user** if impact analysis returns HIGH or CRITICAL risk before proceeding with edits.
- When exploring unfamiliar code, use `gitnexus_query({query: "concept"})` to find execution flows instead of grepping. It returns process-grouped results ranked by relevance.
- When you need full context on a specific symbol — callers, callees, which execution flows it participates in — use `gitnexus_context({name: "symbolName"})`.
- When exploring unfamiliar code, use `query({query: "concept"})` to find execution flows instead of grepping. It returns process-grouped results ranked by relevance.
- When you need full context on a specific symbol — callers, callees, which execution flows it participates in — use `context({name: "symbolName"})`.
## Never Do
- NEVER edit a function, class, or method without first running `gitnexus_impact` on it.
- NEVER edit a function, class, or method without first running `impact` on it.
- NEVER ignore HIGH or CRITICAL risk warnings from impact analysis.
- NEVER rename symbols with find-and-replace — use `gitnexus_rename` which understands the call graph.
- NEVER commit changes without running `gitnexus_detect_changes()` to check affected scope.
- NEVER rename symbols with find-and-replace — use `rename` which understands the call graph.
- NEVER commit changes without running `detect_changes()` to check affected scope.
## Resources

View file

@ -15,7 +15,7 @@ Monorepo: **CLI/MCP** (`gitnexus/`) + **browser UI** (`gitnexus-web/`).
## End-to-end flow: index → graph → tools
1. **Ingestion** — `analyze.ts` → `runFullAnalysis` (`run-analyze.ts`) → `runPipelineFromRepo` (`pipeline.ts`). DAG of 12 phases builds a `KnowledgeGraph` in memory, then loads into LadybugDB under `.gitnexus/`. Repo registered in `~/.gitnexus/registry.json` for MCP discovery.
1. **Ingestion** — `analyze.ts` → `runFullAnalysis` (`run-analyze.ts`) → `runPipelineFromRepo` (`pipeline.ts`). DAG of 14 phases builds a `KnowledgeGraph` in memory, then loads into LadybugDB under `.gitnexus/`. Repo registered in `~/.gitnexus/registry.json` for MCP discovery.
2. **Persistence** — `repo-manager.ts` (paths, registry, KuzuDB cleanup). `lbug-adapter.ts` (graph load, queries, embedding batches).
@ -77,11 +77,11 @@ Monorepo: **CLI/MCP** (`gitnexus/`) + **browser UI** (`gitnexus-web/`).
## Pipeline Phase DAG
12 phases defined in `gitnexus/src/core/ingestion/pipeline-phases/`, each with explicit `deps` and typed output.
14 phases defined in `gitnexus/src/core/ingestion/pipeline-phases/`, each with explicit `deps` and typed output.
```
scan → structure → [markdown, cobol] → parse → [routes, tools, orm]
→ crossFile → mro → communities → processes
→ crossFile → scopeResolution → pruneLocalSymbols → mro → communities → processes
```
| Phase | File | Deps | Output |
@ -95,11 +95,13 @@ scan → structure → [markdown, cobol] → parse → [routes, tools, orm]
| `tools` | `tools.ts` | `parse` | Tool nodes + HANDLES_TOOL edges |
| `orm` | `orm.ts` | `parse` | QUERIES edges (Prisma, Supabase) |
| `crossFile` | `cross-file.ts` + `cross-file-impl.ts` | `parse`, `routes`, `tools`, `orm` | Cross-file type propagation in topological import order |
| `mro` | `mro.ts` | `crossFile`, `structure` | METHOD_OVERRIDES + METHOD_IMPLEMENTS edges |
| `communities` | `communities.ts` | `mro`, `structure` | Community nodes + MEMBER_OF edges (Leiden algorithm) |
| `processes` | `processes.ts` | `communities`, `routes`, `tools`, `structure` | Process nodes + STEP_IN_PROCESS edges |
| `scopeResolution` | `scope-resolution/pipeline/phase.ts` | `parse`, `crossFile`, `structure` | Binding/reference + inheritance edges; disposes BindingAccumulator |
| `pruneLocalSymbols` | `prune-local-symbols.ts` | `scopeResolution` | Drops inert block-local `Const`/`Variable`/`Static` nodes (only a `File→DEFINES` edge) post-resolution |
| `mro` | `mro.ts` | `crossFile`, `scopeResolution`, `pruneLocalSymbols`, `structure` | METHOD_OVERRIDES + METHOD_IMPLEMENTS edges |
| `communities` | `communities.ts` | `mro`, `pruneLocalSymbols`, `structure` | Community nodes + MEMBER_OF edges (Leiden algorithm) |
| `processes` | `processes.ts` | `communities`, `routes`, `tools`, `pruneLocalSymbols`, `structure` | Process nodes + STEP_IN_PROCESS edges |
**Non-phase files in the same directory:** `parse-impl.ts`, `cross-file-impl.ts` (implementation), `wildcard-synthesis.ts` (whole-module import expansion), `orm-extraction.ts` (sequential ORM fallback), `types.ts`, `runner.ts`, `index.ts`.
**Non-phase files in the same directory:** `parse-impl.ts`, `cross-file-impl.ts` (implementation), `wildcard-synthesis.ts` (whole-module import expansion), `types.ts`, `runner.ts`, `index.ts`.
### DAG runner
@ -119,7 +121,8 @@ scan → structure → [markdown, cobol] → parse → [routes, tools, orm]
- **Single graph accumulator** — all phases mutate the same `KnowledgeGraph` in `ctx`; the graph is the primary output.
- **Typed phase access** — `getPhaseOutput<T>(deps, 'name')` for type-safe upstream results.
- **Binding accumulator lifecycle** — created in `parse`, disposed by `crossFile` (in `finally`). No other phase should take ownership.
- **Skippable phases** — `skipGraphPhases` omits MRO/communities/processes (faster tests). `skipWorkers` forces sequential parsing.
- **Skippable phases** — `skipGraphPhases` omits MRO/communities/processes (faster tests); `pruneLocalSymbols` still runs (it is graph cleanup, not analysis). `skipWorkers` is no longer a sequential escape hatch — it (like `--workers 0` / `GITNEXUS_WORKER_POOL_SIZE=0`) is rejected with an actionable error, since the worker pool is the sole parse path (§ Chunked parse-and-resolve).
- **Local-symbol pruning** — `pruneLocalSymbols` removes inert block-local value symbols after scope resolution has consumed them. Opt out per-call with `PipelineOptions.keepLocalValueSymbols` or globally with the `GITNEXUS_KEEP_LOCAL_VALUE_SYMBOLS` env var.
### How to add a new phase
@ -199,7 +202,7 @@ Language-agnostic scope-resolution resolver. This is the resolution path for eve
```
Orchestrator: `runScopeResolution(input, provider)` in `scope-resolution/pipeline/run.ts`.
Pipeline phase: `scopeResolutionPhase` in `scope-resolution/pipeline/phase.ts` — iterates the registered `SCOPE_RESOLVERS`, reads per-file Trees from the parse phase's `scopeTreeCache`, disposes the cache at the end.
Pipeline phase: `scopeResolutionPhase` in `scope-resolution/pipeline/phase.ts` — iterates the registered `SCOPE_RESOLVERS` over the worker-serialized `ParsedFile`s. (Per-language `emitScopeCaptures` hooks may reuse a cached Tree via the orchestrator's `treeCache`, but in worker-pool runs that cache is empty — Trees can't cross MessageChannels — so they consume the pre-extracted `ParsedFile` instead; § Performance notes.)
### `ScopeResolver` contract
@ -248,7 +251,7 @@ CI auto-discovers the set via `tsx`. No workflow edit required.
### Performance notes
- **Cross-phase Tree cache**: parse phase writes Trees into `scopeTreeCache` (separate from the chunk-local `astCache`) ONLY for languages with `emitScopeCaptures`. Scope-resolution reads from it to skip the second parse. Cleared at end of the phase. Workers leave the cache empty — Trees can't cross MessageChannels; cache miss = fresh parse. `PROF_SCOPE_RESOLUTION=1` emits hit/miss counters and a worker-engaged warning.
- **Cross-phase Tree cache**: the orchestrator's `treeCache` (`RunScopeResolutionInput.treeCache`) lets a scope-resolution per-language hook (`emitScopeCaptures`) reuse a tree instead of re-parsing. Workers leave it empty — Trees can't cross MessageChannels — so in normal (worker-pool) runs scope-resolution does NOT rely on it: workers serialize each file's `ParsedFile` (+ capture side-channel) and stream them in, so scope-resolution consumes the pre-extracted artifact rather than re-parsing on the main thread (§ Chunked parse-and-resolve). `PROF_SCOPE_RESOLUTION=1` emits hit/miss counters and a worker-engaged warning.
- **Typed relationship iteration**: heritage + MRO walk only the EXTENDS / IMPLEMENTS / HAS_METHOD edges via `iterRelationshipsByType`, not the full relationship map.
- **Workspace-resolution-index**: O(1) `findOwnedMember` / `findExportedDef` / `classScopeByDefId` built once per run.
- **SCC-ordered cross-file return-type propagation** (PR #1050): `propagateImportedReturnTypes` walks `indexes.sccs` in reverse-topological order (leaves first), so multi-hop alias chains like `models.User → service.user → app.user` collapse to the terminal class in a single linear pass. Within each importer, the source module's `typeBindings` is chain-followed BEFORE mirroring (so we mirror terminal types, not intermediate refs), and the importer's own `typeBindings` is chain-followed AFTER mirroring (so local `const x = importedFn()` resolves before downstream importers run). Cyclic SCCs reach a partial fixpoint within a single pass without iterating to convergence — see the `ts-circular` cross-file-binding fixture which only asserts pipeline-no-throw. PROF output (`PROF_SCOPE_RESOLUTION=1`) splits `finalize` from `propagate` so quadratic regressions in the chain-follow surface independently.
@ -311,7 +314,7 @@ Unified 3-tier algorithm (`model/resolution-context.ts`), per-language `importSe
### Chunked parse-and-resolve
`parse` processes files in ~20 MB byte-budget chunks to bound memory. Per chunk:
1. Worker pool dispatches files (or sequential fallback via `skipWorkers`)
1. Worker pool dispatches files (the sole parse path — there is no sequential fallback; `skipWorkers`, `--workers 0`, and `GITNEXUS_WORKER_POOL_SIZE=0` are rejected with an actionable error)
2. Each worker: detect language → load grammar → run queries → return unified `ParseWorkerResult`
3. Synthesize wildcard bindings (`wildcard-synthesis.ts`)
4. Resolve imports
@ -321,6 +324,8 @@ Inheritance edges are emitted later, by the scope-resolution phase (`preEmitInhe
Workers: `workers/worker-pool.ts`, `workers/parse-worker.ts`.
**Worker-serialized ParsedFiles (#2038).** To index very large repos (e.g. the Linux kernel) without OOM, the worker pool is the *sole* parse path and workers serialize each file's `ParsedFile` (plus its capture side-channel) in parallel, streaming them to scope-resolution through a disk-backed store. Scope-resolution consumes the pre-extracted artifact instead of re-parsing every file on the main thread — tree-sitter's native input buffers are not GC-reclaimable, so the former main-thread re-parse leaked native memory until the process died. Pool creation is lazy / cache-miss-gated, so a warm all-cache-hit run replays cached worker output without spawning a worker (hence `usedWorkerPool` can be false even when the repo has parseable files).
### Inheritance and MRO
Inheritance is captured by the `@reference.inherits` tag and emitted by the scope-resolution phase: `preEmitInheritanceEdges` resolves each base in scope, then `emitHeritageEdges` writes the `EXTENDS`/`IMPLEMENTS` edges. The phase then computes method resolution order via each `ScopeResolver`'s `buildMro` hook, feeding a `MethodDispatchIndex` used for owner-scoped lookups. Per-language strategy:

View file

@ -62,18 +62,18 @@ This project is indexed by GitNexus as **GitNexus** (26675 symbols, 35395 relati
## Always Do
- **MUST run impact analysis before editing any symbol.** Before modifying a function, class, or method, run `gitnexus_impact({target: "symbolName", direction: "upstream"})` and report the blast radius (direct callers, affected processes, risk level) to the user.
- **MUST run `gitnexus_detect_changes()` before committing** to verify your changes only affect expected symbols and execution flows.
- **MUST run impact analysis before editing any symbol.** Before modifying a function, class, or method, run `impact({target: "symbolName", direction: "upstream"})` and report the blast radius (direct callers, affected processes, risk level) to the user.
- **MUST run `detect_changes()` before committing** to verify your changes only affect expected symbols and execution flows.
- **MUST warn the user** if impact analysis returns HIGH or CRITICAL risk before proceeding with edits.
- When exploring unfamiliar code, use `gitnexus_query({query: "concept"})` to find execution flows instead of grepping. It returns process-grouped results ranked by relevance.
- When you need full context on a specific symbol — callers, callees, which execution flows it participates in — use `gitnexus_context({name: "symbolName"})`.
- When exploring unfamiliar code, use `query({query: "concept"})` to find execution flows instead of grepping. It returns process-grouped results ranked by relevance.
- When you need full context on a specific symbol — callers, callees, which execution flows it participates in — use `context({name: "symbolName"})`.
## Never Do
- NEVER edit a function, class, or method without first running `gitnexus_impact` on it.
- NEVER edit a function, class, or method without first running `impact` on it.
- NEVER ignore HIGH or CRITICAL risk warnings from impact analysis.
- NEVER rename symbols with find-and-replace — use `gitnexus_rename` which understands the call graph.
- NEVER commit changes without running `gitnexus_detect_changes()` to check affected scope.
- NEVER rename symbols with find-and-replace — use `rename` which understands the call graph.
- NEVER commit changes without running `detect_changes()` to check affected scope.
## Resources

View file

@ -157,7 +157,12 @@ routes between two modes based on the triggering event:
suffix; RC tags are excluded at trigger via a negative glob). Publishes to
the `latest` dist-tag with a changelog-backed GitHub release. Maintainers
are expected to tag from `main` as a convention; the workflow itself does
not enforce branch reachability. No Docker build (RC-only).
not enforce branch reachability. No Docker build (RC-only). Before cutting a
stable release, keep `gitnexus/package.json`,
`gitnexus-claude-plugin/.claude-plugin/plugin.json`,
`.claude-plugin/marketplace.json`, and the matching `CHANGELOG.md` entry in
lockstep — the always-on `gitnexus` unit suite now fails if those manifest
versions drift.
- **Release-candidate mode** — runs on every push to `main` (typically a
merged PR) plus manual `workflow_dispatch`. Docs-only changes are skipped
via `paths-ignore`. Publishes to the `rc` dist-tag with version

View file

@ -236,7 +236,7 @@ gitnexus analyze --embeddings [limit] # Enable embedding generation (slower, be
gitnexus analyze --verbose # Log skipped files when parsers are unavailable
gitnexus analyze --worker-timeout 60 # Increase worker idle timeout for slow parses
gitnexus analyze --wal-checkpoint-threshold 67108864 # 64 MiB. Control LadybugDB WAL auto-checkpoint threshold (default: 67108864 = 64 MiB; -1 keeps Ladybug stock ~16 MiB)
gitnexus analyze --workers <n> # Parse worker pool size (default: cores-1, capped at 16; 0 = sequential)
gitnexus analyze --workers <n> # Parse worker pool size (>=1; default: cores-1, capped at 16, auto-sized to the repo). 0 is rejected — there is no sequential mode.
gitnexus mcp # Start MCP server (stdio) — serves all indexed repos
gitnexus serve # Start local HTTP server (multi-repo) for web UI connection
gitnexus list # List all indexed repositories
@ -314,7 +314,7 @@ Most `analyze` knobs are also CLI flags (`--workers`, `--worker-timeout`, `--max
| Variable | Default | Effect | Tune when… |
| -------------------------------------- | ------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------- |
| `GITNEXUS_WORKER_POOL_SIZE` | `cores - 1`, capped at 16 | Parse worker pool size. `0` disables the pool (sequential fallback). Equivalent to `--workers <n>`. | Constrained containers (cgroup CPU limits), CI runners with explicit quotas, or debugging a worker-only crash via `0`. |
| `GITNEXUS_WORKER_POOL_SIZE` | `cores - 1`, capped at 16 | Parse worker pool size (must be ≥ 1). Equivalent to `--workers <n>`. The worker pool is the sole parse path — there is no sequential parser, so `0` is rejected with an actionable error (the pool self-heals via quarantine + respawn). | Constrained containers (cgroup CPU limits) or CI runners with explicit quotas. To narrow down a worker crash set `1` for a single-worker pool — not `0`. |
| `GITNEXUS_PARSE_CHUNK_CONCURRENCY` | `2` | Number of chunks whose file contents may be read into memory in parallel while the pool dispatches the current chunk. Worker dispatch itself stays serial. | Repos large enough to chunk (multi-MB total source) where disk I/O is a measurable fraction of analyze wall-clock. |
| `GITNEXUS_VERBOSE` | unset | When `1`, enables verbose ingestion logs (skipped-file warnings, per-chunk throughput, parse-cache stats). Equivalent to `--verbose`. | Debugging an analyze that "completed" but seems to have missed files; tuning `--workers` / chunk concurrency against observable throughput. |
| `GITNEXUS_PROFILE_DEFERRED` | unset | When `1`, emits `[deferred-profile]` timing/progress logs for the post-chunk deferred resolution band (imports → heritage → buildHeritageMap → legacy call resolution). Implied by `GITNEXUS_VERBOSE`. | Diagnosing analyze stalls in "Resolving calls (all chunks)" on large Java/Kotlin repos (issue #1741) without the full verbose ingestion noise. |

View file

@ -1,7 +1,7 @@
{
"name": "gitnexus",
"description": "Code intelligence powered by a knowledge graph. Provides execution flow tracing, blast radius analysis, and augmented search across your codebase.",
"version": "1.3.6",
"version": "1.6.6",
"author": {
"name": "GitNexus"
},

View file

@ -16,10 +16,10 @@ description: "Use when the user is debugging a bug, tracing an error, or asking
## Workflow
```
1. gitnexus_query({query: "<error or symptom>"}) → Find related execution flows
2. gitnexus_context({name: "<suspect>"}) → See callers/callees/processes
1. query({query: "<error or symptom>"}) → Find related execution flows
2. context({name: "<suspect>"}) → See callers/callees/processes
3. READ gitnexus://repo/{name}/process/{name} → Trace execution flow
4. gitnexus_cypher({query: "MATCH path..."}) → Custom traces if needed
4. cypher({query: "MATCH path..."}) → Custom traces if needed
```
> If "Index is stale" → run `node .gitnexus/run.cjs analyze` in terminal.
@ -28,11 +28,11 @@ description: "Use when the user is debugging a bug, tracing an error, or asking
```
- [ ] Understand the symptom (error message, unexpected behavior)
- [ ] gitnexus_query for error text or related code
- [ ] query for error text or related code
- [ ] Identify the suspect function from returned processes
- [ ] gitnexus_context to see callers and callees
- [ ] context to see callers and callees
- [ ] Trace execution flow via process resource if applicable
- [ ] gitnexus_cypher for custom call chain traces if needed
- [ ] cypher for custom call chain traces if needed
- [ ] Read source files to confirm root cause
```
@ -40,7 +40,7 @@ description: "Use when the user is debugging a bug, tracing an error, or asking
| Symptom | GitNexus Approach |
| -------------------- | ---------------------------------------------------------- |
| Error message | `gitnexus_query` for error text → `context` on throw sites |
| Error message | `query` for error text → `context` on throw sites |
| Wrong return value | `context` on the function → trace callees for data flow |
| Intermittent failure | `context` → look for external calls, async deps |
| Performance issue | `context` → find symbols with many callers (hot paths) |
@ -48,24 +48,24 @@ description: "Use when the user is debugging a bug, tracing an error, or asking
## Tools
**gitnexus_query** — find code related to error:
**query** — find code related to error:
```
gitnexus_query({query: "payment validation error"})
query({query: "payment validation error"})
→ Processes: CheckoutFlow, ErrorHandling
→ Symbols: validatePayment, handlePaymentError, PaymentException
```
**gitnexus_context** — full context for a suspect:
**context** — full context for a suspect:
```
gitnexus_context({name: "validatePayment"})
context({name: "validatePayment"})
→ Incoming calls: processCheckout, webhookHandler
→ Outgoing calls: verifyCard, fetchRates (external API!)
→ Processes: CheckoutFlow (step 3/7)
```
**gitnexus_cypher** — custom call chain traces:
**cypher** — custom call chain traces:
```cypher
MATCH path = (a)-[:CodeRelation {type: 'CALLS'}*1..2]->(b:Function {name: "validatePayment"})
@ -75,11 +75,11 @@ RETURN [n IN nodes(path) | n.name] AS chain
## Example: "Payment endpoint returns 500 intermittently"
```
1. gitnexus_query({query: "payment error handling"})
1. query({query: "payment error handling"})
→ Processes: CheckoutFlow, ErrorHandling
→ Symbols: validatePayment, handlePaymentError
2. gitnexus_context({name: "validatePayment"})
2. context({name: "validatePayment"})
→ Outgoing calls: verifyCard, fetchRates (external API!)
3. READ gitnexus://repo/my-app/process/CheckoutFlow

View file

@ -18,8 +18,8 @@ description: "Use when the user asks how code works, wants to understand archite
```
1. READ gitnexus://repos → Discover indexed repos
2. READ gitnexus://repo/{name}/context → Codebase overview, check staleness
3. gitnexus_query({query: "<what you want to understand>"}) → Find related execution flows
4. gitnexus_context({name: "<symbol>"}) → Deep dive on specific symbol
3. query({query: "<what you want to understand>"}) → Find related execution flows
4. context({name: "<symbol>"}) → Deep dive on specific symbol
5. READ gitnexus://repo/{name}/process/{name} → Trace full execution flow
```
@ -29,9 +29,9 @@ description: "Use when the user asks how code works, wants to understand archite
```
- [ ] READ gitnexus://repo/{name}/context
- [ ] gitnexus_query for the concept you want to understand
- [ ] query for the concept you want to understand
- [ ] Review returned processes (execution flows)
- [ ] gitnexus_context on key symbols for callers/callees
- [ ] context on key symbols for callers/callees
- [ ] READ process resource for full execution traces
- [ ] Read source files for implementation details
```
@ -47,18 +47,18 @@ description: "Use when the user asks how code works, wants to understand archite
## Tools
**gitnexus_query** — find execution flows related to a concept:
**query** — find execution flows related to a concept:
```
gitnexus_query({query: "payment processing"})
query({query: "payment processing"})
→ Processes: CheckoutFlow, RefundFlow, WebhookHandler
→ Symbols grouped by flow with file locations
```
**gitnexus_context** — 360-degree view of a symbol:
**context** — 360-degree view of a symbol:
```
gitnexus_context({name: "validateUser"})
context({name: "validateUser"})
→ Incoming calls: loginHandler, apiMiddleware
→ Outgoing calls: checkToken, getUserById
→ Processes: LoginFlow (step 2/5), TokenRefresh (step 1/3)
@ -68,10 +68,10 @@ gitnexus_context({name: "validateUser"})
```
1. READ gitnexus://repo/my-app/context → 918 symbols, 45 processes
2. gitnexus_query({query: "payment processing"})
2. query({query: "payment processing"})
→ CheckoutFlow: processPayment → validateCard → chargeStripe
→ RefundFlow: initiateRefund → calculateRefund → processRefund
3. gitnexus_context({name: "processPayment"})
3. context({name: "processPayment"})
→ Incoming: checkoutHandler, webhookHandler
→ Outgoing: validateCard, chargeStripe, saveTransaction
4. Read src/payments/processor.ts for implementation details

View file

@ -17,9 +17,9 @@ description: "Use when the user wants to know what will break if they change som
## Workflow
```
1. gitnexus_impact({target: "X", direction: "upstream"}) → What depends on this
1. impact({target: "X", direction: "upstream"}) → What depends on this
2. READ gitnexus://repo/{name}/processes → Check affected execution flows
3. gitnexus_detect_changes() → Map current git changes to affected flows
3. detect_changes() → Map current git changes to affected flows
4. Assess risk and report to user
```
@ -28,11 +28,11 @@ description: "Use when the user wants to know what will break if they change som
## Checklist
```
- [ ] gitnexus_impact({target, direction: "upstream"}) to find dependents
- [ ] impact({target, direction: "upstream"}) to find dependents
- [ ] Review d=1 items first (these WILL BREAK)
- [ ] Check high-confidence (>0.8) dependencies
- [ ] READ processes to check affected execution flows
- [ ] gitnexus_detect_changes() for pre-commit check
- [ ] detect_changes() for pre-commit check
- [ ] Assess risk level and report to user
```
@ -55,10 +55,10 @@ description: "Use when the user wants to know what will break if they change som
## Tools
**gitnexus_impact** — the primary tool for symbol blast radius:
**impact** — the primary tool for symbol blast radius:
```
gitnexus_impact({
impact({
target: "validateUser",
direction: "upstream",
minConfidence: 0.8,
@ -73,10 +73,10 @@ gitnexus_impact({
- authRouter (src/routes/auth.ts:22) [CALLS, 95%]
```
**gitnexus_detect_changes** — git-diff based impact analysis:
**detect_changes** — git-diff based impact analysis:
```
gitnexus_detect_changes({scope: "staged"})
detect_changes({scope: "staged"})
→ Changed: 5 symbols in 3 files
→ Affected: LoginFlow, TokenRefresh, APIMiddlewarePipeline
@ -86,7 +86,7 @@ gitnexus_detect_changes({scope: "staged"})
## Example: "What breaks if I change validateUser?"
```
1. gitnexus_impact({target: "validateUser", direction: "upstream"})
1. impact({target: "validateUser", direction: "upstream"})
→ d=1: loginHandler, apiMiddleware (WILL BREAK)
→ d=2: authRouter, sessionManager (LIKELY AFFECTED)

View file

@ -18,10 +18,10 @@ description: "Use when the user wants to review a pull request, understand what
```
1. gh pr diff <number> → Get the raw diff
2. gitnexus_detect_changes({scope: "compare", base_ref: "main"}) → Map diff to affected flows
2. detect_changes({scope: "compare", base_ref: "main"}) → Map diff to affected flows
3. For each changed symbol:
gitnexus_impact({target: "<symbol>", direction: "upstream"}) → Blast radius per change
4. gitnexus_context({name: "<key symbol>"}) → Understand callers/callees
impact({target: "<symbol>", direction: "upstream"}) → Blast radius per change
4. context({name: "<key symbol>"}) → Understand callers/callees
5. READ gitnexus://repo/{name}/processes → Check affected execution flows
6. Summarize findings with risk assessment
```
@ -32,10 +32,10 @@ description: "Use when the user wants to review a pull request, understand what
```
- [ ] Fetch PR diff (gh pr diff or git diff base...head)
- [ ] gitnexus_detect_changes to map changes to affected execution flows
- [ ] gitnexus_impact on each non-trivial changed symbol
- [ ] detect_changes to map changes to affected execution flows
- [ ] impact on each non-trivial changed symbol
- [ ] Review d=1 items (WILL BREAK) — are callers updated?
- [ ] gitnexus_context on key changed symbols to understand full picture
- [ ] context on key changed symbols to understand full picture
- [ ] Check if affected processes have test coverage
- [ ] Assess overall risk level
- [ ] Write review summary with findings
@ -63,20 +63,20 @@ description: "Use when the user wants to review a pull request, understand what
## Tools
**gitnexus_detect_changes** — map PR diff to affected execution flows:
**detect_changes** — map PR diff to affected execution flows:
```
gitnexus_detect_changes({scope: "compare", base_ref: "main"})
detect_changes({scope: "compare", base_ref: "main"})
→ Changed: 8 symbols in 4 files
→ Affected processes: CheckoutFlow, RefundFlow, WebhookHandler
→ Risk: MEDIUM
```
**gitnexus_impact** — blast radius per changed symbol:
**impact** — blast radius per changed symbol:
```
gitnexus_impact({target: "validatePayment", direction: "upstream"})
impact({target: "validatePayment", direction: "upstream"})
→ d=1 (WILL BREAK):
- processCheckout (src/checkout.ts:42) [CALLS, 100%]
@ -86,20 +86,20 @@ gitnexus_impact({target: "validatePayment", direction: "upstream"})
- checkoutRouter (src/routes/checkout.ts:22) [CALLS, 95%]
```
**gitnexus_impact with tests** — check test coverage:
**impact with tests** — check test coverage:
```
gitnexus_impact({target: "validatePayment", direction: "upstream", includeTests: true})
impact({target: "validatePayment", direction: "upstream", includeTests: true})
→ Tests that cover this symbol:
- validatePayment.test.ts [direct]
- checkout.integration.test.ts [via processCheckout]
```
**gitnexus_context** — understand a changed symbol's role:
**context** — understand a changed symbol's role:
```
gitnexus_context({name: "validatePayment"})
context({name: "validatePayment"})
→ Incoming calls: processCheckout, webhookHandler
→ Outgoing calls: verifyCard, fetchRates
@ -112,20 +112,20 @@ gitnexus_context({name: "validatePayment"})
1. gh pr diff 42 > /tmp/pr42.diff
→ 4 files changed: payments.ts, checkout.ts, types.ts, utils.ts
2. gitnexus_detect_changes({scope: "compare", base_ref: "main"})
2. detect_changes({scope: "compare", base_ref: "main"})
→ Changed symbols: validatePayment, PaymentInput, formatAmount
→ Affected processes: CheckoutFlow, RefundFlow
→ Risk: MEDIUM
3. gitnexus_impact({target: "validatePayment", direction: "upstream"})
3. impact({target: "validatePayment", direction: "upstream"})
→ d=1: processCheckout, webhookHandler (WILL BREAK)
→ webhookHandler is NOT in the PR diff — potential breakage!
4. gitnexus_impact({target: "PaymentInput", direction: "upstream"})
4. impact({target: "PaymentInput", direction: "upstream"})
→ d=1: validatePayment (in PR), createPayment (NOT in PR)
→ createPayment uses the old PaymentInput shape — breaking change!
5. gitnexus_context({name: "formatAmount"})
5. context({name: "formatAmount"})
→ Called by 12 functions — but change is backwards-compatible (added optional param)
6. Review summary:

View file

@ -16,9 +16,9 @@ description: "Use when the user wants to rename, extract, split, move, or restru
## Workflow
```
1. gitnexus_impact({target: "X", direction: "upstream"}) → Map all dependents
2. gitnexus_query({query: "X"}) → Find execution flows involving X
3. gitnexus_context({name: "X"}) → See all incoming/outgoing refs
1. impact({target: "X", direction: "upstream"}) → Map all dependents
2. query({query: "X"}) → Find execution flows involving X
3. context({name: "X"}) → See all incoming/outgoing refs
4. Plan update order: interfaces → implementations → callers → tests
```
@ -29,65 +29,65 @@ description: "Use when the user wants to rename, extract, split, move, or restru
### Rename Symbol
```
- [ ] gitnexus_rename({symbol_name: "oldName", new_name: "newName", dry_run: true}) — preview all edits
- [ ] rename({symbol_name: "oldName", new_name: "newName", dry_run: true}) — preview all edits
- [ ] Review graph edits (high confidence) and ast_search edits (review carefully)
- [ ] If satisfied: gitnexus_rename({..., dry_run: false}) — apply edits
- [ ] gitnexus_detect_changes() — verify only expected files changed
- [ ] If satisfied: rename({..., dry_run: false}) — apply edits
- [ ] detect_changes() — verify only expected files changed
- [ ] Run tests for affected processes
```
### Extract Module
```
- [ ] gitnexus_context({name: target}) — see all incoming/outgoing refs
- [ ] gitnexus_impact({target, direction: "upstream"}) — find all external callers
- [ ] context({name: target}) — see all incoming/outgoing refs
- [ ] impact({target, direction: "upstream"}) — find all external callers
- [ ] Define new module interface
- [ ] Extract code, update imports
- [ ] gitnexus_detect_changes() — verify affected scope
- [ ] detect_changes() — verify affected scope
- [ ] Run tests for affected processes
```
### Split Function/Service
```
- [ ] gitnexus_context({name: target}) — understand all callees
- [ ] context({name: target}) — understand all callees
- [ ] Group callees by responsibility
- [ ] gitnexus_impact({target, direction: "upstream"}) — map callers to update
- [ ] impact({target, direction: "upstream"}) — map callers to update
- [ ] Create new functions/services
- [ ] Update callers
- [ ] gitnexus_detect_changes() — verify affected scope
- [ ] detect_changes() — verify affected scope
- [ ] Run tests for affected processes
```
## Tools
**gitnexus_rename** — automated multi-file rename:
**rename** — automated multi-file rename:
```
gitnexus_rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: true})
rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: true})
→ 12 edits across 8 files
→ 10 graph edits (high confidence), 2 ast_search edits (review)
→ Changes: [{file_path, edits: [{line, old_text, new_text, confidence}]}]
```
**gitnexus_impact** — map all dependents first:
**impact** — map all dependents first:
```
gitnexus_impact({target: "validateUser", direction: "upstream"})
impact({target: "validateUser", direction: "upstream"})
→ d=1: loginHandler, apiMiddleware, testUtils
→ Affected Processes: LoginFlow, TokenRefresh
```
**gitnexus_detect_changes** — verify your changes after refactoring:
**detect_changes** — verify your changes after refactoring:
```
gitnexus_detect_changes({scope: "all"})
detect_changes({scope: "all"})
→ Changed: 8 files, 12 symbols
→ Affected processes: LoginFlow, TokenRefresh
→ Risk: MEDIUM
```
**gitnexus_cypher** — custom reference queries:
**cypher** — custom reference queries:
```cypher
MATCH (caller)-[:CodeRelation {type: 'CALLS'}]->(f:Function {name: "validateUser"})
@ -98,24 +98,24 @@ RETURN caller.name, caller.filePath ORDER BY caller.filePath
| Risk Factor | Mitigation |
| ------------------- | ----------------------------------------- |
| Many callers (>5) | Use gitnexus_rename for automated updates |
| Many callers (>5) | Use rename for automated updates |
| Cross-area refs | Use detect_changes after to verify scope |
| String/dynamic refs | gitnexus_query to find them |
| String/dynamic refs | query to find them |
| External/public API | Version and deprecate properly |
## Example: Rename `validateUser` to `authenticateUser`
```
1. gitnexus_rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: true})
1. rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: true})
→ 12 edits: 10 graph (safe), 2 ast_search (review)
→ Files: validator.ts, login.ts, middleware.ts, config.json...
2. Review ast_search edits (config.json: dynamic reference!)
3. gitnexus_rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: false})
3. rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: false})
→ Applied 12 edits across 8 files
4. gitnexus_detect_changes({scope: "all"})
4. detect_changes({scope: "all"})
→ Affected: LoginFlow, TokenRefresh
→ Risk: MEDIUM — run tests for these flows
```

View file

@ -15,10 +15,10 @@ description: Trace bugs through call chains using knowledge graph
## Workflow
```
1. gitnexus_query({query: "<error or symptom>"}) → Find related execution flows
2. gitnexus_context({name: "<suspect>"}) → See callers/callees/processes
1. query({query: "<error or symptom>"}) → Find related execution flows
2. context({name: "<suspect>"}) → See callers/callees/processes
3. READ gitnexus://repo/{name}/process/{name} → Trace execution flow
4. gitnexus_cypher({query: "MATCH path..."}) → Custom traces if needed
4. cypher({query: "MATCH path..."}) → Custom traces if needed
```
> If "Index is stale" → run `node .gitnexus/run.cjs analyze` in terminal.
@ -27,11 +27,11 @@ description: Trace bugs through call chains using knowledge graph
```
- [ ] Understand the symptom (error message, unexpected behavior)
- [ ] gitnexus_query for error text or related code
- [ ] query for error text or related code
- [ ] Identify the suspect function from returned processes
- [ ] gitnexus_context to see callers and callees
- [ ] context to see callers and callees
- [ ] Trace execution flow via process resource if applicable
- [ ] gitnexus_cypher for custom call chain traces if needed
- [ ] cypher for custom call chain traces if needed
- [ ] Read source files to confirm root cause
```
@ -39,7 +39,7 @@ description: Trace bugs through call chains using knowledge graph
| Symptom | GitNexus Approach |
|---------|-------------------|
| Error message | `gitnexus_query` for error text → `context` on throw sites |
| Error message | `query` for error text → `context` on throw sites |
| Wrong return value | `context` on the function → trace callees for data flow |
| Intermittent failure | `context` → look for external calls, async deps |
| Performance issue | `context` → find symbols with many callers (hot paths) |
@ -47,22 +47,22 @@ description: Trace bugs through call chains using knowledge graph
## Tools
**gitnexus_query** — find code related to error:
**query** — find code related to error:
```
gitnexus_query({query: "payment validation error"})
query({query: "payment validation error"})
→ Processes: CheckoutFlow, ErrorHandling
→ Symbols: validatePayment, handlePaymentError, PaymentException
```
**gitnexus_context** — full context for a suspect:
**context** — full context for a suspect:
```
gitnexus_context({name: "validatePayment"})
context({name: "validatePayment"})
→ Incoming calls: processCheckout, webhookHandler
→ Outgoing calls: verifyCard, fetchRates (external API!)
→ Processes: CheckoutFlow (step 3/7)
```
**gitnexus_cypher** — custom call chain traces:
**cypher** — custom call chain traces:
```cypher
MATCH path = (a)-[:CodeRelation {type: 'CALLS'}*1..2]->(b:Function {name: "validatePayment"})
RETURN [n IN nodes(path) | n.name] AS chain
@ -71,11 +71,11 @@ RETURN [n IN nodes(path) | n.name] AS chain
## Example: "Payment endpoint returns 500 intermittently"
```
1. gitnexus_query({query: "payment error handling"})
1. query({query: "payment error handling"})
→ Processes: CheckoutFlow, ErrorHandling
→ Symbols: validatePayment, handlePaymentError
2. gitnexus_context({name: "validatePayment"})
2. context({name: "validatePayment"})
→ Outgoing calls: verifyCard, fetchRates (external API!)
3. READ gitnexus://repo/my-app/process/CheckoutFlow

View file

@ -17,8 +17,8 @@ description: Navigate unfamiliar code using GitNexus knowledge graph
```
1. READ gitnexus://repos → Discover indexed repos
2. READ gitnexus://repo/{name}/context → Codebase overview, check staleness
3. gitnexus_query({query: "<what you want to understand>"}) → Find related execution flows
4. gitnexus_context({name: "<symbol>"}) → Deep dive on specific symbol
3. query({query: "<what you want to understand>"}) → Find related execution flows
4. context({name: "<symbol>"}) → Deep dive on specific symbol
5. READ gitnexus://repo/{name}/process/{name} → Trace full execution flow
```
@ -28,9 +28,9 @@ description: Navigate unfamiliar code using GitNexus knowledge graph
```
- [ ] READ gitnexus://repo/{name}/context
- [ ] gitnexus_query for the concept you want to understand
- [ ] query for the concept you want to understand
- [ ] Review returned processes (execution flows)
- [ ] gitnexus_context on key symbols for callers/callees
- [ ] context on key symbols for callers/callees
- [ ] READ process resource for full execution traces
- [ ] Read source files for implementation details
```
@ -46,16 +46,16 @@ description: Navigate unfamiliar code using GitNexus knowledge graph
## Tools
**gitnexus_query** — find execution flows related to a concept:
**query** — find execution flows related to a concept:
```
gitnexus_query({query: "payment processing"})
query({query: "payment processing"})
→ Processes: CheckoutFlow, RefundFlow, WebhookHandler
→ Symbols grouped by flow with file locations
```
**gitnexus_context** — 360-degree view of a symbol:
**context** — 360-degree view of a symbol:
```
gitnexus_context({name: "validateUser"})
context({name: "validateUser"})
→ Incoming calls: loginHandler, apiMiddleware
→ Outgoing calls: checkToken, getUserById
→ Processes: LoginFlow (step 2/5), TokenRefresh (step 1/3)
@ -65,10 +65,10 @@ gitnexus_context({name: "validateUser"})
```
1. READ gitnexus://repo/my-app/context → 918 symbols, 45 processes
2. gitnexus_query({query: "payment processing"})
2. query({query: "payment processing"})
→ CheckoutFlow: processPayment → validateCard → chargeStripe
→ RefundFlow: initiateRefund → calculateRefund → processRefund
3. gitnexus_context({name: "processPayment"})
3. context({name: "processPayment"})
→ Incoming: checkoutHandler, webhookHandler
→ Outgoing: validateCard, chargeStripe, saveTransaction
4. Read src/payments/processor.ts for implementation details

View file

@ -16,9 +16,9 @@ description: Analyze blast radius before making code changes
## Workflow
```
1. gitnexus_impact({target: "X", direction: "upstream"}) → What depends on this
1. impact({target: "X", direction: "upstream"}) → What depends on this
2. READ gitnexus://repo/{name}/processes → Check affected execution flows
3. gitnexus_detect_changes() → Map current git changes to affected flows
3. detect_changes() → Map current git changes to affected flows
4. Assess risk and report to user
```
@ -27,11 +27,11 @@ description: Analyze blast radius before making code changes
## Checklist
```
- [ ] gitnexus_impact({target, direction: "upstream"}) to find dependents
- [ ] impact({target, direction: "upstream"}) to find dependents
- [ ] Review d=1 items first (these WILL BREAK)
- [ ] Check high-confidence (>0.8) dependencies
- [ ] READ processes to check affected execution flows
- [ ] gitnexus_detect_changes() for pre-commit check
- [ ] detect_changes() for pre-commit check
- [ ] Assess risk level and report to user
```
@ -54,9 +54,9 @@ description: Analyze blast radius before making code changes
## Tools
**gitnexus_impact** — the primary tool for symbol blast radius:
**impact** — the primary tool for symbol blast radius:
```
gitnexus_impact({
impact({
target: "validateUser",
direction: "upstream",
minConfidence: 0.8,
@ -71,9 +71,9 @@ gitnexus_impact({
- authRouter (src/routes/auth.ts:22) [CALLS, 95%]
```
**gitnexus_detect_changes** — git-diff based impact analysis:
**detect_changes** — git-diff based impact analysis:
```
gitnexus_detect_changes({scope: "staged"})
detect_changes({scope: "staged"})
→ Changed: 5 symbols in 3 files
→ Affected: LoginFlow, TokenRefresh, APIMiddlewarePipeline
@ -83,7 +83,7 @@ gitnexus_detect_changes({scope: "staged"})
## Example: "What breaks if I change validateUser?"
```
1. gitnexus_impact({target: "validateUser", direction: "upstream"})
1. impact({target: "validateUser", direction: "upstream"})
→ d=1: loginHandler, apiMiddleware (WILL BREAK)
→ d=2: authRouter, sessionManager (LIKELY AFFECTED)

View file

@ -18,10 +18,10 @@ description: "Use when the user wants to review a pull request, understand what
```
1. gh pr diff <number> → Get the raw diff
2. gitnexus_detect_changes({scope: "compare", base_ref: "main"}) → Map diff to affected flows
2. detect_changes({scope: "compare", base_ref: "main"}) → Map diff to affected flows
3. For each changed symbol:
gitnexus_impact({target: "<symbol>", direction: "upstream"}) → Blast radius per change
4. gitnexus_context({name: "<key symbol>"}) → Understand callers/callees
impact({target: "<symbol>", direction: "upstream"}) → Blast radius per change
4. context({name: "<key symbol>"}) → Understand callers/callees
5. READ gitnexus://repo/{name}/processes → Check affected execution flows
6. Summarize findings with risk assessment
```
@ -32,10 +32,10 @@ description: "Use when the user wants to review a pull request, understand what
```
- [ ] Fetch PR diff (gh pr diff or git diff base...head)
- [ ] gitnexus_detect_changes to map changes to affected execution flows
- [ ] gitnexus_impact on each non-trivial changed symbol
- [ ] detect_changes to map changes to affected execution flows
- [ ] impact on each non-trivial changed symbol
- [ ] Review d=1 items (WILL BREAK) — are callers updated?
- [ ] gitnexus_context on key changed symbols to understand full picture
- [ ] context on key changed symbols to understand full picture
- [ ] Check if affected processes have test coverage
- [ ] Assess overall risk level
- [ ] Write review summary with findings
@ -63,20 +63,20 @@ description: "Use when the user wants to review a pull request, understand what
## Tools
**gitnexus_detect_changes** — map PR diff to affected execution flows:
**detect_changes** — map PR diff to affected execution flows:
```
gitnexus_detect_changes({scope: "compare", base_ref: "main"})
detect_changes({scope: "compare", base_ref: "main"})
→ Changed: 8 symbols in 4 files
→ Affected processes: CheckoutFlow, RefundFlow, WebhookHandler
→ Risk: MEDIUM
```
**gitnexus_impact** — blast radius per changed symbol:
**impact** — blast radius per changed symbol:
```
gitnexus_impact({target: "validatePayment", direction: "upstream"})
impact({target: "validatePayment", direction: "upstream"})
→ d=1 (WILL BREAK):
- processCheckout (src/checkout.ts:42) [CALLS, 100%]
@ -86,20 +86,20 @@ gitnexus_impact({target: "validatePayment", direction: "upstream"})
- checkoutRouter (src/routes/checkout.ts:22) [CALLS, 95%]
```
**gitnexus_impact with tests** — check test coverage:
**impact with tests** — check test coverage:
```
gitnexus_impact({target: "validatePayment", direction: "upstream", includeTests: true})
impact({target: "validatePayment", direction: "upstream", includeTests: true})
→ Tests that cover this symbol:
- validatePayment.test.ts [direct]
- checkout.integration.test.ts [via processCheckout]
```
**gitnexus_context** — understand a changed symbol's role:
**context** — understand a changed symbol's role:
```
gitnexus_context({name: "validatePayment"})
context({name: "validatePayment"})
→ Incoming calls: processCheckout, webhookHandler
→ Outgoing calls: verifyCard, fetchRates
@ -112,20 +112,20 @@ gitnexus_context({name: "validatePayment"})
1. gh pr diff 42 > /tmp/pr42.diff
→ 4 files changed: payments.ts, checkout.ts, types.ts, utils.ts
2. gitnexus_detect_changes({scope: "compare", base_ref: "main"})
2. detect_changes({scope: "compare", base_ref: "main"})
→ Changed symbols: validatePayment, PaymentInput, formatAmount
→ Affected processes: CheckoutFlow, RefundFlow
→ Risk: MEDIUM
3. gitnexus_impact({target: "validatePayment", direction: "upstream"})
3. impact({target: "validatePayment", direction: "upstream"})
→ d=1: processCheckout, webhookHandler (WILL BREAK)
→ webhookHandler is NOT in the PR diff — potential breakage!
4. gitnexus_impact({target: "PaymentInput", direction: "upstream"})
4. impact({target: "PaymentInput", direction: "upstream"})
→ d=1: validatePayment (in PR), createPayment (NOT in PR)
→ createPayment uses the old PaymentInput shape — breaking change!
5. gitnexus_context({name: "formatAmount"})
5. context({name: "formatAmount"})
→ Called by 12 functions — but change is backwards-compatible (added optional param)
6. Review summary:

View file

@ -15,9 +15,9 @@ description: Plan safe refactors using blast radius and dependency mapping
## Workflow
```
1. gitnexus_impact({target: "X", direction: "upstream"}) → Map all dependents
2. gitnexus_query({query: "X"}) → Find execution flows involving X
3. gitnexus_context({name: "X"}) → See all incoming/outgoing refs
1. impact({target: "X", direction: "upstream"}) → Map all dependents
2. query({query: "X"}) → Find execution flows involving X
3. context({name: "X"}) → See all incoming/outgoing refs
4. Plan update order: interfaces → implementations → callers → tests
```
@ -27,60 +27,60 @@ description: Plan safe refactors using blast radius and dependency mapping
### Rename Symbol
```
- [ ] gitnexus_rename({symbol_name: "oldName", new_name: "newName", dry_run: true}) — preview all edits
- [ ] rename({symbol_name: "oldName", new_name: "newName", dry_run: true}) — preview all edits
- [ ] Review graph edits (high confidence) and ast_search edits (review carefully)
- [ ] If satisfied: gitnexus_rename({..., dry_run: false}) — apply edits
- [ ] gitnexus_detect_changes() — verify only expected files changed
- [ ] If satisfied: rename({..., dry_run: false}) — apply edits
- [ ] detect_changes() — verify only expected files changed
- [ ] Run tests for affected processes
```
### Extract Module
```
- [ ] gitnexus_context({name: target}) — see all incoming/outgoing refs
- [ ] gitnexus_impact({target, direction: "upstream"}) — find all external callers
- [ ] context({name: target}) — see all incoming/outgoing refs
- [ ] impact({target, direction: "upstream"}) — find all external callers
- [ ] Define new module interface
- [ ] Extract code, update imports
- [ ] gitnexus_detect_changes() — verify affected scope
- [ ] detect_changes() — verify affected scope
- [ ] Run tests for affected processes
```
### Split Function/Service
```
- [ ] gitnexus_context({name: target}) — understand all callees
- [ ] context({name: target}) — understand all callees
- [ ] Group callees by responsibility
- [ ] gitnexus_impact({target, direction: "upstream"}) — map callers to update
- [ ] impact({target, direction: "upstream"}) — map callers to update
- [ ] Create new functions/services
- [ ] Update callers
- [ ] gitnexus_detect_changes() — verify affected scope
- [ ] detect_changes() — verify affected scope
- [ ] Run tests for affected processes
```
## Tools
**gitnexus_rename** — automated multi-file rename:
**rename** — automated multi-file rename:
```
gitnexus_rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: true})
rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: true})
→ 12 edits across 8 files
→ 10 graph edits (high confidence), 2 ast_search edits (review)
→ Changes: [{file_path, edits: [{line, old_text, new_text, confidence}]}]
```
**gitnexus_impact** — map all dependents first:
**impact** — map all dependents first:
```
gitnexus_impact({target: "validateUser", direction: "upstream"})
impact({target: "validateUser", direction: "upstream"})
→ d=1: loginHandler, apiMiddleware, testUtils
→ Affected Processes: LoginFlow, TokenRefresh
```
**gitnexus_detect_changes** — verify your changes after refactoring:
**detect_changes** — verify your changes after refactoring:
```
gitnexus_detect_changes({scope: "all"})
detect_changes({scope: "all"})
→ Changed: 8 files, 12 symbols
→ Affected processes: LoginFlow, TokenRefresh
→ Risk: MEDIUM
```
**gitnexus_cypher** — custom reference queries:
**cypher** — custom reference queries:
```cypher
MATCH (caller)-[:CodeRelation {type: 'CALLS'}]->(f:Function {name: "validateUser"})
RETURN caller.name, caller.filePath ORDER BY caller.filePath
@ -90,24 +90,24 @@ RETURN caller.name, caller.filePath ORDER BY caller.filePath
| Risk Factor | Mitigation |
|-------------|------------|
| Many callers (>5) | Use gitnexus_rename for automated updates |
| Many callers (>5) | Use rename for automated updates |
| Cross-area refs | Use detect_changes after to verify scope |
| String/dynamic refs | gitnexus_query to find them |
| String/dynamic refs | query to find them |
| External/public API | Version and deprecate properly |
## Example: Rename `validateUser` to `authenticateUser`
```
1. gitnexus_rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: true})
1. rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: true})
→ 12 edits: 10 graph (safe), 2 ast_search (review)
→ Files: validator.ts, login.ts, middleware.ts, config.json...
2. Review ast_search edits (config.json: dynamic reference!)
3. gitnexus_rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: false})
3. rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: false})
→ Applied 12 edits across 8 files
4. gitnexus_detect_changes({scope: "all"})
4. detect_changes({scope: "all"})
→ Affected: LoginFlow, TokenRefresh
→ Risk: MEDIUM — run tests for these flows
```

View file

@ -44,7 +44,10 @@ export type NodeLabel =
| 'Template'
| 'Section'
| 'Route'
| 'Tool';
| 'Tool'
// Taint/PDG substrate (issue #2080). Intra-procedural control-flow node.
// Emitted by no phase yet — M1 (#2081) populates these behind an opt-in.
| 'BasicBlock';
export type NodeProperties = {
name: string;
@ -89,6 +92,8 @@ export type NodeProperties = {
responseKeys?: string[];
errorKeys?: string[];
middleware?: string[];
// BasicBlock (taint/PDG substrate, issue #2080) — reuses filePath/startLine/endLine.
text?: string;
// Extensible
[key: string]: unknown;
};
@ -131,7 +136,28 @@ export type RelationshipType =
* `reason` encodes the event name: `vue-emit: <eventName>`.
* Complements `BINDS_EVENT_HANDLER`; a Cypher query joining on the
* component File node reveals all (emitter, handler) pairs. */
| 'EMITS_EVENT';
| 'EMITS_EVENT'
// ── Taint/PDG substrate (issue #2080) ────────────────────────────────────
// Reserved edge types for the taint-first PDG substrate. No phase emits any
// of these yet; they are populated behind an opt-in by later milestones
// (CFG → M1 #2081, REACHING_DEF → M2 #2082, TAINTED/SANITIZES/TAINT_PATH →
// M3/M4 #2083/#2084). Adding them here keeps the shared schema stable so
// downstream work does not re-ripple the exhaustiveness sites.
/** Control-flow edge between two BasicBlock nodes (intra-procedural CFG). */
| 'CFG'
/** Data-dependence edge: a definition of `variable` reaches a use of it.
* The `variable` name is stored in the relation's existing `reason` column
* (M0/S1 verdict: LadybugDB has no secondary index on relationship
* properties, so a dedicated indexed column would not speed the
* variable-filtered path query). */
| 'REACHING_DEF'
/** A tainted value flows from source toward sink. */
| 'TAINTED'
/** A sanitizer clears taint along a flow. */
| 'SANITIZES'
/** Materialized source→sink taint path. Working name — final name/representation
* is confirmed when M3/M4 emits it; no persisted edge exists before then. */
| 'TAINT_PATH';
export interface GraphNode {
id: string;

View file

@ -183,13 +183,3 @@ export {
stripGitSuffix,
} from './integrations/understand-quickly.js';
export type { UqDispatchPayload } from './integrations/understand-quickly.js';
// Shadow-mode diff + aggregation (RFC §6.3; Ring 2 SHARED #918)
export { diffResolutions } from './scope-resolution/shadow/diff.js';
export type {
ShadowAgreement,
ShadowCallsite,
ShadowDiff,
} from './scope-resolution/shadow/diff.js';
export { aggregateDiffs } from './scope-resolution/shadow/aggregate.js';
export type { LanguageParityRow, ShadowParityReport } from './scope-resolution/shadow/aggregate.js';

View file

@ -40,6 +40,8 @@ export const NODE_TABLES = [
'Module',
'Route',
'Tool',
// Taint/PDG substrate (issue #2080) — inert until M1 (#2081) emits blocks.
'BasicBlock',
] as const;
export type NodeTableName = (typeof NODE_TABLES)[number];
@ -67,6 +69,14 @@ export const REL_TYPES = [
'ENTRY_POINT_OF',
'WRAPS',
'QUERIES',
// Taint/PDG substrate (issue #2080) — reserved edge types, emitted by no
// phase yet (CFG → M1, REACHING_DEF → M2, TAINTED/SANITIZES/TAINT_PATH →
// M3/M4). REACHING_DEF's variable name rides the relation's `reason` column.
'CFG',
'REACHING_DEF',
'TAINTED',
'SANITIZES',
'TAINT_PATH',
] as const;
export type RelType = (typeof REL_TYPES)[number];

View file

@ -8,8 +8,7 @@
*
* Part of RFC #909 Ring 2 SHARED — #913.
*
* Consumed by: #915 (SCC finalize link pass), #923 (shadow harness when
* resolving callsite file → enclosing module).
* Consumed by: #915 (SCC finalize link pass).
*/
import type { ScopeId } from './types.js';

View file

@ -74,4 +74,28 @@ export interface ParsedFile {
*/
readonly localDefs: readonly SymbolDefinition[];
readonly referenceSites: readonly ReferenceSite[];
/**
* Opaque, language-private serialization of capture-time side-channel
* state that a provider's `emitScopeCaptures` populates into module-level
* maps as a SIDE EFFECT (not onto the scopes/defs of this `ParsedFile`).
*
* Such state is computed inside the parse worker (where `emitScopeCaptures`
* runs) and would otherwise be lost across the worker→main MessageChannel
* and the disk store, because scope-resolution reuses the serialized
* `ParsedFile` and SKIPS re-extraction on the main thread (#1983 — the
* whole point is to avoid a main-thread tree-sitter re-parse). Carrying the
* data here lets the main thread repopulate those maps WITHOUT re-parsing.
*
* Shared / ingestion code treats this as opaque (`unknown`) per AGENTS.md
* (no language names in shared code). The producing language fills it via
* the `LanguageProvider.collectCaptureSideChannel` hook (worker side) and
* consumes it via the `ScopeResolver.applyCaptureSideChannel` hook
* (main-thread resolution side). It MUST be plain JSON-serializable data
* (objects / arrays / primitives) so it round-trips through the disk-backed
* `parsedfile-store` (JSON.stringify + interning reviver).
*
* Optional: providers whose `emitScopeCaptures` is pure (no module-level
* side effects — the contract default) leave this undefined.
*/
readonly captureSideChannel?: unknown;
}

View file

@ -57,8 +57,7 @@ export interface RawSignals {
*
* Emission order mirrors the `EvidenceWeights` layout: where-found →
* type-binding → corroborators → arity → degraded. Stable order makes
* the per-signal contributions easy to reason about in tests and in the
* shadow-mode parity dashboard.
* the per-signal contributions easy to reason about in tests.
*/
export function composeEvidence(signals: RawSignals): readonly ResolutionEvidence[] {
const out: ResolutionEvidence[] = [];
@ -141,7 +140,7 @@ export function composeEvidence(signals: RawSignals): readonly ResolutionEvidenc
/**
* Sum evidence weights and clamp to `[0, 1]`. Separate from `composeEvidence`
* so tests and the parity dashboard can inspect the raw evidence list.
* so tests can inspect the raw evidence list.
*/
export function confidenceFromEvidence(evidence: readonly ResolutionEvidence[]): number {
let sum = 0;

View file

@ -1,188 +0,0 @@
/**
* Shadow-mode aggregation — per-language parity %, per-evidence-kind
* breakdown of divergences. Consumed by the parity dashboard (RING2-PKG-5).
*
* Pure functions; no I/O. The harness persists per-run JSON; the dashboard
* reads `.gitnexus/shadow-parity/latest.json` and renders.
*
* Related types — `ShadowAgreement`, `ShadowCallsite`, `ShadowDiff` — are
* defined alongside `diffResolutions` in `./diff.ts` and re-exported
* through the top-level `gitnexus-shared` barrel. Consumers import all
* three from `gitnexus-shared`, not from this module.
*
* Part of RFC #909 Ring 2 SHARED — #918.
*/
import type { SupportedLanguages } from '../../languages.js';
import type { ResolutionEvidence } from '../types.js';
import type { ShadowAgreement, ShadowDiff } from './diff.js';
// ─── Aggregated report shape ────────────────────────────────────────────────
export interface LanguageParityRow {
readonly language: SupportedLanguages;
readonly totalCalls: number;
readonly bothAgree: number;
readonly onlyLegacy: number;
readonly onlyNew: number;
readonly bothDisagree: number;
readonly bothEmpty: number;
/**
* Fraction in [0, 1]. Numerator = `bothAgree`; denominator = "calls where
* at least one side resolved" = `totalCalls - bothEmpty`.
*
* When the denominator is 0 (all calls for this language were
* `both-empty`), returns 0. Callers rendering the dashboard should treat
* a 0 parity alongside `totalCalls === bothEmpty` as "no signal" rather
* than "total disagreement".
*/
readonly parity: number;
/**
* Divergence signals broken down by `ResolutionEvidence.kind`. Sourced
* from `ShadowDiff.evidenceDelta` on non-agreeing rows only — `both-agree`
* and `both-empty` do not contribute.
*/
readonly evidenceBreakdown: ReadonlyMap<ResolutionEvidence['kind'], number>;
}
export interface ShadowParityReport {
readonly generatedAt: string; // ISO 8601
readonly perLanguage: readonly LanguageParityRow[];
readonly overall: Omit<LanguageParityRow, 'language' | 'evidenceBreakdown'>;
}
// ─── Public API ─────────────────────────────────────────────────────────────
/**
* Aggregate a stream of `ShadowDiff` records into a `ShadowParityReport`,
* bucketed by language. Pure function.
*
* - `perLanguage` rows are sorted alphabetically by `SupportedLanguages`
* value for stable JSON output (the dashboard reads
* `.gitnexus/shadow-parity/latest.json` and diffing snapshots is useful).
* - `overall` is the column-wise sum across languages.
* - `generatedAt` is injected via the `now` parameter so tests stay
* deterministic; production callers let it default to `new Date()`.
*/
export function aggregateDiffs(
diffs: readonly { readonly language: SupportedLanguages; readonly diff: ShadowDiff }[],
now: Date = new Date(),
): ShadowParityReport {
const perLanguageMap = new Map<SupportedLanguages, MutableCounts>();
for (const { language, diff } of diffs) {
let counts = perLanguageMap.get(language);
if (!counts) {
counts = makeEmptyCounts();
perLanguageMap.set(language, counts);
}
tallyDiff(counts, diff);
}
const perLanguage: LanguageParityRow[] = Array.from(perLanguageMap.entries())
.map(([language, counts]) => buildRow(language, counts))
.sort((a, b) => a.language.localeCompare(b.language));
const overall = buildOverallRow(perLanguage);
return {
generatedAt: now.toISOString(),
perLanguage,
overall,
};
}
// ─── Internal helpers ───────────────────────────────────────────────────────
interface MutableCounts {
totalCalls: number;
bothAgree: number;
onlyLegacy: number;
onlyNew: number;
bothDisagree: number;
bothEmpty: number;
evidenceBreakdown: Map<ResolutionEvidence['kind'], number>;
}
function makeEmptyCounts(): MutableCounts {
return {
totalCalls: 0,
bothAgree: 0,
onlyLegacy: 0,
onlyNew: 0,
bothDisagree: 0,
bothEmpty: 0,
evidenceBreakdown: new Map(),
};
}
function tallyDiff(counts: MutableCounts, diff: ShadowDiff): void {
counts.totalCalls += 1;
incrementAgreement(counts, diff.agreement);
if (diff.agreement === 'both-agree' || diff.agreement === 'both-empty') return;
for (const ev of diff.evidenceDelta) {
counts.evidenceBreakdown.set(ev.kind, (counts.evidenceBreakdown.get(ev.kind) ?? 0) + 1);
}
}
function incrementAgreement(counts: MutableCounts, agreement: ShadowAgreement): void {
switch (agreement) {
case 'both-agree':
counts.bothAgree += 1;
return;
case 'only-legacy':
counts.onlyLegacy += 1;
return;
case 'only-new':
counts.onlyNew += 1;
return;
case 'both-disagree':
counts.bothDisagree += 1;
return;
case 'both-empty':
counts.bothEmpty += 1;
return;
}
}
function buildRow(language: SupportedLanguages, counts: MutableCounts): LanguageParityRow {
const resolved = counts.totalCalls - counts.bothEmpty;
const parity = resolved > 0 ? counts.bothAgree / resolved : 0;
return {
language,
totalCalls: counts.totalCalls,
bothAgree: counts.bothAgree,
onlyLegacy: counts.onlyLegacy,
onlyNew: counts.onlyNew,
bothDisagree: counts.bothDisagree,
bothEmpty: counts.bothEmpty,
parity,
// Freeze via `new Map` on a sorted-kind copy so downstream consumers
// can't mutate the aggregator's internal state.
evidenceBreakdown: new Map(
Array.from(counts.evidenceBreakdown.entries()).sort(([a], [b]) => a.localeCompare(b)),
),
};
}
function buildOverallRow(
perLanguage: readonly LanguageParityRow[],
): Omit<LanguageParityRow, 'language' | 'evidenceBreakdown'> {
let totalCalls = 0;
let bothAgree = 0;
let onlyLegacy = 0;
let onlyNew = 0;
let bothDisagree = 0;
let bothEmpty = 0;
for (const row of perLanguage) {
totalCalls += row.totalCalls;
bothAgree += row.bothAgree;
onlyLegacy += row.onlyLegacy;
onlyNew += row.onlyNew;
bothDisagree += row.bothDisagree;
bothEmpty += row.bothEmpty;
}
const resolved = totalCalls - bothEmpty;
const parity = resolved > 0 ? bothAgree / resolved : 0;
return { totalCalls, bothAgree, onlyLegacy, onlyNew, bothDisagree, bothEmpty, parity };
}

View file

@ -1,126 +0,0 @@
/**
* Shadow-mode diff logic — RFC §6.3.
*
* Pure comparison logic for shadow mode. Takes two `Resolution[]` (legacy
* DAG result + new scope-based registry result) and produces a structured
* diff record for the parity dashboard.
*
* Consumed by the Ring 2 PKG shadow harness (#923), which dual-runs each
* call through legacy + new paths, diffs results, and persists per-run JSON
* for the parity dashboard.
*
* Part of RFC #909 Ring 2 SHARED — #918.
*/
import type { Resolution, ResolutionEvidence } from '../types.js';
// ─── Diff record shape ──────────────────────────────────────────────────────
export type ShadowAgreement =
| 'both-agree' // top match identical (same DefId)
| 'only-legacy' // legacy resolved; new did not
| 'only-new' // new resolved; legacy did not
| 'both-disagree' // both resolved, but to different targets
| 'both-empty'; // both returned empty
export interface ShadowDiff {
readonly callsite: ShadowCallsite;
readonly legacy: Resolution | null;
readonly newResult: Resolution | null;
readonly agreement: ShadowAgreement;
/**
* Symmetric difference of the two top resolutions' `evidence` arrays,
* keyed on `ResolutionEvidence.kind`.
*
* - For `'both-agree'` and `'both-empty'` agreements, always empty.
* - For `'both-disagree'`, contains evidence kinds present on exactly one
* side (not in both).
* - For `'only-legacy'`, contains all of legacy's top evidence.
* - For `'only-new'`, contains all of new's top evidence.
*/
readonly evidenceDelta: readonly ResolutionEvidence[];
}
export interface ShadowCallsite {
readonly filePath: string;
readonly line: number;
readonly col: number;
readonly calledName: string;
}
// ─── Public API ─────────────────────────────────────────────────────────────
/**
* Compare two `Resolution[]` arrays (top matches at `[0]`) and produce a
* `ShadowDiff`. Pure function.
*
* Agreement rules:
* - both arrays empty → `'both-empty'`, `evidenceDelta: []`
* - legacy empty, new non-empty → `'only-new'`, `evidenceDelta` = new's top evidence
* - legacy non-empty, new empty → `'only-legacy'`, `evidenceDelta` = legacy's top evidence
* - both non-empty, same top `def.nodeId` → `'both-agree'`, `evidenceDelta: []`
* - both non-empty, different top `def.nodeId` → `'both-disagree'`,
* `evidenceDelta` = symmetric difference by `ResolutionEvidence.kind`
* (first occurrence of a kind-only-on-legacy then kind-only-on-new; order
* preserved from input arrays)
*
* Evidence-delta rationale: callers aggregating divergences want to know
* which signal kinds explain a disagreement. Keying on `kind` (not full
* equality over `weight`/`note`) avoids spurious deltas when the same
* signal fires with slightly different calibration weights on each side.
*/
export function diffResolutions(
callsite: ShadowCallsite,
legacy: readonly Resolution[],
newResult: readonly Resolution[],
): ShadowDiff {
const legacyTop: Resolution | null = legacy.length > 0 ? legacy[0] : null;
const newTop: Resolution | null = newResult.length > 0 ? newResult[0] : null;
const agreement: ShadowAgreement = (() => {
if (legacyTop === null && newTop === null) return 'both-empty';
if (legacyTop === null) return 'only-new';
if (newTop === null) return 'only-legacy';
return legacyTop.def.nodeId === newTop.def.nodeId ? 'both-agree' : 'both-disagree';
})();
const evidenceDelta = computeEvidenceDelta(legacyTop, newTop, agreement);
return {
callsite,
legacy: legacyTop,
newResult: newTop,
agreement,
evidenceDelta,
};
}
// ─── Internal helpers ───────────────────────────────────────────────────────
/**
* Symmetric difference of two evidence arrays, keyed on
* `ResolutionEvidence.kind`. Preserves input order: legacy-only signals
* first (in legacy's original order), then new-only signals (in new's order).
*
* For `'both-agree'` / `'both-empty'` the delta is empty by contract. For
* `'only-legacy'` / `'only-new'` one side's evidence is the delta (nothing to
* subtract against).
*/
function computeEvidenceDelta(
legacy: Resolution | null,
newResult: Resolution | null,
agreement: ShadowAgreement,
): readonly ResolutionEvidence[] {
if (agreement === 'both-agree' || agreement === 'both-empty') return [];
if (agreement === 'only-legacy') return legacy!.evidence;
if (agreement === 'only-new') return newResult!.evidence;
// both-disagree: symmetric difference keyed on `kind`
const legacyKinds = new Set(legacy!.evidence.map((e) => e.kind));
const newKinds = new Set(newResult!.evidence.map((e) => e.kind));
const onlyInLegacy = legacy!.evidence.filter((e) => !newKinds.has(e.kind));
const onlyInNew = newResult!.evidence.filter((e) => !legacyKinds.has(e.kind));
return [...onlyInLegacy, ...onlyInNew];
}

View file

@ -38,6 +38,7 @@ export const NODE_COLORS: Record<NodeLabel, string> = {
Template: '#a78bfa', // Violet light - like Type
Route: '#f43f5e', // Rose - like Process
Tool: '#a855f7', // Purple - like Project
BasicBlock: '#475569', // Slate darker - control-flow node (muted, taint/PDG substrate)
};
// Node sizes by type - clear visual hierarchy with dramatic size differences
@ -79,6 +80,7 @@ export const NODE_SIZES: Record<NodeLabel, number> = {
Template: 3, // Like Type
Route: 5, // Like Enum
Tool: 5, // Like Enum
BasicBlock: 2, // Tiny - control-flow node (taint/PDG substrate)
};
// Community color palette for cluster-based coloring

View file

@ -4,6 +4,75 @@ All notable changes to GitNexus will be documented in this file.
## [Unreleased]
### Added
- **Taint/PDG substrate (M0)** — foundational schema + seams for reliable taint analysis on a PDG-expandable substrate (#2080, Epic #2087). Adds the `BasicBlock` node label and `CFG` / `REACHING_DEF` / `TAINTED` / `SANITIZES` / `TAINT_PATH` relationship types to the graph schema (round-trip through the bulk-COPY path), a phase-registry seam (`registerPhase` / `enabledWhen`) generalising the graph-phase opt-in guard, and a per-language source/sink/sanitizer config registry seam. All additive and inert — no phase emits the new nodes/edges yet, and a default `analyze` run is byte-identical to before. De-risking spikes (LadybugDB rel-property indexing, post-dominator feasibility) recorded on the issue.
## [1.6.6] - 2026-06-08
### Added
- **Scope-resolution (RFC #909) migrations completed across the language matrix** — Rust (#1639), JavaScript (#1640), Ruby (#1831), Swift (#937, #1948), Vue SFC (#940, #1950), Dart (#939, #1970), COBOL (#941, #1835, #1842), and Kotlin (#1727, #1746, #1782) now run on the registry-primary path; Java reached 100% scope-resolution parity and joined `MIGRATED_LANGUAGES` (#1805); per-language progress reporting added to the scope-resolution phase (#1813)
- **HTTP route & consumer contract extraction (group mode)** — Spring interface routes attributed to controllers (#1743); named/positional Java Spring route args (#1834); Kotlin Spring HTTP route, consumer, and WebClient long-form extraction (#1849, #1855, #1884); Java HTTP consumer contracts (#1872); OpenFeign `@RequestLine` consumer contracts incl. plain interfaces without `@FeignClient` (#1904, #1917); FastAPI `include_router(prefix=...)` cross-file routes (#1877); indirect call patterns via FastAPI `Depends()` and frontend HTTP consumers (#1852); gRPC consumer FQN derivation from Java imports for client-jar consumers (#1889)
- **C++ overload & template resolution** — operator-call resolution (#1754), template partial ordering (#1885), user-defined conversion ranking (#1829), nullptr/ellipsis pointer conversion ranks (#1708), SFINAE filter (#1623), expanded `type_traits` constraint registry (#1648), structured resolver-suppression outcomes (#1785), function-type ADL entities (#1822), and a parameter-type class sidecar (#1642)
- **Go enhancements** — structural interface implementation inference (#1966) and a `builtInNames` set for the Go language provider (#1886)
- **Self-healing worker pool** — automatic worker replacement plus deferred-resolution observability and verbose progress logging (#1741, #1773, #1947)
- **`.gitnexusrc` config file and `gitnexus analyze --default-branch`** (#243, #1996)
- **CLI / MCP impact ergonomics** — `--uid/--file/--kind` disambiguation flags (#1907, #1914), `limit/offset/summaryOnly` pagination on the impact tool (#1818), and a per-symbol `processes` field on `byDepth` items (#1867)
- **`gitnexus analyze --repair-fts`** — enforces FTS verification with hardened repair safeguards (#1720)
- **Web viewer** — Tree View and Circles View (#1799), GitLab repository URLs (#1565), `GITNEXUS_BACKEND_URL` env var for Docker deployments (#1286), and web + CLI internationalization (#1748)
- **Wiki** — local Claude/Codex providers (#1769), an opencode local provider (#2039), and `gitnexus wiki --lang <lang>` for multilanguage wiki generation (#1613)
- **`detect-changes` git-worktree support** (#1654)
- **DeepSeek V4 API support** (#1594)
- **Devcontainer for the Claude / Codex / Cursor CLIs** (#1875) and antigravity integration setup + hook adapter (#1730)
- **Object-literal methods linked to exported bindings** (#1718)
- **`eval-server --host`** for a user-configured bind IP (#1667)
- **PR reviewer swarm agents** (#1851)
- **tree-sitter node-type/field validation gate** — validates against the grammar and removes dead literal handling (#1937)
### Fixed
- **Parsing-layer coverage gaps closed across the language matrix** (umbrella #1919) — remaining open gaps (#2072) plus Java F35/F38/F41 (#1928, #2045), PHP F53/F54/F55 (#1931, #1989), COBOL F17–F23 (#1925, #1959), Rust F66/F68/F71/F72 (#1934, #1974), Python F57/F58/F61 (#1932, #1964), JS/TS F44/F83/F85/F86/F87 (#1929, #1968), and Ruby F62 (#1933, #1972)
- **Fully-qualified nested-type identity for C++ and Ruby** — distinct nodes for union-, anonymous-namespace-, and same-tail-nested types (#1978, #1981, #2004, #2005); cross-namespace same-tail inheritance bases resolved (#1993, #2005); Ruby same-tail nested mixin modules qualified with `IMPLEMENTS` routed by scope (#1991, #2006); shared codec for `__heritage__`/`__property__` markers (#1994, #2007); graph nodes materialized for scoped class/module/impl declarations (#1975, #1977); generic Rust inherent-impl methods owned through the mod-qualified `Impl` node (#1992, #2003)
- **C# resolution & memory** — global-namespace `typeBindings` O(files²) OOM eliminated (#1871, #1954) and namespace-siblings OOM with worker-path re-parse removed (#1905); qualified/alias constructor names, `:base`/`:this` initializers, and generic type-arg stripping (#2046); primary-base receiver type normalization (#2036); spurious `IMPORTS` edges from ungated `using` resolution stopped (#1881, #1908)
- **C++ dependent-base and member lookup** — resolution across nested/inline namespaces (#1634, #1814), base-specifier qualifier threading (#1815, #1819), call-site types threaded into qualified member lookup (#1632, #1810), variadic pack dependent lookup (#1909), uninitialized multi-declarators (#1965), and typedef-enum / anonymous-struct declarations (#1941)
- **Kotlin type resolution** — smart-cast refinement for `when/is` and `if/is` (#1758, #1774), overload target-id by parameter types (#1761, #1777), cross-file iterable return propagation (#1759, #1775), method-chain fixpoint receiver types (#1760, #1776), virtual dispatch via constructor type override (#1762, #1778), interface default-method dispatch via implements-split MRO (#1763, #1779), and default-parameter arity detection (#2034)
- **Go declarations** — multi-name declaration capture (#2032), fixed-array parameter binding normalization (#1988), and generic composite-literal constructor inference F33 (#1976)
- **Rust / PHP / Vue / Java parsing** — Rust `struct_expression` name pattern split (#2051); PHP import decomposition, namespace-less `.phtml` module scopes, and Blade-template exclusion (#1801, #1790, #1989); Vue JSDoc, dual-script merge, and lang plumbing F89/F90/F92 (#1936, #2050); Java inherited `RequestMapping` prefix deduplication (#2057) and same-module type resolution for duplicate FQNs (#1712)
- **TypeScript** — HOC pattern false positives fixed with `export default` HOC support (#1943) and suffix-index reuse in the scope resolver (#1840)
- **Inheritance on the worker path** — all languages' inheritance migrated to scope-resolution in worker mode (#1951, #1956); centralized heritage supertype matching (#1921, #1922, #1940); `File->Member` `DEFINES` edges skipped for class members (#1949); phantom `Function` defs for array-method callbacks no longer emitted (#1906)
- **MCP** — sibling-clone repo-ID collisions prevented and generated MCP tool names corrected (#2067); orphan processes avoided by handling stdin close/end and the startup race (#2049); duplicate-name repo resolution disambiguated for worktrees (#1753); Windows setup fallback when global `gitnexus` resolves to a non-spawnable shim (#1694)
- **Worker pool** — resilient zero-copy ingestion worker pool prevents analyze hangs on TS-root-scale loads (#1693); cache-hit native workers no longer abort (#1751, #1833); worker-pool docs drift corrected and worker-side stack surfaced on crash (#2068, #2070)
- **LadybugDB** — FTS loaded in the Windows read pool (#2040) and probed-then-loaded on Windows (#1690, #1692); non-ASCII KuzuDB paths resolved on Windows (#1811, #1817); WAL corruption detected in schema init with recovery surfaced (#1647, #1650); WAL checkpoint-threshold control (#1772); init lock skipped for read-only opens (#1783, #1784); `serve` kept stable when sidecars are missing (#1747)
- **Server / API** — `gitnexus serve` startup restored under Express 5 (#1749); `/api/graph`, `/api/search`, `/api/grep` opened read-only (#1686); native read-only enforcement and prepared statements for Cypher query paths (#1655); `eval-server` localhost binding left to the OS (#1722)
- **Embeddings** — local ONNX runtime guarded on macOS Intel before the transformers.js import (#1987)
- **Web agent** — Nexus AI agent system prompt aligned with registered tools (#1984) and the agent stopped cleanly on user Stop (#1820)
- **Group / contracts** — HTTP graph and source contracts unioned (#1709); `httpx` `AsyncClient` alias imports detected (#1687); Node gRPC `loadPackageDefinition` gate no longer matches every member call (#1916); manifest/workspace extraction moved before `closeLbug` (#1802, #1807)
- **Hooks / install** — `gitnexus` resolved on `PATH` via a pure-Node, all-OS scan (#1938, #1980); offline-first extension installs (#1161); actionable error and docs for the `pnpm dlx`/`pnpx` native-load crash (#307, #1967); `onnxruntime-common` declared as a runtime dependency (#2074); vendored grammars materialized to fix Windows EPERM (#1728, #1729)
- **CLI** — missing LadybugDB native binary detected at startup with actionable guidance (#835, #1837); `--no-stats` applied to the keep-marker stats line (#1706, #1765); skipped large-file paths surfaced by default (#1659, #1661); build.js skipped when running outside the monorepo (#1795, #1816); auto-heap raised to 16 GB with tightened cross-platform OOM guidance for UE5-scale repos (#1652)
- **Wiki** — hidden 60s default timeout removed with timeout/retry flag validation and surfaced timeout errors (#1651); budget-aware grouping to prevent context overflow on large repos (#627, #1832)
- **`detect-changes`** — `resolveWorktreeCwd` guarded against overriding a separately-indexed worktree (#1691)
- **Windows reliability** — `windowsHide:true` passed to every `child_process` spawn-family call (#1794)
### Changed
- **Legacy resolution deletion (Ring 4)** — removed the legacy call-resolution DAG + heritage processor (RING4-1, #942, #2023), the legacy resolution-context + tiered-lookup plumbing (RING4-2, #943, #2033), and the shadow-mode parity harness (RING4-3, #944, #2071)
- **CONTRIBUTING** — clarified local development setup (#2024)
- **Tests / CI** — cli-e2e made read-only and eval-server tests hardened under load (#2000, #1786, #1838, #1688); parity shards consolidated and the cross-platform matrix narrowed (#1798); devcontainer smoke build hardened against Docker Hub flakes (#1969); gitleaks stabilized (#2027)
### Performance
- **Linux-kernel-scale analysis overhaul** — worker-pool parse, finalize O(n²), and the scope-resolution memory wall (#1983, #2038)
- **Scope-capture linearized across all languages (O(n²)→O(n))** plus Python import-resolution linearization (#1918), the Go-specific re-walk fix (#1848, #1915), and owner-keyed lookup for Step 2 member resolution (#1657)
- **C++ ADL candidates indexed once instead of per-site rescans** (#1990)
- **Inert local value symbols pruned** during ingestion (#2065)
### Chore / Dependencies
- `@ladybugdb/core` bump in /gitnexus (#2056)
- Routine dependency bumps across /gitnexus, /gitnexus-web, /eval, and GitHub Actions — incl. `hono`, `vitest`, `@vitest/coverage-v8`, `tsx`, `lru-cache`, `express`/`@types/express`, `express-rate-limit`, `qs`, `node-addon-api`, `brace-expansion`, `langchain`, `i18next`, `dompurify`, `lucide-react`, `axios`, `zod`, `@langchain/langgraph`, `@vercel/node`, `langsmith`, `aiohttp`, `idna`, and the `docker/*` / `github/codeql-action` / `release-drafter` / `dependency-review-action` actions (#2056, #2044, #2043, #2042, #2016, #2015, #2013, #2012, #2011, #2010, #2009, #2008, #2018, #2019, #2017, #2020, #1986, #1911, #1864, #1863, #1861, #1860, #1866, #1844, #1845, #1826, #1825, #1824, #1791, #1789, #1768, #1767, #1739, #1740, #1738, #1736, #1735, #1734, #1731, #1713, #1698, #1697, #1696, #1689, #1604, #1552, #1464, #872)
- **Security** — `@vercel/node` upgraded in /gitnexus-web with transitive advisories remediated (#1705)
## [1.6.5] - 2026-05-16
### Added

View file

@ -400,7 +400,7 @@ Values above **32768 KB (32 MB)** are clamped to the tree-sitter parser ceiling;
### Analyze reports a worker timeout
Worker parse timeouts are recoverable. GitNexus retries stalled worker jobs with backoff, splits large jobs to isolate slow files, and falls back to the sequential parser when needed. If a large repository needs more time per worker job, use either:
Worker parse timeouts are recoverable. GitNexus retries stalled worker jobs with backoff, splits large jobs to isolate slow files, and quarantines a file that repeatedly crashes its worker (respawning the slot so the pool keeps going). If a large repository needs more time per worker job, use either:
```bash
# CLI flag, in seconds
@ -423,6 +423,16 @@ Three env vars expose the pool's resilience layers (respawn budget, cumulative-t
| `GITNEXUS_WORKER_MAX_CUMULATIVE_TIMEOUT_MS` | `5 × subBatchTimeoutMs` | Total retry wall-time budget per job before quarantining. Bounds exponentially-growing retry waits. |
| `GITNEXUS_WORKER_CONSECUTIVE_FAILURE_THRESHOLD` | `max(3, poolSize)` | Per-slot consecutive deaths before the pool's circuit breaker trips. After tripping, dispatches require a fresh pool. |
### Graph cleanup tuning
After scope resolution, analyze prunes inert block-local value symbols (a function-local `const`/`let`/`var` that ends up with only its structural `File→DEFINES` edge) to keep the graph focused on cross-symbol relationships. Module/file-scope symbols, class members, and any local with a real edge are always kept.
| Variable | Default | Effect |
| ------------------------------------ | ------- | ------------------------------------------------------------------------------------------------------- |
| `GITNEXUS_KEEP_LOCAL_VALUE_SYMBOLS` | unset | Set to `1`/`true` to keep inert block-local value symbols instead of pruning them. |
Programmatic callers can pass `keepLocalValueSymbols: true` in `PipelineOptions` instead of setting the env var.
## Privacy
- All processing happens locally on your machine

View file

@ -70,10 +70,17 @@ this doc, run it under instrumentation:
```bash
# From the gitnexus/ subdir:
cd gitnexus
# Single-threaded baseline (sequential fallback):
npx vitest run test/integration/parse-impl-large-fixture.test.ts --reporter=verbose
# The worker pool is the sole parse path, so every run needs the dist worker
# (`npm run build`) and a pool size pinned via GITNEXUS_WORKER_POOL_SIZE.
# Worker-pool path (requires built dist/ — pre-built by `npm run build`):
# Single-worker-pool baseline (closest analog to the old single-threaded run —
# sequential parsing was removed, so a 1-worker pool is the floor):
npm run build && \
GITNEXUS_WORKER_POOL_SIZE=1 \
GITNEXUS_VERBOSE=1 \
npx vitest run test/integration/parse-impl-large-fixture.test.ts --reporter=verbose
# Multi-worker path:
npm run build && \
GITNEXUS_WORKER_POOL_SIZE=4 \
GITNEXUS_PARSE_CHUNK_CONCURRENCY=2 \
@ -97,22 +104,26 @@ node --inspect=0 \
## Latest measurement
> _No measurement data has been collected yet — this file is the
> methodology + harness scaffold. The single recorded data point is the
> U6 wall-clock smoke baseline below; the worker-pool rows are
> placeholders for future bench-pass output._
> methodology + harness scaffold. The U6 smoke test confirms the
> worker-pool path stays well within its wall-clock budget, but every
> throughput/heap cell below is a `_TBD_` placeholder for a future
> bench-pass._
The U6 integration test (`gitnexus/test/integration/parse-impl-large-fixture.test.ts`)
was observed completing the synthetic fixture in **~6 seconds** under
the sequential path (`skipWorkers: true`) on the development machine,
well under the 30 s `Promise.race` wall-clock budget. That number is a
smoke baseline only — recorded here for reference, not as a regression
target.
runs the worker pool — the sole parse path now that sequential parsing
has been removed (disabling the pool on a repo with parseable files
raises a hard `WorkerPoolDisabledError`). It completes the synthetic
fixture well within the 30 s `Promise.race` wall-clock budget on the
development machine, but no worker-pool throughput/heap numbers have been
captured yet, so the rows below are all `_TBD_`. (An earlier ~6 s figure
recorded here was measured on the now-removed sequential path; it has
been dropped rather than relabelled as a worker-pool baseline, since the
two paths are not comparable.)
| Path | files/s | wall-clock | peak heap | chunks | quarantined |
| ------------------------------------------ | ------- | -------------------- | --------- | ------ | ----------- |
| Sequential (`skipWorkers: true`, U6 smoke) | _TBD_ | ~6 s _(observation)_ | _TBD_ | 17 | 0 |
| Worker pool, `--workers 4`, concurrency 2 | _TBD_ | _TBD_ | _TBD_ | _TBD_ | 0 |
| Worker pool, `--workers 1`, concurrency 1 | _TBD_ | _TBD_ | _TBD_ | _TBD_ | 0 |
| Path | files/s | wall-clock | peak heap | chunks | quarantined |
| ------------------------------------------------------------------------- | ------- | ---------- | --------- | ------ | ----------- |
| Worker pool, `--workers 1` (`GITNEXUS_WORKER_POOL_SIZE=1`), concurrency 1 | _TBD_ | _TBD_ | _TBD_ | _TBD_ | 0 |
| Worker pool, `--workers 4`, concurrency 2 | _TBD_ | _TBD_ | _TBD_ | _TBD_ | 0 |
**Hardware:** _TBD — record OS, CPU, RAM, Node version, gitnexus SHA at
the time of the bench-pass that populates the table above._

View file

@ -11,16 +11,18 @@
"_note": "Updated for F17-F23 fixes (P2: TIMES guard, ADD GIVING, SQL AS alias). See PR #1959."
},
"c": {
"fingerprint": "0de009bdbfe095f530fa87eb32bce6ab83092c904f26b3c8fe8d8ab587cf6dc9",
"fingerprint": "12a196b2d6249c8d86a931b12ecebc2a0cdf8d6f47683acdd0d8e9d8bc7657f5",
"scaling_budget": 1.5,
"_added": "#1956: c added to the scope-capture bench (was UNBENCHED). C has no inheritance \u2014 flat scale source. Adding it exposed + fixed a pre-existing O(n^2) findNodeAtRange root-walk in c/captures.ts (threaded c.node, byte-identical over c-* fixtures); scaling 3.475 -> 0.96."
"_added": "#1956: c added to the scope-capture bench (was UNBENCHED). C has no inheritance — flat scale source. Adding it exposed + fixed a pre-existing O(n^2) findNodeAtRange root-walk in c/captures.ts (threaded c.node, byte-identical over c-* fixtures); scaling 3.475 -> 0.96.",
"_note": "#1983: + c-static-linkage-worker fixture (caller.c/lib.c/lib.h/local.c — worker-path static-linkage side-channel test). Pure fixture-corpus drift: no c/captures.ts or query change branch-vs-main, existing fixtures' captures byte-identical (c-captures.test.ts 45/45), scaling stays linear (~0.97). The baseline was missed when the fixture landed; regenerated here. fingerprint 0de009b->39f3a83.",
"_rebaselined": "#1919 open-language coverage: new lang-resolution fixtures + intended capture additions (F5/F9 c-cpp, F26/F28/F29 dart, F47/F48/F49/F51/F52 kotlin, F75/F79 swift). Fingerprint-only drift; scaling_ratio ~1.0 (linear, no perf regression)."
},
"cpp": {
"fingerprint": "6d6207ae1df3943c5fae28983e0c294e55225456e7cf39af1d46fda21b6787c4",
"fingerprint": "f56625342f73e182170e2c964d538e316c079fa6e9466a7f076bff2ebcf8aac4",
"scaling_budget": 1.5,
"_added": "#1956: cpp added to the scope-capture bench (was UNBENCHED). Heritage-bearing scale source (: public Base, public Mixin) drives emitCppInheritanceCaptures at scale. Adding it exposed + fixed a pre-existing O(n^2) findNodeAtRange root-walk in cpp/captures.ts (~12 sites, threaded c.node, byte-identical over 263 cpp-* fixtures); scaling 2.30 -> 1.12.",
"_rebaselined": "#1965 / #1923 F4: uninitialized non-leading multi-declarators now emit @declaration.variable captures; cpp-adl-inner-callable-outer-noncallable data::Pair a, b adds the legitimate fixture drift. Linear (~1.06).",
"_note": "#1975: + cpp-out-of-line-class fixture, fixture_count 263->265. #1990: + cpp-adl-ns-plus-hidden-friend-same-name fixture (ADL hidden-friend + namespace-callable merge parity test). Pure fixture-corpus drift \u2014 no scope-extractor change; existing fixtures' captures byte-identical. fixture_count 265->267. #1995: + cpp-union-nested-tail-collision and cpp-anon-ns-tail-collision fixtures \u2014 pure fixture-corpus drift; fixture_count 270->272, fingerprint 538e8be->d63ded6. #1993: + cpp-cross-namespace-same-tail fixture \u2014 pure fixture-corpus drift; fixture_count 272->273, fingerprint d63ded6->6d6207ae."
"_rebaselined": "#1919 open-language coverage: new lang-resolution fixtures + intended capture additions (F5/F9 c-cpp, F26/F28/F29 dart, F47/F48/F49/F51/F52 kotlin, F75/F79 swift). Fingerprint-only drift; scaling_ratio ~1.0 (linear, no perf regression).",
"_note": "#1975: + cpp-out-of-line-class fixture, fixture_count 263->265. #1990: + cpp-adl-ns-plus-hidden-friend-same-name fixture (ADL hidden-friend + namespace-callable merge parity test). Pure fixture-corpus drift — no scope-extractor change; existing fixtures' captures byte-identical. fixture_count 265->267. #1995: + cpp-union-nested-tail-collision and cpp-anon-ns-tail-collision fixtures — pure fixture-corpus drift; fixture_count 270->272, fingerprint 538e8be->d63ded6. #1993: + cpp-cross-namespace-same-tail fixture — pure fixture-corpus drift; fixture_count 272->273, fingerprint d63ded6->6d6207ae. #2077 review follow-up: cpp-member-lattice adds cross-file, qualified-base, nested-template, inherited-using, this-receiver, and non-virtual-override regressions; fixture_count 274->275. Capture scaling remains linear (1.134 < 1.5)."
},
"csharp": {
"_rebaselined": "#1956 synth-widening: + csharp-qualified-base fixture; the synth now walks record_declaration + struct_declaration base_lists and handles alias_qualified_name (matching the #1940 legacy leg), so record/struct heritage now emits. csharp-record-base gains a record inherits capture. (record->record SAME-namespace EXTENDS is a separate registry resolution gap, tracked as follow-up.) Linear (~1.00). (Earlier #1956: heritage-bearing scale source.) | #942: scope-resolution-only cleanup reworded fixture comments; capture byte-positions shift, capture LOGIC unchanged. | #1924 F16: record primary-constructor base bindings now exclude constructor arguments; capture fingerprint changes, scaling remains linear. | #2036 review follow-up: csharp-record-base now exercises primary-constructor base dispatch end to end; +2 capture groups, scaling remains linear.",
@ -31,8 +33,8 @@
"rust": {
"fingerprint": "ac610bbe97666bf285923479dd7b43a2fe4c5354aae8df1bcbafdc04fb220f82",
"scaling_budget": 1.5,
"_rebaselined": "#1956 tri-review U1: rust-qualified-trait fixture (scoped + generic-of-scoped impl trait paths); bareTypeIdentifier now resolves scoped_type_identifier bases by their name: tail (additive, no existing-fixture drift); linear (~1.04). #1975: + rust-scoped-impl fixture (impl a::Inner / b::Inner inherent scoped impls) \u2014 legacy @definition.impl scoped arm + findEnclosingClassInfo inherent-impl scoped target; rust scope-extractor captures byte-identical. | #942: scope-resolution-only cleanup reworded fixture comments; capture byte-positions shift, capture LOGIC unchanged.",
"_note": "PR #1934: F66/F68 let-binding pattern narrowing; F71 union (Struct-labeled, now materialized via legacy @definition.struct + resolvable); F72 macro FULLY WIRED \u2014 @declaration.macro/@reference.macro + MacroRegistry \u2192 USES edges to Macro nodes (never a same-named fn). + rust-macro / rust-union fixtures and merged with origin/main #1975 rust-scoped-impl; fingerprint re-baselined (scaling ~0.99, fixture_count 126). #1992: + rust-nested-tail-collision-generic and rust-generic-impl-same-method-name (F3) fixtures \u2014 pure fixture-corpus drift, no scope-extractor change; fixture_count 127->129, fingerprint 56ffc1c0->b00aea0f."
"_rebaselined": "#1956 tri-review U1: rust-qualified-trait fixture (scoped + generic-of-scoped impl trait paths); bareTypeIdentifier now resolves scoped_type_identifier bases by their name: tail (additive, no existing-fixture drift); linear (~1.04). #1975: + rust-scoped-impl fixture (impl a::Inner / b::Inner inherent scoped impls) — legacy @definition.impl scoped arm + findEnclosingClassInfo inherent-impl scoped target; rust scope-extractor captures byte-identical. | #942: scope-resolution-only cleanup reworded fixture comments; capture byte-positions shift, capture LOGIC unchanged.",
"_note": "PR #1934: F66/F68 let-binding pattern narrowing; F71 union (Struct-labeled, now materialized via legacy @definition.struct + resolvable); F72 macro FULLY WIRED — @declaration.macro/@reference.macro + MacroRegistry → USES edges to Macro nodes (never a same-named fn). + rust-macro / rust-union fixtures and merged with origin/main #1975 rust-scoped-impl; fingerprint re-baselined (scaling ~0.99, fixture_count 126). #1992: + rust-nested-tail-collision-generic and rust-generic-impl-same-method-name (F3) fixtures — pure fixture-corpus drift, no scope-extractor change; fixture_count 127->129, fingerprint 56ffc1c0->b00aea0f."
},
"php": {
"fingerprint": "bc2c27c5ba26d5aea61142a2a99fb772222f5b969205260eb7a71b4c0bd73cdb",
@ -44,18 +46,18 @@
"fingerprint": "b5ea93bb3d0469c3821a8c70f5d5991c6f326e41097c119ad691154301dcc753",
"scaling_budget": 1.5,
"_rebaselined": "#1956 synth-widening: + ruby-qualified-base fixture; synth now reduces a scope_resolution superclass (class C < Mod::Super) to its trailing constant (matching the #1940 legacy leg), at parity. Linear (~1.03). (Earlier #1956: heritage-bearing scale source.) | #942: scope-resolution-only cleanup reworded fixture comments; capture byte-positions shift, capture LOGIC unchanged.",
"_note": "F62: + scope_resolution class/module declaration captures \u2014 fixture count 78\u219281, fingerprint drift expected. #1975: + ruby-tail-collision fixture (Foo::Bar vs Baz::Bar stay distinct nodes) \u2014 pure fixture-corpus drift, scope-extractor captures unchanged; 81\u219282. #1991: + ruby-nested-mixin-tail-collision fixture (85\u219286). Recomputed on the #942 merge (fixture-comment rewording shifts capture byte-positions, capture LOGIC unchanged): bf6b13a -> b5ea93bb."
"_note": "F62: + scope_resolution class/module declaration captures — fixture count 78→81, fingerprint drift expected. #1975: + ruby-tail-collision fixture (Foo::Bar vs Baz::Bar stay distinct nodes) — pure fixture-corpus drift, scope-extractor captures unchanged; 81→82. #1991: + ruby-nested-mixin-tail-collision fixture (85→86). Recomputed on the #942 merge (fixture-comment rewording shifts capture byte-positions, capture LOGIC unchanged): bf6b13a -> b5ea93bb."
},
"swift": {
"fingerprint": "53325c6345161c5a495f997297af5a24fb718fd3e6647040160f8ab2a2c8e4c0",
"fingerprint": "180ac68e780bdf6f9089d53f51cbb9a66aed3e7774631cc3fcbaae5020213998",
"scaling_budget": 1.5,
"_rebaselined": "#1956: swift-qualified-base fixture + heritage-bearing scale source (class: Base, Serviceable \u2014 extends + protocol conformance); linear (~1.03)."
"_rebaselined": "#1919 open-language coverage: new lang-resolution fixtures + intended capture additions (F5/F9 c-cpp, F26/F28/F29 dart, F47/F48/F49/F51/F52 kotlin, F75/F79 swift). Fingerprint-only drift; scaling_ratio ~1.0 (linear, no perf regression)."
},
"dart": {
"fingerprint": "a9e882b537765e8fd0ddfcd33b38b253dd86fc5ddffa6e4bf5a85ed8ee615eaa",
"fingerprint": "94bf2c26e1ba96f4211634aa572c0a989b503e717e75dfc5df04f66c417de80f",
"scaling_budget": 1.5,
"_added": "#939: dart added to the scope-capture bench with the registry-primary migration. Heritage-bearing scale source (Entity extends Base implements Marker) gates the @reference.inherits synth + the postfix-chain reference walk at scale. emitDartScopeCaptures threads tree-sitter captured nodes (no findNodeAtRange root-walk), so it is linear (~1.0).",
"_rebaselined": "#1970 review + tri-review follow-ups: constructor-call retag, cascade calls, built-in suppression, enum scope, #1926 F24/F25, named-ctor dedup (crash fix), container-name binding suppression; heritage file-affinity resolution. Fixtures: member-call-contexts, constructor-body, named-constructor-body, heritage-name-collision, construct-cascade."
"_rebaselined": "#1919 review CF3 fix: extended kotlin-local-property-owner (init/accessor destructuring) + new dart-accessor-owner fixture (getter/setter ownership). Fingerprint-only corpus drift; scaling ~1.0."
},
"java": {
"fingerprint": "9b29cafe32873b4902bda311bd089ffc04efe08f13557b966d29544be514080a",
@ -66,8 +68,8 @@
"typescript": {
"fingerprint": "3f44a4a6892698df2d145c8ff2812c3b318807648983c88aca28fbd694f172f9",
"scaling_budget": 1.5,
"_rebaselined": "#1962: F44 (class scope@), F85 (enum member declarations), F87 (optional_parameter type annotations) add new captures \u2014 fingerprint drift expected.",
"_note": "#1968: F44, F85, F87 \u2014 fingerprint drift expected."
"_rebaselined": "#1962: F44 (class scope@), F85 (enum member declarations), F87 (optional_parameter type annotations) add new captures — fingerprint drift expected.",
"_note": "#1968: F44, F85, F87 — fingerprint drift expected."
},
"javascript": {
"fingerprint": "d72f03c6c502235d2d4b74d66baa5c7d361f040d7a1b72e84acad61210d05ae8",
@ -76,9 +78,9 @@
"_rebaselined": "#1956 synth-widening: + javascript-qualified-base fixture; synthesizeJsInheritanceReferences now handles a member_expression base (class S extends ns.Base -> Base), matching the #1940 legacy leg + the TS terminalTsTypeNameNode property_identifier case, at parity. Linear (~1.05). | #942: scope-resolution-only cleanup reworded fixture comments; capture byte-positions shift, capture LOGIC unchanged."
},
"kotlin": {
"fingerprint": "a16400622892183581b8f5f8fa01f07842d19b8cf49ed021df52bd17009d749f",
"fingerprint": "90aa832978d9744e50058e77a04748390a7e34e36b309f6c1d178eb07280b7ea",
"scaling_budget": 1.5,
"_added": "#1951: bench coverage added (was ungated); scale source heritage-bearing (: Base()); js/kotlin O(n^2) findNodeAtRange-per-match fixed to threaded captured node, now linear.",
"_rebaselined": "#1956 synth-widening: + kotlin-qualified-base fixture; synthesizeKotlinInheritanceReferences now handles the explicit_delegation form (class F : Iface by d -> Iface), matching the #1940 legacy leg, at parity. Linear (~0.87). | #942: scope-resolution-only cleanup reworded fixture comments; capture byte-positions shift, capture LOGIC unchanged. | #1930 F45: default parameters now emit optional-arity metadata; capture fingerprint changes, scaling remains linear."
"_rebaselined": "#1919 review CF3 fix: extended kotlin-local-property-owner (init/accessor destructuring) + new dart-accessor-owner fixture (getter/setter ownership). Fingerprint-only corpus drift; scaling ~1.0."
}
}

View file

@ -1,17 +1,17 @@
{
"name": "gitnexus",
"version": "1.6.5",
"version": "1.6.6",
"lockfileVersion": 3,
"requires": true,
"packages": {
"": {
"name": "gitnexus",
"version": "1.6.5",
"version": "1.6.6",
"hasInstallScript": true,
"license": "PolyForm-Noncommercial-1.0.0",
"dependencies": {
"@huggingface/transformers": "^4.1.0",
"@ladybugdb/core": "^0.16.1",
"@ladybugdb/core": "^0.17.0",
"@modelcontextprotocol/sdk": "^1.0.0",
"@scarf/scarf": "^1.4.0",
"cli-progress": "^3.12.0",
@ -26,8 +26,8 @@
"ignore": "^7.0.5",
"js-yaml": "^4.1.1",
"jsonc-parser": "^3.3.1",
"lru-cache": "^11.0.0",
"mnemonist": "^0.40.3",
"onnxruntime-common": "^1.26.0",
"onnxruntime-node": "^1.24.0",
"pandemonium": "^2.4.0",
"pino": "^10.3.1",
@ -1159,27 +1159,28 @@
}
},
"node_modules/@ladybugdb/core": {
"version": "0.16.1",
"resolved": "https://registry.npmjs.org/@ladybugdb/core/-/core-0.16.1.tgz",
"integrity": "sha512-qwuEcR8CVMKb6tNDaHtq7Ux8hT/XbPC0db+vwutX6JxNAejyx7YomHKPSy9XAKURhYK8mezZe3UN8rf+xpHOjQ==",
"version": "0.17.1",
"resolved": "https://registry.npmjs.org/@ladybugdb/core/-/core-0.17.1.tgz",
"integrity": "sha512-K1bHnQrRy3bxkyrFHlxGqKUyIUS1LsRXKOSt14XGY/msBZHaDat/uBrlHiWpM4/24OtfOq/qwTqcTCXannnEjw==",
"hasInstallScript": true,
"license": "MIT",
"dependencies": {
"apache-arrow": "^21.1.0",
"cmake-js": "^8.0.0",
"node-addon-api": "^6.0.0"
},
"optionalDependencies": {
"@ladybugdb/core-darwin-arm64": "0.16.1",
"@ladybugdb/core-darwin-x64": "0.16.1",
"@ladybugdb/core-linux-arm64": "0.16.1",
"@ladybugdb/core-linux-x64": "0.16.1",
"@ladybugdb/core-win32-x64": "0.16.1"
"@ladybugdb/core-darwin-arm64": "0.17.1",
"@ladybugdb/core-darwin-x64": "0.17.1",
"@ladybugdb/core-linux-arm64": "0.17.1",
"@ladybugdb/core-linux-x64": "0.17.1",
"@ladybugdb/core-win32-x64": "0.17.1"
}
},
"node_modules/@ladybugdb/core-darwin-arm64": {
"version": "0.16.1",
"resolved": "https://registry.npmjs.org/@ladybugdb/core-darwin-arm64/-/core-darwin-arm64-0.16.1.tgz",
"integrity": "sha512-Nl+Cf70rD+HaC9IBHv+oeUwqX9plghXD7PN9tyMzMohRVPvcGEbqWPB6YcdJa8rR7qRqCCbmaNMDen5wg4rY2w==",
"version": "0.17.1",
"resolved": "https://registry.npmjs.org/@ladybugdb/core-darwin-arm64/-/core-darwin-arm64-0.17.1.tgz",
"integrity": "sha512-JG/uzmolEh3wXJ/ME1EaTH5LTDQ9Cs+Q3Czul8pW2eWbWQZghQU3jjM++7ST7Bla5BX/WITqwPqPoC+sL+slfA==",
"cpu": [
"arm64"
],
@ -1190,9 +1191,9 @@
]
},
"node_modules/@ladybugdb/core-darwin-x64": {
"version": "0.16.1",
"resolved": "https://registry.npmjs.org/@ladybugdb/core-darwin-x64/-/core-darwin-x64-0.16.1.tgz",
"integrity": "sha512-4eAjfimAAQRSmDfUUkGrl9OhefxcW1ziA9tl0eljBlGoUseE7dL02+RSqjGohYMcQ+lzuHAq1QWb0XRlMA8YTQ==",
"version": "0.17.1",
"resolved": "https://registry.npmjs.org/@ladybugdb/core-darwin-x64/-/core-darwin-x64-0.17.1.tgz",
"integrity": "sha512-Enjm+/V9/jpKmtzF2PB0muVkgpFUGHEvA7r16eJWxVRA/BeO8VPmngTKy9rf/4Yc6TWexjoHRug04BbTXEmerg==",
"cpu": [
"x64"
],
@ -1203,9 +1204,9 @@
]
},
"node_modules/@ladybugdb/core-linux-arm64": {
"version": "0.16.1",
"resolved": "https://registry.npmjs.org/@ladybugdb/core-linux-arm64/-/core-linux-arm64-0.16.1.tgz",
"integrity": "sha512-zkctksev+hsPFrNxHHdq4lYK5OWdLhWfRdQzjzkgDyaHayHU6yCL2fgD6uPGQ8TRQ6/2DxMErb4p3FzGW85Ubw==",
"version": "0.17.1",
"resolved": "https://registry.npmjs.org/@ladybugdb/core-linux-arm64/-/core-linux-arm64-0.17.1.tgz",
"integrity": "sha512-P+xM9o4I3JAQtXpX19ZuLj9EeO2gppa+IdmAqhpI8tuhyA3/a85Eaxby1fXOjsbrnOAEyFJczUdyoDkhCPSyiw==",
"cpu": [
"arm64"
],
@ -1216,9 +1217,9 @@
]
},
"node_modules/@ladybugdb/core-linux-x64": {
"version": "0.16.1",
"resolved": "https://registry.npmjs.org/@ladybugdb/core-linux-x64/-/core-linux-x64-0.16.1.tgz",
"integrity": "sha512-5rAb9T5vif8WKhHwhobosu2/aiOwJkWb/ViybvUc5GFKunKl8VI6RmZQVeufT9zUzRktUwrxBrxblCxsnamXJw==",
"version": "0.17.1",
"resolved": "https://registry.npmjs.org/@ladybugdb/core-linux-x64/-/core-linux-x64-0.17.1.tgz",
"integrity": "sha512-N2ujE0CrsToBpVBpou1iWwEkK7CgVxucnUNxteySrnDccZwICXFP5BlcFpKE0qq3Eqmqszh4ptR4GuSi6rKPGw==",
"cpu": [
"x64"
],
@ -1229,9 +1230,9 @@
]
},
"node_modules/@ladybugdb/core-win32-x64": {
"version": "0.16.1",
"resolved": "https://registry.npmjs.org/@ladybugdb/core-win32-x64/-/core-win32-x64-0.16.1.tgz",
"integrity": "sha512-ShOUTrIuZKQ63J95tcRJxKf1cvg8yi2FSYx9kMTSercc1FdQZPV+zxUN0myMq3MTWOl7xDxsVMmdp/t80O29UQ==",
"version": "0.17.1",
"resolved": "https://registry.npmjs.org/@ladybugdb/core-win32-x64/-/core-win32-x64-0.17.1.tgz",
"integrity": "sha512-9i3xNfFAMqFRuQG3F1hOCWYGna6eTg8HJ/XYhWVDGkeFJNUV3IdneEiYttF5B2qAtQYUd4sAikScsImrMRw+6g==",
"cpu": [
"x64"
],
@ -1675,6 +1676,15 @@
"dev": true,
"license": "MIT"
},
"node_modules/@swc/helpers": {
"version": "0.5.23",
"resolved": "https://registry.npmjs.org/@swc/helpers/-/helpers-0.5.23.tgz",
"integrity": "sha512-5lSsMOTXURePglDfvuAQUqkGek9Hg2kksOYay2m0+XR++b2NWYL/4sWyuvVBIs8oKnJaxkdi9whaL/sqN13afw==",
"license": "Apache-2.0",
"dependencies": {
"tslib": "^2.8.0"
}
},
"node_modules/@tybys/wasm-util": {
"version": "0.10.2",
"resolved": "https://registry.npmjs.org/@tybys/wasm-util/-/wasm-util-0.10.2.tgz",
@ -1718,6 +1728,18 @@
"@types/node": "*"
}
},
"node_modules/@types/command-line-args": {
"version": "5.2.3",
"resolved": "https://registry.npmjs.org/@types/command-line-args/-/command-line-args-5.2.3.tgz",
"integrity": "sha512-uv0aG6R0Y8WHZLTamZwtfsDLVRnOa+n+n5rEvFWL5Na5gZ8V2Teab/duDPFzIIIhs9qizDpcavCusCLJZu62Kw==",
"license": "MIT"
},
"node_modules/@types/command-line-usage": {
"version": "5.0.4",
"resolved": "https://registry.npmjs.org/@types/command-line-usage/-/command-line-usage-5.0.4.tgz",
"integrity": "sha512-BwR5KP3Es/CSht0xqBcUXS3qCAUVXwpRKsV2+arxeb65atasuXG9LykC9Ab10Cw3s2raH92ZqOeILaQbsB2ACg==",
"license": "MIT"
},
"node_modules/@types/connect": {
"version": "3.4.38",
"resolved": "https://registry.npmjs.org/@types/connect/-/connect-3.4.38.tgz",
@ -2069,12 +2091,56 @@
"url": "https://github.com/chalk/ansi-styles?sponsor=1"
}
},
"node_modules/apache-arrow": {
"version": "21.1.0",
"resolved": "https://registry.npmjs.org/apache-arrow/-/apache-arrow-21.1.0.tgz",
"integrity": "sha512-kQrYLxhC+NTVVZ4CCzGF6L/uPVOzJmD1T3XgbiUnP7oTeVFOFgEUu6IKNwCDkpFoBVqDKQivlX4RUFqqnWFlEA==",
"license": "Apache-2.0",
"dependencies": {
"@swc/helpers": "^0.5.11",
"@types/command-line-args": "^5.2.3",
"@types/command-line-usage": "^5.0.4",
"@types/node": "^24.0.3",
"command-line-args": "^6.0.1",
"command-line-usage": "^7.0.1",
"flatbuffers": "^25.1.24",
"json-bignum": "^0.0.3",
"tslib": "^2.6.2"
},
"bin": {
"arrow2csv": "bin/arrow2csv.js"
}
},
"node_modules/apache-arrow/node_modules/@types/node": {
"version": "24.13.0",
"resolved": "https://registry.npmjs.org/@types/node/-/node-24.13.0.tgz",
"integrity": "sha512-5vtOqGQr4NJKeEzV441FcOi2MeG9UTWq9LqVLGneDdu4vlX17H8kQ2PA2UmNwCUGPVDj4oBjNhS7ReVEIWJJrg==",
"license": "MIT",
"dependencies": {
"undici-types": "~7.18.0"
}
},
"node_modules/apache-arrow/node_modules/undici-types": {
"version": "7.18.2",
"resolved": "https://registry.npmjs.org/undici-types/-/undici-types-7.18.2.tgz",
"integrity": "sha512-AsuCzffGHJybSaRrmr5eHr81mwJU3kjw6M+uprWvCXiNeN9SOGwQ3Jn8jb8m3Z6izVgknn1R0FTCEAP2QrLY/w==",
"license": "MIT"
},
"node_modules/argparse": {
"version": "2.0.1",
"resolved": "https://registry.npmjs.org/argparse/-/argparse-2.0.1.tgz",
"integrity": "sha512-8+9WqebbFzpX9OR+Wa6O29asIogeRMzcGtAINdpMHHyAg10f05aSFVBbcEqGf/PXw1EjAZ+q2/bEBg3DvurK3Q==",
"license": "Python-2.0"
},
"node_modules/array-back": {
"version": "6.2.3",
"resolved": "https://registry.npmjs.org/array-back/-/array-back-6.2.3.tgz",
"integrity": "sha512-SGDvmg6QTYiTxCBkYVmThcoa67uLl35pyzRHdpCGBOcqFy6BtwnphoFPk7LhJshD+Yk1Kt35WGWeZPTgwR4Fhw==",
"license": "MIT",
"engines": {
"node": ">=12.17"
}
},
"node_modules/assertion-error": {
"version": "2.0.1",
"resolved": "https://registry.npmjs.org/assertion-error/-/assertion-error-2.0.1.tgz",
@ -2199,6 +2265,37 @@
"node": ">=18"
}
},
"node_modules/chalk": {
"version": "4.1.2",
"resolved": "https://registry.npmjs.org/chalk/-/chalk-4.1.2.tgz",
"integrity": "sha512-oKnbhFyRIXpUuez8iBMmyEa4nbj4IOQyuhc/wy9kY7/WVPcwIO9VA668Pu8RkO7+0G76SLROeyw9CpQ061i4mA==",
"license": "MIT",
"dependencies": {
"ansi-styles": "^4.1.0",
"supports-color": "^7.1.0"
},
"engines": {
"node": ">=10"
},
"funding": {
"url": "https://github.com/chalk/chalk?sponsor=1"
}
},
"node_modules/chalk-template": {
"version": "0.4.0",
"resolved": "https://registry.npmjs.org/chalk-template/-/chalk-template-0.4.0.tgz",
"integrity": "sha512-/ghrgmhfY8RaSdeo43hNXxpoHAtxdbskUHjPpfqUWGttFgycUhYPGx3YZBCnUCvOa7Doivn1IZec3DEGFoMgLg==",
"license": "MIT",
"dependencies": {
"chalk": "^4.1.2"
},
"engines": {
"node": ">=12"
},
"funding": {
"url": "https://github.com/chalk/chalk-template?sponsor=1"
}
},
"node_modules/chownr": {
"version": "3.0.0",
"resolved": "https://registry.npmjs.org/chownr/-/chownr-3.0.0.tgz",
@ -2281,6 +2378,44 @@
"integrity": "sha512-IfEDxwoWIjkeXL1eXcDiow4UbKjhLdq6/EuSVR9GMN7KVH3r9gQ83e73hsz1Nd1T3ijd5xv1wcWRYO+D6kCI2w==",
"license": "MIT"
},
"node_modules/command-line-args": {
"version": "6.0.2",
"resolved": "https://registry.npmjs.org/command-line-args/-/command-line-args-6.0.2.tgz",
"integrity": "sha512-AIjYVxrV9X752LmPDLbVYv8aMCuHPSLZJXEo2qo/xJfv+NYhaZ4sMSF01rM+gHPaMgvPM0l5D/F+Qx+i2WfSmQ==",
"license": "MIT",
"dependencies": {
"array-back": "^6.2.3",
"find-replace": "^5.0.2",
"lodash.camelcase": "^4.3.0",
"typical": "^7.3.0"
},
"engines": {
"node": ">=12.20"
},
"peerDependencies": {
"@75lb/nature": "latest"
},
"peerDependenciesMeta": {
"@75lb/nature": {
"optional": true
}
}
},
"node_modules/command-line-usage": {
"version": "7.0.4",
"resolved": "https://registry.npmjs.org/command-line-usage/-/command-line-usage-7.0.4.tgz",
"integrity": "sha512-85UdvzTNx/+s5CkSgBm/0hzP80RFHAa7PsfeADE5ezZF3uHz3/Tqj9gIKGT9PTtpycc3Ua64T0oVulGfKxzfqg==",
"license": "MIT",
"dependencies": {
"array-back": "^6.2.2",
"chalk-template": "^0.4.0",
"table-layout": "^4.1.1",
"typical": "^7.3.0"
},
"engines": {
"node": ">=12.20.0"
}
},
"node_modules/commander": {
"version": "14.0.3",
"resolved": "https://registry.npmjs.org/commander/-/commander-14.0.3.tgz",
@ -2819,6 +2954,23 @@
"url": "https://opencollective.com/express"
}
},
"node_modules/find-replace": {
"version": "5.0.2",
"resolved": "https://registry.npmjs.org/find-replace/-/find-replace-5.0.2.tgz",
"integrity": "sha512-Y45BAiE3mz2QsrN2fb5QEtO4qb44NcS7en/0y9PEVsg351HsLeVclP8QPMH79Le9sH3rs5RSwJu99W0WPZO43Q==",
"license": "MIT",
"engines": {
"node": ">=14"
},
"peerDependencies": {
"@75lb/nature": "latest"
},
"peerDependenciesMeta": {
"@75lb/nature": {
"optional": true
}
}
},
"node_modules/flatbuffers": {
"version": "25.9.23",
"resolved": "https://registry.npmjs.org/flatbuffers/-/flatbuffers-25.9.23.tgz",
@ -3057,7 +3209,6 @@
"version": "4.0.0",
"resolved": "https://registry.npmjs.org/has-flag/-/has-flag-4.0.0.tgz",
"integrity": "sha512-EykJT/Q1KjTWctppgIAgfSO0tKVuZUjhgMr17kqTumMl6Afv3EISleU7qZUzoXDFTAHTDC4NOoG/ZxU3EvlMPQ==",
"dev": true,
"license": "MIT",
"engines": {
"node": ">=8"
@ -3296,6 +3447,14 @@
"js-yaml": "bin/js-yaml.js"
}
},
"node_modules/json-bignum": {
"version": "0.0.3",
"resolved": "https://registry.npmjs.org/json-bignum/-/json-bignum-0.0.3.tgz",
"integrity": "sha512-2WHyXj3OfHSgNyuzDbSxI1w2jgw5gkWSWhS7Qg4bWXx1nLk3jnbwfUeS0PSba3IzpTUWdHxBieELUzXRjQB2zg==",
"engines": {
"node": ">=0.8"
}
},
"node_modules/json-schema-traverse": {
"version": "1.0.0",
"resolved": "https://registry.npmjs.org/json-schema-traverse/-/json-schema-traverse-1.0.0.tgz",
@ -3587,6 +3746,12 @@
"url": "https://opencollective.com/parcel"
}
},
"node_modules/lodash.camelcase": {
"version": "4.3.0",
"resolved": "https://registry.npmjs.org/lodash.camelcase/-/lodash.camelcase-4.3.0.tgz",
"integrity": "sha512-TwuEnCnxbc3rAvhf/LbG7tJUDzhqXyFnv3dtzLOPgCG/hODL7WFnsbwktkD7yUV0RrreP/l1PALq/YSg6VvjlA==",
"license": "MIT"
},
"node_modules/long": {
"version": "5.3.2",
"resolved": "https://registry.npmjs.org/long/-/long-5.3.2.tgz",
@ -4693,7 +4858,6 @@
"version": "7.2.0",
"resolved": "https://registry.npmjs.org/supports-color/-/supports-color-7.2.0.tgz",
"integrity": "sha512-qpCAvRl9stuOHveKsn7HncJRvv501qIacKzQlO/+Lwxc9+0q2wLyv4Dfvt80/DPn2pqOBsJdDiogXGR9+OvwRw==",
"dev": true,
"license": "MIT",
"dependencies": {
"has-flag": "^4.0.0"
@ -4702,6 +4866,19 @@
"node": ">=8"
}
},
"node_modules/table-layout": {
"version": "4.1.1",
"resolved": "https://registry.npmjs.org/table-layout/-/table-layout-4.1.1.tgz",
"integrity": "sha512-iK5/YhZxq5GO5z8wb0bY1317uDF3Zjpha0QFFLA8/trAoiLbQD0HUbMesEaxyzUgDxi2QlcbM8IvqOlEjgoXBA==",
"license": "MIT",
"dependencies": {
"array-back": "^6.2.2",
"wordwrapjs": "^5.1.0"
},
"engines": {
"node": ">=12.17"
}
},
"node_modules/tar": {
"version": "7.5.13",
"resolved": "https://registry.npmjs.org/tar/-/tar-7.5.13.tgz",
@ -5035,8 +5212,7 @@
"version": "2.8.1",
"resolved": "https://registry.npmjs.org/tslib/-/tslib-2.8.1.tgz",
"integrity": "sha512-oJFu94HQb+KVduSUQL7wnpmqnfmLsOA/nAh6b6EH0wCEoK0/mPeXU6c3wKDV83MkOuHPRHtSXKKU99IBazS/2w==",
"license": "0BSD",
"optional": true
"license": "0BSD"
},
"node_modules/tsx": {
"version": "4.22.4",
@ -5114,6 +5290,15 @@
"node": ">=14.17"
}
},
"node_modules/typical": {
"version": "7.3.0",
"resolved": "https://registry.npmjs.org/typical/-/typical-7.3.0.tgz",
"integrity": "sha512-ya4mg/30vm+DOWfBg4YK3j2WD6TWtRkCbasOJr40CseYENzCUby/7rIvXA99JGsQHeNxLbnXdyLLxKSv3tauFw==",
"license": "MIT",
"engines": {
"node": ">=12.17"
}
},
"node_modules/undici-types": {
"version": "7.24.6",
"resolved": "https://registry.npmjs.org/undici-types/-/undici-types-7.24.6.tgz",
@ -5366,6 +5551,15 @@
"node": ">=8"
}
},
"node_modules/wordwrapjs": {
"version": "5.1.1",
"resolved": "https://registry.npmjs.org/wordwrapjs/-/wordwrapjs-5.1.1.tgz",
"integrity": "sha512-0yweIbkINJodk27gX9LBGMzyQdBDan3s/dEAiwBOj+Mf0PPyWL6/rikalkv8EeD0E8jm4o5RXEOrFTP3NXbhJg==",
"license": "MIT",
"engines": {
"node": ">=12.17"
}
},
"node_modules/wrap-ansi": {
"version": "7.0.0",
"resolved": "https://registry.npmjs.org/wrap-ansi/-/wrap-ansi-7.0.0.tgz",

View file

@ -1,6 +1,6 @@
{
"name": "gitnexus",
"version": "1.6.5",
"version": "1.6.6",
"description": "Graph-powered code intelligence for AI agents. Index any codebase, query via MCP or CLI.",
"author": "Abhigyan Patwari",
"license": "PolyForm-Noncommercial-1.0.0",
@ -55,7 +55,7 @@
},
"dependencies": {
"@huggingface/transformers": "^4.1.0",
"@ladybugdb/core": "^0.16.1",
"@ladybugdb/core": "^0.17.0",
"@modelcontextprotocol/sdk": "^1.0.0",
"@scarf/scarf": "^1.4.0",
"cli-progress": "^3.12.0",
@ -70,8 +70,8 @@
"ignore": "^7.0.5",
"js-yaml": "^4.1.1",
"jsonc-parser": "^3.3.1",
"lru-cache": "^11.0.0",
"mnemonist": "^0.40.3",
"onnxruntime-common": "^1.26.0",
"onnxruntime-node": "^1.24.0",
"pandemonium": "^2.4.0",
"pino": "^10.3.1",

View file

@ -0,0 +1,151 @@
/**
* Spike S1 (issue #2080, M0) — THROWAWAY benchmark. Not part of the build
* (scripts/ is excluded from tsconfig) or the test suite.
*
* Question: can LadybugDB serve the headline REACHING_DEF query
* [:REACHING_DEF*1..5 {variable}]
* fast enough, and what is the right storage shape for the `variable`?
*
* What it does:
* 1. Builds a synthetic ~100K-edge graph of BasicBlock nodes + REACHING_DEF
* edges (variable carried in the CodeRelation `reason` column) with a
* realistic per-variable fan-out distribution, and loads it through the
* real bulk-COPY path (loadGraphToLbug).
* 2. Probes whether LadybugDB supports a secondary index on a relationship
* property (the crux of the "edge property vs side table" decision).
* 3. Times the variable-filtered bounded var-length path query.
*
* Run: npx tsx scripts/spikes/s1-reaching-def-index-bench.ts [edgeCount]
*/
import fs from 'fs/promises';
import path from 'path';
import os from 'os';
import { performance } from 'node:perf_hooks';
import { createKnowledgeGraph } from '../../src/core/graph/graph.js';
import type { KnowledgeGraph } from '../../src/core/graph/types.js';
const EDGE_COUNT = Number(process.argv[2] ?? 30_000);
// Realistic-ish def-use shape: many short chains, variables reused across them.
const CHAIN_LEN = 6; // blocks per function-ish chain
const DISTINCT_VARS = Math.max(1, Math.floor(EDGE_COUNT / 20)); // ~20 edges/variable fan-out
const log = (m: string) => process.stdout.write(m + '\n');
function buildSynthGraph(edgeCount: number): KnowledgeGraph {
const g = createKnowledgeGraph();
let edges = 0;
let chain = 0;
while (edges < edgeCount) {
const base = `BasicBlock:synth/f${chain}.ts`;
for (let i = 0; i <= CHAIN_LEN; i++) {
g.addNode({
id: `${base}:${i}`,
label: 'BasicBlock',
properties: {
name: '',
filePath: `synth/f${chain}.ts`,
startLine: i,
endLine: i,
text: '',
},
});
}
for (let i = 0; i < CHAIN_LEN && edges < edgeCount; i++) {
const variable = `v${edges % DISTINCT_VARS}`;
g.addRelationship({
id: `${base}:${i}->${i + 1}:${variable}`,
sourceId: `${base}:${i}`,
targetId: `${base}:${i + 1}`,
type: 'REACHING_DEF',
confidence: 1.0,
reason: variable, // M0 storage: variable rides `reason`
});
edges++;
}
chain++;
}
return g;
}
async function main() {
const tmp = path.join(os.tmpdir(), `s1-spike-${Date.now()}`);
const storagePath = path.join(tmp, '.gitnexus');
const dbPath = path.join(storagePath, 'lbug');
await fs.mkdir(dbPath, { recursive: true });
const adapter = await import('../../src/core/lbug/lbug-adapter.js');
await adapter.initLbug(dbPath);
log(
`[S1] building synthetic graph: ~${EDGE_COUNT} REACHING_DEF edges, ` +
`${DISTINCT_VARS} distinct variables (~20 edges/var fan-out), chains of ${CHAIN_LEN}`,
);
const g = buildSynthGraph(EDGE_COUNT);
let t = performance.now();
await adapter.loadGraphToLbug(g, tmp, storagePath);
const loadMs = performance.now() - t;
const stats = await adapter.getLbugStats();
log(`[S1] bulk-COPY load: ${loadMs.toFixed(0)}ms (nodes=${stats.nodes}, edges=${stats.edges})`);
// (2) Probe: does LadybugDB support a secondary index on a REL property?
let relIndexSupported = false;
let relIndexErr = '';
for (const stmt of [
"CALL CREATE_REL_INDEX('CodeRelation', 'cr_reason_idx', 'reason')",
'CREATE INDEX cr_reason_idx ON CodeRelation(reason)',
]) {
try {
await adapter.executeQuery(stmt);
relIndexSupported = true;
break;
} catch (e: any) {
relIndexErr = String(e?.message ?? e).split('\n')[0];
}
}
log(
`[S1] rel-property secondary index supported? ${relIndexSupported} ` +
`(last error: ${relIndexErr})`,
);
// (3a) Single-hop variable filter — the common case M3 runs most.
const probeVar = 'v0';
t = performance.now();
const single = await adapter.executeQuery(
`MATCH (a:BasicBlock)-[r:CodeRelation {type: 'REACHING_DEF', reason: '${probeVar}'}]->(b:BasicBlock)
RETURN count(r) AS c`,
);
const singleMs = performance.now() - t;
log(`[S1] single-hop variable filter → ${single[0]?.c} edges in ${singleMs.toFixed(0)}ms`);
// (3b) SOURCE-ANCHORED bounded var-length path — the realistic taint query
// (anchor the source block, then walk REACHING_DEF up to 5 hops). The
// UNANCHORED global form ([:REACHING_DEF*1..5] from every block) is
// impractical at scale (path explosion) — that is itself an S1 finding:
// taint queries MUST be scoped to a source block, not run graph-wide.
const srcId = 'BasicBlock:synth/f0.ts:0';
t = performance.now();
const anchored = await adapter.executeQuery(
`MATCH p = (a:BasicBlock)-[:CodeRelation*1..5 {type: 'REACHING_DEF'}]->(b:BasicBlock)
WHERE a.id = '${srcId}' AND all(rel IN relationships(p) WHERE rel.reason = '${probeVar}')
RETURN count(p) AS paths`,
);
const pathMs = performance.now() - t;
log(
`[S1] source-anchored [:REACHING_DEF*1..5 {reason='${probeVar}'}] from one block → ` +
`${anchored[0]?.paths} paths in ${pathMs.toFixed(0)}ms`,
);
await adapter.closeLbug();
await fs.rm(tmp, { recursive: true, force: true });
log('\n[S1] VERDICT INPUTS:');
log(
` load_ms=${loadMs.toFixed(0)} single_hop_ms=${singleMs.toFixed(0)} anchored_path_ms=${pathMs.toFixed(0)} rel_index=${relIndexSupported}`,
);
}
main().catch((e) => {
console.error('[S1] FAILED:', e);
process.exit(1);
});

View file

@ -0,0 +1,162 @@
/**
* Spike S2 (issue #2080, M0) — THROWAWAY post-dominator feasibility prototype.
* Not part of the build (scripts/ excluded from tsconfig) or the test suite.
*
* Question (per maintainer review): does the post-dominator algorithm Epic B
* (#2085, CDG) depends on hold up on real TS/JS control-flow shapes — the
* classic CFG hazards — before Epic B commits to it?
*
* Scope boundary: post-dominators operate on a CFG, not on the AST directly.
* This prototype validates the ALGORITHM (iterative dataflow on the reverse
* CFG, EXIT-rooted, → immediate-post-dominator tree) against CFGs that model
* each hazard's real TS control flow (the TS source each CFG represents is
* shown inline). Building the CFG from a tree-sitter AST is M1's job (#2081);
* this spike deliberately does not reimplement it.
*
* Run: npx tsx scripts/spikes/s2-postdom-prototype.ts
*/
type CFG = {
name: string;
tsSource: string;
entry: string;
exit: string;
// adjacency: block -> successors
succ: Record<string, string[]>;
hazard: string;
};
// Iterative post-dominator dataflow on the reverse CFG.
// PostDom(EXIT) = {EXIT}; PostDom(n) = {n} ∪ (⋂ PostDom(s) for s ∈ succ(n)).
// Monotone over a finite lattice (powerset of blocks) ⇒ guaranteed to converge.
function postDominators(cfg: CFG): { pdom: Record<string, Set<string>>; iterations: number } {
const blocks = Object.keys(cfg.succ);
const all = new Set(blocks);
const pdom: Record<string, Set<string>> = {};
for (const b of blocks) pdom[b] = b === cfg.exit ? new Set([cfg.exit]) : new Set(all);
let changed = true;
let iterations = 0;
while (changed) {
changed = false;
iterations++;
for (const b of blocks) {
if (b === cfg.exit) continue;
const succs = cfg.succ[b] ?? [];
let inter: Set<string> | null = null;
for (const s of succs) {
if (inter === null) inter = new Set(pdom[s]);
else inter = new Set([...inter].filter((x) => pdom[s].has(x)));
}
const next = new Set<string>(inter ?? []);
next.add(b);
if (next.size !== pdom[b].size || [...next].some((x) => !pdom[b].has(x))) {
pdom[b] = next;
changed = true;
}
}
if (iterations > blocks.length + 5)
throw new Error('post-dom did not converge (suspected bug)');
}
return { pdom, iterations };
}
// Immediate post-dominator: the closest strict post-dominator.
function ipdom(cfg: CFG, pdom: Record<string, Set<string>>): Record<string, string | null> {
const res: Record<string, string | null> = {};
for (const b of Object.keys(cfg.succ)) {
if (b === cfg.exit) {
res[b] = null;
continue;
}
const strict = [...pdom[b]].filter((x) => x !== b);
// ipdom = the strict post-dom that does not post-dominate any other strict post-dom.
res[b] =
strict.find((cand) => strict.every((other) => other === cand || !pdom[other].has(cand))) ??
null;
}
return res;
}
const CFGS: CFG[] = [
{
name: 'early-return',
hazard: 'early return / multiple paths to EXIT',
tsSource: `function f(x){ if (x) { return 1; } g(); return 2; }`,
entry: 'ENTRY',
exit: 'EXIT',
succ: { ENTRY: ['ret1', 'g'], ret1: ['EXIT'], g: ['ret2'], ret2: ['EXIT'], EXIT: [] },
},
{
name: 'try-throw-finally',
hazard: 'try/throw/finally with multiple exits through finally',
tsSource: `function f(){ try { risky(); } catch(e){ handle(e); } finally { cleanup(); } done(); }`,
entry: 'ENTRY',
exit: 'EXIT',
// try → (normal | throw→catch) → finally → done → EXIT; finally also reached on rethrow
succ: {
ENTRY: ['try'],
try: ['finally', 'catch'],
catch: ['finally'],
finally: ['done', 'EXIT'],
done: ['EXIT'],
EXIT: [],
},
},
{
name: 'labeled-break',
hazard: 'labeled break/continue across nested loops',
tsSource: `outer: for(;;){ for(;;){ if (a) break outer; if (b) continue outer; work(); } }`,
entry: 'ENTRY',
exit: 'EXIT',
succ: {
ENTRY: ['outerHead'],
outerHead: ['innerHead', 'EXIT'],
innerHead: ['breakOuter', 'afterIf1'],
breakOuter: ['EXIT'],
afterIf1: ['contOuter', 'work'],
contOuter: ['outerHead'],
work: ['innerHead'],
EXIT: [],
},
},
{
name: 'if-else-diamond',
hazard: 'baseline reducible diamond (sanity)',
tsSource: `function f(x){ if (x) { a(); } else { b(); } c(); }`,
entry: 'ENTRY',
exit: 'EXIT',
succ: { ENTRY: ['a', 'b'], a: ['c'], b: ['c'], c: ['EXIT'], EXIT: [] },
},
];
function main() {
let allOk = true;
for (const cfg of CFGS) {
try {
const { pdom, iterations } = postDominators(cfg);
const idom = ipdom(cfg, pdom);
// Sanity invariants: EXIT post-dominates every block; ipdom tree reaches EXIT.
const exitPostDomsAll = Object.keys(cfg.succ).every((b) => pdom[b].has(cfg.exit));
console.log(`\n[S2] ${cfg.name} — ${cfg.hazard}`);
console.log(` TS: ${cfg.tsSource}`);
console.log(
` converged in ${iterations} iters; EXIT post-dominates all blocks: ${exitPostDomsAll}`,
);
console.log(
` ipdom tree: ${Object.entries(idom)
.map(([b, p]) => `${b}->${p ?? '∅'}`)
.join(' ')}`,
);
if (!exitPostDomsAll) allOk = false;
} catch (e) {
allOk = false;
console.log(`\n[S2] ${cfg.name} FAILED: ${(e as Error).message}`);
}
}
console.log(
`\n[S2] VERDICT INPUT: all hazard CFGs converged + EXIT post-dominates all = ${allOk}`,
);
}
main();

View file

@ -1,291 +0,0 @@
<!doctype html>
<html lang="en">
<head>
<meta charset="utf-8" />
<meta name="viewport" content="width=device-width,initial-scale=1" />
<title>GitNexus — Shadow Parity Dashboard</title>
<!--
Static dashboard for the RFC #909 shadow-mode parity report.
Reads `latest.json` from this directory and renders a per-language
parity table. Zero build step, zero runtime dependencies — a
single file that any browser or file:// context can open.
Usage:
# from repo root, after a shadow-mode run
cp .gitnexus/shadow-parity/latest.json gitnexus/shadow-parity-dashboard/
open gitnexus/shadow-parity-dashboard/index.html
CI artifact wiring (follow-up): the CI job publishes a snapshot
of this directory + latest.json as a downloadable bundle per run.
-->
<style>
:root {
color-scheme: light dark;
--fg: #1f2937;
--fg-muted: #6b7280;
--bg: #ffffff;
--bg-muted: #f9fafb;
--border: #e5e7eb;
--good: #16a34a;
--warn: #d97706;
--bad: #dc2626;
--primary-tag-legacy: #7c3aed;
--primary-tag-registry: #0ea5e9;
}
@media (prefers-color-scheme: dark) {
:root {
--fg: #e5e7eb;
--fg-muted: #9ca3af;
--bg: #111827;
--bg-muted: #1f2937;
--border: #374151;
}
}
html,
body {
margin: 0;
padding: 0;
background: var(--bg);
color: var(--fg);
font:
14px/1.45 system-ui,
-apple-system,
sans-serif;
}
main {
max-width: 1200px;
margin: 0 auto;
padding: 24px 16px;
}
h1 {
font-size: 20px;
margin: 0 0 4px;
}
.meta {
color: var(--fg-muted);
font-size: 12px;
margin-bottom: 20px;
}
.cards {
display: grid;
grid-template-columns: repeat(auto-fit, minmax(180px, 1fr));
gap: 10px;
margin-bottom: 20px;
}
.card {
border: 1px solid var(--border);
border-radius: 6px;
padding: 10px 12px;
background: var(--bg-muted);
}
.card .k {
color: var(--fg-muted);
font-size: 11px;
text-transform: uppercase;
letter-spacing: 0.04em;
}
.card .v {
font-size: 20px;
font-weight: 600;
}
table {
width: 100%;
border-collapse: collapse;
font-variant-numeric: tabular-nums;
}
th,
td {
padding: 6px 10px;
text-align: right;
border-bottom: 1px solid var(--border);
}
th:first-child,
td:first-child {
text-align: left;
}
thead th {
font-weight: 600;
color: var(--fg-muted);
font-size: 12px;
background: var(--bg-muted);
}
tbody tr:hover {
background: var(--bg-muted);
}
.parity {
font-weight: 600;
}
.parity.good {
color: var(--good);
}
.parity.warn {
color: var(--warn);
}
.parity.bad {
color: var(--bad);
}
.tag {
display: inline-block;
padding: 1px 6px;
border-radius: 10px;
font-size: 10px;
margin-left: 6px;
color: white;
}
.tag.legacy {
background: var(--primary-tag-legacy);
}
.tag.registry {
background: var(--primary-tag-registry);
}
.empty {
padding: 40px;
text-align: center;
color: var(--fg-muted);
}
code {
font-family: ui-monospace, SFMono-Regular, Menlo, monospace;
background: var(--bg-muted);
padding: 1px 4px;
border-radius: 3px;
}
</style>
</head>
<body>
<main>
<h1>Shadow Parity — RFC #909</h1>
<div class="meta" id="meta">loading <code>latest.json</code>…</div>
<div class="cards" id="cards"></div>
<table id="per-language">
<thead>
<tr>
<th>Language</th>
<th>Total</th>
<th>Agree</th>
<th>Only legacy</th>
<th>Only new</th>
<th>Disagree</th>
<th>Both empty</th>
<th>Parity</th>
</tr>
</thead>
<tbody></tbody>
</table>
<div id="empty" class="empty" style="display: none">
No records yet. Enable <code>GITNEXUS_SHADOW_MODE=1</code> and run ingestion to populate.
</div>
</main>
<script>
/* global fetch, document */
(async function () {
const tbody = document.querySelector('#per-language tbody');
const cards = document.getElementById('cards');
const meta = document.getElementById('meta');
const empty = document.getElementById('empty');
const table = document.getElementById('per-language');
let payload;
try {
const r = await fetch('./latest.json', { cache: 'no-store' });
if (!r.ok) throw new Error('HTTP ' + r.status);
payload = await r.json();
} catch (err) {
meta.textContent = 'Failed to load latest.json: ' + err.message;
table.style.display = 'none';
empty.style.display = 'block';
return;
}
const primary = payload.primaryByLanguage || {};
const report = payload.report || {};
const perLang = report.perLanguage || [];
const overall = report.overall || {};
meta.textContent =
'Run ' +
payload.runId +
' — generated ' +
payload.generatedAt +
' (schema v' +
payload.schemaVersion +
')';
// Overall summary cards.
cards.innerHTML = '';
const overallParity = overall.parity !== undefined ? overall.parity : 0;
cards.appendChild(makeCard('Total calls', overall.totalCalls ?? 0));
cards.appendChild(makeCard('Both agree', overall.bothAgree ?? 0));
cards.appendChild(makeCard('Disagree', overall.bothDisagree ?? 0));
cards.appendChild(makeCard('Overall parity', formatPct(overallParity)));
if (!perLang.length) {
table.style.display = 'none';
empty.style.display = 'block';
return;
}
for (const row of perLang) {
const tr = document.createElement('tr');
const primaryTag = primary[row.language];
const tag = primaryTag
? '<span class="tag ' + primaryTag + '">primary: ' + primaryTag + '</span>'
: '';
const parityClass = parityClassFor(row.parity);
tr.innerHTML =
'<td>' +
escape(row.language) +
tag +
'</td>' +
'<td>' +
row.totalCalls +
'</td>' +
'<td>' +
row.bothAgree +
'</td>' +
'<td>' +
row.onlyLegacy +
'</td>' +
'<td>' +
row.onlyNew +
'</td>' +
'<td>' +
row.bothDisagree +
'</td>' +
'<td>' +
row.bothEmpty +
'</td>' +
'<td class="parity ' +
parityClass +
'">' +
formatPct(row.parity) +
'</td>';
tbody.appendChild(tr);
}
function makeCard(k, v) {
const div = document.createElement('div');
div.className = 'card';
div.innerHTML =
'<div class="k">' + escape(k) + '</div><div class="v">' + escape(String(v)) + '</div>';
return div;
}
function formatPct(x) {
if (typeof x !== 'number' || !isFinite(x)) return '—';
return (x * 100).toFixed(1) + '%';
}
function parityClassFor(x) {
if (typeof x !== 'number') return '';
if (x >= 0.95) return 'good';
if (x >= 0.8) return 'warn';
return 'bad';
}
function escape(s) {
return String(s).replace(/[&<>"']/g, function (c) {
return { '&': '&amp;', '<': '&lt;', '>': '&gt;', '"': '&quot;', "'": '&#39;' }[c];
});
}
})();
</script>
</body>
</html>

View file

@ -16,10 +16,10 @@ description: "Use when the user is debugging a bug, tracing an error, or asking
## Workflow
```
1. gitnexus_query({query: "<error or symptom>"}) → Find related execution flows
2. gitnexus_context({name: "<suspect>"}) → See callers/callees/processes
1. query({query: "<error or symptom>"}) → Find related execution flows
2. context({name: "<suspect>"}) → See callers/callees/processes
3. READ gitnexus://repo/{name}/process/{name} → Trace execution flow
4. gitnexus_cypher({query: "MATCH path..."}) → Custom traces if needed
4. cypher({query: "MATCH path..."}) → Custom traces if needed
```
> If "Index is stale" → run `node .gitnexus/run.cjs analyze` in terminal.
@ -28,11 +28,11 @@ description: "Use when the user is debugging a bug, tracing an error, or asking
```
- [ ] Understand the symptom (error message, unexpected behavior)
- [ ] gitnexus_query for error text or related code
- [ ] query for error text or related code
- [ ] Identify the suspect function from returned processes
- [ ] gitnexus_context to see callers and callees
- [ ] context to see callers and callees
- [ ] Trace execution flow via process resource if applicable
- [ ] gitnexus_cypher for custom call chain traces if needed
- [ ] cypher for custom call chain traces if needed
- [ ] Read source files to confirm root cause
```
@ -40,7 +40,7 @@ description: "Use when the user is debugging a bug, tracing an error, or asking
| Symptom | GitNexus Approach |
| -------------------- | ---------------------------------------------------------- |
| Error message | `gitnexus_query` for error text → `context` on throw sites |
| Error message | `query` for error text → `context` on throw sites |
| Wrong return value | `context` on the function → trace callees for data flow |
| Intermittent failure | `context` → look for external calls, async deps |
| Performance issue | `context` → find symbols with many callers (hot paths) |
@ -48,24 +48,24 @@ description: "Use when the user is debugging a bug, tracing an error, or asking
## Tools
**gitnexus_query** — find code related to error:
**query** — find code related to error:
```
gitnexus_query({query: "payment validation error"})
query({query: "payment validation error"})
→ Processes: CheckoutFlow, ErrorHandling
→ Symbols: validatePayment, handlePaymentError, PaymentException
```
**gitnexus_context** — full context for a suspect:
**context** — full context for a suspect:
```
gitnexus_context({name: "validatePayment"})
context({name: "validatePayment"})
→ Incoming calls: processCheckout, webhookHandler
→ Outgoing calls: verifyCard, fetchRates (external API!)
→ Processes: CheckoutFlow (step 3/7)
```
**gitnexus_cypher** — custom call chain traces:
**cypher** — custom call chain traces:
```cypher
MATCH path = (a)-[:CodeRelation {type: 'CALLS'}*1..2]->(b:Function {name: "validatePayment"})
@ -75,11 +75,11 @@ RETURN [n IN nodes(path) | n.name] AS chain
## Example: "Payment endpoint returns 500 intermittently"
```
1. gitnexus_query({query: "payment error handling"})
1. query({query: "payment error handling"})
→ Processes: CheckoutFlow, ErrorHandling
→ Symbols: validatePayment, handlePaymentError
2. gitnexus_context({name: "validatePayment"})
2. context({name: "validatePayment"})
→ Outgoing calls: verifyCard, fetchRates (external API!)
3. READ gitnexus://repo/my-app/process/CheckoutFlow

View file

@ -18,8 +18,8 @@ description: "Use when the user asks how code works, wants to understand archite
```
1. READ gitnexus://repos → Discover indexed repos
2. READ gitnexus://repo/{name}/context → Codebase overview, check staleness
3. gitnexus_query({query: "<what you want to understand>"}) → Find related execution flows
4. gitnexus_context({name: "<symbol>"}) → Deep dive on specific symbol
3. query({query: "<what you want to understand>"}) → Find related execution flows
4. context({name: "<symbol>"}) → Deep dive on specific symbol
5. READ gitnexus://repo/{name}/process/{name} → Trace full execution flow
```
@ -29,9 +29,9 @@ description: "Use when the user asks how code works, wants to understand archite
```
- [ ] READ gitnexus://repo/{name}/context
- [ ] gitnexus_query for the concept you want to understand
- [ ] query for the concept you want to understand
- [ ] Review returned processes (execution flows)
- [ ] gitnexus_context on key symbols for callers/callees
- [ ] context on key symbols for callers/callees
- [ ] READ process resource for full execution traces
- [ ] Read source files for implementation details
```
@ -47,18 +47,18 @@ description: "Use when the user asks how code works, wants to understand archite
## Tools
**gitnexus_query** — find execution flows related to a concept:
**query** — find execution flows related to a concept:
```
gitnexus_query({query: "payment processing"})
query({query: "payment processing"})
→ Processes: CheckoutFlow, RefundFlow, WebhookHandler
→ Symbols grouped by flow with file locations
```
**gitnexus_context** — 360-degree view of a symbol:
**context** — 360-degree view of a symbol:
```
gitnexus_context({name: "validateUser"})
context({name: "validateUser"})
→ Incoming calls: loginHandler, apiMiddleware
→ Outgoing calls: checkToken, getUserById
→ Processes: LoginFlow (step 2/5), TokenRefresh (step 1/3)
@ -68,10 +68,10 @@ gitnexus_context({name: "validateUser"})
```
1. READ gitnexus://repo/my-app/context → 918 symbols, 45 processes
2. gitnexus_query({query: "payment processing"})
2. query({query: "payment processing"})
→ CheckoutFlow: processPayment → validateCard → chargeStripe
→ RefundFlow: initiateRefund → calculateRefund → processRefund
3. gitnexus_context({name: "processPayment"})
3. context({name: "processPayment"})
→ Incoming: checkoutHandler, webhookHandler
→ Outgoing: validateCard, chargeStripe, saveTransaction
4. Read src/payments/processor.ts for implementation details

View file

@ -17,9 +17,9 @@ description: "Use when the user wants to know what will break if they change som
## Workflow
```
1. gitnexus_impact({target: "X", direction: "upstream"}) → What depends on this
1. impact({target: "X", direction: "upstream"}) → What depends on this
2. READ gitnexus://repo/{name}/processes → Check affected execution flows
3. gitnexus_detect_changes() → Map current git changes to affected flows
3. detect_changes() → Map current git changes to affected flows
4. Assess risk and report to user
```
@ -28,11 +28,11 @@ description: "Use when the user wants to know what will break if they change som
## Checklist
```
- [ ] gitnexus_impact({target, direction: "upstream"}) to find dependents
- [ ] impact({target, direction: "upstream"}) to find dependents
- [ ] Review d=1 items first (these WILL BREAK)
- [ ] Check high-confidence (>0.8) dependencies
- [ ] READ processes to check affected execution flows
- [ ] gitnexus_detect_changes() for pre-commit check
- [ ] detect_changes() for pre-commit check
- [ ] Assess risk level and report to user
```
@ -55,10 +55,10 @@ description: "Use when the user wants to know what will break if they change som
## Tools
**gitnexus_impact** — the primary tool for symbol blast radius:
**impact** — the primary tool for symbol blast radius:
```
gitnexus_impact({
impact({
target: "validateUser",
direction: "upstream",
minConfidence: 0.8,
@ -73,10 +73,10 @@ gitnexus_impact({
- authRouter (src/routes/auth.ts:22) [CALLS, 95%]
```
**gitnexus_detect_changes** — git-diff based impact analysis:
**detect_changes** — git-diff based impact analysis:
```
gitnexus_detect_changes({scope: "staged"})
detect_changes({scope: "staged"})
→ Changed: 5 symbols in 3 files
→ Affected: LoginFlow, TokenRefresh, APIMiddlewarePipeline
@ -86,7 +86,7 @@ gitnexus_detect_changes({scope: "staged"})
## Example: "What breaks if I change validateUser?"
```
1. gitnexus_impact({target: "validateUser", direction: "upstream"})
1. impact({target: "validateUser", direction: "upstream"})
→ d=1: loginHandler, apiMiddleware (WILL BREAK)
→ d=2: authRouter, sessionManager (LIKELY AFFECTED)

View file

@ -18,10 +18,10 @@ description: "Use when the user wants to review a pull request, understand what
```
1. gh pr diff <number> → Get the raw diff
2. gitnexus_detect_changes({scope: "compare", base_ref: "main"}) → Map diff to affected flows
2. detect_changes({scope: "compare", base_ref: "main"}) → Map diff to affected flows
3. For each changed symbol:
gitnexus_impact({target: "<symbol>", direction: "upstream"}) → Blast radius per change
4. gitnexus_context({name: "<key symbol>"}) → Understand callers/callees
impact({target: "<symbol>", direction: "upstream"}) → Blast radius per change
4. context({name: "<key symbol>"}) → Understand callers/callees
5. READ gitnexus://repo/{name}/processes → Check affected execution flows
6. Summarize findings with risk assessment
```
@ -32,10 +32,10 @@ description: "Use when the user wants to review a pull request, understand what
```
- [ ] Fetch PR diff (gh pr diff or git diff base...head)
- [ ] gitnexus_detect_changes to map changes to affected execution flows
- [ ] gitnexus_impact on each non-trivial changed symbol
- [ ] detect_changes to map changes to affected execution flows
- [ ] impact on each non-trivial changed symbol
- [ ] Review d=1 items (WILL BREAK) — are callers updated?
- [ ] gitnexus_context on key changed symbols to understand full picture
- [ ] context on key changed symbols to understand full picture
- [ ] Check if affected processes have test coverage
- [ ] Assess overall risk level
- [ ] Write review summary with findings
@ -63,20 +63,20 @@ description: "Use when the user wants to review a pull request, understand what
## Tools
**gitnexus_detect_changes** — map PR diff to affected execution flows:
**detect_changes** — map PR diff to affected execution flows:
```
gitnexus_detect_changes({scope: "compare", base_ref: "main"})
detect_changes({scope: "compare", base_ref: "main"})
→ Changed: 8 symbols in 4 files
→ Affected processes: CheckoutFlow, RefundFlow, WebhookHandler
→ Risk: MEDIUM
```
**gitnexus_impact** — blast radius per changed symbol:
**impact** — blast radius per changed symbol:
```
gitnexus_impact({target: "validatePayment", direction: "upstream"})
impact({target: "validatePayment", direction: "upstream"})
→ d=1 (WILL BREAK):
- processCheckout (src/checkout.ts:42) [CALLS, 100%]
@ -86,20 +86,20 @@ gitnexus_impact({target: "validatePayment", direction: "upstream"})
- checkoutRouter (src/routes/checkout.ts:22) [CALLS, 95%]
```
**gitnexus_impact with tests** — check test coverage:
**impact with tests** — check test coverage:
```
gitnexus_impact({target: "validatePayment", direction: "upstream", includeTests: true})
impact({target: "validatePayment", direction: "upstream", includeTests: true})
→ Tests that cover this symbol:
- validatePayment.test.ts [direct]
- checkout.integration.test.ts [via processCheckout]
```
**gitnexus_context** — understand a changed symbol's role:
**context** — understand a changed symbol's role:
```
gitnexus_context({name: "validatePayment"})
context({name: "validatePayment"})
→ Incoming calls: processCheckout, webhookHandler
→ Outgoing calls: verifyCard, fetchRates
@ -112,20 +112,20 @@ gitnexus_context({name: "validatePayment"})
1. gh pr diff 42 > /tmp/pr42.diff
→ 4 files changed: payments.ts, checkout.ts, types.ts, utils.ts
2. gitnexus_detect_changes({scope: "compare", base_ref: "main"})
2. detect_changes({scope: "compare", base_ref: "main"})
→ Changed symbols: validatePayment, PaymentInput, formatAmount
→ Affected processes: CheckoutFlow, RefundFlow
→ Risk: MEDIUM
3. gitnexus_impact({target: "validatePayment", direction: "upstream"})
3. impact({target: "validatePayment", direction: "upstream"})
→ d=1: processCheckout, webhookHandler (WILL BREAK)
→ webhookHandler is NOT in the PR diff — potential breakage!
4. gitnexus_impact({target: "PaymentInput", direction: "upstream"})
4. impact({target: "PaymentInput", direction: "upstream"})
→ d=1: validatePayment (in PR), createPayment (NOT in PR)
→ createPayment uses the old PaymentInput shape — breaking change!
5. gitnexus_context({name: "formatAmount"})
5. context({name: "formatAmount"})
→ Called by 12 functions — but change is backwards-compatible (added optional param)
6. Review summary:

View file

@ -16,9 +16,9 @@ description: "Use when the user wants to rename, extract, split, move, or restru
## Workflow
```
1. gitnexus_impact({target: "X", direction: "upstream"}) → Map all dependents
2. gitnexus_query({query: "X"}) → Find execution flows involving X
3. gitnexus_context({name: "X"}) → See all incoming/outgoing refs
1. impact({target: "X", direction: "upstream"}) → Map all dependents
2. query({query: "X"}) → Find execution flows involving X
3. context({name: "X"}) → See all incoming/outgoing refs
4. Plan update order: interfaces → implementations → callers → tests
```
@ -29,65 +29,65 @@ description: "Use when the user wants to rename, extract, split, move, or restru
### Rename Symbol
```
- [ ] gitnexus_rename({symbol_name: "oldName", new_name: "newName", dry_run: true}) — preview all edits
- [ ] rename({symbol_name: "oldName", new_name: "newName", dry_run: true}) — preview all edits
- [ ] Review graph edits (high confidence) and ast_search edits (review carefully)
- [ ] If satisfied: gitnexus_rename({..., dry_run: false}) — apply edits
- [ ] gitnexus_detect_changes() — verify only expected files changed
- [ ] If satisfied: rename({..., dry_run: false}) — apply edits
- [ ] detect_changes() — verify only expected files changed
- [ ] Run tests for affected processes
```
### Extract Module
```
- [ ] gitnexus_context({name: target}) — see all incoming/outgoing refs
- [ ] gitnexus_impact({target, direction: "upstream"}) — find all external callers
- [ ] context({name: target}) — see all incoming/outgoing refs
- [ ] impact({target, direction: "upstream"}) — find all external callers
- [ ] Define new module interface
- [ ] Extract code, update imports
- [ ] gitnexus_detect_changes() — verify affected scope
- [ ] detect_changes() — verify affected scope
- [ ] Run tests for affected processes
```
### Split Function/Service
```
- [ ] gitnexus_context({name: target}) — understand all callees
- [ ] context({name: target}) — understand all callees
- [ ] Group callees by responsibility
- [ ] gitnexus_impact({target, direction: "upstream"}) — map callers to update
- [ ] impact({target, direction: "upstream"}) — map callers to update
- [ ] Create new functions/services
- [ ] Update callers
- [ ] gitnexus_detect_changes() — verify affected scope
- [ ] detect_changes() — verify affected scope
- [ ] Run tests for affected processes
```
## Tools
**gitnexus_rename** — automated multi-file rename:
**rename** — automated multi-file rename:
```
gitnexus_rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: true})
rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: true})
→ 12 edits across 8 files
→ 10 graph edits (high confidence), 2 ast_search edits (review)
→ Changes: [{file_path, edits: [{line, old_text, new_text, confidence}]}]
```
**gitnexus_impact** — map all dependents first:
**impact** — map all dependents first:
```
gitnexus_impact({target: "validateUser", direction: "upstream"})
impact({target: "validateUser", direction: "upstream"})
→ d=1: loginHandler, apiMiddleware, testUtils
→ Affected Processes: LoginFlow, TokenRefresh
```
**gitnexus_detect_changes** — verify your changes after refactoring:
**detect_changes** — verify your changes after refactoring:
```
gitnexus_detect_changes({scope: "all"})
detect_changes({scope: "all"})
→ Changed: 8 files, 12 symbols
→ Affected processes: LoginFlow, TokenRefresh
→ Risk: MEDIUM
```
**gitnexus_cypher** — custom reference queries:
**cypher** — custom reference queries:
```cypher
MATCH (caller)-[:CodeRelation {type: 'CALLS'}]->(f:Function {name: "validateUser"})
@ -98,24 +98,24 @@ RETURN caller.name, caller.filePath ORDER BY caller.filePath
| Risk Factor | Mitigation |
| ------------------- | ----------------------------------------- |
| Many callers (>5) | Use gitnexus_rename for automated updates |
| Many callers (>5) | Use rename for automated updates |
| Cross-area refs | Use detect_changes after to verify scope |
| String/dynamic refs | gitnexus_query to find them |
| String/dynamic refs | query to find them |
| External/public API | Version and deprecate properly |
## Example: Rename `validateUser` to `authenticateUser`
```
1. gitnexus_rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: true})
1. rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: true})
→ 12 edits: 10 graph (safe), 2 ast_search (review)
→ Files: validator.ts, login.ts, middleware.ts, config.json...
2. Review ast_search edits (config.json: dynamic reference!)
3. gitnexus_rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: false})
3. rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: false})
→ Applied 12 edits across 8 files
4. gitnexus_detect_changes({scope: "all"})
4. detect_changes({scope: "all"})
→ Affected: LoginFlow, TokenRefresh
→ Risk: MEDIUM — run tests for these flows
```

View file

@ -174,18 +174,18 @@ This project is indexed by GitNexus as **${projectName}**${noStats ? '' : ` (${s
## Always Do
- **MUST run impact analysis before editing any symbol.** Before modifying a function, class, or method, run \`gitnexus_impact({target: "symbolName", direction: "upstream"})\` and report the blast radius (direct callers, affected processes, risk level) to the user.
- **MUST run \`gitnexus_detect_changes()\` before committing** to verify your changes only affect expected symbols and execution flows. For regression review, compare against the default branch: \`gitnexus_detect_changes({scope: "compare", base_ref: ${JSON.stringify(markdownSafeBranch(defaultBranch))}})\`.
- **MUST run impact analysis before editing any symbol.** Before modifying a function, class, or method, run \`impact({target: "symbolName", direction: "upstream"})\` and report the blast radius (direct callers, affected processes, risk level) to the user.
- **MUST run \`detect_changes()\` before committing** to verify your changes only affect expected symbols and execution flows. For regression review, compare against the default branch: \`detect_changes({scope: "compare", base_ref: ${JSON.stringify(markdownSafeBranch(defaultBranch))}})\`.
- **MUST warn the user** if impact analysis returns HIGH or CRITICAL risk before proceeding with edits.
- When exploring unfamiliar code, use \`gitnexus_query({query: "concept"})\` to find execution flows instead of grepping. It returns process-grouped results ranked by relevance.
- When you need full context on a specific symbol — callers, callees, which execution flows it participates in — use \`gitnexus_context({name: "symbolName"})\`.
- When exploring unfamiliar code, use \`query({query: "concept"})\` to find execution flows instead of grepping. It returns process-grouped results ranked by relevance.
- When you need full context on a specific symbol — callers, callees, which execution flows it participates in — use \`context({name: "symbolName"})\`.
## Never Do
- NEVER edit a function, class, or method without first running \`gitnexus_impact\` on it.
- NEVER edit a function, class, or method without first running \`impact\` on it.
- NEVER ignore HIGH or CRITICAL risk warnings from impact analysis.
- NEVER rename symbols with find-and-replace — use \`gitnexus_rename\` which understands the call graph.
- NEVER commit changes without running \`gitnexus_detect_changes()\` to check affected scope.
- NEVER rename symbols with find-and-replace — use \`rename\` which understands the call graph.
- NEVER commit changes without running \`detect_changes()\` to check affected scope.
## Resources

View file

@ -9,6 +9,7 @@
*/
import path from 'path';
import os from 'os';
import { spawn } from 'child_process';
import v8 from 'v8';
import cliProgress from 'cli-progress';
@ -36,7 +37,7 @@ import {
} from './analyze-config.js';
import { runFullAnalysis } from '../core/run-analyze.js';
import { getMaxFileSizeBannerMessage } from '../core/ingestion/utils/max-file-size.js';
import { warnMissingOptionalGrammars } from './optional-grammars.js';
import { warnMissingOptionalGrammars, getOptionalGrammarExtensions } from './optional-grammars.js';
import { glob } from 'glob';
import fs from 'fs/promises';
import { cliError } from './cli-message.js';
@ -59,6 +60,22 @@ const writeFatalToStderr = (label: string, err: unknown): void => {
const message = isErr ? err.message : String(err);
realStderrWrite(`\n ${label}: ${message}\n`);
if (isErr && err.stack) realStderrWrite(`${err.stack}\n`);
// Walk and print the `cause` chain. The phase runner wraps the underlying
// failure as `new Error("Phase 'X' failed: …", { cause })`, so the original
// error (e.g. a WorkerPoolDispatchError carrying the worker-side stack from
// #2068) is only reachable via `.cause`. Without this the user sees the
// wrapper's main-thread stack and never the real frame. `cause.stack` already
// begins with the cause's message, so we print the stack alone (not message +
// stack) to avoid repeating it. Depth-bounded so a cyclic `cause` can't loop
// (the phase runner wraps one level; the bound leaves headroom for future
// nesting); uses realStderrWrite so the redirected console.error's ANSI
// clear-line wrapping can't erase it (#1169).
const MAX_CAUSE_DEPTH = 5;
let cause: unknown = isErr ? (err as { cause?: unknown }).cause : undefined;
for (let depth = 0; depth < MAX_CAUSE_DEPTH && cause instanceof Error; depth++) {
realStderrWrite(`\n Caused by: ${cause.stack ?? cause.message}\n`);
cause = (cause as { cause?: unknown }).cause;
}
};
let fatalHandlersInstalled = false;
@ -84,13 +101,45 @@ const installFatalHandlers = (): void => {
});
};
const HEAP_MB = 16384;
/** Historical floor for the re-exec heap cap — the auto-sizer never goes below
* this, so small boxes / CI never regress. */
const DEFAULT_HEAP_MB = 16384;
/**
* RAM-aware re-exec heap cap (MB): `0.75 × effective RAM`, clamped to
* `>= DEFAULT_HEAP_MB`. Kept BELOW physical RAM on purpose — a cap `>=` RAM makes
* V8 collect lazily and inflate the heap into swap-thrash (observed analyzing the
* Linux kernel at a 30GB cap on a 31GB box). `constrainedBytes` is the cgroup
* limit or `null`; it is honored only as a real, smaller-than-physical cap, because
* `process.constrainedMemory()` returns a huge sentinel when UNCONSTRAINED.
*/
export function computeHeapCapMb(totalBytes: number, constrainedBytes: number | null): number {
const effectiveBytes =
constrainedBytes !== null && constrainedBytes > 0 && constrainedBytes < totalBytes
? constrainedBytes
: totalBytes;
const effectiveMb = Math.floor(effectiveBytes / (1024 * 1024));
return Math.max(DEFAULT_HEAP_MB, Math.floor(0.75 * effectiveMb));
}
function readConstrainedBytes(): number | null {
if (typeof process.constrainedMemory !== 'function') return null;
const c = process.constrainedMemory();
return typeof c === 'number' && c > 0 ? c : null;
}
const HEAP_MB = computeHeapCapMb(os.totalmem(), readConstrainedBytes());
const TEST_RESPAWN_HEAP_MB = Number(process.env.GITNEXUS_TEST_RESPAWN_HEAP_MB);
const RESPAWN_HEAP_MB =
Number.isFinite(TEST_RESPAWN_HEAP_MB) && TEST_RESPAWN_HEAP_MB > 0
? Math.floor(TEST_RESPAWN_HEAP_MB)
: HEAP_MB;
const HEAP_FLAG = `--max-old-space-size=${RESPAWN_HEAP_MB}`;
/** Larger semi-space (young-gen) cuts minor-GC frequency + promotion churn during
* the multi-million-node graph build/emit. Allowed in NODE_OPTIONS (unlike
* --stack-size), so it propagates to the re-exec env cleanly. */
const SEMI_SPACE_MB = 128;
const SEMI_FLAG = `--max-semi-space-size=${SEMI_SPACE_MB}`;
/** Increase default stack size (KB) to prevent stack overflow on deep class hierarchies. */
const STACK_KB = 4096;
const STACK_FLAG = `--stack-size=${STACK_KB}`;
@ -440,7 +489,8 @@ const forceHeapOOMForTestIfEnabled = (): void => {
// `gitnexus/src/core/lbug/lbug-config.ts` in sync with this value.
const RECOMMENDED_WAL_CHECKPOINT_THRESHOLD = 64 * 1024 * 1024;
/** Re-exec the process with a 16GB heap and larger stack if we're currently below that. */
/** Re-exec the process with the RAM-aware auto heap cap + larger semi-space/stack
* if we're currently below that. A user-supplied NODE_OPTIONS heap wins (no re-exec). */
async function ensureHeap(): Promise<boolean> {
const nodeOpts = process.env.NODE_OPTIONS || '';
if (nodeOpts.includes('--max-old-space-size')) return false;
@ -448,25 +498,26 @@ async function ensureHeap(): Promise<boolean> {
const v8Heap = v8.getHeapStatistics().heap_size_limit;
if (v8Heap >= HEAP_MB * 1024 * 1024 * 0.9) return false;
// --stack-size is a V8 flag not allowed in NODE_OPTIONS on Node 24+,
// so pass it only as a direct CLI argument, not via the environment.
const cliFlags = [HEAP_FLAG];
// --stack-size is a V8 flag not allowed in NODE_OPTIONS on Node 24+, so pass it
// only as a direct CLI argument. --max-semi-space-size IS allowed in NODE_OPTIONS.
const cliFlags = [HEAP_FLAG, SEMI_FLAG];
if (!nodeOpts.includes('--stack-size')) cliFlags.push(STACK_FLAG);
const childArgs = [...cliFlags, ...process.argv.slice(1)];
const childEnv = {
...process.env,
NODE_OPTIONS: `${nodeOpts} ${HEAP_FLAG}`.trim(),
NODE_OPTIONS: `${nodeOpts} ${HEAP_FLAG} ${SEMI_FLAG}`.trim(),
};
if (shouldBridgeRespawnProgressTty()) childEnv[RESPAWN_PROGRESS_ENV] = '1';
const childExit = await runRespawnedAnalyze(childArgs, childEnv);
if (childExit.status !== 0 || childExit.signal) {
if (childProcessLikelyOom(childExit)) {
cliError(
` Analysis likely ran out of memory.\n` +
` Retry with a larger heap if your machine allows it:\n` +
` NODE_OPTIONS="--max-old-space-size=24576" gitnexus analyze [your-args]\n` +
` (Windows: set NODE_OPTIONS=--max-old-space-size=24576 && gitnexus analyze [your-args])\n` +
` Analysis likely ran out of memory (heap cap auto-sized to ${RESPAWN_HEAP_MB}MB ≈ 0.75x RAM).\n` +
` This repository's working set exceeds available RAM. Use a machine with more RAM,\n` +
` or override the cap (a cap above physical RAM causes swap-thrash — use with care):\n` +
` NODE_OPTIONS="--max-old-space-size=<MB>" gitnexus analyze [your-args]\n` +
` (Windows: set NODE_OPTIONS=--max-old-space-size=<MB> && gitnexus analyze [your-args])\n` +
` If this persists, it may be a native crash unrelated to heap size.\n`,
{ recoveryHint: 'heap-oom-respawn' },
);
@ -474,8 +525,7 @@ async function ensureHeap(): Promise<boolean> {
cliError(
` Analysis aborted in a native worker or native binding path.\n` +
` Try one of these recovery paths:\n` +
` gitnexus analyze --workers 0\n` +
` npm uninstall -g gitnexus && npm install -g gitnexus@latest\n` +
` npm uninstall -g gitnexus && npm install -g gitnexus@latest (rebuilds native bindings)\n` +
` Use Node 22 LTS if you are on a newer non-LTS runtime.\n`,
{ recoveryHint: 'native-worker-abort' },
);
@ -500,6 +550,7 @@ const ANALYZE_CLI_ENV_KEYS = [
'GITNEXUS_VERBOSE',
'GITNEXUS_PROFILE_DEFERRED',
'GITNEXUS_PROFILE_DEFERRED_SLOW_MS',
'GITNEXUS_DEBUG_HEAP',
'GITNEXUS_MAX_FILE_SIZE',
'GITNEXUS_WORKER_SUB_BATCH_TIMEOUT_MS',
'GITNEXUS_WAL_CHECKPOINT_THRESHOLD',
@ -598,7 +649,7 @@ export interface AnalyzeOptions {
workerTimeout?: string;
/** Control LadybugDB WAL auto-checkpoint threshold during analyze. */
walCheckpointThreshold?: string;
/** Parse worker pool size; 0 disables workers (sequential fallback). */
/** Parse worker pool size (>=1); 0 is rejected (no sequential mode). */
workers?: string;
embeddingThreads?: string;
embeddingBatchSize?: string;
@ -793,10 +844,11 @@ const analyzeCommandImpl = async (
let workerPoolSize: number | undefined;
if (options.workers !== undefined) {
const parsedWorkers = Number(options.workers);
if (!Number.isInteger(parsedWorkers) || parsedWorkers < 0) {
if (!Number.isInteger(parsedWorkers) || parsedWorkers < 1) {
cliError(
' --workers must be a non-negative integer. ' +
'Pass 0 to disable the worker pool (sequential fallback).\n',
' --workers must be a positive integer (>= 1). ' +
'GitNexus parses through a worker pool only — there is no sequential ' +
'mode, so 0 is not allowed. Omit --workers for an auto-sized pool.\n',
);
process.exitCode = 1;
return;
@ -891,11 +943,13 @@ const analyzeCommandImpl = async (
}
// If the target repo contains files an optional grammar would parse but
// that grammar's native binding is absent, warn before analysis so users
// learn why those files end up unparsed instead of silently getting a
// degraded index.
// that grammar's native binding is absent (or disabled via
// GITNEXUS_SKIP_OPTIONAL_GRAMMARS), warn before analysis so users learn why
// those files end up unparsed instead of silently getting a degraded index.
// The extension set is derived from OPTIONAL_GRAMMARS so it can't drift.
try {
const matches = await glob(['**/*.dart', '**/*.proto'], {
const optionalGlobs = getOptionalGrammarExtensions().map((e) => `**/*${e}`);
const matches = await glob(optionalGlobs, {
cwd: repoPath,
ignore: ['**/node_modules/**', '**/.git/**', '**/dist/**', '**/build/**'],
dot: false,

View file

@ -175,7 +175,7 @@ export const en = {
'help.option.analyze.walCheckpointThreshold':
'LadybugDB WAL auto-checkpoint threshold in bytes during analyze (integer >= -1; default: 67108864 = 64 MiB; -1 keeps Ladybug stock ~16 MiB).',
'help.option.analyze.workers':
'Parse worker pool size. Default: cores-1 capped at 16. Pass 0 to disable workers (sequential).',
'Parse worker pool size (>=1). Default: cores-1 capped at 16, auto-sized to the repo.',
'help.option.analyze.embeddingThreads': 'Limit local ONNX embedding CPU threads',
'help.option.analyze.embeddingBatchSize': 'Number of nodes per embedding batch',
'help.option.analyze.embeddingSubBatchSize': 'Number of chunks per embedding model call',

View file

@ -164,7 +164,7 @@ export const zhCN = {
'help.option.analyze.walCheckpointThreshold':
'analyze 期间 LadybugDB WAL 自动 checkpoint 阈值(字节,整数 >= -1;默认:67108864 = 64 MiB;-1 保持 Ladybug 默认约 16 MiB)。',
'help.option.analyze.workers':
'解析 worker 池大小。默认:cores-1,最多 16。传 0 禁用 worker(顺序执行)。',
'解析 worker 池大小(>=1)。默认:cores-1,最多 16,按仓库规模自适应。',
'help.option.analyze.embeddingThreads': '限制本地 ONNX 嵌入 CPU 线程数',
'help.option.analyze.embeddingBatchSize': '每个嵌入批次的节点数',
'help.option.analyze.embeddingSubBatchSize': '每次嵌入模型调用的分块数',

View file

@ -87,7 +87,7 @@ program
)
.option(
'--workers <n>',
'Parse worker pool size. Default: cores-1 capped at 16. Pass 0 to disable workers (sequential).',
'Parse worker pool size (>=1). Default: cores-1 capped at 16, auto-sized to the repo.',
)
.option('--embedding-threads <n>', 'Limit local ONNX embedding CPU threads')
.option('--embedding-batch-size <n>', 'Number of nodes per embedding batch')

View file

@ -4,18 +4,22 @@
* tree-sitter-dart, tree-sitter-proto, and tree-sitter-swift are vendored
* under vendor/ and materialized into node_modules/ at postinstall. Dart
* and Proto are built from source with node-gyp; Swift ships platform
* prebuilds activated via node-gyp-build. All three can be skipped via
* prebuilds activated via node-gyp-build. tree-sitter-kotlin is a declared
* optionalDependency (not vendored). All can be skipped via
* GITNEXUS_SKIP_OPTIONAL_GRAMMARS=1 (postinstall scripts), or can silently
* soft-fail when the toolchain is missing (Dart/Proto) or no prebuild
* matches the host platform (Swift).
* soft-fail when the toolchain is missing (Dart/Proto), when no prebuild
* matches the host platform (Swift), or when the optional install was
* skipped or its native build failed (Kotlin).
*
* Either path produces the same observable: the .node binding is absent
* at runtime. This helper detects that condition and surfaces a single
* stderr line per missing grammar so users learn why .dart/.proto/.swift
* stderr line per missing grammar so users learn why .dart/.proto/.swift/.kt
* support is unavailable instead of silently getting a degraded index.
*/
import { createRequire } from 'module';
import { SupportedLanguages } from 'gitnexus-shared';
import { isGrammarRuntimeSkipped } from '../core/tree-sitter/parser-loader.js';
import { cliWarn } from './cli-message.js';
const _require = createRequire(import.meta.url);
@ -27,17 +31,55 @@ interface OptionalGrammar {
pkg: string;
/** File extensions this grammar parses */
extensions: string[];
/**
* SupportedLanguages id, when this grammar backs an ingestion language.
* Used to ask `isGrammarRuntimeSkipped` whether the grammar was disabled via
* `GITNEXUS_SKIP_OPTIONAL_GRAMMARS` (vs. genuinely missing). Omitted for
* `.proto`, which is a gRPC-extractor concern, not a SupportedLanguages.
*/
language?: SupportedLanguages;
}
const OPTIONAL_GRAMMARS: OptionalGrammar[] = [
{ name: 'tree-sitter-dart', pkg: 'tree-sitter-dart', extensions: ['.dart'] },
{
name: 'tree-sitter-dart',
pkg: 'tree-sitter-dart',
extensions: ['.dart'],
language: SupportedLanguages.Dart,
},
{ name: 'tree-sitter-proto', pkg: 'tree-sitter-proto', extensions: ['.proto'] },
{ name: 'tree-sitter-swift', pkg: 'tree-sitter-swift', extensions: ['.swift'] },
{
name: 'tree-sitter-swift',
pkg: 'tree-sitter-swift',
extensions: ['.swift'],
language: SupportedLanguages.Swift,
},
{
name: 'tree-sitter-kotlin',
pkg: 'tree-sitter-kotlin',
extensions: ['.kt', '.kts'],
language: SupportedLanguages.Kotlin,
},
];
/**
* The file extensions backed by an optional grammar — the single source for
* the `analyze` preflight glob (so the glob can't drift from this list).
*/
export function getOptionalGrammarExtensions(): string[] {
return [...new Set(OPTIONAL_GRAMMARS.flatMap((g) => g.extensions))];
}
export interface MissingGrammar {
name: string;
extensions: string[];
/**
* `missing` — the native binding could not be loaded (not installed / build
* soft-failed / no prebuild). `skipped` — the binding is fine but the user
* disabled it via `GITNEXUS_SKIP_OPTIONAL_GRAMMARS`. Drives the warning text
* so a deliberate opt-out is not told to reinstall.
*/
reason: 'missing' | 'skipped';
}
/**
@ -59,6 +101,13 @@ export interface MissingGrammar {
export function detectMissingOptionalGrammars(): MissingGrammar[] {
const missing: MissingGrammar[] = [];
for (const g of OPTIONAL_GRAMMARS) {
// Deliberate runtime opt-out comes first: even an installed binding is
// treated as unavailable, with a `skipped` reason so the warning says so
// instead of suggesting a reinstall (#2101 review).
if (g.language !== undefined && isGrammarRuntimeSkipped(g.language)) {
missing.push({ name: g.name, extensions: g.extensions, reason: 'skipped' });
continue;
}
try {
_require(g.pkg);
} catch (err) {
@ -80,7 +129,7 @@ export function detectMissingOptionalGrammars(): MissingGrammar[] {
{ grammar: g.name, extensions: g.extensions, error: msg },
);
}
missing.push({ name: g.name, extensions: g.extensions });
missing.push({ name: g.name, extensions: g.extensions, reason: 'missing' });
}
}
return missing;
@ -110,9 +159,16 @@ export function warnMissingOptionalGrammars(opts?: {
if (relevantExtensions && !g.extensions.some((e) => relevantExtensions.has(e))) {
continue;
}
cliWarn(
`GitNexus${ctx}: optional grammar "${g.name}" is unavailable — ${g.extensions.join('/')} files will not be parsed. Reinstall without GITNEXUS_SKIP_OPTIONAL_GRAMMARS=1 (and ensure python3, make, g++) to enable.`,
{ grammar: g.name, extensions: g.extensions, context: opts?.context },
);
const exts = g.extensions.join('/');
const message =
g.reason === 'skipped'
? `GitNexus${ctx}: optional grammar "${g.name}" is disabled via GITNEXUS_SKIP_OPTIONAL_GRAMMARS — ${exts} files will not be parsed. Unset the variable to re-enable.`
: `GitNexus${ctx}: optional grammar "${g.name}" is unavailable — ${exts} files will not be parsed. Reinstall without GITNEXUS_SKIP_OPTIONAL_GRAMMARS=1 (and ensure python3, make, g++) to enable.`;
cliWarn(message, {
grammar: g.name,
extensions: g.extensions,
reason: g.reason,
context: opts?.context,
});
}
}

View file

@ -649,9 +649,9 @@ const renderSkillMarkdown = (
: community.label;
lines.push('## How to Explore');
lines.push('');
lines.push(`1. \`gitnexus_context({name: "${firstEntry}"})\` \u2014 see callers and callees`);
lines.push(`1. \`context({name: "${firstEntry}"})\` \u2014 see callers and callees`);
lines.push(
`2. \`gitnexus_query({query: "${community.label.toLowerCase()}"})\` \u2014 find related execution flows`,
`2. \`query({query: "${community.label.toLowerCase()}"})\` \u2014 find related execution flows`,
);
lines.push('3. Read key files listed above for implementation details');
lines.push('');

View file

@ -55,6 +55,7 @@ const METHOD_ANNOTATION_TO_HTTP: Record<string, string> = {
interface SpringRouteBinding {
method: string;
path: string;
ownerPrefix?: string;
}
interface SpringMethodInfo {
@ -395,6 +396,25 @@ function joinPath(prefix: string, methodPath: string): string {
return `/${cleanPrefix}/${cleanSub}`;
}
function joinInheritedSpringPath(
controllerPrefix: string,
inheritedPath: string,
inheritedOwnerPrefix = '',
): string {
const joined = joinPath(controllerPrefix, inheritedPath);
const cleanPrefix = controllerPrefix.replace(/^\/+/, '').replace(/\/+$/, '');
const cleanOwnerPrefix = inheritedOwnerPrefix.replace(/^\/+/, '').replace(/\/+$/, '');
const cleanInherited = inheritedPath.replace(/^\/+/, '');
if (!cleanPrefix) return joined;
if (
cleanPrefix === cleanOwnerPrefix &&
(cleanInherited === cleanPrefix || cleanInherited.startsWith(`${cleanPrefix}/`))
) {
return `/${cleanInherited}`;
}
return joined;
}
function getNodeName(node: Parser.SyntaxNode): string | null {
return node.childForFieldName('name')?.text ?? null;
}
@ -634,6 +654,7 @@ function scanSpringProject(files: readonly HttpScanInput[]): HttpFileDetections[
const routes = method.routes.map((route) => ({
method: route.method,
path: type.classPrefix ? joinPath(type.classPrefix, route.path) : route.path,
ownerPrefix: type.classPrefix,
}));
if (routes.length > 0) methodMap.set(method.name, routes);
}
@ -651,7 +672,7 @@ function scanSpringProject(files: readonly HttpScanInput[]): HttpFileDetections[
const routes = routeMap.get(method.name) ?? [];
return routes.map((route) => ({
method: route.method,
path: joinPath(type.classPrefix, route.path),
path: joinInheritedSpringPath(type.classPrefix, route.path, route.ownerPrefix),
}));
});

View file

@ -1,77 +0,0 @@
import { LRUCache } from 'lru-cache';
import Parser from 'tree-sitter';
import { logger } from '../logger.js';
/**
* Minimal structural shape consumers need when reading Trees back
* through a phase-dependency boundary. Declared here so phases that
* receive ASTCache via `getPhaseOutput<...>` don't hand-roll their
* own inline structural types that silently drift when ASTCache's
* contract changes.
*
* Typed as `unknown` at the Tree boundary because consumers on the
* other side of the phase-output map don't share tree-sitter's type
* graph (e.g. COBOL's standalone processor).
*/
export interface ASTCacheReader {
get(filePath: string): unknown;
clear(): void;
}
// Define the interface for the Cache
export interface ASTCache extends ASTCacheReader {
get: (filePath: string) => Parser.Tree | undefined;
set: (filePath: string, tree: Parser.Tree) => void;
clear: () => void;
stats: () => { size: number; maxSize: number };
}
export const createASTCache = (maxSize: number = 50): ASTCache => {
const effectiveMax = Math.max(maxSize, 1);
// Initialize the cache with a 'dispose' handler
// This is the magic: When an item is evicted (dropped), this runs automatically.
const cache = new LRUCache<string, Parser.Tree>({
max: effectiveMax,
dispose: (tree) => {
try {
// NOTE: web-tree-sitter has tree.delete(); native tree-sitter
// trees are GC-managed and .delete is absent (no-op here).
//
// Single-owner invariant (load-bearing under WASM): a given
// Parser.Tree reference must live in AT MOST ONE ASTCache
// that disposes. The parse-phase chunk-local cache clears
// between chunks; the cross-phase `scopeTreeCache` (also an
// ASTCache today) holds the same Tree by reference. Under
// native tree-sitter this is benign (dispose is a no-op).
// If/when GitNexus adopts web-tree-sitter for sequential
// parsing, the cross-phase cache must either (a) skip
// writing Trees that are already owned by a disposing cache,
// or (b) use tree.copy() per entry. Failing to pick one
// will hand freed memory to scope-resolution.
(tree as unknown as { delete?: () => void }).delete?.();
} catch (e) {
logger.warn({ e }, 'Failed to delete tree from WASM memory');
}
},
});
return {
get: (filePath: string) => {
const tree = cache.get(filePath);
return tree; // Returns undefined if not found
},
set: (filePath: string, tree: Parser.Tree) => {
cache.set(filePath, tree);
},
clear: () => {
cache.clear();
},
stats: () => ({
size: cache.size,
maxSize: effectiveMax,
}),
};
};

View file

@ -9,25 +9,17 @@
*
* - `processRoutesFromExtracted` — CALLS edges from framework routes
* (e.g. Laravel) to their controller methods.
* - `processNextjsFetchRoutes` / `extractFetchCallsFromFiles` /
* `extractConsumerAccessedKeys` — FETCHES edges from `fetch()` calls to
* Next.js Route nodes.
* - `processNextjsFetchRoutes` / `extractConsumerAccessedKeys` — FETCHES edges
* from `fetch()` calls to Next.js Route nodes.
* - `buildExportedTypeMapFromGraph` — exported symbol → return/declared type
* map, consumed by the cross-file enrichment pass.
*/
import Parser from 'tree-sitter';
import { KnowledgeGraph } from '../graph/types.js';
import { ASTCache } from './ast-cache.js';
import type { SemanticModel, SymbolTableReader } from './model/index.js';
import { isLanguageAvailable, loadParser, loadLanguage } from '../tree-sitter/parser-loader.js';
import { getProvider } from './languages/index.js';
import { generateId } from '../../lib/utils.js';
import { getLanguageFromFilename } from 'gitnexus-shared';
import type { SymbolDefinition } from 'gitnexus-shared';
import { yieldToEventLoop } from './utils/event-loop.js';
import { parseSourceSafe } from '../tree-sitter/safe-parse.js';
import { getTreeSitterBufferSize } from './constants.js';
import type { ExtractedRoute, ExtractedFetchCall } from './workers/parse-worker.js';
import { normalizeFetchURL, routeMatches } from './route-extractors/nextjs.js';
import { extractReturnTypeName } from './type-extractors/shared.js';
@ -39,6 +31,34 @@ const MAX_TYPE_NAME_LENGTH = 256;
* Consumed by the cross-file re-resolution / enrichment pass. */
export type ExportedTypeMap = Map<string, Map<string, string>>;
/** Record one exported graph node into the incremental ExportedTypeMap. */
export const accumulateExportedTypesFromParsedNode = (
result: ExportedTypeMap,
node: { id: string; properties?: Record<string, unknown> },
symbolTable: SymbolTableReader,
): void => {
if (!node.properties?.isExported) return;
if (!node.properties?.filePath || !node.properties?.name) return;
const filePath = node.properties.filePath as string;
const name = node.properties.name as string;
if (!name || name.length > MAX_TYPE_NAME_LENGTH) return;
const defs = symbolTable.lookupExactAll(filePath, name);
const def = defs.find((d) => d.nodeId === node.id) ?? defs[0];
if (!def) return;
const typeName = def.returnType ?? def.declaredType;
if (!typeName || typeName.length > MAX_TYPE_NAME_LENGTH) return;
const simpleType = extractReturnTypeName(typeName) ?? typeName;
if (!simpleType) return;
let fileExports = result.get(filePath);
if (!fileExports) {
fileExports = new Map();
result.set(filePath, fileExports);
}
if (fileExports.size < MAX_EXPORTS_PER_FILE) {
fileExports.set(name, simpleType);
}
};
/** Build ExportedTypeMap from graph nodes — used for the worker path where the
* sequential TypeEnv is not available in the main thread. Collects
* returnType/declaredType from exported symbols with known types. */
@ -48,29 +68,7 @@ export function buildExportedTypeMapFromGraph(
): ExportedTypeMap {
const result: ExportedTypeMap = new Map();
graph.forEachNode((node) => {
if (!node.properties?.isExported) return;
if (!node.properties?.filePath || !node.properties?.name) return;
const filePath = node.properties.filePath as string;
const name = node.properties.name as string;
if (!name || name.length > MAX_TYPE_NAME_LENGTH) return;
// For callable symbols, use returnType; for properties/variables, use declaredType.
// Use lookupExactAll + nodeId match to handle same-name methods in different classes.
const defs = symbolTable.lookupExactAll(filePath, name);
const def = defs.find((d) => d.nodeId === node.id) ?? defs[0];
if (!def) return;
const typeName = def.returnType ?? def.declaredType;
if (!typeName || typeName.length > MAX_TYPE_NAME_LENGTH) return;
// Extract simple type name (strip Promise<>, etc.) — reuse shared utility
const simpleType = extractReturnTypeName(typeName) ?? typeName;
if (!simpleType) return;
let fileExports = result.get(filePath);
if (!fileExports) {
fileExports = new Map();
result.set(filePath, fileExports);
}
if (fileExports.size < MAX_EXPORTS_PER_FILE) {
fileExports.set(name, simpleType);
}
accumulateExportedTypesFromParsedNode(result, node, symbolTable);
});
return result;
}
@ -448,79 +446,3 @@ export const processNextjsFetchRoutes = (
}
}
};
/**
* Extract fetch() calls from source files (sequential path).
* Workers handle this via tree-sitter captures in parse-worker; this function
* provides the same extraction for the sequential fallback path.
*/
export const extractFetchCallsFromFiles = async (
files: { path: string; content: string }[],
astCache: ASTCache,
): Promise<ExtractedFetchCall[]> => {
const parser = await loadParser();
const result: ExtractedFetchCall[] = [];
for (const file of files) {
const language = getLanguageFromFilename(file.path);
if (!language) continue;
if (!isLanguageAvailable(language)) continue;
const provider = getProvider(language);
const queryStr = provider.treeSitterQueries;
if (!queryStr) continue;
await loadLanguage(language, file.path);
let tree = astCache.get(file.path);
if (!tree) {
const parseContent = provider.preprocessSource?.(file.content, file.path) ?? file.content;
try {
tree = parseSourceSafe(parser, parseContent, undefined, {
bufferSize: getTreeSitterBufferSize(parseContent),
});
} catch {
continue;
}
astCache.set(file.path, tree);
}
let matches;
try {
const lang = parser.getLanguage();
const query = new Parser.Query(lang, queryStr);
matches = query.matches(tree.rootNode);
} catch {
continue;
}
for (const match of matches) {
const captureMap: Record<string, any> = {};
match.captures.forEach((c) => (captureMap[c.name] = c.node));
if (captureMap['route.fetch']) {
const urlNode = captureMap['route.url'] ?? captureMap['route.template_url'];
if (urlNode) {
result.push({
filePath: file.path,
fetchURL: urlNode.text,
lineNumber: captureMap['route.fetch'].startPosition.row,
});
}
} else if (captureMap['http_client'] && captureMap['http_client.url']) {
const method = captureMap['http_client.method']?.text;
const url = captureMap['http_client.url'].text;
const HTTP_CLIENT_ONLY = new Set(['head', 'options', 'request', 'ajax']);
if (method && HTTP_CLIENT_ONLY.has(method) && url.startsWith('/')) {
result.push({
filePath: file.path,
fetchURL: url,
lineNumber: captureMap['http_client'].startPosition.row,
});
}
}
}
}
return result;
};

View file

@ -4,7 +4,8 @@
* Determines whether a symbol (function, class, etc.) is exported/public
* in its language. This is a pure function — safe for use in worker threads.
*
* Shared between parse-worker.ts (worker pool) and parsing-processor.ts (sequential fallback).
* Used by the language providers during worker parsing (parse-worker.ts) — the
* sole parse path. (Sequential parsing was removed.)
*/
import { findSiblingChild, type SyntaxNode } from './utils/ast-helpers.js';

View file

@ -2,15 +2,59 @@
import { SupportedLanguages } from 'gitnexus-shared';
import type { FieldExtractionConfig } from '../generic.js';
import type { FieldVisibility } from '../../field-types.js';
import type { SyntaxNode } from '../../utils/ast-helpers.js';
import { hasKeyword } from './helpers.js';
import { extractSimpleTypeName } from '../../type-extractors/shared.js';
/**
* Dart field extraction config.
*
* Dart class fields appear as declaration nodes inside class_body.
* Dart class fields appear as `declaration` nodes inside `class_body`.
* Two shapes carry the field name(s):
* - instance / plain fields → `initialized_identifier_list`
* (`int z = 0;`, `int a = 1, b = 2;`)
* - `static const` / `static final` / `const` fields → `static_final_declaration_list`
* (`static const a = 1;`, `static final String b = 'x', c = 'y';`)
* Both shapes may declare SEVERAL fields in one declaration, so name extraction
* is multi-name (`extractNames`). The structure query (`DART_QUERIES`) emits one
* `@definition.property` per name for both shapes; this config enriches each.
*
* Visibility is convention-based: underscore prefix = private.
*/
/** All field names declared by a `declaration` node, across both Dart shapes. */
function extractDartFieldNames(node: SyntaxNode): string[] {
const names: string[] = [];
for (let i = 0; i < node.namedChildCount; i++) {
const child = node.namedChild(i);
if (!child) continue;
// instance / plain fields: initialized_identifier_list > initialized_identifier > identifier
if (child.type === 'initialized_identifier_list') {
for (let j = 0; j < child.namedChildCount; j++) {
const init = child.namedChild(j);
if (init?.type === 'initialized_identifier') {
const ident = init.firstNamedChild;
if (ident?.type === 'identifier') names.push(ident.text);
}
}
}
// static const / final fields: static_final_declaration_list > static_final_declaration > identifier
if (child.type === 'static_final_declaration_list') {
for (let j = 0; j < child.namedChildCount; j++) {
const decl = child.namedChild(j);
if (decl?.type === 'static_final_declaration') {
const ident = decl.firstNamedChild;
if (ident?.type === 'identifier') names.push(ident.text);
}
}
}
}
return names;
}
export const dartConfig: FieldExtractionConfig = {
language: SupportedLanguages.Dart,
typeDeclarationNodes: ['class_definition'],
@ -18,31 +62,20 @@ export const dartConfig: FieldExtractionConfig = {
bodyNodeTypes: ['class_body'],
defaultVisibility: 'public',
// One AST `declaration` node may declare several fields (`int a, b;`,
// `static final String b = 'x', c = 'y';`), so use the multi-name path.
extractName(node) {
// declaration > initialized_identifier_list > initialized_identifier > identifier
for (let i = 0; i < node.namedChildCount; i++) {
const child = node.namedChild(i);
if (child?.type === 'initialized_identifier_list') {
for (let j = 0; j < child.namedChildCount; j++) {
const init = child.namedChild(j);
if (init?.type === 'initialized_identifier') {
const ident = init.firstNamedChild;
if (ident?.type === 'identifier') return ident.text;
}
}
}
if (child?.type === 'initialized_identifier') {
const ident = child.firstNamedChild;
if (ident?.type === 'identifier') return ident.text;
}
}
// fallback: look for direct identifier
const name = node.childForFieldName('name');
return name?.text;
return extractDartFieldNames(node)[0];
},
extractNames(node) {
return extractDartFieldNames(node);
},
extractType(node) {
// declaration > type_identifier (first named child usually)
// declaration > type_identifier (the type annotation, present for both the
// instance-field shape and `static final String b = …`). `static const a = 1;`
// has no annotation → undefined (untyped).
for (let i = 0; i < node.namedChildCount; i++) {
const child = node.namedChild(i);
if (child && (child.type === 'type_identifier' || child.type === 'function_type')) {
@ -52,22 +85,16 @@ export const dartConfig: FieldExtractionConfig = {
return undefined;
},
extractVisibility(node) {
// Dart uses _ prefix for private
// Walk to find the identifier name
for (let i = 0; i < node.namedChildCount; i++) {
const child = node.namedChild(i);
if (child?.type === 'initialized_identifier_list') {
for (let j = 0; j < child.namedChildCount; j++) {
const init = child.namedChild(j);
if (init?.type === 'initialized_identifier') {
const ident = init.firstNamedChild;
if (ident?.text?.startsWith('_')) return 'private';
}
}
}
}
return 'public';
// Per-name: Dart convention is underscore-prefixed = private. A single
// declaration can mix visibilities (`static const _p = 1, q = 2;`), so the
// decision is keyed on the individual field name.
extractVisibilityForName(_node, name): FieldVisibility {
return name.startsWith('_') ? 'private' : 'public';
},
extractVisibility(node): FieldVisibility {
const first = extractDartFieldNames(node)[0];
return first?.startsWith('_') ? 'private' : 'public';
},
isStatic(node) {
@ -75,6 +102,8 @@ export const dartConfig: FieldExtractionConfig = {
},
isReadonly(node) {
// `final` / `const` (both `final_builtin`/`const_builtin` nodes whose text
// is `final`/`const`) are read-only.
return hasKeyword(node, 'final') || hasKeyword(node, 'const');
},
};

View file

@ -3,6 +3,8 @@
import { SupportedLanguages } from 'gitnexus-shared';
import type { FieldExtractionConfig } from '../generic.js';
import { extractSimpleTypeName } from '../../type-extractors/shared.js';
import type { FieldVisibility } from '../../field-types.js';
import type { SyntaxNode } from '../../utils/ast-helpers.js';
/**
* Go field extraction config.
@ -13,14 +15,52 @@ import { extractSimpleTypeName } from '../../type-extractors/shared.js';
* Visibility in Go is based on the first character: uppercase = exported (public),
* lowercase = unexported (package).
*/
function goVisibilityForName(name: string): FieldVisibility {
const first = name.charAt(0);
return first === first.toUpperCase() && first !== first.toLowerCase() ? 'public' : 'package';
}
function extractGoFieldNames(node: SyntaxNode): string[] {
const names: string[] = [];
for (let i = 0; i < node.namedChildCount; i++) {
const child = node.namedChild(i);
if (child?.type === 'field_identifier') names.push(child.text);
}
return names;
}
export const goConfig: FieldExtractionConfig = {
language: SupportedLanguages.Go,
typeDeclarationNodes: ['type_declaration'],
typeDeclarationNodes: ['type_declaration', 'struct_type'],
fieldNodeTypes: ['field_declaration'],
bodyNodeTypes: ['field_declaration_list'],
defaultVisibility: 'package',
extractOwnerName(node) {
if (node.type === 'struct_type') {
return node.parent?.type === 'type_spec'
? node.parent.childForFieldName('name')?.text
: undefined;
}
const typeSpec = node.namedChildren.find((child) => child.type === 'type_spec');
return typeSpec?.childForFieldName('name')?.text;
},
findBodyNodes(node) {
if (node.type === 'struct_type') {
const body = node.namedChildren.find((child) => child.type === 'field_declaration_list');
return body ? [body] : [];
}
const typeSpec = node.namedChildren.find((child) => child.type === 'type_spec');
const typeNode = typeSpec?.childForFieldName('type');
const body = typeNode?.namedChildren.find((child) => child.type === 'field_declaration_list');
return body ? [body] : [];
},
extractName(node) {
const firstName = extractGoFieldNames(node)[0];
if (firstName) return firstName;
// field_declaration > name:(field_identifier)
const name = node.childForFieldName('name');
if (name) return name.text;
@ -32,6 +72,8 @@ export const goConfig: FieldExtractionConfig = {
return undefined;
},
extractNames: extractGoFieldNames,
extractType(node) {
// field_declaration > type:(type_identifier | pointer_type | ...)
const typeNode = node.childForFieldName('type');
@ -54,6 +96,10 @@ export const goConfig: FieldExtractionConfig = {
return 'package';
},
extractVisibilityForName(_node, name) {
return goVisibilityForName(name);
},
isStatic(_node) {
return false; // Go has no static fields
},

View file

@ -5,6 +5,7 @@ import type { FieldExtractionConfig } from '../generic.js';
import { findVisibility, hasKeyword, hasModifier, typeFromField } from './helpers.js';
import { extractSimpleTypeName } from '../../type-extractors/shared.js';
import type { FieldVisibility } from '../../field-types.js';
import type { SyntaxNode } from '../../utils/ast-helpers.js';
// ---------------------------------------------------------------------------
// Java
@ -73,13 +74,49 @@ export const javaConfig: FieldExtractionConfig = {
const KOTLIN_VIS = new Set<FieldVisibility>(['public', 'private', 'protected', 'internal']);
/** A property_declaration is a companion-object member when its nearest
* class-body ancestor is the body of a companion_object (F52, issue #1919).
* Companion members are addressed statically through the enclosing class
* (`C.TAG`), so they are marked static. */
function isInsideKotlinCompanion(node: SyntaxNode): boolean {
for (let cur = node.parent; cur !== null; cur = cur.parent) {
if (cur.type === 'class_body') return cur.parent?.type === 'companion_object';
if (cur.type === 'companion_object') return true;
}
return false;
}
export const kotlinConfig: FieldExtractionConfig = {
language: SupportedLanguages.Kotlin,
typeDeclarationNodes: ['class_declaration', 'object_declaration'],
// F52: include companion_object so a companion property's innermost
// class-container owner (findEnclosingClassNode returns the companion_object)
// is recognized as a type declaration and its nested class_body is walked.
// The structure query already creates the Property node and owns it on the
// ENCLOSING class for anonymous companions / on the named companion Class —
// this entry only drives field-metadata enrichment, so it does NOT change
// ownership or emit a second node (no double-count).
typeDeclarationNodes: ['class_declaration', 'object_declaration', 'companion_object'],
fieldNodeTypes: ['property_declaration'],
bodyNodeTypes: ['class_body'],
defaultVisibility: 'public',
// F52: an anonymous `companion object { ... }` has no name child, so the
// generic factory's `childForFieldName('name')` owner lookup is empty and
// `extract()` would bail before walking the body. Supply a stable owner
// name (the named companion's identifier, else "Companion") so the body IS
// walked; the resulting FieldInfo map is keyed by field NAME only, so the
// owner name does not affect which Property node gets enriched.
extractOwnerName(node) {
const typeIdentifierText = node.namedChildren.find((c) => c.type === 'type_identifier')?.text;
if (node.type === 'companion_object') {
// Anonymous companions have no type_identifier — fall back to "Companion".
return typeIdentifierText ?? 'Companion';
}
const name = node.childForFieldName('name');
if (name) return name.text;
return typeIdentifierText;
},
extractName(node) {
// property_declaration > variable_declaration > simple_identifier
for (let i = 0; i < node.namedChildCount; i++) {
@ -124,9 +161,11 @@ export const kotlinConfig: FieldExtractionConfig = {
return findVisibility(node, KOTLIN_VIS, 'public', 'modifiers');
},
isStatic(_node) {
// Kotlin doesn't have static; companion object members are handled separately
return false;
isStatic(node) {
// Kotlin has no `static`, but companion-object members are accessed
// statically through the enclosing class (`C.TAG`) — mark them static
// so the field metadata reflects that (F52).
return isInsideKotlinCompanion(node);
},
isReadonly(node) {

View file

@ -2,7 +2,7 @@
import { SupportedLanguages } from 'gitnexus-shared';
import type { FieldExtractionConfig } from '../generic.js';
import { hasKeyword, findVisibility } from './helpers.js';
import { hasKeyword, hasModifier, findVisibility } from './helpers.js';
import { extractSimpleTypeName } from '../../type-extractors/shared.js';
import type { FieldVisibility } from '../../field-types.js';
@ -17,18 +17,33 @@ const SWIFT_VIS = new Set<FieldVisibility>([
/**
* Swift field extraction config.
*
* Handles property_declaration inside class_body / protocol_body.
* Handles property_declaration inside class_body / protocol_body and
* protocol_property_declaration inside protocol_body (F75 — protocol property
* requirements like "var title: String { get }").
*
* tree-sitter-swift uses property_declaration for stored/computed properties.
* A protocol property requirement parses to its own node type,
* protocol_property_declaration, whose name lives in a "name:" pattern field
* (pattern > value_binding_pattern + simple_identifier(bound_identifier)), its
* type in a sibling type_annotation, and its "{ get }" / "{ get set }" in a
* protocol_property_requirements child. Note: Swift reuses the "name:" field
* across many positions (func name, every parameter label, parameter/return
* type), so the name is synthesized from the simple_identifier inside the
* pattern rather than read blindly off "name:".
*/
export const swiftConfig: FieldExtractionConfig = {
language: SupportedLanguages.Swift,
typeDeclarationNodes: ['class_declaration', 'protocol_declaration'],
fieldNodeTypes: ['property_declaration'],
fieldNodeTypes: ['property_declaration', 'protocol_property_declaration'],
bodyNodeTypes: ['class_body', 'protocol_body'],
defaultVisibility: 'internal',
extractName(node) {
// property_declaration > pattern > simple_identifier
// property_declaration > pattern > simple_identifier, and
// protocol_property_declaration > name: (pattern ... simple_identifier).
// For protocol_property_declaration the pattern wraps a leading
// value_binding_pattern ("var") plus the simple_identifier — the loop
// below skips the binding keyword and returns the identifier.
for (let i = 0; i < node.namedChildCount; i++) {
const child = node.namedChild(i);
if (child?.type === 'pattern') {
@ -62,7 +77,19 @@ export const swiftConfig: FieldExtractionConfig = {
},
isStatic(node) {
return hasKeyword(node, 'static') || hasKeyword(node, 'class');
// `static`/`class` (type-level) modifiers live inside a `modifiers`
// wrapper for both property_declaration and protocol_property_declaration
// (e.g. `static var shared: P { get }`), so check the wrapper too.
// `hasKeyword` compares each direct child by `.text` equality: it matches a
// single-modifier wrapper (`modifiers.text === 'static'`) but fails for a
// multi-modifier wrapper (`private static` → `modifiers.text === 'private static'`),
// which `hasModifier` handles by descending into the wrapper's children.
return (
hasKeyword(node, 'static') ||
hasKeyword(node, 'class') ||
hasModifier(node, 'modifiers', 'static') ||
hasModifier(node, 'modifiers', 'class')
);
},
isReadonly(node) {

View file

@ -33,6 +33,10 @@ export interface FieldExtractionConfig {
bodyNodeTypes: string[];
/** Default visibility when no modifier is present */
defaultVisibility: FieldVisibility;
/** Extract owner type name from a type declaration node. */
extractOwnerName?: (node: SyntaxNode) => string | undefined;
/** Find body nodes inside a type declaration node. */
findBodyNodes?: (node: SyntaxNode) => SyntaxNode[];
/**
* Extract field name from a field declaration node.
* Use this for nodes that declare exactly one field.
@ -49,6 +53,8 @@ export interface FieldExtractionConfig {
extractType: (node: SyntaxNode) => string | undefined;
/** Extract visibility from a field declaration node */
extractVisibility: (node: SyntaxNode) => FieldVisibility;
/** Extract visibility for one field name from a multi-name declaration. */
extractVisibilityForName?: (node: SyntaxNode, name: string) => FieldVisibility;
/** Check if a field is static */
isStatic: (node: SyntaxNode) => boolean;
/** Check if a field is readonly/final/const */
@ -84,10 +90,9 @@ export function createFieldExtractor(config: FieldExtractionConfig): FieldExtrac
extract(node: SyntaxNode, context: FieldExtractorContext): ExtractedFields | null {
if (!this.isTypeDeclaration(node)) return null;
const nameNode = node.childForFieldName('name');
if (!nameNode) return null;
const ownerFqn = config.extractOwnerName?.(node) ?? node.childForFieldName('name')?.text;
if (!ownerFqn) return null;
const ownerFqn = nameNode.text;
const fields: FieldInfo[] = [];
// Find body container(s)
@ -110,6 +115,8 @@ export function createFieldExtractor(config: FieldExtractionConfig): FieldExtrac
// ------------------------------------------------------------------
private findBodies(node: SyntaxNode): SyntaxNode[] {
if (config.findBodyNodes) return config.findBodyNodes(node);
const result: SyntaxNode[] = [];
// Try named 'body' field first
const bodyField = node.childForFieldName('body');
@ -179,7 +186,7 @@ export function createFieldExtractor(config: FieldExtractionConfig): FieldExtrac
return {
name,
type,
visibility: config.extractVisibility(node),
visibility: config.extractVisibilityForName?.(node, name) ?? config.extractVisibility(node),
isStatic: config.isStatic(node),
isReadonly: config.isReadonly(node),
sourceFile: context.filePath,

View file

@ -6,10 +6,6 @@ import { glob } from 'glob';
import { createIgnoreFilter } from '../../config/ignore-service.js';
import { logger } from '../logger.js';
export interface FileEntry {
path: string;
content: string;
}
/** Lightweight entry — path + size from stat, no content in memory */
export interface ScannedFile {
@ -153,21 +149,3 @@ export const readFileContents = async (
return contents;
};
/**
* Legacy API — scans and reads everything into memory.
* Used by sequential fallback path only.
*/
export const walkRepository = async (
repoPath: string,
onProgress?: (current: number, total: number, filePath: string) => void,
): Promise<FileEntry[]> => {
const scanned = await walkRepositoryPaths(repoPath, onProgress);
const contents = await readFileContents(
repoPath,
scanned.map((f) => f.path),
);
return scanned
.filter((f) => contents.has(f.path))
.map((f) => ({ path: f.path, content: contents.get(f.path)! }));
};

View file

@ -51,6 +51,8 @@ import {
finalize,
} from 'gitnexus-shared';
import type { ScopeResolutionIndexes } from './model/scope-resolution-indexes.js';
import { parseTruthyEnv } from './utils/env.js';
import { TransitionalScopeTree } from '../../storage/scope-index-store.js';
// ─── Public entry point ─────────────────────────────────────────────────────
@ -114,7 +116,13 @@ export function finalizeScopeModel(
moduleEntries.push({ filePath: file.filePath, moduleScopeId: file.moduleScope });
}
const scopeTree = buildScopeTree(allScopes);
// Out-of-core scope index: when enabled, build a TransitionalScopeTree
// (validated + fully resident now; sealed to disk by run.ts just before emit so
// the heavy Scope.bindings payload is reclaimed). Default off → the in-heap
// buildScopeTree result exactly, byte-identical.
const scopeTree = parseTruthyEnv(process.env.GITNEXUS_DISK_SCOPE_INDEX)
? new TransitionalScopeTree(allScopes)
: buildScopeTree(allScopes);
const defs = buildDefIndex(allDefs);
const qualifiedNames = buildQualifiedNameIndex(allDefs);
const moduleScopes = buildModuleScopeIndex(moduleEntries);

View file

@ -311,6 +311,27 @@ interface LanguageProviderConfig {
},
) => readonly CaptureMatch[];
/**
* Snapshot the capture-time side-channel state that this provider's
* `emitScopeCaptures` just populated for `filePath` into module-level maps,
* returning a plain JSON-serializable value (or `undefined` when there is
* nothing to carry).
*
* Called in the parse worker IMMEDIATELY after `emitScopeCaptures` runs for
* a file (see `parse-worker.ts`), and the result is stored on the produced
* `ParsedFile.captureSideChannel`. Scope-resolution on the main thread reuses
* that serialized `ParsedFile` and skips re-extraction (#1983), so this hook
* is how the worker-computed marks survive the worker→main boundary and the
* disk store WITHOUT a main-thread re-parse. The main thread restores them
* via the matching `ScopeResolver.applyCaptureSideChannel` hook.
*
* MUST return plain data (objects / arrays / primitives) so it round-trips
* through `JSON.stringify` + the parsedfile-store interning reviver.
*
* Default: undefined (provider has no capture-time module-level side effects).
*/
readonly collectCaptureSideChannel?: (filePath: string) => unknown;
/**
* Interpret a raw `@import.statement` capture group into a `ParsedImport`.
* The central finalize algorithm resolves `ParsedImport.targetRaw` to a

View file

@ -53,6 +53,7 @@ import {
cBindingScopeFor,
cImportOwningScope,
cReceiverBinding,
collectCStaticLinkageSideChannel,
} from './c/index.js';
import {
emitCppScopeCaptures,
@ -62,6 +63,7 @@ import {
cppBindingScopeFor,
cppImportOwningScope,
cppReceiverBinding,
collectCppCaptureSideChannel,
} from './cpp/index.js';
import { extractCppTemplateConstraints } from './cpp/constraint-extractor.js';
@ -395,6 +397,15 @@ export const cProvider = defineLanguage({
// ── RFC #909 Ring 3: scope-based resolution hooks (RFC §5) ──────────
emitScopeCaptures: emitCScopeCaptures,
// Worker-side: snapshot the module-level `static`-linkage marks
// `emitCScopeCaptures` just populated for this file (`markStaticName` →
// `staticNames`) into plain data on `ParsedFile.captureSideChannel`, so the
// main thread can restore them via `applyCaptureSideChannel` WITHOUT a
// re-parse (#1983 — the worker is the sole parse path). Without this, C
// `static` functions look non-file-local on the main thread and leak into
// cross-file global free-call resolution / wildcard imports. See
// `c/capture-side-channel.ts`.
collectCaptureSideChannel: collectCStaticLinkageSideChannel,
interpretImport: interpretCImport,
interpretTypeBinding: interpretCTypeBinding,
bindingScopeFor: cBindingScopeFor,
@ -465,6 +476,11 @@ export const cppProvider = defineLanguage({
// ── RFC #909 Ring 3: scope-based resolution hooks (RFC §5) ──────────
emitScopeCaptures: emitCppScopeCaptures,
// Worker-side: snapshot the module-level capture marks `emitCppScopeCaptures`
// just populated for this file into plain data on `ParsedFile.captureSideChannel`,
// so the main thread can restore them via `applyCaptureSideChannel` WITHOUT a
// re-parse (#1983). See `cpp/capture-side-channel.ts`.
collectCaptureSideChannel: collectCppCaptureSideChannel,
interpretImport: interpretCppImport,
interpretTypeBinding: interpretCppTypeBinding,
bindingScopeFor: cppBindingScopeFor,

View file

@ -0,0 +1,80 @@
/**
* C capture-time side-channel serialization (#1983).
*
* `emitCScopeCaptures` populates one MODULE-LEVEL, per-file map as a side
* effect that is NOT part of the returned `ParsedFile`'s scopes/defs:
*
* - `staticNames` (static-linkage.ts) — the simple names of functions
* declared with `static` storage class (file-local / translation-unit
* linkage in C), recorded via `markStaticName` from the
* `@declaration.name` capture when the function node has a `static`
* storage-class specifier.
*
* On the worker path that map is filled in the WORKER process and lost across
* the worker→main MessageChannel (and the disk-backed parsedfile-store),
* because scope-resolution reuses the serialized `ParsedFile` and SKIPS the
* main-thread re-extraction (the #1983 fix that avoids a main-thread
* tree-sitter re-parse / OOM on huge repos — e.g. the Linux kernel). The main
* thread then reads the map empty in `isStaticName` (consulted by
* `isFileLocalDef` in `c/scope-resolver.ts` and by `expandCWildcardNames` in
* static-linkage.ts) — so file-local `static` functions become eligible for
* cross-file global free-call resolution (false CALLS edges) and `#include`
* wildcard imports over-expose them.
*
* This module snapshots the per-file slice of that map into a plain,
* JSON-serializable object (carried on `ParsedFile.captureSideChannel`) and
* restores it on the main thread WITHOUT any parse. It mirrors the C++ pattern
* in `cpp/capture-side-channel.ts` and the Kotlin pattern in
* `kotlin/capture-side-channel.ts`.
*
* The single generic `ParsedFile.captureSideChannel` field is shared with C++
* and Kotlin, which is safe because each file is one language (a `.c` file uses
* the C provider). The payload is self-describing (`{ kind: 'c', staticNames }`)
* so `applyCStaticLinkageSideChannel` only restores C state and ignores a
* foreign-shaped snapshot.
*/
import type { ParsedFile } from 'gitnexus-shared';
import { getStaticNamesForFile, markStaticName } from './static-linkage.js';
/**
* Plain JSON-serializable snapshot of the per-file C capture-time
* side-channel. Carried opaquely on `ParsedFile.captureSideChannel`. The
* `kind` tag makes the payload self-describing so `apply` can distinguish a C
* snapshot from another language's (C++ and Kotlin share the same field).
*/
export interface CCaptureSideChannel {
readonly kind: 'c';
/** Simple names of `static` (file-local linkage) functions in this file. */
readonly staticNames: readonly string[];
}
/**
* `LanguageProvider.collectCaptureSideChannel` implementation for C.
* Returns `undefined` when this file recorded no static names at all, so the
* produced `ParsedFile` carries the field only when there's data to ship.
*/
export function collectCStaticLinkageSideChannel(
filePath: string,
): CCaptureSideChannel | undefined {
const staticNames = getStaticNamesForFile(filePath);
if (staticNames.length === 0) return undefined;
return { kind: 'c', staticNames };
}
/**
* `ScopeResolver.applyCaptureSideChannel` implementation for C. Reads the
* worker-serialized snapshot from `parsed.captureSideChannel` and re-populates
* the module-level static-linkage map via `markStaticName`. Tolerant of
* `undefined` (file carried no data) and of an unexpected / foreign shape
* (defensive — the `kind` tag guards against restoring a non-C payload).
* Does NO tree-sitter parse.
*/
export function applyCStaticLinkageSideChannel(parsed: ParsedFile): void {
const data = parsed.captureSideChannel as CCaptureSideChannel | undefined;
if (data === undefined || data === null || typeof data !== 'object') return;
if (data.kind !== 'c' || !Array.isArray(data.staticNames)) return;
for (const name of data.staticNames) {
markStaticName(parsed.filePath, name);
}
}

View file

@ -5,10 +5,20 @@ import { nodeToCapture, syntheticCapture, type SyntaxNode } from '../../utils/as
* Decompose a `preproc_include` node into a CaptureMatch with structured
* import captures. C #include maps to a wildcard import (all symbols
* from the header are visible).
*
* Only literal include paths are emitted as import sources:
* #include <stdio.h> → system_lib_string
* #include "local.h" → string_literal
* A computed include like `#include HEADER_MACRO` carries an `identifier`
* path node (the macro name, not a header path). Emitting it as an import
* source produces a garbage literal edge, so we skip it entirely — matching
* the convention in interpretCImport, which drops imports with no resolvable
* source (issue #1919 F5).
*/
export function splitCInclude(node: SyntaxNode): CaptureMatch | null {
// node.type === 'preproc_include'
// path field: (string_literal (string_content)) | (system_lib_string)
// | (identifier) ← computed macro include, NOT a header path
const pathNode = node.childForFieldName?.('path') ?? null;
if (pathNode === null) {
// Fallback: scan children
@ -24,7 +34,13 @@ export function splitCInclude(node: SyntaxNode): CaptureMatch | null {
return buildIncludeCapture(node, pathNode);
}
function buildIncludeCapture(node: SyntaxNode, pathNode: SyntaxNode): CaptureMatch {
function buildIncludeCapture(node: SyntaxNode, pathNode: SyntaxNode): CaptureMatch | null {
// Skip computed includes (`#include MACRO`) — the path is an `identifier`,
// not a literal header path. Emitting it would create a garbage import.
if (pathNode.type !== 'string_literal' && pathNode.type !== 'system_lib_string') {
return null;
}
let raw: string;
if (pathNode.type === 'string_literal') {
// string_literal has children: `"`, string_content, `"`

View file

@ -1,5 +1,54 @@
import { dirname, join } from 'path';
/**
* A workspace file path pre-decomposed for the suffix-match fallback:
* `original` is returned verbatim (preserving the prior `bestMatch = filePath`
* contract); `normalized` and `depth` are precomputed so the hot path does no
* per-element regex/`split`.
*/
interface CSuffixCandidate {
original: string;
normalized: string;
depth: number;
}
/**
* Per-pass memo: workspace paths bucketed by basename (last path segment),
* keyed on the `allFilePaths` set identity.
*
* `resolveCImportTarget` is called once per (quoted) C/C++ `#include` with the
* same `allFilePaths` set per pass (the augmented set is itself memoized in
* the C resolver). The old suffix-match fallback scanned ALL workspace paths
* per include — with a per-element `.replace`/`.split` and no early exit
* (the fewest-path-components tie-break forces a full scan) — i.e.
* O(R_suffix × (F+H)). A path can satisfy `endsWith('/'+target)` (or equal
* the target) ONLY IF its basename equals the target's last segment, so we
* pre-bucket by basename once (O(F+H), `normalized`/`depth` precomputed) and
* the fallback inspects a single small bucket → O(F+H) build + ~O(1)/include.
* `WeakMap`-keyed so it is reclaimed with the pass (no cross-pass staleness).
* Shared by C and C++ (`resolveCppImportTarget` delegates here).
*/
const suffixIndexByPaths = new WeakMap<ReadonlySet<string>, Map<string, CSuffixCandidate[]>>();
function suffixIndex(allFilePaths: ReadonlySet<string>): Map<string, CSuffixCandidate[]> {
let index = suffixIndexByPaths.get(allFilePaths);
if (index === undefined) {
index = new Map<string, CSuffixCandidate[]>();
for (const original of allFilePaths) {
const normalized = original.replace(/\\/g, '/');
const basename = normalized.slice(normalized.lastIndexOf('/') + 1);
let bucket = index.get(basename);
if (bucket === undefined) {
bucket = [];
index.set(basename, bucket);
}
bucket.push({ original, normalized, depth: normalized.split('/').length });
}
suffixIndexByPaths.set(allFilePaths, index);
}
return index;
}
/**
* Resolve a C #include path to a file in the workspace.
*
@ -41,21 +90,31 @@ export function resolveCImportTarget(
// Exact match (path as-is in the workspace)
if (allFilePaths.has(normalizedTarget)) return normalizedTarget;
// Suffix match: find files ending with /targetRaw or equal to targetRaw
// Suffix match: find files ending with /targetRaw or equal to targetRaw.
// A path can only match `=== normalizedTarget` or `endsWith('/'+target)` if
// its basename equals the target's last segment, so we inspect only that
// basename bucket (built once per pass) instead of scanning every workspace
// path. Match condition + tie-break (fewest path components, then
// lexicographic on the normalized path) are byte-identical to the prior scan.
const suffix = '/' + normalizedTarget;
const targetBasename = normalizedTarget.slice(normalizedTarget.lastIndexOf('/') + 1);
const bucket = suffixIndex(allFilePaths).get(targetBasename);
if (bucket === undefined) return null;
let bestMatch: string | null = null;
let bestDepth = Infinity;
let bestNormalized = '';
for (const filePath of allFilePaths) {
const normalized = filePath.replace(/\\/g, '/');
if (normalized === normalizedTarget || normalized.endsWith(suffix)) {
for (const cand of bucket) {
if (cand.normalized === normalizedTarget || cand.normalized.endsWith(suffix)) {
// Prefer shortest path (closest match)
const depth = normalized.split('/').length;
if (depth < bestDepth || (depth === bestDepth && normalized < bestNormalized)) {
bestDepth = depth;
bestMatch = filePath;
bestNormalized = normalized;
if (
cand.depth < bestDepth ||
(cand.depth === bestDepth && cand.normalized < bestNormalized)
) {
bestDepth = cand.depth;
bestMatch = cand.original;
bestNormalized = cand.normalized;
}
}
}

View file

@ -13,4 +13,9 @@ export {
isStaticName,
clearStaticNames,
expandCWildcardNames,
getStaticNamesForFile,
} from './static-linkage.js';
export {
collectCStaticLinkageSideChannel,
applyCStaticLinkageSideChannel,
} from './capture-side-channel.js';

View file

@ -7,6 +7,43 @@ import { cProvider } from '../c-cpp.js';
import { cArityCompatibility, cMergeBindings, resolveCImportTarget } from './index.js';
import { scanHeaderFiles } from './header-scan.js';
import { expandCWildcardNames, isStaticName, clearStaticNames } from './static-linkage.js';
import { applyCStaticLinkageSideChannel } from './capture-side-channel.js';
/**
* Per-pass memo of the augmented `#include`-resolution file set
* (`allFilePaths` ∪ header `.h` paths), keyed on the two stable source sets.
*
* `resolveImportTarget` is called once per C `#include`; the old code rebuilt
* a fresh ~F-entry `Set` on EVERY call (O(R × (F+H)) inserts + GC churn) and,
* worse, defeated `resolveCImportTarget`'s own per-set suffix-index memo by
* handing it a new set identity each time. Both `allFilePaths` (built once in
* scope-resolution `run.ts`) and the header set (`loadResolutionConfig`
* result) are stable per pass, so the union is built once and reused.
* `WeakMap`-keyed → reclaimed with the pass (no cross-pass staleness).
*/
const augmentedPathsByPass = new WeakMap<
ReadonlySet<string>,
WeakMap<ReadonlySet<string>, ReadonlySet<string>>
>();
function augmentedFilePaths(
allFilePaths: ReadonlySet<string>,
headerPaths: ReadonlySet<string>,
): ReadonlySet<string> {
let byHeaders = augmentedPathsByPass.get(allFilePaths);
if (byHeaders === undefined) {
byHeaders = new WeakMap();
augmentedPathsByPass.set(allFilePaths, byHeaders);
}
let augmented = byHeaders.get(headerPaths);
if (augmented === undefined) {
const set = new Set(allFilePaths);
for (const h of headerPaths) set.add(h);
augmented = set;
byHeaders.set(headerPaths, augmented);
}
return augmented;
}
/**
* C `ScopeResolver` registered in `SCOPE_RESOLVERS` and consumed by
@ -31,15 +68,34 @@ export const cScopeResolver: ScopeResolver = {
return scanHeaderFiles(repoPath);
},
// Worker-boundary restore (see `ScopeResolver.applyCaptureSideChannel`).
// `emitCScopeCaptures` records per-file `static`-linkage names
// (`markStaticName` → `staticNames`) as a SIDE EFFECT — that state is NOT
// serialized onto the returned ParsedFile's scopes/defs. On the worker path
// those marks are populated in the worker process and lost across the
// MessageChannel / disk store; the main thread reuses the serialized
// ParsedFile and skips `extractParsedFile`, so `isStaticName` (read by
// `isFileLocalDef` and `expandCWildcardNames`) sees an empty map and C
// `static` functions leak into cross-file global free-call resolution
// (false CALLS edges) and `#include` wildcard imports. The worker stashed a
// plain-data snapshot on `parsed.captureSideChannel` via
// `cProvider.collectCaptureSideChannel`; this restores it into the module
// map WITHOUT any tree-sitter re-parse (the #1983 fix). The
// freshly-extracted leg never calls this — its marks were just populated in
// this process. Runs BEFORE `populateOwners`.
applyCaptureSideChannel: applyCStaticLinkageSideChannel,
resolveImportTarget: (targetRaw, fromFile, allFilePaths, resolutionConfig) => {
// Augment allFilePaths with .h files discovered via loadResolutionConfig
// since the phase only passes .c files to the C resolver but #include
// targets .h files classified as C++ in language detection.
const headerPaths = resolutionConfig as ReadonlySet<string> | undefined;
if (headerPaths !== undefined && headerPaths.size > 0) {
const augmented = new Set(allFilePaths);
for (const h of headerPaths) augmented.add(h);
return resolveCImportTarget(targetRaw, fromFile, augmented);
return resolveCImportTarget(
targetRaw,
fromFile,
augmentedFilePaths(allFilePaths, headerPaths),
);
}
return resolveCImportTarget(targetRaw, fromFile, allFilePaths);
},

View file

@ -29,11 +29,58 @@ export function isStaticName(filePath: string, name: string): boolean {
return staticNames.get(filePath)?.has(name) ?? false;
}
/**
* Return the `static` (file-local) names recorded for the given file as a
* plain array (empty when none). Used to snapshot the per-file slice of the
* module-level `staticNames` map into `ParsedFile.captureSideChannel` so it
* survives the worker→main boundary (#1983 — the worker is the sole parse
* path). See `c/capture-side-channel.ts`.
*/
export function getStaticNamesForFile(filePath: string): string[] {
const names = staticNames.get(filePath);
return names === undefined ? [] : [...names];
}
/** Clear tracked static names (for testing). */
export function clearStaticNames(): void {
staticNames.clear();
}
/**
* Per-pass memo: `moduleScope` → owning `ParsedFile`, keyed on the
* `parsedFiles` array identity.
*
* The shared finalize Phase-4 loop calls `expandsWildcardTo`
* (→ `expandCWildcardNames`) ONCE PER RESOLVED `#include` edge, every time
* with the SAME `parsedFiles` reference (wired at scope-resolution
* `run.ts` — `allFilePaths`/`parsedFiles` are built once per pass). The old
* `parsedFiles.find(...)` therefore did a full O(F) scan per edge →
* O(R_include × F) overall; at Linux-kernel scale (F ≈ 63k C files, tens of
* thousands of resolved includes) that is ~10^10+ comparisons on a single
* thread — the dominant term in the scope-resolution finalize grind.
*
* Building the lookup once collapses it to O(R_include + F). `WeakMap`-keyed
* on the array so the index is reclaimed with the pass — no cross-pass
* staleness (mirrors the {@link clearStaticNames} discipline for server-mode
* / multi-repo reuse), and a fresh array transparently rebuilds.
*/
const moduleScopeIndexByPass = new WeakMap<readonly ParsedFile[], Map<ScopeId, ParsedFile>>();
function moduleScopeIndex(parsedFiles: readonly ParsedFile[]): Map<ScopeId, ParsedFile> {
let index = moduleScopeIndexByPass.get(parsedFiles);
if (index === undefined) {
index = new Map<ScopeId, ParsedFile>();
// First-wins to preserve `Array.find` semantics (returns the first match).
// `moduleScope` is unique per file in practice, so collisions are absent;
// the guard only formalises identical behaviour to the prior `.find`.
for (const p of parsedFiles) {
if (!index.has(p.moduleScope)) index.set(p.moduleScope, p);
}
moduleScopeIndexByPass.set(parsedFiles, index);
}
return index;
}
/**
* Return the names visible through a C wildcard import (`#include`).
* All module-scope defs from the target file are visible EXCEPT those
@ -43,7 +90,7 @@ export function expandCWildcardNames(
targetModuleScope: ScopeId,
parsedFiles: readonly ParsedFile[],
): readonly string[] {
const target = parsedFiles.find((p) => p.moduleScope === targetModuleScope);
const target = moduleScopeIndex(parsedFiles).get(targetModuleScope);
if (target === undefined) return [];
const seen = new Set<string>();

View file

@ -108,6 +108,38 @@ const argInfoBySite = new Map<string, readonly CppAdlArgInfo[]>();
const noAdlSites = new Set<string>();
const classToNamespaceQualifiedName = new Map<string, string>();
/**
* Per-`filePath` index of the site keys this file contributed to
* `argInfoBySite` / `noAdlSites`, kept in **strict lockstep** with those two
* maps (#1983 perf). Without it, `collectCppAdlSideChannel(filePath)` had to
* scan the ENTIRE module-level maps (every site of every file the worker
* parsed in the current sub-batch) and `parseSiteKey` each entry just to pick
* out one file's slice — O(F²) per sub-batch (~100M `parseSiteKey` calls
* across the Linux kernel). These indexes turn collect into
* O(entries-for-this-file).
*
* Lockstep invariant: a key is pushed here at most once, exactly when it is
* first inserted into the corresponding map, and both indexes are cleared
* wherever `argInfoBySite` / `noAdlSites` are cleared (`clearCppAdlState` and
* the per-file restore in `applyCppAdlSideChannel`). The "first insert only"
* guard mirrors the maps' own de-dup (`Map.set` / `Set.add` are idempotent on
* the key), so iterating an index yields each of this file's keys exactly once
* — byte-identical to the old filtered full scan.
*/
const argInfoSiteKeysByFile = new Map<string, string[]>();
const noAdlSiteKeysByFile = new Map<string, string[]>();
/** Push `key` into the per-file index `idx[filePath]` (creating the bucket on
* first use). Callers guard against duplicate keys so each key appears once. */
function pushFileSiteKey(idx: Map<string, string[]>, filePath: string, key: string): void {
let keys = idx.get(filePath);
if (keys === undefined) {
keys = [];
idx.set(filePath, keys);
}
keys.push(key);
}
/**
* ADL candidate index — built **once** per pipeline run from
* `(scopes, parsedFiles)` and reused by every call site.
@ -370,13 +402,94 @@ export function markCppAdlSiteArgs(
col: number,
args: readonly CppAdlArgInfo[],
): void {
argInfoBySite.set(siteKey(filePath, line, col), args);
const key = siteKey(filePath, line, col);
// Lockstep with `argInfoSiteKeysByFile`: index the key only on first insert
// (a re-mark overwrites the value but must NOT duplicate the index entry).
if (!argInfoBySite.has(key)) pushFileSiteKey(argInfoSiteKeysByFile, filePath, key);
argInfoBySite.set(key, args);
}
/** Mark a call site as ADL-suppressed (function child wrapped in
* `parenthesized_expression`, e.g. `(f)(s)`). */
export function markCppAdlSiteNoAdl(filePath: string, line: number, col: number): void {
noAdlSites.add(siteKey(filePath, line, col));
const key = siteKey(filePath, line, col);
// Lockstep with `noAdlSiteKeysByFile`: index the key only on first insert.
if (!noAdlSites.has(key)) pushFileSiteKey(noAdlSiteKeysByFile, filePath, key);
noAdlSites.add(key);
}
/**
* Plain-data, JSON-serializable snapshot of the per-file ADL capture state
* (`argInfoBySite` entries for this file + `noAdlSites` keys for this file).
* Carried on `ParsedFile.captureSideChannel` across the worker→main boundary
* (#1983); the call-site key's `line:col` are stored per-entry so the full
* `filePath:line:col` key can be reconstructed without parsing.
*/
export interface CppAdlSideChannel {
/** Per-call-site arg info: `[line, col, args]` for sites in this file. */
readonly argInfoBySite: readonly [number, number, readonly CppAdlArgInfo[]][];
/** ADL-suppressed sites in this file: `[line, col]`. */
readonly noAdlSites: readonly [number, number][];
}
const SITE_KEY_RE = /^(.*):(\d+):(\d+)$/;
/** Split a `filePath:line:col` site key, tolerating colons in the path. */
function parseSiteKey(key: string): { filePath: string; line: number; col: number } | undefined {
const m = SITE_KEY_RE.exec(key);
if (m === null) return undefined;
return { filePath: m[1], line: Number(m[2]), col: Number(m[3]) };
}
/**
* Snapshot this file's ADL capture state for the worker→main side-channel.
*
* Uses the per-file `argInfoSiteKeysByFile` / `noAdlSiteKeysByFile` indexes to
* touch only THIS file's entries — O(entries-for-this-file) — instead of the
* old O(all-entries) full scan over `argInfoBySite` / `noAdlSites` (#1983).
* The output order, and therefore the serialized JSON shape, is byte-identical
* to the old filtered scan: the index records keys in the same insertion order
* the maps' own iteration would have yielded for this file, and each key is
* indexed exactly once (mark guards on first insert), so the same per-file
* subsequence is produced.
*
* `parseSiteKey` is still used to recover `line:col` from each key, but now
* only for this file's keys (a bounded handful), never for the whole batch.
*/
export function collectCppAdlSideChannel(filePath: string): CppAdlSideChannel {
const args: [number, number, readonly CppAdlArgInfo[]][] = [];
for (const key of argInfoSiteKeysByFile.get(filePath) ?? []) {
const value = argInfoBySite.get(key);
const parsed = parseSiteKey(key);
if (value !== undefined && parsed !== undefined) {
args.push([parsed.line, parsed.col, value]);
}
}
const noAdl: [number, number][] = [];
for (const key of noAdlSiteKeysByFile.get(filePath) ?? []) {
const parsed = parseSiteKey(key);
if (parsed !== undefined) {
noAdl.push([parsed.line, parsed.col]);
}
}
return { argInfoBySite: args, noAdlSites: noAdl };
}
/** Restore this file's ADL capture state from the side-channel (no parse).
* Keeps the per-file site-key indexes in lockstep with `argInfoBySite` /
* `noAdlSites` (first-insert-only) so a later `collectCppAdlSideChannel` on
* the same process would still produce a correct, duplicate-free snapshot. */
export function applyCppAdlSideChannel(filePath: string, data: CppAdlSideChannel): void {
for (const [line, col, value] of data.argInfoBySite) {
const key = siteKey(filePath, line, col);
if (!argInfoBySite.has(key)) pushFileSiteKey(argInfoSiteKeysByFile, filePath, key);
argInfoBySite.set(key, value);
}
for (const [line, col] of data.noAdlSites) {
const key = siteKey(filePath, line, col);
if (!noAdlSites.has(key)) pushFileSiteKey(noAdlSiteKeysByFile, filePath, key);
noAdlSites.add(key);
}
}
/** Clear ADL state. Called from `cppScopeResolver.loadResolutionConfig`
@ -385,6 +498,11 @@ export function markCppAdlSiteNoAdl(filePath: string, line: number, col: number)
export function clearCppAdlState(): void {
argInfoBySite.clear();
noAdlSites.clear();
// Lockstep: the per-file site-key indexes mirror argInfoBySite/noAdlSites and
// MUST be cleared together — a stale index would resurrect a prior pass's
// (or prior file's, after a re-key) keys into the next snapshot.
argInfoSiteKeysByFile.clear();
noAdlSiteKeysByFile.clear();
classToNamespaceQualifiedName.clear();
adlIndex = undefined;
adlIndexSource = undefined;

View file

@ -0,0 +1,123 @@
/**
* C++ capture-time side-channel serialization (#1983).
*
* `emitCppScopeCaptures` populates several MODULE-LEVEL maps as a side effect
* that are NOT part of the returned `ParsedFile`'s scopes/defs:
*
* - `argInfoBySite` / `noAdlSites` (adl.ts)
* - `inlineNamespaceRangesByFile` (inline-namespaces.ts)
* - `fileLocalNames` / `anonymousNamespaceRangesByFile` (file-local-linkage.ts)
* - `dependentBasesByFile` / `dependentPackBaseClassesByFile` (two-phase-lookup.ts)
*
* On the worker path those maps are filled in the WORKER process and lost
* across the worker→main MessageChannel (and the disk-backed parsedfile-store),
* because scope-resolution reuses the serialized `ParsedFile` and SKIPS the
* main-thread re-extraction — the entire point of #1983 is to avoid a
* main-thread tree-sitter re-parse on huge `.h`/`.cpp` repos (the OOM).
*
* This module snapshots the per-file slice of those maps into a plain,
* JSON-serializable object (carried on `ParsedFile.captureSideChannel`) and
* restores it on the main thread WITHOUT any parse. It is the data-only
* replacement for the removed re-parse `replayCaptureSideChannel` hook.
*
* The derived state each `populateOwners` / `populateWorkspaceOwners` pass
* builds (resolved scope-id Sets, `dependentBaseNodeIds`, etc.) is recomputed
* on the main thread from these restored capture-time maps, so only the
* capture-time maps need to cross the boundary.
*/
import type { ParsedFile } from 'gitnexus-shared';
import { collectCppAdlSideChannel, applyCppAdlSideChannel, type CppAdlSideChannel } from './adl.js';
import {
collectCppInlineNamespaceSideChannel,
applyCppInlineNamespaceSideChannel,
} from './inline-namespaces.js';
import {
collectCppFileLocalSideChannel,
applyCppFileLocalSideChannel,
type CppFileLocalSideChannel,
} from './file-local-linkage.js';
import {
collectCppTwoPhaseSideChannel,
applyCppTwoPhaseSideChannel,
type CppTwoPhaseSideChannel,
} from './two-phase-lookup.js';
import {
applyCppMemberLookupSideChannel,
collectCppMemberLookupSideChannel,
type CppMemberLookupSideChannel,
} from './member-lookup.js';
/**
* Plain JSON-serializable composite of every C++ capture-time side-channel
* slice for one file. Carried opaquely on `ParsedFile.captureSideChannel`.
*/
export interface CppCaptureSideChannel {
/**
* Discriminant tag — the single generic `ParsedFile.captureSideChannel`
* field is shared with C (`{ kind: 'c' }`) and Kotlin (`{ kind: 'kotlin' }`).
* `applyCppCaptureSideChannel` checks this first so a foreign-language
* payload reaching the C++ apply (or vice-versa) is cleanly ignored. In
* practice apply only runs for the matching provider (one language per file),
* but the tag makes it robust and consistent with the C/Kotlin snapshots.
*/
readonly kind: 'cpp';
readonly adl: CppAdlSideChannel;
/** Inline-namespace source-range keys recorded for this file. */
readonly inlineNamespaceRanges: readonly string[];
readonly fileLocal: CppFileLocalSideChannel;
readonly twoPhase: CppTwoPhaseSideChannel;
readonly memberLookup: CppMemberLookupSideChannel;
}
/**
* `LanguageProvider.collectCaptureSideChannel` implementation for C++.
* Returns `undefined` when this file recorded no side-channel state at all, so
* the produced `ParsedFile` carries the field only when there's data to ship.
*/
export function collectCppCaptureSideChannel(filePath: string): CppCaptureSideChannel | undefined {
const adl = collectCppAdlSideChannel(filePath);
const inlineNamespaceRanges = collectCppInlineNamespaceSideChannel(filePath);
const fileLocal = collectCppFileLocalSideChannel(filePath);
const twoPhase = collectCppTwoPhaseSideChannel(filePath);
const memberLookup = collectCppMemberLookupSideChannel(filePath);
const isEmpty =
adl.argInfoBySite.length === 0 &&
adl.noAdlSites.length === 0 &&
inlineNamespaceRanges.length === 0 &&
fileLocal.fileLocalNames.length === 0 &&
fileLocal.anonymousNamespaceRanges.length === 0 &&
twoPhase.dependentBases.length === 0 &&
twoPhase.dependentPackBaseClasses.length === 0 &&
memberLookup.baseEdges.length === 0 &&
memberLookup.memberUsings.length === 0;
if (isEmpty) return undefined;
return { kind: 'cpp', adl, inlineNamespaceRanges, fileLocal, twoPhase, memberLookup };
}
/**
* `ScopeResolver.applyCaptureSideChannel` implementation for C++. Reads the
* worker-serialized snapshot from `parsed.captureSideChannel` and writes it
* back into the module-level maps. Tolerant of `undefined` (file carried no
* data) and of an unexpected shape (defensive — never throws on a malformed
* snapshot). Does NO tree-sitter parse.
*/
export function applyCppCaptureSideChannel(parsed: ParsedFile): void {
const data = parsed.captureSideChannel as CppCaptureSideChannel | undefined;
if (data === undefined || data === null || typeof data !== 'object') return;
// Discriminant guard — the generic `captureSideChannel` field is shared
// with C (`{ kind: 'c' }`) and Kotlin (`{ kind: 'kotlin' }`); cleanly
// ignore a non-C++ payload rather than mis-applying it.
if (data.kind !== 'cpp') return;
if (data.adl !== undefined) applyCppAdlSideChannel(parsed.filePath, data.adl);
if (data.inlineNamespaceRanges !== undefined) {
applyCppInlineNamespaceSideChannel(parsed.filePath, data.inlineNamespaceRanges);
}
if (data.fileLocal !== undefined) applyCppFileLocalSideChannel(parsed.filePath, data.fileLocal);
if (data.twoPhase !== undefined) applyCppTwoPhaseSideChannel(parsed.filePath, data.twoPhase);
if (data.memberLookup !== undefined) {
applyCppMemberLookupSideChannel(parsed.filePath, data.memberLookup);
}
}

View file

@ -20,6 +20,7 @@ import { markCppDependentBase, markCppDependentPackBase } from './two-phase-look
import { markCppAdlSiteArgs, markCppAdlSiteNoAdl, type CppAdlArgInfo } from './adl.js';
import { markCppInlineNamespaceRange } from './inline-namespaces.js';
import { extractCppTemplateConstraints } from './constraint-extractor.js';
import { captureCppMemberLookupFacts } from './member-lookup.js';
export function emitCppScopeCaptures(
sourceText: string,
@ -464,6 +465,7 @@ export function emitCppScopeCaptures(
// and the resolver can suppress unqualified-call binding to those
// bases per ISO C++ two-phase lookup.
detectCppDependentBases(tree.rootNode, filePath);
captureCppMemberLookupFacts(tree.rootNode, filePath);
return out;
}

View file

@ -95,6 +95,47 @@ export function isCppAnonymousNamespaceScope(scopeId: ScopeId): boolean {
return anonymousNamespaceScopeIds.has(scopeId);
}
/**
* Plain-data, JSON-serializable snapshot of the per-file capture-time
* file-local-linkage state. Carried on `ParsedFile.captureSideChannel` across
* the worker→main boundary (#1983). The derived sets (`nonGloballyVisibleNodeIds`,
* `anonymousNamespaceScopeIds`) are recomputed by `populateCppNonGloballyVisible`
* / `populateCppAnonymousNamespaceScopes` during `populateOwners`, so only the
* two capture-time maps cross the boundary.
*/
export interface CppFileLocalSideChannel {
/** File-local symbol names (static / anonymous-namespace) in this file. */
readonly fileLocalNames: readonly string[];
/** Anonymous-namespace source-range keys recorded for this file. */
readonly anonymousNamespaceRanges: readonly string[];
}
/** Snapshot this file's file-local-linkage capture state for the side-channel. */
export function collectCppFileLocalSideChannel(filePath: string): CppFileLocalSideChannel {
const names = fileLocalNames.get(filePath);
const anon = anonymousNamespaceRangesByFile.get(filePath);
return {
fileLocalNames: names === undefined ? [] : [...names],
anonymousNamespaceRanges: anon === undefined ? [] : [...anon],
};
}
/** Restore this file's file-local-linkage capture state from the side-channel. */
export function applyCppFileLocalSideChannel(
filePath: string,
data: CppFileLocalSideChannel,
): void {
for (const name of data.fileLocalNames) markFileLocal(filePath, name);
if (data.anonymousNamespaceRanges.length > 0) {
let set = anonymousNamespaceRangesByFile.get(filePath);
if (set === undefined) {
set = new Set();
anonymousNamespaceRangesByFile.set(filePath, set);
}
for (const r of data.anonymousNamespaceRanges) set.add(r);
}
}
/** Clear tracked file-local names (call at start of each resolution pass). */
export function clearFileLocalNames(): void {
fileLocalNames.clear();
@ -235,11 +276,37 @@ export function isCppDefGloballyVisible(filePath: string, nodeId: string): boole
* does, mirror this filter or harden registration so class/namespace
* members never enter `localDefs` unqualified.
*/
/**
* Per-pass memo: `moduleScope` → owning `ParsedFile`, keyed on the
* `parsedFiles` array identity. The shared finalize Phase-4 loop calls
* `expandsWildcardTo` (→ this) ONCE PER RESOLVED `#include` edge with the same
* `parsedFiles` reference; the old `parsedFiles.find(...)` was therefore O(F)
* per edge → O(R·F) overall (at kernel scale the ~25–30k `.h` headers are
* classified C++, so this fires hard — the C twin in `c/static-linkage.ts`).
* Building the lookup once collapses it to O(R+F). `WeakMap`-keyed so it is
* reclaimed with the pass (no cross-pass staleness; mirrors
* {@link clearFileLocalNames}).
*/
const moduleScopeIndexByPass = new WeakMap<readonly ParsedFile[], Map<ScopeId, ParsedFile>>();
function moduleScopeIndex(parsedFiles: readonly ParsedFile[]): Map<ScopeId, ParsedFile> {
let index = moduleScopeIndexByPass.get(parsedFiles);
if (index === undefined) {
index = new Map<ScopeId, ParsedFile>();
// First-wins to preserve `Array.find` semantics (returns the first match).
for (const p of parsedFiles) {
if (!index.has(p.moduleScope)) index.set(p.moduleScope, p);
}
moduleScopeIndexByPass.set(parsedFiles, index);
}
return index;
}
export function expandCppWildcardNames(
targetModuleScope: ScopeId,
parsedFiles: readonly ParsedFile[],
): readonly string[] {
const target = parsedFiles.find((p) => p.moduleScope === targetModuleScope);
const target = moduleScopeIndex(parsedFiles).get(targetModuleScope);
if (target === undefined) return [];
// Build nodeId → owning Scope map from the structural scope tree.

View file

@ -5,6 +5,13 @@ import { nodeToCapture, syntheticCapture, type SyntaxNode } from '../../utils/as
* Decompose a `preproc_include` node into a CaptureMatch with structured
* import captures. C++ #include maps to a wildcard import (all symbols
* from the header are visible). Identical to C's splitCInclude.
*
* Only literal include paths are emitted as import sources:
* #include <map> → system_lib_string
* #include "User.h" → string_literal
* A computed include like `#include HEADER_MACRO` carries an `identifier`
* path node (the macro name, not a header path); we skip it so it never
* becomes a garbage literal import source (issue #1919 F5).
*/
export function splitCppInclude(node: SyntaxNode): CaptureMatch | null {
const pathNode = node.childForFieldName?.('path') ?? null;
@ -21,7 +28,13 @@ export function splitCppInclude(node: SyntaxNode): CaptureMatch | null {
return buildIncludeCapture(node, pathNode);
}
function buildIncludeCapture(node: SyntaxNode, pathNode: SyntaxNode): CaptureMatch {
function buildIncludeCapture(node: SyntaxNode, pathNode: SyntaxNode): CaptureMatch | null {
// Skip computed includes (`#include MACRO`) — the path is an `identifier`,
// not a literal header path. Emitting it would create a garbage import.
if (pathNode.type !== 'string_literal' && pathNode.type !== 'system_lib_string') {
return null;
}
let raw: string;
if (pathNode.type === 'string_literal') {
const content = pathNode.namedChildren.find((c) => c.type === 'string_content');
@ -60,6 +73,12 @@ function buildIncludeCapture(node: SyntaxNode, pathNode: SyntaxNode): CaptureMat
*/
export function splitCppUsingDecl(node: SyntaxNode): CaptureMatch | null {
if (node.type !== 'using_declaration') return null;
// A class-scope `using Base::member;` changes the derived class's member
// lookup set; it is not a namespace import. The C++ member-lookup sidecar
// captures it separately, so suppress import decomposition here.
for (let parent = node.parent; parent !== null; parent = parent.parent) {
if (parent.type === 'class_specifier' || parent.type === 'struct_specifier') return null;
}
// Check for "namespace" keyword among anonymous children
let hasNamespaceKeyword = false;

View file

@ -14,3 +14,7 @@ export {
clearFileLocalNames,
expandCppWildcardNames,
} from './file-local-linkage.js';
export {
collectCppCaptureSideChannel,
applyCppCaptureSideChannel,
} from './capture-side-channel.js';

View file

@ -61,6 +61,30 @@ export function markCppInlineNamespaceRange(filePath: string, range: RangeKey):
set.add(rangeKey(range));
}
/** Snapshot this file's captured inline-namespace ranges for the worker→main
* side-channel (#1983). `populateCppInlineNamespaceScopes` (in `populateOwners`)
* later resolves these range keys to ScopeIds on the main thread, so only the
* capture-time ranges need to cross the boundary. Returns the rangeKey strings
* as a plain array (empty when this file recorded none). */
export function collectCppInlineNamespaceSideChannel(filePath: string): readonly string[] {
const set = inlineNamespaceRangesByFile.get(filePath);
return set === undefined ? [] : [...set];
}
/** Restore this file's captured inline-namespace ranges from the side-channel. */
export function applyCppInlineNamespaceSideChannel(
filePath: string,
ranges: readonly string[],
): void {
if (ranges.length === 0) return;
let set = inlineNamespaceRangesByFile.get(filePath);
if (set === undefined) {
set = new Set();
inlineNamespaceRangesByFile.set(filePath, set);
}
for (const r of ranges) set.add(r);
}
/** Clear all inline-namespace state. Called from `clearFileLocalNames`. */
export function clearCppInlineNamespaces(): void {
inlineNamespaceRangesByFile.clear();

View file

@ -0,0 +1,616 @@
import type { ParsedFile, ReferenceSite, SymbolDefinition } from 'gitnexus-shared';
import type { KnowledgeGraph } from '../../../graph/types.js';
import type { GraphNodeLookup } from '../../scope-resolution/graph-bridge/node-lookup.js';
import { resolveDefGraphId } from '../../scope-resolution/graph-bridge/ids.js';
import type { ScopeResolutionIndexes } from '../../model/scope-resolution-indexes.js';
import type { SemanticModel } from '../../model/semantic-model.js';
import type { ReceiverMemberResolution } from '../../scope-resolution/contract/scope-resolver.js';
import { buildMro, defaultLinearize } from '../../scope-resolution/passes/mro.js';
import {
isOverloadAmbiguousAfterNormalization,
narrowOverloadCandidates,
} from '../../scope-resolution/passes/overload-narrowing.js';
import { isClassLike } from '../../scope-resolution/scope/walkers.js';
import type { SyntaxNode } from '../../utils/ast-helpers.js';
import { cppConstraintCompatibility } from './constraint-filter.js';
import { cppConversionRank } from './conversion-rank.js';
interface CapturedBaseEdge {
readonly childName: string;
readonly childQualifiedName?: string;
readonly baseName: string;
readonly baseQualifiedName?: string;
readonly isVirtual: boolean;
}
interface CapturedMemberUsing {
readonly childName: string;
readonly childQualifiedName?: string;
readonly baseName: string;
readonly baseQualifiedName?: string;
readonly memberName: string;
}
export interface CppMemberLookupSideChannel {
readonly baseEdges: readonly CapturedBaseEdge[];
readonly memberUsings: readonly CapturedMemberUsing[];
}
const capturedByFile = new Map<string, CppMemberLookupSideChannel>();
let directParentsByDefId = new Map<string, readonly string[]>();
let virtualEdges = new Set<string>();
let ancestorsByDefId = new Map<string, ReadonlySet<string>>();
let memberUsingsByDefId = new Map<
string,
readonly { readonly baseDefId: string; readonly memberName: string }[]
>();
let inheritedLookupCache = new Map<string, CachedInheritedLookup>();
const MAX_INHERITANCE_VISITS = 4096;
type CachedInheritedLookup =
| { readonly kind: 'none' }
| { readonly kind: 'candidates'; readonly definitions: readonly SymbolDefinition[] }
| { readonly kind: 'ambiguous'; readonly candidateIds: readonly string[] };
export function clearCppMemberLookupState(): void {
capturedByFile.clear();
directParentsByDefId = new Map();
virtualEdges = new Set();
ancestorsByDefId = new Map();
memberUsingsByDefId = new Map();
inheritedLookupCache = new Map();
}
export function captureCppMemberLookupFacts(root: SyntaxNode, filePath: string): void {
const baseEdges: CapturedBaseEdge[] = [];
const memberUsings: CapturedMemberUsing[] = [];
const stack: SyntaxNode[] = [root];
while (stack.length > 0) {
const node = stack.pop()!;
if (node.type === 'class_specifier' || node.type === 'struct_specifier') {
const childName = classNameOf(node);
const childQualifiedName = classQualifiedNameOf(node);
if (childName !== '') {
const baseClause = directChildOfType(node, 'base_class_clause');
if (baseClause !== null) {
captureBaseEdges(baseClause, childName, childQualifiedName, baseEdges);
}
const body = directChildOfType(node, 'field_declaration_list');
if (body !== null) {
for (let i = 0; i < body.namedChildCount; i++) {
const child = body.namedChild(i);
if (child?.type !== 'using_declaration') continue;
const parsed = parseMemberUsing(child, childName, childQualifiedName);
if (parsed !== undefined) memberUsings.push(parsed);
}
}
}
}
for (let i = 0; i < node.childCount; i++) {
const child = node.child(i);
if (child !== null) stack.push(child);
}
}
if (baseEdges.length === 0 && memberUsings.length === 0) {
capturedByFile.delete(filePath);
} else {
capturedByFile.set(filePath, { baseEdges, memberUsings });
}
}
export function collectCppMemberLookupSideChannel(filePath: string): CppMemberLookupSideChannel {
return capturedByFile.get(filePath) ?? { baseEdges: [], memberUsings: [] };
}
export function applyCppMemberLookupSideChannel(
filePath: string,
data: CppMemberLookupSideChannel,
): void {
if (!Array.isArray(data.baseEdges) || !Array.isArray(data.memberUsings)) return;
if (data.baseEdges.length === 0 && data.memberUsings.length === 0) {
capturedByFile.delete(filePath);
return;
}
capturedByFile.set(filePath, {
baseEdges: data.baseEdges.slice(),
memberUsings: data.memberUsings.slice(),
});
}
export function buildCppMemberLookupMro(
graph: KnowledgeGraph,
parsedFiles: readonly ParsedFile[],
nodeLookup: GraphNodeLookup,
): Map<string, string[]> {
populateResolvedHierarchy(graph, parsedFiles, nodeLookup);
return buildMro(graph, parsedFiles, nodeLookup, defaultLinearize);
}
export function resolveCppReceiverMember(
ownerDef: SymbolDefinition,
memberName: string,
callsite: ReferenceSite,
_scopes: ScopeResolutionIndexes,
model: SemanticModel,
): ReceiverMemberResolution | undefined {
if (callsite.kind !== 'call') return undefined;
const ownMethods = model.methods.lookupAllByOwner(ownerDef.nodeId, memberName);
const introduced = introducedDefinitions(ownerDef.nodeId, memberName, model);
if (introduced.length > 0) {
return chooseOverload(uniqueDefinitions([...ownMethods, ...introduced]), callsite);
}
// Direct declarations hide every base declaration. Let the shared path
// retain its existing overload/static filtering for this common case.
if (ownMethods.length > 0) return undefined;
const lookup = inheritedLookupSet(ownerDef.nodeId, memberName, model);
if (lookup.kind === 'none') return undefined;
if (lookup.kind === 'ambiguous') return lookup;
return chooseOverload(lookup.definitions, callsite);
}
interface MemberOccurrence {
readonly ownerDefId: string;
readonly definitions: readonly SymbolDefinition[];
readonly path: readonly string[];
readonly virtualAnchor?: string;
}
function collectInheritedOccurrences(
ownerDefId: string,
memberName: string,
model: SemanticModel,
path: readonly string[],
virtualAnchor: string | undefined,
active: Set<string>,
budget: { remaining: number; truncated: boolean },
): MemberOccurrence[] {
if (budget.remaining <= 0) {
budget.truncated = true;
return [];
}
budget.remaining--;
if (active.has(ownerDefId)) return [];
const nextActive = new Set(active);
nextActive.add(ownerDefId);
const definitions = uniqueDefinitions([
...model.methods.lookupAllByOwner(ownerDefId, memberName),
...introducedDefinitions(ownerDefId, memberName, model),
]);
if (definitions.length > 0) {
return [{ ownerDefId, definitions, path, virtualAnchor }];
}
const results: MemberOccurrence[] = [];
for (const parentDefId of directParentsByDefId.get(ownerDefId) ?? []) {
const edgeKey = `${ownerDefId}\0${parentDefId}`;
results.push(
...collectInheritedOccurrences(
parentDefId,
memberName,
model,
[...path, parentDefId],
virtualEdges.has(edgeKey) ? parentDefId : virtualAnchor,
nextActive,
budget,
),
);
}
return results;
}
function inheritedLookupSet(
ownerDefId: string,
memberName: string,
model: SemanticModel,
): CachedInheritedLookup {
const cacheKey = `${ownerDefId}\0${memberName}`;
const cached = inheritedLookupCache.get(cacheKey);
if (cached !== undefined) return cached;
const budget = { remaining: MAX_INHERITANCE_VISITS, truncated: false };
const occurrences = collectInheritedOccurrences(
ownerDefId,
memberName,
model,
[],
undefined,
new Set(),
budget,
);
if (budget.truncated) {
const conservative: CachedInheritedLookup = {
kind: 'ambiguous',
candidateIds: uniqueDefinitions(occurrences.flatMap((entry) => entry.definitions)).map(
(definition) => definition.nodeId,
),
};
inheritedLookupCache.set(cacheKey, conservative);
return conservative;
}
if (occurrences.length === 0) {
const none: CachedInheritedLookup = { kind: 'none' };
inheritedLookupCache.set(cacheKey, none);
return none;
}
// A declaration can dominate another lookup set only when the latter is
// reached through a shared virtual subobject. Ordinary ancestry alone is
// insufficient: declarations in one non-virtual branch do not hide members
// reached through a sibling base subobject.
const undominated = occurrences.filter(
(candidate) =>
!(
candidate.virtualAnchor !== undefined &&
occurrences.some(
(other) =>
other.ownerDefId !== candidate.ownerDefId &&
isAncestor(candidate.ownerDefId, other.ownerDefId),
)
),
);
const groups = new Map<string, MemberOccurrence[]>();
for (const occurrence of undominated) {
const key =
occurrence.virtualAnchor !== undefined
? `virtual:${occurrence.virtualAnchor}:${occurrence.ownerDefId}`
: `path:${occurrence.path.join('>')}:${occurrence.ownerDefId}`;
const bucket = groups.get(key);
if (bucket === undefined) groups.set(key, [occurrence]);
else bucket.push(occurrence);
}
let result: CachedInheritedLookup;
if (groups.size !== 1) {
result = {
kind: 'ambiguous',
candidateIds: uniqueDefinitions(undominated.flatMap((entry) => entry.definitions)).map(
(definition) => definition.nodeId,
),
};
} else {
result = {
kind: 'candidates',
definitions: groups.values().next().value?.[0]?.definitions ?? [],
};
}
inheritedLookupCache.set(cacheKey, result);
return result;
}
function introducedDefinitions(
ownerDefId: string,
memberName: string,
model: SemanticModel,
): SymbolDefinition[] {
const definitions: SymbolDefinition[] = [];
for (const entry of memberUsingsByDefId.get(ownerDefId) ?? []) {
if (entry.memberName !== memberName) continue;
definitions.push(...model.methods.lookupAllByOwner(entry.baseDefId, memberName));
}
return definitions;
}
function uniqueDefinitions(definitions: readonly SymbolDefinition[]): SymbolDefinition[] {
return [...new Map(definitions.map((definition) => [definition.nodeId, definition])).values()];
}
function chooseOverload(
candidates: readonly SymbolDefinition[],
callsite: ReferenceSite,
): ReceiverMemberResolution | undefined {
if (candidates.length === 0) return undefined;
const narrowed = narrowOverloadCandidates(candidates, callsite.arity, callsite.argumentTypes, {
argumentTypeClasses: callsite.argumentTypeClasses,
conversionRankFn: cppConversionRank,
constraintCompatibility: cppConstraintCompatibility,
});
if (narrowed.length === 1) return { kind: 'resolved', definition: narrowed[0]! };
if (narrowed.length > 1 || isOverloadAmbiguousAfterNormalization(narrowed, callsite.arity)) {
return {
kind: 'ambiguous',
candidateIds: narrowed.map((candidate) => candidate.nodeId),
};
}
return undefined;
}
function populateResolvedHierarchy(
graph: KnowledgeGraph,
parsedFiles: readonly ParsedFile[],
nodeLookup: GraphNodeLookup,
): void {
const defByGraphId = new Map<string, SymbolDefinition>();
const defById = new Map<string, SymbolDefinition>();
const defsByFileAndName = new Map<string, SymbolDefinition[]>();
for (const parsed of parsedFiles) {
for (const def of parsed.localDefs) {
if (!isClassLike(def.type)) continue;
const graphId = resolveDefGraphId(parsed.filePath, def, nodeLookup);
if (graphId === undefined) continue;
defByGraphId.set(graphId, def);
defById.set(def.nodeId, def);
const names = new Set([simpleName(def), definitionQualifiedName(def)]);
for (const name of names) {
if (name === '') continue;
const key = `${parsed.filePath}\0${name}`;
const bucket = defsByFileAndName.get(key);
if (bucket === undefined) defsByFileAndName.set(key, [def]);
else bucket.push(def);
}
}
}
const parents = new Map<string, string[]>();
for (const rel of graph.iterRelationshipsByType('EXTENDS')) {
const child = defByGraphId.get(rel.sourceId);
const parent = defByGraphId.get(rel.targetId);
if (child === undefined || parent === undefined) continue;
const bucket = parents.get(child.nodeId);
if (bucket === undefined) parents.set(child.nodeId, [parent.nodeId]);
else bucket.push(parent.nodeId);
}
directParentsByDefId = parents;
ancestorsByDefId = buildAncestorClosure(parents);
inheritedLookupCache = new Map();
const nextVirtualEdges = new Set<string>();
const nextUsings = new Map<
string,
{ readonly baseDefId: string; readonly memberName: string }[]
>();
for (const parsed of parsedFiles) {
const captured = capturedByFile.get(parsed.filePath);
if (captured === undefined) continue;
for (const edge of captured.baseEdges) {
if (!edge.isVirtual) continue;
for (const child of matchingChildren(
parsed.filePath,
edge.childName,
edge.childQualifiedName,
defsByFileAndName,
)) {
const parent = findCapturedParent(
parents.get(child.nodeId) ?? [],
edge.baseName,
edge.baseQualifiedName,
defById,
);
if (parent !== undefined) nextVirtualEdges.add(`${child.nodeId}\0${parent.nodeId}`);
}
}
for (const using of captured.memberUsings) {
const children = matchingChildren(
parsed.filePath,
using.childName,
using.childQualifiedName,
defsByFileAndName,
);
for (const child of children) {
const baseDef = findCapturedParent(
parents.get(child.nodeId) ?? [],
using.baseName,
using.baseQualifiedName,
defById,
);
if (baseDef === undefined) continue;
const bucket = nextUsings.get(child.nodeId);
const entry = { baseDefId: baseDef.nodeId, memberName: using.memberName };
if (bucket === undefined) nextUsings.set(child.nodeId, [entry]);
else bucket.push(entry);
}
}
}
virtualEdges = nextVirtualEdges;
memberUsingsByDefId = nextUsings;
}
function captureBaseEdges(
baseClause: SyntaxNode,
childName: string,
childQualifiedName: string,
output: CapturedBaseEdge[],
): void {
let segmentStart = 0;
for (let i = 0; i < baseClause.childCount; i++) {
const child = baseClause.child(i);
if (child === null) continue;
if (child.type === ',' || child.text === ',') {
segmentStart = i + 1;
continue;
}
if (
child.type !== 'type_identifier' &&
child.type !== 'template_type' &&
child.type !== 'qualified_identifier'
) {
continue;
}
let isVirtual = false;
for (let j = segmentStart; j < i; j++) {
const modifier = baseClause.child(j);
if (modifier?.text === 'virtual') isVirtual = true;
}
const baseQualifiedName = qualifiedTypeName(child.text);
const baseName = baseQualifiedName.split('.').at(-1) ?? '';
if (baseName !== '') {
output.push({
childName,
...(childQualifiedName !== childName ? { childQualifiedName } : {}),
baseName,
...(baseQualifiedName !== baseName ? { baseQualifiedName } : {}),
isVirtual,
});
}
}
}
function parseMemberUsing(
node: SyntaxNode,
childName: string,
childQualifiedName: string,
): CapturedMemberUsing | undefined {
const qualified = node.namedChildren.find((child) => child.type === 'qualified_identifier');
if (qualified === undefined) return undefined;
const parts = splitQualifiedSegments(qualified.text);
if (parts.length < 2) return undefined;
const memberName = stripTemplateSuffix(parts.at(-1) ?? '');
const baseParts = parts.slice(0, -1).map(stripTemplateSuffix).filter(Boolean);
const baseName = baseParts.at(-1) ?? '';
const baseQualifiedName = baseParts.join('.');
if (baseName === '' || memberName === '') return undefined;
return {
childName,
...(childQualifiedName !== childName ? { childQualifiedName } : {}),
baseName,
...(baseQualifiedName !== baseName ? { baseQualifiedName } : {}),
memberName,
};
}
function classNameOf(node: SyntaxNode): string {
const name = node.childForFieldName?.('name');
return name === null || name === undefined ? '' : trailingIdentifier(name.text);
}
function classQualifiedNameOf(node: SyntaxNode): string {
const parts = [classNameOf(node)];
let current = node.parent;
while (current !== null) {
if (current.type === 'class_specifier' || current.type === 'struct_specifier') {
const name = classNameOf(current);
if (name !== '') parts.unshift(name);
} else if (current.type === 'namespace_definition') {
const name = current.childForFieldName?.('name');
if (name !== null && name !== undefined) {
parts.unshift(
...splitQualifiedSegments(name.text).map(stripTemplateSuffix).filter(Boolean),
);
}
}
current = current.parent;
}
return parts.filter(Boolean).join('.');
}
function directChildOfType(node: SyntaxNode, type: string): SyntaxNode | null {
for (let i = 0; i < node.namedChildCount; i++) {
const child = node.namedChild(i);
if (child?.type === type) return child;
}
return null;
}
function trailingIdentifier(value: string): string {
return stripTemplateSuffix(splitQualifiedSegments(value).at(-1) ?? '');
}
function qualifiedTypeName(value: string): string {
return splitQualifiedSegments(value).map(stripTemplateSuffix).filter(Boolean).join('.');
}
function splitQualifiedSegments(value: string): string[] {
const parts: string[] = [];
let angleDepth = 0;
let segmentStart = 0;
for (let i = 0; i < value.length; i++) {
const char = value[i];
if (char === '<') angleDepth++;
else if (char === '>' && angleDepth > 0) angleDepth--;
else if (char === ':' && value[i + 1] === ':' && angleDepth === 0) {
const segment = value.slice(segmentStart, i).trim();
if (segment !== '') parts.push(segment);
segmentStart = i + 2;
i++;
}
}
const tail = value.slice(segmentStart).trim();
if (tail !== '') parts.push(tail);
return parts;
}
function stripTemplateSuffix(value: string): string {
const templateStart = value.indexOf('<');
return (templateStart >= 0 ? value.slice(0, templateStart) : value).trim();
}
function simpleName(def: SymbolDefinition): string {
return def.qualifiedName?.split('.').at(-1) ?? '';
}
function definitionQualifiedName(def: SymbolDefinition): string {
const name = def.qualifiedName ?? '';
if (name === '' || def.namespacePrefix === undefined || def.namespacePrefix === '') return name;
return name.startsWith(`${def.namespacePrefix}.`) ? name : `${def.namespacePrefix}.${name}`;
}
function matchingChildren(
filePath: string,
childName: string,
childQualifiedName: string | undefined,
defsByFileAndName: ReadonlyMap<string, readonly SymbolDefinition[]>,
): readonly SymbolDefinition[] {
if (childQualifiedName !== undefined) {
const qualified = defsByFileAndName.get(`${filePath}\0${childQualifiedName}`) ?? [];
if (qualified.length > 0) return qualified;
}
const simple = defsByFileAndName.get(`${filePath}\0${childName}`) ?? [];
return simple.length === 1 ? simple : [];
}
function findCapturedParent(
parentIds: readonly string[],
baseName: string,
baseQualifiedName: string | undefined,
defById: ReadonlyMap<string, SymbolDefinition>,
): SymbolDefinition | undefined {
const candidates = parentIds
.map((id) => defById.get(id))
.filter((definition): definition is SymbolDefinition => definition !== undefined);
if (baseQualifiedName !== undefined) {
const qualified = candidates.filter((definition) => {
const name = definitionQualifiedName(definition);
return name === baseQualifiedName || name.endsWith(`.${baseQualifiedName}`);
});
if (qualified.length === 1) return qualified[0];
return undefined;
}
const simple = candidates.filter((definition) => simpleName(definition) === baseName);
return simple.length === 1 ? simple[0] : undefined;
}
function buildAncestorClosure(
parents: ReadonlyMap<string, readonly string[]>,
): Map<string, ReadonlySet<string>> {
const closure = new Map<string, ReadonlySet<string>>();
const visiting = new Set<string>();
const ancestorsOf = (defId: string): ReadonlySet<string> => {
const cached = closure.get(defId);
if (cached !== undefined) return cached;
if (visiting.has(defId)) return new Set();
visiting.add(defId);
const ancestors = new Set<string>();
for (const parent of parents.get(defId) ?? []) {
ancestors.add(parent);
for (const ancestor of ancestorsOf(parent)) ancestors.add(ancestor);
}
visiting.delete(defId);
closure.set(defId, ancestors);
return ancestors;
};
for (const defId of parents.keys()) ancestorsOf(defId);
return closure;
}
function isAncestor(ancestorDefId: string, descendantDefId: string): boolean {
return ancestorsByDefId.get(descendantDefId)?.has(ancestorDefId) === true;
}

View file

@ -4,7 +4,6 @@ import {
findEnclosingClassDef,
} from '../../scope-resolution/scope/walkers.js';
import { SupportedLanguages } from 'gitnexus-shared';
import { buildMro, defaultLinearize } from '../../scope-resolution/passes/mro.js';
import {
populateClassOwnedMembers,
tagNamespacePrefixes,
@ -30,6 +29,7 @@ import {
isCppDependentBaseMember,
} from './two-phase-lookup.js';
import { populateCppAssociatedNamespaces, clearCppAdlState, pickCppAdlCandidates } from './adl.js';
import { applyCppCaptureSideChannel } from './capture-side-channel.js';
import {
clearCppInlineNamespaces,
populateCppInlineNamespaceScopes,
@ -41,6 +41,45 @@ import {
clearCppUserDefinedConversions,
populateCppUserDefinedConversions,
} from './user-defined-conversions.js';
import {
buildCppMemberLookupMro,
clearCppMemberLookupState,
resolveCppReceiverMember,
} from './member-lookup.js';
/**
* Per-pass memo of the augmented `#include`-resolution file set
* (`allFilePaths` ∪ header paths), keyed on the two stable source sets.
* `resolveImportTarget` is called once per C++ `#include`; the old code rebuilt
* a fresh ~F-entry `Set` on every call AND defeated the shared
* `resolveCImportTarget` suffix-index memo (in `c/import-target.ts`) by handing
* it a new set identity each time. Both inputs are stable per pass, so the
* union is built once and reused. `WeakMap`-keyed → reclaimed with the pass.
* (Twin of the C resolver's `augmentedFilePaths`.)
*/
const augmentedPathsByPass = new WeakMap<
ReadonlySet<string>,
WeakMap<ReadonlySet<string>, ReadonlySet<string>>
>();
function augmentedFilePaths(
allFilePaths: ReadonlySet<string>,
headerPaths: ReadonlySet<string>,
): ReadonlySet<string> {
let byHeaders = augmentedPathsByPass.get(allFilePaths);
if (byHeaders === undefined) {
byHeaders = new WeakMap();
augmentedPathsByPass.set(allFilePaths, byHeaders);
}
let augmented = byHeaders.get(headerPaths);
if (augmented === undefined) {
const set = new Set(allFilePaths);
for (const h of headerPaths) set.add(h);
augmented = set;
byHeaders.set(headerPaths, augmented);
}
return augmented;
}
/**
* C++ `ScopeResolver` registered in `SCOPE_RESOLVERS` and consumed by
@ -69,6 +108,7 @@ export const cppScopeResolver: ScopeResolver = {
clearCppAdlState();
clearCppInlineNamespaces();
clearCppUserDefinedConversions();
clearCppMemberLookupState();
return scanCppHeaderFiles(repoPath);
},
@ -78,9 +118,11 @@ export const cppScopeResolver: ScopeResolver = {
// detection but are importable from .cpp files via #include.
const headerPaths = resolutionConfig as ReadonlySet<string> | undefined;
if (headerPaths !== undefined && headerPaths.size > 0) {
const augmented = new Set(allFilePaths);
for (const h of headerPaths) augmented.add(h);
return resolveCppImportTarget(targetRaw, fromFile, augmented);
return resolveCppImportTarget(
targetRaw,
fromFile,
augmentedFilePaths(allFilePaths, headerPaths),
);
}
return resolveCppImportTarget(targetRaw, fromFile, allFilePaths);
},
@ -100,8 +142,25 @@ export const cppScopeResolver: ScopeResolver = {
// `'unknown'` keeps the candidate, preserving "degrade not lie".
constraintCompatibility: cppConstraintCompatibility,
buildMro: (graph, parsedFiles, nodeLookup) =>
buildMro(graph, parsedFiles, nodeLookup, defaultLinearize),
buildMro: buildCppMemberLookupMro,
// Worker-boundary restore (see `ScopeResolver.applyCaptureSideChannel`).
// `emitCppScopeCaptures` records per-file ADL call-site arg shapes
// (`markCppAdlSiteArgs`/`markCppAdlSiteNoAdl`), inline-/anonymous-namespace
// ranges (`markCppInlineNamespaceRange`/`markCppAnonymousNamespaceRange`),
// dependent-base names (`markCppDependentBase`/`markCppDependentPackBase`),
// and file-local linkage (`markFileLocal`) into module-level maps as a SIDE
// EFFECT — none of it is serialized onto the returned ParsedFile's scopes/defs.
// On the worker path those marks are populated in the worker process and lost
// across the MessageChannel / disk store; the main thread reuses the
// serialized ParsedFile and skips `extractParsedFile`, so `populateOwners` +
// the ADL / two-phase-lookup passes would see empty maps and emit zero edges.
// The worker stashed a plain-data snapshot on `parsed.captureSideChannel` via
// `cppProvider.collectCaptureSideChannel`; this restores it into the module
// maps WITHOUT any tree-sitter re-parse (the #1983 fix — the old re-parse
// replay re-OOM'd huge `.h`/`.cpp` repos). The freshly-extracted leg never
// calls this — its marks were just populated in this process.
applyCaptureSideChannel: applyCppCaptureSideChannel,
populateOwners: (parsed: ParsedFile) => {
populateClassOwnedMembers(parsed);
@ -206,6 +265,7 @@ export const cppScopeResolver: ScopeResolver = {
hoistTypeBindingsToModule: true,
// Enable receiver-bound explicit-`this` fallback only for C++.
resolveThisViaEnclosingClass: true,
resolveReceiverMember: resolveCppReceiverMember,
// The `isFileLocalDef` hook on the global free-call fallback names
// file-local linkage historically, but semantically gates "logically
// invisible cross-file" defs. C++ extends this to also reject class-

View file

@ -104,6 +104,56 @@ export function markCppDependentPackBase(filePath: string, className: string): v
perFile.add(className);
}
/**
* Plain-data, JSON-serializable snapshot of the per-file capture-time
* two-phase-lookup state. Carried on `ParsedFile.captureSideChannel` across the
* worker→main boundary (#1983). The resolved `dependentBaseNodeIds` index is
* rebuilt by `populateCppDependentBases` (workspace pass) after all files have
* their `populateOwners` applied, so only the two capture-time maps cross.
*
* Nested `Map`/`Set` are flattened to arrays here so the snapshot stays plain
* JSON (avoids relying on the parsedfile-store's Map/Set replacer for nested
* structures): `dependentBases` is `[className, [baseName, qualifiers[]][]][]`.
*/
export interface CppTwoPhaseSideChannel {
readonly dependentBases: readonly [string, readonly [string, readonly string[]][]][];
readonly dependentPackBaseClasses: readonly string[];
}
/** Snapshot this file's two-phase-lookup capture state for the side-channel. */
export function collectCppTwoPhaseSideChannel(filePath: string): CppTwoPhaseSideChannel {
const perFile = dependentBasesByFile.get(filePath);
const dependentBases: [string, [string, string[]][]][] = [];
if (perFile !== undefined) {
for (const [className, bases] of perFile) {
const baseEntries: [string, string[]][] = [];
for (const [baseName, quals] of bases) {
baseEntries.push([baseName, [...quals]]);
}
dependentBases.push([className, baseEntries]);
}
}
const pack = dependentPackBaseClassesByFile.get(filePath);
return {
dependentBases,
dependentPackBaseClasses: pack === undefined ? [] : [...pack],
};
}
/** Restore this file's two-phase-lookup capture state from the side-channel. */
export function applyCppTwoPhaseSideChannel(filePath: string, data: CppTwoPhaseSideChannel): void {
for (const [className, baseEntries] of data.dependentBases) {
for (const [baseName, quals] of baseEntries) {
for (const qualifier of quals) {
markCppDependentBase(filePath, className, baseName, qualifier);
}
}
}
for (const className of data.dependentPackBaseClasses) {
markCppDependentPackBase(filePath, className);
}
}
/** Clear two-phase-lookup state. Called from `clearFileLocalNames`. */
export function clearCppDependentBases(): void {
dependentBasesByFile.clear();

View file

@ -86,10 +86,11 @@ export function emitCsharpScopeCaptures(
_filePath: string,
cachedTree?: unknown,
): readonly CaptureMatch[] {
// Skip the parse when the caller (parse phase's scopeTreeCache)
// already produced a Tree for this source. Cache miss = re-parse,
// same as before. The cachedTree parameter is typed as `unknown` at
// the LanguageProvider contract layer; cast here at the use site.
// Reuse a pre-parsed Tree when the caller passes one via `cachedTree`; a
// miss re-parses. (The cache is currently always empty — its only producer,
// the sequential parser, was removed — so this re-parses in practice.) The
// cachedTree parameter is typed `unknown` at the LanguageProvider contract
// layer; cast here at the use site.
let tree = cachedTree as ReturnType<ReturnType<typeof getCsharpParser>['parse']> | undefined;
if (tree === undefined) {
tree = parseSourceSafe(getCsharpParser(), sourceText, undefined, {

View file

@ -68,10 +68,9 @@
* `using static X = Y.Z;`, attributes, and preprocessor-gated
* declarations are all recognized correctly.
*
* Shadow-harness corpus parity is the authoritative signal for which
* of these matter in practice. The CI parity gate blocks any PR that
* regresses either the legacy or registry-primary run of
* `test/integration/resolvers/csharp.test.ts`.
* The `test/integration/resolvers/csharp.test.ts` resolver suite is the
* authoritative signal for which of these matter in practice; it runs in
* the standard CI test workflow, so a regression blocks the merge.
*/
export { emitCsharpScopeCaptures } from './captures.js';

View file

@ -356,8 +356,9 @@ function extractFileStructure(content: string, cachedTree: unknown): CsharpFileS
/** Content + (optional) pre-parsed tree-sitter trees keyed by filePath.
* The orchestrator builds `fileContents` from the pipeline's file list;
* `treeCache` is the same `scopeTreeCache` already populated by the
* parse phase, so cache hits avoid a second `parser.parse()`. */
* `treeCache` is currently always empty (its only producer, the sequential
* parser, was removed), so the providers re-parse. Kept as an extension
* point that would let cache hits avoid a second `parser.parse()`. */
export interface CsharpSiblingInputs {
readonly fileContents: ReadonlyMap<string, string>;
readonly treeCache?: { get(filePath: string): unknown };

View file

@ -23,7 +23,15 @@
*/
import Parser from 'tree-sitter';
import Dart from 'tree-sitter-dart';
import { SupportedLanguages } from 'gitnexus-shared';
// `tree-sitter-dart` is an optional/vendored grammar that may be absent on a
// default install. Loaded lazily + guarded via parser-loader rather than
// statically imported: this module is pulled onto the main thread eagerly by
// the scope-resolution registry and the language-provider index, so a top-level
// `import Dart from 'tree-sitter-dart'` would throw ERR_MODULE_NOT_FOUND at
// module-load and crash `analyze` even for repos with no Dart files (#2091,
// #2093). The grammar is only ever needed inside the lazy getters below.
import { getLanguageGrammar } from '../../../tree-sitter/parser-loader.js';
const DART_SCOPE_QUERY = `
; ── Scopes ───────────────────────────────────────────────────────────────────
@ -39,6 +47,46 @@ const DART_SCOPE_QUERY = `
(extension_declaration name: (identifier) @declaration.name) @declaration.class
(enum_declaration name: (identifier) @declaration.name) @declaration.enum
; ── Declarations — type aliases (old-style + new-style function typedefs) ────
; Both forms parse as type_alias; the name position differs, and a generic
; <T> parameter list intervenes for the generic variants. Per #1919 review CF2,
; a generic type_parameters node sits between the name and the next anchor, so
; the non-generic adjacency patterns silently drop the generic forms. Four
; standalone patterns (NOT one alternation — the tree-sitter 0.21 hazard drops
; sibling branches) keep the name capture unambiguous and single-match per form:
; non-generic old-style typedef int Cmp(int a, int b);
; children: return-type, NAME, formal_parameter_list
; generic old-style typedef int Cmp<T>(T a, T b); (CF2)
; children: return-type, NAME, type_parameters, formal_parameter_list
; non-generic new-style typedef Pred = bool Function(int);
; children: NAME, "=", function_type
; generic new-style typedef Mapper<T> = T Function(T);
; children: NAME, type_parameters, "=", function_type
; The alias name is the type_identifier immediately before the param list (old)
; or before "=" (new); for the generic forms it is the one immediately before
; the intervening type_parameters. Mirrors Kotlin's @declaration.type_alias
; rule; the generic scope-extractor maps "type_alias" → TypeAlias.
(type_alias
(type_identifier) @declaration.name
.
(formal_parameter_list)) @declaration.type_alias
(type_alias
(type_identifier) @declaration.name
.
(type_parameters)
.
(formal_parameter_list)) @declaration.type_alias
(type_alias
(type_identifier) @declaration.name
.
"=") @declaration.type_alias
(type_alias
(type_identifier) @declaration.name
.
(type_parameters)
.
"=") @declaration.type_alias
; ── Declarations — top-level functions (parent is program, not method) ───────
(program
(function_signature
@ -94,14 +142,19 @@ let _query: Parser.Query | null = null;
export function getDartParser(): Parser {
if (_parser === null) {
_parser = new Parser();
_parser.setLanguage(Dart as Parameters<Parser['setLanguage']>[0]);
_parser.setLanguage(
getLanguageGrammar(SupportedLanguages.Dart) as Parameters<Parser['setLanguage']>[0],
);
}
return _parser;
}
export function getDartScopeQuery(): Parser.Query {
if (_query === null) {
_query = new Parser.Query(Dart as Parameters<Parser['setLanguage']>[0], DART_SCOPE_QUERY);
_query = new Parser.Query(
getLanguageGrammar(SupportedLanguages.Dart) as Parameters<Parser['setLanguage']>[0],
DART_SCOPE_QUERY,
);
}
return _query;
}

View file

@ -53,11 +53,15 @@ const GO_SCOPE_QUERY = `
;; Declarations — variables
(var_declaration
(var_spec
name: (identifier) @declaration.name)) @declaration.variable
(identifier) @declaration.name)) @declaration.variable
(var_declaration
(var_spec_list
(var_spec
(identifier) @declaration.name))) @declaration.variable
(const_declaration
(const_spec
name: (identifier) @declaration.name)) @declaration.const
(identifier) @declaration.name)) @declaration.const
(short_var_declaration
left: (expression_list (identifier) @declaration.name)) @declaration.variable

View file

@ -28,6 +28,7 @@ import { kotlinMethodConfig } from '../method-extractors/configs/jvm.js';
import { createVariableExtractor } from '../variable-extractors/generic.js';
import { kotlinVariableConfig } from '../variable-extractors/configs/jvm.js';
import {
collectKotlinCaptureSideChannel,
emitKotlinScopeCaptures,
interpretKotlinImport,
interpretKotlinTypeBinding,
@ -175,6 +176,13 @@ export const kotlinProvider = defineLanguage({
// ── RFC #909 Ring 3: scope-based resolution hooks ──
emitScopeCaptures: emitKotlinScopeCaptures,
// Worker-side: snapshot the module-level companion-scope marks
// `emitKotlinScopeCaptures` just populated for this file (`markCompanionScope`
// → `companionScopesByFile`) into plain data on `ParsedFile.captureSideChannel`,
// so the main thread can restore them via `applyCaptureSideChannel` WITHOUT a
// re-parse (#1983). Without this, companion/static dispatch emits no CALLS
// edges on the worker path. See `kotlin/capture-side-channel.ts`.
collectCaptureSideChannel: collectKotlinCaptureSideChannel,
interpretImport: interpretKotlinImport,
interpretTypeBinding: interpretKotlinTypeBinding,
bindingScopeFor: kotlinBindingScopeFor,

View file

@ -0,0 +1,75 @@
/**
* Kotlin capture-time side-channel serialization (#1983).
*
* `emitKotlinScopeCaptures` populates one MODULE-LEVEL, per-file map as a side
* effect that is NOT part of the returned `ParsedFile`'s scopes/defs:
*
* - `companionScopesByFile` (companion-scopes.ts) — the `ScopeId`s that came
* from a `companion_object` AST node, recorded via `markCompanionScope`
* from the `@scope.companion` marker capture.
*
* On the worker path that map is filled in the WORKER process and lost across
* the worker→main MessageChannel (and the disk-backed parsedfile-store),
* because scope-resolution reuses the serialized `ParsedFile` and SKIPS the
* main-thread re-extraction (the #1983 fix that avoids a main-thread
* tree-sitter re-parse / OOM on huge repos). The main thread then reads the map
* empty in `isKotlinStaticOnly` / `populateCompanionMembersOnEnclosingClass`
* (owners.ts) — so companion methods aren't identified as static and
* companion/static dispatch emits no CALLS edges.
*
* This module snapshots the per-file slice of that map into a plain,
* JSON-serializable object (carried on `ParsedFile.captureSideChannel`) and
* restores it on the main thread WITHOUT any parse. It mirrors the C++ pattern
* in `cpp/capture-side-channel.ts`.
*
* The single generic `ParsedFile.captureSideChannel` field is shared with C++,
* which is safe because each file is one language (a `.kt` file uses the kotlin
* provider, a `.cpp` file the cpp provider). The payload is self-describing
* (`{ kind: 'kotlin', companionScopes }`) so `applyKotlinCaptureSideChannel`
* only restores kotlin state and ignores a foreign-shaped snapshot.
*/
import type { ParsedFile, ScopeId } from 'gitnexus-shared';
import { getCompanionScopesForFile, markCompanionScope } from './companion-scopes.js';
/**
* Plain JSON-serializable snapshot of the per-file Kotlin capture-time
* side-channel. Carried opaquely on `ParsedFile.captureSideChannel`. The
* `kind` tag makes the payload self-describing so `apply` can distinguish a
* kotlin snapshot from another language's (C++ shares the same field).
*/
export interface KotlinCaptureSideChannel {
readonly kind: 'kotlin';
/** Companion-object scope ids recorded for this file. */
readonly companionScopes: readonly ScopeId[];
}
/**
* `LanguageProvider.collectCaptureSideChannel` implementation for Kotlin.
* Returns `undefined` when this file recorded no companion scopes at all, so
* the produced `ParsedFile` carries the field only when there's data to ship.
*/
export function collectKotlinCaptureSideChannel(
filePath: string,
): KotlinCaptureSideChannel | undefined {
const companionScopes = getCompanionScopesForFile(filePath);
if (companionScopes.length === 0) return undefined;
return { kind: 'kotlin', companionScopes };
}
/**
* `ScopeResolver.applyCaptureSideChannel` implementation for Kotlin. Reads the
* worker-serialized snapshot from `parsed.captureSideChannel` and re-populates
* the module-level companion-scope map via `markCompanionScope`. Tolerant of
* `undefined` (file carried no data) and of an unexpected / foreign shape
* (defensive — the `kind` tag guards against restoring a non-kotlin payload).
* Does NO tree-sitter parse.
*/
export function applyKotlinCaptureSideChannel(parsed: ParsedFile): void {
const data = parsed.captureSideChannel as KotlinCaptureSideChannel | undefined;
if (data === undefined || data === null || typeof data !== 'object') return;
if (data.kind !== 'kotlin' || !Array.isArray(data.companionScopes)) return;
for (const scopeId of data.companionScopes) {
markCompanionScope(parsed.filePath, scopeId);
}
}

View file

@ -39,6 +39,7 @@ export function emitKotlinScopeCaptures(
out.push(...synthesizeKotlinSmartCastBindings(tree.rootNode));
out.push(...synthesizeKotlinLambdaBindings(tree.rootNode, returnTypes));
out.push(...synthesizeKotlinInheritanceReferences(tree.rootNode));
out.push(...synthesizeKotlinSecondaryConstructorDeclarations(tree.rootNode));
for (const match of getKotlinScopeQuery().matches(tree.rootNode)) {
const grouped: Record<string, Capture> = {};
@ -87,6 +88,40 @@ export function emitKotlinScopeCaptures(
}
}
// Callable references (`::method`, `Type::new`, `obj::m`) — F47 (#1919).
// The query captures the referenced member as `@reference.name`, an
// optional receiver type as `@reference.receiver`, and the whole node as
// `@reference.callable`. Rewrite into a call reference so it participates
// in call-graph resolution: a bare `::member` resolves as a free call;
// a `Receiver::member` resolves as a member call against the receiver
// type. The function/constructor is referenced (not invoked), so no
// arity/argument metadata is attached.
if (grouped['@reference.callable'] !== undefined) {
const nameCap = grouped['@reference.name'];
const callableNode = groupedNodes['@reference.callable'];
if (nameCap !== undefined && callableNode !== undefined) {
const receiverCap = grouped['@reference.receiver'];
// The anchor Capture must carry the call-form tag as its `name` —
// the scope-extractor reads `Capture.name` (not the map key) to
// classify the reference kind, so re-wrap via nodeToCapture rather
// than reusing the `@reference.callable`-named Capture (whose head
// `callable` resolves to no ReferenceKind and silently drops it).
if (receiverCap !== undefined) {
out.push({
'@reference.call.member': nodeToCapture('@reference.call.member', callableNode),
'@reference.name': nameCap,
'@reference.receiver': receiverCap,
});
} else {
out.push({
'@reference.call.free': nodeToCapture('@reference.call.free', callableNode),
'@reference.name': nameCap,
});
}
}
continue;
}
if (
grouped['@reference.call.free'] !== undefined &&
grouped['@reference.receiver'] !== undefined
@ -253,6 +288,100 @@ function synthesizeKotlinInheritanceReferences(rootNode: SyntaxNode): CaptureMat
return out;
}
/**
* The enclosing type name for a node nested in a class/object/companion body.
* Walks up to the first `class_declaration` / `object_declaration` /
* `companion_object` ancestor and returns its `type_identifier` name node.
* Used to qualify a secondary-constructor declaration as `<ClassName>.constructor`.
*/
function kotlinEnclosingTypeNameNode(node: SyntaxNode): SyntaxNode | null {
for (let cur: SyntaxNode | null = node.parent; cur !== null; cur = cur.parent) {
if (
cur.type === 'class_declaration' ||
cur.type === 'object_declaration' ||
cur.type === 'companion_object'
) {
const nameNode = cur.namedChildren.find((c) => c.type === 'type_identifier');
return nameNode ?? null;
}
}
return null;
}
/**
* Synthesize a `@declaration.constructor` capture for each Kotlin
* `secondary_constructor` (issue #1919 review CF1). The structure phase already
* materializes a `Constructor` graph node (`Constructor:file:Class.constructor#<arity>`),
* but the registry-primary scope-resolution path had no Constructor *def* in the
* scope tree — so a call inside the constructor body resolved its caller anchor
* up to the enclosing Class def, mis-attributing the CALLS edge to the class.
*
* Paired with `(secondary_constructor) @scope.function` in query.ts: that rule
* makes the constructor body its own Function scope; this declaration places a
* Constructor def in that scope so `pickCallerCallableDef` anchors calls on the
* Constructor. The def is keyed to match the structure-phase node id:
* - `@declaration.qualified_name` = `<ClassName>.constructor` so the bridge's
* qualified key (`<q>:file::Constructor::Class.constructor`) hits the node.
* - `@declaration.parameter-types` so two same-name secondary constructors are
* disambiguated by the bridge's parameter-types key (`~Int,Int`), matching
* the `#<arity>`-suffixed structure node for the overload with the same
* parameter shape. (The zero-arg overload carries no parameter types and
* resolves via the qualified/simple key to the `#0` node.)
*
* The anchor spans the whole `secondary_constructor` node — same range as the
* `@scope.function` it pairs with — so the def is owned by that Function scope
* and the constructor name auto-hoists to the enclosing class scope (exactly the
* binding shape a normal method declaration produces).
*/
function synthesizeKotlinSecondaryConstructorDeclarations(rootNode: SyntaxNode): CaptureMatch[] {
const out: CaptureMatch[] = [];
for (const ctorNode of descendantsOfType(rootNode, 'secondary_constructor')) {
const keyword = ctorNode.namedChildren.find((c) => c.type === 'constructor');
// The `constructor` keyword is an anonymous token; fall back to the node
// itself for the name capture position when the named-child lookup misses.
const nameAnchor = keyword ?? ctorNode;
const classNameNode = kotlinEnclosingTypeNameNode(ctorNode);
const qualifiedName =
classNameNode !== null ? `${classNameNode.text}.constructor` : 'constructor';
const match: Record<string, Capture> = {
'@declaration.constructor': nodeToCapture('@declaration.constructor', ctorNode),
'@declaration.name': syntheticCapture('@declaration.name', nameAnchor, 'constructor'),
'@declaration.qualified_name': syntheticCapture(
'@declaration.qualified_name',
ctorNode,
qualifiedName,
),
};
const arity = computeKotlinArityMetadata(ctorNode);
if (arity.parameterCount !== undefined) {
match['@declaration.parameter-count'] = syntheticCapture(
'@declaration.parameter-count',
ctorNode,
String(arity.parameterCount),
);
}
if (arity.requiredParameterCount !== undefined) {
match['@declaration.required-parameter-count'] = syntheticCapture(
'@declaration.required-parameter-count',
ctorNode,
String(arity.requiredParameterCount),
);
}
if (arity.parameterTypes !== undefined) {
match['@declaration.parameter-types'] = syntheticCapture(
'@declaration.parameter-types',
ctorNode,
JSON.stringify(arity.parameterTypes),
);
}
out.push(match);
}
return out;
}
/**
* The bare simple-name `type_identifier` of a `user_type`. Strips generic
* type arguments (`Base<T>` → `Base`) and qualifier tails (`pkg.Base` → `Base`)

View file

@ -55,6 +55,17 @@ export function isCompanionScope(filePath: string, scopeId: ScopeId): boolean {
return companionScopesByFile.get(filePath)?.has(scopeId) ?? false;
}
/**
* Snapshot the companion-object scope ids recorded for `filePath` as a plain
* array (for the worker→main capture side-channel, #1983). Returns an empty
* array when the file recorded no companion scopes. See
* `capture-side-channel.ts`.
*/
export function getCompanionScopesForFile(filePath: string): ScopeId[] {
const scopes = companionScopesByFile.get(filePath);
return scopes === undefined ? [] : [...scopes];
}
/** Clear all tracked companion scopes (for testing). */
export function clearCompanionScopes(): void {
companionScopesByFile.clear();

View file

@ -1,4 +1,9 @@
export { emitKotlinScopeCaptures } from './captures.js';
export {
collectKotlinCaptureSideChannel,
applyKotlinCaptureSideChannel,
type KotlinCaptureSideChannel,
} from './capture-side-channel.js';
export { getKotlinCaptureCacheStats, resetKotlinCaptureCacheStats } from './cache-stats.js';
export { interpretKotlinImport, interpretKotlinTypeBinding } from './interpret.js';
export { kotlinArityCompatibility } from './arity.js';

View file

@ -1,5 +1,13 @@
import Parser from 'tree-sitter';
import Kotlin from 'tree-sitter-kotlin';
import { SupportedLanguages } from 'gitnexus-shared';
// `tree-sitter-kotlin` is an optionalDependency that may be absent on a default
// install (or fail its native build). Loaded lazily + guarded via parser-loader
// rather than statically imported: this module is pulled onto the main thread
// eagerly by the scope-resolution registry and the language-provider index, so
// a top-level `import Kotlin from 'tree-sitter-kotlin'` would throw
// ERR_MODULE_NOT_FOUND at module-load and crash `analyze` even for repos with no
// Kotlin files (#2091, #2093). The grammar is only ever needed in the getters.
import { getLanguageGrammar } from '../../../tree-sitter/parser-loader.js';
const KOTLIN_SCOPE_QUERY = `
;; Scopes
@ -9,6 +17,16 @@ const KOTLIN_SCOPE_QUERY = `
(companion_object) @scope.class
(function_declaration) @scope.function
;; Secondary-constructor body scope (issue #1919 review CF1). A
;; secondary constructor's "constructor(...) { ... }" body executes statements
;; just like a method body, so it must be its OWN Function scope — otherwise a
;; call inside the body resolves its caller anchor up to the enclosing Class
;; scope (the class's Class def), mis-attributing the CALLS edge to the class
;; rather than the Constructor. The matching @declaration.constructor is
;; synthesized in captures.ts (synthesizeKotlinSecondaryConstructorDeclarations)
;; so this scope owns a Constructor def keyed to the structure-phase node id.
(secondary_constructor) @scope.function
;; Companion-object marker (issue #1756 / U4). Side-channel capture that
;; lets populateCompanionMembersOnEnclosingClass distinguish a companion
;; Class scope from a regular Class scope without inspecting ownedDefs.
@ -117,6 +135,26 @@ const KOTLIN_SCOPE_QUERY = `
(function_value_parameters)
[(user_type) (nullable_type) (function_type)] @type-binding.type) @type-binding.return
;; References — callable references ("::method", "Type::new", "obj::m") — F47.
;; A "callable_reference" references a function/constructor as a value (no
;; call_suffix), so the registry-primary call path never saw it. Real-parse
;; (issue #1919) shows the canonical shape inside a function body is:
;; "::topLevelFn" -> (callable_reference :: (simple_identifier)) member only
;; "String::length" -> (callable_reference (type_identifier) :: (simple_identifier))
;; "obj::method" -> (callable_reference (type_identifier) :: (simple_identifier))
;; "Type::new" -> (callable_reference (type_identifier) :: (simple_identifier))
;; The receiver (real type OR object) is always a "type_identifier"; the
;; referenced member is the LAST "simple_identifier". One rule with an
;; optional receiver and an end-anchored member covers all four forms with
;; exactly one match per callable_reference (no sibling-branch double-match).
;; (NOTE: a qualified "A.B::m" parses as a nested navigation_expression, not a
;; callable_reference, and is already captured by the read.member rule below.)
;; emitKotlinScopeCaptures rewrites this into a free/member call reference.
(callable_reference
(type_identifier)? @reference.receiver
(simple_identifier) @reference.name
.) @reference.callable
;; References — direct calls / constructor syntax
(call_expression
(simple_identifier) @reference.name) @reference.call.free
@ -149,14 +187,19 @@ let query: Parser.Query | null = null;
export function getKotlinParser(): Parser {
if (parser === null) {
parser = new Parser();
parser.setLanguage(Kotlin as Parameters<Parser['setLanguage']>[0]);
parser.setLanguage(
getLanguageGrammar(SupportedLanguages.Kotlin) as Parameters<Parser['setLanguage']>[0],
);
}
return parser;
}
export function getKotlinScopeQuery(): Parser.Query {
if (query === null) {
query = new Parser.Query(Kotlin as Parameters<Parser['setLanguage']>[0], KOTLIN_SCOPE_QUERY);
query = new Parser.Query(
getLanguageGrammar(SupportedLanguages.Kotlin) as Parameters<Parser['setLanguage']>[0],
KOTLIN_SCOPE_QUERY,
);
}
return query;
}

View file

@ -14,6 +14,7 @@ import {
type KotlinResolveContext,
} from './index.js';
import { clearCompanionScopes } from './companion-scopes.js';
import { applyKotlinCaptureSideChannel } from './capture-side-channel.js';
import { isKotlinStaticOnly } from './owners.js';
/**
@ -84,6 +85,23 @@ export const kotlinScopeResolver: ScopeResolver = {
buildMro: (graph, parsedFiles, nodeLookup) => buildKotlinMro(graph, parsedFiles, nodeLookup),
// Worker-boundary restore (see `ScopeResolver.applyCaptureSideChannel`).
// `emitKotlinScopeCaptures` records per-file companion-object scope ids
// (`markCompanionScope` → `companionScopesByFile`) as a SIDE EFFECT — that
// state is NOT serialized onto the returned ParsedFile's scopes/defs. On the
// worker path those marks are populated in the worker process and lost across
// the MessageChannel / disk store; the main thread reuses the serialized
// ParsedFile and skips `extractParsedFile`, so `isKotlinStaticOnly` and
// `populateCompanionMembersOnEnclosingClass` (owners.ts) would see an empty
// map and companion/static dispatch would emit zero CALLS edges. The worker
// stashed a plain-data snapshot on `parsed.captureSideChannel` via
// `kotlinProvider.collectCaptureSideChannel`; this restores it into the
// module map WITHOUT any tree-sitter re-parse (the #1983 fix). The
// freshly-extracted leg never calls this — its marks were just populated in
// this process. Runs BEFORE `populateOwners` so the restored companion map is
// visible to it.
applyCaptureSideChannel: applyKotlinCaptureSideChannel,
populateOwners: (parsed: ParsedFile) => populateKotlinOwners(parsed),
isSuperReceiver: (text) => text.trim() === 'super',

View file

@ -54,10 +54,9 @@
* 6. **Intersection types in parameters** — `T&U $param` takes the first
* named part (`T`). This matches the legacy type-extractor's behavior.
*
* Shadow-harness corpus parity is the authoritative signal for which of
* these matter in practice. The CI parity gate blocks any PR that regresses
* either the legacy or registry-primary run of
* `test/integration/resolvers/php.test.ts`.
* The `test/integration/resolvers/php.test.ts` resolver suite is the
* authoritative signal for which of these matter in practice; it runs in
* the standard CI test workflow, so a regression blocks the merge.
*/
export { emitPhpScopeCaptures } from './captures.js';

View file

@ -38,9 +38,10 @@ export function emitPythonScopeCaptures(
_filePath: string,
cachedTree?: unknown,
): readonly CaptureMatch[] {
// Skip the parse when the caller (parse phase's ASTCache) already
// produced a Tree for this source. Cache miss = re-parse, same as
// before. The cachedTree parameter is typed as `unknown` at the
// Skip the parse when the caller (the scope-resolution orchestrator's
// `treeCache`) already produced a Tree for this source — empty under
// worker-pool runs, so cache miss = re-parse. The cachedTree parameter
// is typed as `unknown` at the
// contract layer (see `LanguageProvider.emitScopeCaptures`); cast
// here at the use site.
let tree = cachedTree as ReturnType<ReturnType<typeof getPythonParser>['parse']> | undefined;

View file

@ -66,10 +66,9 @@
* site where the enclosing class can't be statically determined
* is left unresolved.
*
* Shadow-harness corpus parity is the authoritative signal for which
* of these matter in practice. The CI parity gate blocks any PR that
* regresses either the legacy or registry-primary run of
* `test/integration/resolvers/python.test.ts`.
* The `test/integration/resolvers/python.test.ts` resolver suite is the
* authoritative signal for which of these matter in practice; it runs in
* the standard CI test workflow, so a regression blocks the merge.
*/
export { emitPythonScopeCaptures } from './captures.js';

View file

@ -43,7 +43,15 @@
*/
import Parser from 'tree-sitter';
import Swift from 'tree-sitter-swift';
import { SupportedLanguages } from 'gitnexus-shared';
// `tree-sitter-swift` is an optional/vendored grammar that may be absent on a
// default install. It is loaded lazily + guarded via parser-loader rather than
// statically imported: this module is pulled onto the main thread eagerly by
// the scope-resolution registry and the language-provider index, so a top-level
// `import Swift from 'tree-sitter-swift'` would throw ERR_MODULE_NOT_FOUND at
// module-load and crash `analyze` even for repos with no Swift files (#2091,
// #2093). The grammar is only ever needed inside the lazy getters below.
import { getLanguageGrammar } from '../../../tree-sitter/parser-loader.js';
const SWIFT_SCOPE_QUERY = `
;; ── Scopes ──────────────────────────────────────────────────────────
@ -186,14 +194,19 @@ let _query: Parser.Query | null = null;
export function getSwiftParser(): Parser {
if (_parser === null) {
_parser = new Parser();
_parser.setLanguage(Swift as Parameters<Parser['setLanguage']>[0]);
_parser.setLanguage(
getLanguageGrammar(SupportedLanguages.Swift) as Parameters<Parser['setLanguage']>[0],
);
}
return _parser;
}
export function getSwiftScopeQuery(): Parser.Query {
if (_query === null) {
_query = new Parser.Query(Swift as Parameters<Parser['setLanguage']>[0], SWIFT_SCOPE_QUERY);
_query = new Parser.Query(
getLanguageGrammar(SupportedLanguages.Swift) as Parameters<Parser['setLanguage']>[0],
SWIFT_SCOPE_QUERY,
);
}
return _query;
}

View file

@ -163,10 +163,11 @@ export function emitTsScopeCaptures(
filePath: string,
cachedTree?: unknown,
): readonly CaptureMatch[] {
// Skip the parse when the caller (parse phase's scopeTreeCache) already
// produced a Tree for this source. Cache miss = re-parse, same as before.
// The cachedTree parameter is typed as `unknown` at the LanguageProvider
// contract layer; cast here at the use site.
// Reuse a pre-parsed Tree when the caller passes one via `cachedTree`; a
// miss re-parses. (The cache is currently always empty — its only producer,
// the sequential parser, was removed — so this re-parses in practice.) The
// cachedTree parameter is typed `unknown` at the LanguageProvider contract
// layer; cast here at the use site.
//
// Grammar selection: `.tsx` files are parsed with the TSX grammar,
// `.ts` files with the TypeScript grammar. The two grammars have

View file

@ -80,10 +80,9 @@
* identifiers are narrowed (`user instanceof User`). Member paths
* such as `user.address instanceof Address` remain unresolved.
*
* Shadow-harness corpus parity on `test/integration/resolvers/
* typescript.test.ts` is the authoritative signal for which of these
* matter in practice. The CI parity gate blocks any PR that regresses
* either the legacy or registry-primary run.
* The `test/integration/resolvers/typescript.test.ts` resolver suite is
* the authoritative signal for which of these matter in practice; it runs
* in the standard CI test workflow, so a regression blocks the merge.
*/
export { emitTsScopeCaptures } from './captures.js';

View file

@ -22,6 +22,7 @@
import type { CaptureMatch } from 'gitnexus-shared';
import { extractVueScript } from '../../vue-sfc-extractor.js';
import { emitTsScopeCaptures } from '../typescript/captures.js';
import { emitJsScopeCaptures } from '../javascript/captures.js';
/**
* Emit scope captures for a Vue SFC.
@ -31,11 +32,11 @@ import { emitTsScopeCaptures } from '../typescript/captures.js';
* 1. **Full SFC content** (sequential path, <15 files): `sourceText`
* contains the whole `.vue` file with `<template>`, `<script>`, etc.
* `extractVueScript` extracts the script block and we delegate to
* `emitTsScopeCaptures` with that extracted content.
* `emitTsScopeCaptures` or `emitJsScopeCaptures` based on `lang`.
*
* 2. **Already-extracted script content** (worker-mode path, ≥15 files):
* the parse worker calls `extractVueScript` itself before calling
* `extractParsedFile`, so `sourceText` is already the bare TypeScript
* `extractParsedFile`, so `sourceText` is already the bare script
* text with no `<script>` tags. The caller marks this explicitly via
* `sourceMeta.sourceKind === 'pre-extracted-script'`.
*
@ -48,7 +49,7 @@ export function emitVueScopeCaptures(
sourceText: string,
filePath: string,
cachedTree?: unknown,
sourceMeta?: { sourceKind?: 'full-file' | 'pre-extracted-script' },
sourceMeta?: { sourceKind?: 'full-file' | 'pre-extracted-script'; setupLang?: string },
): readonly CaptureMatch[] {
// Vue resolver may include supporting TS/JS files in the same run to
// preserve cross-file import/type context for `.vue` callers. These are
@ -58,10 +59,19 @@ export function emitVueScopeCaptures(
}
if (sourceMeta?.sourceKind === 'pre-extracted-script') {
// Worker-mode path: the parse worker always uses TypeScript grammar for
// .vue files. Lang-based grammar selection is a sequential-path feature.
return emitTsScopeCaptures(sourceText, filePath, cachedTree);
}
const extracted = extractVueScript(sourceText);
if (extracted === null) return [];
// Select captures based on script lang attribute.
// Use TS grammar unless ALL blocks explicitly request JS/JSX.
// Mixed-lang: TS handles JS natively; JS grammar chokes on TS syntax.
if (extracted.lang === 'js' || extracted.lang === 'jsx') {
return emitJsScopeCaptures(extracted.scriptContent, filePath, cachedTree);
}
return emitTsScopeCaptures(extracted.scriptContent, filePath, cachedTree);
}

Some files were not shown because too many files have changed in this diff Show more