diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json index 719473def..4dc2f3260 100644 --- a/.claude-plugin/marketplace.json +++ b/.claude-plugin/marketplace.json @@ -11,7 +11,7 @@ "plugins": [ { "name": "gitnexus", - "version": "1.3.3", + "version": "1.6.6", "source": "./gitnexus-claude-plugin", "description": "Code intelligence powered by a knowledge graph. Provides execution flow tracing, blast radius analysis, and augmented search across your codebase." } diff --git a/.claude/skills/gitnexus/gitnexus-debugging/SKILL.md b/.claude/skills/gitnexus/gitnexus-debugging/SKILL.md index 937b5e2a4..9834f94b7 100644 --- a/.claude/skills/gitnexus/gitnexus-debugging/SKILL.md +++ b/.claude/skills/gitnexus/gitnexus-debugging/SKILL.md @@ -16,10 +16,10 @@ description: "Use when the user is debugging a bug, tracing an error, or asking ## Workflow ``` -1. gitnexus_query({query: ""}) → Find related execution flows -2. gitnexus_context({name: ""}) → See callers/callees/processes +1. query({query: ""}) → Find related execution flows +2. context({name: ""}) → See callers/callees/processes 3. READ gitnexus://repo/{name}/process/{name} → Trace execution flow -4. gitnexus_cypher({query: "MATCH path..."}) → Custom traces if needed +4. cypher({query: "MATCH path..."}) → Custom traces if needed ``` > If "Index is stale" → run `node .gitnexus/run.cjs analyze` in terminal. @@ -28,11 +28,11 @@ description: "Use when the user is debugging a bug, tracing an error, or asking ``` - [ ] Understand the symptom (error message, unexpected behavior) -- [ ] gitnexus_query for error text or related code +- [ ] query for error text or related code - [ ] Identify the suspect function from returned processes -- [ ] gitnexus_context to see callers and callees +- [ ] context to see callers and callees - [ ] Trace execution flow via process resource if applicable -- [ ] gitnexus_cypher for custom call chain traces if needed +- [ ] cypher for custom call chain traces if needed - [ ] Read source files to confirm root cause ``` @@ -40,7 +40,7 @@ description: "Use when the user is debugging a bug, tracing an error, or asking | Symptom | GitNexus Approach | | -------------------- | ---------------------------------------------------------- | -| Error message | `gitnexus_query` for error text → `context` on throw sites | +| Error message | `query` for error text → `context` on throw sites | | Wrong return value | `context` on the function → trace callees for data flow | | Intermittent failure | `context` → look for external calls, async deps | | Performance issue | `context` → find symbols with many callers (hot paths) | @@ -48,24 +48,24 @@ description: "Use when the user is debugging a bug, tracing an error, or asking ## Tools -**gitnexus_query** — find code related to error: +**query** — find code related to error: ``` -gitnexus_query({query: "payment validation error"}) +query({query: "payment validation error"}) → Processes: CheckoutFlow, ErrorHandling → Symbols: validatePayment, handlePaymentError, PaymentException ``` -**gitnexus_context** — full context for a suspect: +**context** — full context for a suspect: ``` -gitnexus_context({name: "validatePayment"}) +context({name: "validatePayment"}) → Incoming calls: processCheckout, webhookHandler → Outgoing calls: verifyCard, fetchRates (external API!) → Processes: CheckoutFlow (step 3/7) ``` -**gitnexus_cypher** — custom call chain traces: +**cypher** — custom call chain traces: ```cypher MATCH path = (a)-[:CodeRelation {type: 'CALLS'}*1..2]->(b:Function {name: "validatePayment"}) @@ -75,11 +75,11 @@ RETURN [n IN nodes(path) | n.name] AS chain ## Example: "Payment endpoint returns 500 intermittently" ``` -1. gitnexus_query({query: "payment error handling"}) +1. query({query: "payment error handling"}) → Processes: CheckoutFlow, ErrorHandling → Symbols: validatePayment, handlePaymentError -2. gitnexus_context({name: "validatePayment"}) +2. context({name: "validatePayment"}) → Outgoing calls: verifyCard, fetchRates (external API!) 3. READ gitnexus://repo/my-app/process/CheckoutFlow diff --git a/.claude/skills/gitnexus/gitnexus-exploring/SKILL.md b/.claude/skills/gitnexus/gitnexus-exploring/SKILL.md index 2dcf7b578..ccf684c28 100644 --- a/.claude/skills/gitnexus/gitnexus-exploring/SKILL.md +++ b/.claude/skills/gitnexus/gitnexus-exploring/SKILL.md @@ -18,8 +18,8 @@ description: "Use when the user asks how code works, wants to understand archite ``` 1. READ gitnexus://repos → Discover indexed repos 2. READ gitnexus://repo/{name}/context → Codebase overview, check staleness -3. gitnexus_query({query: ""}) → Find related execution flows -4. gitnexus_context({name: ""}) → Deep dive on specific symbol +3. query({query: ""}) → Find related execution flows +4. context({name: ""}) → Deep dive on specific symbol 5. READ gitnexus://repo/{name}/process/{name} → Trace full execution flow ``` @@ -29,9 +29,9 @@ description: "Use when the user asks how code works, wants to understand archite ``` - [ ] READ gitnexus://repo/{name}/context -- [ ] gitnexus_query for the concept you want to understand +- [ ] query for the concept you want to understand - [ ] Review returned processes (execution flows) -- [ ] gitnexus_context on key symbols for callers/callees +- [ ] context on key symbols for callers/callees - [ ] READ process resource for full execution traces - [ ] Read source files for implementation details ``` @@ -47,18 +47,18 @@ description: "Use when the user asks how code works, wants to understand archite ## Tools -**gitnexus_query** — find execution flows related to a concept: +**query** — find execution flows related to a concept: ``` -gitnexus_query({query: "payment processing"}) +query({query: "payment processing"}) → Processes: CheckoutFlow, RefundFlow, WebhookHandler → Symbols grouped by flow with file locations ``` -**gitnexus_context** — 360-degree view of a symbol: +**context** — 360-degree view of a symbol: ``` -gitnexus_context({name: "validateUser"}) +context({name: "validateUser"}) → Incoming calls: loginHandler, apiMiddleware → Outgoing calls: checkToken, getUserById → Processes: LoginFlow (step 2/5), TokenRefresh (step 1/3) @@ -68,10 +68,10 @@ gitnexus_context({name: "validateUser"}) ``` 1. READ gitnexus://repo/my-app/context → 918 symbols, 45 processes -2. gitnexus_query({query: "payment processing"}) +2. query({query: "payment processing"}) → CheckoutFlow: processPayment → validateCard → chargeStripe → RefundFlow: initiateRefund → calculateRefund → processRefund -3. gitnexus_context({name: "processPayment"}) +3. context({name: "processPayment"}) → Incoming: checkoutHandler, webhookHandler → Outgoing: validateCard, chargeStripe, saveTransaction 4. Read src/payments/processor.ts for implementation details diff --git a/.claude/skills/gitnexus/gitnexus-impact-analysis/SKILL.md b/.claude/skills/gitnexus/gitnexus-impact-analysis/SKILL.md index 7206ca506..45eb7ce87 100644 --- a/.claude/skills/gitnexus/gitnexus-impact-analysis/SKILL.md +++ b/.claude/skills/gitnexus/gitnexus-impact-analysis/SKILL.md @@ -17,9 +17,9 @@ description: "Use when the user wants to know what will break if they change som ## Workflow ``` -1. gitnexus_impact({target: "X", direction: "upstream"}) → What depends on this +1. impact({target: "X", direction: "upstream"}) → What depends on this 2. READ gitnexus://repo/{name}/processes → Check affected execution flows -3. gitnexus_detect_changes() → Map current git changes to affected flows +3. detect_changes() → Map current git changes to affected flows 4. Assess risk and report to user ``` @@ -28,11 +28,11 @@ description: "Use when the user wants to know what will break if they change som ## Checklist ``` -- [ ] gitnexus_impact({target, direction: "upstream"}) to find dependents +- [ ] impact({target, direction: "upstream"}) to find dependents - [ ] Review d=1 items first (these WILL BREAK) - [ ] Check high-confidence (>0.8) dependencies - [ ] READ processes to check affected execution flows -- [ ] gitnexus_detect_changes() for pre-commit check +- [ ] detect_changes() for pre-commit check - [ ] Assess risk level and report to user ``` @@ -55,10 +55,10 @@ description: "Use when the user wants to know what will break if they change som ## Tools -**gitnexus_impact** — the primary tool for symbol blast radius: +**impact** — the primary tool for symbol blast radius: ``` -gitnexus_impact({ +impact({ target: "validateUser", direction: "upstream", minConfidence: 0.8, @@ -73,10 +73,10 @@ gitnexus_impact({ - authRouter (src/routes/auth.ts:22) [CALLS, 95%] ``` -**gitnexus_detect_changes** — git-diff based impact analysis: +**detect_changes** — git-diff based impact analysis: ``` -gitnexus_detect_changes({scope: "staged"}) +detect_changes({scope: "staged"}) → Changed: 5 symbols in 3 files → Affected: LoginFlow, TokenRefresh, APIMiddlewarePipeline @@ -86,7 +86,7 @@ gitnexus_detect_changes({scope: "staged"}) ## Example: "What breaks if I change validateUser?" ``` -1. gitnexus_impact({target: "validateUser", direction: "upstream"}) +1. impact({target: "validateUser", direction: "upstream"}) → d=1: loginHandler, apiMiddleware (WILL BREAK) → d=2: authRouter, sessionManager (LIKELY AFFECTED) diff --git a/.claude/skills/gitnexus/gitnexus-pr-review/SKILL.md b/.claude/skills/gitnexus/gitnexus-pr-review/SKILL.md index 319c063f9..9f1d362e5 100644 --- a/.claude/skills/gitnexus/gitnexus-pr-review/SKILL.md +++ b/.claude/skills/gitnexus/gitnexus-pr-review/SKILL.md @@ -18,10 +18,10 @@ description: "Use when the user wants to review a pull request, understand what ``` 1. gh pr diff → Get the raw diff -2. gitnexus_detect_changes({scope: "compare", base_ref: "main"}) → Map diff to affected flows +2. detect_changes({scope: "compare", base_ref: "main"}) → Map diff to affected flows 3. For each changed symbol: - gitnexus_impact({target: "", direction: "upstream"}) → Blast radius per change -4. gitnexus_context({name: ""}) → Understand callers/callees + impact({target: "", direction: "upstream"}) → Blast radius per change +4. context({name: ""}) → Understand callers/callees 5. READ gitnexus://repo/{name}/processes → Check affected execution flows 6. Summarize findings with risk assessment ``` @@ -32,10 +32,10 @@ description: "Use when the user wants to review a pull request, understand what ``` - [ ] Fetch PR diff (gh pr diff or git diff base...head) -- [ ] gitnexus_detect_changes to map changes to affected execution flows -- [ ] gitnexus_impact on each non-trivial changed symbol +- [ ] detect_changes to map changes to affected execution flows +- [ ] impact on each non-trivial changed symbol - [ ] Review d=1 items (WILL BREAK) — are callers updated? -- [ ] gitnexus_context on key changed symbols to understand full picture +- [ ] context on key changed symbols to understand full picture - [ ] Check if affected processes have test coverage - [ ] Assess overall risk level - [ ] Write review summary with findings @@ -63,20 +63,20 @@ description: "Use when the user wants to review a pull request, understand what ## Tools -**gitnexus_detect_changes** — map PR diff to affected execution flows: +**detect_changes** — map PR diff to affected execution flows: ``` -gitnexus_detect_changes({scope: "compare", base_ref: "main"}) +detect_changes({scope: "compare", base_ref: "main"}) → Changed: 8 symbols in 4 files → Affected processes: CheckoutFlow, RefundFlow, WebhookHandler → Risk: MEDIUM ``` -**gitnexus_impact** — blast radius per changed symbol: +**impact** — blast radius per changed symbol: ``` -gitnexus_impact({target: "validatePayment", direction: "upstream"}) +impact({target: "validatePayment", direction: "upstream"}) → d=1 (WILL BREAK): - processCheckout (src/checkout.ts:42) [CALLS, 100%] @@ -86,20 +86,20 @@ gitnexus_impact({target: "validatePayment", direction: "upstream"}) - checkoutRouter (src/routes/checkout.ts:22) [CALLS, 95%] ``` -**gitnexus_impact with tests** — check test coverage: +**impact with tests** — check test coverage: ``` -gitnexus_impact({target: "validatePayment", direction: "upstream", includeTests: true}) +impact({target: "validatePayment", direction: "upstream", includeTests: true}) → Tests that cover this symbol: - validatePayment.test.ts [direct] - checkout.integration.test.ts [via processCheckout] ``` -**gitnexus_context** — understand a changed symbol's role: +**context** — understand a changed symbol's role: ``` -gitnexus_context({name: "validatePayment"}) +context({name: "validatePayment"}) → Incoming calls: processCheckout, webhookHandler → Outgoing calls: verifyCard, fetchRates @@ -112,20 +112,20 @@ gitnexus_context({name: "validatePayment"}) 1. gh pr diff 42 > /tmp/pr42.diff → 4 files changed: payments.ts, checkout.ts, types.ts, utils.ts -2. gitnexus_detect_changes({scope: "compare", base_ref: "main"}) +2. detect_changes({scope: "compare", base_ref: "main"}) → Changed symbols: validatePayment, PaymentInput, formatAmount → Affected processes: CheckoutFlow, RefundFlow → Risk: MEDIUM -3. gitnexus_impact({target: "validatePayment", direction: "upstream"}) +3. impact({target: "validatePayment", direction: "upstream"}) → d=1: processCheckout, webhookHandler (WILL BREAK) → webhookHandler is NOT in the PR diff — potential breakage! -4. gitnexus_impact({target: "PaymentInput", direction: "upstream"}) +4. impact({target: "PaymentInput", direction: "upstream"}) → d=1: validatePayment (in PR), createPayment (NOT in PR) → createPayment uses the old PaymentInput shape — breaking change! -5. gitnexus_context({name: "formatAmount"}) +5. context({name: "formatAmount"}) → Called by 12 functions — but change is backwards-compatible (added optional param) 6. Review summary: diff --git a/.claude/skills/gitnexus/gitnexus-refactoring/SKILL.md b/.claude/skills/gitnexus/gitnexus-refactoring/SKILL.md index c749eb384..e13c04e14 100644 --- a/.claude/skills/gitnexus/gitnexus-refactoring/SKILL.md +++ b/.claude/skills/gitnexus/gitnexus-refactoring/SKILL.md @@ -16,9 +16,9 @@ description: "Use when the user wants to rename, extract, split, move, or restru ## Workflow ``` -1. gitnexus_impact({target: "X", direction: "upstream"}) → Map all dependents -2. gitnexus_query({query: "X"}) → Find execution flows involving X -3. gitnexus_context({name: "X"}) → See all incoming/outgoing refs +1. impact({target: "X", direction: "upstream"}) → Map all dependents +2. query({query: "X"}) → Find execution flows involving X +3. context({name: "X"}) → See all incoming/outgoing refs 4. Plan update order: interfaces → implementations → callers → tests ``` @@ -29,65 +29,65 @@ description: "Use when the user wants to rename, extract, split, move, or restru ### Rename Symbol ``` -- [ ] gitnexus_rename({symbol_name: "oldName", new_name: "newName", dry_run: true}) — preview all edits +- [ ] rename({symbol_name: "oldName", new_name: "newName", dry_run: true}) — preview all edits - [ ] Review graph edits (high confidence) and ast_search edits (review carefully) -- [ ] If satisfied: gitnexus_rename({..., dry_run: false}) — apply edits -- [ ] gitnexus_detect_changes() — verify only expected files changed +- [ ] If satisfied: rename({..., dry_run: false}) — apply edits +- [ ] detect_changes() — verify only expected files changed - [ ] Run tests for affected processes ``` ### Extract Module ``` -- [ ] gitnexus_context({name: target}) — see all incoming/outgoing refs -- [ ] gitnexus_impact({target, direction: "upstream"}) — find all external callers +- [ ] context({name: target}) — see all incoming/outgoing refs +- [ ] impact({target, direction: "upstream"}) — find all external callers - [ ] Define new module interface - [ ] Extract code, update imports -- [ ] gitnexus_detect_changes() — verify affected scope +- [ ] detect_changes() — verify affected scope - [ ] Run tests for affected processes ``` ### Split Function/Service ``` -- [ ] gitnexus_context({name: target}) — understand all callees +- [ ] context({name: target}) — understand all callees - [ ] Group callees by responsibility -- [ ] gitnexus_impact({target, direction: "upstream"}) — map callers to update +- [ ] impact({target, direction: "upstream"}) — map callers to update - [ ] Create new functions/services - [ ] Update callers -- [ ] gitnexus_detect_changes() — verify affected scope +- [ ] detect_changes() — verify affected scope - [ ] Run tests for affected processes ``` ## Tools -**gitnexus_rename** — automated multi-file rename: +**rename** — automated multi-file rename: ``` -gitnexus_rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: true}) +rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: true}) → 12 edits across 8 files → 10 graph edits (high confidence), 2 ast_search edits (review) → Changes: [{file_path, edits: [{line, old_text, new_text, confidence}]}] ``` -**gitnexus_impact** — map all dependents first: +**impact** — map all dependents first: ``` -gitnexus_impact({target: "validateUser", direction: "upstream"}) +impact({target: "validateUser", direction: "upstream"}) → d=1: loginHandler, apiMiddleware, testUtils → Affected Processes: LoginFlow, TokenRefresh ``` -**gitnexus_detect_changes** — verify your changes after refactoring: +**detect_changes** — verify your changes after refactoring: ``` -gitnexus_detect_changes({scope: "all"}) +detect_changes({scope: "all"}) → Changed: 8 files, 12 symbols → Affected processes: LoginFlow, TokenRefresh → Risk: MEDIUM ``` -**gitnexus_cypher** — custom reference queries: +**cypher** — custom reference queries: ```cypher MATCH (caller)-[:CodeRelation {type: 'CALLS'}]->(f:Function {name: "validateUser"}) @@ -98,24 +98,24 @@ RETURN caller.name, caller.filePath ORDER BY caller.filePath | Risk Factor | Mitigation | | ------------------- | ----------------------------------------- | -| Many callers (>5) | Use gitnexus_rename for automated updates | +| Many callers (>5) | Use rename for automated updates | | Cross-area refs | Use detect_changes after to verify scope | -| String/dynamic refs | gitnexus_query to find them | +| String/dynamic refs | query to find them | | External/public API | Version and deprecate properly | ## Example: Rename `validateUser` to `authenticateUser` ``` -1. gitnexus_rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: true}) +1. rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: true}) → 12 edits: 10 graph (safe), 2 ast_search (review) → Files: validator.ts, login.ts, middleware.ts, config.json... 2. Review ast_search edits (config.json: dynamic reference!) -3. gitnexus_rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: false}) +3. rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: false}) → Applied 12 edits across 8 files -4. gitnexus_detect_changes({scope: "all"}) +4. detect_changes({scope: "all"}) → Affected: LoginFlow, TokenRefresh → Risk: MEDIUM — run tests for these flows ``` diff --git a/AGENTS.md b/AGENTS.md index 1e31004e4..d4b09c8ce 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -80,18 +80,18 @@ This project is indexed by GitNexus as **GitNexus** (26675 symbols, 35395 relati ## Always Do -- **MUST run impact analysis before editing any symbol.** Before modifying a function, class, or method, run `gitnexus_impact({target: "symbolName", direction: "upstream"})` and report the blast radius (direct callers, affected processes, risk level) to the user. -- **MUST run `gitnexus_detect_changes()` before committing** to verify your changes only affect expected symbols and execution flows. +- **MUST run impact analysis before editing any symbol.** Before modifying a function, class, or method, run `impact({target: "symbolName", direction: "upstream"})` and report the blast radius (direct callers, affected processes, risk level) to the user. +- **MUST run `detect_changes()` before committing** to verify your changes only affect expected symbols and execution flows. - **MUST warn the user** if impact analysis returns HIGH or CRITICAL risk before proceeding with edits. -- When exploring unfamiliar code, use `gitnexus_query({query: "concept"})` to find execution flows instead of grepping. It returns process-grouped results ranked by relevance. -- When you need full context on a specific symbol — callers, callees, which execution flows it participates in — use `gitnexus_context({name: "symbolName"})`. +- When exploring unfamiliar code, use `query({query: "concept"})` to find execution flows instead of grepping. It returns process-grouped results ranked by relevance. +- When you need full context on a specific symbol — callers, callees, which execution flows it participates in — use `context({name: "symbolName"})`. ## Never Do -- NEVER edit a function, class, or method without first running `gitnexus_impact` on it. +- NEVER edit a function, class, or method without first running `impact` on it. - NEVER ignore HIGH or CRITICAL risk warnings from impact analysis. -- NEVER rename symbols with find-and-replace — use `gitnexus_rename` which understands the call graph. -- NEVER commit changes without running `gitnexus_detect_changes()` to check affected scope. +- NEVER rename symbols with find-and-replace — use `rename` which understands the call graph. +- NEVER commit changes without running `detect_changes()` to check affected scope. ## Resources diff --git a/ARCHITECTURE.md b/ARCHITECTURE.md index f65c175f9..01013c71c 100644 --- a/ARCHITECTURE.md +++ b/ARCHITECTURE.md @@ -15,7 +15,7 @@ Monorepo: **CLI/MCP** (`gitnexus/`) + **browser UI** (`gitnexus-web/`). ## End-to-end flow: index → graph → tools -1. **Ingestion** — `analyze.ts` → `runFullAnalysis` (`run-analyze.ts`) → `runPipelineFromRepo` (`pipeline.ts`). DAG of 12 phases builds a `KnowledgeGraph` in memory, then loads into LadybugDB under `.gitnexus/`. Repo registered in `~/.gitnexus/registry.json` for MCP discovery. +1. **Ingestion** — `analyze.ts` → `runFullAnalysis` (`run-analyze.ts`) → `runPipelineFromRepo` (`pipeline.ts`). DAG of 14 phases builds a `KnowledgeGraph` in memory, then loads into LadybugDB under `.gitnexus/`. Repo registered in `~/.gitnexus/registry.json` for MCP discovery. 2. **Persistence** — `repo-manager.ts` (paths, registry, KuzuDB cleanup). `lbug-adapter.ts` (graph load, queries, embedding batches). @@ -77,11 +77,11 @@ Monorepo: **CLI/MCP** (`gitnexus/`) + **browser UI** (`gitnexus-web/`). ## Pipeline Phase DAG -12 phases defined in `gitnexus/src/core/ingestion/pipeline-phases/`, each with explicit `deps` and typed output. +14 phases defined in `gitnexus/src/core/ingestion/pipeline-phases/`, each with explicit `deps` and typed output. ``` scan → structure → [markdown, cobol] → parse → [routes, tools, orm] - → crossFile → mro → communities → processes + → crossFile → scopeResolution → pruneLocalSymbols → mro → communities → processes ``` | Phase | File | Deps | Output | @@ -95,11 +95,13 @@ scan → structure → [markdown, cobol] → parse → [routes, tools, orm] | `tools` | `tools.ts` | `parse` | Tool nodes + HANDLES_TOOL edges | | `orm` | `orm.ts` | `parse` | QUERIES edges (Prisma, Supabase) | | `crossFile` | `cross-file.ts` + `cross-file-impl.ts` | `parse`, `routes`, `tools`, `orm` | Cross-file type propagation in topological import order | -| `mro` | `mro.ts` | `crossFile`, `structure` | METHOD_OVERRIDES + METHOD_IMPLEMENTS edges | -| `communities` | `communities.ts` | `mro`, `structure` | Community nodes + MEMBER_OF edges (Leiden algorithm) | -| `processes` | `processes.ts` | `communities`, `routes`, `tools`, `structure` | Process nodes + STEP_IN_PROCESS edges | +| `scopeResolution` | `scope-resolution/pipeline/phase.ts` | `parse`, `crossFile`, `structure` | Binding/reference + inheritance edges; disposes BindingAccumulator | +| `pruneLocalSymbols` | `prune-local-symbols.ts` | `scopeResolution` | Drops inert block-local `Const`/`Variable`/`Static` nodes (only a `File→DEFINES` edge) post-resolution | +| `mro` | `mro.ts` | `crossFile`, `scopeResolution`, `pruneLocalSymbols`, `structure` | METHOD_OVERRIDES + METHOD_IMPLEMENTS edges | +| `communities` | `communities.ts` | `mro`, `pruneLocalSymbols`, `structure` | Community nodes + MEMBER_OF edges (Leiden algorithm) | +| `processes` | `processes.ts` | `communities`, `routes`, `tools`, `pruneLocalSymbols`, `structure` | Process nodes + STEP_IN_PROCESS edges | -**Non-phase files in the same directory:** `parse-impl.ts`, `cross-file-impl.ts` (implementation), `wildcard-synthesis.ts` (whole-module import expansion), `orm-extraction.ts` (sequential ORM fallback), `types.ts`, `runner.ts`, `index.ts`. +**Non-phase files in the same directory:** `parse-impl.ts`, `cross-file-impl.ts` (implementation), `wildcard-synthesis.ts` (whole-module import expansion), `types.ts`, `runner.ts`, `index.ts`. ### DAG runner @@ -119,7 +121,8 @@ scan → structure → [markdown, cobol] → parse → [routes, tools, orm] - **Single graph accumulator** — all phases mutate the same `KnowledgeGraph` in `ctx`; the graph is the primary output. - **Typed phase access** — `getPhaseOutput(deps, 'name')` for type-safe upstream results. - **Binding accumulator lifecycle** — created in `parse`, disposed by `crossFile` (in `finally`). No other phase should take ownership. -- **Skippable phases** — `skipGraphPhases` omits MRO/communities/processes (faster tests). `skipWorkers` forces sequential parsing. +- **Skippable phases** — `skipGraphPhases` omits MRO/communities/processes (faster tests); `pruneLocalSymbols` still runs (it is graph cleanup, not analysis). `skipWorkers` is no longer a sequential escape hatch — it (like `--workers 0` / `GITNEXUS_WORKER_POOL_SIZE=0`) is rejected with an actionable error, since the worker pool is the sole parse path (§ Chunked parse-and-resolve). +- **Local-symbol pruning** — `pruneLocalSymbols` removes inert block-local value symbols after scope resolution has consumed them. Opt out per-call with `PipelineOptions.keepLocalValueSymbols` or globally with the `GITNEXUS_KEEP_LOCAL_VALUE_SYMBOLS` env var. ### How to add a new phase @@ -199,7 +202,7 @@ Language-agnostic scope-resolution resolver. This is the resolution path for eve ``` Orchestrator: `runScopeResolution(input, provider)` in `scope-resolution/pipeline/run.ts`. -Pipeline phase: `scopeResolutionPhase` in `scope-resolution/pipeline/phase.ts` — iterates the registered `SCOPE_RESOLVERS`, reads per-file Trees from the parse phase's `scopeTreeCache`, disposes the cache at the end. +Pipeline phase: `scopeResolutionPhase` in `scope-resolution/pipeline/phase.ts` — iterates the registered `SCOPE_RESOLVERS` over the worker-serialized `ParsedFile`s. (Per-language `emitScopeCaptures` hooks may reuse a cached Tree via the orchestrator's `treeCache`, but in worker-pool runs that cache is empty — Trees can't cross MessageChannels — so they consume the pre-extracted `ParsedFile` instead; § Performance notes.) ### `ScopeResolver` contract @@ -248,7 +251,7 @@ CI auto-discovers the set via `tsx`. No workflow edit required. ### Performance notes -- **Cross-phase Tree cache**: parse phase writes Trees into `scopeTreeCache` (separate from the chunk-local `astCache`) ONLY for languages with `emitScopeCaptures`. Scope-resolution reads from it to skip the second parse. Cleared at end of the phase. Workers leave the cache empty — Trees can't cross MessageChannels; cache miss = fresh parse. `PROF_SCOPE_RESOLUTION=1` emits hit/miss counters and a worker-engaged warning. +- **Cross-phase Tree cache**: the orchestrator's `treeCache` (`RunScopeResolutionInput.treeCache`) lets a scope-resolution per-language hook (`emitScopeCaptures`) reuse a tree instead of re-parsing. Workers leave it empty — Trees can't cross MessageChannels — so in normal (worker-pool) runs scope-resolution does NOT rely on it: workers serialize each file's `ParsedFile` (+ capture side-channel) and stream them in, so scope-resolution consumes the pre-extracted artifact rather than re-parsing on the main thread (§ Chunked parse-and-resolve). `PROF_SCOPE_RESOLUTION=1` emits hit/miss counters and a worker-engaged warning. - **Typed relationship iteration**: heritage + MRO walk only the EXTENDS / IMPLEMENTS / HAS_METHOD edges via `iterRelationshipsByType`, not the full relationship map. - **Workspace-resolution-index**: O(1) `findOwnedMember` / `findExportedDef` / `classScopeByDefId` built once per run. - **SCC-ordered cross-file return-type propagation** (PR #1050): `propagateImportedReturnTypes` walks `indexes.sccs` in reverse-topological order (leaves first), so multi-hop alias chains like `models.User → service.user → app.user` collapse to the terminal class in a single linear pass. Within each importer, the source module's `typeBindings` is chain-followed BEFORE mirroring (so we mirror terminal types, not intermediate refs), and the importer's own `typeBindings` is chain-followed AFTER mirroring (so local `const x = importedFn()` resolves before downstream importers run). Cyclic SCCs reach a partial fixpoint within a single pass without iterating to convergence — see the `ts-circular` cross-file-binding fixture which only asserts pipeline-no-throw. PROF output (`PROF_SCOPE_RESOLUTION=1`) splits `finalize` from `propagate` so quadratic regressions in the chain-follow surface independently. @@ -311,7 +314,7 @@ Unified 3-tier algorithm (`model/resolution-context.ts`), per-language `importSe ### Chunked parse-and-resolve `parse` processes files in ~20 MB byte-budget chunks to bound memory. Per chunk: -1. Worker pool dispatches files (or sequential fallback via `skipWorkers`) +1. Worker pool dispatches files (the sole parse path — there is no sequential fallback; `skipWorkers`, `--workers 0`, and `GITNEXUS_WORKER_POOL_SIZE=0` are rejected with an actionable error) 2. Each worker: detect language → load grammar → run queries → return unified `ParseWorkerResult` 3. Synthesize wildcard bindings (`wildcard-synthesis.ts`) 4. Resolve imports @@ -321,6 +324,8 @@ Inheritance edges are emitted later, by the scope-resolution phase (`preEmitInhe Workers: `workers/worker-pool.ts`, `workers/parse-worker.ts`. +**Worker-serialized ParsedFiles (#2038).** To index very large repos (e.g. the Linux kernel) without OOM, the worker pool is the *sole* parse path and workers serialize each file's `ParsedFile` (plus its capture side-channel) in parallel, streaming them to scope-resolution through a disk-backed store. Scope-resolution consumes the pre-extracted artifact instead of re-parsing every file on the main thread — tree-sitter's native input buffers are not GC-reclaimable, so the former main-thread re-parse leaked native memory until the process died. Pool creation is lazy / cache-miss-gated, so a warm all-cache-hit run replays cached worker output without spawning a worker (hence `usedWorkerPool` can be false even when the repo has parseable files). + ### Inheritance and MRO Inheritance is captured by the `@reference.inherits` tag and emitted by the scope-resolution phase: `preEmitInheritanceEdges` resolves each base in scope, then `emitHeritageEdges` writes the `EXTENDS`/`IMPLEMENTS` edges. The phase then computes method resolution order via each `ScopeResolver`'s `buildMro` hook, feeding a `MethodDispatchIndex` used for owner-scoped lookups. Per-language strategy: diff --git a/CLAUDE.md b/CLAUDE.md index bbb991589..f2bf1e487 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -62,18 +62,18 @@ This project is indexed by GitNexus as **GitNexus** (26675 symbols, 35395 relati ## Always Do -- **MUST run impact analysis before editing any symbol.** Before modifying a function, class, or method, run `gitnexus_impact({target: "symbolName", direction: "upstream"})` and report the blast radius (direct callers, affected processes, risk level) to the user. -- **MUST run `gitnexus_detect_changes()` before committing** to verify your changes only affect expected symbols and execution flows. +- **MUST run impact analysis before editing any symbol.** Before modifying a function, class, or method, run `impact({target: "symbolName", direction: "upstream"})` and report the blast radius (direct callers, affected processes, risk level) to the user. +- **MUST run `detect_changes()` before committing** to verify your changes only affect expected symbols and execution flows. - **MUST warn the user** if impact analysis returns HIGH or CRITICAL risk before proceeding with edits. -- When exploring unfamiliar code, use `gitnexus_query({query: "concept"})` to find execution flows instead of grepping. It returns process-grouped results ranked by relevance. -- When you need full context on a specific symbol — callers, callees, which execution flows it participates in — use `gitnexus_context({name: "symbolName"})`. +- When exploring unfamiliar code, use `query({query: "concept"})` to find execution flows instead of grepping. It returns process-grouped results ranked by relevance. +- When you need full context on a specific symbol — callers, callees, which execution flows it participates in — use `context({name: "symbolName"})`. ## Never Do -- NEVER edit a function, class, or method without first running `gitnexus_impact` on it. +- NEVER edit a function, class, or method without first running `impact` on it. - NEVER ignore HIGH or CRITICAL risk warnings from impact analysis. -- NEVER rename symbols with find-and-replace — use `gitnexus_rename` which understands the call graph. -- NEVER commit changes without running `gitnexus_detect_changes()` to check affected scope. +- NEVER rename symbols with find-and-replace — use `rename` which understands the call graph. +- NEVER commit changes without running `detect_changes()` to check affected scope. ## Resources diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index ddefad384..848884be4 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -157,7 +157,12 @@ routes between two modes based on the triggering event: suffix; RC tags are excluded at trigger via a negative glob). Publishes to the `latest` dist-tag with a changelog-backed GitHub release. Maintainers are expected to tag from `main` as a convention; the workflow itself does - not enforce branch reachability. No Docker build (RC-only). + not enforce branch reachability. No Docker build (RC-only). Before cutting a + stable release, keep `gitnexus/package.json`, + `gitnexus-claude-plugin/.claude-plugin/plugin.json`, + `.claude-plugin/marketplace.json`, and the matching `CHANGELOG.md` entry in + lockstep — the always-on `gitnexus` unit suite now fails if those manifest + versions drift. - **Release-candidate mode** — runs on every push to `main` (typically a merged PR) plus manual `workflow_dispatch`. Docs-only changes are skipped via `paths-ignore`. Publishes to the `rc` dist-tag with version diff --git a/README.md b/README.md index 28229c98c..2df823d9a 100644 --- a/README.md +++ b/README.md @@ -236,7 +236,7 @@ gitnexus analyze --embeddings [limit] # Enable embedding generation (slower, be gitnexus analyze --verbose # Log skipped files when parsers are unavailable gitnexus analyze --worker-timeout 60 # Increase worker idle timeout for slow parses gitnexus analyze --wal-checkpoint-threshold 67108864 # 64 MiB. Control LadybugDB WAL auto-checkpoint threshold (default: 67108864 = 64 MiB; -1 keeps Ladybug stock ~16 MiB) -gitnexus analyze --workers # Parse worker pool size (default: cores-1, capped at 16; 0 = sequential) +gitnexus analyze --workers # Parse worker pool size (>=1; default: cores-1, capped at 16, auto-sized to the repo). 0 is rejected — there is no sequential mode. gitnexus mcp # Start MCP server (stdio) — serves all indexed repos gitnexus serve # Start local HTTP server (multi-repo) for web UI connection gitnexus list # List all indexed repositories @@ -314,7 +314,7 @@ Most `analyze` knobs are also CLI flags (`--workers`, `--worker-timeout`, `--max | Variable | Default | Effect | Tune when… | | -------------------------------------- | ------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------- | -| `GITNEXUS_WORKER_POOL_SIZE` | `cores - 1`, capped at 16 | Parse worker pool size. `0` disables the pool (sequential fallback). Equivalent to `--workers `. | Constrained containers (cgroup CPU limits), CI runners with explicit quotas, or debugging a worker-only crash via `0`. | +| `GITNEXUS_WORKER_POOL_SIZE` | `cores - 1`, capped at 16 | Parse worker pool size (must be ≥ 1). Equivalent to `--workers `. The worker pool is the sole parse path — there is no sequential parser, so `0` is rejected with an actionable error (the pool self-heals via quarantine + respawn). | Constrained containers (cgroup CPU limits) or CI runners with explicit quotas. To narrow down a worker crash set `1` for a single-worker pool — not `0`. | | `GITNEXUS_PARSE_CHUNK_CONCURRENCY` | `2` | Number of chunks whose file contents may be read into memory in parallel while the pool dispatches the current chunk. Worker dispatch itself stays serial. | Repos large enough to chunk (multi-MB total source) where disk I/O is a measurable fraction of analyze wall-clock. | | `GITNEXUS_VERBOSE` | unset | When `1`, enables verbose ingestion logs (skipped-file warnings, per-chunk throughput, parse-cache stats). Equivalent to `--verbose`. | Debugging an analyze that "completed" but seems to have missed files; tuning `--workers` / chunk concurrency against observable throughput. | | `GITNEXUS_PROFILE_DEFERRED` | unset | When `1`, emits `[deferred-profile]` timing/progress logs for the post-chunk deferred resolution band (imports → heritage → buildHeritageMap → legacy call resolution). Implied by `GITNEXUS_VERBOSE`. | Diagnosing analyze stalls in "Resolving calls (all chunks)" on large Java/Kotlin repos (issue #1741) without the full verbose ingestion noise. | diff --git a/gitnexus-claude-plugin/.claude-plugin/plugin.json b/gitnexus-claude-plugin/.claude-plugin/plugin.json index bd4b8c426..da6b42a7b 100644 --- a/gitnexus-claude-plugin/.claude-plugin/plugin.json +++ b/gitnexus-claude-plugin/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "gitnexus", "description": "Code intelligence powered by a knowledge graph. Provides execution flow tracing, blast radius analysis, and augmented search across your codebase.", - "version": "1.3.6", + "version": "1.6.6", "author": { "name": "GitNexus" }, diff --git a/gitnexus-claude-plugin/skills/gitnexus-debugging/SKILL.md b/gitnexus-claude-plugin/skills/gitnexus-debugging/SKILL.md index 937b5e2a4..9834f94b7 100644 --- a/gitnexus-claude-plugin/skills/gitnexus-debugging/SKILL.md +++ b/gitnexus-claude-plugin/skills/gitnexus-debugging/SKILL.md @@ -16,10 +16,10 @@ description: "Use when the user is debugging a bug, tracing an error, or asking ## Workflow ``` -1. gitnexus_query({query: ""}) → Find related execution flows -2. gitnexus_context({name: ""}) → See callers/callees/processes +1. query({query: ""}) → Find related execution flows +2. context({name: ""}) → See callers/callees/processes 3. READ gitnexus://repo/{name}/process/{name} → Trace execution flow -4. gitnexus_cypher({query: "MATCH path..."}) → Custom traces if needed +4. cypher({query: "MATCH path..."}) → Custom traces if needed ``` > If "Index is stale" → run `node .gitnexus/run.cjs analyze` in terminal. @@ -28,11 +28,11 @@ description: "Use when the user is debugging a bug, tracing an error, or asking ``` - [ ] Understand the symptom (error message, unexpected behavior) -- [ ] gitnexus_query for error text or related code +- [ ] query for error text or related code - [ ] Identify the suspect function from returned processes -- [ ] gitnexus_context to see callers and callees +- [ ] context to see callers and callees - [ ] Trace execution flow via process resource if applicable -- [ ] gitnexus_cypher for custom call chain traces if needed +- [ ] cypher for custom call chain traces if needed - [ ] Read source files to confirm root cause ``` @@ -40,7 +40,7 @@ description: "Use when the user is debugging a bug, tracing an error, or asking | Symptom | GitNexus Approach | | -------------------- | ---------------------------------------------------------- | -| Error message | `gitnexus_query` for error text → `context` on throw sites | +| Error message | `query` for error text → `context` on throw sites | | Wrong return value | `context` on the function → trace callees for data flow | | Intermittent failure | `context` → look for external calls, async deps | | Performance issue | `context` → find symbols with many callers (hot paths) | @@ -48,24 +48,24 @@ description: "Use when the user is debugging a bug, tracing an error, or asking ## Tools -**gitnexus_query** — find code related to error: +**query** — find code related to error: ``` -gitnexus_query({query: "payment validation error"}) +query({query: "payment validation error"}) → Processes: CheckoutFlow, ErrorHandling → Symbols: validatePayment, handlePaymentError, PaymentException ``` -**gitnexus_context** — full context for a suspect: +**context** — full context for a suspect: ``` -gitnexus_context({name: "validatePayment"}) +context({name: "validatePayment"}) → Incoming calls: processCheckout, webhookHandler → Outgoing calls: verifyCard, fetchRates (external API!) → Processes: CheckoutFlow (step 3/7) ``` -**gitnexus_cypher** — custom call chain traces: +**cypher** — custom call chain traces: ```cypher MATCH path = (a)-[:CodeRelation {type: 'CALLS'}*1..2]->(b:Function {name: "validatePayment"}) @@ -75,11 +75,11 @@ RETURN [n IN nodes(path) | n.name] AS chain ## Example: "Payment endpoint returns 500 intermittently" ``` -1. gitnexus_query({query: "payment error handling"}) +1. query({query: "payment error handling"}) → Processes: CheckoutFlow, ErrorHandling → Symbols: validatePayment, handlePaymentError -2. gitnexus_context({name: "validatePayment"}) +2. context({name: "validatePayment"}) → Outgoing calls: verifyCard, fetchRates (external API!) 3. READ gitnexus://repo/my-app/process/CheckoutFlow diff --git a/gitnexus-claude-plugin/skills/gitnexus-exploring/SKILL.md b/gitnexus-claude-plugin/skills/gitnexus-exploring/SKILL.md index 2dcf7b578..ccf684c28 100644 --- a/gitnexus-claude-plugin/skills/gitnexus-exploring/SKILL.md +++ b/gitnexus-claude-plugin/skills/gitnexus-exploring/SKILL.md @@ -18,8 +18,8 @@ description: "Use when the user asks how code works, wants to understand archite ``` 1. READ gitnexus://repos → Discover indexed repos 2. READ gitnexus://repo/{name}/context → Codebase overview, check staleness -3. gitnexus_query({query: ""}) → Find related execution flows -4. gitnexus_context({name: ""}) → Deep dive on specific symbol +3. query({query: ""}) → Find related execution flows +4. context({name: ""}) → Deep dive on specific symbol 5. READ gitnexus://repo/{name}/process/{name} → Trace full execution flow ``` @@ -29,9 +29,9 @@ description: "Use when the user asks how code works, wants to understand archite ``` - [ ] READ gitnexus://repo/{name}/context -- [ ] gitnexus_query for the concept you want to understand +- [ ] query for the concept you want to understand - [ ] Review returned processes (execution flows) -- [ ] gitnexus_context on key symbols for callers/callees +- [ ] context on key symbols for callers/callees - [ ] READ process resource for full execution traces - [ ] Read source files for implementation details ``` @@ -47,18 +47,18 @@ description: "Use when the user asks how code works, wants to understand archite ## Tools -**gitnexus_query** — find execution flows related to a concept: +**query** — find execution flows related to a concept: ``` -gitnexus_query({query: "payment processing"}) +query({query: "payment processing"}) → Processes: CheckoutFlow, RefundFlow, WebhookHandler → Symbols grouped by flow with file locations ``` -**gitnexus_context** — 360-degree view of a symbol: +**context** — 360-degree view of a symbol: ``` -gitnexus_context({name: "validateUser"}) +context({name: "validateUser"}) → Incoming calls: loginHandler, apiMiddleware → Outgoing calls: checkToken, getUserById → Processes: LoginFlow (step 2/5), TokenRefresh (step 1/3) @@ -68,10 +68,10 @@ gitnexus_context({name: "validateUser"}) ``` 1. READ gitnexus://repo/my-app/context → 918 symbols, 45 processes -2. gitnexus_query({query: "payment processing"}) +2. query({query: "payment processing"}) → CheckoutFlow: processPayment → validateCard → chargeStripe → RefundFlow: initiateRefund → calculateRefund → processRefund -3. gitnexus_context({name: "processPayment"}) +3. context({name: "processPayment"}) → Incoming: checkoutHandler, webhookHandler → Outgoing: validateCard, chargeStripe, saveTransaction 4. Read src/payments/processor.ts for implementation details diff --git a/gitnexus-claude-plugin/skills/gitnexus-impact-analysis/SKILL.md b/gitnexus-claude-plugin/skills/gitnexus-impact-analysis/SKILL.md index 7206ca506..45eb7ce87 100644 --- a/gitnexus-claude-plugin/skills/gitnexus-impact-analysis/SKILL.md +++ b/gitnexus-claude-plugin/skills/gitnexus-impact-analysis/SKILL.md @@ -17,9 +17,9 @@ description: "Use when the user wants to know what will break if they change som ## Workflow ``` -1. gitnexus_impact({target: "X", direction: "upstream"}) → What depends on this +1. impact({target: "X", direction: "upstream"}) → What depends on this 2. READ gitnexus://repo/{name}/processes → Check affected execution flows -3. gitnexus_detect_changes() → Map current git changes to affected flows +3. detect_changes() → Map current git changes to affected flows 4. Assess risk and report to user ``` @@ -28,11 +28,11 @@ description: "Use when the user wants to know what will break if they change som ## Checklist ``` -- [ ] gitnexus_impact({target, direction: "upstream"}) to find dependents +- [ ] impact({target, direction: "upstream"}) to find dependents - [ ] Review d=1 items first (these WILL BREAK) - [ ] Check high-confidence (>0.8) dependencies - [ ] READ processes to check affected execution flows -- [ ] gitnexus_detect_changes() for pre-commit check +- [ ] detect_changes() for pre-commit check - [ ] Assess risk level and report to user ``` @@ -55,10 +55,10 @@ description: "Use when the user wants to know what will break if they change som ## Tools -**gitnexus_impact** — the primary tool for symbol blast radius: +**impact** — the primary tool for symbol blast radius: ``` -gitnexus_impact({ +impact({ target: "validateUser", direction: "upstream", minConfidence: 0.8, @@ -73,10 +73,10 @@ gitnexus_impact({ - authRouter (src/routes/auth.ts:22) [CALLS, 95%] ``` -**gitnexus_detect_changes** — git-diff based impact analysis: +**detect_changes** — git-diff based impact analysis: ``` -gitnexus_detect_changes({scope: "staged"}) +detect_changes({scope: "staged"}) → Changed: 5 symbols in 3 files → Affected: LoginFlow, TokenRefresh, APIMiddlewarePipeline @@ -86,7 +86,7 @@ gitnexus_detect_changes({scope: "staged"}) ## Example: "What breaks if I change validateUser?" ``` -1. gitnexus_impact({target: "validateUser", direction: "upstream"}) +1. impact({target: "validateUser", direction: "upstream"}) → d=1: loginHandler, apiMiddleware (WILL BREAK) → d=2: authRouter, sessionManager (LIKELY AFFECTED) diff --git a/gitnexus-claude-plugin/skills/gitnexus-pr-review/SKILL.md b/gitnexus-claude-plugin/skills/gitnexus-pr-review/SKILL.md index 319c063f9..9f1d362e5 100644 --- a/gitnexus-claude-plugin/skills/gitnexus-pr-review/SKILL.md +++ b/gitnexus-claude-plugin/skills/gitnexus-pr-review/SKILL.md @@ -18,10 +18,10 @@ description: "Use when the user wants to review a pull request, understand what ``` 1. gh pr diff → Get the raw diff -2. gitnexus_detect_changes({scope: "compare", base_ref: "main"}) → Map diff to affected flows +2. detect_changes({scope: "compare", base_ref: "main"}) → Map diff to affected flows 3. For each changed symbol: - gitnexus_impact({target: "", direction: "upstream"}) → Blast radius per change -4. gitnexus_context({name: ""}) → Understand callers/callees + impact({target: "", direction: "upstream"}) → Blast radius per change +4. context({name: ""}) → Understand callers/callees 5. READ gitnexus://repo/{name}/processes → Check affected execution flows 6. Summarize findings with risk assessment ``` @@ -32,10 +32,10 @@ description: "Use when the user wants to review a pull request, understand what ``` - [ ] Fetch PR diff (gh pr diff or git diff base...head) -- [ ] gitnexus_detect_changes to map changes to affected execution flows -- [ ] gitnexus_impact on each non-trivial changed symbol +- [ ] detect_changes to map changes to affected execution flows +- [ ] impact on each non-trivial changed symbol - [ ] Review d=1 items (WILL BREAK) — are callers updated? -- [ ] gitnexus_context on key changed symbols to understand full picture +- [ ] context on key changed symbols to understand full picture - [ ] Check if affected processes have test coverage - [ ] Assess overall risk level - [ ] Write review summary with findings @@ -63,20 +63,20 @@ description: "Use when the user wants to review a pull request, understand what ## Tools -**gitnexus_detect_changes** — map PR diff to affected execution flows: +**detect_changes** — map PR diff to affected execution flows: ``` -gitnexus_detect_changes({scope: "compare", base_ref: "main"}) +detect_changes({scope: "compare", base_ref: "main"}) → Changed: 8 symbols in 4 files → Affected processes: CheckoutFlow, RefundFlow, WebhookHandler → Risk: MEDIUM ``` -**gitnexus_impact** — blast radius per changed symbol: +**impact** — blast radius per changed symbol: ``` -gitnexus_impact({target: "validatePayment", direction: "upstream"}) +impact({target: "validatePayment", direction: "upstream"}) → d=1 (WILL BREAK): - processCheckout (src/checkout.ts:42) [CALLS, 100%] @@ -86,20 +86,20 @@ gitnexus_impact({target: "validatePayment", direction: "upstream"}) - checkoutRouter (src/routes/checkout.ts:22) [CALLS, 95%] ``` -**gitnexus_impact with tests** — check test coverage: +**impact with tests** — check test coverage: ``` -gitnexus_impact({target: "validatePayment", direction: "upstream", includeTests: true}) +impact({target: "validatePayment", direction: "upstream", includeTests: true}) → Tests that cover this symbol: - validatePayment.test.ts [direct] - checkout.integration.test.ts [via processCheckout] ``` -**gitnexus_context** — understand a changed symbol's role: +**context** — understand a changed symbol's role: ``` -gitnexus_context({name: "validatePayment"}) +context({name: "validatePayment"}) → Incoming calls: processCheckout, webhookHandler → Outgoing calls: verifyCard, fetchRates @@ -112,20 +112,20 @@ gitnexus_context({name: "validatePayment"}) 1. gh pr diff 42 > /tmp/pr42.diff → 4 files changed: payments.ts, checkout.ts, types.ts, utils.ts -2. gitnexus_detect_changes({scope: "compare", base_ref: "main"}) +2. detect_changes({scope: "compare", base_ref: "main"}) → Changed symbols: validatePayment, PaymentInput, formatAmount → Affected processes: CheckoutFlow, RefundFlow → Risk: MEDIUM -3. gitnexus_impact({target: "validatePayment", direction: "upstream"}) +3. impact({target: "validatePayment", direction: "upstream"}) → d=1: processCheckout, webhookHandler (WILL BREAK) → webhookHandler is NOT in the PR diff — potential breakage! -4. gitnexus_impact({target: "PaymentInput", direction: "upstream"}) +4. impact({target: "PaymentInput", direction: "upstream"}) → d=1: validatePayment (in PR), createPayment (NOT in PR) → createPayment uses the old PaymentInput shape — breaking change! -5. gitnexus_context({name: "formatAmount"}) +5. context({name: "formatAmount"}) → Called by 12 functions — but change is backwards-compatible (added optional param) 6. Review summary: diff --git a/gitnexus-claude-plugin/skills/gitnexus-refactoring/SKILL.md b/gitnexus-claude-plugin/skills/gitnexus-refactoring/SKILL.md index c749eb384..e13c04e14 100644 --- a/gitnexus-claude-plugin/skills/gitnexus-refactoring/SKILL.md +++ b/gitnexus-claude-plugin/skills/gitnexus-refactoring/SKILL.md @@ -16,9 +16,9 @@ description: "Use when the user wants to rename, extract, split, move, or restru ## Workflow ``` -1. gitnexus_impact({target: "X", direction: "upstream"}) → Map all dependents -2. gitnexus_query({query: "X"}) → Find execution flows involving X -3. gitnexus_context({name: "X"}) → See all incoming/outgoing refs +1. impact({target: "X", direction: "upstream"}) → Map all dependents +2. query({query: "X"}) → Find execution flows involving X +3. context({name: "X"}) → See all incoming/outgoing refs 4. Plan update order: interfaces → implementations → callers → tests ``` @@ -29,65 +29,65 @@ description: "Use when the user wants to rename, extract, split, move, or restru ### Rename Symbol ``` -- [ ] gitnexus_rename({symbol_name: "oldName", new_name: "newName", dry_run: true}) — preview all edits +- [ ] rename({symbol_name: "oldName", new_name: "newName", dry_run: true}) — preview all edits - [ ] Review graph edits (high confidence) and ast_search edits (review carefully) -- [ ] If satisfied: gitnexus_rename({..., dry_run: false}) — apply edits -- [ ] gitnexus_detect_changes() — verify only expected files changed +- [ ] If satisfied: rename({..., dry_run: false}) — apply edits +- [ ] detect_changes() — verify only expected files changed - [ ] Run tests for affected processes ``` ### Extract Module ``` -- [ ] gitnexus_context({name: target}) — see all incoming/outgoing refs -- [ ] gitnexus_impact({target, direction: "upstream"}) — find all external callers +- [ ] context({name: target}) — see all incoming/outgoing refs +- [ ] impact({target, direction: "upstream"}) — find all external callers - [ ] Define new module interface - [ ] Extract code, update imports -- [ ] gitnexus_detect_changes() — verify affected scope +- [ ] detect_changes() — verify affected scope - [ ] Run tests for affected processes ``` ### Split Function/Service ``` -- [ ] gitnexus_context({name: target}) — understand all callees +- [ ] context({name: target}) — understand all callees - [ ] Group callees by responsibility -- [ ] gitnexus_impact({target, direction: "upstream"}) — map callers to update +- [ ] impact({target, direction: "upstream"}) — map callers to update - [ ] Create new functions/services - [ ] Update callers -- [ ] gitnexus_detect_changes() — verify affected scope +- [ ] detect_changes() — verify affected scope - [ ] Run tests for affected processes ``` ## Tools -**gitnexus_rename** — automated multi-file rename: +**rename** — automated multi-file rename: ``` -gitnexus_rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: true}) +rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: true}) → 12 edits across 8 files → 10 graph edits (high confidence), 2 ast_search edits (review) → Changes: [{file_path, edits: [{line, old_text, new_text, confidence}]}] ``` -**gitnexus_impact** — map all dependents first: +**impact** — map all dependents first: ``` -gitnexus_impact({target: "validateUser", direction: "upstream"}) +impact({target: "validateUser", direction: "upstream"}) → d=1: loginHandler, apiMiddleware, testUtils → Affected Processes: LoginFlow, TokenRefresh ``` -**gitnexus_detect_changes** — verify your changes after refactoring: +**detect_changes** — verify your changes after refactoring: ``` -gitnexus_detect_changes({scope: "all"}) +detect_changes({scope: "all"}) → Changed: 8 files, 12 symbols → Affected processes: LoginFlow, TokenRefresh → Risk: MEDIUM ``` -**gitnexus_cypher** — custom reference queries: +**cypher** — custom reference queries: ```cypher MATCH (caller)-[:CodeRelation {type: 'CALLS'}]->(f:Function {name: "validateUser"}) @@ -98,24 +98,24 @@ RETURN caller.name, caller.filePath ORDER BY caller.filePath | Risk Factor | Mitigation | | ------------------- | ----------------------------------------- | -| Many callers (>5) | Use gitnexus_rename for automated updates | +| Many callers (>5) | Use rename for automated updates | | Cross-area refs | Use detect_changes after to verify scope | -| String/dynamic refs | gitnexus_query to find them | +| String/dynamic refs | query to find them | | External/public API | Version and deprecate properly | ## Example: Rename `validateUser` to `authenticateUser` ``` -1. gitnexus_rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: true}) +1. rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: true}) → 12 edits: 10 graph (safe), 2 ast_search (review) → Files: validator.ts, login.ts, middleware.ts, config.json... 2. Review ast_search edits (config.json: dynamic reference!) -3. gitnexus_rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: false}) +3. rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: false}) → Applied 12 edits across 8 files -4. gitnexus_detect_changes({scope: "all"}) +4. detect_changes({scope: "all"}) → Affected: LoginFlow, TokenRefresh → Risk: MEDIUM — run tests for these flows ``` diff --git a/gitnexus-cursor-integration/skills/gitnexus-debugging/SKILL.md b/gitnexus-cursor-integration/skills/gitnexus-debugging/SKILL.md index a7e250647..a88b76430 100644 --- a/gitnexus-cursor-integration/skills/gitnexus-debugging/SKILL.md +++ b/gitnexus-cursor-integration/skills/gitnexus-debugging/SKILL.md @@ -15,10 +15,10 @@ description: Trace bugs through call chains using knowledge graph ## Workflow ``` -1. gitnexus_query({query: ""}) → Find related execution flows -2. gitnexus_context({name: ""}) → See callers/callees/processes +1. query({query: ""}) → Find related execution flows +2. context({name: ""}) → See callers/callees/processes 3. READ gitnexus://repo/{name}/process/{name} → Trace execution flow -4. gitnexus_cypher({query: "MATCH path..."}) → Custom traces if needed +4. cypher({query: "MATCH path..."}) → Custom traces if needed ``` > If "Index is stale" → run `node .gitnexus/run.cjs analyze` in terminal. @@ -27,11 +27,11 @@ description: Trace bugs through call chains using knowledge graph ``` - [ ] Understand the symptom (error message, unexpected behavior) -- [ ] gitnexus_query for error text or related code +- [ ] query for error text or related code - [ ] Identify the suspect function from returned processes -- [ ] gitnexus_context to see callers and callees +- [ ] context to see callers and callees - [ ] Trace execution flow via process resource if applicable -- [ ] gitnexus_cypher for custom call chain traces if needed +- [ ] cypher for custom call chain traces if needed - [ ] Read source files to confirm root cause ``` @@ -39,7 +39,7 @@ description: Trace bugs through call chains using knowledge graph | Symptom | GitNexus Approach | |---------|-------------------| -| Error message | `gitnexus_query` for error text → `context` on throw sites | +| Error message | `query` for error text → `context` on throw sites | | Wrong return value | `context` on the function → trace callees for data flow | | Intermittent failure | `context` → look for external calls, async deps | | Performance issue | `context` → find symbols with many callers (hot paths) | @@ -47,22 +47,22 @@ description: Trace bugs through call chains using knowledge graph ## Tools -**gitnexus_query** — find code related to error: +**query** — find code related to error: ``` -gitnexus_query({query: "payment validation error"}) +query({query: "payment validation error"}) → Processes: CheckoutFlow, ErrorHandling → Symbols: validatePayment, handlePaymentError, PaymentException ``` -**gitnexus_context** — full context for a suspect: +**context** — full context for a suspect: ``` -gitnexus_context({name: "validatePayment"}) +context({name: "validatePayment"}) → Incoming calls: processCheckout, webhookHandler → Outgoing calls: verifyCard, fetchRates (external API!) → Processes: CheckoutFlow (step 3/7) ``` -**gitnexus_cypher** — custom call chain traces: +**cypher** — custom call chain traces: ```cypher MATCH path = (a)-[:CodeRelation {type: 'CALLS'}*1..2]->(b:Function {name: "validatePayment"}) RETURN [n IN nodes(path) | n.name] AS chain @@ -71,11 +71,11 @@ RETURN [n IN nodes(path) | n.name] AS chain ## Example: "Payment endpoint returns 500 intermittently" ``` -1. gitnexus_query({query: "payment error handling"}) +1. query({query: "payment error handling"}) → Processes: CheckoutFlow, ErrorHandling → Symbols: validatePayment, handlePaymentError -2. gitnexus_context({name: "validatePayment"}) +2. context({name: "validatePayment"}) → Outgoing calls: verifyCard, fetchRates (external API!) 3. READ gitnexus://repo/my-app/process/CheckoutFlow diff --git a/gitnexus-cursor-integration/skills/gitnexus-exploring/SKILL.md b/gitnexus-cursor-integration/skills/gitnexus-exploring/SKILL.md index 4549505e5..73df1353a 100644 --- a/gitnexus-cursor-integration/skills/gitnexus-exploring/SKILL.md +++ b/gitnexus-cursor-integration/skills/gitnexus-exploring/SKILL.md @@ -17,8 +17,8 @@ description: Navigate unfamiliar code using GitNexus knowledge graph ``` 1. READ gitnexus://repos → Discover indexed repos 2. READ gitnexus://repo/{name}/context → Codebase overview, check staleness -3. gitnexus_query({query: ""}) → Find related execution flows -4. gitnexus_context({name: ""}) → Deep dive on specific symbol +3. query({query: ""}) → Find related execution flows +4. context({name: ""}) → Deep dive on specific symbol 5. READ gitnexus://repo/{name}/process/{name} → Trace full execution flow ``` @@ -28,9 +28,9 @@ description: Navigate unfamiliar code using GitNexus knowledge graph ``` - [ ] READ gitnexus://repo/{name}/context -- [ ] gitnexus_query for the concept you want to understand +- [ ] query for the concept you want to understand - [ ] Review returned processes (execution flows) -- [ ] gitnexus_context on key symbols for callers/callees +- [ ] context on key symbols for callers/callees - [ ] READ process resource for full execution traces - [ ] Read source files for implementation details ``` @@ -46,16 +46,16 @@ description: Navigate unfamiliar code using GitNexus knowledge graph ## Tools -**gitnexus_query** — find execution flows related to a concept: +**query** — find execution flows related to a concept: ``` -gitnexus_query({query: "payment processing"}) +query({query: "payment processing"}) → Processes: CheckoutFlow, RefundFlow, WebhookHandler → Symbols grouped by flow with file locations ``` -**gitnexus_context** — 360-degree view of a symbol: +**context** — 360-degree view of a symbol: ``` -gitnexus_context({name: "validateUser"}) +context({name: "validateUser"}) → Incoming calls: loginHandler, apiMiddleware → Outgoing calls: checkToken, getUserById → Processes: LoginFlow (step 2/5), TokenRefresh (step 1/3) @@ -65,10 +65,10 @@ gitnexus_context({name: "validateUser"}) ``` 1. READ gitnexus://repo/my-app/context → 918 symbols, 45 processes -2. gitnexus_query({query: "payment processing"}) +2. query({query: "payment processing"}) → CheckoutFlow: processPayment → validateCard → chargeStripe → RefundFlow: initiateRefund → calculateRefund → processRefund -3. gitnexus_context({name: "processPayment"}) +3. context({name: "processPayment"}) → Incoming: checkoutHandler, webhookHandler → Outgoing: validateCard, chargeStripe, saveTransaction 4. Read src/payments/processor.ts for implementation details diff --git a/gitnexus-cursor-integration/skills/gitnexus-impact-analysis/SKILL.md b/gitnexus-cursor-integration/skills/gitnexus-impact-analysis/SKILL.md index 0733b09ac..139b897e4 100644 --- a/gitnexus-cursor-integration/skills/gitnexus-impact-analysis/SKILL.md +++ b/gitnexus-cursor-integration/skills/gitnexus-impact-analysis/SKILL.md @@ -16,9 +16,9 @@ description: Analyze blast radius before making code changes ## Workflow ``` -1. gitnexus_impact({target: "X", direction: "upstream"}) → What depends on this +1. impact({target: "X", direction: "upstream"}) → What depends on this 2. READ gitnexus://repo/{name}/processes → Check affected execution flows -3. gitnexus_detect_changes() → Map current git changes to affected flows +3. detect_changes() → Map current git changes to affected flows 4. Assess risk and report to user ``` @@ -27,11 +27,11 @@ description: Analyze blast radius before making code changes ## Checklist ``` -- [ ] gitnexus_impact({target, direction: "upstream"}) to find dependents +- [ ] impact({target, direction: "upstream"}) to find dependents - [ ] Review d=1 items first (these WILL BREAK) - [ ] Check high-confidence (>0.8) dependencies - [ ] READ processes to check affected execution flows -- [ ] gitnexus_detect_changes() for pre-commit check +- [ ] detect_changes() for pre-commit check - [ ] Assess risk level and report to user ``` @@ -54,9 +54,9 @@ description: Analyze blast radius before making code changes ## Tools -**gitnexus_impact** — the primary tool for symbol blast radius: +**impact** — the primary tool for symbol blast radius: ``` -gitnexus_impact({ +impact({ target: "validateUser", direction: "upstream", minConfidence: 0.8, @@ -71,9 +71,9 @@ gitnexus_impact({ - authRouter (src/routes/auth.ts:22) [CALLS, 95%] ``` -**gitnexus_detect_changes** — git-diff based impact analysis: +**detect_changes** — git-diff based impact analysis: ``` -gitnexus_detect_changes({scope: "staged"}) +detect_changes({scope: "staged"}) → Changed: 5 symbols in 3 files → Affected: LoginFlow, TokenRefresh, APIMiddlewarePipeline @@ -83,7 +83,7 @@ gitnexus_detect_changes({scope: "staged"}) ## Example: "What breaks if I change validateUser?" ``` -1. gitnexus_impact({target: "validateUser", direction: "upstream"}) +1. impact({target: "validateUser", direction: "upstream"}) → d=1: loginHandler, apiMiddleware (WILL BREAK) → d=2: authRouter, sessionManager (LIKELY AFFECTED) diff --git a/gitnexus-cursor-integration/skills/gitnexus-pr-review/SKILL.md b/gitnexus-cursor-integration/skills/gitnexus-pr-review/SKILL.md index 319c063f9..9f1d362e5 100644 --- a/gitnexus-cursor-integration/skills/gitnexus-pr-review/SKILL.md +++ b/gitnexus-cursor-integration/skills/gitnexus-pr-review/SKILL.md @@ -18,10 +18,10 @@ description: "Use when the user wants to review a pull request, understand what ``` 1. gh pr diff → Get the raw diff -2. gitnexus_detect_changes({scope: "compare", base_ref: "main"}) → Map diff to affected flows +2. detect_changes({scope: "compare", base_ref: "main"}) → Map diff to affected flows 3. For each changed symbol: - gitnexus_impact({target: "", direction: "upstream"}) → Blast radius per change -4. gitnexus_context({name: ""}) → Understand callers/callees + impact({target: "", direction: "upstream"}) → Blast radius per change +4. context({name: ""}) → Understand callers/callees 5. READ gitnexus://repo/{name}/processes → Check affected execution flows 6. Summarize findings with risk assessment ``` @@ -32,10 +32,10 @@ description: "Use when the user wants to review a pull request, understand what ``` - [ ] Fetch PR diff (gh pr diff or git diff base...head) -- [ ] gitnexus_detect_changes to map changes to affected execution flows -- [ ] gitnexus_impact on each non-trivial changed symbol +- [ ] detect_changes to map changes to affected execution flows +- [ ] impact on each non-trivial changed symbol - [ ] Review d=1 items (WILL BREAK) — are callers updated? -- [ ] gitnexus_context on key changed symbols to understand full picture +- [ ] context on key changed symbols to understand full picture - [ ] Check if affected processes have test coverage - [ ] Assess overall risk level - [ ] Write review summary with findings @@ -63,20 +63,20 @@ description: "Use when the user wants to review a pull request, understand what ## Tools -**gitnexus_detect_changes** — map PR diff to affected execution flows: +**detect_changes** — map PR diff to affected execution flows: ``` -gitnexus_detect_changes({scope: "compare", base_ref: "main"}) +detect_changes({scope: "compare", base_ref: "main"}) → Changed: 8 symbols in 4 files → Affected processes: CheckoutFlow, RefundFlow, WebhookHandler → Risk: MEDIUM ``` -**gitnexus_impact** — blast radius per changed symbol: +**impact** — blast radius per changed symbol: ``` -gitnexus_impact({target: "validatePayment", direction: "upstream"}) +impact({target: "validatePayment", direction: "upstream"}) → d=1 (WILL BREAK): - processCheckout (src/checkout.ts:42) [CALLS, 100%] @@ -86,20 +86,20 @@ gitnexus_impact({target: "validatePayment", direction: "upstream"}) - checkoutRouter (src/routes/checkout.ts:22) [CALLS, 95%] ``` -**gitnexus_impact with tests** — check test coverage: +**impact with tests** — check test coverage: ``` -gitnexus_impact({target: "validatePayment", direction: "upstream", includeTests: true}) +impact({target: "validatePayment", direction: "upstream", includeTests: true}) → Tests that cover this symbol: - validatePayment.test.ts [direct] - checkout.integration.test.ts [via processCheckout] ``` -**gitnexus_context** — understand a changed symbol's role: +**context** — understand a changed symbol's role: ``` -gitnexus_context({name: "validatePayment"}) +context({name: "validatePayment"}) → Incoming calls: processCheckout, webhookHandler → Outgoing calls: verifyCard, fetchRates @@ -112,20 +112,20 @@ gitnexus_context({name: "validatePayment"}) 1. gh pr diff 42 > /tmp/pr42.diff → 4 files changed: payments.ts, checkout.ts, types.ts, utils.ts -2. gitnexus_detect_changes({scope: "compare", base_ref: "main"}) +2. detect_changes({scope: "compare", base_ref: "main"}) → Changed symbols: validatePayment, PaymentInput, formatAmount → Affected processes: CheckoutFlow, RefundFlow → Risk: MEDIUM -3. gitnexus_impact({target: "validatePayment", direction: "upstream"}) +3. impact({target: "validatePayment", direction: "upstream"}) → d=1: processCheckout, webhookHandler (WILL BREAK) → webhookHandler is NOT in the PR diff — potential breakage! -4. gitnexus_impact({target: "PaymentInput", direction: "upstream"}) +4. impact({target: "PaymentInput", direction: "upstream"}) → d=1: validatePayment (in PR), createPayment (NOT in PR) → createPayment uses the old PaymentInput shape — breaking change! -5. gitnexus_context({name: "formatAmount"}) +5. context({name: "formatAmount"}) → Called by 12 functions — but change is backwards-compatible (added optional param) 6. Review summary: diff --git a/gitnexus-cursor-integration/skills/gitnexus-refactoring/SKILL.md b/gitnexus-cursor-integration/skills/gitnexus-refactoring/SKILL.md index a49b58be4..76c9d3351 100644 --- a/gitnexus-cursor-integration/skills/gitnexus-refactoring/SKILL.md +++ b/gitnexus-cursor-integration/skills/gitnexus-refactoring/SKILL.md @@ -15,9 +15,9 @@ description: Plan safe refactors using blast radius and dependency mapping ## Workflow ``` -1. gitnexus_impact({target: "X", direction: "upstream"}) → Map all dependents -2. gitnexus_query({query: "X"}) → Find execution flows involving X -3. gitnexus_context({name: "X"}) → See all incoming/outgoing refs +1. impact({target: "X", direction: "upstream"}) → Map all dependents +2. query({query: "X"}) → Find execution flows involving X +3. context({name: "X"}) → See all incoming/outgoing refs 4. Plan update order: interfaces → implementations → callers → tests ``` @@ -27,60 +27,60 @@ description: Plan safe refactors using blast radius and dependency mapping ### Rename Symbol ``` -- [ ] gitnexus_rename({symbol_name: "oldName", new_name: "newName", dry_run: true}) — preview all edits +- [ ] rename({symbol_name: "oldName", new_name: "newName", dry_run: true}) — preview all edits - [ ] Review graph edits (high confidence) and ast_search edits (review carefully) -- [ ] If satisfied: gitnexus_rename({..., dry_run: false}) — apply edits -- [ ] gitnexus_detect_changes() — verify only expected files changed +- [ ] If satisfied: rename({..., dry_run: false}) — apply edits +- [ ] detect_changes() — verify only expected files changed - [ ] Run tests for affected processes ``` ### Extract Module ``` -- [ ] gitnexus_context({name: target}) — see all incoming/outgoing refs -- [ ] gitnexus_impact({target, direction: "upstream"}) — find all external callers +- [ ] context({name: target}) — see all incoming/outgoing refs +- [ ] impact({target, direction: "upstream"}) — find all external callers - [ ] Define new module interface - [ ] Extract code, update imports -- [ ] gitnexus_detect_changes() — verify affected scope +- [ ] detect_changes() — verify affected scope - [ ] Run tests for affected processes ``` ### Split Function/Service ``` -- [ ] gitnexus_context({name: target}) — understand all callees +- [ ] context({name: target}) — understand all callees - [ ] Group callees by responsibility -- [ ] gitnexus_impact({target, direction: "upstream"}) — map callers to update +- [ ] impact({target, direction: "upstream"}) — map callers to update - [ ] Create new functions/services - [ ] Update callers -- [ ] gitnexus_detect_changes() — verify affected scope +- [ ] detect_changes() — verify affected scope - [ ] Run tests for affected processes ``` ## Tools -**gitnexus_rename** — automated multi-file rename: +**rename** — automated multi-file rename: ``` -gitnexus_rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: true}) +rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: true}) → 12 edits across 8 files → 10 graph edits (high confidence), 2 ast_search edits (review) → Changes: [{file_path, edits: [{line, old_text, new_text, confidence}]}] ``` -**gitnexus_impact** — map all dependents first: +**impact** — map all dependents first: ``` -gitnexus_impact({target: "validateUser", direction: "upstream"}) +impact({target: "validateUser", direction: "upstream"}) → d=1: loginHandler, apiMiddleware, testUtils → Affected Processes: LoginFlow, TokenRefresh ``` -**gitnexus_detect_changes** — verify your changes after refactoring: +**detect_changes** — verify your changes after refactoring: ``` -gitnexus_detect_changes({scope: "all"}) +detect_changes({scope: "all"}) → Changed: 8 files, 12 symbols → Affected processes: LoginFlow, TokenRefresh → Risk: MEDIUM ``` -**gitnexus_cypher** — custom reference queries: +**cypher** — custom reference queries: ```cypher MATCH (caller)-[:CodeRelation {type: 'CALLS'}]->(f:Function {name: "validateUser"}) RETURN caller.name, caller.filePath ORDER BY caller.filePath @@ -90,24 +90,24 @@ RETURN caller.name, caller.filePath ORDER BY caller.filePath | Risk Factor | Mitigation | |-------------|------------| -| Many callers (>5) | Use gitnexus_rename for automated updates | +| Many callers (>5) | Use rename for automated updates | | Cross-area refs | Use detect_changes after to verify scope | -| String/dynamic refs | gitnexus_query to find them | +| String/dynamic refs | query to find them | | External/public API | Version and deprecate properly | ## Example: Rename `validateUser` to `authenticateUser` ``` -1. gitnexus_rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: true}) +1. rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: true}) → 12 edits: 10 graph (safe), 2 ast_search (review) → Files: validator.ts, login.ts, middleware.ts, config.json... 2. Review ast_search edits (config.json: dynamic reference!) -3. gitnexus_rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: false}) +3. rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: false}) → Applied 12 edits across 8 files -4. gitnexus_detect_changes({scope: "all"}) +4. detect_changes({scope: "all"}) → Affected: LoginFlow, TokenRefresh → Risk: MEDIUM — run tests for these flows ``` diff --git a/gitnexus-shared/src/graph/types.ts b/gitnexus-shared/src/graph/types.ts index ede8bc906..86abc9eba 100644 --- a/gitnexus-shared/src/graph/types.ts +++ b/gitnexus-shared/src/graph/types.ts @@ -44,7 +44,10 @@ export type NodeLabel = | 'Template' | 'Section' | 'Route' - | 'Tool'; + | 'Tool' + // Taint/PDG substrate (issue #2080). Intra-procedural control-flow node. + // Emitted by no phase yet — M1 (#2081) populates these behind an opt-in. + | 'BasicBlock'; export type NodeProperties = { name: string; @@ -89,6 +92,8 @@ export type NodeProperties = { responseKeys?: string[]; errorKeys?: string[]; middleware?: string[]; + // BasicBlock (taint/PDG substrate, issue #2080) — reuses filePath/startLine/endLine. + text?: string; // Extensible [key: string]: unknown; }; @@ -131,7 +136,28 @@ export type RelationshipType = * `reason` encodes the event name: `vue-emit: `. * Complements `BINDS_EVENT_HANDLER`; a Cypher query joining on the * component File node reveals all (emitter, handler) pairs. */ - | 'EMITS_EVENT'; + | 'EMITS_EVENT' + // ── Taint/PDG substrate (issue #2080) ──────────────────────────────────── + // Reserved edge types for the taint-first PDG substrate. No phase emits any + // of these yet; they are populated behind an opt-in by later milestones + // (CFG → M1 #2081, REACHING_DEF → M2 #2082, TAINTED/SANITIZES/TAINT_PATH → + // M3/M4 #2083/#2084). Adding them here keeps the shared schema stable so + // downstream work does not re-ripple the exhaustiveness sites. + /** Control-flow edge between two BasicBlock nodes (intra-procedural CFG). */ + | 'CFG' + /** Data-dependence edge: a definition of `variable` reaches a use of it. + * The `variable` name is stored in the relation's existing `reason` column + * (M0/S1 verdict: LadybugDB has no secondary index on relationship + * properties, so a dedicated indexed column would not speed the + * variable-filtered path query). */ + | 'REACHING_DEF' + /** A tainted value flows from source toward sink. */ + | 'TAINTED' + /** A sanitizer clears taint along a flow. */ + | 'SANITIZES' + /** Materialized source→sink taint path. Working name — final name/representation + * is confirmed when M3/M4 emits it; no persisted edge exists before then. */ + | 'TAINT_PATH'; export interface GraphNode { id: string; diff --git a/gitnexus-shared/src/index.ts b/gitnexus-shared/src/index.ts index d732d2633..1b1d9c9a9 100644 --- a/gitnexus-shared/src/index.ts +++ b/gitnexus-shared/src/index.ts @@ -183,13 +183,3 @@ export { stripGitSuffix, } from './integrations/understand-quickly.js'; export type { UqDispatchPayload } from './integrations/understand-quickly.js'; - -// Shadow-mode diff + aggregation (RFC §6.3; Ring 2 SHARED #918) -export { diffResolutions } from './scope-resolution/shadow/diff.js'; -export type { - ShadowAgreement, - ShadowCallsite, - ShadowDiff, -} from './scope-resolution/shadow/diff.js'; -export { aggregateDiffs } from './scope-resolution/shadow/aggregate.js'; -export type { LanguageParityRow, ShadowParityReport } from './scope-resolution/shadow/aggregate.js'; diff --git a/gitnexus-shared/src/lbug/schema-constants.ts b/gitnexus-shared/src/lbug/schema-constants.ts index 656ffe552..d022ba5c4 100644 --- a/gitnexus-shared/src/lbug/schema-constants.ts +++ b/gitnexus-shared/src/lbug/schema-constants.ts @@ -40,6 +40,8 @@ export const NODE_TABLES = [ 'Module', 'Route', 'Tool', + // Taint/PDG substrate (issue #2080) — inert until M1 (#2081) emits blocks. + 'BasicBlock', ] as const; export type NodeTableName = (typeof NODE_TABLES)[number]; @@ -67,6 +69,14 @@ export const REL_TYPES = [ 'ENTRY_POINT_OF', 'WRAPS', 'QUERIES', + // Taint/PDG substrate (issue #2080) — reserved edge types, emitted by no + // phase yet (CFG → M1, REACHING_DEF → M2, TAINTED/SANITIZES/TAINT_PATH → + // M3/M4). REACHING_DEF's variable name rides the relation's `reason` column. + 'CFG', + 'REACHING_DEF', + 'TAINTED', + 'SANITIZES', + 'TAINT_PATH', ] as const; export type RelType = (typeof REL_TYPES)[number]; diff --git a/gitnexus-shared/src/scope-resolution/module-scope-index.ts b/gitnexus-shared/src/scope-resolution/module-scope-index.ts index a57c02d27..bb1b4bab2 100644 --- a/gitnexus-shared/src/scope-resolution/module-scope-index.ts +++ b/gitnexus-shared/src/scope-resolution/module-scope-index.ts @@ -8,8 +8,7 @@ * * Part of RFC #909 Ring 2 SHARED — #913. * - * Consumed by: #915 (SCC finalize link pass), #923 (shadow harness when - * resolving callsite file → enclosing module). + * Consumed by: #915 (SCC finalize link pass). */ import type { ScopeId } from './types.js'; diff --git a/gitnexus-shared/src/scope-resolution/parsed-file.ts b/gitnexus-shared/src/scope-resolution/parsed-file.ts index 50eb5a795..98a327ad4 100644 --- a/gitnexus-shared/src/scope-resolution/parsed-file.ts +++ b/gitnexus-shared/src/scope-resolution/parsed-file.ts @@ -74,4 +74,28 @@ export interface ParsedFile { */ readonly localDefs: readonly SymbolDefinition[]; readonly referenceSites: readonly ReferenceSite[]; + /** + * Opaque, language-private serialization of capture-time side-channel + * state that a provider's `emitScopeCaptures` populates into module-level + * maps as a SIDE EFFECT (not onto the scopes/defs of this `ParsedFile`). + * + * Such state is computed inside the parse worker (where `emitScopeCaptures` + * runs) and would otherwise be lost across the worker→main MessageChannel + * and the disk store, because scope-resolution reuses the serialized + * `ParsedFile` and SKIPS re-extraction on the main thread (#1983 — the + * whole point is to avoid a main-thread tree-sitter re-parse). Carrying the + * data here lets the main thread repopulate those maps WITHOUT re-parsing. + * + * Shared / ingestion code treats this as opaque (`unknown`) per AGENTS.md + * (no language names in shared code). The producing language fills it via + * the `LanguageProvider.collectCaptureSideChannel` hook (worker side) and + * consumes it via the `ScopeResolver.applyCaptureSideChannel` hook + * (main-thread resolution side). It MUST be plain JSON-serializable data + * (objects / arrays / primitives) so it round-trips through the disk-backed + * `parsedfile-store` (JSON.stringify + interning reviver). + * + * Optional: providers whose `emitScopeCaptures` is pure (no module-level + * side effects — the contract default) leave this undefined. + */ + readonly captureSideChannel?: unknown; } diff --git a/gitnexus-shared/src/scope-resolution/registries/evidence.ts b/gitnexus-shared/src/scope-resolution/registries/evidence.ts index cabeb6a95..855d1f15a 100644 --- a/gitnexus-shared/src/scope-resolution/registries/evidence.ts +++ b/gitnexus-shared/src/scope-resolution/registries/evidence.ts @@ -57,8 +57,7 @@ export interface RawSignals { * * Emission order mirrors the `EvidenceWeights` layout: where-found → * type-binding → corroborators → arity → degraded. Stable order makes - * the per-signal contributions easy to reason about in tests and in the - * shadow-mode parity dashboard. + * the per-signal contributions easy to reason about in tests. */ export function composeEvidence(signals: RawSignals): readonly ResolutionEvidence[] { const out: ResolutionEvidence[] = []; @@ -141,7 +140,7 @@ export function composeEvidence(signals: RawSignals): readonly ResolutionEvidenc /** * Sum evidence weights and clamp to `[0, 1]`. Separate from `composeEvidence` - * so tests and the parity dashboard can inspect the raw evidence list. + * so tests can inspect the raw evidence list. */ export function confidenceFromEvidence(evidence: readonly ResolutionEvidence[]): number { let sum = 0; diff --git a/gitnexus-shared/src/scope-resolution/shadow/aggregate.ts b/gitnexus-shared/src/scope-resolution/shadow/aggregate.ts deleted file mode 100644 index 27c24ff92..000000000 --- a/gitnexus-shared/src/scope-resolution/shadow/aggregate.ts +++ /dev/null @@ -1,188 +0,0 @@ -/** - * Shadow-mode aggregation — per-language parity %, per-evidence-kind - * breakdown of divergences. Consumed by the parity dashboard (RING2-PKG-5). - * - * Pure functions; no I/O. The harness persists per-run JSON; the dashboard - * reads `.gitnexus/shadow-parity/latest.json` and renders. - * - * Related types — `ShadowAgreement`, `ShadowCallsite`, `ShadowDiff` — are - * defined alongside `diffResolutions` in `./diff.ts` and re-exported - * through the top-level `gitnexus-shared` barrel. Consumers import all - * three from `gitnexus-shared`, not from this module. - * - * Part of RFC #909 Ring 2 SHARED — #918. - */ - -import type { SupportedLanguages } from '../../languages.js'; -import type { ResolutionEvidence } from '../types.js'; -import type { ShadowAgreement, ShadowDiff } from './diff.js'; - -// ─── Aggregated report shape ──────────────────────────────────────────────── - -export interface LanguageParityRow { - readonly language: SupportedLanguages; - readonly totalCalls: number; - readonly bothAgree: number; - readonly onlyLegacy: number; - readonly onlyNew: number; - readonly bothDisagree: number; - readonly bothEmpty: number; - /** - * Fraction in [0, 1]. Numerator = `bothAgree`; denominator = "calls where - * at least one side resolved" = `totalCalls - bothEmpty`. - * - * When the denominator is 0 (all calls for this language were - * `both-empty`), returns 0. Callers rendering the dashboard should treat - * a 0 parity alongside `totalCalls === bothEmpty` as "no signal" rather - * than "total disagreement". - */ - readonly parity: number; - /** - * Divergence signals broken down by `ResolutionEvidence.kind`. Sourced - * from `ShadowDiff.evidenceDelta` on non-agreeing rows only — `both-agree` - * and `both-empty` do not contribute. - */ - readonly evidenceBreakdown: ReadonlyMap; -} - -export interface ShadowParityReport { - readonly generatedAt: string; // ISO 8601 - readonly perLanguage: readonly LanguageParityRow[]; - readonly overall: Omit; -} - -// ─── Public API ───────────────────────────────────────────────────────────── - -/** - * Aggregate a stream of `ShadowDiff` records into a `ShadowParityReport`, - * bucketed by language. Pure function. - * - * - `perLanguage` rows are sorted alphabetically by `SupportedLanguages` - * value for stable JSON output (the dashboard reads - * `.gitnexus/shadow-parity/latest.json` and diffing snapshots is useful). - * - `overall` is the column-wise sum across languages. - * - `generatedAt` is injected via the `now` parameter so tests stay - * deterministic; production callers let it default to `new Date()`. - */ -export function aggregateDiffs( - diffs: readonly { readonly language: SupportedLanguages; readonly diff: ShadowDiff }[], - now: Date = new Date(), -): ShadowParityReport { - const perLanguageMap = new Map(); - - for (const { language, diff } of diffs) { - let counts = perLanguageMap.get(language); - if (!counts) { - counts = makeEmptyCounts(); - perLanguageMap.set(language, counts); - } - tallyDiff(counts, diff); - } - - const perLanguage: LanguageParityRow[] = Array.from(perLanguageMap.entries()) - .map(([language, counts]) => buildRow(language, counts)) - .sort((a, b) => a.language.localeCompare(b.language)); - - const overall = buildOverallRow(perLanguage); - - return { - generatedAt: now.toISOString(), - perLanguage, - overall, - }; -} - -// ─── Internal helpers ─────────────────────────────────────────────────────── - -interface MutableCounts { - totalCalls: number; - bothAgree: number; - onlyLegacy: number; - onlyNew: number; - bothDisagree: number; - bothEmpty: number; - evidenceBreakdown: Map; -} - -function makeEmptyCounts(): MutableCounts { - return { - totalCalls: 0, - bothAgree: 0, - onlyLegacy: 0, - onlyNew: 0, - bothDisagree: 0, - bothEmpty: 0, - evidenceBreakdown: new Map(), - }; -} - -function tallyDiff(counts: MutableCounts, diff: ShadowDiff): void { - counts.totalCalls += 1; - incrementAgreement(counts, diff.agreement); - if (diff.agreement === 'both-agree' || diff.agreement === 'both-empty') return; - for (const ev of diff.evidenceDelta) { - counts.evidenceBreakdown.set(ev.kind, (counts.evidenceBreakdown.get(ev.kind) ?? 0) + 1); - } -} - -function incrementAgreement(counts: MutableCounts, agreement: ShadowAgreement): void { - switch (agreement) { - case 'both-agree': - counts.bothAgree += 1; - return; - case 'only-legacy': - counts.onlyLegacy += 1; - return; - case 'only-new': - counts.onlyNew += 1; - return; - case 'both-disagree': - counts.bothDisagree += 1; - return; - case 'both-empty': - counts.bothEmpty += 1; - return; - } -} - -function buildRow(language: SupportedLanguages, counts: MutableCounts): LanguageParityRow { - const resolved = counts.totalCalls - counts.bothEmpty; - const parity = resolved > 0 ? counts.bothAgree / resolved : 0; - return { - language, - totalCalls: counts.totalCalls, - bothAgree: counts.bothAgree, - onlyLegacy: counts.onlyLegacy, - onlyNew: counts.onlyNew, - bothDisagree: counts.bothDisagree, - bothEmpty: counts.bothEmpty, - parity, - // Freeze via `new Map` on a sorted-kind copy so downstream consumers - // can't mutate the aggregator's internal state. - evidenceBreakdown: new Map( - Array.from(counts.evidenceBreakdown.entries()).sort(([a], [b]) => a.localeCompare(b)), - ), - }; -} - -function buildOverallRow( - perLanguage: readonly LanguageParityRow[], -): Omit { - let totalCalls = 0; - let bothAgree = 0; - let onlyLegacy = 0; - let onlyNew = 0; - let bothDisagree = 0; - let bothEmpty = 0; - for (const row of perLanguage) { - totalCalls += row.totalCalls; - bothAgree += row.bothAgree; - onlyLegacy += row.onlyLegacy; - onlyNew += row.onlyNew; - bothDisagree += row.bothDisagree; - bothEmpty += row.bothEmpty; - } - const resolved = totalCalls - bothEmpty; - const parity = resolved > 0 ? bothAgree / resolved : 0; - return { totalCalls, bothAgree, onlyLegacy, onlyNew, bothDisagree, bothEmpty, parity }; -} diff --git a/gitnexus-shared/src/scope-resolution/shadow/diff.ts b/gitnexus-shared/src/scope-resolution/shadow/diff.ts deleted file mode 100644 index a1c8755c6..000000000 --- a/gitnexus-shared/src/scope-resolution/shadow/diff.ts +++ /dev/null @@ -1,126 +0,0 @@ -/** - * Shadow-mode diff logic — RFC §6.3. - * - * Pure comparison logic for shadow mode. Takes two `Resolution[]` (legacy - * DAG result + new scope-based registry result) and produces a structured - * diff record for the parity dashboard. - * - * Consumed by the Ring 2 PKG shadow harness (#923), which dual-runs each - * call through legacy + new paths, diffs results, and persists per-run JSON - * for the parity dashboard. - * - * Part of RFC #909 Ring 2 SHARED — #918. - */ - -import type { Resolution, ResolutionEvidence } from '../types.js'; - -// ─── Diff record shape ────────────────────────────────────────────────────── - -export type ShadowAgreement = - | 'both-agree' // top match identical (same DefId) - | 'only-legacy' // legacy resolved; new did not - | 'only-new' // new resolved; legacy did not - | 'both-disagree' // both resolved, but to different targets - | 'both-empty'; // both returned empty - -export interface ShadowDiff { - readonly callsite: ShadowCallsite; - readonly legacy: Resolution | null; - readonly newResult: Resolution | null; - readonly agreement: ShadowAgreement; - /** - * Symmetric difference of the two top resolutions' `evidence` arrays, - * keyed on `ResolutionEvidence.kind`. - * - * - For `'both-agree'` and `'both-empty'` agreements, always empty. - * - For `'both-disagree'`, contains evidence kinds present on exactly one - * side (not in both). - * - For `'only-legacy'`, contains all of legacy's top evidence. - * - For `'only-new'`, contains all of new's top evidence. - */ - readonly evidenceDelta: readonly ResolutionEvidence[]; -} - -export interface ShadowCallsite { - readonly filePath: string; - readonly line: number; - readonly col: number; - readonly calledName: string; -} - -// ─── Public API ───────────────────────────────────────────────────────────── - -/** - * Compare two `Resolution[]` arrays (top matches at `[0]`) and produce a - * `ShadowDiff`. Pure function. - * - * Agreement rules: - * - both arrays empty → `'both-empty'`, `evidenceDelta: []` - * - legacy empty, new non-empty → `'only-new'`, `evidenceDelta` = new's top evidence - * - legacy non-empty, new empty → `'only-legacy'`, `evidenceDelta` = legacy's top evidence - * - both non-empty, same top `def.nodeId` → `'both-agree'`, `evidenceDelta: []` - * - both non-empty, different top `def.nodeId` → `'both-disagree'`, - * `evidenceDelta` = symmetric difference by `ResolutionEvidence.kind` - * (first occurrence of a kind-only-on-legacy then kind-only-on-new; order - * preserved from input arrays) - * - * Evidence-delta rationale: callers aggregating divergences want to know - * which signal kinds explain a disagreement. Keying on `kind` (not full - * equality over `weight`/`note`) avoids spurious deltas when the same - * signal fires with slightly different calibration weights on each side. - */ -export function diffResolutions( - callsite: ShadowCallsite, - legacy: readonly Resolution[], - newResult: readonly Resolution[], -): ShadowDiff { - const legacyTop: Resolution | null = legacy.length > 0 ? legacy[0] : null; - const newTop: Resolution | null = newResult.length > 0 ? newResult[0] : null; - - const agreement: ShadowAgreement = (() => { - if (legacyTop === null && newTop === null) return 'both-empty'; - if (legacyTop === null) return 'only-new'; - if (newTop === null) return 'only-legacy'; - return legacyTop.def.nodeId === newTop.def.nodeId ? 'both-agree' : 'both-disagree'; - })(); - - const evidenceDelta = computeEvidenceDelta(legacyTop, newTop, agreement); - - return { - callsite, - legacy: legacyTop, - newResult: newTop, - agreement, - evidenceDelta, - }; -} - -// ─── Internal helpers ─────────────────────────────────────────────────────── - -/** - * Symmetric difference of two evidence arrays, keyed on - * `ResolutionEvidence.kind`. Preserves input order: legacy-only signals - * first (in legacy's original order), then new-only signals (in new's order). - * - * For `'both-agree'` / `'both-empty'` the delta is empty by contract. For - * `'only-legacy'` / `'only-new'` one side's evidence is the delta (nothing to - * subtract against). - */ -function computeEvidenceDelta( - legacy: Resolution | null, - newResult: Resolution | null, - agreement: ShadowAgreement, -): readonly ResolutionEvidence[] { - if (agreement === 'both-agree' || agreement === 'both-empty') return []; - if (agreement === 'only-legacy') return legacy!.evidence; - if (agreement === 'only-new') return newResult!.evidence; - - // both-disagree: symmetric difference keyed on `kind` - const legacyKinds = new Set(legacy!.evidence.map((e) => e.kind)); - const newKinds = new Set(newResult!.evidence.map((e) => e.kind)); - - const onlyInLegacy = legacy!.evidence.filter((e) => !newKinds.has(e.kind)); - const onlyInNew = newResult!.evidence.filter((e) => !legacyKinds.has(e.kind)); - - return [...onlyInLegacy, ...onlyInNew]; -} diff --git a/gitnexus-web/src/lib/constants.ts b/gitnexus-web/src/lib/constants.ts index fe0505483..2f717cab3 100644 --- a/gitnexus-web/src/lib/constants.ts +++ b/gitnexus-web/src/lib/constants.ts @@ -38,6 +38,7 @@ export const NODE_COLORS: Record = { Template: '#a78bfa', // Violet light - like Type Route: '#f43f5e', // Rose - like Process Tool: '#a855f7', // Purple - like Project + BasicBlock: '#475569', // Slate darker - control-flow node (muted, taint/PDG substrate) }; // Node sizes by type - clear visual hierarchy with dramatic size differences @@ -79,6 +80,7 @@ export const NODE_SIZES: Record = { Template: 3, // Like Type Route: 5, // Like Enum Tool: 5, // Like Enum + BasicBlock: 2, // Tiny - control-flow node (taint/PDG substrate) }; // Community color palette for cluster-based coloring diff --git a/gitnexus/CHANGELOG.md b/gitnexus/CHANGELOG.md index 39db0eb90..5fe3e70f2 100644 --- a/gitnexus/CHANGELOG.md +++ b/gitnexus/CHANGELOG.md @@ -4,6 +4,75 @@ All notable changes to GitNexus will be documented in this file. ## [Unreleased] +### Added + +- **Taint/PDG substrate (M0)** — foundational schema + seams for reliable taint analysis on a PDG-expandable substrate (#2080, Epic #2087). Adds the `BasicBlock` node label and `CFG` / `REACHING_DEF` / `TAINTED` / `SANITIZES` / `TAINT_PATH` relationship types to the graph schema (round-trip through the bulk-COPY path), a phase-registry seam (`registerPhase` / `enabledWhen`) generalising the graph-phase opt-in guard, and a per-language source/sink/sanitizer config registry seam. All additive and inert — no phase emits the new nodes/edges yet, and a default `analyze` run is byte-identical to before. De-risking spikes (LadybugDB rel-property indexing, post-dominator feasibility) recorded on the issue. + +## [1.6.6] - 2026-06-08 + +### Added + +- **Scope-resolution (RFC #909) migrations completed across the language matrix** — Rust (#1639), JavaScript (#1640), Ruby (#1831), Swift (#937, #1948), Vue SFC (#940, #1950), Dart (#939, #1970), COBOL (#941, #1835, #1842), and Kotlin (#1727, #1746, #1782) now run on the registry-primary path; Java reached 100% scope-resolution parity and joined `MIGRATED_LANGUAGES` (#1805); per-language progress reporting added to the scope-resolution phase (#1813) +- **HTTP route & consumer contract extraction (group mode)** — Spring interface routes attributed to controllers (#1743); named/positional Java Spring route args (#1834); Kotlin Spring HTTP route, consumer, and WebClient long-form extraction (#1849, #1855, #1884); Java HTTP consumer contracts (#1872); OpenFeign `@RequestLine` consumer contracts incl. plain interfaces without `@FeignClient` (#1904, #1917); FastAPI `include_router(prefix=...)` cross-file routes (#1877); indirect call patterns via FastAPI `Depends()` and frontend HTTP consumers (#1852); gRPC consumer FQN derivation from Java imports for client-jar consumers (#1889) +- **C++ overload & template resolution** — operator-call resolution (#1754), template partial ordering (#1885), user-defined conversion ranking (#1829), nullptr/ellipsis pointer conversion ranks (#1708), SFINAE filter (#1623), expanded `type_traits` constraint registry (#1648), structured resolver-suppression outcomes (#1785), function-type ADL entities (#1822), and a parameter-type class sidecar (#1642) +- **Go enhancements** — structural interface implementation inference (#1966) and a `builtInNames` set for the Go language provider (#1886) +- **Self-healing worker pool** — automatic worker replacement plus deferred-resolution observability and verbose progress logging (#1741, #1773, #1947) +- **`.gitnexusrc` config file and `gitnexus analyze --default-branch`** (#243, #1996) +- **CLI / MCP impact ergonomics** — `--uid/--file/--kind` disambiguation flags (#1907, #1914), `limit/offset/summaryOnly` pagination on the impact tool (#1818), and a per-symbol `processes` field on `byDepth` items (#1867) +- **`gitnexus analyze --repair-fts`** — enforces FTS verification with hardened repair safeguards (#1720) +- **Web viewer** — Tree View and Circles View (#1799), GitLab repository URLs (#1565), `GITNEXUS_BACKEND_URL` env var for Docker deployments (#1286), and web + CLI internationalization (#1748) +- **Wiki** — local Claude/Codex providers (#1769), an opencode local provider (#2039), and `gitnexus wiki --lang ` for multilanguage wiki generation (#1613) +- **`detect-changes` git-worktree support** (#1654) +- **DeepSeek V4 API support** (#1594) +- **Devcontainer for the Claude / Codex / Cursor CLIs** (#1875) and antigravity integration setup + hook adapter (#1730) +- **Object-literal methods linked to exported bindings** (#1718) +- **`eval-server --host`** for a user-configured bind IP (#1667) +- **PR reviewer swarm agents** (#1851) +- **tree-sitter node-type/field validation gate** — validates against the grammar and removes dead literal handling (#1937) + +### Fixed + +- **Parsing-layer coverage gaps closed across the language matrix** (umbrella #1919) — remaining open gaps (#2072) plus Java F35/F38/F41 (#1928, #2045), PHP F53/F54/F55 (#1931, #1989), COBOL F17–F23 (#1925, #1959), Rust F66/F68/F71/F72 (#1934, #1974), Python F57/F58/F61 (#1932, #1964), JS/TS F44/F83/F85/F86/F87 (#1929, #1968), and Ruby F62 (#1933, #1972) +- **Fully-qualified nested-type identity for C++ and Ruby** — distinct nodes for union-, anonymous-namespace-, and same-tail-nested types (#1978, #1981, #2004, #2005); cross-namespace same-tail inheritance bases resolved (#1993, #2005); Ruby same-tail nested mixin modules qualified with `IMPLEMENTS` routed by scope (#1991, #2006); shared codec for `__heritage__`/`__property__` markers (#1994, #2007); graph nodes materialized for scoped class/module/impl declarations (#1975, #1977); generic Rust inherent-impl methods owned through the mod-qualified `Impl` node (#1992, #2003) +- **C# resolution & memory** — global-namespace `typeBindings` O(files²) OOM eliminated (#1871, #1954) and namespace-siblings OOM with worker-path re-parse removed (#1905); qualified/alias constructor names, `:base`/`:this` initializers, and generic type-arg stripping (#2046); primary-base receiver type normalization (#2036); spurious `IMPORTS` edges from ungated `using` resolution stopped (#1881, #1908) +- **C++ dependent-base and member lookup** — resolution across nested/inline namespaces (#1634, #1814), base-specifier qualifier threading (#1815, #1819), call-site types threaded into qualified member lookup (#1632, #1810), variadic pack dependent lookup (#1909), uninitialized multi-declarators (#1965), and typedef-enum / anonymous-struct declarations (#1941) +- **Kotlin type resolution** — smart-cast refinement for `when/is` and `if/is` (#1758, #1774), overload target-id by parameter types (#1761, #1777), cross-file iterable return propagation (#1759, #1775), method-chain fixpoint receiver types (#1760, #1776), virtual dispatch via constructor type override (#1762, #1778), interface default-method dispatch via implements-split MRO (#1763, #1779), and default-parameter arity detection (#2034) +- **Go declarations** — multi-name declaration capture (#2032), fixed-array parameter binding normalization (#1988), and generic composite-literal constructor inference F33 (#1976) +- **Rust / PHP / Vue / Java parsing** — Rust `struct_expression` name pattern split (#2051); PHP import decomposition, namespace-less `.phtml` module scopes, and Blade-template exclusion (#1801, #1790, #1989); Vue JSDoc, dual-script merge, and lang plumbing F89/F90/F92 (#1936, #2050); Java inherited `RequestMapping` prefix deduplication (#2057) and same-module type resolution for duplicate FQNs (#1712) +- **TypeScript** — HOC pattern false positives fixed with `export default` HOC support (#1943) and suffix-index reuse in the scope resolver (#1840) +- **Inheritance on the worker path** — all languages' inheritance migrated to scope-resolution in worker mode (#1951, #1956); centralized heritage supertype matching (#1921, #1922, #1940); `File->Member` `DEFINES` edges skipped for class members (#1949); phantom `Function` defs for array-method callbacks no longer emitted (#1906) +- **MCP** — sibling-clone repo-ID collisions prevented and generated MCP tool names corrected (#2067); orphan processes avoided by handling stdin close/end and the startup race (#2049); duplicate-name repo resolution disambiguated for worktrees (#1753); Windows setup fallback when global `gitnexus` resolves to a non-spawnable shim (#1694) +- **Worker pool** — resilient zero-copy ingestion worker pool prevents analyze hangs on TS-root-scale loads (#1693); cache-hit native workers no longer abort (#1751, #1833); worker-pool docs drift corrected and worker-side stack surfaced on crash (#2068, #2070) +- **LadybugDB** — FTS loaded in the Windows read pool (#2040) and probed-then-loaded on Windows (#1690, #1692); non-ASCII KuzuDB paths resolved on Windows (#1811, #1817); WAL corruption detected in schema init with recovery surfaced (#1647, #1650); WAL checkpoint-threshold control (#1772); init lock skipped for read-only opens (#1783, #1784); `serve` kept stable when sidecars are missing (#1747) +- **Server / API** — `gitnexus serve` startup restored under Express 5 (#1749); `/api/graph`, `/api/search`, `/api/grep` opened read-only (#1686); native read-only enforcement and prepared statements for Cypher query paths (#1655); `eval-server` localhost binding left to the OS (#1722) +- **Embeddings** — local ONNX runtime guarded on macOS Intel before the transformers.js import (#1987) +- **Web agent** — Nexus AI agent system prompt aligned with registered tools (#1984) and the agent stopped cleanly on user Stop (#1820) +- **Group / contracts** — HTTP graph and source contracts unioned (#1709); `httpx` `AsyncClient` alias imports detected (#1687); Node gRPC `loadPackageDefinition` gate no longer matches every member call (#1916); manifest/workspace extraction moved before `closeLbug` (#1802, #1807) +- **Hooks / install** — `gitnexus` resolved on `PATH` via a pure-Node, all-OS scan (#1938, #1980); offline-first extension installs (#1161); actionable error and docs for the `pnpm dlx`/`pnpx` native-load crash (#307, #1967); `onnxruntime-common` declared as a runtime dependency (#2074); vendored grammars materialized to fix Windows EPERM (#1728, #1729) +- **CLI** — missing LadybugDB native binary detected at startup with actionable guidance (#835, #1837); `--no-stats` applied to the keep-marker stats line (#1706, #1765); skipped large-file paths surfaced by default (#1659, #1661); build.js skipped when running outside the monorepo (#1795, #1816); auto-heap raised to 16 GB with tightened cross-platform OOM guidance for UE5-scale repos (#1652) +- **Wiki** — hidden 60s default timeout removed with timeout/retry flag validation and surfaced timeout errors (#1651); budget-aware grouping to prevent context overflow on large repos (#627, #1832) +- **`detect-changes`** — `resolveWorktreeCwd` guarded against overriding a separately-indexed worktree (#1691) +- **Windows reliability** — `windowsHide:true` passed to every `child_process` spawn-family call (#1794) + +### Changed + +- **Legacy resolution deletion (Ring 4)** — removed the legacy call-resolution DAG + heritage processor (RING4-1, #942, #2023), the legacy resolution-context + tiered-lookup plumbing (RING4-2, #943, #2033), and the shadow-mode parity harness (RING4-3, #944, #2071) +- **CONTRIBUTING** — clarified local development setup (#2024) +- **Tests / CI** — cli-e2e made read-only and eval-server tests hardened under load (#2000, #1786, #1838, #1688); parity shards consolidated and the cross-platform matrix narrowed (#1798); devcontainer smoke build hardened against Docker Hub flakes (#1969); gitleaks stabilized (#2027) + +### Performance + +- **Linux-kernel-scale analysis overhaul** — worker-pool parse, finalize O(n²), and the scope-resolution memory wall (#1983, #2038) +- **Scope-capture linearized across all languages (O(n²)→O(n))** plus Python import-resolution linearization (#1918), the Go-specific re-walk fix (#1848, #1915), and owner-keyed lookup for Step 2 member resolution (#1657) +- **C++ ADL candidates indexed once instead of per-site rescans** (#1990) +- **Inert local value symbols pruned** during ingestion (#2065) + +### Chore / Dependencies + +- `@ladybugdb/core` bump in /gitnexus (#2056) +- Routine dependency bumps across /gitnexus, /gitnexus-web, /eval, and GitHub Actions — incl. `hono`, `vitest`, `@vitest/coverage-v8`, `tsx`, `lru-cache`, `express`/`@types/express`, `express-rate-limit`, `qs`, `node-addon-api`, `brace-expansion`, `langchain`, `i18next`, `dompurify`, `lucide-react`, `axios`, `zod`, `@langchain/langgraph`, `@vercel/node`, `langsmith`, `aiohttp`, `idna`, and the `docker/*` / `github/codeql-action` / `release-drafter` / `dependency-review-action` actions (#2056, #2044, #2043, #2042, #2016, #2015, #2013, #2012, #2011, #2010, #2009, #2008, #2018, #2019, #2017, #2020, #1986, #1911, #1864, #1863, #1861, #1860, #1866, #1844, #1845, #1826, #1825, #1824, #1791, #1789, #1768, #1767, #1739, #1740, #1738, #1736, #1735, #1734, #1731, #1713, #1698, #1697, #1696, #1689, #1604, #1552, #1464, #872) +- **Security** — `@vercel/node` upgraded in /gitnexus-web with transitive advisories remediated (#1705) + ## [1.6.5] - 2026-05-16 ### Added diff --git a/gitnexus/README.md b/gitnexus/README.md index 7544508ba..7c84087ea 100644 --- a/gitnexus/README.md +++ b/gitnexus/README.md @@ -400,7 +400,7 @@ Values above **32768 KB (32 MB)** are clamped to the tree-sitter parser ceiling; ### Analyze reports a worker timeout -Worker parse timeouts are recoverable. GitNexus retries stalled worker jobs with backoff, splits large jobs to isolate slow files, and falls back to the sequential parser when needed. If a large repository needs more time per worker job, use either: +Worker parse timeouts are recoverable. GitNexus retries stalled worker jobs with backoff, splits large jobs to isolate slow files, and quarantines a file that repeatedly crashes its worker (respawning the slot so the pool keeps going). If a large repository needs more time per worker job, use either: ```bash # CLI flag, in seconds @@ -423,6 +423,16 @@ Three env vars expose the pool's resilience layers (respawn budget, cumulative-t | `GITNEXUS_WORKER_MAX_CUMULATIVE_TIMEOUT_MS` | `5 × subBatchTimeoutMs` | Total retry wall-time budget per job before quarantining. Bounds exponentially-growing retry waits. | | `GITNEXUS_WORKER_CONSECUTIVE_FAILURE_THRESHOLD` | `max(3, poolSize)` | Per-slot consecutive deaths before the pool's circuit breaker trips. After tripping, dispatches require a fresh pool. | +### Graph cleanup tuning + +After scope resolution, analyze prunes inert block-local value symbols (a function-local `const`/`let`/`var` that ends up with only its structural `File→DEFINES` edge) to keep the graph focused on cross-symbol relationships. Module/file-scope symbols, class members, and any local with a real edge are always kept. + +| Variable | Default | Effect | +| ------------------------------------ | ------- | ------------------------------------------------------------------------------------------------------- | +| `GITNEXUS_KEEP_LOCAL_VALUE_SYMBOLS` | unset | Set to `1`/`true` to keep inert block-local value symbols instead of pruning them. | + +Programmatic callers can pass `keepLocalValueSymbols: true` in `PipelineOptions` instead of setting the env var. + ## Privacy - All processing happens locally on your machine diff --git a/gitnexus/bench/parse-throughput.md b/gitnexus/bench/parse-throughput.md index 24a7c98b1..4d53cd2dc 100644 --- a/gitnexus/bench/parse-throughput.md +++ b/gitnexus/bench/parse-throughput.md @@ -70,10 +70,17 @@ this doc, run it under instrumentation: ```bash # From the gitnexus/ subdir: cd gitnexus -# Single-threaded baseline (sequential fallback): -npx vitest run test/integration/parse-impl-large-fixture.test.ts --reporter=verbose +# The worker pool is the sole parse path, so every run needs the dist worker +# (`npm run build`) and a pool size pinned via GITNEXUS_WORKER_POOL_SIZE. -# Worker-pool path (requires built dist/ — pre-built by `npm run build`): +# Single-worker-pool baseline (closest analog to the old single-threaded run — +# sequential parsing was removed, so a 1-worker pool is the floor): +npm run build && \ + GITNEXUS_WORKER_POOL_SIZE=1 \ + GITNEXUS_VERBOSE=1 \ + npx vitest run test/integration/parse-impl-large-fixture.test.ts --reporter=verbose + +# Multi-worker path: npm run build && \ GITNEXUS_WORKER_POOL_SIZE=4 \ GITNEXUS_PARSE_CHUNK_CONCURRENCY=2 \ @@ -97,22 +104,26 @@ node --inspect=0 \ ## Latest measurement > _No measurement data has been collected yet — this file is the -> methodology + harness scaffold. The single recorded data point is the -> U6 wall-clock smoke baseline below; the worker-pool rows are -> placeholders for future bench-pass output._ +> methodology + harness scaffold. The U6 smoke test confirms the +> worker-pool path stays well within its wall-clock budget, but every +> throughput/heap cell below is a `_TBD_` placeholder for a future +> bench-pass._ The U6 integration test (`gitnexus/test/integration/parse-impl-large-fixture.test.ts`) -was observed completing the synthetic fixture in **~6 seconds** under -the sequential path (`skipWorkers: true`) on the development machine, -well under the 30 s `Promise.race` wall-clock budget. That number is a -smoke baseline only — recorded here for reference, not as a regression -target. +runs the worker pool — the sole parse path now that sequential parsing +has been removed (disabling the pool on a repo with parseable files +raises a hard `WorkerPoolDisabledError`). It completes the synthetic +fixture well within the 30 s `Promise.race` wall-clock budget on the +development machine, but no worker-pool throughput/heap numbers have been +captured yet, so the rows below are all `_TBD_`. (An earlier ~6 s figure +recorded here was measured on the now-removed sequential path; it has +been dropped rather than relabelled as a worker-pool baseline, since the +two paths are not comparable.) -| Path | files/s | wall-clock | peak heap | chunks | quarantined | -| ------------------------------------------ | ------- | -------------------- | --------- | ------ | ----------- | -| Sequential (`skipWorkers: true`, U6 smoke) | _TBD_ | ~6 s _(observation)_ | _TBD_ | 17 | 0 | -| Worker pool, `--workers 4`, concurrency 2 | _TBD_ | _TBD_ | _TBD_ | _TBD_ | 0 | -| Worker pool, `--workers 1`, concurrency 1 | _TBD_ | _TBD_ | _TBD_ | _TBD_ | 0 | +| Path | files/s | wall-clock | peak heap | chunks | quarantined | +| ------------------------------------------------------------------------- | ------- | ---------- | --------- | ------ | ----------- | +| Worker pool, `--workers 1` (`GITNEXUS_WORKER_POOL_SIZE=1`), concurrency 1 | _TBD_ | _TBD_ | _TBD_ | _TBD_ | 0 | +| Worker pool, `--workers 4`, concurrency 2 | _TBD_ | _TBD_ | _TBD_ | _TBD_ | 0 | **Hardware:** _TBD — record OS, CPU, RAM, Node version, gitnexus SHA at the time of the bench-pass that populates the table above._ diff --git a/gitnexus/bench/scope-capture/baselines.json b/gitnexus/bench/scope-capture/baselines.json index 225238dd8..bb0a3f5f4 100644 --- a/gitnexus/bench/scope-capture/baselines.json +++ b/gitnexus/bench/scope-capture/baselines.json @@ -11,16 +11,18 @@ "_note": "Updated for F17-F23 fixes (P2: TIMES guard, ADD GIVING, SQL AS alias). See PR #1959." }, "c": { - "fingerprint": "0de009bdbfe095f530fa87eb32bce6ab83092c904f26b3c8fe8d8ab587cf6dc9", + "fingerprint": "12a196b2d6249c8d86a931b12ecebc2a0cdf8d6f47683acdd0d8e9d8bc7657f5", "scaling_budget": 1.5, - "_added": "#1956: c added to the scope-capture bench (was UNBENCHED). C has no inheritance \u2014 flat scale source. Adding it exposed + fixed a pre-existing O(n^2) findNodeAtRange root-walk in c/captures.ts (threaded c.node, byte-identical over c-* fixtures); scaling 3.475 -> 0.96." + "_added": "#1956: c added to the scope-capture bench (was UNBENCHED). C has no inheritance — flat scale source. Adding it exposed + fixed a pre-existing O(n^2) findNodeAtRange root-walk in c/captures.ts (threaded c.node, byte-identical over c-* fixtures); scaling 3.475 -> 0.96.", + "_note": "#1983: + c-static-linkage-worker fixture (caller.c/lib.c/lib.h/local.c — worker-path static-linkage side-channel test). Pure fixture-corpus drift: no c/captures.ts or query change branch-vs-main, existing fixtures' captures byte-identical (c-captures.test.ts 45/45), scaling stays linear (~0.97). The baseline was missed when the fixture landed; regenerated here. fingerprint 0de009b->39f3a83.", + "_rebaselined": "#1919 open-language coverage: new lang-resolution fixtures + intended capture additions (F5/F9 c-cpp, F26/F28/F29 dart, F47/F48/F49/F51/F52 kotlin, F75/F79 swift). Fingerprint-only drift; scaling_ratio ~1.0 (linear, no perf regression)." }, "cpp": { - "fingerprint": "6d6207ae1df3943c5fae28983e0c294e55225456e7cf39af1d46fda21b6787c4", + "fingerprint": "f56625342f73e182170e2c964d538e316c079fa6e9466a7f076bff2ebcf8aac4", "scaling_budget": 1.5, "_added": "#1956: cpp added to the scope-capture bench (was UNBENCHED). Heritage-bearing scale source (: public Base, public Mixin) drives emitCppInheritanceCaptures at scale. Adding it exposed + fixed a pre-existing O(n^2) findNodeAtRange root-walk in cpp/captures.ts (~12 sites, threaded c.node, byte-identical over 263 cpp-* fixtures); scaling 2.30 -> 1.12.", - "_rebaselined": "#1965 / #1923 F4: uninitialized non-leading multi-declarators now emit @declaration.variable captures; cpp-adl-inner-callable-outer-noncallable data::Pair a, b adds the legitimate fixture drift. Linear (~1.06).", - "_note": "#1975: + cpp-out-of-line-class fixture, fixture_count 263->265. #1990: + cpp-adl-ns-plus-hidden-friend-same-name fixture (ADL hidden-friend + namespace-callable merge parity test). Pure fixture-corpus drift \u2014 no scope-extractor change; existing fixtures' captures byte-identical. fixture_count 265->267. #1995: + cpp-union-nested-tail-collision and cpp-anon-ns-tail-collision fixtures \u2014 pure fixture-corpus drift; fixture_count 270->272, fingerprint 538e8be->d63ded6. #1993: + cpp-cross-namespace-same-tail fixture \u2014 pure fixture-corpus drift; fixture_count 272->273, fingerprint d63ded6->6d6207ae." + "_rebaselined": "#1919 open-language coverage: new lang-resolution fixtures + intended capture additions (F5/F9 c-cpp, F26/F28/F29 dart, F47/F48/F49/F51/F52 kotlin, F75/F79 swift). Fingerprint-only drift; scaling_ratio ~1.0 (linear, no perf regression).", + "_note": "#1975: + cpp-out-of-line-class fixture, fixture_count 263->265. #1990: + cpp-adl-ns-plus-hidden-friend-same-name fixture (ADL hidden-friend + namespace-callable merge parity test). Pure fixture-corpus drift — no scope-extractor change; existing fixtures' captures byte-identical. fixture_count 265->267. #1995: + cpp-union-nested-tail-collision and cpp-anon-ns-tail-collision fixtures — pure fixture-corpus drift; fixture_count 270->272, fingerprint 538e8be->d63ded6. #1993: + cpp-cross-namespace-same-tail fixture — pure fixture-corpus drift; fixture_count 272->273, fingerprint d63ded6->6d6207ae. #2077 review follow-up: cpp-member-lattice adds cross-file, qualified-base, nested-template, inherited-using, this-receiver, and non-virtual-override regressions; fixture_count 274->275. Capture scaling remains linear (1.134 < 1.5)." }, "csharp": { "_rebaselined": "#1956 synth-widening: + csharp-qualified-base fixture; the synth now walks record_declaration + struct_declaration base_lists and handles alias_qualified_name (matching the #1940 legacy leg), so record/struct heritage now emits. csharp-record-base gains a record inherits capture. (record->record SAME-namespace EXTENDS is a separate registry resolution gap, tracked as follow-up.) Linear (~1.00). (Earlier #1956: heritage-bearing scale source.) | #942: scope-resolution-only cleanup reworded fixture comments; capture byte-positions shift, capture LOGIC unchanged. | #1924 F16: record primary-constructor base bindings now exclude constructor arguments; capture fingerprint changes, scaling remains linear. | #2036 review follow-up: csharp-record-base now exercises primary-constructor base dispatch end to end; +2 capture groups, scaling remains linear.", @@ -31,8 +33,8 @@ "rust": { "fingerprint": "ac610bbe97666bf285923479dd7b43a2fe4c5354aae8df1bcbafdc04fb220f82", "scaling_budget": 1.5, - "_rebaselined": "#1956 tri-review U1: rust-qualified-trait fixture (scoped + generic-of-scoped impl trait paths); bareTypeIdentifier now resolves scoped_type_identifier bases by their name: tail (additive, no existing-fixture drift); linear (~1.04). #1975: + rust-scoped-impl fixture (impl a::Inner / b::Inner inherent scoped impls) \u2014 legacy @definition.impl scoped arm + findEnclosingClassInfo inherent-impl scoped target; rust scope-extractor captures byte-identical. | #942: scope-resolution-only cleanup reworded fixture comments; capture byte-positions shift, capture LOGIC unchanged.", - "_note": "PR #1934: F66/F68 let-binding pattern narrowing; F71 union (Struct-labeled, now materialized via legacy @definition.struct + resolvable); F72 macro FULLY WIRED \u2014 @declaration.macro/@reference.macro + MacroRegistry \u2192 USES edges to Macro nodes (never a same-named fn). + rust-macro / rust-union fixtures and merged with origin/main #1975 rust-scoped-impl; fingerprint re-baselined (scaling ~0.99, fixture_count 126). #1992: + rust-nested-tail-collision-generic and rust-generic-impl-same-method-name (F3) fixtures \u2014 pure fixture-corpus drift, no scope-extractor change; fixture_count 127->129, fingerprint 56ffc1c0->b00aea0f." + "_rebaselined": "#1956 tri-review U1: rust-qualified-trait fixture (scoped + generic-of-scoped impl trait paths); bareTypeIdentifier now resolves scoped_type_identifier bases by their name: tail (additive, no existing-fixture drift); linear (~1.04). #1975: + rust-scoped-impl fixture (impl a::Inner / b::Inner inherent scoped impls) — legacy @definition.impl scoped arm + findEnclosingClassInfo inherent-impl scoped target; rust scope-extractor captures byte-identical. | #942: scope-resolution-only cleanup reworded fixture comments; capture byte-positions shift, capture LOGIC unchanged.", + "_note": "PR #1934: F66/F68 let-binding pattern narrowing; F71 union (Struct-labeled, now materialized via legacy @definition.struct + resolvable); F72 macro FULLY WIRED — @declaration.macro/@reference.macro + MacroRegistry → USES edges to Macro nodes (never a same-named fn). + rust-macro / rust-union fixtures and merged with origin/main #1975 rust-scoped-impl; fingerprint re-baselined (scaling ~0.99, fixture_count 126). #1992: + rust-nested-tail-collision-generic and rust-generic-impl-same-method-name (F3) fixtures — pure fixture-corpus drift, no scope-extractor change; fixture_count 127->129, fingerprint 56ffc1c0->b00aea0f." }, "php": { "fingerprint": "bc2c27c5ba26d5aea61142a2a99fb772222f5b969205260eb7a71b4c0bd73cdb", @@ -44,18 +46,18 @@ "fingerprint": "b5ea93bb3d0469c3821a8c70f5d5991c6f326e41097c119ad691154301dcc753", "scaling_budget": 1.5, "_rebaselined": "#1956 synth-widening: + ruby-qualified-base fixture; synth now reduces a scope_resolution superclass (class C < Mod::Super) to its trailing constant (matching the #1940 legacy leg), at parity. Linear (~1.03). (Earlier #1956: heritage-bearing scale source.) | #942: scope-resolution-only cleanup reworded fixture comments; capture byte-positions shift, capture LOGIC unchanged.", - "_note": "F62: + scope_resolution class/module declaration captures \u2014 fixture count 78\u219281, fingerprint drift expected. #1975: + ruby-tail-collision fixture (Foo::Bar vs Baz::Bar stay distinct nodes) \u2014 pure fixture-corpus drift, scope-extractor captures unchanged; 81\u219282. #1991: + ruby-nested-mixin-tail-collision fixture (85\u219286). Recomputed on the #942 merge (fixture-comment rewording shifts capture byte-positions, capture LOGIC unchanged): bf6b13a -> b5ea93bb." + "_note": "F62: + scope_resolution class/module declaration captures — fixture count 78→81, fingerprint drift expected. #1975: + ruby-tail-collision fixture (Foo::Bar vs Baz::Bar stay distinct nodes) — pure fixture-corpus drift, scope-extractor captures unchanged; 81→82. #1991: + ruby-nested-mixin-tail-collision fixture (85→86). Recomputed on the #942 merge (fixture-comment rewording shifts capture byte-positions, capture LOGIC unchanged): bf6b13a -> b5ea93bb." }, "swift": { - "fingerprint": "53325c6345161c5a495f997297af5a24fb718fd3e6647040160f8ab2a2c8e4c0", + "fingerprint": "180ac68e780bdf6f9089d53f51cbb9a66aed3e7774631cc3fcbaae5020213998", "scaling_budget": 1.5, - "_rebaselined": "#1956: swift-qualified-base fixture + heritage-bearing scale source (class: Base, Serviceable \u2014 extends + protocol conformance); linear (~1.03)." + "_rebaselined": "#1919 open-language coverage: new lang-resolution fixtures + intended capture additions (F5/F9 c-cpp, F26/F28/F29 dart, F47/F48/F49/F51/F52 kotlin, F75/F79 swift). Fingerprint-only drift; scaling_ratio ~1.0 (linear, no perf regression)." }, "dart": { - "fingerprint": "a9e882b537765e8fd0ddfcd33b38b253dd86fc5ddffa6e4bf5a85ed8ee615eaa", + "fingerprint": "94bf2c26e1ba96f4211634aa572c0a989b503e717e75dfc5df04f66c417de80f", "scaling_budget": 1.5, "_added": "#939: dart added to the scope-capture bench with the registry-primary migration. Heritage-bearing scale source (Entity extends Base implements Marker) gates the @reference.inherits synth + the postfix-chain reference walk at scale. emitDartScopeCaptures threads tree-sitter captured nodes (no findNodeAtRange root-walk), so it is linear (~1.0).", - "_rebaselined": "#1970 review + tri-review follow-ups: constructor-call retag, cascade calls, built-in suppression, enum scope, #1926 F24/F25, named-ctor dedup (crash fix), container-name binding suppression; heritage file-affinity resolution. Fixtures: member-call-contexts, constructor-body, named-constructor-body, heritage-name-collision, construct-cascade." + "_rebaselined": "#1919 review CF3 fix: extended kotlin-local-property-owner (init/accessor destructuring) + new dart-accessor-owner fixture (getter/setter ownership). Fingerprint-only corpus drift; scaling ~1.0." }, "java": { "fingerprint": "9b29cafe32873b4902bda311bd089ffc04efe08f13557b966d29544be514080a", @@ -66,8 +68,8 @@ "typescript": { "fingerprint": "3f44a4a6892698df2d145c8ff2812c3b318807648983c88aca28fbd694f172f9", "scaling_budget": 1.5, - "_rebaselined": "#1962: F44 (class scope@), F85 (enum member declarations), F87 (optional_parameter type annotations) add new captures \u2014 fingerprint drift expected.", - "_note": "#1968: F44, F85, F87 \u2014 fingerprint drift expected." + "_rebaselined": "#1962: F44 (class scope@), F85 (enum member declarations), F87 (optional_parameter type annotations) add new captures — fingerprint drift expected.", + "_note": "#1968: F44, F85, F87 — fingerprint drift expected." }, "javascript": { "fingerprint": "d72f03c6c502235d2d4b74d66baa5c7d361f040d7a1b72e84acad61210d05ae8", @@ -76,9 +78,9 @@ "_rebaselined": "#1956 synth-widening: + javascript-qualified-base fixture; synthesizeJsInheritanceReferences now handles a member_expression base (class S extends ns.Base -> Base), matching the #1940 legacy leg + the TS terminalTsTypeNameNode property_identifier case, at parity. Linear (~1.05). | #942: scope-resolution-only cleanup reworded fixture comments; capture byte-positions shift, capture LOGIC unchanged." }, "kotlin": { - "fingerprint": "a16400622892183581b8f5f8fa01f07842d19b8cf49ed021df52bd17009d749f", + "fingerprint": "90aa832978d9744e50058e77a04748390a7e34e36b309f6c1d178eb07280b7ea", "scaling_budget": 1.5, "_added": "#1951: bench coverage added (was ungated); scale source heritage-bearing (: Base()); js/kotlin O(n^2) findNodeAtRange-per-match fixed to threaded captured node, now linear.", - "_rebaselined": "#1956 synth-widening: + kotlin-qualified-base fixture; synthesizeKotlinInheritanceReferences now handles the explicit_delegation form (class F : Iface by d -> Iface), matching the #1940 legacy leg, at parity. Linear (~0.87). | #942: scope-resolution-only cleanup reworded fixture comments; capture byte-positions shift, capture LOGIC unchanged. | #1930 F45: default parameters now emit optional-arity metadata; capture fingerprint changes, scaling remains linear." + "_rebaselined": "#1919 review CF3 fix: extended kotlin-local-property-owner (init/accessor destructuring) + new dart-accessor-owner fixture (getter/setter ownership). Fingerprint-only corpus drift; scaling ~1.0." } } diff --git a/gitnexus/package-lock.json b/gitnexus/package-lock.json index 280d5b366..e4fee45fc 100644 --- a/gitnexus/package-lock.json +++ b/gitnexus/package-lock.json @@ -1,17 +1,17 @@ { "name": "gitnexus", - "version": "1.6.5", + "version": "1.6.6", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "gitnexus", - "version": "1.6.5", + "version": "1.6.6", "hasInstallScript": true, "license": "PolyForm-Noncommercial-1.0.0", "dependencies": { "@huggingface/transformers": "^4.1.0", - "@ladybugdb/core": "^0.16.1", + "@ladybugdb/core": "^0.17.0", "@modelcontextprotocol/sdk": "^1.0.0", "@scarf/scarf": "^1.4.0", "cli-progress": "^3.12.0", @@ -26,8 +26,8 @@ "ignore": "^7.0.5", "js-yaml": "^4.1.1", "jsonc-parser": "^3.3.1", - "lru-cache": "^11.0.0", "mnemonist": "^0.40.3", + "onnxruntime-common": "^1.26.0", "onnxruntime-node": "^1.24.0", "pandemonium": "^2.4.0", "pino": "^10.3.1", @@ -1159,27 +1159,28 @@ } }, "node_modules/@ladybugdb/core": { - "version": "0.16.1", - "resolved": "https://registry.npmjs.org/@ladybugdb/core/-/core-0.16.1.tgz", - "integrity": "sha512-qwuEcR8CVMKb6tNDaHtq7Ux8hT/XbPC0db+vwutX6JxNAejyx7YomHKPSy9XAKURhYK8mezZe3UN8rf+xpHOjQ==", + "version": "0.17.1", + "resolved": "https://registry.npmjs.org/@ladybugdb/core/-/core-0.17.1.tgz", + "integrity": "sha512-K1bHnQrRy3bxkyrFHlxGqKUyIUS1LsRXKOSt14XGY/msBZHaDat/uBrlHiWpM4/24OtfOq/qwTqcTCXannnEjw==", "hasInstallScript": true, "license": "MIT", "dependencies": { + "apache-arrow": "^21.1.0", "cmake-js": "^8.0.0", "node-addon-api": "^6.0.0" }, "optionalDependencies": { - "@ladybugdb/core-darwin-arm64": "0.16.1", - "@ladybugdb/core-darwin-x64": "0.16.1", - "@ladybugdb/core-linux-arm64": "0.16.1", - "@ladybugdb/core-linux-x64": "0.16.1", - "@ladybugdb/core-win32-x64": "0.16.1" + "@ladybugdb/core-darwin-arm64": "0.17.1", + "@ladybugdb/core-darwin-x64": "0.17.1", + "@ladybugdb/core-linux-arm64": "0.17.1", + "@ladybugdb/core-linux-x64": "0.17.1", + "@ladybugdb/core-win32-x64": "0.17.1" } }, "node_modules/@ladybugdb/core-darwin-arm64": { - "version": "0.16.1", - "resolved": "https://registry.npmjs.org/@ladybugdb/core-darwin-arm64/-/core-darwin-arm64-0.16.1.tgz", - "integrity": "sha512-Nl+Cf70rD+HaC9IBHv+oeUwqX9plghXD7PN9tyMzMohRVPvcGEbqWPB6YcdJa8rR7qRqCCbmaNMDen5wg4rY2w==", + "version": "0.17.1", + "resolved": "https://registry.npmjs.org/@ladybugdb/core-darwin-arm64/-/core-darwin-arm64-0.17.1.tgz", + "integrity": "sha512-JG/uzmolEh3wXJ/ME1EaTH5LTDQ9Cs+Q3Czul8pW2eWbWQZghQU3jjM++7ST7Bla5BX/WITqwPqPoC+sL+slfA==", "cpu": [ "arm64" ], @@ -1190,9 +1191,9 @@ ] }, "node_modules/@ladybugdb/core-darwin-x64": { - "version": "0.16.1", - "resolved": "https://registry.npmjs.org/@ladybugdb/core-darwin-x64/-/core-darwin-x64-0.16.1.tgz", - "integrity": "sha512-4eAjfimAAQRSmDfUUkGrl9OhefxcW1ziA9tl0eljBlGoUseE7dL02+RSqjGohYMcQ+lzuHAq1QWb0XRlMA8YTQ==", + "version": "0.17.1", + "resolved": "https://registry.npmjs.org/@ladybugdb/core-darwin-x64/-/core-darwin-x64-0.17.1.tgz", + "integrity": "sha512-Enjm+/V9/jpKmtzF2PB0muVkgpFUGHEvA7r16eJWxVRA/BeO8VPmngTKy9rf/4Yc6TWexjoHRug04BbTXEmerg==", "cpu": [ "x64" ], @@ -1203,9 +1204,9 @@ ] }, "node_modules/@ladybugdb/core-linux-arm64": { - "version": "0.16.1", - "resolved": "https://registry.npmjs.org/@ladybugdb/core-linux-arm64/-/core-linux-arm64-0.16.1.tgz", - "integrity": "sha512-zkctksev+hsPFrNxHHdq4lYK5OWdLhWfRdQzjzkgDyaHayHU6yCL2fgD6uPGQ8TRQ6/2DxMErb4p3FzGW85Ubw==", + "version": "0.17.1", + "resolved": "https://registry.npmjs.org/@ladybugdb/core-linux-arm64/-/core-linux-arm64-0.17.1.tgz", + "integrity": "sha512-P+xM9o4I3JAQtXpX19ZuLj9EeO2gppa+IdmAqhpI8tuhyA3/a85Eaxby1fXOjsbrnOAEyFJczUdyoDkhCPSyiw==", "cpu": [ "arm64" ], @@ -1216,9 +1217,9 @@ ] }, "node_modules/@ladybugdb/core-linux-x64": { - "version": "0.16.1", - "resolved": "https://registry.npmjs.org/@ladybugdb/core-linux-x64/-/core-linux-x64-0.16.1.tgz", - "integrity": "sha512-5rAb9T5vif8WKhHwhobosu2/aiOwJkWb/ViybvUc5GFKunKl8VI6RmZQVeufT9zUzRktUwrxBrxblCxsnamXJw==", + "version": "0.17.1", + "resolved": "https://registry.npmjs.org/@ladybugdb/core-linux-x64/-/core-linux-x64-0.17.1.tgz", + "integrity": "sha512-N2ujE0CrsToBpVBpou1iWwEkK7CgVxucnUNxteySrnDccZwICXFP5BlcFpKE0qq3Eqmqszh4ptR4GuSi6rKPGw==", "cpu": [ "x64" ], @@ -1229,9 +1230,9 @@ ] }, "node_modules/@ladybugdb/core-win32-x64": { - "version": "0.16.1", - "resolved": "https://registry.npmjs.org/@ladybugdb/core-win32-x64/-/core-win32-x64-0.16.1.tgz", - "integrity": "sha512-ShOUTrIuZKQ63J95tcRJxKf1cvg8yi2FSYx9kMTSercc1FdQZPV+zxUN0myMq3MTWOl7xDxsVMmdp/t80O29UQ==", + "version": "0.17.1", + "resolved": "https://registry.npmjs.org/@ladybugdb/core-win32-x64/-/core-win32-x64-0.17.1.tgz", + "integrity": "sha512-9i3xNfFAMqFRuQG3F1hOCWYGna6eTg8HJ/XYhWVDGkeFJNUV3IdneEiYttF5B2qAtQYUd4sAikScsImrMRw+6g==", "cpu": [ "x64" ], @@ -1675,6 +1676,15 @@ "dev": true, "license": "MIT" }, + "node_modules/@swc/helpers": { + "version": "0.5.23", + "resolved": "https://registry.npmjs.org/@swc/helpers/-/helpers-0.5.23.tgz", + "integrity": "sha512-5lSsMOTXURePglDfvuAQUqkGek9Hg2kksOYay2m0+XR++b2NWYL/4sWyuvVBIs8oKnJaxkdi9whaL/sqN13afw==", + "license": "Apache-2.0", + "dependencies": { + "tslib": "^2.8.0" + } + }, "node_modules/@tybys/wasm-util": { "version": "0.10.2", "resolved": "https://registry.npmjs.org/@tybys/wasm-util/-/wasm-util-0.10.2.tgz", @@ -1718,6 +1728,18 @@ "@types/node": "*" } }, + "node_modules/@types/command-line-args": { + "version": "5.2.3", + "resolved": "https://registry.npmjs.org/@types/command-line-args/-/command-line-args-5.2.3.tgz", + "integrity": "sha512-uv0aG6R0Y8WHZLTamZwtfsDLVRnOa+n+n5rEvFWL5Na5gZ8V2Teab/duDPFzIIIhs9qizDpcavCusCLJZu62Kw==", + "license": "MIT" + }, + "node_modules/@types/command-line-usage": { + "version": "5.0.4", + "resolved": "https://registry.npmjs.org/@types/command-line-usage/-/command-line-usage-5.0.4.tgz", + "integrity": "sha512-BwR5KP3Es/CSht0xqBcUXS3qCAUVXwpRKsV2+arxeb65atasuXG9LykC9Ab10Cw3s2raH92ZqOeILaQbsB2ACg==", + "license": "MIT" + }, "node_modules/@types/connect": { "version": "3.4.38", "resolved": "https://registry.npmjs.org/@types/connect/-/connect-3.4.38.tgz", @@ -2069,12 +2091,56 @@ "url": "https://github.com/chalk/ansi-styles?sponsor=1" } }, + "node_modules/apache-arrow": { + "version": "21.1.0", + "resolved": "https://registry.npmjs.org/apache-arrow/-/apache-arrow-21.1.0.tgz", + "integrity": "sha512-kQrYLxhC+NTVVZ4CCzGF6L/uPVOzJmD1T3XgbiUnP7oTeVFOFgEUu6IKNwCDkpFoBVqDKQivlX4RUFqqnWFlEA==", + "license": "Apache-2.0", + "dependencies": { + "@swc/helpers": "^0.5.11", + "@types/command-line-args": "^5.2.3", + "@types/command-line-usage": "^5.0.4", + "@types/node": "^24.0.3", + "command-line-args": "^6.0.1", + "command-line-usage": "^7.0.1", + "flatbuffers": "^25.1.24", + "json-bignum": "^0.0.3", + "tslib": "^2.6.2" + }, + "bin": { + "arrow2csv": "bin/arrow2csv.js" + } + }, + "node_modules/apache-arrow/node_modules/@types/node": { + "version": "24.13.0", + "resolved": "https://registry.npmjs.org/@types/node/-/node-24.13.0.tgz", + "integrity": "sha512-5vtOqGQr4NJKeEzV441FcOi2MeG9UTWq9LqVLGneDdu4vlX17H8kQ2PA2UmNwCUGPVDj4oBjNhS7ReVEIWJJrg==", + "license": "MIT", + "dependencies": { + "undici-types": "~7.18.0" + } + }, + "node_modules/apache-arrow/node_modules/undici-types": { + "version": "7.18.2", + "resolved": "https://registry.npmjs.org/undici-types/-/undici-types-7.18.2.tgz", + "integrity": "sha512-AsuCzffGHJybSaRrmr5eHr81mwJU3kjw6M+uprWvCXiNeN9SOGwQ3Jn8jb8m3Z6izVgknn1R0FTCEAP2QrLY/w==", + "license": "MIT" + }, "node_modules/argparse": { "version": "2.0.1", "resolved": "https://registry.npmjs.org/argparse/-/argparse-2.0.1.tgz", "integrity": "sha512-8+9WqebbFzpX9OR+Wa6O29asIogeRMzcGtAINdpMHHyAg10f05aSFVBbcEqGf/PXw1EjAZ+q2/bEBg3DvurK3Q==", "license": "Python-2.0" }, + "node_modules/array-back": { + "version": "6.2.3", + "resolved": "https://registry.npmjs.org/array-back/-/array-back-6.2.3.tgz", + "integrity": "sha512-SGDvmg6QTYiTxCBkYVmThcoa67uLl35pyzRHdpCGBOcqFy6BtwnphoFPk7LhJshD+Yk1Kt35WGWeZPTgwR4Fhw==", + "license": "MIT", + "engines": { + "node": ">=12.17" + } + }, "node_modules/assertion-error": { "version": "2.0.1", "resolved": "https://registry.npmjs.org/assertion-error/-/assertion-error-2.0.1.tgz", @@ -2199,6 +2265,37 @@ "node": ">=18" } }, + "node_modules/chalk": { + "version": "4.1.2", + "resolved": "https://registry.npmjs.org/chalk/-/chalk-4.1.2.tgz", + "integrity": "sha512-oKnbhFyRIXpUuez8iBMmyEa4nbj4IOQyuhc/wy9kY7/WVPcwIO9VA668Pu8RkO7+0G76SLROeyw9CpQ061i4mA==", + "license": "MIT", + "dependencies": { + "ansi-styles": "^4.1.0", + "supports-color": "^7.1.0" + }, + "engines": { + "node": ">=10" + }, + "funding": { + "url": "https://github.com/chalk/chalk?sponsor=1" + } + }, + "node_modules/chalk-template": { + "version": "0.4.0", + "resolved": "https://registry.npmjs.org/chalk-template/-/chalk-template-0.4.0.tgz", + "integrity": "sha512-/ghrgmhfY8RaSdeo43hNXxpoHAtxdbskUHjPpfqUWGttFgycUhYPGx3YZBCnUCvOa7Doivn1IZec3DEGFoMgLg==", + "license": "MIT", + "dependencies": { + "chalk": "^4.1.2" + }, + "engines": { + "node": ">=12" + }, + "funding": { + "url": "https://github.com/chalk/chalk-template?sponsor=1" + } + }, "node_modules/chownr": { "version": "3.0.0", "resolved": "https://registry.npmjs.org/chownr/-/chownr-3.0.0.tgz", @@ -2281,6 +2378,44 @@ "integrity": "sha512-IfEDxwoWIjkeXL1eXcDiow4UbKjhLdq6/EuSVR9GMN7KVH3r9gQ83e73hsz1Nd1T3ijd5xv1wcWRYO+D6kCI2w==", "license": "MIT" }, + "node_modules/command-line-args": { + "version": "6.0.2", + "resolved": "https://registry.npmjs.org/command-line-args/-/command-line-args-6.0.2.tgz", + "integrity": "sha512-AIjYVxrV9X752LmPDLbVYv8aMCuHPSLZJXEo2qo/xJfv+NYhaZ4sMSF01rM+gHPaMgvPM0l5D/F+Qx+i2WfSmQ==", + "license": "MIT", + "dependencies": { + "array-back": "^6.2.3", + "find-replace": "^5.0.2", + "lodash.camelcase": "^4.3.0", + "typical": "^7.3.0" + }, + "engines": { + "node": ">=12.20" + }, + "peerDependencies": { + "@75lb/nature": "latest" + }, + "peerDependenciesMeta": { + "@75lb/nature": { + "optional": true + } + } + }, + "node_modules/command-line-usage": { + "version": "7.0.4", + "resolved": "https://registry.npmjs.org/command-line-usage/-/command-line-usage-7.0.4.tgz", + "integrity": "sha512-85UdvzTNx/+s5CkSgBm/0hzP80RFHAa7PsfeADE5ezZF3uHz3/Tqj9gIKGT9PTtpycc3Ua64T0oVulGfKxzfqg==", + "license": "MIT", + "dependencies": { + "array-back": "^6.2.2", + "chalk-template": "^0.4.0", + "table-layout": "^4.1.1", + "typical": "^7.3.0" + }, + "engines": { + "node": ">=12.20.0" + } + }, "node_modules/commander": { "version": "14.0.3", "resolved": "https://registry.npmjs.org/commander/-/commander-14.0.3.tgz", @@ -2819,6 +2954,23 @@ "url": "https://opencollective.com/express" } }, + "node_modules/find-replace": { + "version": "5.0.2", + "resolved": "https://registry.npmjs.org/find-replace/-/find-replace-5.0.2.tgz", + "integrity": "sha512-Y45BAiE3mz2QsrN2fb5QEtO4qb44NcS7en/0y9PEVsg351HsLeVclP8QPMH79Le9sH3rs5RSwJu99W0WPZO43Q==", + "license": "MIT", + "engines": { + "node": ">=14" + }, + "peerDependencies": { + "@75lb/nature": "latest" + }, + "peerDependenciesMeta": { + "@75lb/nature": { + "optional": true + } + } + }, "node_modules/flatbuffers": { "version": "25.9.23", "resolved": "https://registry.npmjs.org/flatbuffers/-/flatbuffers-25.9.23.tgz", @@ -3057,7 +3209,6 @@ "version": "4.0.0", "resolved": "https://registry.npmjs.org/has-flag/-/has-flag-4.0.0.tgz", "integrity": "sha512-EykJT/Q1KjTWctppgIAgfSO0tKVuZUjhgMr17kqTumMl6Afv3EISleU7qZUzoXDFTAHTDC4NOoG/ZxU3EvlMPQ==", - "dev": true, "license": "MIT", "engines": { "node": ">=8" @@ -3296,6 +3447,14 @@ "js-yaml": "bin/js-yaml.js" } }, + "node_modules/json-bignum": { + "version": "0.0.3", + "resolved": "https://registry.npmjs.org/json-bignum/-/json-bignum-0.0.3.tgz", + "integrity": "sha512-2WHyXj3OfHSgNyuzDbSxI1w2jgw5gkWSWhS7Qg4bWXx1nLk3jnbwfUeS0PSba3IzpTUWdHxBieELUzXRjQB2zg==", + "engines": { + "node": ">=0.8" + } + }, "node_modules/json-schema-traverse": { "version": "1.0.0", "resolved": "https://registry.npmjs.org/json-schema-traverse/-/json-schema-traverse-1.0.0.tgz", @@ -3587,6 +3746,12 @@ "url": "https://opencollective.com/parcel" } }, + "node_modules/lodash.camelcase": { + "version": "4.3.0", + "resolved": "https://registry.npmjs.org/lodash.camelcase/-/lodash.camelcase-4.3.0.tgz", + "integrity": "sha512-TwuEnCnxbc3rAvhf/LbG7tJUDzhqXyFnv3dtzLOPgCG/hODL7WFnsbwktkD7yUV0RrreP/l1PALq/YSg6VvjlA==", + "license": "MIT" + }, "node_modules/long": { "version": "5.3.2", "resolved": "https://registry.npmjs.org/long/-/long-5.3.2.tgz", @@ -4693,7 +4858,6 @@ "version": "7.2.0", "resolved": "https://registry.npmjs.org/supports-color/-/supports-color-7.2.0.tgz", "integrity": "sha512-qpCAvRl9stuOHveKsn7HncJRvv501qIacKzQlO/+Lwxc9+0q2wLyv4Dfvt80/DPn2pqOBsJdDiogXGR9+OvwRw==", - "dev": true, "license": "MIT", "dependencies": { "has-flag": "^4.0.0" @@ -4702,6 +4866,19 @@ "node": ">=8" } }, + "node_modules/table-layout": { + "version": "4.1.1", + "resolved": "https://registry.npmjs.org/table-layout/-/table-layout-4.1.1.tgz", + "integrity": "sha512-iK5/YhZxq5GO5z8wb0bY1317uDF3Zjpha0QFFLA8/trAoiLbQD0HUbMesEaxyzUgDxi2QlcbM8IvqOlEjgoXBA==", + "license": "MIT", + "dependencies": { + "array-back": "^6.2.2", + "wordwrapjs": "^5.1.0" + }, + "engines": { + "node": ">=12.17" + } + }, "node_modules/tar": { "version": "7.5.13", "resolved": "https://registry.npmjs.org/tar/-/tar-7.5.13.tgz", @@ -5035,8 +5212,7 @@ "version": "2.8.1", "resolved": "https://registry.npmjs.org/tslib/-/tslib-2.8.1.tgz", "integrity": "sha512-oJFu94HQb+KVduSUQL7wnpmqnfmLsOA/nAh6b6EH0wCEoK0/mPeXU6c3wKDV83MkOuHPRHtSXKKU99IBazS/2w==", - "license": "0BSD", - "optional": true + "license": "0BSD" }, "node_modules/tsx": { "version": "4.22.4", @@ -5114,6 +5290,15 @@ "node": ">=14.17" } }, + "node_modules/typical": { + "version": "7.3.0", + "resolved": "https://registry.npmjs.org/typical/-/typical-7.3.0.tgz", + "integrity": "sha512-ya4mg/30vm+DOWfBg4YK3j2WD6TWtRkCbasOJr40CseYENzCUby/7rIvXA99JGsQHeNxLbnXdyLLxKSv3tauFw==", + "license": "MIT", + "engines": { + "node": ">=12.17" + } + }, "node_modules/undici-types": { "version": "7.24.6", "resolved": "https://registry.npmjs.org/undici-types/-/undici-types-7.24.6.tgz", @@ -5366,6 +5551,15 @@ "node": ">=8" } }, + "node_modules/wordwrapjs": { + "version": "5.1.1", + "resolved": "https://registry.npmjs.org/wordwrapjs/-/wordwrapjs-5.1.1.tgz", + "integrity": "sha512-0yweIbkINJodk27gX9LBGMzyQdBDan3s/dEAiwBOj+Mf0PPyWL6/rikalkv8EeD0E8jm4o5RXEOrFTP3NXbhJg==", + "license": "MIT", + "engines": { + "node": ">=12.17" + } + }, "node_modules/wrap-ansi": { "version": "7.0.0", "resolved": "https://registry.npmjs.org/wrap-ansi/-/wrap-ansi-7.0.0.tgz", diff --git a/gitnexus/package.json b/gitnexus/package.json index 53eb7dedd..cd065cef0 100644 --- a/gitnexus/package.json +++ b/gitnexus/package.json @@ -1,6 +1,6 @@ { "name": "gitnexus", - "version": "1.6.5", + "version": "1.6.6", "description": "Graph-powered code intelligence for AI agents. Index any codebase, query via MCP or CLI.", "author": "Abhigyan Patwari", "license": "PolyForm-Noncommercial-1.0.0", @@ -55,7 +55,7 @@ }, "dependencies": { "@huggingface/transformers": "^4.1.0", - "@ladybugdb/core": "^0.16.1", + "@ladybugdb/core": "^0.17.0", "@modelcontextprotocol/sdk": "^1.0.0", "@scarf/scarf": "^1.4.0", "cli-progress": "^3.12.0", @@ -70,8 +70,8 @@ "ignore": "^7.0.5", "js-yaml": "^4.1.1", "jsonc-parser": "^3.3.1", - "lru-cache": "^11.0.0", "mnemonist": "^0.40.3", + "onnxruntime-common": "^1.26.0", "onnxruntime-node": "^1.24.0", "pandemonium": "^2.4.0", "pino": "^10.3.1", diff --git a/gitnexus/scripts/spikes/s1-reaching-def-index-bench.ts b/gitnexus/scripts/spikes/s1-reaching-def-index-bench.ts new file mode 100644 index 000000000..98eb40bfc --- /dev/null +++ b/gitnexus/scripts/spikes/s1-reaching-def-index-bench.ts @@ -0,0 +1,151 @@ +/** + * Spike S1 (issue #2080, M0) — THROWAWAY benchmark. Not part of the build + * (scripts/ is excluded from tsconfig) or the test suite. + * + * Question: can LadybugDB serve the headline REACHING_DEF query + * [:REACHING_DEF*1..5 {variable}] + * fast enough, and what is the right storage shape for the `variable`? + * + * What it does: + * 1. Builds a synthetic ~100K-edge graph of BasicBlock nodes + REACHING_DEF + * edges (variable carried in the CodeRelation `reason` column) with a + * realistic per-variable fan-out distribution, and loads it through the + * real bulk-COPY path (loadGraphToLbug). + * 2. Probes whether LadybugDB supports a secondary index on a relationship + * property (the crux of the "edge property vs side table" decision). + * 3. Times the variable-filtered bounded var-length path query. + * + * Run: npx tsx scripts/spikes/s1-reaching-def-index-bench.ts [edgeCount] + */ +import fs from 'fs/promises'; +import path from 'path'; +import os from 'os'; +import { performance } from 'node:perf_hooks'; +import { createKnowledgeGraph } from '../../src/core/graph/graph.js'; +import type { KnowledgeGraph } from '../../src/core/graph/types.js'; + +const EDGE_COUNT = Number(process.argv[2] ?? 30_000); +// Realistic-ish def-use shape: many short chains, variables reused across them. +const CHAIN_LEN = 6; // blocks per function-ish chain +const DISTINCT_VARS = Math.max(1, Math.floor(EDGE_COUNT / 20)); // ~20 edges/variable fan-out + +const log = (m: string) => process.stdout.write(m + '\n'); + +function buildSynthGraph(edgeCount: number): KnowledgeGraph { + const g = createKnowledgeGraph(); + let edges = 0; + let chain = 0; + while (edges < edgeCount) { + const base = `BasicBlock:synth/f${chain}.ts`; + for (let i = 0; i <= CHAIN_LEN; i++) { + g.addNode({ + id: `${base}:${i}`, + label: 'BasicBlock', + properties: { + name: '', + filePath: `synth/f${chain}.ts`, + startLine: i, + endLine: i, + text: '', + }, + }); + } + for (let i = 0; i < CHAIN_LEN && edges < edgeCount; i++) { + const variable = `v${edges % DISTINCT_VARS}`; + g.addRelationship({ + id: `${base}:${i}->${i + 1}:${variable}`, + sourceId: `${base}:${i}`, + targetId: `${base}:${i + 1}`, + type: 'REACHING_DEF', + confidence: 1.0, + reason: variable, // M0 storage: variable rides `reason` + }); + edges++; + } + chain++; + } + return g; +} + +async function main() { + const tmp = path.join(os.tmpdir(), `s1-spike-${Date.now()}`); + const storagePath = path.join(tmp, '.gitnexus'); + const dbPath = path.join(storagePath, 'lbug'); + await fs.mkdir(dbPath, { recursive: true }); + + const adapter = await import('../../src/core/lbug/lbug-adapter.js'); + await adapter.initLbug(dbPath); + + log( + `[S1] building synthetic graph: ~${EDGE_COUNT} REACHING_DEF edges, ` + + `${DISTINCT_VARS} distinct variables (~20 edges/var fan-out), chains of ${CHAIN_LEN}`, + ); + const g = buildSynthGraph(EDGE_COUNT); + + let t = performance.now(); + await adapter.loadGraphToLbug(g, tmp, storagePath); + const loadMs = performance.now() - t; + const stats = await adapter.getLbugStats(); + log(`[S1] bulk-COPY load: ${loadMs.toFixed(0)}ms (nodes=${stats.nodes}, edges=${stats.edges})`); + + // (2) Probe: does LadybugDB support a secondary index on a REL property? + let relIndexSupported = false; + let relIndexErr = ''; + for (const stmt of [ + "CALL CREATE_REL_INDEX('CodeRelation', 'cr_reason_idx', 'reason')", + 'CREATE INDEX cr_reason_idx ON CodeRelation(reason)', + ]) { + try { + await adapter.executeQuery(stmt); + relIndexSupported = true; + break; + } catch (e: any) { + relIndexErr = String(e?.message ?? e).split('\n')[0]; + } + } + log( + `[S1] rel-property secondary index supported? ${relIndexSupported} ` + + `(last error: ${relIndexErr})`, + ); + + // (3a) Single-hop variable filter — the common case M3 runs most. + const probeVar = 'v0'; + t = performance.now(); + const single = await adapter.executeQuery( + `MATCH (a:BasicBlock)-[r:CodeRelation {type: 'REACHING_DEF', reason: '${probeVar}'}]->(b:BasicBlock) + RETURN count(r) AS c`, + ); + const singleMs = performance.now() - t; + log(`[S1] single-hop variable filter → ${single[0]?.c} edges in ${singleMs.toFixed(0)}ms`); + + // (3b) SOURCE-ANCHORED bounded var-length path — the realistic taint query + // (anchor the source block, then walk REACHING_DEF up to 5 hops). The + // UNANCHORED global form ([:REACHING_DEF*1..5] from every block) is + // impractical at scale (path explosion) — that is itself an S1 finding: + // taint queries MUST be scoped to a source block, not run graph-wide. + const srcId = 'BasicBlock:synth/f0.ts:0'; + t = performance.now(); + const anchored = await adapter.executeQuery( + `MATCH p = (a:BasicBlock)-[:CodeRelation*1..5 {type: 'REACHING_DEF'}]->(b:BasicBlock) + WHERE a.id = '${srcId}' AND all(rel IN relationships(p) WHERE rel.reason = '${probeVar}') + RETURN count(p) AS paths`, + ); + const pathMs = performance.now() - t; + log( + `[S1] source-anchored [:REACHING_DEF*1..5 {reason='${probeVar}'}] from one block → ` + + `${anchored[0]?.paths} paths in ${pathMs.toFixed(0)}ms`, + ); + + await adapter.closeLbug(); + await fs.rm(tmp, { recursive: true, force: true }); + + log('\n[S1] VERDICT INPUTS:'); + log( + ` load_ms=${loadMs.toFixed(0)} single_hop_ms=${singleMs.toFixed(0)} anchored_path_ms=${pathMs.toFixed(0)} rel_index=${relIndexSupported}`, + ); +} + +main().catch((e) => { + console.error('[S1] FAILED:', e); + process.exit(1); +}); diff --git a/gitnexus/scripts/spikes/s2-postdom-prototype.ts b/gitnexus/scripts/spikes/s2-postdom-prototype.ts new file mode 100644 index 000000000..da06d8e7e --- /dev/null +++ b/gitnexus/scripts/spikes/s2-postdom-prototype.ts @@ -0,0 +1,162 @@ +/** + * Spike S2 (issue #2080, M0) — THROWAWAY post-dominator feasibility prototype. + * Not part of the build (scripts/ excluded from tsconfig) or the test suite. + * + * Question (per maintainer review): does the post-dominator algorithm Epic B + * (#2085, CDG) depends on hold up on real TS/JS control-flow shapes — the + * classic CFG hazards — before Epic B commits to it? + * + * Scope boundary: post-dominators operate on a CFG, not on the AST directly. + * This prototype validates the ALGORITHM (iterative dataflow on the reverse + * CFG, EXIT-rooted, → immediate-post-dominator tree) against CFGs that model + * each hazard's real TS control flow (the TS source each CFG represents is + * shown inline). Building the CFG from a tree-sitter AST is M1's job (#2081); + * this spike deliberately does not reimplement it. + * + * Run: npx tsx scripts/spikes/s2-postdom-prototype.ts + */ + +type CFG = { + name: string; + tsSource: string; + entry: string; + exit: string; + // adjacency: block -> successors + succ: Record; + hazard: string; +}; + +// Iterative post-dominator dataflow on the reverse CFG. +// PostDom(EXIT) = {EXIT}; PostDom(n) = {n} ∪ (⋂ PostDom(s) for s ∈ succ(n)). +// Monotone over a finite lattice (powerset of blocks) ⇒ guaranteed to converge. +function postDominators(cfg: CFG): { pdom: Record>; iterations: number } { + const blocks = Object.keys(cfg.succ); + const all = new Set(blocks); + const pdom: Record> = {}; + for (const b of blocks) pdom[b] = b === cfg.exit ? new Set([cfg.exit]) : new Set(all); + + let changed = true; + let iterations = 0; + while (changed) { + changed = false; + iterations++; + for (const b of blocks) { + if (b === cfg.exit) continue; + const succs = cfg.succ[b] ?? []; + let inter: Set | null = null; + for (const s of succs) { + if (inter === null) inter = new Set(pdom[s]); + else inter = new Set([...inter].filter((x) => pdom[s].has(x))); + } + const next = new Set(inter ?? []); + next.add(b); + if (next.size !== pdom[b].size || [...next].some((x) => !pdom[b].has(x))) { + pdom[b] = next; + changed = true; + } + } + if (iterations > blocks.length + 5) + throw new Error('post-dom did not converge (suspected bug)'); + } + return { pdom, iterations }; +} + +// Immediate post-dominator: the closest strict post-dominator. +function ipdom(cfg: CFG, pdom: Record>): Record { + const res: Record = {}; + for (const b of Object.keys(cfg.succ)) { + if (b === cfg.exit) { + res[b] = null; + continue; + } + const strict = [...pdom[b]].filter((x) => x !== b); + // ipdom = the strict post-dom that does not post-dominate any other strict post-dom. + res[b] = + strict.find((cand) => strict.every((other) => other === cand || !pdom[other].has(cand))) ?? + null; + } + return res; +} + +const CFGS: CFG[] = [ + { + name: 'early-return', + hazard: 'early return / multiple paths to EXIT', + tsSource: `function f(x){ if (x) { return 1; } g(); return 2; }`, + entry: 'ENTRY', + exit: 'EXIT', + succ: { ENTRY: ['ret1', 'g'], ret1: ['EXIT'], g: ['ret2'], ret2: ['EXIT'], EXIT: [] }, + }, + { + name: 'try-throw-finally', + hazard: 'try/throw/finally with multiple exits through finally', + tsSource: `function f(){ try { risky(); } catch(e){ handle(e); } finally { cleanup(); } done(); }`, + entry: 'ENTRY', + exit: 'EXIT', + // try → (normal | throw→catch) → finally → done → EXIT; finally also reached on rethrow + succ: { + ENTRY: ['try'], + try: ['finally', 'catch'], + catch: ['finally'], + finally: ['done', 'EXIT'], + done: ['EXIT'], + EXIT: [], + }, + }, + { + name: 'labeled-break', + hazard: 'labeled break/continue across nested loops', + tsSource: `outer: for(;;){ for(;;){ if (a) break outer; if (b) continue outer; work(); } }`, + entry: 'ENTRY', + exit: 'EXIT', + succ: { + ENTRY: ['outerHead'], + outerHead: ['innerHead', 'EXIT'], + innerHead: ['breakOuter', 'afterIf1'], + breakOuter: ['EXIT'], + afterIf1: ['contOuter', 'work'], + contOuter: ['outerHead'], + work: ['innerHead'], + EXIT: [], + }, + }, + { + name: 'if-else-diamond', + hazard: 'baseline reducible diamond (sanity)', + tsSource: `function f(x){ if (x) { a(); } else { b(); } c(); }`, + entry: 'ENTRY', + exit: 'EXIT', + succ: { ENTRY: ['a', 'b'], a: ['c'], b: ['c'], c: ['EXIT'], EXIT: [] }, + }, +]; + +function main() { + let allOk = true; + for (const cfg of CFGS) { + try { + const { pdom, iterations } = postDominators(cfg); + const idom = ipdom(cfg, pdom); + // Sanity invariants: EXIT post-dominates every block; ipdom tree reaches EXIT. + const exitPostDomsAll = Object.keys(cfg.succ).every((b) => pdom[b].has(cfg.exit)); + console.log(`\n[S2] ${cfg.name} — ${cfg.hazard}`); + console.log(` TS: ${cfg.tsSource}`); + console.log( + ` converged in ${iterations} iters; EXIT post-dominates all blocks: ${exitPostDomsAll}`, + ); + console.log( + ` ipdom tree: ${Object.entries(idom) + .map(([b, p]) => `${b}->${p ?? '∅'}`) + .join(' ')}`, + ); + if (!exitPostDomsAll) allOk = false; + } catch (e) { + allOk = false; + console.log(`\n[S2] ${cfg.name} FAILED: ${(e as Error).message}`); + } + } + console.log( + `\n[S2] VERDICT INPUT: all hazard CFGs converged + EXIT post-dominates all = ${allOk}`, + ); +} + +main(); diff --git a/gitnexus/shadow-parity-dashboard/index.html b/gitnexus/shadow-parity-dashboard/index.html deleted file mode 100644 index 104d7b026..000000000 --- a/gitnexus/shadow-parity-dashboard/index.html +++ /dev/null @@ -1,291 +0,0 @@ - - - - - - GitNexus — Shadow Parity Dashboard - - - - -
-

Shadow Parity — RFC #909

-
loading latest.json…
-
- - - - - - - - - - - - - - -
LanguageTotalAgreeOnly legacyOnly newDisagreeBoth emptyParity
- -
- - - diff --git a/gitnexus/skills/gitnexus-debugging.md b/gitnexus/skills/gitnexus-debugging.md index 937b5e2a4..9834f94b7 100644 --- a/gitnexus/skills/gitnexus-debugging.md +++ b/gitnexus/skills/gitnexus-debugging.md @@ -16,10 +16,10 @@ description: "Use when the user is debugging a bug, tracing an error, or asking ## Workflow ``` -1. gitnexus_query({query: ""}) → Find related execution flows -2. gitnexus_context({name: ""}) → See callers/callees/processes +1. query({query: ""}) → Find related execution flows +2. context({name: ""}) → See callers/callees/processes 3. READ gitnexus://repo/{name}/process/{name} → Trace execution flow -4. gitnexus_cypher({query: "MATCH path..."}) → Custom traces if needed +4. cypher({query: "MATCH path..."}) → Custom traces if needed ``` > If "Index is stale" → run `node .gitnexus/run.cjs analyze` in terminal. @@ -28,11 +28,11 @@ description: "Use when the user is debugging a bug, tracing an error, or asking ``` - [ ] Understand the symptom (error message, unexpected behavior) -- [ ] gitnexus_query for error text or related code +- [ ] query for error text or related code - [ ] Identify the suspect function from returned processes -- [ ] gitnexus_context to see callers and callees +- [ ] context to see callers and callees - [ ] Trace execution flow via process resource if applicable -- [ ] gitnexus_cypher for custom call chain traces if needed +- [ ] cypher for custom call chain traces if needed - [ ] Read source files to confirm root cause ``` @@ -40,7 +40,7 @@ description: "Use when the user is debugging a bug, tracing an error, or asking | Symptom | GitNexus Approach | | -------------------- | ---------------------------------------------------------- | -| Error message | `gitnexus_query` for error text → `context` on throw sites | +| Error message | `query` for error text → `context` on throw sites | | Wrong return value | `context` on the function → trace callees for data flow | | Intermittent failure | `context` → look for external calls, async deps | | Performance issue | `context` → find symbols with many callers (hot paths) | @@ -48,24 +48,24 @@ description: "Use when the user is debugging a bug, tracing an error, or asking ## Tools -**gitnexus_query** — find code related to error: +**query** — find code related to error: ``` -gitnexus_query({query: "payment validation error"}) +query({query: "payment validation error"}) → Processes: CheckoutFlow, ErrorHandling → Symbols: validatePayment, handlePaymentError, PaymentException ``` -**gitnexus_context** — full context for a suspect: +**context** — full context for a suspect: ``` -gitnexus_context({name: "validatePayment"}) +context({name: "validatePayment"}) → Incoming calls: processCheckout, webhookHandler → Outgoing calls: verifyCard, fetchRates (external API!) → Processes: CheckoutFlow (step 3/7) ``` -**gitnexus_cypher** — custom call chain traces: +**cypher** — custom call chain traces: ```cypher MATCH path = (a)-[:CodeRelation {type: 'CALLS'}*1..2]->(b:Function {name: "validatePayment"}) @@ -75,11 +75,11 @@ RETURN [n IN nodes(path) | n.name] AS chain ## Example: "Payment endpoint returns 500 intermittently" ``` -1. gitnexus_query({query: "payment error handling"}) +1. query({query: "payment error handling"}) → Processes: CheckoutFlow, ErrorHandling → Symbols: validatePayment, handlePaymentError -2. gitnexus_context({name: "validatePayment"}) +2. context({name: "validatePayment"}) → Outgoing calls: verifyCard, fetchRates (external API!) 3. READ gitnexus://repo/my-app/process/CheckoutFlow diff --git a/gitnexus/skills/gitnexus-exploring.md b/gitnexus/skills/gitnexus-exploring.md index 2dcf7b578..ccf684c28 100644 --- a/gitnexus/skills/gitnexus-exploring.md +++ b/gitnexus/skills/gitnexus-exploring.md @@ -18,8 +18,8 @@ description: "Use when the user asks how code works, wants to understand archite ``` 1. READ gitnexus://repos → Discover indexed repos 2. READ gitnexus://repo/{name}/context → Codebase overview, check staleness -3. gitnexus_query({query: ""}) → Find related execution flows -4. gitnexus_context({name: ""}) → Deep dive on specific symbol +3. query({query: ""}) → Find related execution flows +4. context({name: ""}) → Deep dive on specific symbol 5. READ gitnexus://repo/{name}/process/{name} → Trace full execution flow ``` @@ -29,9 +29,9 @@ description: "Use when the user asks how code works, wants to understand archite ``` - [ ] READ gitnexus://repo/{name}/context -- [ ] gitnexus_query for the concept you want to understand +- [ ] query for the concept you want to understand - [ ] Review returned processes (execution flows) -- [ ] gitnexus_context on key symbols for callers/callees +- [ ] context on key symbols for callers/callees - [ ] READ process resource for full execution traces - [ ] Read source files for implementation details ``` @@ -47,18 +47,18 @@ description: "Use when the user asks how code works, wants to understand archite ## Tools -**gitnexus_query** — find execution flows related to a concept: +**query** — find execution flows related to a concept: ``` -gitnexus_query({query: "payment processing"}) +query({query: "payment processing"}) → Processes: CheckoutFlow, RefundFlow, WebhookHandler → Symbols grouped by flow with file locations ``` -**gitnexus_context** — 360-degree view of a symbol: +**context** — 360-degree view of a symbol: ``` -gitnexus_context({name: "validateUser"}) +context({name: "validateUser"}) → Incoming calls: loginHandler, apiMiddleware → Outgoing calls: checkToken, getUserById → Processes: LoginFlow (step 2/5), TokenRefresh (step 1/3) @@ -68,10 +68,10 @@ gitnexus_context({name: "validateUser"}) ``` 1. READ gitnexus://repo/my-app/context → 918 symbols, 45 processes -2. gitnexus_query({query: "payment processing"}) +2. query({query: "payment processing"}) → CheckoutFlow: processPayment → validateCard → chargeStripe → RefundFlow: initiateRefund → calculateRefund → processRefund -3. gitnexus_context({name: "processPayment"}) +3. context({name: "processPayment"}) → Incoming: checkoutHandler, webhookHandler → Outgoing: validateCard, chargeStripe, saveTransaction 4. Read src/payments/processor.ts for implementation details diff --git a/gitnexus/skills/gitnexus-impact-analysis.md b/gitnexus/skills/gitnexus-impact-analysis.md index 7206ca506..45eb7ce87 100644 --- a/gitnexus/skills/gitnexus-impact-analysis.md +++ b/gitnexus/skills/gitnexus-impact-analysis.md @@ -17,9 +17,9 @@ description: "Use when the user wants to know what will break if they change som ## Workflow ``` -1. gitnexus_impact({target: "X", direction: "upstream"}) → What depends on this +1. impact({target: "X", direction: "upstream"}) → What depends on this 2. READ gitnexus://repo/{name}/processes → Check affected execution flows -3. gitnexus_detect_changes() → Map current git changes to affected flows +3. detect_changes() → Map current git changes to affected flows 4. Assess risk and report to user ``` @@ -28,11 +28,11 @@ description: "Use when the user wants to know what will break if they change som ## Checklist ``` -- [ ] gitnexus_impact({target, direction: "upstream"}) to find dependents +- [ ] impact({target, direction: "upstream"}) to find dependents - [ ] Review d=1 items first (these WILL BREAK) - [ ] Check high-confidence (>0.8) dependencies - [ ] READ processes to check affected execution flows -- [ ] gitnexus_detect_changes() for pre-commit check +- [ ] detect_changes() for pre-commit check - [ ] Assess risk level and report to user ``` @@ -55,10 +55,10 @@ description: "Use when the user wants to know what will break if they change som ## Tools -**gitnexus_impact** — the primary tool for symbol blast radius: +**impact** — the primary tool for symbol blast radius: ``` -gitnexus_impact({ +impact({ target: "validateUser", direction: "upstream", minConfidence: 0.8, @@ -73,10 +73,10 @@ gitnexus_impact({ - authRouter (src/routes/auth.ts:22) [CALLS, 95%] ``` -**gitnexus_detect_changes** — git-diff based impact analysis: +**detect_changes** — git-diff based impact analysis: ``` -gitnexus_detect_changes({scope: "staged"}) +detect_changes({scope: "staged"}) → Changed: 5 symbols in 3 files → Affected: LoginFlow, TokenRefresh, APIMiddlewarePipeline @@ -86,7 +86,7 @@ gitnexus_detect_changes({scope: "staged"}) ## Example: "What breaks if I change validateUser?" ``` -1. gitnexus_impact({target: "validateUser", direction: "upstream"}) +1. impact({target: "validateUser", direction: "upstream"}) → d=1: loginHandler, apiMiddleware (WILL BREAK) → d=2: authRouter, sessionManager (LIKELY AFFECTED) diff --git a/gitnexus/skills/gitnexus-pr-review.md b/gitnexus/skills/gitnexus-pr-review.md index 319c063f9..9f1d362e5 100644 --- a/gitnexus/skills/gitnexus-pr-review.md +++ b/gitnexus/skills/gitnexus-pr-review.md @@ -18,10 +18,10 @@ description: "Use when the user wants to review a pull request, understand what ``` 1. gh pr diff → Get the raw diff -2. gitnexus_detect_changes({scope: "compare", base_ref: "main"}) → Map diff to affected flows +2. detect_changes({scope: "compare", base_ref: "main"}) → Map diff to affected flows 3. For each changed symbol: - gitnexus_impact({target: "", direction: "upstream"}) → Blast radius per change -4. gitnexus_context({name: ""}) → Understand callers/callees + impact({target: "", direction: "upstream"}) → Blast radius per change +4. context({name: ""}) → Understand callers/callees 5. READ gitnexus://repo/{name}/processes → Check affected execution flows 6. Summarize findings with risk assessment ``` @@ -32,10 +32,10 @@ description: "Use when the user wants to review a pull request, understand what ``` - [ ] Fetch PR diff (gh pr diff or git diff base...head) -- [ ] gitnexus_detect_changes to map changes to affected execution flows -- [ ] gitnexus_impact on each non-trivial changed symbol +- [ ] detect_changes to map changes to affected execution flows +- [ ] impact on each non-trivial changed symbol - [ ] Review d=1 items (WILL BREAK) — are callers updated? -- [ ] gitnexus_context on key changed symbols to understand full picture +- [ ] context on key changed symbols to understand full picture - [ ] Check if affected processes have test coverage - [ ] Assess overall risk level - [ ] Write review summary with findings @@ -63,20 +63,20 @@ description: "Use when the user wants to review a pull request, understand what ## Tools -**gitnexus_detect_changes** — map PR diff to affected execution flows: +**detect_changes** — map PR diff to affected execution flows: ``` -gitnexus_detect_changes({scope: "compare", base_ref: "main"}) +detect_changes({scope: "compare", base_ref: "main"}) → Changed: 8 symbols in 4 files → Affected processes: CheckoutFlow, RefundFlow, WebhookHandler → Risk: MEDIUM ``` -**gitnexus_impact** — blast radius per changed symbol: +**impact** — blast radius per changed symbol: ``` -gitnexus_impact({target: "validatePayment", direction: "upstream"}) +impact({target: "validatePayment", direction: "upstream"}) → d=1 (WILL BREAK): - processCheckout (src/checkout.ts:42) [CALLS, 100%] @@ -86,20 +86,20 @@ gitnexus_impact({target: "validatePayment", direction: "upstream"}) - checkoutRouter (src/routes/checkout.ts:22) [CALLS, 95%] ``` -**gitnexus_impact with tests** — check test coverage: +**impact with tests** — check test coverage: ``` -gitnexus_impact({target: "validatePayment", direction: "upstream", includeTests: true}) +impact({target: "validatePayment", direction: "upstream", includeTests: true}) → Tests that cover this symbol: - validatePayment.test.ts [direct] - checkout.integration.test.ts [via processCheckout] ``` -**gitnexus_context** — understand a changed symbol's role: +**context** — understand a changed symbol's role: ``` -gitnexus_context({name: "validatePayment"}) +context({name: "validatePayment"}) → Incoming calls: processCheckout, webhookHandler → Outgoing calls: verifyCard, fetchRates @@ -112,20 +112,20 @@ gitnexus_context({name: "validatePayment"}) 1. gh pr diff 42 > /tmp/pr42.diff → 4 files changed: payments.ts, checkout.ts, types.ts, utils.ts -2. gitnexus_detect_changes({scope: "compare", base_ref: "main"}) +2. detect_changes({scope: "compare", base_ref: "main"}) → Changed symbols: validatePayment, PaymentInput, formatAmount → Affected processes: CheckoutFlow, RefundFlow → Risk: MEDIUM -3. gitnexus_impact({target: "validatePayment", direction: "upstream"}) +3. impact({target: "validatePayment", direction: "upstream"}) → d=1: processCheckout, webhookHandler (WILL BREAK) → webhookHandler is NOT in the PR diff — potential breakage! -4. gitnexus_impact({target: "PaymentInput", direction: "upstream"}) +4. impact({target: "PaymentInput", direction: "upstream"}) → d=1: validatePayment (in PR), createPayment (NOT in PR) → createPayment uses the old PaymentInput shape — breaking change! -5. gitnexus_context({name: "formatAmount"}) +5. context({name: "formatAmount"}) → Called by 12 functions — but change is backwards-compatible (added optional param) 6. Review summary: diff --git a/gitnexus/skills/gitnexus-refactoring.md b/gitnexus/skills/gitnexus-refactoring.md index c749eb384..e13c04e14 100644 --- a/gitnexus/skills/gitnexus-refactoring.md +++ b/gitnexus/skills/gitnexus-refactoring.md @@ -16,9 +16,9 @@ description: "Use when the user wants to rename, extract, split, move, or restru ## Workflow ``` -1. gitnexus_impact({target: "X", direction: "upstream"}) → Map all dependents -2. gitnexus_query({query: "X"}) → Find execution flows involving X -3. gitnexus_context({name: "X"}) → See all incoming/outgoing refs +1. impact({target: "X", direction: "upstream"}) → Map all dependents +2. query({query: "X"}) → Find execution flows involving X +3. context({name: "X"}) → See all incoming/outgoing refs 4. Plan update order: interfaces → implementations → callers → tests ``` @@ -29,65 +29,65 @@ description: "Use when the user wants to rename, extract, split, move, or restru ### Rename Symbol ``` -- [ ] gitnexus_rename({symbol_name: "oldName", new_name: "newName", dry_run: true}) — preview all edits +- [ ] rename({symbol_name: "oldName", new_name: "newName", dry_run: true}) — preview all edits - [ ] Review graph edits (high confidence) and ast_search edits (review carefully) -- [ ] If satisfied: gitnexus_rename({..., dry_run: false}) — apply edits -- [ ] gitnexus_detect_changes() — verify only expected files changed +- [ ] If satisfied: rename({..., dry_run: false}) — apply edits +- [ ] detect_changes() — verify only expected files changed - [ ] Run tests for affected processes ``` ### Extract Module ``` -- [ ] gitnexus_context({name: target}) — see all incoming/outgoing refs -- [ ] gitnexus_impact({target, direction: "upstream"}) — find all external callers +- [ ] context({name: target}) — see all incoming/outgoing refs +- [ ] impact({target, direction: "upstream"}) — find all external callers - [ ] Define new module interface - [ ] Extract code, update imports -- [ ] gitnexus_detect_changes() — verify affected scope +- [ ] detect_changes() — verify affected scope - [ ] Run tests for affected processes ``` ### Split Function/Service ``` -- [ ] gitnexus_context({name: target}) — understand all callees +- [ ] context({name: target}) — understand all callees - [ ] Group callees by responsibility -- [ ] gitnexus_impact({target, direction: "upstream"}) — map callers to update +- [ ] impact({target, direction: "upstream"}) — map callers to update - [ ] Create new functions/services - [ ] Update callers -- [ ] gitnexus_detect_changes() — verify affected scope +- [ ] detect_changes() — verify affected scope - [ ] Run tests for affected processes ``` ## Tools -**gitnexus_rename** — automated multi-file rename: +**rename** — automated multi-file rename: ``` -gitnexus_rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: true}) +rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: true}) → 12 edits across 8 files → 10 graph edits (high confidence), 2 ast_search edits (review) → Changes: [{file_path, edits: [{line, old_text, new_text, confidence}]}] ``` -**gitnexus_impact** — map all dependents first: +**impact** — map all dependents first: ``` -gitnexus_impact({target: "validateUser", direction: "upstream"}) +impact({target: "validateUser", direction: "upstream"}) → d=1: loginHandler, apiMiddleware, testUtils → Affected Processes: LoginFlow, TokenRefresh ``` -**gitnexus_detect_changes** — verify your changes after refactoring: +**detect_changes** — verify your changes after refactoring: ``` -gitnexus_detect_changes({scope: "all"}) +detect_changes({scope: "all"}) → Changed: 8 files, 12 symbols → Affected processes: LoginFlow, TokenRefresh → Risk: MEDIUM ``` -**gitnexus_cypher** — custom reference queries: +**cypher** — custom reference queries: ```cypher MATCH (caller)-[:CodeRelation {type: 'CALLS'}]->(f:Function {name: "validateUser"}) @@ -98,24 +98,24 @@ RETURN caller.name, caller.filePath ORDER BY caller.filePath | Risk Factor | Mitigation | | ------------------- | ----------------------------------------- | -| Many callers (>5) | Use gitnexus_rename for automated updates | +| Many callers (>5) | Use rename for automated updates | | Cross-area refs | Use detect_changes after to verify scope | -| String/dynamic refs | gitnexus_query to find them | +| String/dynamic refs | query to find them | | External/public API | Version and deprecate properly | ## Example: Rename `validateUser` to `authenticateUser` ``` -1. gitnexus_rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: true}) +1. rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: true}) → 12 edits: 10 graph (safe), 2 ast_search (review) → Files: validator.ts, login.ts, middleware.ts, config.json... 2. Review ast_search edits (config.json: dynamic reference!) -3. gitnexus_rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: false}) +3. rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: false}) → Applied 12 edits across 8 files -4. gitnexus_detect_changes({scope: "all"}) +4. detect_changes({scope: "all"}) → Affected: LoginFlow, TokenRefresh → Risk: MEDIUM — run tests for these flows ``` diff --git a/gitnexus/src/cli/ai-context.ts b/gitnexus/src/cli/ai-context.ts index 6470c6430..641e7ee94 100644 --- a/gitnexus/src/cli/ai-context.ts +++ b/gitnexus/src/cli/ai-context.ts @@ -174,18 +174,18 @@ This project is indexed by GitNexus as **${projectName}**${noStats ? '' : ` (${s ## Always Do -- **MUST run impact analysis before editing any symbol.** Before modifying a function, class, or method, run \`gitnexus_impact({target: "symbolName", direction: "upstream"})\` and report the blast radius (direct callers, affected processes, risk level) to the user. -- **MUST run \`gitnexus_detect_changes()\` before committing** to verify your changes only affect expected symbols and execution flows. For regression review, compare against the default branch: \`gitnexus_detect_changes({scope: "compare", base_ref: ${JSON.stringify(markdownSafeBranch(defaultBranch))}})\`. +- **MUST run impact analysis before editing any symbol.** Before modifying a function, class, or method, run \`impact({target: "symbolName", direction: "upstream"})\` and report the blast radius (direct callers, affected processes, risk level) to the user. +- **MUST run \`detect_changes()\` before committing** to verify your changes only affect expected symbols and execution flows. For regression review, compare against the default branch: \`detect_changes({scope: "compare", base_ref: ${JSON.stringify(markdownSafeBranch(defaultBranch))}})\`. - **MUST warn the user** if impact analysis returns HIGH or CRITICAL risk before proceeding with edits. -- When exploring unfamiliar code, use \`gitnexus_query({query: "concept"})\` to find execution flows instead of grepping. It returns process-grouped results ranked by relevance. -- When you need full context on a specific symbol — callers, callees, which execution flows it participates in — use \`gitnexus_context({name: "symbolName"})\`. +- When exploring unfamiliar code, use \`query({query: "concept"})\` to find execution flows instead of grepping. It returns process-grouped results ranked by relevance. +- When you need full context on a specific symbol — callers, callees, which execution flows it participates in — use \`context({name: "symbolName"})\`. ## Never Do -- NEVER edit a function, class, or method without first running \`gitnexus_impact\` on it. +- NEVER edit a function, class, or method without first running \`impact\` on it. - NEVER ignore HIGH or CRITICAL risk warnings from impact analysis. -- NEVER rename symbols with find-and-replace — use \`gitnexus_rename\` which understands the call graph. -- NEVER commit changes without running \`gitnexus_detect_changes()\` to check affected scope. +- NEVER rename symbols with find-and-replace — use \`rename\` which understands the call graph. +- NEVER commit changes without running \`detect_changes()\` to check affected scope. ## Resources diff --git a/gitnexus/src/cli/analyze.ts b/gitnexus/src/cli/analyze.ts index 6801f3f77..a3bf6cd5a 100644 --- a/gitnexus/src/cli/analyze.ts +++ b/gitnexus/src/cli/analyze.ts @@ -9,6 +9,7 @@ */ import path from 'path'; +import os from 'os'; import { spawn } from 'child_process'; import v8 from 'v8'; import cliProgress from 'cli-progress'; @@ -36,7 +37,7 @@ import { } from './analyze-config.js'; import { runFullAnalysis } from '../core/run-analyze.js'; import { getMaxFileSizeBannerMessage } from '../core/ingestion/utils/max-file-size.js'; -import { warnMissingOptionalGrammars } from './optional-grammars.js'; +import { warnMissingOptionalGrammars, getOptionalGrammarExtensions } from './optional-grammars.js'; import { glob } from 'glob'; import fs from 'fs/promises'; import { cliError } from './cli-message.js'; @@ -59,6 +60,22 @@ const writeFatalToStderr = (label: string, err: unknown): void => { const message = isErr ? err.message : String(err); realStderrWrite(`\n ${label}: ${message}\n`); if (isErr && err.stack) realStderrWrite(`${err.stack}\n`); + // Walk and print the `cause` chain. The phase runner wraps the underlying + // failure as `new Error("Phase 'X' failed: …", { cause })`, so the original + // error (e.g. a WorkerPoolDispatchError carrying the worker-side stack from + // #2068) is only reachable via `.cause`. Without this the user sees the + // wrapper's main-thread stack and never the real frame. `cause.stack` already + // begins with the cause's message, so we print the stack alone (not message + + // stack) to avoid repeating it. Depth-bounded so a cyclic `cause` can't loop + // (the phase runner wraps one level; the bound leaves headroom for future + // nesting); uses realStderrWrite so the redirected console.error's ANSI + // clear-line wrapping can't erase it (#1169). + const MAX_CAUSE_DEPTH = 5; + let cause: unknown = isErr ? (err as { cause?: unknown }).cause : undefined; + for (let depth = 0; depth < MAX_CAUSE_DEPTH && cause instanceof Error; depth++) { + realStderrWrite(`\n Caused by: ${cause.stack ?? cause.message}\n`); + cause = (cause as { cause?: unknown }).cause; + } }; let fatalHandlersInstalled = false; @@ -84,13 +101,45 @@ const installFatalHandlers = (): void => { }); }; -const HEAP_MB = 16384; +/** Historical floor for the re-exec heap cap — the auto-sizer never goes below + * this, so small boxes / CI never regress. */ +const DEFAULT_HEAP_MB = 16384; + +/** + * RAM-aware re-exec heap cap (MB): `0.75 × effective RAM`, clamped to + * `>= DEFAULT_HEAP_MB`. Kept BELOW physical RAM on purpose — a cap `>=` RAM makes + * V8 collect lazily and inflate the heap into swap-thrash (observed analyzing the + * Linux kernel at a 30GB cap on a 31GB box). `constrainedBytes` is the cgroup + * limit or `null`; it is honored only as a real, smaller-than-physical cap, because + * `process.constrainedMemory()` returns a huge sentinel when UNCONSTRAINED. + */ +export function computeHeapCapMb(totalBytes: number, constrainedBytes: number | null): number { + const effectiveBytes = + constrainedBytes !== null && constrainedBytes > 0 && constrainedBytes < totalBytes + ? constrainedBytes + : totalBytes; + const effectiveMb = Math.floor(effectiveBytes / (1024 * 1024)); + return Math.max(DEFAULT_HEAP_MB, Math.floor(0.75 * effectiveMb)); +} + +function readConstrainedBytes(): number | null { + if (typeof process.constrainedMemory !== 'function') return null; + const c = process.constrainedMemory(); + return typeof c === 'number' && c > 0 ? c : null; +} + +const HEAP_MB = computeHeapCapMb(os.totalmem(), readConstrainedBytes()); const TEST_RESPAWN_HEAP_MB = Number(process.env.GITNEXUS_TEST_RESPAWN_HEAP_MB); const RESPAWN_HEAP_MB = Number.isFinite(TEST_RESPAWN_HEAP_MB) && TEST_RESPAWN_HEAP_MB > 0 ? Math.floor(TEST_RESPAWN_HEAP_MB) : HEAP_MB; const HEAP_FLAG = `--max-old-space-size=${RESPAWN_HEAP_MB}`; +/** Larger semi-space (young-gen) cuts minor-GC frequency + promotion churn during + * the multi-million-node graph build/emit. Allowed in NODE_OPTIONS (unlike + * --stack-size), so it propagates to the re-exec env cleanly. */ +const SEMI_SPACE_MB = 128; +const SEMI_FLAG = `--max-semi-space-size=${SEMI_SPACE_MB}`; /** Increase default stack size (KB) to prevent stack overflow on deep class hierarchies. */ const STACK_KB = 4096; const STACK_FLAG = `--stack-size=${STACK_KB}`; @@ -440,7 +489,8 @@ const forceHeapOOMForTestIfEnabled = (): void => { // `gitnexus/src/core/lbug/lbug-config.ts` in sync with this value. const RECOMMENDED_WAL_CHECKPOINT_THRESHOLD = 64 * 1024 * 1024; -/** Re-exec the process with a 16GB heap and larger stack if we're currently below that. */ +/** Re-exec the process with the RAM-aware auto heap cap + larger semi-space/stack + * if we're currently below that. A user-supplied NODE_OPTIONS heap wins (no re-exec). */ async function ensureHeap(): Promise { const nodeOpts = process.env.NODE_OPTIONS || ''; if (nodeOpts.includes('--max-old-space-size')) return false; @@ -448,25 +498,26 @@ async function ensureHeap(): Promise { const v8Heap = v8.getHeapStatistics().heap_size_limit; if (v8Heap >= HEAP_MB * 1024 * 1024 * 0.9) return false; - // --stack-size is a V8 flag not allowed in NODE_OPTIONS on Node 24+, - // so pass it only as a direct CLI argument, not via the environment. - const cliFlags = [HEAP_FLAG]; + // --stack-size is a V8 flag not allowed in NODE_OPTIONS on Node 24+, so pass it + // only as a direct CLI argument. --max-semi-space-size IS allowed in NODE_OPTIONS. + const cliFlags = [HEAP_FLAG, SEMI_FLAG]; if (!nodeOpts.includes('--stack-size')) cliFlags.push(STACK_FLAG); const childArgs = [...cliFlags, ...process.argv.slice(1)]; const childEnv = { ...process.env, - NODE_OPTIONS: `${nodeOpts} ${HEAP_FLAG}`.trim(), + NODE_OPTIONS: `${nodeOpts} ${HEAP_FLAG} ${SEMI_FLAG}`.trim(), }; if (shouldBridgeRespawnProgressTty()) childEnv[RESPAWN_PROGRESS_ENV] = '1'; const childExit = await runRespawnedAnalyze(childArgs, childEnv); if (childExit.status !== 0 || childExit.signal) { if (childProcessLikelyOom(childExit)) { cliError( - ` Analysis likely ran out of memory.\n` + - ` Retry with a larger heap if your machine allows it:\n` + - ` NODE_OPTIONS="--max-old-space-size=24576" gitnexus analyze [your-args]\n` + - ` (Windows: set NODE_OPTIONS=--max-old-space-size=24576 && gitnexus analyze [your-args])\n` + + ` Analysis likely ran out of memory (heap cap auto-sized to ${RESPAWN_HEAP_MB}MB ≈ 0.75x RAM).\n` + + ` This repository's working set exceeds available RAM. Use a machine with more RAM,\n` + + ` or override the cap (a cap above physical RAM causes swap-thrash — use with care):\n` + + ` NODE_OPTIONS="--max-old-space-size=" gitnexus analyze [your-args]\n` + + ` (Windows: set NODE_OPTIONS=--max-old-space-size= && gitnexus analyze [your-args])\n` + ` If this persists, it may be a native crash unrelated to heap size.\n`, { recoveryHint: 'heap-oom-respawn' }, ); @@ -474,8 +525,7 @@ async function ensureHeap(): Promise { cliError( ` Analysis aborted in a native worker or native binding path.\n` + ` Try one of these recovery paths:\n` + - ` gitnexus analyze --workers 0\n` + - ` npm uninstall -g gitnexus && npm install -g gitnexus@latest\n` + + ` npm uninstall -g gitnexus && npm install -g gitnexus@latest (rebuilds native bindings)\n` + ` Use Node 22 LTS if you are on a newer non-LTS runtime.\n`, { recoveryHint: 'native-worker-abort' }, ); @@ -500,6 +550,7 @@ const ANALYZE_CLI_ENV_KEYS = [ 'GITNEXUS_VERBOSE', 'GITNEXUS_PROFILE_DEFERRED', 'GITNEXUS_PROFILE_DEFERRED_SLOW_MS', + 'GITNEXUS_DEBUG_HEAP', 'GITNEXUS_MAX_FILE_SIZE', 'GITNEXUS_WORKER_SUB_BATCH_TIMEOUT_MS', 'GITNEXUS_WAL_CHECKPOINT_THRESHOLD', @@ -598,7 +649,7 @@ export interface AnalyzeOptions { workerTimeout?: string; /** Control LadybugDB WAL auto-checkpoint threshold during analyze. */ walCheckpointThreshold?: string; - /** Parse worker pool size; 0 disables workers (sequential fallback). */ + /** Parse worker pool size (>=1); 0 is rejected (no sequential mode). */ workers?: string; embeddingThreads?: string; embeddingBatchSize?: string; @@ -793,10 +844,11 @@ const analyzeCommandImpl = async ( let workerPoolSize: number | undefined; if (options.workers !== undefined) { const parsedWorkers = Number(options.workers); - if (!Number.isInteger(parsedWorkers) || parsedWorkers < 0) { + if (!Number.isInteger(parsedWorkers) || parsedWorkers < 1) { cliError( - ' --workers must be a non-negative integer. ' + - 'Pass 0 to disable the worker pool (sequential fallback).\n', + ' --workers must be a positive integer (>= 1). ' + + 'GitNexus parses through a worker pool only — there is no sequential ' + + 'mode, so 0 is not allowed. Omit --workers for an auto-sized pool.\n', ); process.exitCode = 1; return; @@ -891,11 +943,13 @@ const analyzeCommandImpl = async ( } // If the target repo contains files an optional grammar would parse but - // that grammar's native binding is absent, warn before analysis so users - // learn why those files end up unparsed instead of silently getting a - // degraded index. + // that grammar's native binding is absent (or disabled via + // GITNEXUS_SKIP_OPTIONAL_GRAMMARS), warn before analysis so users learn why + // those files end up unparsed instead of silently getting a degraded index. + // The extension set is derived from OPTIONAL_GRAMMARS so it can't drift. try { - const matches = await glob(['**/*.dart', '**/*.proto'], { + const optionalGlobs = getOptionalGrammarExtensions().map((e) => `**/*${e}`); + const matches = await glob(optionalGlobs, { cwd: repoPath, ignore: ['**/node_modules/**', '**/.git/**', '**/dist/**', '**/build/**'], dot: false, diff --git a/gitnexus/src/cli/i18n/en.ts b/gitnexus/src/cli/i18n/en.ts index 6b57c2aa6..040008570 100644 --- a/gitnexus/src/cli/i18n/en.ts +++ b/gitnexus/src/cli/i18n/en.ts @@ -175,7 +175,7 @@ export const en = { 'help.option.analyze.walCheckpointThreshold': 'LadybugDB WAL auto-checkpoint threshold in bytes during analyze (integer >= -1; default: 67108864 = 64 MiB; -1 keeps Ladybug stock ~16 MiB).', 'help.option.analyze.workers': - 'Parse worker pool size. Default: cores-1 capped at 16. Pass 0 to disable workers (sequential).', + 'Parse worker pool size (>=1). Default: cores-1 capped at 16, auto-sized to the repo.', 'help.option.analyze.embeddingThreads': 'Limit local ONNX embedding CPU threads', 'help.option.analyze.embeddingBatchSize': 'Number of nodes per embedding batch', 'help.option.analyze.embeddingSubBatchSize': 'Number of chunks per embedding model call', diff --git a/gitnexus/src/cli/i18n/zh-CN.ts b/gitnexus/src/cli/i18n/zh-CN.ts index 0eec71c44..6d1efb77a 100644 --- a/gitnexus/src/cli/i18n/zh-CN.ts +++ b/gitnexus/src/cli/i18n/zh-CN.ts @@ -164,7 +164,7 @@ export const zhCN = { 'help.option.analyze.walCheckpointThreshold': 'analyze 期间 LadybugDB WAL 自动 checkpoint 阈值(字节,整数 >= -1;默认:67108864 = 64 MiB;-1 保持 Ladybug 默认约 16 MiB)。', 'help.option.analyze.workers': - '解析 worker 池大小。默认:cores-1,最多 16。传 0 禁用 worker(顺序执行)。', + '解析 worker 池大小(>=1)。默认:cores-1,最多 16,按仓库规模自适应。', 'help.option.analyze.embeddingThreads': '限制本地 ONNX 嵌入 CPU 线程数', 'help.option.analyze.embeddingBatchSize': '每个嵌入批次的节点数', 'help.option.analyze.embeddingSubBatchSize': '每次嵌入模型调用的分块数', diff --git a/gitnexus/src/cli/index.ts b/gitnexus/src/cli/index.ts index 8ae99455d..a2a88bd83 100644 --- a/gitnexus/src/cli/index.ts +++ b/gitnexus/src/cli/index.ts @@ -87,7 +87,7 @@ program ) .option( '--workers ', - 'Parse worker pool size. Default: cores-1 capped at 16. Pass 0 to disable workers (sequential).', + 'Parse worker pool size (>=1). Default: cores-1 capped at 16, auto-sized to the repo.', ) .option('--embedding-threads ', 'Limit local ONNX embedding CPU threads') .option('--embedding-batch-size ', 'Number of nodes per embedding batch') diff --git a/gitnexus/src/cli/optional-grammars.ts b/gitnexus/src/cli/optional-grammars.ts index e12b471e0..994016a56 100644 --- a/gitnexus/src/cli/optional-grammars.ts +++ b/gitnexus/src/cli/optional-grammars.ts @@ -4,18 +4,22 @@ * tree-sitter-dart, tree-sitter-proto, and tree-sitter-swift are vendored * under vendor/ and materialized into node_modules/ at postinstall. Dart * and Proto are built from source with node-gyp; Swift ships platform - * prebuilds activated via node-gyp-build. All three can be skipped via + * prebuilds activated via node-gyp-build. tree-sitter-kotlin is a declared + * optionalDependency (not vendored). All can be skipped via * GITNEXUS_SKIP_OPTIONAL_GRAMMARS=1 (postinstall scripts), or can silently - * soft-fail when the toolchain is missing (Dart/Proto) or no prebuild - * matches the host platform (Swift). + * soft-fail when the toolchain is missing (Dart/Proto), when no prebuild + * matches the host platform (Swift), or when the optional install was + * skipped or its native build failed (Kotlin). * * Either path produces the same observable: the .node binding is absent * at runtime. This helper detects that condition and surfaces a single - * stderr line per missing grammar so users learn why .dart/.proto/.swift + * stderr line per missing grammar so users learn why .dart/.proto/.swift/.kt * support is unavailable instead of silently getting a degraded index. */ import { createRequire } from 'module'; +import { SupportedLanguages } from 'gitnexus-shared'; +import { isGrammarRuntimeSkipped } from '../core/tree-sitter/parser-loader.js'; import { cliWarn } from './cli-message.js'; const _require = createRequire(import.meta.url); @@ -27,17 +31,55 @@ interface OptionalGrammar { pkg: string; /** File extensions this grammar parses */ extensions: string[]; + /** + * SupportedLanguages id, when this grammar backs an ingestion language. + * Used to ask `isGrammarRuntimeSkipped` whether the grammar was disabled via + * `GITNEXUS_SKIP_OPTIONAL_GRAMMARS` (vs. genuinely missing). Omitted for + * `.proto`, which is a gRPC-extractor concern, not a SupportedLanguages. + */ + language?: SupportedLanguages; } const OPTIONAL_GRAMMARS: OptionalGrammar[] = [ - { name: 'tree-sitter-dart', pkg: 'tree-sitter-dart', extensions: ['.dart'] }, + { + name: 'tree-sitter-dart', + pkg: 'tree-sitter-dart', + extensions: ['.dart'], + language: SupportedLanguages.Dart, + }, { name: 'tree-sitter-proto', pkg: 'tree-sitter-proto', extensions: ['.proto'] }, - { name: 'tree-sitter-swift', pkg: 'tree-sitter-swift', extensions: ['.swift'] }, + { + name: 'tree-sitter-swift', + pkg: 'tree-sitter-swift', + extensions: ['.swift'], + language: SupportedLanguages.Swift, + }, + { + name: 'tree-sitter-kotlin', + pkg: 'tree-sitter-kotlin', + extensions: ['.kt', '.kts'], + language: SupportedLanguages.Kotlin, + }, ]; +/** + * The file extensions backed by an optional grammar — the single source for + * the `analyze` preflight glob (so the glob can't drift from this list). + */ +export function getOptionalGrammarExtensions(): string[] { + return [...new Set(OPTIONAL_GRAMMARS.flatMap((g) => g.extensions))]; +} + export interface MissingGrammar { name: string; extensions: string[]; + /** + * `missing` — the native binding could not be loaded (not installed / build + * soft-failed / no prebuild). `skipped` — the binding is fine but the user + * disabled it via `GITNEXUS_SKIP_OPTIONAL_GRAMMARS`. Drives the warning text + * so a deliberate opt-out is not told to reinstall. + */ + reason: 'missing' | 'skipped'; } /** @@ -59,6 +101,13 @@ export interface MissingGrammar { export function detectMissingOptionalGrammars(): MissingGrammar[] { const missing: MissingGrammar[] = []; for (const g of OPTIONAL_GRAMMARS) { + // Deliberate runtime opt-out comes first: even an installed binding is + // treated as unavailable, with a `skipped` reason so the warning says so + // instead of suggesting a reinstall (#2101 review). + if (g.language !== undefined && isGrammarRuntimeSkipped(g.language)) { + missing.push({ name: g.name, extensions: g.extensions, reason: 'skipped' }); + continue; + } try { _require(g.pkg); } catch (err) { @@ -80,7 +129,7 @@ export function detectMissingOptionalGrammars(): MissingGrammar[] { { grammar: g.name, extensions: g.extensions, error: msg }, ); } - missing.push({ name: g.name, extensions: g.extensions }); + missing.push({ name: g.name, extensions: g.extensions, reason: 'missing' }); } } return missing; @@ -110,9 +159,16 @@ export function warnMissingOptionalGrammars(opts?: { if (relevantExtensions && !g.extensions.some((e) => relevantExtensions.has(e))) { continue; } - cliWarn( - `GitNexus${ctx}: optional grammar "${g.name}" is unavailable — ${g.extensions.join('/')} files will not be parsed. Reinstall without GITNEXUS_SKIP_OPTIONAL_GRAMMARS=1 (and ensure python3, make, g++) to enable.`, - { grammar: g.name, extensions: g.extensions, context: opts?.context }, - ); + const exts = g.extensions.join('/'); + const message = + g.reason === 'skipped' + ? `GitNexus${ctx}: optional grammar "${g.name}" is disabled via GITNEXUS_SKIP_OPTIONAL_GRAMMARS — ${exts} files will not be parsed. Unset the variable to re-enable.` + : `GitNexus${ctx}: optional grammar "${g.name}" is unavailable — ${exts} files will not be parsed. Reinstall without GITNEXUS_SKIP_OPTIONAL_GRAMMARS=1 (and ensure python3, make, g++) to enable.`; + cliWarn(message, { + grammar: g.name, + extensions: g.extensions, + reason: g.reason, + context: opts?.context, + }); } } diff --git a/gitnexus/src/cli/skill-gen.ts b/gitnexus/src/cli/skill-gen.ts index e4e66e85b..f2fb42838 100644 --- a/gitnexus/src/cli/skill-gen.ts +++ b/gitnexus/src/cli/skill-gen.ts @@ -649,9 +649,9 @@ const renderSkillMarkdown = ( : community.label; lines.push('## How to Explore'); lines.push(''); - lines.push(`1. \`gitnexus_context({name: "${firstEntry}"})\` \u2014 see callers and callees`); + lines.push(`1. \`context({name: "${firstEntry}"})\` \u2014 see callers and callees`); lines.push( - `2. \`gitnexus_query({query: "${community.label.toLowerCase()}"})\` \u2014 find related execution flows`, + `2. \`query({query: "${community.label.toLowerCase()}"})\` \u2014 find related execution flows`, ); lines.push('3. Read key files listed above for implementation details'); lines.push(''); diff --git a/gitnexus/src/core/group/extractors/http-patterns/java.ts b/gitnexus/src/core/group/extractors/http-patterns/java.ts index 920d499aa..5d452fd1f 100644 --- a/gitnexus/src/core/group/extractors/http-patterns/java.ts +++ b/gitnexus/src/core/group/extractors/http-patterns/java.ts @@ -55,6 +55,7 @@ const METHOD_ANNOTATION_TO_HTTP: Record = { interface SpringRouteBinding { method: string; path: string; + ownerPrefix?: string; } interface SpringMethodInfo { @@ -395,6 +396,25 @@ function joinPath(prefix: string, methodPath: string): string { return `/${cleanPrefix}/${cleanSub}`; } +function joinInheritedSpringPath( + controllerPrefix: string, + inheritedPath: string, + inheritedOwnerPrefix = '', +): string { + const joined = joinPath(controllerPrefix, inheritedPath); + const cleanPrefix = controllerPrefix.replace(/^\/+/, '').replace(/\/+$/, ''); + const cleanOwnerPrefix = inheritedOwnerPrefix.replace(/^\/+/, '').replace(/\/+$/, ''); + const cleanInherited = inheritedPath.replace(/^\/+/, ''); + if (!cleanPrefix) return joined; + if ( + cleanPrefix === cleanOwnerPrefix && + (cleanInherited === cleanPrefix || cleanInherited.startsWith(`${cleanPrefix}/`)) + ) { + return `/${cleanInherited}`; + } + return joined; +} + function getNodeName(node: Parser.SyntaxNode): string | null { return node.childForFieldName('name')?.text ?? null; } @@ -634,6 +654,7 @@ function scanSpringProject(files: readonly HttpScanInput[]): HttpFileDetections[ const routes = method.routes.map((route) => ({ method: route.method, path: type.classPrefix ? joinPath(type.classPrefix, route.path) : route.path, + ownerPrefix: type.classPrefix, })); if (routes.length > 0) methodMap.set(method.name, routes); } @@ -651,7 +672,7 @@ function scanSpringProject(files: readonly HttpScanInput[]): HttpFileDetections[ const routes = routeMap.get(method.name) ?? []; return routes.map((route) => ({ method: route.method, - path: joinPath(type.classPrefix, route.path), + path: joinInheritedSpringPath(type.classPrefix, route.path, route.ownerPrefix), })); }); diff --git a/gitnexus/src/core/ingestion/ast-cache.ts b/gitnexus/src/core/ingestion/ast-cache.ts deleted file mode 100644 index 454c60df2..000000000 --- a/gitnexus/src/core/ingestion/ast-cache.ts +++ /dev/null @@ -1,77 +0,0 @@ -import { LRUCache } from 'lru-cache'; -import Parser from 'tree-sitter'; - -import { logger } from '../logger.js'; -/** - * Minimal structural shape consumers need when reading Trees back - * through a phase-dependency boundary. Declared here so phases that - * receive ASTCache via `getPhaseOutput<...>` don't hand-roll their - * own inline structural types that silently drift when ASTCache's - * contract changes. - * - * Typed as `unknown` at the Tree boundary because consumers on the - * other side of the phase-output map don't share tree-sitter's type - * graph (e.g. COBOL's standalone processor). - */ -export interface ASTCacheReader { - get(filePath: string): unknown; - clear(): void; -} - -// Define the interface for the Cache -export interface ASTCache extends ASTCacheReader { - get: (filePath: string) => Parser.Tree | undefined; - set: (filePath: string, tree: Parser.Tree) => void; - clear: () => void; - stats: () => { size: number; maxSize: number }; -} - -export const createASTCache = (maxSize: number = 50): ASTCache => { - const effectiveMax = Math.max(maxSize, 1); - // Initialize the cache with a 'dispose' handler - // This is the magic: When an item is evicted (dropped), this runs automatically. - const cache = new LRUCache({ - max: effectiveMax, - dispose: (tree) => { - try { - // NOTE: web-tree-sitter has tree.delete(); native tree-sitter - // trees are GC-managed and .delete is absent (no-op here). - // - // Single-owner invariant (load-bearing under WASM): a given - // Parser.Tree reference must live in AT MOST ONE ASTCache - // that disposes. The parse-phase chunk-local cache clears - // between chunks; the cross-phase `scopeTreeCache` (also an - // ASTCache today) holds the same Tree by reference. Under - // native tree-sitter this is benign (dispose is a no-op). - // If/when GitNexus adopts web-tree-sitter for sequential - // parsing, the cross-phase cache must either (a) skip - // writing Trees that are already owned by a disposing cache, - // or (b) use tree.copy() per entry. Failing to pick one - // will hand freed memory to scope-resolution. - (tree as unknown as { delete?: () => void }).delete?.(); - } catch (e) { - logger.warn({ e }, 'Failed to delete tree from WASM memory'); - } - }, - }); - - return { - get: (filePath: string) => { - const tree = cache.get(filePath); - return tree; // Returns undefined if not found - }, - - set: (filePath: string, tree: Parser.Tree) => { - cache.set(filePath, tree); - }, - - clear: () => { - cache.clear(); - }, - - stats: () => ({ - size: cache.size, - maxSize: effectiveMax, - }), - }; -}; diff --git a/gitnexus/src/core/ingestion/call-processor.ts b/gitnexus/src/core/ingestion/call-processor.ts index 4a9fba9af..b5b258dcd 100644 --- a/gitnexus/src/core/ingestion/call-processor.ts +++ b/gitnexus/src/core/ingestion/call-processor.ts @@ -9,25 +9,17 @@ * * - `processRoutesFromExtracted` — CALLS edges from framework routes * (e.g. Laravel) to their controller methods. - * - `processNextjsFetchRoutes` / `extractFetchCallsFromFiles` / - * `extractConsumerAccessedKeys` — FETCHES edges from `fetch()` calls to - * Next.js Route nodes. + * - `processNextjsFetchRoutes` / `extractConsumerAccessedKeys` — FETCHES edges + * from `fetch()` calls to Next.js Route nodes. * - `buildExportedTypeMapFromGraph` — exported symbol → return/declared type * map, consumed by the cross-file enrichment pass. */ -import Parser from 'tree-sitter'; import { KnowledgeGraph } from '../graph/types.js'; -import { ASTCache } from './ast-cache.js'; import type { SemanticModel, SymbolTableReader } from './model/index.js'; -import { isLanguageAvailable, loadParser, loadLanguage } from '../tree-sitter/parser-loader.js'; -import { getProvider } from './languages/index.js'; import { generateId } from '../../lib/utils.js'; -import { getLanguageFromFilename } from 'gitnexus-shared'; import type { SymbolDefinition } from 'gitnexus-shared'; import { yieldToEventLoop } from './utils/event-loop.js'; -import { parseSourceSafe } from '../tree-sitter/safe-parse.js'; -import { getTreeSitterBufferSize } from './constants.js'; import type { ExtractedRoute, ExtractedFetchCall } from './workers/parse-worker.js'; import { normalizeFetchURL, routeMatches } from './route-extractors/nextjs.js'; import { extractReturnTypeName } from './type-extractors/shared.js'; @@ -39,6 +31,34 @@ const MAX_TYPE_NAME_LENGTH = 256; * Consumed by the cross-file re-resolution / enrichment pass. */ export type ExportedTypeMap = Map>; +/** Record one exported graph node into the incremental ExportedTypeMap. */ +export const accumulateExportedTypesFromParsedNode = ( + result: ExportedTypeMap, + node: { id: string; properties?: Record }, + symbolTable: SymbolTableReader, +): void => { + if (!node.properties?.isExported) return; + if (!node.properties?.filePath || !node.properties?.name) return; + const filePath = node.properties.filePath as string; + const name = node.properties.name as string; + if (!name || name.length > MAX_TYPE_NAME_LENGTH) return; + const defs = symbolTable.lookupExactAll(filePath, name); + const def = defs.find((d) => d.nodeId === node.id) ?? defs[0]; + if (!def) return; + const typeName = def.returnType ?? def.declaredType; + if (!typeName || typeName.length > MAX_TYPE_NAME_LENGTH) return; + const simpleType = extractReturnTypeName(typeName) ?? typeName; + if (!simpleType) return; + let fileExports = result.get(filePath); + if (!fileExports) { + fileExports = new Map(); + result.set(filePath, fileExports); + } + if (fileExports.size < MAX_EXPORTS_PER_FILE) { + fileExports.set(name, simpleType); + } +}; + /** Build ExportedTypeMap from graph nodes — used for the worker path where the * sequential TypeEnv is not available in the main thread. Collects * returnType/declaredType from exported symbols with known types. */ @@ -48,29 +68,7 @@ export function buildExportedTypeMapFromGraph( ): ExportedTypeMap { const result: ExportedTypeMap = new Map(); graph.forEachNode((node) => { - if (!node.properties?.isExported) return; - if (!node.properties?.filePath || !node.properties?.name) return; - const filePath = node.properties.filePath as string; - const name = node.properties.name as string; - if (!name || name.length > MAX_TYPE_NAME_LENGTH) return; - // For callable symbols, use returnType; for properties/variables, use declaredType. - // Use lookupExactAll + nodeId match to handle same-name methods in different classes. - const defs = symbolTable.lookupExactAll(filePath, name); - const def = defs.find((d) => d.nodeId === node.id) ?? defs[0]; - if (!def) return; - const typeName = def.returnType ?? def.declaredType; - if (!typeName || typeName.length > MAX_TYPE_NAME_LENGTH) return; - // Extract simple type name (strip Promise<>, etc.) — reuse shared utility - const simpleType = extractReturnTypeName(typeName) ?? typeName; - if (!simpleType) return; - let fileExports = result.get(filePath); - if (!fileExports) { - fileExports = new Map(); - result.set(filePath, fileExports); - } - if (fileExports.size < MAX_EXPORTS_PER_FILE) { - fileExports.set(name, simpleType); - } + accumulateExportedTypesFromParsedNode(result, node, symbolTable); }); return result; } @@ -448,79 +446,3 @@ export const processNextjsFetchRoutes = ( } } }; - -/** - * Extract fetch() calls from source files (sequential path). - * Workers handle this via tree-sitter captures in parse-worker; this function - * provides the same extraction for the sequential fallback path. - */ -export const extractFetchCallsFromFiles = async ( - files: { path: string; content: string }[], - astCache: ASTCache, -): Promise => { - const parser = await loadParser(); - const result: ExtractedFetchCall[] = []; - - for (const file of files) { - const language = getLanguageFromFilename(file.path); - if (!language) continue; - if (!isLanguageAvailable(language)) continue; - - const provider = getProvider(language); - const queryStr = provider.treeSitterQueries; - if (!queryStr) continue; - - await loadLanguage(language, file.path); - - let tree = astCache.get(file.path); - if (!tree) { - const parseContent = provider.preprocessSource?.(file.content, file.path) ?? file.content; - try { - tree = parseSourceSafe(parser, parseContent, undefined, { - bufferSize: getTreeSitterBufferSize(parseContent), - }); - } catch { - continue; - } - astCache.set(file.path, tree); - } - - let matches; - try { - const lang = parser.getLanguage(); - const query = new Parser.Query(lang, queryStr); - matches = query.matches(tree.rootNode); - } catch { - continue; - } - - for (const match of matches) { - const captureMap: Record = {}; - match.captures.forEach((c) => (captureMap[c.name] = c.node)); - - if (captureMap['route.fetch']) { - const urlNode = captureMap['route.url'] ?? captureMap['route.template_url']; - if (urlNode) { - result.push({ - filePath: file.path, - fetchURL: urlNode.text, - lineNumber: captureMap['route.fetch'].startPosition.row, - }); - } - } else if (captureMap['http_client'] && captureMap['http_client.url']) { - const method = captureMap['http_client.method']?.text; - const url = captureMap['http_client.url'].text; - const HTTP_CLIENT_ONLY = new Set(['head', 'options', 'request', 'ajax']); - if (method && HTTP_CLIENT_ONLY.has(method) && url.startsWith('/')) { - result.push({ - filePath: file.path, - fetchURL: url, - lineNumber: captureMap['http_client'].startPosition.row, - }); - } - } - } - } - - return result; -}; diff --git a/gitnexus/src/core/ingestion/export-detection.ts b/gitnexus/src/core/ingestion/export-detection.ts index 31d0722f4..17494e7bb 100644 --- a/gitnexus/src/core/ingestion/export-detection.ts +++ b/gitnexus/src/core/ingestion/export-detection.ts @@ -4,7 +4,8 @@ * Determines whether a symbol (function, class, etc.) is exported/public * in its language. This is a pure function — safe for use in worker threads. * - * Shared between parse-worker.ts (worker pool) and parsing-processor.ts (sequential fallback). + * Used by the language providers during worker parsing (parse-worker.ts) — the + * sole parse path. (Sequential parsing was removed.) */ import { findSiblingChild, type SyntaxNode } from './utils/ast-helpers.js'; diff --git a/gitnexus/src/core/ingestion/field-extractors/configs/dart.ts b/gitnexus/src/core/ingestion/field-extractors/configs/dart.ts index 52f4c0ca7..00c89b607 100644 --- a/gitnexus/src/core/ingestion/field-extractors/configs/dart.ts +++ b/gitnexus/src/core/ingestion/field-extractors/configs/dart.ts @@ -2,15 +2,59 @@ import { SupportedLanguages } from 'gitnexus-shared'; import type { FieldExtractionConfig } from '../generic.js'; +import type { FieldVisibility } from '../../field-types.js'; +import type { SyntaxNode } from '../../utils/ast-helpers.js'; import { hasKeyword } from './helpers.js'; import { extractSimpleTypeName } from '../../type-extractors/shared.js'; /** * Dart field extraction config. * - * Dart class fields appear as declaration nodes inside class_body. + * Dart class fields appear as `declaration` nodes inside `class_body`. + * Two shapes carry the field name(s): + * - instance / plain fields → `initialized_identifier_list` + * (`int z = 0;`, `int a = 1, b = 2;`) + * - `static const` / `static final` / `const` fields → `static_final_declaration_list` + * (`static const a = 1;`, `static final String b = 'x', c = 'y';`) + * Both shapes may declare SEVERAL fields in one declaration, so name extraction + * is multi-name (`extractNames`). The structure query (`DART_QUERIES`) emits one + * `@definition.property` per name for both shapes; this config enriches each. + * * Visibility is convention-based: underscore prefix = private. */ + +/** All field names declared by a `declaration` node, across both Dart shapes. */ +function extractDartFieldNames(node: SyntaxNode): string[] { + const names: string[] = []; + for (let i = 0; i < node.namedChildCount; i++) { + const child = node.namedChild(i); + if (!child) continue; + + // instance / plain fields: initialized_identifier_list > initialized_identifier > identifier + if (child.type === 'initialized_identifier_list') { + for (let j = 0; j < child.namedChildCount; j++) { + const init = child.namedChild(j); + if (init?.type === 'initialized_identifier') { + const ident = init.firstNamedChild; + if (ident?.type === 'identifier') names.push(ident.text); + } + } + } + + // static const / final fields: static_final_declaration_list > static_final_declaration > identifier + if (child.type === 'static_final_declaration_list') { + for (let j = 0; j < child.namedChildCount; j++) { + const decl = child.namedChild(j); + if (decl?.type === 'static_final_declaration') { + const ident = decl.firstNamedChild; + if (ident?.type === 'identifier') names.push(ident.text); + } + } + } + } + return names; +} + export const dartConfig: FieldExtractionConfig = { language: SupportedLanguages.Dart, typeDeclarationNodes: ['class_definition'], @@ -18,31 +62,20 @@ export const dartConfig: FieldExtractionConfig = { bodyNodeTypes: ['class_body'], defaultVisibility: 'public', + // One AST `declaration` node may declare several fields (`int a, b;`, + // `static final String b = 'x', c = 'y';`), so use the multi-name path. extractName(node) { - // declaration > initialized_identifier_list > initialized_identifier > identifier - for (let i = 0; i < node.namedChildCount; i++) { - const child = node.namedChild(i); - if (child?.type === 'initialized_identifier_list') { - for (let j = 0; j < child.namedChildCount; j++) { - const init = child.namedChild(j); - if (init?.type === 'initialized_identifier') { - const ident = init.firstNamedChild; - if (ident?.type === 'identifier') return ident.text; - } - } - } - if (child?.type === 'initialized_identifier') { - const ident = child.firstNamedChild; - if (ident?.type === 'identifier') return ident.text; - } - } - // fallback: look for direct identifier - const name = node.childForFieldName('name'); - return name?.text; + return extractDartFieldNames(node)[0]; + }, + + extractNames(node) { + return extractDartFieldNames(node); }, extractType(node) { - // declaration > type_identifier (first named child usually) + // declaration > type_identifier (the type annotation, present for both the + // instance-field shape and `static final String b = …`). `static const a = 1;` + // has no annotation → undefined (untyped). for (let i = 0; i < node.namedChildCount; i++) { const child = node.namedChild(i); if (child && (child.type === 'type_identifier' || child.type === 'function_type')) { @@ -52,22 +85,16 @@ export const dartConfig: FieldExtractionConfig = { return undefined; }, - extractVisibility(node) { - // Dart uses _ prefix for private - // Walk to find the identifier name - for (let i = 0; i < node.namedChildCount; i++) { - const child = node.namedChild(i); - if (child?.type === 'initialized_identifier_list') { - for (let j = 0; j < child.namedChildCount; j++) { - const init = child.namedChild(j); - if (init?.type === 'initialized_identifier') { - const ident = init.firstNamedChild; - if (ident?.text?.startsWith('_')) return 'private'; - } - } - } - } - return 'public'; + // Per-name: Dart convention is underscore-prefixed = private. A single + // declaration can mix visibilities (`static const _p = 1, q = 2;`), so the + // decision is keyed on the individual field name. + extractVisibilityForName(_node, name): FieldVisibility { + return name.startsWith('_') ? 'private' : 'public'; + }, + + extractVisibility(node): FieldVisibility { + const first = extractDartFieldNames(node)[0]; + return first?.startsWith('_') ? 'private' : 'public'; }, isStatic(node) { @@ -75,6 +102,8 @@ export const dartConfig: FieldExtractionConfig = { }, isReadonly(node) { + // `final` / `const` (both `final_builtin`/`const_builtin` nodes whose text + // is `final`/`const`) are read-only. return hasKeyword(node, 'final') || hasKeyword(node, 'const'); }, }; diff --git a/gitnexus/src/core/ingestion/field-extractors/configs/go.ts b/gitnexus/src/core/ingestion/field-extractors/configs/go.ts index b51f1f046..0e37b89c4 100644 --- a/gitnexus/src/core/ingestion/field-extractors/configs/go.ts +++ b/gitnexus/src/core/ingestion/field-extractors/configs/go.ts @@ -3,6 +3,8 @@ import { SupportedLanguages } from 'gitnexus-shared'; import type { FieldExtractionConfig } from '../generic.js'; import { extractSimpleTypeName } from '../../type-extractors/shared.js'; +import type { FieldVisibility } from '../../field-types.js'; +import type { SyntaxNode } from '../../utils/ast-helpers.js'; /** * Go field extraction config. @@ -13,14 +15,52 @@ import { extractSimpleTypeName } from '../../type-extractors/shared.js'; * Visibility in Go is based on the first character: uppercase = exported (public), * lowercase = unexported (package). */ +function goVisibilityForName(name: string): FieldVisibility { + const first = name.charAt(0); + return first === first.toUpperCase() && first !== first.toLowerCase() ? 'public' : 'package'; +} + +function extractGoFieldNames(node: SyntaxNode): string[] { + const names: string[] = []; + for (let i = 0; i < node.namedChildCount; i++) { + const child = node.namedChild(i); + if (child?.type === 'field_identifier') names.push(child.text); + } + return names; +} + export const goConfig: FieldExtractionConfig = { language: SupportedLanguages.Go, - typeDeclarationNodes: ['type_declaration'], + typeDeclarationNodes: ['type_declaration', 'struct_type'], fieldNodeTypes: ['field_declaration'], bodyNodeTypes: ['field_declaration_list'], defaultVisibility: 'package', + extractOwnerName(node) { + if (node.type === 'struct_type') { + return node.parent?.type === 'type_spec' + ? node.parent.childForFieldName('name')?.text + : undefined; + } + const typeSpec = node.namedChildren.find((child) => child.type === 'type_spec'); + return typeSpec?.childForFieldName('name')?.text; + }, + + findBodyNodes(node) { + if (node.type === 'struct_type') { + const body = node.namedChildren.find((child) => child.type === 'field_declaration_list'); + return body ? [body] : []; + } + const typeSpec = node.namedChildren.find((child) => child.type === 'type_spec'); + const typeNode = typeSpec?.childForFieldName('type'); + const body = typeNode?.namedChildren.find((child) => child.type === 'field_declaration_list'); + return body ? [body] : []; + }, + extractName(node) { + const firstName = extractGoFieldNames(node)[0]; + if (firstName) return firstName; + // field_declaration > name:(field_identifier) const name = node.childForFieldName('name'); if (name) return name.text; @@ -32,6 +72,8 @@ export const goConfig: FieldExtractionConfig = { return undefined; }, + extractNames: extractGoFieldNames, + extractType(node) { // field_declaration > type:(type_identifier | pointer_type | ...) const typeNode = node.childForFieldName('type'); @@ -54,6 +96,10 @@ export const goConfig: FieldExtractionConfig = { return 'package'; }, + extractVisibilityForName(_node, name) { + return goVisibilityForName(name); + }, + isStatic(_node) { return false; // Go has no static fields }, diff --git a/gitnexus/src/core/ingestion/field-extractors/configs/jvm.ts b/gitnexus/src/core/ingestion/field-extractors/configs/jvm.ts index 9d6cdbf07..37015a998 100644 --- a/gitnexus/src/core/ingestion/field-extractors/configs/jvm.ts +++ b/gitnexus/src/core/ingestion/field-extractors/configs/jvm.ts @@ -5,6 +5,7 @@ import type { FieldExtractionConfig } from '../generic.js'; import { findVisibility, hasKeyword, hasModifier, typeFromField } from './helpers.js'; import { extractSimpleTypeName } from '../../type-extractors/shared.js'; import type { FieldVisibility } from '../../field-types.js'; +import type { SyntaxNode } from '../../utils/ast-helpers.js'; // --------------------------------------------------------------------------- // Java @@ -73,13 +74,49 @@ export const javaConfig: FieldExtractionConfig = { const KOTLIN_VIS = new Set(['public', 'private', 'protected', 'internal']); +/** A property_declaration is a companion-object member when its nearest + * class-body ancestor is the body of a companion_object (F52, issue #1919). + * Companion members are addressed statically through the enclosing class + * (`C.TAG`), so they are marked static. */ +function isInsideKotlinCompanion(node: SyntaxNode): boolean { + for (let cur = node.parent; cur !== null; cur = cur.parent) { + if (cur.type === 'class_body') return cur.parent?.type === 'companion_object'; + if (cur.type === 'companion_object') return true; + } + return false; +} + export const kotlinConfig: FieldExtractionConfig = { language: SupportedLanguages.Kotlin, - typeDeclarationNodes: ['class_declaration', 'object_declaration'], + // F52: include companion_object so a companion property's innermost + // class-container owner (findEnclosingClassNode returns the companion_object) + // is recognized as a type declaration and its nested class_body is walked. + // The structure query already creates the Property node and owns it on the + // ENCLOSING class for anonymous companions / on the named companion Class — + // this entry only drives field-metadata enrichment, so it does NOT change + // ownership or emit a second node (no double-count). + typeDeclarationNodes: ['class_declaration', 'object_declaration', 'companion_object'], fieldNodeTypes: ['property_declaration'], bodyNodeTypes: ['class_body'], defaultVisibility: 'public', + // F52: an anonymous `companion object { ... }` has no name child, so the + // generic factory's `childForFieldName('name')` owner lookup is empty and + // `extract()` would bail before walking the body. Supply a stable owner + // name (the named companion's identifier, else "Companion") so the body IS + // walked; the resulting FieldInfo map is keyed by field NAME only, so the + // owner name does not affect which Property node gets enriched. + extractOwnerName(node) { + const typeIdentifierText = node.namedChildren.find((c) => c.type === 'type_identifier')?.text; + if (node.type === 'companion_object') { + // Anonymous companions have no type_identifier — fall back to "Companion". + return typeIdentifierText ?? 'Companion'; + } + const name = node.childForFieldName('name'); + if (name) return name.text; + return typeIdentifierText; + }, + extractName(node) { // property_declaration > variable_declaration > simple_identifier for (let i = 0; i < node.namedChildCount; i++) { @@ -124,9 +161,11 @@ export const kotlinConfig: FieldExtractionConfig = { return findVisibility(node, KOTLIN_VIS, 'public', 'modifiers'); }, - isStatic(_node) { - // Kotlin doesn't have static; companion object members are handled separately - return false; + isStatic(node) { + // Kotlin has no `static`, but companion-object members are accessed + // statically through the enclosing class (`C.TAG`) — mark them static + // so the field metadata reflects that (F52). + return isInsideKotlinCompanion(node); }, isReadonly(node) { diff --git a/gitnexus/src/core/ingestion/field-extractors/configs/swift.ts b/gitnexus/src/core/ingestion/field-extractors/configs/swift.ts index 75c70ab95..6e27ff702 100644 --- a/gitnexus/src/core/ingestion/field-extractors/configs/swift.ts +++ b/gitnexus/src/core/ingestion/field-extractors/configs/swift.ts @@ -2,7 +2,7 @@ import { SupportedLanguages } from 'gitnexus-shared'; import type { FieldExtractionConfig } from '../generic.js'; -import { hasKeyword, findVisibility } from './helpers.js'; +import { hasKeyword, hasModifier, findVisibility } from './helpers.js'; import { extractSimpleTypeName } from '../../type-extractors/shared.js'; import type { FieldVisibility } from '../../field-types.js'; @@ -17,18 +17,33 @@ const SWIFT_VIS = new Set([ /** * Swift field extraction config. * - * Handles property_declaration inside class_body / protocol_body. + * Handles property_declaration inside class_body / protocol_body and + * protocol_property_declaration inside protocol_body (F75 — protocol property + * requirements like "var title: String { get }"). + * * tree-sitter-swift uses property_declaration for stored/computed properties. + * A protocol property requirement parses to its own node type, + * protocol_property_declaration, whose name lives in a "name:" pattern field + * (pattern > value_binding_pattern + simple_identifier(bound_identifier)), its + * type in a sibling type_annotation, and its "{ get }" / "{ get set }" in a + * protocol_property_requirements child. Note: Swift reuses the "name:" field + * across many positions (func name, every parameter label, parameter/return + * type), so the name is synthesized from the simple_identifier inside the + * pattern rather than read blindly off "name:". */ export const swiftConfig: FieldExtractionConfig = { language: SupportedLanguages.Swift, typeDeclarationNodes: ['class_declaration', 'protocol_declaration'], - fieldNodeTypes: ['property_declaration'], + fieldNodeTypes: ['property_declaration', 'protocol_property_declaration'], bodyNodeTypes: ['class_body', 'protocol_body'], defaultVisibility: 'internal', extractName(node) { - // property_declaration > pattern > simple_identifier + // property_declaration > pattern > simple_identifier, and + // protocol_property_declaration > name: (pattern ... simple_identifier). + // For protocol_property_declaration the pattern wraps a leading + // value_binding_pattern ("var") plus the simple_identifier — the loop + // below skips the binding keyword and returns the identifier. for (let i = 0; i < node.namedChildCount; i++) { const child = node.namedChild(i); if (child?.type === 'pattern') { @@ -62,7 +77,19 @@ export const swiftConfig: FieldExtractionConfig = { }, isStatic(node) { - return hasKeyword(node, 'static') || hasKeyword(node, 'class'); + // `static`/`class` (type-level) modifiers live inside a `modifiers` + // wrapper for both property_declaration and protocol_property_declaration + // (e.g. `static var shared: P { get }`), so check the wrapper too. + // `hasKeyword` compares each direct child by `.text` equality: it matches a + // single-modifier wrapper (`modifiers.text === 'static'`) but fails for a + // multi-modifier wrapper (`private static` → `modifiers.text === 'private static'`), + // which `hasModifier` handles by descending into the wrapper's children. + return ( + hasKeyword(node, 'static') || + hasKeyword(node, 'class') || + hasModifier(node, 'modifiers', 'static') || + hasModifier(node, 'modifiers', 'class') + ); }, isReadonly(node) { diff --git a/gitnexus/src/core/ingestion/field-extractors/generic.ts b/gitnexus/src/core/ingestion/field-extractors/generic.ts index 4cb4a5b1b..68f77dc8b 100644 --- a/gitnexus/src/core/ingestion/field-extractors/generic.ts +++ b/gitnexus/src/core/ingestion/field-extractors/generic.ts @@ -33,6 +33,10 @@ export interface FieldExtractionConfig { bodyNodeTypes: string[]; /** Default visibility when no modifier is present */ defaultVisibility: FieldVisibility; + /** Extract owner type name from a type declaration node. */ + extractOwnerName?: (node: SyntaxNode) => string | undefined; + /** Find body nodes inside a type declaration node. */ + findBodyNodes?: (node: SyntaxNode) => SyntaxNode[]; /** * Extract field name from a field declaration node. * Use this for nodes that declare exactly one field. @@ -49,6 +53,8 @@ export interface FieldExtractionConfig { extractType: (node: SyntaxNode) => string | undefined; /** Extract visibility from a field declaration node */ extractVisibility: (node: SyntaxNode) => FieldVisibility; + /** Extract visibility for one field name from a multi-name declaration. */ + extractVisibilityForName?: (node: SyntaxNode, name: string) => FieldVisibility; /** Check if a field is static */ isStatic: (node: SyntaxNode) => boolean; /** Check if a field is readonly/final/const */ @@ -84,10 +90,9 @@ export function createFieldExtractor(config: FieldExtractionConfig): FieldExtrac extract(node: SyntaxNode, context: FieldExtractorContext): ExtractedFields | null { if (!this.isTypeDeclaration(node)) return null; - const nameNode = node.childForFieldName('name'); - if (!nameNode) return null; + const ownerFqn = config.extractOwnerName?.(node) ?? node.childForFieldName('name')?.text; + if (!ownerFqn) return null; - const ownerFqn = nameNode.text; const fields: FieldInfo[] = []; // Find body container(s) @@ -110,6 +115,8 @@ export function createFieldExtractor(config: FieldExtractionConfig): FieldExtrac // ------------------------------------------------------------------ private findBodies(node: SyntaxNode): SyntaxNode[] { + if (config.findBodyNodes) return config.findBodyNodes(node); + const result: SyntaxNode[] = []; // Try named 'body' field first const bodyField = node.childForFieldName('body'); @@ -179,7 +186,7 @@ export function createFieldExtractor(config: FieldExtractionConfig): FieldExtrac return { name, type, - visibility: config.extractVisibility(node), + visibility: config.extractVisibilityForName?.(node, name) ?? config.extractVisibility(node), isStatic: config.isStatic(node), isReadonly: config.isReadonly(node), sourceFile: context.filePath, diff --git a/gitnexus/src/core/ingestion/filesystem-walker.ts b/gitnexus/src/core/ingestion/filesystem-walker.ts index 9ba959ea6..0af28a958 100644 --- a/gitnexus/src/core/ingestion/filesystem-walker.ts +++ b/gitnexus/src/core/ingestion/filesystem-walker.ts @@ -6,10 +6,6 @@ import { glob } from 'glob'; import { createIgnoreFilter } from '../../config/ignore-service.js'; import { logger } from '../logger.js'; -export interface FileEntry { - path: string; - content: string; -} /** Lightweight entry — path + size from stat, no content in memory */ export interface ScannedFile { @@ -153,21 +149,3 @@ export const readFileContents = async ( return contents; }; - -/** - * Legacy API — scans and reads everything into memory. - * Used by sequential fallback path only. - */ -export const walkRepository = async ( - repoPath: string, - onProgress?: (current: number, total: number, filePath: string) => void, -): Promise => { - const scanned = await walkRepositoryPaths(repoPath, onProgress); - const contents = await readFileContents( - repoPath, - scanned.map((f) => f.path), - ); - return scanned - .filter((f) => contents.has(f.path)) - .map((f) => ({ path: f.path, content: contents.get(f.path)! })); -}; diff --git a/gitnexus/src/core/ingestion/finalize-orchestrator.ts b/gitnexus/src/core/ingestion/finalize-orchestrator.ts index 558672d7d..02717a0b9 100644 --- a/gitnexus/src/core/ingestion/finalize-orchestrator.ts +++ b/gitnexus/src/core/ingestion/finalize-orchestrator.ts @@ -51,6 +51,8 @@ import { finalize, } from 'gitnexus-shared'; import type { ScopeResolutionIndexes } from './model/scope-resolution-indexes.js'; +import { parseTruthyEnv } from './utils/env.js'; +import { TransitionalScopeTree } from '../../storage/scope-index-store.js'; // ─── Public entry point ───────────────────────────────────────────────────── @@ -114,7 +116,13 @@ export function finalizeScopeModel( moduleEntries.push({ filePath: file.filePath, moduleScopeId: file.moduleScope }); } - const scopeTree = buildScopeTree(allScopes); + // Out-of-core scope index: when enabled, build a TransitionalScopeTree + // (validated + fully resident now; sealed to disk by run.ts just before emit so + // the heavy Scope.bindings payload is reclaimed). Default off → the in-heap + // buildScopeTree result exactly, byte-identical. + const scopeTree = parseTruthyEnv(process.env.GITNEXUS_DISK_SCOPE_INDEX) + ? new TransitionalScopeTree(allScopes) + : buildScopeTree(allScopes); const defs = buildDefIndex(allDefs); const qualifiedNames = buildQualifiedNameIndex(allDefs); const moduleScopes = buildModuleScopeIndex(moduleEntries); diff --git a/gitnexus/src/core/ingestion/language-provider.ts b/gitnexus/src/core/ingestion/language-provider.ts index c979102e5..b375050fa 100644 --- a/gitnexus/src/core/ingestion/language-provider.ts +++ b/gitnexus/src/core/ingestion/language-provider.ts @@ -311,6 +311,27 @@ interface LanguageProviderConfig { }, ) => readonly CaptureMatch[]; + /** + * Snapshot the capture-time side-channel state that this provider's + * `emitScopeCaptures` just populated for `filePath` into module-level maps, + * returning a plain JSON-serializable value (or `undefined` when there is + * nothing to carry). + * + * Called in the parse worker IMMEDIATELY after `emitScopeCaptures` runs for + * a file (see `parse-worker.ts`), and the result is stored on the produced + * `ParsedFile.captureSideChannel`. Scope-resolution on the main thread reuses + * that serialized `ParsedFile` and skips re-extraction (#1983), so this hook + * is how the worker-computed marks survive the worker→main boundary and the + * disk store WITHOUT a main-thread re-parse. The main thread restores them + * via the matching `ScopeResolver.applyCaptureSideChannel` hook. + * + * MUST return plain data (objects / arrays / primitives) so it round-trips + * through `JSON.stringify` + the parsedfile-store interning reviver. + * + * Default: undefined (provider has no capture-time module-level side effects). + */ + readonly collectCaptureSideChannel?: (filePath: string) => unknown; + /** * Interpret a raw `@import.statement` capture group into a `ParsedImport`. * The central finalize algorithm resolves `ParsedImport.targetRaw` to a diff --git a/gitnexus/src/core/ingestion/languages/c-cpp.ts b/gitnexus/src/core/ingestion/languages/c-cpp.ts index 4ce17a9c5..84523fc70 100644 --- a/gitnexus/src/core/ingestion/languages/c-cpp.ts +++ b/gitnexus/src/core/ingestion/languages/c-cpp.ts @@ -53,6 +53,7 @@ import { cBindingScopeFor, cImportOwningScope, cReceiverBinding, + collectCStaticLinkageSideChannel, } from './c/index.js'; import { emitCppScopeCaptures, @@ -62,6 +63,7 @@ import { cppBindingScopeFor, cppImportOwningScope, cppReceiverBinding, + collectCppCaptureSideChannel, } from './cpp/index.js'; import { extractCppTemplateConstraints } from './cpp/constraint-extractor.js'; @@ -395,6 +397,15 @@ export const cProvider = defineLanguage({ // ── RFC #909 Ring 3: scope-based resolution hooks (RFC §5) ────────── emitScopeCaptures: emitCScopeCaptures, + // Worker-side: snapshot the module-level `static`-linkage marks + // `emitCScopeCaptures` just populated for this file (`markStaticName` → + // `staticNames`) into plain data on `ParsedFile.captureSideChannel`, so the + // main thread can restore them via `applyCaptureSideChannel` WITHOUT a + // re-parse (#1983 — the worker is the sole parse path). Without this, C + // `static` functions look non-file-local on the main thread and leak into + // cross-file global free-call resolution / wildcard imports. See + // `c/capture-side-channel.ts`. + collectCaptureSideChannel: collectCStaticLinkageSideChannel, interpretImport: interpretCImport, interpretTypeBinding: interpretCTypeBinding, bindingScopeFor: cBindingScopeFor, @@ -465,6 +476,11 @@ export const cppProvider = defineLanguage({ // ── RFC #909 Ring 3: scope-based resolution hooks (RFC §5) ────────── emitScopeCaptures: emitCppScopeCaptures, + // Worker-side: snapshot the module-level capture marks `emitCppScopeCaptures` + // just populated for this file into plain data on `ParsedFile.captureSideChannel`, + // so the main thread can restore them via `applyCaptureSideChannel` WITHOUT a + // re-parse (#1983). See `cpp/capture-side-channel.ts`. + collectCaptureSideChannel: collectCppCaptureSideChannel, interpretImport: interpretCppImport, interpretTypeBinding: interpretCppTypeBinding, bindingScopeFor: cppBindingScopeFor, diff --git a/gitnexus/src/core/ingestion/languages/c/capture-side-channel.ts b/gitnexus/src/core/ingestion/languages/c/capture-side-channel.ts new file mode 100644 index 000000000..2f619e5f3 --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/c/capture-side-channel.ts @@ -0,0 +1,80 @@ +/** + * C capture-time side-channel serialization (#1983). + * + * `emitCScopeCaptures` populates one MODULE-LEVEL, per-file map as a side + * effect that is NOT part of the returned `ParsedFile`'s scopes/defs: + * + * - `staticNames` (static-linkage.ts) — the simple names of functions + * declared with `static` storage class (file-local / translation-unit + * linkage in C), recorded via `markStaticName` from the + * `@declaration.name` capture when the function node has a `static` + * storage-class specifier. + * + * On the worker path that map is filled in the WORKER process and lost across + * the worker→main MessageChannel (and the disk-backed parsedfile-store), + * because scope-resolution reuses the serialized `ParsedFile` and SKIPS the + * main-thread re-extraction (the #1983 fix that avoids a main-thread + * tree-sitter re-parse / OOM on huge repos — e.g. the Linux kernel). The main + * thread then reads the map empty in `isStaticName` (consulted by + * `isFileLocalDef` in `c/scope-resolver.ts` and by `expandCWildcardNames` in + * static-linkage.ts) — so file-local `static` functions become eligible for + * cross-file global free-call resolution (false CALLS edges) and `#include` + * wildcard imports over-expose them. + * + * This module snapshots the per-file slice of that map into a plain, + * JSON-serializable object (carried on `ParsedFile.captureSideChannel`) and + * restores it on the main thread WITHOUT any parse. It mirrors the C++ pattern + * in `cpp/capture-side-channel.ts` and the Kotlin pattern in + * `kotlin/capture-side-channel.ts`. + * + * The single generic `ParsedFile.captureSideChannel` field is shared with C++ + * and Kotlin, which is safe because each file is one language (a `.c` file uses + * the C provider). The payload is self-describing (`{ kind: 'c', staticNames }`) + * so `applyCStaticLinkageSideChannel` only restores C state and ignores a + * foreign-shaped snapshot. + */ + +import type { ParsedFile } from 'gitnexus-shared'; +import { getStaticNamesForFile, markStaticName } from './static-linkage.js'; + +/** + * Plain JSON-serializable snapshot of the per-file C capture-time + * side-channel. Carried opaquely on `ParsedFile.captureSideChannel`. The + * `kind` tag makes the payload self-describing so `apply` can distinguish a C + * snapshot from another language's (C++ and Kotlin share the same field). + */ +export interface CCaptureSideChannel { + readonly kind: 'c'; + /** Simple names of `static` (file-local linkage) functions in this file. */ + readonly staticNames: readonly string[]; +} + +/** + * `LanguageProvider.collectCaptureSideChannel` implementation for C. + * Returns `undefined` when this file recorded no static names at all, so the + * produced `ParsedFile` carries the field only when there's data to ship. + */ +export function collectCStaticLinkageSideChannel( + filePath: string, +): CCaptureSideChannel | undefined { + const staticNames = getStaticNamesForFile(filePath); + if (staticNames.length === 0) return undefined; + return { kind: 'c', staticNames }; +} + +/** + * `ScopeResolver.applyCaptureSideChannel` implementation for C. Reads the + * worker-serialized snapshot from `parsed.captureSideChannel` and re-populates + * the module-level static-linkage map via `markStaticName`. Tolerant of + * `undefined` (file carried no data) and of an unexpected / foreign shape + * (defensive — the `kind` tag guards against restoring a non-C payload). + * Does NO tree-sitter parse. + */ +export function applyCStaticLinkageSideChannel(parsed: ParsedFile): void { + const data = parsed.captureSideChannel as CCaptureSideChannel | undefined; + if (data === undefined || data === null || typeof data !== 'object') return; + if (data.kind !== 'c' || !Array.isArray(data.staticNames)) return; + for (const name of data.staticNames) { + markStaticName(parsed.filePath, name); + } +} diff --git a/gitnexus/src/core/ingestion/languages/c/import-decomposer.ts b/gitnexus/src/core/ingestion/languages/c/import-decomposer.ts index 5cef27430..77b54f120 100644 --- a/gitnexus/src/core/ingestion/languages/c/import-decomposer.ts +++ b/gitnexus/src/core/ingestion/languages/c/import-decomposer.ts @@ -5,10 +5,20 @@ import { nodeToCapture, syntheticCapture, type SyntaxNode } from '../../utils/as * Decompose a `preproc_include` node into a CaptureMatch with structured * import captures. C #include maps to a wildcard import (all symbols * from the header are visible). + * + * Only literal include paths are emitted as import sources: + * #include → system_lib_string + * #include "local.h" → string_literal + * A computed include like `#include HEADER_MACRO` carries an `identifier` + * path node (the macro name, not a header path). Emitting it as an import + * source produces a garbage literal edge, so we skip it entirely — matching + * the convention in interpretCImport, which drops imports with no resolvable + * source (issue #1919 F5). */ export function splitCInclude(node: SyntaxNode): CaptureMatch | null { // node.type === 'preproc_include' // path field: (string_literal (string_content)) | (system_lib_string) + // | (identifier) ← computed macro include, NOT a header path const pathNode = node.childForFieldName?.('path') ?? null; if (pathNode === null) { // Fallback: scan children @@ -24,7 +34,13 @@ export function splitCInclude(node: SyntaxNode): CaptureMatch | null { return buildIncludeCapture(node, pathNode); } -function buildIncludeCapture(node: SyntaxNode, pathNode: SyntaxNode): CaptureMatch { +function buildIncludeCapture(node: SyntaxNode, pathNode: SyntaxNode): CaptureMatch | null { + // Skip computed includes (`#include MACRO`) — the path is an `identifier`, + // not a literal header path. Emitting it would create a garbage import. + if (pathNode.type !== 'string_literal' && pathNode.type !== 'system_lib_string') { + return null; + } + let raw: string; if (pathNode.type === 'string_literal') { // string_literal has children: `"`, string_content, `"` diff --git a/gitnexus/src/core/ingestion/languages/c/import-target.ts b/gitnexus/src/core/ingestion/languages/c/import-target.ts index 0cb9c2fb4..495846030 100644 --- a/gitnexus/src/core/ingestion/languages/c/import-target.ts +++ b/gitnexus/src/core/ingestion/languages/c/import-target.ts @@ -1,5 +1,54 @@ import { dirname, join } from 'path'; +/** + * A workspace file path pre-decomposed for the suffix-match fallback: + * `original` is returned verbatim (preserving the prior `bestMatch = filePath` + * contract); `normalized` and `depth` are precomputed so the hot path does no + * per-element regex/`split`. + */ +interface CSuffixCandidate { + original: string; + normalized: string; + depth: number; +} + +/** + * Per-pass memo: workspace paths bucketed by basename (last path segment), + * keyed on the `allFilePaths` set identity. + * + * `resolveCImportTarget` is called once per (quoted) C/C++ `#include` with the + * same `allFilePaths` set per pass (the augmented set is itself memoized in + * the C resolver). The old suffix-match fallback scanned ALL workspace paths + * per include — with a per-element `.replace`/`.split` and no early exit + * (the fewest-path-components tie-break forces a full scan) — i.e. + * O(R_suffix × (F+H)). A path can satisfy `endsWith('/'+target)` (or equal + * the target) ONLY IF its basename equals the target's last segment, so we + * pre-bucket by basename once (O(F+H), `normalized`/`depth` precomputed) and + * the fallback inspects a single small bucket → O(F+H) build + ~O(1)/include. + * `WeakMap`-keyed so it is reclaimed with the pass (no cross-pass staleness). + * Shared by C and C++ (`resolveCppImportTarget` delegates here). + */ +const suffixIndexByPaths = new WeakMap, Map>(); + +function suffixIndex(allFilePaths: ReadonlySet): Map { + let index = suffixIndexByPaths.get(allFilePaths); + if (index === undefined) { + index = new Map(); + for (const original of allFilePaths) { + const normalized = original.replace(/\\/g, '/'); + const basename = normalized.slice(normalized.lastIndexOf('/') + 1); + let bucket = index.get(basename); + if (bucket === undefined) { + bucket = []; + index.set(basename, bucket); + } + bucket.push({ original, normalized, depth: normalized.split('/').length }); + } + suffixIndexByPaths.set(allFilePaths, index); + } + return index; +} + /** * Resolve a C #include path to a file in the workspace. * @@ -41,21 +90,31 @@ export function resolveCImportTarget( // Exact match (path as-is in the workspace) if (allFilePaths.has(normalizedTarget)) return normalizedTarget; - // Suffix match: find files ending with /targetRaw or equal to targetRaw + // Suffix match: find files ending with /targetRaw or equal to targetRaw. + // A path can only match `=== normalizedTarget` or `endsWith('/'+target)` if + // its basename equals the target's last segment, so we inspect only that + // basename bucket (built once per pass) instead of scanning every workspace + // path. Match condition + tie-break (fewest path components, then + // lexicographic on the normalized path) are byte-identical to the prior scan. const suffix = '/' + normalizedTarget; + const targetBasename = normalizedTarget.slice(normalizedTarget.lastIndexOf('/') + 1); + const bucket = suffixIndex(allFilePaths).get(targetBasename); + if (bucket === undefined) return null; + let bestMatch: string | null = null; let bestDepth = Infinity; let bestNormalized = ''; - for (const filePath of allFilePaths) { - const normalized = filePath.replace(/\\/g, '/'); - if (normalized === normalizedTarget || normalized.endsWith(suffix)) { + for (const cand of bucket) { + if (cand.normalized === normalizedTarget || cand.normalized.endsWith(suffix)) { // Prefer shortest path (closest match) - const depth = normalized.split('/').length; - if (depth < bestDepth || (depth === bestDepth && normalized < bestNormalized)) { - bestDepth = depth; - bestMatch = filePath; - bestNormalized = normalized; + if ( + cand.depth < bestDepth || + (cand.depth === bestDepth && cand.normalized < bestNormalized) + ) { + bestDepth = cand.depth; + bestMatch = cand.original; + bestNormalized = cand.normalized; } } } diff --git a/gitnexus/src/core/ingestion/languages/c/index.ts b/gitnexus/src/core/ingestion/languages/c/index.ts index c6900ecba..8ebc2c98e 100644 --- a/gitnexus/src/core/ingestion/languages/c/index.ts +++ b/gitnexus/src/core/ingestion/languages/c/index.ts @@ -13,4 +13,9 @@ export { isStaticName, clearStaticNames, expandCWildcardNames, + getStaticNamesForFile, } from './static-linkage.js'; +export { + collectCStaticLinkageSideChannel, + applyCStaticLinkageSideChannel, +} from './capture-side-channel.js'; diff --git a/gitnexus/src/core/ingestion/languages/c/scope-resolver.ts b/gitnexus/src/core/ingestion/languages/c/scope-resolver.ts index 90d54e072..cb4a6424f 100644 --- a/gitnexus/src/core/ingestion/languages/c/scope-resolver.ts +++ b/gitnexus/src/core/ingestion/languages/c/scope-resolver.ts @@ -7,6 +7,43 @@ import { cProvider } from '../c-cpp.js'; import { cArityCompatibility, cMergeBindings, resolveCImportTarget } from './index.js'; import { scanHeaderFiles } from './header-scan.js'; import { expandCWildcardNames, isStaticName, clearStaticNames } from './static-linkage.js'; +import { applyCStaticLinkageSideChannel } from './capture-side-channel.js'; + +/** + * Per-pass memo of the augmented `#include`-resolution file set + * (`allFilePaths` ∪ header `.h` paths), keyed on the two stable source sets. + * + * `resolveImportTarget` is called once per C `#include`; the old code rebuilt + * a fresh ~F-entry `Set` on EVERY call (O(R × (F+H)) inserts + GC churn) and, + * worse, defeated `resolveCImportTarget`'s own per-set suffix-index memo by + * handing it a new set identity each time. Both `allFilePaths` (built once in + * scope-resolution `run.ts`) and the header set (`loadResolutionConfig` + * result) are stable per pass, so the union is built once and reused. + * `WeakMap`-keyed → reclaimed with the pass (no cross-pass staleness). + */ +const augmentedPathsByPass = new WeakMap< + ReadonlySet, + WeakMap, ReadonlySet> +>(); + +function augmentedFilePaths( + allFilePaths: ReadonlySet, + headerPaths: ReadonlySet, +): ReadonlySet { + let byHeaders = augmentedPathsByPass.get(allFilePaths); + if (byHeaders === undefined) { + byHeaders = new WeakMap(); + augmentedPathsByPass.set(allFilePaths, byHeaders); + } + let augmented = byHeaders.get(headerPaths); + if (augmented === undefined) { + const set = new Set(allFilePaths); + for (const h of headerPaths) set.add(h); + augmented = set; + byHeaders.set(headerPaths, augmented); + } + return augmented; +} /** * C `ScopeResolver` registered in `SCOPE_RESOLVERS` and consumed by @@ -31,15 +68,34 @@ export const cScopeResolver: ScopeResolver = { return scanHeaderFiles(repoPath); }, + // Worker-boundary restore (see `ScopeResolver.applyCaptureSideChannel`). + // `emitCScopeCaptures` records per-file `static`-linkage names + // (`markStaticName` → `staticNames`) as a SIDE EFFECT — that state is NOT + // serialized onto the returned ParsedFile's scopes/defs. On the worker path + // those marks are populated in the worker process and lost across the + // MessageChannel / disk store; the main thread reuses the serialized + // ParsedFile and skips `extractParsedFile`, so `isStaticName` (read by + // `isFileLocalDef` and `expandCWildcardNames`) sees an empty map and C + // `static` functions leak into cross-file global free-call resolution + // (false CALLS edges) and `#include` wildcard imports. The worker stashed a + // plain-data snapshot on `parsed.captureSideChannel` via + // `cProvider.collectCaptureSideChannel`; this restores it into the module + // map WITHOUT any tree-sitter re-parse (the #1983 fix). The + // freshly-extracted leg never calls this — its marks were just populated in + // this process. Runs BEFORE `populateOwners`. + applyCaptureSideChannel: applyCStaticLinkageSideChannel, + resolveImportTarget: (targetRaw, fromFile, allFilePaths, resolutionConfig) => { // Augment allFilePaths with .h files discovered via loadResolutionConfig // since the phase only passes .c files to the C resolver but #include // targets .h files classified as C++ in language detection. const headerPaths = resolutionConfig as ReadonlySet | undefined; if (headerPaths !== undefined && headerPaths.size > 0) { - const augmented = new Set(allFilePaths); - for (const h of headerPaths) augmented.add(h); - return resolveCImportTarget(targetRaw, fromFile, augmented); + return resolveCImportTarget( + targetRaw, + fromFile, + augmentedFilePaths(allFilePaths, headerPaths), + ); } return resolveCImportTarget(targetRaw, fromFile, allFilePaths); }, diff --git a/gitnexus/src/core/ingestion/languages/c/static-linkage.ts b/gitnexus/src/core/ingestion/languages/c/static-linkage.ts index 5cfd166b1..2cc195205 100644 --- a/gitnexus/src/core/ingestion/languages/c/static-linkage.ts +++ b/gitnexus/src/core/ingestion/languages/c/static-linkage.ts @@ -29,11 +29,58 @@ export function isStaticName(filePath: string, name: string): boolean { return staticNames.get(filePath)?.has(name) ?? false; } +/** + * Return the `static` (file-local) names recorded for the given file as a + * plain array (empty when none). Used to snapshot the per-file slice of the + * module-level `staticNames` map into `ParsedFile.captureSideChannel` so it + * survives the worker→main boundary (#1983 — the worker is the sole parse + * path). See `c/capture-side-channel.ts`. + */ +export function getStaticNamesForFile(filePath: string): string[] { + const names = staticNames.get(filePath); + return names === undefined ? [] : [...names]; +} + /** Clear tracked static names (for testing). */ export function clearStaticNames(): void { staticNames.clear(); } +/** + * Per-pass memo: `moduleScope` → owning `ParsedFile`, keyed on the + * `parsedFiles` array identity. + * + * The shared finalize Phase-4 loop calls `expandsWildcardTo` + * (→ `expandCWildcardNames`) ONCE PER RESOLVED `#include` edge, every time + * with the SAME `parsedFiles` reference (wired at scope-resolution + * `run.ts` — `allFilePaths`/`parsedFiles` are built once per pass). The old + * `parsedFiles.find(...)` therefore did a full O(F) scan per edge → + * O(R_include × F) overall; at Linux-kernel scale (F ≈ 63k C files, tens of + * thousands of resolved includes) that is ~10^10+ comparisons on a single + * thread — the dominant term in the scope-resolution finalize grind. + * + * Building the lookup once collapses it to O(R_include + F). `WeakMap`-keyed + * on the array so the index is reclaimed with the pass — no cross-pass + * staleness (mirrors the {@link clearStaticNames} discipline for server-mode + * / multi-repo reuse), and a fresh array transparently rebuilds. + */ +const moduleScopeIndexByPass = new WeakMap>(); + +function moduleScopeIndex(parsedFiles: readonly ParsedFile[]): Map { + let index = moduleScopeIndexByPass.get(parsedFiles); + if (index === undefined) { + index = new Map(); + // First-wins to preserve `Array.find` semantics (returns the first match). + // `moduleScope` is unique per file in practice, so collisions are absent; + // the guard only formalises identical behaviour to the prior `.find`. + for (const p of parsedFiles) { + if (!index.has(p.moduleScope)) index.set(p.moduleScope, p); + } + moduleScopeIndexByPass.set(parsedFiles, index); + } + return index; +} + /** * Return the names visible through a C wildcard import (`#include`). * All module-scope defs from the target file are visible EXCEPT those @@ -43,7 +90,7 @@ export function expandCWildcardNames( targetModuleScope: ScopeId, parsedFiles: readonly ParsedFile[], ): readonly string[] { - const target = parsedFiles.find((p) => p.moduleScope === targetModuleScope); + const target = moduleScopeIndex(parsedFiles).get(targetModuleScope); if (target === undefined) return []; const seen = new Set(); diff --git a/gitnexus/src/core/ingestion/languages/cpp/adl.ts b/gitnexus/src/core/ingestion/languages/cpp/adl.ts index 565c23125..7a1bd9f66 100644 --- a/gitnexus/src/core/ingestion/languages/cpp/adl.ts +++ b/gitnexus/src/core/ingestion/languages/cpp/adl.ts @@ -108,6 +108,38 @@ const argInfoBySite = new Map(); const noAdlSites = new Set(); const classToNamespaceQualifiedName = new Map(); +/** + * Per-`filePath` index of the site keys this file contributed to + * `argInfoBySite` / `noAdlSites`, kept in **strict lockstep** with those two + * maps (#1983 perf). Without it, `collectCppAdlSideChannel(filePath)` had to + * scan the ENTIRE module-level maps (every site of every file the worker + * parsed in the current sub-batch) and `parseSiteKey` each entry just to pick + * out one file's slice — O(F²) per sub-batch (~100M `parseSiteKey` calls + * across the Linux kernel). These indexes turn collect into + * O(entries-for-this-file). + * + * Lockstep invariant: a key is pushed here at most once, exactly when it is + * first inserted into the corresponding map, and both indexes are cleared + * wherever `argInfoBySite` / `noAdlSites` are cleared (`clearCppAdlState` and + * the per-file restore in `applyCppAdlSideChannel`). The "first insert only" + * guard mirrors the maps' own de-dup (`Map.set` / `Set.add` are idempotent on + * the key), so iterating an index yields each of this file's keys exactly once + * — byte-identical to the old filtered full scan. + */ +const argInfoSiteKeysByFile = new Map(); +const noAdlSiteKeysByFile = new Map(); + +/** Push `key` into the per-file index `idx[filePath]` (creating the bucket on + * first use). Callers guard against duplicate keys so each key appears once. */ +function pushFileSiteKey(idx: Map, filePath: string, key: string): void { + let keys = idx.get(filePath); + if (keys === undefined) { + keys = []; + idx.set(filePath, keys); + } + keys.push(key); +} + /** * ADL candidate index — built **once** per pipeline run from * `(scopes, parsedFiles)` and reused by every call site. @@ -370,13 +402,94 @@ export function markCppAdlSiteArgs( col: number, args: readonly CppAdlArgInfo[], ): void { - argInfoBySite.set(siteKey(filePath, line, col), args); + const key = siteKey(filePath, line, col); + // Lockstep with `argInfoSiteKeysByFile`: index the key only on first insert + // (a re-mark overwrites the value but must NOT duplicate the index entry). + if (!argInfoBySite.has(key)) pushFileSiteKey(argInfoSiteKeysByFile, filePath, key); + argInfoBySite.set(key, args); } /** Mark a call site as ADL-suppressed (function child wrapped in * `parenthesized_expression`, e.g. `(f)(s)`). */ export function markCppAdlSiteNoAdl(filePath: string, line: number, col: number): void { - noAdlSites.add(siteKey(filePath, line, col)); + const key = siteKey(filePath, line, col); + // Lockstep with `noAdlSiteKeysByFile`: index the key only on first insert. + if (!noAdlSites.has(key)) pushFileSiteKey(noAdlSiteKeysByFile, filePath, key); + noAdlSites.add(key); +} + +/** + * Plain-data, JSON-serializable snapshot of the per-file ADL capture state + * (`argInfoBySite` entries for this file + `noAdlSites` keys for this file). + * Carried on `ParsedFile.captureSideChannel` across the worker→main boundary + * (#1983); the call-site key's `line:col` are stored per-entry so the full + * `filePath:line:col` key can be reconstructed without parsing. + */ +export interface CppAdlSideChannel { + /** Per-call-site arg info: `[line, col, args]` for sites in this file. */ + readonly argInfoBySite: readonly [number, number, readonly CppAdlArgInfo[]][]; + /** ADL-suppressed sites in this file: `[line, col]`. */ + readonly noAdlSites: readonly [number, number][]; +} + +const SITE_KEY_RE = /^(.*):(\d+):(\d+)$/; + +/** Split a `filePath:line:col` site key, tolerating colons in the path. */ +function parseSiteKey(key: string): { filePath: string; line: number; col: number } | undefined { + const m = SITE_KEY_RE.exec(key); + if (m === null) return undefined; + return { filePath: m[1], line: Number(m[2]), col: Number(m[3]) }; +} + +/** + * Snapshot this file's ADL capture state for the worker→main side-channel. + * + * Uses the per-file `argInfoSiteKeysByFile` / `noAdlSiteKeysByFile` indexes to + * touch only THIS file's entries — O(entries-for-this-file) — instead of the + * old O(all-entries) full scan over `argInfoBySite` / `noAdlSites` (#1983). + * The output order, and therefore the serialized JSON shape, is byte-identical + * to the old filtered scan: the index records keys in the same insertion order + * the maps' own iteration would have yielded for this file, and each key is + * indexed exactly once (mark guards on first insert), so the same per-file + * subsequence is produced. + * + * `parseSiteKey` is still used to recover `line:col` from each key, but now + * only for this file's keys (a bounded handful), never for the whole batch. + */ +export function collectCppAdlSideChannel(filePath: string): CppAdlSideChannel { + const args: [number, number, readonly CppAdlArgInfo[]][] = []; + for (const key of argInfoSiteKeysByFile.get(filePath) ?? []) { + const value = argInfoBySite.get(key); + const parsed = parseSiteKey(key); + if (value !== undefined && parsed !== undefined) { + args.push([parsed.line, parsed.col, value]); + } + } + const noAdl: [number, number][] = []; + for (const key of noAdlSiteKeysByFile.get(filePath) ?? []) { + const parsed = parseSiteKey(key); + if (parsed !== undefined) { + noAdl.push([parsed.line, parsed.col]); + } + } + return { argInfoBySite: args, noAdlSites: noAdl }; +} + +/** Restore this file's ADL capture state from the side-channel (no parse). + * Keeps the per-file site-key indexes in lockstep with `argInfoBySite` / + * `noAdlSites` (first-insert-only) so a later `collectCppAdlSideChannel` on + * the same process would still produce a correct, duplicate-free snapshot. */ +export function applyCppAdlSideChannel(filePath: string, data: CppAdlSideChannel): void { + for (const [line, col, value] of data.argInfoBySite) { + const key = siteKey(filePath, line, col); + if (!argInfoBySite.has(key)) pushFileSiteKey(argInfoSiteKeysByFile, filePath, key); + argInfoBySite.set(key, value); + } + for (const [line, col] of data.noAdlSites) { + const key = siteKey(filePath, line, col); + if (!noAdlSites.has(key)) pushFileSiteKey(noAdlSiteKeysByFile, filePath, key); + noAdlSites.add(key); + } } /** Clear ADL state. Called from `cppScopeResolver.loadResolutionConfig` @@ -385,6 +498,11 @@ export function markCppAdlSiteNoAdl(filePath: string, line: number, col: number) export function clearCppAdlState(): void { argInfoBySite.clear(); noAdlSites.clear(); + // Lockstep: the per-file site-key indexes mirror argInfoBySite/noAdlSites and + // MUST be cleared together — a stale index would resurrect a prior pass's + // (or prior file's, after a re-key) keys into the next snapshot. + argInfoSiteKeysByFile.clear(); + noAdlSiteKeysByFile.clear(); classToNamespaceQualifiedName.clear(); adlIndex = undefined; adlIndexSource = undefined; diff --git a/gitnexus/src/core/ingestion/languages/cpp/capture-side-channel.ts b/gitnexus/src/core/ingestion/languages/cpp/capture-side-channel.ts new file mode 100644 index 000000000..ad9766d46 --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/cpp/capture-side-channel.ts @@ -0,0 +1,123 @@ +/** + * C++ capture-time side-channel serialization (#1983). + * + * `emitCppScopeCaptures` populates several MODULE-LEVEL maps as a side effect + * that are NOT part of the returned `ParsedFile`'s scopes/defs: + * + * - `argInfoBySite` / `noAdlSites` (adl.ts) + * - `inlineNamespaceRangesByFile` (inline-namespaces.ts) + * - `fileLocalNames` / `anonymousNamespaceRangesByFile` (file-local-linkage.ts) + * - `dependentBasesByFile` / `dependentPackBaseClassesByFile` (two-phase-lookup.ts) + * + * On the worker path those maps are filled in the WORKER process and lost + * across the worker→main MessageChannel (and the disk-backed parsedfile-store), + * because scope-resolution reuses the serialized `ParsedFile` and SKIPS the + * main-thread re-extraction — the entire point of #1983 is to avoid a + * main-thread tree-sitter re-parse on huge `.h`/`.cpp` repos (the OOM). + * + * This module snapshots the per-file slice of those maps into a plain, + * JSON-serializable object (carried on `ParsedFile.captureSideChannel`) and + * restores it on the main thread WITHOUT any parse. It is the data-only + * replacement for the removed re-parse `replayCaptureSideChannel` hook. + * + * The derived state each `populateOwners` / `populateWorkspaceOwners` pass + * builds (resolved scope-id Sets, `dependentBaseNodeIds`, etc.) is recomputed + * on the main thread from these restored capture-time maps, so only the + * capture-time maps need to cross the boundary. + */ + +import type { ParsedFile } from 'gitnexus-shared'; +import { collectCppAdlSideChannel, applyCppAdlSideChannel, type CppAdlSideChannel } from './adl.js'; +import { + collectCppInlineNamespaceSideChannel, + applyCppInlineNamespaceSideChannel, +} from './inline-namespaces.js'; +import { + collectCppFileLocalSideChannel, + applyCppFileLocalSideChannel, + type CppFileLocalSideChannel, +} from './file-local-linkage.js'; +import { + collectCppTwoPhaseSideChannel, + applyCppTwoPhaseSideChannel, + type CppTwoPhaseSideChannel, +} from './two-phase-lookup.js'; +import { + applyCppMemberLookupSideChannel, + collectCppMemberLookupSideChannel, + type CppMemberLookupSideChannel, +} from './member-lookup.js'; + +/** + * Plain JSON-serializable composite of every C++ capture-time side-channel + * slice for one file. Carried opaquely on `ParsedFile.captureSideChannel`. + */ +export interface CppCaptureSideChannel { + /** + * Discriminant tag — the single generic `ParsedFile.captureSideChannel` + * field is shared with C (`{ kind: 'c' }`) and Kotlin (`{ kind: 'kotlin' }`). + * `applyCppCaptureSideChannel` checks this first so a foreign-language + * payload reaching the C++ apply (or vice-versa) is cleanly ignored. In + * practice apply only runs for the matching provider (one language per file), + * but the tag makes it robust and consistent with the C/Kotlin snapshots. + */ + readonly kind: 'cpp'; + readonly adl: CppAdlSideChannel; + /** Inline-namespace source-range keys recorded for this file. */ + readonly inlineNamespaceRanges: readonly string[]; + readonly fileLocal: CppFileLocalSideChannel; + readonly twoPhase: CppTwoPhaseSideChannel; + readonly memberLookup: CppMemberLookupSideChannel; +} + +/** + * `LanguageProvider.collectCaptureSideChannel` implementation for C++. + * Returns `undefined` when this file recorded no side-channel state at all, so + * the produced `ParsedFile` carries the field only when there's data to ship. + */ +export function collectCppCaptureSideChannel(filePath: string): CppCaptureSideChannel | undefined { + const adl = collectCppAdlSideChannel(filePath); + const inlineNamespaceRanges = collectCppInlineNamespaceSideChannel(filePath); + const fileLocal = collectCppFileLocalSideChannel(filePath); + const twoPhase = collectCppTwoPhaseSideChannel(filePath); + const memberLookup = collectCppMemberLookupSideChannel(filePath); + + const isEmpty = + adl.argInfoBySite.length === 0 && + adl.noAdlSites.length === 0 && + inlineNamespaceRanges.length === 0 && + fileLocal.fileLocalNames.length === 0 && + fileLocal.anonymousNamespaceRanges.length === 0 && + twoPhase.dependentBases.length === 0 && + twoPhase.dependentPackBaseClasses.length === 0 && + memberLookup.baseEdges.length === 0 && + memberLookup.memberUsings.length === 0; + if (isEmpty) return undefined; + + return { kind: 'cpp', adl, inlineNamespaceRanges, fileLocal, twoPhase, memberLookup }; +} + +/** + * `ScopeResolver.applyCaptureSideChannel` implementation for C++. Reads the + * worker-serialized snapshot from `parsed.captureSideChannel` and writes it + * back into the module-level maps. Tolerant of `undefined` (file carried no + * data) and of an unexpected shape (defensive — never throws on a malformed + * snapshot). Does NO tree-sitter parse. + */ +export function applyCppCaptureSideChannel(parsed: ParsedFile): void { + const data = parsed.captureSideChannel as CppCaptureSideChannel | undefined; + if (data === undefined || data === null || typeof data !== 'object') return; + // Discriminant guard — the generic `captureSideChannel` field is shared + // with C (`{ kind: 'c' }`) and Kotlin (`{ kind: 'kotlin' }`); cleanly + // ignore a non-C++ payload rather than mis-applying it. + if (data.kind !== 'cpp') return; + if (data.adl !== undefined) applyCppAdlSideChannel(parsed.filePath, data.adl); + if (data.inlineNamespaceRanges !== undefined) { + applyCppInlineNamespaceSideChannel(parsed.filePath, data.inlineNamespaceRanges); + } + if (data.fileLocal !== undefined) applyCppFileLocalSideChannel(parsed.filePath, data.fileLocal); + if (data.twoPhase !== undefined) applyCppTwoPhaseSideChannel(parsed.filePath, data.twoPhase); + if (data.memberLookup !== undefined) { + applyCppMemberLookupSideChannel(parsed.filePath, data.memberLookup); + } +} diff --git a/gitnexus/src/core/ingestion/languages/cpp/captures.ts b/gitnexus/src/core/ingestion/languages/cpp/captures.ts index 265db8d5d..883f571c5 100644 --- a/gitnexus/src/core/ingestion/languages/cpp/captures.ts +++ b/gitnexus/src/core/ingestion/languages/cpp/captures.ts @@ -20,6 +20,7 @@ import { markCppDependentBase, markCppDependentPackBase } from './two-phase-look import { markCppAdlSiteArgs, markCppAdlSiteNoAdl, type CppAdlArgInfo } from './adl.js'; import { markCppInlineNamespaceRange } from './inline-namespaces.js'; import { extractCppTemplateConstraints } from './constraint-extractor.js'; +import { captureCppMemberLookupFacts } from './member-lookup.js'; export function emitCppScopeCaptures( sourceText: string, @@ -464,6 +465,7 @@ export function emitCppScopeCaptures( // and the resolver can suppress unqualified-call binding to those // bases per ISO C++ two-phase lookup. detectCppDependentBases(tree.rootNode, filePath); + captureCppMemberLookupFacts(tree.rootNode, filePath); return out; } diff --git a/gitnexus/src/core/ingestion/languages/cpp/file-local-linkage.ts b/gitnexus/src/core/ingestion/languages/cpp/file-local-linkage.ts index dd5fc8c0a..c6b383bf0 100644 --- a/gitnexus/src/core/ingestion/languages/cpp/file-local-linkage.ts +++ b/gitnexus/src/core/ingestion/languages/cpp/file-local-linkage.ts @@ -95,6 +95,47 @@ export function isCppAnonymousNamespaceScope(scopeId: ScopeId): boolean { return anonymousNamespaceScopeIds.has(scopeId); } +/** + * Plain-data, JSON-serializable snapshot of the per-file capture-time + * file-local-linkage state. Carried on `ParsedFile.captureSideChannel` across + * the worker→main boundary (#1983). The derived sets (`nonGloballyVisibleNodeIds`, + * `anonymousNamespaceScopeIds`) are recomputed by `populateCppNonGloballyVisible` + * / `populateCppAnonymousNamespaceScopes` during `populateOwners`, so only the + * two capture-time maps cross the boundary. + */ +export interface CppFileLocalSideChannel { + /** File-local symbol names (static / anonymous-namespace) in this file. */ + readonly fileLocalNames: readonly string[]; + /** Anonymous-namespace source-range keys recorded for this file. */ + readonly anonymousNamespaceRanges: readonly string[]; +} + +/** Snapshot this file's file-local-linkage capture state for the side-channel. */ +export function collectCppFileLocalSideChannel(filePath: string): CppFileLocalSideChannel { + const names = fileLocalNames.get(filePath); + const anon = anonymousNamespaceRangesByFile.get(filePath); + return { + fileLocalNames: names === undefined ? [] : [...names], + anonymousNamespaceRanges: anon === undefined ? [] : [...anon], + }; +} + +/** Restore this file's file-local-linkage capture state from the side-channel. */ +export function applyCppFileLocalSideChannel( + filePath: string, + data: CppFileLocalSideChannel, +): void { + for (const name of data.fileLocalNames) markFileLocal(filePath, name); + if (data.anonymousNamespaceRanges.length > 0) { + let set = anonymousNamespaceRangesByFile.get(filePath); + if (set === undefined) { + set = new Set(); + anonymousNamespaceRangesByFile.set(filePath, set); + } + for (const r of data.anonymousNamespaceRanges) set.add(r); + } +} + /** Clear tracked file-local names (call at start of each resolution pass). */ export function clearFileLocalNames(): void { fileLocalNames.clear(); @@ -235,11 +276,37 @@ export function isCppDefGloballyVisible(filePath: string, nodeId: string): boole * does, mirror this filter or harden registration so class/namespace * members never enter `localDefs` unqualified. */ +/** + * Per-pass memo: `moduleScope` → owning `ParsedFile`, keyed on the + * `parsedFiles` array identity. The shared finalize Phase-4 loop calls + * `expandsWildcardTo` (→ this) ONCE PER RESOLVED `#include` edge with the same + * `parsedFiles` reference; the old `parsedFiles.find(...)` was therefore O(F) + * per edge → O(R·F) overall (at kernel scale the ~25–30k `.h` headers are + * classified C++, so this fires hard — the C twin in `c/static-linkage.ts`). + * Building the lookup once collapses it to O(R+F). `WeakMap`-keyed so it is + * reclaimed with the pass (no cross-pass staleness; mirrors + * {@link clearFileLocalNames}). + */ +const moduleScopeIndexByPass = new WeakMap>(); + +function moduleScopeIndex(parsedFiles: readonly ParsedFile[]): Map { + let index = moduleScopeIndexByPass.get(parsedFiles); + if (index === undefined) { + index = new Map(); + // First-wins to preserve `Array.find` semantics (returns the first match). + for (const p of parsedFiles) { + if (!index.has(p.moduleScope)) index.set(p.moduleScope, p); + } + moduleScopeIndexByPass.set(parsedFiles, index); + } + return index; +} + export function expandCppWildcardNames( targetModuleScope: ScopeId, parsedFiles: readonly ParsedFile[], ): readonly string[] { - const target = parsedFiles.find((p) => p.moduleScope === targetModuleScope); + const target = moduleScopeIndex(parsedFiles).get(targetModuleScope); if (target === undefined) return []; // Build nodeId → owning Scope map from the structural scope tree. diff --git a/gitnexus/src/core/ingestion/languages/cpp/import-decomposer.ts b/gitnexus/src/core/ingestion/languages/cpp/import-decomposer.ts index eb6b252ce..4b0820589 100644 --- a/gitnexus/src/core/ingestion/languages/cpp/import-decomposer.ts +++ b/gitnexus/src/core/ingestion/languages/cpp/import-decomposer.ts @@ -5,6 +5,13 @@ import { nodeToCapture, syntheticCapture, type SyntaxNode } from '../../utils/as * Decompose a `preproc_include` node into a CaptureMatch with structured * import captures. C++ #include maps to a wildcard import (all symbols * from the header are visible). Identical to C's splitCInclude. + * + * Only literal include paths are emitted as import sources: + * #include → system_lib_string + * #include "User.h" → string_literal + * A computed include like `#include HEADER_MACRO` carries an `identifier` + * path node (the macro name, not a header path); we skip it so it never + * becomes a garbage literal import source (issue #1919 F5). */ export function splitCppInclude(node: SyntaxNode): CaptureMatch | null { const pathNode = node.childForFieldName?.('path') ?? null; @@ -21,7 +28,13 @@ export function splitCppInclude(node: SyntaxNode): CaptureMatch | null { return buildIncludeCapture(node, pathNode); } -function buildIncludeCapture(node: SyntaxNode, pathNode: SyntaxNode): CaptureMatch { +function buildIncludeCapture(node: SyntaxNode, pathNode: SyntaxNode): CaptureMatch | null { + // Skip computed includes (`#include MACRO`) — the path is an `identifier`, + // not a literal header path. Emitting it would create a garbage import. + if (pathNode.type !== 'string_literal' && pathNode.type !== 'system_lib_string') { + return null; + } + let raw: string; if (pathNode.type === 'string_literal') { const content = pathNode.namedChildren.find((c) => c.type === 'string_content'); @@ -60,6 +73,12 @@ function buildIncludeCapture(node: SyntaxNode, pathNode: SyntaxNode): CaptureMat */ export function splitCppUsingDecl(node: SyntaxNode): CaptureMatch | null { if (node.type !== 'using_declaration') return null; + // A class-scope `using Base::member;` changes the derived class's member + // lookup set; it is not a namespace import. The C++ member-lookup sidecar + // captures it separately, so suppress import decomposition here. + for (let parent = node.parent; parent !== null; parent = parent.parent) { + if (parent.type === 'class_specifier' || parent.type === 'struct_specifier') return null; + } // Check for "namespace" keyword among anonymous children let hasNamespaceKeyword = false; diff --git a/gitnexus/src/core/ingestion/languages/cpp/index.ts b/gitnexus/src/core/ingestion/languages/cpp/index.ts index c4d208d76..4bc52dd32 100644 --- a/gitnexus/src/core/ingestion/languages/cpp/index.ts +++ b/gitnexus/src/core/ingestion/languages/cpp/index.ts @@ -14,3 +14,7 @@ export { clearFileLocalNames, expandCppWildcardNames, } from './file-local-linkage.js'; +export { + collectCppCaptureSideChannel, + applyCppCaptureSideChannel, +} from './capture-side-channel.js'; diff --git a/gitnexus/src/core/ingestion/languages/cpp/inline-namespaces.ts b/gitnexus/src/core/ingestion/languages/cpp/inline-namespaces.ts index eec44c68f..c7280bf46 100644 --- a/gitnexus/src/core/ingestion/languages/cpp/inline-namespaces.ts +++ b/gitnexus/src/core/ingestion/languages/cpp/inline-namespaces.ts @@ -61,6 +61,30 @@ export function markCppInlineNamespaceRange(filePath: string, range: RangeKey): set.add(rangeKey(range)); } +/** Snapshot this file's captured inline-namespace ranges for the worker→main + * side-channel (#1983). `populateCppInlineNamespaceScopes` (in `populateOwners`) + * later resolves these range keys to ScopeIds on the main thread, so only the + * capture-time ranges need to cross the boundary. Returns the rangeKey strings + * as a plain array (empty when this file recorded none). */ +export function collectCppInlineNamespaceSideChannel(filePath: string): readonly string[] { + const set = inlineNamespaceRangesByFile.get(filePath); + return set === undefined ? [] : [...set]; +} + +/** Restore this file's captured inline-namespace ranges from the side-channel. */ +export function applyCppInlineNamespaceSideChannel( + filePath: string, + ranges: readonly string[], +): void { + if (ranges.length === 0) return; + let set = inlineNamespaceRangesByFile.get(filePath); + if (set === undefined) { + set = new Set(); + inlineNamespaceRangesByFile.set(filePath, set); + } + for (const r of ranges) set.add(r); +} + /** Clear all inline-namespace state. Called from `clearFileLocalNames`. */ export function clearCppInlineNamespaces(): void { inlineNamespaceRangesByFile.clear(); diff --git a/gitnexus/src/core/ingestion/languages/cpp/member-lookup.ts b/gitnexus/src/core/ingestion/languages/cpp/member-lookup.ts new file mode 100644 index 000000000..a681c4952 --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/cpp/member-lookup.ts @@ -0,0 +1,616 @@ +import type { ParsedFile, ReferenceSite, SymbolDefinition } from 'gitnexus-shared'; +import type { KnowledgeGraph } from '../../../graph/types.js'; +import type { GraphNodeLookup } from '../../scope-resolution/graph-bridge/node-lookup.js'; +import { resolveDefGraphId } from '../../scope-resolution/graph-bridge/ids.js'; +import type { ScopeResolutionIndexes } from '../../model/scope-resolution-indexes.js'; +import type { SemanticModel } from '../../model/semantic-model.js'; +import type { ReceiverMemberResolution } from '../../scope-resolution/contract/scope-resolver.js'; +import { buildMro, defaultLinearize } from '../../scope-resolution/passes/mro.js'; +import { + isOverloadAmbiguousAfterNormalization, + narrowOverloadCandidates, +} from '../../scope-resolution/passes/overload-narrowing.js'; +import { isClassLike } from '../../scope-resolution/scope/walkers.js'; +import type { SyntaxNode } from '../../utils/ast-helpers.js'; +import { cppConstraintCompatibility } from './constraint-filter.js'; +import { cppConversionRank } from './conversion-rank.js'; + +interface CapturedBaseEdge { + readonly childName: string; + readonly childQualifiedName?: string; + readonly baseName: string; + readonly baseQualifiedName?: string; + readonly isVirtual: boolean; +} + +interface CapturedMemberUsing { + readonly childName: string; + readonly childQualifiedName?: string; + readonly baseName: string; + readonly baseQualifiedName?: string; + readonly memberName: string; +} + +export interface CppMemberLookupSideChannel { + readonly baseEdges: readonly CapturedBaseEdge[]; + readonly memberUsings: readonly CapturedMemberUsing[]; +} + +const capturedByFile = new Map(); +let directParentsByDefId = new Map(); +let virtualEdges = new Set(); +let ancestorsByDefId = new Map>(); +let memberUsingsByDefId = new Map< + string, + readonly { readonly baseDefId: string; readonly memberName: string }[] +>(); +let inheritedLookupCache = new Map(); + +const MAX_INHERITANCE_VISITS = 4096; + +type CachedInheritedLookup = + | { readonly kind: 'none' } + | { readonly kind: 'candidates'; readonly definitions: readonly SymbolDefinition[] } + | { readonly kind: 'ambiguous'; readonly candidateIds: readonly string[] }; + +export function clearCppMemberLookupState(): void { + capturedByFile.clear(); + directParentsByDefId = new Map(); + virtualEdges = new Set(); + ancestorsByDefId = new Map(); + memberUsingsByDefId = new Map(); + inheritedLookupCache = new Map(); +} + +export function captureCppMemberLookupFacts(root: SyntaxNode, filePath: string): void { + const baseEdges: CapturedBaseEdge[] = []; + const memberUsings: CapturedMemberUsing[] = []; + const stack: SyntaxNode[] = [root]; + + while (stack.length > 0) { + const node = stack.pop()!; + if (node.type === 'class_specifier' || node.type === 'struct_specifier') { + const childName = classNameOf(node); + const childQualifiedName = classQualifiedNameOf(node); + if (childName !== '') { + const baseClause = directChildOfType(node, 'base_class_clause'); + if (baseClause !== null) { + captureBaseEdges(baseClause, childName, childQualifiedName, baseEdges); + } + const body = directChildOfType(node, 'field_declaration_list'); + if (body !== null) { + for (let i = 0; i < body.namedChildCount; i++) { + const child = body.namedChild(i); + if (child?.type !== 'using_declaration') continue; + const parsed = parseMemberUsing(child, childName, childQualifiedName); + if (parsed !== undefined) memberUsings.push(parsed); + } + } + } + } + for (let i = 0; i < node.childCount; i++) { + const child = node.child(i); + if (child !== null) stack.push(child); + } + } + + if (baseEdges.length === 0 && memberUsings.length === 0) { + capturedByFile.delete(filePath); + } else { + capturedByFile.set(filePath, { baseEdges, memberUsings }); + } +} + +export function collectCppMemberLookupSideChannel(filePath: string): CppMemberLookupSideChannel { + return capturedByFile.get(filePath) ?? { baseEdges: [], memberUsings: [] }; +} + +export function applyCppMemberLookupSideChannel( + filePath: string, + data: CppMemberLookupSideChannel, +): void { + if (!Array.isArray(data.baseEdges) || !Array.isArray(data.memberUsings)) return; + if (data.baseEdges.length === 0 && data.memberUsings.length === 0) { + capturedByFile.delete(filePath); + return; + } + capturedByFile.set(filePath, { + baseEdges: data.baseEdges.slice(), + memberUsings: data.memberUsings.slice(), + }); +} + +export function buildCppMemberLookupMro( + graph: KnowledgeGraph, + parsedFiles: readonly ParsedFile[], + nodeLookup: GraphNodeLookup, +): Map { + populateResolvedHierarchy(graph, parsedFiles, nodeLookup); + return buildMro(graph, parsedFiles, nodeLookup, defaultLinearize); +} + +export function resolveCppReceiverMember( + ownerDef: SymbolDefinition, + memberName: string, + callsite: ReferenceSite, + _scopes: ScopeResolutionIndexes, + model: SemanticModel, +): ReceiverMemberResolution | undefined { + if (callsite.kind !== 'call') return undefined; + const ownMethods = model.methods.lookupAllByOwner(ownerDef.nodeId, memberName); + const introduced = introducedDefinitions(ownerDef.nodeId, memberName, model); + + if (introduced.length > 0) { + return chooseOverload(uniqueDefinitions([...ownMethods, ...introduced]), callsite); + } + + // Direct declarations hide every base declaration. Let the shared path + // retain its existing overload/static filtering for this common case. + if (ownMethods.length > 0) return undefined; + + const lookup = inheritedLookupSet(ownerDef.nodeId, memberName, model); + if (lookup.kind === 'none') return undefined; + if (lookup.kind === 'ambiguous') return lookup; + return chooseOverload(lookup.definitions, callsite); +} + +interface MemberOccurrence { + readonly ownerDefId: string; + readonly definitions: readonly SymbolDefinition[]; + readonly path: readonly string[]; + readonly virtualAnchor?: string; +} + +function collectInheritedOccurrences( + ownerDefId: string, + memberName: string, + model: SemanticModel, + path: readonly string[], + virtualAnchor: string | undefined, + active: Set, + budget: { remaining: number; truncated: boolean }, +): MemberOccurrence[] { + if (budget.remaining <= 0) { + budget.truncated = true; + return []; + } + budget.remaining--; + if (active.has(ownerDefId)) return []; + const nextActive = new Set(active); + nextActive.add(ownerDefId); + + const definitions = uniqueDefinitions([ + ...model.methods.lookupAllByOwner(ownerDefId, memberName), + ...introducedDefinitions(ownerDefId, memberName, model), + ]); + if (definitions.length > 0) { + return [{ ownerDefId, definitions, path, virtualAnchor }]; + } + + const results: MemberOccurrence[] = []; + for (const parentDefId of directParentsByDefId.get(ownerDefId) ?? []) { + const edgeKey = `${ownerDefId}\0${parentDefId}`; + results.push( + ...collectInheritedOccurrences( + parentDefId, + memberName, + model, + [...path, parentDefId], + virtualEdges.has(edgeKey) ? parentDefId : virtualAnchor, + nextActive, + budget, + ), + ); + } + return results; +} + +function inheritedLookupSet( + ownerDefId: string, + memberName: string, + model: SemanticModel, +): CachedInheritedLookup { + const cacheKey = `${ownerDefId}\0${memberName}`; + const cached = inheritedLookupCache.get(cacheKey); + if (cached !== undefined) return cached; + + const budget = { remaining: MAX_INHERITANCE_VISITS, truncated: false }; + const occurrences = collectInheritedOccurrences( + ownerDefId, + memberName, + model, + [], + undefined, + new Set(), + budget, + ); + if (budget.truncated) { + const conservative: CachedInheritedLookup = { + kind: 'ambiguous', + candidateIds: uniqueDefinitions(occurrences.flatMap((entry) => entry.definitions)).map( + (definition) => definition.nodeId, + ), + }; + inheritedLookupCache.set(cacheKey, conservative); + return conservative; + } + if (occurrences.length === 0) { + const none: CachedInheritedLookup = { kind: 'none' }; + inheritedLookupCache.set(cacheKey, none); + return none; + } + + // A declaration can dominate another lookup set only when the latter is + // reached through a shared virtual subobject. Ordinary ancestry alone is + // insufficient: declarations in one non-virtual branch do not hide members + // reached through a sibling base subobject. + const undominated = occurrences.filter( + (candidate) => + !( + candidate.virtualAnchor !== undefined && + occurrences.some( + (other) => + other.ownerDefId !== candidate.ownerDefId && + isAncestor(candidate.ownerDefId, other.ownerDefId), + ) + ), + ); + const groups = new Map(); + for (const occurrence of undominated) { + const key = + occurrence.virtualAnchor !== undefined + ? `virtual:${occurrence.virtualAnchor}:${occurrence.ownerDefId}` + : `path:${occurrence.path.join('>')}:${occurrence.ownerDefId}`; + const bucket = groups.get(key); + if (bucket === undefined) groups.set(key, [occurrence]); + else bucket.push(occurrence); + } + + let result: CachedInheritedLookup; + if (groups.size !== 1) { + result = { + kind: 'ambiguous', + candidateIds: uniqueDefinitions(undominated.flatMap((entry) => entry.definitions)).map( + (definition) => definition.nodeId, + ), + }; + } else { + result = { + kind: 'candidates', + definitions: groups.values().next().value?.[0]?.definitions ?? [], + }; + } + inheritedLookupCache.set(cacheKey, result); + return result; +} + +function introducedDefinitions( + ownerDefId: string, + memberName: string, + model: SemanticModel, +): SymbolDefinition[] { + const definitions: SymbolDefinition[] = []; + for (const entry of memberUsingsByDefId.get(ownerDefId) ?? []) { + if (entry.memberName !== memberName) continue; + definitions.push(...model.methods.lookupAllByOwner(entry.baseDefId, memberName)); + } + return definitions; +} + +function uniqueDefinitions(definitions: readonly SymbolDefinition[]): SymbolDefinition[] { + return [...new Map(definitions.map((definition) => [definition.nodeId, definition])).values()]; +} + +function chooseOverload( + candidates: readonly SymbolDefinition[], + callsite: ReferenceSite, +): ReceiverMemberResolution | undefined { + if (candidates.length === 0) return undefined; + const narrowed = narrowOverloadCandidates(candidates, callsite.arity, callsite.argumentTypes, { + argumentTypeClasses: callsite.argumentTypeClasses, + conversionRankFn: cppConversionRank, + constraintCompatibility: cppConstraintCompatibility, + }); + if (narrowed.length === 1) return { kind: 'resolved', definition: narrowed[0]! }; + if (narrowed.length > 1 || isOverloadAmbiguousAfterNormalization(narrowed, callsite.arity)) { + return { + kind: 'ambiguous', + candidateIds: narrowed.map((candidate) => candidate.nodeId), + }; + } + return undefined; +} + +function populateResolvedHierarchy( + graph: KnowledgeGraph, + parsedFiles: readonly ParsedFile[], + nodeLookup: GraphNodeLookup, +): void { + const defByGraphId = new Map(); + const defById = new Map(); + const defsByFileAndName = new Map(); + + for (const parsed of parsedFiles) { + for (const def of parsed.localDefs) { + if (!isClassLike(def.type)) continue; + const graphId = resolveDefGraphId(parsed.filePath, def, nodeLookup); + if (graphId === undefined) continue; + defByGraphId.set(graphId, def); + defById.set(def.nodeId, def); + const names = new Set([simpleName(def), definitionQualifiedName(def)]); + for (const name of names) { + if (name === '') continue; + const key = `${parsed.filePath}\0${name}`; + const bucket = defsByFileAndName.get(key); + if (bucket === undefined) defsByFileAndName.set(key, [def]); + else bucket.push(def); + } + } + } + + const parents = new Map(); + for (const rel of graph.iterRelationshipsByType('EXTENDS')) { + const child = defByGraphId.get(rel.sourceId); + const parent = defByGraphId.get(rel.targetId); + if (child === undefined || parent === undefined) continue; + const bucket = parents.get(child.nodeId); + if (bucket === undefined) parents.set(child.nodeId, [parent.nodeId]); + else bucket.push(parent.nodeId); + } + directParentsByDefId = parents; + ancestorsByDefId = buildAncestorClosure(parents); + inheritedLookupCache = new Map(); + + const nextVirtualEdges = new Set(); + const nextUsings = new Map< + string, + { readonly baseDefId: string; readonly memberName: string }[] + >(); + for (const parsed of parsedFiles) { + const captured = capturedByFile.get(parsed.filePath); + if (captured === undefined) continue; + for (const edge of captured.baseEdges) { + if (!edge.isVirtual) continue; + for (const child of matchingChildren( + parsed.filePath, + edge.childName, + edge.childQualifiedName, + defsByFileAndName, + )) { + const parent = findCapturedParent( + parents.get(child.nodeId) ?? [], + edge.baseName, + edge.baseQualifiedName, + defById, + ); + if (parent !== undefined) nextVirtualEdges.add(`${child.nodeId}\0${parent.nodeId}`); + } + } + for (const using of captured.memberUsings) { + const children = matchingChildren( + parsed.filePath, + using.childName, + using.childQualifiedName, + defsByFileAndName, + ); + for (const child of children) { + const baseDef = findCapturedParent( + parents.get(child.nodeId) ?? [], + using.baseName, + using.baseQualifiedName, + defById, + ); + if (baseDef === undefined) continue; + const bucket = nextUsings.get(child.nodeId); + const entry = { baseDefId: baseDef.nodeId, memberName: using.memberName }; + if (bucket === undefined) nextUsings.set(child.nodeId, [entry]); + else bucket.push(entry); + } + } + } + virtualEdges = nextVirtualEdges; + memberUsingsByDefId = nextUsings; +} + +function captureBaseEdges( + baseClause: SyntaxNode, + childName: string, + childQualifiedName: string, + output: CapturedBaseEdge[], +): void { + let segmentStart = 0; + for (let i = 0; i < baseClause.childCount; i++) { + const child = baseClause.child(i); + if (child === null) continue; + if (child.type === ',' || child.text === ',') { + segmentStart = i + 1; + continue; + } + if ( + child.type !== 'type_identifier' && + child.type !== 'template_type' && + child.type !== 'qualified_identifier' + ) { + continue; + } + let isVirtual = false; + for (let j = segmentStart; j < i; j++) { + const modifier = baseClause.child(j); + if (modifier?.text === 'virtual') isVirtual = true; + } + const baseQualifiedName = qualifiedTypeName(child.text); + const baseName = baseQualifiedName.split('.').at(-1) ?? ''; + if (baseName !== '') { + output.push({ + childName, + ...(childQualifiedName !== childName ? { childQualifiedName } : {}), + baseName, + ...(baseQualifiedName !== baseName ? { baseQualifiedName } : {}), + isVirtual, + }); + } + } +} + +function parseMemberUsing( + node: SyntaxNode, + childName: string, + childQualifiedName: string, +): CapturedMemberUsing | undefined { + const qualified = node.namedChildren.find((child) => child.type === 'qualified_identifier'); + if (qualified === undefined) return undefined; + const parts = splitQualifiedSegments(qualified.text); + if (parts.length < 2) return undefined; + const memberName = stripTemplateSuffix(parts.at(-1) ?? ''); + const baseParts = parts.slice(0, -1).map(stripTemplateSuffix).filter(Boolean); + const baseName = baseParts.at(-1) ?? ''; + const baseQualifiedName = baseParts.join('.'); + if (baseName === '' || memberName === '') return undefined; + return { + childName, + ...(childQualifiedName !== childName ? { childQualifiedName } : {}), + baseName, + ...(baseQualifiedName !== baseName ? { baseQualifiedName } : {}), + memberName, + }; +} + +function classNameOf(node: SyntaxNode): string { + const name = node.childForFieldName?.('name'); + return name === null || name === undefined ? '' : trailingIdentifier(name.text); +} + +function classQualifiedNameOf(node: SyntaxNode): string { + const parts = [classNameOf(node)]; + let current = node.parent; + while (current !== null) { + if (current.type === 'class_specifier' || current.type === 'struct_specifier') { + const name = classNameOf(current); + if (name !== '') parts.unshift(name); + } else if (current.type === 'namespace_definition') { + const name = current.childForFieldName?.('name'); + if (name !== null && name !== undefined) { + parts.unshift( + ...splitQualifiedSegments(name.text).map(stripTemplateSuffix).filter(Boolean), + ); + } + } + current = current.parent; + } + return parts.filter(Boolean).join('.'); +} + +function directChildOfType(node: SyntaxNode, type: string): SyntaxNode | null { + for (let i = 0; i < node.namedChildCount; i++) { + const child = node.namedChild(i); + if (child?.type === type) return child; + } + return null; +} + +function trailingIdentifier(value: string): string { + return stripTemplateSuffix(splitQualifiedSegments(value).at(-1) ?? ''); +} + +function qualifiedTypeName(value: string): string { + return splitQualifiedSegments(value).map(stripTemplateSuffix).filter(Boolean).join('.'); +} + +function splitQualifiedSegments(value: string): string[] { + const parts: string[] = []; + let angleDepth = 0; + let segmentStart = 0; + for (let i = 0; i < value.length; i++) { + const char = value[i]; + if (char === '<') angleDepth++; + else if (char === '>' && angleDepth > 0) angleDepth--; + else if (char === ':' && value[i + 1] === ':' && angleDepth === 0) { + const segment = value.slice(segmentStart, i).trim(); + if (segment !== '') parts.push(segment); + segmentStart = i + 2; + i++; + } + } + const tail = value.slice(segmentStart).trim(); + if (tail !== '') parts.push(tail); + return parts; +} + +function stripTemplateSuffix(value: string): string { + const templateStart = value.indexOf('<'); + return (templateStart >= 0 ? value.slice(0, templateStart) : value).trim(); +} + +function simpleName(def: SymbolDefinition): string { + return def.qualifiedName?.split('.').at(-1) ?? ''; +} + +function definitionQualifiedName(def: SymbolDefinition): string { + const name = def.qualifiedName ?? ''; + if (name === '' || def.namespacePrefix === undefined || def.namespacePrefix === '') return name; + return name.startsWith(`${def.namespacePrefix}.`) ? name : `${def.namespacePrefix}.${name}`; +} + +function matchingChildren( + filePath: string, + childName: string, + childQualifiedName: string | undefined, + defsByFileAndName: ReadonlyMap, +): readonly SymbolDefinition[] { + if (childQualifiedName !== undefined) { + const qualified = defsByFileAndName.get(`${filePath}\0${childQualifiedName}`) ?? []; + if (qualified.length > 0) return qualified; + } + const simple = defsByFileAndName.get(`${filePath}\0${childName}`) ?? []; + return simple.length === 1 ? simple : []; +} + +function findCapturedParent( + parentIds: readonly string[], + baseName: string, + baseQualifiedName: string | undefined, + defById: ReadonlyMap, +): SymbolDefinition | undefined { + const candidates = parentIds + .map((id) => defById.get(id)) + .filter((definition): definition is SymbolDefinition => definition !== undefined); + if (baseQualifiedName !== undefined) { + const qualified = candidates.filter((definition) => { + const name = definitionQualifiedName(definition); + return name === baseQualifiedName || name.endsWith(`.${baseQualifiedName}`); + }); + if (qualified.length === 1) return qualified[0]; + return undefined; + } + const simple = candidates.filter((definition) => simpleName(definition) === baseName); + return simple.length === 1 ? simple[0] : undefined; +} + +function buildAncestorClosure( + parents: ReadonlyMap, +): Map> { + const closure = new Map>(); + const visiting = new Set(); + + const ancestorsOf = (defId: string): ReadonlySet => { + const cached = closure.get(defId); + if (cached !== undefined) return cached; + if (visiting.has(defId)) return new Set(); + visiting.add(defId); + const ancestors = new Set(); + for (const parent of parents.get(defId) ?? []) { + ancestors.add(parent); + for (const ancestor of ancestorsOf(parent)) ancestors.add(ancestor); + } + visiting.delete(defId); + closure.set(defId, ancestors); + return ancestors; + }; + + for (const defId of parents.keys()) ancestorsOf(defId); + return closure; +} + +function isAncestor(ancestorDefId: string, descendantDefId: string): boolean { + return ancestorsByDefId.get(descendantDefId)?.has(ancestorDefId) === true; +} diff --git a/gitnexus/src/core/ingestion/languages/cpp/scope-resolver.ts b/gitnexus/src/core/ingestion/languages/cpp/scope-resolver.ts index 459b313c6..3ef89bc07 100644 --- a/gitnexus/src/core/ingestion/languages/cpp/scope-resolver.ts +++ b/gitnexus/src/core/ingestion/languages/cpp/scope-resolver.ts @@ -4,7 +4,6 @@ import { findEnclosingClassDef, } from '../../scope-resolution/scope/walkers.js'; import { SupportedLanguages } from 'gitnexus-shared'; -import { buildMro, defaultLinearize } from '../../scope-resolution/passes/mro.js'; import { populateClassOwnedMembers, tagNamespacePrefixes, @@ -30,6 +29,7 @@ import { isCppDependentBaseMember, } from './two-phase-lookup.js'; import { populateCppAssociatedNamespaces, clearCppAdlState, pickCppAdlCandidates } from './adl.js'; +import { applyCppCaptureSideChannel } from './capture-side-channel.js'; import { clearCppInlineNamespaces, populateCppInlineNamespaceScopes, @@ -41,6 +41,45 @@ import { clearCppUserDefinedConversions, populateCppUserDefinedConversions, } from './user-defined-conversions.js'; +import { + buildCppMemberLookupMro, + clearCppMemberLookupState, + resolveCppReceiverMember, +} from './member-lookup.js'; + +/** + * Per-pass memo of the augmented `#include`-resolution file set + * (`allFilePaths` ∪ header paths), keyed on the two stable source sets. + * `resolveImportTarget` is called once per C++ `#include`; the old code rebuilt + * a fresh ~F-entry `Set` on every call AND defeated the shared + * `resolveCImportTarget` suffix-index memo (in `c/import-target.ts`) by handing + * it a new set identity each time. Both inputs are stable per pass, so the + * union is built once and reused. `WeakMap`-keyed → reclaimed with the pass. + * (Twin of the C resolver's `augmentedFilePaths`.) + */ +const augmentedPathsByPass = new WeakMap< + ReadonlySet, + WeakMap, ReadonlySet> +>(); + +function augmentedFilePaths( + allFilePaths: ReadonlySet, + headerPaths: ReadonlySet, +): ReadonlySet { + let byHeaders = augmentedPathsByPass.get(allFilePaths); + if (byHeaders === undefined) { + byHeaders = new WeakMap(); + augmentedPathsByPass.set(allFilePaths, byHeaders); + } + let augmented = byHeaders.get(headerPaths); + if (augmented === undefined) { + const set = new Set(allFilePaths); + for (const h of headerPaths) set.add(h); + augmented = set; + byHeaders.set(headerPaths, augmented); + } + return augmented; +} /** * C++ `ScopeResolver` registered in `SCOPE_RESOLVERS` and consumed by @@ -69,6 +108,7 @@ export const cppScopeResolver: ScopeResolver = { clearCppAdlState(); clearCppInlineNamespaces(); clearCppUserDefinedConversions(); + clearCppMemberLookupState(); return scanCppHeaderFiles(repoPath); }, @@ -78,9 +118,11 @@ export const cppScopeResolver: ScopeResolver = { // detection but are importable from .cpp files via #include. const headerPaths = resolutionConfig as ReadonlySet | undefined; if (headerPaths !== undefined && headerPaths.size > 0) { - const augmented = new Set(allFilePaths); - for (const h of headerPaths) augmented.add(h); - return resolveCppImportTarget(targetRaw, fromFile, augmented); + return resolveCppImportTarget( + targetRaw, + fromFile, + augmentedFilePaths(allFilePaths, headerPaths), + ); } return resolveCppImportTarget(targetRaw, fromFile, allFilePaths); }, @@ -100,8 +142,25 @@ export const cppScopeResolver: ScopeResolver = { // `'unknown'` keeps the candidate, preserving "degrade not lie". constraintCompatibility: cppConstraintCompatibility, - buildMro: (graph, parsedFiles, nodeLookup) => - buildMro(graph, parsedFiles, nodeLookup, defaultLinearize), + buildMro: buildCppMemberLookupMro, + + // Worker-boundary restore (see `ScopeResolver.applyCaptureSideChannel`). + // `emitCppScopeCaptures` records per-file ADL call-site arg shapes + // (`markCppAdlSiteArgs`/`markCppAdlSiteNoAdl`), inline-/anonymous-namespace + // ranges (`markCppInlineNamespaceRange`/`markCppAnonymousNamespaceRange`), + // dependent-base names (`markCppDependentBase`/`markCppDependentPackBase`), + // and file-local linkage (`markFileLocal`) into module-level maps as a SIDE + // EFFECT — none of it is serialized onto the returned ParsedFile's scopes/defs. + // On the worker path those marks are populated in the worker process and lost + // across the MessageChannel / disk store; the main thread reuses the + // serialized ParsedFile and skips `extractParsedFile`, so `populateOwners` + + // the ADL / two-phase-lookup passes would see empty maps and emit zero edges. + // The worker stashed a plain-data snapshot on `parsed.captureSideChannel` via + // `cppProvider.collectCaptureSideChannel`; this restores it into the module + // maps WITHOUT any tree-sitter re-parse (the #1983 fix — the old re-parse + // replay re-OOM'd huge `.h`/`.cpp` repos). The freshly-extracted leg never + // calls this — its marks were just populated in this process. + applyCaptureSideChannel: applyCppCaptureSideChannel, populateOwners: (parsed: ParsedFile) => { populateClassOwnedMembers(parsed); @@ -206,6 +265,7 @@ export const cppScopeResolver: ScopeResolver = { hoistTypeBindingsToModule: true, // Enable receiver-bound explicit-`this` fallback only for C++. resolveThisViaEnclosingClass: true, + resolveReceiverMember: resolveCppReceiverMember, // The `isFileLocalDef` hook on the global free-call fallback names // file-local linkage historically, but semantically gates "logically // invisible cross-file" defs. C++ extends this to also reject class- diff --git a/gitnexus/src/core/ingestion/languages/cpp/two-phase-lookup.ts b/gitnexus/src/core/ingestion/languages/cpp/two-phase-lookup.ts index 7ec55bc9a..739cceeb5 100644 --- a/gitnexus/src/core/ingestion/languages/cpp/two-phase-lookup.ts +++ b/gitnexus/src/core/ingestion/languages/cpp/two-phase-lookup.ts @@ -104,6 +104,56 @@ export function markCppDependentPackBase(filePath: string, className: string): v perFile.add(className); } +/** + * Plain-data, JSON-serializable snapshot of the per-file capture-time + * two-phase-lookup state. Carried on `ParsedFile.captureSideChannel` across the + * worker→main boundary (#1983). The resolved `dependentBaseNodeIds` index is + * rebuilt by `populateCppDependentBases` (workspace pass) after all files have + * their `populateOwners` applied, so only the two capture-time maps cross. + * + * Nested `Map`/`Set` are flattened to arrays here so the snapshot stays plain + * JSON (avoids relying on the parsedfile-store's Map/Set replacer for nested + * structures): `dependentBases` is `[className, [baseName, qualifiers[]][]][]`. + */ +export interface CppTwoPhaseSideChannel { + readonly dependentBases: readonly [string, readonly [string, readonly string[]][]][]; + readonly dependentPackBaseClasses: readonly string[]; +} + +/** Snapshot this file's two-phase-lookup capture state for the side-channel. */ +export function collectCppTwoPhaseSideChannel(filePath: string): CppTwoPhaseSideChannel { + const perFile = dependentBasesByFile.get(filePath); + const dependentBases: [string, [string, string[]][]][] = []; + if (perFile !== undefined) { + for (const [className, bases] of perFile) { + const baseEntries: [string, string[]][] = []; + for (const [baseName, quals] of bases) { + baseEntries.push([baseName, [...quals]]); + } + dependentBases.push([className, baseEntries]); + } + } + const pack = dependentPackBaseClassesByFile.get(filePath); + return { + dependentBases, + dependentPackBaseClasses: pack === undefined ? [] : [...pack], + }; +} + +/** Restore this file's two-phase-lookup capture state from the side-channel. */ +export function applyCppTwoPhaseSideChannel(filePath: string, data: CppTwoPhaseSideChannel): void { + for (const [className, baseEntries] of data.dependentBases) { + for (const [baseName, quals] of baseEntries) { + for (const qualifier of quals) { + markCppDependentBase(filePath, className, baseName, qualifier); + } + } + } + for (const className of data.dependentPackBaseClasses) { + markCppDependentPackBase(filePath, className); + } +} + /** Clear two-phase-lookup state. Called from `clearFileLocalNames`. */ export function clearCppDependentBases(): void { dependentBasesByFile.clear(); diff --git a/gitnexus/src/core/ingestion/languages/csharp/captures.ts b/gitnexus/src/core/ingestion/languages/csharp/captures.ts index b3b720587..3f28e6be9 100644 --- a/gitnexus/src/core/ingestion/languages/csharp/captures.ts +++ b/gitnexus/src/core/ingestion/languages/csharp/captures.ts @@ -86,10 +86,11 @@ export function emitCsharpScopeCaptures( _filePath: string, cachedTree?: unknown, ): readonly CaptureMatch[] { - // Skip the parse when the caller (parse phase's scopeTreeCache) - // already produced a Tree for this source. Cache miss = re-parse, - // same as before. The cachedTree parameter is typed as `unknown` at - // the LanguageProvider contract layer; cast here at the use site. + // Reuse a pre-parsed Tree when the caller passes one via `cachedTree`; a + // miss re-parses. (The cache is currently always empty — its only producer, + // the sequential parser, was removed — so this re-parses in practice.) The + // cachedTree parameter is typed `unknown` at the LanguageProvider contract + // layer; cast here at the use site. let tree = cachedTree as ReturnType['parse']> | undefined; if (tree === undefined) { tree = parseSourceSafe(getCsharpParser(), sourceText, undefined, { diff --git a/gitnexus/src/core/ingestion/languages/csharp/index.ts b/gitnexus/src/core/ingestion/languages/csharp/index.ts index 9f06ce87d..5e8fa99ea 100644 --- a/gitnexus/src/core/ingestion/languages/csharp/index.ts +++ b/gitnexus/src/core/ingestion/languages/csharp/index.ts @@ -68,10 +68,9 @@ * `using static X = Y.Z;`, attributes, and preprocessor-gated * declarations are all recognized correctly. * - * Shadow-harness corpus parity is the authoritative signal for which - * of these matter in practice. The CI parity gate blocks any PR that - * regresses either the legacy or registry-primary run of - * `test/integration/resolvers/csharp.test.ts`. + * The `test/integration/resolvers/csharp.test.ts` resolver suite is the + * authoritative signal for which of these matter in practice; it runs in + * the standard CI test workflow, so a regression blocks the merge. */ export { emitCsharpScopeCaptures } from './captures.js'; diff --git a/gitnexus/src/core/ingestion/languages/csharp/namespace-siblings.ts b/gitnexus/src/core/ingestion/languages/csharp/namespace-siblings.ts index 634afdeec..5dab83afe 100644 --- a/gitnexus/src/core/ingestion/languages/csharp/namespace-siblings.ts +++ b/gitnexus/src/core/ingestion/languages/csharp/namespace-siblings.ts @@ -356,8 +356,9 @@ function extractFileStructure(content: string, cachedTree: unknown): CsharpFileS /** Content + (optional) pre-parsed tree-sitter trees keyed by filePath. * The orchestrator builds `fileContents` from the pipeline's file list; - * `treeCache` is the same `scopeTreeCache` already populated by the - * parse phase, so cache hits avoid a second `parser.parse()`. */ + * `treeCache` is currently always empty (its only producer, the sequential + * parser, was removed), so the providers re-parse. Kept as an extension + * point that would let cache hits avoid a second `parser.parse()`. */ export interface CsharpSiblingInputs { readonly fileContents: ReadonlyMap; readonly treeCache?: { get(filePath: string): unknown }; diff --git a/gitnexus/src/core/ingestion/languages/dart/query.ts b/gitnexus/src/core/ingestion/languages/dart/query.ts index c00cfbeb5..5314b0c8b 100644 --- a/gitnexus/src/core/ingestion/languages/dart/query.ts +++ b/gitnexus/src/core/ingestion/languages/dart/query.ts @@ -23,7 +23,15 @@ */ import Parser from 'tree-sitter'; -import Dart from 'tree-sitter-dart'; +import { SupportedLanguages } from 'gitnexus-shared'; +// `tree-sitter-dart` is an optional/vendored grammar that may be absent on a +// default install. Loaded lazily + guarded via parser-loader rather than +// statically imported: this module is pulled onto the main thread eagerly by +// the scope-resolution registry and the language-provider index, so a top-level +// `import Dart from 'tree-sitter-dart'` would throw ERR_MODULE_NOT_FOUND at +// module-load and crash `analyze` even for repos with no Dart files (#2091, +// #2093). The grammar is only ever needed inside the lazy getters below. +import { getLanguageGrammar } from '../../../tree-sitter/parser-loader.js'; const DART_SCOPE_QUERY = ` ; ── Scopes ─────────────────────────────────────────────────────────────────── @@ -39,6 +47,46 @@ const DART_SCOPE_QUERY = ` (extension_declaration name: (identifier) @declaration.name) @declaration.class (enum_declaration name: (identifier) @declaration.name) @declaration.enum +; ── Declarations — type aliases (old-style + new-style function typedefs) ──── +; Both forms parse as type_alias; the name position differs, and a generic +; parameter list intervenes for the generic variants. Per #1919 review CF2, +; a generic type_parameters node sits between the name and the next anchor, so +; the non-generic adjacency patterns silently drop the generic forms. Four +; standalone patterns (NOT one alternation — the tree-sitter 0.21 hazard drops +; sibling branches) keep the name capture unambiguous and single-match per form: +; non-generic old-style typedef int Cmp(int a, int b); +; children: return-type, NAME, formal_parameter_list +; generic old-style typedef int Cmp(T a, T b); (CF2) +; children: return-type, NAME, type_parameters, formal_parameter_list +; non-generic new-style typedef Pred = bool Function(int); +; children: NAME, "=", function_type +; generic new-style typedef Mapper = T Function(T); +; children: NAME, type_parameters, "=", function_type +; The alias name is the type_identifier immediately before the param list (old) +; or before "=" (new); for the generic forms it is the one immediately before +; the intervening type_parameters. Mirrors Kotlin's @declaration.type_alias +; rule; the generic scope-extractor maps "type_alias" → TypeAlias. +(type_alias + (type_identifier) @declaration.name + . + (formal_parameter_list)) @declaration.type_alias +(type_alias + (type_identifier) @declaration.name + . + (type_parameters) + . + (formal_parameter_list)) @declaration.type_alias +(type_alias + (type_identifier) @declaration.name + . + "=") @declaration.type_alias +(type_alias + (type_identifier) @declaration.name + . + (type_parameters) + . + "=") @declaration.type_alias + ; ── Declarations — top-level functions (parent is program, not method) ─────── (program (function_signature @@ -94,14 +142,19 @@ let _query: Parser.Query | null = null; export function getDartParser(): Parser { if (_parser === null) { _parser = new Parser(); - _parser.setLanguage(Dart as Parameters[0]); + _parser.setLanguage( + getLanguageGrammar(SupportedLanguages.Dart) as Parameters[0], + ); } return _parser; } export function getDartScopeQuery(): Parser.Query { if (_query === null) { - _query = new Parser.Query(Dart as Parameters[0], DART_SCOPE_QUERY); + _query = new Parser.Query( + getLanguageGrammar(SupportedLanguages.Dart) as Parameters[0], + DART_SCOPE_QUERY, + ); } return _query; } diff --git a/gitnexus/src/core/ingestion/languages/go/query.ts b/gitnexus/src/core/ingestion/languages/go/query.ts index 48387582f..4246ee3b4 100644 --- a/gitnexus/src/core/ingestion/languages/go/query.ts +++ b/gitnexus/src/core/ingestion/languages/go/query.ts @@ -53,11 +53,15 @@ const GO_SCOPE_QUERY = ` ;; Declarations — variables (var_declaration (var_spec - name: (identifier) @declaration.name)) @declaration.variable + (identifier) @declaration.name)) @declaration.variable +(var_declaration + (var_spec_list + (var_spec + (identifier) @declaration.name))) @declaration.variable (const_declaration (const_spec - name: (identifier) @declaration.name)) @declaration.const + (identifier) @declaration.name)) @declaration.const (short_var_declaration left: (expression_list (identifier) @declaration.name)) @declaration.variable diff --git a/gitnexus/src/core/ingestion/languages/kotlin.ts b/gitnexus/src/core/ingestion/languages/kotlin.ts index 31116663e..6ee7b1dda 100644 --- a/gitnexus/src/core/ingestion/languages/kotlin.ts +++ b/gitnexus/src/core/ingestion/languages/kotlin.ts @@ -28,6 +28,7 @@ import { kotlinMethodConfig } from '../method-extractors/configs/jvm.js'; import { createVariableExtractor } from '../variable-extractors/generic.js'; import { kotlinVariableConfig } from '../variable-extractors/configs/jvm.js'; import { + collectKotlinCaptureSideChannel, emitKotlinScopeCaptures, interpretKotlinImport, interpretKotlinTypeBinding, @@ -175,6 +176,13 @@ export const kotlinProvider = defineLanguage({ // ── RFC #909 Ring 3: scope-based resolution hooks ── emitScopeCaptures: emitKotlinScopeCaptures, + // Worker-side: snapshot the module-level companion-scope marks + // `emitKotlinScopeCaptures` just populated for this file (`markCompanionScope` + // → `companionScopesByFile`) into plain data on `ParsedFile.captureSideChannel`, + // so the main thread can restore them via `applyCaptureSideChannel` WITHOUT a + // re-parse (#1983). Without this, companion/static dispatch emits no CALLS + // edges on the worker path. See `kotlin/capture-side-channel.ts`. + collectCaptureSideChannel: collectKotlinCaptureSideChannel, interpretImport: interpretKotlinImport, interpretTypeBinding: interpretKotlinTypeBinding, bindingScopeFor: kotlinBindingScopeFor, diff --git a/gitnexus/src/core/ingestion/languages/kotlin/capture-side-channel.ts b/gitnexus/src/core/ingestion/languages/kotlin/capture-side-channel.ts new file mode 100644 index 000000000..375e0f679 --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/kotlin/capture-side-channel.ts @@ -0,0 +1,75 @@ +/** + * Kotlin capture-time side-channel serialization (#1983). + * + * `emitKotlinScopeCaptures` populates one MODULE-LEVEL, per-file map as a side + * effect that is NOT part of the returned `ParsedFile`'s scopes/defs: + * + * - `companionScopesByFile` (companion-scopes.ts) — the `ScopeId`s that came + * from a `companion_object` AST node, recorded via `markCompanionScope` + * from the `@scope.companion` marker capture. + * + * On the worker path that map is filled in the WORKER process and lost across + * the worker→main MessageChannel (and the disk-backed parsedfile-store), + * because scope-resolution reuses the serialized `ParsedFile` and SKIPS the + * main-thread re-extraction (the #1983 fix that avoids a main-thread + * tree-sitter re-parse / OOM on huge repos). The main thread then reads the map + * empty in `isKotlinStaticOnly` / `populateCompanionMembersOnEnclosingClass` + * (owners.ts) — so companion methods aren't identified as static and + * companion/static dispatch emits no CALLS edges. + * + * This module snapshots the per-file slice of that map into a plain, + * JSON-serializable object (carried on `ParsedFile.captureSideChannel`) and + * restores it on the main thread WITHOUT any parse. It mirrors the C++ pattern + * in `cpp/capture-side-channel.ts`. + * + * The single generic `ParsedFile.captureSideChannel` field is shared with C++, + * which is safe because each file is one language (a `.kt` file uses the kotlin + * provider, a `.cpp` file the cpp provider). The payload is self-describing + * (`{ kind: 'kotlin', companionScopes }`) so `applyKotlinCaptureSideChannel` + * only restores kotlin state and ignores a foreign-shaped snapshot. + */ + +import type { ParsedFile, ScopeId } from 'gitnexus-shared'; +import { getCompanionScopesForFile, markCompanionScope } from './companion-scopes.js'; + +/** + * Plain JSON-serializable snapshot of the per-file Kotlin capture-time + * side-channel. Carried opaquely on `ParsedFile.captureSideChannel`. The + * `kind` tag makes the payload self-describing so `apply` can distinguish a + * kotlin snapshot from another language's (C++ shares the same field). + */ +export interface KotlinCaptureSideChannel { + readonly kind: 'kotlin'; + /** Companion-object scope ids recorded for this file. */ + readonly companionScopes: readonly ScopeId[]; +} + +/** + * `LanguageProvider.collectCaptureSideChannel` implementation for Kotlin. + * Returns `undefined` when this file recorded no companion scopes at all, so + * the produced `ParsedFile` carries the field only when there's data to ship. + */ +export function collectKotlinCaptureSideChannel( + filePath: string, +): KotlinCaptureSideChannel | undefined { + const companionScopes = getCompanionScopesForFile(filePath); + if (companionScopes.length === 0) return undefined; + return { kind: 'kotlin', companionScopes }; +} + +/** + * `ScopeResolver.applyCaptureSideChannel` implementation for Kotlin. Reads the + * worker-serialized snapshot from `parsed.captureSideChannel` and re-populates + * the module-level companion-scope map via `markCompanionScope`. Tolerant of + * `undefined` (file carried no data) and of an unexpected / foreign shape + * (defensive — the `kind` tag guards against restoring a non-kotlin payload). + * Does NO tree-sitter parse. + */ +export function applyKotlinCaptureSideChannel(parsed: ParsedFile): void { + const data = parsed.captureSideChannel as KotlinCaptureSideChannel | undefined; + if (data === undefined || data === null || typeof data !== 'object') return; + if (data.kind !== 'kotlin' || !Array.isArray(data.companionScopes)) return; + for (const scopeId of data.companionScopes) { + markCompanionScope(parsed.filePath, scopeId); + } +} diff --git a/gitnexus/src/core/ingestion/languages/kotlin/captures.ts b/gitnexus/src/core/ingestion/languages/kotlin/captures.ts index 2b14266a6..16d2a3db8 100644 --- a/gitnexus/src/core/ingestion/languages/kotlin/captures.ts +++ b/gitnexus/src/core/ingestion/languages/kotlin/captures.ts @@ -39,6 +39,7 @@ export function emitKotlinScopeCaptures( out.push(...synthesizeKotlinSmartCastBindings(tree.rootNode)); out.push(...synthesizeKotlinLambdaBindings(tree.rootNode, returnTypes)); out.push(...synthesizeKotlinInheritanceReferences(tree.rootNode)); + out.push(...synthesizeKotlinSecondaryConstructorDeclarations(tree.rootNode)); for (const match of getKotlinScopeQuery().matches(tree.rootNode)) { const grouped: Record = {}; @@ -87,6 +88,40 @@ export function emitKotlinScopeCaptures( } } + // Callable references (`::method`, `Type::new`, `obj::m`) — F47 (#1919). + // The query captures the referenced member as `@reference.name`, an + // optional receiver type as `@reference.receiver`, and the whole node as + // `@reference.callable`. Rewrite into a call reference so it participates + // in call-graph resolution: a bare `::member` resolves as a free call; + // a `Receiver::member` resolves as a member call against the receiver + // type. The function/constructor is referenced (not invoked), so no + // arity/argument metadata is attached. + if (grouped['@reference.callable'] !== undefined) { + const nameCap = grouped['@reference.name']; + const callableNode = groupedNodes['@reference.callable']; + if (nameCap !== undefined && callableNode !== undefined) { + const receiverCap = grouped['@reference.receiver']; + // The anchor Capture must carry the call-form tag as its `name` — + // the scope-extractor reads `Capture.name` (not the map key) to + // classify the reference kind, so re-wrap via nodeToCapture rather + // than reusing the `@reference.callable`-named Capture (whose head + // `callable` resolves to no ReferenceKind and silently drops it). + if (receiverCap !== undefined) { + out.push({ + '@reference.call.member': nodeToCapture('@reference.call.member', callableNode), + '@reference.name': nameCap, + '@reference.receiver': receiverCap, + }); + } else { + out.push({ + '@reference.call.free': nodeToCapture('@reference.call.free', callableNode), + '@reference.name': nameCap, + }); + } + } + continue; + } + if ( grouped['@reference.call.free'] !== undefined && grouped['@reference.receiver'] !== undefined @@ -253,6 +288,100 @@ function synthesizeKotlinInheritanceReferences(rootNode: SyntaxNode): CaptureMat return out; } +/** + * The enclosing type name for a node nested in a class/object/companion body. + * Walks up to the first `class_declaration` / `object_declaration` / + * `companion_object` ancestor and returns its `type_identifier` name node. + * Used to qualify a secondary-constructor declaration as `.constructor`. + */ +function kotlinEnclosingTypeNameNode(node: SyntaxNode): SyntaxNode | null { + for (let cur: SyntaxNode | null = node.parent; cur !== null; cur = cur.parent) { + if ( + cur.type === 'class_declaration' || + cur.type === 'object_declaration' || + cur.type === 'companion_object' + ) { + const nameNode = cur.namedChildren.find((c) => c.type === 'type_identifier'); + return nameNode ?? null; + } + } + return null; +} + +/** + * Synthesize a `@declaration.constructor` capture for each Kotlin + * `secondary_constructor` (issue #1919 review CF1). The structure phase already + * materializes a `Constructor` graph node (`Constructor:file:Class.constructor#`), + * but the registry-primary scope-resolution path had no Constructor *def* in the + * scope tree — so a call inside the constructor body resolved its caller anchor + * up to the enclosing Class def, mis-attributing the CALLS edge to the class. + * + * Paired with `(secondary_constructor) @scope.function` in query.ts: that rule + * makes the constructor body its own Function scope; this declaration places a + * Constructor def in that scope so `pickCallerCallableDef` anchors calls on the + * Constructor. The def is keyed to match the structure-phase node id: + * - `@declaration.qualified_name` = `.constructor` so the bridge's + * qualified key (`:file::Constructor::Class.constructor`) hits the node. + * - `@declaration.parameter-types` so two same-name secondary constructors are + * disambiguated by the bridge's parameter-types key (`~Int,Int`), matching + * the `#`-suffixed structure node for the overload with the same + * parameter shape. (The zero-arg overload carries no parameter types and + * resolves via the qualified/simple key to the `#0` node.) + * + * The anchor spans the whole `secondary_constructor` node — same range as the + * `@scope.function` it pairs with — so the def is owned by that Function scope + * and the constructor name auto-hoists to the enclosing class scope (exactly the + * binding shape a normal method declaration produces). + */ +function synthesizeKotlinSecondaryConstructorDeclarations(rootNode: SyntaxNode): CaptureMatch[] { + const out: CaptureMatch[] = []; + for (const ctorNode of descendantsOfType(rootNode, 'secondary_constructor')) { + const keyword = ctorNode.namedChildren.find((c) => c.type === 'constructor'); + // The `constructor` keyword is an anonymous token; fall back to the node + // itself for the name capture position when the named-child lookup misses. + const nameAnchor = keyword ?? ctorNode; + const classNameNode = kotlinEnclosingTypeNameNode(ctorNode); + const qualifiedName = + classNameNode !== null ? `${classNameNode.text}.constructor` : 'constructor'; + + const match: Record = { + '@declaration.constructor': nodeToCapture('@declaration.constructor', ctorNode), + '@declaration.name': syntheticCapture('@declaration.name', nameAnchor, 'constructor'), + '@declaration.qualified_name': syntheticCapture( + '@declaration.qualified_name', + ctorNode, + qualifiedName, + ), + }; + + const arity = computeKotlinArityMetadata(ctorNode); + if (arity.parameterCount !== undefined) { + match['@declaration.parameter-count'] = syntheticCapture( + '@declaration.parameter-count', + ctorNode, + String(arity.parameterCount), + ); + } + if (arity.requiredParameterCount !== undefined) { + match['@declaration.required-parameter-count'] = syntheticCapture( + '@declaration.required-parameter-count', + ctorNode, + String(arity.requiredParameterCount), + ); + } + if (arity.parameterTypes !== undefined) { + match['@declaration.parameter-types'] = syntheticCapture( + '@declaration.parameter-types', + ctorNode, + JSON.stringify(arity.parameterTypes), + ); + } + + out.push(match); + } + return out; +} + /** * The bare simple-name `type_identifier` of a `user_type`. Strips generic * type arguments (`Base` → `Base`) and qualifier tails (`pkg.Base` → `Base`) diff --git a/gitnexus/src/core/ingestion/languages/kotlin/companion-scopes.ts b/gitnexus/src/core/ingestion/languages/kotlin/companion-scopes.ts index ac4034620..f1c45c65c 100644 --- a/gitnexus/src/core/ingestion/languages/kotlin/companion-scopes.ts +++ b/gitnexus/src/core/ingestion/languages/kotlin/companion-scopes.ts @@ -55,6 +55,17 @@ export function isCompanionScope(filePath: string, scopeId: ScopeId): boolean { return companionScopesByFile.get(filePath)?.has(scopeId) ?? false; } +/** + * Snapshot the companion-object scope ids recorded for `filePath` as a plain + * array (for the worker→main capture side-channel, #1983). Returns an empty + * array when the file recorded no companion scopes. See + * `capture-side-channel.ts`. + */ +export function getCompanionScopesForFile(filePath: string): ScopeId[] { + const scopes = companionScopesByFile.get(filePath); + return scopes === undefined ? [] : [...scopes]; +} + /** Clear all tracked companion scopes (for testing). */ export function clearCompanionScopes(): void { companionScopesByFile.clear(); diff --git a/gitnexus/src/core/ingestion/languages/kotlin/index.ts b/gitnexus/src/core/ingestion/languages/kotlin/index.ts index 206128e52..206254540 100644 --- a/gitnexus/src/core/ingestion/languages/kotlin/index.ts +++ b/gitnexus/src/core/ingestion/languages/kotlin/index.ts @@ -1,4 +1,9 @@ export { emitKotlinScopeCaptures } from './captures.js'; +export { + collectKotlinCaptureSideChannel, + applyKotlinCaptureSideChannel, + type KotlinCaptureSideChannel, +} from './capture-side-channel.js'; export { getKotlinCaptureCacheStats, resetKotlinCaptureCacheStats } from './cache-stats.js'; export { interpretKotlinImport, interpretKotlinTypeBinding } from './interpret.js'; export { kotlinArityCompatibility } from './arity.js'; diff --git a/gitnexus/src/core/ingestion/languages/kotlin/query.ts b/gitnexus/src/core/ingestion/languages/kotlin/query.ts index 9209655d8..f3a749086 100644 --- a/gitnexus/src/core/ingestion/languages/kotlin/query.ts +++ b/gitnexus/src/core/ingestion/languages/kotlin/query.ts @@ -1,5 +1,13 @@ import Parser from 'tree-sitter'; -import Kotlin from 'tree-sitter-kotlin'; +import { SupportedLanguages } from 'gitnexus-shared'; +// `tree-sitter-kotlin` is an optionalDependency that may be absent on a default +// install (or fail its native build). Loaded lazily + guarded via parser-loader +// rather than statically imported: this module is pulled onto the main thread +// eagerly by the scope-resolution registry and the language-provider index, so +// a top-level `import Kotlin from 'tree-sitter-kotlin'` would throw +// ERR_MODULE_NOT_FOUND at module-load and crash `analyze` even for repos with no +// Kotlin files (#2091, #2093). The grammar is only ever needed in the getters. +import { getLanguageGrammar } from '../../../tree-sitter/parser-loader.js'; const KOTLIN_SCOPE_QUERY = ` ;; Scopes @@ -9,6 +17,16 @@ const KOTLIN_SCOPE_QUERY = ` (companion_object) @scope.class (function_declaration) @scope.function +;; Secondary-constructor body scope (issue #1919 review CF1). A +;; secondary constructor's "constructor(...) { ... }" body executes statements +;; just like a method body, so it must be its OWN Function scope — otherwise a +;; call inside the body resolves its caller anchor up to the enclosing Class +;; scope (the class's Class def), mis-attributing the CALLS edge to the class +;; rather than the Constructor. The matching @declaration.constructor is +;; synthesized in captures.ts (synthesizeKotlinSecondaryConstructorDeclarations) +;; so this scope owns a Constructor def keyed to the structure-phase node id. +(secondary_constructor) @scope.function + ;; Companion-object marker (issue #1756 / U4). Side-channel capture that ;; lets populateCompanionMembersOnEnclosingClass distinguish a companion ;; Class scope from a regular Class scope without inspecting ownedDefs. @@ -117,6 +135,26 @@ const KOTLIN_SCOPE_QUERY = ` (function_value_parameters) [(user_type) (nullable_type) (function_type)] @type-binding.type) @type-binding.return +;; References — callable references ("::method", "Type::new", "obj::m") — F47. +;; A "callable_reference" references a function/constructor as a value (no +;; call_suffix), so the registry-primary call path never saw it. Real-parse +;; (issue #1919) shows the canonical shape inside a function body is: +;; "::topLevelFn" -> (callable_reference :: (simple_identifier)) member only +;; "String::length" -> (callable_reference (type_identifier) :: (simple_identifier)) +;; "obj::method" -> (callable_reference (type_identifier) :: (simple_identifier)) +;; "Type::new" -> (callable_reference (type_identifier) :: (simple_identifier)) +;; The receiver (real type OR object) is always a "type_identifier"; the +;; referenced member is the LAST "simple_identifier". One rule with an +;; optional receiver and an end-anchored member covers all four forms with +;; exactly one match per callable_reference (no sibling-branch double-match). +;; (NOTE: a qualified "A.B::m" parses as a nested navigation_expression, not a +;; callable_reference, and is already captured by the read.member rule below.) +;; emitKotlinScopeCaptures rewrites this into a free/member call reference. +(callable_reference + (type_identifier)? @reference.receiver + (simple_identifier) @reference.name + .) @reference.callable + ;; References — direct calls / constructor syntax (call_expression (simple_identifier) @reference.name) @reference.call.free @@ -149,14 +187,19 @@ let query: Parser.Query | null = null; export function getKotlinParser(): Parser { if (parser === null) { parser = new Parser(); - parser.setLanguage(Kotlin as Parameters[0]); + parser.setLanguage( + getLanguageGrammar(SupportedLanguages.Kotlin) as Parameters[0], + ); } return parser; } export function getKotlinScopeQuery(): Parser.Query { if (query === null) { - query = new Parser.Query(Kotlin as Parameters[0], KOTLIN_SCOPE_QUERY); + query = new Parser.Query( + getLanguageGrammar(SupportedLanguages.Kotlin) as Parameters[0], + KOTLIN_SCOPE_QUERY, + ); } return query; } diff --git a/gitnexus/src/core/ingestion/languages/kotlin/scope-resolver.ts b/gitnexus/src/core/ingestion/languages/kotlin/scope-resolver.ts index bd63c1e86..6bd48f0d1 100644 --- a/gitnexus/src/core/ingestion/languages/kotlin/scope-resolver.ts +++ b/gitnexus/src/core/ingestion/languages/kotlin/scope-resolver.ts @@ -14,6 +14,7 @@ import { type KotlinResolveContext, } from './index.js'; import { clearCompanionScopes } from './companion-scopes.js'; +import { applyKotlinCaptureSideChannel } from './capture-side-channel.js'; import { isKotlinStaticOnly } from './owners.js'; /** @@ -84,6 +85,23 @@ export const kotlinScopeResolver: ScopeResolver = { buildMro: (graph, parsedFiles, nodeLookup) => buildKotlinMro(graph, parsedFiles, nodeLookup), + // Worker-boundary restore (see `ScopeResolver.applyCaptureSideChannel`). + // `emitKotlinScopeCaptures` records per-file companion-object scope ids + // (`markCompanionScope` → `companionScopesByFile`) as a SIDE EFFECT — that + // state is NOT serialized onto the returned ParsedFile's scopes/defs. On the + // worker path those marks are populated in the worker process and lost across + // the MessageChannel / disk store; the main thread reuses the serialized + // ParsedFile and skips `extractParsedFile`, so `isKotlinStaticOnly` and + // `populateCompanionMembersOnEnclosingClass` (owners.ts) would see an empty + // map and companion/static dispatch would emit zero CALLS edges. The worker + // stashed a plain-data snapshot on `parsed.captureSideChannel` via + // `kotlinProvider.collectCaptureSideChannel`; this restores it into the + // module map WITHOUT any tree-sitter re-parse (the #1983 fix). The + // freshly-extracted leg never calls this — its marks were just populated in + // this process. Runs BEFORE `populateOwners` so the restored companion map is + // visible to it. + applyCaptureSideChannel: applyKotlinCaptureSideChannel, + populateOwners: (parsed: ParsedFile) => populateKotlinOwners(parsed), isSuperReceiver: (text) => text.trim() === 'super', diff --git a/gitnexus/src/core/ingestion/languages/php/index.ts b/gitnexus/src/core/ingestion/languages/php/index.ts index 9b549bc3d..de7b6345b 100644 --- a/gitnexus/src/core/ingestion/languages/php/index.ts +++ b/gitnexus/src/core/ingestion/languages/php/index.ts @@ -54,10 +54,9 @@ * 6. **Intersection types in parameters** — `T&U $param` takes the first * named part (`T`). This matches the legacy type-extractor's behavior. * - * Shadow-harness corpus parity is the authoritative signal for which of - * these matter in practice. The CI parity gate blocks any PR that regresses - * either the legacy or registry-primary run of - * `test/integration/resolvers/php.test.ts`. + * The `test/integration/resolvers/php.test.ts` resolver suite is the + * authoritative signal for which of these matter in practice; it runs in + * the standard CI test workflow, so a regression blocks the merge. */ export { emitPhpScopeCaptures } from './captures.js'; diff --git a/gitnexus/src/core/ingestion/languages/python/captures.ts b/gitnexus/src/core/ingestion/languages/python/captures.ts index 761404ebb..756066382 100644 --- a/gitnexus/src/core/ingestion/languages/python/captures.ts +++ b/gitnexus/src/core/ingestion/languages/python/captures.ts @@ -38,9 +38,10 @@ export function emitPythonScopeCaptures( _filePath: string, cachedTree?: unknown, ): readonly CaptureMatch[] { - // Skip the parse when the caller (parse phase's ASTCache) already - // produced a Tree for this source. Cache miss = re-parse, same as - // before. The cachedTree parameter is typed as `unknown` at the + // Skip the parse when the caller (the scope-resolution orchestrator's + // `treeCache`) already produced a Tree for this source — empty under + // worker-pool runs, so cache miss = re-parse. The cachedTree parameter + // is typed as `unknown` at the // contract layer (see `LanguageProvider.emitScopeCaptures`); cast // here at the use site. let tree = cachedTree as ReturnType['parse']> | undefined; diff --git a/gitnexus/src/core/ingestion/languages/python/index.ts b/gitnexus/src/core/ingestion/languages/python/index.ts index 166c3a7ae..dc7e84f64 100644 --- a/gitnexus/src/core/ingestion/languages/python/index.ts +++ b/gitnexus/src/core/ingestion/languages/python/index.ts @@ -66,10 +66,9 @@ * site where the enclosing class can't be statically determined * is left unresolved. * - * Shadow-harness corpus parity is the authoritative signal for which - * of these matter in practice. The CI parity gate blocks any PR that - * regresses either the legacy or registry-primary run of - * `test/integration/resolvers/python.test.ts`. + * The `test/integration/resolvers/python.test.ts` resolver suite is the + * authoritative signal for which of these matter in practice; it runs in + * the standard CI test workflow, so a regression blocks the merge. */ export { emitPythonScopeCaptures } from './captures.js'; diff --git a/gitnexus/src/core/ingestion/languages/swift/query.ts b/gitnexus/src/core/ingestion/languages/swift/query.ts index cf2699423..78c1efa41 100644 --- a/gitnexus/src/core/ingestion/languages/swift/query.ts +++ b/gitnexus/src/core/ingestion/languages/swift/query.ts @@ -43,7 +43,15 @@ */ import Parser from 'tree-sitter'; -import Swift from 'tree-sitter-swift'; +import { SupportedLanguages } from 'gitnexus-shared'; +// `tree-sitter-swift` is an optional/vendored grammar that may be absent on a +// default install. It is loaded lazily + guarded via parser-loader rather than +// statically imported: this module is pulled onto the main thread eagerly by +// the scope-resolution registry and the language-provider index, so a top-level +// `import Swift from 'tree-sitter-swift'` would throw ERR_MODULE_NOT_FOUND at +// module-load and crash `analyze` even for repos with no Swift files (#2091, +// #2093). The grammar is only ever needed inside the lazy getters below. +import { getLanguageGrammar } from '../../../tree-sitter/parser-loader.js'; const SWIFT_SCOPE_QUERY = ` ;; ── Scopes ────────────────────────────────────────────────────────── @@ -186,14 +194,19 @@ let _query: Parser.Query | null = null; export function getSwiftParser(): Parser { if (_parser === null) { _parser = new Parser(); - _parser.setLanguage(Swift as Parameters[0]); + _parser.setLanguage( + getLanguageGrammar(SupportedLanguages.Swift) as Parameters[0], + ); } return _parser; } export function getSwiftScopeQuery(): Parser.Query { if (_query === null) { - _query = new Parser.Query(Swift as Parameters[0], SWIFT_SCOPE_QUERY); + _query = new Parser.Query( + getLanguageGrammar(SupportedLanguages.Swift) as Parameters[0], + SWIFT_SCOPE_QUERY, + ); } return _query; } diff --git a/gitnexus/src/core/ingestion/languages/typescript/captures.ts b/gitnexus/src/core/ingestion/languages/typescript/captures.ts index b3e338242..9d466c3f9 100644 --- a/gitnexus/src/core/ingestion/languages/typescript/captures.ts +++ b/gitnexus/src/core/ingestion/languages/typescript/captures.ts @@ -163,10 +163,11 @@ export function emitTsScopeCaptures( filePath: string, cachedTree?: unknown, ): readonly CaptureMatch[] { - // Skip the parse when the caller (parse phase's scopeTreeCache) already - // produced a Tree for this source. Cache miss = re-parse, same as before. - // The cachedTree parameter is typed as `unknown` at the LanguageProvider - // contract layer; cast here at the use site. + // Reuse a pre-parsed Tree when the caller passes one via `cachedTree`; a + // miss re-parses. (The cache is currently always empty — its only producer, + // the sequential parser, was removed — so this re-parses in practice.) The + // cachedTree parameter is typed `unknown` at the LanguageProvider contract + // layer; cast here at the use site. // // Grammar selection: `.tsx` files are parsed with the TSX grammar, // `.ts` files with the TypeScript grammar. The two grammars have diff --git a/gitnexus/src/core/ingestion/languages/typescript/index.ts b/gitnexus/src/core/ingestion/languages/typescript/index.ts index cb4bd4023..120566f48 100644 --- a/gitnexus/src/core/ingestion/languages/typescript/index.ts +++ b/gitnexus/src/core/ingestion/languages/typescript/index.ts @@ -80,10 +80,9 @@ * identifiers are narrowed (`user instanceof User`). Member paths * such as `user.address instanceof Address` remain unresolved. * - * Shadow-harness corpus parity on `test/integration/resolvers/ - * typescript.test.ts` is the authoritative signal for which of these - * matter in practice. The CI parity gate blocks any PR that regresses - * either the legacy or registry-primary run. + * The `test/integration/resolvers/typescript.test.ts` resolver suite is + * the authoritative signal for which of these matter in practice; it runs + * in the standard CI test workflow, so a regression blocks the merge. */ export { emitTsScopeCaptures } from './captures.js'; diff --git a/gitnexus/src/core/ingestion/languages/vue/captures.ts b/gitnexus/src/core/ingestion/languages/vue/captures.ts index b8031fcdd..b8ec7a8a4 100644 --- a/gitnexus/src/core/ingestion/languages/vue/captures.ts +++ b/gitnexus/src/core/ingestion/languages/vue/captures.ts @@ -22,6 +22,7 @@ import type { CaptureMatch } from 'gitnexus-shared'; import { extractVueScript } from '../../vue-sfc-extractor.js'; import { emitTsScopeCaptures } from '../typescript/captures.js'; +import { emitJsScopeCaptures } from '../javascript/captures.js'; /** * Emit scope captures for a Vue SFC. @@ -31,11 +32,11 @@ import { emitTsScopeCaptures } from '../typescript/captures.js'; * 1. **Full SFC content** (sequential path, <15 files): `sourceText` * contains the whole `.vue` file with `