diff --git a/.claude/README-gitnexus-reviewer-swarm.md b/.claude/README-gitnexus-reviewer-swarm.md new file mode 100644 index 000000000..ce54d1d2f --- /dev/null +++ b/.claude/README-gitnexus-reviewer-swarm.md @@ -0,0 +1,43 @@ +# GitNexus PR Reviewer Swarm — Claude Code adapter + +This is the **Claude Code** entrypoint for the cross-CLI GitNexus PR reviewer swarm. The +review logic itself is CLI-neutral and lives in **[`pr-swarm-review/`](../pr-swarm-review/README.md)** +— that README is the canonical guide and covers every CLI (Claude Code, Gemini, Copilot, +Cursor, Codex, and any AGENTS.md-aware agent). + +## Invocation (Claude Code) + +``` +/gitnexus-pr-swarm-review +``` + +Runs in **Swarm mode**: the coordinator skill dispatches the seven `gitnexus-*` subagents in +parallel (lanes 1–2 first, 3–6 in parallel, lane 7 last as a hard gate). + +## Files in this adapter + +| File | Role | +|------|------| +| `.claude/skills/gitnexus-pr-swarm-review/SKILL.md` | Coordinator — runs Swarm mode per `pr-swarm-review/orchestration.md` | +| `.claude/agents/gitnexus-*.md` | Seven thin subagent wrappers; each reads its canonical persona in `pr-swarm-review/personas/` | + +Each subagent keeps valid Claude Code frontmatter (model, tools, etc.); the mechanical +verifier lanes (`test-ci-verifier`, `branch-hygiene-reviewer`) run on Haiku, the analytical +lanes on Sonnet. + +## Key properties + +- **Read-only.** Tools limited to Read/Grep/Glob/Bash, and every persona enforces an + explicit permitted/prohibited Bash list. No agent edits files, commits, or posts. +- **Evidence-grounded**; **missing visibility becomes verification work**; **manually invoked.** + +## Editing + +Edit review behavior in the canonical files under `pr-swarm-review/` (orchestration + +personas), **not** in these wrappers. After adding or editing files in `.claude/agents/`, +restart Claude Code so it reloads the agent definitions. + +## Relationship to `/gitnexus-pr-review` + +Coexists with the single-agent `/gitnexus-pr-review` skill (a linear checklist using GitNexus +MCP tools). This swarm is the multi-persona deep production-readiness review. diff --git a/.claude/agents/gitnexus-branch-hygiene-reviewer.md b/.claude/agents/gitnexus-branch-hygiene-reviewer.md new file mode 100644 index 000000000..77bbba944 --- /dev/null +++ b/.claude/agents/gitnexus-branch-hygiene-reviewer.md @@ -0,0 +1,24 @@ +--- +name: gitnexus-branch-hygiene-reviewer +description: "GitNexus branch hygiene and mergeability reviewer. Use to classify merge state, conflicts, stale branches, merge-from-main commits, unrelated churn, mixed domains, and whether rebase or split is required." +tools: + - Read + - Grep + - Glob + - Bash +model: claude-haiku-4-5-20251001 +maxTurns: 30 +--- + +# GitNexus Branch Hygiene & Mergeability Reviewer + +Your complete operating spec — role, what to inspect, classifications, and the required output sections — lives in the canonical, CLI-neutral persona file: + +**`pr-swarm-review/personas/02-branch-hygiene-reviewer.md`** + +Read that file now with the Read tool and follow it exactly. It is the single source of truth shared across all AI CLIs; this subagent only adapts it to Claude Code. The orchestration contract (lane order, Swarm vs Solo execution, output structure) is in `pr-swarm-review/orchestration.md`. + +## Rules (always enforced) + +- **Do not edit files.** You are read-only. +- **Bash is read-only.** Permitted: `git log`, `git diff`, `git show`, `git grep`, `git ls-files`, `gh pr view`, `gh pr diff`, `gh pr checks`, `gh issue view`, and inspection tools (`grep`, `cat`, `find`, `ls`). Prohibited: any command that writes files, modifies git state (`git commit`, `git add`, `git checkout -- `), posts to GitHub (`gh pr comment`, `gh pr review`, `gh issue comment`), installs packages, or runs arbitrary scripts. diff --git a/.claude/agents/gitnexus-docs-dod-reviewer.md b/.claude/agents/gitnexus-docs-dod-reviewer.md new file mode 100644 index 000000000..bb38d0095 --- /dev/null +++ b/.claude/agents/gitnexus-docs-dod-reviewer.md @@ -0,0 +1,24 @@ +--- +name: gitnexus-docs-dod-reviewer +description: "GitNexus docs and Definition-of-Done reviewer. Use to translate repo guidance, linked issues, changed domains, docs requirements, release notes, and acceptance criteria into a PR-specific DoD." +tools: + - Read + - Grep + - Glob + - Bash +model: claude-sonnet-4-6 +maxTurns: 30 +--- + +# GitNexus Docs & Definition-of-Done Reviewer + +Your complete operating spec — role, what to inspect, classifications, and the required output sections — lives in the canonical, CLI-neutral persona file: + +**`pr-swarm-review/personas/06-docs-dod-reviewer.md`** + +Read that file now with the Read tool and follow it exactly. It is the single source of truth shared across all AI CLIs; this subagent only adapts it to Claude Code. The orchestration contract (lane order, Swarm vs Solo execution, output structure) is in `pr-swarm-review/orchestration.md`. + +## Rules (always enforced) + +- **Do not edit files.** You are read-only. +- **Bash is read-only.** Permitted: `git log`, `git diff`, `git show`, `git grep`, `git ls-files`, `gh pr view`, `gh pr diff`, `gh pr checks`, `gh issue view`, and inspection tools (`grep`, `cat`, `find`, `ls`). Prohibited: any command that writes files, modifies git state (`git commit`, `git add`, `git checkout -- `), posts to GitHub (`gh pr comment`, `gh pr review`, `gh issue comment`), installs packages, or runs arbitrary scripts. diff --git a/.claude/agents/gitnexus-pr-facts-historian.md b/.claude/agents/gitnexus-pr-facts-historian.md new file mode 100644 index 000000000..6ee95412b --- /dev/null +++ b/.claude/agents/gitnexus-pr-facts-historian.md @@ -0,0 +1,24 @@ +--- +name: gitnexus-pr-facts-historian +description: "GitNexus PR facts and repository-history investigator. Use to gather PR identity, visible GitHub state, changed files, commits, linked issues, related PRs, historical fixes, regressions, stale follow-ups, and missing visibility." +tools: + - Read + - Grep + - Glob + - Bash +model: claude-sonnet-4-6 +maxTurns: 40 +--- + +# GitNexus PR Facts & Repository-History Investigator + +Your complete operating spec — role, what to inspect, classifications, and the required output sections — lives in the canonical, CLI-neutral persona file: + +**`pr-swarm-review/personas/01-pr-facts-historian.md`** + +Read that file now with the Read tool and follow it exactly. It is the single source of truth shared across all AI CLIs; this subagent only adapts it to Claude Code. The orchestration contract (lane order, Swarm vs Solo execution, output structure) is in `pr-swarm-review/orchestration.md`. + +## Rules (always enforced) + +- **Do not edit files.** You are read-only. +- **Bash is read-only.** Permitted: `git log`, `git diff`, `git show`, `git grep`, `git ls-files`, `gh pr view`, `gh pr diff`, `gh pr checks`, `gh issue view`, and inspection tools (`grep`, `cat`, `find`, `ls`). Prohibited: any command that writes files, modifies git state (`git commit`, `git add`, `git checkout -- `), posts to GitHub (`gh pr comment`, `gh pr review`, `gh issue comment`), installs packages, or runs arbitrary scripts. diff --git a/.claude/agents/gitnexus-risk-architect.md b/.claude/agents/gitnexus-risk-architect.md new file mode 100644 index 000000000..39b6d9675 --- /dev/null +++ b/.claude/agents/gitnexus-risk-architect.md @@ -0,0 +1,24 @@ +--- +name: gitnexus-risk-architect +description: "GitNexus production-risk reviewer. Use for risk-model-first review of changed files, runtime behavior, multi-domain changes, user impact, failure modes, compatibility, and merge-blocking risk." +tools: + - Read + - Grep + - Glob + - Bash +model: claude-sonnet-4-6 +maxTurns: 40 +--- + +# GitNexus Production-Risk Architect + +Your complete operating spec — role, what to inspect, classifications, and the required output sections — lives in the canonical, CLI-neutral persona file: + +**`pr-swarm-review/personas/03-risk-architect.md`** + +Read that file now with the Read tool and follow it exactly. It is the single source of truth shared across all AI CLIs; this subagent only adapts it to Claude Code. The orchestration contract (lane order, Swarm vs Solo execution, output structure) is in `pr-swarm-review/orchestration.md`. + +## Rules (always enforced) + +- **Do not edit files.** You are read-only. +- **Bash is read-only.** Permitted: `git log`, `git diff`, `git show`, `git grep`, `git ls-files`, `gh pr view`, `gh pr diff`, `gh pr checks`, `gh issue view`, and inspection tools (`grep`, `cat`, `find`, `ls`). Prohibited: any command that writes files, modifies git state (`git commit`, `git add`, `git checkout -- `), posts to GitHub (`gh pr comment`, `gh pr review`, `gh issue comment`), installs packages, or runs arbitrary scripts. diff --git a/.claude/agents/gitnexus-security-boundary-reviewer.md b/.claude/agents/gitnexus-security-boundary-reviewer.md new file mode 100644 index 000000000..c932f9826 --- /dev/null +++ b/.claude/agents/gitnexus-security-boundary-reviewer.md @@ -0,0 +1,24 @@ +--- +name: gitnexus-security-boundary-reviewer +description: "GitNexus security and trust-boundary reviewer. Use for auth, permissions, secrets, injection, unsafe parsing, external input handling, hidden Unicode, YAML/Docker/workflow risks, and suspicious non-ASCII hygiene." +tools: + - Read + - Grep + - Glob + - Bash +model: claude-sonnet-4-6 +maxTurns: 35 +--- + +# GitNexus Security & Trust-Boundary Reviewer + +Your complete operating spec — role, what to inspect, classifications, and the required output sections — lives in the canonical, CLI-neutral persona file: + +**`pr-swarm-review/personas/05-security-boundary-reviewer.md`** + +Read that file now with the Read tool and follow it exactly. It is the single source of truth shared across all AI CLIs; this subagent only adapts it to Claude Code. The orchestration contract (lane order, Swarm vs Solo execution, output structure) is in `pr-swarm-review/orchestration.md`. + +## Rules (always enforced) + +- **Do not edit files.** You are read-only. +- **Bash is read-only.** Permitted: `git log`, `git diff`, `git show`, `git grep`, `git ls-files`, `gh pr view`, `gh pr diff`, `gh pr checks`, `gh issue view`, and inspection tools (`grep`, `cat`, `find`, `ls`). Prohibited: any command that writes files, modifies git state (`git commit`, `git add`, `git checkout -- `), posts to GitHub (`gh pr comment`, `gh pr review`, `gh issue comment`), installs packages, or runs arbitrary scripts. diff --git a/.claude/agents/gitnexus-synthesis-critic.md b/.claude/agents/gitnexus-synthesis-critic.md new file mode 100644 index 000000000..1c7ec0b5b --- /dev/null +++ b/.claude/agents/gitnexus-synthesis-critic.md @@ -0,0 +1,24 @@ +--- +name: gitnexus-synthesis-critic +description: "GitNexus final review synthesis critic. Use to check whether the final PR review is evidence-grounded, risk-prioritized, GitNexus-specific, non-generic, and follows required verdict rules." +tools: + - Read + - Grep + - Glob + - Bash +model: claude-sonnet-4-6 +maxTurns: 25 +--- + +# GitNexus Final-Review Synthesis Critic + +Your complete operating spec — role, what to inspect, classifications, and the required output sections — lives in the canonical, CLI-neutral persona file: + +**`pr-swarm-review/personas/07-synthesis-critic.md`** + +Read that file now with the Read tool and follow it exactly. It is the single source of truth shared across all AI CLIs; this subagent only adapts it to Claude Code. The orchestration contract (lane order, Swarm vs Solo execution, output structure) is in `pr-swarm-review/orchestration.md`. + +## Rules (always enforced) + +- **Do not edit files.** You are read-only. +- **Bash is read-only.** Permitted: `git log`, `git diff`, `git show`, `git grep`, `git ls-files`, `gh pr view`, `gh pr diff`, `gh pr checks`, `gh issue view`, and inspection tools (`grep`, `cat`, `find`, `ls`). Prohibited: any command that writes files, modifies git state (`git commit`, `git add`, `git checkout -- `), posts to GitHub (`gh pr comment`, `gh pr review`, `gh issue comment`), installs packages, or runs arbitrary scripts. diff --git a/.claude/agents/gitnexus-test-ci-verifier.md b/.claude/agents/gitnexus-test-ci-verifier.md new file mode 100644 index 000000000..66e90d767 --- /dev/null +++ b/.claude/agents/gitnexus-test-ci-verifier.md @@ -0,0 +1,24 @@ +--- +name: gitnexus-test-ci-verifier +description: "GitNexus test and CI reviewer. Use to verify whether changed behavior is covered by targeted tests, whether CI actually runs those tests, and whether workflow changes weaken validation." +tools: + - Read + - Grep + - Glob + - Bash +model: claude-haiku-4-5-20251001 +maxTurns: 35 +--- + +# GitNexus Test & CI Verifier + +Your complete operating spec — role, what to inspect, classifications, and the required output sections — lives in the canonical, CLI-neutral persona file: + +**`pr-swarm-review/personas/04-test-ci-verifier.md`** + +Read that file now with the Read tool and follow it exactly. It is the single source of truth shared across all AI CLIs; this subagent only adapts it to Claude Code. The orchestration contract (lane order, Swarm vs Solo execution, output structure) is in `pr-swarm-review/orchestration.md`. + +## Rules (always enforced) + +- **Do not edit files.** You are read-only. +- **Bash is read-only.** Permitted: `git log`, `git diff`, `git show`, `git grep`, `git ls-files`, `gh pr view`, `gh pr diff`, `gh pr checks`, `gh issue view`, and inspection tools (`grep`, `cat`, `find`, `ls`). Prohibited: any command that writes files, modifies git state (`git commit`, `git add`, `git checkout -- `), posts to GitHub (`gh pr comment`, `gh pr review`, `gh issue comment`), installs packages, or runs arbitrary scripts. diff --git a/.claude/skills/gitnexus-pr-swarm-review/SKILL.md b/.claude/skills/gitnexus-pr-swarm-review/SKILL.md new file mode 100644 index 000000000..3ed78399c --- /dev/null +++ b/.claude/skills/gitnexus-pr-swarm-review/SKILL.md @@ -0,0 +1,31 @@ +--- +name: gitnexus-pr-swarm-review +description: "Run a GitNexus production-readiness pull request review using a coordinated reviewer swarm." +--- + +# GitNexus PR Swarm Review (Claude Code adapter) + +Use this skill to review a GitNexus pull request and produce a production-readiness review. + +``` +/gitnexus-pr-swarm-review +``` + +You are the **swarm coordinator**. The full review contract — lanes, dependencies, +classifications, output structure, finding format, hidden-Unicode checks, and behavior +rules — is the canonical, CLI-neutral spec: + +**`pr-swarm-review/orchestration.md`** — read it now and follow it. + +This adapter only pins the Claude Code specifics: + +- **Run in Swarm mode.** Dispatch each lane as its own subagent via the Agent tool. The + seven subagents are the project agents named `gitnexus-*` (one per persona); each reads + its canonical persona under `pr-swarm-review/personas/`. Run lanes 1–2 first, lanes 3–6 + in parallel after, and lane 7 last on the draft. +- **Lane 7 is a hard gate.** Do not emit the final review while the synthesis critic's + "Required corrections before posting" section is non-empty — revise and re-run it. +- Stay **read-only**: investigate and report; never edit, commit, or post. + +Do not flatten the review into a generic checklist; delegate to the subagents and +synthesize per `orchestration.md`. diff --git a/.cursor/commands/gitnexus-pr-swarm-review.md b/.cursor/commands/gitnexus-pr-swarm-review.md new file mode 100644 index 000000000..4acec179c --- /dev/null +++ b/.cursor/commands/gitnexus-pr-swarm-review.md @@ -0,0 +1,17 @@ +# GitNexus PR Swarm Review + +You are the GitNexus PR review coordinator. Review the pull request named after this command +(a PR URL or number for `https://github.com/abhigyanpatwari/GitNexus`). If none was given, +ask for one. + +Read `pr-swarm-review/orchestration.md` in this repository and follow it exactly — it is the +canonical, CLI-neutral review contract (lanes, classifications, output structure, finding +format, hidden-Unicode checks, behavior rules). + +Run in **Solo mode**: you are a single agent, so perform all seven lanes yourself in +dependency order, adopting each persona in `pr-swarm-review/personas/0N-*.md` in turn +(lanes 1–2 first, then 3–6, then lane 7). Keep every lane's findings in context. Lane 7 +(synthesis critic) is a hard gate: do not emit the final review until its "Required +corrections before posting" section is empty. + +Stay strictly read-only: investigate and report; never edit files, commit, or post to GitHub. diff --git a/.gemini/commands/gitnexus-pr-swarm-review.toml b/.gemini/commands/gitnexus-pr-swarm-review.toml new file mode 100644 index 000000000..20a847fea --- /dev/null +++ b/.gemini/commands/gitnexus-pr-swarm-review.toml @@ -0,0 +1,19 @@ +description = "GitNexus production-readiness PR swarm review (Solo mode)" + +prompt = """ +You are the GitNexus PR review coordinator. Review this pull request: {{args}} +(a PR URL or number for https://github.com/abhigyanpatwari/GitNexus). If no target was +given, ask for one. + +Read `pr-swarm-review/orchestration.md` in this repository and follow it exactly. It is the +canonical, CLI-neutral review contract (lanes, classifications, output structure, finding +format, hidden-Unicode checks, behavior rules). + +Run in **Solo mode**: you are a single agent, so perform all seven lanes yourself in +dependency order, adopting each persona in `pr-swarm-review/personas/0N-*.md` in turn +(lanes 1-2 first, then 3-6, then lane 7). Keep every lane's findings in context. Lane 7 +(synthesis critic) is a hard gate: do not emit the final review until its "Required +corrections before posting" section is empty — revise and re-run it otherwise. + +Stay strictly read-only: investigate and report; never edit files, commit, or post to GitHub. +""" diff --git a/.github/prompts/gitnexus-pr-swarm-review.prompt.md b/.github/prompts/gitnexus-pr-swarm-review.prompt.md new file mode 100644 index 000000000..8df6737f4 --- /dev/null +++ b/.github/prompts/gitnexus-pr-swarm-review.prompt.md @@ -0,0 +1,19 @@ +--- +description: 'GitNexus production-readiness PR swarm review (Solo mode)' +mode: 'agent' +--- + +You are the GitNexus PR review coordinator. Review the pull request the user names (a PR URL +or number for `https://github.com/abhigyanpatwari/GitNexus`). If none was given, ask for one. + +Read `pr-swarm-review/orchestration.md` in this repository and follow it exactly — it is the +canonical, CLI-neutral review contract (lanes, classifications, output structure, finding +format, hidden-Unicode checks, behavior rules). + +Run in **Solo mode**: you are a single agent, so perform all seven lanes yourself in +dependency order, adopting each persona in `pr-swarm-review/personas/0N-*.md` in turn +(lanes 1–2 first, then 3–6, then lane 7). Keep every lane's findings in context. Lane 7 +(synthesis critic) is a hard gate: do not emit the final review until its "Required +corrections before posting" section is empty. + +Stay strictly read-only: investigate and report; never edit files, commit, or post to GitHub. diff --git a/.gitignore b/.gitignore index 9ca3d0d09..11f2743c7 100644 --- a/.gitignore +++ b/.gitignore @@ -91,11 +91,13 @@ gitnexus/vendor/**/node_modules/ .claude-flow/ -.claude/agents/ +.claude/agents/* +!.claude/agents/gitnexus-*.md .claude/commands/ .claude/helpers -.claude/skills/ +.claude/skills/* !.claude/skills/gitnexus/ +!.claude/skills/gitnexus-pr-swarm-review/ .history/ diff --git a/AGENTS.md b/AGENTS.md index 7971154aa..5b0fb162d 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -44,6 +44,18 @@ Commands and gotchas live under **Repo reference** below and in **[CONTRIBUTING. - **Cursor:** `.cursor/index.mdc` (always-on); `.cursor/rules/*.mdc` (glob-scoped). Legacy `.cursorrules` deprecated. - **GitNexus:** skills in `.claude/skills/gitnexus/`; MCP rules in `gitnexus:start` block below. +## PR Swarm Review (cross-CLI) + +To run a production-readiness review of a GitNexus pull request from **any** AI CLI, follow +the canonical, CLI-neutral spec **[`pr-swarm-review/orchestration.md`](pr-swarm-review/orchestration.md)** +(seven read-only review personas under `pr-swarm-review/personas/`). It defines two +execution modes with the same output contract: **Swarm mode** (parallel subagents, e.g. +Claude Code) and **Solo mode** (one agent runs all lanes sequentially — Codex, Gemini, +Cursor, Copilot, or any agent reading this file). Per-CLI entrypoints are thin wrappers +listed in [`pr-swarm-review/README.md`](pr-swarm-review/README.md); edit review logic only +in the canonical files, never in the wrappers. The review is read-only — it never edits, +commits, or posts. + ## Changelog | Date | Version | Change | diff --git a/gitnexus/src/cli/analyze.ts b/gitnexus/src/cli/analyze.ts index bde83b621..18d4bd641 100644 --- a/gitnexus/src/cli/analyze.ts +++ b/gitnexus/src/cli/analyze.ts @@ -1096,6 +1096,17 @@ const analyzeCommandImpl = async (inputPath?: string, options?: AnalyzeOptions): ); console.log(` ${repoPath}`); + // Persistent (non-scrolling) warning when FTS indexing was skipped — the + // progress-bar log() that fired mid-run has already scrolled away, so the + // degraded-search state must also appear in the final summary (#1161). + if (result.ftsSkipped) { + console.log( + `\n Warning: full-text/BM25 search is disabled — the LadybugDB FTS extension was unavailable.\n` + + ` Install it once with network access (GITNEXUS_LBUG_EXTENSION_INSTALL=auto) then rerun, or\n` + + ` run \`gitnexus analyze --repair-fts\` when connected. Run \`gitnexus doctor\` for details.`, + ); + } + try { await fs.access(getGlobalRegistryPath()); } catch { diff --git a/gitnexus/src/cli/doctor.ts b/gitnexus/src/cli/doctor.ts index 0fb2a98a0..a45471fc1 100644 --- a/gitnexus/src/cli/doctor.ts +++ b/gitnexus/src/cli/doctor.ts @@ -2,6 +2,7 @@ import { getRuntimeCapabilities, getRuntimeFingerprint } from '../core/platform/ import { resolveEmbeddingConfig } from '../core/embeddings/config.js'; import { isHttpMode } from '../core/embeddings/http-client.js'; import { checkLbugNative } from '../core/lbug/native-check.js'; +import { getExtensionInstallPolicy } from '../core/lbug/extension-loader.js'; import { t } from './i18n/index.js'; function isCombiningMark(codePoint: number): boolean { @@ -74,6 +75,17 @@ export const doctorCommand = async () => { console.log(` ${label('doctor.labels.fullTextSearch', 18)}${capabilities.fts}`); console.log(` ${label('doctor.labels.vectorIndex', 18)}${capabilities.vector}`); console.log(` ${label('doctor.labels.semanticMode', 18)}${capabilities.semanticMode}`); + // Surface the optional-extension install policy so offline users can see + // whether analyze/query will reach the network (extension.ladybugdb.com). + // Literal label (like the 'native' line) to avoid adding i18n keys. + const installPolicy = getExtensionInstallPolicy(); + const policyHint = + installPolicy === 'load-only' + ? ' (offline; load only, no network install)' + : installPolicy === 'never' + ? ' (optional extensions disabled)' + : ' (installs missing extensions over network)'; + console.log(` ${padDisplayEnd('Ext install:', 18)}${installPolicy}${policyHint}`); console.log( ` ${label('doctor.labels.exactScanLimit', 18)}${t('doctor.chunks', { count: capabilities.exactScanLimit })}`, ); diff --git a/gitnexus/src/core/embeddings/embedding-pipeline.ts b/gitnexus/src/core/embeddings/embedding-pipeline.ts index cc2d38d9b..e394659f4 100644 --- a/gitnexus/src/core/embeddings/embedding-pipeline.ts +++ b/gitnexus/src/core/embeddings/embedding-pipeline.ts @@ -43,20 +43,38 @@ import { STALE_HASH_SENTINEL, } from '../lbug/schema.js'; import { loadVectorExtension } from '../lbug/lbug-adapter.js'; +import type { ExtensionInstallPolicy } from '../lbug/extension-loader.js'; import { getExactScanLimit } from '../platform/capabilities.js'; import { logger } from '../logger.js'; const isDev = process.env.NODE_ENV === 'development'; const vectorUnavailableMessage = - 'VECTOR extension is unavailable for this LadybugDB runtime; semantic search will use exact scan when embeddings exist.'; + 'VECTOR extension unavailable; semantic embeddings fall back to exact scan. ' + + 'To enable vector search, install it once with network access ' + + '(GITNEXUS_LBUG_EXTENSION_INSTALL=auto), or pre-install it for offline use. ' + + 'Set GITNEXUS_LBUG_EXTENSION_INSTALL=never to skip installs and silence this.'; + +/** + * Resolve the extension-install policy for the embedding WRITE path (analyze). + * + * Generating embeddings is an explicit opt-in to a feature that requires the + * VECTOR extension, so when the operator has NOT pinned a policy we default to + * `auto` (one bounded, out-of-process INSTALL) — matching the documented + * "auto = default for analyze" intent in extension-loader.ts. An explicit + * GITNEXUS_LBUG_EXTENSION_INSTALL=load-only|never|auto always wins, so an + * offline or locked-down operator is never silently forced onto the network + * (the #1153 regression caused by hard-coding `auto` here). Read on every call + * (not memoized) so test env stubbing works. + */ +export const resolveEmbeddingInstallPolicy = (): ExtensionInstallPolicy => { + const raw = process.env.GITNEXUS_LBUG_EXTENSION_INSTALL; + if (raw === 'load-only' || raw === 'never' || raw === 'auto') return raw; + return 'auto'; +}; const ensureVectorExtensionAvailable = async (): Promise => { - const vectorReady = await loadVectorExtension(); - if (!vectorReady) { - return false; - } - return true; + return loadVectorExtension(undefined, { policy: resolveEmbeddingInstallPolicy() }); }; /** * Bump this when the embedding text template changes in a way that should @@ -257,7 +275,7 @@ export const runEmbeddingPipeline = async ( try { const vectorAvailable = await ensureVectorExtensionAvailable(); - if (!vectorAvailable && isDev) { + if (!vectorAvailable) { logger.warn(vectorUnavailableMessage); } @@ -584,7 +602,11 @@ export const semanticSearch = async ( string, { distance: number; chunkIndex: number; startLine: number; endLine: number } >(); - if (await loadVectorExtension()) { + // Query/read path: NEVER spawn a network INSTALL on a user query. If the + // VECTOR extension was not pre-installed, fall back to exact scan rather than + // blocking the query on a download (offline-first; see extension-loader.ts + // "load-only" — used by all serve/MCP query paths). + if (await loadVectorExtension(undefined, { policy: 'load-only' })) { try { bestChunks = await collectBestChunks(k, async (fetchLimit) => { const vectorQuery = ` diff --git a/gitnexus/src/core/group/extractors/grpc-extractor.ts b/gitnexus/src/core/group/extractors/grpc-extractor.ts index 56107d8ee..6d70873c0 100644 --- a/gitnexus/src/core/group/extractors/grpc-extractor.ts +++ b/gitnexus/src/core/group/extractors/grpc-extractor.ts @@ -188,6 +188,16 @@ function makeContract( export interface ProtoServiceInfo { package: string; + /** + * Optional. Value of `option java_package = "..."` declared in the + * same `.proto` file, when present and different from `package`. + * Empty string when the option is absent or equals `package`. Used by + * `detectionToContract()` to translate a Java import path back to the + * proto package whenever the proto explicitly publishes its generated + * Java code under a different namespace (a common pattern in + * Google-style protobuf projects). + */ + javaPackage: string; serviceName: string; methods: string[]; protoPath: string; @@ -207,6 +217,19 @@ function extractProtoImports(content: string): string[] { return imports; } +/** + * Extract `option java_package = "..."` from a `.proto` file, if any. + * The Java code generator places generated `XxxGrpc.java` classes under + * this package (instead of the proto `package` declaration) when the + * option is set. Real-world projects (Google Cloud Java APIs, internal + * shaded SDKs) routinely use this to publish their Java artifacts under + * a corporate namespace different from the wire-protocol package. + */ +function extractJavaPackageOption(content: string): string { + const m = content.match(/^\s*option\s+java_package\s*=\s*"([\w.]+)"\s*;/m); + return m?.[1] ?? ''; +} + function longestSharedSegmentRun(aPath: string, bPath: string): number { const a = aPath.split('/').filter(Boolean); const b = bPath.split('/').filter(Boolean); @@ -228,8 +251,18 @@ function longestSharedSegmentRun(aPath: string, bPath: string): number { async function buildProtoContext(repoPath: string): Promise<{ packagesByProto: Map; servicesByName: Map; + /** + * Reverse index: `option java_package` value → ProtoServiceInfo[] + * declared in `.proto` files that ship under that Java namespace. + * Only populated when `java_package` is set AND differs from + * `package`. Lets `detectionToContract()` translate an import-derived + * Java package back to its source proto package whenever the proto + * is in the same repository. + */ + servicesByJavaPackage: Map; }> { const servicesByName = new Map(); + const servicesByJavaPackage = new Map(); // `.gitnexusignore` / `.gitignore` honoured via the shared IgnoreService — // see `filesystem-walker.ts` for the canonical pattern. Replaces a // hardcoded `[node_modules, .git, vendor]` array; those names plus the @@ -292,6 +325,13 @@ async function buildProtoContext(repoPath: string): Promise<{ const content = contents.get(normalizedRel); if (!content) continue; const pkg = resolvePackage(normalizedRel); + const javaPkgOption = extractJavaPackageOption(content); + // Only retain `javaPackage` when it actively diverges from `pkg`. + // When equal (or absent), the import-derived path produces the + // same FQN as the proto-derived path, so no translation is needed + // and we keep the field empty to avoid populating the reverse + // index with redundant entries. + const javaPackage = javaPkgOption && javaPkgOption !== pkg ? javaPkgOption : ''; const serviceBlocks = extractServiceBlocks(content); for (const block of serviceBlocks) { @@ -303,6 +343,7 @@ async function buildProtoContext(repoPath: string): Promise<{ } const info: ProtoServiceInfo = { package: pkg, + javaPackage, serviceName: block.name, methods, protoPath: normalizedRel, @@ -310,10 +351,16 @@ async function buildProtoContext(repoPath: string): Promise<{ const existing = servicesByName.get(block.name) ?? []; existing.push(info); servicesByName.set(block.name, existing); + + if (javaPackage) { + const byJava = servicesByJavaPackage.get(javaPackage) ?? []; + byJava.push(info); + servicesByJavaPackage.set(javaPackage, byJava); + } } } - return { packagesByProto, servicesByName }; + return { packagesByProto, servicesByName, servicesByJavaPackage }; } export async function buildProtoMap(repoPath: string): Promise> { @@ -377,6 +424,7 @@ export class GrpcExtractor implements ContractExtractor { const out: ExtractedContract[] = []; const protoContext = await buildProtoContext(repoPath); const protoMap = protoContext.servicesByName; + const javaPackageMap = protoContext.servicesByJavaPackage; // ─── Proto files — definitive provider source ───────────────── // When tree-sitter-proto is available, .proto files are handled by @@ -435,7 +483,7 @@ export class GrpcExtractor implements ContractExtractor { continue; } for (const d of detections) { - const contract = this.detectionToContract(d, rel, protoMap); + const contract = this.detectionToContract(d, rel, protoMap, javaPackageMap); if (contract) out.push(contract); } } @@ -449,12 +497,163 @@ export class GrpcExtractor implements ContractExtractor { * either a service-level (`grpc::pkg.Svc/*`) or method-level * (`grpc::pkg.Svc/Method`) contract id, and selecting confidence * based on whether the proto map had an entry. + * + * Resolution order for the package prefix: + * + * 1. **Java-package translation** (when detection + * supplied a `protoPackage` from a Java import). + * A `.proto` in the SAME repo may set `option + * java_package = "..."` to publish its generated + * Java classes under a namespace different from + * the proto `package`. Real-world projects (e.g. + * Google Cloud Java APIs) routinely do this. + * When the import-derived package matches that + * `java_package` value, translate back to the + * proto `package` so the resulting contract id + * is wire-correct rather than Java-namespace. + * + * 2. **Per-repo proto map check** (when the same + * service name has `.proto` candidates in this + * repo). The proto file is the authoritative + * source. If the proto's `package` agrees with + * the import's `protoPackage`, both paths produce + * the same FQN — emit it. If they DISAGREE (e.g. + * a typo'd Java import, or a mismatched + * java_package the reverse index didn't catch), + * trust the proto map and warn — the import + * MUST NOT silently overwrite an authoritative + * proto package. + * + * 3. **Import-derived FQN fallback** (when neither + * a `java_package` translation nor a proto map + * candidate exists in this repo). Typical for the + * "client-jar" pattern, where a consumer repo + * depends on a published stub jar and never + * carries the originating `.proto`. Use the + * import path verbatim as the proto package. Note + * the known limitation: when the published proto + * sets `option java_package` differing from + * `package`, the resulting FQN reflects the Java + * namespace rather than the proto namespace and + * will not match a provider repo's contract id — + * we cannot translate without sight of the proto. + * + * 4. **Per-repo proto map (no import)** — the legacy + * path. Used when the plugin didn't supply + * `protoPackage` (no import statement, wildcard + * import only, or non-Java languages that haven't + * been retrofitted yet). + * + * 5. **Short-name fallback** — when none of the + * above resolves a package, emit a service-only + * short-name contract id (`grpc::Svc/*`), + * preserving the pre-fix behaviour. */ private detectionToContract( d: GrpcDetection, filePath: string, protoMap: Map, + javaPackageMap: Map, ): ExtractedContract | null { + if (d.protoPackage) { + // Step 1: java_package translation. The import-derived package + // may be the `option java_package` value of a `.proto` in the + // SAME repo. Look it up and, if found for the same service name, + // use the underlying proto `package` to build a wire-correct + // contract id. + const javaCandidates = javaPackageMap.get(d.protoPackage) ?? []; + const javaTranslated = javaCandidates.find((p) => p.serviceName === d.serviceName); + if (javaTranslated) { + const cid = d.methodName + ? contractId(javaTranslated.package, d.serviceName, d.methodName) + : serviceContractId(javaTranslated.package, d.serviceName); + const meta: Record = { + service: d.serviceName, + source: d.source, + package: javaTranslated.package, + protoPackageSource: 'import-translated', + }; + if (d.methodName) meta.method = d.methodName; + return makeContract(cid, d.role, filePath, d.symbolName, d.confidenceWithProto, meta); + } + + // Step 2: proto map cross-check. When this repo also carries a + // `.proto` defining the same short service name, the proto is + // authoritative and decides the package. The import is only used + // to disambiguate among same-short-name candidates when the + // resolution heuristic can't pick a unique winner on path alone. + const candidates = protoMap.get(d.serviceName) ?? []; + if (candidates.length > 0) { + const proto = resolveProtoConflict(d.serviceName, filePath, candidates); + if (proto === null) { + // Ambiguous proto resolution; resolveProtoConflict already warned. + return null; + } + const protoPkg = proto.package; + if (protoPkg === d.protoPackage) { + // Both paths agree. + const cid = d.methodName + ? contractId(protoPkg, d.serviceName, d.methodName) + : serviceContractId(protoPkg, d.serviceName); + const meta: Record = { + service: d.serviceName, + source: d.source, + package: protoPkg, + protoPackageSource: 'import', + }; + if (d.methodName) meta.method = d.methodName; + return makeContract(cid, d.role, filePath, d.symbolName, d.confidenceWithProto, meta); + } + // Disagreement. Trust the proto file and emit a warning so + // operators can investigate the import. This protects against + // the symmetric Finding 2 case: a stale or typo'd Java import + // silently corrupting the contract id of a service whose + // `.proto` lives in the same repo. + logger.warn( + `[grpc-extractor] Java import package "${d.protoPackage}" for service ` + + `"${d.serviceName}" disagrees with local proto package "${protoPkg}" at ` + + `${filePath}; using proto package as authoritative source`, + ); + const cid = d.methodName + ? contractId(protoPkg, d.serviceName, d.methodName) + : serviceContractId(protoPkg, d.serviceName); + const meta: Record = { + service: d.serviceName, + source: d.source, + package: protoPkg, + protoPackageSource: 'proto-override', + importPackage: d.protoPackage, + }; + if (d.methodName) meta.method = d.methodName; + return makeContract(cid, d.role, filePath, d.symbolName, d.confidenceWithProto, meta); + } + + // Step 3: import-derived fallback. No `.proto` in this repo + // names the service, and no `java_package` reverse-lookup + // matched. Emit the FQN with the import-derived package. This + // is the typical client-jar consumer path. + // + // Known limitation: when the published proto sets + // `option java_package` to a value that differs from + // `package`, this path produces a contract id that reflects + // the Java namespace, not the proto namespace, and will not + // match a provider repo. Resolving that case requires + // group-level proto knowledge, which is intentionally out of + // scope for this fix. + const cid = d.methodName + ? contractId(d.protoPackage, d.serviceName, d.methodName) + : serviceContractId(d.protoPackage, d.serviceName); + const meta: Record = { + service: d.serviceName, + source: d.source, + package: d.protoPackage, + protoPackageSource: 'import', + }; + if (d.methodName) meta.method = d.methodName; + return makeContract(cid, d.role, filePath, d.symbolName, d.confidenceWithProto, meta); + } + + // Steps 4 + 5: legacy per-repo proto map resolution (no import). const candidates = protoMap.get(d.serviceName) ?? []; const proto = resolveProtoConflict(d.serviceName, filePath, candidates); // If there were proto candidates but resolution was ambiguous, skip diff --git a/gitnexus/src/core/group/extractors/grpc-patterns/java.ts b/gitnexus/src/core/group/extractors/grpc-patterns/java.ts index bf1cf4816..eeacdb4c6 100644 --- a/gitnexus/src/core/group/extractors/grpc-patterns/java.ts +++ b/gitnexus/src/core/group/extractors/grpc-patterns/java.ts @@ -78,6 +78,33 @@ const STUB_PATTERNS = compilePatterns({ ], } satisfies LanguagePatterns>); +// `import .;` — captures the proto package of the +// imported gRPC class (e.g. `cn.unipus.ucf.admin.proto.client.service` +// for `import cn.unipus.ucf.admin.proto.client.service.ContentRpcServiceGrpc`). +// Used by `scan` to build a per-file `XxxGrpc → fullPackage` map so +// consumer-side detections can carry a fully-qualified contract id +// even when the consumer repo does not contain any `.proto` files. +// +// `import static …` is excluded by tree-sitter shape: the `name:` +// field is only present on the non-static form. `import w.x.*;` is +// also excluded for the same reason — wildcard imports have an +// `asterisk` child instead of a named identifier. +const GRPC_CLASS_IMPORT_PATTERNS = compilePatterns({ + name: 'java-grpc-class-import', + language: Java, + patterns: [ + { + meta: {}, + query: ` + (import_declaration + (scoped_identifier + scope: (_) @import_pkg + name: (identifier) @import_name (#match? @import_name "Grpc$"))) + `, + }, + ], +} satisfies LanguagePatterns>); + /** * Check whether a `class_declaration` node has a `@GrpcService` * annotation in its modifiers list. In tree-sitter-java, class-level @@ -118,6 +145,39 @@ export const JAVA_GRPC_PLUGIN: GrpcLanguagePlugin = { const out: GrpcDetection[] = []; const emittedClassIds = new Set(); + // ─── Build per-file gRPC class import map ─────────────────────── + // Maps `XxxGrpc` (short class name) → fully-qualified proto package + // (e.g. `cn.unipus.ucf.admin.proto.client.service`). Used below to + // tag both provider and consumer detections with a `protoPackage` + // so the orchestrator can build a fully-qualified contract id + // without depending on the current repo carrying any `.proto` + // files. This is the key fix for client-jar consumer repos. + // + // Same-short-name disambiguation: when two distinct `import` lines + // bring different `XxxGrpc` classes from different packages into + // the same file (rare for grpc — the second import would be a + // compile error in Java), the last one wins. Java's compiler + // forbids that case so we don't bother modelling it. + const grpcClassImports = new Map(); + for (const match of runCompiledPatterns(GRPC_CLASS_IMPORT_PATTERNS, tree)) { + const pkgNode = match.captures.import_pkg; + const nameNode = match.captures.import_name; + if (!pkgNode || !nameNode) continue; + grpcClassImports.set(nameNode.text, pkgNode.text); + } + + /** + * Resolve the fully-qualified proto package for a short service + * name in this file. Looks up `Grpc` in the import + * map; returns `undefined` when the class is referenced via a + * fully-qualified name on every call site (no import line) or + * when only a wildcard import is present. The orchestrator falls + * back to the per-repo proto map in that case, preserving the + * pre-fix behaviour. + */ + const protoPackageFor = (serviceName: string): string | undefined => + grpcClassImports.get(`${serviceName}Grpc`); + // ─── Providers: scoped form (`...Grpc.XxxImplBase`) ───────────── for (const match of runCompiledPatterns(SCOPED_IMPL_BASE_PATTERNS, tree)) { const classNode = match.captures.class; @@ -127,6 +187,7 @@ export const JAVA_GRPC_PLUGIN: GrpcLanguagePlugin = { if (!serviceName) continue; emittedClassIds.add(classNode.id); const annotated = hasGrpcServiceAnnotation(classNode); + const protoPackage = protoPackageFor(serviceName); out.push({ role: 'provider', serviceName, @@ -134,6 +195,7 @@ export const JAVA_GRPC_PLUGIN: GrpcLanguagePlugin = { source: annotated ? 'java_grpc_service' : 'java_impl_base', confidenceWithProto: 0.8, confidenceWithoutProto: 0.65, + ...(protoPackage ? { protoPackage } : {}), }); } @@ -147,6 +209,7 @@ export const JAVA_GRPC_PLUGIN: GrpcLanguagePlugin = { if (!serviceName) continue; emittedClassIds.add(classNode.id); const annotated = hasGrpcServiceAnnotation(classNode); + const protoPackage = protoPackageFor(serviceName); out.push({ role: 'provider', serviceName, @@ -154,6 +217,7 @@ export const JAVA_GRPC_PLUGIN: GrpcLanguagePlugin = { source: annotated ? 'java_grpc_service' : 'java_impl_base', confidenceWithProto: 0.8, confidenceWithoutProto: 0.65, + ...(protoPackage ? { protoPackage } : {}), }); } @@ -164,6 +228,7 @@ export const JAVA_GRPC_PLUGIN: GrpcLanguagePlugin = { const grpcMatch = GRPC_SUFFIX_RE.exec(grpcClsNode.text); if (!grpcMatch) continue; const serviceName = grpcMatch[1]; + const protoPackage = protoPackageFor(serviceName); out.push({ role: 'consumer', serviceName, @@ -171,6 +236,7 @@ export const JAVA_GRPC_PLUGIN: GrpcLanguagePlugin = { source: 'java_stub', confidenceWithProto: 0.75, confidenceWithoutProto: 0.55, + ...(protoPackage ? { protoPackage } : {}), }); } diff --git a/gitnexus/src/core/group/extractors/grpc-patterns/types.ts b/gitnexus/src/core/group/extractors/grpc-patterns/types.ts index 606d9629b..dd94a4e93 100644 --- a/gitnexus/src/core/group/extractors/grpc-patterns/types.ts +++ b/gitnexus/src/core/group/extractors/grpc-patterns/types.ts @@ -36,6 +36,18 @@ export interface GrpcDetection { confidenceWithProto: number; /** Confidence when the proto map has no entry. */ confidenceWithoutProto: number; + /** + * Optional. Fully-qualified proto package the detection's service + * belongs to (e.g. `cn.unipus.ucf.admin.proto.client.service`), + * derived directly from the source file's import statements when + * available. When set, the orchestrator uses this package to build + * the contract id INSTEAD of consulting the per-repo proto map — + * letting consumer repos that don't carry `.proto` files (the + * client-jar architecture used by most Java gRPC microservices) + * still emit a fully-qualified contract id that matches the + * provider repo's contract id verbatim. + */ + protoPackage?: string; } /** diff --git a/gitnexus/src/core/group/extractors/http-patterns/index.ts b/gitnexus/src/core/group/extractors/http-patterns/index.ts index 4cc758218..e5c03b68c 100644 --- a/gitnexus/src/core/group/extractors/http-patterns/index.ts +++ b/gitnexus/src/core/group/extractors/http-patterns/index.ts @@ -8,7 +8,13 @@ import { PYTHON_HTTP_PLUGIN } from './python.js'; import { PHP_HTTP_PLUGIN } from './php.js'; import { JAVASCRIPT_HTTP_PLUGIN, TYPESCRIPT_HTTP_PLUGIN, TSX_HTTP_PLUGIN } from './node.js'; -export type { HttpDetection, HttpLanguagePlugin, HttpRole } from './types.js'; +export type { + HttpDetection, + HttpFileDetections, + HttpLanguagePlugin, + HttpRole, + HttpScanInput, +} from './types.js'; /** * File-extension → HTTP language plugin registry. The top-level diff --git a/gitnexus/src/core/group/extractors/http-patterns/java.ts b/gitnexus/src/core/group/extractors/http-patterns/java.ts index b3ced5920..48da46765 100644 --- a/gitnexus/src/core/group/extractors/http-patterns/java.ts +++ b/gitnexus/src/core/group/extractors/http-patterns/java.ts @@ -6,13 +6,21 @@ import { unquoteLiteral, type LanguagePatterns, } from '../tree-sitter-scanner.js'; -import type { HttpDetection, HttpLanguagePlugin } from './types.js'; +import type { + HttpDetection, + HttpFileDetections, + HttpLanguagePlugin, + HttpScanInput, +} from './types.js'; /** * Java HTTP plugin. Handles: * - Spring `@RequestMapping` class prefixes + `@(Get|Post|...)Mapping` method annotations - * - Spring `RestTemplate.getForObject/...`, `WebClient.method(HttpMethod.X, ...)` + * - Spring `RestTemplate.getForObject/...`, `exchange(...)` + * - Spring `WebClient.method(HttpMethod.X, ...)`, `WebClient.get().uri(...)` * - OkHttp `new Request.Builder().url("...")` + * - OpenFeign interfaces with Spring MVC method annotations + * - Java / Apache HttpClient literal request construction * * The plugin runs two pattern bundles: one to collect class-level * `@RequestMapping` prefixes keyed by the enclosing class node, and a @@ -43,31 +51,132 @@ const METHOD_ANNOTATION_TO_HTTP: Record = { // route prefixes — e.g. `produces = "application/json"` would corrupt // every method route under that controller). The sibling // `topic-patterns/java.ts` uses the same `key:` constraint approach. -const SPRING_CLASS_PREFIX_PATTERNS = compilePatterns({ - name: 'java-spring-class-prefix', +interface SpringRouteBinding { + method: string; + path: string; +} + +interface SpringMethodInfo { + name: string; + routes: SpringRouteBinding[]; +} + +interface SpringTypeInfo { + filePath: string; + kind: 'class' | 'interface'; + name: string; + classPrefix: string; + implementedInterfaces: string[]; + isController: boolean; + methods: SpringMethodInfo[]; +} + +// ─── Provider: Spring class/interface-level @RequestMapping prefix ─── +const SPRING_TYPE_PREFIX_PATTERNS = compilePatterns({ + name: 'java-spring-type-prefix', language: Java, patterns: [ { meta: {}, query: ` - (class_declaration - (modifiers - (annotation - name: (identifier) @ann (#eq? @ann "RequestMapping") - arguments: (annotation_argument_list (string_literal) @prefix)))) @class + [ + (class_declaration + (modifiers + (annotation + name: (identifier) @ann (#eq? @ann "RequestMapping") + arguments: (annotation_argument_list (string_literal) @prefix)))) @type + (interface_declaration + (modifiers + (annotation + name: (identifier) @ann (#eq? @ann "RequestMapping") + arguments: (annotation_argument_list (string_literal) @prefix)))) @type + ] `, }, { meta: {}, query: ` - (class_declaration + [ + (class_declaration + (modifiers + (annotation + name: (identifier) @ann (#eq? @ann "RequestMapping") + arguments: (annotation_argument_list + (element_value_pair + key: (identifier) @key (#match? @key "^(path|value)$") + value: (string_literal) @prefix))))) @type + (interface_declaration + (modifiers + (annotation + name: (identifier) @ann (#eq? @ann "RequestMapping") + arguments: (annotation_argument_list + (element_value_pair + key: (identifier) @key (#match? @key "^(path|value)$") + value: (string_literal) @prefix))))) @type + ] + `, + }, + ], +} satisfies LanguagePatterns>); + +const SPRING_TYPE_DECLARATION_PATTERNS = compilePatterns({ + name: 'java-spring-type-declaration', + language: Java, + patterns: [ + { + meta: {}, + query: ` + [ + (class_declaration name: (identifier) @type_name) @type + (interface_declaration name: (identifier) @type_name) @type + ] + `, + }, + ], +} satisfies LanguagePatterns>); + +// ─── Consumer: OpenFeign interface-level prefixes ─────────────────── +// Feign's `name`/`value` attributes identify a service, not an HTTP path, +// so only `path` is used as a URL prefix. `@RequestMapping` on a Feign +// interface is also common and does carry a path prefix. +const FEIGN_INTERFACE_PREFIX_PATTERNS = compilePatterns({ + name: 'java-feign-interface-prefix', + language: Java, + patterns: [ + { + meta: {}, + query: ` + (interface_declaration + (modifiers + (annotation + name: (identifier) @ann (#eq? @ann "FeignClient") + arguments: (annotation_argument_list + (element_value_pair + key: (identifier) @key (#eq? @key "path") + value: (string_literal) @prefix))))) @interface + `, + }, + { + meta: {}, + query: ` + (interface_declaration + (modifiers + (annotation + name: (identifier) @ann (#eq? @ann "RequestMapping") + arguments: (annotation_argument_list (string_literal) @prefix)))) @interface + `, + }, + { + meta: {}, + query: ` + (interface_declaration (modifiers (annotation name: (identifier) @ann (#eq? @ann "RequestMapping") arguments: (annotation_argument_list (element_value_pair key: (identifier) @key (#match? @key "^(path|value)$") - value: (string_literal) @prefix))))) @class + value: (string_literal) @prefix))))) @interface `, }, ], @@ -116,6 +225,8 @@ const SPRING_METHOD_ROUTE_PATTERNS = compilePatterns({ // RestTemplate.put → PUT // RestTemplate.delete → DELETE // RestTemplate.patchForObject → PATCH +// Source-scan only: receiver must be named exactly `restTemplate`. +// Fields, `this.restTemplate`, aliases, and other injection names are deferred. const REST_TEMPLATE_TO_HTTP: Record = { getForObject: 'GET', getForEntity: 'GET', @@ -146,22 +257,48 @@ const REST_TEMPLATE_PATTERNS = compilePatterns({ ], } satisfies LanguagePatterns); -// ─── Consumer: Spring WebClient — webClient.method(HttpMethod.X, "path") ─ -const WEB_CLIENT_PATTERNS = compilePatterns({ - name: 'java-web-client', +const REST_TEMPLATE_EXCHANGE_PATTERNS = compilePatterns({ + name: 'java-rest-template-exchange', + language: Java, + patterns: [ + { + meta: { framework: 'spring-rest-template' }, + query: ` + (method_invocation + object: (identifier) @obj (#eq? @obj "restTemplate") + name: (identifier) @method (#eq? @method "exchange") + arguments: (argument_list + . (string_literal) @path + (field_access + object: (identifier) @httpMethodCls (#eq? @httpMethodCls "HttpMethod") + field: (identifier) @http_method))) + `, + }, + ], +} satisfies LanguagePatterns); + +const WEB_CLIENT_SHORT_TO_HTTP: Record = { + get: 'GET', + post: 'POST', + put: 'PUT', + delete: 'DELETE', + patch: 'PATCH', +}; + +const WEB_CLIENT_SHORT_FORM_PATTERNS = compilePatterns({ + name: 'java-web-client-short-form', language: Java, patterns: [ { meta: {}, query: ` (method_invocation - object: (identifier) @obj (#eq? @obj "webClient") - name: (identifier) @method (#eq? @method "method") - arguments: (argument_list - (field_access - object: (identifier) @httpMethodCls (#eq? @httpMethodCls "HttpMethod") - field: (identifier) @http_method) - (string_literal) @path)) + object: (method_invocation + object: (identifier) @obj (#eq? @obj "webClient") + name: (identifier) @verb (#match? @verb "^(get|post|put|delete|patch)$") + arguments: (argument_list)) + name: (identifier) @uri_method (#eq? @uri_method "uri") + arguments: (argument_list . (string_literal) @path)) `, }, ], @@ -188,10 +325,58 @@ const OK_HTTP_PATTERNS = compilePatterns({ ], } satisfies LanguagePatterns>); +const JAVA_HTTP_CLIENT_PATTERNS = compilePatterns({ + name: 'java-http-client', + language: Java, + patterns: [ + { + meta: {}, + query: ` + (method_invocation + object: (method_invocation + object: (method_invocation + object: (identifier) @builderCls (#eq? @builderCls "HttpRequest") + name: (identifier) @newBuilder (#eq? @newBuilder "newBuilder") + arguments: (argument_list)) + name: (identifier) @uri_method (#eq? @uri_method "uri") + arguments: (argument_list + (method_invocation + object: (identifier) @uriCls (#eq? @uriCls "URI") + name: (identifier) @create (#eq? @create "create") + arguments: (argument_list . (string_literal) @path)))) + name: (identifier) @http_method (#match? @http_method "^(GET|POST|PUT|DELETE)$")) + `, + }, + ], +} satisfies LanguagePatterns>); + +const APACHE_HTTP_CLIENT_TO_HTTP: Record = { + HttpGet: 'GET', + HttpPost: 'POST', + HttpPut: 'PUT', + HttpDelete: 'DELETE', + HttpPatch: 'PATCH', +}; + +const APACHE_HTTP_CLIENT_PATTERNS = compilePatterns({ + name: 'java-apache-http-client', + language: Java, + patterns: [ + { + meta: {}, + query: ` + (object_creation_expression + type: (type_identifier) @type (#match? @type "^Http(Get|Post|Put|Delete|Patch)$") + arguments: (argument_list . (string_literal) @path)) + `, + }, + ], +} satisfies LanguagePatterns>); + /** - * Find the nearest enclosing class_declaration ancestor for a node, or - * null if the node is top-level. Tree-sitter's SyntaxNode.parent walks - * one level at a time. + * Find the nearest enclosing class/interface declaration ancestor for + * a node, or null if the node is top-level. Tree-sitter's + * SyntaxNode.parent walks one level at a time. */ function findEnclosingClass(node: Parser.SyntaxNode): Parser.SyntaxNode | null { let cur: Parser.SyntaxNode | null = node.parent; @@ -202,6 +387,15 @@ function findEnclosingClass(node: Parser.SyntaxNode): Parser.SyntaxNode | null { return null; } +function findEnclosingInterface(node: Parser.SyntaxNode): Parser.SyntaxNode | null { + let cur: Parser.SyntaxNode | null = node.parent; + while (cur) { + if (cur.type === 'interface_declaration') return cur; + cur = cur.parent; + } + return null; +} + /** * Join a class-level prefix and a method-level path into a single URL * path. Mirrors the semantics of the original regex implementation: @@ -215,6 +409,184 @@ function joinPath(prefix: string, methodPath: string): string { return `/${cleanPrefix}/${cleanSub}`; } +function getNodeName(node: Parser.SyntaxNode): string | null { + return node.childForFieldName('name')?.text ?? null; +} + +function hasAnnotation(node: Parser.SyntaxNode, names: string | readonly string[]): boolean { + const modifiers = node.namedChildren.find((child) => child.type === 'modifiers'); + if (!modifiers) return false; + const allowed = new Set(typeof names === 'string' ? [names] : names); + const stack = [...modifiers.namedChildren]; + while (stack.length > 0) { + const cur = stack.pop()!; + const annotationName = cur.childForFieldName('name')?.text ?? ''; + const simpleName = annotationName.split('.').pop() ?? annotationName; + if ( + (cur.type === 'annotation' || cur.type === 'marker_annotation') && + (allowed.has(annotationName) || allowed.has(simpleName)) + ) { + return true; + } + stack.push(...cur.namedChildren); + } + return false; +} + +function collectTypePrefixes(tree: Parser.Tree): Map { + const prefixByTypeId = new Map(); + for (const match of runCompiledPatterns(SPRING_TYPE_PREFIX_PATTERNS, tree)) { + const prefixNode = match.captures.prefix; + const typeNode = match.captures.type; + if (!prefixNode || !typeNode) continue; + const prefix = unquoteLiteral(prefixNode.text); + if (prefix !== null) prefixByTypeId.set(typeNode.id, prefix); + } + return prefixByTypeId; +} + +function collectMethodRoutes(tree: Parser.Tree): Map { + const routesByMethodId = new Map(); + for (const match of runCompiledPatterns(SPRING_METHOD_ROUTE_PATTERNS, tree)) { + const annNode = match.captures.ann; + const pathNode = match.captures.path; + const methodNode = match.captures.method; + if (!annNode || !pathNode || !methodNode) continue; + const httpMethod = METHOD_ANNOTATION_TO_HTTP[annNode.text]; + if (!httpMethod) continue; + const rawPath = unquoteLiteral(pathNode.text); + if (rawPath === null) continue; + const routes = routesByMethodId.get(methodNode.id) ?? []; + routes.push({ method: httpMethod, path: rawPath }); + routesByMethodId.set(methodNode.id, routes); + } + return routesByMethodId; +} + +function collectDirectMethods(typeNode: Parser.SyntaxNode): Parser.SyntaxNode[] { + const out: Parser.SyntaxNode[] = []; + const visit = (node: Parser.SyntaxNode): void => { + for (const child of node.namedChildren) { + if (child.type === 'method_declaration') { + out.push(child); + continue; + } + if ( + child !== typeNode && + (child.type === 'class_declaration' || child.type === 'interface_declaration') + ) { + continue; + } + visit(child); + } + }; + visit(typeNode); + return out; +} + +function collectImplementedInterfaces(typeNode: Parser.SyntaxNode): string[] { + const interfacesNode = typeNode.childForFieldName('interfaces'); + if (!interfacesNode) return []; + const out: string[] = []; + const visit = (node: Parser.SyntaxNode): void => { + if (node.type === 'type_identifier' || node.type === 'scoped_type_identifier') { + out.push(node.text.split('.').pop() ?? node.text); + return; + } + for (const child of node.namedChildren) visit(child); + }; + visit(interfacesNode); + return out; +} + +function collectSpringTypes(filePath: string, tree: Parser.Tree): SpringTypeInfo[] { + const prefixByTypeId = collectTypePrefixes(tree); + const routesByMethodId = collectMethodRoutes(tree); + const out: SpringTypeInfo[] = []; + + for (const match of runCompiledPatterns(SPRING_TYPE_DECLARATION_PATTERNS, tree)) { + const typeNode = match.captures.type; + const typeNameNode = match.captures.type_name; + if (!typeNode || !typeNameNode) continue; + const kind = typeNode.type === 'interface_declaration' ? 'interface' : 'class'; + const methods = collectDirectMethods(typeNode) + .map((methodNode) => ({ + name: getNodeName(methodNode), + routes: routesByMethodId.get(methodNode.id) ?? [], + })) + .filter((method): method is SpringMethodInfo => method.name !== null); + + out.push({ + filePath, + kind, + name: typeNameNode.text, + classPrefix: prefixByTypeId.get(typeNode.id) ?? '', + implementedInterfaces: kind === 'class' ? collectImplementedInterfaces(typeNode) : [], + isController: kind === 'class' && hasAnnotation(typeNode, ['RestController', 'Controller']), + methods, + }); + } + + return out; +} + +function scanSpringProject(files: readonly HttpScanInput[]): HttpFileDetections[] { + const types = files.flatMap((file) => collectSpringTypes(file.filePath, file.tree)); + const interfaceRoutes = new Map | null>(); + + for (const type of types) { + if (type.kind !== 'interface') continue; + if (interfaceRoutes.has(type.name)) { + interfaceRoutes.set(type.name, null); + continue; + } + const methodMap = new Map(); + for (const method of type.methods) { + const routes = method.routes.map((route) => ({ + method: route.method, + path: type.classPrefix ? joinPath(type.classPrefix, route.path) : route.path, + })); + if (routes.length > 0) methodMap.set(method.name, routes); + } + interfaceRoutes.set(type.name, methodMap); + } + + const detectionsByFile = new Map(); + for (const type of types) { + if (type.kind !== 'class' || !type.isController) continue; + for (const method of type.methods) { + if (method.routes.length > 0) continue; + const inheritedRoutes = type.implementedInterfaces.flatMap((interfaceName) => { + const routeMap = interfaceRoutes.get(interfaceName); + if (!routeMap) return []; + const routes = routeMap.get(method.name) ?? []; + return routes.map((route) => ({ + method: route.method, + path: joinPath(type.classPrefix, route.path), + })); + }); + + for (const route of inheritedRoutes) { + const detections = detectionsByFile.get(type.filePath) ?? []; + detections.push({ + role: 'provider', + framework: 'spring', + method: route.method, + path: route.path, + name: method.name, + confidence: 0.8, + }); + detectionsByFile.set(type.filePath, detections); + } + } + } + + return [...detectionsByFile.entries()].map(([filePath, detections]) => ({ + filePath, + detections, + })); +} + export const JAVA_HTTP_PLUGIN: HttpLanguagePlugin = { name: 'java-http', language: Java, @@ -222,13 +594,16 @@ export const JAVA_HTTP_PLUGIN: HttpLanguagePlugin = { const out: HttpDetection[] = []; // ─── Providers: Spring class prefix + method annotations ──────── - const prefixByClassId = new Map(); - for (const match of runCompiledPatterns(SPRING_CLASS_PREFIX_PATTERNS, tree)) { + const prefixByTypeId = collectTypePrefixes(tree); + + const feignPrefixByInterfaceId = new Map(); + for (const match of runCompiledPatterns(FEIGN_INTERFACE_PREFIX_PATTERNS, tree)) { const prefixNode = match.captures.prefix; - const classNode = match.captures.class; - if (!prefixNode || !classNode) continue; + const interfaceNode = match.captures.interface; + if (!prefixNode || !interfaceNode) continue; const prefix = unquoteLiteral(prefixNode.text); - if (prefix !== null) prefixByClassId.set(classNode.id, prefix); + if (prefix !== null && !feignPrefixByInterfaceId.has(interfaceNode.id)) + feignPrefixByInterfaceId.set(interfaceNode.id, prefix); } for (const match of runCompiledPatterns(SPRING_METHOD_ROUTE_PATTERNS, tree)) { @@ -241,8 +616,23 @@ export const JAVA_HTTP_PLUGIN: HttpLanguagePlugin = { if (!httpMethod) continue; const rawPath = unquoteLiteral(pathNode.text); if (rawPath === null) continue; + const enclosingInterface = findEnclosingInterface(methodNode); + if (enclosingInterface && hasAnnotation(enclosingInterface, 'FeignClient')) { + const prefix = feignPrefixByInterfaceId.get(enclosingInterface.id) ?? ''; + const fullPath = joinPath(prefix, rawPath); + out.push({ + role: 'consumer', + framework: 'openfeign', + method: httpMethod, + path: fullPath, + name: nameNode?.text ?? null, + confidence: 0.7, + }); + continue; + } const enclosingClass = findEnclosingClass(methodNode); - const prefix = enclosingClass ? (prefixByClassId.get(enclosingClass.id) ?? '') : ''; + if (!enclosingClass) continue; + const prefix = prefixByTypeId.get(enclosingClass.id) ?? ''; const fullPath = joinPath(prefix, rawPath); out.push({ role: 'provider', @@ -273,8 +663,7 @@ export const JAVA_HTTP_PLUGIN: HttpLanguagePlugin = { }); } - // ─── Consumers: WebClient.method(HttpMethod.X, "path") ────────── - for (const match of runCompiledPatterns(WEB_CLIENT_PATTERNS, tree)) { + for (const match of runCompiledPatterns(REST_TEMPLATE_EXCHANGE_PATTERNS, tree)) { const httpMethodNode = match.captures.http_method; const pathNode = match.captures.path; if (!httpMethodNode || !pathNode) continue; @@ -282,7 +671,7 @@ export const JAVA_HTTP_PLUGIN: HttpLanguagePlugin = { if (path === null) continue; out.push({ role: 'consumer', - framework: 'spring-web-client', + framework: 'spring-rest-template', method: httpMethodNode.text.toUpperCase(), path, name: null, @@ -290,6 +679,28 @@ export const JAVA_HTTP_PLUGIN: HttpLanguagePlugin = { }); } + // ─── Consumers: WebClient.get().uri("path") short form ───────── + // Source-scan only: receiver must be named exactly `webClient`. + // The real long-form chain `webClient.method(HttpMethod.X).uri("/x")` + // needs multi-hop chain analysis and is intentionally deferred. + for (const match of runCompiledPatterns(WEB_CLIENT_SHORT_FORM_PATTERNS, tree)) { + const verbNode = match.captures.verb; + const pathNode = match.captures.path; + if (!verbNode || !pathNode) continue; + const httpMethod = WEB_CLIENT_SHORT_TO_HTTP[verbNode.text]; + if (!httpMethod) continue; + const path = unquoteLiteral(pathNode.text); + if (path === null) continue; + out.push({ + role: 'consumer', + framework: 'spring-web-client', + method: httpMethod, + path, + name: null, + confidence: 0.7, + }); + } + // ─── Consumers: OkHttp Request.Builder().url("path") ──────────── for (const match of runCompiledPatterns(OK_HTTP_PATTERNS, tree)) { const pathNode = match.captures.path; @@ -306,6 +717,45 @@ export const JAVA_HTTP_PLUGIN: HttpLanguagePlugin = { }); } + // ─── Consumers: Java HttpClient request builder ───────────────── + // Java's builder exposes GET/POST/PUT/DELETE helpers. PATCH uses + // `.method("PATCH", body)`, which is intentionally deferred. + for (const match of runCompiledPatterns(JAVA_HTTP_CLIENT_PATTERNS, tree)) { + const httpMethodNode = match.captures.http_method; + const pathNode = match.captures.path; + if (!httpMethodNode || !pathNode) continue; + const path = unquoteLiteral(pathNode.text); + if (path === null) continue; + out.push({ + role: 'consumer', + framework: 'java-http-client', + method: httpMethodNode.text.toUpperCase(), + path, + name: null, + confidence: 0.65, + }); + } + + // ─── Consumers: Apache HttpClient request constructors ────────── + for (const match of runCompiledPatterns(APACHE_HTTP_CLIENT_PATTERNS, tree)) { + const typeNode = match.captures.type; + const pathNode = match.captures.path; + if (!typeNode || !pathNode) continue; + const httpMethod = APACHE_HTTP_CLIENT_TO_HTTP[typeNode.text]; + if (!httpMethod) continue; + const path = unquoteLiteral(pathNode.text); + if (path === null) continue; + out.push({ + role: 'consumer', + framework: 'apache-http-client', + method: httpMethod, + path, + name: null, + confidence: 0.65, + }); + } + return out; }, + scanProject: scanSpringProject, }; diff --git a/gitnexus/src/core/group/extractors/http-patterns/kotlin.ts b/gitnexus/src/core/group/extractors/http-patterns/kotlin.ts index 0bafb7a7e..0e56b554b 100644 --- a/gitnexus/src/core/group/extractors/http-patterns/kotlin.ts +++ b/gitnexus/src/core/group/extractors/http-patterns/kotlin.ts @@ -17,18 +17,22 @@ import type { HttpDetection, HttpLanguagePlugin } from './types.js'; * named annotation arguments (`@GetMapping(value = "/x")` and * `@GetMapping(path = "/x")`) are supported. * - * **Consumers** (this PR) — three call-site patterns common in Kotlin + * **Consumers** — four call-site patterns common in Kotlin * Spring projects: * - * 1. `restTemplate.getForObject("/x", ...)` and friends - * 2. `webClient.get().uri("/x")` (short form, 1 verb hop + 1 uri hop) - * 3. `Request.Builder().url("/x")` (OkHttp) + * 1. `restTemplate.getForObject("/x", ...)` and friends (#1855) + * 2. `webClient.get().uri("/x")` — short form (#1855) + * 3. `Request.Builder().url("/x")` — OkHttp (#1855) + * 4. `webClient.method(HttpMethod.X).uri("/y")` — long form (this PR) * - * The long-form `webClient.method(HttpMethod.X).uri("/y")` chain is - * intentionally deferred to a follow-up: it requires walk-up logic - * to recover the verb from a sibling `call_expression`, and we can - * land 80% of real-world Kotlin Spring consumer coverage with the - * three simpler patterns above. + * The long form puts the verb on a sibling `call_expression` two hops + * away from the path. Rather than introducing imperative walk-up logic, + * we use a single deeper tree-sitter query that matches the full chain + * structurally — see `WEB_CLIENT_LONG_PATTERNS` below. The verb is + * captured directly as the `simple_identifier` of `HttpMethod.X`, so + * variable-bound verbs (`val verb = HttpMethod.PATCH; webClient.method(verb)...`) + * are intentionally NOT picked up — those need a graph-aware resolver + * and are out of scope for source-scan. * * tree-sitter-kotlin (fwcd) AST shapes used here: * class_declaration @@ -109,6 +113,16 @@ const WEB_CLIENT_SHORT_TO_HTTP: Record = { patch: 'PATCH', }; +/** + * Allowed HTTP verbs for the WebClient long-form path + * `webClient.method(HttpMethod.X).uri("/y")`. Compiled once at module + * load (instead of inside the scan loop) per maintainer feedback on + * PR #1884. Mirrors the keys of `WEB_CLIENT_SHORT_TO_HTTP` above — + * keeping HEAD/OPTIONS/TRACE intentionally excluded for symmetry + * with the short form and the Java plugin. + */ +const WEB_CLIENT_LONG_VERB_RE = /^(GET|POST|PUT|DELETE|PATCH)$/; + /** * Build the plugin only if the Kotlin grammar is available. Compiling * the queries against a null grammar would throw at module load time @@ -265,8 +279,9 @@ function buildKotlinPlugin(language: unknown): HttpLanguagePlugin { // - outer call's first value_argument is a string literal // // The long-form `webClient.method(HttpMethod.GET).uri("/x")` chain - // uses an extra navigation hop and an enum field access — it's - // intentionally out of scope here (see file header). + // uses an extra navigation hop and an enum field access — handled + // by `WEB_CLIENT_LONG_PATTERNS` below, separately so each query is + // straightforward to reason about. const WEB_CLIENT_SHORT_PATTERNS = compilePatterns({ name: 'kotlin-web-client-short', language, @@ -290,6 +305,59 @@ function buildKotlinPlugin(language: unknown): HttpLanguagePlugin { ], } satisfies LanguagePatterns>); + // ─── Consumer: Spring WebClient (long form) ─────────────────────────── + // The fluent long form passes the verb as a `HttpMethod.X` enum field + // access through `.method(...)`, then carries the path on a separate + // `.uri(...)` hop further down the chain: + // + // webClient.method(HttpMethod.GET).uri("/x").retrieve().awaitBody() + // + // Compared to the short form there are two extra structural hops: + // - the inner `.method(...)` `call_expression` has a `value_argument` + // whose payload is itself a `navigation_expression` (HttpMethod → .GET) + // - the outer `.uri(...)` is reached via one more + // `navigation_expression` wrapping that inner call + // + // We capture the verb at the `simple_identifier` under `HttpMethod`'s + // `navigation_suffix`. That `simple_identifier` is the literal field + // name (`GET`, `POST`, ...) used in source — Kotlin enum fields by + // convention are upper-case, matching `HttpMethod` from + // `org.springframework.http`. We forward the captured text as-is. + // + // Variable-bound verbs (`val verb = HttpMethod.PATCH; webClient.method(verb)...`) + // do NOT match — they fail the `(navigation_expression ...)` shape + // because the value_argument carries a bare `simple_identifier` instead + // of a `HttpMethod.X` field access. This is intentional: source-scan + // can't follow the binding without graph context. Pinned by an + // anti-overreach test in the consumer suite. + const WEB_CLIENT_LONG_PATTERNS = compilePatterns({ + name: 'kotlin-web-client-long', + language, + patterns: [ + { + meta: {}, + query: ` + (call_expression + (navigation_expression + (call_expression + (navigation_expression + (simple_identifier) @obj (#eq? @obj "webClient") + (navigation_suffix + (simple_identifier) @method_call (#eq? @method_call "method"))) + (call_suffix + (value_arguments + . (value_argument + (navigation_expression + (simple_identifier) @httpMethodCls (#eq? @httpMethodCls "HttpMethod") + (navigation_suffix (simple_identifier) @verb)))))) + (navigation_suffix (simple_identifier) @uri (#eq? @uri "uri"))) + (call_suffix + (value_arguments . (value_argument . (string_literal) @path)))) + `, + }, + ], + } satisfies LanguagePatterns>); + // ─── Consumer: OkHttp Request.Builder().url("/x") ───────────────────── // Kotlin parses `Request.Builder()` as a `call_expression` whose // callee is a `navigation_expression` (Request → .Builder), NOT as @@ -437,6 +505,33 @@ function buildKotlinPlugin(language: unknown): HttpLanguagePlugin { }); } + // ─── Consumers: WebClient long form (.method(HttpMethod.X) → .uri) ─ + for (const match of runCompiledPatterns(WEB_CLIENT_LONG_PATTERNS, tree)) { + const verbNode = match.captures.verb; + const pathNode = match.captures.path; + if (!verbNode || !pathNode) continue; + // The captured text is the literal `HttpMethod.X` field name. + // Spring's `org.springframework.http.HttpMethod` defines GET, + // POST, PUT, DELETE, PATCH, HEAD, OPTIONS, TRACE — we only + // emit for the five verbs we already handle elsewhere, so + // exotic ones are silently skipped (consistent with the + // short form's WEB_CLIENT_SHORT_TO_HTTP guard). The accepted + // verb regex is hoisted to module scope (see + // `WEB_CLIENT_LONG_VERB_RE` near the top of this file). + const verbText = verbNode.text; + if (!WEB_CLIENT_LONG_VERB_RE.test(verbText)) continue; + const path = unquoteLiteral(pathNode.text); + if (path === null) continue; + out.push({ + role: 'consumer', + framework: 'spring-web-client', + method: verbText, + path, + name: null, + confidence: 0.7, + }); + } + // ─── Consumers: OkHttp Request.Builder().url("path") ──────────── for (const match of runCompiledPatterns(OK_HTTP_PATTERNS, tree)) { const pathNode = match.captures.path; diff --git a/gitnexus/src/core/group/extractors/http-patterns/python.ts b/gitnexus/src/core/group/extractors/http-patterns/python.ts index 1667d0de9..5dc314d35 100644 --- a/gitnexus/src/core/group/extractors/http-patterns/python.ts +++ b/gitnexus/src/core/group/extractors/http-patterns/python.ts @@ -6,7 +6,7 @@ import { unquoteLiteral, type LanguagePatterns, } from '../tree-sitter-scanner.js'; -import type { HttpDetection, HttpLanguagePlugin } from './types.js'; +import type { HttpDetection, HttpLanguagePlugin, RepoContext } from './types.js'; /** * Python HTTP plugin. Handles: @@ -29,9 +29,13 @@ const FASTAPI_VERBS: Record = { patch: 'PATCH', }; -// ─── Provider: FastAPI @app.get/... ────────────────────────────────── -const FASTAPI_PATTERNS = compilePatterns({ - name: 'python-fastapi', +// ─── Provider: FastAPI @app. / @router. ────────────────── +// Two separate patterns so we can tag detections by decorator object. +// Only `@router.*` detections participate in `include_router(prefix=)` +// path-prefix joining (see `PythonRepoContext` + `joinPrefix`); `@app.*` +// routes already carry their final path verbatim. +const FASTAPI_APP_PATTERNS = compilePatterns({ + name: 'python-fastapi-app', language: Python, patterns: [ { @@ -48,6 +52,138 @@ const FASTAPI_PATTERNS = compilePatterns({ ], } satisfies LanguagePatterns>); +const FASTAPI_ROUTER_PATTERNS = compilePatterns({ + name: 'python-fastapi-router', + language: Python, + patterns: [ + { + meta: {}, + query: ` + (decorator + (call + function: (attribute + object: (identifier) @obj (#eq? @obj "router") + attribute: (identifier) @method (#match? @method "^(get|post|put|delete|patch)$")) + arguments: (argument_list . (string) @path))) + `, + }, + ], +} satisfies LanguagePatterns>); + +// ─── include_router(, prefix='/x') across the repo ──────── +// Two shapes are common: +// app.include_router(assistant.router, prefix='/ai') +// app.include_router(my_router, prefix='/ai') +// The first names the originating module via `.router`; the second +// references a name imported into the host file. We capture both. +const INCLUDE_ROUTER_ATTR_PATTERNS = compilePatterns({ + name: 'python-fastapi-include-router-attr', + language: Python, + patterns: [ + { + meta: {}, + // Match any `.include_router(.router, ..., prefix='/x')` + // call. We deliberately do NOT pin `` to the literal name `app` + // — production code routinely uses `api`, `application`, `asgi_app`, + // etc. The shape (`include_router` invoked with a router argument and + // a `prefix=` keyword) is specific enough on its own; restricting the + // host produces false negatives without removing meaningful false + // positives. + query: ` + (call + function: (attribute + attribute: (identifier) @incl (#eq? @incl "include_router")) + arguments: (argument_list + (attribute + object: (identifier) @router_module + attribute: (identifier) @router_attr (#eq? @router_attr "router")) + (keyword_argument + name: (identifier) @kw (#eq? @kw "prefix") + value: (string) @prefix))) + `, + }, + ], +} satisfies LanguagePatterns>); + +const INCLUDE_ROUTER_NAME_PATTERNS = compilePatterns({ + name: 'python-fastapi-include-router-name', + language: Python, + patterns: [ + { + meta: {}, + // Same `` rationale as INCLUDE_ROUTER_ATTR_PATTERNS — see above. + query: ` + (call + function: (attribute + attribute: (identifier) @incl (#eq? @incl "include_router")) + arguments: (argument_list + (identifier) @router_name + (keyword_argument + name: (identifier) @kw (#eq? @kw "prefix") + value: (string) @prefix))) + `, + }, + ], +} satisfies LanguagePatterns>); + +// `from .api.assistant import router` style — used together with +// INCLUDE_ROUTER_NAME so we can map a local name back to its module +// path, then back to the file the router was declared in. +const FROM_IMPORT_ROUTER_PATTERNS = compilePatterns({ + name: 'python-fastapi-from-import-router', + language: Python, + patterns: [ + { + meta: {}, + query: ` + (import_from_statement + module_name: (_) @module + name: (dotted_name (identifier) @imported (#eq? @imported "router"))) + `, + }, + { + meta: {}, + query: ` + (import_from_statement + module_name: (_) @module + name: (aliased_import + name: (dotted_name (identifier) @imported (#eq? @imported "router")) + alias: (identifier) @alias)) + `, + }, + ], +} satisfies LanguagePatterns>); + +// `from api import users` / `from api import users as u` — module-level +// imports where the imported name is itself the module that owns +// `.router`. Lets Shape A (`.include_router(.router, …)`) +// look up the full package path of `` and pin the prefix onto the +// exact file (`api/users.py`) rather than every file basenamed `users.py`. +const FROM_IMPORT_MODULE_PATTERNS = compilePatterns({ + name: 'python-fastapi-from-import-module', + language: Python, + patterns: [ + { + meta: {}, + query: ` + (import_from_statement + module_name: (_) @module + name: (dotted_name (identifier) @imported)) + `, + }, + { + meta: {}, + query: ` + (import_from_statement + module_name: (_) @module + name: (aliased_import + name: (dotted_name (identifier) @imported) + alias: (identifier) @alias)) + `, + }, + ], +} satisfies LanguagePatterns>); + // ─── Consumer: requests.get/post/... ────────────────────────────────── const REQUESTS_VERB_PATTERNS = compilePatterns({ name: 'python-requests-verb', @@ -447,15 +583,226 @@ const HTTPX_ASYNC_CLIENT_GENERIC_PATTERNS = compilePatterns({ ], } satisfies LanguagePatterns>); +// ─── prepareRepo: build router-module → prefix list map ───────────── +// +// FastAPI splits route declarations across files: handler decorators +// live in `api/.py` while `app.include_router(.router, +// prefix='/ai')` lives in `main.py`. A per-file plugin scan therefore +// can't see the prefix that ought to be applied. We resolve this by +// running a one-shot pre-pass over the repo: for every file that +// hosts an `app.include_router(...)` we record the module the router +// came from (either via `module.router` attribute access, or via a +// local name resolved through a `from import router` import) +// together with the prefix string. At scan time the python plugin +// looks up the current file's module key in this map and joins each +// prefix with each `@router.` decorator's path. +// +// Multiple prefixes for the same module are kept and emitted as +// separate detections — this matches FastAPI's behaviour when one +// router is mounted under several prefixes. +// +// Module keying is two-tiered to avoid prefix bleed between same-named +// files in different packages (e.g. `api/users.py` vs `admin/users.py`): +// • short key — file basename without `.py` (`users`) +// • long key — `/` (`api/users`) +// The pre-pass records prefixes against the long key whenever the import +// site supplies enough context (`from api.users import router as ...` → +// long key `api/users`); otherwise it falls back to the short key. +// At scan time the file's own long key is consulted first; only when no +// long-key entry targets this file do we look up the short key. This +// preserves the previous coarse-grained behaviour where context is +// missing while delivering precision wherever the import statement +// gives us a multi-segment module path. +interface PythonRepoContext { + /** `/` → set of prefixes (precise, package-aware) */ + prefixesByLongKey: Map>; + /** stem only → set of prefixes (basename fallback, may collide) */ + prefixesByShortKey: Map>; +} + +/** Strip `.py` and return the bare basename (e.g. `api/users.py` → `users`). */ +function fileShortKey(rel: string): string { + const slash = rel.lastIndexOf('/'); + const file = slash >= 0 ? rel.slice(slash + 1) : rel; + return file.endsWith('.py') ? file.slice(0, -3) : file; +} + +/** + * Long key for a `.py` file: parent directory + stem, joined with `/`. + * Files at the repo root return the empty string (no parent), in which + * case callers should fall back to the short key. + */ +function fileLongKey(rel: string): string { + const noExt = rel.endsWith('.py') ? rel.slice(0, -3) : rel; + const lastSlash = noExt.lastIndexOf('/'); + if (lastSlash < 0) return ''; + const beforeLast = noExt.slice(0, lastSlash); + const stem = noExt.slice(lastSlash + 1); + const prevSlash = beforeLast.lastIndexOf('/'); + const parent = prevSlash >= 0 ? beforeLast.slice(prevSlash + 1) : beforeLast; + return `${parent}/${stem}`; +} + +/** Last `.`-separated segment of a (possibly relative) module path. */ +function lastSegmentOfDotted(text: string): string { + const stripped = text.replace(/^\.+/, ''); + if (!stripped) return ''; + const dot = stripped.lastIndexOf('.'); + return dot >= 0 ? stripped.slice(dot + 1) : stripped; +} + +/** + * Last two `.`-separated segments of a (possibly relative) module path + * joined with `/`, e.g. `api.users` → `api/users`. Single-segment paths + * and pure-dot inputs return the empty string; callers should fall back + * to the short key in that case. + */ +function lastTwoSegmentsAsLongKey(text: string): string { + const stripped = text.replace(/^\.+/, ''); + if (!stripped) return ''; + const last = stripped.lastIndexOf('.'); + if (last <= 0) return ''; + const beforeLast = stripped.slice(0, last); + const stem = stripped.slice(last + 1); + const prev = beforeLast.lastIndexOf('.'); + const parent = prev >= 0 ? beforeLast.slice(prev + 1) : beforeLast; + return `${parent}/${stem}`; +} + +function recordPrefix(target: Map>, key: string, prefix: string): void { + const set = target.get(key) ?? new Set(); + set.add(prefix); + target.set(key, set); +} + +function buildPythonRepoContext( + files: string[], + parser: Parser, + readFile: (rel: string) => string | null, + parseSource: (parser: Parser, src: string) => Parser.Tree | null, +): PythonRepoContext { + const prefixesByLongKey = new Map>(); + const prefixesByShortKey = new Map>(); + + // Pre-pass over .py files. We deliberately run this even on files + // that don't contain `include_router` — the cost of an extra parse + // is bounded by the file count, and detecting `include_router` + // beforehand would require its own grep/scan. + for (const rel of files) { + if (!rel.endsWith('.py')) continue; + const src = readFile(rel); + if (!src) continue; + if (!src.includes('include_router')) continue; + parser.setLanguage(Python); + const tree = parseSource(parser, src); + if (!tree) continue; + + // Local name → (short, long) map for the current file, populated + // from `from import router [as ]` statements. The + // alias (or 'router' when there is no alias) is the local name + // we'll later see passed to `.include_router`. + interface LocalImport { + moduleShort: string; + moduleLong: string; + } + const localNameToModule = new Map(); + for (const m of runCompiledPatterns(FROM_IMPORT_ROUTER_PATTERNS, tree)) { + const moduleNode = m.captures.module; + const aliasNode = m.captures.alias; + const importedNode = m.captures.imported; + if (!moduleNode || !importedNode) continue; + const localName = aliasNode?.text ?? importedNode.text; + const moduleShort = lastSegmentOfDotted(moduleNode.text); + if (!moduleShort) continue; + const moduleLong = lastTwoSegmentsAsLongKey(moduleNode.text); + localNameToModule.set(localName, { moduleShort, moduleLong }); + } + + // Module-alias map: name imported from a multi-segment package → + // long key. Lets Shape A look up the precise file for `.router` + // even when `` collides with another package's basename. + const localNameToModuleAlias = new Map(); + for (const m of runCompiledPatterns(FROM_IMPORT_MODULE_PATTERNS, tree)) { + const moduleNode = m.captures.module; + const importedNode = m.captures.imported; + const aliasNode = m.captures.alias; + if (!moduleNode || !importedNode) continue; + // Skip the `router` shape — already handled by FROM_IMPORT_ROUTER_PATTERNS + // above and stored under its router-aware semantics. + if (importedNode.text === 'router') continue; + const moduleLong = lastTwoSegmentsAsLongKey(`${moduleNode.text}.${importedNode.text}`); + if (!moduleLong) continue; + const localName = aliasNode?.text ?? importedNode.text; + localNameToModuleAlias.set(localName, moduleLong); + } + + // Shape A: `.include_router(.router, prefix='/x')`. + // The call site gives us only a short module name. We promote to a + // long key when the same file imports `` via either + // `from import ` (recorded in `localNameToModuleAlias` + // — the typical pattern) or, less commonly, a router-aware import + // statement. Only fall back to the basename short key when neither + // alias is available. + for (const m of runCompiledPatterns(INCLUDE_ROUTER_ATTR_PATTERNS, tree)) { + const modNode = m.captures.router_module; + const prefixNode = m.captures.prefix; + if (!modNode || !prefixNode) continue; + const prefix = unquoteLiteral(prefixNode.text); + if (prefix === null) continue; + const moduleShort = modNode.text; + const aliasLong = localNameToModuleAlias.get(moduleShort); + const sameFileImport = localNameToModule.get(moduleShort); + const longKey = aliasLong ?? sameFileImport?.moduleLong; + if (longKey) { + recordPrefix(prefixesByLongKey, longKey, prefix); + } else { + recordPrefix(prefixesByShortKey, moduleShort, prefix); + } + } + + // Shape B: `.include_router(my_router, prefix='/x')` — resolve + // `my_router` via the import map built above. Whenever the import + // statement supplied a multi-segment module path the long key is + // recorded, eliminating cross-package collisions. + for (const m of runCompiledPatterns(INCLUDE_ROUTER_NAME_PATTERNS, tree)) { + const nameNode = m.captures.router_name; + const prefixNode = m.captures.prefix; + if (!nameNode || !prefixNode) continue; + const localImp = localNameToModule.get(nameNode.text); + if (!localImp) continue; + const prefix = unquoteLiteral(prefixNode.text); + if (prefix === null) continue; + if (localImp.moduleLong) { + recordPrefix(prefixesByLongKey, localImp.moduleLong, prefix); + } else { + recordPrefix(prefixesByShortKey, localImp.moduleShort, prefix); + } + } + } + + return { prefixesByLongKey, prefixesByShortKey }; +} + +function joinPrefix(prefix: string, route: string): string { + // Mirror FastAPI's path joining: trim trailing slash off prefix, + // ensure exactly one leading slash on the result. + const p = prefix.replace(/\/+$/, ''); + const r = route.startsWith('/') ? route : `/${route}`; + return `${p}${r}`; +} export const PYTHON_HTTP_PLUGIN: HttpLanguagePlugin = { name: 'python-http', language: Python, - scan(tree) { + prepareRepo({ files, parser, readFile, parseSource }): RepoContext { + return buildPythonRepoContext(files, parser, readFile, parseSource); + }, + scan(tree, repoContext, fileRel) { const out: HttpDetection[] = []; const httpxAsyncClients = collectHttpxAsyncClients(tree); + const ctx = repoContext as PythonRepoContext | undefined; - // Providers: FastAPI - for (const match of runCompiledPatterns(FASTAPI_PATTERNS, tree)) { + // Providers: FastAPI @app.("/path") — already absolute path. + for (const match of runCompiledPatterns(FASTAPI_APP_PATTERNS, tree)) { const methodNode = match.captures.method; const pathNode = match.captures.path; if (!methodNode || !pathNode) continue; @@ -473,6 +820,47 @@ export const PYTHON_HTTP_PLUGIN: HttpLanguagePlugin = { }); } + // Providers: FastAPI @router.("/path") — must be joined + // with the prefix(es) declared at the include_router site. When + // no prefix is found we still emit the unprefixed path so this + // change is strictly additive vs. the prior @app-only behaviour; + // when the same router is mounted under multiple prefixes we emit + // one detection per prefix. + for (const match of runCompiledPatterns(FASTAPI_ROUTER_PATTERNS, tree)) { + const methodNode = match.captures.method; + const pathNode = match.captures.path; + if (!methodNode || !pathNode) continue; + const httpMethod = FASTAPI_VERBS[methodNode.text]; + if (!httpMethod) continue; + const rawPath = unquoteLiteral(pathNode.text); + if (rawPath === null) continue; + + // Long key first (precise, package-aware), short key as fallback. + // Mirrors the ingestion-side resolution in parse-impl.ts so the + // graph nodes and group contracts agree on which prefix applies. + const longKey = fileRel ? fileLongKey(fileRel) : ''; + const longPrefixes = longKey ? ctx?.prefixesByLongKey.get(longKey) : undefined; + const shortKey = fileRel ? fileShortKey(fileRel) : ''; + const shortPrefixes = + longPrefixes || !shortKey ? undefined : ctx?.prefixesByShortKey.get(shortKey); + const prefixSet = longPrefixes ?? shortPrefixes; + const paths = + prefixSet && prefixSet.size > 0 + ? [...prefixSet].map((p) => joinPrefix(p, rawPath)) + : [rawPath]; + + for (const p of paths) { + out.push({ + role: 'provider', + framework: 'fastapi', + method: httpMethod, + path: p, + name: null, + confidence: 0.8, + }); + } + } + // Consumers: requests. for (const match of runCompiledPatterns(REQUESTS_VERB_PATTERNS, tree)) { const methodNode = match.captures.method; diff --git a/gitnexus/src/core/group/extractors/http-patterns/types.ts b/gitnexus/src/core/group/extractors/http-patterns/types.ts index 6df0ede28..fb4ab09cb 100644 --- a/gitnexus/src/core/group/extractors/http-patterns/types.ts +++ b/gitnexus/src/core/group/extractors/http-patterns/types.ts @@ -40,6 +40,16 @@ export interface HttpDetection { confidence: number; } +export interface HttpScanInput { + filePath: string; + tree: Parser.Tree; +} + +export interface HttpFileDetections { + filePath: string; + detections: HttpDetection[]; +} + /** * One language-scoped HTTP plugin. The plugin owns the tree-sitter * grammar and the `scan` function that translates a parsed tree into @@ -51,15 +61,54 @@ export interface HttpDetection { * `LanguagePatterns.language` in `tree-sitter-scanner.ts` — the * grammar modules export different shapes. */ +/** + * Per-repo state a plugin can build during a `prepareRepo` pass before + * any per-file `scan` is invoked. The orchestrator threads this opaque + * value back into each `scan` call so plugins can resolve cross-file + * facts (e.g. FastAPI `app.include_router(prefix=...)` mappings live + * in `main.py` but apply to handlers declared in `api/*.py`). + * + * Plugins that have no cross-file state can omit `prepareRepo` and + * receive `undefined`. + */ +export type RepoContext = unknown; + export interface HttpLanguagePlugin { /** Human-readable plugin name for diagnostics. */ name: string; /** tree-sitter grammar object (passed to the shared parser). */ language: unknown; + /** + * Optional pre-pass: walk the relevant files in the repo and produce + * an opaque context that `scan` can use to resolve cross-file facts. + * Implementations must not throw — return undefined on any error so + * the orchestrator falls back to context-less scanning. + */ + prepareRepo?(args: { + repoPath: string; + files: string[]; + parser: Parser; + readFile: (rel: string) => string | null; + parseSource: (parser: Parser, src: string) => Parser.Tree | null; + }): RepoContext | undefined; /** * Scan a parsed tree and return zero or more HTTP detections. Plugins * must not throw — they should swallow per-match errors so a single * malformed construct does not abort the whole file. + * + * `repoContext` is whatever the plugin's `prepareRepo` produced (or + * `undefined` if there is no `prepareRepo`). + * + * `fileRel` is the repo-relative path of the file being scanned; + * plugins that resolve cross-file facts (e.g. FastAPI router prefix + * joining) need it to key into `repoContext`. Optional so existing + * single-file plugins can keep their unary `scan(tree)` shape. */ - scan(tree: Parser.Tree): HttpDetection[]; + scan(tree: Parser.Tree, repoContext?: RepoContext, fileRel?: string): HttpDetection[]; + /** + * Optional project-level scan hook for language rules that require + * multiple files, such as Java controllers inheriting Spring mappings + * from annotated interfaces. + */ + scanProject?(files: readonly HttpScanInput[]): HttpFileDetections[]; } diff --git a/gitnexus/src/core/group/extractors/http-route-extractor.ts b/gitnexus/src/core/group/extractors/http-route-extractor.ts index 54aeb9150..e14b9b701 100644 --- a/gitnexus/src/core/group/extractors/http-route-extractor.ts +++ b/gitnexus/src/core/group/extractors/http-route-extractor.ts @@ -6,7 +6,13 @@ import type { ContractExtractor, CypherExecutor } from '../contract-extractor.js import type { ExtractedContract, RepoHandle } from '../types.js'; import { readSafe } from './fs-utils.js'; import { parseSourceSafe } from '../../tree-sitter/safe-parse.js'; -import { getPluginForFile, HTTP_SCAN_GLOB, type HttpDetection } from './http-patterns/index.js'; +import { + getPluginForFile, + HTTP_SCAN_GLOB, + type HttpDetection, + type HttpLanguagePlugin, + type HttpScanInput, +} from './http-patterns/index.js'; /** * Language-agnostic orchestrator for HTTP route (provider + consumer) @@ -160,31 +166,85 @@ export class HttpRouteExtractor implements ContractExtractor { // both graph-assisted enrichment and source-scan emission. const parser = new Parser(); const cachedDetections = new Map(); - const getDetections = (rel: string): HttpDetection[] => { - const cached = cachedDetections.get(rel); - if (cached) return cached; + const cachedInputs = new Map< + string, + { plugin: HttpLanguagePlugin; input: HttpScanInput; repoContext: unknown } | null + >(); + const projectDetections = new Map(); + let projectScanComplete = false; + + // Per-plugin cross-file context (e.g. Python's FastAPI router → + // include_router(prefix=...) map). Built lazily on first + // `getDetections` call for a file the plugin handles, scoped to the + // file list returned by `getScannedFiles`. Stored by plugin name so + // a repo with multiple languages keeps each plugin's context + // independent. + const repoContextByPlugin = new Map(); + const ensureRepoContext = async ( + plugin: ReturnType, + ): Promise => { + if (!plugin || typeof plugin.prepareRepo !== 'function') return undefined; + if (repoContextByPlugin.has(plugin.name)) return repoContextByPlugin.get(plugin.name); + try { + const ctx = plugin.prepareRepo({ + repoPath, + files: await getScannedFiles(), + parser, + readFile: (rel) => readSafe(repoPath, rel), + parseSource: (p, src) => parseSourceSafe(p, src), + }); + repoContextByPlugin.set(plugin.name, ctx); + return ctx; + } catch { + repoContextByPlugin.set(plugin.name, undefined); + return undefined; + } + }; + + const getScanInput = async ( + rel: string, + ): Promise<{ + plugin: HttpLanguagePlugin; + input: HttpScanInput; + repoContext: unknown; + } | null> => { + if (cachedInputs.has(rel)) return cachedInputs.get(rel) ?? null; const plugin = getPluginForFile(rel); if (!plugin) { - cachedDetections.set(rel, []); - return []; + cachedInputs.set(rel, null); + return null; } + const repoContext = await ensureRepoContext(plugin); const content = readSafe(repoPath, rel); if (!content) { - cachedDetections.set(rel, []); - return []; + cachedInputs.set(rel, null); + return null; } try { parser.setLanguage(plugin.language); const tree = parseSourceSafe(parser, content); - const detections = plugin.scan(tree); - cachedDetections.set(rel, detections); - return detections; + const input = { filePath: rel, tree }; + const item = { plugin, input, repoContext }; + cachedInputs.set(rel, item); + return item; } catch { - cachedDetections.set(rel, []); - return []; + cachedInputs.set(rel, null); + return null; } }; + const getDetections = async (rel: string): Promise => { + const cached = cachedDetections.get(rel); + if (cached) return cached; + const scanInput = await getScanInput(rel); + const ownDetections = scanInput + ? scanInput.plugin.scan(scanInput.input.tree, scanInput.repoContext, rel) + : []; + const detections = [...ownDetections, ...(projectDetections.get(rel) ?? [])]; + cachedDetections.set(rel, detections); + return detections; + }; + // Glob the source-scan file list at most once per extract() — // both provider and consumer fallback paths share the same list. let scannedFiles: string[] | null = null; @@ -194,20 +254,46 @@ export class HttpRouteExtractor implements ContractExtractor { return scannedFiles; }; + const collectProjectDetections = async (files: string[]): Promise => { + if (projectScanComplete) return; + projectScanComplete = true; + const byPlugin = new Map(); + for (const rel of files) { + const scanInput = await getScanInput(rel); + if (!scanInput?.plugin.scanProject) continue; + const items = byPlugin.get(scanInput.plugin) ?? []; + items.push(scanInput.input); + byPlugin.set(scanInput.plugin, items); + } + + for (const [plugin, inputs] of byPlugin) { + const results = plugin.scanProject?.(inputs) ?? []; + for (const result of results) { + const existing = projectDetections.get(result.filePath) ?? []; + projectDetections.set(result.filePath, [...existing, ...result.detections]); + } + } + + cachedDetections.clear(); + }; + + const files = await getScannedFiles(); + await collectProjectDetections(files); + const graphProviders = dbExecutor != null ? await this.extractProvidersGraph(dbExecutor, getDetections) : []; // Source scan always runs to capture routes in languages/files not covered // by graph edges; the glob and per-file parse results are cached above. const providers = this.mergeGraphAndSourceContracts( graphProviders, - this.extractProvidersSourceScan(await getScannedFiles(), getDetections), + await this.extractProvidersSourceScan(files, getDetections), ); const graphConsumers = dbExecutor != null ? await this.extractConsumersGraph(dbExecutor, getDetections) : []; const consumers = this.mergeGraphAndSourceContracts( graphConsumers, - this.extractConsumersSourceScan(await getScannedFiles(), getDetections), + await this.extractConsumersSourceScan(files, getDetections), ); return [...providers, ...consumers]; @@ -232,7 +318,7 @@ export class HttpRouteExtractor implements ContractExtractor { private async extractProvidersGraph( db: CypherExecutor, - getDetections: (rel: string) => HttpDetection[], + getDetections: (rel: string) => Promise, ): Promise { const out: ExtractedContract[] = []; let rows: Record[]; @@ -254,7 +340,7 @@ export class HttpRouteExtractor implements ContractExtractor { // helpers — tree-sitter gives both pieces of information // structurally. Always run the lookup: even when method is set by // `methodFromRouteReason`, we still need the handler name. - const detections = filePath ? getDetections(filePath) : []; + const detections = filePath ? await getDetections(filePath) : []; const providerDetections = detections.filter((d) => d.role === 'provider'); let handlerName: string | null = null; const normalizedRoute = normalizeHttpPath(routePath); @@ -331,13 +417,13 @@ export class HttpRouteExtractor implements ContractExtractor { // ─── Source-scan providers ───────────────────────────────────────── - private extractProvidersSourceScan( + private async extractProvidersSourceScan( files: string[], - getDetections: (rel: string) => HttpDetection[], - ): ExtractedContract[] { + getDetections: (rel: string) => Promise, + ): Promise { const out: ExtractedContract[] = []; for (const rel of files) { - const detections = getDetections(rel); + const detections = await getDetections(rel); for (const d of detections) { if (d.role !== 'provider') continue; const pathNorm = normalizeHttpPath(d.path); @@ -366,7 +452,7 @@ export class HttpRouteExtractor implements ContractExtractor { private async extractConsumersGraph( db: CypherExecutor, - getDetections: (rel: string) => HttpDetection[], + getDetections: (rel: string) => Promise, ): Promise { const out: ExtractedContract[] = []; let rows: Record[]; @@ -382,7 +468,7 @@ export class HttpRouteExtractor implements ContractExtractor { let method = 'GET'; // Prefer the plugin's detected method if we can find a matching // fetch/axios call in the same file. - const detections = filePath ? getDetections(filePath) : []; + const detections = filePath ? await getDetections(filePath) : []; // Symmetric to the provider path: if multiple consumer calls in // the same file share the same normalized path (e.g. a GET // fetch AND a POST fetch to `/api/orders`), `.find()` silently @@ -436,13 +522,13 @@ export class HttpRouteExtractor implements ContractExtractor { // ─── Source-scan consumers ───────────────────────────────────────── - private extractConsumersSourceScan( + private async extractConsumersSourceScan( files: string[], - getDetections: (rel: string) => HttpDetection[], - ): ExtractedContract[] { + getDetections: (rel: string) => Promise, + ): Promise { const out: ExtractedContract[] = []; for (const rel of files) { - const detections = getDetections(rel); + const detections = await getDetections(rel); for (const d of detections) { if (d.role !== 'consumer') continue; const pathNorm = normalizeConsumerPath(d.path); diff --git a/gitnexus/src/core/ingestion/cobol-processor.ts b/gitnexus/src/core/ingestion/cobol-processor.ts index 2e0551770..b7f3835b1 100644 --- a/gitnexus/src/core/ingestion/cobol-processor.ts +++ b/gitnexus/src/core/ingestion/cobol-processor.ts @@ -150,9 +150,24 @@ export const processCobol = ( const entry = copybookMap.get(name.toUpperCase()); return entry ? entry.path : null; }; + // Memoize preprocessed copybook content for the duration of this + // processCobol call. A single copybook is COPYed by many programs (and at + // many COPY sites within a program); without this cache + // preprocessCobolSource would re-run once per COPY site — + // O(programs × copybooks) preprocessing passes over the same content. + // Keyed by the resolved copybook path. REPLACING is applied later by the + // expander on the returned (pre-REPLACING) content (see + // cobol-copy-expander.ts readFile→applyReplacing), so caching the + // pre-REPLACING preprocessed text here is safe and per-call-scoped. + const preprocessedCopyCache = new Map(); const readCopy = (copyPath: string): string | null => { + const cached = preprocessedCopyCache.get(copyPath); + if (cached !== undefined) return cached; const content = copybookByPath.get(copyPath); - return content ? preprocessCobolSource(content) : null; + if (!content) return null; // preserves original falsy→null (missing/empty) + const preprocessed = preprocessCobolSource(content); + preprocessedCopyCache.set(copyPath, preprocessed); + return preprocessed; }; // Track module names for cross-program CALL resolution diff --git a/gitnexus/src/core/ingestion/languages/cobol/captures.ts b/gitnexus/src/core/ingestion/languages/cobol/captures.ts index 69a6c80b9..04906f7a5 100644 --- a/gitnexus/src/core/ingestion/languages/cobol/captures.ts +++ b/gitnexus/src/core/ingestion/languages/cobol/captures.ts @@ -80,7 +80,11 @@ export function emitCobolScopeCaptures( : rangeOf(startLine, startCol, endLine, endCol); const grouped: Record = { - '@scope.module': capture('@scope.module', nameRange, name), + '@scope.module': capture( + '@scope.module', + rangeOf(startLine, startCol, endLine, endCol), + name, + ), '@declaration.program': capture( '@declaration.program', rangeOf(startLine, startCol, endLine, endCol), @@ -118,7 +122,11 @@ export function emitCobolScopeCaptures( : rangeOf(startLine, startCol, endLine, endCol); const grouped: Record = { - '@scope.module': capture('@scope.module', nameRange, prog.name), + '@scope.module': capture( + '@scope.module', + rangeOf(startLine, startCol, endLine, endCol), + prog.name, + ), '@declaration.program': capture( '@declaration.program', rangeOf(startLine, startCol, endLine, endCol), diff --git a/gitnexus/src/core/ingestion/languages/cpp/adl.ts b/gitnexus/src/core/ingestion/languages/cpp/adl.ts index 0bcda9322..9c23c73cc 100644 --- a/gitnexus/src/core/ingestion/languages/cpp/adl.ts +++ b/gitnexus/src/core/ingestion/languages/cpp/adl.ts @@ -24,22 +24,18 @@ * V2 additionally walks class ancestors (via MRO), so base-class enclosing * namespaces also contribute associated namespaces. * - * **GitNexus approximation (not strict ISO C++ ADL):** passing a qualified - * function reference like `utils::worker` contributes `utils` to the associated - * set, enabling resolution of unqualified calls like `with_callback(utils::worker)` - * to `utils::with_callback`. Under ISO C++ `[basic.lookup.argdep]`, associated - * entities for function-type arguments come from the **parameter types and return - * type** of each function in the overload set — NOT the function's enclosing - * namespace. For `void worker()`, the standard-compliant associated set is empty. - * GitNexus instead contributes the enclosing namespace of any Function/Method - * def whose simple name matches, because it enables the dominant real-world ADL - * pattern at reasonable precision cost. + * Function-reference arguments follow ISO C++ `[basic.lookup.argdep]`: + * associated entities come from the parameter types and return type of each + * referenced function in the overload set, not from the function's enclosing + * namespace. For `void worker()`, the associated set is empty. For + * `void worker(api::Token)` or `api::Token make_token()`, `api` is associated + * through `Token`. * - * For qualified refs (e.g. `utils::worker`) the namespace is confirmed via a - * workspace lookup (only contributed when a Function/Method named `worker` exists - * in `utils`). For unqualified refs the workspace is searched for any Function - * def with that simple name. Locally-declared function-pointer variables - * (e.g. `void (*g)()`) and function parameters are excluded from this path. + * For qualified refs (e.g. `utils::worker`) the workspace lookup is restricted + * to functions/methods named `worker` in `utils`; for unqualified refs the + * workspace is searched for matching functions/methods by simple name. Locally + * declared function-pointer variables and function parameters are excluded + * from this path. * * ADL candidates are merged with ordinary unqualified-lookup candidates * in the free-call fallback before overload narrowing. @@ -70,6 +66,7 @@ import type { ParsedFile, ScopeId, SymbolDefinition } from 'gitnexus-shared'; import type { ScopeResolutionIndexes } from '../../model/scope-resolution-indexes.js'; +import { normalizeCppParamType } from './arity-metadata.js'; import { isCppInlineNamespaceScope } from './inline-namespaces.js'; /** @@ -97,11 +94,8 @@ export interface CppAdlArgInfo { /** When set, the arg is a potential free-function reference (not a locally- * declared function-pointer variable or function parameter). Contains the * identifier text as written in source (e.g. `"utils::worker"` or - * `"worker"`). GitNexus approximation: the function's enclosing namespace - * is contributed to the ADL associated set. For qualified refs a workspace - * lookup confirms a Function/Method with that simple name exists in the - * namespace before contributing; for unqualified refs every namespace - * containing a matching Function/Method def is contributed. */ + * `"worker"`). Resolution contributes associated namespaces from each + * referenced Function/Method def's parameter and return types. */ readonly functionRefText?: string; } @@ -207,7 +201,12 @@ export function pickCppAdlCandidates( for (const arg of args) { collectAssociatedNamespacesForAdlArg(arg, scopes, associatedNamespaces); if (arg.functionRefText !== undefined) { - collectFunctionRefNamespaces(arg.functionRefText, parsedFiles, associatedNamespaces); + collectFunctionTypeAssociatedNamespaces( + arg.functionRefText, + scopes, + parsedFiles, + associatedNamespaces, + ); } } if (associatedNamespaces.size === 0) return undefined; @@ -472,23 +471,12 @@ function findCppClassDefBySimpleName( } /** - * Contribute associated namespaces for a function-reference argument. - * - * - **Qualified refs** (`utils::worker`, `outer::inner::fn`): the namespace - * is extracted from the qualifier text (converting `::` to `.` for dot-joined - * QName matching). A workspace lookup then **verifies** that a Function or - * Method def named `worker` (the simple name after the last `::`) actually - * exists in the extracted namespace. This prevents false positives from - * namespace-qualified variables, enum values, and static data members, which - * also produce `qualified_identifier` AST nodes in tree-sitter-cpp (the - * AST node type alone does not distinguish functions from non-function names). - * - **Unqualified refs** (`worker`): the workspace is searched for any - * Function/Method def whose simple name matches. Every distinct enclosing - * namespace found is added — overloads across the same namespace produce - * a single entry; GitNexus does not select a specific overload at this stage. + * Contribute associated namespaces for a function-reference argument by walking + * the referenced overload set's parameter and return types. */ -function collectFunctionRefNamespaces( +function collectFunctionTypeAssociatedNamespaces( refText: string, + scopes: ScopeResolutionIndexes, parsedFiles: readonly ParsedFile[], out: Set, ): void { @@ -511,30 +499,130 @@ function collectFunctionRefNamespaces( for (const def of scope.ownedDefs) { if (def.type !== 'Function' && def.type !== 'Method') continue; const simple = def.qualifiedName?.split('.').pop() ?? def.qualifiedName ?? ''; - if (simple === simpleName) { - out.add(nsText); - return; // Namespace confirmed; no need to scan further files. - } + if (simple === simpleName) collectAssociatedNamespacesForFunctionDef(def, scopes, out); } } } return; } - // Unqualified: search all namespace scopes for a Function def with this - // simple name and contribute its enclosing namespace. + // Unqualified function references are approximated workspace-wide, matching + // the previous V1 lookup scope. The stricter part of this PR is what each + // overload contributes: only namespaces from parameter/return types, never + // the function's own enclosing namespace. for (const parsed of parsedFiles) { - const scopesById = new Map(); - for (const sc of parsed.scopes) scopesById.set(sc.id, sc); for (const scope of parsed.scopes) { if (scope.kind !== 'Namespace') continue; for (const def of scope.ownedDefs) { if (def.type !== 'Function' && def.type !== 'Method') continue; const simple = def.qualifiedName?.split('.').pop() ?? def.qualifiedName ?? ''; if (simple !== refText) continue; - const nsQName = computeNamespaceQName(scope, scopesById); - if (nsQName !== '') out.add(nsQName); + collectAssociatedNamespacesForFunctionDef(def, scopes, out); } } } } + +function collectAssociatedNamespacesForFunctionDef( + def: SymbolDefinition, + scopes: ScopeResolutionIndexes, + out: Set, +): void { + const parameterTypes = def.parameterTypeClasses?.map((typeClass) => typeClass.base); + for (const paramType of parameterTypes ?? def.parameterTypes ?? []) { + collectAssociatedNamespacesForFunctionTypeText(paramType, scopes, out); + } + if (def.returnType !== undefined) { + collectAssociatedNamespacesForFunctionTypeText(def.returnType, scopes, out); + } +} + +function collectAssociatedNamespacesForFunctionTypeText( + typeText: string, + scopes: ScopeResolutionIndexes, + out: Set, +): void { + for (const token of extractCppTypeNameTokens(typeText)) { + if (isIgnoredCppAdlNamespace(token.namespaceName)) continue; + addAssociatedNamespaceForClassName(token.simpleName, scopes, out); + if (token.namespaceName !== '') out.add(token.namespaceName); + } +} + +function extractCppTypeNameTokens(typeText: string): readonly { + readonly simpleName: string; + readonly namespaceName: string; +}[] { + const cleaned = normalizeCppParamType(typeText); + if (cleaned === '' || isPrimitiveCppAdlType(cleaned)) return []; + const out: { simpleName: string; namespaceName: string }[] = []; + const seen = new Set(); + const tokenSource = typeText.includes('<') ? `${cleaned} ${typeText}` : cleaned; + for (const rawToken of tokenSource.match(/[A-Za-z_]\w*(?:::[A-Za-z_]\w*)*/g) ?? []) { + if (isPrimitiveCppAdlType(rawToken)) continue; + const segments = rawToken.split('::').filter((part) => part.length > 0); + const simpleName = segments.at(-1) ?? ''; + if (simpleName === '' || isPrimitiveCppAdlType(simpleName)) continue; + const namespaceName = segments.length > 1 ? segments.slice(0, -1).join('.') : ''; + const key = `${namespaceName}\0${simpleName}`; + if (seen.has(key)) continue; + seen.add(key); + out.push({ + simpleName, + namespaceName, + }); + } + return out; +} + +const CPP_ADL_PRIMITIVE_OR_KEYWORD_TYPES = new Set([ + 'alignas', + 'alignof', + 'auto', + 'bool', + 'char', + 'char8_t', + 'char16_t', + 'char32_t', + 'class', + 'const', + 'consteval', + 'constexpr', + 'constinit', + 'decltype', + 'double', + 'enum', + 'explicit', + 'extern', + 'float', + 'inline', + 'int', + 'long', + 'mutable', + 'noexcept', + 'null', + 'register', + 'short', + 'signed', + 'static', + 'string', + 'struct', + 'template', + 'thread_local', + 'typename', + 'union', + 'unknown', + 'unsigned', + 'void', + 'volatile', + 'wchar_t', + '...', +]); + +function isPrimitiveCppAdlType(typeText: string): boolean { + return CPP_ADL_PRIMITIVE_OR_KEYWORD_TYPES.has(typeText); +} + +function isIgnoredCppAdlNamespace(namespaceName: string): boolean { + return namespaceName === 'std' || namespaceName.startsWith('std.'); +} diff --git a/gitnexus/src/core/ingestion/languages/cpp/captures.ts b/gitnexus/src/core/ingestion/languages/cpp/captures.ts index 354f6ce4c..86384a759 100644 --- a/gitnexus/src/core/ingestion/languages/cpp/captures.ts +++ b/gitnexus/src/core/ingestion/languages/cpp/captures.ts @@ -126,6 +126,14 @@ export function emitCppScopeCaptures( JSON.stringify(arity.parameterTypeClasses), ); } + const returnType = extractCppDeclarationReturnType(fnNode); + if (returnType !== undefined) { + grouped['@declaration.return-type'] = syntheticCapture( + '@declaration.return-type', + fnNode, + returnType, + ); + } if (hasExplicitSpecifier(fnNode)) { grouped['@declaration.is-explicit'] = syntheticCapture( '@declaration.is-explicit', @@ -417,6 +425,30 @@ export function emitCppScopeCaptures( return out; } +function extractCppDeclarationReturnType(fnNode: SyntaxNode): string | undefined { + const typeNode = fnNode.childForFieldName('type'); + if (typeNode === null) return undefined; + const funcDeclarator = findFunctionDeclarator(fnNode); + if (funcDeclarator !== null && isCppUnsupportedReturnTypeDeclarator(funcDeclarator)) { + return undefined; + } + const typeText = typeNode.text.trim(); + if (typeText !== 'auto') return typeText.length > 0 ? typeText : undefined; + if (funcDeclarator === null) return typeText; + for (let i = 0; i < funcDeclarator.namedChildCount; i++) { + const child = funcDeclarator.namedChild(i); + if (child?.type !== 'trailing_return_type') continue; + const typeDesc = child.firstNamedChild; + return typeDesc?.text.trim() || typeText; + } + return typeText; +} + +function isCppUnsupportedReturnTypeDeclarator(funcDeclarator: SyntaxNode): boolean { + const text = funcDeclarator.text; + return /\boperator\b/.test(text) || /(^|[(:\s])~\s*[A-Za-z_]\w*/.test(text); +} + /** * Walk every C++ class/struct base clause and emit `@reference.inherits` * captures for each base so scope resolution can resolve them into EXTENDS diff --git a/gitnexus/src/core/ingestion/languages/csharp/namespace-siblings.ts b/gitnexus/src/core/ingestion/languages/csharp/namespace-siblings.ts index b6de589e2..f4d29d7e7 100644 --- a/gitnexus/src/core/ingestion/languages/csharp/namespace-siblings.ts +++ b/gitnexus/src/core/ingestion/languages/csharp/namespace-siblings.ts @@ -28,17 +28,19 @@ * aliased `using static X = Y.Z;`, attributed namespace declarations, * and preprocessor-guarded declarations correctly because the * tree-sitter grammar parses them as real nodes (not textual - * coincidences). + * coincidences). When the orchestrator's `treeCache` has no Tree for a + * file — the worker path, where native Trees can't cross MessageChannels + * — `extractFileStructure` falls back to a line scanner rather than + * re-parsing every file from scratch (that re-parse dominated worker-mode + * scope-resolution time). See `extractCsharpStructureViaScanner`. */ import type { SyntaxNode } from 'tree-sitter'; import type { BindingRef, ParsedFile, Scope, ScopeId, SymbolDefinition } from 'gitnexus-shared'; import type { ScopeResolutionIndexes } from '../../model/scope-resolution-indexes.js'; import { getCsharpParser } from './query.js'; -import { getTreeSitterBufferSize } from '../../constants.js'; -import { parseSourceSafe } from '../../../tree-sitter/safe-parse.js'; -interface CsharpFileStructure { +export interface CsharpFileStructure { /** Declared namespace names in file source order. Empty array means * the file has no `namespace X;` / `namespace X { }` declaration * and sits in the default (global) namespace. */ @@ -48,18 +50,174 @@ interface CsharpFileStructure { readonly usingStaticPaths: readonly string[]; } -/** Build a structural view of a C# file by walking the tree-sitter - * AST. Prefers `cachedTree` (handed in via `treeCache`) so we don't - * re-parse files the orchestrator already parsed for `extractParsedFile`; - * falls back to a fresh parse on cache miss. Parser singleton is - * shared across calls. */ +// Line-anchored matchers for the worker-path fallback (see +// `extractCsharpStructureViaScanner`). Anchored at line start (after +// indentation); the scanner additionally tracks block-comment / string +// state across lines so a keyword at the start of a line inside one of +// those regions is skipped. +const CS_NAMESPACE_RE = /^[ \t]*namespace[ \t]+([A-Za-z_@][A-Za-z0-9_.]*)/; +// `global using static`, plain `using static`, and the aliased +// `using static Alias = NS.Type;` form (the AST keeps the RHS path, so +// the optional `Alias =` is skipped and only the dotted path captured). +const CS_USING_STATIC_RE = + /^[ \t]*(?:global[ \t]+)?using[ \t]+static[ \t]+(?:[A-Za-z_@][A-Za-z0-9_]*[ \t]*=[ \t]*)?([A-Za-z_@][A-Za-z0-9_.]*)/; + +/** Multi-line lexical state carried line-to-line by the scanner. */ +type CsScanState = 'code' | 'block' | 'verbatim' | 'raw'; + +/** Advance the scanner's lexical state across one line, consuming block + * comments (slash-star), line comments (`//`), single-line regular / + * interpolated strings, verbatim strings (`@"…"`), and raw string literals + * (`"""…"""`, fence length tracked in `rawFence`). Returns the state and + * raw-fence length in effect at the START of the next line. Single-line + * strings and `//` comments resolve back to `code` before end of line; only + * block comments and multi-line strings carry state forward. */ +function advanceCsScanState( + line: string, + state: CsScanState, + rawFence: number, +): [CsScanState, number] { + const n = line.length; + let i = 0; + while (i < n) { + if (state === 'block') { + const end = line.indexOf('*/', i); + if (end === -1) return ['block', rawFence]; + i = end + 2; + state = 'code'; + } else if (state === 'verbatim') { + // Ends at a `"` that is not doubled (`""` is an escaped quote). + while (i < n) { + if (line[i] === '"') { + if (line[i + 1] === '"') { + i += 2; + continue; + } + break; + } + i++; + } + if (i >= n) return ['verbatim', rawFence]; + i += 1; + state = 'code'; + } else if (state === 'raw') { + // Ends at a run of `"` at least `rawFence` long. + let closed = false; + while (i < n) { + if (line[i] === '"') { + let k = i; + while (k < n && line[k] === '"') k++; + if (k - i >= rawFence) { + i = k; + state = 'code'; + rawFence = 0; + closed = true; + break; + } + i = k; + } else { + i++; + } + } + if (!closed) return ['raw', rawFence]; + } else { + const c = line[i]; + const next = line[i + 1]; + if (c === '/' && next === '/') return ['code', rawFence]; // line comment to EOL + if (c === '/' && next === '*') { + state = 'block'; + i += 2; + } else if (c === '@' && next === '"') { + state = 'verbatim'; + i += 2; + } else if ((c === '$' && next === '@') || (c === '@' && next === '$')) { + if (line[i + 2] === '"') { + state = 'verbatim'; // interpolated verbatim ($@"…" / @$"…") + i += 3; + } else { + i++; + } + } else if (c === '"') { + let k = i; + while (k < n && line[k] === '"') k++; + const run = k - i; + if (run >= 3) { + state = 'raw'; + rawFence = run; + i = k; + } else if (run === 2) { + i = k; // "" — empty string + } else { + // single-line regular / interpolated string; consume to closer + let j = i + 1; + while (j < n) { + if (line[j] === '\\') { + j += 2; + continue; + } + if (line[j] === '"') break; + j++; + } + i = j >= n ? n : j + 1; + } + } else { + i++; + } + } + } + return [state, rawFence]; +} + +/** Line-scanner used when no cached tree is available (worker-parsed files + * can't transfer native tree-sitter Trees across MessageChannels, so + * `treeCache` is empty for them). Re-parsing every C# file here with + * tree-sitter was the dominant scope-resolution cost on large worker-mode + * runs — for a multi-thousand-file solution this loop alone re-parsed the + * whole repo a second time. The scanner extracts the same `namespaces` / + * `usingStaticPaths` the AST walk produces for line-anchored declarations, + * while tracking block-comment and string state across lines (via + * `advanceCsScanState`) so a `namespace` / `using static` keyword at the + * start of a line inside a block comment, verbatim string, or raw string + * literal is NOT mistaken for a declaration. The remaining trade-off vs the + * AST is a declaration whose keyword is not at the start of a code line + * (split across lines, or sharing a line with a comment/string closer). + * Mirrors PHP's `extractNamespaceViaScanner` (issue #1741). */ +export function extractCsharpStructureViaScanner(content: string): CsharpFileStructure { + const namespaces: string[] = []; + const usingStaticPaths: string[] = []; + let state: CsScanState = 'code'; + let rawFence = 0; + for (const line of content.split('\n')) { + // Only match when the line START is real code — keywords reached while + // inside a block comment / multi-line string are skipped. + if (state === 'code') { + const ns = CS_NAMESPACE_RE.exec(line); + if (ns !== null) { + namespaces.push(ns[1]!); + } else { + const us = CS_USING_STATIC_RE.exec(line); + if (us !== null) usingStaticPaths.push(us[1]!); + } + } + [state, rawFence] = advanceCsScanState(line, state, rawFence); + } + return { namespaces, usingStaticPaths }; +} + +/** Build a structural view of a C# file. Prefers `cachedTree` (handed in + * via `treeCache`) and walks the tree-sitter AST — the authoritative + * path that sees `global using static`, aliased `using static X = Y.Z;`, + * attributed namespace declarations, and preprocessor-guarded nodes + * correctly. On cache miss (worker-parsed files, whose native Trees + * can't cross MessageChannels) it falls back to the line scanner instead + * of a fresh tree-sitter parse — the parse here dominated worker-mode + * scope-resolution time. Parser singleton is shared across calls. */ function extractFileStructure(content: string, cachedTree: unknown): CsharpFileStructure { + if (!cachedTree) { + return extractCsharpStructureViaScanner(content); + } type CsharpTree = ReturnType['parse']>; - const tree = - (cachedTree as CsharpTree | undefined) ?? - parseSourceSafe(getCsharpParser(), content, undefined, { - bufferSize: getTreeSitterBufferSize(content), - }); + const tree = cachedTree as CsharpTree; const namespaces: string[] = []; const usingStaticPaths: string[] = []; @@ -277,11 +435,17 @@ export function populateCsharpNamespaceSiblings( // scope, so `Record(...)` (without `Logger.` qualifier) resolves // to `Logger.Record`. AST walk above captured these (including // `global using static` and aliased forms). + // Pre-index files by path once: the member-injection lookup below would + // otherwise be an O(files) scan per `using static` import. + const fileByPath = new Map(parsedFiles.map((p) => [p.filePath, p])); for (const parsed of parsedFiles) { const struct = structureByFile.get(parsed.filePath); if (struct === undefined) continue; const moduleScope = parsed.scopes.find((s) => s.kind === 'Module'); if (moduleScope === undefined) continue; + // Per-file de-dup sets keyed by simple name, seeded lazily from the + // augmentation bucket — replaces the per-member O(A) `.some` scan below. + const seenByName = new Map>(); for (const fullPath of struct.usingStaticPaths) { const lastDot = fullPath.lastIndexOf('.'); @@ -302,7 +466,7 @@ export function populateCsharpNamespaceSiblings( // Inject the class's member methods into the importer's module // scope. `memberByOwner` wasn't built yet here, so we walk the // file's localDefs to find members with `ownerId === targetDef.nodeId`. - const targetFile = parsedFiles.find((p) => p.filePath === targetDef.filePath); + const targetFile = fileByPath.get(targetDef.filePath); if (targetFile === undefined) continue; for (const memberDef of targetFile.localDefs) { if ((memberDef as { ownerId?: string }).ownerId !== targetDef.nodeId) continue; @@ -316,7 +480,14 @@ export function populateCsharpNamespaceSiblings( // `lookupBindingsAt`, which fans out across `bindings` + // `bindingAugmentations`. const bucketArr = getAugmentationBucket(augmentations, moduleScope.id, simpleName); - if (bucketArr.some((b) => b.def.nodeId === memberDef.nodeId)) continue; + let seen = seenByName.get(simpleName); + if (seen === undefined) { + seen = new Set(); + for (const b of bucketArr) seen.add(b.def.nodeId); + seenByName.set(simpleName, seen); + } + if (seen.has(memberDef.nodeId)) continue; + seen.add(memberDef.nodeId); bucketArr.push({ def: memberDef, origin: 'import' }); } } @@ -332,6 +503,9 @@ export function populateCsharpNamespaceSiblings( for (const parsed of parsedFiles) { const moduleScope = parsed.scopes.find((s) => s.kind === 'Module'); if (moduleScope === undefined) continue; + // Per-file de-dup sets keyed by simple name, seeded lazily from the + // augmentation bucket — replaces the per-def O(A) `.some` scan below. + const seenByName = new Map>(); for (const imp of parsed.parsedImports) { if (imp.kind !== 'namespace') continue; const targetNs = imp.targetRaw; @@ -344,41 +518,113 @@ export function populateCsharpNamespaceSiblings( const simpleName = q.includes('.') ? q.slice(q.lastIndexOf('.') + 1) : q; if (simpleName === '') continue; const bucketArr = getAugmentationBucket(augmentations, moduleScope.id, simpleName); - if (bucketArr.some((b) => b.def.nodeId === def.nodeId)) continue; + let seen = seenByName.get(simpleName); + if (seen === undefined) { + seen = new Set(); + for (const b of bucketArr) seen.add(b.def.nodeId); + seenByName.set(simpleName, seen); + } + if (seen.has(def.nodeId)) continue; + seen.add(def.nodeId); bucketArr.push({ def, origin: 'namespace' }); } } } - for (const [, bucket] of buckets) { - // De-dup by (nodeId, filePath) across multiple declarations (e.g. - // partial classes declaring the same name in two files — we take - // both and leave de-dup to downstream consumers of bindings). + // Workspace-level binding channel for global-namespace types (see the + // global fast-path below). `lookupBindingsAt` consults this as a third + // source after finalized + per-scope augmented bindings. Its inner arrays + // are mutable by contract (append-only, like `bindingAugmentations` — see + // the ScopeResolutionIndexes doc + validateBindingsImmutability), so the + // ReadonlyMap→Map cast is localized to this one line and all writes go + // through `getWorkspaceBucket`. + const workspace = indexes.workspaceFqnBindings as Map; + + for (const [nsName, bucket] of buckets) { + // Group sibling defs by simple name. Append in place — the previous + // `[...prev, def]` copy made this O(D²) per bucket, which on the + // global (`''`) namespace bucket of a large Unity solution (tens of + // thousands of type defs) was a primary slowness/OOM source. We keep + // every declaration (e.g. partial classes across files) and leave + // de-dup to downstream consumers. const defsByName = new Map(); for (const def of bucket.classDefs) { // Simple name = last segment of qualifiedName (e.g. `App.User` → `User`). const q = def.qualifiedName ?? ''; const key = q.includes('.') ? q.slice(q.lastIndexOf('.') + 1) : q; if (key === '') continue; - const arr = [...(defsByName.get(key) ?? [])]; + let arr = defsByName.get(key); + if (arr === undefined) { + arr = []; + defsByName.set(key, arr); + } arr.push(def); - defsByName.set(key, arr); + } + + // Global-namespace fast path (Unity OOM guard). Types declared in the + // default (global) namespace are visible from EVERY file in C# — the + // global namespace is always implicitly in scope — so one workspace- + // level entry per simple name is both semantically correct and O(D) + // instead of the O(S·D) per-scope augmentation that materialized + // billions of BindingRefs on large Unity solutions (tens of thousands + // of global types × tens of thousands of scopes). `walkScopeChain` + // checks local `scope.bindings` first, so local declarations still + // shadow these workspace entries; a file resolving its own global type + // hits the local binding before this map. Dedup by `def.nodeId` keeps + // partial-class / duplicate declarations from double-emitting. + if (nsName === '') { + for (const [name, defs] of defsByName) { + const bucket = getWorkspaceBucket(workspace, name); + const seen = new Set(); + for (const b of bucket) seen.add(b.def.nodeId); + for (const def of defs) { + if (seen.has(def.nodeId)) continue; // dedup by nodeId (keeps partials, drops re-emits) + seen.add(def.nodeId); + bucket.push({ def, origin: 'namespace' }); + } + } + continue; + } + + // Pre-index the first scope per file once (O(S)) instead of an + // O(S) `.find` re-run for every (scope, name) pair, which made the + // injection loop O(S²·D) and was the dominant cost on large buckets. + // Multiple scopes share a filePath (Module + Namespace); the local + // shadow check only needs that file's lexical `Scope.bindings`, which + // is identical regardless of which of those scopes we read. + const firstScopeByFile = new Map(); + for (const s of bucket.scopes) { + if (!firstScopeByFile.has(s.filePath)) firstScopeByFile.set(s.filePath, s.scope); } for (const { scopeId, filePath } of bucket.scopes) { + const localScope = firstScopeByFile.get(filePath); for (const [name, defs] of defsByName) { // Skip names already present locally — `origin: 'local'` in // scope.bindings would naturally shadow the cross-file // namespace entry, but we also keep this index lean. - const local = bucket.scopes.find((s) => s.filePath === filePath)?.scope.bindings.get(name); + const local = localScope?.bindings.get(name); if (local !== undefined && local.some((b) => b.origin === 'local')) continue; - let bucketArr: BindingRef[] | null = null; + // Bind the augmentation bucket and its seeded de-dup set together + // under one nullable lifecycle, so neither needs a non-null + // assertion (they are always set or unset as a pair). Stays lazy: + // nothing is allocated for a name with no cross-file defs. + let inject: { bucket: BindingRef[]; seen: Set } | null = null; for (const def of defs) { if (def.filePath === filePath) continue; // don't self-reference - if (bucketArr === null) bucketArr = getAugmentationBucket(augmentations, scopeId, name); - if (bucketArr.some((b) => b.def.nodeId === def.nodeId)) continue; - bucketArr.push({ def, origin: 'namespace' }); + if (inject === null) { + const bucket = getAugmentationBucket(augmentations, scopeId, name); + // Seed the de-dup set from any entries an earlier pass + // (using-static / cross-namespace imports) already added, + // replacing the per-def O(A) `.some` scan. + const seen = new Set(); + for (const b of bucket) seen.add(b.def.nodeId); + inject = { bucket, seen }; + } + if (inject.seen.has(def.nodeId)) continue; + inject.seen.add(def.nodeId); + inject.bucket.push({ def, origin: 'namespace' }); } } } @@ -409,6 +655,22 @@ function getAugmentationBucket( return bucketArr; } +/** Get-or-create a mutable inner bucket inside the `workspaceFqnBindings` + * channel (the scope-independent third channel; see + * `ScopeResolutionIndexes.workspaceFqnBindings`). Like + * `getAugmentationBucket`, the inner arrays are mutable by contract — + * callers `push` directly. Keeping the get-or-create here means the one + * ReadonlyMap→Map cast at the call site is the only place the mutable + * view is taken. */ +function getWorkspaceBucket(workspace: Map, name: string): BindingRef[] { + let bucketArr = workspace.get(name); + if (bucketArr === undefined) { + bucketArr = []; + workspace.set(name, bucketArr); + } + return bucketArr; +} + function isTypeDef(def: SymbolDefinition): boolean { return ( def.type === 'Class' || diff --git a/gitnexus/src/core/ingestion/languages/go.ts b/gitnexus/src/core/ingestion/languages/go.ts index 5c9aef039..b1f8f8ce5 100644 --- a/gitnexus/src/core/ingestion/languages/go.ts +++ b/gitnexus/src/core/ingestion/languages/go.ts @@ -39,6 +39,56 @@ import { interpretGoTypeBinding, } from './go/index.js'; +const GO_BUILT_INS: ReadonlySet = new Set([ + // built-in functions + 'make', + 'new', + 'len', + 'cap', + 'append', + 'copy', + 'delete', + 'close', + 'panic', + 'recover', + 'print', + 'println', + 'complex', + 'real', + 'imag', + 'clear', + 'min', + 'max', + // built-in types + 'error', + 'bool', + 'string', + 'int', + 'int8', + 'int16', + 'int32', + 'int64', + 'uint', + 'uint8', + 'uint16', + 'uint32', + 'uint64', + 'uintptr', + 'float32', + 'float64', + 'complex64', + 'complex128', + 'byte', + 'rune', + 'any', + 'comparable', + // built-in values + 'true', + 'false', + 'nil', + 'iota', +]); + export const goProvider = defineLanguage({ id: SupportedLanguages.Go, extensions: ['.go'], @@ -92,6 +142,7 @@ export const goProvider = defineLanguage({ variableExtractor: createVariableExtractor(goVariableConfig), classExtractor: createClassExtractor(goClassConfig), heritageExtractor: createHeritageExtractor(goHeritageConfig), + builtInNames: GO_BUILT_INS, // ── RFC #909 Ring 3: scope-based resolution hooks ────────── emitScopeCaptures: emitGoScopeCaptures, diff --git a/gitnexus/src/core/ingestion/languages/javascript/captures.ts b/gitnexus/src/core/ingestion/languages/javascript/captures.ts index a04844059..dc8175bdf 100644 --- a/gitnexus/src/core/ingestion/languages/javascript/captures.ts +++ b/gitnexus/src/core/ingestion/languages/javascript/captures.ts @@ -38,6 +38,7 @@ import { splitImportStatement } from '../typescript/import-decomposer.js'; import { getJsParser, getJsScopeQuery, jsCachedTreeMatchesGrammar } from './query.js'; import { computeTsArityMetadata } from '../typescript/arity-metadata.js'; import { synthesizeTsReceiverBinding } from '../typescript/receiver-binding.js'; +import { isArrayMethodCallbackArrow } from '../typescript/array-callback.js'; import { getTreeSitterBufferSize } from '../../constants.js'; import { parseSourceSafe } from '../../../tree-sitter/safe-parse.js'; @@ -640,6 +641,21 @@ export function emitJsScopeCaptures( } } + // #1876: drop @declaration.function for array higher-order-method + // callbacks (`const x = arr.map(a => …)`). The HOC-wrapped-arrow + // pattern matches them, but the binding holds a value, not a callable. + // The binding keeps its separate @declaration.const / .variable match, + // and the arrow's own @scope.function match (a different pattern) is + // untouched, so inner-call attribution falls through to the enclosing + // scope instead of a phantom Function. + const fnDeclAnchor = grouped['@declaration.function']; + if (fnDeclAnchor !== undefined) { + const arrowNode = findFunctionNode(tree.rootNode, fnDeclAnchor.range); + if (arrowNode !== null && isArrayMethodCallbackArrow(arrowNode)) { + continue; + } + } + // Synthesize arity metadata on function-like declarations. const declAnchor = pickFirstDefined(grouped, FUNCTION_DECL_TAGS); if (declAnchor !== undefined) { diff --git a/gitnexus/src/core/ingestion/languages/javascript/query.ts b/gitnexus/src/core/ingestion/languages/javascript/query.ts index 20cfc4f2e..42b043fb0 100644 --- a/gitnexus/src/core/ingestion/languages/javascript/query.ts +++ b/gitnexus/src/core/ingestion/languages/javascript/query.ts @@ -148,6 +148,12 @@ const JAVASCRIPT_SCOPE_QUERY = ` ;; HOC-wrapped variable declarations: const X = HOC((args) => { ... }). ;; Covers React.forwardRef, memo, useCallback, useMemo, observer, ;; debounce, and any user-defined HOC factory. +;; +;; #1876: this shape also matches array higher-order-method callbacks +;; (const x = arr.map(a => ...)), where x is a value, not a function. +;; Those are filtered out emit-side in captures.ts via +;; isArrayMethodCallbackArrow (member-expression callee whose property +;; is a known Array method), so only the @declaration.const survives. (lexical_declaration (variable_declarator name: (identifier) @declaration.name diff --git a/gitnexus/src/core/ingestion/languages/typescript/array-callback.ts b/gitnexus/src/core/ingestion/languages/typescript/array-callback.ts new file mode 100644 index 000000000..14daa9a4c --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/typescript/array-callback.ts @@ -0,0 +1,98 @@ +/** + * Array higher-order-method callback detection (issue #1876). + * + * The HOC-wrapped-arrow declaration pattern in the JS/TS scope queries + * (`const X = call((args) => …)`) was added for React idioms + * (`forwardRef` / `memo` / `useCallback`). It has the same AST shape as + * an array higher-order-method call (`const x = arr.map(a => …)`), so + * those callbacks also match and produce a spurious `@declaration.function` + * named after the binding — duplicating the `@declaration.const` / + * `@declaration.variable` def that the same binding already gets. + * + * For an array-method callback the binding holds a *value* (the method's + * result), not a callable, so the `Function` def is semantically wrong. + * `isArrayMethodCallbackArrow` lets the emitter (`captures.ts`) drop that + * `@declaration.function` match, leaving only the value def. + * + * Shared by both the JavaScript and TypeScript capture emitters — the + * relevant grammar nodes (`arrow_function`, `function_expression`, + * `arguments`, `call_expression`, `member_expression`, + * `property_identifier`) are identical across `tree-sitter-javascript` + * and `tree-sitter-typescript`. + * + * Pure given the input node. No I/O, no globals. + */ + +import type { SyntaxNode } from '../../utils/ast-helpers.js'; + +/** + * Array prototype higher-order methods whose result is a value, not a + * function. A callback passed to one of these is an anonymous callback, + * never a top-level function definition. Identifier-callee HOCs + * (`forwardRef(...)`, `useCallback(...)`, custom factories) are + * deliberately NOT listed — they keep their `Function` classification. + * + * Trade-off (unchanged from before #1876): a custom *fluent-API* member + * call with a callback whose method name is not in this set + * (`qb.where(x => …)`) still classifies as `Function`. There is no clean + * syntactic line beyond the well-known Array surface, so the set is + * intentionally closed and easy to extend. + * + * Receiver-blind, by design: the match keys on the method NAME only, never + * the receiver type (tree-sitter has no type information here). So an in-set + * name on a NON-array receiver — `Map`/`Set` `.forEach`, an RxJS + * `observable.map(…)`, a query builder `.sort(…)`, a lodash chain + * `.filter(…)` — is ALSO treated as a callback and has its + * `@declaration.function` dropped. This is an accepted limitation, not a + * regression: those bindings hold the call's *result value*, not a callable, + * so a value def is the correct classification anyway. The only genuine loss + * is a bespoke DSL whose in-set-named method returns something callable — + * rare enough to accept rather than guard with type inference. Pinned by the + * "in-set method on a non-array receiver" case in `*-captures.test.ts`. + */ +export const ARRAY_CALLBACK_METHODS: ReadonlySet = new Set([ + 'map', + 'filter', + 'find', + 'findIndex', + 'findLast', + 'findLastIndex', + 'forEach', + 'reduce', + 'reduceRight', + 'some', + 'every', + 'flatMap', + 'sort', +]); + +/** + * True when `node` (an `arrow_function` / `function_expression`) is the + * callback argument of an array higher-order-method call, i.e. the + * enclosing call's callee is a `member_expression` whose property is one + * of {@link ARRAY_CALLBACK_METHODS}. + * + * Returns false for direct assignments (`const fn = () => {}` — parent is + * `variable_declarator`, not `arguments`) and for identifier-callee HOCs + * (`forwardRef(() => …)` — callee is an `identifier`, not a + * `member_expression`), so neither is ever suppressed. + * + * Intentional non-suppressing gaps (preserve current behavior, no + * regression): parenthesized callee `(arr.map)(cb)` (`parenthesized_expression`) + * and computed callee `arr['map'](cb)` (`subscript_expression`). + */ +export function isArrayMethodCallbackArrow(node: SyntaxNode): boolean { + const args = node.parent; + if (args === null || args.type !== 'arguments') return false; + + const call = args.parent; + if (call === null || call.type !== 'call_expression') return false; + + const callee = call.childForFieldName('function'); + if (callee === null || callee.type !== 'member_expression') return false; + + const property = callee.childForFieldName('property'); + if (property === null || property.type !== 'property_identifier') return false; + + return ARRAY_CALLBACK_METHODS.has(property.text); +} diff --git a/gitnexus/src/core/ingestion/languages/typescript/captures.ts b/gitnexus/src/core/ingestion/languages/typescript/captures.ts index 22f48e80d..bc9d1a4f7 100644 --- a/gitnexus/src/core/ingestion/languages/typescript/captures.ts +++ b/gitnexus/src/core/ingestion/languages/typescript/captures.ts @@ -37,6 +37,7 @@ import { getTsParser, getTsScopeQuery, tsCachedTreeMatchesGrammar } from './quer import { recordCacheHit, recordCacheMiss } from './cache-stats.js'; import { synthesizeTsReceiverBinding } from './receiver-binding.js'; import { computeTsArityMetadata } from './arity-metadata.js'; +import { isArrayMethodCallbackArrow } from './array-callback.js'; import { getTreeSitterBufferSize } from '../../constants.js'; import { parseSourceSafe } from '../../../tree-sitter/safe-parse.js'; @@ -252,6 +253,25 @@ export function emitTsScopeCaptures( } } + // #1876: drop @declaration.function for array higher-order-method + // callbacks (`const x = arr.map(a => …)`). The HOC-wrapped-arrow + // pattern matches them, but the binding holds a value, not a callable. + // The binding keeps its separate @declaration.const / .variable match, + // and the arrow's own @scope.function match (a different pattern) is + // untouched, so inner-call attribution falls through to the enclosing + // scope instead of a phantom Function. + const fnDeclAnchor = grouped['@declaration.function']; + if (fnDeclAnchor !== undefined) { + const arrowNode = findFunctionNode( + tree.rootNode, + fnDeclAnchor.range, + groupedNodes['@declaration.function'], + ); + if (arrowNode !== null && isArrayMethodCallbackArrow(arrowNode)) { + continue; + } + } + // Synthesize arity metadata on function-like declaration anchors // before pushing the match. The registry uses these to narrow // overloads — TypeScript supports overload signatures via diff --git a/gitnexus/src/core/ingestion/languages/typescript/query.ts b/gitnexus/src/core/ingestion/languages/typescript/query.ts index 9e0d9b809..9d2ebe5db 100644 --- a/gitnexus/src/core/ingestion/languages/typescript/query.ts +++ b/gitnexus/src/core/ingestion/languages/typescript/query.ts @@ -250,20 +250,22 @@ const TYPESCRIPT_SCOPE_QUERY = ` ;; that promotes the binding to the parent scope (where \`const X\` ;; lives). ;; -;; Trade-off — chained array-method form: \`const x = arr.find((y) => p(y))\` -;; has the same syntactic shape and would also match, naming the -;; \`.find\` callback as \`x\`. The resulting \`Function:x\` is mostly -;; harmless: \`x\` is consumed as a value (\`if (x) { ... }\`), never -;; invoked as a function, so it gets zero incoming \`CALLS\` edges. The -;; one outgoing edge \`Function:x → p\` is a minor mis-attribution that -;; could in principle be fixed by adding a \`function: [(identifier) -;; (member_expression)]\` predicate that excludes property-identifiers -;; matching a known array-method blocklist (\`map\` / \`filter\` / \`find\` -;; / \`reduce\` / \`forEach\` / \`some\` / \`every\`). We don't do that here -;; because (a) the false-positive cost is negligible, (b) the blocklist -;; would need maintenance, and (c) any user-defined fluent-API method -;; with a callback argument would still false-positive — there's no -;; clean syntactic line. +;; #1876 — chained array-method form: \`const x = arr.find((y) => p(y))\` +;; has the same syntactic shape and matches here too, naming the +;; \`.find\` callback as \`x\`. Because \`x\` holds a value (the method +;; result), not a callable, the spurious \`Function:x\` def is dropped +;; emit-side in captures.ts: \`isArrayMethodCallbackArrow\` skips any +;; \`@declaration.function\` whose enclosing call has a member-expression +;; callee with a known Array-method property (\`ARRAY_CALLBACK_METHODS\`: +;; \`map\` / \`filter\` / \`find\` / \`reduce\` / \`forEach\` / \`some\` / +;; \`every\` / …). Only the \`@declaration.variable\` survives, so the +;; binding is a single value def and calls inside the callback attribute +;; to the enclosing scope rather than \`Function:x\`. +;; +;; Residual (intentional): a user-defined fluent-API method with a +;; callback (\`qb.where(x => …)\`) is NOT in the blocklist and still +;; classifies as \`Function\` — there's no clean syntactic line beyond +;; the well-known Array surface, so the set is closed and easy to extend. ;; ;; Trade-off — multi-arrow arguments: \`const x = call(arrow1, arrow2)\` ;; would emit TWO matches with the same name \`x\`. tree-sitter-query diff --git a/gitnexus/src/core/ingestion/model/scope-resolution-indexes.ts b/gitnexus/src/core/ingestion/model/scope-resolution-indexes.ts index 561595ac5..71e78f567 100644 --- a/gitnexus/src/core/ingestion/model/scope-resolution-indexes.ts +++ b/gitnexus/src/core/ingestion/model/scope-resolution-indexes.ts @@ -77,11 +77,15 @@ export interface ScopeResolutionIndexes { * are returned first and win duplicate `def.nodeId` metadata, with * unique augmentations appended after. See I8. */ readonly bindingAugmentations: ReadonlyMap>; - /** Workspace-level FQN binding lookup. Populated by PHP namespace- - * siblings Step 3b as a shared map instead of per-scope duplication. - * Consulted by `lookupBindingsAt` as a third source after finalized - * and per-scope augmented bindings. Keys are backslash-separated FQNs - * (e.g. `App\Models\User`). */ + /** Workspace-level binding lookup, shared instead of per-scope + * duplication. Consulted by `lookupBindingsAt` as a third source after + * finalized and per-scope augmented bindings. Language-specific + * namespace-sibling hooks populate it with disjoint key formats that + * never collide — e.g. backslash-separated FQNs (`App\Models\User`) for + * backslash-namespace languages, and bare simple names (`User`) for + * global-/default-namespace types that are visible from every file. The + * shared map gives those workspace-wide names one entry each instead of + * O(scopes × defs) per-scope augmentation. */ readonly workspaceFqnBindings: ReadonlyMap; /** Pre-resolution usage facts; consumed by the resolution phase. */ readonly referenceSites: readonly ReferenceSite[]; diff --git a/gitnexus/src/core/ingestion/parsing-processor.ts b/gitnexus/src/core/ingestion/parsing-processor.ts index 39d16461a..6b69f75c3 100644 --- a/gitnexus/src/core/ingestion/parsing-processor.ts +++ b/gitnexus/src/core/ingestion/parsing-processor.ts @@ -55,6 +55,11 @@ import type { ExtractedORMQuery, FetchWrapperDef, } from './workers/parse-worker.js'; +import type { + ExtractedRouterImport, + ExtractedRouterInclude, + ExtractedRouterModuleAlias, +} from './route-extractors/fastapi-router-bindings.js'; import { getTreeSitterBufferSize, getTreeSitterContentByteLength, @@ -72,6 +77,9 @@ export interface WorkerExtractedData { fetchCalls: ExtractedFetchCall[]; fetchWrapperDefs: FetchWrapperDef[]; decoratorRoutes: ExtractedDecoratorRoute[]; + routerIncludes: ExtractedRouterInclude[]; + routerImports: ExtractedRouterImport[]; + routerModuleAliases: ExtractedRouterModuleAlias[]; toolDefs: ExtractedToolDef[]; ormQueries: ExtractedORMQuery[]; constructorBindings: FileConstructorBindings[]; @@ -114,6 +122,9 @@ export const mergeChunkResults = ( const allFetchCalls: ExtractedFetchCall[] = []; const allFetchWrapperDefs: FetchWrapperDef[] = []; const allDecoratorRoutes: ExtractedDecoratorRoute[] = []; + const allRouterIncludes: ExtractedRouterInclude[] = []; + const allRouterImports: ExtractedRouterImport[] = []; + const allRouterModuleAliases: ExtractedRouterModuleAlias[] = []; const allToolDefs: ExtractedToolDef[] = []; const allORMQueries: ExtractedORMQuery[] = []; const allConstructorBindings: FileConstructorBindings[] = []; @@ -152,6 +163,9 @@ export const mergeChunkResults = ( for (const item of result.fetchCalls) allFetchCalls.push(item); for (const item of result.fetchWrapperDefs ?? []) allFetchWrapperDefs.push(item); for (const item of result.decoratorRoutes) allDecoratorRoutes.push(item); + for (const item of result.routerIncludes ?? []) allRouterIncludes.push(item); + for (const item of result.routerImports ?? []) allRouterImports.push(item); + for (const item of result.routerModuleAliases ?? []) allRouterModuleAliases.push(item); for (const item of result.toolDefs) allToolDefs.push(item); if (result.ormQueries) for (const item of result.ormQueries) allORMQueries.push(item); for (const item of result.constructorBindings) allConstructorBindings.push(item); @@ -169,6 +183,9 @@ export const mergeChunkResults = ( fetchCalls: allFetchCalls, fetchWrapperDefs: allFetchWrapperDefs, decoratorRoutes: allDecoratorRoutes, + routerIncludes: allRouterIncludes, + routerImports: allRouterImports, + routerModuleAliases: allRouterModuleAliases, toolDefs: allToolDefs, ormQueries: allORMQueries, constructorBindings: allConstructorBindings, @@ -210,6 +227,9 @@ const processParsingWithWorkers = async ( fetchCalls: [], fetchWrapperDefs: [], decoratorRoutes: [], + routerIncludes: [], + routerImports: [], + routerModuleAliases: [], toolDefs: [], ormQueries: [], constructorBindings: [], diff --git a/gitnexus/src/core/ingestion/pipeline-phases/parse-impl.ts b/gitnexus/src/core/ingestion/pipeline-phases/parse-impl.ts index a37d040e6..dc8a8ab00 100644 --- a/gitnexus/src/core/ingestion/pipeline-phases/parse-impl.ts +++ b/gitnexus/src/core/ingestion/pipeline-phases/parse-impl.ts @@ -63,6 +63,11 @@ import type { FileConstructorBindings, FetchWrapperDef, } from '../workers/parse-worker.js'; +import type { + ExtractedRouterImport, + ExtractedRouterInclude, + ExtractedRouterModuleAlias, +} from '../route-extractors/fastapi-router-bindings.js'; import type { ExtractedHeritage } from '../model/heritage-map.js'; import type { KnowledgeGraph } from '../../graph/types.js'; import type { PipelineOptions } from '../pipeline.js'; @@ -357,6 +362,9 @@ export async function runChunkedParseAndResolve( const allFetchWrapperDefs: FetchWrapperDef[] = []; const allExtractedRoutes: ExtractedRoute[] = []; const allDecoratorRoutes: ExtractedDecoratorRoute[] = []; + const allRouterIncludes: ExtractedRouterInclude[] = []; + const allRouterImports: ExtractedRouterImport[] = []; + const allRouterModuleAliases: ExtractedRouterModuleAlias[] = []; const allToolDefs: ExtractedToolDef[] = []; const allORMQueries: ExtractedORMQuery[] = []; const deferredWorkerCalls: ExtractedCall[] = []; @@ -675,6 +683,15 @@ export async function runChunkedParseAndResolve( if (chunkWorkerData.decoratorRoutes?.length) { for (const item of chunkWorkerData.decoratorRoutes) allDecoratorRoutes.push(item); } + if (chunkWorkerData.routerIncludes?.length) { + for (const item of chunkWorkerData.routerIncludes) allRouterIncludes.push(item); + } + if (chunkWorkerData.routerImports?.length) { + for (const item of chunkWorkerData.routerImports) allRouterImports.push(item); + } + if (chunkWorkerData.routerModuleAliases?.length) { + for (const item of chunkWorkerData.routerModuleAliases) allRouterModuleAliases.push(item); + } if (chunkWorkerData.toolDefs?.length) { for (const item of chunkWorkerData.toolDefs) allToolDefs.push(item); } @@ -1085,6 +1102,157 @@ export async function runChunkedParseAndResolve( importCtx.index = EMPTY_INDEX; importCtx.normalizedFileList = []; + // FastAPI router-prefix resolution (cross-file). + // + // Workers emit two kinds of records per Python file: + // • `routerIncludes` — every `app.include_router(, prefix='/x')` + // site, where `routerExpr` is either `.router` (Shape A) or a + // bare local name (Shape B). + // • `routerImports` — every `from import router [as ]`, + // mapping a local name to a module key (the basename of the source + // module). These let us resolve Shape-B router includes back to the + // module that defines the router. + // + // We build `module-basename → Set` and then walk + // `allDecoratorRoutes`: any decorator route emitted from a `router.` + // decorator inherits its file-basename's prefix. When a router is mounted + // under multiple prefixes we duplicate the route entry, mirroring FastAPI's + // runtime behaviour. + if (allRouterIncludes.length > 0 && allDecoratorRoutes.length > 0) { + // Group `routerImports` by file so we can resolve Shape-B locals against + // imports declared in the SAME file as the include_router call. We carry + // both the short module key (file basename) and, when available, the long + // key (`/`) so cross-package same-name modules don't blur + // their prefixes together. `routerModuleAliases` lifts the same long-key + // information for Shape-A includes whose receiving module was imported + // via `from import `. + interface LocalImport { + moduleKey: string; + moduleKeyLong: string | undefined; + } + const importsByFile = new Map>(); + for (const imp of allRouterImports) { + let m = importsByFile.get(imp.filePath); + if (!m) { + m = new Map(); + importsByFile.set(imp.filePath, m); + } + m.set(imp.localName, { + moduleKey: imp.moduleKey, + moduleKeyLong: imp.moduleKeyLong, + }); + } + // Module-alias map keyed by file: `localName` (the imported module + // identifier in this file) → long key. Shape-A receivers like + // `users.router` are matched against this map; the long key, when + // present, scopes the prefix to the precise source file. + const moduleAliasesByFile = new Map>(); + for (const alias of allRouterModuleAliases) { + let m = moduleAliasesByFile.get(alias.filePath); + if (!m) { + m = new Map(); + moduleAliasesByFile.set(alias.filePath, m); + } + m.set(alias.localName, alias.moduleKeyLong); + } + + // Two parallel maps: long-key (precise) and short-key (basename + // fallback). Long-key entries are preferred when the file's own long + // key matches; short-key entries match any file with that basename and + // remain the fallback when no long key is known (e.g. Shape A includes + // without a corresponding import statement). + const prefixesByLongKey = new Map>(); + const prefixesByShortKey = new Map>(); + + const recordPrefix = (target: Map>, key: string, prefix: string): void => { + let set = target.get(key); + if (!set) { + set = new Set(); + target.set(key, set); + } + set.add(prefix); + }; + + for (const inc of allRouterIncludes) { + // Shape A: `.router`. The worker emits `routerExpr` already + // including `.router`, so split it back. We only know a short module + // key here — the call site doesn't carry the dotted package path. If + // the same file imports `` via `from import ` + // (recorded in `allRouterModuleAliases`) we promote to a long key. + const dotIdx = inc.routerExpr.indexOf('.router'); + if (dotIdx > 0) { + const moduleShort = inc.routerExpr.slice(0, dotIdx); + const aliasLong = moduleAliasesByFile.get(inc.filePath)?.get(moduleShort); + if (aliasLong) { + recordPrefix(prefixesByLongKey, aliasLong, inc.prefix); + } else { + recordPrefix(prefixesByShortKey, moduleShort, inc.prefix); + } + continue; + } + + // Shape B: bare local name. Resolve through this file's imports. The + // import line gives us a long key whenever the module path was multi- + // segment, so cross-package collisions are eliminated for Shape B. + const localImp = importsByFile.get(inc.filePath)?.get(inc.routerExpr); + if (!localImp) continue; + if (localImp.moduleKeyLong) { + recordPrefix(prefixesByLongKey, localImp.moduleKeyLong, inc.prefix); + } else { + recordPrefix(prefixesByShortKey, localImp.moduleKey, inc.prefix); + } + } + + if (prefixesByLongKey.size > 0 || prefixesByShortKey.size > 0) { + const fileLongKey = (rel: string): string => { + // Strip `.py`, then take the last two path segments. `api/users.py` + // → `api/users`. Files at the repo root return the empty string, + // which can never match a long-key entry (those always include a + // parent directory) and so fall through to the short-key lookup. + const noExt = rel.endsWith('.py') ? rel.slice(0, -3) : rel; + const lastSlash = noExt.lastIndexOf('/'); + if (lastSlash < 0) return ''; + const beforeLast = noExt.slice(0, lastSlash); + const stem = noExt.slice(lastSlash + 1); + const prevSlash = beforeLast.lastIndexOf('/'); + const parent = prevSlash >= 0 ? beforeLast.slice(prevSlash + 1) : beforeLast; + return `${parent}/${stem}`; + }; + + const fileShortKey = (rel: string): string => { + const slash = rel.lastIndexOf('/'); + const file = slash >= 0 ? rel.slice(slash + 1) : rel; + return file.endsWith('.py') ? file.slice(0, -3) : file; + }; + + const expanded: ExtractedDecoratorRoute[] = []; + for (const dr of allDecoratorRoutes) { + if (dr.decoratorReceiver !== 'router' || !dr.filePath.endsWith('.py')) { + expanded.push(dr); + continue; + } + // Long-key lookup first; only fall back to the short key when no + // long-key prefix targets this file. This avoids prefix leakage + // between e.g. `api/users.py` and `admin/users.py`. + const longKey = fileLongKey(dr.filePath); + const longPrefixes = longKey ? prefixesByLongKey.get(longKey) : undefined; + const shortPrefixes = longPrefixes + ? undefined + : prefixesByShortKey.get(fileShortKey(dr.filePath)); + const prefixes = longPrefixes ?? shortPrefixes; + if (!prefixes || prefixes.size === 0) { + expanded.push(dr); + continue; + } + for (const prefix of prefixes) { + expanded.push({ ...dr, prefix }); + } + } + allDecoratorRoutes.length = 0; + for (const dr of expanded) allDecoratorRoutes.push(dr); + } + } + return { exportedTypeMap, allFetchCalls, diff --git a/gitnexus/src/core/ingestion/pipeline-phases/routes.ts b/gitnexus/src/core/ingestion/pipeline-phases/routes.ts index a87b9576d..8c0a67ac5 100644 --- a/gitnexus/src/core/ingestion/pipeline-phases/routes.ts +++ b/gitnexus/src/core/ingestion/pipeline-phases/routes.ts @@ -198,7 +198,6 @@ export const routesPhase: PipelinePhase = { } } - const ensureSlash = (path: string) => (path.startsWith('/') ? path : '/' + path); let duplicateRoutes = 0; const namedRouteRegistry = new Map(); const addRoute = (url: string, entry: RouteEntry) => { @@ -220,7 +219,8 @@ export const routesPhase: PipelinePhase = { } } for (const dr of allDecoratorRoutes) { - addRoute(ensureSlash(dr.routePath), { + const url = normalizeExtractedRoutePath(dr.routePath, dr.prefix ?? null); + addRoute(url, { filePath: dr.filePath, source: `decorator-${dr.decoratorName}`, }); diff --git a/gitnexus/src/core/ingestion/registry-primary-flag.ts b/gitnexus/src/core/ingestion/registry-primary-flag.ts index e552a4600..cbed4dde1 100644 --- a/gitnexus/src/core/ingestion/registry-primary-flag.ts +++ b/gitnexus/src/core/ingestion/registry-primary-flag.ts @@ -81,6 +81,7 @@ export const MIGRATED_LANGUAGES: ReadonlySet = new Set.include_router(, prefix='/x')` site, where + * `` is either `.router` (Shape A) or a bare + * local name (Shape B). `` is intentionally unconstrained: + * production code uses `app`, `api`, `application`, `asgi_app`, + * etc., and the call shape (`include_router` invoked with a + * `prefix=` keyword) is specific enough on its own. + * + * • {@link ExtractedRouterImport} — every + * `from import router [as ]`, captured for both + * absolute and relative module paths (`from .calls import …`). + * parse-impl uses the imports to resolve Shape-B local names back + * to the file that declares the router. + * + * Module keying is two-tiered to avoid prefix bleed between same-named + * files in different packages (e.g. `api/users.py` vs `admin/users.py`): + * + * • short key — basename without `.py` (`users`) + * • long key — `/` (`api/users`) + * + * Imports always carry the short key and, when the module path was + * multi-segment, also the long key. parse-impl matches against the + * long key first and falls back to the short key, so cross-package + * collisions are eliminated for Shape B and minimised for Shape A. + * + * The functions in this module are pure (no Worker / parentPort + * dependency) so they can be unit-tested directly without booting a + * worker thread. + */ + +/** + * One `.include_router(, prefix='/x')` site. + * + * `routerExpr` is the raw text of the first argument — either + * `.router` (Shape A) or a bare local name (Shape B). + * parse-impl resolves Shape B against {@link ExtractedRouterImport} + * records emitted by the same file. + */ +export interface ExtractedRouterInclude { + filePath: string; + routerExpr: string; + prefix: string; + lineNumber: number; +} + +/** + * One `from import router [as ]` discovered in a + * Python file. + * + * `moduleKey` is the short key (last `.`-segment of the module path, + * e.g. `api.users` → `users`). `moduleKeyLong` is the long key (last + * two segments joined with `/`, e.g. `api/users`); it is the empty + * string / undefined when the import is single-segment (e.g. + * `from users import router`) or pure-dots (e.g. `from . import + * router`). The long key, when present, gives parse-impl a precise + * way to bind a Shape-B `include_router` call to exactly one Python + * file even when other packages contain a same-named module. + */ +export interface ExtractedRouterImport { + filePath: string; + localName: string; + moduleKey: string; + moduleKeyLong?: string; +} + +/** + * One `from import ` discovered in a Python file + * where `` is later used as a Shape-A include receiver + * (`.include_router(.router, prefix='/x')`). Without + * this record parse-impl would have to fall back to the short key + * ``, which collides between e.g. `api/users.py` and + * `admin/users.py`. The record carries the long key + * (`/`) so parse-impl can pin the prefix onto the + * exact source file. + * + * Only emitted when the import path was multi-segment (a single + * `from users import users` would yield no long key). All fields + * carry the same module-key semantics as + * {@link ExtractedRouterImport}. + */ +export interface ExtractedRouterModuleAlias { + filePath: string; + /** Local name in the importing file (== imported name or its alias). */ + localName: string; + /** Long key (`/`) — non-empty for every emitted record. */ + moduleKeyLong: string; +} + +// `.include_router(.router, ..., prefix='/x')` (Shape A). +// `` is left unrestricted — common production names include +// `app`, `api`, `application`, `asgi_app`. Pinning to the literal +// `app` would silently drop these. +const INCLUDE_ROUTER_ATTR_RE = + /\b(?:[A-Za-z_][\w.]*)\.include_router\s*\(\s*([A-Za-z_][\w]*)\.router\b[^)]*?\bprefix\s*=\s*(['"])([^'"]*)\2/g; + +// `.include_router(, ..., prefix='/x')` (Shape B). +const INCLUDE_ROUTER_NAME_RE = + /\b(?:[A-Za-z_][\w.]*)\.include_router\s*\(\s*([A-Za-z_][\w]*)\b[^)]*?\bprefix\s*=\s*(['"])([^'"]*)\2/g; + +// Module path: a sequence of dots (`.`, `..`, `...`) for "current +// package" imports, OR an optional leading-dot prefix followed by a +// dotted identifier (`api.users`, `.api.users`, `..siblings.users`). +// The latter is the common case and the only one we can map back to +// a module stem. +const FROM_IMPORT_ROUTER_RE = /^\s*from\s+(\.+|\.*[A-Za-z_][\w.]*)\s+import\s+([^#\n]+)/gm; + +/** + * Last `.`-separated segment of a (possibly relative) Python module + * path. Strips any leading dots first so `from .api.assistant import + * …` and `from api.assistant import …` both yield `assistant`. + * Pure-dot inputs (`.`, `..`) have no segment and return the empty + * string; callers should skip empty results. + */ +export function lastDottedSegment(text: string): string { + const stripped = text.replace(/^\.+/, ''); + if (!stripped) return ''; + const dot = stripped.lastIndexOf('.'); + return dot >= 0 ? stripped.slice(dot + 1) : stripped; +} + +/** + * Last two `.`-separated segments of a (possibly relative) module + * path joined with `/`, e.g. `api.users` → `api/users`. Mirrors the + * long-key shape used for files (`api/users.py` → `api/users`). + * Returns the empty string when no parent segment is available + * (single-segment imports or pure dots); callers should fall back + * to the short key in that case. + */ +export function lastTwoSegmentsAsPath(text: string): string { + const stripped = text.replace(/^\.+/, ''); + if (!stripped) return ''; + const last = stripped.lastIndexOf('.'); + if (last <= 0) return ''; + const beforeLast = stripped.slice(0, last); + const stem = stripped.slice(last + 1); + const prev = beforeLast.lastIndexOf('.'); + const parent = prev >= 0 ? beforeLast.slice(prev + 1) : beforeLast; + return `${parent}/${stem}`; +} + +/** + * Scan a single Python file's source text for FastAPI router + * `include_router` sites and `from import router` imports, + * appending raw records to the supplied collectors. + * + * `outModuleAliases` is optional: when supplied, every multi-segment + * `from import ` (other than `router` itself) is recorded + * as a module alias so parse-impl can pin Shape-A + * `.include_router(...)` calls onto the exact module file. When + * omitted, the function preserves the pre-existing behaviour and + * skips the alias collection — this keeps the function signature + * back-compat with older callers (and the parse-cache replay path). + */ +export function extractFastAPIRouterBindings( + filePath: string, + content: string, + outIncludes: ExtractedRouterInclude[], + outImports: ExtractedRouterImport[], + outModuleAliases?: ExtractedRouterModuleAlias[], +): void { + if (!content.includes('include_router') && !content.includes('router')) return; + + // `from import router [as ]`. We capture every name + // in the import list. `router` (with or without an `as` alias) maps + // to outImports; every other name lands in outModuleAliases when a + // long key is available, so Shape-A `.router` includes can be + // pinned to the exact module file. + if (content.includes(' import ')) { + FROM_IMPORT_ROUTER_RE.lastIndex = 0; + let m: RegExpExecArray | null; + while ((m = FROM_IMPORT_ROUTER_RE.exec(content)) !== null) { + const moduleText = m[1]; + const importList = m[2]; + const moduleShort = lastDottedSegment(moduleText); + if (!moduleShort) continue; + // Long key for the imported MODULE itself (used by router + // imports — `from api.users import router` sets + // `moduleKeyLong = api/users`). + const moduleLong = lastTwoSegmentsAsPath(moduleText); + // Strip surrounding parens / trailing whitespace; split on + // commas. (Multiline import groups already have their newlines + // present in the captured list.) + const cleaned = importList.replace(/[()]/g, '').trim(); + for (const rawPart of cleaned.split(',')) { + const part = rawPart.trim(); + if (!part) continue; + + // `router` or `router as foo` → ExtractedRouterImport. + const routerAlias = /^router(?:\s+as\s+([A-Za-z_]\w*))?$/.exec(part); + if (routerAlias) { + const localName = routerAlias[1] ?? 'router'; + outImports.push({ + filePath, + localName, + moduleKey: moduleShort, + ...(moduleLong ? { moduleKeyLong: moduleLong } : {}), + }); + continue; + } + + // Any other `` or ` as ` — recorded as a + // module alias so parse-impl can pin Shape-A includes. The + // long key here is computed against the IMPORTED MODULE PATH + // (`.`), not the package path that `` + // was imported FROM. `from api import users` therefore yields + // `api/users`, the same long key as the file it points at. + if (!outModuleAliases) continue; + const otherAlias = /^([A-Za-z_]\w*)(?:\s+as\s+([A-Za-z_]\w*))?$/.exec(part); + if (!otherAlias) continue; + const importedName = otherAlias[1]; + const localName = otherAlias[2] ?? importedName; + const aliasLong = lastTwoSegmentsAsPath(`${moduleText}.${importedName}`); + if (!aliasLong) continue; + outModuleAliases.push({ + filePath, + localName, + moduleKeyLong: aliasLong, + }); + } + } + } + + if (!content.includes('include_router')) return; + + // Shape A: `.include_router(.router, prefix='/x')`. + INCLUDE_ROUTER_ATTR_RE.lastIndex = 0; + let m: RegExpExecArray | null; + while ((m = INCLUDE_ROUTER_ATTR_RE.exec(content)) !== null) { + outIncludes.push({ + filePath, + routerExpr: `${m[1]}.router`, + prefix: m[3], + lineNumber: content.substring(0, m.index).split('\n').length, + }); + } + + // Shape B: `.include_router(my_router, prefix='/x')`. + // Resolution to a module key happens in parse-impl using + // outImports from the same file. + INCLUDE_ROUTER_NAME_RE.lastIndex = 0; + while ((m = INCLUDE_ROUTER_NAME_RE.exec(content)) !== null) { + // Skip cases that already matched Shape A — INCLUDE_ROUTER_NAME_RE + // is intentionally permissive and would re-capture `.router` + // as the bare name `mod`. Discriminate by re-checking the + // immediate source around the captured argument position. + const argStart = m.index + m[0].indexOf(m[1]); + const dotProbe = content.slice(argStart + m[1].length, argStart + m[1].length + 8); + if (/^\s*\.\s*router/.test(dotProbe)) continue; + outIncludes.push({ + filePath, + routerExpr: m[1], + prefix: m[3], + lineNumber: content.substring(0, m.index).split('\n').length, + }); + } +} diff --git a/gitnexus/src/core/ingestion/scope-extractor.ts b/gitnexus/src/core/ingestion/scope-extractor.ts index 963cf4862..31a59d2f5 100644 --- a/gitnexus/src/core/ingestion/scope-extractor.ts +++ b/gitnexus/src/core/ingestion/scope-extractor.ts @@ -750,6 +750,61 @@ function normalizeNodeLabel(kindStr: string): SymbolDefinition['type'] | undefin } } +/** Function-like labels: callable defs that must keep incoming CALLS edges. */ +const NODE_BEARING_FUNCTION_LABELS: ReadonlySet = new Set([ + 'Function', + 'Method', + 'Constructor', +]); + +/** Value labels: non-callable bindings (a `const`/`let`/`var` holds a value). */ +const NODE_BEARING_VALUE_LABELS: ReadonlySet = new Set([ + 'Const', + 'Variable', +]); + +/** + * Collapse rule for the deferred node-creation migration (#1876). + * + * When graph-node creation moves from the legacy DAG onto the + * registry-primary path, a single source binding can carry more than one + * `SymbolDefinition` for the same name in the same scope — e.g. a direct + * arrow `const fn = () => {}` is classified BOTH as a `Function` (the + * arrow) and a `Variable` (the binding). Emitting one graph node per def + * would reproduce exactly the duplicate-node bug this issue tracks. + * + * `selectNodeBearingDef` picks the ONE def that should bear the graph node + * for such a binding group: + * + * 1. a function-like def (`Function` / `Method` / `Constructor`) if any — + * the binding is callable and must keep incoming `CALLS` edges; + * 2. otherwise a value def (`Const` / `Variable`) — the binding holds a + * value (e.g. an array-method result after the U1/U2 narrowing); + * 3. otherwise the first def — deterministic fallback for label sets this + * rule does not rank. + * + * INPUT CONTRACT: `group` must be the defs bound to ONE name within ONE + * scope (a binding group). It deliberately does NOT dedup by range — + * `SymbolDefinition` carries no range and `makeDefId` encodes only the + * start position, so containment is uncomputable here; the caller forms the + * group (e.g. from a scope's `ownedDefs` keyed by name) before calling. + * + * Pure. No production call site yet — this dead export is intentional and + * tracked by #1876 (the deferred node-creation migration); it is the + * executable contract that follow-up will consume, pinned today by the + * scope-extractor unit test. + */ +export function selectNodeBearingDef( + group: readonly SymbolDefinition[], +): SymbolDefinition | undefined { + if (group.length === 0) return undefined; + const functionLike = group.find((def) => NODE_BEARING_FUNCTION_LABELS.has(def.type)); + if (functionLike !== undefined) return functionLike; + const value = group.find((def) => NODE_BEARING_VALUE_LABELS.has(def.type)); + if (value !== undefined) return value; + return group[0]; +} + function makeDefId( filePath: string, range: Range, @@ -1087,6 +1142,7 @@ const KNOWN_SUB_TAGS: ReadonlySet = new Set([ '@declaration.required-parameter-count', '@declaration.parameter-types', '@declaration.parameter-type-classes', + '@declaration.return-type', '@declaration.template-constraints', '@declaration.is-explicit', ]); diff --git a/gitnexus/src/core/ingestion/scope-resolution/pipeline/phase.ts b/gitnexus/src/core/ingestion/scope-resolution/pipeline/phase.ts index 5e10750b2..ae7847e68 100644 --- a/gitnexus/src/core/ingestion/scope-resolution/pipeline/phase.ts +++ b/gitnexus/src/core/ingestion/scope-resolution/pipeline/phase.ts @@ -116,6 +116,20 @@ export const scopeResolutionPhase: PipelinePhase = { preExtractedByPath.set(pf.filePath, pf); } + // Drop pre-extracted entries for standalone providers — these + // languages are skipped by the canonical guard below (line 164) + // and never consume preExtractedByPath, so holding onto their + // entries leaks memory until the cleanup loop at 262-264 which + // also never runs for skipped providers. + for (const [path] of preExtractedByPath) { + const lang = getLanguageFromFilename(path); + if (lang === null) continue; + const provider = SCOPE_RESOLVERS.get(lang); + if (provider?.languageProvider.parseStrategy === 'standalone') { + preExtractedByPath.delete(path); + } + } + let totalFiles = 0; let totalImports = 0; let totalRefs = 0; @@ -158,6 +172,14 @@ export const scopeResolutionPhase: PipelinePhase = { for (const [lang, provider] of SCOPE_RESOLVERS) { if (!isRegistryPrimary(lang)) continue; + // Standalone providers (COBOL, JCL) don't emit graph edges yet + // through the scope-resolution path. This is the canonical guard: + // runScopeResolution is never called for standalone providers, which + // keeps cobolPhase as the sole IMPORTS edge producer. Keep this guard + // in sync with any additional standalone providers added to + // SCOPE_RESOLVERS. + if (provider.languageProvider.parseStrategy === 'standalone') continue; + const langFiles = scannedFiles.filter((f) => getLanguageFromFilename(f.path) === lang); if (langFiles.length === 0) continue; diff --git a/gitnexus/src/core/ingestion/scope-resolution/pipeline/validate-bindings-immutability.ts b/gitnexus/src/core/ingestion/scope-resolution/pipeline/validate-bindings-immutability.ts index 60e5fe179..5ccba91ac 100644 --- a/gitnexus/src/core/ingestion/scope-resolution/pipeline/validate-bindings-immutability.ts +++ b/gitnexus/src/core/ingestion/scope-resolution/pipeline/validate-bindings-immutability.ts @@ -1,6 +1,6 @@ /** - * Dev-mode runtime validator for the two-channel binding lifecycle - * (Contract Invariant I8 in `contract/scope-resolver.ts`). + * Dev-mode runtime validator for the post-finalize binding-channel + * lifecycle (Contract Invariant I8 in `contract/scope-resolver.ts`). * * The two channels: * - `indexes.bindings` — finalize-output channel. After @@ -74,5 +74,21 @@ export function validateBindingsImmutability( } } + // Third channel: `workspaceFqnBindings` (scope-independent, shared map + // populated by language namespace-sibling hooks — PHP FQN keys, C# + // global-namespace simple names). Like bindingAugmentations its inner + // arrays are mutable by contract (hooks `push()` directly), so freezing + // one is the same defect as freezing an augmentation bucket. + for (const [name, bucket] of indexes.workspaceFqnBindings) { + if (Object.isFrozen(bucket)) { + onWarn( + `binding-immutability: indexes.workspaceFqnBindings[${name}] is FROZEN — ` + + `the workspace channel is mutable by contract; freezing it defeats the ` + + `append-only purpose. See ScopeResolver Invariant I8.`, + ); + violations++; + } + } + return violations; } diff --git a/gitnexus/src/core/ingestion/scope-resolution/scope/walkers.ts b/gitnexus/src/core/ingestion/scope-resolution/scope/walkers.ts index df6d259b5..6bf967fff 100644 --- a/gitnexus/src/core/ingestion/scope-resolution/scope/walkers.ts +++ b/gitnexus/src/core/ingestion/scope-resolution/scope/walkers.ts @@ -99,6 +99,14 @@ const EMPTY_NAMES: Iterable = Object.freeze([]) as readonly string[]; * Fast paths (zero allocation) when at most one channel is populated: * returns the underlying `Map.keys()` iterator directly. Only when both * channels carry names do we materialize a `Set` for deduplication. + * + * Scope: enumerates only the per-scope `bindings` and `bindingAugmentations` + * channels. It deliberately EXCLUDES the scope-independent + * `workspaceFqnBindings` channel (PHP FQN keys, C# global-namespace simple + * names). `lookupBindingsAt` consults that third channel when resolving a + * specific name, but name *enumeration* here does not — those names apply at + * every scope and would flood per-scope callers. Callers that need + * workspace-level names must read `workspaceFqnBindings` directly. */ export function namesAtScope(scopeId: ScopeId, scopes: ScopeResolutionIndexes): Iterable { const finalized = scopes.bindings.get(scopeId); diff --git a/gitnexus/src/core/ingestion/workers/parse-worker.ts b/gitnexus/src/core/ingestion/workers/parse-worker.ts index f092d4a1a..a20d86d92 100644 --- a/gitnexus/src/core/ingestion/workers/parse-worker.ts +++ b/gitnexus/src/core/ingestion/workers/parse-worker.ts @@ -23,6 +23,11 @@ import { import { parseSourceSafe } from '../../tree-sitter/safe-parse.js'; import type { SymbolTableReader } from '../model/symbol-table.js'; import type { ExtractedHeritage } from '../model/heritage-map.js'; +import type { + ExtractedRouterInclude, + ExtractedRouterImport, + ExtractedRouterModuleAlias, +} from '../route-extractors/fastapi-router-bindings.js'; /** Language grammar type accepted by Parser.setLanguage(). */ type TreeSitterLanguage = Parameters[0]; @@ -209,6 +214,19 @@ export interface ExtractedDecoratorRoute { httpMethod: string; decoratorName: string; lineNumber: number; + /** + * Decorator receiver identifier (e.g. `router` for `@router.get(...)`, + * `app` for `@app.get(...)`). Used by parse-impl to decide which routes + * participate in `include_router(prefix=...)` joining. + */ + decoratorReceiver?: string; + /** + * FastAPI `app.include_router(prefix='/x')` prefix that applies to + * this route. Filled by parse-impl after cross-file aggregation; the + * routes phase joins it via `normalizeExtractedRoutePath`. `null` / + * absent ⇒ no prefix applies. + */ + prefix?: string | null; } export interface ExtractedToolDef { @@ -275,6 +293,18 @@ export interface ParseWorkerResult { fetchCalls: ExtractedFetchCall[]; fetchWrapperDefs: FetchWrapperDef[]; decoratorRoutes: ExtractedDecoratorRoute[]; + routerIncludes: ExtractedRouterInclude[]; + routerImports: ExtractedRouterImport[]; + /** + * Optional. `from import ` records from Python files + * where `` is later used as a Shape-A include receiver + * (`.include_router(.router, prefix='/x')`). parse-impl + * uses these to promote Shape-A short-key entries to long keys, so + * same-named modules in different packages don't share prefixes. + * Optional for cache backward compatibility (older cache entries + * predate the field; consumers must guard with `if (… ?? [])`). + */ + routerModuleAliases?: ExtractedRouterModuleAlias[]; toolDefs: ExtractedToolDef[]; ormQueries: ExtractedORMQuery[]; constructorBindings: FileConstructorBindings[]; @@ -740,6 +770,9 @@ const processBatch = ( fetchCalls: [], fetchWrapperDefs: [], decoratorRoutes: [], + routerIncludes: [], + routerImports: [], + routerModuleAliases: [], toolDefs: [], ormQueries: [], constructorBindings: [], @@ -779,9 +812,34 @@ const processBatch = ( for (const [language, langFiles] of byLanguage) { const provider = getProvider(language); const queryString = provider.treeSitterQueries; - if (!queryString) continue; - - // Track if we need to handle tsx separately + if (!queryString) { + // Standalone providers (regex-based, no tree-sitter) that implement + // emitScopeCaptures feed into the scope-resolution pipeline via + // extractParsedFile directly — no tree-sitter involved. + if (provider.emitScopeCaptures) { + for (const file of langFiles) { + const parsedFile = extractParsedFile( + provider, + file.content, + file.path, + (message) => { + if (parentPort) { + parentPort.postMessage({ type: 'warning', message }); + } else { + logger.warn(message); + } + }, + undefined, // no cachedTree for standalone providers + ); + if (parsedFile !== undefined) { + result.parsedFiles.push(parsedFile); + result.fileCount++; + onFileProcessed?.(); + } + } + } + continue; + } const tsxFiles: ParseWorkerInput[] = []; const regularFiles: ParseWorkerInput[] = []; @@ -968,6 +1026,18 @@ export function extractORMQueries( } } +// ============================================================================ +// FastAPI router prefix detection (Python) +// ============================================================================ +// +// The extraction lives in `../route-extractors/fastapi-router-bindings` +// (a pure-function module — NOT a worker, no `worker_threads`, no +// `parentPort`). It's imported here only so the worker entry can call it +// per file; this module does not re-export it. Downstream consumers +// import the function and its types directly from `route-extractors/`. + +import { extractFastAPIRouterBindings } from '../route-extractors/fastapi-router-bindings.js'; + const processFileGroup = ( files: ParseWorkerInput[], language: SupportedLanguages, @@ -1200,6 +1270,7 @@ const processFileGroup = ( if (captureMap['decorator'] && captureMap['decorator.name']) { const decoratorName = captureMap['decorator.name'].text; const decoratorArg = captureMap['decorator.arg']?.text; + const decoratorReceiver = captureMap['decorator.receiver']?.text; const decoratorNode = captureMap['decorator']; // Store by the decorator's end line — the definition follows immediately after fileDecorators.set(decoratorNode.endPosition.row, { @@ -1219,6 +1290,7 @@ const processFileGroup = ( httpMethod, decoratorName, lineNumber: decoratorNode.startPosition.row + lineOffset, + ...(decoratorReceiver ? { decoratorReceiver } : {}), }); } // MCP/RPC tool detection: @mcp.tool(), @app.tool(), @server.tool() @@ -1994,6 +2066,20 @@ const processFileGroup = ( // Extract ORM queries (Prisma, Supabase) extractORMQueries(file.path, parseContent, result.ormQueries); + // Extract FastAPI include_router(prefix=...) and `from import router` + // sites. parse-impl aggregates these into a per-module prefix map and + // injects the resolved prefix onto each ExtractedDecoratorRoute that + // came from a `@router.` decorator. Python-only. + if (language === SupportedLanguages.Python) { + extractFastAPIRouterBindings( + file.path, + parseContent, + result.routerIncludes, + result.routerImports, + (result.routerModuleAliases ??= []), + ); + } + // Vue: emit CALLS edges for components used in