Merge remote-tracking branch 'origin/main' into feat/phase8-9-implementation

# Conflicts:
#	gitnexus/package-lock.json
This commit is contained in:
Gergo Magyar 2026-03-26 08:01:05 +00:00
commit 48056460a5
190 changed files with 12642 additions and 4585 deletions

21
.cursor/index.mdc Normal file
View file

@ -0,0 +1,21 @@
---
alwaysApply: true
---
# GitNexus — Cursor project rules
Last reviewed: 2026-03-24
Canonical agent instructions: **[AGENTS.md](../AGENTS.md)** (GitNexus MCP rules, monorepo commands, Cursor Cloud notes). **[CLAUDE.md](../CLAUDE.md)** adds Claude Code-specific notes and points back to AGENTS.md for GitNexus.
## Non-negotiables (always apply)
- NEVER edit a function/class/method without running `gitnexus_impact` first.
- NEVER rename symbols with find-and-replace — use `gitnexus_rename`.
- NEVER commit without running `gitnexus_detect_changes()`.
- NEVER ignore HIGH/CRITICAL risk warnings from impact analysis.
- NEVER run `npx gitnexus analyze` without `--embeddings` if `.gitnexus/meta.json` shows stored embeddings.
Full rules: **[AGENTS.md](../AGENTS.md)** (`gitnexus:start` block, Cursor Cloud section).
**Rule architecture:** Prefer this file plus optional `.cursor/rules/*.mdc` globs (YAML `globs` in frontmatter). Legacy `.cursorrules` is deprecated; content lives here.

View file

@ -0,0 +1,12 @@
---
globs:
- "gitnexus/**"
- "gitnexus-web/**"
---
# GitNexus build/test quick refs
- CLI (`gitnexus/`): `npm test`; `npm run test:integration`; `npx tsc --noEmit`.
- Web (`gitnexus-web/`): `npm test`; `npm run dev`; `npx tsc -b --noEmit`; `E2E=1 npx playwright test` (needs servers).
- `npm install` in `gitnexus/` runs `prepare` (tsc build) and `postinstall` (tree-sitter patches); needs `python3`, `make`, `g++`.
- LadybugDB locking tests may fail in containerized environments because of `/tmp` file locks (known issue, not a code bug).

View file

@ -0,0 +1,14 @@
---
globs:
- "eval/**"
---
# GitNexus eval harness (Python)
- **Run tests**: `cd eval && uv run pytest tests/`
- **Run with coverage**: `cd eval && uv run coverage run -m pytest tests/ && uv run coverage report`
- **Lint**: `cd eval && uv run ruff check .`
- **Run eval**: `cd eval && uv run python run_eval.py --config configs/<config>.yaml`
- Shared constants live in `eval/constants.py`; tool specs in `eval/tool_registry.py`.
- Error logging uses `utils/errors.py` — set `GITNEXUS_EVAL_DEBUG=1` for full tracebacks.
- Property-based tests use Hypothesis (`eval/tests/test_property_based.py`).

View file

@ -1,5 +1,5 @@
# AI Agent Rules
# Deprecated for Cursor Agent Mode
Follow .gitnexus/RULES.md for all project context and coding guidelines.
Use **`.cursor/index.mdc`** (`alwaysApply: true`) for project rules. See [AGENTS.md](AGENTS.md).
This project uses GitNexus MCP for code intelligence. See .gitnexus/RULES.md for available tools and best practices.
This file is kept only as a breadcrumb for older workflows.

86
.github/ISSUE_TEMPLATE/bug_report.yml vendored Normal file
View file

@ -0,0 +1,86 @@
name: Bug report
description: Report unexpected behavior or a regression
labels: [bug]
body:
- type: markdown
attributes:
value: |
**Goal:** capture enough context to reproduce and fix the issue quickly.
Use **one issue per bug**; split unrelated problems.
- type: dropdown
id: area
attributes:
label: Area
description: Where does the problem show up?
options:
- gitnexus (CLI / core / indexing / MCP server)
- gitnexus-web (browser UI / WASM / workers)
- CI / GitHub Actions
- Documentation / developer experience
- Other
validations:
required: true
- type: textarea
id: summary
attributes:
label: Summary
description: One sentence — what went wrong?
validations:
required: true
- type: textarea
id: context
attributes:
label: Context
description: What were you trying to do? Any relevant links, PRs, or commits?
validations:
required: false
- type: textarea
id: expected
attributes:
label: Expected behavior
validations:
required: true
- type: textarea
id: actual
attributes:
label: Actual behavior
validations:
required: true
- type: textarea
id: reproduce
attributes:
label: Steps to reproduce
description: Ordered steps, sample repo or minimal case, commands run.
placeholder: |
1. …
2. …
3. …
validations:
required: true
- type: textarea
id: environment
attributes:
label: Environment
description: OS, Node version, browser (if web), GitNexus version or commit SHA.
placeholder: |
- OS:
- Node:
- Browser (if applicable):
- Commit / version:
validations:
required: false
- type: textarea
id: logs
attributes:
label: Logs / screenshots
description: Paste errors, stack traces, or attach screenshots (redact secrets).
validations:
required: false

1
.github/ISSUE_TEMPLATE/config.yml vendored Normal file
View file

@ -0,0 +1 @@
blank_issues_enabled: true

View file

@ -0,0 +1,74 @@
name: Feature request
description: Propose a new capability or improvement
labels: [enhancement]
body:
- type: markdown
attributes:
value: |
**Goal:** describe the problem and desired outcome so maintainers can size and prioritize.
Prefer **small, shippable** requests; split large ideas into phases.
- type: dropdown
id: area
attributes:
label: Area
description: Primary part of the monorepo this relates to.
options:
- gitnexus (CLI / core / indexing / MCP server)
- gitnexus-web (browser UI / WASM / workers)
- CI / release / packaging
- Documentation / developer experience
- Other
validations:
required: true
- type: textarea
id: problem
attributes:
label: Problem or opportunity
description: What pain point or gap exists today?
validations:
required: true
- type: textarea
id: proposal
attributes:
label: Proposed solution
description: What should happen instead? User-visible behavior, APIs, or UX.
validations:
required: true
- type: textarea
id: alternatives
attributes:
label: Alternatives considered
description: Other approaches you considered and why this one is preferred.
validations:
required: false
- type: textarea
id: acceptance
attributes:
label: Acceptance criteria
description: Testable conditions for “done” (bullets or checkboxes in prose).
placeholder: |
- When … then …
- Documentation / tests updated where appropriate
validations:
required: false
- type: textarea
id: constraints
attributes:
label: Constraints
description: Compatibility, performance, security, or “must not change” boundaries.
validations:
required: false
- type: checkboxes
id: willing
attributes:
label: Contribution
options:
- label: I am willing to open a PR for this (may need design discussion first).
required: false

52
.github/PULL_REQUEST_TEMPLATE.md vendored Normal file
View file

@ -0,0 +1,52 @@
## Summary
<!-- One or two sentences: what does this PR change? -->
## Motivation / context
<!-- Why is this change needed? Link issues, ADRs, or prior discussion. -->
## Areas touched
<!-- Check all that apply -->
- [ ] `gitnexus/` (CLI / core / MCP server)
- [ ] `gitnexus-web/` (Vite / React UI)
- [ ] `.github/` (workflows, actions)
- [ ] `eval/` or other tooling
- [ ] Docs / agent config only (`AGENTS.md`, `CLAUDE.md`, `.cursor/`, `llms.txt`, etc.)
## Scope & constraints
**In scope**
- <!-- bullets -->
**Explicitly out of scope / not done here**
- <!-- bullets — prevents reviewers assuming missing work is an oversight -->
## Implementation notes
<!-- Optional: design choices, tradeoffs, follow-ups -->
## Testing & verification
<!-- What you ran; paste commands. Omit sections that do not apply. -->
- [ ] `cd gitnexus && npm test`
- [ ] `cd gitnexus && npm run test:integration` *(if core/indexing/MCP paths changed)*
- [ ] `cd gitnexus && npx tsc --noEmit`
- [ ] `cd gitnexus-web && npm test` *(if web changed)*
- [ ] `cd gitnexus-web && npx tsc -b --noEmit` *(if web changed)*
- [ ] Manual / Playwright E2E *(note environment — see `gitnexus-web/e2e/`)*
## Risk & rollout
<!-- Breaking changes, migrations, index refresh (`npx gitnexus analyze`), release notes -->
## Checklist
- [ ] PR body meets repo minimum length (workflow may label short descriptions)
- [ ] If `AGENTS.md` / overlays changed: headers, scope block, and changelog updated per project conventions
- [ ] No secrets, tokens, or machine-specific paths committed

14
.gitignore vendored
View file

@ -81,4 +81,16 @@ GitNexus.sln
# Git worktrees
.worktrees/
/github/scripts/triage/__pycache__/
/github/scripts/triage/__pycache__/
.claude-flow/
.claude/agents/
.claude/commands/
.claude/helpers
.claude/skills/
!.claude/skills/gitnexus/
.history/
.swarm/

105
AGENTS.md
View file

@ -1,7 +1,69 @@
<!-- version: 1.2.0 -->
<!--
Metadata: version, last reviewed, scope, model policy, reference docs, changelog.
Last updated: 2026-03-22
-->
Last reviewed: 2026-03-24
**Project:** GitNexus · **Environment:** dev · **Maintainer:** repository maintainers (see GitHub)
This file uses a standard agent header (version, scope, model policy, reference docs, changelog), adapted for this **TypeScript/JavaScript monorepo**.
## Scope
| | |
|--|--|
| **Reads** | Repository tree as needed for the task: `gitnexus/`, `gitnexus-web/`, `eval/`, plugin packages, `.github/`, `.gitnexus/` when present, and docs. |
| **Writes** | Only paths required for the requested change; keep diffs minimal. Update lockfiles when dependencies change. |
| **Executes** | `npm`, `npx`, `node` under `gitnexus/` and `gitnexus-web/`; `uv run` for Python under `eval/` when applicable; shell utilities for documented CI/dev workflows. |
| **Off-limits** | User secrets (e.g. real `.env`), production deployment credentials, unrelated repositories, destructive git history operations without explicit human confirmation. |
## Model Configuration
- **Primary:** Pin in **Cursor** (Settings → model). Use a **named** model (e.g. GPT-5.2, Claude Sonnet 4.x). Avoid relying on **Auto** when reproducibility or audit trail matters.
- **Fallback:** As configured in Cursor or your organization (do not encode `latest` or wildcards in automation configs).
- **Notes:** The open-source GitNexus CLI indexer does not call an LLM. Optional Nexus AI in the web UI uses end-user provider keys and models.
## Execution Sequence (complex tasks)
Long sessions dilute instructions. For **multi-step** work, state up front:
1. Which rules in this file and **[GUARDRAILS.md](GUARDRAILS.md)** apply (and any relevant Signs).
2. Current **Scope** boundaries (Reads / Writes / Off-limits).
3. Which **validation commands** you will run (e.g. `cd gitnexus && npm test`, `npx tsc --noEmit`).
On very long threads, the human may add *“Remember: apply all AGENTS.md rules”* to re-weight rule tokens against context dilution.
## Claude Code hooks
Hooks enforce gates that prompts cannot. In **Claude Code**, **PreToolUse** hooks can block tools such as `git_commit` until checks pass. Adapt to this repo: e.g. `cd gitnexus && npm test` before commit.
## Context budget (Cursor / standards)
Generic “core standards” playbooks are often long and stack-specific. For this monorepo, commands and gotchas live under **Cursor Cloud specific instructions** below and in **[CONTRIBUTING.md](CONTRIBUTING.md)**. If always-on rules grow, split domain rules into **`.cursor/rules/*.mdc`** (globs). **Cursor:** project-wide rules live in **`.cursor/index.mdc`** (YAML frontmatter with `alwaysApply: true`). **Claude Code:** optionally load a **`STANDARDS.md`** only when needed (e.g. *“When writing new code, read STANDARDS.md”*) to save context.
## Reference Documentation
- **This repository:** **[ARCHITECTURE.md](ARCHITECTURE.md)**, **[CONTRIBUTING.md](CONTRIBUTING.md)**, **[GUARDRAILS.md](GUARDRAILS.md)**.
- **Cursor:** `.cursor/index.mdc` (always-on rules); optional `.cursor/rules/*.mdc` (glob-scoped). Legacy `.cursorrules` is deprecated — see `.cursor/index.mdc`.
- **Optional local files:** `NOTES.md` (short vendor-neutral project snapshot). For handoffs, keep notes local (e.g., a scratch file outside the repo) rather than committing `HANDOFF.md`.
- **GitNexus:** skills under `.claude/skills/gitnexus/`; machine-oriented rules in the `gitnexus:start` … `gitnexus:end` block below.
## Changelog
| Date | Version | Change |
|------|---------|--------|
| 2026-03-24 | 1.2.0 | Fixed gitnexus:start block duplication (was inlined in Reference Docs bullet). |
| 2026-03-23 | 1.1.0 | Updated agent instructions (sections, references, Cursor layout). |
| 2026-03-22 | 1.0.0 | Added structured agent header and changelog. |
---
<!-- gitnexus:start -->
# GitNexus — Code Intelligence
This project is indexed by GitNexus as **GitNexus** (2298 symbols, 5501 relationships, 175 execution flows). Use the GitNexus MCP tools to understand code, assess impact, and navigate safely.
This project is indexed by GitNexus as **GitNexus**. Use the GitNexus MCP tools to understand code, assess impact, and navigate safely. For current symbol stats, run `npx gitnexus analyze` and inspect `.gitnexus/meta.json`.
> If any GitNexus tool warns the index is stale, run `npx gitnexus analyze` in terminal first.
@ -99,3 +161,44 @@ To check whether embeddings exist, inspect `.gitnexus/meta.json` — the `stats.
| Index, status, clean, wiki CLI commands | `.claude/skills/gitnexus/gitnexus-cli/SKILL.md` |
<!-- gitnexus:end -->
## Cursor Cloud specific instructions
### Repository structure
This is a monorepo with two main products and supporting config packages:
| Component | Path | Purpose |
|-----------|------|---------|
| **GitNexus CLI/Core** | `gitnexus/` | Main product — TypeScript CLI, indexing pipeline, MCP server. Published to npm. |
| **GitNexus Web UI** | `gitnexus-web/` | React/Vite browser app — graph explorer + AI chat. Runs entirely in WASM. |
| Claude Plugin | `gitnexus-claude-plugin/` | Static config for Claude marketplace (no build). |
| Cursor Integration | `gitnexus-cursor-integration/` | Static config for Cursor editor (no build). |
| SWE-bench Eval | `eval/` | Python evaluation harness (optional; needs Docker + LLM API keys). |
### Running services
- **CLI/Core**: `cd gitnexus && npm run dev` (tsx watch mode) or `npm run build && node dist/cli/index.js <command>`
- **Web UI**: `cd gitnexus-web && npm run dev` (Vite on port 5173)
- **Backend mode**: `cd <indexed-repo> && node /workspace/gitnexus/dist/cli/index.js serve` (HTTP API on port 3741 by default)
### Testing
**CLI / Core (`gitnexus/`)**
- **Unit tests**: `cd gitnexus && npm test` (vitest, ~2000 tests)
- **Integration tests**: `cd gitnexus && npm run test:integration` (vitest, ~1850 tests). Two LadybugDB file-locking tests (`lbug-core-adapter`, `search-core`) may fail in containerized environments due to `/tmp` locking limitations — this is a known environment issue, not a code bug.
- **TypeScript check**: `cd gitnexus && npx tsc --noEmit`
**Web UI (`gitnexus-web/`)**
- **Unit tests**: `cd gitnexus-web && npm test` (vitest, ~200 tests)
- **E2E tests**: `cd gitnexus-web && E2E=1 npx playwright test` (Playwright, 5 tests — requires `gitnexus serve` + `npm run dev` running)
- **TypeScript check**: `cd gitnexus-web && npx tsc -b --noEmit`
No separate lint command is configured; TypeScript strict checking serves as the primary static analysis.
### Gotchas
- `npm install` in `gitnexus/` triggers `prepare` (builds via `tsc`) and `postinstall` (patches tree-sitter-swift). Native tree-sitter bindings require `python3`, `make`, and `g++` to be present.
- `tree-sitter-kotlin` and `tree-sitter-swift` are optional dependencies — install warnings for these are expected and non-blocking.
- The Web UI uses `vite-plugin-wasm` and requires `Cross-Origin-Opener-Policy`/`Cross-Origin-Embedder-Policy` headers for `SharedArrayBuffer` (handled automatically by Vite dev server).
- There is no ESLint/Prettier configuration in this repo.

66
ARCHITECTURE.md Normal file
View file

@ -0,0 +1,66 @@
# Architecture — GitNexus
This repository is a **monorepo** with two main products: the **CLI / MCP package** (`gitnexus/`) and the **browser UI** (`gitnexus-web/`). Supporting folders ship editor integrations and plugins without changing the core graph engine.
## Repository layout
| Path | Role |
|------|------|
| `gitnexus/` | Published npm package `gitnexus`: CLI, MCP server (stdio), local HTTP API for bridge mode, ingestion pipeline, LadybugDB graph, embeddings (optional). |
| `gitnexus-web/` | Vite + React UI: in-browser indexing (WASM), graph visualization, optional connection to `gitnexus serve`. |
| `.claude/`, `gitnexus-claude-plugin/`, `gitnexus-cursor-integration/` | Packaged **skills** and plugin metadata so agents discover the same workflows as documented in `AGENTS.md`. |
| `eval/` | Evaluation harnesses and docs for benchmarking tool usage. |
| `.github/` | CI workflows (quality, unit, integration, E2E) and composite actions. |
## End-to-end flow: index → graph → tools
1. **Ingestion** (`gitnexus analyze`)
- Entry: `gitnexus/src/cli/analyze.ts` → `runPipelineFromRepo` in `gitnexus/src/core/ingestion/pipeline.ts`.
- Walks the git working tree, parses supported languages via **Tree-sitter**, resolves imports/calls/inheritance, detects **communities** and **processes** (execution flows), and builds an in-memory **knowledge graph** (`gitnexus/src/core/graph/`).
- Output is loaded into **LadybugDB** under **`.gitnexus/`** at the repo root (`lbug/`, `meta.json`, etc.). Optional **FTS** indexes and **embeddings** attach to the same store.
- The repo is registered in **`~/.gitnexus/registry.json`** so MCP can find it from any working directory.
2. **Persistence & metadata**
- `gitnexus/src/storage/repo-manager.ts` — paths, registry, cleanup of legacy Kuzu artifacts.
- `gitnexus/src/core/lbug/lbug-adapter.ts` — graph load, queries, embedding restore batches.
3. **Query & agents**
- **MCP (stdio):** `gitnexus/src/cli/mcp.ts` → `startMCPServer` → `LocalBackend` (`gitnexus/src/mcp/local/local-backend.ts`) opens registered repos and serves **tools** from `gitnexus/src/mcp/tools.ts` and **resources** from `gitnexus/src/mcp/resources.ts`.
- **Bridge HTTP:** `gitnexus/src/cli/serve.ts` → Express app in `gitnexus/src/server/api.ts` (CORS-limited) exposes REST + MCP-over-HTTP for the web UI.
- **CLI tools (no MCP):** `gitnexus query`, `context`, `impact`, `cypher` in `gitnexus/src/cli/tool.ts` call the same backend for scripts and CI.
4. **Staleness**
- `gitnexus/src/mcp/staleness.ts` compares indexed `lastCommit` to `HEAD` and surfaces hints when the graph is behind git.
## MCP tools (summary)
| Tool | Purpose |
|------|---------|
| `list_repos` | Discover indexed repositories when more than one is registered. |
| `query` | Natural-language / keyword search over the graph (hybrid BM25 + optional vectors). |
| `cypher` | Ad hoc **Cypher** against the schema (see resource `gitnexus://repo/{name}/schema`). |
| `context` | Callers, callees, processes for one symbol (with disambiguation). |
| `impact` | Blast radius (upstream/downstream) with depth and risk summary. |
| `detect_changes` | Map git diffs to affected symbols and processes. |
| `rename` | Graph-assisted rename with `dry_run` preview (`graph` vs `text_search` confidence). |
## Where to change what
| If you are changing… | Start in… |
|----------------------|-----------|
| CLI commands / flags | `gitnexus/src/cli/` (`index.ts`, per-command modules). |
| Parsing or graph construction | `gitnexus/src/core/ingestion/` (pipeline, processors, resolvers, type-extractors). |
| Graph schema / DB access | `gitnexus/src/core/lbug/` (`schema.ts`, `lbug-adapter.ts`), `gitnexus/src/mcp/core/lbug-adapter.ts` if MCP-specific. |
| MCP protocol, tools, resources | `gitnexus/src/mcp/server.ts`, `tools.ts`, `resources.ts`. |
| Search ranking | `gitnexus/src/core/search/` (BM25, hybrid fusion). |
| Embeddings | `gitnexus/src/core/embeddings/`, phases in `analyze.ts`. |
| Wiki generation | `gitnexus/src/core/wiki/`. |
| Web UI behavior | `gitnexus-web/src/` (components, workers, graph client). |
| CI | `.github/workflows/*.yml`, `.github/actions/setup-gitnexus/`. |
## Related docs
- [RUNBOOK.md](RUNBOOK.md) — operational commands and recovery.
- [GUARDRAILS.md](GUARDRAILS.md) — safety boundaries for humans and agents.
- [TESTING.md](TESTING.md) — how to run tests.
- `AGENTS.md` / `CLAUDE.md` — agent workflows and tool usage expectations for **this** repo when indexed by GitNexus.

113
CLAUDE.md
View file

@ -1,101 +1,52 @@
<!-- gitnexus:start -->
# GitNexus — Code Intelligence
<!-- version: 1.2.0 -->
<!--
Metadata: version, last reviewed, scope, model policy, reference docs, changelog.
Last updated: 2026-03-22
-->
This project is indexed by GitNexus as **GitNexus** (2298 symbols, 5501 relationships, 175 execution flows). Use the GitNexus MCP tools to understand code, assess impact, and navigate safely.
Last reviewed: 2026-03-24
> If any GitNexus tool warns the index is stale, run `npx gitnexus analyze` in terminal first.
**Project:** GitNexus · **Environment:** dev · **Maintainer:** repository maintainers (see GitHub)
## Always Do
Follow **AGENTS.md** for the canonical rules; this file adds Claude Code–specific deltas. Cursor-specific notes live only in `AGENTS.md`.
- **MUST run impact analysis before editing any symbol.** Before modifying a function, class, or method, run `gitnexus_impact({target: "symbolName", direction: "upstream"})` and report the blast radius (direct callers, affected processes, risk level) to the user.
- **MUST run `gitnexus_detect_changes()` before committing** to verify your changes only affect expected symbols and execution flows.
- **MUST warn the user** if impact analysis returns HIGH or CRITICAL risk before proceeding with edits.
- When exploring unfamiliar code, use `gitnexus_query({query: "concept"})` to find execution flows instead of grepping. It returns process-grouped results ranked by relevance.
- When you need full context on a specific symbol — callers, callees, which execution flows it participates in — use `gitnexus_context({name: "symbolName"})`.
## Scope
## When Debugging
See the **Scope** table in [AGENTS.md](AGENTS.md) for read/write/execute/off-limits boundaries. Cursor-specific workflow notes also live only in AGENTS.md.
1. `gitnexus_query({query: "<error or symptom>"})` — find execution flows related to the issue
2. `gitnexus_context({name: "<suspect function>"})` — see all callers, callees, and process participation
3. `READ gitnexus://repo/GitNexus/process/{processName}` — trace the full execution flow step by step
4. For regressions: `gitnexus_detect_changes({scope: "compare", base_ref: "main"})` — see what your branch changed
## Model Configuration
## When Refactoring
- **Primary:** Pin per **Claude Code** / Anthropic org policy (explicit model id). Do not rely on an unversioned `latest` alias for governed workflows.
- **Fallback:** As configured in Claude Code (organization default or user override).
- **Notes:** The GitNexus CLI analyzer does not call an LLM.
- **Renaming**: MUST use `gitnexus_rename({symbol_name: "old", new_name: "new", dry_run: true})` first. Review the preview — graph edits are safe, text_search edits need manual review. Then run with `dry_run: false`.
- **Extracting/Splitting**: MUST run `gitnexus_context({name: "target"})` to see all incoming/outgoing refs, then `gitnexus_impact({target: "target", direction: "upstream"})` to find all external callers before moving code.
- After any refactor: run `gitnexus_detect_changes({scope: "all"})` to verify only expected files changed.
## Execution Sequence (complex tasks)
## Never Do
Same discipline as [AGENTS.md](AGENTS.md): before large multi-step work, state which **AGENTS.md** / **GUARDRAILS.md** rules apply, current **Scope**, and planned validation commands (`npm test`, `tsc`, etc.). When pausing, summarize progress in the chat or a **local** scratch file (do not add `HANDOFF.md` to the repo), then `/clear` and resume with that summary.
- NEVER edit a function, class, or method without first running `gitnexus_impact` on it.
- NEVER ignore HIGH or CRITICAL risk warnings from impact analysis.
- NEVER rename symbols with find-and-replace — use `gitnexus_rename` which understands the call graph.
- NEVER commit changes without running `gitnexus_detect_changes()` to check affected scope.
## Claude Code hooks
## Tools Quick Reference
Prefer **PreToolUse** hooks for hard gates (e.g. tests before `git_commit`). Adapt hook commands to `gitnexus/` npm scripts.
| Tool | When to use | Command |
|------|-------------|---------|
| `query` | Find code by concept | `gitnexus_query({query: "auth validation"})` |
| `context` | 360-degree view of one symbol | `gitnexus_context({name: "validateUser"})` |
| `impact` | Blast radius before editing | `gitnexus_impact({target: "X", direction: "upstream"})` |
| `detect_changes` | Pre-commit scope check | `gitnexus_detect_changes({scope: "staged"})` |
| `rename` | Safe multi-file rename | `gitnexus_rename({symbol_name: "old", new_name: "new", dry_run: true})` |
| `cypher` | Custom graph queries | `gitnexus_cypher({query: "MATCH ..."})` |
## Context budget
## Impact Risk Levels
If always-on instructions grow, load deep conventions via conditional reads (e.g. *“When writing new code, read STANDARDS.md”*) instead of pasting long blocks here. In Cursor, prefer `.cursor/index.mdc` plus optional `.cursor/rules/*.mdc` globs (see [AGENTS.md](AGENTS.md) § Context budget).
| Depth | Meaning | Action |
|-------|---------|--------|
| d=1 | WILL BREAK — direct callers/importers | MUST update these |
| d=2 | LIKELY AFFECTED — indirect deps | Should test |
| d=3 | MAY NEED TESTING — transitive | Test if critical path |
## Reference Documentation
## Resources
- **This repository:** [AGENTS.md](AGENTS.md) (Cursor + monorepo notes), [ARCHITECTURE.md](ARCHITECTURE.md), [CONTRIBUTING.md](CONTRIBUTING.md), [GUARDRAILS.md](GUARDRAILS.md).
- **GitNexus:** `.claude/skills/gitnexus/`; MCP and indexed-repo rules live only in [AGENTS.md](AGENTS.md) (`gitnexus:start` … `gitnexus:end`). See **GitNexus rules** below.
| Resource | Use for |
|----------|---------|
| `gitnexus://repo/GitNexus/context` | Codebase overview, check index freshness |
| `gitnexus://repo/GitNexus/clusters` | All functional areas |
| `gitnexus://repo/GitNexus/processes` | All execution flows |
| `gitnexus://repo/GitNexus/process/{name}` | Step-by-step execution trace |
## Changelog
## Self-Check Before Finishing
| Date | Version | Change |
|------|---------|--------|
| 2026-03-24 | 1.2.0 | Removed duplicated gitnexus:start block and scope table; replaced with pointers to AGENTS.md. |
| 2026-03-23 | 1.1.0 | Updated agent instructions to match AGENTS.md. |
| 2026-03-22 | 1.0.0 | Added structured header and changelog. |
Before completing any code modification task, verify:
1. `gitnexus_impact` was run for all modified symbols
2. No HIGH/CRITICAL risk warnings were ignored
3. `gitnexus_detect_changes()` confirms changes match expected scope
4. All d=1 (WILL BREAK) dependents were updated
---
## Keeping the Index Fresh
## GitNexus rules
After committing code changes, the GitNexus index becomes stale. Re-run analyze to update it:
```bash
npx gitnexus analyze
```
If the index previously included embeddings, preserve them by adding `--embeddings`:
```bash
npx gitnexus analyze --embeddings
```
To check whether embeddings exist, inspect `.gitnexus/meta.json` — the `stats.embeddings` field shows the count (0 means no embeddings). **Running analyze without `--embeddings` will delete any previously generated embeddings.**
> Claude Code users: A PostToolUse hook handles this automatically after `git commit` and `git merge`.
## CLI
| Task | Read this skill file |
|------|---------------------|
| Understand architecture / "How does X work?" | `.claude/skills/gitnexus/gitnexus-exploring/SKILL.md` |
| Blast radius / "What breaks if I change X?" | `.claude/skills/gitnexus/gitnexus-impact-analysis/SKILL.md` |
| Trace bugs / "Why is X failing?" | `.claude/skills/gitnexus/gitnexus-debugging/SKILL.md` |
| Rename / extract / split / refactor | `.claude/skills/gitnexus/gitnexus-refactoring/SKILL.md` |
| Tools, resources, schema reference | `.claude/skills/gitnexus/gitnexus-guide/SKILL.md` |
| Index, status, clean, wiki CLI commands | `.claude/skills/gitnexus/gitnexus-cli/SKILL.md` |
<!-- gitnexus:end -->
GitNexus MCP rules are in the `<!-- gitnexus:start -->` … `<!-- gitnexus:end -->` block in **[AGENTS.md](AGENTS.md)** — load that section when working with MCP tools or the graph index.

50
CONTRIBUTING.md Normal file
View file

@ -0,0 +1,50 @@
# Contributing to GitNexus
How to propose changes, run checks locally, and open pull requests.
## License
This project uses the [PolyForm Noncommercial License 1.0.0](https://polyformproject.org/licenses/noncommercial/1.0.0/). By contributing, you agree your contributions are licensed under the same terms unless stated otherwise.
## Where to discuss
- **Issues & feature ideas:** use [GitHub Issues](https://github.com/abhigyanpatwari/GitNexus/issues) for the upstream repo, or your fork’s tracker if you work from a fork.
- **Community:** see the Discord link in the root [README.md](README.md).
## Development setup
1. Clone the repository.
2. **CLI / MCP package:** `cd gitnexus && npm install && npm run build`
3. **Web UI (if needed):** `cd gitnexus-web && npm install`
4. Run tests as described in [TESTING.md](TESTING.md).
## Branch and pull requests
- Use short-lived branches off the default branch of the repo you are targeting.
- Prefer **conventional commits** (short prefix + description), for example:
```text
feat: add graph export option
fix: correct MCP tool schema for query
test: cover cluster merge edge case
docs: clarify analyze flags
```
- **PR title:** `[area] Short description` (e.g. `[cli] Fix index refresh race`).
- **PR description:** what changed, why, how to verify (commands), and any risk or rollback notes.
## Before you open a PR
- [ ] Tests pass for the packages you touched (`gitnexus` and/or `gitnexus-web`).
- [ ] Typecheck passes: `npx tsc --noEmit` in `gitnexus/` and `npx tsc -b --noEmit` in `gitnexus-web/`.
- [ ] No secrets, tokens, or machine-specific paths committed.
- [ ] Documentation updated if behavior or public CLI/MCP contract changes.
- [ ] Pre-commit hook runs clean (`.husky/pre-commit` — typecheck + unit tests for staged packages).
## Code review
Maintainers may request changes for correctness, tests, performance, or consistency with existing patterns. Keeping diffs focused makes review faster.
## AI-assisted contributions
If you use coding agents, follow project context files (e.g. `AGENTS.md`, `CLAUDE.md`) and avoid drive-by refactors unrelated to the issue. Prefer incremental, test-backed changes.

88
GUARDRAILS.md Normal file
View file

@ -0,0 +1,88 @@
# Guardrails — GitNexus (repo + agents)
Rules for **human contributors** and **AI agents** working on this codebase or publishing artifacts. These complement `AGENTS.md` / `CLAUDE.md` (which focus on GitNexus-in-GitNexus workflows).
## Scope (typical agent session)
When automating changes in this repository, treat scope as **least privilege**:
- **Read:** Source, tests, docs, public config as needed for the task.
- **Write:** Only files required for the requested fix or feature; avoid unrelated formatting or refactors.
- **Execute:** Tests, typecheck, and documented CLI commands; do not run destructive commands on user data outside the repo without explicit approval.
- **Off-limits:** Other people’s machines, production deployments you don’t own, and credentials you didn’t receive permission to use.
Adjust explicitly if the maintainer defines a different scope for a task.
---
## Non-negotiables
1. **Never commit secrets** — API keys, tokens, `.env` with real values, private URLs, or session cookies. Use `.env.example` with placeholders only.
2. **Never rename symbols with blind find-and-replace** when working in a GitNexus-indexed project — use the **`rename` MCP tool** with **`dry_run: true` first**, then review `graph` vs `text_search` edits. (There is no separate `gitnexus rename` CLI; renaming goes through MCP or editor integration.)
3. **Run impact analysis before editing shared symbols** — use **`impact`** (upstream) for functions/classes/methods others call; do not ignore **HIGH** / **CRITICAL** risk without maintainer sign-off.
4. **Prefer `detect_changes` before commit** — confirm diffs map to expected symbols/processes when the graph is available.
5. **Preserve embeddings** — if `.gitnexus/meta.json` shows embeddings, run `npx gitnexus analyze --embeddings` when refreshing the index; plain `analyze` can drop them.
---
## Signs (recurring failure patterns)
Use this format: **Trigger → Instruction → Reason**.
Append new Signs here when the same mistake repeats (e.g. CI broken twice the same way).
### Sign: Stale graph after edits
- **Trigger:** MCP or resources warn the index is behind `HEAD`, or code search doesn’t match latest commit.
- **Instruction:** Run `npx gitnexus analyze` from the repo root (plus `--embeddings` if the project used them).
- **Reason:** Tools query LadybugDB built at last analyze; git changes are invisible until re-indexed.
### Sign: Embeddings vanished after analyze
- **Trigger:** Semantic search quality drops; `stats.embeddings` in `.gitnexus/meta.json` is 0 after a refresh.
- **Instruction:** Re-run `npx gitnexus analyze --embeddings` and confirm `meta.json` reflects stored embeddings.
- **Reason:** Embedding generation is opt-in; analyze without the flag does not preserve prior vectors.
### Sign: MCP lists no repos
- **Trigger:** MCP stderr says no indexed repos.
- **Instruction:** Run `npx gitnexus analyze` in the target repository; verify `npx gitnexus list` shows it.
- **Reason:** The MCP server discovers repos via `~/.gitnexus/registry.json`, populated by analyze.
### Sign: Wrong repo in multi-repo setups
- **Trigger:** Query/impact results clearly belong to another project.
- **Instruction:** Call `list_repos`, then pass **`repo`** on subsequent tools (or use per-workspace MCP config).
- **Reason:** Default target may be ambiguous when multiple repos are registered.
### Sign: LadybugDB lock / “database busy”
- **Trigger:** Errors opening `.gitnexus/lbug` while MCP and analyze both run.
- **Instruction:** Stop overlapping processes; one writer at a time. Retry analyze or restart MCP.
- **Reason:** Embedded DB expects single-process ownership of the store.
---
## Publishing & supply chain
- **npm:** Do not publish from unreviewed automation; follow maintainer release process. Bump version intentionally; tag releases to match `package.json`.
- **Dependencies:** Prefer minimal, auditable changes to `package.json`; run tests and CI after lockfile updates.
- **License:** This project ships under **PolyForm Noncommercial 1.0.0** — do not relicense or imply a different license in docs or metadata without maintainer approval.
---
## Escalation
Stop and ask a **human maintainer** when:
- Impact analysis shows **HIGH** / **CRITICAL** risk and the task still requires the change.
- You need to alter **CI**, **release**, or **security-sensitive** config.
- Requirements conflict (e.g. “speed up analyze” vs “must keep all embeddings on huge repo”).
- You are unsure whether data loss is acceptable (`clean`, forced migrations, schema changes).
---
## Related docs
- [ARCHITECTURE.md](ARCHITECTURE.md) — components and data flow.
- [RUNBOOK.md](RUNBOOK.md) — commands for recovery.
- [CONTRIBUTING.md](CONTRIBUTING.md) — PR and commit expectations.

View file

@ -59,6 +59,14 @@ https://github.com/user-attachments/assets/172685ba-8e54-4ea7-9ad1-e31a3398da72
---
## Development
- [ARCHITECTURE.md](ARCHITECTURE.md) — packages, index → graph → MCP flow, where to change code
- [RUNBOOK.md](RUNBOOK.md) — analyze, embeddings, stale index, MCP recovery, CI snippets
- [GUARDRAILS.md](GUARDRAILS.md) — safety rules and operational “Signs” for contributors and agents
- [CONTRIBUTING.md](CONTRIBUTING.md) — license, setup, commits, and pull requests
- [TESTING.md](TESTING.md) — test commands for `gitnexus` and `gitnexus-web`
## CLI + MCP (recommended)
The CLI indexes your repository and runs an MCP server that gives AI agents deep codebase awareness.

163
RUNBOOK.md Normal file
View file

@ -0,0 +1,163 @@
# Runbook — GitNexus
Short, copy-paste operations for **local development**, **MCP**, and **CI**. Commands assume a Unix shell; on Windows use Git Bash or equivalent paths.
## Prerequisites
- **Node.js** ≥ 20 (`gitnexus-web/package.json` `engines`).
- **Git** (analyze requires a git repository).
- From repo root, install and build the CLI package:
```bash
cd gitnexus
npm install
npm run build
```
Use `npx gitnexus …` from any path after global/published install, or `node dist/cli/index.js …` when developing from `gitnexus/` with a local build.
---
## Index out of date / “stale” tools
**Symptom:** MCP or resources warn the index is behind `HEAD`, or results don’t reflect recent commits.
**Fix (from the target repo root):**
```bash
npx gitnexus analyze
```
**Force full rebuild** (same commit but suspect corruption or changed ignore rules):
```bash
npx gitnexus analyze --force
```
**Check status:**
```bash
npx gitnexus status
```
**List what MCP knows about:**
```bash
npx gitnexus list
```
---
## Embeddings
**First time with vectors** (slower, more disk/RAM):
```bash
npx gitnexus analyze --embeddings
```
**Important:** If you already had embeddings, **always** pass `--embeddings` on later analyzes, or they can be dropped. See `stats.embeddings` in `.gitnexus/meta.json` (0 means none).
**Large repos:** Analyze may skip or limit embedding work when node counts are very high; watch CLI output.
---
## MCP: no repos / empty tools
**Symptom:** `GitNexus: No indexed repos yet` on stderr when starting MCP.
**Fix:** In each project you want indexed:
```bash
cd /path/to/repo
npx gitnexus analyze
```
Restart the editor MCP session if needed. The server **refreshes the registry lazily**; new analyzes are picked up without necessarily reinstalling MCP.
**Symptom:** Wrong repo when multiple are indexed — pass `repo` on tools or use `list_repos` first.
---
## Clean slate (corrupt or huge `.gitnexus`)
**Current repo only** (prompts for confirmation):
```bash
npx gitnexus clean
```
**Skip confirmation:**
```bash
npx gitnexus clean --force
```
**All registered repos:**
```bash
npx gitnexus clean --all --force
```
Then re-run `npx gitnexus analyze` (and `--embeddings` if you need vectors).
---
## Local bridge for the web UI
```bash
cd gitnexus
npx gitnexus serve
# default http://127.0.0.1:4747 — see serve --help for port/host
```
Use when the browser UI should talk to **local** indexed repos instead of WASM-only mode.
---
## CLI equivalents of MCP tools
Useful for debugging without an editor:
```bash
cd gitnexus
npx gitnexus query "authentication flow" --repo MyRepo
npx gitnexus context SomeSymbol --repo MyRepo
npx gitnexus impact SomeSymbol --direction upstream --repo MyRepo
npx gitnexus cypher "MATCH (n) RETURN count(n) LIMIT 1" --repo MyRepo
```
---
## CI failures (contributors)
Orchestrator: `.github/workflows/ci.yml`.
| Job | Typical local repro |
|-----|---------------------|
| **quality** | `cd gitnexus && npx tsc --noEmit` |
| **unit-tests** | `cd gitnexus && npx vitest run test/unit` |
| **integration** | `cd gitnexus && npx vitest run test/integration` (see workflow matrix for groups) |
| **e2e** | Triggered when `gitnexus-web/` changes; `cd gitnexus-web && E2E=1 npx playwright test` (requires `gitnexus serve` + `npm run dev`) |
**Note:** Pushes that touch only certain markdown paths may be skipped by `paths-ignore` in CI — see workflow file for exact patterns.
---
## Memory / analyze crashes
Analyze re-execs Node with a **large old-space heap** when needed (`analyze.ts`). If you still OOM on huge repos, close other processes, avoid `--embeddings` for a first pass, or analyze a smaller path if supported by your workflow.
---
## LadybugDB / lock errors
Only one process should open a repo’s `.gitnexus/lbug` store at a time. If MCP and a second `analyze` run conflict, stop one process, then retry `analyze` or restart MCP.
---
## Where to dig deeper
- Architecture overview: [ARCHITECTURE.md](ARCHITECTURE.md)
- Agent safety rules: [GUARDRAILS.md](GUARDRAILS.md)
- Tests: [TESTING.md](TESTING.md)

95
TESTING.md Normal file
View file

@ -0,0 +1,95 @@
# Testing — GitNexus
How we structure tests and which commands to run locally and in CI.
## Packages
| Package | Path | Runner | Notes |
| -------------- | -------------- | -------- | ------------------------------ |
| CLI + MCP core | `gitnexus/` | Vitest | Primary test surface in CI |
| Web UI | `gitnexus-web/`| Vitest | Unit/component tests |
| Web UI E2E | `gitnexus-web/`| Playwright | Run when changing UI flows |
## Commands (local)
From repository root, unless noted:
**`gitnexus` (CLI / library)**
```bash
cd gitnexus
npm install
npm run build
npm test # unit: vitest run test/unit
npm run test:integration # integration suite
npm run test:all
npm run test:coverage
npx tsc --noEmit # typecheck (matches CI)
```
**`gitnexus-web`**
```bash
cd gitnexus-web
npm install
npm test # unit tests (vitest)
npx tsc -b --noEmit # typecheck (matches CI)
npm run test:coverage
npm run test:e2e # Playwright (requires gitnexus serve + npm run dev)
```
## Pre-commit hook
A husky pre-commit hook (`.husky/pre-commit`) runs automatically on every `git commit`:
- **`gitnexus-web/` files staged** → `tsc -b --noEmit` + `vitest run`
- **`gitnexus/` files staged** → `tsc --noEmit` + `vitest run --project default`
Skip with `git commit --no-verify` (use sparingly).
## Test categories
- **Unit** — Pure logic, parsers, graph/query helpers; fast; no network.
- **Integration** — Real combinations (filesystem, MCP wiring, larger pipelines) as already organized under `gitnexus/test/integration`.
- **Eval-style / golden sets** — For agent- or classification-style behavior, keep labeled inputs and expected outputs (JSON or table-driven tests) and run them in CI when relevant.
- **E2E (web)** — Critical user paths only; prefer `data-testid` attributes for stable selectors. Tests run against real backend (`gitnexus serve`) and Vite dev server.
## Performance metrics (targets)
Set targets to match team expectations, then tune to this repo’s CI reality:
| Metric | Target (initial) | Notes |
| ------------------- | ---------------- | ------------------------------------------ |
| Unit coverage | Align with CI | CI runs Vitest with coverage in `gitnexus` |
| Unit wall time | Fast PR feedback | Use `vitest run test/unit` for tight loop |
| Integration duration| &lt; few minutes | Guard heavy tests with env flags if needed |
## Regression testing
Re-run the full relevant suite when:
- Prompt or agent-behavior documentation changes (if tests encode behavior)
- Model or embedding-related code paths change
- Graph schema, query contracts, or MCP tool shapes change
- Dependencies with parsing or runtime impact upgrade
## CI integration
GitHub Actions (`.github/workflows/ci.yml`) orchestrate:
- **`ci-quality.yml`** — `tsc --noEmit` for `gitnexus/` + `tsc -b --noEmit` for `gitnexus-web/`
- **`ci-tests.yml`** — `vitest run` with coverage (ubuntu) + cross-platform (macOS, Windows)
- **`ci-e2e.yml`** — Playwright E2E tests, gated on `gitnexus-web/**` changes
Local checks before pushing:
```bash
cd gitnexus && npx tsc --noEmit && npm test
cd ../gitnexus-web && npx tsc -b --noEmit && npm test
```
Or rely on the pre-commit hook which runs these automatically for staged files.
## User acceptance / beta (optional)
For staged releases or UI betas: deploy to a staging environment, collect structured feedback, watch errors and latency, then iterate before a wider release.

View file

@ -50,6 +50,10 @@ All models are routed through **OpenRouter** by default, so a single `OPENROUTER
docker pull swebench/sweb.eval.x86_64.django_1776_django-16527:latest
```
### Debug logging
Set `GITNEXUS_EVAL_DEBUG=1` to include full Python tracebacks in run summaries and logs. By default, errors are sanitized to avoid leaking host paths or stack traces.
## Quick Start
### Debug a single instance

View file

@ -22,8 +22,10 @@ import time
from enum import Enum
from pathlib import Path
from constants import AUGMENT_TIMEOUT_SECONDS
from minisweagent import Environment, Model
from minisweagent.agents.default import AgentConfig, DefaultAgent
from tool_registry import BINARIES_BY_KEY, TOOL_METRIC_KEYS
logger = logging.getLogger("gitnexus_agent")
@ -40,7 +42,7 @@ class GitNexusMode(str, Enum):
class GitNexusAgentConfig(AgentConfig):
"""Extended config for GitNexus evaluation agent."""
gitnexus_mode: GitNexusMode = GitNexusMode.BASELINE
augment_timeout: float = 5.0
augment_timeout: float = AUGMENT_TIMEOUT_SECONDS
augment_min_pattern_length: int = 3
track_gitnexus_usage: bool = True
@ -152,16 +154,10 @@ class GitNexusAgent(DefaultAgent):
"""Track which GitNexus tools the agent uses."""
for action in message.get("extra", {}).get("actions", []):
command = action.get("command", "")
if "gitnexus-query" in command:
self.gitnexus_metrics.tool_calls["query"] += 1
elif "gitnexus-context" in command:
self.gitnexus_metrics.tool_calls["context"] += 1
elif "gitnexus-impact" in command:
self.gitnexus_metrics.tool_calls["impact"] += 1
elif "gitnexus-cypher" in command:
self.gitnexus_metrics.tool_calls["cypher"] += 1
elif "gitnexus-overview" in command:
self.gitnexus_metrics.tool_calls["overview"] += 1
for key, binary in BINARIES_BY_KEY.items():
if binary in command and key in self.gitnexus_metrics.tool_calls:
self.gitnexus_metrics.tool_calls[key] += 1
break
def serialize(self, *extra_dicts) -> dict:
"""Serialize with GitNexus-specific metrics."""
@ -180,13 +176,7 @@ class GitNexusMetrics:
"""Tracks GitNexus-specific metrics during evaluation."""
def __init__(self):
self.tool_calls: dict[str, int] = {
"query": 0,
"context": 0,
"impact": 0,
"cypher": 0,
"overview": 0,
}
self.tool_calls: dict[str, int] = {key: 0 for key in TOOL_METRIC_KEYS}
self.augmentation_calls: int = 0
self.augmentation_hits: int = 0
self.augmentation_errors: int = 0

View file

@ -27,6 +27,8 @@ import typer
from rich.console import Console
from rich.table import Table
from tool_registry import TOOL_METRIC_KEYS
logger = logging.getLogger("analyze_results")
console = Console()
app = typer.Typer(rich_markup_mode="rich", add_completion=False)
@ -77,13 +79,20 @@ def load_run_results(results_dir: Path) -> dict[str, dict]:
def parse_run_id(run_id: str) -> tuple[str, str]:
"""Parse 'model_mode' into (model, mode)."""
# Handle multi-word model names like 'minimax-2.5'
# Modes are: baseline, mcp, augment, full
known_modes = {"baseline", "mcp", "augment", "full"}
parts = run_id.rsplit("_", 1)
if len(parts) == 2 and parts[1] in known_modes:
return parts[0], parts[1]
"""Parse 'model_mode' into (model, mode) using known suffixes."""
# Match the longest known suffix first to avoid hyphen collisions in model names.
known_modes = [
"native_augment",
"native",
"baseline",
"mcp",
"augment",
"full",
]
for mode in known_modes:
suffix = f"_{mode}"
if run_id.endswith(suffix):
return run_id[: -len(suffix)], mode
return run_id, "unknown"
@ -266,12 +275,17 @@ def compare_modes(
_, mode = parse_run_id(run_id)
metrics[mode] = compute_metrics(run_data)
mode_order = [
mode
for mode in ["baseline", "native", "native_augment", "mcp", "augment", "full"]
if mode in metrics
] or sorted(metrics.keys())
# Print comparison table
table = Table(title=f"Mode Comparison: {model}")
table.add_column("Metric", style="bold")
for mode in ["baseline", "mcp", "augment", "full"]:
if mode in metrics:
table.add_column(mode, justify="right")
for mode in mode_order:
table.add_column(mode, justify="right")
rows = [
("Instances", "n_instances", "d"),
@ -288,7 +302,7 @@ def compare_modes(
for label, key, fmt in rows:
values = []
for mode in ["baseline", "mcp", "augment", "full"]:
for mode in mode_order:
if mode in metrics:
v = metrics[mode].get(key, 0)
if fmt == ".1%":
@ -307,8 +321,8 @@ def compare_modes(
baseline_calls = metrics["baseline"]["avg_api_calls"]
table.add_section()
for mode in ["mcp", "augment", "full"]:
if mode not in metrics:
for mode in mode_order:
if mode == "baseline":
continue
mode_cost = metrics[mode]["avg_cost"]
mode_calls = metrics[mode]["avg_api_calls"]
@ -340,10 +354,8 @@ def gitnexus_usage(
table = Table(title="Tool Usage by Run")
table.add_column("Run", style="bold")
table.add_column("query", justify="right")
table.add_column("context", justify="right")
table.add_column("impact", justify="right")
table.add_column("cypher", justify="right")
for key in TOOL_METRIC_KEYS:
table.add_column(key, justify="right")
table.add_column("Total", justify="right")
table.add_column("Augment Hits", justify="right")
@ -353,7 +365,7 @@ def gitnexus_usage(
continue
# Aggregate tool calls across trajectories
tool_totals: dict[str, int] = {"query": 0, "context": 0, "impact": 0, "cypher": 0, "overview": 0}
tool_totals: dict[str, int] = {key: 0 for key in TOOL_METRIC_KEYS}
augment_hits = 0
for traj in run_data.get("trajectories", {}).values():
@ -373,10 +385,7 @@ def gitnexus_usage(
if total > 0 or augment_hits > 0:
table.add_row(
run_id,
str(tool_totals.get("query", 0)),
str(tool_totals.get("context", 0)),
str(tool_totals.get("impact", 0)),
str(tool_totals.get("cypher", 0)),
*[str(tool_totals.get(key, 0)) for key in TOOL_METRIC_KEYS],
str(total),
str(augment_hits),
)

View file

@ -17,6 +17,14 @@ import time
from pathlib import Path
from typing import Any
from constants import (
MCP_FIND_GITNEXUS_FALLBACK_TIMEOUT_SECONDS,
MCP_FIND_GITNEXUS_TIMEOUT_SECONDS,
MCP_READ_TIMEOUT_SECONDS,
MCP_STOP_WAIT_SECONDS,
)
from utils.errors import is_debug_enabled, log_safe_exception
logger = logging.getLogger("mcp_bridge")
@ -78,7 +86,7 @@ class MCPBridge:
return True
except Exception as e:
logger.error(f"Failed to start MCP bridge: {e}")
log_safe_exception(logger, "Failed to start MCP bridge", e, include_debug=is_debug_enabled())
self.stop()
return False
@ -86,9 +94,14 @@ class MCPBridge:
"""Stop the MCP server subprocess."""
if self.process:
try:
self.process.stdin.close()
if self.process.stdin:
self.process.stdin.close()
if self.process.stdout:
self.process.stdout.close()
if self.process.stderr:
self.process.stderr.close()
self.process.terminate()
self.process.wait(timeout=5)
self.process.wait(timeout=MCP_STOP_WAIT_SECONDS)
except Exception:
try:
self.process.kill()
@ -146,7 +159,9 @@ class MCPBridge:
try:
result = subprocess.run(
[cmd, "gitnexus", "--version"],
capture_output=True, text=True, timeout=15,
capture_output=True,
text=True,
timeout=MCP_FIND_GITNEXUS_TIMEOUT_SECONDS,
cwd=self.repo_path,
)
if result.returncode == 0:
@ -158,7 +173,9 @@ class MCPBridge:
try:
result = subprocess.run(
["gitnexus", "--version"],
capture_output=True, text=True, timeout=10,
capture_output=True,
text=True,
timeout=MCP_FIND_GITNEXUS_FALLBACK_TIMEOUT_SECONDS,
)
if result.returncode == 0:
return "gitnexus"
@ -194,7 +211,7 @@ class MCPBridge:
self.process.stdin.flush()
# Read response
response = self._read_response(timeout=30)
response = self._read_response(timeout=MCP_READ_TIMEOUT_SECONDS)
if response and response.get("id") == request_id:
if "error" in response:
logger.error(f"MCP error: {response['error']}")
@ -203,7 +220,7 @@ class MCPBridge:
return None
except Exception as e:
logger.error(f"MCP request failed: {e}")
log_safe_exception(logger, "MCP request failed", e, include_debug=is_debug_enabled())
return None
def _send_notification(self, method: str, params: dict):
@ -224,42 +241,67 @@ class MCPBridge:
self.process.stdin.write(message.encode("utf-8"))
self.process.stdin.flush()
except Exception as e:
logger.error(f"MCP notification failed: {e}")
log_safe_exception(logger, "MCP notification failed", e, include_debug=is_debug_enabled())
def _read_response(self, timeout: float = 30) -> dict | None:
def _read_content_length(self, deadline: float) -> int | None:
"""Read Content-Length header, returning the byte length or None."""
if not self.process or not self.process.stdout:
return None
header_line = b""
while time.time() < deadline:
byte = self.process.stdout.read(1)
if not byte:
return None
header_line += byte
if header_line.endswith(b"\r\n\r\n") or header_line.endswith(b"\n\n"):
break
if not header_line:
return None
header_str = header_line.decode("utf-8").strip()
for line in header_str.split("\r\n"):
if line.lower().startswith("content-length:"):
try:
return int(line.split(":", 1)[1].strip())
except (ValueError, IndexError):
return None
return None
def _read_body(self, content_length: int, deadline: float) -> bytes | None:
"""Read a response body of the expected length before deadline."""
if not self.process or not self.process.stdout:
return None
remaining = content_length
chunks: list[bytes] = []
while remaining > 0 and time.time() < deadline:
chunk = self.process.stdout.read(remaining)
if not chunk:
return None
chunks.append(chunk)
remaining -= len(chunk)
if remaining > 0:
return None
return b"".join(chunks)
def _read_response(self, timeout: float = MCP_READ_TIMEOUT_SECONDS) -> dict | None:
"""Read a JSON-RPC response from the MCP server."""
if not self.process or not self.process.stdout:
return None
start = time.time()
try:
while time.time() - start < timeout:
# Read Content-Length header
header_line = b""
while True:
byte = self.process.stdout.read(1)
if not byte:
return None
header_line += byte
if header_line.endswith(b"\r\n\r\n"):
break
if header_line.endswith(b"\n\n"):
break
# Parse content length
header_str = header_line.decode("utf-8").strip()
content_length = None
for line in header_str.split("\r\n"):
if line.lower().startswith("content-length:"):
content_length = int(line.split(":")[1].strip())
break
deadline = time.time() + timeout
while time.time() < deadline:
content_length = self._read_content_length(deadline)
if content_length is None:
continue
# Read body
body = self.process.stdout.read(content_length)
body = self._read_body(content_length, deadline)
if not body:
return None
@ -272,7 +314,7 @@ class MCPBridge:
return None
except Exception as e:
logger.error(f"Error reading MCP response: {e}")
log_safe_exception(logger, "Error reading MCP response", e, include_debug=is_debug_enabled())
return None

15
eval/constants.py Normal file
View file

@ -0,0 +1,15 @@
DEBUG_ENV_VAR = "GITNEXUS_EVAL_DEBUG"
# GitNexus eval-server health checks
EVAL_SERVER_HEALTH_RETRIES = 30
EVAL_SERVER_HEALTH_INTERVAL_SECONDS = 0.5
EVAL_SERVER_HEALTH_TIMEOUT_SECONDS = 3
# MCP bridge timeouts
MCP_FIND_GITNEXUS_TIMEOUT_SECONDS = 15
MCP_FIND_GITNEXUS_FALLBACK_TIMEOUT_SECONDS = 10
MCP_READ_TIMEOUT_SECONDS = 30
MCP_STOP_WAIT_SECONDS = 5
# Agent defaults
AUGMENT_TIMEOUT_SECONDS = 5.0

View file

@ -26,72 +26,20 @@ import shutil
import time
from pathlib import Path
from constants import (
EVAL_SERVER_HEALTH_INTERVAL_SECONDS,
EVAL_SERVER_HEALTH_RETRIES,
EVAL_SERVER_HEALTH_TIMEOUT_SECONDS,
)
from minisweagent.environments.docker import DockerEnvironment
from tool_registry import TOOL_SPECS, ToolScriptSpec
from utils.errors import is_debug_enabled, log_safe_exception
logger = logging.getLogger("gitnexus_docker")
DEFAULT_CACHE_DIR = Path.home() / ".gitnexus-eval-cache"
EVAL_SERVER_PORT = 4848
# Standalone tool scripts installed into /usr/local/bin/ inside the container.
# Each script calls the eval-server via curl, with a CLI fallback.
# These are standalone — no sourcing, no env inheritance needed.
TOOL_SCRIPT_QUERY = r'''#!/bin/bash
PORT="${GITNEXUS_EVAL_PORT:-__PORT__}"
query="$1"; task_ctx="${2:-}"; goal="${3:-}"
[ -z "$query" ] && echo "Usage: gitnexus-query <query> [task_context] [goal]" && exit 1
args="{\"query\": \"$query\""
[ -n "$task_ctx" ] && args="$args, \"task_context\": \"$task_ctx\""
[ -n "$goal" ] && args="$args, \"goal\": \"$goal\""
args="$args}"
result=$(curl -sf -X POST "http://127.0.0.1:${PORT}/tool/query" -H "Content-Type: application/json" -d "$args" 2>/dev/null)
if [ $? -eq 0 ] && [ -n "$result" ]; then echo "$result"; exit 0; fi
cd /testbed && npx gitnexus query "$query" 2>&1
'''
TOOL_SCRIPT_CONTEXT = r'''#!/bin/bash
PORT="${GITNEXUS_EVAL_PORT:-__PORT__}"
name="$1"; file_path="${2:-}"
[ -z "$name" ] && echo "Usage: gitnexus-context <symbol_name> [file_path]" && exit 1
args="{\"name\": \"$name\""
[ -n "$file_path" ] && args="$args, \"file_path\": \"$file_path\""
args="$args}"
result=$(curl -sf -X POST "http://127.0.0.1:${PORT}/tool/context" -H "Content-Type: application/json" -d "$args" 2>/dev/null)
if [ $? -eq 0 ] && [ -n "$result" ]; then echo "$result"; exit 0; fi
cd /testbed && npx gitnexus context "$name" 2>&1
'''
TOOL_SCRIPT_IMPACT = r'''#!/bin/bash
PORT="${GITNEXUS_EVAL_PORT:-__PORT__}"
target="$1"; direction="${2:-upstream}"
[ -z "$target" ] && echo "Usage: gitnexus-impact <symbol_name> [upstream|downstream]" && exit 1
result=$(curl -sf -X POST "http://127.0.0.1:${PORT}/tool/impact" -H "Content-Type: application/json" -d "{\"target\": \"$target\", \"direction\": \"$direction\"}" 2>/dev/null)
if [ $? -eq 0 ] && [ -n "$result" ]; then echo "$result"; exit 0; fi
cd /testbed && npx gitnexus impact "$target" --direction "$direction" 2>&1
'''
TOOL_SCRIPT_CYPHER = r'''#!/bin/bash
PORT="${GITNEXUS_EVAL_PORT:-__PORT__}"
query="$1"
[ -z "$query" ] && echo "Usage: gitnexus-cypher <cypher_query>" && exit 1
result=$(curl -sf -X POST "http://127.0.0.1:${PORT}/tool/cypher" -H "Content-Type: application/json" -d "{\"query\": \"$query\"}" 2>/dev/null)
if [ $? -eq 0 ] && [ -n "$result" ]; then echo "$result"; exit 0; fi
cd /testbed && npx gitnexus cypher "$query" 2>&1
'''
TOOL_SCRIPT_AUGMENT = r'''#!/bin/bash
cd /testbed && npx gitnexus augment "$1" 2>&1 || true
'''
TOOL_SCRIPT_OVERVIEW = r'''#!/bin/bash
PORT="${GITNEXUS_EVAL_PORT:-__PORT__}"
echo "=== Code Knowledge Graph Overview ==="
result=$(curl -sf -X POST "http://127.0.0.1:${PORT}/tool/list_repos" -H "Content-Type: application/json" -d "{}" 2>/dev/null)
if [ $? -eq 0 ] && [ -n "$result" ]; then echo "$result"; exit 0; fi
cd /testbed && npx gitnexus list 2>&1
'''
class GitNexusDockerEnvironment(DockerEnvironment):
"""
@ -133,7 +81,13 @@ class GitNexusDockerEnvironment(DockerEnvironment):
try:
self._setup_gitnexus()
except Exception as e:
logger.warning(f"GitNexus setup failed, continuing without it: {e}")
log_safe_exception(
logger,
"GitNexus setup failed, continuing without it",
e,
include_debug=is_debug_enabled(),
level="warning",
)
self._gitnexus_ready = False
return result
@ -222,27 +176,59 @@ class GitNexusDockerEnvironment(DockerEnvironment):
"timeout": 5,
})
# Wait for the server to be ready (up to 15s for KuzuDB init)
for i in range(30):
time.sleep(0.5)
# Wait for the server to be ready (up to ~15s for KuzuDB init)
for i in range(EVAL_SERVER_HEALTH_RETRIES):
time.sleep(EVAL_SERVER_HEALTH_INTERVAL_SECONDS)
health = self.execute({
"command": f"curl -sf http://127.0.0.1:{self.eval_server_port}/health 2>/dev/null || echo 'NOT_READY'",
"timeout": 3,
"timeout": EVAL_SERVER_HEALTH_TIMEOUT_SECONDS,
})
output = health.get("output", "").strip()
if "NOT_READY" not in output and "ok" in output:
logger.info(f"Eval-server ready after {(i + 1) * 0.5:.1f}s")
logger.info(
f"Eval-server ready after {(i + 1) * EVAL_SERVER_HEALTH_INTERVAL_SECONDS:.1f}s"
)
return
log_output = self.execute({
"command": "cat /tmp/gitnexus-eval-server.log 2>/dev/null | tail -20",
})
logger.warning(
f"Eval-server didn't become ready in 15s. "
f"Eval-server didn't become ready in "
f"{EVAL_SERVER_HEALTH_RETRIES * EVAL_SERVER_HEALTH_INTERVAL_SECONDS:.1f}s. "
f"Tools will fall back to direct CLI.\n"
f"Server log: {log_output.get('output', 'N/A')}"
)
@staticmethod
def _render_tool_script(spec: ToolScriptSpec, port: str) -> str:
"""
Render a standalone bash script for a GitNexus tool.
Scripts call the eval-server fast path when an endpoint is present,
and fall back to the CLI otherwise.
"""
lines = ["#!/bin/bash"]
if spec.endpoint:
lines.append(f'PORT="${{GITNEXUS_EVAL_PORT:-{port}}}"')
if spec.header:
lines.append(spec.header.strip())
if spec.payload_builder:
lines.append(spec.payload_builder.strip())
if spec.endpoint:
lines.append(
f'result=$(curl -sf -X POST "http://127.0.0.1:${{PORT}}{spec.endpoint}" '
'-H "Content-Type: application/json" -d "$payload" 2>/dev/null)'
)
lines.append('if [ $? -eq 0 ] && [ -n "$result" ]; then echo "$result"; exit 0; fi')
lines.append(spec.fallback.strip())
return "\n".join(lines)
def _install_tools(self):
"""
Install standalone GitNexus tool scripts in /usr/local/bin/.
@ -259,24 +245,20 @@ class GitNexusDockerEnvironment(DockerEnvironment):
"""
port = str(self.eval_server_port)
tools = {
"gitnexus-query": TOOL_SCRIPT_QUERY,
"gitnexus-context": TOOL_SCRIPT_CONTEXT,
"gitnexus-impact": TOOL_SCRIPT_IMPACT,
"gitnexus-cypher": TOOL_SCRIPT_CYPHER,
"gitnexus-augment": TOOL_SCRIPT_AUGMENT,
"gitnexus-overview": TOOL_SCRIPT_OVERVIEW,
}
for name, script in tools.items():
script_content = script.replace("__PORT__", port).strip()
for spec in TOOL_SPECS.values():
script_content = self._render_tool_script(spec, port).strip()
# Use heredoc with quoted delimiter — prevents all variable expansion and quoting issues
self.execute({
"command": f"cat << 'GITNEXUS_SCRIPT_EOF' > /usr/local/bin/{name}\n{script_content}\nGITNEXUS_SCRIPT_EOF\nchmod +x /usr/local/bin/{name}",
"command": (
f"cat << 'GITNEXUS_SCRIPT_EOF' > /usr/local/bin/{spec.bin_name}\n"
f"{script_content}\n"
"GITNEXUS_SCRIPT_EOF\n"
f"chmod +x /usr/local/bin/{spec.bin_name}"
),
"timeout": 5,
})
logger.info(f"Installed {len(tools)} GitNexus tool scripts in /usr/local/bin/")
logger.info(f"Installed {len(TOOL_SPECS)} GitNexus tool scripts in /usr/local/bin/")
def _get_repo_info(self) -> dict:
"""Get repository identity info from the container."""
@ -325,7 +307,13 @@ class GitNexusDockerEnvironment(DockerEnvironment):
logger.info(f"Cached GitNexus index: {cache_path}")
except Exception as e:
logger.warning(f"Failed to cache GitNexus index: {e}")
log_safe_exception(
logger,
"Failed to cache GitNexus index",
e,
include_debug=is_debug_enabled(),
level="warning",
)
if cache_path.exists():
shutil.rmtree(cache_path, ignore_errors=True)
@ -361,7 +349,13 @@ class GitNexusDockerEnvironment(DockerEnvironment):
logger.info("GitNexus index restored from cache")
except Exception as e:
logger.warning(f"Failed to restore cache, re-indexing: {e}")
log_safe_exception(
logger,
"Failed to restore cache, re-indexing",
e,
include_debug=is_debug_enabled(),
level="warning",
)
self._index_repository()
def stop(self) -> dict:

View file

@ -20,6 +20,8 @@ dependencies = [
dev = [
"pytest>=8.0.0",
"ruff>=0.5.0",
"hypothesis>=6.88.0",
"coverage>=7.6.0",
]
[project.scripts]
@ -31,8 +33,8 @@ requires = ["hatchling"]
build-backend = "hatchling.build"
[tool.hatch.build.targets.wheel]
packages = ["agents", "environments", "analysis", "bridge"]
extra-files = ["run_eval.py"]
packages = ["agents", "environments", "analysis", "bridge", "utils"]
extra-files = ["run_eval.py", "tool_registry.py", "constants.py"]
[tool.ruff]
line-length = 120

View file

@ -25,7 +25,6 @@ import logging
import os
import threading
import time
import traceback
from itertools import product
from pathlib import Path
from typing import Any
@ -36,6 +35,8 @@ from rich.console import Console
from rich.live import Live
from rich.table import Table
from utils.errors import is_debug_enabled, log_safe_exception
# Load .env file from eval/ directory
_env_file = Path(__file__).parent / ".env"
if _env_file.exists():
@ -138,6 +139,65 @@ def get_swebench_docker_image(instance: dict) -> str:
return image_name
def _build_model(config: dict):
"""Construct the model from config."""
from minisweagent.models import get_model
return get_model(config=config.get("model", {}))
def _build_environment(config: dict, instance: dict):
"""Construct the environment for the instance."""
env_config = dict(config.get("environment", {}))
env_class_name = env_config.pop("environment_class", "docker")
if env_class_name == "eval.environments.gitnexus_docker.GitNexusDockerEnvironment":
from environments.gitnexus_docker import GitNexusDockerEnvironment
env_config["image"] = get_swebench_docker_image(instance)
return GitNexusDockerEnvironment(**env_config)
from minisweagent.environments.docker import DockerEnvironment
return DockerEnvironment(image=get_swebench_docker_image(instance), **env_config)
def _build_agent(config: dict, model, env, instance_dir: Path, instance_id: str):
"""Construct the GitNexus agent with trajectory output configured."""
from agents.gitnexus_agent import GitNexusAgent
agent_config = dict(config.get("agent", {}))
agent_config.pop("agent_class", "eval.agents.gitnexus_agent.GitNexusAgent")
traj_path = instance_dir / f"{instance_id}.traj.json"
agent_config["output_path"] = traj_path
return GitNexusAgent(model, env, **agent_config)
def _extract_submission(env, info: dict, run_id: str) -> str:
"""Pull the git diff patch from the container, falling back to the agent submission."""
try:
patch_output = env.execute({"command": "cd /testbed && git diff"})
return patch_output.get("output", "").strip()
except Exception as patch_err:
logger.warning(f"[{run_id}] Failed to extract patch: {patch_err}")
return info.get("submission", "")
def _record_failure(run_id: str, instance_id: str, result: dict, error: Exception):
sanitized = log_safe_exception(
logger,
f"[{run_id}] Error on {instance_id}",
error,
include_debug=is_debug_enabled(),
)
result["exit_status"] = sanitized["error_type"]
result["error_type"] = sanitized["error_type"]
result["error_message"] = sanitized["error_message"]
result["error"] = sanitized["error_message"]
if "error_detail_debug" in sanitized:
result["error_detail_debug"] = sanitized["error_detail_debug"]
def process_instance(
instance: dict,
config: dict,
@ -149,8 +209,6 @@ def process_instance(
Process a single SWE-bench instance with the given config.
Returns result dict with instance_id, exit_status, submission, metrics.
"""
from minisweagent.models import get_model
instance_id = instance["instance_id"]
run_id = f"{model_name}_{mode_name}"
instance_dir = output_dir / run_id / instance_id
@ -168,31 +226,12 @@ def process_instance(
}
agent = None
env = None
try:
# Build model
model = get_model(config=config.get("model", {}))
# Build environment
env_config = dict(config.get("environment", {}))
env_class_name = env_config.pop("environment_class", "docker")
if env_class_name == "eval.environments.gitnexus_docker.GitNexusDockerEnvironment":
from environments.gitnexus_docker import GitNexusDockerEnvironment
env_config["image"] = get_swebench_docker_image(instance)
env = GitNexusDockerEnvironment(**env_config)
else:
from minisweagent.environments.docker import DockerEnvironment
env = DockerEnvironment(image=get_swebench_docker_image(instance), **env_config)
# Build agent
agent_config = dict(config.get("agent", {}))
agent_class_name = agent_config.pop("agent_class", "eval.agents.gitnexus_agent.GitNexusAgent")
from agents.gitnexus_agent import GitNexusAgent
traj_path = instance_dir / f"{instance_id}.traj.json"
agent_config["output_path"] = traj_path
agent = GitNexusAgent(model, env, **agent_config)
model = _build_model(config)
env = _build_environment(config, instance)
agent = _build_agent(config, model, env, instance_dir, instance_id)
# Run
logger.info(f"[{run_id}] Starting {instance_id}")
@ -204,18 +243,10 @@ def process_instance(
result["gitnexus_metrics"] = agent.gitnexus_metrics.to_dict()
# Extract git diff patch from the container (SWE-bench needs the model_patch)
try:
patch_output = env.execute({"command": "cd /testbed && git diff"})
result["submission"] = patch_output.get("output", "").strip()
except Exception as patch_err:
logger.warning(f"[{run_id}] Failed to extract patch: {patch_err}")
result["submission"] = info.get("submission", "")
result["submission"] = _extract_submission(env, info, run_id)
except Exception as e:
logger.error(f"[{run_id}] Error on {instance_id}: {e}")
result["exit_status"] = type(e).__name__
result["error"] = str(e)
result["traceback"] = traceback.format_exc()
_record_failure(run_id, instance_id, result, e)
finally:
if agent:
@ -287,7 +318,12 @@ def run_configuration(
results.append(future.result())
except Exception as e:
iid = futures[future]
logger.error(f"[{run_id}] Uncaught error for {iid}: {e}")
log_safe_exception(
logger,
f"[{run_id}] Uncaught error for {iid}",
e,
include_debug=is_debug_enabled(),
)
# Save run summary
summary = {

1
eval/tests/__init__.py Normal file
View file

@ -0,0 +1 @@
"""Tests for the GitNexus eval harness."""

6
eval/tests/conftest.py Normal file
View file

@ -0,0 +1,6 @@
import sys
from pathlib import Path
ROOT = Path(__file__).resolve().parents[1]
if str(ROOT) not in sys.path:
sys.path.insert(0, str(ROOT))

29
eval/tests/test_errors.py Normal file
View file

@ -0,0 +1,29 @@
from utils.errors import sanitize_exception
def _raise_value_error():
raise ValueError("boom")
def test_sanitize_exception_without_debug(monkeypatch):
monkeypatch.delenv("GITNEXUS_EVAL_DEBUG", raising=False)
try:
_raise_value_error()
except Exception as exc: # noqa: BLE001
data = sanitize_exception(exc, include_debug=False)
assert data["error_type"] == "ValueError"
assert data["error_message"] == "boom"
assert "error_detail_debug" not in data
def test_sanitize_exception_with_debug(monkeypatch):
monkeypatch.setenv("GITNEXUS_EVAL_DEBUG", "1")
try:
_raise_value_error()
except Exception as exc: # noqa: BLE001
data = sanitize_exception(exc)
assert data["error_type"] == "ValueError"
assert "error_detail_debug" in data
assert "ValueError" in data["error_detail_debug"]

View file

@ -0,0 +1,20 @@
import pytest
analysis_module = pytest.importorskip("analysis.analyze_results")
parse_run_id = analysis_module.parse_run_id
def test_parse_run_id_native_augment():
model, mode = parse_run_id("claude-sonnet_native_augment")
assert model == "claude-sonnet"
assert mode == "native_augment"
def test_parse_run_id_hyphenated_model():
model, mode = parse_run_id("glm-4.7_native")
assert model == "glm-4.7"
assert mode == "native"
def test_parse_run_id_unknown():
assert parse_run_id("custom_model") == ("custom_model", "unknown")

View file

@ -0,0 +1,63 @@
from __future__ import annotations
from hypothesis import given
from hypothesis import strategies as st
from analysis.analyze_results import parse_run_id
from environments.gitnexus_docker import GitNexusDockerEnvironment
from tool_registry import TOOL_SPECS
from utils.errors import sanitize_exception
KNOWN_MODES = [
"native_augment",
"native",
"baseline",
"mcp",
"augment",
"full",
]
def _model_strategy():
base_chars = st.characters(
blacklist_categories=("Cs",),
blacklist_characters={" ", "\n", "\t"},
)
text = st.text(alphabet=base_chars, min_size=1)
return text.filter(lambda s: not any(s.endswith(f"_{m}") for m in KNOWN_MODES))
@given(_model_strategy(), st.sampled_from(KNOWN_MODES))
def test_parse_run_id_round_trip(model: str, mode: str) -> None:
run_id = f"{model}_{mode}"
parsed_model, parsed_mode = parse_run_id(run_id)
assert parsed_model == model
assert parsed_mode == mode
@given(st.text())
def test_sanitize_exception_respects_debug_flag(message: str) -> None:
exc = ValueError(message)
data = sanitize_exception(exc, include_debug=False)
assert data["error_type"] == "ValueError"
assert data["error_message"] == (message or "ValueError")
assert "error_detail_debug" not in data
data_debug = sanitize_exception(exc, include_debug=True)
assert data_debug["error_type"] == "ValueError"
assert "error_detail_debug" in data_debug
assert data_debug["error_detail_debug"]
@given(st.sampled_from(list(TOOL_SPECS.values())), st.integers(min_value=1, max_value=99999))
def test_render_tool_script_contains_expected_paths(spec, port: int) -> None:
script = GitNexusDockerEnvironment._render_tool_script(spec, str(port))
assert spec.fallback.strip() in script
if spec.endpoint:
assert spec.endpoint in script
assert f"${{GITNEXUS_EVAL_PORT:-{port}}}" in script
assert "curl" in script
else:
assert "curl" not in script

View file

@ -0,0 +1,21 @@
import pytest
GitNexusDockerEnvironment = pytest.importorskip(
"environments.gitnexus_docker"
).GitNexusDockerEnvironment
tool_registry = pytest.importorskip("tool_registry")
TOOL_SPECS = tool_registry.TOOL_SPECS
def test_render_query_script_uses_endpoint_and_fallback():
script = GitNexusDockerEnvironment._render_tool_script(TOOL_SPECS["query"], "4848")
assert "/tool/query" in script
assert "gitnexus query" in script
assert "GITNEXUS_EVAL_PORT" in script
def test_render_augment_script_skips_curl():
script = GitNexusDockerEnvironment._render_tool_script(TOOL_SPECS["augment"], "4848")
assert "/tool/" not in script
assert "curl" not in script
assert "gitnexus augment" in script

79
eval/tool_registry.py Normal file
View file

@ -0,0 +1,79 @@
from __future__ import annotations
from dataclasses import dataclass
from typing import Dict, Tuple
@dataclass(frozen=True)
class ToolScriptSpec:
key: str
bin_name: str
endpoint: str | None
payload_builder: str
fallback: str
header: str | None = None
TOOL_METRIC_KEYS: Tuple[str, ...] = ("query", "context", "impact", "cypher", "overview")
TOOL_SPECS: Dict[str, ToolScriptSpec] = {
"query": ToolScriptSpec(
key="query",
bin_name="gitnexus-query",
endpoint="/tool/query",
payload_builder=r'''query="$1"; task_ctx="${2:-}"; goal="${3:-}"
[ -z "$query" ] && echo "Usage: gitnexus-query <query> [task_context] [goal]" && exit 1
payload="{\"query\": \"$query\""
[ -n "$task_ctx" ] && payload="$payload, \"task_context\": \"$task_ctx\""
[ -n "$goal" ] && payload="$payload, \"goal\": \"$goal\""
payload="$payload}"''',
fallback='cd /testbed && npx gitnexus query "$query" 2>&1',
),
"context": ToolScriptSpec(
key="context",
bin_name="gitnexus-context",
endpoint="/tool/context",
payload_builder=r'''name="$1"; file_path="${2:-}"
[ -z "$name" ] && echo "Usage: gitnexus-context <symbol_name> [file_path]" && exit 1
payload="{\"name\": \"$name\""
[ -n "$file_path" ] && payload="$payload, \"file_path\": \"$file_path\""
payload="$payload}"''',
fallback='cd /testbed && npx gitnexus context "$name" 2>&1',
),
"impact": ToolScriptSpec(
key="impact",
bin_name="gitnexus-impact",
endpoint="/tool/impact",
payload_builder=r'''target="$1"; direction="${2:-upstream}"
[ -z "$target" ] && echo "Usage: gitnexus-impact <symbol_name> [upstream|downstream]" && exit 1
payload="{\"target\": \"$target\", \"direction\": \"$direction\"}"''',
fallback='cd /testbed && npx gitnexus impact "$target" --direction "$direction" 2>&1',
),
"cypher": ToolScriptSpec(
key="cypher",
bin_name="gitnexus-cypher",
endpoint="/tool/cypher",
payload_builder=r'''query="$1"
[ -z "$query" ] && echo "Usage: gitnexus-cypher <cypher_query>" && exit 1
payload="{\"query\": \"$query\"}"''',
fallback='cd /testbed && npx gitnexus cypher "$query" 2>&1',
),
"overview": ToolScriptSpec(
key="overview",
bin_name="gitnexus-overview",
endpoint="/tool/list_repos",
header='echo "=== Code Knowledge Graph Overview ==="',
payload_builder='payload="{}"',
fallback='cd /testbed && npx gitnexus list 2>&1',
),
"augment": ToolScriptSpec(
key="augment",
bin_name="gitnexus-augment",
endpoint=None,
payload_builder="",
fallback='cd /testbed && npx gitnexus augment "$1" 2>&1 || true',
),
}
BINARIES_BY_KEY: Dict[str, str] = {spec.key: spec.bin_name for spec in TOOL_SPECS.values()}
ENDPOINTS_BY_KEY: Dict[str, str | None] = {spec.key: spec.endpoint for spec in TOOL_SPECS.values()}

1
eval/utils/__init__.py Normal file
View file

@ -0,0 +1 @@
"""Utility package for the eval harness."""

60
eval/utils/errors.py Normal file
View file

@ -0,0 +1,60 @@
from __future__ import annotations
import os
import traceback
from typing import Any, Callable
from constants import DEBUG_ENV_VAR
def is_debug_enabled() -> bool:
"""Return True when debug output (full tracebacks) should be emitted."""
return os.getenv(DEBUG_ENV_VAR, "").strip().lower() in {"1", "true", "yes", "on"}
def sanitize_exception(exc: BaseException, *, include_debug: bool | None = None) -> dict[str, str]:
"""
Produce a log-safe, JSON-friendly view of an exception.
- Always returns error_type and error_message.
- Only includes error_detail_debug (full traceback) when debug is enabled.
"""
debug = is_debug_enabled() if include_debug is None else include_debug
error_type = type(exc).__name__
message = str(exc) or error_type
data: dict[str, str] = {
"error_type": error_type,
"error_message": message,
}
if debug:
tb = "".join(traceback.format_exception(type(exc), exc, exc.__traceback__))
if tb:
data["error_detail_debug"] = tb
return data
def log_safe_exception(
logger: Any,
prefix: str,
exc: BaseException,
*,
include_debug: bool | None = None,
level: str = "error",
) -> dict[str, str]:
"""
Log an exception without leaking stack traces unless debug is enabled.
Returns the sanitized dict so callers can persist it.
"""
data = sanitize_exception(exc, include_debug=include_debug)
debug = "error_detail_debug" in data
log_fn: Callable[..., None] = getattr(logger, level, logger.error)
message = f"{prefix}: {data['error_type']}: {data['error_message']}"
log_kwargs = {"exc_info": True} if debug else {}
log_fn(message, **log_kwargs)
return data

2529
eval/uv.lock generated Normal file

File diff suppressed because it is too large Load diff

View file

@ -14,6 +14,9 @@ import {
Variable,
Hash,
Target,
List,
AtSign,
Type,
} from '@/lib/lucide-icons';
import { useAppState } from '../hooks/useAppState';
import { FILTERABLE_LABELS, NODE_COLORS, ALL_EDGE_TYPES, EDGE_INFO, type EdgeType } from '../lib/constants';
@ -187,7 +190,11 @@ const getNodeTypeIcon = (label: NodeLabel) => {
case 'Function': return Braces;
case 'Method': return Braces;
case 'Interface': return Hash;
case 'Enum': return List;
case 'Type': return Type;
case 'Decorator': return AtSign;
case 'Import': return FileCode;
case 'Variable': return Variable;
default: return Variable;
}
};
@ -501,7 +508,7 @@ export const FileTreePanel = ({ onFocusNode }: FileTreePanelProps) => {
Color Legend
</h3>
<div className="grid grid-cols-2 gap-2">
{(['Folder', 'File', 'Class', 'Function', 'Interface', 'Method'] as NodeLabel[]).map(label => (
{(['Folder', 'File', 'Class', 'Interface', 'Enum', 'Type', 'Function', 'Method', 'Variable', 'Decorator'] as NodeLabel[]).map(label => (
<div key={label} className="flex items-center gap-1.5">
<div
className="w-2.5 h-2.5 rounded-full"

View file

@ -112,15 +112,18 @@ export const DEFAULT_VISIBLE_LABELS: NodeLabel[] = [
'Type',
];
// All filterable labels
// All filterable labels (in display order)
export const FILTERABLE_LABELS: NodeLabel[] = [
'Folder',
'File',
'Class',
'Interface',
'Enum',
'Type',
'Function',
'Method',
'Variable',
'Interface',
'Decorator',
'Import',
];

View file

@ -10,6 +10,7 @@ export {
AlertCircle,
AlertTriangle,
ArrowRight,
AtSign,
Brain,
Box,
Braces,
@ -39,6 +40,7 @@ export {
Layers,
Lightbulb,
LightbulbOff,
List,
Loader2,
Maximize2,
MousePointerClick,
@ -63,6 +65,7 @@ export {
Target,
Terminal,
Trash2,
Type,
Upload,
User,
Variable,

View file

@ -59,6 +59,41 @@ describe('DEFAULT_VISIBLE_LABELS', () => {
});
});
describe('FILTERABLE_LABELS', () => {
it('includes all newly added node types', () => {
expect(FILTERABLE_LABELS).toContain('Enum');
expect(FILTERABLE_LABELS).toContain('Type');
expect(FILTERABLE_LABELS).toContain('Decorator');
expect(FILTERABLE_LABELS).toContain('Variable');
});
it('every filterable label has a defined color in NODE_COLORS', () => {
for (const label of FILTERABLE_LABELS) {
expect(NODE_COLORS).toHaveProperty(label);
expect(NODE_COLORS[label as keyof typeof NODE_COLORS]).toMatch(/^#[0-9a-f]{6}$/i);
}
});
it('every filterable label has a defined size in NODE_SIZES', () => {
for (const label of FILTERABLE_LABELS) {
expect(NODE_SIZES).toHaveProperty(label);
expect(NODE_SIZES[label as keyof typeof NODE_SIZES]).toBeGreaterThan(0);
}
});
it('has no duplicate entries', () => {
const unique = new Set(FILTERABLE_LABELS);
expect(unique.size).toBe(FILTERABLE_LABELS.length);
});
it('is a subset of DEFAULT_VISIBLE_LABELS plus togglable labels', () => {
const allKnown = new Set(Object.keys(NODE_COLORS));
for (const label of FILTERABLE_LABELS) {
expect(allKnown.has(label)).toBe(true);
}
});
});
describe('edge types', () => {
it('ALL_EDGE_TYPES contains all EDGE_INFO keys', () => {
const edgeInfoKeys = Object.keys(EDGE_INFO).sort();

View file

@ -0,0 +1,81 @@
import { describe, expect, it } from 'vitest';
import { FILTERABLE_LABELS, NODE_COLORS } from '../../src/lib/constants';
import type { NodeLabel } from '../../src/core/graph/types';
import * as lucideIcons from '../../src/lib/lucide-icons';
const LEGEND_LABELS: NodeLabel[] = [
'Folder', 'File', 'Class', 'Interface', 'Enum', 'Type',
'Function', 'Method', 'Variable', 'Decorator',
];
const ICON_MAP: Record<string, string> = {
Folder: 'Folder',
File: 'FileCode',
Class: 'Box',
Function: 'Braces',
Method: 'Braces',
Interface: 'Hash',
Enum: 'List',
Type: 'Type',
Decorator: 'AtSign',
Import: 'FileCode',
Variable: 'Variable',
};
describe('filter panel icon mappings', () => {
it('every filterable label has a mapped icon', () => {
for (const label of FILTERABLE_LABELS) {
expect(ICON_MAP).toHaveProperty(label);
}
});
it('every mapped icon is exported from lucide-icons', () => {
const exportedNames = new Set(Object.keys(lucideIcons));
const requiredIcons = new Set(Object.values(ICON_MAP));
for (const iconName of requiredIcons) {
expect(exportedNames.has(iconName), `${iconName} should be exported from lucide-icons`).toBe(true);
}
});
it('newly added node types have distinct icons', () => {
expect(ICON_MAP.Enum).toBe('List');
expect(ICON_MAP.Type).toBe('Type');
expect(ICON_MAP.Decorator).toBe('AtSign');
});
});
describe('color legend', () => {
it('includes all newly added node types', () => {
expect(LEGEND_LABELS).toContain('Enum');
expect(LEGEND_LABELS).toContain('Type');
expect(LEGEND_LABELS).toContain('Decorator');
expect(LEGEND_LABELS).toContain('Variable');
});
it('every legend label has a color defined', () => {
for (const label of LEGEND_LABELS) {
expect(NODE_COLORS).toHaveProperty(label);
expect(NODE_COLORS[label]).toMatch(/^#[0-9a-f]{6}$/i);
}
});
it('legend labels match the order used in FileTreePanel', () => {
const expected: NodeLabel[] = [
'Folder', 'File', 'Class', 'Interface', 'Enum', 'Type',
'Function', 'Method', 'Variable', 'Decorator',
];
expect(LEGEND_LABELS).toEqual(expected);
});
it('legend labels are a subset of filterable labels plus Import', () => {
const filterableSet = new Set(FILTERABLE_LABELS);
for (const label of LEGEND_LABELS) {
expect(filterableSet.has(label), `${label} should be in FILTERABLE_LABELS`).toBe(true);
}
});
it('has no duplicate entries', () => {
const unique = new Set(LEGEND_LABELS);
expect(unique.size).toBe(LEGEND_LABELS.length);
});
});

View file

@ -156,6 +156,7 @@ gitnexus analyze --embeddings # Enable embedding generation (slower, better
gitnexus analyze --verbose # Log skipped files when parsers are unavailable
gitnexus mcp # Start MCP server (stdio) — serves all indexed repos
gitnexus serve # Start local HTTP server (multi-repo) for web UI
gitnexus index # Register an existing .gitnexus/ folder into the global registry
gitnexus list # List all indexed repositories
gitnexus status # Show index status for current repo
gitnexus clean # Delete index for current repo

File diff suppressed because it is too large Load diff

View file

@ -46,7 +46,6 @@
"test:watch": "vitest",
"test:coverage": "vitest run --coverage",
"prepare": "npm run build",
"postinstall": "node scripts/patch-tree-sitter-swift.cjs",
"prepack": "npm run build && chmod +x dist/cli/index.js"
},
"dependencies": {
@ -66,18 +65,18 @@
"mnemonist": "^0.39.0",
"onnxruntime-node": "^1.24.0",
"pandemonium": "^2.4.0",
"tree-sitter": "0.22.4",
"tree-sitter-c": "^0.21.0",
"tree-sitter-c-sharp": "^0.21.0",
"tree-sitter-cpp": "^0.22.0",
"tree-sitter-go": "^0.21.0",
"tree-sitter-java": "^0.21.0",
"tree-sitter-javascript": "^0.21.0",
"tree-sitter-php": "^0.23.12",
"tree-sitter-python": "^0.21.0",
"tree-sitter": "^0.25.0",
"tree-sitter-c": "^0.24.1",
"tree-sitter-c-sharp": "^0.23.1",
"tree-sitter-cpp": "^0.23.4",
"tree-sitter-go": "^0.25.0",
"tree-sitter-java": "^0.23.5",
"tree-sitter-javascript": "^0.25.0",
"tree-sitter-php": "^0.24.2",
"tree-sitter-python": "^0.25.0",
"tree-sitter-ruby": "^0.23.1",
"tree-sitter-rust": "^0.21.0",
"tree-sitter-typescript": "^0.21.0",
"tree-sitter-rust": "^0.24.0",
"tree-sitter-typescript": "^0.23.2",
"uuid": "^13.0.0"
},
"optionalDependencies": {
@ -100,7 +99,7 @@
"@huggingface/transformers": {
"onnxruntime-node": "$onnxruntime-node"
},
"tree-sitter": "0.22.4"
"tree-sitter": "^0.25.0"
},
"engines": {
"node": ">=20.0.0"

View file

@ -1,76 +0,0 @@
#!/usr/bin/env node
/**
* WORKAROUND: tree-sitter-swift@0.6.0 binding.gyp build failure
*
* Background:
* tree-sitter-swift@0.6.0's binding.gyp contains an "actions" array that
* invokes `tree-sitter generate` to regenerate parser.c from grammar.js.
* This is intended for grammar developers, but the published npm package
* already ships pre-generated parser files (parser.c, scanner.c), so the
* actions are unnecessary for consumers. Since consumers don't have
* tree-sitter-cli installed, the actions always fail during `npm install`.
*
* Why we can't just upgrade:
* tree-sitter-swift@0.7.1 fixes this (removes postinstall, ships prebuilds),
* but it requires tree-sitter@^0.22.1. The upstream project pins tree-sitter
* to ^0.21.0 and all other grammar packages depend on that version.
* Upgrading tree-sitter would be a separate breaking change.
*
* How this workaround works:
* 1. tree-sitter-swift's own postinstall fails (npm warns but continues)
* 2. This script runs as gitnexus's postinstall
* 3. It removes the "actions" array from binding.gyp
* 4. It rebuilds the native binding with the cleaned binding.gyp
*
* TODO: Remove this script when tree-sitter is upgraded to ^0.22.x,
* which allows using tree-sitter-swift@0.7.1+ directly.
*/
const fs = require('fs');
const path = require('path');
const { execSync } = require('child_process');
const swiftDir = path.join(__dirname, '..', 'node_modules', 'tree-sitter-swift');
const bindingPath = path.join(swiftDir, 'binding.gyp');
try {
if (!fs.existsSync(bindingPath)) {
process.exit(0);
}
const content = fs.readFileSync(bindingPath, 'utf8');
let needsRebuild = false;
if (content.includes('"actions"')) {
// Strip Python-style comments (#) and trailing commas before JSON parsing
const cleaned = content
.replace(/#[^\n]*/g, '') // Remove # comments
.replace(/,(\s*[\]}])/g, '$1'); // Remove trailing commas before ] or }
const gyp = JSON.parse(cleaned);
if (gyp.targets && gyp.targets[0] && gyp.targets[0].actions) {
delete gyp.targets[0].actions;
fs.writeFileSync(bindingPath, JSON.stringify(gyp, null, 2) + '\n');
console.log('[tree-sitter-swift] Patched binding.gyp (removed actions array)');
needsRebuild = true;
}
}
// Check if native binding exists
const bindingNode = path.join(swiftDir, 'build', 'Release', 'tree_sitter_swift_binding.node');
if (!fs.existsSync(bindingNode)) {
needsRebuild = true;
}
if (needsRebuild) {
console.log('[tree-sitter-swift] Rebuilding native binding...');
execSync('npx node-gyp rebuild', {
cwd: swiftDir,
stdio: 'pipe',
timeout: 120000,
});
console.log('[tree-sitter-swift] Native binding built successfully');
}
} catch (err) {
console.warn('[tree-sitter-swift] Could not build native binding:', err.message);
console.warn('[tree-sitter-swift] You may need to manually run: cd node_modules/tree-sitter-swift && npx node-gyp rebuild');
}

View file

@ -0,0 +1,137 @@
/**
* Index Command
*
* Registers an existing .gitnexus/ folder into the global registry so the
* MCP server can discover the repo without running a full `gitnexus analyze`.
*
* Useful when a pre-built .gitnexus/ directory is already present (e.g. after
* cloning a repo that ships its index, restoring from backup, or using a
* shared team index).
*/
import path from "path";
import fs from "fs/promises";
import {
getStoragePaths,
loadMeta,
addToGitignore,
registerRepo,
} from "../storage/repo-manager.js";
import { getGitRoot, isGitRepo } from "../storage/git.js";
export interface IndexOptions {
force?: boolean;
allowNonGit?: boolean;
}
export const indexCommand = async (
inputPathParts?: string[],
options?: IndexOptions,
) => {
console.log("\n GitNexus Index\n");
const inputPath = inputPathParts?.length
? inputPathParts.join(" ")
: undefined;
if (inputPathParts && inputPathParts.length > 1) {
const resolvedCombinedPath = path.resolve(inputPath);
try {
await fs.access(resolvedCombinedPath);
} catch {
console.log(" The `index` command accepts a single path only.");
console.log(" If your path contains spaces, wrap it in quotes.");
console.log(` Received multiple path parts: ${inputPathParts.join(", ")}`);
console.log("");
process.exitCode = 1;
return;
}
}
let repoPath: string;
if (inputPath) {
repoPath = path.resolve(inputPath);
} else {
const gitRoot = getGitRoot(process.cwd());
if (!gitRoot) {
console.log(" Not inside a git repository, try to run git init\n");
process.exitCode = 1;
return;
}
repoPath = gitRoot;
}
if (!options?.allowNonGit && !isGitRepo(repoPath)) {
console.log(` Not a git repository: ${repoPath}`);
console.log(" Initialize one with `git init` or choose a valid repo path.\n");
console.log(" Or use --allow-non-git to register an existing .gitnexus index anyway.\n");
process.exitCode = 1;
return;
}
const { storagePath, lbugPath } = getStoragePaths(repoPath);
// ── Verify .gitnexus/ exists ──────────────────────────────────────
try {
await fs.access(storagePath);
} catch {
console.log(` No .gitnexus/ folder found at: ${storagePath}`);
console.log(" Run `gitnexus analyze` to build the index first.\n");
process.exitCode = 1;
return;
}
// ── Verify lbug database exists ───────────────────────────────────
try {
await fs.access(lbugPath);
} catch {
console.log(` .gitnexus/ folder exists but contains no LadybugDB index.`);
console.log(" Run `gitnexus analyze` to build the index.\n");
process.exitCode = 1;
return;
}
// ── Load or reconstruct meta ──────────────────────────────────────
let meta = await loadMeta(storagePath);
if (!meta) {
if (!options?.force) {
console.log(` .gitnexus/ exists but meta.json is missing.`);
console.log(" Use --force to register anyway (stats will be empty),");
console.log(" or run `gitnexus analyze` to rebuild properly.\n");
process.exitCode = 1;
return;
}
// --force: build a minimal meta so the repo can be registered
meta = {
repoPath,
lastCommit: "",
indexedAt: new Date().toISOString(),
};
}
// ── Register in global registry ───────────────────────────────────
await registerRepo(repoPath, meta);
await addToGitignore(repoPath);
const projectName = path.basename(repoPath);
const { stats } = meta;
console.log(` Repository registered: ${projectName}`);
if (stats) {
const parts: string[] = [];
if (stats.nodes != null) {
parts.push(`${stats.nodes.toLocaleString()} nodes`);
}
if (stats.edges != null) {
parts.push(`${stats.edges.toLocaleString()} edges`);
}
if (stats.communities != null) parts.push(`${stats.communities} clusters`);
if (stats.processes != null) parts.push(`${stats.processes} flows`);
if (parts.length) console.log(` ${parts.join(" | ")}`);
}
console.log(` ${repoPath}`);
console.log("");
};

View file

@ -33,6 +33,13 @@ program
.addHelpText('after', '\nEnvironment variables:\n GITNEXUS_NO_GITIGNORE=1 Skip .gitignore parsing (still reads .gitnexusignore)')
.action(createLazyAction(() => import('./analyze.js'), 'analyzeCommand'));
program
.command('index [path...]')
.description('Register an existing .gitnexus/ folder into the global registry (no re-analysis needed)')
.option('-f, --force', 'Register even if meta.json is missing (stats will be empty)')
.option('--allow-non-git', 'Allow registering folders that are not Git repositories')
.action(createLazyAction(() => import('./index-repo.js'), 'indexCommand'));
program
.command('serve')
.description('Start local HTTP server for web UI connection')
@ -66,11 +73,14 @@ program
.command('wiki [path]')
.description('Generate repository wiki from knowledge graph')
.option('-f, --force', 'Force full regeneration even if up to date')
.option('--model <model>', 'LLM model name (default: minimax/minimax-m2.5)')
.option('--base-url <url>', 'LLM API base URL (default: OpenAI)')
.option('--provider <provider>', 'LLM provider: openai or cursor (default: openai)')
.option('--model <model>', 'LLM model name (default depends on provider)')
.option('--base-url <url>', 'LLM API base URL (for openai provider)')
.option('--api-key <key>', 'LLM API key (saved to ~/.gitnexus/config.json)')
.option('--concurrency <n>', 'Parallel LLM calls (default: 3)', '3')
.option('--gist', 'Publish wiki as a public GitHub Gist after generation')
.option('-v, --verbose', 'Enable verbose output (show LLM commands and responses)')
.option('--review', 'Stop after grouping to review module structure before generating pages')
.action(createLazyAction(() => import('./wiki.js'), 'wikiCommand'));
program

View file

@ -12,7 +12,8 @@ import cliProgress from 'cli-progress';
import { getGitRoot, isGitRepo } from '../storage/git.js';
import { getStoragePaths, loadMeta, loadCLIConfig, saveCLIConfig } from '../storage/repo-manager.js';
import { WikiGenerator, type WikiOptions } from '../core/wiki/generator.js';
import { resolveLLMConfig } from '../core/wiki/llm-client.js';
import { resolveLLMConfig, type LLMProvider } from '../core/wiki/llm-client.js';
import { detectCursorCLI } from '../core/wiki/cursor-client.js';
export interface WikiCommandOptions {
force?: boolean;
@ -21,6 +22,9 @@ export interface WikiCommandOptions {
apiKey?: string;
concurrency?: string;
gist?: boolean;
provider?: LLMProvider;
verbose?: boolean;
review?: boolean;
}
/**
@ -78,6 +82,11 @@ export const wikiCommand = async (
inputPath?: string,
options?: WikiCommandOptions,
) => {
// Set verbose mode globally for cursor-client to pick up
if (options?.verbose) {
process.env.GITNEXUS_VERBOSE = '1';
}
console.log('\n GitNexus Wiki Generator\n');
// ── Resolve repo path ───────────────────────────────────────────────
@ -113,98 +122,135 @@ export const wikiCommand = async (
// ── Resolve LLM config (with interactive fallback) ─────────────────
// Save any CLI overrides immediately
if (options?.apiKey || options?.model || options?.baseUrl) {
if (options?.apiKey || options?.model || options?.baseUrl || options?.provider) {
const existing = await loadCLIConfig();
const updates: Record<string, string> = {};
if (options.apiKey) updates.apiKey = options.apiKey;
if (options.model) updates.model = options.model;
if (options.baseUrl) updates.baseUrl = options.baseUrl;
if (options.provider) updates.provider = options.provider;
// Save model to appropriate field based on provider
if (options.model) {
if (options.provider === 'cursor') {
updates.cursorModel = options.model;
} else {
updates.model = options.model;
}
}
await saveCLIConfig({ ...existing, ...updates });
console.log(' Config saved to ~/.gitnexus/config.json\n');
}
const savedConfig = await loadCLIConfig();
const hasSavedConfig = !!(savedConfig.apiKey && savedConfig.baseUrl);
const hasCLIOverrides = !!(options?.apiKey || options?.model || options?.baseUrl);
const hasSavedConfig = !!(savedConfig.provider === 'cursor' || (savedConfig.apiKey && savedConfig.baseUrl));
const hasCLIOverrides = !!(options?.apiKey || options?.model || options?.baseUrl || options?.provider);
let llmConfig = await resolveLLMConfig({
model: options?.model,
baseUrl: options?.baseUrl,
apiKey: options?.apiKey,
provider: options?.provider,
});
// Run interactive setup if no saved config and no CLI flags provided
// (even if env vars exist — let user explicitly choose their provider)
if (!hasSavedConfig && !hasCLIOverrides) {
if (!process.stdin.isTTY) {
if (!llmConfig.apiKey) {
// Non-interactive mode — need either API key or Cursor CLI
if (!llmConfig.apiKey && llmConfig.provider !== 'cursor') {
console.log(' Error: No LLM API key found.');
console.log(' Set OPENAI_API_KEY or GITNEXUS_API_KEY environment variable,');
console.log(' or pass --api-key <key>.\n');
console.log(' or pass --api-key <key>, or use --provider cursor.\n');
process.exitCode = 1;
return;
}
// Non-interactive with env var — just use it
// Non-interactive with env var or cursor — just use it
} else {
console.log(' No LLM configured. Let\'s set it up.\n');
console.log(' Supports OpenAI, OpenRouter, or any OpenAI-compatible API.\n');
console.log(' Supports OpenAI, OpenRouter, any OpenAI-compatible API, or Cursor CLI.\n');
// Check if Cursor CLI is available
const hasCursor = detectCursorCLI();
// Provider selection
console.log(' [1] OpenAI (api.openai.com)');
console.log(' [2] OpenRouter (openrouter.ai)');
console.log(' [3] Custom endpoint\n');
console.log(' [3] Custom endpoint');
if (hasCursor) {
console.log(' [4] Cursor CLI (local, uses your Cursor subscription)');
}
console.log('');
const choice = await prompt(' Select provider (1/2/3): ');
const maxChoice = hasCursor ? '4' : '3';
const choice = await prompt(` Select provider (1/${maxChoice}): `);
let baseUrl: string;
let defaultModel: string;
let provider: LLMProvider = 'openai';
let key = '';
if (choice === '2') {
baseUrl = 'https://openrouter.ai/api/v1';
defaultModel = 'minimax/minimax-m2.5';
} else if (choice === '3') {
baseUrl = await prompt(' Base URL (e.g. http://localhost:11434/v1): ');
if (!baseUrl) {
console.log('\n No URL provided. Aborting.\n');
process.exitCode = 1;
return;
}
defaultModel = 'gpt-4o-mini';
if (choice === '4' && hasCursor) {
// Cursor CLI selected - model defaults to 'auto' (Cursor's default)
provider = 'cursor';
baseUrl = '';
const modelInput = await prompt(' Model (leave empty for auto): ');
const model = modelInput || '';
// Save config for Cursor
const cursorConfig: Record<string, string> = { provider: 'cursor' };
if (model) cursorConfig.cursorModel = model;
await saveCLIConfig(cursorConfig);
console.log(' Config saved to ~/.gitnexus/config.json\n');
llmConfig = { ...llmConfig, provider: 'cursor', model, apiKey: '', baseUrl: '' };
} else {
baseUrl = 'https://api.openai.com/v1';
defaultModel = 'gpt-4o-mini';
}
// OpenAI-compatible provider
if (choice === '2') {
baseUrl = 'https://openrouter.ai/api/v1';
defaultModel = 'minimax/minimax-m2.5';
} else if (choice === '3') {
baseUrl = await prompt(' Base URL (e.g. http://localhost:11434/v1): ');
if (!baseUrl) {
console.log('\n No URL provided. Aborting.\n');
process.exitCode = 1;
return;
}
defaultModel = 'gpt-4o-mini';
} else {
baseUrl = 'https://api.openai.com/v1';
defaultModel = 'gpt-4o-mini';
}
// Model
const modelInput = await prompt(` Model (default: ${defaultModel}): `);
const model = modelInput || defaultModel;
// Model
const modelInput = await prompt(` Model (default: ${defaultModel}): `);
const model = modelInput || defaultModel;
// API key — pre-fill hint if env var exists
const envKey = process.env.GITNEXUS_API_KEY || process.env.OPENAI_API_KEY || '';
let key: string;
if (envKey) {
const masked = envKey.slice(0, 6) + '...' + envKey.slice(-4);
const useEnv = await prompt(` Use existing env key (${masked})? (Y/n): `);
if (!useEnv || useEnv.toLowerCase() === 'y' || useEnv.toLowerCase() === 'yes') {
key = envKey;
// API key — pre-fill hint if env var exists
const envKey = process.env.GITNEXUS_API_KEY || process.env.OPENAI_API_KEY || '';
if (envKey) {
const masked = envKey.slice(0, 6) + '...' + envKey.slice(-4);
const useEnv = await prompt(` Use existing env key (${masked})? (Y/n): `);
if (!useEnv || useEnv.toLowerCase() === 'y' || useEnv.toLowerCase() === 'yes') {
key = envKey;
} else {
key = await prompt(' API key: ', true);
}
} else {
key = await prompt(' API key: ', true);
}
} else {
key = await prompt(' API key: ', true);
if (!key) {
console.log('\n No key provided. Aborting.\n');
process.exitCode = 1;
return;
}
// Save
await saveCLIConfig({ apiKey: key, baseUrl, model, provider: 'openai' });
console.log(' Config saved to ~/.gitnexus/config.json\n');
llmConfig = { ...llmConfig, apiKey: key, baseUrl, model, provider: 'openai' };
}
if (!key) {
console.log('\n No key provided. Aborting.\n');
process.exitCode = 1;
return;
}
// Save
await saveCLIConfig({ apiKey: key, baseUrl, model });
console.log(' Config saved to ~/.gitnexus/config.json\n');
llmConfig = { ...llmConfig, apiKey: key, baseUrl, model };
}
}
@ -239,9 +285,8 @@ export const wikiCommand = async (
// ── Run generator ───────────────────────────────────────────────────
const wikiOptions: WikiOptions = {
force: options?.force,
model: options?.model,
baseUrl: options?.baseUrl,
concurrency: options?.concurrency ? parseInt(options.concurrency, 10) : undefined,
reviewOnly: options?.review,
};
const generator = new WikiGenerator(
@ -264,13 +309,116 @@ export const wikiCommand = async (
const result = await generator.run();
clearInterval(elapsedTimer);
bar.update(100, { phase: 'Done' });
bar.stop();
const elapsed = ((Date.now() - t0) / 1000).toFixed(1);
const wikiDir = path.join(storagePath, 'wiki');
const viewerPath = path.join(wikiDir, 'index.html');
const treeFile = path.join(wikiDir, 'module_tree.json');
// Review mode: show module tree and ask for confirmation
if (options?.review && result.moduleTree) {
console.log(`\n Module structure ready for review (${elapsed}s)\n`);
console.log(' Modules to generate:\n');
const printTree = (nodes: typeof result.moduleTree, indent = 0) => {
for (const node of nodes) {
const prefix = ' '.repeat(indent + 2);
const fileCount = node.files?.length || 0;
const childCount = node.children?.length || 0;
const suffix = fileCount > 0 ? ` (${fileCount} files)` : childCount > 0 ? ` (${childCount} children)` : '';
console.log(`${prefix}- ${node.name}${suffix}`);
if (node.children && node.children.length > 0) {
printTree(node.children, indent + 1);
}
}
};
printTree(result.moduleTree);
console.log(`\n Tree saved to: ${treeFile}`);
console.log(' You can edit this file to remove/rename modules.\n');
// Ask for confirmation (auto-continue in non-interactive environments)
if (!process.stdin.isTTY) {
console.log(' Non-interactive mode — auto-continuing with generation.\n');
}
const answer = process.stdin.isTTY
? await prompt(' Continue with generation? (Y/n/edit): ')
: 'y';
const choice = answer.trim().toLowerCase();
if (choice === 'n' || choice === 'no') {
console.log('\n Generation cancelled. Run `gitnexus wiki` later to generate.\n');
return;
}
if (choice === 'edit' || choice === 'e') {
// Open editor for the user
const editor = process.env.EDITOR || process.env.VISUAL || 'vi';
console.log(`\n Opening ${treeFile} in ${editor}...`);
console.log(' Save and close the editor when done.\n');
try {
execSync(`${editor} "${treeFile}"`, { stdio: 'inherit' });
} catch {
console.log(` Could not open editor. Please edit manually:\n ${treeFile}\n`);
console.log(' Then run `gitnexus wiki` to continue.\n');
return;
}
}
// Continue with generation using the (possibly edited) tree
console.log('\n Continuing with wiki generation...\n');
bar.start(100, 30, { phase: 'Generating pages...' });
// Re-run generator without reviewOnly flag
const continueOptions: WikiOptions = {
...wikiOptions,
reviewOnly: false,
};
const continueGenerator = new WikiGenerator(
repoPath,
storagePath,
lbugPath,
llmConfig,
continueOptions,
(phase, percent, detail) => {
const label = detail || phase;
if (label !== lastPhase) {
lastPhase = label;
phaseStart = Date.now();
}
bar.update(percent, { phase: label });
},
);
const continueResult = await continueGenerator.run();
bar.update(100, { phase: 'Done' });
bar.stop();
const totalElapsed = ((Date.now() - t0) / 1000).toFixed(1);
console.log(`\n Wiki generated successfully (${totalElapsed}s)\n`);
console.log(` Mode: ${continueResult.mode}`);
console.log(` Pages: ${continueResult.pagesGenerated}`);
console.log(` Output: ${wikiDir}`);
console.log(` Viewer: ${viewerPath}`);
if (continueResult.failedModules && continueResult.failedModules.length > 0) {
console.log(`\n Failed modules (${continueResult.failedModules.length}):`);
for (const mod of continueResult.failedModules) {
console.log(` - ${mod}`);
}
}
console.log('');
await maybePublishGist(viewerPath, options?.gist);
return;
}
bar.update(100, { phase: 'Done' });
if (result.mode === 'up-to-date' && !options?.force) {
console.log('\n Wiki is already up to date.');
@ -322,7 +470,7 @@ export const wikiCommand = async (
}
} else {
console.log(`\n Error: ${err.message}\n`);
if (process.env.DEBUG) {
if (process.env.GITNEXUS_VERBOSE) {
console.error(err);
}
}

View file

@ -9,14 +9,13 @@
* ----------------------------------|------------------------------------------|---------------------------
* tree-sitter-queries.ts | Query string + LANGUAGE_QUERIES entry | (required)
* export-detection.ts | ExportChecker function + table entry | (required)
* import-resolution.ts | Resolver in importResolvers | resolveStandard(...)
* import-resolution.ts | namedBindingExtractors entry | undefined
* call-routing.ts | callRouters entry | noRouting
* import-resolvers/<lang>.ts | Exported resolve<Lang>Import function | resolveStandard(...)
* call-routing.ts | CallRouter function (or noRouting) | noRouting
* entry-point-scoring.ts | ENTRY_POINT_PATTERNS entry | []
* framework-detection.ts | AST_FRAMEWORK_PATTERNS entry | []
* type-extractors/<lang>.ts | New file + index.ts import | (required)
* resolvers/<lang>.ts | Resolver file (if non-standard) | (only if resolveStandard insufficient)
* named-binding-extraction.ts | Extractor (if named imports) | (only if language has named imports)
* named-bindings/<lang>.ts | Extractor (if named imports) | (only if language has named imports)
*
* 4. Also check these files for language-specific if-checks (no compile-time guard):
* - mro-processor.ts (MRO strategy selection)

View file

@ -5,34 +5,30 @@ import Parser from 'tree-sitter';
import type { ResolutionContext } from './resolution-context.js';
import { TIER_CONFIDENCE, type ResolutionTier } from './resolution-context.js';
import { isLanguageAvailable, loadParser, loadLanguage } from '../tree-sitter/parser-loader.js';
import { LANGUAGE_QUERIES } from './tree-sitter-queries.js';
import { getProvider } from './languages/index.js';
import { generateId } from '../../lib/utils.js';
import { getLanguageFromFilename } from './utils/language-detection.js';
import { isVerboseIngestionEnabled } from './utils/verbose.js';
import { yieldToEventLoop } from './utils/event-loop.js';
import { FUNCTION_NODE_TYPES, extractFunctionName, findEnclosingClassId } from './utils/ast-helpers.js';
import { isBuiltInOrNoise } from './utils/noise-filter.js';
import {
getLanguageFromFilename,
isVerboseIngestionEnabled,
yieldToEventLoop,
FUNCTION_NODE_TYPES,
extractFunctionName,
isBuiltInOrNoise,
countCallArguments,
inferCallForm,
extractReceiverName,
extractReceiverNode,
findEnclosingClassId,
CALL_EXPRESSION_TYPES,
extractMixedChain,
type MixedChainStep,
} from './utils.js';
} from './utils/call-analysis.js';
import { buildTypeEnv, isSubclassOf } from './type-env.js';
import type { ConstructorBinding } from './type-env.js';
import { getTreeSitterBufferSize } from './constants.js';
import type { ExtractedCall, ExtractedAssignment, ExtractedHeritage, ExtractedRoute, ExtractedFetchCall, FileConstructorBindings } from './workers/parse-worker.js';
import { normalizeFetchURL, routeMatches } from './route-extractors/nextjs.js';
import { callRouters } from './call-routing.js';
import { extractReturnTypeName, stripNullable } from './type-extractors/shared.js';
import { typeConfigs } from './type-extractors/index.js';
import type { LiteralTypeInferrer } from './type-extractors/types.js';
import type { SyntaxNode } from './utils.js';
import type { SyntaxNode } from './utils/ast-helpers.js';
/** Per-file resolved type bindings for exported symbols.
* Populated during call processing, consumed by Phase 14 re-resolution pass. */
@ -85,12 +81,12 @@ export function buildImportedRawReturnTypes(
/** Collect resolved type bindings for exported file-scope symbols.
* Uses graph node isExported flag — does NOT require isExported on SymbolDefinition. */
function collectExportedBindings(
typeEnv: { readonly env: ReadonlyMap<string, ReadonlyMap<string, string>> },
typeEnv: { fileScope(): ReadonlyMap<string, string> },
filePath: string,
symbolTable: { lookupExact(filePath: string, name: string): string | undefined },
graph: { getNode(id: string): { properties?: { isExported?: boolean } } | undefined },
): Map<string, string> | null {
const fileScope = typeEnv.env.get('');
const fileScope = typeEnv.fileScope();
if (!fileScope || fileScope.size === 0) return null;
const exported = new Map<string, string>();
@ -190,7 +186,8 @@ const TYPE_PRESERVING_METHODS = new Set([
const findEnclosingFunction = (
node: SyntaxNode,
filePath: string,
ctx: ResolutionContext
ctx: ResolutionContext,
provider: import('./language-provider.js').LanguageProvider,
): string | null => {
let current = node.parent;
@ -204,7 +201,13 @@ const findEnclosingFunction = (
return resolved.candidates[0].nodeId;
}
return generateId(label, `${filePath}:${funcName}`);
// Apply labelOverride so label matches the definition phase (single source of truth).
let finalLabel = label;
if (provider.labelOverride) {
const override = provider.labelOverride(current, label);
if (override !== null) finalLabel = override;
}
return generateId(finalLabel, `${filePath}:${funcName}`);
}
}
current = current.parent;
@ -313,7 +316,8 @@ export const processCalls = async (
continue;
}
const queryStr = LANGUAGE_QUERIES[language];
const provider = getProvider(language);
const queryStr = provider.treeSitterQueries;
if (!queryStr) continue;
await loadLanguage(language, file.path);
@ -339,8 +343,6 @@ export const processCalls = async (
continue;
}
const lang = getLanguageFromFilename(file.path);
// Pre-pass: extract heritage from query matches to build parentMap for buildTypeEnv.
// Heritage-processor runs in PARALLEL, so graph edges don't exist when buildTypeEnv runs.
const fileParentMap = new Map<string, string[]>();
@ -374,19 +376,20 @@ export const processCalls = async (
const importedBindings = importedBindingsMap?.get(file.path);
const importedReturnTypes = importedReturnTypesMap?.get(file.path);
const importedRawReturnTypes = importedRawReturnTypesMap?.get(file.path);
const typeEnv = lang ? buildTypeEnv(tree, lang, { symbolTable: ctx.symbols, parentMap, importedBindings, importedReturnTypes, importedRawReturnTypes }) : null;
const typeEnv = buildTypeEnv(tree, language, { symbolTable: ctx.symbols, parentMap, importedBindings, importedReturnTypes, importedRawReturnTypes });
if (typeEnv && exportedTypeMap) {
const fileExports = collectExportedBindings(typeEnv, file.path, ctx.symbols, graph);
if (fileExports) exportedTypeMap.set(file.path, fileExports);
}
const callRouter = callRouters[language];
const callRouter = provider.callRouter;
const verifiedReceivers = typeEnv && typeEnv.constructorBindings.length > 0
const verifiedReceivers = typeEnv.constructorBindings.length > 0
? verifyConstructorBindings(typeEnv.constructorBindings, file.path, ctx)
: new Map<string, string>();
const receiverIndex = buildReceiverTypeIndex(verifiedReceivers);
ctx.enableCache(file.path);
const widenCache: WidenCache = new Map();
matches.forEach(match => {
const captureMap: Record<string, any> = {};
@ -403,7 +406,7 @@ export const processCalls = async (
}
// Fall back to verified constructor bindings (mirrors CALLS resolution tier 2)
if (!receiverTypeName && receiverText && receiverIndex.size > 0) {
const enclosing = findEnclosingFunction(captureMap['assignment'], file.path, ctx);
const enclosing = findEnclosingFunction(captureMap['assignment'], file.path, ctx, provider);
const funcName = enclosing ? extractFuncNameFromSourceId(enclosing) : '';
receiverTypeName = lookupReceiverType(receiverIndex, funcName, receiverText);
}
@ -417,7 +420,7 @@ export const processCalls = async (
}
}
if (receiverTypeName) {
const enclosing = findEnclosingFunction(captureMap['assignment'], file.path, ctx);
const enclosing = findEnclosingFunction(captureMap['assignment'], file.path, ctx, provider);
const srcId = enclosing || generateId('File', file.path);
// Defer resolution: Ruby attr_accessor properties are registered during
// this same loop, so cross-file lookups fail if the declaring file hasn't
@ -436,7 +439,7 @@ export const processCalls = async (
const calledName = nameNode.text;
const routed = callRouter(calledName, captureMap['call']);
const routed = callRouter?.(calledName, captureMap['call']);
if (routed) {
switch (routed.kind) {
case 'skip':
@ -537,7 +540,7 @@ export const processCalls = async (
}
// Fall back to verified constructor bindings for return type inference
if (!receiverTypeName && receiverName && receiverIndex.size > 0) {
const enclosingFunc = findEnclosingFunction(callNode, file.path, ctx);
const enclosingFunc = findEnclosingFunction(callNode, file.path, ctx, provider);
const funcName = enclosingFunc ? extractFuncNameFromSourceId(enclosingFunc) : '';
receiverTypeName = lookupReceiverType(receiverIndex, funcName, receiverName);
}
@ -553,7 +556,7 @@ export const processCalls = async (
}
}
// Hoist sourceId so it's available for ACCESSES edge emission during chain walk.
const enclosingFuncId = findEnclosingFunction(callNode, file.path, ctx);
const enclosingFuncId = findEnclosingFunction(callNode, file.path, ctx, provider);
const sourceId = enclosingFuncId || generateId('File', file.path);
// Fall back to mixed chain resolution when the receiver is a complex expression
@ -591,7 +594,7 @@ export const processCalls = async (
// Build overload hints for languages with inferLiteralType (Java/Kotlin/C#/C++).
// Only used when multiple candidates survive arity filtering — ~1-3% of calls.
const langConfig = lang ? typeConfigs[lang as keyof typeof typeConfigs] : undefined;
const langConfig = provider.typeConfig;
const hints: OverloadHints | undefined = langConfig?.inferLiteralType
? { callNode, inferLiteralType: langConfig.inferLiteralType }
: undefined;
@ -602,7 +605,7 @@ export const processCalls = async (
callForm,
receiverTypeName,
receiverName,
}, file.path, ctx, hints);
}, file.path, ctx, hints, widenCache);
if (!resolved) return;
const relId = generateId('CALLS', `${sourceId}:${calledName}->${resolved.nodeId}`);
@ -820,11 +823,15 @@ const tryOverloadDisambiguation = (
*
* If filtering still leaves multiple candidates, refuse to emit a CALLS edge.
*/
/** Per-file cache for the widen path's lookupFuzzy calls. Cleared between files. */
type WidenCache = Map<string, readonly SymbolDefinition[]>;
const resolveCallTarget = (
call: Pick<ExtractedCall, 'calledName' | 'argCount' | 'callForm' | 'receiverTypeName' | 'receiverName'>,
currentFile: string,
ctx: ResolutionContext,
overloadHints?: OverloadHints,
widenCache?: WidenCache,
): ResolveResult | null => {
const tiered = ctx.resolve(call.calledName, currentFile);
if (!tiered) return null;
@ -851,16 +858,33 @@ const resolveCallTarget = (
filteredCandidates = filterCallableCandidates(tiered.candidates, call.argCount, 'constructor');
}
// Module-alias disambiguation: Python `import auth; auth.User()` — when both models.py and
// auth.py export User, receiverName='auth' selects auth.py via moduleAliasMap.
// Runs when multiple candidates survive filtering and the receiver is a known module alias.
if (filteredCandidates.length > 1 && call.callForm === 'member' && call.receiverName) {
// Module-alias disambiguation: Python `import auth; auth.User()` — receiverName='auth'
// selects auth.py via moduleAliasMap. Runs for ALL member calls with a known module alias,
// not just ambiguous ones — same-file tier may shadow the correct cross-module target when
// the caller defines a function with the same name as the callee (Issue #417).
if (call.callForm === 'member' && call.receiverName) {
const aliasMap = ctx.moduleAliasMap?.get(currentFile);
if (aliasMap) {
const moduleFile = aliasMap.get(call.receiverName);
if (moduleFile) {
const aliasFiltered = filteredCandidates.filter(c => c.filePath === moduleFile);
if (aliasFiltered.length > 0) filteredCandidates = aliasFiltered;
if (aliasFiltered.length > 0) {
filteredCandidates = aliasFiltered;
} else {
// Same-file tier returned a local match, but the alias points elsewhere.
// Widen to global candidates and filter to the aliased module's file.
// Use per-file widenCache to avoid repeated lookupFuzzy for the same
// calledName+moduleFile from multiple call sites in the same file.
const cacheKey = `${call.calledName}\0${moduleFile}`;
let fuzzyDefs = widenCache?.get(cacheKey);
if (!fuzzyDefs) {
fuzzyDefs = ctx.symbols.lookupFuzzy(call.calledName);
widenCache?.set(cacheKey, fuzzyDefs);
}
const widened = filterCallableCandidates(fuzzyDefs, call.argCount, call.callForm)
.filter(c => c.filePath === moduleFile);
if (widened.length > 0) filteredCandidates = widened;
}
}
}
}
@ -1198,6 +1222,7 @@ export const processCallsFromExtracted = async (
}
ctx.enableCache(filePath);
const widenCache: WidenCache = new Map();
const receiverMap = fileReceiverTypes.get(filePath);
for (const call of calls) {
@ -1251,7 +1276,7 @@ export const processCallsFromExtracted = async (
}
}
const resolved = resolveCallTarget(effectiveCall, effectiveCall.filePath, ctx);
const resolved = resolveCallTarget(effectiveCall, effectiveCall.filePath, ctx, undefined, widenCache);
if (!resolved) continue;
const relId = generateId('CALLS', `${effectiveCall.sourceId}:${effectiveCall.calledName}->${resolved.nodeId}`);
@ -1401,13 +1426,30 @@ export const processRoutesFromExtracted = async (
*/
/** Common method names on response/data objects that are NOT property accesses */
const RESPONSE_METHOD_BLOCKLIST = new Set([
'json', 'text', 'blob', 'arrayBuffer', 'formData', 'ok', 'status', 'headers',
'then', 'catch', 'finally', 'clone',
// Properties/methods to ignore when extracting consumer accessed keys from `data.X` patterns.
// Avoids false positives from Fetch API, Array, Object, Promise, and DOM access on variables
// that happen to share names with response variables (data, result, response, etc.).
const RESPONSE_ACCESS_BLOCKLIST = new Set([
// Fetch/Response API
'json', 'text', 'blob', 'arrayBuffer', 'formData', 'ok', 'status', 'headers', 'clone',
// Promise
'then', 'catch', 'finally',
// Array
'map', 'filter', 'forEach', 'reduce', 'find', 'some', 'every',
'length', 'toString', 'valueOf',
'push', 'pop', 'shift', 'unshift', 'splice', 'slice', 'concat', 'join',
'sort', 'reverse', 'includes', 'indexOf', 'keys', 'values', 'entries',
'sort', 'reverse', 'includes', 'indexOf',
// Object
'length', 'toString', 'valueOf', 'keys', 'values', 'entries',
// DOM methods — file-download patterns often reuse `data`/`response` variable names
'appendChild', 'removeChild', 'insertBefore', 'replaceChild', 'replaceChildren',
'createElement', 'getElementById', 'querySelector', 'querySelectorAll',
'setAttribute', 'getAttribute', 'removeAttribute', 'hasAttribute',
'addEventListener', 'removeEventListener', 'dispatchEvent',
'classList', 'className',
'parentNode', 'parentElement', 'childNodes', 'children',
'nextSibling', 'previousSibling', 'firstChild', 'lastChild',
'click', 'focus', 'blur', 'submit', 'reset',
'innerHTML', 'outerHTML', 'textContent', 'innerText',
]);
export const extractConsumerAccessedKeys = (content: string): string[] => {
@ -1446,7 +1488,7 @@ export const extractConsumerAccessedKeys = (content: string): string[] => {
while ((match = propAccessPattern.exec(content)) !== null) {
const key = match[1];
// Skip common method calls that aren't property accesses
if (!RESPONSE_METHOD_BLOCKLIST.has(key)) {
if (!RESPONSE_ACCESS_BLOCKLIST.has(key)) {
keys.add(key);
}
}
@ -1536,7 +1578,8 @@ export const extractFetchCallsFromFiles = async (
if (!language) continue;
if (!isLanguageAvailable(language)) continue;
const queryStr = LANGUAGE_QUERIES[language];
const provider = getProvider(language);
const queryStr = provider.treeSitterQueries;
if (!queryStr) continue;
await loadLanguage(language, file.path);

View file

@ -11,7 +11,7 @@
* Keep both copies in sync until a shared package is introduced.
*/
import { SupportedLanguages } from '../../config/supported-languages.js';
import type { SyntaxNode } from './utils/ast-helpers.js';
// ── Call routing dispatch table ─────────────────────────────────────────────
@ -26,29 +26,9 @@ export type CallRoutingResult = RubyCallRouting | null;
*/
export type CallRouter = (
calledName: string,
callNode: any,
callNode: SyntaxNode,
) => CallRoutingResult;
/** No-op router: returns null for every call (passthrough to normal processing) */
const noRouting: CallRouter = () => null;
/** Per-language call routing. noRouting = no special routing (normal call processing) */
export const callRouters = {
[SupportedLanguages.JavaScript]: noRouting,
[SupportedLanguages.TypeScript]: noRouting,
[SupportedLanguages.Python]: noRouting,
[SupportedLanguages.Java]: noRouting,
[SupportedLanguages.Kotlin]: noRouting,
[SupportedLanguages.Go]: noRouting,
[SupportedLanguages.Rust]: noRouting,
[SupportedLanguages.CSharp]: noRouting,
[SupportedLanguages.PHP]: noRouting,
[SupportedLanguages.Swift]: noRouting,
[SupportedLanguages.CPlusPlus]: noRouting,
[SupportedLanguages.C]: noRouting,
[SupportedLanguages.Ruby]: routeRubyCall,
} satisfies Record<SupportedLanguages, CallRouter>;
// ── Result types ────────────────────────────────────────────────────────────
export type RubyCallRouting =
@ -91,7 +71,7 @@ const MAX_PARENT_DEPTH = 50;
* @param callNode - The tree-sitter `call` AST node
* @returns A discriminated union describing the call's semantic role
*/
export function routeRubyCall(calledName: string, callNode: any): RubyCallRouting {
export function routeRubyCall(calledName: string, callNode: SyntaxNode): RubyCallRouting {
// ── require / require_relative → import ─────────────────────────────────
if (calledName === 'require' || calledName === 'require_relative') {
const argList = callNode.childForFieldName?.('arguments');

View file

@ -40,7 +40,7 @@ const UNIVERSAL_ENTRY_POINT_PATTERNS: RegExp[] = [
/^emit[A-Z]/, // emitEvent
];
const ENTRY_POINT_PATTERNS = {
export const ENTRY_POINT_PATTERNS = {
// JavaScript/TypeScript
[SupportedLanguages.JavaScript]: [
/^use[A-Z]/, // React hooks (useEffect, etc.)
@ -216,9 +216,9 @@ const ENTRY_POINT_PATTERNS = {
/** Pre-computed merged patterns (universal + language-specific) to avoid per-call array allocation. */
const MERGED_ENTRY_POINT_PATTERNS = Object.fromEntries(
(Object.keys(ENTRY_POINT_PATTERNS) as SupportedLanguages[]).map(lang => [
Object.values(SupportedLanguages).map(lang => [
lang,
[...UNIVERSAL_ENTRY_POINT_PATTERNS, ...ENTRY_POINT_PATTERNS[lang]],
[...UNIVERSAL_ENTRY_POINT_PATTERNS, ...(ENTRY_POINT_PATTERNS[lang] ?? [])],
])
) as Record<SupportedLanguages, RegExp[]>;

View file

@ -7,18 +7,17 @@
* Shared between parse-worker.ts (worker pool) and parsing-processor.ts (sequential fallback).
*/
import { findSiblingChild, SyntaxNode } from './utils.js';
import { SupportedLanguages } from '../../config/supported-languages.js';
import { findSiblingChild, type SyntaxNode } from './utils/ast-helpers.js';
/** Handler type: given a node and symbol name, return true if the symbol is exported/public. */
type ExportChecker = (node: SyntaxNode, name: string) => boolean;
export type ExportChecker = (node: SyntaxNode, name: string) => boolean;
// ============================================================================
// Per-language export checkers
// ============================================================================
/** JS/TS: walk ancestors looking for export_statement or export_specifier. */
const tsExportChecker: ExportChecker = (node, _name) => {
export const tsExportChecker: ExportChecker = (node, _name) => {
let current: SyntaxNode | null = node;
while (current) {
const type = current.type;
@ -37,10 +36,10 @@ const tsExportChecker: ExportChecker = (node, _name) => {
};
/** Python: public if no leading underscore (convention). */
const pythonExportChecker: ExportChecker = (_node, name) => !name.startsWith('_');
export const pythonExportChecker: ExportChecker = (_node, name) => !name.startsWith('_');
/** Java: check for 'public' modifier — modifiers are siblings of the name node, not parents. */
const javaExportChecker: ExportChecker = (node, _name) => {
export const javaExportChecker: ExportChecker = (node, _name) => {
let current: SyntaxNode | null = node;
while (current) {
if (current.parent) {
@ -76,7 +75,7 @@ const CSHARP_DECL_TYPES = new Set([
* C#: modifier nodes are SIBLINGS of the name node inside the declaration.
* Walk up to the declaration node, then scan its direct children.
*/
const csharpExportChecker: ExportChecker = (node, _name) => {
export const csharpExportChecker: ExportChecker = (node, _name) => {
let current: SyntaxNode | null = node;
while (current) {
if (CSHARP_DECL_TYPES.has(current.type)) {
@ -92,7 +91,7 @@ const csharpExportChecker: ExportChecker = (node, _name) => {
};
/** Go: uppercase first letter = exported. */
const goExportChecker: ExportChecker = (_node, name) => {
export const goExportChecker: ExportChecker = (_node, name) => {
if (name.length === 0) return false;
const first = name[0];
return first === first.toUpperCase() && first !== first.toLowerCase();
@ -110,7 +109,7 @@ const RUST_DECL_TYPES = new Set([
* (function_item, struct_item, etc.), not a parent. Walk up to the declaration node,
* then scan its direct children.
*/
const rustExportChecker: ExportChecker = (node, _name) => {
export const rustExportChecker: ExportChecker = (node, _name) => {
let current: SyntaxNode | null = node;
while (current) {
if (RUST_DECL_TYPES.has(current.type)) {
@ -129,7 +128,7 @@ const rustExportChecker: ExportChecker = (node, _name) => {
* Kotlin: default visibility is public (unlike Java).
* visibility_modifier is inside modifiers, a sibling of the name node within the declaration.
*/
const kotlinExportChecker: ExportChecker = (node, _name) => {
export const kotlinExportChecker: ExportChecker = (node, _name) => {
let current: SyntaxNode | null = node;
while (current) {
if (current.parent) {
@ -152,7 +151,7 @@ const kotlinExportChecker: ExportChecker = (node, _name) => {
* marked 'static' are file-scoped (not exported). C++ anonymous namespaces
* (namespace { ... }) also give internal linkage.
*/
const cCppExportChecker: ExportChecker = (node, _name) => {
export const cCppExportChecker: ExportChecker = (node, _name) => {
let cur: SyntaxNode | null = node;
while (cur) {
if (cur.type === 'function_definition' || cur.type === 'declaration') {
@ -174,7 +173,7 @@ const cCppExportChecker: ExportChecker = (node, _name) => {
};
/** PHP: check for visibility modifier or top-level scope. */
const phpExportChecker: ExportChecker = (node, _name) => {
export const phpExportChecker: ExportChecker = (node, _name) => {
let current: SyntaxNode | null = node;
while (current) {
if (current.type === 'class_declaration' ||
@ -200,7 +199,7 @@ const phpExportChecker: ExportChecker = (node, _name) => {
* `internal` symbols should be treated as exported (cross-file visible).
* Only `private` and `fileprivate` symbols are truly file-scoped.
*/
const swiftExportChecker: ExportChecker = (node, _name) => {
export const swiftExportChecker: ExportChecker = (node, _name) => {
let current: SyntaxNode | null = node;
while (current) {
if (current.type === 'modifiers' || current.type === 'visibility_modifier') {
@ -215,39 +214,6 @@ const swiftExportChecker: ExportChecker = (node, _name) => {
return true;
};
// ============================================================================
// Exhaustive dispatch table — satisfies enforces all SupportedLanguages are covered
// ============================================================================
/** Ruby: all top-level definitions are public (no export syntax). */
export const rubyExportChecker: ExportChecker = (_node, _name) => true;
const exportCheckers = {
[SupportedLanguages.JavaScript]: tsExportChecker,
[SupportedLanguages.TypeScript]: tsExportChecker,
[SupportedLanguages.Python]: pythonExportChecker,
[SupportedLanguages.Java]: javaExportChecker,
[SupportedLanguages.CSharp]: csharpExportChecker,
[SupportedLanguages.Go]: goExportChecker,
[SupportedLanguages.Rust]: rustExportChecker,
[SupportedLanguages.Kotlin]: kotlinExportChecker,
[SupportedLanguages.C]: cCppExportChecker,
[SupportedLanguages.CPlusPlus]: cCppExportChecker,
[SupportedLanguages.PHP]: phpExportChecker,
[SupportedLanguages.Swift]: swiftExportChecker,
[SupportedLanguages.Ruby]: (_node, _name) => true,
} satisfies Record<SupportedLanguages, ExportChecker>;
// ============================================================================
// Public API
// ============================================================================
/**
* Check if a tree-sitter node is exported/public in its language.
* @param node - The tree-sitter AST node
* @param name - The symbol name
* @param language - The programming language
* @returns true if the symbol is exported/public
*/
export const isNodeExported = (node: SyntaxNode, name: string, language: SupportedLanguages): boolean => {
const checker = exportCheckers[language];
if (!checker) return false;
return checker(node, name);
};

View file

@ -62,10 +62,25 @@ export function detectFrameworkFromPath(filePath: string): FrameworkHint | null
}
// Next.js - Layout files (moderate - they're entry-ish but not the main entry)
if (p.includes('/app/') && (p.endsWith('layout.tsx') || p.endsWith('layout.ts'))) {
if (p.includes('/app/') && !p.includes('_layout') && (p.endsWith('layout.tsx') || p.endsWith('layout.ts'))) {
return { framework: 'nextjs-app', entryPointMultiplier: 2.0, reason: 'nextjs-layout' };
}
// Expo Router - screen/layout/api files in app/ directory
if (p.includes('/app/') && (p.endsWith('.tsx') || p.endsWith('.ts') || p.endsWith('.jsx') || p.endsWith('.js'))) {
const fn = p.split('/').pop() || '';
if (fn.startsWith('_layout')) {
return { framework: 'expo-router', entryPointMultiplier: 2.0, reason: 'expo-layout' };
}
if (fn.startsWith('+') && !fn.startsWith('+api')) {
return { framework: 'expo-router', entryPointMultiplier: 1.5, reason: 'expo-special-route' };
}
if (fn.endsWith('+api.ts') || fn.endsWith('+api.tsx')) {
return { framework: 'expo-router', entryPointMultiplier: 3.0, reason: 'expo-api-route' };
}
return { framework: 'expo-router', entryPointMultiplier: 2.5, reason: 'expo-screen' };
}
// Express / Node.js routes
if (p.includes('/routes/') && (p.endsWith('.ts') || p.endsWith('.js'))) {
return { framework: 'express', entryPointMultiplier: 2.5, reason: 'routes-folder' };
@ -416,6 +431,7 @@ export function detectFrameworkFromPath(filePath: string): FrameworkHint | null
export const FRAMEWORK_AST_PATTERNS = {
// JavaScript/TypeScript decorators
'nestjs': ['@Controller', '@Get', '@Post', '@Put', '@Delete', '@Patch'],
'expo-router': ['router.push', 'router.replace', 'router.navigate', 'useRouter', 'useLocalSearchParams', 'useSegments', 'expo-router'],
'express': ['app.get', 'app.post', 'app.put', 'app.delete', 'router.get', 'router.post'],
// Python decorators
@ -471,12 +487,14 @@ interface AstFrameworkPatternConfig {
patterns: string[];
}
const AST_FRAMEWORK_PATTERNS_BY_LANGUAGE = {
export const AST_FRAMEWORK_PATTERNS_BY_LANGUAGE = {
[SupportedLanguages.JavaScript]: [
{ framework: 'nestjs', entryPointMultiplier: 3.2, reason: 'nestjs-decorator', patterns: FRAMEWORK_AST_PATTERNS.nestjs },
{ framework: 'expo-router', entryPointMultiplier: 2.5, reason: 'expo-router-navigation', patterns: FRAMEWORK_AST_PATTERNS['expo-router'] },
],
[SupportedLanguages.TypeScript]: [
{ framework: 'nestjs', entryPointMultiplier: 3.2, reason: 'nestjs-decorator', patterns: FRAMEWORK_AST_PATTERNS.nestjs },
{ framework: 'expo-router', entryPointMultiplier: 2.5, reason: 'expo-router-navigation', patterns: FRAMEWORK_AST_PATTERNS['expo-router'] },
],
[SupportedLanguages.Python]: [
{ framework: 'fastapi', entryPointMultiplier: 3.0, reason: 'fastapi-decorator', patterns: FRAMEWORK_AST_PATTERNS.fastapi },

View file

@ -18,24 +18,23 @@ import { KnowledgeGraph } from '../graph/types.js';
import { ASTCache } from './ast-cache.js';
import Parser from 'tree-sitter';
import { isLanguageAvailable, loadParser, loadLanguage } from '../tree-sitter/parser-loader.js';
import { LANGUAGE_QUERIES } from './tree-sitter-queries.js';
import { generateId } from '../../lib/utils.js';
import { getLanguageFromFilename, isVerboseIngestionEnabled, yieldToEventLoop } from './utils.js';
import { getLanguageFromFilename } from './utils/language-detection.js';
import { isVerboseIngestionEnabled } from './utils/verbose.js';
import { yieldToEventLoop } from './utils/event-loop.js';
import { SupportedLanguages } from '../../config/supported-languages.js';
import { getProvider } from './languages/index.js';
import { getTreeSitterBufferSize } from './constants.js';
import type { ExtractedHeritage } from './workers/parse-worker.js';
import type { ResolutionContext } from './resolution-context.js';
import { TIER_CONFIDENCE } from './resolution-context.js';
/** C#/Java convention: interfaces start with I followed by an uppercase letter */
const INTERFACE_NAME_RE = /^I[A-Z]/;
/**
* Determine whether a heritage.extends capture is actually an IMPLEMENTS relationship.
* Uses the symbol table first (authoritative — Tier 1); falls back to a language-gated
* heuristic for external symbols not present in the graph:
* - C# / Java: `I[A-Z]` naming convention
* - Swift: default IMPLEMENTS (protocol conformance is the norm)
* Uses the symbol table first (authoritative — Tier 1); falls back to provider-defined
* heuristics for external symbols not present in the graph:
* - interfaceNamePattern: matched against parent name (e.g., /^I[A-Z]/ for C#/Java)
* - heritageDefaultEdge: 'IMPLEMENTS' causes all unresolved parents to map to IMPLEMENTS
* - All others: default EXTENDS
*/
const resolveExtendsType = (
@ -51,13 +50,12 @@ const resolveExtendsType = (
? { type: 'IMPLEMENTS', idPrefix: 'Interface' }
: { type: 'EXTENDS', idPrefix: 'Class' };
}
// Unresolved symbol — fall back to language-specific heuristic
if (language === SupportedLanguages.CSharp || language === SupportedLanguages.Java) {
if (INTERFACE_NAME_RE.test(parentName)) {
return { type: 'IMPLEMENTS', idPrefix: 'Interface' };
}
} else if (language === SupportedLanguages.Swift) {
// Protocol conformance is far more common than class inheritance in Swift
// Unresolved symbol — fall back to provider-defined heuristics
const provider = getProvider(language);
if (provider.interfaceNamePattern?.test(parentName)) {
return { type: 'IMPLEMENTS', idPrefix: 'Interface' };
}
if (provider.heritageDefaultEdge === 'IMPLEMENTS') {
return { type: 'IMPLEMENTS', idPrefix: 'Interface' };
}
return { type: 'EXTENDS', idPrefix: 'Class' };
@ -117,7 +115,8 @@ export const processHeritage = async (
continue;
}
const queryStr = LANGUAGE_QUERIES[language];
const provider = getProvider(language);
const queryStr = provider.treeSitterQueries;
if (!queryStr) continue;
// 2. Load the language

View file

@ -2,29 +2,22 @@ import { KnowledgeGraph } from '../graph/types.js';
import { ASTCache } from './ast-cache.js';
import Parser from 'tree-sitter';
import { isLanguageAvailable, loadParser, loadLanguage } from '../tree-sitter/parser-loader.js';
import { LANGUAGE_QUERIES } from './tree-sitter-queries.js';
import { getProvider, getProviderForFile, providersWithImplicitWiring } from './languages/index.js';
import type { LanguageProvider } from './language-provider.js';
import { generateId } from '../../lib/utils.js';
import { getLanguageFromFilename, isVerboseIngestionEnabled, yieldToEventLoop } from './utils.js';
import { SupportedLanguages } from '../../config/supported-languages.js';
import type { SwiftPackageConfig } from './language-config.js';
import { getLanguageFromFilename } from './utils/language-detection.js';
import { isVerboseIngestionEnabled } from './utils/verbose.js';
import { yieldToEventLoop } from './utils/event-loop.js';
import type { ExtractedImport } from './workers/parse-worker.js';
import { getTreeSitterBufferSize } from './constants.js';
import { loadImportConfigs } from './language-config.js';
import { buildSuffixIndex } from './resolvers/index.js';
import { callRouters } from './call-routing.js';
import { buildSuffixIndex } from './import-resolvers/utils.js';
import type { ResolutionContext, ModuleAliasMap } from './resolution-context.js';
import type { SuffixIndex } from './resolvers/index.js';
import { importResolvers, namedBindingExtractors, preprocessImportPath } from './import-resolution.js';
import type { ImportResult, ResolveCtx, NamedBinding } from './import-resolution.js';
import type { SuffixIndex } from './import-resolvers/utils.js';
import type { ImportResult, ResolveCtx, ImportResolutionContext } from './import-resolvers/types.js';
import type { NamedBinding } from './named-bindings/types.js';
import type { SyntaxNode } from './utils/ast-helpers.js';
// Re-export resolver types for consumers
export type {
SuffixIndex,
TsconfigPaths,
GoModuleConfig,
CSharpProjectConfig,
ComposerConfig
} from './resolvers/index.js';
const isDev = process.env.NODE_ENV === 'development';
@ -32,6 +25,32 @@ const isDev = process.env.NODE_ENV === 'development';
// Stores all files that a given file imports from
export type ImportMap = Map<string, Set<string>>;
/** Group files by provider (only those with implicit import wiring), then call each wirer
* with its own language's files. O(n) over files, O(1) per provider lookup. */
function wireImplicitImports(
files: string[],
importMap: Map<string, Set<string>>,
addImportEdge: (src: string, target: string) => void,
projectConfig: unknown,
): void {
if (providersWithImplicitWiring.length === 0) return;
const grouped = new Map<LanguageProvider, string[]>();
for (const file of files) {
const provider = getProviderForFile(file);
if (!provider?.implicitImportWirer) continue;
let list = grouped.get(provider);
if (!list) { list = []; grouped.set(provider, list); }
list.push(file);
}
for (const [provider, langFiles] of grouped) {
if (langFiles.length > 1) {
provider.implicitImportWirer(langFiles, importMap, addImportEdge, projectConfig);
}
}
}
// Type: Map<FilePath, Set<PackageDirSuffix>>
// Stores Go package directory suffixes imported by a file (e.g., "/internal/auth/").
// Avoids expanding every Go package import into N individual ImportMap edges.
@ -58,14 +77,7 @@ export function isFileInPackageDir(filePath: string, dirSuffix: string): boolean
return !afterDir.includes('/');
}
/** Pre-built lookup structures for import resolution. Build once, reuse across chunks. */
export interface ImportResolutionContext {
allFilePaths: Set<string>;
allFileList: string[];
normalizedFileList: string[];
index: SuffixIndex;
resolveCache: Map<string, string | null>;
}
// ImportResolutionContext is defined in ./import-resolvers/types.ts — re-exported here for consumers.
export function buildImportResolutionContext(allPaths: string[]): ImportResolutionContext {
const allFileList = allPaths;
@ -76,7 +88,30 @@ export function buildImportResolutionContext(allPaths: string[]): ImportResoluti
}
// Config loaders extracted to ./language-config.ts (Phase 2 refactor)
// Resolver dispatch tables are in ./import-resolution.ts — imported above
// Resolver types are in ./import-resolvers/types.ts; named binding types in ./named-bindings/types.ts
// ============================================================================
// Import path preprocessing
// ============================================================================
/**
* Clean and preprocess a raw import source text into a resolved import path.
* Strips quotes/angle brackets (universal) and applies provider-specific
* transformations (currently only Kotlin wildcard import detection).
*/
export function preprocessImportPath(
sourceText: string,
importNode: SyntaxNode,
provider: LanguageProvider,
): string | null {
const cleaned = sourceText.replace(/['"<>]/g, '');
// Defense-in-depth: reject null bytes and control characters (matches Ruby call-routing pattern)
if (!cleaned || cleaned.length > 2048 || /[\x00-\x1f]/.test(cleaned)) return null;
if (provider.importPathPreprocessor) {
return provider.importPathPreprocessor(cleaned, importNode);
}
return cleaned;
}
/** Create IMPORTS edge helpers that share a resolved-count tracker. */
function createImportEdgeHelpers(graph: KnowledgeGraph, importMap: ImportMap) {
@ -99,82 +134,6 @@ function createImportEdgeHelpers(graph: KnowledgeGraph, importMap: ImportMap) {
return { addImportEdge, addImportGraphEdge, getResolvedCount: () => totalImportsResolved };
}
/**
* Group Swift files by target for implicit module visibility.
*
* If SwiftPackageConfig is available, use SPM target → directory mappings.
* Otherwise, group all Swift files under a single "default" target
* (assumes a single-module Xcode project).
*/
function groupSwiftFilesByTarget(
swiftFiles: string[],
swiftPackageConfig: SwiftPackageConfig | null,
): Map<string, string[]> {
const groups = new Map<string, string[]>();
if (swiftPackageConfig && swiftPackageConfig.targets.size > 0) {
for (const file of swiftFiles) {
const normalized = file.replace(/\\/g, '/');
let assigned = false;
for (const [targetName, targetDir] of swiftPackageConfig.targets) {
const dirPrefix = targetDir + '/';
const idx = normalized.indexOf(dirPrefix);
if (idx === 0 || (idx > 0 && normalized[idx - 1] === '/')) {
if (!groups.has(targetName)) groups.set(targetName, []);
groups.get(targetName)!.push(file);
assigned = true;
break;
}
}
if (!assigned) {
if (!groups.has('__default__')) groups.set('__default__', []);
groups.get('__default__')!.push(file);
}
}
} else {
groups.set('__default__', [...swiftFiles]);
}
return groups;
}
/**
* Add implicit IMPORTS edges between all Swift files in the same module/target.
* Swift has no file-level imports — all files in a module see each other.
*/
function addSwiftImplicitImports(
files: string[] | { path: string }[],
swiftPackageConfig: SwiftPackageConfig | null,
importMap: Map<string, Set<string>>,
addImportEdge: (src: string, target: string) => void,
logSuffix = '',
): void {
const paths = typeof files[0] === 'string'
? files as string[]
: (files as { path: string }[]).map(f => f.path);
const swiftFiles = paths
.filter(f => getLanguageFromFilename(f) === SupportedLanguages.Swift);
if (swiftFiles.length <= 1) return;
const targetGroups = groupSwiftFilesByTarget(swiftFiles, swiftPackageConfig);
for (const group of targetGroups.values()) {
for (const srcFile of group) {
const existing = importMap.get(srcFile);
for (const otherFile of group) {
if (srcFile === otherFile) continue;
if (existing?.has(otherFile)) continue;
addImportEdge(srcFile, otherFile);
}
}
}
if (isDev) {
console.log(`📊 Swift: ${swiftFiles.length} files in ${targetGroups.size} target group(s), implicit imports added${logSuffix}`);
}
}
/**
* Apply an ImportResult: emit graph edges and update ImportMap/PackageMap.
* If namedBindings are provided and the import resolves to a single file,
@ -320,7 +279,8 @@ export const processImports = async (
continue;
}
const queryStr = LANGUAGE_QUERIES[language];
const provider = getProvider(language);
const queryStr = provider.treeSitterQueries;
if (!queryStr) continue;
// 2. ALWAYS load the language before querying (parser is stateful)
@ -376,12 +336,12 @@ export const processImports = async (
return;
}
const rawImportPath = preprocessImportPath(sourceNode.text, captureMap['import'], language);
const rawImportPath = preprocessImportPath(sourceNode.text, captureMap['import'], provider);
if (!rawImportPath) return;
totalImportsFound++;
const result = importResolvers[language](rawImportPath, file.path, resolveCtx);
const extractor = namedBindingExtractors[language];
const result = provider.importResolver(rawImportPath, file.path, resolveCtx);
const extractor = provider.namedBindingExtractor;
const bindings = namedImportMap && extractor ? extractor(captureMap['import']) : undefined;
applyImportResult(result, file.path, importMap, packageMap, addImportEdge, addImportGraphEdge, bindings, namedImportMap, moduleAliasMap);
}
@ -390,11 +350,10 @@ export const processImports = async (
if (captureMap['call']) {
const callNameNode = captureMap['call.name'];
if (callNameNode) {
const callRouter = callRouters[language];
const routed = callRouter(callNameNode.text, captureMap['call']);
const routed = provider.callRouter?.(callNameNode.text, captureMap['call']);
if (routed && routed.kind === 'import') {
totalImportsFound++;
const result = importResolvers[language](routed.importPath, file.path, resolveCtx);
const result = provider.importResolver(routed.importPath, file.path, resolveCtx);
applyImportResult(result, file.path, importMap, packageMap, addImportEdge, addImportGraphEdge);
}
}
@ -404,7 +363,7 @@ export const processImports = async (
// Tree is now owned by the LRU cache — no manual delete needed
}
addSwiftImplicitImports(allFileList, configs.swiftPackageConfig, importMap, addImportEdge);
wireImplicitImports(allFileList, importMap, addImportEdge, configs);
if (skippedByLang && skippedByLang.size > 0) {
for (const [lang, count] of skippedByLang.entries()) {
@ -469,14 +428,15 @@ export const processImportsFromExtracted = async (
for (const imp of fileImports) {
totalImportsFound++;
const result = importResolvers[imp.language](imp.rawImportPath, filePath, resolveCtx);
const provider = getProvider(imp.language);
const result = provider.importResolver(imp.rawImportPath, filePath, resolveCtx);
applyImportResult(result, filePath, importMap, packageMap, addImportEdge, addImportGraphEdge, imp.namedBindings, namedImportMap, moduleAliasMap);
}
}
onProgress?.(totalFiles, totalFiles);
addSwiftImplicitImports(files, configs.swiftPackageConfig, importMap, addImportEdge, ' (fast path)');
wireImplicitImports(files.map(f => f.path), importMap, addImportEdge, configs);
if (isDev) {
console.log(`📊 Import processing (fast path): ${getResolvedCount()}/${totalImportsFound} imports resolved to graph edges`);

View file

@ -1,385 +0,0 @@
/**
* Import Resolution Dispatch
*
* Per-language dispatch table for import resolution and named binding extraction.
* Replaces the 120-line if-chain in resolveLanguageImport() and the 7-branch
* dispatch in extractNamedBindings() with a single table lookup each.
*
* Follows the existing ExportChecker / CallRouter pattern:
* - Function aliases (not interfaces) to avoid megamorphic inline-cache issues
* - `satisfies Record<SupportedLanguages, ...>` for compile-time exhaustiveness
* - Const dispatch table — configs are accessed via ctx.configs at call time
*/
import { SupportedLanguages } from '../../config/supported-languages.js';
import type { SyntaxNode } from './utils.js';
import {
KOTLIN_EXTENSIONS,
appendKotlinWildcard,
resolveJvmWildcard,
resolveJvmMemberImport,
resolveGoPackageDir,
resolveGoPackage,
resolveCSharpImport as resolveCSharpImportHelper,
resolveCSharpNamespaceDir,
resolvePhpImport as resolvePhpImportHelper,
resolveRustImport as resolveRustImportHelper,
resolveRubyImport as resolveRubyImportHelper,
resolvePythonImport as resolvePythonImportHelper,
resolveImportPath,
} from './resolvers/index.js';
import type {
SuffixIndex,
TsconfigPaths,
GoModuleConfig,
CSharpProjectConfig,
ComposerConfig,
} from './resolvers/index.js';
import type { SwiftPackageConfig } from './language-config.js';
import {
extractTsNamedBindings,
extractPythonNamedBindings,
extractKotlinNamedBindings,
extractRustNamedBindings,
extractPhpNamedBindings,
extractCsharpNamedBindings,
extractJavaNamedBindings,
} from './named-binding-extraction.js';
import type { ImportResolutionContext } from './import-processor.js';
// ============================================================================
// Types
// ============================================================================
/**
* Result of resolving an import via language-specific dispatch.
* - 'files': resolved to one or more files -> add to ImportMap
* - 'package': resolved to a directory -> add graph edges + store dirSuffix in PackageMap
* - null: no resolution (external dependency, etc.)
*/
export type ImportResult =
| { kind: 'files'; files: string[] }
| { kind: 'package'; files: string[]; dirSuffix: string }
| null;
/** Bundled language-specific configs loaded once per ingestion run. */
export interface ImportConfigs {
tsconfigPaths: TsconfigPaths | null;
goModule: GoModuleConfig | null;
composerConfig: ComposerConfig | null;
swiftPackageConfig: SwiftPackageConfig | null;
csharpConfigs: CSharpProjectConfig[];
}
/** Full context for import resolution: file lookups + language configs. */
export interface ResolveCtx extends ImportResolutionContext {
configs: ImportConfigs;
}
/** Per-language import resolver -- function alias matching ExportChecker/CallRouter pattern. */
export type ImportResolverFn = (
rawImportPath: string,
filePath: string,
resolveCtx: ResolveCtx,
) => ImportResult;
/** A single named import binding: local name in the importing file and exported name from the source.
* When `isModuleAlias` is true, the binding represents a Python `import X as Y` module alias
* and is routed to moduleAliasMap instead of namedImportMap during import processing. */
export interface NamedBinding { local: string; exported: string; isModuleAlias?: boolean }
/** Per-language named binding extractor -- optional (returns undefined if language has no named imports). */
type NamedBindingExtractorFn = (importNode: SyntaxNode) => NamedBinding[] | undefined;
// ============================================================================
// Import path preprocessing
// ============================================================================
/**
* Clean and preprocess a raw import source text into a resolved import path.
* Strips quotes/angle brackets (universal) and applies language-specific
* transformations (currently only Kotlin wildcard import detection).
*/
export function preprocessImportPath(
sourceText: string,
importNode: SyntaxNode,
language: SupportedLanguages,
): string | null {
const cleaned = sourceText.replace(/['"<>]/g, '');
// Defense-in-depth: reject null bytes and control characters (matches Ruby call-routing pattern)
if (!cleaned || cleaned.length > 2048 || /[\x00-\x1f]/.test(cleaned)) return null;
if (language === SupportedLanguages.Kotlin) {
return appendKotlinWildcard(cleaned, importNode);
}
return cleaned;
}
// ============================================================================
// Per-language resolver functions
// ============================================================================
/**
* Standard single-file resolution (TS/JS/C/C++ and fallback for other languages).
* Handles relative imports, tsconfig path aliases, and suffix matching.
*/
function resolveStandard(
rawImportPath: string,
filePath: string,
ctx: ResolveCtx,
language: SupportedLanguages,
): ImportResult {
const resolvedPath = resolveImportPath(
filePath,
rawImportPath,
ctx.allFilePaths,
ctx.allFileList,
ctx.normalizedFileList,
ctx.resolveCache,
language,
ctx.configs.tsconfigPaths,
ctx.index,
);
return resolvedPath ? { kind: 'files', files: [resolvedPath] } : null;
}
/** Java: JVM wildcard -> member import -> standard fallthrough */
function resolveJavaImport(
rawImportPath: string,
filePath: string,
ctx: ResolveCtx,
): ImportResult {
if (rawImportPath.endsWith('.*')) {
const matchedFiles = resolveJvmWildcard(rawImportPath, ctx.normalizedFileList, ctx.allFileList, ['.java'], ctx.index);
if (matchedFiles.length > 0) return { kind: 'files', files: matchedFiles };
} else {
const memberResolved = resolveJvmMemberImport(rawImportPath, ctx.normalizedFileList, ctx.allFileList, ['.java'], ctx.index);
if (memberResolved) return { kind: 'files', files: [memberResolved] };
}
return resolveStandard(rawImportPath, filePath, ctx, SupportedLanguages.Java);
}
/**
* Kotlin: JVM wildcard/member with Java-interop fallback -> top-level function imports -> standard.
* Kotlin can import from .kt/.kts files OR from .java files (Java interop).
*/
function resolveKotlinImport(
rawImportPath: string,
filePath: string,
ctx: ResolveCtx,
): ImportResult {
if (rawImportPath.endsWith('.*')) {
const matchedFiles = resolveJvmWildcard(rawImportPath, ctx.normalizedFileList, ctx.allFileList, KOTLIN_EXTENSIONS, ctx.index);
if (matchedFiles.length === 0) {
const javaMatches = resolveJvmWildcard(rawImportPath, ctx.normalizedFileList, ctx.allFileList, ['.java'], ctx.index);
if (javaMatches.length > 0) return { kind: 'files', files: javaMatches };
}
if (matchedFiles.length > 0) return { kind: 'files', files: matchedFiles };
} else {
let memberResolved = resolveJvmMemberImport(rawImportPath, ctx.normalizedFileList, ctx.allFileList, KOTLIN_EXTENSIONS, ctx.index);
if (!memberResolved) {
memberResolved = resolveJvmMemberImport(rawImportPath, ctx.normalizedFileList, ctx.allFileList, ['.java'], ctx.index);
}
if (memberResolved) return { kind: 'files', files: [memberResolved] };
// Kotlin: top-level function imports (e.g. import models.getUser) have only 2 segments,
// which resolveJvmMemberImport skips (requires >=3). Fall back to package-directory scan
// for lowercase last segments (function/property imports). Uppercase last segments
// (class imports like models.User) fall through to standard suffix resolution.
const segments = rawImportPath.split('.');
const lastSeg = segments[segments.length - 1];
if (segments.length >= 2 && lastSeg[0] && lastSeg[0] === lastSeg[0].toLowerCase()) {
const pkgWildcard = segments.slice(0, -1).join('.') + '.*';
let dirFiles = resolveJvmWildcard(pkgWildcard, ctx.normalizedFileList, ctx.allFileList, KOTLIN_EXTENSIONS, ctx.index);
if (dirFiles.length === 0) {
dirFiles = resolveJvmWildcard(pkgWildcard, ctx.normalizedFileList, ctx.allFileList, ['.java'], ctx.index);
}
if (dirFiles.length > 0) return { kind: 'files', files: dirFiles };
}
}
return resolveStandard(rawImportPath, filePath, ctx, SupportedLanguages.Kotlin);
}
/** Go: package-level imports via go.mod module path. */
function resolveGoImport(
rawImportPath: string,
filePath: string,
ctx: ResolveCtx,
): ImportResult {
const goModule = ctx.configs.goModule;
if (goModule && rawImportPath.startsWith(goModule.modulePath)) {
const pkgSuffix = resolveGoPackageDir(rawImportPath, goModule);
if (pkgSuffix) {
const pkgFiles = resolveGoPackage(rawImportPath, goModule, ctx.normalizedFileList, ctx.allFileList);
if (pkgFiles.length > 0) {
return { kind: 'package', files: pkgFiles, dirSuffix: pkgSuffix };
}
}
// Fall through if no files found (package might be external)
}
return resolveStandard(rawImportPath, filePath, ctx, SupportedLanguages.Go);
}
/** C#: namespace-based resolution via .csproj configs, with suffix-match fallback. */
function resolveCSharpImportDispatch(
rawImportPath: string,
filePath: string,
ctx: ResolveCtx,
): ImportResult {
const csharpConfigs = ctx.configs.csharpConfigs;
if (csharpConfigs.length > 0) {
const resolvedFiles = resolveCSharpImportHelper(rawImportPath, csharpConfigs, ctx.normalizedFileList, ctx.allFileList, ctx.index);
if (resolvedFiles.length > 1) {
const dirSuffix = resolveCSharpNamespaceDir(rawImportPath, csharpConfigs);
if (dirSuffix) {
return { kind: 'package', files: resolvedFiles, dirSuffix };
}
}
if (resolvedFiles.length > 0) return { kind: 'files', files: resolvedFiles };
}
return resolveStandard(rawImportPath, filePath, ctx, SupportedLanguages.CSharp);
}
/** PHP: namespace-based resolution via composer.json PSR-4. */
function resolvePhpImportDispatch(
rawImportPath: string,
_filePath: string,
ctx: ResolveCtx,
): ImportResult {
const resolved = resolvePhpImportHelper(rawImportPath, ctx.configs.composerConfig, ctx.allFilePaths, ctx.normalizedFileList, ctx.allFileList, ctx.index);
return resolved ? { kind: 'files', files: [resolved] } : null;
}
/** Swift: module imports via Package.swift target map. */
function resolveSwiftImportDispatch(
rawImportPath: string,
_filePath: string,
ctx: ResolveCtx,
): ImportResult {
const swiftPackageConfig = ctx.configs.swiftPackageConfig;
if (swiftPackageConfig) {
const targetDir = swiftPackageConfig.targets.get(rawImportPath);
if (targetDir) {
const dirPrefix = targetDir + '/';
const files: string[] = [];
for (let i = 0; i < ctx.normalizedFileList.length; i++) {
if (ctx.normalizedFileList[i].startsWith(dirPrefix) && ctx.normalizedFileList[i].endsWith('.swift')) {
files.push(ctx.allFileList[i]);
}
}
if (files.length > 0) return { kind: 'files', files };
}
}
return null; // External framework (Foundation, UIKit, etc.)
}
/**
* Python: relative imports (PEP 328) + proximity-based bare imports.
* Falls through to standard suffix resolution when proximity finds no match.
*/
function resolvePythonImportDispatch(
rawImportPath: string,
filePath: string,
ctx: ResolveCtx,
): ImportResult {
const resolved = resolvePythonImportHelper(filePath, rawImportPath, ctx.allFilePaths);
if (resolved) return { kind: 'files', files: [resolved] };
if (rawImportPath.startsWith('.')) return null; // relative but unresolved -- don't suffix-match
return resolveStandard(rawImportPath, filePath, ctx, SupportedLanguages.Python);
}
/** Ruby: require / require_relative. */
function resolveRubyImportDispatch(
rawImportPath: string,
_filePath: string,
ctx: ResolveCtx,
): ImportResult {
const resolved = resolveRubyImportHelper(rawImportPath, ctx.normalizedFileList, ctx.allFileList, ctx.index);
return resolved ? { kind: 'files', files: [resolved] } : null;
}
/** Rust: expand grouped imports: use {crate::a, crate::b} and use crate::models::{User, Repo}. */
function resolveRustImportDispatch(
rawImportPath: string,
filePath: string,
ctx: ResolveCtx,
): ImportResult {
// Top-level grouped: use {crate::a, crate::b}
if (rawImportPath.startsWith('{') && rawImportPath.endsWith('}')) {
const inner = rawImportPath.slice(1, -1);
const parts = inner.split(',').map(p => p.trim()).filter(Boolean);
const resolved: string[] = [];
for (const part of parts) {
const r = resolveRustImportHelper(filePath, part, ctx.allFilePaths);
if (r) resolved.push(r);
}
return resolved.length > 0 ? { kind: 'files', files: resolved } : null;
}
// Scoped grouped: use crate::models::{User, Repo}
const braceIdx = rawImportPath.indexOf('::{');
if (braceIdx !== -1 && rawImportPath.endsWith('}')) {
const pathPrefix = rawImportPath.substring(0, braceIdx);
const braceContent = rawImportPath.substring(braceIdx + 3, rawImportPath.length - 1);
const items = braceContent.split(',').map(s => s.trim()).filter(Boolean);
const resolved: string[] = [];
for (const item of items) {
// Handle `use crate::models::{User, Repo as R}` — strip alias for resolution
const itemName = item.includes(' as ') ? item.split(' as ')[0].trim() : item;
const r = resolveRustImportHelper(filePath, `${pathPrefix}::${itemName}`, ctx.allFilePaths);
if (r) resolved.push(r);
}
if (resolved.length > 0) return { kind: 'files', files: resolved };
// Fallback: resolve the prefix path itself (e.g. crate::models -> models.rs)
const prefixResult = resolveRustImportHelper(filePath, pathPrefix, ctx.allFilePaths);
if (prefixResult) return { kind: 'files', files: [prefixResult] };
}
return resolveStandard(rawImportPath, filePath, ctx, SupportedLanguages.Rust);
}
// ============================================================================
// Dispatch tables
// ============================================================================
/**
* Per-language import resolver dispatch table.
* Configs are accessed via ctx.configs at call time — no factory closure needed.
* Each resolver encapsulates the full resolution flow for its language, including
* fallthrough to standard resolution where appropriate.
*/
export const importResolvers = {
[SupportedLanguages.JavaScript]: (raw, fp, ctx) => resolveStandard(raw, fp, ctx, SupportedLanguages.JavaScript),
[SupportedLanguages.TypeScript]: (raw, fp, ctx) => resolveStandard(raw, fp, ctx, SupportedLanguages.TypeScript),
[SupportedLanguages.Python]: (raw, fp, ctx) => resolvePythonImportDispatch(raw, fp, ctx),
[SupportedLanguages.Java]: (raw, fp, ctx) => resolveJavaImport(raw, fp, ctx),
[SupportedLanguages.C]: (raw, fp, ctx) => resolveStandard(raw, fp, ctx, SupportedLanguages.C),
[SupportedLanguages.CPlusPlus]: (raw, fp, ctx) => resolveStandard(raw, fp, ctx, SupportedLanguages.CPlusPlus),
[SupportedLanguages.CSharp]: (raw, fp, ctx) => resolveCSharpImportDispatch(raw, fp, ctx),
[SupportedLanguages.Go]: (raw, fp, ctx) => resolveGoImport(raw, fp, ctx),
[SupportedLanguages.Ruby]: (raw, fp, ctx) => resolveRubyImportDispatch(raw, fp, ctx),
[SupportedLanguages.Rust]: (raw, fp, ctx) => resolveRustImportDispatch(raw, fp, ctx),
[SupportedLanguages.PHP]: (raw, fp, ctx) => resolvePhpImportDispatch(raw, fp, ctx),
[SupportedLanguages.Kotlin]: (raw, fp, ctx) => resolveKotlinImport(raw, fp, ctx),
[SupportedLanguages.Swift]: (raw, fp, ctx) => resolveSwiftImportDispatch(raw, fp, ctx),
} satisfies Record<SupportedLanguages, ImportResolverFn>;
/**
* Per-language named binding extractor dispatch table.
* Languages with whole-module import semantics (Go, Ruby, C/C++, Swift) return undefined --
* their bindings are synthesized post-parse by synthesizeWildcardImportBindings() in pipeline.ts.
*/
export const namedBindingExtractors = {
[SupportedLanguages.JavaScript]: extractTsNamedBindings,
[SupportedLanguages.TypeScript]: extractTsNamedBindings,
[SupportedLanguages.Python]: extractPythonNamedBindings,
[SupportedLanguages.Java]: extractJavaNamedBindings,
[SupportedLanguages.C]: undefined,
[SupportedLanguages.CPlusPlus]: undefined,
[SupportedLanguages.CSharp]: extractCsharpNamedBindings,
[SupportedLanguages.Go]: undefined,
[SupportedLanguages.Ruby]: undefined,
[SupportedLanguages.Rust]: extractRustNamedBindings,
[SupportedLanguages.PHP]: extractPhpNamedBindings,
[SupportedLanguages.Kotlin]: extractKotlinNamedBindings,
[SupportedLanguages.Swift]: undefined,
} satisfies Record<SupportedLanguages, NamedBindingExtractorFn | undefined>;

View file

@ -5,20 +5,16 @@
import type { SuffixIndex } from './utils.js';
import { suffixResolve } from './utils.js';
/** C# project config parsed from .csproj files */
export interface CSharpProjectConfig {
/** Root namespace from <RootNamespace> or assembly name (default: project directory name) */
rootNamespace: string;
/** Directory containing the .csproj file */
projectDir: string;
}
import { SupportedLanguages } from '../../../config/supported-languages.js';
import type { ImportResult, ResolveCtx } from './types.js';
import { resolveStandard } from './standard.js';
import type { CSharpProjectConfig } from '../language-config.js';
/**
* Resolve a C# using-directive import path to matching .cs files.
* Resolve a C# using-directive import path to matching .cs files (low-level helper).
* Tries single-file match first, then directory match for namespace imports.
*/
export function resolveCSharpImport(
export function resolveCSharpImportInternal(
importPath: string,
csharpConfigs: CSharpProjectConfig[],
normalizedFileList: string[],
@ -126,3 +122,23 @@ export function resolveCSharpNamespaceDir(
return null;
}
/** C#: namespace-based resolution via .csproj configs, with suffix-match fallback. */
export function resolveCSharpImport(
rawImportPath: string,
filePath: string,
ctx: ResolveCtx,
): ImportResult {
const csharpConfigs = ctx.configs.csharpConfigs;
if (csharpConfigs.length > 0) {
const resolvedFiles = resolveCSharpImportInternal(rawImportPath, csharpConfigs, ctx.normalizedFileList, ctx.allFileList, ctx.index);
if (resolvedFiles.length > 1) {
const dirSuffix = resolveCSharpNamespaceDir(rawImportPath, csharpConfigs);
if (dirSuffix) {
return { kind: 'package', files: resolvedFiles, dirSuffix };
}
}
if (resolvedFiles.length > 0) return { kind: 'files', files: resolvedFiles };
}
return resolveStandard(rawImportPath, filePath, ctx, SupportedLanguages.CSharp);
}

View file

@ -3,11 +3,10 @@
* Handles Go module path-based package imports.
*/
/** Go module config parsed from go.mod */
export interface GoModuleConfig {
/** Module path (e.g., "github.com/user/repo") */
modulePath: string;
}
import { SupportedLanguages } from '../../../config/supported-languages.js';
import type { ImportResult, ResolveCtx } from './types.js';
import { resolveStandard } from './standard.js';
import type { GoModuleConfig } from '../language-config.js';
/**
* Extract the package directory suffix from a Go import path.
@ -56,3 +55,23 @@ export function resolveGoPackage(
return matches;
}
/** Go: package-level imports via go.mod module path. */
export function resolveGoImport(
rawImportPath: string,
filePath: string,
ctx: ResolveCtx,
): ImportResult {
const goModule = ctx.configs.goModule;
if (goModule && rawImportPath.startsWith(goModule.modulePath)) {
const pkgSuffix = resolveGoPackageDir(rawImportPath, goModule);
if (pkgSuffix) {
const pkgFiles = resolveGoPackage(rawImportPath, goModule, ctx.normalizedFileList, ctx.allFileList);
if (pkgFiles.length > 0) {
return { kind: 'package', files: pkgFiles, dirSuffix: pkgSuffix };
}
}
// Fall through if no files found (package might be external)
}
return resolveStandard(rawImportPath, filePath, ctx, SupportedLanguages.Go);
}

View file

@ -4,7 +4,10 @@
*/
import type { SuffixIndex } from './utils.js';
import type { SyntaxNode } from '../utils.js';
import type { SyntaxNode } from '../utils/ast-helpers.js';
import { SupportedLanguages } from '../../../config/supported-languages.js';
import type { ImportResult, ResolveCtx } from './types.js';
import { resolveStandard } from './standard.js';
/** Kotlin file extensions for JVM resolver reuse */
export const KOTLIN_EXTENSIONS: readonly string[] = ['.kt', '.kts'];
@ -118,3 +121,60 @@ export function resolveJvmMemberImport(
return null;
}
/** Java: JVM wildcard -> member import -> standard fallthrough */
export function resolveJavaImport(
rawImportPath: string,
filePath: string,
ctx: ResolveCtx,
): ImportResult {
if (rawImportPath.endsWith('.*')) {
const matchedFiles = resolveJvmWildcard(rawImportPath, ctx.normalizedFileList, ctx.allFileList, ['.java'], ctx.index);
if (matchedFiles.length > 0) return { kind: 'files', files: matchedFiles };
} else {
const memberResolved = resolveJvmMemberImport(rawImportPath, ctx.normalizedFileList, ctx.allFileList, ['.java'], ctx.index);
if (memberResolved) return { kind: 'files', files: [memberResolved] };
}
return resolveStandard(rawImportPath, filePath, ctx, SupportedLanguages.Java);
}
/**
* Kotlin: JVM wildcard/member with Java-interop fallback -> top-level function imports -> standard.
* Kotlin can import from .kt/.kts files OR from .java files (Java interop).
*/
export function resolveKotlinImport(
rawImportPath: string,
filePath: string,
ctx: ResolveCtx,
): ImportResult {
if (rawImportPath.endsWith('.*')) {
const matchedFiles = resolveJvmWildcard(rawImportPath, ctx.normalizedFileList, ctx.allFileList, KOTLIN_EXTENSIONS, ctx.index);
if (matchedFiles.length === 0) {
const javaMatches = resolveJvmWildcard(rawImportPath, ctx.normalizedFileList, ctx.allFileList, ['.java'], ctx.index);
if (javaMatches.length > 0) return { kind: 'files', files: javaMatches };
}
if (matchedFiles.length > 0) return { kind: 'files', files: matchedFiles };
} else {
let memberResolved = resolveJvmMemberImport(rawImportPath, ctx.normalizedFileList, ctx.allFileList, KOTLIN_EXTENSIONS, ctx.index);
if (!memberResolved) {
memberResolved = resolveJvmMemberImport(rawImportPath, ctx.normalizedFileList, ctx.allFileList, ['.java'], ctx.index);
}
if (memberResolved) return { kind: 'files', files: [memberResolved] };
// Kotlin: top-level function imports (e.g. import models.getUser) have only 2 segments,
// which resolveJvmMemberImport skips (requires >=3). Fall back to package-directory scan
// for lowercase last segments (function/property imports). Uppercase last segments
// (class imports like models.User) fall through to standard suffix resolution.
const segments = rawImportPath.split('.');
const lastSeg = segments[segments.length - 1];
if (segments.length >= 2 && lastSeg[0] && lastSeg[0] === lastSeg[0].toLowerCase()) {
const pkgWildcard = segments.slice(0, -1).join('.') + '.*';
let dirFiles = resolveJvmWildcard(pkgWildcard, ctx.normalizedFileList, ctx.allFileList, KOTLIN_EXTENSIONS, ctx.index);
if (dirFiles.length === 0) {
dirFiles = resolveJvmWildcard(pkgWildcard, ctx.normalizedFileList, ctx.allFileList, ['.java'], ctx.index);
}
if (dirFiles.length > 0) return { kind: 'files', files: dirFiles };
}
}
return resolveStandard(rawImportPath, filePath, ctx, SupportedLanguages.Kotlin);
}

View file

@ -5,15 +5,8 @@
import type { SuffixIndex } from './utils.js';
import { suffixResolve } from './utils.js';
/** PHP Composer PSR-4 autoload config */
export interface ComposerConfig {
/** Map of namespace prefix -> directory (e.g., "App\\" -> "app/") */
psr4: Map<string, string>;
/** PSR-4 entries sorted by namespace length descending (longest match wins).
* Cached once at config load time to avoid re-sorting on every import. */
psr4Sorted?: readonly [string, string][];
}
import type { ImportResult, ResolveCtx } from './types.js';
import type { ComposerConfig } from '../language-config.js';
/** Get or compute the sorted PSR-4 entries (cached after first call). */
function getSortedPsr4(config: ComposerConfig): readonly [string, string][] {
@ -25,7 +18,7 @@ function getSortedPsr4(config: ComposerConfig): readonly [string, string][] {
}
/**
* Resolve a PHP use-statement import path using PSR-4 mappings.
* Resolve a PHP use-statement import path using PSR-4 mappings (low-level helper).
* e.g. "App\Http\Controllers\UserController" -> "app/Http/Controllers/UserController.php"
*
* For function/constant imports (use function App\Models\getUser), the last
@ -39,7 +32,7 @@ function getSortedPsr4(config: ComposerConfig): readonly [string, string][] {
* a known limitation — PHP function imports cannot be resolved to a specific file
* without parsing all candidate files.
*/
export function resolvePhpImport(
export function resolvePhpImportInternal(
importPath: string,
composerConfig: ComposerConfig | null,
allFiles: Set<string>,
@ -96,3 +89,13 @@ export function resolvePhpImport(
const pathParts = normalized.split('/').filter(Boolean);
return suffixResolve(pathParts, normalizedFileList, allFileList, index);
}
/** PHP: namespace-based resolution via composer.json PSR-4. */
export function resolvePhpImport(
rawImportPath: string,
_filePath: string,
ctx: ResolveCtx,
): ImportResult {
const resolved = resolvePhpImportInternal(rawImportPath, ctx.configs.composerConfig, ctx.allFilePaths, ctx.normalizedFileList, ctx.allFileList, ctx.index);
return resolved ? { kind: 'files', files: [resolved] } : null;
}

View file

@ -4,9 +4,12 @@
*/
import { tryResolveWithExtensions } from './utils.js';
import { SupportedLanguages } from '../../../config/supported-languages.js';
import type { ImportResult, ResolveCtx } from './types.js';
import { resolveStandard } from './standard.js';
/**
* Resolve a Python import to a file path.
* Resolve a Python import to a file path (low-level helper).
*
* 1. Relative (PEP 328): `.module`, `..module` — 1 dot = current package, each extra dot goes up one level.
* 2. Proximity bare import: static heuristic — checks the importer's own directory first.
@ -15,11 +18,11 @@ import { tryResolveWithExtensions } from './utils.js';
* Checks package (__init__.py) before module (.py), matching CPython's finder order (PEP 451 §4).
* Coexistence of both is physically impossible (same name = file vs directory), so the order
* only matters for spec compliance.
* Note: namespace packages (PEP 420, directory without __init__.py) are not handled.
* Note: implicit namespace packages (Python 3.3+, directory without __init__.py) are not handled.
*
* Returns null to let the caller fall through to suffixResolve.
*/
export function resolvePythonImport(
export function resolvePythonImportInternal(
currentFile: string,
importPath: string,
allFiles: Set<string>,
@ -55,5 +58,38 @@ export function resolvePythonImport(
if (allFiles.has(`${importerDir}/${pathLike}/__init__.py`)) return `${importerDir}/${pathLike}/__init__.py`;
if (allFiles.has(`${importerDir}/${pathLike}.py`)) return `${importerDir}/${pathLike}.py`;
// Ancestor directory walk — Python resolves bare imports against sys.path entries,
// which typically includes the project root and package directories. Walk up from the
// importer's directory to find the module in an ancestor, preferring the closest match.
// This prevents cross-language misresolution (e.g., Python `from middleware import X`
// resolving to a TypeScript middleware.ts via suffix matching). Issue #417.
const dirParts = importerDir.split('/');
for (let i = dirParts.length - 1; i >= 0; i--) {
const ancestorDir = dirParts.slice(0, i).join('/');
const prefix = ancestorDir ? `${ancestorDir}/` : '';
if (allFiles.has(`${prefix}${pathLike}/__init__.py`)) return `${prefix}${pathLike}/__init__.py`;
if (allFiles.has(`${prefix}${pathLike}.py`)) return `${prefix}${pathLike}.py`;
}
return null;
}
/**
* Python: relative imports (PEP 328) + proximity-based bare imports.
* Falls through to standard suffix resolution when proximity finds no match.
*/
export function resolvePythonImport(
rawImportPath: string,
filePath: string,
ctx: ResolveCtx,
): ImportResult {
const resolved = resolvePythonImportInternal(filePath, rawImportPath, ctx.allFilePaths);
if (resolved) {
// Store in resolveCache so other files importing the same module skip the
// ancestor walk. The cache key matches resolveStandard's convention.
ctx.resolveCache.set(`${filePath}::${rawImportPath}`, resolved);
return { kind: 'files', files: [resolved] };
}
if (rawImportPath.startsWith('.')) return null; // relative but unresolved -- don't suffix-match
return resolveStandard(rawImportPath, filePath, ctx, SupportedLanguages.Python);
}

View file

@ -5,14 +5,15 @@
import type { SuffixIndex } from './utils.js';
import { suffixResolve } from './utils.js';
import type { ImportResult, ResolveCtx } from './types.js';
/**
* Resolve a Ruby require/require_relative path to a matching .rb file.
* Resolve a Ruby require/require_relative path to a matching .rb file (low-level helper).
*
* require_relative paths are pre-normalized to './' prefix by the caller.
* require paths use suffix matching (gem-style paths like 'json', 'net/http').
*/
export function resolveRubyImport(
export function resolveRubyImportInternal(
importPath: string,
normalizedFileList: string[],
allFileList: string[],
@ -21,3 +22,13 @@ export function resolveRubyImport(
const pathParts = importPath.replace(/^\.\//, '').split('/').filter(Boolean);
return suffixResolve(pathParts, normalizedFileList, allFileList, index);
}
/** Ruby: require / require_relative. */
export function resolveRubyImport(
rawImportPath: string,
_filePath: string,
ctx: ResolveCtx,
): ImportResult {
const resolved = resolveRubyImportInternal(rawImportPath, ctx.normalizedFileList, ctx.allFileList, ctx.index);
return resolved ? { kind: 'files', files: [resolved] } : null;
}

View file

@ -3,11 +3,15 @@
* Handles crate::, super::, self:: prefix paths and :: separators.
*/
import { SupportedLanguages } from '../../../config/supported-languages.js';
import type { ImportResult, ResolveCtx } from './types.js';
import { resolveStandard } from './standard.js';
/**
* Resolve Rust use-path to a file.
* Resolve Rust use-path to a file (low-level helper).
* Handles crate::, super::, self:: prefixes and :: path separators.
*/
export function resolveRustImport(
export function resolveRustImportInternal(
currentFile: string,
importPath: string,
allFiles: Set<string>,
@ -80,3 +84,43 @@ export function tryRustModulePath(modulePath: string, allFiles: Set<string>): st
return null;
}
/** Rust: expand grouped imports: use {crate::a, crate::b} and use crate::models::{User, Repo}. */
export function resolveRustImport(
rawImportPath: string,
filePath: string,
ctx: ResolveCtx,
): ImportResult {
// Top-level grouped: use {crate::a, crate::b}
if (rawImportPath.startsWith('{') && rawImportPath.endsWith('}')) {
const inner = rawImportPath.slice(1, -1);
const parts = inner.split(',').map(p => p.trim()).filter(Boolean);
const resolved: string[] = [];
for (const part of parts) {
const r = resolveRustImportInternal(filePath, part, ctx.allFilePaths);
if (r) resolved.push(r);
}
return resolved.length > 0 ? { kind: 'files', files: resolved } : null;
}
// Scoped grouped: use crate::models::{User, Repo}
const braceIdx = rawImportPath.indexOf('::{');
if (braceIdx !== -1 && rawImportPath.endsWith('}')) {
const pathPrefix = rawImportPath.substring(0, braceIdx);
const braceContent = rawImportPath.substring(braceIdx + 3, rawImportPath.length - 1);
const items = braceContent.split(',').map(s => s.trim()).filter(Boolean);
const resolved: string[] = [];
for (const item of items) {
// Handle `use crate::models::{User, Repo as R}` — strip alias for resolution
const itemName = item.includes(' as ') ? item.split(' as ')[0].trim() : item;
const r = resolveRustImportInternal(filePath, `${pathPrefix}::${itemName}`, ctx.allFilePaths);
if (r) resolved.push(r);
}
if (resolved.length > 0) return { kind: 'files', files: resolved };
// Fallback: resolve the prefix path itself (e.g. crate::models -> models.rs)
const prefixResult = resolveRustImportInternal(filePath, pathPrefix, ctx.allFilePaths);
if (prefixResult) return { kind: 'files', files: [prefixResult] };
}
return resolveStandard(rawImportPath, filePath, ctx, SupportedLanguages.Rust);
}

View file

@ -6,16 +6,10 @@
import type { SuffixIndex } from './utils.js';
import { tryResolveWithExtensions, suffixResolve } from './utils.js';
import { resolveRustImport } from './rust.js';
import { resolveRustImportInternal } from './rust.js';
import { SupportedLanguages } from '../../../config/supported-languages.js';
/** TypeScript path alias config parsed from tsconfig.json */
export interface TsconfigPaths {
/** Map of alias prefix -> target prefix (e.g., "@/" -> "src/") */
aliases: Map<string, string>;
/** Base URL for path resolution (relative to repo root) */
baseUrl: string;
}
import type { ImportResult, ImportResolverFn, ResolveCtx } from './types.js';
import type { TsconfigPaths } from '../language-config.js';
/** Max entries in the resolve cache. Beyond this, entries are evicted.
* 100K entries ≈ 15MB — covers the most common import patterns. */
@ -102,13 +96,13 @@ export const resolveImportPath = (
const inner = importPath.slice(1, -1);
const parts = inner.split(',').map(p => p.trim()).filter(Boolean);
for (const part of parts) {
const partResult = resolveRustImport(currentFile, part, allFiles);
const partResult = resolveRustImportInternal(currentFile, part, allFiles);
if (partResult) return cache(partResult);
}
return cache(null);
}
const rustResult = resolveRustImport(currentFile, rustImportPath, allFiles);
const rustResult = resolveRustImportInternal(currentFile, rustImportPath, allFiles);
if (rustResult) return cache(rustResult);
// Fall through to generic resolution if Rust-specific didn't match
}
@ -149,3 +143,47 @@ export const resolveImportPath = (
const resolved = suffixResolve(pathParts, normalizedFileList, allFileList, index);
return cache(resolved);
};
// ============================================================================
// Per-language dispatch functions (moved from import-resolution.ts)
// ============================================================================
/**
* Standard single-file resolution (TS/JS/C/C++ and fallback for other languages).
* Handles relative imports, tsconfig path aliases, and suffix matching.
*/
export function resolveStandard(
rawImportPath: string,
filePath: string,
ctx: ResolveCtx,
language: SupportedLanguages,
): ImportResult {
const resolvedPath = resolveImportPath(
filePath,
rawImportPath,
ctx.allFilePaths,
ctx.allFileList,
ctx.normalizedFileList,
ctx.resolveCache,
language,
ctx.configs.tsconfigPaths,
ctx.index,
);
return resolvedPath ? { kind: 'files', files: [resolvedPath] } : null;
}
/** JavaScript: standard single-file resolution. */
export const resolveJavascriptImport: ImportResolverFn = (raw, fp, ctx) =>
resolveStandard(raw, fp, ctx, SupportedLanguages.JavaScript);
/** TypeScript: standard single-file resolution. */
export const resolveTypescriptImport: ImportResolverFn = (raw, fp, ctx) =>
resolveStandard(raw, fp, ctx, SupportedLanguages.TypeScript);
/** C: standard single-file resolution for #include directives. */
export const resolveCImport: ImportResolverFn = (raw, fp, ctx) =>
resolveStandard(raw, fp, ctx, SupportedLanguages.C);
/** C++: standard single-file resolution for #include directives. */
export const resolveCppImport: ImportResolverFn = (raw, fp, ctx) =>
resolveStandard(raw, fp, ctx, SupportedLanguages.CPlusPlus);

View file

@ -0,0 +1,29 @@
/**
* Swift module import resolution.
* Handles module imports via Package.swift target map.
*/
import type { ImportResult, ResolveCtx } from './types.js';
/** Swift: module imports via Package.swift target map. */
export function resolveSwiftImport(
rawImportPath: string,
_filePath: string,
ctx: ResolveCtx,
): ImportResult {
const swiftPackageConfig = ctx.configs.swiftPackageConfig;
if (swiftPackageConfig) {
const targetDir = swiftPackageConfig.targets.get(rawImportPath);
if (targetDir) {
const dirPrefix = targetDir + '/';
const files: string[] = [];
for (let i = 0; i < ctx.normalizedFileList.length; i++) {
if (ctx.normalizedFileList[i].startsWith(dirPrefix) && ctx.normalizedFileList[i].endsWith('.swift')) {
files.push(ctx.allFileList[i]);
}
}
if (files.length > 0) return { kind: 'files', files };
}
}
return null; // External framework (Foundation, UIKit, etc.)
}

View file

@ -0,0 +1,50 @@
/**
* Import resolution types — shared across all per-language resolvers.
*
* Extracted from import-resolution.ts to co-locate types with their consumers.
*/
import type { TsconfigPaths, GoModuleConfig, CSharpProjectConfig, ComposerConfig } from '../language-config.js';
import type { SwiftPackageConfig } from '../language-config.js';
import type { SuffixIndex } from './utils.js';
/**
* Result of resolving an import via language-specific dispatch.
* - 'files': resolved to one or more files -> add to ImportMap
* - 'package': resolved to a directory -> add graph edges + store dirSuffix in PackageMap
* - null: no resolution (external dependency, etc.)
*/
export type ImportResult =
| { kind: 'files'; files: string[] }
| { kind: 'package'; files: string[]; dirSuffix: string }
| null;
/** Bundled language-specific configs loaded once per ingestion run. */
export interface ImportConfigs {
tsconfigPaths: TsconfigPaths | null;
goModule: GoModuleConfig | null;
composerConfig: ComposerConfig | null;
swiftPackageConfig: SwiftPackageConfig | null;
csharpConfigs: CSharpProjectConfig[];
}
/** Pre-built lookup structures for import resolution. Build once, reuse across chunks. */
export interface ImportResolutionContext {
allFilePaths: Set<string>;
allFileList: string[];
normalizedFileList: string[];
index: SuffixIndex;
resolveCache: Map<string, string | null>;
}
/** Full context for import resolution: file lookups + language configs. */
export interface ResolveCtx extends ImportResolutionContext {
configs: ImportConfigs;
}
/** Per-language import resolver -- function alias matching ExportChecker/CallRouter pattern. */
export type ImportResolverFn = (
rawImportPath: string,
filePath: string,
resolveCtx: ResolveCtx,
) => ImportResult;

View file

@ -3,7 +3,7 @@
* Extracted from import-processor.ts to reduce file size.
*/
import type { SyntaxNode } from '../utils.js';
import type { SyntaxNode } from '../utils/ast-helpers.js';
/** All file extensions to try during resolution */
export const EXTENSIONS = [
@ -168,11 +168,3 @@ export function suffixResolve(
return null;
}
/** Find the first direct named child of a tree-sitter node matching the given type. */
export function findChild(node: SyntaxNode, type: string): SyntaxNode | null {
for (let i = 0; i < node.namedChildCount; i++) {
const child = node.namedChild(i);
if (child?.type === type) return child;
}
return null;
}

View file

@ -1,6 +1,6 @@
import fs from 'fs/promises';
import path from 'path';
import type { ImportConfigs } from './import-resolution.js';
import type { ImportConfigs } from './import-resolvers/types.js';
const isDev = process.env.NODE_ENV === 'development';
@ -26,6 +26,9 @@ export interface GoModuleConfig {
export interface ComposerConfig {
/** Map of namespace prefix -> directory (e.g., "App\\" -> "app/") */
psr4: Map<string, string>;
/** PSR-4 entries sorted by namespace length descending (longest match wins).
* Cached once at config load time to avoid re-sorting on every import. */
psr4Sorted?: readonly [string, string][];
}
/** C# project config parsed from .csproj files */

View file

@ -0,0 +1,133 @@
/**
* Language Provider interface — the complete capability contract for a supported language.
*
* Each language implements this interface in a single file under `languages/`.
* The pipeline accesses all per-language behavior through this interface.
*
* Design pattern: Strategy pattern with compile-time exhaustiveness.
* The providers table in `languages/index.ts` uses `satisfies Record<SupportedLanguages, LanguageProvider>`
* so adding a language to the enum without creating a provider is a compiler error.
*/
import type { SupportedLanguages } from '../../config/supported-languages.js';
import type { LanguageTypeConfig } from './type-extractors/types.js';
import type { CallRouter } from './call-routing.js';
import type { ExportChecker } from './export-detection.js';
import type { ImportResolverFn } from './import-resolvers/types.js';
import type { NamedBindingExtractorFn } from './named-bindings/types.js';
import type { SyntaxNode } from './utils/ast-helpers.js';
import type { NodeLabel } from '../graph/types.js';
// ── Shared type aliases ────────────────────────────────────────────────────
/** Tree-sitter query captures: capture name → AST node (or undefined if not captured). */
export type CaptureMap = Record<string, SyntaxNode | undefined>;
// ── Strategy tag types ─────────────────────────────────────────────────────
/** MRO strategy for multiple inheritance resolution. */
export type MroStrategy = 'first-wins' | 'c3' | 'leftmost-base' | 'implements-split' | 'qualified-syntax';
/** How a language handles imports — determines wildcard synthesis behavior. */
export type ImportSemantics = 'named' | 'wildcard' | 'namespace';
/**
* Everything a language needs to provide.
* Required fields must be explicitly set; optional fields have defaults
* applied by defineLanguage().
*/
interface LanguageProviderConfig {
// ── Identity ──────────────────────────────────────────────────────
readonly id: SupportedLanguages;
/** File extensions that map to this language (e.g., ['.ts', '.tsx']) */
readonly extensions: readonly string[];
// ── Parser ────────────────────────────────────────────────────────
/** Tree-sitter query strings for definitions, imports, calls, heritage */
readonly treeSitterQueries: string;
// ── Core (required) ───────────────────────────────────────────────
/** Type extraction: declarations, initializers, for-loop bindings */
readonly typeConfig: LanguageTypeConfig;
/** Export detection: is this AST node a public/exported symbol? */
readonly exportChecker: ExportChecker;
/** Import resolution: resolves raw import path to file system path */
readonly importResolver: ImportResolverFn;
// ── Calls & Imports (optional) ────────────────────────────────────
/** Call routing for languages that express imports/heritage as calls (e.g., Ruby).
* Default: no routing (all calls are normal call expressions). */
readonly callRouter?: CallRouter;
/** Named binding extraction from import statements.
* Default: undefined (language uses wildcard/whole-module imports). */
readonly namedBindingExtractor?: NamedBindingExtractorFn;
/** How this language handles imports.
* - 'named': per-symbol imports (JS/TS, Java, C#, Rust, PHP, Kotlin)
* - 'wildcard': whole-module imports, needs synthesis (Go, Ruby, C/C++, Swift)
* - 'namespace': namespace imports, needs moduleAliasMap (Python)
* Default: 'named'. */
readonly importSemantics?: ImportSemantics;
/** Language-specific transformation of raw import path text before resolution.
* Called after sanitization. E.g., Kotlin appends wildcard suffixes.
* Default: undefined (no preprocessing). */
readonly importPathPreprocessor?: (cleaned: string, importNode: SyntaxNode) => string;
/** Wire implicit inter-file imports for languages where all files in a module
* see each other (e.g., Swift targets, C header inclusion units).
* Called with only THIS language's files (pre-grouped by the processor).
* Default: undefined (no implicit imports). */
readonly implicitImportWirer?: (
languageFiles: string[],
importMap: ReadonlyMap<string, ReadonlySet<string>>,
addImportEdge: (src: string, target: string) => void,
projectConfig: unknown,
) => void;
// ── Labels ────────────────────────────────────────────────────────
/** Override the default node label for definition.function captures.
* Return null to skip (C/C++ duplicate), a different label to reclassify
* (e.g., 'Method' for Kotlin), or defaultLabel to keep as-is.
* Default: undefined (standard label assignment). */
readonly labelOverride?: (functionNode: SyntaxNode, defaultLabel: NodeLabel) => NodeLabel | null;
// ── Heritage & MRO ────────────────────────────────────────────────
/** Default edge type when parent symbol is ambiguous (interface vs class).
* Default: 'EXTENDS'. */
readonly heritageDefaultEdge?: 'EXTENDS' | 'IMPLEMENTS';
/** Regex to detect interface names by convention (e.g., /^I[A-Z]/ for C#/Java).
* When matched, IMPLEMENTS edge is used instead of heritageDefaultEdge. */
readonly interfaceNamePattern?: RegExp;
/** MRO strategy for multiple inheritance resolution.
* Default: 'first-wins'. */
readonly mroStrategy?: MroStrategy;
// ── Language-specific extraction hooks ────────────────────────────
/** Extract a semantic description for a definition node (e.g., PHP Eloquent
* property arrays, relation method descriptions).
* Default: undefined (no description extraction). */
readonly descriptionExtractor?: (
nodeLabel: NodeLabel,
nodeName: string,
captureMap: CaptureMap,
) => string | undefined;
/** Detect if a file contains framework route definitions (e.g., Laravel routes.php).
* When true, the worker extracts routes via the language's route extraction logic.
* Default: undefined (no route files). */
readonly isRouteFile?: (filePath: string) => boolean;
}
/** Runtime type — same as LanguageProviderConfig but with defaults guaranteed present. */
export interface LanguageProvider extends Omit<LanguageProviderConfig,
'importSemantics' | 'heritageDefaultEdge' | 'mroStrategy'
> {
readonly importSemantics: ImportSemantics;
readonly heritageDefaultEdge: 'EXTENDS' | 'IMPLEMENTS';
readonly mroStrategy: MroStrategy;
}
const DEFAULTS: Pick<LanguageProvider, 'importSemantics' | 'heritageDefaultEdge' | 'mroStrategy'> = {
importSemantics: 'named',
heritageDefaultEdge: 'EXTENDS',
mroStrategy: 'first-wins',
};
/** Define a language provider — required fields must be supplied, optional fields get sensible defaults. */
export function defineLanguage(config: LanguageProviderConfig): LanguageProvider {
return { ...DEFAULTS, ...config };
}

View file

@ -0,0 +1,49 @@
/**
* C and C++ language providers.
*
* Both languages use wildcard import semantics (headers expose all symbols
* via #include). Neither language has named binding extraction.
*
* C uses 'first-wins' MRO (no inheritance). C++ uses 'leftmost-base' MRO
* for its left-to-right multiple inheritance resolution order.
*/
import { SupportedLanguages } from '../../../config/supported-languages.js';
import { defineLanguage } from '../language-provider.js';
import { typeConfig as cCppConfig } from '../type-extractors/c-cpp.js';
import { cCppExportChecker } from '../export-detection.js';
import { resolveCImport, resolveCppImport } from '../import-resolvers/standard.js';
import { C_QUERIES, CPP_QUERIES } from '../tree-sitter-queries.js';
import { isCppInsideClassOrStruct } from '../utils/ast-helpers.js';
import type { LanguageProvider } from '../language-provider.js';
/** Label override shared by C and C++: skip function_definition captures inside class/struct
* bodies (they're duplicates of definition.method captures). */
const cppLabelOverride: NonNullable<LanguageProvider['labelOverride']> = (functionNode, defaultLabel) => {
if (defaultLabel !== 'Function') return defaultLabel;
return isCppInsideClassOrStruct(functionNode) ? null : defaultLabel;
};
export const cProvider = defineLanguage({
id: SupportedLanguages.C,
extensions: ['.c'],
treeSitterQueries: C_QUERIES,
typeConfig: cCppConfig,
exportChecker: cCppExportChecker,
importResolver: resolveCImport,
importSemantics: 'wildcard',
labelOverride: cppLabelOverride,
});
export const cppProvider = defineLanguage({
id: SupportedLanguages.CPlusPlus,
extensions: ['.cpp', '.cc', '.cxx', '.h', '.hpp', '.hxx', '.hh'],
treeSitterQueries: CPP_QUERIES,
typeConfig: cCppConfig,
exportChecker: cCppExportChecker,
importResolver: resolveCppImport,
importSemantics: 'wildcard',
mroStrategy: 'leftmost-base',
labelOverride: cppLabelOverride,
});

View file

@ -0,0 +1,27 @@
/**
* C# language provider.
*
* C# uses named imports (using directives), modifier-based export detection,
* and an implements-split MRO strategy for multiple interface implementation.
* Interface names follow the I-prefix convention (e.g., IDisposable).
*/
import { SupportedLanguages } from '../../../config/supported-languages.js';
import { defineLanguage } from '../language-provider.js';
import { typeConfig as csharpConfig } from '../type-extractors/csharp.js';
import { csharpExportChecker } from '../export-detection.js';
import { resolveCSharpImport } from '../import-resolvers/csharp.js';
import { extractCSharpNamedBindings } from '../named-bindings/csharp.js';
import { CSHARP_QUERIES } from '../tree-sitter-queries.js';
export const csharpProvider = defineLanguage({
id: SupportedLanguages.CSharp,
extensions: ['.cs'],
treeSitterQueries: CSHARP_QUERIES,
typeConfig: csharpConfig,
exportChecker: csharpExportChecker,
importResolver: resolveCSharpImport,
namedBindingExtractor: extractCSharpNamedBindings,
interfaceNamePattern: /^I[A-Z]/,
mroStrategy: 'implements-split',
});

View file

@ -0,0 +1,27 @@
/**
* Go Language Provider
*
* Assembles all Go-specific ingestion capabilities into a single
* LanguageProvider, following the Strategy pattern used by the pipeline.
*
* Key Go traits:
* - importSemantics: 'wildcard' (Go imports entire packages)
* - callRouter: present (Go method calls may need routing)
*/
import { SupportedLanguages } from '../../../config/supported-languages.js';
import { defineLanguage } from '../language-provider.js';
import { typeConfig as goConfig } from '../type-extractors/go.js';
import { goExportChecker } from '../export-detection.js';
import { resolveGoImport } from '../import-resolvers/go.js';
import { GO_QUERIES } from '../tree-sitter-queries.js';
export const goProvider = defineLanguage({
id: SupportedLanguages.Go,
extensions: ['.go'],
treeSitterQueries: GO_QUERIES,
typeConfig: goConfig,
exportChecker: goExportChecker,
importResolver: resolveGoImport,
importSemantics: 'wildcard',
});

View file

@ -0,0 +1,70 @@
/**
* Language Provider Registry — compile-time exhaustive provider table.
*
* To add a new language:
* 1. Add enum member to SupportedLanguages
* 2. Create `languages/<lang>.ts` exporting a LanguageProvider
* 3. Add one line to the `providers` table below
* 4. Run `tsc --noEmit` to verify
*/
import { SupportedLanguages } from '../../../config/supported-languages.js';
import type { LanguageProvider } from '../language-provider.js';
import { typescriptProvider, javascriptProvider } from './typescript.js';
import { pythonProvider } from './python.js';
import { javaProvider } from './java.js';
import { kotlinProvider } from './kotlin.js';
import { goProvider } from './go.js';
import { rustProvider } from './rust.js';
import { csharpProvider } from './csharp.js';
import { cProvider, cppProvider } from './c-cpp.js';
import { phpProvider } from './php.js';
import { rubyProvider } from './ruby.js';
import { swiftProvider } from './swift.js';
export const providers = {
[SupportedLanguages.JavaScript]: javascriptProvider,
[SupportedLanguages.TypeScript]: typescriptProvider,
[SupportedLanguages.Python]: pythonProvider,
[SupportedLanguages.Java]: javaProvider,
[SupportedLanguages.Kotlin]: kotlinProvider,
[SupportedLanguages.Go]: goProvider,
[SupportedLanguages.Rust]: rustProvider,
[SupportedLanguages.CSharp]: csharpProvider,
[SupportedLanguages.C]: cProvider,
[SupportedLanguages.CPlusPlus]: cppProvider,
[SupportedLanguages.PHP]: phpProvider,
[SupportedLanguages.Ruby]: rubyProvider,
[SupportedLanguages.Swift]: swiftProvider,
} satisfies Record<SupportedLanguages, LanguageProvider>;
/** Get provider by language enum (always succeeds for SupportedLanguages). */
export function getProvider(language: SupportedLanguages): LanguageProvider {
return providers[language];
}
/** Pre-built extension → provider lookup (built once at module load). */
const extensionMap = new Map<string, LanguageProvider>();
for (const provider of Object.values(providers)) {
for (const ext of provider.extensions) {
extensionMap.set(ext, provider);
}
}
/** Look up a language provider from a file path by extension.
* Returns null if the file extension is not recognized. */
export function getProviderForFile(filePath: string): LanguageProvider | null {
const lastDot = filePath.lastIndexOf('.');
const ext = lastDot >= 0 ? filePath.slice(lastDot).toLowerCase() : '';
const basename = filePath.slice(filePath.lastIndexOf('/') + 1);
return extensionMap.get(ext) ?? extensionMap.get(basename) ?? null;
}
/** Pre-computed list of providers that have implicit import wiring (e.g., Swift).
* Built once at module load — avoids iterating all 13 providers per call. */
export const providersWithImplicitWiring = Object.values(providers)
.filter((p): p is LanguageProvider & { implicitImportWirer: NonNullable<LanguageProvider['implicitImportWirer']> } =>
p.implicitImportWirer != null
);

View file

@ -0,0 +1,28 @@
/**
* Java language provider.
*
* Java uses named imports, JVM wildcard/member import resolution,
* and a 'public' modifier-based export checker. Heritage uses
* EXTENDS by default with implements-split MRO for multiple
* interface implementation.
*/
import { SupportedLanguages } from '../../../config/supported-languages.js';
import { defineLanguage } from '../language-provider.js';
import { javaTypeConfig } from '../type-extractors/jvm.js';
import { javaExportChecker } from '../export-detection.js';
import { resolveJavaImport } from '../import-resolvers/jvm.js';
import { extractJavaNamedBindings } from '../named-bindings/java.js';
import { JAVA_QUERIES } from '../tree-sitter-queries.js';
export const javaProvider = defineLanguage({
id: SupportedLanguages.Java,
extensions: ['.java'],
treeSitterQueries: JAVA_QUERIES,
typeConfig: javaTypeConfig,
exportChecker: javaExportChecker,
importResolver: resolveJavaImport,
namedBindingExtractor: extractJavaNamedBindings,
interfaceNamePattern: /^I[A-Z]/,
mroStrategy: 'implements-split',
});

View file

@ -0,0 +1,35 @@
/**
* Kotlin language provider.
*
* Kotlin uses named imports with JVM wildcard/member resolution and
* Java-interop fallback. Default visibility is public (no modifier needed).
* Heritage uses EXTENDS by default with implements-split MRO for
* multiple interface implementation.
*/
import { SupportedLanguages } from '../../../config/supported-languages.js';
import { defineLanguage } from '../language-provider.js';
import { kotlinTypeConfig } from '../type-extractors/jvm.js';
import { kotlinExportChecker } from '../export-detection.js';
import { resolveKotlinImport } from '../import-resolvers/jvm.js';
import { extractKotlinNamedBindings } from '../named-bindings/kotlin.js';
import { appendKotlinWildcard } from '../import-resolvers/jvm.js';
import { KOTLIN_QUERIES } from '../tree-sitter-queries.js';
import { isKotlinClassMethod } from '../utils/ast-helpers.js';
export const kotlinProvider = defineLanguage({
id: SupportedLanguages.Kotlin,
extensions: ['.kt', '.kts'],
treeSitterQueries: KOTLIN_QUERIES,
typeConfig: kotlinTypeConfig,
exportChecker: kotlinExportChecker,
importResolver: resolveKotlinImport,
namedBindingExtractor: extractKotlinNamedBindings,
importPathPreprocessor: appendKotlinWildcard,
mroStrategy: 'implements-split',
labelOverride: (functionNode, defaultLabel) => {
if (defaultLabel !== 'Function') return defaultLabel;
if (isKotlinClassMethod(functionNode)) return 'Method';
return defaultLabel;
},
});

View file

@ -0,0 +1,133 @@
/**
* PHP language provider.
*
* PHP uses named imports (use statements for classes/functions/constants),
* and standard export/import resolution. PHP files can use a variety of
* extensions from legacy versions through modern PHP 8.
*/
import { SupportedLanguages } from '../../../config/supported-languages.js';
import { defineLanguage } from '../language-provider.js';
import { typeConfig as phpConfig } from '../type-extractors/php.js';
import { phpExportChecker } from '../export-detection.js';
import { resolvePhpImport } from '../import-resolvers/php.js';
import { extractPhpNamedBindings } from '../named-bindings/php.js';
import { PHP_QUERIES } from '../tree-sitter-queries.js';
import { findDescendant, extractStringContent } from '../utils/ast-helpers.js';
import type { NodeLabel } from '../../graph/types.js';
/** Eloquent model properties whose array values are worth indexing. */
const ELOQUENT_ARRAY_PROPS = new Set(['fillable', 'casts', 'hidden', 'guarded', 'with', 'appends']);
/** Eloquent relationship method names. */
const ELOQUENT_RELATIONS = new Set([
'hasMany', 'hasOne', 'belongsTo', 'belongsToMany',
'morphTo', 'morphMany', 'morphOne', 'morphToMany', 'morphedByMany',
'hasManyThrough', 'hasOneThrough',
]);
/**
* For a PHP property_declaration node, extract array values as a description string.
* Returns null if not an Eloquent model property or no array values found.
*/
function extractPhpPropertyDescription(propName: string, propDeclNode: any): string | null {
if (!ELOQUENT_ARRAY_PROPS.has(propName)) return null;
const arrayNode = findDescendant(propDeclNode, 'array_creation_expression');
if (!arrayNode) return null;
const items: string[] = [];
for (const child of (arrayNode.children ?? [])) {
if (child.type !== 'array_element_initializer') continue;
const children = child.children ?? [];
const arrowIdx = children.findIndex((c: any) => c.type === '=>');
if (arrowIdx !== -1) {
const key = extractStringContent(children[arrowIdx - 1]);
const val = extractStringContent(children[arrowIdx + 1]);
if (key && val) items.push(`${key}:${val}`);
} else {
const val = extractStringContent(children[0]);
if (val) items.push(val);
}
}
return items.length > 0 ? items.join(', ') : null;
}
/**
* For a PHP method_declaration node, detect if it defines an Eloquent relationship.
* Returns description like "hasMany(Post)" or null.
*/
function extractEloquentRelationDescription(methodNode: any): string | null {
function findRelationCall(node: any): any {
if (node.type === 'member_call_expression') {
const children = node.children ?? [];
const objectNode = children.find((c: any) => c.type === 'variable_name' && c.text === '$this');
const nameNode = children.find((c: any) => c.type === 'name');
if (objectNode && nameNode && ELOQUENT_RELATIONS.has(nameNode.text)) return node;
}
for (const child of (node.children ?? [])) {
const found = findRelationCall(child);
if (found) return found;
}
return null;
}
const callNode = findRelationCall(methodNode);
if (!callNode) return null;
const relType = callNode.children?.find((c: any) => c.type === 'name')?.text;
const argsNode = callNode.children?.find((c: any) => c.type === 'arguments');
let targetModel: string | null = null;
if (argsNode) {
const firstArg = argsNode.children?.find((c: any) => c.type === 'argument');
if (firstArg) {
const classConstant = firstArg.children?.find((c: any) =>
c.type === 'class_constant_access_expression'
);
if (classConstant) {
targetModel = classConstant.children?.find((c: any) => c.type === 'name')?.text ?? null;
}
}
}
if (relType && targetModel) return `${relType}(${targetModel})`;
if (relType) return relType;
return null;
}
/**
* LanguageProvider.descriptionExtractor implementation for PHP.
* Extracts Eloquent model property metadata and relationship descriptions.
*/
function phpDescriptionExtractor(
nodeLabel: NodeLabel,
nodeName: string,
captureMap: Record<string, any>,
): string | undefined {
if (nodeLabel === 'Property' && captureMap['definition.property']) {
return extractPhpPropertyDescription(nodeName, captureMap['definition.property']) ?? undefined;
}
if (nodeLabel === 'Method' && captureMap['definition.method']) {
return extractEloquentRelationDescription(captureMap['definition.method']) ?? undefined;
}
return undefined;
}
/** Detect Laravel route files by path convention. */
function isPhpRouteFile(filePath: string): boolean {
return filePath.endsWith('.php') &&
(filePath.includes('/routes/') || filePath.startsWith('routes/'));
}
export const phpProvider = defineLanguage({
id: SupportedLanguages.PHP,
extensions: ['.php', '.phtml', '.php3', '.php4', '.php5', '.php8'],
treeSitterQueries: PHP_QUERIES,
typeConfig: phpConfig,
exportChecker: phpExportChecker,
importResolver: resolvePhpImport,
namedBindingExtractor: extractPhpNamedBindings,
descriptionExtractor: phpDescriptionExtractor,
isRouteFile: isPhpRouteFile,
});

View file

@ -0,0 +1,31 @@
/**
* Python Language Provider
*
* Assembles all Python-specific ingestion capabilities into a single
* LanguageProvider, following the Strategy pattern used by the pipeline.
*
* Key Python traits:
* - importSemantics: 'namespace' (Python uses namespace imports, not wildcard)
* - mroStrategy: 'c3' (Python C3 linearization for multiple inheritance)
* - namedBindingExtractor: present (from X import Y)
*/
import { SupportedLanguages } from '../../../config/supported-languages.js';
import { defineLanguage } from '../language-provider.js';
import { typeConfig as pythonConfig } from '../type-extractors/python.js';
import { pythonExportChecker } from '../export-detection.js';
import { resolvePythonImport } from '../import-resolvers/python.js';
import { extractPythonNamedBindings } from '../named-bindings/python.js';
import { PYTHON_QUERIES } from '../tree-sitter-queries.js';
export const pythonProvider = defineLanguage({
id: SupportedLanguages.Python,
extensions: ['.py'],
treeSitterQueries: PYTHON_QUERIES,
typeConfig: pythonConfig,
exportChecker: pythonExportChecker,
importResolver: resolvePythonImport,
namedBindingExtractor: extractPythonNamedBindings,
importSemantics: 'namespace',
mroStrategy: 'c3',
});

View file

@ -0,0 +1,27 @@
/**
* Ruby language provider.
*
* Ruby uses wildcard import semantics (require/require_relative bring
* everything into scope). Ruby has SPECIAL call routing via routeRubyCall
* to handle require, include/extend (heritage), and attr_accessor/
* attr_reader/attr_writer (property definitions) as call expressions.
*/
import { SupportedLanguages } from '../../../config/supported-languages.js';
import { defineLanguage } from '../language-provider.js';
import { typeConfig as rubyConfig } from '../type-extractors/ruby.js';
import { routeRubyCall } from '../call-routing.js';
import { rubyExportChecker } from '../export-detection.js';
import { resolveRubyImport } from '../import-resolvers/ruby.js';
import { RUBY_QUERIES } from '../tree-sitter-queries.js';
export const rubyProvider = defineLanguage({
id: SupportedLanguages.Ruby,
extensions: ['.rb', '.rake', '.gemspec'],
treeSitterQueries: RUBY_QUERIES,
typeConfig: rubyConfig,
exportChecker: rubyExportChecker,
importResolver: resolveRubyImport,
callRouter: routeRubyCall,
importSemantics: 'wildcard',
});

View file

@ -0,0 +1,30 @@
/**
* Rust Language Provider
*
* Assembles all Rust-specific ingestion capabilities into a single
* LanguageProvider, following the Strategy pattern used by the pipeline.
*
* Key Rust traits:
* - importSemantics: 'named' (Rust has use X::{a, b})
* - mroStrategy: 'qualified-syntax' (Rust uses trait qualification, not MRO)
* - namedBindingExtractor: present (use X::{a, b} extracts named bindings)
*/
import { SupportedLanguages } from '../../../config/supported-languages.js';
import { defineLanguage } from '../language-provider.js';
import { typeConfig as rustConfig } from '../type-extractors/rust.js';
import { rustExportChecker } from '../export-detection.js';
import { resolveRustImport } from '../import-resolvers/rust.js';
import { extractRustNamedBindings } from '../named-bindings/rust.js';
import { RUST_QUERIES } from '../tree-sitter-queries.js';
export const rustProvider = defineLanguage({
id: SupportedLanguages.Rust,
extensions: ['.rs'],
treeSitterQueries: RUST_QUERIES,
typeConfig: rustConfig,
exportChecker: rustExportChecker,
importResolver: resolveRustImport,
namedBindingExtractor: extractRustNamedBindings,
mroStrategy: 'qualified-syntax',
});

View file

@ -0,0 +1,114 @@
/**
* Swift Language Provider
*
* Assembles all Swift-specific ingestion capabilities into a single
* LanguageProvider, following the Strategy pattern used by the pipeline.
*
* Key Swift traits:
* - importSemantics: 'wildcard' (Swift imports entire modules)
* - heritageDefaultEdge: 'IMPLEMENTS' (protocols are more common than class inheritance)
* - implicitImportWirer: all files in the same SPM target see each other
*/
import { SupportedLanguages } from '../../../config/supported-languages.js';
import { defineLanguage } from '../language-provider.js';
import { typeConfig as swiftConfig } from '../type-extractors/swift.js';
import { swiftExportChecker } from '../export-detection.js';
import { resolveSwiftImport } from '../import-resolvers/swift.js';
import { SWIFT_QUERIES } from '../tree-sitter-queries.js';
import type { SwiftPackageConfig } from '../language-config.js';
/**
* Group Swift files by SPM target for implicit module visibility.
* If SwiftPackageConfig is available, use target -> directory mappings.
* Otherwise, group all Swift files under a single "default" target
* (assumes a single-module Xcode project).
*/
function groupSwiftFilesByTarget(
swiftFiles: string[],
swiftPackageConfig: SwiftPackageConfig | null,
): Map<string, string[]> {
// No SPM config -> single target (common for Xcode projects)
if (!swiftPackageConfig || swiftPackageConfig.targets.size === 0) {
return new Map([['__default__', swiftFiles]]);
}
// Pre-convert target dirs to normalized prefix format once
const targets = [...swiftPackageConfig.targets.entries()].map(
([name, dir]) => ({ name, prefix: dir.replace(/\\/g, '/') + '/' }),
);
const groups = new Map<string, string[]>();
const defaultGroup: string[] = [];
for (const file of swiftFiles) {
const normalized = file.includes('\\') ? file.replace(/\\/g, '/') : file;
let assigned = false;
for (const { name, prefix } of targets) {
const idx = normalized.indexOf(prefix);
if (idx === 0 || (idx > 0 && normalized[idx - 1] === '/')) {
let group = groups.get(name);
if (!group) { group = []; groups.set(name, group); }
group.push(file);
assigned = true;
break;
}
}
if (!assigned) defaultGroup.push(file);
}
if (defaultGroup.length > 0) groups.set('__default__', defaultGroup);
return groups;
}
/**
* Wire implicit inter-file imports for Swift.
* All files in the same SPM target see each other (full module visibility).
* Two fast paths avoid unnecessary work:
* 1. No existing imports for src -> emit all (m-1) edges without Set.has checks
* 2. Existing imports present -> skip already-connected pairs
*/
function wireSwiftImplicitImports(
swiftFiles: string[],
importMap: ReadonlyMap<string, ReadonlySet<string>>,
addImportEdge: (src: string, target: string) => void,
projectConfig: unknown,
): void {
const configs = projectConfig as { swiftPackageConfig?: SwiftPackageConfig | null } | null;
const targetGroups = groupSwiftFilesByTarget(swiftFiles, configs?.swiftPackageConfig ?? null);
for (const group of targetGroups.values()) {
const m = group.length;
if (m <= 1) continue;
// All-pairs implicit edges: O(m²) is inherent for full module visibility.
for (let i = 0; i < m; i++) {
const src = group[i];
const existing = importMap.get(src);
if (!existing || existing.size === 0) {
// Fast path: no prior imports — emit all peers unconditionally
for (let j = 0; j < m; j++) {
if (i !== j) addImportEdge(src, group[j]);
}
} else {
// Dedup path: skip already-connected pairs
for (let j = 0; j < m; j++) {
if (i !== j && !existing.has(group[j])) {
addImportEdge(src, group[j]);
}
}
}
}
}
}
export const swiftProvider = defineLanguage({
id: SupportedLanguages.Swift,
extensions: ['.swift'],
treeSitterQueries: SWIFT_QUERIES,
typeConfig: swiftConfig,
exportChecker: swiftExportChecker,
importResolver: resolveSwiftImport,
importSemantics: 'wildcard',
heritageDefaultEdge: 'IMPLEMENTS',
implicitImportWirer: wireSwiftImplicitImports,
});

View file

@ -0,0 +1,36 @@
/**
* TypeScript and JavaScript language providers.
*
* Both languages share the same type extraction config (typescriptConfig),
* export checker (tsExportChecker), and named binding extractor
* (extractTsNamedBindings). They differ in file extensions, tree-sitter
* queries (TypeScript grammar has interface/type nodes), and language ID.
*/
import { SupportedLanguages } from '../../../config/supported-languages.js';
import { defineLanguage } from '../language-provider.js';
import { typeConfig as typescriptConfig } from '../type-extractors/typescript.js';
import { tsExportChecker } from '../export-detection.js';
import { resolveTypescriptImport, resolveJavascriptImport } from '../import-resolvers/standard.js';
import { extractTsNamedBindings } from '../named-bindings/typescript.js';
import { TYPESCRIPT_QUERIES, JAVASCRIPT_QUERIES } from '../tree-sitter-queries.js';
export const typescriptProvider = defineLanguage({
id: SupportedLanguages.TypeScript,
extensions: ['.ts', '.tsx'],
treeSitterQueries: TYPESCRIPT_QUERIES,
typeConfig: typescriptConfig,
exportChecker: tsExportChecker,
importResolver: resolveTypescriptImport,
namedBindingExtractor: extractTsNamedBindings,
});
export const javascriptProvider = defineLanguage({
id: SupportedLanguages.JavaScript,
extensions: ['.js', '.jsx'],
treeSitterQueries: JAVASCRIPT_QUERIES,
typeConfig: typescriptConfig,
exportChecker: tsExportChecker,
importResolver: resolveJavascriptImport,
namedBindingExtractor: extractTsNamedBindings,
});

View file

@ -22,6 +22,7 @@
import { KnowledgeGraph, GraphRelationship } from '../graph/types.js';
import { generateId } from '../../lib/utils.js';
import { SupportedLanguages } from '../../config/supported-languages.js';
import { getProvider } from './languages/index.js';
// ---------------------------------------------------------------------------
// Public types
@ -298,9 +299,10 @@ export function computeMRO(graph: KnowledgeGraph): MROResult {
if (!language) continue;
const className = classNode.properties.name;
// Compute linearized MRO depending on language
// Compute linearized MRO depending on language strategy
const provider = getProvider(language);
let mroOrder: string[];
if (language === SupportedLanguages.Python) {
if (provider.mroStrategy === 'c3') {
const c3Result = c3Linearize(classId, parentMap, c3Cache);
mroOrder = c3Result ?? gatherAncestors(classId, parentMap);
} else {
@ -345,8 +347,8 @@ export function computeMRO(graph: KnowledgeGraph): MROResult {
// Detect collisions: methods defined in 2+ different ancestors
const ambiguities: MethodAmbiguity[] = [];
// Compute transitive edge types once per class (only needed for C#/Java)
const needsEdgeTypes = language === SupportedLanguages.CSharp || language === SupportedLanguages.Java || language === SupportedLanguages.Kotlin;
// Compute transitive edge types once per class (only needed for implements-split languages)
const needsEdgeTypes = provider.mroStrategy === 'implements-split';
const classEdgeTypes = needsEdgeTypes
? buildTransitiveEdgeTypes(classId, parentMap, parentEdgeType)
: undefined;
@ -364,22 +366,20 @@ export function computeMRO(graph: KnowledgeGraph): MROResult {
let resolution: Resolution;
switch (language) {
case SupportedLanguages.CPlusPlus:
resolution = resolveByMroOrder(methodName, defs, mroOrder, 'C++ leftmost base');
switch (provider.mroStrategy) {
case 'leftmost-base':
resolution = resolveByMroOrder(methodName, defs, mroOrder, 'leftmost base');
break;
case SupportedLanguages.CSharp:
case SupportedLanguages.Java:
case SupportedLanguages.Kotlin:
case 'implements-split':
resolution = resolveCsharpJava(methodName, defs, classEdgeTypes);
break;
case SupportedLanguages.Python:
resolution = resolveByMroOrder(methodName, defs, mroOrder, 'Python C3 MRO');
case 'c3':
resolution = resolveByMroOrder(methodName, defs, mroOrder, 'C3 MRO');
break;
case SupportedLanguages.Rust:
case 'qualified-syntax':
resolution = {
resolvedTo: null,
reason: `Rust requires qualified syntax: <Type as Trait>::${methodName}()`,
reason: `requires qualified syntax: <Type as Trait>::${methodName}()`,
confidence: 0.5,
};
break;

View file

@ -1,396 +0,0 @@
import type { SymbolTable, SymbolDefinition } from './symbol-table.js';
import type { NamedImportMap } from './import-processor.js';
import type { NamedBinding } from './import-resolution.js';
import type { SyntaxNode } from './utils.js';
import { findChild } from './resolvers/utils.js';
/**
* Walk a named-binding re-export chain through NamedImportMap.
*
* When file A imports { User } from B, and B re-exports { User } from C,
* the NamedImportMap for A points to B, but B has no User definition.
* This function follows the chain: A→B→C until a definition is found.
*
* Returns the definitions found at the end of the chain, or null if the
* chain breaks (missing binding, circular reference, or depth exceeded).
* Max depth 5 to prevent infinite loops.
*
* @param allDefs Pre-computed `symbolTable.lookupFuzzy(name)` result — must be the
* complete unfiltered result. Passing a file-filtered subset will cause
* silent misses at depth=0 for non-aliased bindings.
*/
export function walkBindingChain(
name: string,
currentFilePath: string,
symbolTable: SymbolTable,
namedImportMap: NamedImportMap,
allDefs: SymbolDefinition[],
): SymbolDefinition[] | null {
let lookupFile = currentFilePath;
let lookupName = name;
const visited = new Set<string>();
for (let depth = 0; depth < 5; depth++) {
const bindings = namedImportMap.get(lookupFile);
if (!bindings) return null;
const binding = bindings.get(lookupName);
if (!binding) return null;
const key = `${binding.sourcePath}:${binding.exportedName}`;
if (visited.has(key)) return null; // circular
visited.add(key);
const targetName = binding.exportedName;
const resolvedDefs = targetName !== lookupName || depth > 0
? symbolTable.lookupFuzzy(targetName).filter(def => def.filePath === binding.sourcePath)
: allDefs.filter(def => def.filePath === binding.sourcePath);
if (resolvedDefs.length > 0) return resolvedDefs;
// No definition in source file → follow re-export chain
lookupFile = binding.sourcePath;
lookupName = targetName;
}
return null;
}
export function extractTsNamedBindings(importNode: SyntaxNode): NamedBinding[] | undefined {
// import_statement > import_clause > named_imports > import_specifier*
const importClause = findChild(importNode, 'import_clause');
if (importClause) {
const namedImports = findChild(importClause, 'named_imports');
if (!namedImports) return undefined; // default import, namespace import, or side-effect
const bindings: NamedBinding[] = [];
for (let i = 0; i < namedImports.namedChildCount; i++) {
const specifier = namedImports.namedChild(i);
if (specifier?.type !== 'import_specifier') continue;
const identifiers: string[] = [];
for (let j = 0; j < specifier.namedChildCount; j++) {
const child = specifier.namedChild(j);
if (child?.type === 'identifier') identifiers.push(child.text);
}
if (identifiers.length === 1) {
bindings.push({ local: identifiers[0], exported: identifiers[0] });
} else if (identifiers.length === 2) {
// import { Foo as Bar } → exported='Foo', local='Bar'
bindings.push({ local: identifiers[1], exported: identifiers[0] });
}
}
return bindings.length > 0 ? bindings : undefined;
}
// Re-export: export { X } from './y' → export_statement > export_clause > export_specifier
const exportClause = findChild(importNode, 'export_clause');
if (exportClause) {
const bindings: NamedBinding[] = [];
for (let i = 0; i < exportClause.namedChildCount; i++) {
const specifier = exportClause.namedChild(i);
if (specifier?.type !== 'export_specifier') continue;
const identifiers: string[] = [];
for (let j = 0; j < specifier.namedChildCount; j++) {
const child = specifier.namedChild(j);
if (child?.type === 'identifier') identifiers.push(child.text);
}
if (identifiers.length === 1) {
// export { User } from './base' → re-exports User as User
bindings.push({ local: identifiers[0], exported: identifiers[0] });
} else if (identifiers.length === 2) {
// export { Repo as Repository } from './models' → name=Repo, alias=Repository
// For re-exports, the first id is the source name, second is what's exported
// When another file imports { Repository }, they get Repo from the source
bindings.push({ local: identifiers[1], exported: identifiers[0] });
}
}
return bindings.length > 0 ? bindings : undefined;
}
return undefined;
}
export function extractPythonNamedBindings(importNode: SyntaxNode): NamedBinding[] | undefined {
// Handle: from x import User, Repo as R
if (importNode.type === 'import_from_statement') {
const bindings: NamedBinding[] = [];
for (let i = 0; i < importNode.namedChildCount; i++) {
const child = importNode.namedChild(i);
if (!child) continue;
if (child.type === 'dotted_name') {
// Skip the module_name (first dotted_name is the source module)
const fieldName = importNode.childForFieldName?.('module_name');
if (fieldName && child.startIndex === fieldName.startIndex) continue;
// This is an imported name: from x import User
const name = child.text;
if (name) bindings.push({ local: name, exported: name });
}
if (child.type === 'aliased_import') {
// from x import Repo as R
const dottedName = findChild(child, 'dotted_name');
const aliasIdent = findChild(child, 'identifier');
if (dottedName && aliasIdent) {
bindings.push({ local: aliasIdent.text, exported: dottedName.text });
}
}
}
return bindings.length > 0 ? bindings : undefined;
}
// Handle: import numpy as np (import_statement with aliased_import child)
// Tagged with isModuleAlias so applyImportResult routes these directly to
// moduleAliasMap (e.g. "np" → "numpy.py") instead of namedImportMap.
if (importNode.type === 'import_statement') {
const bindings: NamedBinding[] = [];
for (let i = 0; i < importNode.namedChildCount; i++) {
const child = importNode.namedChild(i);
if (!child || child.type !== 'aliased_import') continue;
const dottedName = findChild(child, 'dotted_name');
const aliasIdent = findChild(child, 'identifier');
if (dottedName && aliasIdent) {
bindings.push({ local: aliasIdent.text, exported: dottedName.text, isModuleAlias: true });
}
}
return bindings.length > 0 ? bindings : undefined;
}
return undefined;
}
export function extractKotlinNamedBindings(importNode: SyntaxNode): NamedBinding[] | undefined {
// import_header > identifier + import_alias > simple_identifier
if (importNode.type !== 'import_header') return undefined;
const fullIdent = findChild(importNode, 'identifier');
if (!fullIdent) return undefined;
const fullText = fullIdent.text;
const exportedName = fullText.includes('.') ? fullText.split('.').pop()! : fullText;
const importAlias = findChild(importNode, 'import_alias');
if (importAlias) {
// Aliased: import com.example.User as U
const aliasIdent = findChild(importAlias, 'simple_identifier');
if (!aliasIdent) return undefined;
return [{ local: aliasIdent.text, exported: exportedName }];
}
// Non-aliased: import com.example.User → local="User", exported="User"
// Also handles top-level function imports: import models.getUser → local="getUser"
// Skip wildcard imports (ending in *)
if (fullText.endsWith('.*') || fullText.endsWith('*')) return undefined;
// Skip class-member imports (e.g., import util.OneArg.writeAudit) where the
// second-to-last segment is PascalCase (a class name). Multiple member imports
// with the same function name would collide in NamedImportMap, breaking
// arity-based disambiguation. Top-level function imports (import models.getUser)
// and class imports (import models.User) have package-only prefixes.
const segments = fullText.split('.');
if (segments.length >= 3) {
const parentSegment = segments[segments.length - 2];
if (parentSegment[0] && parentSegment[0] === parentSegment[0].toUpperCase()) return undefined;
}
return [{ local: exportedName, exported: exportedName }];
}
export function extractRustNamedBindings(importNode: SyntaxNode): NamedBinding[] | undefined {
// use_declaration may contain use_as_clause at any depth
if (importNode.type !== 'use_declaration') return undefined;
const bindings: NamedBinding[] = [];
collectRustBindings(importNode, bindings);
return bindings.length > 0 ? bindings : undefined;
}
function collectRustBindings(node: SyntaxNode, bindings: NamedBinding[]): void {
if (node.type === 'use_as_clause') {
// First identifier = exported name, second identifier = local alias
const idents: string[] = [];
for (let i = 0; i < node.namedChildCount; i++) {
const child = node.namedChild(i);
if (child?.type === 'identifier') idents.push(child.text);
// For scoped_identifier, extract the last segment
if (child?.type === 'scoped_identifier') {
const nameNode = child.childForFieldName?.('name');
if (nameNode) idents.push(nameNode.text);
}
}
if (idents.length === 2) {
bindings.push({ local: idents[1], exported: idents[0] });
}
return;
}
// Terminal identifier in a use_list: use crate::models::{User, Repo}
if (node.type === 'identifier' && node.parent?.type === 'use_list') {
bindings.push({ local: node.text, exported: node.text });
return;
}
// Skip scoped_identifier that serves as path prefix in scoped_use_list
// e.g. use crate::models::{User, Repo} — the path node "crate::models" is not an importable symbol
if (node.type === 'scoped_identifier' && node.parent?.type === 'scoped_use_list') {
return; // path prefix — the use_list sibling handles the actual symbols
}
// Terminal scoped_identifier: use crate::models::User;
// Only extract if this is a leaf (no deeper use_list/use_as_clause/scoped_use_list)
if (node.type === 'scoped_identifier') {
let hasDeeper = false;
for (let i = 0; i < node.namedChildCount; i++) {
const child = node.namedChild(i);
if (child?.type === 'use_list' || child?.type === 'use_as_clause' || child?.type === 'scoped_use_list') {
hasDeeper = true;
break;
}
}
if (!hasDeeper) {
const nameNode = node.childForFieldName?.('name');
if (nameNode) {
bindings.push({ local: nameNode.text, exported: nameNode.text });
}
return;
}
}
// Recurse into children
for (let i = 0; i < node.namedChildCount; i++) {
const child = node.namedChild(i);
if (child) collectRustBindings(child, bindings);
}
}
export function extractPhpNamedBindings(importNode: SyntaxNode): NamedBinding[] | undefined {
// namespace_use_declaration > namespace_use_clause* (flat)
// namespace_use_declaration > namespace_use_group > namespace_use_clause* (grouped)
if (importNode.type !== 'namespace_use_declaration') return undefined;
// Skip 'use function' and 'use const' declarations — these import callables/constants,
// not class types, and should not be added to namedImportMap as type bindings.
const useTypeNode = importNode.childForFieldName?.('type');
if (useTypeNode && (useTypeNode.text === 'function' || useTypeNode.text === 'const')) {
return undefined;
}
const bindings: NamedBinding[] = [];
// Collect all clauses — from direct children AND from namespace_use_group
const clauses: SyntaxNode[] = [];
for (let i = 0; i < importNode.namedChildCount; i++) {
const child = importNode.namedChild(i);
if (child?.type === 'namespace_use_clause') {
clauses.push(child);
} else if (child?.type === 'namespace_use_group') {
for (let j = 0; j < child.namedChildCount; j++) {
const groupChild = child.namedChild(j);
if (groupChild?.type === 'namespace_use_clause') clauses.push(groupChild);
}
}
}
for (const clause of clauses) {
// Flat imports: qualified_name + name (alias)
let qualifiedName: SyntaxNode | null = null;
const names: SyntaxNode[] = [];
for (let j = 0; j < clause.namedChildCount; j++) {
const child = clause.namedChild(j);
if (child?.type === 'qualified_name') qualifiedName = child;
else if (child?.type === 'name') names.push(child);
}
if (qualifiedName && names.length > 0) {
// Flat aliased import: use App\Models\Repo as R;
const fullText = qualifiedName.text;
const exportedName = fullText.includes('\\') ? fullText.split('\\').pop()! : fullText;
bindings.push({ local: names[0].text, exported: exportedName });
} else if (qualifiedName && names.length === 0) {
// Flat non-aliased import: use App\Models\User;
const fullText = qualifiedName.text;
const lastSegment = fullText.includes('\\') ? fullText.split('\\').pop()! : fullText;
bindings.push({ local: lastSegment, exported: lastSegment });
} else if (!qualifiedName && names.length >= 2) {
// Grouped aliased import: {Repo as R} — first name = exported, second = alias
bindings.push({ local: names[1].text, exported: names[0].text });
} else if (!qualifiedName && names.length === 1) {
// Grouped non-aliased import: {User} in use App\Models\{User, Repo as R}
bindings.push({ local: names[0].text, exported: names[0].text });
}
}
return bindings.length > 0 ? bindings : undefined;
}
export function extractCsharpNamedBindings(importNode: SyntaxNode): NamedBinding[] | undefined {
// using_directive — three forms:
// using Alias = NS.Type; → aliasIdent + qualifiedName
// using static NS.Type; → static + qualifiedName (no alias)
// using NS; → qualifiedName only (namespace, not capturable)
if (importNode.type !== 'using_directive') return undefined;
let aliasIdent: SyntaxNode | null = null;
let qualifiedName: SyntaxNode | null = null;
let isStatic = false;
for (let i = 0; i < importNode.childCount; i++) {
const child = importNode.child(i);
if (child?.text === 'static') isStatic = true;
}
for (let i = 0; i < importNode.namedChildCount; i++) {
const child = importNode.namedChild(i);
if (child?.type === 'identifier' && !aliasIdent) aliasIdent = child;
else if (child?.type === 'qualified_name') qualifiedName = child;
}
// Form 1: using Alias = NS.Type;
if (aliasIdent && qualifiedName) {
const fullText = qualifiedName.text;
const exportedName = fullText.includes('.') ? fullText.split('.').pop()! : fullText;
return [{ local: aliasIdent.text, exported: exportedName }];
}
// Form 2: using static NS.Type; — last segment is the class name
if (isStatic && qualifiedName) {
const fullText = qualifiedName.text;
const lastSegment = fullText.includes('.') ? fullText.split('.').pop()! : fullText;
return [{ local: lastSegment, exported: lastSegment }];
}
// Form 3: using NS; — namespace import, can't resolve to per-symbol bindings
return undefined;
}
export function extractJavaNamedBindings(importNode: SyntaxNode): NamedBinding[] | undefined {
// import_declaration > scoped_identifier "com.example.models.User"
// Wildcard imports (.*) don't produce named bindings
if (importNode.type !== 'import_declaration') return undefined;
// Check for asterisk (wildcard import) and static modifier
let isStatic = false;
for (let i = 0; i < importNode.childCount; i++) {
const child = importNode.child(i);
if (child?.type === 'asterisk') return undefined;
if (child?.text === 'static') isStatic = true;
}
const scopedId = findChild(importNode, 'scoped_identifier');
if (!scopedId) return undefined;
const fullText = scopedId.text;
const lastDot = fullText.lastIndexOf('.');
if (lastDot === -1) return undefined;
const name = fullText.slice(lastDot + 1);
// Non-static: skip lowercase names — those are package imports, not class imports.
// Static: allow lowercase — `import static models.UserFactory.getUser` imports a method.
if (!isStatic && name[0] && name[0] === name[0].toLowerCase()) return undefined;
return [{ local: name, exported: name }];
}

View file

@ -0,0 +1,54 @@
import type { SymbolTable, SymbolDefinition } from './symbol-table.js';
import type { NamedImportMap } from './import-processor.js';
/**
* Walk a named-binding re-export chain through NamedImportMap.
*
* When file A imports { User } from B, and B re-exports { User } from C,
* the NamedImportMap for A points to B, but B has no User definition.
* This function follows the chain: A→B→C until a definition is found.
*
* Returns the definitions found at the end of the chain, or null if the
* chain breaks (missing binding, circular reference, or depth exceeded).
* Max depth 5 to prevent infinite loops.
*
* @param allDefs Pre-computed `symbolTable.lookupFuzzy(name)` result — must be the
* complete unfiltered result. Passing a file-filtered subset will cause
* silent misses at depth=0 for non-aliased bindings.
*/
export function walkBindingChain(
name: string,
currentFilePath: string,
symbolTable: SymbolTable,
namedImportMap: NamedImportMap,
allDefs: SymbolDefinition[],
): SymbolDefinition[] | null {
let lookupFile = currentFilePath;
let lookupName = name;
const visited = new Set<string>();
for (let depth = 0; depth < 5; depth++) {
const bindings = namedImportMap.get(lookupFile);
if (!bindings) return null;
const binding = bindings.get(lookupName);
if (!binding) return null;
const key = `${binding.sourcePath}:${binding.exportedName}`;
if (visited.has(key)) return null; // circular
visited.add(key);
const targetName = binding.exportedName;
const resolvedDefs = targetName !== lookupName || depth > 0
? symbolTable.lookupFuzzy(targetName).filter(def => def.filePath === binding.sourcePath)
: allDefs.filter(def => def.filePath === binding.sourcePath);
if (resolvedDefs.length > 0) return resolvedDefs;
// No definition in source file → follow re-export chain
lookupFile = binding.sourcePath;
lookupName = targetName;
}
return null;
}

View file

@ -0,0 +1,40 @@
import type { SyntaxNode } from '../utils/ast-helpers.js';
import type { NamedBinding } from './types.js';
export function extractCSharpNamedBindings(importNode: SyntaxNode): NamedBinding[] | undefined {
// using_directive — three forms:
// using Alias = NS.Type; → aliasIdent + qualifiedName
// using static NS.Type; → static + qualifiedName (no alias)
// using NS; → qualifiedName only (namespace, not capturable)
if (importNode.type !== 'using_directive') return undefined;
let aliasIdent: SyntaxNode | null = null;
let qualifiedName: SyntaxNode | null = null;
let isStatic = false;
for (let i = 0; i < importNode.childCount; i++) {
const child = importNode.child(i);
if (child?.text === 'static') isStatic = true;
}
for (let i = 0; i < importNode.namedChildCount; i++) {
const child = importNode.namedChild(i);
if (child?.type === 'identifier' && !aliasIdent) aliasIdent = child;
else if (child?.type === 'qualified_name') qualifiedName = child;
}
// Form 1: using Alias = NS.Type;
if (aliasIdent && qualifiedName) {
const fullText = qualifiedName.text;
const exportedName = fullText.includes('.') ? fullText.split('.').pop()! : fullText;
return [{ local: aliasIdent.text, exported: exportedName }];
}
// Form 2: using static NS.Type; — last segment is the class name
if (isStatic && qualifiedName) {
const fullText = qualifiedName.text;
const lastSegment = fullText.includes('.') ? fullText.split('.').pop()! : fullText;
return [{ local: lastSegment, exported: lastSegment }];
}
// Form 3: using NS; — namespace import, can't resolve to per-symbol bindings
return undefined;
}

View file

@ -0,0 +1,30 @@
import { findChild, type SyntaxNode } from '../utils/ast-helpers.js';
import type { NamedBinding } from './types.js';
export function extractJavaNamedBindings(importNode: SyntaxNode): NamedBinding[] | undefined {
// import_declaration > scoped_identifier "com.example.models.User"
// Wildcard imports (.*) don't produce named bindings
if (importNode.type !== 'import_declaration') return undefined;
// Check for asterisk (wildcard import) and static modifier
let isStatic = false;
for (let i = 0; i < importNode.childCount; i++) {
const child = importNode.child(i);
if (child?.type === 'asterisk') return undefined;
if (child?.text === 'static') isStatic = true;
}
const scopedId = findChild(importNode, 'scoped_identifier');
if (!scopedId) return undefined;
const fullText = scopedId.text;
const lastDot = fullText.lastIndexOf('.');
if (lastDot === -1) return undefined;
const name = fullText.slice(lastDot + 1);
// Non-static: skip lowercase names — those are package imports, not class imports.
// Static: allow lowercase — `import static models.UserFactory.getUser` imports a method.
if (!isStatic && name[0] && name[0] === name[0].toLowerCase()) return undefined;
return [{ local: name, exported: name }];
}

View file

@ -0,0 +1,37 @@
import { findChild, type SyntaxNode } from '../utils/ast-helpers.js';
import type { NamedBinding } from './types.js';
export function extractKotlinNamedBindings(importNode: SyntaxNode): NamedBinding[] | undefined {
// import_header > identifier + import_alias > simple_identifier
if (importNode.type !== 'import_header') return undefined;
const fullIdent = findChild(importNode, 'identifier');
if (!fullIdent) return undefined;
const fullText = fullIdent.text;
const exportedName = fullText.includes('.') ? fullText.split('.').pop()! : fullText;
const importAlias = findChild(importNode, 'import_alias');
if (importAlias) {
// Aliased: import com.example.User as U
const aliasIdent = findChild(importAlias, 'simple_identifier');
if (!aliasIdent) return undefined;
return [{ local: aliasIdent.text, exported: exportedName }];
}
// Non-aliased: import com.example.User → local="User", exported="User"
// Also handles top-level function imports: import models.getUser → local="getUser"
// Skip wildcard imports (ending in *)
if (fullText.endsWith('.*') || fullText.endsWith('*')) return undefined;
// Skip class-member imports (e.g., import util.OneArg.writeAudit) where the
// second-to-last segment is PascalCase (a class name). Multiple member imports
// with the same function name would collide in NamedImportMap, breaking
// arity-based disambiguation. Top-level function imports (import models.getUser)
// and class imports (import models.User) have package-only prefixes.
const segments = fullText.split('.');
if (segments.length >= 3) {
const parentSegment = segments[segments.length - 2];
if (parentSegment[0] && parentSegment[0] === parentSegment[0].toUpperCase()) return undefined;
}
return [{ local: exportedName, exported: exportedName }];
}

View file

@ -0,0 +1,61 @@
import type { SyntaxNode } from '../utils/ast-helpers.js';
import type { NamedBinding } from './types.js';
export function extractPhpNamedBindings(importNode: SyntaxNode): NamedBinding[] | undefined {
// namespace_use_declaration > namespace_use_clause* (flat)
// namespace_use_declaration > namespace_use_group > namespace_use_clause* (grouped)
if (importNode.type !== 'namespace_use_declaration') return undefined;
// Skip 'use function' and 'use const' declarations — these import callables/constants,
// not class types, and should not be added to namedImportMap as type bindings.
const useTypeNode = importNode.childForFieldName?.('type');
if (useTypeNode && (useTypeNode.text === 'function' || useTypeNode.text === 'const')) {
return undefined;
}
const bindings: NamedBinding[] = [];
// Collect all clauses — from direct children AND from namespace_use_group
const clauses: SyntaxNode[] = [];
for (let i = 0; i < importNode.namedChildCount; i++) {
const child = importNode.namedChild(i);
if (child?.type === 'namespace_use_clause') {
clauses.push(child);
} else if (child?.type === 'namespace_use_group') {
for (let j = 0; j < child.namedChildCount; j++) {
const groupChild = child.namedChild(j);
if (groupChild?.type === 'namespace_use_clause') clauses.push(groupChild);
}
}
}
for (const clause of clauses) {
// Flat imports: qualified_name + name (alias)
let qualifiedName: SyntaxNode | null = null;
const names: SyntaxNode[] = [];
for (let j = 0; j < clause.namedChildCount; j++) {
const child = clause.namedChild(j);
if (child?.type === 'qualified_name') qualifiedName = child;
else if (child?.type === 'name') names.push(child);
}
if (qualifiedName && names.length > 0) {
// Flat aliased import: use App\Models\Repo as R;
const fullText = qualifiedName.text;
const exportedName = fullText.includes('\\') ? fullText.split('\\').pop()! : fullText;
bindings.push({ local: names[0].text, exported: exportedName });
} else if (qualifiedName && names.length === 0) {
// Flat non-aliased import: use App\Models\User;
const fullText = qualifiedName.text;
const lastSegment = fullText.includes('\\') ? fullText.split('\\').pop()! : fullText;
bindings.push({ local: lastSegment, exported: lastSegment });
} else if (!qualifiedName && names.length >= 2) {
// Grouped aliased import: {Repo as R} — first name = exported, second = alias
bindings.push({ local: names[1].text, exported: names[0].text });
} else if (!qualifiedName && names.length === 1) {
// Grouped non-aliased import: {User} in use App\Models\{User, Repo as R}
bindings.push({ local: names[0].text, exported: names[0].text });
}
}
return bindings.length > 0 ? bindings : undefined;
}

View file

@ -0,0 +1,55 @@
import { findChild, type SyntaxNode } from '../utils/ast-helpers.js';
import type { NamedBinding } from './types.js';
export function extractPythonNamedBindings(importNode: SyntaxNode): NamedBinding[] | undefined {
// Handle: from x import User, Repo as R
if (importNode.type === 'import_from_statement') {
const bindings: NamedBinding[] = [];
for (let i = 0; i < importNode.namedChildCount; i++) {
const child = importNode.namedChild(i);
if (!child) continue;
if (child.type === 'dotted_name') {
// Skip the module_name (first dotted_name is the source module)
const fieldName = importNode.childForFieldName?.('module_name');
if (fieldName && child.startIndex === fieldName.startIndex) continue;
// This is an imported name: from x import User
const name = child.text;
if (name) bindings.push({ local: name, exported: name });
}
if (child.type === 'aliased_import') {
// from x import Repo as R
const dottedName = findChild(child, 'dotted_name');
const aliasIdent = findChild(child, 'identifier');
if (dottedName && aliasIdent) {
bindings.push({ local: aliasIdent.text, exported: dottedName.text });
}
}
}
return bindings.length > 0 ? bindings : undefined;
}
// Handle: import numpy as np (import_statement with aliased_import child)
// Tagged with isModuleAlias so applyImportResult routes these directly to
// moduleAliasMap (e.g. "np" → "numpy.py") instead of namedImportMap.
if (importNode.type === 'import_statement') {
const bindings: NamedBinding[] = [];
for (let i = 0; i < importNode.namedChildCount; i++) {
const child = importNode.namedChild(i);
if (!child || child.type !== 'aliased_import') continue;
const dottedName = findChild(child, 'dotted_name');
const aliasIdent = findChild(child, 'identifier');
if (dottedName && aliasIdent) {
bindings.push({ local: aliasIdent.text, exported: dottedName.text, isModuleAlias: true });
}
}
return bindings.length > 0 ? bindings : undefined;
}
return undefined;
}

View file

@ -0,0 +1,69 @@
import type { SyntaxNode } from '../utils/ast-helpers.js';
import type { NamedBinding } from './types.js';
export function extractRustNamedBindings(importNode: SyntaxNode): NamedBinding[] | undefined {
// use_declaration may contain use_as_clause at any depth
if (importNode.type !== 'use_declaration') return undefined;
const bindings: NamedBinding[] = [];
collectRustBindings(importNode, bindings);
return bindings.length > 0 ? bindings : undefined;
}
function collectRustBindings(node: SyntaxNode, bindings: NamedBinding[]): void {
if (node.type === 'use_as_clause') {
// First identifier = exported name, second identifier = local alias
const idents: string[] = [];
for (let i = 0; i < node.namedChildCount; i++) {
const child = node.namedChild(i);
if (child?.type === 'identifier') idents.push(child.text);
// For scoped_identifier, extract the last segment
if (child?.type === 'scoped_identifier') {
const nameNode = child.childForFieldName?.('name');
if (nameNode) idents.push(nameNode.text);
}
}
if (idents.length === 2) {
bindings.push({ local: idents[1], exported: idents[0] });
}
return;
}
// Terminal identifier in a use_list: use crate::models::{User, Repo}
if (node.type === 'identifier' && node.parent?.type === 'use_list') {
bindings.push({ local: node.text, exported: node.text });
return;
}
// Skip scoped_identifier that serves as path prefix in scoped_use_list
// e.g. use crate::models::{User, Repo} — the path node "crate::models" is not an importable symbol
if (node.type === 'scoped_identifier' && node.parent?.type === 'scoped_use_list') {
return; // path prefix — the use_list sibling handles the actual symbols
}
// Terminal scoped_identifier: use crate::models::User;
// Only extract if this is a leaf (no deeper use_list/use_as_clause/scoped_use_list)
if (node.type === 'scoped_identifier') {
let hasDeeper = false;
for (let i = 0; i < node.namedChildCount; i++) {
const child = node.namedChild(i);
if (child?.type === 'use_list' || child?.type === 'use_as_clause' || child?.type === 'scoped_use_list') {
hasDeeper = true;
break;
}
}
if (!hasDeeper) {
const nameNode = node.childForFieldName?.('name');
if (nameNode) {
bindings.push({ local: nameNode.text, exported: nameNode.text });
}
return;
}
}
// Recurse into children
for (let i = 0; i < node.namedChildCount; i++) {
const child = node.namedChild(i);
if (child) collectRustBindings(child, bindings);
}
}

View file

@ -0,0 +1,15 @@
/**
* Named binding types — shared across all per-language binding extractors.
*
* Extracted from import-resolution.ts to co-locate types with their consumers.
*/
import type { SyntaxNode } from '../utils/ast-helpers.js';
/** A single named import binding: local name in the importing file and exported name from the source.
* When `isModuleAlias` is true, the binding represents a Python `import X as Y` module alias
* and is routed to moduleAliasMap instead of namedImportMap during import processing. */
export interface NamedBinding { local: string; exported: string; isModuleAlias?: boolean }
/** Per-language named binding extractor -- optional (returns undefined if language has no named imports). */
export type NamedBindingExtractorFn = (importNode: SyntaxNode) => NamedBinding[] | undefined;

View file

@ -0,0 +1,60 @@
import { findChild, type SyntaxNode } from '../utils/ast-helpers.js';
import type { NamedBinding } from './types.js';
export function extractTsNamedBindings(importNode: SyntaxNode): NamedBinding[] | undefined {
// import_statement > import_clause > named_imports > import_specifier*
const importClause = findChild(importNode, 'import_clause');
if (importClause) {
const namedImports = findChild(importClause, 'named_imports');
if (!namedImports) return undefined; // default import, namespace import, or side-effect
const bindings: NamedBinding[] = [];
for (let i = 0; i < namedImports.namedChildCount; i++) {
const specifier = namedImports.namedChild(i);
if (specifier?.type !== 'import_specifier') continue;
const identifiers: string[] = [];
for (let j = 0; j < specifier.namedChildCount; j++) {
const child = specifier.namedChild(j);
if (child?.type === 'identifier') identifiers.push(child.text);
}
if (identifiers.length === 1) {
bindings.push({ local: identifiers[0], exported: identifiers[0] });
} else if (identifiers.length === 2) {
// import { Foo as Bar } → exported='Foo', local='Bar'
bindings.push({ local: identifiers[1], exported: identifiers[0] });
}
}
return bindings.length > 0 ? bindings : undefined;
}
// Re-export: export { X } from './y' → export_statement > export_clause > export_specifier
const exportClause = findChild(importNode, 'export_clause');
if (exportClause) {
const bindings: NamedBinding[] = [];
for (let i = 0; i < exportClause.namedChildCount; i++) {
const specifier = exportClause.namedChild(i);
if (specifier?.type !== 'export_specifier') continue;
const identifiers: string[] = [];
for (let j = 0; j < specifier.namedChildCount; j++) {
const child = specifier.namedChild(j);
if (child?.type === 'identifier') identifiers.push(child.text);
}
if (identifiers.length === 1) {
// export { User } from './base' → re-exports User as User
bindings.push({ local: identifiers[0], exported: identifiers[0] });
} else if (identifiers.length === 2) {
// export { Repo as Repository } from './models' → name=Repo, alias=Repository
// For re-exports, the first id is the source name, second is what's exported
// When another file imports { Repository }, they get Repo from the source
bindings.push({ local: identifiers[1], exported: identifiers[0] });
}
}
return bindings.length > 0 ? bindings : undefined;
}
return undefined;
}

View file

@ -1,15 +1,15 @@
import { KnowledgeGraph, GraphNode, GraphRelationship, type NodeLabel } from '../graph/types.js';
import Parser from 'tree-sitter';
import { loadParser, loadLanguage, isLanguageAvailable } from '../tree-sitter/parser-loader.js';
import { LANGUAGE_QUERIES } from './tree-sitter-queries.js';
import { getProvider } from './languages/index.js';
import { generateId } from '../../lib/utils.js';
import { SymbolTable } from './symbol-table.js';
import { ASTCache } from './ast-cache.js';
import { getLanguageFromFilename, yieldToEventLoop, getDefinitionNodeFromCaptures, findEnclosingClassId, extractMethodSignature, getLabelFromCaptures } from './utils.js';
import { getLanguageFromFilename } from './utils/language-detection.js';
import { yieldToEventLoop } from './utils/event-loop.js';
import { getDefinitionNodeFromCaptures, findEnclosingClassId, extractMethodSignature, getLabelFromCaptures } from './utils/ast-helpers.js';
import { extractPropertyDeclaredType } from './type-extractors/shared.js';
import { isNodeExported } from './export-detection.js';
import { detectFrameworkFromAST } from './framework-detection.js';
import { typeConfigs } from './type-extractors/index.js';
import { WorkerPool } from './workers/worker-pool.js';
import type { ParseWorkerResult, ParseWorkerInput, ExtractedImport, ExtractedCall, ExtractedAssignment, ExtractedHeritage, ExtractedRoute, ExtractedFetchCall, ExtractedDecoratorRoute, ExtractedToolDef, FileConstructorBindings, FileTypeEnvBindings } from './workers/parse-worker.js';
import { getTreeSitterBufferSize, TREE_SITTER_MAX_BUFFER } from './constants.js';
@ -130,6 +130,27 @@ const processParsingWithWorkers = async (
// Sequential fallback (original implementation)
// ============================================================================
// Inline caches to avoid repeated parent-walks per node (same pattern as parse-worker.ts).
// Keyed by tree-sitter node reference — cleared at the start of each file.
const classIdCache = new Map<any, string | null>();
const exportCache = new Map<any, boolean>();
const cachedFindEnclosingClassId = (node: any, filePath: string): string | null => {
const cached = classIdCache.get(node);
if (cached !== undefined) return cached;
const result = findEnclosingClassId(node, filePath);
classIdCache.set(node, result);
return result;
};
const cachedExportCheck = (checker: (node: any, name: string) => boolean, node: any, name: string): boolean => {
const cached = exportCache.get(node);
if (cached !== undefined) return cached;
const result = checker(node, name);
exportCache.set(node, result);
return result;
};
const processParsingSequential = async (
graph: KnowledgeGraph,
files: { path: string; content: string }[],
@ -144,6 +165,10 @@ const processParsingSequential = async (
for (let i = 0; i < files.length; i++) {
const file = files[i];
// Reset memoization before each new file (node refs are per-tree)
classIdCache.clear();
exportCache.clear();
onFileProgress?.(i + 1, total, file.path);
if (i % 20 === 0) await yieldToEventLoop();
@ -177,7 +202,8 @@ const processParsingSequential = async (
astCache.set(file.path, tree);
const queryString = LANGUAGE_QUERIES[language];
const provider = getProvider(language);
const queryString = provider.treeSitterQueries;
if (!queryString) {
continue;
}
@ -200,7 +226,7 @@ const processParsingSequential = async (
captureMap[c.name] = c.node;
});
const nodeLabel = getLabelFromCaptures(captureMap, language);
const nodeLabel = getLabelFromCaptures(captureMap, provider);
if (!nodeLabel) return;
const nameNode = captureMap['name'];
@ -225,7 +251,7 @@ const processParsingSequential = async (
// Language-specific return type fallback (e.g. Ruby YARD @return [Type])
// Also upgrades uninformative AST types like PHP `array` with PHPDoc `@return User[]`
if (methodSig && (!methodSig.returnType || methodSig.returnType === 'array' || methodSig.returnType === 'iterable') && definitionNode) {
const tc = typeConfigs[language as keyof typeof typeConfigs];
const tc = provider.typeConfig;
if (tc?.extractReturnType) {
const docReturn = tc.extractReturnType(definitionNode);
if (docReturn) methodSig.returnType = docReturn;
@ -241,7 +267,7 @@ const processParsingSequential = async (
startLine: definitionNodeForRange ? definitionNodeForRange.startPosition.row : startLine,
endLine: definitionNodeForRange ? definitionNodeForRange.endPosition.row : startLine,
language: language,
isExported: isNodeExported(nameNode || definitionNodeForRange, nodeName, language),
isExported: cachedExportCheck(provider.exportChecker, nameNode || definitionNodeForRange, nodeName),
...(frameworkHint ? {
astFrameworkMultiplier: frameworkHint.entryPointMultiplier,
astFrameworkReason: frameworkHint.reason,
@ -260,7 +286,7 @@ const processParsingSequential = async (
// Compute enclosing class for Method/Constructor/Property/Function — used for both ownerId and HAS_METHOD
// Function is included because Kotlin/Rust/Python capture class methods as Function nodes
const needsOwner = nodeLabel === 'Method' || nodeLabel === 'Constructor' || nodeLabel === 'Property' || nodeLabel === 'Function';
const enclosingClassId = needsOwner ? findEnclosingClassId(nameNode || definitionNodeForRange, file.path) : null;
const enclosingClassId = needsOwner ? cachedFindEnclosingClassId(nameNode || definitionNodeForRange, file.path) : null;
// Extract declared type for Property nodes (field/property type annotations)
const declaredType = (nodeLabel === 'Property' && definitionNode)

File diff suppressed because it is too large Load diff

View file

@ -17,7 +17,7 @@ import type { SymbolTable, SymbolDefinition } from './symbol-table.js';
import { createSymbolTable } from './symbol-table.js';
import type { NamedImportBinding } from './import-processor.js';
import { isFileInPackageDir } from './import-processor.js';
import { walkBindingChain } from './named-binding-extraction.js';
import { walkBindingChain } from './named-binding-processor.js';
/** Resolution tier for tracking, logging, and test assertions. */
export type ResolutionTier = 'same-file' | 'import-scoped' | 'global';

View file

@ -1,27 +0,0 @@
/**
* Language-specific import resolvers.
* Extracted from import-processor.ts for maintainability.
*/
export { EXTENSIONS, tryResolveWithExtensions, buildSuffixIndex, suffixResolve, EMPTY_INDEX } from './utils.js';
export type { SuffixIndex } from './utils.js';
export { KOTLIN_EXTENSIONS, appendKotlinWildcard, resolveJvmWildcard, resolveJvmMemberImport } from './jvm.js';
export { resolveGoPackageDir, resolveGoPackage } from './go.js';
export type { GoModuleConfig } from './go.js';
export { resolveCSharpImport, resolveCSharpNamespaceDir } from './csharp.js';
export type { CSharpProjectConfig } from './csharp.js';
export { resolvePhpImport } from './php.js';
export type { ComposerConfig } from './php.js';
export { resolveRustImport, tryRustModulePath } from './rust.js';
export { resolveRubyImport } from './ruby.js';
export { resolvePythonImport } from './python.js';
export { resolveImportPath, RESOLVE_CACHE_CAP } from './standard.js';
export type { TsconfigPaths } from './standard.js';

View file

@ -0,0 +1,41 @@
// Expo Router route extraction utilities.
export function expoFileToRouteURL(filePath: string): string | null {
const normalized = filePath.replace(/\\/g, '/');
// Skip TypeScript declaration files
if (/\.d\.tsx?$/.test(normalized)) return null;
// Must be inside an app/ directory
const appMatch = normalized.match(/app\/(.+)\.(tsx?|jsx?)$/);
if (!appMatch) return null;
const segments = appMatch[1];
const fileName = segments.split('/').pop() || '';
// Skip layout files (_layout.tsx)
if (fileName.startsWith('_')) return null;
// Skip special Expo files (+not-found.tsx, +html.tsx) — but NOT +api files
if (fileName.startsWith('+') && !fileName.startsWith('+api')) return null;
// Handle Expo API routes: users+api.ts → /users
if (fileName.endsWith('+api')) {
const apiSegments = segments.replace(/\+api$/, '');
const route = '/' + stripRouteGroups(apiSegments);
return stripIndex(route);
}
// Regular screen route
const route = '/' + stripRouteGroups(segments);
return stripIndex(route);
}
function stripRouteGroups(path: string): string {
return path.replace(/\([^)]+\)\/?/g, '');
}
function stripIndex(route: string): string {
if (route === '/index') return '/';
return route.replace(/\/index$/, '') || '/';
}

View file

@ -3,6 +3,10 @@
* Detects wrapper patterns like: export const POST = withA(withB(withC(handler)))
*/
/** Keywords that terminate middleware chain walking (not wrapper function names) */
/** Names that are composition wrappers, not middleware functions themselves. */
const COMPOSER_NAMES = new Set(['middleware', 'default', 'chain', 'compose']);
/** Keywords that terminate middleware chain walking (not wrapper function names) */
export const MIDDLEWARE_STOP_KEYWORDS = new Set([
'async', 'await', 'function', 'new', 'return', 'if', 'for', 'while', 'switch',
@ -10,6 +14,22 @@ export const MIDDLEWARE_STOP_KEYWORDS = new Set([
'event', 'ctx', 'context', 'next',
]);
/** Walk nested wrapper calls starting at `pos` in `content`, returning function names. */
function walkNestedWrappers(content: string, pos: number): string[] {
const names: string[] = [];
const nestedRe = /^\s*(\w+)\s*\(/;
let remaining = content.slice(pos);
let nested;
while ((nested = nestedRe.exec(remaining)) !== null) {
if (MIDDLEWARE_STOP_KEYWORDS.has(nested[1])) break;
names.push(nested[1]);
pos += nested[0].length;
remaining = content.slice(pos);
}
return names;
}
/**
* Extract middleware wrapper chain from a route handler file.
* Detects patterns like: export const POST = withA(withB(withC(handler)))
@ -23,20 +43,123 @@ export function extractMiddlewareChain(content: string): { chain: string[]; meth
const method = mwMatch[1] ?? 'default';
const firstWrapper = mwMatch[2];
const chain: string[] = [firstWrapper];
let pos = mwMatch.index + mwMatch[0].length;
const nestedPattern = /^\s*(\w+)\s*\(/;
let remaining = content.slice(pos);
let nestedMatch;
while ((nestedMatch = nestedPattern.exec(remaining)) !== null) {
const name = nestedMatch[1];
if (MIDDLEWARE_STOP_KEYWORDS.has(name)) break;
chain.push(name);
pos += nestedMatch[0].length;
remaining = content.slice(pos);
}
chain.push(...walkNestedWrappers(content, mwMatch.index + mwMatch[0].length));
if (chain.length >= 2 || (chain.length === 1 && /^with[A-Z]/.test(chain[0]))) {
return { chain, method };
}
}
return undefined;
}
// ---------------------------------------------------------------------------
// Next.js project-level middleware.ts extraction
// ---------------------------------------------------------------------------
export interface NextjsMiddlewareConfig {
matchers: string[];
exportedName: string;
wrappedFunctions: string[];
}
/**
* Parse a Next.js project-level middleware.ts file and extract:
* - config.matcher patterns (string or string[])
* - the exported middleware function name
* - wrapper composition (e.g. chain([withAuth, withI18n]))
*/
export function extractNextjsMiddlewareConfig(content: string): NextjsMiddlewareConfig | undefined {
const matchers: string[] = [];
const matcherArrayRe = /config\s*=\s*\{[^}]*matcher\s*:\s*\[([^\]]*)\]/s;
const matcherStringRe = /config\s*=\s*\{[^}]*matcher\s*:\s*(['"`])([^'"`]+)\1/s;
const arrMatch = matcherArrayRe.exec(content);
if (arrMatch) {
const items = arrMatch[1];
const strRe = /(['"`])((?:[^'"`\\\\]|\\\\.)*)\1/g;
let m;
while ((m = strRe.exec(items)) !== null) {
matchers.push(m[2]);
}
} else {
const strMatch = matcherStringRe.exec(content);
if (strMatch) {
matchers.push(strMatch[2]);
}
}
let exportedName = 'middleware';
const isNamedMw = /export\s+(?:async\s+)?function\s+middleware\b/.test(content);
const isConstMw = /export\s+const\s+middleware\s*=/.test(content);
const defaultFunctionMatch = /export\s+default\s+(?:async\s+)?function(?:\s+(\w+))?/.exec(content);
const defaultIdentifierMatch = /export\s+default\s+(?!function\b)(\w+)/.exec(content);
if (!isNamedMw && !isConstMw) {
if (defaultFunctionMatch) {
exportedName = defaultFunctionMatch[1] ?? 'middleware';
} else if (defaultIdentifierMatch) {
exportedName = defaultIdentifierMatch[1];
}
}
// --- wrapper composition ---
const wrappedFunctions: string[] = [];
// Pattern: chain([fn1, fn2]) or compose(fn1, fn2)
const chainRe = /(?:chain|compose)\s*\(\s*\[([^\]]+)\]/;
const chainMatch = chainRe.exec(content);
if (chainMatch) {
const fns = chainMatch[1].split(',').map(s => s.trim()).filter(Boolean);
wrappedFunctions.push(...fns);
}
// Pattern: export default withA(withB(handler))
const wrapperRe = /export\s+default\s+(\w+)\s*\(/;
const wrapperMatch = wrapperRe.exec(content);
if (wrapperMatch && wrappedFunctions.length === 0) {
const name = wrapperMatch[1];
if (name !== 'function' && name !== 'async') {
wrappedFunctions.push(name);
wrappedFunctions.push(...walkNestedWrappers(content, wrapperMatch.index + wrapperMatch[0].length));
}
}
if (!COMPOSER_NAMES.has(exportedName) && !wrappedFunctions.includes(exportedName)) {
wrappedFunctions.unshift(exportedName);
}
const hasExport = isNamedMw || isConstMw || !!defaultFunctionMatch || !!defaultIdentifierMatch;
if (!hasExport && matchers.length === 0 && wrappedFunctions.length === 0) return undefined;
return { matchers, exportedName, wrappedFunctions };
}
/** Pre-compiled matcher for efficient per-route testing. */
export type CompiledMatcher =
| { type: 'prefix'; prefix: string }
| { type: 'regex'; re: RegExp }
| { type: 'exact'; value: string };
/**
* Compile a Next.js middleware matcher pattern into a reusable matcher.
* Call once per pattern, then use compiledMatcherMatchesRoute per route.
*/
export function compileMatcher(matcher: string): CompiledMatcher | null {
const paramWild = matcher.replace(/\/:path\*$/, '');
if (paramWild !== matcher) return { type: 'prefix', prefix: paramWild };
if (matcher.includes('(')) {
try { return { type: 'regex', re: new RegExp('^' + matcher + '$') }; }
catch { return null; }
}
return { type: 'exact', value: matcher };
}
/** Test a route URL against a pre-compiled matcher. */
export function compiledMatcherMatchesRoute(cm: CompiledMatcher, routeURL: string): boolean {
switch (cm.type) {
case 'prefix': return routeURL === cm.prefix || routeURL.startsWith(cm.prefix + '/');
case 'regex': return cm.re.test(routeURL);
case 'exact': return routeURL === cm.value;
}
}
export function middlewareMatcherMatchesRoute(matcher: string, routeURL: string): boolean {
const cm = compileMatcher(matcher);
return cm ? compiledMatcherMatchesRoute(cm, routeURL) : false;
}

View file

@ -1,57 +1,57 @@
/**
* Response shape extraction from route handler file content.
* Detects .json() calls, extracts top-level keys, and classifies by HTTP status code.
* Detects .json() calls (JS/TS) and json_encode() calls (PHP),
* extracts top-level keys, and classifies by HTTP status code.
*/
/** Return the status code (group 1) from the last match, or undefined. */
function lastMatchGroup(text: string, pattern: RegExp): number | undefined {
const matches = [...text.matchAll(pattern)];
if (matches.length === 0) return undefined;
return parseInt(matches[matches.length - 1][1], 10);
}
/** Build the {responseKeys, errorKeys} result, deduplicating and omitting empty. */
function buildShapeResult(
successKeys: string[],
errKeys: string[],
): { responseKeys?: string[]; errorKeys?: string[] } {
return {
...(successKeys.length > 0 ? { responseKeys: [...new Set(successKeys)] } : {}),
...(errKeys.length > 0 ? { errorKeys: [...new Set(errKeys)] } : {}),
};
}
/**
* Detect an HTTP status code associated with a .json() call.
* Looks for three patterns:
* 1. `.status(N).json(` — Express style (look backwards from .json match)
* 2. `.json({...}, { status: N })` — NextResponse style (look after closing brace of first arg)
* 3. `new Response(JSON.stringify({...}), { status: N })` — raw Response constructor
*
* Returns the numeric status code, or undefined if none found.
*/
export function detectStatusCode(content: string, jsonMatchPos: number, closingBracePos: number): number | undefined {
// Pattern 1: .status(N).json( — look backwards from .json
// Check the ~200 chars before .json for .status(NNN) (generous window for chained calls)
const lookbackStart = Math.max(0, jsonMatchPos - 200);
const before = content.slice(lookbackStart, jsonMatchPos);
const statusChainMatch = before.match(/\.status\s*\(\s*(\d{3})\s*\)\s*$/);
if (statusChainMatch) {
return parseInt(statusChainMatch[1], 10);
}
// Pattern 2: .json({...}, { status: N }) — look after closing brace for second arg
if (closingBracePos > 0) {
// After the first arg's closing brace, look for ", { status: N" within ~100 chars
const afterFirstArg = content.slice(closingBracePos + 1, closingBracePos + 150);
const secondArgMatch = afterFirstArg.match(/^\s*,\s*\{[^}]*status\s*:\s*(\d{3})/);
if (secondArgMatch) {
return parseInt(secondArgMatch[1], 10);
}
}
// Pattern 3: new Response(JSON.stringify({...}), { status: N }) — look before .json for JSON.stringify
// This is a less common pattern; we check if the .json is actually part of JSON.stringify
// by looking for "new Response" further back
const extendedBefore = content.slice(Math.max(0, jsonMatchPos - 300), jsonMatchPos);
if (/new\s+Response\s*\(\s*JSON\s*\.stringify\s*$/.test(extendedBefore) && closingBracePos > 0) {
// Look for ), { status: N }) after the stringify's closing paren
const afterStringify = content.slice(closingBracePos + 1, closingBracePos + 200);
const respStatusMatch = afterStringify.match(/^\s*\)\s*,\s*\{[^}]*status\s*:\s*(\d{3})/);
if (respStatusMatch) {
return parseInt(respStatusMatch[1], 10);
}
}
return undefined;
}
/**
* Extract response shapes from handler file content.
* Finds all .json({...}) calls, extracts top-level keys using brace-depth counting,
* and classifies into success (responseKeys) vs error (errorKeys) by HTTP status code.
* Extract response shapes from JS/TS handler file content.
*/
export function extractResponseShapes(content: string): { responseKeys?: string[]; errorKeys?: string[] } {
const successKeys: string[] = [];
@ -76,7 +76,31 @@ export function extractResponseShapes(content: string): { responseKeys?: string[
if (ch === inString) inString = null;
continue;
}
if (ch === '"' || ch === "'" || ch === '`') { inString = ch; continue; }
if (ch === '"' || ch === "'" || ch === '`') {
// Quoted string at depth 1 before ':' is a property key (e.g., { 'courses': data })
// The original parser only handled unquoted identifiers.
if (depth === 1 && keyStart === -1) {
const quote = ch;
const strStart = j + 1;
let strEnd = -1;
for (let s = strStart; s < content.length; s++) {
if (content[s] === '\\') { s++; continue; }
if (content[s] === quote) { strEnd = s; break; }
}
if (strEnd !== -1) {
// Scan forward for ':' without allocating a substring
let p = strEnd + 1;
while (p < content.length && (content[p] === ' ' || content[p] === '\t' || content[p] === '\n' || content[p] === '\r')) p++;
if (content[p] === ':') {
callKeys.push(content.slice(strStart, strEnd));
}
j = strEnd;
continue;
}
}
inString = ch;
continue;
}
if (ch === '{') { depth++; continue; }
if (ch === '}') { depth--; if (depth === 0) { closingBracePos = j; break; } continue; }
if (depth !== 1) continue;
@ -99,8 +123,119 @@ export function extractResponseShapes(content: string): { responseKeys?: string[
successKeys.push(...callKeys);
}
}
return {
...(successKeys.length > 0 ? { responseKeys: [...new Set(successKeys)] } : {}),
...(errKeys.length > 0 ? { errorKeys: [...new Set(errKeys)] } : {}),
};
return buildShapeResult(successKeys, errKeys);
}
/**
* Find the last exit/die boundary in a string.
* Matches: exit; exit(N); die; die('msg'); die($var);
* Returns the index AFTER the boundary.
*/
function findLastExitBoundary(text: string): number {
const pattern = /\b(exit|die)\s*(\([^)]*\))?\s*;/g;
let lastEnd = -1;
let m;
while ((m = pattern.exec(text)) !== null) {
lastEnd = m.index + m[0].length;
}
return lastEnd;
}
function detectPHPStatusCode(content: string, jsonEncodePos: number): number | undefined {
const lookbackStart = Math.max(0, jsonEncodePos - 300);
let before = content.slice(lookbackStart, jsonEncodePos);
const boundaryEnd = findLastExitBoundary(before);
if (boundaryEnd !== -1) {
before = before.slice(boundaryEnd);
}
return lastMatchGroup(before, /http_response_code\s*\(\s*(\d{3})\s*\)/g)
?? lastMatchGroup(before, /header\s*\(\s*['"]HTTP\/[\d.]+\s+(\d{3})/g)
// CGI/FastCGI format
?? lastMatchGroup(before, /header\s*\(\s*['"]Status:\s*(\d{3})/g);
}
function findMatchingBracket(content: string, openPos: number, open: string, close: string): number {
let depth = 0;
let inString: string | null = null;
for (let j = openPos; j < content.length; j++) {
const ch = content[j];
if (inString) {
if (ch === '\\') { j++; continue; }
if (ch === inString) inString = null;
continue;
}
if (ch === '"' || ch === "'") { inString = ch; continue; }
if (ch === open) { depth++; continue; }
if (ch === close) { depth--; if (depth === 0) return j; continue; }
}
return -1;
}
function extractPHPArrayKeys(arrayContent: string): string[] {
const keys: string[] = [];
let depth = 0;
let inString: string | null = null;
const topLevelRanges: Array<[number, number]> = [];
let rangeStart = 0;
for (let i = 0; i < arrayContent.length; i++) {
const ch = arrayContent[i];
if (inString) {
if (ch === '\\') { i++; continue; }
if (ch === inString) inString = null;
continue;
}
if (ch === '"' || ch === "'") { inString = ch; continue; }
if (ch === '[' || ch === '(' || ch === '{') {
if (depth === 0) topLevelRanges.push([rangeStart, i]);
depth++;
} else if (ch === ']' || ch === ')' || ch === '}') {
depth--;
if (depth === 0) rangeStart = i + 1;
}
}
if (depth === 0) topLevelRanges.push([rangeStart, arrayContent.length]);
for (const [start, end] of topLevelRanges) {
const segment = arrayContent.slice(start, end);
const localPattern = /(['"])([a-zA-Z_][a-zA-Z0-9_]*)\1\s*=>/g;
let m;
while ((m = localPattern.exec(segment)) !== null) {
keys.push(m[2]);
}
}
return keys;
}
export function extractPHPResponseShapes(content: string): { responseKeys?: string[]; errorKeys?: string[] } {
const successKeys: string[] = [];
const errKeys: string[] = [];
const jsonEncodePattern = /json_encode\s*\(/g;
let match;
while ((match = jsonEncodePattern.exec(content)) !== null) {
const matchPos = match.index;
const startIdx = matchPos + match[0].length;
let i = startIdx;
while (i < content.length && /\s/.test(content[i])) i++;
if (i >= content.length) continue;
let arrayEnd = -1;
const openChar = content[i];
if (openChar === '[') {
arrayEnd = findMatchingBracket(content, i, '[', ']');
} else if (content.slice(i, i + 6) === 'array(') {
i += 5;
arrayEnd = findMatchingBracket(content, i, '(', ')');
} else {
continue;
}
if (arrayEnd === -1) continue;
const arrayContent = content.slice(i + 1, arrayEnd);
const callKeys = extractPHPArrayKeys(arrayContent);
if (callKeys.length === 0) continue;
const status = detectPHPStatusCode(content, matchPos);
if (status !== undefined && status >= 400) {
errKeys.push(...callKeys);
} else {
successKeys.push(...callKeys);
}
}
return buildShapeResult(successKeys, errKeys);
}

View file

@ -1,4 +1,3 @@
import { SupportedLanguages } from '../../config/supported-languages.js';
/*
* Tree-sitter queries for extracting code definitions.
@ -1002,19 +1001,4 @@ export const SWIFT_QUERIES = `
`;
export const LANGUAGE_QUERIES: Record<SupportedLanguages, string> = {
[SupportedLanguages.TypeScript]: TYPESCRIPT_QUERIES,
[SupportedLanguages.JavaScript]: JAVASCRIPT_QUERIES,
[SupportedLanguages.Python]: PYTHON_QUERIES,
[SupportedLanguages.Java]: JAVA_QUERIES,
[SupportedLanguages.C]: C_QUERIES,
[SupportedLanguages.Go]: GO_QUERIES,
[SupportedLanguages.CPlusPlus]: CPP_QUERIES,
[SupportedLanguages.CSharp]: CSHARP_QUERIES,
[SupportedLanguages.Ruby]: RUBY_QUERIES,
[SupportedLanguages.Rust]: RUST_QUERIES,
[SupportedLanguages.PHP]: PHP_QUERIES,
[SupportedLanguages.Kotlin]: KOTLIN_QUERIES,
[SupportedLanguages.Swift]: SWIFT_QUERIES,
};

Some files were not shown because too many files have changed in this diff Show more