Merge upstream/main and fix critical module instance mismatch

Address PR #300 review findings:

[CRITICAL] hasOrtCudaProvider() was checking the top-level
onnxruntime-node@1.24.3 but @huggingface/transformers loads its own
nested onnxruntime-node@1.21.0 at runtime. The guard inspected the
wrong binary, so the native crash was not prevented.

Fix: resolve onnxruntime-node from transformers' own module scope
(createRequire from transformers' package.json) so the guard always
checks the same binary that will be dlopen'd at runtime.

Also:
- Add npm overrides to force @huggingface/transformers to use our
  onnxruntime-node@^1.24.0 (works for global installs where gitnexus
  is the root package; npx installs get safety from the resolve fix)
- Replace hardcoded 'x64' with process.arch for arm64 support
- Remove dead napi-v3 path check (ORT 1.21.0 never shipped CUDA .so)

Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
This commit is contained in:
Jim Park 2026-03-21 16:45:10 -07:00
commit 60265c1d0d
704 changed files with 32234 additions and 2831 deletions

View file

@ -1,196 +0,0 @@
name: Integration Tests
on:
workflow_call:
inputs:
collect-coverage:
description: 'Whether to run the coverage collection job (only needed for PR reports)'
required: false
default: true
type: boolean
jobs:
# ── Integration test matrix ─────────────────────────────────────────
# Each test-group runs on a SEPARATE runner per OS, giving full process
# isolation for the LadybugDB native C++ addon.
# 3 OS x 4 groups = 12 parallel jobs.
#
# Groups:
# lbug-db — 7 files using withTestLbugDB / lbug-adapter (native addon)
# Each file runs as its own `vitest run` invocation for full
# process isolation. LadybugDB's native N-API addon registers
# persistent handles that prevent fork workers from exiting
# on Linux, and its C++ destructors segfault during
# process.exit(). Running each file in its own process lets
# the OS reclaim all resources cleanly.
# pipeline — 12 files: ingestion pipeline + csv + 9 resolver tests
# e2e — 2 files: child-process only (spawnSync), no in-process lbug
# standalone — 4 files: pure logic, no lbug, no child processes
test-matrix:
name: integration (${{ matrix.os }} / ${{ matrix.test-group }})
strategy:
fail-fast: false
matrix:
os: [ubuntu-latest, windows-latest, macos-latest]
test-group: [lbug-db, pipeline, e2e, standalone]
include:
- test-group: lbug-db
# Marker — actual files are listed in the run step below
test-glob: ''
- test-group: pipeline
test-glob: >-
test/integration/pipeline.test.ts
test/integration/csv-pipeline.test.ts
test/integration/parsing.test.ts
test/integration/resolvers/typescript.test.ts
test/integration/resolvers/csharp.test.ts
test/integration/resolvers/cpp.test.ts
test/integration/resolvers/java.test.ts
test/integration/resolvers/python.test.ts
test/integration/resolvers/rust.test.ts
test/integration/resolvers/go.test.ts
test/integration/resolvers/kotlin.test.ts
test/integration/resolvers/php.test.ts
test/integration/resolvers/ruby.test.ts
test/integration/resolvers/swift.test.ts
- test-group: e2e
test-glob: >-
test/integration/cli-e2e.test.ts
test/integration/hooks-e2e.test.ts
test/integration/skills-e2e.test.ts
- test-group: standalone
test-glob: >-
test/integration/filesystem-walker.test.ts
test/integration/enrichment.test.ts
test/integration/tree-sitter-languages.test.ts
test/integration/worker-pool.test.ts
runs-on: ${{ matrix.os }}
timeout-minutes: 25
steps:
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
- uses: ./.github/actions/setup-gitnexus
with:
build: 'true'
# lbug-db: run each file in its own vitest process for full isolation.
# LadybugDB's native addon hangs fork workers on Linux — process isolation
# is the only reliable fix boundary.
- name: Run integration tests — lbug-db (process-isolated)
if: matrix.test-group == 'lbug-db'
working-directory: gitnexus
shell: bash
run: |
set -e
files=(
test/integration/lbug-core-adapter.test.ts
test/integration/lbug-pool.test.ts
test/integration/local-backend.test.ts
test/integration/local-backend-calltool.test.ts
test/integration/search-core.test.ts
test/integration/search-pool.test.ts
test/integration/augmentation.test.ts
)
exit_code=0
for f in "${files[@]}"; do
echo "::group::$f"
if ! npx vitest run --reporter=verbose --pool=forks "$f"; then
exit_code=1
echo "::error::Test file failed: $f"
fi
echo "::endgroup::"
done
exit $exit_code
# Non-lbug groups: run all files in a single vitest invocation
- name: Run integration tests — ${{ matrix.test-group }}
if: matrix.test-group != 'lbug-db'
shell: bash
env:
TEST_GLOB: ${{ matrix.test-glob }}
run: npx vitest run --reporter=verbose $TEST_GLOB
working-directory: gitnexus
# ── Coverage collection (ubuntu only) ─────────────────────────────────
# Runs non-lbug integration tests with coverage enabled so the PR report
# can merge integration + unit coverage for a combined view.
# lbug-db tests are excluded because each file must run in its own vitest
# process (native addon isolation) which prevents single-run coverage merge.
coverage:
name: integration (ubuntu / coverage)
if: inputs.collect-coverage
runs-on: ubuntu-latest
timeout-minutes: 15
steps:
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
- uses: ./.github/actions/setup-gitnexus
with:
build: 'true'
- name: Run integration tests with coverage
working-directory: gitnexus
run: >-
npx vitest run
--reporter=default
--reporter=json
--outputFile=integration-results.json
--coverage
--coverage.reporter=json-summary
--coverage.reporter=json
--coverage.reporter=text
--coverage.thresholdAutoUpdate=false
--coverage.reportOnFailure=true
--coverage.thresholds.statements=0
--coverage.thresholds.branches=0
--coverage.thresholds.functions=0
--coverage.thresholds.lines=0
test/integration/pipeline.test.ts
test/integration/csv-pipeline.test.ts
test/integration/parsing.test.ts
test/integration/cli-e2e.test.ts
test/integration/hooks-e2e.test.ts
test/integration/filesystem-walker.test.ts
test/integration/enrichment.test.ts
test/integration/tree-sitter-languages.test.ts
test/integration/worker-pool.test.ts
test/integration/resolvers/typescript.test.ts
test/integration/resolvers/csharp.test.ts
test/integration/resolvers/cpp.test.ts
test/integration/resolvers/java.test.ts
test/integration/resolvers/python.test.ts
test/integration/resolvers/rust.test.ts
test/integration/resolvers/go.test.ts
test/integration/resolvers/kotlin.test.ts
test/integration/resolvers/php.test.ts
test/integration/resolvers/ruby.test.ts
test/integration/resolvers/swift.test.ts
- name: Upload integration coverage
if: always()
uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4
with:
name: integration-reports
path: |
gitnexus/coverage/coverage-summary.json
gitnexus/coverage/coverage-final.json
gitnexus/integration-results.json
retention-days: 5
# ── Unified status gate ──────────────────────────────────────────────
# Branch protection should require THIS job, not the matrix jobs directly.
# ci.yml's needs.integration.result aggregates through this gate.
status:
name: integration (all groups)
needs: test-matrix
if: always()
runs-on: ubuntu-latest
timeout-minutes: 5
steps:
- name: Check all matrix jobs passed
shell: bash
env:
RESULT: ${{ needs.test-matrix.result }}
run: |
if [[ "$RESULT" != "success" ]]; then
echo "::error::Integration matrix failed or cancelled: $RESULT"
exit 1
fi

View file

@ -1,127 +1,132 @@
name: CI Report
# Triggered after the CI workflow completes. Because workflow_run
# always runs code from the *default branch*, it receives a read/write
# GITHUB_TOKEN — even when the triggering PR comes from a fork.
on:
workflow_run:
workflows: ["CI"]
workflows: ['CI']
types: [completed]
permissions:
actions: read # needed to list/download workflow run artifacts
contents: read # needed for sparse checkout of vitest.config.ts
pull-requests: write # needed to post sticky PR comment
actions: read
contents: read
pull-requests: write
jobs:
pr-report:
name: PR Report
# Only run for pull-request CI runs
if: >-
github.event.workflow_run.event == 'pull_request' &&
github.event.workflow_run.conclusion != 'cancelled'
runs-on: ubuntu-latest
timeout-minutes: 5
steps:
# ── Download artifacts from the CI run ────────────────────────
- name: Download artifacts
- name: Download PR metadata
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7
with:
script: |
const fs = require('fs');
const path = require('path');
const runId = context.payload.workflow_run.id;
const allArtifacts = await github.rest.actions.listWorkflowRunArtifacts({
const artifacts = await github.rest.actions.listWorkflowRunArtifacts({
owner: context.repo.owner,
repo: context.repo.repo,
run_id: runId,
run_id: ${{ github.event.workflow_run.id }},
});
async function downloadArtifact(name, dest) {
const match = allArtifacts.data.artifacts.find(a => a.name === name);
if (!match) {
core.warning(`Artifact "${name}" not found`);
return false;
}
const zip = await github.rest.actions.downloadArtifact({
owner: context.repo.owner,
repo: context.repo.repo,
artifact_id: match.id,
archive_format: 'zip',
});
fs.mkdirSync(dest, { recursive: true });
fs.writeFileSync(path.join(dest, `${name}.zip`), Buffer.from(zip.data));
return true;
const meta = artifacts.data.artifacts.find(a => a.name === 'pr-meta');
if (!meta) {
core.setFailed('pr-meta artifact not found — skipping report');
return;
}
const temp = process.env.RUNNER_TEMP;
await downloadArtifact('pr-meta', path.join(temp, 'dl'));
await downloadArtifact('test-reports', path.join(temp, 'dl'));
await downloadArtifact('integration-reports', path.join(temp, 'dl'));
const zip = await github.rest.actions.downloadArtifact({
owner: context.repo.owner,
repo: context.repo.repo,
artifact_id: meta.id,
archive_format: 'zip',
});
- name: Extract artifacts
shell: bash
run: |
cd "$RUNNER_TEMP/dl"
# Extract each artifact into its own directory to avoid filename collisions
for z in *.zip; do
[ -f "$z" ] || continue
name="${z%.zip}"
mkdir -p "$RUNNER_TEMP/artifacts/$name"
unzip -o "$z" -d "$RUNNER_TEMP/artifacts/$name"
done
const dest = path.join(process.env.RUNNER_TEMP, 'pr-meta');
fs.mkdirSync(dest, { recursive: true });
fs.writeFileSync(path.join(dest, 'pr-meta.zip'), Buffer.from(zip.data));
- name: Read PR metadata
- name: Extract PR metadata
id: meta
shell: bash
run: |
DIR="$RUNNER_TEMP/artifacts/pr-meta"
if [ ! -f "$DIR/pr_number" ]; then
echo "skip=true" >> "$GITHUB_OUTPUT"
echo "::warning::pr_number artifact missing — skipping report"
exit 0
cd "$RUNNER_TEMP/pr-meta"
unzip -o pr-meta.zip
PR_NUMBER=$(cat pr-number | tr -d '[:space:]')
if ! [[ "$PR_NUMBER" =~ ^[0-9]+$ ]]; then
echo "::error::Invalid PR number: '$PR_NUMBER'"
exit 1
fi
# Validate PR number is a positive integer (artifact comes from
# untrusted fork code, so treat contents defensively).
PR_NUM=$(cat "$DIR/pr_number" | tr -d '[:space:]')
if ! [[ "$PR_NUM" =~ ^[0-9]+$ ]]; then
echo "skip=true" >> "$GITHUB_OUTPUT"
echo "::error::Invalid PR number in artifact: '$PR_NUM'"
exit 0
fi
echo "pr-number=$PR_NUMBER" >> "$GITHUB_OUTPUT"
echo "quality=$(cat quality-result | tr -d '[:space:]')" >> "$GITHUB_OUTPUT"
echo "tests=$(cat tests-result | tr -d '[:space:]')" >> "$GITHUB_OUTPUT"
echo "skip=false" >> "$GITHUB_OUTPUT"
echo "pr_number=$PR_NUM" >> "$GITHUB_OUTPUT"
# Validate job-result strings against known GitHub Actions values.
# Artifact contents come from the PR workflow (potentially untrusted
# fork code), so we whitelist to prevent newline injection into
# GITHUB_OUTPUT.
validate_result() {
local val
val=$(cat "$1" | tr -d '[:space:]')
case "$val" in
success|failure|cancelled|skipped) echo "$val" ;;
*) echo "unknown" ;;
esac
}
echo "quality=$(validate_result "$DIR/quality_result")" >> "$GITHUB_OUTPUT"
echo "unit=$(validate_result "$DIR/unit_result")" >> "$GITHUB_OUTPUT"
echo "integration=$(validate_result "$DIR/integration_result")" >> "$GITHUB_OUTPUT"
- name: Checkout (for vitest config)
if: steps.meta.outputs.skip != 'true'
uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
- name: Download test reports
id: download-test-reports
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7
with:
sparse-checkout: gitnexus/vitest.config.ts
sparse-checkout-cone-mode: false
script: |
const fs = require('fs');
const path = require('path');
const artifacts = await github.rest.actions.listWorkflowRunArtifacts({
owner: context.repo.owner,
repo: context.repo.repo,
run_id: ${{ github.event.workflow_run.id }},
});
const reports = artifacts.data.artifacts.find(a => a.name === 'test-reports');
if (!reports) {
core.warning('test-reports artifact not found');
return;
}
const zip = await github.rest.actions.downloadArtifact({
owner: context.repo.owner,
repo: context.repo.repo,
artifact_id: reports.id,
archive_format: 'zip',
});
const dest = path.join(process.env.RUNNER_TEMP, 'test-reports');
fs.mkdirSync(dest, { recursive: true });
fs.writeFileSync(path.join(dest, 'test-reports.zip'), Buffer.from(zip.data));
- name: Extract test reports
if: steps.download-test-reports.outcome == 'success'
shell: bash
run: |
cd "$RUNNER_TEMP/test-reports"
unzip -o test-reports.zip || true
- name: Fetch cross-platform job results
id: jobs
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7
with:
script: |
const jobs = await github.rest.actions.listJobsForWorkflowRun({
owner: context.repo.owner,
repo: context.repo.repo,
run_id: ${{ github.event.workflow_run.id }},
per_page: 50,
});
const results = {};
for (const job of jobs.data.jobs) {
if (job.name.includes('ubuntu')) results.ubuntu = job.conclusion || 'pending';
else if (job.name.includes('windows')) results.windows = job.conclusion || 'pending';
else if (job.name.includes('macos')) results.macos = job.conclusion || 'pending';
}
core.setOutput('ubuntu', results.ubuntu || 'unknown');
core.setOutput('windows', results.windows || 'unknown');
core.setOutput('macos', results.macos || 'unknown');
# ── Fetch base branch coverage for delta reporting ───────────
- name: Fetch base branch coverage
if: steps.meta.outputs.skip != 'true'
id: base-coverage
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7
with:
@ -129,7 +134,6 @@ jobs:
const fs = require('fs');
const path = require('path');
// Find the latest successful CI run on main
const runs = await github.rest.actions.listWorkflowRuns({
owner: context.repo.owner,
repo: context.repo.repo,
@ -141,7 +145,6 @@ jobs:
if (runs.data.workflow_runs.length === 0) {
core.setOutput('found', 'false');
core.info('No successful main branch CI runs found');
return;
}
@ -155,7 +158,6 @@ jobs:
const testReports = artifacts.data.artifacts.find(a => a.name === 'test-reports');
if (!testReports) {
core.setOutput('found', 'false');
core.info('No test-reports artifact on main branch');
return;
}
@ -173,350 +175,174 @@ jobs:
core.setOutput('dir', dest);
- name: Extract base coverage
if: steps.meta.outputs.skip != 'true' && steps.base-coverage.outputs.found == 'true'
if: steps.base-coverage.outputs.found == 'true'
shell: bash
run: |
cd "${{ steps.base-coverage.outputs.dir }}"
unzip -o base.zip -d base
# ── Merge coverage from unit + integration ─────────────────────
- name: Setup Node.js
if: steps.meta.outputs.skip != 'true'
uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4
with:
node-version: 20
- name: Install coverage merge tools
if: steps.meta.outputs.skip != 'true'
run: npm install --no-save istanbul-lib-coverage istanbul-lib-report istanbul-reports
- name: Merge coverage reports
if: steps.meta.outputs.skip != 'true'
id: coverage
shell: bash
run: |
DIR="$RUNNER_TEMP/artifacts"
UNIT_COV=$(find "$DIR/test-reports" -name "coverage-final.json" -type f 2>/dev/null | head -1)
INTEG_COV=$(find "$DIR/integration-reports" -name "coverage-final.json" -type f 2>/dev/null | head -1)
MERGED_DIR="$RUNNER_TEMP/merged-coverage"
mkdir -p "$MERGED_DIR"
if [ -n "$UNIT_COV" ] && [ -n "$INTEG_COV" ]; then
echo "has_merged=true" >> "$GITHUB_OUTPUT"
# Merge using Node.js + istanbul-lib-coverage.
# Paths are passed via env vars to avoid shell interpolation
# inside the script string.
UNIT_COV_PATH="$UNIT_COV" \
INTEG_COV_PATH="$INTEG_COV" \
MERGED_OUT_DIR="$MERGED_DIR" \
node -e "
const libCoverage = require('istanbul-lib-coverage');
const libReport = require('istanbul-lib-report');
const reports = require('istanbul-reports');
const fs = require('fs');
const map = libCoverage.createCoverageMap({});
map.merge(JSON.parse(fs.readFileSync(process.env.UNIT_COV_PATH, 'utf8')));
map.merge(JSON.parse(fs.readFileSync(process.env.INTEG_COV_PATH, 'utf8')));
const context = libReport.createContext({
coverageMap: map,
dir: process.env.MERGED_OUT_DIR,
});
reports.create('json-summary').execute(context);
console.log('Merged coverage written to ' + process.env.MERGED_OUT_DIR + '/coverage-summary.json');
"
elif [ -n "$UNIT_COV" ]; then
echo "has_merged=false" >> "$GITHUB_OUTPUT"
echo "::warning::Integration coverage not found — using unit coverage only"
else
echo "has_merged=false" >> "$GITHUB_OUTPUT"
echo "::warning::No coverage data found"
fi
- name: Build report
if: steps.meta.outputs.skip != 'true'
id: report
shell: bash
- name: Build and post report
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7
env:
PR_NUMBER: ${{ steps.meta.outputs.pr-number }}
QUALITY: ${{ steps.meta.outputs.quality }}
UNIT: ${{ steps.meta.outputs.unit }}
INTEG: ${{ steps.meta.outputs.integration }}
HAS_MERGED: ${{ steps.coverage.outputs.has_merged }}
TESTS: ${{ steps.meta.outputs.tests }}
UBUNTU: ${{ steps.jobs.outputs.ubuntu }}
WINDOWS: ${{ steps.jobs.outputs.windows }}
MACOS: ${{ steps.jobs.outputs.macos }}
BASE_FOUND: ${{ steps.base-coverage.outputs.found }}
BASE_DIR: ${{ steps.base-coverage.outputs.dir }}
RUN_URL: ${{ github.event.workflow_run.html_url }}
run: |
DIR="$RUNNER_TEMP/artifacts"
MERGED_DIR="$RUNNER_TEMP/merged-coverage"
RUN_ID: ${{ github.event.workflow_run.id }}
HEAD_SHA: ${{ github.event.workflow_run.head_sha }}
with:
script: |
const fs = require('fs');
const path = require('path');
# ── Helper: read coverage summary into prefixed vars ──
read_cov() {
local prefix=$1 file=$2
if [ -n "$file" ] && [ -f "$file" ]; then
local val
val=$(jq -r '.total.statements.pct // "N/A"' "$file" 2>/dev/null) || val="N/A"
printf -v "${prefix}_STMTS" '%s' "$val"
val=$(jq -r '.total.branches.pct // "N/A"' "$file" 2>/dev/null) || val="N/A"
printf -v "${prefix}_BRANCH" '%s' "$val"
val=$(jq -r '.total.functions.pct // "N/A"' "$file" 2>/dev/null) || val="N/A"
printf -v "${prefix}_FUNCS" '%s' "$val"
val=$(jq -r '.total.lines.pct // "N/A"' "$file" 2>/dev/null) || val="N/A"
printf -v "${prefix}_LINES" '%s' "$val"
val=$(jq -r '"\(.total.statements.covered)/\(.total.statements.total)"' "$file" 2>/dev/null) || val=""
printf -v "${prefix}_STMTS_COV" '%s' "$val"
val=$(jq -r '"\(.total.branches.covered)/\(.total.branches.total)"' "$file" 2>/dev/null) || val=""
printf -v "${prefix}_BRANCH_COV" '%s' "$val"
val=$(jq -r '"\(.total.functions.covered)/\(.total.functions.total)"' "$file" 2>/dev/null) || val=""
printf -v "${prefix}_FUNCS_COV" '%s' "$val"
val=$(jq -r '"\(.total.lines.covered)/\(.total.lines.total)"' "$file" 2>/dev/null) || val=""
printf -v "${prefix}_LINES_COV" '%s' "$val"
return 0
else
printf -v "${prefix}_STMTS" '%s' "N/A"
printf -v "${prefix}_BRANCH" '%s' "N/A"
printf -v "${prefix}_FUNCS" '%s' "N/A"
printf -v "${prefix}_LINES" '%s' "N/A"
printf -v "${prefix}_STMTS_COV" '%s' ""
printf -v "${prefix}_BRANCH_COV" '%s' ""
printf -v "${prefix}_FUNCS_COV" '%s' ""
printf -v "${prefix}_LINES_COV" '%s' ""
return 1
fi
}
const icon = (s) => ({ success: '✅', failure: '❌', cancelled: '⏭️' }[s] || '❓');
const temp = process.env.RUNNER_TEMP;
# ── Read all coverage reports ──
UNIT_SUMMARY=$(find "$DIR/test-reports" -name "coverage-summary.json" -type f 2>/dev/null | head -1)
INTEG_SUMMARY=$(find "$DIR/integration-reports" -name "coverage-summary.json" -type f 2>/dev/null | head -1)
MERGED_SUMMARY="$MERGED_DIR/coverage-summary.json"
read_cov "U" "$UNIT_SUMMARY"
HAS_UNIT=$?
read_cov "I" "$INTEG_SUMMARY"
HAS_INTEG=$?
read_cov "M" "$MERGED_SUMMARY"
# ── Read base branch coverage (main) ──
BASE_SUMMARY=""
if [ "$BASE_FOUND" = "true" ] && [ -n "$BASE_DIR" ]; then
BASE_SUMMARY=$(find "$BASE_DIR/base" -name "coverage-summary.json" -type f 2>/dev/null | head -1)
fi
read_cov "B" "$BASE_SUMMARY"
# ── Locate test results ──
RESULTS_FILE=$(find "$DIR/test-reports" -name "test-results.json" -type f 2>/dev/null | head -1)
INTEG_RESULTS=$(find "$DIR/integration-reports" -name "integration-results.json" -type f 2>/dev/null | head -1)
if [ -n "$RESULTS_FILE" ]; then
U_TOTAL=$(jq -r '.numTotalTests' "$RESULTS_FILE" 2>/dev/null || echo 0)
U_PASSED=$(jq -r '.numPassedTests' "$RESULTS_FILE" 2>/dev/null || echo 0)
U_FAILED=$(jq -r '.numFailedTests' "$RESULTS_FILE" 2>/dev/null || echo 0)
U_SKIPPED=$(jq -r '.numPendingTests' "$RESULTS_FILE" 2>/dev/null || echo 0)
U_SUITES=$(jq -r '.numTotalTestSuites' "$RESULTS_FILE" 2>/dev/null || echo 0)
U_DURATION=$(jq -r '((.testResults | map(.endTime) | max) - (.startTime)) / 1000 | floor' "$RESULTS_FILE" 2>/dev/null || echo 0)
else
U_TOTAL=0; U_PASSED=0; U_FAILED=0; U_SKIPPED=0; U_SUITES=0; U_DURATION=0
fi
if [ -n "$INTEG_RESULTS" ]; then
I_TOTAL=$(jq -r '.numTotalTests' "$INTEG_RESULTS" 2>/dev/null || echo 0)
I_PASSED=$(jq -r '.numPassedTests' "$INTEG_RESULTS" 2>/dev/null || echo 0)
I_FAILED=$(jq -r '.numFailedTests' "$INTEG_RESULTS" 2>/dev/null || echo 0)
I_SKIPPED=$(jq -r '.numPendingTests' "$INTEG_RESULTS" 2>/dev/null || echo 0)
I_SUITES=$(jq -r '.numTotalTestSuites' "$INTEG_RESULTS" 2>/dev/null || echo 0)
I_DURATION=$(jq -r '((.testResults | map(.endTime) | max) - (.startTime)) / 1000 | floor' "$INTEG_RESULTS" 2>/dev/null || echo 0)
else
I_TOTAL=0; I_PASSED=0; I_FAILED=0; I_SKIPPED=0; I_SUITES=0; I_DURATION=0
fi
# ── Sum test results ──
TOTAL=$((U_TOTAL + I_TOTAL))
PASSED=$((U_PASSED + I_PASSED))
FAILED=$((U_FAILED + I_FAILED))
SKIPPED=$((U_SKIPPED + I_SKIPPED))
SUITES=$((U_SUITES + I_SUITES))
# ── Status helpers ──
status_icon() {
case "$1" in
success) echo "✅" ;;
failure) echo "❌" ;;
cancelled) echo "⏭️" ;;
*) echo "❓" ;;
esac
}
cov_delta() {
local pct=$1 base=$2
if [ "$pct" = "N/A" ] || [ "$base" = "N/A" ]; then echo "—"; return; fi
local diff
diff=$(awk "BEGIN { printf \"%.1f\", $pct - $base }")
if [ "$(awk "BEGIN { print ($pct > $base) ? 1 : 0 }")" = "1" ]; then
echo "📈 +${diff}"
elif [ "$(awk "BEGIN { print ($pct < $base) ? 1 : 0 }")" = "1" ]; then
echo "📉 ${diff}"
else
echo "= ${diff}"
fi
}
cov_bar() {
local pct=$1 base=$2
if [ "$pct" = "N/A" ]; then echo "—"; return; fi
local filled
filled=$(awk "BEGIN { printf \"%d\", $pct / 5 }")
(( filled < 0 )) && filled=0
(( filled > 20 )) && filled=20
local empty=$((20 - filled))
local bar=""
for ((i=0; i<filled; i++)); do bar+="█"; done
for ((i=0; i<empty; i++)); do bar+="░"; done
# Green if >= base (or base unavailable), red if dropped
if [ "$base" = "N/A" ] || [ "$(awk "BEGIN { print ($pct >= $base) ? 1 : 0 }")" = "1" ]; then
echo "🟢 ${bar}"
else
echo "🔴 ${bar}"
fi
}
# ── Overall status ──
if [[ "$QUALITY" == "success" && "$UNIT" == "success" && "$INTEG" == "success" ]]; then
OVERALL="✅ **All checks passed**"
else
OVERALL="❌ **Some checks failed**"
fi
# ── Build markdown ──
{
echo "body<<GITNEXUS_CI_REPORT_EOF_7f3a"
echo "## CI Report"
echo ""
echo "${OVERALL}"
echo ""
echo "### Pipeline Status"
echo ""
echo "| Stage | Status | Details |"
echo "|-------|--------|---------|"
echo "| $(status_icon "$QUALITY") Typecheck | \`${QUALITY}\` | tsc --noEmit |"
echo "| $(status_icon "$UNIT") Unit Tests | \`${UNIT}\` | 3 platforms |"
echo "| $(status_icon "$INTEG") Integration | \`${INTEG}\` | 3 OS x 4 groups = 12 jobs |"
echo ""
if [ "$TOTAL" -gt 0 ] 2>/dev/null; then
echo "### Test Results"
echo ""
echo "| Suite | Tests | Passed | Failed | Skipped | Duration |"
echo "|-------|-------|--------|--------|---------|----------|"
if [ "$U_TOTAL" -gt 0 ] 2>/dev/null; then
echo "| Unit | ${U_TOTAL} | ${U_PASSED} | ${U_FAILED} | ${U_SKIPPED} | ${U_DURATION}s |"
fi
if [ "$I_TOTAL" -gt 0 ] 2>/dev/null; then
echo "| Integration | ${I_TOTAL} | ${I_PASSED} | ${I_FAILED} | ${I_SKIPPED} | ${I_DURATION}s |"
fi
echo "| **Total** | **${TOTAL}** | **${PASSED}** | **${FAILED}** | **${SKIPPED}** | **$((U_DURATION + I_DURATION))s** |"
echo ""
if [ "$FAILED" = "0" ]; then
echo "✅ All **${PASSED}** tests passed"
else
echo "❌ **${FAILED}** failed / **${PASSED}** passed"
fi
if [ "$SKIPPED" != "0" ]; then
echo ""
echo "<details>"
echo "<summary>${SKIPPED} test(s) skipped — expand for details</summary>"
echo ""
# Extract skipped test names from integration results
if [ -n "$INTEG_RESULTS" ] && [ "$I_SKIPPED" -gt 0 ] 2>/dev/null; then
echo "**Integration:**"
jq -r '
.testResults[]
| .assertionResults[]?
| select(.status == "pending" or .status == "skipped")
| "- \(.ancestorTitles | join(" > ")) > \(.title)"
' "$INTEG_RESULTS" 2>/dev/null || echo "- _(unable to parse skipped test details)_"
fi
# Extract skipped test names from unit results
if [ -n "$RESULTS_FILE" ] && [ "$U_SKIPPED" -gt 0 ] 2>/dev/null; then
echo ""
echo "**Unit:**"
jq -r '
.testResults[]
| .assertionResults[]?
| select(.status == "pending" or .status == "skipped")
| "- \(.ancestorTitles | join(" > ")) > \(.title)"
' "$RESULTS_FILE" 2>/dev/null || echo "- _(unable to parse skipped test details)_"
fi
echo ""
echo "</details>"
fi
echo ""
fi
# ── Coverage table helper ──
cov_table() {
local label=$1 s=$2 b=$3 f=$4 l=$5 sc=$6 bc=$7 fc=$8 lc=$9
shift 9
local bs=$1 bb=$2 bf=$3 bl=$4
echo "#### ${label}"
echo ""
echo "| Metric | Coverage | Covered | Base | Delta | Status |"
echo "|--------|----------|---------|------|-------|--------|"
echo "| Statements | **${s}%** | ${sc} | ${bs}% | $(cov_delta "$s" "$bs") | $(cov_bar "$s" "$bs") |"
echo "| Branches | **${b}%** | ${bc} | ${bb}% | $(cov_delta "$b" "$bb") | $(cov_bar "$b" "$bb") |"
echo "| Functions | **${f}%** | ${fc} | ${bf}% | $(cov_delta "$f" "$bf") | $(cov_bar "$f" "$bf") |"
echo "| Lines | **${l}%** | ${lc} | ${bl}% | $(cov_delta "$l" "$bl") | $(cov_bar "$l" "$bl") |"
echo ""
// ── Read coverage ──
function readCov(dir) {
const out = { stmts: 'N/A', branch: 'N/A', funcs: 'N/A', lines: 'N/A',
stmtsCov: '', branchCov: '', funcsCov: '', linesCov: '' };
try {
const files = require('child_process')
.execSync(`find "${dir}" -name coverage-summary.json -type f`, { encoding: 'utf8' })
.trim().split('\n').filter(Boolean);
if (!files.length) return out;
const d = JSON.parse(fs.readFileSync(files[0], 'utf8')).total;
out.stmts = d.statements.pct; out.branch = d.branches.pct;
out.funcs = d.functions.pct; out.lines = d.lines.pct;
out.stmtsCov = `${d.statements.covered}/${d.statements.total}`;
out.branchCov = `${d.branches.covered}/${d.branches.total}`;
out.funcsCov = `${d.functions.covered}/${d.functions.total}`;
out.linesCov = `${d.lines.covered}/${d.lines.total}`;
} catch {}
return out;
}
if [ "$M_STMTS" != "N/A" ]; then
echo "### Code Coverage"
echo ""
cov_table "Combined (Unit + Integration)" \
"$M_STMTS" "$M_BRANCH" "$M_FUNCS" "$M_LINES" \
"$M_STMTS_COV" "$M_BRANCH_COV" "$M_FUNCS_COV" "$M_LINES_COV" \
"$B_STMTS" "$B_BRANCH" "$B_FUNCS" "$B_LINES"
const cov = readCov(path.join(temp, 'test-reports'));
const base = process.env.BASE_FOUND === 'true'
? readCov(path.join(process.env.BASE_DIR, 'base'))
: { stmts: 'N/A', branch: 'N/A', funcs: 'N/A', lines: 'N/A' };
echo "<details>"
echo "<summary>Coverage breakdown by test suite</summary>"
echo ""
if [ "$U_STMTS" != "N/A" ]; then
cov_table "Unit Tests" \
"$U_STMTS" "$U_BRANCH" "$U_FUNCS" "$U_LINES" \
"$U_STMTS_COV" "$U_BRANCH_COV" "$U_FUNCS_COV" "$U_LINES_COV" \
"$B_STMTS" "$B_BRANCH" "$B_FUNCS" "$B_LINES"
fi
if [ "$I_STMTS" != "N/A" ]; then
cov_table "Integration Tests" \
"$I_STMTS" "$I_BRANCH" "$I_FUNCS" "$I_LINES" \
"$I_STMTS_COV" "$I_BRANCH_COV" "$I_FUNCS_COV" "$I_LINES_COV" \
"$B_STMTS" "$B_BRANCH" "$B_FUNCS" "$B_LINES"
fi
echo "</details>"
echo ""
elif [ "$U_STMTS" != "N/A" ]; then
echo "### Code Coverage (Unit only)"
echo ""
cov_table "Unit Tests" \
"$U_STMTS" "$U_BRANCH" "$U_FUNCS" "$U_LINES" \
"$U_STMTS_COV" "$U_BRANCH_COV" "$U_FUNCS_COV" "$U_LINES_COV" \
"$B_STMTS" "$B_BRANCH" "$B_FUNCS" "$B_LINES"
else
echo "### Code Coverage"
echo ""
echo "⚠️ Coverage data unavailable - check the [unit test job](${RUN_URL}) for details."
echo ""
fi
// ── Read test results ──
let total = 0, passed = 0, failed = 0, skipped = 0, suites = 0, duration = '0s';
let skippedTests = [];
try {
const files = require('child_process')
.execSync(`find "${path.join(temp, 'test-reports')}" -name test-results.json -type f`, { encoding: 'utf8' })
.trim().split('\n').filter(Boolean);
if (files.length) {
const r = JSON.parse(fs.readFileSync(files[0], 'utf8'));
total = r.numTotalTests || 0;
passed = r.numPassedTests || 0;
failed = r.numFailedTests || 0;
skipped = r.numPendingTests || 0;
suites = r.numTotalTestSuites || 0;
const durS = Math.floor((Math.max(...r.testResults.map(t => t.endTime)) - r.startTime) / 1000);
duration = durS >= 60 ? `${Math.floor(durS / 60)}m ${durS % 60}s` : `${durS}s`;
// Collect skipped test names
for (const suite of r.testResults) {
for (const t of (suite.assertionResults || [])) {
if (t.status === 'pending' || t.status === 'skipped') {
skippedTests.push(`- ${t.ancestorTitles.join(' > ')} > ${t.title}`);
}
}
}
}
} catch {}
echo "---"
echo "<sub>📋 [View full run](${RUN_URL}) · Generated by CI</sub>"
echo "GITNEXUS_CI_REPORT_EOF_7f3a"
} >> "$GITHUB_OUTPUT"
// ── Coverage delta ──
function delta(pct, basePct) {
if (pct === 'N/A' || basePct === 'N/A') return '—';
const d = (pct - basePct).toFixed(1);
const dNum = parseFloat(d);
if (dNum > 0) return `📈 +${d}%`;
if (dNum < 0) return `📉 ${d}%`;
return '=';
}
- name: Comment on PR
if: steps.meta.outputs.skip != 'true'
uses: marocchino/sticky-pull-request-comment@773744901bac0e8cbb5a0dc842800d45e9b2b405 # v2
with:
header: ci-report
number: ${{ steps.meta.outputs.pr_number }}
message: ${{ steps.report.outputs.body }}
// ── Build markdown ──
const { PR_NUMBER, QUALITY, TESTS, UBUNTU, WINDOWS, MACOS, RUN_ID, HEAD_SHA } = process.env;
const prNumber = parseInt(PR_NUMBER, 10);
const overall = (QUALITY === 'success' && TESTS === 'success')
? '✅ **All checks passed**' : '❌ **Some checks failed**';
const sha = HEAD_SHA.slice(0, 7);
let body = `## CI Report\n\n${overall} &ensp; \`${sha}\`\n\n`;
body += `### Pipeline\n\n`;
body += `| Stage | Status | Ubuntu | Windows | macOS |\n`;
body += `|-------|--------|--------|---------|-------|\n`;
body += `| Typecheck | ${icon(QUALITY)} \`${QUALITY}\` | — | — | — |\n`;
body += `| Tests | ${icon(TESTS)} \`${TESTS}\` | ${icon(UBUNTU)} | ${icon(WINDOWS)} | ${icon(MACOS)} |\n\n`;
if (total > 0) {
body += `### Tests\n\n`;
body += `| Metric | Value |\n|--------|-------|\n`;
body += `| Total | **${total}** |\n`;
body += `| Passed | **${passed}** |\n`;
if (failed > 0) body += `| Failed | **${failed}** |\n`;
if (skipped > 0) body += `| Skipped | ${skipped} |\n`;
body += `| Files | ${suites} |\n`;
body += `| Duration | ${duration} |\n\n`;
if (failed === 0) {
body += `✅ All **${passed}** tests passed across **${suites}** files\n`;
} else {
body += `❌ **${failed}** failed / **${passed}** passed\n`;
}
if (skippedTests.length > 0) {
body += `\n<details>\n<summary>${skipped} test(s) skipped</summary>\n\n`;
body += skippedTests.join('\n') + '\n\n</details>\n';
}
body += '\n';
}
if (cov.stmts !== 'N/A') {
body += `### Coverage\n\n`;
body += `| Metric | Coverage | Covered | Base (main) | Delta |\n`;
body += `|--------|----------|---------|-------------|-------|\n`;
body += `| Statements | **${cov.stmts}%** | ${cov.stmtsCov} | ${base.stmts}% | ${delta(cov.stmts, base.stmts)} |\n`;
body += `| Branches | **${cov.branch}%** | ${cov.branchCov} | ${base.branch}% | ${delta(cov.branch, base.branch)} |\n`;
body += `| Functions | **${cov.funcs}%** | ${cov.funcsCov} | ${base.funcs}% | ${delta(cov.funcs, base.funcs)} |\n`;
body += `| Lines | **${cov.lines}%** | ${cov.linesCov} | ${base.lines}% | ${delta(cov.lines, base.lines)} |\n\n`;
} else {
const runUrl = `${context.serverUrl}/${context.repo.owner}/${context.repo.repo}/actions/runs/${RUN_ID}`;
body += `### Coverage\n\n⚠️ Coverage data unavailable — check the [test job](${runUrl}) for details.\n\n`;
}
const runUrl = `${context.serverUrl}/${context.repo.owner}/${context.repo.repo}/actions/runs/${RUN_ID}`;
body += `---\n<sub>📋 [Full run](${runUrl}) · Coverage from Ubuntu · Generated by CI</sub>`;
// ── Post sticky comment ──
const { data: comments } = await github.rest.issues.listComments({
owner: context.repo.owner,
repo: context.repo.repo,
issue_number: prNumber,
per_page: 100,
direction: 'desc',
});
const marker = '<!-- ci-report -->';
const existing = comments.find(c => c.body?.includes(marker));
const fullBody = marker + '\n' + body;
if (existing) {
await github.rest.issues.updateComment({
owner: context.repo.owner,
repo: context.repo.repo,
comment_id: existing.id,
body: fullBody,
});
} else {
await github.rest.issues.createComment({
owner: context.repo.owner,
repo: context.repo.repo,
issue_number: prNumber,
body: fullBody,
});
}

View file

@ -1,20 +1,22 @@
name: Unit Tests
name: Tests
on:
workflow_call:
jobs:
unit-tests:
name: unit (ubuntu / coverage)
tests:
name: ubuntu / coverage
runs-on: ubuntu-latest
timeout-minutes: 15
timeout-minutes: 25
steps:
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
- uses: ./.github/actions/setup-gitnexus
with:
build: 'true'
- name: Run unit tests with coverage
- name: Run all tests with coverage
run: >-
npx vitest run test/unit
npx vitest run
--reporter=default
--reporter=json
--outputFile=test-results.json
@ -38,16 +40,18 @@ jobs:
retention-days: 5
cross-platform:
name: unit (${{ matrix.os }})
name: ${{ matrix.os }}
strategy:
fail-fast: false
matrix:
# Ubuntu already covered by the coverage job above
os: [windows-latest, macos-latest]
runs-on: ${{ matrix.os }}
timeout-minutes: 15
timeout-minutes: 25
steps:
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
- uses: ./.github/actions/setup-gitnexus
- run: npx vitest run test/unit
with:
build: 'true'
- run: npx vitest run
working-directory: gitnexus

View file

@ -16,10 +16,8 @@ concurrency:
# ── Reusable workflow orchestration ─────────────────────────────────
# Each concern lives in its own workflow file for maintainability:
# ci-quality.yml — typecheck (tsc --noEmit)
# ci-unit-tests.yml — unit tests with coverage + cross-platform
# ci-integration.yml — integration test matrix (3 OS x 4 groups)
#
# Shared setup is DRY via .github/actions/setup-gitnexus composite action.
# ci-tests.yml — all tests with coverage (ubuntu) + cross-platform
# ci-report.yml — PR comment (workflow_run trigger for fork write access)
jobs:
quality:
@ -27,56 +25,16 @@ jobs:
permissions:
contents: read
unit-tests:
uses: ./.github/workflows/ci-unit-tests.yml
tests:
uses: ./.github/workflows/ci-tests.yml
permissions:
contents: read
integration:
uses: ./.github/workflows/ci-integration.yml
with:
collect-coverage: ${{ github.event_name == 'pull_request' }}
permissions:
contents: read
# ── Save PR metadata for the reporting workflow ─────────────────
# The ci-report.yml workflow (triggered by workflow_run) needs the
# PR number and job results to post a comment. We save them as an
# artifact because workflow_run context doesn't reliably carry PR
# info for fork PRs.
save-pr-meta:
name: Save PR Metadata
if: always() && github.event_name == 'pull_request'
needs: [quality, unit-tests, integration]
runs-on: ubuntu-latest
timeout-minutes: 5
steps:
- name: Write metadata
shell: bash
env:
PR_NUMBER: ${{ github.event.number }}
QUALITY: ${{ needs.quality.result }}
UNIT: ${{ needs.unit-tests.result }}
INTEG: ${{ needs.integration.result }}
run: |
mkdir -p pr-meta
echo "$PR_NUMBER" > pr-meta/pr_number
echo "$QUALITY" > pr-meta/quality_result
echo "$UNIT" > pr-meta/unit_result
echo "$INTEG" > pr-meta/integration_result
- name: Upload PR metadata
uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4
with:
name: pr-meta
path: pr-meta/
retention-days: 1
# ── Unified CI gate ──────────────────────────────────────────────
# Single required check for branch protection.
ci-status:
name: CI Gate
needs: [quality, unit-tests, integration]
needs: [quality, tests]
if: always()
runs-on: ubuntu-latest
timeout-minutes: 5
@ -85,15 +43,41 @@ jobs:
shell: bash
env:
QUALITY: ${{ needs.quality.result }}
UNIT: ${{ needs.unit-tests.result }}
INTEG: ${{ needs.integration.result }}
TESTS: ${{ needs.tests.result }}
run: |
echo "Quality: $QUALITY"
echo "Unit Tests: $UNIT"
echo "Integration: $INTEG"
echo "Quality: $QUALITY"
echo "Tests: $TESTS"
if [[ "$QUALITY" != "success" ]] ||
[[ "$UNIT" != "success" ]] ||
[[ "$INTEG" != "success" ]]; then
[[ "$TESTS" != "success" ]]; then
echo "::error::One or more CI jobs failed"
exit 1
fi
# ── PR metadata for ci-report.yml ────────────────────────────────
# Saves PR number and job results so the workflow_run-triggered
# report can post comments with a write token (works for forks).
save-pr-meta:
name: Save PR Metadata
if: always() && github.event_name == 'pull_request'
needs: [quality, tests]
runs-on: ubuntu-latest
timeout-minutes: 5
steps:
- name: Write PR metadata
shell: bash
env:
PR_NUMBER: ${{ github.event.pull_request.number }}
QUALITY: ${{ needs.quality.result }}
TESTS: ${{ needs.tests.result }}
run: |
mkdir -p pr-meta
echo "$PR_NUMBER" > pr-meta/pr-number
echo "$QUALITY" > pr-meta/quality-result
echo "$TESTS" > pr-meta/tests-result
- name: Upload PR metadata
uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4
with:
name: pr-meta
path: pr-meta/
retention-days: 1

View file

@ -2,8 +2,9 @@ name: Claude Code Review
# Uses pull_request_target so the workflow runs as defined on the default branch,
# which allows access to secrets for posting review comments on fork PRs.
# SECURITY: The checkout below uses the PR head SHA to review the correct code.
# The claude-code-action sandboxes execution — it does NOT run arbitrary code
# SECURITY: The checkout pins the fork's HEAD SHA (not the branch name) to
# prevent TOCTOU races (force-push between trigger and checkout). The
# claude-code-action sandboxes execution — it does NOT run arbitrary code
# from the checked-out source.
on:
@ -15,6 +16,11 @@ on:
issue_comment:
types: [created]
# Serialize per-PR to avoid racing review comments.
concurrency:
group: claude-review-${{ github.event.issue.number || github.event.pull_request.number }}
cancel-in-progress: false
jobs:
claude-review:
# Run only when:
@ -41,13 +47,13 @@ jobs:
runs-on: ubuntu-latest
timeout-minutes: 30
permissions:
contents: write # needed to push fork branch to origin
contents: read
pull-requests: write
issues: read
id-token: write
steps:
# For issue_comment triggers, resolve the PR number, head SHA, and branch name
# For issue_comment triggers, resolve the PR number, head SHA, and fork repo
- name: Resolve PR context
id: pr
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7
@ -66,32 +72,24 @@ jobs:
}
core.setOutput('number', pr.number);
core.setOutput('sha', pr.head.sha);
core.setOutput('repo', pr.head.repo.full_name);
core.setOutput('branch', pr.head.ref);
core.setOutput('is_fork', String(pr.head.repo.full_name !== pr.base.repo.full_name));
- name: Checkout PR head
uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
with:
repository: ${{ steps.pr.outputs.repo }}
ref: ${{ steps.pr.outputs.sha }}
fetch-depth: 1
# claude-code-action fetches branches by name from origin, which fails
# for fork PRs. Work around by pushing the fork branch to origin so
# the action can find it. Cleaned up in the post step below.
- name: Push fork branch to origin
if: steps.pr.outputs.is_fork == 'true'
run: git push origin HEAD:refs/heads/${{ steps.pr.outputs.branch }}
- name: Run Claude Code Review
id: claude-review
uses: anthropics/claude-code-action@9469d113c6afd29550c402740f22d1a97dd1209b # v1
with:
claude_code_oauth_token: ${{ secrets.CLAUDE_CODE_OAUTH_TOKEN }}
github_token: ${{ secrets.GITHUB_TOKEN }}
allowed_non_write_users: '*'
show_full_output: true
plugin_marketplaces: 'https://github.com/anthropics/claude-code.git'
plugins: 'code-review@claude-code-plugins'
prompt: '/code-review:code-review ${{ github.repository }}/pull/${{ steps.pr.outputs.number }}'
# Clean up the temporary branch we pushed for fork PRs
- name: Delete fork branch from origin
if: always() && steps.pr.outputs.is_fork == 'true'
run: git push origin --delete refs/heads/${{ steps.pr.outputs.branch }} || true

View file

@ -10,13 +10,42 @@ on:
pull_request_review:
types: [submitted]
# Serialize per-PR/issue to avoid racing comments.
concurrency:
group: claude-code-${{ github.event.issue.number || github.event.pull_request.number || github.event.issue.id }}
cancel-in-progress: false
jobs:
claude:
if: |
(github.event_name == 'issue_comment' && contains(github.event.comment.body, '@claude')) ||
(github.event_name == 'pull_request_review_comment' && contains(github.event.comment.body, '@claude')) ||
(github.event_name == 'pull_request_review' && contains(github.event.review.body, '@claude')) ||
(github.event_name == 'issues' && (contains(github.event.issue.body, '@claude') || contains(github.event.issue.title, '@claude')))
(
github.event_name == 'issue_comment' &&
contains(github.event.comment.body, '@claude') &&
(github.event.comment.author_association == 'OWNER' ||
github.event.comment.author_association == 'MEMBER' ||
github.event.comment.author_association == 'COLLABORATOR')
) ||
(
github.event_name == 'pull_request_review_comment' &&
contains(github.event.comment.body, '@claude') &&
(github.event.comment.author_association == 'OWNER' ||
github.event.comment.author_association == 'MEMBER' ||
github.event.comment.author_association == 'COLLABORATOR')
) ||
(
github.event_name == 'pull_request_review' &&
contains(github.event.review.body, '@claude') &&
(github.event.review.author_association == 'OWNER' ||
github.event.review.author_association == 'MEMBER' ||
github.event.review.author_association == 'COLLABORATOR')
) ||
(
github.event_name == 'issues' &&
(contains(github.event.issue.body, '@claude') || contains(github.event.issue.title, '@claude')) &&
(github.event.issue.author_association == 'OWNER' ||
github.event.issue.author_association == 'MEMBER' ||
github.event.issue.author_association == 'COLLABORATOR')
)
runs-on: ubuntu-latest
timeout-minutes: 30
permissions:
@ -24,11 +53,47 @@ jobs:
pull-requests: write
issues: write
id-token: write
actions: read # Required for Claude to read CI results on PRs
actions: read # required for Claude to read CI results on PRs
steps:
# For PR-related triggers, resolve the fork repo so we can checkout correctly.
- name: Resolve PR context
id: pr
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7
with:
script: |
// Determine if this event is PR-related
let prNumber = null;
if (context.eventName === 'issue_comment' && context.payload.issue.pull_request) {
prNumber = context.payload.issue.number;
} else if (context.eventName === 'pull_request_review_comment') {
prNumber = context.payload.pull_request.number;
} else if (context.eventName === 'pull_request_review') {
prNumber = context.payload.pull_request.number;
}
if (!prNumber) {
core.setOutput('is_pr', 'false');
return;
}
const resp = await github.rest.pulls.get({
owner: context.repo.owner,
repo: context.repo.repo,
pull_number: prNumber,
});
const pr = resp.data;
core.setOutput('is_pr', 'true');
core.setOutput('number', String(prNumber));
core.setOutput('sha', pr.head.sha);
core.setOutput('repo', pr.head.repo.full_name);
core.setOutput('branch', pr.head.ref);
- name: Checkout repository
uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
with:
repository: ${{ steps.pr.outputs.is_pr == 'true' && steps.pr.outputs.repo || github.repository }}
ref: ${{ steps.pr.outputs.is_pr == 'true' && steps.pr.outputs.sha || '' }}
fetch-depth: 1
- name: Run Claude Code
@ -36,6 +101,9 @@ jobs:
uses: anthropics/claude-code-action@9469d113c6afd29550c402740f22d1a97dd1209b # v1
with:
claude_code_oauth_token: ${{ secrets.CLAUDE_CODE_OAUTH_TOKEN }}
github_token: ${{ secrets.GITHUB_TOKEN }}
allowed_non_write_users: '*'
show_full_output: true
# This is an optional setting that allows Claude to read CI results on PRs
additional_permissions: |

View file

@ -12,6 +12,7 @@ jobs:
uses: ./.github/workflows/ci.yml
permissions:
contents: read
actions: read
pull-requests: write
publish:

4
.gitignore vendored
View file

@ -67,4 +67,6 @@ gitnexus/test/fixtures/mini-repo/.gitignore
# Ignore csharp generated obj and bin folders
gitnexus/test/fixtures/lang-resolution/**/obj
gitnexus/test/fixtures/lang-resolution/**/bin
GitNexus.sln
GitNexus.sln
# Git worktrees
.worktrees/

View file

@ -0,0 +1,33 @@
import { defineConfig } from 'vitest/config';
export default defineConfig({
test: {
globalSetup: ['test/global-setup.ts'],
include: ['test/**/*.test.ts'],
testTimeout: 30000,
hookTimeout: 120000,
pool: 'forks',
globals: true,
setupFiles: ['test/setup.ts'],
teardownTimeout: 3000,
dangerouslyIgnoreUnhandledErrors: true, // LadybugDB N-API destructor segfaults on fork exit — not a test failure
coverage: {
provider: 'v8',
include: ['src/**/*.ts'],
exclude: [
'src/cli/index.ts', // CLI entry point (commander wiring)
'src/server/**', // HTTP server (requires network)
'src/core/wiki/**', // Wiki generation (requires LLM)
],
// Auto-ratchet: vitest bumps thresholds when coverage exceeds them.
// CI will fail if a PR drops below these floors.
thresholds: {
statements: 26,
branches: 23,
functions: 28,
lines: 27,
autoUpdate: true,
},
},
},
});

View file

@ -1,7 +1,7 @@
<!-- gitnexus:start -->
# GitNexus — Code Intelligence
This project is indexed by GitNexus as **GitNexus** (1999 symbols, 4681 relationships, 149 execution flows). Use the GitNexus MCP tools to understand code, assess impact, and navigate safely.
This project is indexed by GitNexus as **GitNexus** (2273 symbols, 5419 relationships, 174 execution flows). Use the GitNexus MCP tools to understand code, assess impact, and navigate safely.
> If any GitNexus tool warns the index is stale, run `npx gitnexus analyze` in terminal first.

View file

@ -1,7 +1,7 @@
<!-- gitnexus:start -->
# GitNexus — Code Intelligence
This project is indexed by GitNexus as **GitNexus** (1999 symbols, 4681 relationships, 149 execution flows). Use the GitNexus MCP tools to understand code, assess impact, and navigate safely.
This project is indexed by GitNexus as **GitNexus** (2273 symbols, 5419 relationships, 174 execution flows). Use the GitNexus MCP tools to understand code, assess impact, and navigate safely.
> If any GitNexus tool warns the index is stale, run `npx gitnexus analyze` in terminal first.

View file

@ -34,7 +34,7 @@ https://github.com/user-attachments/assets/172685ba-8e54-4ea7-9ad1-e31a3398da72
> *Like DeepWiki, but deeper.* DeepWiki helps you *understand* code. GitNexus lets you *analyze* it — because a knowledge graph tracks every relationship, not just descriptions.
**TL;DR:** The **Web UI** is a quick way to chat with any repo. The **CLI + MCP** is how you make your AI agent actually reliable — it gives Cursor, Claude Code, and friends a deep architectural view of your codebase so they stop missing dependencies, breaking call chains, and shipping blind edits. Even smaller models get full architectural clarity, making it compete with goliath models.
**TL;DR:** The **Web UI** is a quick way to chat with any repo. The **CLI + MCP** is how you make your AI agent actually reliable — it gives Cursor, Claude Code, Codex, and friends a deep architectural view of your codebase so they stop missing dependencies, breaking call chains, and shipping blind edits. Even smaller models get full architectural clarity, making it compete with goliath models.
---
@ -48,7 +48,7 @@ https://github.com/user-attachments/assets/172685ba-8e54-4ea7-9ad1-e31a3398da72
| | **CLI + MCP** | **Web UI** |
| ----------------- | -------------------------------------------------------------- | ------------------------------------------------------------ |
| **What** | Index repos locally, connect AI agents via MCP | Visual graph explorer + AI chat in browser |
| **For** | Daily development with Cursor, Claude Code, Windsurf, OpenCode | Quick exploration, demos, one-off analysis |
| **For** | Daily development with Cursor, Claude Code, Codex, Windsurf, OpenCode | Quick exploration, demos, one-off analysis |
| **Scale** | Full repos, any size | Limited by browser memory (~5k files), or unlimited via backend mode |
| **Install** | `npm install -g gitnexus` | No install —[gitnexus.vercel.app](https://gitnexus.vercel.app) |
| **Storage** | LadybugDB native (fast, persistent) | LadybugDB WASM (in-memory, per session) |
@ -84,16 +84,23 @@ To configure MCP for your editor, run `npx gitnexus setup` once — or set it up
| --------------------- | --- | ------ | -------------------- | -------------- |
| **Claude Code** | Yes | Yes | Yes (PreToolUse + PostToolUse) | **Full** |
| **Cursor** | Yes | Yes | — | MCP + Skills |
| **Codex** | Yes | Yes | — | MCP + Skills |
| **Windsurf** | Yes | — | — | MCP |
| **OpenCode** | Yes | Yes | — | MCP + Skills |
| **Codex** | Yes | — | — | MCP |
> **Claude Code** gets the deepest integration: MCP tools + agent skills + PreToolUse hooks that enrich searches with graph context + PostToolUse hooks that auto-reindex after commits.
### Community Integrations
## Community Integrations
| Agent | Install | Source |
|-------|---------|--------|
| [pi](https://pi.dev) | `pi install npm:pi-gitnexus` | [pi-gitnexus](https://github.com/tintinweb/pi-gitnexus) |
Built by the community — not officially maintained, but worth checking out.
| Project | Author | Description |
|---------|--------|-------------|
| [pi-gitnexus](https://github.com/tintinweb/pi-gitnexus) | [@tintinweb](https://github.com/tintinweb) | GitNexus plugin for [pi](https://pi.dev) — `pi install npm:pi-gitnexus` |
| [gitnexus-stable-ops](https://github.com/ShunsukeHayashi/gitnexus-stable-ops) | [@ShunsukeHayashi](https://github.com/ShunsukeHayashi) | Stable ops & deployment workflows (Miyabi ecosystem) |
> Have a project built on GitNexus? Open a PR to add it here!
If you prefer manual configuration:
@ -103,6 +110,12 @@ If you prefer manual configuration:
claude mcp add gitnexus -- npx -y gitnexus@latest mcp
```
**Codex** (full support — MCP + skills):
```bash
codex mcp add gitnexus -- npx -y gitnexus@latest mcp
```
**Cursor** (`~/.cursor/mcp.json` — global, works for all projects):
```json
@ -129,6 +142,14 @@ claude mcp add gitnexus -- npx -y gitnexus@latest mcp
}
```
**Codex** (`~/.codex/config.toml` for system scope, or `.codex/config.toml` for project scope):
```toml
[mcp_servers.gitnexus]
command = "npx"
args = ["-y", "gitnexus@latest", "mcp"]
```
### CLI Commands
```bash
@ -271,7 +292,7 @@ The web UI uses the same indexing pipeline as the CLI but runs entirely in WebAs
## The Problem GitNexus Solves
Tools like **Cursor**, **Claude Code**, **Cline**, **Roo Code**, and **Windsurf** are powerful — but they don't truly know your codebase structure.
Tools like **Cursor**, **Claude Code**, **Codex**, **Cline**, **Roo Code**, and **Windsurf** are powerful — but they don't truly know your codebase structure.
**What happens:**

View file

@ -0,0 +1,57 @@
---
review_agents: [kieran-typescript-reviewer, pattern-recognition-specialist, architecture-strategist, data-integrity-guardian, security-sentinel, performance-oracle, code-simplicity-reviewer]
plan_review_agents: [kieran-typescript-reviewer, architecture-strategist, code-simplicity-reviewer]
voltagent_agents: [voltagent-lang:typescript-pro, voltagent-qa-sec:security-auditor, voltagent-data-ai:database-optimizer]
---
# Review Context
## Project Overview
GitNexus is a code intelligence tool that builds a knowledge graph from source code using tree-sitter AST parsing across 12 languages and KuzuDB for graph storage. Two packages: `gitnexus/` (CLI/MCP, TypeScript) and `gitnexus-web/` (browser).
## Cross-Language Pattern Consistency (pattern-recognition-specialist)
- 12 language-specific type extractors in `gitnexus/src/core/ingestion/type-extractors/` must follow identical patterns for: async unwrapping, constructor binding, namespace handling, nullable type stripping, for-loop element typing.
- Past bugs: C#/Rust missing `await_expression` unwrapping that TypeScript handled correctly; PHP backslash namespace splitting inconsistent with other languages' `::` / `.` splitting.
- When reviewing type extractor changes, verify the same pattern exists in ALL applicable language files — asymmetry is the #1 source of bugs.
## Data Integrity (data-integrity-guardian)
- KuzuDB graph operations: schema in `gitnexus/src/core/kuzu/schema.ts`, adapter in `kuzu-adapter.ts`.
- The ingestion pipeline writes symbols and relationships to the graph — changes to node/relation schemas or the ingestion pipeline can corrupt the index.
- Known issue: KuzuDB `close()` hangs on Linux due to C++ destructor — use `detachKuzu()` pattern.
- `lbug-adapter.ts` fallback path needs quote/newline escaping for Cypher injection prevention.
## Security (security-sentinel)
- Cypher query construction in `lbug-adapter.ts` and `kuzu-adapter.ts` — watch for injection via unescaped user-provided symbol names.
- CLI accepts `--repo` parameter and file paths — validate against path traversal.
- MCP server exposes tools to external AI agents — all tool inputs are untrusted.
## Performance (performance-oracle)
- Tree-sitter buffer size is adaptive (512KB–32MB) via `getTreeSitterBufferSize()` in `constants.ts`.
- The ingestion pipeline processes entire repositories — O(n) per file with potential O(n²) in cross-file resolution.
- KuzuDB batch inserts vs individual inserts matter for large repos.
## Architecture (architecture-strategist)
- Ingestion pipeline phases: structure → parsing → imports → calls → heritage → processes → type resolution.
- Shared modules: `export-detection.ts`, `constants.ts`, `utils.ts` — changes here have wide blast radius.
- `gitnexus-web` package drifts behind CLI — flag if a change should be mirrored.
## Voltagent Supplementary Agents
Invoke these via the Agent tool alongside `/ce:review` for deeper specialist analysis. These cover gaps that compound-engineering agents don't:
### voltagent-lang:typescript-pro
**When:** Changes touch type-resolution logic, generics, conditional types, or complex type-level programming in `type-env.ts`, `type-extractors/*.ts`, or `types.ts`.
**Why:** The type resolution system uses advanced TypeScript patterns (discriminated unions, mapped types, recursive generics) that benefit from deep TS type-system review beyond what kieran-typescript-reviewer covers.
### voltagent-qa-sec:security-auditor
**When:** Changes touch MCP tool handlers, Cypher query construction, CLI argument parsing, or any code that processes external input.
**Why:** GitNexus is an MCP server — all tool inputs come from untrusted AI agents. Systematic OWASP-level audit catches injection vectors that spot-checking misses. Past finding: `lbug-adapter.ts` fallback path had unescaped newlines in Cypher queries.
### voltagent-data-ai:database-optimizer
**When:** Changes touch `kuzu-adapter.ts`, `schema.ts`, `lbug-adapter.ts`, or any Cypher query construction/execution.
**Why:** No CE agent specializes in graph database optimization. KuzuDB batch insert patterns, index usage, and query planning directly affect analysis speed on large repos.
## Review Tooling
- Use `gitnexus_impact()` before approving changes to any symbol — check d=1 (WILL BREAK) callers.
- Use `gitnexus_detect_changes({scope: "compare", base_ref: "main"})` to map PR diffs to affected execution flows.
- Use claude-mem to surface past architectural decisions relevant to the code under review.

View file

@ -2,7 +2,7 @@
MCP Bridge for GitNexus
Starts the GitNexus MCP server as a subprocess and provides a Python interface
to call MCP tools. Used by the bash wrapper scripts and the augmentation layer.
to call MCP tools. Used by the bash wrapper scripts and the augmentation layer..
The bridge communicates with the MCP server via stdio using the JSON-RPC protocol.
"""

View file

@ -33,7 +33,7 @@
"graphology-layout-noverlap": "^0.4.2",
"isomorphic-git": "^1.36.1",
"jszip": "^3.10.1",
"@ladybugdb/wasm-core": "^0.15.1",
"@ladybugdb/wasm-core": "^0.15.2",
"langchain": "^1.2.10",
"lru-cache": "^11.2.4",
"lucide-react": "^0.562.0",

View file

@ -40,6 +40,7 @@ const AppContent = () => {
availableRepos,
setAvailableRepos,
switchRepo,
hydrateWorkerFromServer,
} = useAppState();
const graphCanvasRef = useRef<GraphCanvasHandle>(null);
@ -157,21 +158,31 @@ const AppContent = () => {
// Transition directly to exploring view
setViewMode('exploring');
setProgress(null);
// Initialize agent if LLM is configured
if (getActiveProviderConfig()) {
initializeAgent(projectName);
}
// Hydrate the worker-side DB (LadybugDB + BM25) so Query/Processes/embeddings work
hydrateWorkerFromServer(result.nodes, result.relationships, result.fileContents).then(() => {
// Initialize agent if LLM is configured
if (getActiveProviderConfig()) {
initializeAgent(projectName);
}
// Auto-start embeddings
startEmbeddings().catch((err) => {
if (err?.name === 'WebGPUNotAvailableError' || err?.message?.includes('WebGPU')) {
startEmbeddings('wasm').catch(console.warn);
} else {
console.warn('Embeddings auto-start failed:', err);
// Auto-start embeddings (now that LadybugDB is ready)
startEmbeddings().catch((err) => {
if (err?.name === 'WebGPUNotAvailableError' || err?.message?.includes('WebGPU')) {
startEmbeddings('wasm').catch(console.warn);
} else {
console.warn('Embeddings auto-start failed:', err);
}
});
}).catch((err) => {
console.warn('Worker hydration failed (non-fatal):', err);
// Still initialize agent even if hydration fails
if (getActiveProviderConfig()) {
initializeAgent(projectName);
}
});
}, [setViewMode, setGraph, setFileContents, setProjectName, initializeAgent, startEmbeddings]);
}, [setViewMode, setGraph, setFileContents, setProjectName, setProgress, initializeAgent, startEmbeddings, hydrateWorkerFromServer]);
// Auto-connect when ?server query param is present (bookmarkable shortcut)
const autoConnectRan = useRef(false);

View file

@ -281,7 +281,7 @@ export const SettingsPanel = ({ isOpen, onClose, onSettingsSaved, backendUrl, is
if (!isOpen) return null;
const providers: LLMProvider[] = ['openai', 'gemini', 'anthropic', 'azure-openai', 'ollama', 'openrouter'];
const providers: LLMProvider[] = ['openai', 'gemini', 'anthropic', 'azure-openai', 'ollama', 'openrouter', 'minimax'];
return (
@ -366,7 +366,7 @@ export const SettingsPanel = ({ isOpen, onClose, onSettingsSaved, backendUrl, is
w-8 h-8 rounded-lg flex items-center justify-center text-lg
${settings.activeProvider === provider ? 'bg-accent/20' : 'bg-surface'}
`}>
{provider === 'openai' ? '🤖' : provider === 'gemini' ? '💎' : provider === 'anthropic' ? '🧠' : provider === 'ollama' ? '🦙' : provider === 'openrouter' ? '🌐' : '☁️'}
{provider === 'openai' ? '🤖' : provider === 'gemini' ? '💎' : provider === 'anthropic' ? '🧠' : provider === 'ollama' ? '🦙' : provider === 'openrouter' ? '🌐' : provider === 'minimax' ? '⚡' : '☁️'}
</div>
<span className="font-medium">{getProviderDisplayName(provider)}</span>
</button>
@ -814,7 +814,64 @@ export const SettingsPanel = ({ isOpen, onClose, onSettingsSaved, backendUrl, is
</div>
)}
{/* MiniMax Settings */}
{settings.activeProvider === 'minimax' && (
<div className="space-y-4 animate-fade-in">
<div className="space-y-2">
<label className="flex items-center gap-2 text-sm font-medium text-text-secondary">
<Key className="w-4 h-4" />
API Key
</label>
<div className="relative">
<input
type={showApiKey['minimax'] ? 'text' : 'password'}
value={settings.minimax?.apiKey ?? ''}
onChange={e => setSettings(prev => ({
...prev,
minimax: { ...prev.minimax!, apiKey: e.target.value }
}))}
placeholder="Enter your MiniMax API key"
className="w-full px-4 py-3 pr-12 bg-elevated border border-border-subtle rounded-xl text-text-primary placeholder:text-text-muted focus:border-accent focus:ring-2 focus:ring-accent/20 outline-none transition-all"
/>
<button
type="button"
onClick={() => toggleApiKeyVisibility('minimax')}
className="absolute right-3 top-1/2 -translate-y-1/2 p-1 text-text-muted hover:text-text-primary transition-colors"
>
{showApiKey['minimax'] ? <EyeOff className="w-4 h-4" /> : <Eye className="w-4 h-4" />}
</button>
</div>
<p className="text-xs text-text-muted">
Get your API key from{' '}
<a
href="https://platform.minimax.io"
target="_blank"
rel="noopener noreferrer"
className="text-accent hover:underline"
>
MiniMax Platform
</a>
</p>
</div>
<div className="space-y-2">
<label className="text-sm font-medium text-text-secondary">Model</label>
<input
type="text"
value={settings.minimax?.model ?? 'MiniMax-M2.5'}
onChange={e => setSettings(prev => ({
...prev,
minimax: { ...prev.minimax!, model: e.target.value }
}))}
placeholder="e.g., MiniMax-M2.5, MiniMax-M2.5-highspeed"
className="w-full px-4 py-3 bg-elevated border border-border-subtle rounded-xl text-text-primary placeholder:text-text-muted focus:border-accent focus:ring-2 focus:ring-accent/20 outline-none transition-all font-mono text-sm"
/>
<p className="text-xs text-text-muted">
Available models: MiniMax-M2.5 (default), MiniMax-M2.5-highspeed (faster)
</p>
</div>
</div>
)}
{/* Privacy Note */}
<div className="p-4 bg-elevated/50 border border-border-subtle rounded-xl">

View file

@ -10,5 +10,6 @@ export enum SupportedLanguages {
Rust = 'rust',
PHP = 'php',
Ruby = 'ruby',
Kotlin = 'kotlin',
Swift = 'swift',
}

View file

@ -27,7 +27,7 @@ export type CallRouter = (
const noRouting: CallRouter = () => null;
/** Per-language call routing. noRouting = no special routing (normal call processing) */
export const callRouters: Record<SupportedLanguages, CallRouter> = {
export const callRouters = {
[SupportedLanguages.JavaScript]: noRouting,
[SupportedLanguages.TypeScript]: noRouting,
[SupportedLanguages.Python]: noRouting,
@ -40,7 +40,8 @@ export const callRouters: Record<SupportedLanguages, CallRouter> = {
[SupportedLanguages.CPlusPlus]: noRouting,
[SupportedLanguages.C]: noRouting,
[SupportedLanguages.Ruby]: routeRubyCall,
};
[SupportedLanguages.Kotlin]: noRouting,
} satisfies Record<SupportedLanguages, CallRouter>;
// ── Result types ────────────────────────────────────────────────────────────

View file

@ -495,6 +495,7 @@ export const LANGUAGE_QUERIES: Record<SupportedLanguages, string> = {
[SupportedLanguages.Rust]: RUST_QUERIES,
[SupportedLanguages.PHP]: PHP_QUERIES,
[SupportedLanguages.Ruby]: RUBY_QUERIES,
[SupportedLanguages.Kotlin]: '', // Kotlin WASM parser not yet available for web
[SupportedLanguages.Swift]: SWIFT_QUERIES,
};

View file

@ -190,7 +190,7 @@ export const loadGraphToLbug = async (
for (const tableName of NODE_TABLES) {
try {
const countRes = await conn.query(`MATCH (n:${tableName}) RETURN count(n) AS cnt`);
const countRows = await countRes.getAll();
const countRows = await (countRes.getAll?.() ?? countRes.getAllObjects?.() ?? countRes.getAllRows?.() ?? []);
const countRow = countRows[0];
const count = countRow ? (countRow.cnt ?? countRow[0] ?? 0) : 0;
totalNodes += Number(count);
@ -293,8 +293,8 @@ export const executeQuery = async (cypher: string): Promise<any[]> => {
});
}
// Collect all rows
const allRows = await result.getAll();
// Collect all rows (handle API differences across LadybugDB versions)
const allRows = await (result.getAll?.() ?? result.getAllObjects?.() ?? result.getAllRows?.() ?? []);
const rows: any[] = [];
for (const row of allRows) {
// Convert tuple to named object if we have column names and row is array
@ -331,7 +331,7 @@ export const getLbugStats = async (): Promise<{ nodes: number; edges: number }>
for (const tableName of NODE_TABLES) {
try {
const nodeResult = await conn.query(`MATCH (n:${tableName}) RETURN count(n) AS cnt`);
const nodeRows = await nodeResult.getAll();
const nodeRows = await (nodeResult.getAll?.() ?? nodeResult.getAllObjects?.() ?? nodeResult.getAllRows?.() ?? []);
const nodeRow = nodeRows[0];
totalNodes += Number(nodeRow?.cnt ?? nodeRow?.[0] ?? 0);
} catch {
@ -343,7 +343,7 @@ export const getLbugStats = async (): Promise<{ nodes: number; edges: number }>
let totalEdges = 0;
try {
const edgeResult = await conn.query(`MATCH ()-[r:${REL_TABLE_NAME}]->() RETURN count(r) AS cnt`);
const edgeRows = await edgeResult.getAll();
const edgeRows = await (edgeResult.getAll?.() ?? edgeResult.getAllObjects?.() ?? edgeResult.getAllRows?.() ?? []);
const edgeRow = edgeRows[0];
totalEdges = Number(edgeRow?.cnt ?? edgeRow?.[0] ?? 0);
} catch {
@ -408,7 +408,7 @@ export const executePrepared = async (
const result = await conn.execute(stmt, params);
const rows = await result.getAll();
const rows = await (result.getAll?.() ?? result.getAllObjects?.() ?? result.getAllRows?.() ?? []);
await stmt.close();
return rows;
@ -472,7 +472,7 @@ export const testArrayParams = async (): Promise<{ success: boolean; error?: str
for (const tableName of NODE_TABLES) {
try {
const nodeResult = await conn.query(`MATCH (n:${tableName}) RETURN n.id AS id LIMIT 1`);
const nodeRows = await nodeResult.getAll();
const nodeRows = await (nodeResult.getAll?.() ?? nodeResult.getAllObjects?.() ?? nodeResult.getAllRows?.() ?? []);
const nodeRow = nodeRows[0];
if (nodeRow) {
testNodeId = nodeRow.id ?? nodeRow[0];
@ -509,7 +509,7 @@ export const testArrayParams = async (): Promise<{ success: boolean; error?: str
const verifyResult = await conn.query(
`MATCH (e:${EMBEDDING_TABLE_NAME} {nodeId: '${testNodeId}'}) RETURN e.embedding AS emb`
);
const verifyRows = await verifyResult.getAll();
const verifyRows = await (verifyResult.getAll?.() ?? verifyResult.getAllObjects?.() ?? verifyResult.getAllRows?.() ?? []);
const verifyRow = verifyRows[0];
const storedEmb = verifyRow?.emb ?? verifyRow?.[0];

View file

@ -13,14 +13,15 @@ import { ChatAnthropic } from '@langchain/anthropic';
import { ChatOllama } from '@langchain/ollama';
import type { BaseChatModel } from '@langchain/core/language_models/chat_models';
import { createGraphRAGTools } from './tools';
import type {
ProviderConfig,
import type {
ProviderConfig,
OpenAIConfig,
AzureOpenAIConfig,
AzureOpenAIConfig,
GeminiConfig,
AnthropicConfig,
OllamaConfig,
OpenRouterConfig,
MiniMaxConfig,
AgentStreamChunk,
} from './types';
import {
@ -197,7 +198,7 @@ export const createChatModel = (config: ProviderConfig): BaseChatModel => {
case 'openrouter': {
const openRouterConfig = config as OpenRouterConfig;
// Debug logging
if (import.meta.env.DEV) {
console.log('🌐 OpenRouter config:', {
@ -207,11 +208,11 @@ export const createChatModel = (config: ProviderConfig): BaseChatModel => {
baseUrl: openRouterConfig.baseUrl,
});
}
if (!openRouterConfig.apiKey || openRouterConfig.apiKey.trim() === '') {
throw new Error('OpenRouter API key is required but was not provided');
}
return new ChatOpenAI({
openAIApiKey: openRouterConfig.apiKey,
apiKey: openRouterConfig.apiKey, // Fallback for some versions
@ -225,7 +226,26 @@ export const createChatModel = (config: ProviderConfig): BaseChatModel => {
streaming: true,
});
}
case 'minimax': {
const minimaxConfig = config as MiniMaxConfig;
if (!minimaxConfig.apiKey || minimaxConfig.apiKey.trim() === '') {
throw new Error('MiniMax API key is required but was not provided');
}
return new ChatAnthropic({
anthropicApiKey: minimaxConfig.apiKey,
model: minimaxConfig.model,
temperature: minimaxConfig.temperature ?? 0.1,
maxTokens: minimaxConfig.maxTokens ?? 8192,
streaming: true,
clientOptions: {
baseURL: 'https://api.minimax.io/anthropic',
},
});
}
default:
throw new Error(`Unsupported provider: ${(config as any).provider}`);
}

View file

@ -5,9 +5,9 @@
* All API keys are stored locally - never sent to any server except the LLM provider.
*/
import {
LLMSettings,
DEFAULT_LLM_SETTINGS,
import {
LLMSettings,
DEFAULT_LLM_SETTINGS,
LLMProvider,
OpenAIConfig,
AzureOpenAIConfig,
@ -15,6 +15,7 @@ import {
AnthropicConfig,
OllamaConfig,
OpenRouterConfig,
MiniMaxConfig,
ProviderConfig,
} from './types';
@ -60,6 +61,10 @@ export const loadSettings = (): LLMSettings => {
...DEFAULT_LLM_SETTINGS.openrouter,
...parsed.openrouter,
},
minimax: {
...DEFAULT_LLM_SETTINGS.minimax,
...parsed.minimax,
},
};
} catch (error) {
console.warn('Failed to load LLM settings:', error);
@ -89,6 +94,7 @@ export const updateProviderSettings = <T extends LLMProvider>(
T extends 'gemini' ? Partial<Omit<GeminiConfig, 'provider'>> :
T extends 'anthropic' ? Partial<Omit<AnthropicConfig, 'provider'>> :
T extends 'ollama' ? Partial<Omit<OllamaConfig, 'provider'>> :
T extends 'minimax' ? Partial<Omit<MiniMaxConfig, 'provider'>> :
never
>
): LLMSettings => {
@ -162,6 +168,17 @@ export const updateProviderSettings = <T extends LLMProvider>(
saveSettings(updated);
return updated;
}
case 'minimax': {
const updated: LLMSettings = {
...current,
minimax: {
...(current.minimax ?? {}),
...(updates as Partial<Omit<MiniMaxConfig, 'provider'>>),
},
};
saveSettings(updated);
return updated;
}
default: {
// Should be unreachable due to T extends LLMProvider, but keep a safe fallback
const updated: LLMSettings = { ...current };
@ -245,7 +262,16 @@ export const getActiveProviderConfig = (): ProviderConfig | null => {
temperature: settings.openrouter.temperature,
maxTokens: settings.openrouter.maxTokens,
} as OpenRouterConfig;
case 'minimax':
if (!settings.minimax?.apiKey) {
return null;
}
return {
provider: 'minimax',
...settings.minimax,
} as MiniMaxConfig;
default:
return null;
}
@ -282,6 +308,8 @@ export const getProviderDisplayName = (provider: LLMProvider): string => {
return 'Ollama (Local)';
case 'openrouter':
return 'OpenRouter';
case 'minimax':
return 'MiniMax';
default:
return provider;
}
@ -303,6 +331,8 @@ export const getAvailableModels = (provider: LLMProvider): string[] => {
return ['claude-sonnet-4-20250514', 'claude-3-5-sonnet-20241022', 'claude-3-5-haiku-20241022', 'claude-3-opus-20240229'];
case 'ollama':
return ['llama3.2', 'llama3.1', 'mistral', 'codellama', 'deepseek-coder'];
case 'minimax':
return ['MiniMax-M2.5', 'MiniMax-M2.5-highspeed'];
default:
return [];
}

View file

@ -8,7 +8,7 @@
/**
* Supported LLM providers
*/
export type LLMProvider = 'openai' | 'azure-openai' | 'gemini' | 'anthropic' | 'ollama' | 'openrouter';
export type LLMProvider = 'openai' | 'azure-openai' | 'gemini' | 'anthropic' | 'ollama' | 'openrouter' | 'minimax';
/**
* Base configuration shared by all providers
@ -78,10 +78,19 @@ export interface OpenRouterConfig extends BaseProviderConfig {
baseUrl?: string; // defaults to https://openrouter.ai/api/v1
}
/**
* MiniMax configuration (Anthropic-compatible API)
*/
export interface MiniMaxConfig extends BaseProviderConfig {
provider: 'minimax';
apiKey: string;
model: string; // e.g., 'MiniMax-M2.5', 'MiniMax-M2.5-highspeed'
}
/**
* Union type for all provider configurations
*/
export type ProviderConfig = OpenAIConfig | AzureOpenAIConfig | GeminiConfig | AnthropicConfig | OllamaConfig | OpenRouterConfig;
export type ProviderConfig = OpenAIConfig | AzureOpenAIConfig | GeminiConfig | AnthropicConfig | OllamaConfig | OpenRouterConfig | MiniMaxConfig;
/**
* Stored settings (what goes to localStorage)
@ -98,6 +107,7 @@ export interface LLMSettings {
anthropic?: Partial<Omit<AnthropicConfig, 'provider'>>;
ollama?: Partial<Omit<OllamaConfig, 'provider'>>;
openrouter?: Partial<Omit<OpenRouterConfig, 'provider'>>;
minimax?: Partial<Omit<MiniMaxConfig, 'provider'>>;
// Intelligent Clustering Settings
intelligentClustering: boolean;
@ -148,6 +158,11 @@ export const DEFAULT_LLM_SETTINGS: LLMSettings = {
baseUrl: 'https://openrouter.ai/api/v1',
temperature: 0.1,
},
minimax: {
apiKey: '',
model: 'MiniMax-M2.5',
temperature: 0.1,
},
};
/**

View file

@ -41,6 +41,7 @@ const getWasmPath = (language: SupportedLanguages, filePath?: string): string =>
[SupportedLanguages.Rust]: '/wasm/rust/tree-sitter-rust.wasm',
[SupportedLanguages.PHP]: '/wasm/php/tree-sitter-php.wasm',
[SupportedLanguages.Ruby]: '/wasm/ruby/tree-sitter-ruby.wasm',
[SupportedLanguages.Kotlin]: '', // Kotlin WASM parser not yet available for web
[SupportedLanguages.Swift]: '/wasm/swift/tree-sitter-swift.wasm',
};

View file

@ -125,6 +125,7 @@ interface AppState {
runPipelineFromFiles: (files: FileEntry[], onProgress: (p: PipelineProgress) => void, clusteringConfig?: ProviderConfig) => Promise<PipelineResult>;
runQuery: (cypher: string) => Promise<any[]>;
isDatabaseReady: () => Promise<boolean>;
hydrateWorkerFromServer: (nodes: any[], relationships: any[], fileContents: Record<string, string>) => Promise<void>;
// Embedding state
embeddingStatus: EmbeddingStatus;
@ -482,6 +483,16 @@ export const AppStateProvider = ({ children }: { children: ReactNode }) => {
}
}, []);
const hydrateWorkerFromServer = useCallback(async (
nodes: any[],
relationships: any[],
fileContents: Record<string, string>
): Promise<void> => {
const api = apiRef.current;
if (!api) throw new Error('Worker not initialized');
await api.hydrateFromServerData(nodes, relationships, fileContents);
}, []);
// Embedding methods
const startEmbeddings = useCallback(async (forceDevice?: 'webgpu' | 'wasm'): Promise<void> => {
const api = apiRef.current;
@ -1018,15 +1029,23 @@ export const AppStateProvider = ({ children }: { children: ReactNode }) => {
setFileContents(fileMap);
setViewMode('exploring');
setProgress(null);
if (getActiveProviderConfig()) initializeAgent(pName);
// Hydrate the worker-side DB (LadybugDB + BM25) so Query/Processes/embeddings work
hydrateWorkerFromServer(result.nodes, result.relationships, result.fileContents).then(() => {
if (getActiveProviderConfig()) initializeAgent(pName);
startEmbeddings().catch((err) => {
if (err?.name === 'WebGPUNotAvailableError' || err?.message?.includes('WebGPU')) {
startEmbeddings('wasm').catch(console.warn);
} else {
console.warn('Embeddings auto-start failed:', err);
}
startEmbeddings().catch((err) => {
if (err?.name === 'WebGPUNotAvailableError' || err?.message?.includes('WebGPU')) {
startEmbeddings('wasm').catch(console.warn);
} else {
console.warn('Embeddings auto-start failed:', err);
}
});
}).catch((err) => {
console.warn('Worker hydration failed (non-fatal):', err);
// Still initialize agent even if hydration fails
if (getActiveProviderConfig()) initializeAgent(pName);
});
} catch (err) {
console.error('Repo switch failed:', err);
@ -1037,7 +1056,7 @@ export const AppStateProvider = ({ children }: { children: ReactNode }) => {
});
setTimeout(() => { setViewMode('exploring'); setProgress(null); }, 3000);
}
}, [serverBaseUrl, setProgress, setViewMode, setProjectName, setGraph, setFileContents, initializeAgent, startEmbeddings, setHighlightedNodeIds, clearAIToolHighlights, clearBlastRadius, setSelectedNode, setQueryResult, setCodeReferences, setCodePanelOpen, setCodeReferenceFocus]);
}, [serverBaseUrl, setProgress, setViewMode, setProjectName, setGraph, setFileContents, initializeAgent, startEmbeddings, hydrateWorkerFromServer, setHighlightedNodeIds, clearAIToolHighlights, clearBlastRadius, setSelectedNode, setQueryResult, setCodeReferences, setCodePanelOpen, setCodeReferenceFocus]);
const removeCodeReference = useCallback((id: string) => {
setCodeReferences(prev => {
@ -1142,6 +1161,7 @@ export const AppStateProvider = ({ children }: { children: ReactNode }) => {
runPipelineFromFiles,
runQuery,
isDatabaseReady,
hydrateWorkerFromServer,
// Embedding state and methods
embeddingStatus,
embeddingProgress,

View file

@ -1,5 +1,7 @@
import * as Comlink from 'comlink';
import { runIngestionPipeline, runPipelineFromFiles } from '../core/ingestion/pipeline';
import { createKnowledgeGraph } from '../core/graph/graph';
import type { GraphNode, GraphRelationship } from '../core/graph/types';
import { PipelineProgress, SerializablePipelineResult, serializePipelineResult } from '../types/pipeline';
import { FileEntry } from '../services/zip';
import {
@ -207,6 +209,50 @@ const workerApi = {
return serializePipelineResult(result);
},
/**
* Hydrate the worker-side database and indexes from server-loaded data.
* This is the missing step when using server/bridge mode — the main thread
* builds the React graph, but the worker's LadybugDB + BM25 stay empty.
*/
async hydrateFromServerData(
nodes: GraphNode[],
relationships: GraphRelationship[],
fileContents: Record<string, string>
): Promise<void> {
// 1. Build a KnowledgeGraph the same way the pipeline does
const graph = createKnowledgeGraph();
for (const node of nodes) graph.addNode(node);
for (const rel of relationships) graph.addRelationship(rel);
// 2. Store file contents for grep/read tools
storedFileContents = new Map(Object.entries(fileContents));
// 3. Build BM25 keyword index
const bm25DocCount = buildBM25Index(storedFileContents);
if (import.meta.env.DEV) {
console.log(`🔍 BM25 index built (server mode): ${bm25DocCount} documents`);
}
// 4. Set currentGraphResult so the agent context builder works
currentGraphResult = { graph, fileContents: storedFileContents };
// 5. Load graph into LadybugDB for Cypher queries (optional — gracefully degrades)
try {
const lbug = await getLbugAdapter();
await lbug.loadGraphToLbug(graph, storedFileContents);
if (import.meta.env.DEV) {
const stats = await lbug.getLbugStats();
console.log('✅ LadybugDB hydrated (server mode):', stats);
}
} catch (err) {
// LadybugDB is optional — silently continue without it
if (import.meta.env.DEV) {
console.warn('⚠️ LadybugDB hydration failed (non-fatal):', err);
}
}
},
/**
* Execute a Cypher query against the LadybugDB database
* @param cypher - The Cypher query string

99
gitnexus/CHANGELOG.md Normal file
View file

@ -0,0 +1,99 @@
# Changelog
All notable changes to GitNexus will be documented in this file.
## [1.4.7] - 2026-03-19
### Added
- **Phase 8 field/property type resolution** — ACCESSES edges with `declaredType` for field reads/writes (#354)
- **Phase 9 return-type variable binding** — call-result variable binding across 11 languages (#379)
- `extractPendingAssignment` in per-language type extractors captures `let x = getUser()` patterns
- Unified fixpoint loop resolves variable types from function return types after initial walk
- Field access on call-result variables: `user.name` resolves `name` via return type's class definition
- Method-call-result chaining: `user.getProfile().bio` resolves through intermediate return types
- 22 new test fixtures covering call-result and method-chain binding across all supported languages
- Integration tests added for all 10 language resolver suites
- **ACCESSES edge type** with read/write field access tracking (#372)
- **Python `enumerate()` for-loop support** with nested tuple patterns (#356)
- **MCP tool/resource descriptions** updated to reflect Phase 9 ACCESSES edge semantics and `declaredType` property
### Fixed
- **mcp**: server crashes under parallel tool calls (#326, #349)
- **parsing**: undefined error on languages missing from call routers (#364)
- **web**: add missing Kotlin entries to `Record<SupportedLanguages>` maps
- **rust**: `await` expression unwrapping in `extractPendingAssignment` for async call-result binding
- **tests**: update property edge and write access expectations across multiple language tests
- **docs**: corrected stale "single-pass" claims in type-resolution-system.md to reflect walk+fixpoint architecture
### Changed
- **Upgrade `@ladybugdb/core` to 0.15.2** and remove segfault workarounds (#374)
- **type-resolution-roadmap.md** overhauled — completed phases condensed to summaries, Phases 10–14 added with full engineering specs
## [1.4.6] - 2026-03-18
### Added
- **Phase 7 type resolution** — return-aware loop inference for call-expression iterables (#341)
- `ReturnTypeLookup` interface with `lookupReturnType` / `lookupRawReturnType` split
- `ForLoopExtractorContext` context object replacing positional `(node, env)` signature
- Call-expression iterable resolution across 8 languages (TS/JS, Java, Kotlin, C#, Go, Rust, Python, PHP)
- PHP `$this->property` foreach via `@var` class property scan (Strategy C)
- PHP `function_call_expression` and `member_call_expression` foreach paths
- `extractElementTypeFromString` as canonical raw-string container unwrapper in `shared.ts`
- `extractReturnTypeName` deduplicated from `call-processor.ts` into `shared.ts` (137 lines removed)
- `SKIP_SUBTREE_TYPES` performance optimization with documented `template_string` exclusion
- `pendingCallResults` infrastructure (dormant — Phase 9 work)
### Fixed
- **impact**: return structured error + partial results instead of crashing (#345)
- **impact**: add `HAS_METHOD` and `OVERRIDES` to `VALID_RELATION_TYPES` (#350)
- **cli**: write tool output to stdout via fd 1 instead of stderr (#346)
- **postinstall**: add permission fix for CLI and hook scripts (#348)
- **workflow**: use prefixed temporary branch name for fork PRs to prevent overwriting real branches
- **test**: add `--repo` to CLI e2e tool tests for multi-repo environment
- **php**: add `declaration_list` type guard on `findClassPropertyElementType` fallback
- **docs**: correct `pendingCallResults` description in roadmap and system docs
### Chore
- Add `.worktrees/` to `.gitignore`
## [1.4.5] - 2026-03-17
### Added
- **Ruby language support** for CLI and web (#111)
- **TypeEnvironment API** with constructor inference, self/this/super resolution (#274)
- **Return type inference** with doc-comment parsing (JSDoc, PHPDoc, YARD) and per-language type extractors (#284)
- **Phase 4 type resolution** — nullable unwrapping, for-loop typing, assignment chain propagation (#310)
- **Phase 5 type resolution** — chained calls, pattern matching, class-as-receiver (#315)
- **Phase 6 type resolution** — for-loop Tier 1c, pattern matching, container descriptors, 10-language coverage (#318)
- Container descriptor table for generic type argument resolution (Map keys vs values)
- Method-aware for-loop extractors with integration tests for all languages
- Recursive pattern binding (C# `is` patterns, Kotlin `when/is` smart casts)
- Class field declaration unwrapping for C#/Java
- PHP `$this->property` foreach member access
- C++ pointer dereference range-for
- Java `this.data.values()` field access patterns
- Position-indexed when/is bindings for branch-local narrowing
- **Type resolution system documentation** with architecture guide and roadmap
- `.gitignore` and `.gitnexusignore` support during file discovery (#231)
- Codex MCP configuration documentation in README (#236)
- `skipGraphPhases` pipeline option to skip MRO/community/process phases for faster test runs
- `hookTimeout: 120000` in vitest config for CI beforeAll hooks
### Changed
- **Migrated from KuzuDB to LadybugDB v0.15** (#275)
- Dynamically discover and install agent skills in CLI (#270)
### Performance
- Worker pool threshold — skip worker creation for small repos (<15 files or <512KB total)
- AST walk pruning via `SKIP_SUBTREE_TYPES` for leaf-only nodes (string, comment, number literals)
- Pre-computed `interestingNodeTypes` set — single Set.has() replaces 3 checks per AST node
- `fastStripNullable` — skip full nullable parsing for simple identifiers (90%+ case)
- Replace `.children?.find()` with manual for loops in `extractFunctionName` to eliminate array allocations
### Fixed
- Same-directory Python import resolution (#328)
- Ruby method-level call resolution, HAS_METHOD edges, and dispatch table (#278)
- C++ fixture file casing for case-sensitive CI
- Template string incorrectly included in AST pruning set (contains interpolated expressions)
## [1.4.0] - Previous release

View file

@ -2,7 +2,7 @@
**Graph-powered code intelligence for AI agents.** Index any codebase into a knowledge graph, then query it via MCP or CLI.
Works with **Cursor**, **Claude Code**, **Windsurf**, **Cline**, **OpenCode**, and any MCP-compatible tool.
Works with **Cursor**, **Claude Code**, **Codex**, **Windsurf**, **Cline**, **OpenCode**, and any MCP-compatible tool.
[![npm version](https://img.shields.io/npm/v/gitnexus.svg)](https://www.npmjs.com/package/gitnexus)
[![License: PolyForm Noncommercial](https://img.shields.io/badge/License-PolyForm%20Noncommercial-blue.svg)](https://polyformproject.org/licenses/noncommercial/1.0.0/)
@ -34,6 +34,7 @@ To configure MCP for your editor, run `npx gitnexus setup` once — or set it up
|--------|-----|--------|---------------------|---------|
| **Claude Code** | Yes | Yes | Yes (PreToolUse) | **Full** |
| **Cursor** | Yes | Yes | — | MCP + Skills |
| **Codex** | Yes | Yes | — | MCP + Skills |
| **Windsurf** | Yes | — | — | MCP |
| **OpenCode** | Yes | Yes | — | MCP + Skills |
@ -55,6 +56,12 @@ If you prefer to configure manually instead of using `gitnexus setup`:
claude mcp add gitnexus -- npx -y gitnexus@latest mcp
```
### Codex (full support — MCP + skills)
```bash
codex mcp add gitnexus -- npx -y gitnexus@latest mcp
```
### Cursor / Windsurf
Add to `~/.cursor/mcp.json` (global — works for all projects):

0
gitnexus/hooks/claude/gitnexus-hook.cjs Normal file → Executable file
View file

0
gitnexus/hooks/claude/pre-tool-use.sh Normal file → Executable file
View file

0
gitnexus/hooks/claude/session-start.sh Normal file → Executable file
View file

View file

@ -1,17 +1,17 @@
{
"name": "gitnexus",
"version": "1.4.0",
"version": "1.4.7",
"lockfileVersion": 3,
"requires": true,
"packages": {
"": {
"name": "gitnexus",
"version": "1.4.0",
"version": "1.4.7",
"hasInstallScript": true,
"license": "PolyForm-Noncommercial-1.0.0",
"dependencies": {
"@huggingface/transformers": "^3.0.0",
"@ladybugdb/core": "^0.15.1",
"@ladybugdb/core": "^0.15.2",
"@modelcontextprotocol/sdk": "^1.0.0",
"cli-progress": "^3.12.0",
"commander": "^12.0.0",
@ -21,6 +21,7 @@
"graphology": "^0.25.4",
"graphology-indices": "^0.17.0",
"graphology-utils": "^2.3.0",
"ignore": "^7.0.5",
"lru-cache": "^11.0.0",
"mnemonist": "^0.39.0",
"onnxruntime-node": "^1.24.0",
@ -606,29 +607,6 @@
"sharp": "^0.34.1"
}
},
"node_modules/@huggingface/transformers/node_modules/onnxruntime-common": {
"version": "1.21.0",
"resolved": "https://registry.npmjs.org/onnxruntime-common/-/onnxruntime-common-1.21.0.tgz",
"integrity": "sha512-Q632iLLrtCAVOTO65dh2+mNbQir/QNTVBG3h/QdZBpns7mZ0RYbLRBgGABPbpU9351AgYy7SJf1WaeVwMrBFPQ==",
"license": "MIT"
},
"node_modules/@huggingface/transformers/node_modules/onnxruntime-node": {
"version": "1.21.0",
"resolved": "https://registry.npmjs.org/onnxruntime-node/-/onnxruntime-node-1.21.0.tgz",
"integrity": "sha512-NeaCX6WW2L8cRCSqy3bInlo5ojjQqu2fD3D+9W5qb5irwxhEyWKXeH2vZ8W9r6VxaMPUan+4/7NDwZMtouZxEw==",
"hasInstallScript": true,
"license": "MIT",
"os": [
"win32",
"darwin",
"linux"
],
"dependencies": {
"global-agent": "^3.0.0",
"onnxruntime-common": "1.21.0",
"tar": "^7.0.1"
}
},
"node_modules/@img/colour": {
"version": "1.0.0",
"resolved": "https://registry.npmjs.org/@img/colour/-/colour-1.0.0.tgz",
@ -1173,14 +1151,81 @@
}
},
"node_modules/@ladybugdb/core": {
"version": "0.15.1",
"resolved": "https://registry.npmjs.org/@ladybugdb/core/-/core-0.15.1.tgz",
"integrity": "sha512-a+jhzIlS2+57Y2YWXlta7Dq5A3577dQ8YO7DzPCFZxozeiGIZn0K9v0ROO+ws4PW9BwuQYI5BXQxTEtaa1Otlg==",
"version": "0.15.2",
"resolved": "https://registry.npmjs.org/@ladybugdb/core/-/core-0.15.2.tgz",
"integrity": "sha512-DpseEj9CM/QTV0z+rvBk6nB2mOoG4GVhnKKLiXChGTVddgpH6R/Pv2YiDZB7rUIDnFpJxVQNbQaYEkZ7i1h1KA==",
"hasInstallScript": true,
"license": "MIT",
"dependencies": {
"cmake-js": "^8.0.0",
"node-addon-api": "^6.0.0"
},
"optionalDependencies": {
"@ladybugdb/core-darwin-arm64": "0.15.2",
"@ladybugdb/core-linux-arm64": "0.15.2",
"@ladybugdb/core-linux-x64": "0.15.2",
"@ladybugdb/core-win32-x64": "0.15.2"
}
},
"node_modules/@ladybugdb/core-darwin-arm64": {
"version": "0.15.2",
"resolved": "https://registry.npmjs.org/@ladybugdb/core-darwin-arm64/-/core-darwin-arm64-0.15.2.tgz",
"integrity": "sha512-ifLyUTPzlh2zR1IqkUT5AfldX+X4zfWBzwakmGTgMPxyrEiRNDwUKfnNxHeLQ/TJTOS/nfzYxxLLt5CZf2/FhA==",
"cpu": [
"arm64"
],
"license": "MIT",
"optional": true,
"os": [
"darwin"
]
},
"node_modules/@ladybugdb/core-linux-arm64": {
"version": "0.15.2",
"resolved": "https://registry.npmjs.org/@ladybugdb/core-linux-arm64/-/core-linux-arm64-0.15.2.tgz",
"integrity": "sha512-9537UbHOiuSr/BaTfjcoBsHxEKF4uEXWyXEjm/AQCGXQFocX3nQDVNDYJzuDYjKZ51oJRJ0oSuesAStOCwjolA==",
"cpu": [
"arm64"
],
"license": "MIT",
"optional": true,
"os": [
"linux"
]
},
"node_modules/@ladybugdb/core-linux-x64": {
"version": "0.15.2",
"resolved": "https://registry.npmjs.org/@ladybugdb/core-linux-x64/-/core-linux-x64-0.15.2.tgz",
"integrity": "sha512-1+xLoapjbMQzDHxcPpMPt8Suuvms3nhOIZFNGPDcWz90NwEmLAjWNFQZZHeg8DRz0vG2j8UY292bvGORVcxs8g==",
"cpu": [
"x64"
],
"license": "MIT",
"optional": true,
"os": [
"linux"
]
},
"node_modules/@ladybugdb/core-win32-x64": {
"version": "0.15.2",
"resolved": "https://registry.npmjs.org/@ladybugdb/core-win32-x64/-/core-win32-x64-0.15.2.tgz",
"integrity": "sha512-+LIJVKBNSrf2bGruJO4l0ihrLKZkv5+lNitK8xc3T7gC1bcc+FaYtRMvlgZP6Qh2rEHAjqfbaSKVrdw0M2EXTw==",
"cpu": [
"x64"
],
"license": "MIT",
"optional": true,
"os": [
"win32"
]
},
"node_modules/@ladybugdb/core/node_modules/chownr": {
"version": "3.0.0",
"resolved": "https://registry.npmjs.org/chownr/-/chownr-3.0.0.tgz",
"integrity": "sha512-+IxzY9BZOQd/XuYPRmrvEVjF/nqj5kgT4kEq7VofrDoM1MxoRjEWkrCC3EtLi59TVawxTAn+orJwFQcrqEN1+g==",
"license": "BlueOak-1.0.0",
"engines": {
"node": ">=18"
}
},
"node_modules/@ladybugdb/core/node_modules/cmake-js": {
@ -1232,12 +1277,40 @@
"node": ">=20"
}
},
"node_modules/@ladybugdb/core/node_modules/minizlib": {
"version": "3.1.0",
"resolved": "https://registry.npmjs.org/minizlib/-/minizlib-3.1.0.tgz",
"integrity": "sha512-KZxYo1BUkWD2TVFLr0MQoM8vUUigWD3LlD83a/75BqC+4qE0Hb1Vo5v1FgcfaNXvfXzr+5EhQ6ing/CaBijTlw==",
"license": "MIT",
"dependencies": {
"minipass": "^7.1.2"
},
"engines": {
"node": ">= 18"
}
},
"node_modules/@ladybugdb/core/node_modules/ms": {
"version": "2.1.3",
"resolved": "https://registry.npmjs.org/ms/-/ms-2.1.3.tgz",
"integrity": "sha512-6FlzubTLZG3J2a/NVCAleEhjzq5oxgHyaCU9yYXvcLsvoVaHJq/s5xXI6/XXP6tz7R9xAOtHnSO/tXtF3WRTlA==",
"license": "MIT"
},
"node_modules/@ladybugdb/core/node_modules/tar": {
"version": "7.5.11",
"resolved": "https://registry.npmjs.org/tar/-/tar-7.5.11.tgz",
"integrity": "sha512-ChjMH33/KetonMTAtpYdgUFr0tbz69Fp2v7zWxQfYZX4g5ZN2nOBXm1R2xyA+lMIKrLKIoKAwFj93jE/avX9cQ==",
"license": "BlueOak-1.0.0",
"dependencies": {
"@isaacs/fs-minipass": "^4.0.0",
"chownr": "^3.0.0",
"minipass": "^7.1.2",
"minizlib": "^3.1.0",
"yallist": "^5.0.0"
},
"engines": {
"node": ">=18"
}
},
"node_modules/@ladybugdb/core/node_modules/which": {
"version": "6.0.1",
"resolved": "https://registry.npmjs.org/which/-/which-6.0.1.tgz",
@ -1253,6 +1326,15 @@
"node": "^20.17.0 || >=22.9.0"
}
},
"node_modules/@ladybugdb/core/node_modules/yallist": {
"version": "5.0.0",
"resolved": "https://registry.npmjs.org/yallist/-/yallist-5.0.0.tgz",
"integrity": "sha512-YgvUTfwqyc7UXVMrB+SImsVYSmTS8X/tSrtdNZMImM+n7+QTriRXyXim0mBrTXNeqzVF0KWGgHPeiyViFFrNDw==",
"license": "BlueOak-1.0.0",
"engines": {
"node": ">=18"
}
},
"node_modules/@modelcontextprotocol/sdk": {
"version": "1.25.3",
"resolved": "https://registry.npmjs.org/@modelcontextprotocol/sdk/-/sdk-1.25.3.tgz",
@ -2510,15 +2592,6 @@
"node": ">=18"
}
},
"node_modules/chownr": {
"version": "3.0.0",
"resolved": "https://registry.npmjs.org/chownr/-/chownr-3.0.0.tgz",
"integrity": "sha512-+IxzY9BZOQd/XuYPRmrvEVjF/nqj5kgT4kEq7VofrDoM1MxoRjEWkrCC3EtLi59TVawxTAn+orJwFQcrqEN1+g==",
"license": "BlueOak-1.0.0",
"engines": {
"node": ">=18"
}
},
"node_modules/cli-progress": {
"version": "3.12.0",
"resolved": "https://registry.npmjs.org/cli-progress/-/cli-progress-3.12.0.tgz",
@ -3524,6 +3597,15 @@
"node": ">=0.10.0"
}
},
"node_modules/ignore": {
"version": "7.0.5",
"resolved": "https://registry.npmjs.org/ignore/-/ignore-7.0.5.tgz",
"integrity": "sha512-Hs59xBNfUIunMFgWAbGX5cq6893IbWg4KnrjbYwX3tx0ztorVgTDA6B2sxf8ejHJ4wz8BqGUMYlnzNBer5NvGg==",
"license": "MIT",
"engines": {
"node": ">= 4"
}
},
"node_modules/inherits": {
"version": "2.0.4",
"resolved": "https://registry.npmjs.org/inherits/-/inherits-2.0.4.tgz",
@ -3833,18 +3915,6 @@
"node": ">=16 || 14 >=14.17"
}
},
"node_modules/minizlib": {
"version": "3.1.0",
"resolved": "https://registry.npmjs.org/minizlib/-/minizlib-3.1.0.tgz",
"integrity": "sha512-KZxYo1BUkWD2TVFLr0MQoM8vUUigWD3LlD83a/75BqC+4qE0Hb1Vo5v1FgcfaNXvfXzr+5EhQ6ing/CaBijTlw==",
"license": "MIT",
"dependencies": {
"minipass": "^7.1.2"
},
"engines": {
"node": ">= 18"
}
},
"node_modules/mnemonist": {
"version": "0.39.8",
"resolved": "https://registry.npmjs.org/mnemonist/-/mnemonist-0.39.8.tgz",
@ -4817,22 +4887,6 @@
"node": ">=8"
}
},
"node_modules/tar": {
"version": "7.5.11",
"resolved": "https://registry.npmjs.org/tar/-/tar-7.5.11.tgz",
"integrity": "sha512-ChjMH33/KetonMTAtpYdgUFr0tbz69Fp2v7zWxQfYZX4g5ZN2nOBXm1R2xyA+lMIKrLKIoKAwFj93jE/avX9cQ==",
"license": "BlueOak-1.0.0",
"dependencies": {
"@isaacs/fs-minipass": "^4.0.0",
"chownr": "^3.0.0",
"minipass": "^7.1.2",
"minizlib": "^3.1.0",
"yallist": "^5.0.0"
},
"engines": {
"node": ">=18"
}
},
"node_modules/tinybench": {
"version": "2.9.0",
"resolved": "https://registry.npmjs.org/tinybench/-/tinybench-2.9.0.tgz",
@ -5699,15 +5753,6 @@
"node": ">=10"
}
},
"node_modules/yallist": {
"version": "5.0.0",
"resolved": "https://registry.npmjs.org/yallist/-/yallist-5.0.0.tgz",
"integrity": "sha512-YgvUTfwqyc7UXVMrB+SImsVYSmTS8X/tSrtdNZMImM+n7+QTriRXyXim0mBrTXNeqzVF0KWGgHPeiyViFFrNDw==",
"license": "BlueOak-1.0.0",
"engines": {
"node": ">=18"
}
},
"node_modules/yargs": {
"version": "17.7.2",
"resolved": "https://registry.npmjs.org/yargs/-/yargs-17.7.2.tgz",

View file

@ -1,6 +1,6 @@
{
"name": "gitnexus",
"version": "1.4.0",
"version": "1.4.7",
"description": "Graph-powered code intelligence for AI agents. Index any codebase, query via MCP or CLI.",
"author": "Abhigyan Patwari",
"license": "PolyForm-Noncommercial-1.0.0",
@ -20,6 +20,7 @@
"knowledge-graph",
"cursor",
"claude",
"codex",
"ai-agent",
"gitnexus",
"static-analysis",
@ -39,13 +40,14 @@
"scripts": {
"build": "tsc",
"dev": "tsx watch src/cli/index.ts",
"test": "vitest run test/unit",
"test": "vitest run",
"test:unit": "vitest run test/unit",
"test:integration": "vitest run test/integration",
"test:all": "vitest run",
"test:watch": "vitest",
"test:coverage": "vitest run --coverage",
"prepare": "npm run build",
"postinstall": "node scripts/patch-tree-sitter-swift.cjs"
"postinstall": "node scripts/patch-tree-sitter-swift.cjs",
"prepack": "npm run build && chmod +x dist/cli/index.js"
},
"dependencies": {
"@huggingface/transformers": "^3.0.0",
@ -59,7 +61,8 @@
"graphology": "^0.25.4",
"graphology-indices": "^0.17.0",
"graphology-utils": "^2.3.0",
"@ladybugdb/core": "^0.15.1",
"@ladybugdb/core": "^0.15.2",
"ignore": "^7.0.5",
"lru-cache": "^11.0.0",
"mnemonist": "^0.39.0",
"pandemonium": "^2.4.0",
@ -92,6 +95,11 @@
"typescript": "^5.4.5",
"vitest": "^4.0.18"
},
"overrides": {
"@huggingface/transformers": {
"onnxruntime-node": "$onnxruntime-node"
}
},
"engines": {
"node": ">=18.0.0"
}

0
gitnexus/scripts/patch-tree-sitter-swift.cjs Normal file → Executable file
View file

View file

@ -2,7 +2,7 @@
* AI Context Generator
*
* Creates AGENTS.md and CLAUDE.md with full inline GitNexus context.
* AGENTS.md is the standard read by Cursor, Windsurf, OpenCode, Cline, etc.
* AGENTS.md is the standard read by Cursor, Windsurf, OpenCode, Codex, Cline, etc.
* CLAUDE.md is for Claude Code which only reads that file.
*/
@ -308,4 +308,3 @@ export async function generateAIContextFiles(
return { files: createdFiles };
}

View file

@ -117,6 +117,10 @@ export const analyzeCommand = async (
return;
}
if (process.env.GITNEXUS_NO_GITIGNORE) {
console.log(' GITNEXUS_NO_GITIGNORE is set — skipping .gitignore (still reading .gitnexusignore)\n');
}
// Single progress bar for entire pipeline
const bar = new cliProgress.SingleBar({
format: ' {bar} {percentage}% | {phase}',

View file

@ -25,6 +25,7 @@
*/
import http from 'http';
import { writeSync } from 'node:fs';
import { LocalBackend } from '../mcp/local/local-backend.js';
export interface EvalServerOptions {
@ -142,7 +143,10 @@ export function formatContextResult(result: any): string {
}
export function formatImpactResult(result: any): string {
if (result.error) return `Error: ${result.error}`;
if (result.error) {
const suggestion = result.suggestion ? `\nSuggestion: ${result.suggestion}` : '';
return `Error: ${result.error}${suggestion}`;
}
const target = result.target;
const direction = result.direction;
@ -155,7 +159,11 @@ export function formatImpactResult(result: any): string {
const lines: string[] = [];
const dirLabel = direction === 'upstream' ? 'depends on this (will break if changed)' : 'this depends on';
lines.push(`Blast radius for ${target?.kind || ''} ${target?.name} (${direction}): ${total} symbol(s) ${dirLabel}\n`);
lines.push(`Blast radius for ${target?.kind || ''} ${target?.name} (${direction}): ${total} symbol(s) ${dirLabel}`);
if (result.partial) {
lines.push('⚠️ Partial results — graph traversal was interrupted. Deeper impacts may exist.');
}
lines.push('');
const depthLabels: Record<number, string> = {
1: 'WILL BREAK (direct)',
@ -401,9 +409,10 @@ export async function evalServerCommand(options?: EvalServerOptions): Promise<vo
console.error(` Auto-shutdown after ${idleTimeoutSec}s idle`);
}
try {
process.stdout.write(`GITNEXUS_EVAL_SERVER_READY:${port}\n`);
// Use fd 1 directly — LadybugDB captures process.stdout (#324)
writeSync(1, `GITNEXUS_EVAL_SERVER_READY:${port}\n`);
} catch {
// stdout may not be available
// stdout may not be available (e.g., broken pipe)
}
});

View file

@ -18,9 +18,10 @@ program
program
.command('setup')
.description('One-time setup: configure MCP for Cursor, Claude Code, OpenCode')
.description('One-time setup: configure MCP for Cursor, Claude Code, OpenCode, Codex')
.action(createLazyAction(() => import('./setup.js'), 'setupCommand'));
program
.command('analyze [path]')
.description('Index a repository (full analysis)')
@ -28,6 +29,7 @@ program
.option('--embeddings', 'Enable embedding generation for semantic search (off by default)')
.option('--skills', 'Generate repo-specific skill files from detected communities')
.option('-v, --verbose', 'Enable verbose ingestion warnings (default: false)')
.addHelpText('after', '\nEnvironment variables:\n GITNEXUS_NO_GITIGNORE=1 Skip .gitignore parsing (still reads .gitnexusignore)')
.action(createLazyAction(() => import('./analyze.js'), 'analyzeCommand'));
program

View file

@ -9,12 +9,15 @@
import fs from 'fs/promises';
import path from 'path';
import os from 'os';
import { execFile } from 'child_process';
import { promisify } from 'util';
import { fileURLToPath } from 'url';
import { glob } from 'glob';
import { getGlobalDir } from '../storage/repo-manager.js';
const __filename = fileURLToPath(import.meta.url);
const __dirname = path.dirname(__filename);
const execFileAsync = promisify(execFile);
interface SetupResult {
configured: string[];
@ -239,12 +242,75 @@ async function setupOpenCode(result: SetupResult): Promise<void> {
}
}
/**
* Build a TOML section for Codex MCP config (~/.codex/config.toml).
*/
function getCodexMcpTomlSection(): string {
const entry = getMcpEntry();
const command = JSON.stringify(entry.command);
const args = `[${entry.args.map(arg => JSON.stringify(arg)).join(', ')}]`;
return `[mcp_servers.gitnexus]\ncommand = ${command}\nargs = ${args}\n`;
}
/**
* Append GitNexus MCP server config to Codex's config.toml if missing.
*/
async function upsertCodexConfigToml(configPath: string): Promise<void> {
let existing = '';
try {
existing = await fs.readFile(configPath, 'utf-8');
} catch {
existing = '';
}
if (existing.includes('[mcp_servers.gitnexus]')) {
return;
}
const section = getCodexMcpTomlSection();
const nextContent = existing.trim().length > 0
? `${existing.trimEnd()}\n\n${section}`
: section;
await fs.mkdir(path.dirname(configPath), { recursive: true });
await fs.writeFile(configPath, `${nextContent.trimEnd()}\n`, 'utf-8');
}
async function setupCodex(result: SetupResult): Promise<void> {
const codexDir = path.join(os.homedir(), '.codex');
if (!(await dirExists(codexDir))) {
result.skipped.push('Codex (not installed)');
return;
}
try {
const entry = getMcpEntry();
await execFileAsync(
'codex',
['mcp', 'add', 'gitnexus', '--', entry.command, ...entry.args],
{ shell: process.platform === 'win32' }
);
result.configured.push('Codex');
return;
} catch {
// Fallback for environments where `codex` binary isn't on PATH.
}
try {
const configPath = path.join(codexDir, 'config.toml');
await upsertCodexConfigToml(configPath);
result.configured.push('Codex (MCP added to ~/.codex/config.toml)');
} catch (err: any) {
result.errors.push(`Codex: ${err.message}`);
}
}
// ─── Skill Installation ───────────────────────────────────────────
/**
* Install GitNexus skills to a target directory.
* Each skill is installed as {targetDir}/gitnexus-{skillName}/SKILL.md
* following the Agent Skills standard (both Cursor and Claude Code).
* following the Agent Skills standard (Cursor, Claude Code, and Codex).
*
* Supports two source layouts:
* - Flat file: skills/{name}.md → copied as SKILL.md
@ -353,6 +419,24 @@ async function installOpenCodeSkills(result: SetupResult): Promise<void> {
}
}
/**
* Install global Codex skills to ~/.agents/skills/gitnexus/
*/
async function installCodexSkills(result: SetupResult): Promise<void> {
const codexDir = path.join(os.homedir(), '.codex');
if (!(await dirExists(codexDir))) return;
const skillsDir = path.join(os.homedir(), '.agents', 'skills');
try {
const installed = await installSkillsTo(skillsDir);
if (installed.length > 0) {
result.configured.push(`Codex skills (${installed.length} skills → ~/.agents/skills/)`);
}
} catch (err: any) {
result.errors.push(`Codex skills: ${err.message}`);
}
}
// ─── Main command ──────────────────────────────────────────────────
export const setupCommand = async () => {
@ -375,12 +459,14 @@ export const setupCommand = async () => {
await setupCursor(result);
await setupClaudeCode(result);
await setupOpenCode(result);
await setupCodex(result);
// Install global skills for platforms that support them
await installClaudeCodeSkills(result);
await installClaudeCodeHooks(result);
await installCursorSkills(result);
await installOpenCodeSkills(result);
await installCodexSkills(result);
// Print results
if (result.configured.length > 0) {

View file

@ -10,10 +10,12 @@
* gitnexus impact --target "AuthService" --direction upstream
* gitnexus cypher "MATCH (n:Function) RETURN n.name LIMIT 10"
*
* Note: Output goes to stderr because LadybugDB's native module captures stdout
* at the OS level during init. This is consistent with augment.ts.
* Note: Output goes to stdout via fs.writeSync(fd 1), bypassing LadybugDB's
* native module which captures the Node.js process.stdout stream during init.
* See the output() function for details (#324).
*/
import { writeSync } from 'node:fs';
import { LocalBackend } from '../mcp/local/local-backend.js';
let _backend: LocalBackend | null = null;
@ -29,10 +31,29 @@ async function getBackend(): Promise<LocalBackend> {
return _backend;
}
/**
* Write tool output to stdout using low-level fd write.
*
* LadybugDB's native module captures Node.js process.stdout during init,
* but the underlying OS file descriptor 1 (stdout) remains intact.
* By using fs.writeSync(1, ...) we bypass the Node.js stream layer
* and write directly to the real stdout fd (#324).
*
* Falls back to stderr if the fd write fails (e.g., broken pipe).
*/
function output(data: any): void {
const text = typeof data === 'string' ? data : JSON.stringify(data, null, 2);
// stderr because LadybugDB captures stdout at OS level
process.stderr.write(text + '\n');
try {
writeSync(1, text + '\n');
} catch (err: any) {
if (err?.code === 'EPIPE') {
// Consumer closed the pipe (e.g., `gitnexus cypher ... | head -1`)
// Exit cleanly per Unix convention
process.exit(0);
}
// Fallback: stderr (previous behavior, works on all platforms)
process.stderr.write(text + '\n');
}
}
export async function queryCommand(queryText: string, options?: {
@ -92,15 +113,27 @@ export async function impactCommand(target: string, options?: {
process.exit(1);
}
const backend = await getBackend();
const result = await backend.callTool('impact', {
target,
direction: options?.direction || 'upstream',
maxDepth: options?.depth ? parseInt(options.depth) : undefined,
includeTests: options?.includeTests ?? false,
repo: options?.repo,
});
output(result);
try {
const backend = await getBackend();
const result = await backend.callTool('impact', {
target,
direction: options?.direction || 'upstream',
maxDepth: options?.depth ? parseInt(options.depth, 10) : undefined,
includeTests: options?.includeTests ?? false,
repo: options?.repo,
});
output(result);
} catch (err: unknown) {
// Belt-and-suspenders: catch infrastructure failures (getBackend, callTool transport)
// The backend's impact() already returns structured errors for graph query failures
output({
error: (err instanceof Error ? err.message : String(err)) || 'Impact analysis failed unexpectedly',
target: { name: target },
direction: options?.direction || 'upstream',
suggestion: 'Try reducing --depth or using gitnexus context <symbol> as a fallback',
});
process.exit(1);
}
}
export async function cypherCommand(query: string, options?: {

View file

@ -1,3 +1,8 @@
import ignore, { type Ignore } from 'ignore';
import fs from 'fs/promises';
import nodePath from 'path';
import type { Path } from 'path-scurry';
const DEFAULT_IGNORE_LIST = new Set([
// Version Control
'.git',
@ -186,6 +191,10 @@ const IGNORED_FILES = new Set([
// NOTE: Negation patterns in .gitnexusignore (e.g. `!vendor/`) cannot override
// entries in DEFAULT_IGNORE_LIST — this is intentional. The hardcoded list protects
// against indexing directories that are almost never source code (node_modules, .git, etc.).
// Users who need to include such directories should remove them from the hardcoded list.
export const shouldIgnorePath = (filePath: string): boolean => {
const normalizedPath = filePath.replace(/\\/g, '/');
const parts = normalizedPath.split('/');
@ -237,3 +246,86 @@ export const shouldIgnorePath = (filePath: string): boolean => {
return false;
}
/** Check if a directory name is in the hardcoded ignore list */
export const isHardcodedIgnoredDirectory = (name: string): boolean => {
return DEFAULT_IGNORE_LIST.has(name);
};
/**
* Load .gitignore and .gitnexusignore rules from the repo root.
* Returns an `ignore` instance with all patterns, or null if no files found.
*/
export interface IgnoreOptions {
/** Skip .gitignore parsing, only read .gitnexusignore. Defaults to GITNEXUS_NO_GITIGNORE env var. */
noGitignore?: boolean;
}
export const loadIgnoreRules = async (
repoPath: string,
options?: IgnoreOptions
): Promise<Ignore | null> => {
const ig = ignore();
let hasRules = false;
// Allow users to bypass .gitignore parsing (e.g. when .gitignore accidentally excludes source files)
const skipGitignore = options?.noGitignore ?? !!process.env.GITNEXUS_NO_GITIGNORE;
const filenames = skipGitignore
? ['.gitnexusignore']
: ['.gitignore', '.gitnexusignore'];
for (const filename of filenames) {
try {
const content = await fs.readFile(nodePath.join(repoPath, filename), 'utf-8');
ig.add(content);
hasRules = true;
} catch (err: unknown) {
const code = (err as NodeJS.ErrnoException).code;
if (code !== 'ENOENT') {
console.warn(` Warning: could not read ${filename}: ${(err as Error).message}`);
}
}
}
return hasRules ? ig : null;
};
/**
* Create a glob-compatible ignore filter combining:
* - .gitignore / .gitnexusignore patterns (via `ignore` package)
* - Hardcoded DEFAULT_IGNORE_LIST, IGNORED_EXTENSIONS, IGNORED_FILES
*
* Returns an IgnoreLike object for glob's `ignore` option,
* enabling directory-level pruning during traversal.
*/
export const createIgnoreFilter = async (repoPath: string, options?: IgnoreOptions) => {
const ig = await loadIgnoreRules(repoPath, options);
return {
ignored(p: Path): boolean {
// path-scurry's Path.relative() returns POSIX paths on all platforms,
// which is what the `ignore` package expects. No explicit normalization needed.
const rel = p.relative();
if (!rel) return false;
// Check .gitignore / .gitnexusignore patterns
if (ig && ig.ignores(rel)) return true;
// Fall back to hardcoded rules
return shouldIgnorePath(rel);
},
childrenIgnored(p: Path): boolean {
// Fast path: check directory name against hardcoded list.
// Note: dot-directories (.git, .vscode, etc.) are primarily excluded by
// glob's `dot: false` option in filesystem-walker.ts. This check is
// defense-in-depth — do not remove `dot: false` assuming this covers it.
if (DEFAULT_IGNORE_LIST.has(p.name)) return true;
// Check against .gitignore / .gitnexusignore patterns.
// Test both bare path and path with trailing slash to handle
// bare-name patterns (e.g. `local`) and dir-only patterns (e.g. `local/`).
if (ig) {
const rel = p.relative();
if (rel && (ig.ignores(rel) || ig.ignores(rel + '/'))) return true;
}
return false;
},
};
};

View file

@ -1,3 +1,33 @@
/**
* HOW TO ADD A NEW LANGUAGE:
*
* 1. Add the enum member below (e.g., Scala = 'scala')
* 2. Run `tsc --noEmit` — compiler errors guide you to every dispatch table
* 3. Use this checklist for each file:
*
* FILE | WHAT TO ADD | DEFAULT (simple languages)
* ----------------------------------|------------------------------------------|---------------------------
* tree-sitter-queries.ts | Query string + LANGUAGE_QUERIES entry | (required)
* export-detection.ts | ExportChecker function + table entry | (required)
* import-resolution.ts | Resolver in importResolvers | resolveStandard(...)
* import-resolution.ts | namedBindingExtractors entry | undefined
* call-routing.ts | callRouters entry | noRouting
* entry-point-scoring.ts | ENTRY_POINT_PATTERNS entry | []
* framework-detection.ts | AST_FRAMEWORK_PATTERNS entry | []
* type-extractors/<lang>.ts | New file + index.ts import | (required)
* resolvers/<lang>.ts | Resolver file (if non-standard) | (only if resolveStandard insufficient)
* named-binding-extraction.ts | Extractor (if named imports) | (only if language has named imports)
*
* 4. Also check these files for language-specific if-checks (no compile-time guard):
* - mro-processor.ts (MRO strategy selection)
* - heritage-processor.ts (extends/implements handling)
* - parse-worker.ts (AST edge cases)
* - parsing-processor.ts (node label normalization)
*
* 5. Add tree-sitter-<lang> to package.json dependencies
* 6. Add file extension mapping in utils.ts getLanguageFromFilename()
* 7. Run full test suite
*/
export enum SupportedLanguages {
JavaScript = 'javascript',
TypeScript = 'typescript',

View file

@ -22,17 +22,24 @@ import { createRequire } from 'module';
import { DEFAULT_EMBEDDING_CONFIG, type EmbeddingConfig, type ModelProgress } from './types.js';
/**
* Check whether the onnxruntime-node package ships the CUDA execution provider.
* Versions before 1.24.0 are CPU-only and do not include libonnxruntime_providers_cuda.so.
* Attempting CUDA on those versions causes an uncatchable native crash.
* Check whether the onnxruntime-node package that @huggingface/transformers
* will actually load at runtime ships the CUDA execution provider.
*
* Critical: we resolve from transformers' own module scope, NOT from ours.
* npm may install two copies — a top-level 1.24.x (our dep) and a nested
* 1.21.0 (transformers' pinned dep). The guard must inspect whichever copy
* transformers.js will dlopen, otherwise the check is meaningless.
*/
function hasOrtCudaProvider(): boolean {
try {
const require = createRequire(import.meta.url);
const ortPath = dirname(require.resolve('onnxruntime-node/package.json'));
// Check both napi-v6 (>=1.24) and napi-v3 (<=1.21) layouts
return existsSync(join(ortPath, 'bin', 'napi-v6', 'linux', 'x64', 'libonnxruntime_providers_cuda.so')) ||
existsSync(join(ortPath, 'bin', 'napi-v3', 'linux', 'x64', 'libonnxruntime_providers_cuda.so'));
// Resolve from @huggingface/transformers' scope so we find the same
// onnxruntime-node binary that transformers.js will use at runtime
const transformersDir = dirname(require.resolve('@huggingface/transformers/package.json'));
const ortRequire = createRequire(join(transformersDir, 'package.json'));
const ortPath = dirname(ortRequire.resolve('onnxruntime-node/package.json'));
const arch = process.arch; // x64, arm64, etc.
return existsSync(join(ortPath, 'bin', 'napi-v6', 'linux', arch, 'libonnxruntime_providers_cuda.so'));
} catch {
return false;
}

View file

@ -32,7 +32,8 @@ export type NodeLabel =
| 'Delegate'
| 'Annotation'
| 'Constructor'
| 'Template';
| 'Template'
| 'Section';
import { SupportedLanguages } from '../../config/supported-languages.js';
@ -65,6 +66,8 @@ export type NodeProperties = {
entryPointReason?: string,
// Method signature (for MRO disambiguation)
parameterCount?: number,
// Section-specific (markdown heading level, 1-6)
level?: number,
returnType?: string,
}
@ -80,6 +83,8 @@ export type RelationshipType =
| 'IMPLEMENTS'
| 'EXTENDS'
| 'HAS_METHOD'
| 'HAS_PROPERTY'
| 'ACCESSES'
| 'MEMBER_OF'
| 'STEP_IN_PROCESS'
@ -96,7 +101,7 @@ export interface GraphRelationship {
type: RelationshipType,
/** Confidence score 0-1 (1.0 = certain, lower = uncertain resolution) */
confidence: number,
/** Resolution reason: 'import-resolved', 'same-file', 'fuzzy-global', or empty for non-CALLS */
/** Semantics are edge-type-dependent: CALLS uses resolution tier, ACCESSES uses 'read'/'write', OVERRIDES uses MRO reason */
reason: string,
/** Step number for STEP_IN_PROCESS relationships (1-indexed) */
step?: number,

View file

@ -0,0 +1,710 @@
import type Parser from 'tree-sitter';
import { SupportedLanguages } from '../../config/supported-languages.js';
import type { NodeLabel } from '../graph/types.js';
import { generateId } from '../../lib/utils.js';
import { extractSimpleTypeName } from './type-extractors/shared.js';
/** Tree-sitter AST node. Re-exported for use across ingestion modules. */
export type SyntaxNode = Parser.SyntaxNode;
/**
* Ordered list of definition capture keys for tree-sitter query matches.
* Used to extract the definition node from a capture map.
*/
export const DEFINITION_CAPTURE_KEYS = [
'definition.function',
'definition.class',
'definition.interface',
'definition.method',
'definition.struct',
'definition.enum',
'definition.namespace',
'definition.module',
'definition.trait',
'definition.impl',
'definition.type',
'definition.const',
'definition.static',
'definition.typedef',
'definition.macro',
'definition.union',
'definition.property',
'definition.record',
'definition.delegate',
'definition.annotation',
'definition.constructor',
'definition.template',
] as const;
/** Extract the definition node from a tree-sitter query capture map. */
export const getDefinitionNodeFromCaptures = (captureMap: Record<string, any>): SyntaxNode | null => {
for (const key of DEFINITION_CAPTURE_KEYS) {
if (captureMap[key]) return captureMap[key];
}
return null;
};
/**
* Node types that represent function/method definitions across languages.
* Used to find the enclosing function for a call site.
*/
export const FUNCTION_NODE_TYPES = new Set([
// TypeScript/JavaScript
'function_declaration',
'arrow_function',
'function_expression',
'method_definition',
'generator_function_declaration',
// Python
'function_definition',
// Common async variants
'async_function_declaration',
'async_arrow_function',
// Java
'method_declaration',
'constructor_declaration',
// C/C++
// 'function_definition' already included above
// Go
// 'method_declaration' already included from Java
// C#
'local_function_statement',
// Rust
'function_item',
'impl_item', // Methods inside impl blocks
// PHP
'anonymous_function',
// Kotlin
'lambda_literal',
// Swift
'init_declaration',
'deinit_declaration',
// Ruby
'method', // def foo
'singleton_method', // def self.foo
]);
/**
* Node types for standard function declarations that need C/C++ declarator handling.
* Used by extractFunctionName to determine how to extract the function name.
*/
export const FUNCTION_DECLARATION_TYPES = new Set([
'function_declaration',
'function_definition',
'async_function_declaration',
'generator_function_declaration',
'function_item',
]);
/** AST node types that represent a class-like container (for HAS_METHOD edge extraction) */
export const CLASS_CONTAINER_TYPES = new Set([
'class_declaration', 'abstract_class_declaration',
'interface_declaration', 'struct_declaration', 'record_declaration',
'class_specifier', 'struct_specifier',
'impl_item', 'trait_item', 'struct_item', 'enum_item',
'class_definition',
'trait_declaration',
'protocol_declaration',
// Ruby
'class',
'module',
// Kotlin
'object_declaration',
'companion_object',
]);
export const CONTAINER_TYPE_TO_LABEL: Record<string, string> = {
class_declaration: 'Class',
abstract_class_declaration: 'Class',
interface_declaration: 'Interface',
struct_declaration: 'Struct',
struct_specifier: 'Struct',
class_specifier: 'Class',
class_definition: 'Class',
impl_item: 'Impl',
trait_item: 'Trait',
struct_item: 'Struct',
enum_item: 'Enum',
trait_declaration: 'Trait',
record_declaration: 'Record',
protocol_declaration: 'Interface',
class: 'Class',
module: 'Module',
object_declaration: 'Class',
companion_object: 'Class',
};
/** Check if a Kotlin function_declaration capture is inside a class_body (i.e., a method).
* Kotlin grammar uses function_declaration for both top-level functions and class methods.
* Returns true when the captured definition node has a class_body ancestor. */
export function isKotlinClassMethod(captureNode: { parent?: any } | null | undefined): boolean {
let ancestor = captureNode?.parent;
while (ancestor) {
if (ancestor.type === 'class_body') return true;
ancestor = ancestor.parent;
}
return false;
}
/**
* C/C++: check if a Function capture is inside a class/struct body.
* If true, the function is already captured by @definition.method and should be skipped
* to prevent double-indexing in globalIndex.
*/
export function isCppDuplicateClassFunction(
functionNode: { parent?: any } | null | undefined,
nodeLabel: string,
language: SupportedLanguages,
): boolean {
if (nodeLabel !== 'Function') return false;
if (language !== SupportedLanguages.CPlusPlus && language !== SupportedLanguages.C) return false;
let ancestor = functionNode?.parent;
while (ancestor) {
if (ancestor.type === 'class_specifier' || ancestor.type === 'struct_specifier') return true;
ancestor = ancestor.parent;
}
return false;
}
/**
* Determine the graph node label from a tree-sitter capture map.
* Handles language-specific reclassification (C/C++ duplicate skipping, Kotlin Method promotion).
* Returns null if the capture should be skipped (import, call, C/C++ duplicate, missing name).
*/
export function getLabelFromCaptures(
captureMap: Record<string, any>,
language: SupportedLanguages,
): NodeLabel | null {
if (captureMap['import'] || captureMap['call']) return null;
if (!captureMap['name'] && !captureMap['definition.constructor']) return null;
if (captureMap['definition.function']) {
if (isCppDuplicateClassFunction(captureMap['definition.function'], 'Function', language)) return null;
if (language === SupportedLanguages.Kotlin && isKotlinClassMethod(captureMap['definition.function'])) return 'Method';
return 'Function';
}
if (captureMap['definition.class']) return 'Class';
if (captureMap['definition.interface']) return 'Interface';
if (captureMap['definition.method']) return 'Method';
if (captureMap['definition.struct']) return 'Struct';
if (captureMap['definition.enum']) return 'Enum';
if (captureMap['definition.namespace']) return 'Namespace';
if (captureMap['definition.module']) return 'Module';
if (captureMap['definition.trait']) return 'Trait';
if (captureMap['definition.impl']) return 'Impl';
if (captureMap['definition.type']) return 'TypeAlias';
if (captureMap['definition.const']) return 'Const';
if (captureMap['definition.static']) return 'Static';
if (captureMap['definition.typedef']) return 'Typedef';
if (captureMap['definition.macro']) return 'Macro';
if (captureMap['definition.union']) return 'Union';
if (captureMap['definition.property']) return 'Property';
if (captureMap['definition.record']) return 'Record';
if (captureMap['definition.delegate']) return 'Delegate';
if (captureMap['definition.annotation']) return 'Annotation';
if (captureMap['definition.constructor']) return 'Constructor';
if (captureMap['definition.template']) return 'Template';
return 'CodeElement';
}
/** Walk up AST to find enclosing class/struct/interface/impl, return its generateId or null.
* For Go method_declaration nodes, extracts receiver type (e.g. `func (u *User) Save()` → User struct). */
export const findEnclosingClassId = (node: any, filePath: string): string | null => {
let current = node.parent;
while (current) {
// Go: method_declaration has a receiver parameter with the struct type
if (current.type === 'method_declaration') {
const receiver = current.childForFieldName?.('receiver');
if (receiver) {
// receiver is a parameter_list: (u *User) or (u User)
const paramDecl = receiver.namedChildren?.find?.((c: any) => c.type === 'parameter_declaration');
if (paramDecl) {
const typeNode = paramDecl.childForFieldName?.('type');
if (typeNode) {
// Unwrap pointer_type (*User → User)
const inner = typeNode.type === 'pointer_type' ? typeNode.firstNamedChild : typeNode;
if (inner && (inner.type === 'type_identifier' || inner.type === 'identifier')) {
return generateId('Struct', `${filePath}:${inner.text}`);
}
}
}
}
}
// Go: type_declaration wrapping a struct_type (type User struct { ... })
// field_declaration → field_declaration_list → struct_type → type_spec → type_declaration
if (current.type === 'type_declaration') {
const typeSpec = current.children?.find((c: any) => c.type === 'type_spec');
if (typeSpec) {
const typeBody = typeSpec.childForFieldName?.('type');
if (typeBody?.type === 'struct_type' || typeBody?.type === 'interface_type') {
const nameNode = typeSpec.childForFieldName?.('name');
if (nameNode) {
const label = typeBody.type === 'struct_type' ? 'Struct' : 'Interface';
return generateId(label, `${filePath}:${nameNode.text}`);
}
}
}
}
if (CLASS_CONTAINER_TYPES.has(current.type)) {
// Rust impl_item: for `impl Trait for Struct {}`, pick the type after `for`
if (current.type === 'impl_item') {
const children = current.children ?? [];
const forIdx = children.findIndex((c: any) => c.text === 'for');
if (forIdx !== -1) {
const nameNode = children.slice(forIdx + 1).find((c: any) =>
c.type === 'type_identifier' || c.type === 'identifier'
);
if (nameNode) {
return generateId('Impl', `${filePath}:${nameNode.text}`);
}
}
// Fall through: plain `impl Struct {}` — use first type_identifier below
}
const nameNode = current.childForFieldName?.('name')
?? current.children?.find((c: any) =>
c.type === 'type_identifier' || c.type === 'identifier' || c.type === 'name' || c.type === 'constant'
);
if (nameNode) {
const label = CONTAINER_TYPE_TO_LABEL[current.type] || 'Class';
return generateId(label, `${filePath}:${nameNode.text}`);
}
}
current = current.parent;
}
return null;
};
/**
* Find a child of `childType` within a sibling node of `siblingType`.
* Used for Kotlin AST traversal where visibility_modifier lives inside a modifiers sibling.
*/
export const findSiblingChild = (parent: any, siblingType: string, childType: string): any | null => {
for (let i = 0; i < parent.childCount; i++) {
const sibling = parent.child(i);
if (sibling?.type === siblingType) {
for (let j = 0; j < sibling.childCount; j++) {
const child = sibling.child(j);
if (child?.type === childType) return child;
}
}
}
return null;
};
/**
* Extract function name and label from a function_definition or similar AST node.
* Handles C/C++ qualified_identifier (ClassName::MethodName) and other language patterns.
*/
export const extractFunctionName = (node: SyntaxNode): { funcName: string | null; label: string } => {
let funcName: string | null = null;
let label = 'Function';
// Swift init/deinit
if (node.type === 'init_declaration' || node.type === 'deinit_declaration') {
return {
funcName: node.type === 'init_declaration' ? 'init' : 'deinit',
label: 'Constructor',
};
}
if (FUNCTION_DECLARATION_TYPES.has(node.type)) {
// C/C++: function_definition -> [pointer_declarator ->] function_declarator -> qualified_identifier/identifier
// Unwrap pointer_declarator / reference_declarator wrappers to reach function_declarator
let declarator = node.childForFieldName?.('declarator');
if (!declarator) {
for (let i = 0; i < node.childCount; i++) {
const c = node.child(i);
if (c?.type === 'function_declarator') { declarator = c; break; }
}
}
while (declarator && (declarator.type === 'pointer_declarator' || declarator.type === 'reference_declarator')) {
let nextDeclarator = declarator.childForFieldName?.('declarator');
if (!nextDeclarator) {
for (let i = 0; i < declarator.childCount; i++) {
const c = declarator.child(i);
if (c?.type === 'function_declarator' || c?.type === 'pointer_declarator' || c?.type === 'reference_declarator') { nextDeclarator = c; break; }
}
}
declarator = nextDeclarator;
}
if (declarator) {
let innerDeclarator = declarator.childForFieldName?.('declarator');
if (!innerDeclarator) {
for (let i = 0; i < declarator.childCount; i++) {
const c = declarator.child(i);
if (c?.type === 'qualified_identifier' || c?.type === 'identifier'
|| c?.type === 'field_identifier' || c?.type === 'parenthesized_declarator') { innerDeclarator = c; break; }
}
}
if (innerDeclarator?.type === 'qualified_identifier') {
let nameNode = innerDeclarator.childForFieldName?.('name');
if (!nameNode) {
for (let i = 0; i < innerDeclarator.childCount; i++) {
const c = innerDeclarator.child(i);
if (c?.type === 'identifier') { nameNode = c; break; }
}
}
if (nameNode?.text) {
funcName = nameNode.text;
label = 'Method';
}
} else if (innerDeclarator?.type === 'identifier' || innerDeclarator?.type === 'field_identifier') {
// field_identifier is used for method names inside C++ class bodies
funcName = innerDeclarator.text;
if (innerDeclarator.type === 'field_identifier') label = 'Method';
} else if (innerDeclarator?.type === 'parenthesized_declarator') {
let nestedId: SyntaxNode | null = null;
for (let i = 0; i < innerDeclarator.childCount; i++) {
const c = innerDeclarator.child(i);
if (c?.type === 'qualified_identifier' || c?.type === 'identifier') { nestedId = c; break; }
}
if (nestedId?.type === 'qualified_identifier') {
let nameNode = nestedId.childForFieldName?.('name');
if (!nameNode) {
for (let i = 0; i < nestedId.childCount; i++) {
const c = nestedId.child(i);
if (c?.type === 'identifier') { nameNode = c; break; }
}
}
if (nameNode?.text) {
funcName = nameNode.text;
label = 'Method';
}
} else if (nestedId?.type === 'identifier') {
funcName = nestedId.text;
}
}
}
// Fallback for other languages (Kotlin uses simple_identifier, Swift uses simple_identifier)
if (!funcName) {
let nameNode = node.childForFieldName?.('name');
if (!nameNode) {
for (let i = 0; i < node.childCount; i++) {
const c = node.child(i);
if (c?.type === 'identifier' || c?.type === 'property_identifier' || c?.type === 'simple_identifier') { nameNode = c; break; }
}
}
funcName = nameNode?.text;
// Kotlin: function_declaration inside a class_body is a method, not a top-level function.
// Must match the label assigned in parse-worker.ts for consistent generateId() output.
if (funcName && node.type === 'function_declaration' && isKotlinClassMethod(node)) {
label = 'Method';
}
}
} else if (node.type === 'impl_item') {
let funcItem: SyntaxNode | null = null;
for (let i = 0; i < node.childCount; i++) {
const c = node.child(i);
if (c?.type === 'function_item') { funcItem = c; break; }
}
if (funcItem) {
let nameNode = funcItem.childForFieldName?.('name');
if (!nameNode) {
for (let i = 0; i < funcItem.childCount; i++) {
const c = funcItem.child(i);
if (c?.type === 'identifier') { nameNode = c; break; }
}
}
funcName = nameNode?.text;
label = 'Method';
}
} else if (node.type === 'method_definition') {
let nameNode = node.childForFieldName?.('name');
if (!nameNode) {
for (let i = 0; i < node.childCount; i++) {
const c = node.child(i);
if (c?.type === 'property_identifier') { nameNode = c; break; }
}
}
funcName = nameNode?.text;
label = 'Method';
} else if (node.type === 'method_declaration' || node.type === 'constructor_declaration') {
let nameNode = node.childForFieldName?.('name');
if (!nameNode) {
for (let i = 0; i < node.childCount; i++) {
const c = node.child(i);
if (c?.type === 'identifier') { nameNode = c; break; }
}
}
funcName = nameNode?.text;
label = 'Method';
} else if (node.type === 'arrow_function' || node.type === 'function_expression') {
const parent = node.parent;
if (parent?.type === 'variable_declarator') {
let nameNode = parent.childForFieldName?.('name');
if (!nameNode) {
for (let i = 0; i < parent.childCount; i++) {
const c = parent.child(i);
if (c?.type === 'identifier') { nameNode = c; break; }
}
}
funcName = nameNode?.text;
}
} else if (node.type === 'method' || node.type === 'singleton_method') {
let nameNode = node.childForFieldName?.('name');
if (!nameNode) {
for (let i = 0; i < node.childCount; i++) {
const c = node.child(i);
if (c?.type === 'identifier') { nameNode = c; break; }
}
}
funcName = nameNode?.text;
label = 'Method';
}
return { funcName, label };
};
export interface MethodSignature {
parameterCount: number | undefined;
/** Number of required (non-optional, non-default) parameters.
* Only set when fewer than parameterCount — enables range-based arity filtering.
* undefined means all parameters are required (or metadata unavailable). */
requiredParameterCount: number | undefined;
/** Per-parameter type names extracted via extractSimpleTypeName.
* Only populated for languages with method overloading (Java, Kotlin, C#, C++).
* undefined (not []) when no types are extractable — avoids empty array allocations. */
parameterTypes: string[] | undefined;
returnType: string | undefined;
}
/** Argument list node types shared between extractMethodSignature and countCallArguments. */
export const CALL_ARGUMENT_LIST_TYPES = new Set([
'arguments',
'argument_list',
'value_arguments',
]);
/**
* Extract parameter count and return type text from an AST method/function node.
* Works across languages by looking for common AST patterns.
*/
export const extractMethodSignature = (node: SyntaxNode | null | undefined): MethodSignature => {
let parameterCount: number | undefined = 0;
let requiredCount = 0;
let returnType: string | undefined;
let isVariadic = false;
const paramTypes: string[] = [];
if (!node) return { parameterCount, requiredParameterCount: undefined, parameterTypes: undefined, returnType };
const paramListTypes = new Set([
'formal_parameters', 'parameters', 'parameter_list',
'function_parameters', 'method_parameters', 'function_value_parameters',
]);
// Node types that indicate variadic/rest parameters
const VARIADIC_PARAM_TYPES = new Set([
'variadic_parameter_declaration', // Go: ...string
'variadic_parameter', // Rust: extern "C" fn(...)
'spread_parameter', // Java: Object... args
'list_splat_pattern', // Python: *args
'dictionary_splat_pattern', // Python: **kwargs
]);
/** AST node types that represent parameters with default values. */
const OPTIONAL_PARAM_TYPES = new Set([
'optional_parameter', // TypeScript, Ruby: (x?: number), (x: number = 5), def f(x = 5)
'default_parameter', // Python: def f(x=5)
'typed_default_parameter', // Python: def f(x: int = 5)
'optional_parameter_declaration', // C++: void f(int x = 5)
]);
/** Check if a parameter node has a default value (handles Kotlin, C#, Swift, PHP
* where defaults are expressed as child nodes rather than distinct node types). */
const hasDefaultValue = (paramNode: SyntaxNode): boolean => {
if (OPTIONAL_PARAM_TYPES.has(paramNode.type)) return true;
// C#, Swift, PHP: check for '=' token or equals_value_clause child
for (let i = 0; i < paramNode.childCount; i++) {
const c = paramNode.child(i);
if (!c) continue;
if (c.type === '=' || c.type === 'equals_value_clause') return true;
}
// Kotlin: default values are siblings of the parameter node, not children.
// The AST is: parameter, =, <literal> — all at function_value_parameters level.
// Check if the immediately following sibling is '=' (default value separator).
const sib = paramNode.nextSibling;
if (sib && sib.type === '=') return true;
return false;
};
const findParameterList = (current: SyntaxNode): SyntaxNode | null => {
for (const child of current.children) {
if (paramListTypes.has(child.type)) return child;
}
for (const child of current.children) {
const nested = findParameterList(child);
if (nested) return nested;
}
return null;
};
const parameterList = (
paramListTypes.has(node.type) ? node // node itself IS the parameter list (e.g. C# primary constructors)
: node.childForFieldName?.('parameters')
?? findParameterList(node)
);
if (parameterList && paramListTypes.has(parameterList.type)) {
for (const param of parameterList.namedChildren) {
if (param.type === 'comment') continue;
if (param.text === 'self' || param.text === '&self' || param.text === '&mut self' ||
param.type === 'self_parameter') {
continue;
}
// Kotlin: default values are siblings of the parameter node inside
// function_value_parameters, so they appear as named children (e.g.
// string_literal, integer_literal, boolean_literal, call_expression).
// Skip any named child that isn't a parameter-like or modifier node.
if (param.type.endsWith('_literal') || param.type === 'call_expression'
|| param.type === 'navigation_expression' || param.type === 'prefix_expression'
|| param.type === 'parenthesized_expression') {
continue;
}
// Check for variadic parameter types
if (VARIADIC_PARAM_TYPES.has(param.type)) {
isVariadic = true;
continue;
}
// TypeScript/JavaScript: rest parameter — required_parameter containing rest_pattern
if (param.type === 'required_parameter' || param.type === 'optional_parameter') {
for (const child of param.children) {
if (child.type === 'rest_pattern') {
isVariadic = true;
break;
}
}
if (isVariadic) continue;
}
// Kotlin: vararg modifier on a regular parameter
if (param.type === 'parameter' || param.type === 'formal_parameter') {
const prev = param.previousSibling;
if (prev?.type === 'parameter_modifiers' && prev.text.includes('vararg')) {
isVariadic = true;
}
}
// Extract parameter type name for overload disambiguation.
// Works for Java (formal_parameter), Kotlin (parameter), C# (parameter),
// C++ (parameter_declaration). Uses childForFieldName('type') which is the
// standard tree-sitter field for typed parameters across these languages.
// Kotlin uses positional children instead of 'type' field — fall back to
// searching for user_type/nullable_type/predefined_type children.
const paramTypeNode = param.childForFieldName('type');
if (paramTypeNode) {
const typeName = extractSimpleTypeName(paramTypeNode);
paramTypes.push(typeName ?? 'unknown');
} else {
// Kotlin: parameter → [simple_identifier, user_type|nullable_type]
let found = false;
for (const child of param.namedChildren) {
if (child.type === 'user_type' || child.type === 'nullable_type'
|| child.type === 'type_identifier' || child.type === 'predefined_type') {
const typeName = extractSimpleTypeName(child);
paramTypes.push(typeName ?? 'unknown');
found = true;
break;
}
}
if (!found) paramTypes.push('unknown');
}
if (!hasDefaultValue(param)) requiredCount++;
parameterCount++;
}
// C/C++: bare `...` token in parameter list (not a named child — check all children)
if (!isVariadic) {
for (const child of parameterList.children) {
if (!child.isNamed && child.text === '...') {
isVariadic = true;
break;
}
}
}
}
// Return type extraction — language-specific field names
// Go: 'result' field is either a type_identifier or parameter_list (multi-return)
const goResult = node.childForFieldName?.('result');
if (goResult) {
if (goResult.type === 'parameter_list') {
// Multi-return: extract first parameter's type only (e.g. (*User, error) → *User)
const firstParam = goResult.firstNamedChild;
if (firstParam?.type === 'parameter_declaration') {
const typeNode = firstParam.childForFieldName('type');
if (typeNode) returnType = typeNode.text;
} else if (firstParam) {
// Unnamed return types: (string, error) — first child is a bare type node
returnType = firstParam.text;
}
} else {
returnType = goResult.text;
}
}
// Rust: 'return_type' field — the value IS the type node (e.g. primitive_type, type_identifier).
// Skip if the node is a type_annotation (TS/Python), which is handled by the generic loop below.
if (!returnType) {
const rustReturn = node.childForFieldName?.('return_type');
if (rustReturn && rustReturn.type !== 'type_annotation') {
returnType = rustReturn.text;
}
}
// C/C++: 'type' field on function_definition
if (!returnType) {
const cppType = node.childForFieldName?.('type');
if (cppType && cppType.text !== 'void') {
returnType = cppType.text;
}
}
// C#: 'returns' field on method_declaration
if (!returnType) {
const csReturn = node.childForFieldName?.('returns');
if (csReturn && csReturn.text !== 'void') {
returnType = csReturn.text;
}
}
// TS/Rust/Python/C#/Kotlin: type_annotation or return_type child
if (!returnType) {
for (const child of node.children) {
if (child.type === 'type_annotation' || child.type === 'return_type') {
const typeNode = child.children.find((c) => c.isNamed);
if (typeNode) returnType = typeNode.text;
}
}
}
// Kotlin: fun getUser(): User — return type is a bare user_type child of
// function_declaration. The Kotlin grammar does NOT wrap it in type_annotation
// or return_type; it appears as a direct child after function_value_parameters.
// Note: Kotlin uses function_value_parameters (not a field), so we find it by type.
if (!returnType) {
let paramsEnd = -1;
for (let i = 0; i < node.childCount; i++) {
const child = node.child(i);
if (!child) continue;
if (child.type === 'function_value_parameters' || child.type === 'value_parameters') {
paramsEnd = child.endIndex;
}
if (paramsEnd >= 0 && child.type === 'user_type' && child.startIndex > paramsEnd) {
returnType = child.text;
break;
}
}
}
if (isVariadic) parameterCount = undefined;
// Only include parameterTypes when at least one type was successfully extracted.
// Use undefined (not []) to avoid empty array allocations for untyped parameters.
const hasTypes = paramTypes.length > 0 && paramTypes.some(t => t !== 'unknown');
// Only set requiredParameterCount when it differs from total — saves memory on the common case.
const requiredParameterCount = (!isVariadic && requiredCount < (parameterCount ?? 0))
? requiredCount : undefined;
return { parameterCount, requiredParameterCount, parameterTypes: hasTypes ? paramTypes : undefined, returnType };
};

View file

@ -0,0 +1,539 @@
import type { SyntaxNode } from './ast-helpers.js';
import { CALL_ARGUMENT_LIST_TYPES } from './ast-helpers.js';
/** Node types representing call expressions across supported languages. */
export const CALL_EXPRESSION_TYPES = new Set([
'call_expression', // TS/JS/C/C++/Go/Rust
'method_invocation', // Java
'member_call_expression', // PHP
'nullsafe_member_call_expression', // PHP ?.
'call', // Python/Ruby
'invocation_expression', // C#
]);
/**
* Hard limit on chain depth to prevent runaway recursion.
* For `a.b().c().d()`, the chain has depth 2 (b and c before d).
*/
export const MAX_CHAIN_DEPTH = 3;
/**
* Count direct arguments for a call expression across common tree-sitter grammars.
* Returns undefined when the argument container cannot be located cheaply.
*/
export const countCallArguments = (callNode: SyntaxNode | null | undefined): number | undefined => {
if (!callNode) return undefined;
// Direct field or direct child (most languages)
let argsNode: SyntaxNode | null | undefined = callNode.childForFieldName('arguments')
?? callNode.children.find((child) => CALL_ARGUMENT_LIST_TYPES.has(child.type));
// Kotlin/Swift: call_expression → call_suffix → value_arguments
// Search one level deeper for languages that wrap arguments in a suffix node
if (!argsNode) {
for (const child of callNode.children) {
if (!child.isNamed) continue;
const nested = child.children.find((gc) => CALL_ARGUMENT_LIST_TYPES.has(gc.type));
if (nested) { argsNode = nested; break; }
}
}
if (!argsNode) return undefined;
let count = 0;
for (const child of argsNode.children) {
if (!child.isNamed) continue;
if (child.type === 'comment') continue;
count++;
}
return count;
};
// ── Call-form discrimination (Phase 1, Step D) ─────────────────────────
/**
* AST node types that indicate a member-access wrapper around the callee name.
* When nameNode.parent.type is one of these, the call is a member call.
*/
const MEMBER_ACCESS_NODE_TYPES = new Set([
'member_expression', // TS/JS: obj.method()
'attribute', // Python: obj.method()
'member_access_expression', // C#: obj.Method()
'field_expression', // Rust/C++: obj.method() / ptr->method()
'selector_expression', // Go: obj.Method()
'navigation_suffix', // Kotlin/Swift: obj.method() — nameNode sits inside navigation_suffix
'member_binding_expression', // C#: user?.Method() — null-conditional access
]);
/**
* Call node types that are inherently constructor invocations.
* Only includes patterns that the tree-sitter queries already capture as @call.
*/
const CONSTRUCTOR_CALL_NODE_TYPES = new Set([
'constructor_invocation', // Kotlin: Foo()
'new_expression', // TS/JS/C++: new Foo()
'object_creation_expression', // Java/C#/PHP: new Foo()
'implicit_object_creation_expression', // C# 9: User u = new(...)
'composite_literal', // Go: User{...}
'struct_expression', // Rust: User { ... }
]);
/**
* AST node types for scoped/qualified calls (e.g., Foo::new() in Rust, Foo::bar() in C++).
*/
const SCOPED_CALL_NODE_TYPES = new Set([
'scoped_identifier', // Rust: Foo::new()
'qualified_identifier', // C++: ns::func()
]);
type CallForm = 'free' | 'member' | 'constructor';
/**
* Infer whether a captured call site is a free call, member call, or constructor.
* Returns undefined if the form cannot be determined.
*
* Works by inspecting the AST structure between callNode (@call) and nameNode (@call.name).
* No tree-sitter query changes needed — the distinction is in the node types.
*/
export const inferCallForm = (
callNode: SyntaxNode,
nameNode: SyntaxNode,
): CallForm | undefined => {
// 1. Constructor: callNode itself is a constructor invocation (Kotlin)
if (CONSTRUCTOR_CALL_NODE_TYPES.has(callNode.type)) {
return 'constructor';
}
// 2. Member call: nameNode's parent is a member-access wrapper
const nameParent = nameNode.parent;
if (nameParent && MEMBER_ACCESS_NODE_TYPES.has(nameParent.type)) {
return 'member';
}
// 3. PHP: the callNode itself distinguishes member vs free calls
if (callNode.type === 'member_call_expression' || callNode.type === 'nullsafe_member_call_expression') {
return 'member';
}
if (callNode.type === 'scoped_call_expression') {
return 'member'; // static call Foo::bar()
}
// 4. Java method_invocation: member if it has an 'object' field
if (callNode.type === 'method_invocation' && callNode.childForFieldName('object')) {
return 'member';
}
// 4b. Ruby call with receiver: obj.method
if (callNode.type === 'call' && callNode.childForFieldName('receiver')) {
return 'member';
}
// 5. Scoped calls (Rust Foo::new(), C++ ns::func()): treat as free
// The receiver is a type, not an instance — handled differently in Phase 3
if (nameParent && SCOPED_CALL_NODE_TYPES.has(nameParent.type)) {
return 'free';
}
// 6. Default: if nameNode is a direct child of callNode, it's a free call
if (nameNode.parent === callNode || nameParent?.parent === callNode) {
return 'free';
}
return undefined;
};
/**
* Extract the receiver identifier for member calls.
* Only captures simple identifiers — returns undefined for complex expressions
* like getUser().save() or arr[0].method().
*/
const SIMPLE_RECEIVER_TYPES = new Set([
'identifier',
'simple_identifier',
'variable_name', // PHP $variable (tree-sitter-php)
'name', // PHP name node
'this', // TS/JS/Java/C# this.method()
'self', // Rust/Python self.method()
'super', // TS/JS/Java/Kotlin/Ruby super.method()
'super_expression', // Kotlin wraps super in super_expression
'base', // C# base.Method()
'parent', // PHP parent::method()
'constant', // Ruby CONSTANT.method() (uppercase identifiers)
]);
export const extractReceiverName = (
nameNode: SyntaxNode,
): string | undefined => {
const parent = nameNode.parent;
if (!parent) return undefined;
// PHP: member_call_expression / nullsafe_member_call_expression — receiver is on the callNode
// Java: method_invocation — receiver is the 'object' field on callNode
// For these, parent of nameNode is the call itself, so check the call's object field
const callNode = parent.parent ?? parent;
let receiver: SyntaxNode | null = null;
// Try standard field names used across grammars
receiver = parent.childForFieldName('object') // TS/JS member_expression, Python attribute, PHP, Java
?? parent.childForFieldName('value') // Rust field_expression
?? parent.childForFieldName('operand') // Go selector_expression
?? parent.childForFieldName('expression') // C# member_access_expression
?? parent.childForFieldName('argument'); // C++ field_expression
// Java method_invocation: 'object' field is on the callNode, not on nameNode's parent
if (!receiver && callNode.type === 'method_invocation') {
receiver = callNode.childForFieldName('object');
}
// PHP: member_call_expression has 'object' on the call node
if (!receiver && (callNode.type === 'member_call_expression' || callNode.type === 'nullsafe_member_call_expression')) {
receiver = callNode.childForFieldName('object');
}
// Ruby: call node has 'receiver' field
if (!receiver && parent.type === 'call') {
receiver = parent.childForFieldName('receiver');
}
// PHP scoped_call_expression (parent::method(), self::method()):
// nameNode's direct parent IS the scoped_call_expression (name is a direct child)
if (!receiver && (parent.type === 'scoped_call_expression' || callNode.type === 'scoped_call_expression')) {
const scopedCall = parent.type === 'scoped_call_expression' ? parent : callNode;
receiver = scopedCall.childForFieldName('scope');
// relative_scope wraps 'parent'/'self'/'static' — unwrap to get the keyword
if (receiver?.type === 'relative_scope') {
receiver = receiver.firstChild;
}
}
// C# null-conditional: user?.Save() → conditional_access_expression wraps member_binding_expression
if (!receiver && parent.type === 'member_binding_expression') {
const condAccess = parent.parent;
if (condAccess?.type === 'conditional_access_expression') {
receiver = condAccess.firstNamedChild;
}
}
// Kotlin/Swift: navigation_expression target is the first child
if (!receiver && parent.type === 'navigation_suffix') {
const navExpr = parent.parent;
if (navExpr?.type === 'navigation_expression') {
// First named child is the target (receiver)
for (const child of navExpr.children) {
if (child.isNamed && child !== parent) {
receiver = child;
break;
}
}
}
}
if (!receiver) return undefined;
// Only capture simple identifiers — refuse complex expressions
if (SIMPLE_RECEIVER_TYPES.has(receiver.type)) {
return receiver.text;
}
// Python super().method(): receiver is a call node `super()` — extract the function name
if (receiver.type === 'call') {
const func = receiver.childForFieldName('function');
if (func?.text === 'super') return 'super';
}
return undefined;
};
/**
* Extract the raw receiver AST node for a member call.
* Unlike extractReceiverName, this returns the receiver node regardless of its type —
* including call_expression / method_invocation nodes that appear in chained calls
* like `svc.getUser().save()`.
*
* Returns undefined when the call is not a member call or when no receiver node
* can be found (e.g. top-level free calls).
*/
export const extractReceiverNode = (
nameNode: SyntaxNode,
): SyntaxNode | undefined => {
const parent = nameNode.parent;
if (!parent) return undefined;
const callNode = parent.parent ?? parent;
let receiver: SyntaxNode | null = null;
receiver = parent.childForFieldName('object')
?? parent.childForFieldName('value')
?? parent.childForFieldName('operand')
?? parent.childForFieldName('expression')
?? parent.childForFieldName('argument');
if (!receiver && callNode.type === 'method_invocation') {
receiver = callNode.childForFieldName('object');
}
if (!receiver && (callNode.type === 'member_call_expression' || callNode.type === 'nullsafe_member_call_expression')) {
receiver = callNode.childForFieldName('object');
}
if (!receiver && parent.type === 'call') {
receiver = parent.childForFieldName('receiver');
}
if (!receiver && (parent.type === 'scoped_call_expression' || callNode.type === 'scoped_call_expression')) {
const scopedCall = parent.type === 'scoped_call_expression' ? parent : callNode;
receiver = scopedCall.childForFieldName('scope');
if (receiver?.type === 'relative_scope') {
receiver = receiver.firstChild;
}
}
if (!receiver && parent.type === 'member_binding_expression') {
const condAccess = parent.parent;
if (condAccess?.type === 'conditional_access_expression') {
receiver = condAccess.firstNamedChild;
}
}
if (!receiver && parent.type === 'navigation_suffix') {
const navExpr = parent.parent;
if (navExpr?.type === 'navigation_expression') {
for (const child of navExpr.children) {
if (child.isNamed && child !== parent) {
receiver = child;
break;
}
}
}
}
return receiver ?? undefined;
};
// ── Chained-call extraction ───────────────────────────────────────────────
/** Node types representing member/field access across languages. */
const FIELD_ACCESS_NODE_TYPES = new Set([
'member_expression', // TS/JS
'member_access_expression', // C#
'selector_expression', // Go
'field_expression', // Rust/C++
'field_access', // Java
'attribute', // Python
'navigation_expression', // Kotlin/Swift
'member_binding_expression', // C# null-conditional (user?.Address)
]);
/** One step in a mixed receiver chain. */
export type MixedChainStep = { kind: 'field' | 'call'; name: string };
/**
* Walk a receiver AST node that is itself a call expression, accumulating the
* chain of intermediate method names up to MAX_CHAIN_DEPTH.
*
* For `svc.getUser().save()`, called with the receiver of `save` (getUser() call):
* returns { chain: ['getUser'], baseReceiverName: 'svc' }
*
* For `a.b().c().d()`, called with the receiver of `d` (c() call):
* returns { chain: ['b', 'c'], baseReceiverName: 'a' }
*/
export function extractCallChain(
receiverCallNode: SyntaxNode,
): { chain: string[]; baseReceiverName: string | undefined } | undefined {
const chain: string[] = [];
let current: SyntaxNode = receiverCallNode;
while (CALL_EXPRESSION_TYPES.has(current.type) && chain.length < MAX_CHAIN_DEPTH) {
// Extract the method name from this call node.
const funcNode = current.childForFieldName?.('function')
?? current.childForFieldName?.('name')
?? current.childForFieldName?.('method'); // Ruby `call` node
let methodName: string | undefined;
let innerReceiver: SyntaxNode | null = null;
if (funcNode) {
// member_expression / attribute: last named child is the method identifier
methodName = funcNode.lastNamedChild?.text ?? funcNode.text;
}
// Kotlin/Swift: call_expression exposes callee as firstNamedChild, not a field.
// navigation_expression: method name is in navigation_suffix → simple_identifier.
if (!funcNode && current.type === 'call_expression') {
const callee = current.firstNamedChild;
if (callee?.type === 'navigation_expression') {
const suffix = callee.lastNamedChild;
if (suffix?.type === 'navigation_suffix') {
methodName = suffix.lastNamedChild?.text;
// The receiver is the part of navigation_expression before the suffix
for (let i = 0; i < callee.namedChildCount; i++) {
const child = callee.namedChild(i);
if (child && child.type !== 'navigation_suffix') {
innerReceiver = child;
break;
}
}
}
}
}
if (!methodName) break;
chain.unshift(methodName); // build chain outermost-last
// Walk into the receiver of this call to continue the chain
if (!innerReceiver && funcNode) {
innerReceiver = funcNode.childForFieldName?.('object')
?? funcNode.childForFieldName?.('value')
?? funcNode.childForFieldName?.('operand')
?? funcNode.childForFieldName?.('expression');
}
// Java method_invocation: object field is on the call node
if (!innerReceiver && current.type === 'method_invocation') {
innerReceiver = current.childForFieldName?.('object');
}
// PHP member_call_expression
if (!innerReceiver && (current.type === 'member_call_expression' || current.type === 'nullsafe_member_call_expression')) {
innerReceiver = current.childForFieldName?.('object');
}
// Ruby `call` node: receiver field is on the call node itself
if (!innerReceiver && current.type === 'call') {
innerReceiver = current.childForFieldName?.('receiver');
}
if (!innerReceiver) break;
if (CALL_EXPRESSION_TYPES.has(innerReceiver.type)) {
current = innerReceiver; // continue walking
} else {
// Reached a simple identifier — the base receiver
return { chain, baseReceiverName: innerReceiver.text || undefined };
}
}
return chain.length > 0 ? { chain, baseReceiverName: undefined } : undefined;
}
/**
* Walk a receiver AST node that may interleave field accesses and method calls,
* building a unified chain of steps up to MAX_CHAIN_DEPTH.
*
* For `svc.getUser().address.save()`, called with the receiver of `save`
* (`svc.getUser().address`, a field access node):
* returns { chain: [{ kind:'call', name:'getUser' }, { kind:'field', name:'address' }],
* baseReceiverName: 'svc' }
*
* For `user.getAddress().city.getName()`, called with receiver of `getName`
* (`user.getAddress().city`):
* returns { chain: [{ kind:'call', name:'getAddress' }, { kind:'field', name:'city' }],
* baseReceiverName: 'user' }
*
* Pure field chains and pure call chains are special cases (all steps same kind).
*/
export function extractMixedChain(
receiverNode: SyntaxNode,
): { chain: MixedChainStep[]; baseReceiverName: string | undefined } | undefined {
const chain: MixedChainStep[] = [];
let current: SyntaxNode = receiverNode;
while (chain.length < MAX_CHAIN_DEPTH) {
if (CALL_EXPRESSION_TYPES.has(current.type)) {
// ── Call expression: extract method name + inner receiver ────────────
const funcNode = current.childForFieldName?.('function')
?? current.childForFieldName?.('name')
?? current.childForFieldName?.('method');
let methodName: string | undefined;
let innerReceiver: SyntaxNode | null = null;
if (funcNode) {
methodName = funcNode.lastNamedChild?.text ?? funcNode.text;
}
// Kotlin/Swift: call_expression → navigation_expression
if (!funcNode && current.type === 'call_expression') {
const callee = current.firstNamedChild;
if (callee?.type === 'navigation_expression') {
const suffix = callee.lastNamedChild;
if (suffix?.type === 'navigation_suffix') {
methodName = suffix.lastNamedChild?.text;
for (let i = 0; i < callee.namedChildCount; i++) {
const child = callee.namedChild(i);
if (child && child.type !== 'navigation_suffix') { innerReceiver = child; break; }
}
}
}
}
if (!methodName) break;
chain.unshift({ kind: 'call', name: methodName });
if (!innerReceiver && funcNode) {
innerReceiver = funcNode.childForFieldName?.('object')
?? funcNode.childForFieldName?.('value')
?? funcNode.childForFieldName?.('operand')
?? funcNode.childForFieldName?.('argument') // C/C++ field_expression
?? funcNode.childForFieldName?.('expression')
?? null;
}
if (!innerReceiver && current.type === 'method_invocation') {
innerReceiver = current.childForFieldName?.('object') ?? null;
}
if (!innerReceiver && (current.type === 'member_call_expression' || current.type === 'nullsafe_member_call_expression')) {
innerReceiver = current.childForFieldName?.('object') ?? null;
}
if (!innerReceiver && current.type === 'call') {
innerReceiver = current.childForFieldName?.('receiver') ?? null;
}
if (!innerReceiver) break;
if (CALL_EXPRESSION_TYPES.has(innerReceiver.type) || FIELD_ACCESS_NODE_TYPES.has(innerReceiver.type)) {
current = innerReceiver;
} else {
return { chain, baseReceiverName: innerReceiver.text || undefined };
}
} else if (FIELD_ACCESS_NODE_TYPES.has(current.type)) {
// ── Field/member access: extract property name + inner object ─────────
let propertyName: string | undefined;
let innerObject: SyntaxNode | null = null;
if (current.type === 'navigation_expression') {
for (const child of current.children ?? []) {
if (child.type === 'navigation_suffix') {
for (const sc of child.children ?? []) {
if (sc.isNamed && sc.type !== '.') { propertyName = sc.text; break; }
}
} else if (child.isNamed && !innerObject) {
innerObject = child;
}
}
} else if (current.type === 'attribute') {
innerObject = current.childForFieldName?.('object') ?? null;
propertyName = current.childForFieldName?.('attribute')?.text;
} else {
innerObject = current.childForFieldName?.('object')
?? current.childForFieldName?.('value')
?? current.childForFieldName?.('operand')
?? current.childForFieldName?.('argument') // C/C++ field_expression
?? current.childForFieldName?.('expression')
?? null;
propertyName = (current.childForFieldName?.('property')
?? current.childForFieldName?.('field')
?? current.childForFieldName?.('name'))?.text;
}
if (!propertyName) break;
chain.unshift({ kind: 'field', name: propertyName });
if (!innerObject) break;
if (CALL_EXPRESSION_TYPES.has(innerObject.type) || FIELD_ACCESS_NODE_TYPES.has(innerObject.type)) {
current = innerObject;
} else {
return { chain, baseReceiverName: innerObject.text || undefined };
}
} else {
// Simple identifier — this is the base receiver
return chain.length > 0
? { chain, baseReceiverName: current.text || undefined }
: undefined;
}
}
return chain.length > 0 ? { chain, baseReceiverName: undefined } : undefined;
}

File diff suppressed because it is too large Load diff

View file

@ -18,6 +18,12 @@ import { SupportedLanguages } from '../../config/supported-languages.js';
/** null = this call was not routed; fall through to default call handling */
export type CallRoutingResult = RubyCallRouting | null;
/**
* Per-language call router.
* IMPORTANT: Call-routed imports bypass preprocessImportPath(), so any router that
* returns an importPath MUST validate it independently (length cap, control-char
* rejection). See routeRubyCall for the reference implementation.
*/
export type CallRouter = (
calledName: string,
callNode: any,
@ -27,7 +33,7 @@ export type CallRouter = (
const noRouting: CallRouter = () => null;
/** Per-language call routing. noRouting = no special routing (normal call processing) */
export const callRouters: Record<SupportedLanguages, CallRouter> = {
export const callRouters = {
[SupportedLanguages.JavaScript]: noRouting,
[SupportedLanguages.TypeScript]: noRouting,
[SupportedLanguages.Python]: noRouting,
@ -41,7 +47,7 @@ export const callRouters: Record<SupportedLanguages, CallRouter> = {
[SupportedLanguages.CPlusPlus]: noRouting,
[SupportedLanguages.C]: noRouting,
[SupportedLanguages.Ruby]: routeRubyCall,
};
} satisfies Record<SupportedLanguages, CallRouter>;
// ── Result types ────────────────────────────────────────────────────────────
@ -65,6 +71,8 @@ export interface RubyPropertyItem {
accessorType: RubyAccessorType;
startLine: number;
endLine: number;
/** YARD @return [Type] annotation preceding the attr_accessor call */
declaredType?: string;
}
// ── Pre-allocated singletons for common return values ────────────────────────
@ -129,6 +137,25 @@ export function routeRubyCall(calledName: string, callNode: any): RubyCallRoutin
// ── attr_accessor / attr_reader / attr_writer → property definitions ───
if (calledName === 'attr_accessor' || calledName === 'attr_reader' || calledName === 'attr_writer') {
// Extract YARD @return [Type] from preceding comment (e.g. `# @return [Address]`)
let yardType: string | undefined;
let sibling = callNode.previousSibling;
while (sibling) {
if (sibling.type === 'comment') {
const match = /@return\s+\[([^\]]+)\]/.exec(sibling.text);
if (match) {
const raw = match[1].trim();
// Extract simple type name: "User", "Array<User>" → "User"
const simple = raw.match(/^([A-Z]\w*)/);
if (simple) yardType = simple[1];
break;
}
} else if (sibling.isNamed) {
break; // stop at non-comment named sibling
}
sibling = sibling.previousSibling;
}
const items: RubyPropertyItem[] = [];
const argList = callNode.childForFieldName?.('arguments');
for (const arg of (argList?.children ?? [])) {
@ -138,6 +165,7 @@ export function routeRubyCall(calledName: string, callNode: any): RubyCallRoutin
accessorType: calledName as RubyAccessorType,
startLine: arg.startPosition.row,
endLine: arg.endPosition.row,
...(yardType ? { declaredType: yardType } : {}),
});
}
}

View file

@ -14,30 +14,33 @@ import { detectFrameworkFromPath } from './framework-detection.js';
import { SupportedLanguages } from '../../config/supported-languages.js';
// ============================================================================
// NAME PATTERNS - All 11 supported languages
// NAME PATTERNS - All 13 supported languages
// ============================================================================
/**
* Common entry point naming patterns by language
* These patterns indicate functions that are likely feature entry points
* Common entry point naming patterns by language.
* These patterns indicate functions that are likely feature entry points.
*
* Universal patterns are separated from per-language patterns so the per-language
* table can use `satisfies Record<SupportedLanguages, RegExp[]>` for compile-time
* exhaustiveness — the compiler catches any missing language entry.
*/
const ENTRY_POINT_PATTERNS: Record<string, RegExp[]> = {
// Universal patterns (apply to all languages)
'*': [
/^(main|init|bootstrap|start|run|setup|configure)$/i,
/^handle[A-Z]/, // handleLogin, handleSubmit
/^on[A-Z]/, // onClick, onSubmit
/Handler$/, // RequestHandler
/Controller$/, // UserController
/^process[A-Z]/, // processPayment
/^execute[A-Z]/, // executeQuery
/^perform[A-Z]/, // performAction
/^dispatch[A-Z]/, // dispatchEvent
/^trigger[A-Z]/, // triggerAction
/^fire[A-Z]/, // fireEvent
/^emit[A-Z]/, // emitEvent
],
const UNIVERSAL_ENTRY_POINT_PATTERNS: RegExp[] = [
/^(main|init|bootstrap|start|run|setup|configure)$/i,
/^handle[A-Z]/, // handleLogin, handleSubmit
/^on[A-Z]/, // onClick, onSubmit
/Handler$/, // RequestHandler
/Controller$/, // UserController
/^process[A-Z]/, // processPayment
/^execute[A-Z]/, // executeQuery
/^perform[A-Z]/, // performAction
/^dispatch[A-Z]/, // dispatchEvent
/^trigger[A-Z]/, // triggerAction
/^fire[A-Z]/, // fireEvent
/^emit[A-Z]/, // emitEvent
];
const ENTRY_POINT_PATTERNS = {
// JavaScript/TypeScript
[SupportedLanguages.JavaScript]: [
/^use[A-Z]/, // React hooks (useEffect, etc.)
@ -62,6 +65,17 @@ const ENTRY_POINT_PATTERNS: Record<string, RegExp[]> = {
/Service$/, // UserService
],
// Kotlin
[SupportedLanguages.Kotlin]: [
/^on(Create|Start|Resume|Pause|Stop|Destroy)$/, // Android lifecycle
/^do[A-Z]/, // doGet, doPost (shared JVM Servlet pattern)
/^create[A-Z]/, // Factory patterns
/^build[A-Z]/, // Builder patterns
/ViewModel$/, // MVVM pattern (Android)
/^module$/, // Ktor module entry point
/Service$/, // Service classes
],
// C#
[SupportedLanguages.CSharp]: [
/^(Get|Post|Put|Delete|Patch)/, // ASP.NET action methods
@ -77,7 +91,7 @@ const ENTRY_POINT_PATTERNS: Record<string, RegExp[]> = {
/Service$/, // Service classes
/^Seed/, // Database seeding
],
// Go
[SupportedLanguages.Go]: [
/Handler$/, // http.Handler pattern
@ -85,7 +99,7 @@ const ENTRY_POINT_PATTERNS: Record<string, RegExp[]> = {
/^New[A-Z]/, // Constructor pattern (returns new instance)
/^Make[A-Z]/, // Make functions
],
// Rust
[SupportedLanguages.Rust]: [
/^(get|post|put|delete)_handler$/i,
@ -94,7 +108,7 @@ const ENTRY_POINT_PATTERNS: Record<string, RegExp[]> = {
/^run$/, // run entry point
/^spawn/, // Async spawn
],
// C - explicit main() boost plus common C entry point conventions
[SupportedLanguages.C]: [
/^main$/, // THE entry point
@ -198,15 +212,15 @@ const ENTRY_POINT_PATTERNS: Record<string, RegExp[]> = {
/^perform$/, // Background jobs (Sidekiq, ActiveJob)
/^execute$/, // Command pattern
],
};
} satisfies Record<SupportedLanguages, RegExp[]>;
/** Pre-computed merged patterns (universal + language-specific) to avoid per-call array allocation. */
const MERGED_ENTRY_POINT_PATTERNS: Record<string, RegExp[]> = {};
const UNIVERSAL_PATTERNS = ENTRY_POINT_PATTERNS['*'] || [];
for (const [lang, patterns] of Object.entries(ENTRY_POINT_PATTERNS)) {
if (lang === '*') continue;
MERGED_ENTRY_POINT_PATTERNS[lang] = [...UNIVERSAL_PATTERNS, ...patterns];
}
const MERGED_ENTRY_POINT_PATTERNS = Object.fromEntries(
(Object.keys(ENTRY_POINT_PATTERNS) as SupportedLanguages[]).map(lang => [
lang,
[...UNIVERSAL_ENTRY_POINT_PATTERNS, ...ENTRY_POINT_PATTERNS[lang]],
])
) as Record<SupportedLanguages, RegExp[]>;
// ============================================================================
// UTILITY PATTERNS - Functions that should be penalized
@ -295,7 +309,7 @@ export function calculateEntryPointScore(
reasons.push('utility-pattern');
} else {
// Check positive patterns
const allPatterns = MERGED_ENTRY_POINT_PATTERNS[language] || UNIVERSAL_PATTERNS;
const allPatterns = MERGED_ENTRY_POINT_PATTERNS[language];
if (allPatterns.some(p => p.test(name))) {
nameMultiplier = 1.5; // Bonus for matching entry point pattern

View file

@ -1,7 +1,7 @@
import fs from 'fs/promises';
import path from 'path';
import { glob } from 'glob';
import { shouldIgnorePath } from '../../config/ignore-service.js';
import { createIgnoreFilter } from '../../config/ignore-service.js';
export interface FileEntry {
path: string;
@ -32,13 +32,14 @@ export const walkRepositoryPaths = async (
repoPath: string,
onProgress?: (current: number, total: number, filePath: string) => void
): Promise<ScannedFile[]> => {
const files = await glob('**/*', {
const ignoreFilter = await createIgnoreFilter(repoPath);
const filtered = await glob('**/*', {
cwd: repoPath,
nodir: true,
dot: false,
ignore: ignoreFilter,
});
const filtered = files.filter(file => !shouldIgnorePath(file));
const entries: ScannedFile[] = [];
let processed = 0;
let skippedLarge = 0;

View file

@ -10,6 +10,8 @@
* (no bonus, no penalty) - same behavior as before this feature.
*/
import { SupportedLanguages } from '../../config/supported-languages.js';
// ============================================================================
// TYPES
// ============================================================================
@ -234,8 +236,8 @@ export function detectFrameworkFromPath(filePath: string): FrameworkHint | null
return { framework: 'go-mvc', entryPointMultiplier: 2.5, reason: 'go-controller' };
}
// Go main.go files (THE entry point)
if (p.endsWith('/main.go') || p.endsWith('/cmd/') && p.endsWith('.go')) {
// Go main.go files (THE entry point) — only match main.go, not arbitrary .go files under cmd/
if (p.endsWith('/main.go')) {
return { framework: 'go', entryPointMultiplier: 3.0, reason: 'go-main' };
}
@ -431,25 +433,36 @@ export const FRAMEWORK_AST_PATTERNS = {
'blazor': ['@page', '[Parameter]', '@inject'],
'efcore': ['DbContext', 'DbSet<', 'OnModelCreating'],
// Go patterns (function signatures)
'go-http': ['http.Handler', 'http.HandlerFunc', 'ServeHTTP'],
// Go patterns (function signatures include framework types)
'go-http': ['http.Handler', 'http.HandlerFunc', 'ServeHTTP', 'http.ResponseWriter', 'http.Request'],
'gin': ['gin.Context', 'gin.Default', 'gin.New'],
'echo': ['echo.Context', 'echo.New'],
'fiber': ['fiber.Ctx', 'fiber.New', 'fiber.App'],
'go-grpc': ['grpc.Server', 'RegisterServer', 'pb.Unimplemented'],
// PHP/Laravel
'laravel': ['Route::get', 'Route::post', 'Route::put', 'Route::delete',
'Route::resource', 'Route::apiResource', '#[Route('],
// Rust macros
'actix': ['#[get', '#[post', '#[put', '#[delete'],
'axum': ['Router::new'],
'rocket': ['#[get', '#[post'],
// Rust macros (proc-macro attributes in definition text)
'actix': ['#[get', '#[post', '#[put', '#[delete', '#[actix_web', 'HttpRequest', 'HttpResponse'],
'axum': ['Router::new', 'axum::extract', 'axum::routing'],
'rocket': ['#[get', '#[post', '#[launch', 'rocket::'],
'tokio': ['#[tokio::main]', '#[tokio::test]'],
// C++ patterns (Qt, Boost)
'qt': ['Q_OBJECT', 'Q_INVOKABLE', 'Q_PROPERTY', 'Q_SIGNALS', 'Q_SLOTS', 'Q_SIGNAL', 'Q_SLOT', 'QWidget', 'QApplication'],
// Swift/iOS
'uikit': ['viewDidLoad', 'viewWillAppear', 'viewDidAppear', 'UIViewController'],
'swiftui': ['@main', 'WindowGroup', 'ContentView', '@StateObject', '@ObservedObject'],
'combine': ['sink', 'assign', 'Publisher', 'Subscriber'],
};
'uikit': ['viewDidLoad', 'viewWillAppear', 'viewDidAppear', 'UIViewController', '@IBOutlet', '@IBAction', '@objc'],
'swiftui': ['@main', 'WindowGroup', 'ContentView', '@StateObject', '@ObservedObject', '@EnvironmentObject', '@Published'],
'vapor': ['app.get', 'app.post', 'req.content.decode', 'Vapor'],
import { SupportedLanguages } from '../../config/supported-languages.js';
// Ruby patterns (class-level macros in definition text)
'rails': ['ApplicationController', 'ApplicationRecord', 'ActiveRecord::Base',
'before_action', 'after_action', 'has_many', 'belongs_to', 'has_one', 'validates'],
'sinatra': ['Sinatra::Base', 'Sinatra::Application'],
};
interface AstFrameworkPatternConfig {
framework: string;
@ -458,7 +471,7 @@ interface AstFrameworkPatternConfig {
patterns: string[];
}
const AST_FRAMEWORK_PATTERNS_BY_LANGUAGE: Record<string, AstFrameworkPatternConfig[]> = {
const AST_FRAMEWORK_PATTERNS_BY_LANGUAGE = {
[SupportedLanguages.JavaScript]: [
{ framework: 'nestjs', entryPointMultiplier: 3.2, reason: 'nestjs-decorator', patterns: FRAMEWORK_AST_PATTERNS.nestjs },
],
@ -488,7 +501,33 @@ const AST_FRAMEWORK_PATTERNS_BY_LANGUAGE: Record<string, AstFrameworkPatternConf
[SupportedLanguages.PHP]: [
{ framework: 'laravel', entryPointMultiplier: 3.0, reason: 'php-route-attribute', patterns: FRAMEWORK_AST_PATTERNS.laravel },
],
};
[SupportedLanguages.Go]: [
{ framework: 'go-http', entryPointMultiplier: 2.5, reason: 'go-http-handler', patterns: FRAMEWORK_AST_PATTERNS['go-http'] },
{ framework: 'gin', entryPointMultiplier: 3.0, reason: 'gin-handler', patterns: FRAMEWORK_AST_PATTERNS.gin },
{ framework: 'echo', entryPointMultiplier: 3.0, reason: 'echo-handler', patterns: FRAMEWORK_AST_PATTERNS.echo },
{ framework: 'fiber', entryPointMultiplier: 3.0, reason: 'fiber-handler', patterns: FRAMEWORK_AST_PATTERNS.fiber },
{ framework: 'go-grpc', entryPointMultiplier: 2.8, reason: 'grpc-service', patterns: FRAMEWORK_AST_PATTERNS['go-grpc'] },
],
[SupportedLanguages.Rust]: [
{ framework: 'actix-web', entryPointMultiplier: 3.0, reason: 'actix-attribute', patterns: FRAMEWORK_AST_PATTERNS.actix },
{ framework: 'axum', entryPointMultiplier: 3.0, reason: 'axum-routing', patterns: FRAMEWORK_AST_PATTERNS.axum },
{ framework: 'rocket', entryPointMultiplier: 3.0, reason: 'rocket-attribute', patterns: FRAMEWORK_AST_PATTERNS.rocket },
{ framework: 'tokio', entryPointMultiplier: 2.5, reason: 'tokio-runtime', patterns: FRAMEWORK_AST_PATTERNS.tokio },
],
[SupportedLanguages.C]: [], // C has no framework-specific AST patterns (POSIX/socket patterns are in entry-point-scoring)
[SupportedLanguages.CPlusPlus]: [
{ framework: 'qt', entryPointMultiplier: 2.8, reason: 'qt-macro', patterns: FRAMEWORK_AST_PATTERNS.qt },
],
[SupportedLanguages.Swift]: [
{ framework: 'uikit', entryPointMultiplier: 2.5, reason: 'uikit-lifecycle', patterns: FRAMEWORK_AST_PATTERNS.uikit },
{ framework: 'swiftui', entryPointMultiplier: 2.8, reason: 'swiftui-pattern', patterns: FRAMEWORK_AST_PATTERNS.swiftui },
{ framework: 'vapor', entryPointMultiplier: 3.0, reason: 'vapor-routing', patterns: FRAMEWORK_AST_PATTERNS.vapor },
],
[SupportedLanguages.Ruby]: [
{ framework: 'rails', entryPointMultiplier: 3.0, reason: 'rails-pattern', patterns: FRAMEWORK_AST_PATTERNS.rails },
{ framework: 'sinatra', entryPointMultiplier: 2.8, reason: 'sinatra-pattern', patterns: FRAMEWORK_AST_PATTERNS.sinatra },
],
} satisfies Record<SupportedLanguages, AstFrameworkPatternConfig[]>;
/** Pre-lowercased patterns for O(1) pattern matching at runtime */
const AST_PATTERNS_LOWERED: Record<string, Array<{ framework: string; entryPointMultiplier: number; reason: string; patterns: string[] }>> =

View file

@ -5,42 +5,15 @@ import { isLanguageAvailable, loadParser, loadLanguage } from '../tree-sitter/pa
import { LANGUAGE_QUERIES } from './tree-sitter-queries.js';
import { generateId } from '../../lib/utils.js';
import { getLanguageFromFilename, isVerboseIngestionEnabled, yieldToEventLoop } from './utils.js';
import { SupportedLanguages } from '../../config/supported-languages.js';
import { extractNamedBindings } from './named-binding-extraction.js';
import type { ExtractedImport } from './workers/parse-worker.js';
import { getTreeSitterBufferSize } from './constants.js';
import {
loadTsconfigPaths,
loadGoModulePath,
loadComposerConfig,
loadCSharpProjectConfig,
loadSwiftPackageConfig,
type SwiftPackageConfig,
} from './language-config.js';
import {
buildSuffixIndex,
resolveImportPath,
appendKotlinWildcard,
KOTLIN_EXTENSIONS,
resolveJvmWildcard,
resolveJvmMemberImport,
resolveGoPackageDir,
resolveGoPackage,
resolveCSharpImport,
resolveCSharpNamespaceDir,
resolvePhpImport,
resolveRustImport,
resolveRubyImport,
} from './resolvers/index.js';
import { loadImportConfigs } from './language-config.js';
import { buildSuffixIndex } from './resolvers/index.js';
import { callRouters } from './call-routing.js';
import type { ResolutionContext } from './resolution-context.js';
import type {
SuffixIndex,
TsconfigPaths,
GoModuleConfig,
CSharpProjectConfig,
ComposerConfig
} from './resolvers/index.js';
import type { SuffixIndex } from './resolvers/index.js';
import { importResolvers, namedBindingExtractors, preprocessImportPath } from './import-resolution.js';
import type { ImportResult, ResolveCtx, NamedBinding } from './import-resolution.js';
// Re-export resolver types for consumers
export type {
@ -88,7 +61,7 @@ export interface ImportResolutionContext {
allFilePaths: Set<string>;
allFileList: string[];
normalizedFileList: string[];
suffixIndex: SuffixIndex;
index: SuffixIndex;
resolveCache: Map<string, string | null>;
}
@ -96,161 +69,32 @@ export function buildImportResolutionContext(allPaths: string[]): ImportResoluti
const allFileList = allPaths;
const normalizedFileList = allFileList.map(p => p.replace(/\\/g, '/'));
const allFilePaths = new Set(allFileList);
const suffixIndex = buildSuffixIndex(normalizedFileList, allFileList);
return { allFilePaths, allFileList, normalizedFileList, suffixIndex, resolveCache: new Map() };
const index = buildSuffixIndex(normalizedFileList, allFileList);
return { allFilePaths, allFileList, normalizedFileList, index, resolveCache: new Map() };
}
// Config loaders extracted to ./language-config.ts (Phase 2 refactor)
// Resolver functions are in ./resolvers/ — imported above
// Resolver dispatch tables are in ./import-resolution.ts — imported above
// ============================================================================
// SHARED LANGUAGE DISPATCH
// ============================================================================
/** Create IMPORTS edge helpers that share a resolved-count tracker. */
function createImportEdgeHelpers(graph: KnowledgeGraph, importMap: ImportMap) {
let totalImportsResolved = 0;
/** Bundled language-specific configs loaded once per ingestion run. */
interface LanguageConfigs {
tsconfigPaths: TsconfigPaths | null;
goModule: GoModuleConfig | null;
composerConfig: ComposerConfig | null;
swiftPackageConfig: SwiftPackageConfig | null;
csharpConfigs: CSharpProjectConfig[];
}
const addImportGraphEdge = (filePath: string, resolvedPath: string) => {
const sourceId = generateId('File', filePath);
const targetId = generateId('File', resolvedPath);
const relId = generateId('IMPORTS', `${filePath}->${resolvedPath}`);
totalImportsResolved++;
graph.addRelationship({ id: relId, sourceId, targetId, type: 'IMPORTS', confidence: 1.0, reason: '' });
};
/** Context for import path resolution (file lists, indexes, cache). */
interface ResolveCtx {
allFilePaths: Set<string>;
allFileList: string[];
normalizedFileList: string[];
index: SuffixIndex;
resolveCache: Map<string, string | null>;
}
const addImportEdge = (filePath: string, resolvedPath: string) => {
addImportGraphEdge(filePath, resolvedPath);
if (!importMap.has(filePath)) importMap.set(filePath, new Set());
importMap.get(filePath)!.add(resolvedPath);
};
/**
* Result of resolving an import via language-specific dispatch.
* - 'files': resolved to one or more files → add to ImportMap
* - 'package': resolved to a directory → add graph edges + store dirSuffix in PackageMap
* - null: no resolution (external dependency, etc.)
*/
type ImportResult =
| { kind: 'files'; files: string[] }
| { kind: 'package'; files: string[]; dirSuffix: string }
| null;
/**
* Shared language dispatch for import resolution.
* Used by both processImports and processImportsFromExtracted.
*/
function resolveLanguageImport(
filePath: string,
rawImportPath: string,
language: SupportedLanguages,
configs: LanguageConfigs,
ctx: ResolveCtx,
): ImportResult {
const { allFilePaths, allFileList, normalizedFileList, index, resolveCache } = ctx;
const { tsconfigPaths, goModule, composerConfig, swiftPackageConfig, csharpConfigs } = configs;
// JVM languages (Java + Kotlin): handle wildcards and member imports
if (language === SupportedLanguages.Java || language === SupportedLanguages.Kotlin) {
const exts = language === SupportedLanguages.Java ? ['.java'] : KOTLIN_EXTENSIONS;
if (rawImportPath.endsWith('.*')) {
const matchedFiles = resolveJvmWildcard(rawImportPath, normalizedFileList, allFileList, exts, index);
if (matchedFiles.length === 0 && language === SupportedLanguages.Kotlin) {
const javaMatches = resolveJvmWildcard(rawImportPath, normalizedFileList, allFileList, ['.java'], index);
if (javaMatches.length > 0) return { kind: 'files', files: javaMatches };
}
if (matchedFiles.length > 0) return { kind: 'files', files: matchedFiles };
// Fall through to standard resolution
} else {
let memberResolved = resolveJvmMemberImport(rawImportPath, normalizedFileList, allFileList, exts, index);
if (!memberResolved && language === SupportedLanguages.Kotlin) {
memberResolved = resolveJvmMemberImport(rawImportPath, normalizedFileList, allFileList, ['.java'], index);
}
if (memberResolved) return { kind: 'files', files: [memberResolved] };
// Fall through to standard resolution
}
}
// Go: handle package-level imports
if (language === SupportedLanguages.Go && goModule && rawImportPath.startsWith(goModule.modulePath)) {
const pkgSuffix = resolveGoPackageDir(rawImportPath, goModule);
if (pkgSuffix) {
const pkgFiles = resolveGoPackage(rawImportPath, goModule, normalizedFileList, allFileList);
if (pkgFiles.length > 0) {
return { kind: 'package', files: pkgFiles, dirSuffix: pkgSuffix };
}
}
// Fall through if no files found (package might be external)
}
// C#: handle namespace-based imports (using directives)
if (language === SupportedLanguages.CSharp && csharpConfigs.length > 0) {
const resolvedFiles = resolveCSharpImport(rawImportPath, csharpConfigs, normalizedFileList, allFileList, index);
if (resolvedFiles.length > 1) {
const dirSuffix = resolveCSharpNamespaceDir(rawImportPath, csharpConfigs);
if (dirSuffix) {
return { kind: 'package', files: resolvedFiles, dirSuffix };
}
}
if (resolvedFiles.length > 0) return { kind: 'files', files: resolvedFiles };
return null;
}
// PHP: handle namespace-based imports (use statements)
if (language === SupportedLanguages.PHP) {
const resolved = resolvePhpImport(rawImportPath, composerConfig, allFilePaths, normalizedFileList, allFileList, index);
return resolved ? { kind: 'files', files: [resolved] } : null;
}
// Swift: handle module imports
if (language === SupportedLanguages.Swift && swiftPackageConfig) {
const targetDir = swiftPackageConfig.targets.get(rawImportPath);
if (targetDir) {
const dirPrefix = targetDir + '/';
const files: string[] = [];
for (let i = 0; i < normalizedFileList.length; i++) {
if (normalizedFileList[i].startsWith(dirPrefix) && normalizedFileList[i].endsWith('.swift')) {
files.push(allFileList[i]);
}
}
if (files.length > 0) return { kind: 'files', files };
}
return null; // External framework (Foundation, UIKit, etc.)
}
// Ruby: require / require_relative
if (language === SupportedLanguages.Ruby) {
const resolved = resolveRubyImport(rawImportPath, normalizedFileList, allFileList, index);
return resolved ? { kind: 'files', files: [resolved] } : null;
}
// Rust: expand top-level grouped imports: use {crate::a, crate::b}
if (language === SupportedLanguages.Rust && rawImportPath.startsWith('{') && rawImportPath.endsWith('}')) {
const inner = rawImportPath.slice(1, -1);
const parts = inner.split(',').map(p => p.trim()).filter(Boolean);
const resolved: string[] = [];
for (const part of parts) {
const r = resolveRustImport(filePath, part, allFilePaths);
if (r) resolved.push(r);
}
return resolved.length > 0 ? { kind: 'files', files: resolved } : null;
}
// Standard single-file resolution
const resolvedPath = resolveImportPath(
filePath,
rawImportPath,
allFilePaths,
allFileList,
normalizedFileList,
resolveCache,
language,
tsconfigPaths,
index,
);
return resolvedPath ? { kind: 'files', files: [resolvedPath] } : null;
return { addImportEdge, addImportGraphEdge, getResolvedCount: () => totalImportsResolved };
}
/**
@ -265,7 +109,7 @@ function applyImportResult(
packageMap: PackageMap | undefined,
addImportEdge: (from: string, to: string) => void,
addImportGraphEdge: (from: string, to: string) => void,
namedBindings?: { local: string; exported: string }[],
namedBindings?: NamedBinding[],
namedImportMap?: NamedImportMap,
): void {
if (!result) return;
@ -284,13 +128,44 @@ function applyImportResult(
addImportEdge(filePath, resolvedFile);
}
// Record named bindings for precise Tier 2a resolution
if (namedBindings && namedImportMap && files.length === 1) {
const resolvedFile = files[0];
// Record named bindings for precise Tier 2a resolution.
// If the same local name is imported from multiple files (e.g., Java static imports
// of overloaded methods), remove the entry so resolution falls through to Tier 2a
// import-scoped which sees all candidates and can apply arity narrowing.
if (namedBindings && namedImportMap) {
if (!namedImportMap.has(filePath)) namedImportMap.set(filePath, new Map());
const fileBindings = namedImportMap.get(filePath)!;
for (const binding of namedBindings) {
fileBindings.set(binding.local, { sourcePath: resolvedFile, exportedName: binding.exported });
if (files.length === 1) {
const resolvedFile = files[0];
for (const binding of namedBindings) {
const existing = fileBindings.get(binding.local);
if (existing && existing.sourcePath !== resolvedFile) {
fileBindings.delete(binding.local);
} else {
fileBindings.set(binding.local, { sourcePath: resolvedFile, exportedName: binding.exported });
}
}
} else {
// Multi-file resolution (e.g., Rust `use crate::models::{User, Repo}`).
// Match each binding to a resolved file by comparing the lowercase binding name
// to the file's basename (without extension). If no match, skip the binding.
for (const binding of namedBindings) {
const lowerName = binding.exported.toLowerCase();
const matchedFile = files.find(f => {
const base = f.replace(/\\/g, '/').split('/').pop() ?? '';
const nameWithoutExt = base.substring(0, base.lastIndexOf('.')).toLowerCase();
return nameWithoutExt === lowerName;
});
if (matchedFile) {
const existing = fileBindings.get(binding.local);
if (existing && existing.sourcePath !== matchedFile) {
fileBindings.delete(binding.local);
} else {
fileBindings.set(binding.local, { sourcePath: matchedFile, exportedName: binding.exported });
}
}
}
}
}
}
@ -326,46 +201,11 @@ export const processImports = async (
// Track import statistics
let totalImportsFound = 0;
let totalImportsResolved = 0;
// Load language-specific configs once before the file loop
const effectiveRoot = repoRoot || '';
const configs: LanguageConfigs = {
tsconfigPaths: await loadTsconfigPaths(effectiveRoot),
goModule: await loadGoModulePath(effectiveRoot),
composerConfig: await loadComposerConfig(effectiveRoot),
swiftPackageConfig: await loadSwiftPackageConfig(effectiveRoot),
csharpConfigs: await loadCSharpProjectConfig(effectiveRoot),
};
const resolveCtx: ResolveCtx = { allFilePaths, allFileList, normalizedFileList, index, resolveCache };
// Helper: add an IMPORTS edge to the graph only (no ImportMap update)
const addImportGraphEdge = (filePath: string, resolvedPath: string) => {
const sourceId = generateId('File', filePath);
const targetId = generateId('File', resolvedPath);
const relId = generateId('IMPORTS', `${filePath}->${resolvedPath}`);
totalImportsResolved++;
graph.addRelationship({
id: relId,
sourceId,
targetId,
type: 'IMPORTS',
confidence: 1.0,
reason: '',
});
};
// Helper: add an IMPORTS edge + update import map
const addImportEdge = (filePath: string, resolvedPath: string) => {
addImportGraphEdge(filePath, resolvedPath);
if (!importMap.has(filePath)) {
importMap.set(filePath, new Set());
}
importMap.get(filePath)!.add(resolvedPath);
};
const configs = await loadImportConfigs(repoRoot || '');
const resolveCtx: ResolveCtx = { allFilePaths, allFileList, normalizedFileList, index, resolveCache, configs };
const { addImportEdge, addImportGraphEdge, getResolvedCount } = createImportEdgeHelpers(graph, importMap);
for (let i = 0; i < files.length; i++) {
const file = files[i];
@ -438,14 +278,13 @@ export const processImports = async (
return;
}
// Clean path (remove quotes and angle brackets for C/C++ includes)
const rawImportPath = language === SupportedLanguages.Kotlin
? appendKotlinWildcard(sourceNode.text.replace(/['"<>]/g, ''), captureMap['import'])
: sourceNode.text.replace(/['"<>]/g, '');
const rawImportPath = preprocessImportPath(sourceNode.text, captureMap['import'], language);
if (!rawImportPath) return;
totalImportsFound++;
const result = resolveLanguageImport(file.path, rawImportPath, language, configs, resolveCtx);
const bindings = namedImportMap ? extractNamedBindings(captureMap['import'], language) : undefined;
const result = importResolvers[language](rawImportPath, file.path, resolveCtx);
const extractor = namedBindingExtractors[language];
const bindings = namedImportMap && extractor ? extractor(captureMap['import']) : undefined;
applyImportResult(result, file.path, importMap, packageMap, addImportEdge, addImportGraphEdge, bindings, namedImportMap);
}
@ -457,7 +296,7 @@ export const processImports = async (
const routed = callRouter(callNameNode.text, captureMap['call']);
if (routed && routed.kind === 'import') {
totalImportsFound++;
const result = resolveLanguageImport(file.path, routed.importPath, language, configs, resolveCtx);
const result = importResolvers[language](routed.importPath, file.path, resolveCtx);
applyImportResult(result, file.path, importMap, packageMap, addImportEdge, addImportGraphEdge);
}
}
@ -476,7 +315,7 @@ export const processImports = async (
}
if (isDev) {
console.log(`📊 Import processing complete: ${totalImportsResolved}/${totalImportsFound} imports resolved to graph edges`);
console.log(`📊 Import processing complete: ${getResolvedCount()}/${totalImportsFound} imports resolved to graph edges`);
}
};
@ -497,47 +336,13 @@ export const processImportsFromExtracted = async (
const packageMap = ctx.packageMap;
const namedImportMap = ctx.namedImportMap;
const importCtx = prebuiltCtx ?? buildImportResolutionContext(files.map(f => f.path));
const { allFilePaths, allFileList, normalizedFileList, suffixIndex: index, resolveCache } = importCtx;
const { allFilePaths, allFileList, normalizedFileList, index, resolveCache } = importCtx;
let totalImportsFound = 0;
let totalImportsResolved = 0;
const effectiveRoot = repoRoot || '';
const configs: LanguageConfigs = {
tsconfigPaths: await loadTsconfigPaths(effectiveRoot),
goModule: await loadGoModulePath(effectiveRoot),
composerConfig: await loadComposerConfig(effectiveRoot),
swiftPackageConfig: await loadSwiftPackageConfig(effectiveRoot),
csharpConfigs: await loadCSharpProjectConfig(effectiveRoot),
};
const resolveCtx: ResolveCtx = { allFilePaths, allFileList, normalizedFileList, index, resolveCache };
// Helper: add an IMPORTS edge to the graph only (no ImportMap update)
const addImportGraphEdge = (filePath: string, resolvedPath: string) => {
const sourceId = generateId('File', filePath);
const targetId = generateId('File', resolvedPath);
const relId = generateId('IMPORTS', `${filePath}->${resolvedPath}`);
totalImportsResolved++;
graph.addRelationship({
id: relId,
sourceId,
targetId,
type: 'IMPORTS',
confidence: 1.0,
reason: '',
});
};
const addImportEdge = (filePath: string, resolvedPath: string) => {
addImportGraphEdge(filePath, resolvedPath);
if (!importMap.has(filePath)) {
importMap.set(filePath, new Set());
}
importMap.get(filePath)!.add(resolvedPath);
};
const configs = await loadImportConfigs(repoRoot || '');
const resolveCtx: ResolveCtx = { allFilePaths, allFileList, normalizedFileList, index, resolveCache, configs };
const { addImportEdge, addImportGraphEdge, getResolvedCount } = createImportEdgeHelpers(graph, importMap);
// Group by file for progress reporting (users see file count, not import count)
const importsByFile = new Map<string, ExtractedImport[]>();
@ -563,7 +368,7 @@ export const processImportsFromExtracted = async (
for (const imp of fileImports) {
totalImportsFound++;
const result = resolveLanguageImport(filePath, imp.rawImportPath, imp.language, configs, resolveCtx);
const result = importResolvers[imp.language](imp.rawImportPath, filePath, resolveCtx);
applyImportResult(result, filePath, importMap, packageMap, addImportEdge, addImportGraphEdge, imp.namedBindings, namedImportMap);
}
}
@ -571,6 +376,6 @@ export const processImportsFromExtracted = async (
onProgress?.(totalFiles, totalFiles);
if (isDev) {
console.log(`📊 Import processing (fast path): ${totalImportsResolved}/${totalImportsFound} imports resolved to graph edges`);
console.log(`📊 Import processing (fast path): ${getResolvedCount()}/${totalImportsFound} imports resolved to graph edges`);
}
};

View file

@ -0,0 +1,383 @@
/**
* Import Resolution Dispatch
*
* Per-language dispatch table for import resolution and named binding extraction.
* Replaces the 120-line if-chain in resolveLanguageImport() and the 7-branch
* dispatch in extractNamedBindings() with a single table lookup each.
*
* Follows the existing ExportChecker / CallRouter pattern:
* - Function aliases (not interfaces) to avoid megamorphic inline-cache issues
* - `satisfies Record<SupportedLanguages, ...>` for compile-time exhaustiveness
* - Const dispatch table — configs are accessed via ctx.configs at call time
*/
import { SupportedLanguages } from '../../config/supported-languages.js';
import type { SyntaxNode } from './utils.js';
import {
KOTLIN_EXTENSIONS,
appendKotlinWildcard,
resolveJvmWildcard,
resolveJvmMemberImport,
resolveGoPackageDir,
resolveGoPackage,
resolveCSharpImport as resolveCSharpImportHelper,
resolveCSharpNamespaceDir,
resolvePhpImport as resolvePhpImportHelper,
resolveRustImport as resolveRustImportHelper,
resolveRubyImport as resolveRubyImportHelper,
resolvePythonImport as resolvePythonImportHelper,
resolveImportPath,
} from './resolvers/index.js';
import type {
SuffixIndex,
TsconfigPaths,
GoModuleConfig,
CSharpProjectConfig,
ComposerConfig,
} from './resolvers/index.js';
import type { SwiftPackageConfig } from './language-config.js';
import {
extractTsNamedBindings,
extractPythonNamedBindings,
extractKotlinNamedBindings,
extractRustNamedBindings,
extractPhpNamedBindings,
extractCsharpNamedBindings,
extractJavaNamedBindings,
} from './named-binding-extraction.js';
import type { ImportResolutionContext } from './import-processor.js';
// ============================================================================
// Types
// ============================================================================
/**
* Result of resolving an import via language-specific dispatch.
* - 'files': resolved to one or more files -> add to ImportMap
* - 'package': resolved to a directory -> add graph edges + store dirSuffix in PackageMap
* - null: no resolution (external dependency, etc.)
*/
export type ImportResult =
| { kind: 'files'; files: string[] }
| { kind: 'package'; files: string[]; dirSuffix: string }
| null;
/** Bundled language-specific configs loaded once per ingestion run. */
export interface ImportConfigs {
tsconfigPaths: TsconfigPaths | null;
goModule: GoModuleConfig | null;
composerConfig: ComposerConfig | null;
swiftPackageConfig: SwiftPackageConfig | null;
csharpConfigs: CSharpProjectConfig[];
}
/** Full context for import resolution: file lookups + language configs. */
export interface ResolveCtx extends ImportResolutionContext {
configs: ImportConfigs;
}
/** Per-language import resolver -- function alias matching ExportChecker/CallRouter pattern. */
export type ImportResolverFn = (
rawImportPath: string,
filePath: string,
resolveCtx: ResolveCtx,
) => ImportResult;
/** A single named import binding: local name in the importing file and exported name from the source. */
export interface NamedBinding { local: string; exported: string }
/** Per-language named binding extractor -- optional (returns undefined if language has no named imports). */
type NamedBindingExtractorFn = (importNode: SyntaxNode) => NamedBinding[] | undefined;
// ============================================================================
// Import path preprocessing
// ============================================================================
/**
* Clean and preprocess a raw import source text into a resolved import path.
* Strips quotes/angle brackets (universal) and applies language-specific
* transformations (currently only Kotlin wildcard import detection).
*/
export function preprocessImportPath(
sourceText: string,
importNode: SyntaxNode,
language: SupportedLanguages,
): string | null {
const cleaned = sourceText.replace(/['"<>]/g, '');
// Defense-in-depth: reject null bytes and control characters (matches Ruby call-routing pattern)
if (!cleaned || cleaned.length > 2048 || /[\x00-\x1f]/.test(cleaned)) return null;
if (language === SupportedLanguages.Kotlin) {
return appendKotlinWildcard(cleaned, importNode);
}
return cleaned;
}
// ============================================================================
// Per-language resolver functions
// ============================================================================
/**
* Standard single-file resolution (TS/JS/C/C++ and fallback for other languages).
* Handles relative imports, tsconfig path aliases, and suffix matching.
*/
function resolveStandard(
rawImportPath: string,
filePath: string,
ctx: ResolveCtx,
language: SupportedLanguages,
): ImportResult {
const resolvedPath = resolveImportPath(
filePath,
rawImportPath,
ctx.allFilePaths,
ctx.allFileList,
ctx.normalizedFileList,
ctx.resolveCache,
language,
ctx.configs.tsconfigPaths,
ctx.index,
);
return resolvedPath ? { kind: 'files', files: [resolvedPath] } : null;
}
/** Java: JVM wildcard -> member import -> standard fallthrough */
function resolveJavaImport(
rawImportPath: string,
filePath: string,
ctx: ResolveCtx,
): ImportResult {
if (rawImportPath.endsWith('.*')) {
const matchedFiles = resolveJvmWildcard(rawImportPath, ctx.normalizedFileList, ctx.allFileList, ['.java'], ctx.index);
if (matchedFiles.length > 0) return { kind: 'files', files: matchedFiles };
} else {
const memberResolved = resolveJvmMemberImport(rawImportPath, ctx.normalizedFileList, ctx.allFileList, ['.java'], ctx.index);
if (memberResolved) return { kind: 'files', files: [memberResolved] };
}
return resolveStandard(rawImportPath, filePath, ctx, SupportedLanguages.Java);
}
/**
* Kotlin: JVM wildcard/member with Java-interop fallback -> top-level function imports -> standard.
* Kotlin can import from .kt/.kts files OR from .java files (Java interop).
*/
function resolveKotlinImport(
rawImportPath: string,
filePath: string,
ctx: ResolveCtx,
): ImportResult {
if (rawImportPath.endsWith('.*')) {
const matchedFiles = resolveJvmWildcard(rawImportPath, ctx.normalizedFileList, ctx.allFileList, KOTLIN_EXTENSIONS, ctx.index);
if (matchedFiles.length === 0) {
const javaMatches = resolveJvmWildcard(rawImportPath, ctx.normalizedFileList, ctx.allFileList, ['.java'], ctx.index);
if (javaMatches.length > 0) return { kind: 'files', files: javaMatches };
}
if (matchedFiles.length > 0) return { kind: 'files', files: matchedFiles };
} else {
let memberResolved = resolveJvmMemberImport(rawImportPath, ctx.normalizedFileList, ctx.allFileList, KOTLIN_EXTENSIONS, ctx.index);
if (!memberResolved) {
memberResolved = resolveJvmMemberImport(rawImportPath, ctx.normalizedFileList, ctx.allFileList, ['.java'], ctx.index);
}
if (memberResolved) return { kind: 'files', files: [memberResolved] };
// Kotlin: top-level function imports (e.g. import models.getUser) have only 2 segments,
// which resolveJvmMemberImport skips (requires >=3). Fall back to package-directory scan
// for lowercase last segments (function/property imports). Uppercase last segments
// (class imports like models.User) fall through to standard suffix resolution.
const segments = rawImportPath.split('.');
const lastSeg = segments[segments.length - 1];
if (segments.length >= 2 && lastSeg[0] && lastSeg[0] === lastSeg[0].toLowerCase()) {
const pkgWildcard = segments.slice(0, -1).join('.') + '.*';
let dirFiles = resolveJvmWildcard(pkgWildcard, ctx.normalizedFileList, ctx.allFileList, KOTLIN_EXTENSIONS, ctx.index);
if (dirFiles.length === 0) {
dirFiles = resolveJvmWildcard(pkgWildcard, ctx.normalizedFileList, ctx.allFileList, ['.java'], ctx.index);
}
if (dirFiles.length > 0) return { kind: 'files', files: dirFiles };
}
}
return resolveStandard(rawImportPath, filePath, ctx, SupportedLanguages.Kotlin);
}
/** Go: package-level imports via go.mod module path. */
function resolveGoImport(
rawImportPath: string,
filePath: string,
ctx: ResolveCtx,
): ImportResult {
const goModule = ctx.configs.goModule;
if (goModule && rawImportPath.startsWith(goModule.modulePath)) {
const pkgSuffix = resolveGoPackageDir(rawImportPath, goModule);
if (pkgSuffix) {
const pkgFiles = resolveGoPackage(rawImportPath, goModule, ctx.normalizedFileList, ctx.allFileList);
if (pkgFiles.length > 0) {
return { kind: 'package', files: pkgFiles, dirSuffix: pkgSuffix };
}
}
// Fall through if no files found (package might be external)
}
return resolveStandard(rawImportPath, filePath, ctx, SupportedLanguages.Go);
}
/** C#: namespace-based resolution via .csproj configs, with suffix-match fallback. */
function resolveCSharpImportDispatch(
rawImportPath: string,
filePath: string,
ctx: ResolveCtx,
): ImportResult {
const csharpConfigs = ctx.configs.csharpConfigs;
if (csharpConfigs.length > 0) {
const resolvedFiles = resolveCSharpImportHelper(rawImportPath, csharpConfigs, ctx.normalizedFileList, ctx.allFileList, ctx.index);
if (resolvedFiles.length > 1) {
const dirSuffix = resolveCSharpNamespaceDir(rawImportPath, csharpConfigs);
if (dirSuffix) {
return { kind: 'package', files: resolvedFiles, dirSuffix };
}
}
if (resolvedFiles.length > 0) return { kind: 'files', files: resolvedFiles };
}
return resolveStandard(rawImportPath, filePath, ctx, SupportedLanguages.CSharp);
}
/** PHP: namespace-based resolution via composer.json PSR-4. */
function resolvePhpImportDispatch(
rawImportPath: string,
_filePath: string,
ctx: ResolveCtx,
): ImportResult {
const resolved = resolvePhpImportHelper(rawImportPath, ctx.configs.composerConfig, ctx.allFilePaths, ctx.normalizedFileList, ctx.allFileList, ctx.index);
return resolved ? { kind: 'files', files: [resolved] } : null;
}
/** Swift: module imports via Package.swift target map. */
function resolveSwiftImportDispatch(
rawImportPath: string,
_filePath: string,
ctx: ResolveCtx,
): ImportResult {
const swiftPackageConfig = ctx.configs.swiftPackageConfig;
if (swiftPackageConfig) {
const targetDir = swiftPackageConfig.targets.get(rawImportPath);
if (targetDir) {
const dirPrefix = targetDir + '/';
const files: string[] = [];
for (let i = 0; i < ctx.normalizedFileList.length; i++) {
if (ctx.normalizedFileList[i].startsWith(dirPrefix) && ctx.normalizedFileList[i].endsWith('.swift')) {
files.push(ctx.allFileList[i]);
}
}
if (files.length > 0) return { kind: 'files', files };
}
}
return null; // External framework (Foundation, UIKit, etc.)
}
/**
* Python: relative imports (PEP 328) + proximity-based bare imports.
* Falls through to standard suffix resolution when proximity finds no match.
*/
function resolvePythonImportDispatch(
rawImportPath: string,
filePath: string,
ctx: ResolveCtx,
): ImportResult {
const resolved = resolvePythonImportHelper(filePath, rawImportPath, ctx.allFilePaths);
if (resolved) return { kind: 'files', files: [resolved] };
if (rawImportPath.startsWith('.')) return null; // relative but unresolved -- don't suffix-match
return resolveStandard(rawImportPath, filePath, ctx, SupportedLanguages.Python);
}
/** Ruby: require / require_relative. */
function resolveRubyImportDispatch(
rawImportPath: string,
_filePath: string,
ctx: ResolveCtx,
): ImportResult {
const resolved = resolveRubyImportHelper(rawImportPath, ctx.normalizedFileList, ctx.allFileList, ctx.index);
return resolved ? { kind: 'files', files: [resolved] } : null;
}
/** Rust: expand grouped imports: use {crate::a, crate::b} and use crate::models::{User, Repo}. */
function resolveRustImportDispatch(
rawImportPath: string,
filePath: string,
ctx: ResolveCtx,
): ImportResult {
// Top-level grouped: use {crate::a, crate::b}
if (rawImportPath.startsWith('{') && rawImportPath.endsWith('}')) {
const inner = rawImportPath.slice(1, -1);
const parts = inner.split(',').map(p => p.trim()).filter(Boolean);
const resolved: string[] = [];
for (const part of parts) {
const r = resolveRustImportHelper(filePath, part, ctx.allFilePaths);
if (r) resolved.push(r);
}
return resolved.length > 0 ? { kind: 'files', files: resolved } : null;
}
// Scoped grouped: use crate::models::{User, Repo}
const braceIdx = rawImportPath.indexOf('::{');
if (braceIdx !== -1 && rawImportPath.endsWith('}')) {
const pathPrefix = rawImportPath.substring(0, braceIdx);
const braceContent = rawImportPath.substring(braceIdx + 3, rawImportPath.length - 1);
const items = braceContent.split(',').map(s => s.trim()).filter(Boolean);
const resolved: string[] = [];
for (const item of items) {
// Handle `use crate::models::{User, Repo as R}` — strip alias for resolution
const itemName = item.includes(' as ') ? item.split(' as ')[0].trim() : item;
const r = resolveRustImportHelper(filePath, `${pathPrefix}::${itemName}`, ctx.allFilePaths);
if (r) resolved.push(r);
}
if (resolved.length > 0) return { kind: 'files', files: resolved };
// Fallback: resolve the prefix path itself (e.g. crate::models -> models.rs)
const prefixResult = resolveRustImportHelper(filePath, pathPrefix, ctx.allFilePaths);
if (prefixResult) return { kind: 'files', files: [prefixResult] };
}
return resolveStandard(rawImportPath, filePath, ctx, SupportedLanguages.Rust);
}
// ============================================================================
// Dispatch tables
// ============================================================================
/**
* Per-language import resolver dispatch table.
* Configs are accessed via ctx.configs at call time — no factory closure needed.
* Each resolver encapsulates the full resolution flow for its language, including
* fallthrough to standard resolution where appropriate.
*/
export const importResolvers = {
[SupportedLanguages.JavaScript]: (raw, fp, ctx) => resolveStandard(raw, fp, ctx, SupportedLanguages.JavaScript),
[SupportedLanguages.TypeScript]: (raw, fp, ctx) => resolveStandard(raw, fp, ctx, SupportedLanguages.TypeScript),
[SupportedLanguages.Python]: (raw, fp, ctx) => resolvePythonImportDispatch(raw, fp, ctx),
[SupportedLanguages.Java]: (raw, fp, ctx) => resolveJavaImport(raw, fp, ctx),
[SupportedLanguages.C]: (raw, fp, ctx) => resolveStandard(raw, fp, ctx, SupportedLanguages.C),
[SupportedLanguages.CPlusPlus]: (raw, fp, ctx) => resolveStandard(raw, fp, ctx, SupportedLanguages.CPlusPlus),
[SupportedLanguages.CSharp]: (raw, fp, ctx) => resolveCSharpImportDispatch(raw, fp, ctx),
[SupportedLanguages.Go]: (raw, fp, ctx) => resolveGoImport(raw, fp, ctx),
[SupportedLanguages.Ruby]: (raw, fp, ctx) => resolveRubyImportDispatch(raw, fp, ctx),
[SupportedLanguages.Rust]: (raw, fp, ctx) => resolveRustImportDispatch(raw, fp, ctx),
[SupportedLanguages.PHP]: (raw, fp, ctx) => resolvePhpImportDispatch(raw, fp, ctx),
[SupportedLanguages.Kotlin]: (raw, fp, ctx) => resolveKotlinImport(raw, fp, ctx),
[SupportedLanguages.Swift]: (raw, fp, ctx) => resolveSwiftImportDispatch(raw, fp, ctx),
} satisfies Record<SupportedLanguages, ImportResolverFn>;
/**
* Per-language named binding extractor dispatch table.
* Languages with whole-module import semantics (Go, Ruby, C/C++, Swift) return undefined --
* their bindings are synthesized post-parse by synthesizeWildcardImportBindings() in pipeline.ts.
*/
export const namedBindingExtractors = {
[SupportedLanguages.JavaScript]: extractTsNamedBindings,
[SupportedLanguages.TypeScript]: extractTsNamedBindings,
[SupportedLanguages.Python]: extractPythonNamedBindings,
[SupportedLanguages.Java]: extractJavaNamedBindings,
[SupportedLanguages.C]: undefined,
[SupportedLanguages.CPlusPlus]: undefined,
[SupportedLanguages.CSharp]: extractCsharpNamedBindings,
[SupportedLanguages.Go]: undefined,
[SupportedLanguages.Ruby]: undefined,
[SupportedLanguages.Rust]: extractRustNamedBindings,
[SupportedLanguages.PHP]: extractPhpNamedBindings,
[SupportedLanguages.Kotlin]: extractKotlinNamedBindings,
[SupportedLanguages.Swift]: undefined,
} satisfies Record<SupportedLanguages, NamedBindingExtractorFn | undefined>;

View file

@ -1,5 +1,6 @@
import fs from 'fs/promises';
import path from 'path';
import type { ImportConfigs } from './import-resolution.js';
const isDev = process.env.NODE_ENV === 'development';
@ -213,3 +214,18 @@ export async function loadSwiftPackageConfig(repoRoot: string): Promise<SwiftPac
}
return null;
}
// ============================================================================
// BUNDLED CONFIG LOADER
// ============================================================================
/** Load all language-specific configs once for an ingestion run. */
export async function loadImportConfigs(repoRoot: string): Promise<ImportConfigs> {
return {
tsconfigPaths: await loadTsconfigPaths(repoRoot),
goModule: await loadGoModulePath(repoRoot),
composerConfig: await loadComposerConfig(repoRoot),
swiftPackageConfig: await loadSwiftPackageConfig(repoRoot),
csharpConfigs: await loadCSharpProjectConfig(repoRoot),
};
}

View file

@ -0,0 +1,157 @@
/**
* Markdown Processor
*
* Extracts structure from .md files using regex (no tree-sitter dependency).
* Creates Section nodes for headings with hierarchy, and IMPORTS edges for
* cross-file links.
*/
import path from 'node:path';
import { generateId } from '../../lib/utils.js';
import { KnowledgeGraph, GraphNode, GraphRelationship } from '../graph/types.js';
const HEADING_RE = /^(#{1,6})\s+(.+)$/;
const LINK_RE = /\[([^\]]*)\]\(([^)]+)\)/g;
const MD_EXTENSIONS = new Set(['.md', '.mdx']);
interface MdFile {
path: string;
content: string;
}
export const processMarkdown = (
graph: KnowledgeGraph,
files: MdFile[],
allPathSet: Set<string>,
): { sections: number; links: number } => {
let totalSections = 0;
let totalLinks = 0;
for (const file of files) {
const ext = path.extname(file.path).toLowerCase();
if (!MD_EXTENSIONS.has(ext)) continue;
const fileNodeId = generateId('File', file.path);
// Skip if file node doesn't exist (shouldn't happen, structure-processor creates it)
if (!graph.getNode(fileNodeId)) continue;
const lines = file.content.split('\n');
// --- Extract headings and build hierarchy ---
// First pass: collect all heading positions so we can compute endLine spans
const headings: { level: number; heading: string; lineNum: number }[] = [];
for (let i = 0; i < lines.length; i++) {
const match = lines[i].match(HEADING_RE);
if (!match) continue;
headings.push({
level: match[1].length,
heading: match[2].trim(),
lineNum: i + 1, // 1-indexed
});
}
// Second pass: create nodes with proper endLine spans
const sectionStack: { level: number; id: string }[] = [];
for (let h = 0; h < headings.length; h++) {
const { level, heading, lineNum } = headings[h];
// endLine = line before next heading at same or higher level, or EOF
let endLine = lines.length;
for (let j = h + 1; j < headings.length; j++) {
if (headings[j].level <= level) {
endLine = headings[j].lineNum - 1;
break;
}
}
const sectionId = generateId('Section', `${file.path}:L${lineNum}:${heading}`);
const node: GraphNode = {
id: sectionId,
label: 'Section',
properties: {
name: heading,
filePath: file.path,
startLine: lineNum,
endLine,
level,
description: `h${level}`,
},
};
graph.addNode(node);
totalSections++;
// Find parent: pop stack until we find a level strictly less than current
while (sectionStack.length > 0 && sectionStack[sectionStack.length - 1].level >= level) {
sectionStack.pop();
}
const parentId = sectionStack.length > 0
? sectionStack[sectionStack.length - 1].id
: fileNodeId;
graph.addRelationship({
id: generateId('CONTAINS', `${parentId}->${sectionId}`),
type: 'CONTAINS',
sourceId: parentId,
targetId: sectionId,
confidence: 1.0,
reason: 'markdown-heading',
});
sectionStack.push({ level, id: sectionId });
}
// --- Extract links to other files in the repo ---
const fileDir = path.dirname(file.path);
const seenLinks = new Set<string>();
let linkMatch: RegExpExecArray | null;
LINK_RE.lastIndex = 0;
while ((linkMatch = LINK_RE.exec(file.content)) !== null) {
const href = linkMatch[2];
// Skip external URLs, anchors, and mailto
if (href.startsWith('http://') || href.startsWith('https://') ||
href.startsWith('#') || href.startsWith('mailto:')) {
continue;
}
// Strip anchor fragments from local links
const cleanHref = href.split('#')[0];
if (!cleanHref) continue;
// Resolve relative to the file's directory, then normalize
const resolved = path.posix.normalize(path.posix.join(fileDir, cleanHref));
if (allPathSet.has(resolved)) {
const targetFileId = generateId('File', resolved);
// Skip if target file node doesn't exist
if (!graph.getNode(targetFileId)) continue;
// Dedup: skip if we've already linked this file pair
const linkKey = `${fileNodeId}->${targetFileId}`;
if (seenLinks.has(linkKey)) continue;
seenLinks.add(linkKey);
const relId = generateId('IMPORTS', linkKey);
graph.addRelationship({
id: relId,
type: 'IMPORTS',
sourceId: fileNodeId,
targetId: targetFileId,
confidence: 0.8,
reason: 'markdown-link',
});
totalLinks++;
}
}
}
return { sections: totalSections, links: totalLinks };
};

View file

@ -1,6 +1,8 @@
import { SupportedLanguages } from '../../config/supported-languages.js';
import type { SymbolTable, SymbolDefinition } from './symbol-table.js';
import type { NamedImportMap } from './import-processor.js';
import type { NamedBinding } from './import-resolution.js';
import type { SyntaxNode } from './utils.js';
import { findChild } from './resolvers/utils.js';
/**
* Walk a named-binding re-export chain through NamedImportMap.
@ -54,52 +56,14 @@ export function walkBindingChain(
return null;
}
/**
* Extract named bindings from an import AST node.
* Returns undefined if the import is not a named import (e.g., import * or default).
*
* TS: import { User, Repo as R } from './models'
* → [{local:'User', exported:'User'}, {local:'R', exported:'Repo'}]
*
* Python: from models import User, Repo as R
* → [{local:'User', exported:'User'}, {local:'R', exported:'Repo'}]
*/
export function extractNamedBindings(
importNode: any,
language: SupportedLanguages,
): { local: string; exported: string }[] | undefined {
if (language === SupportedLanguages.TypeScript || language === SupportedLanguages.JavaScript) {
return extractTsNamedBindings(importNode);
}
if (language === SupportedLanguages.Python) {
return extractPythonNamedBindings(importNode);
}
if (language === SupportedLanguages.Kotlin) {
return extractKotlinNamedBindings(importNode);
}
if (language === SupportedLanguages.Rust) {
return extractRustNamedBindings(importNode);
}
if (language === SupportedLanguages.PHP) {
return extractPhpNamedBindings(importNode);
}
if (language === SupportedLanguages.CSharp) {
return extractCsharpNamedBindings(importNode);
}
if (language === SupportedLanguages.Java) {
return extractJavaNamedBindings(importNode);
}
return undefined;
}
export function extractTsNamedBindings(importNode: any): { local: string; exported: string }[] | undefined {
export function extractTsNamedBindings(importNode: SyntaxNode): NamedBinding[] | undefined {
// import_statement > import_clause > named_imports > import_specifier*
const importClause = findChild(importNode, 'import_clause');
if (importClause) {
const namedImports = findChild(importClause, 'named_imports');
if (!namedImports) return undefined; // default import, namespace import, or side-effect
const bindings: { local: string; exported: string }[] = [];
const bindings: NamedBinding[] = [];
for (let i = 0; i < namedImports.namedChildCount; i++) {
const specifier = namedImports.namedChild(i);
if (specifier?.type !== 'import_specifier') continue;
@ -123,7 +87,7 @@ export function extractTsNamedBindings(importNode: any): { local: string; export
// Re-export: export { X } from './y' → export_statement > export_clause > export_specifier
const exportClause = findChild(importNode, 'export_clause');
if (exportClause) {
const bindings: { local: string; exported: string }[] = [];
const bindings: NamedBinding[] = [];
for (let i = 0; i < exportClause.namedChildCount; i++) {
const specifier = exportClause.namedChild(i);
if (specifier?.type !== 'export_specifier') continue;
@ -150,11 +114,11 @@ export function extractTsNamedBindings(importNode: any): { local: string; export
return undefined;
}
export function extractPythonNamedBindings(importNode: any): { local: string; exported: string }[] | undefined {
export function extractPythonNamedBindings(importNode: SyntaxNode): NamedBinding[] | undefined {
// Only from import_from_statement, not plain import_statement
if (importNode.type !== 'import_from_statement') return undefined;
const bindings: { local: string; exported: string }[] = [];
const bindings: NamedBinding[] = [];
for (let i = 0; i < importNode.namedChildCount; i++) {
const child = importNode.namedChild(i);
if (!child) continue;
@ -182,7 +146,7 @@ export function extractPythonNamedBindings(importNode: any): { local: string; ex
return bindings.length > 0 ? bindings : undefined;
}
export function extractKotlinNamedBindings(importNode: any): { local: string; exported: string }[] | undefined {
export function extractKotlinNamedBindings(importNode: SyntaxNode): NamedBinding[] | undefined {
// import_header > identifier + import_alias > simple_identifier
if (importNode.type !== 'import_header') return undefined;
@ -201,26 +165,32 @@ export function extractKotlinNamedBindings(importNode: any): { local: string; ex
}
// Non-aliased: import com.example.User → local="User", exported="User"
// Also handles top-level function imports: import models.getUser → local="getUser"
// Skip wildcard imports (ending in *)
if (fullText.endsWith('.*') || fullText.endsWith('*')) return undefined;
// Skip lowercase last segments — those are member/function imports (e.g.,
// import util.OneArg.writeAudit), not class imports. Multiple member imports
// Skip class-member imports (e.g., import util.OneArg.writeAudit) where the
// second-to-last segment is PascalCase (a class name). Multiple member imports
// with the same function name would collide in NamedImportMap, breaking
// arity-based disambiguation.
if (exportedName[0] && exportedName[0] === exportedName[0].toLowerCase()) return undefined;
// arity-based disambiguation. Top-level function imports (import models.getUser)
// and class imports (import models.User) have package-only prefixes.
const segments = fullText.split('.');
if (segments.length >= 3) {
const parentSegment = segments[segments.length - 2];
if (parentSegment[0] && parentSegment[0] === parentSegment[0].toUpperCase()) return undefined;
}
return [{ local: exportedName, exported: exportedName }];
}
export function extractRustNamedBindings(importNode: any): { local: string; exported: string }[] | undefined {
export function extractRustNamedBindings(importNode: SyntaxNode): NamedBinding[] | undefined {
// use_declaration may contain use_as_clause at any depth
if (importNode.type !== 'use_declaration') return undefined;
const bindings: { local: string; exported: string }[] = [];
const bindings: NamedBinding[] = [];
collectRustBindings(importNode, bindings);
return bindings.length > 0 ? bindings : undefined;
}
function collectRustBindings(node: any, bindings: { local: string; exported: string }[]): void {
function collectRustBindings(node: SyntaxNode, bindings: NamedBinding[]): void {
if (node.type === 'use_as_clause') {
// First identifier = exported name, second identifier = local alias
const idents: string[] = [];
@ -278,15 +248,22 @@ function collectRustBindings(node: any, bindings: { local: string; exported: str
}
}
export function extractPhpNamedBindings(importNode: any): { local: string; exported: string }[] | undefined {
export function extractPhpNamedBindings(importNode: SyntaxNode): NamedBinding[] | undefined {
// namespace_use_declaration > namespace_use_clause* (flat)
// namespace_use_declaration > namespace_use_group > namespace_use_clause* (grouped)
if (importNode.type !== 'namespace_use_declaration') return undefined;
const bindings: { local: string; exported: string }[] = [];
// Skip 'use function' and 'use const' declarations — these import callables/constants,
// not class types, and should not be added to namedImportMap as type bindings.
const useTypeNode = importNode.childForFieldName?.('type');
if (useTypeNode && (useTypeNode.text === 'function' || useTypeNode.text === 'const')) {
return undefined;
}
const bindings: NamedBinding[] = [];
// Collect all clauses — from direct children AND from namespace_use_group
const clauses: any[] = [];
const clauses: SyntaxNode[] = [];
for (let i = 0; i < importNode.namedChildCount; i++) {
const child = importNode.namedChild(i);
if (child?.type === 'namespace_use_clause') {
@ -301,8 +278,8 @@ export function extractPhpNamedBindings(importNode: any): { local: string; expor
for (const clause of clauses) {
// Flat imports: qualified_name + name (alias)
let qualifiedName: any = null;
const names: any[] = [];
let qualifiedName: SyntaxNode | null = null;
const names: SyntaxNode[] = [];
for (let j = 0; j < clause.namedChildCount; j++) {
const child = clause.namedChild(j);
if (child?.type === 'qualified_name') qualifiedName = child;
@ -330,35 +307,55 @@ export function extractPhpNamedBindings(importNode: any): { local: string; expor
return bindings.length > 0 ? bindings : undefined;
}
export function extractCsharpNamedBindings(importNode: any): { local: string; exported: string }[] | undefined {
// using_directive with identifier (alias) + qualified_name (target)
export function extractCsharpNamedBindings(importNode: SyntaxNode): NamedBinding[] | undefined {
// using_directive — three forms:
// using Alias = NS.Type; → aliasIdent + qualifiedName
// using static NS.Type; → static + qualifiedName (no alias)
// using NS; → qualifiedName only (namespace, not capturable)
if (importNode.type !== 'using_directive') return undefined;
let aliasIdent: any = null;
let qualifiedName: any = null;
let aliasIdent: SyntaxNode | null = null;
let qualifiedName: SyntaxNode | null = null;
let isStatic = false;
for (let i = 0; i < importNode.childCount; i++) {
const child = importNode.child(i);
if (child?.text === 'static') isStatic = true;
}
for (let i = 0; i < importNode.namedChildCount; i++) {
const child = importNode.namedChild(i);
if (child?.type === 'identifier' && !aliasIdent) aliasIdent = child;
else if (child?.type === 'qualified_name') qualifiedName = child;
}
if (!aliasIdent || !qualifiedName) return undefined;
// Form 1: using Alias = NS.Type;
if (aliasIdent && qualifiedName) {
const fullText = qualifiedName.text;
const exportedName = fullText.includes('.') ? fullText.split('.').pop()! : fullText;
return [{ local: aliasIdent.text, exported: exportedName }];
}
const fullText = qualifiedName.text;
const exportedName = fullText.includes('.') ? fullText.split('.').pop()! : fullText;
// Form 2: using static NS.Type; — last segment is the class name
if (isStatic && qualifiedName) {
const fullText = qualifiedName.text;
const lastSegment = fullText.includes('.') ? fullText.split('.').pop()! : fullText;
return [{ local: lastSegment, exported: lastSegment }];
}
return [{ local: aliasIdent.text, exported: exportedName }];
// Form 3: using NS; — namespace import, can't resolve to per-symbol bindings
return undefined;
}
export function extractJavaNamedBindings(importNode: any): { local: string; exported: string }[] | undefined {
export function extractJavaNamedBindings(importNode: SyntaxNode): NamedBinding[] | undefined {
// import_declaration > scoped_identifier "com.example.models.User"
// Wildcard imports (.*) don't produce named bindings
if (importNode.type !== 'import_declaration') return undefined;
// Check for asterisk (wildcard import) — skip those
// Check for asterisk (wildcard import) and static modifier
let isStatic = false;
for (let i = 0; i < importNode.childCount; i++) {
const child = importNode.child(i);
if (child?.type === 'asterisk') return undefined;
if (child?.text === 'static') isStatic = true;
}
const scopedId = findChild(importNode, 'scoped_identifier');
@ -368,17 +365,11 @@ export function extractJavaNamedBindings(importNode: any): { local: string; expo
const lastDot = fullText.lastIndexOf('.');
if (lastDot === -1) return undefined;
const className = fullText.slice(lastDot + 1);
// Skip lowercase names — those are package imports, not class imports
if (className[0] && className[0] === className[0].toLowerCase()) return undefined;
const name = fullText.slice(lastDot + 1);
// Non-static: skip lowercase names — those are package imports, not class imports.
// Static: allow lowercase — `import static models.UserFactory.getUser` imports a method.
if (!isStatic && name[0] && name[0] === name[0].toLowerCase()) return undefined;
return [{ local: className, exported: className }];
return [{ local: name, exported: name }];
}
function findChild(node: any, type: string): any {
for (let i = 0; i < node.namedChildCount; i++) {
const child = node.namedChild(i);
if (child?.type === type) return child;
}
return null;
}

View file

@ -1,16 +1,17 @@
import { KnowledgeGraph, GraphNode, GraphRelationship } from '../graph/types.js';
import { KnowledgeGraph, GraphNode, GraphRelationship, type NodeLabel } from '../graph/types.js';
import Parser from 'tree-sitter';
import { loadParser, loadLanguage } from '../tree-sitter/parser-loader.js';
import { loadParser, loadLanguage, isLanguageAvailable } from '../tree-sitter/parser-loader.js';
import { LANGUAGE_QUERIES } from './tree-sitter-queries.js';
import { generateId } from '../../lib/utils.js';
import { SymbolTable } from './symbol-table.js';
import { ASTCache } from './ast-cache.js';
import { getLanguageFromFilename, yieldToEventLoop, DEFINITION_CAPTURE_KEYS, getDefinitionNodeFromCaptures, findEnclosingClassId, extractMethodSignature } from './utils.js';
import { getLanguageFromFilename, yieldToEventLoop, getDefinitionNodeFromCaptures, findEnclosingClassId, extractMethodSignature, getLabelFromCaptures } from './utils.js';
import { extractPropertyDeclaredType } from './type-extractors/shared.js';
import { isNodeExported } from './export-detection.js';
import { detectFrameworkFromAST } from './framework-detection.js';
import { typeConfigs } from './type-extractors/index.js';
import { WorkerPool } from './workers/worker-pool.js';
import type { ParseWorkerResult, ParseWorkerInput, ExtractedImport, ExtractedCall, ExtractedHeritage, ExtractedRoute, FileConstructorBindings } from './workers/parse-worker.js';
import type { ParseWorkerResult, ParseWorkerInput, ExtractedImport, ExtractedCall, ExtractedAssignment, ExtractedHeritage, ExtractedRoute, FileConstructorBindings, FileTypeEnvBindings } from './workers/parse-worker.js';
import { getTreeSitterBufferSize, TREE_SITTER_MAX_BUFFER } from './constants.js';
export type FileProgressCallback = (current: number, total: number, filePath: string) => void;
@ -18,15 +19,13 @@ export type FileProgressCallback = (current: number, total: number, filePath: st
export interface WorkerExtractedData {
imports: ExtractedImport[];
calls: ExtractedCall[];
assignments: ExtractedAssignment[];
heritage: ExtractedHeritage[];
routes: ExtractedRoute[];
constructorBindings: FileConstructorBindings[];
typeEnvBindings: FileTypeEnvBindings[];
}
// isNodeExported imported from ./export-detection.js (shared module)
// Re-export for backward compatibility with any external consumers
export { isNodeExported } from './export-detection.js';
// ============================================================================
// Worker-based parallel parsing
// ============================================================================
@ -46,7 +45,7 @@ const processParsingWithWorkers = async (
if (lang) parseableFiles.push({ path: file.path, content: file.content });
}
if (parseableFiles.length === 0) return { imports: [], calls: [], heritage: [], routes: [], constructorBindings: [] };
if (parseableFiles.length === 0) return { imports: [], calls: [], assignments: [], heritage: [], routes: [], constructorBindings: [], typeEnvBindings: [] };
const total = files.length;
@ -61,9 +60,11 @@ const processParsingWithWorkers = async (
// Merge results from all workers into graph and symbol table
const allImports: ExtractedImport[] = [];
const allCalls: ExtractedCall[] = [];
const allAssignments: ExtractedAssignment[] = [];
const allHeritage: ExtractedHeritage[] = [];
const allRoutes: ExtractedRoute[] = [];
const allConstructorBindings: FileConstructorBindings[] = [];
const allTypeEnvBindings: FileTypeEnvBindings[] = [];
for (const result of chunkResults) {
for (const node of result.nodes) {
graph.addNode({
@ -80,21 +81,40 @@ const processParsingWithWorkers = async (
for (const sym of result.symbols) {
symbolTable.add(sym.filePath, sym.name, sym.nodeId, sym.type, {
parameterCount: sym.parameterCount,
requiredParameterCount: sym.requiredParameterCount,
parameterTypes: sym.parameterTypes,
returnType: sym.returnType,
declaredType: sym.declaredType,
ownerId: sym.ownerId,
});
}
allImports.push(...result.imports);
allCalls.push(...result.calls);
allAssignments.push(...result.assignments);
allHeritage.push(...result.heritage);
allRoutes.push(...result.routes);
allConstructorBindings.push(...result.constructorBindings);
allTypeEnvBindings.push(...result.typeEnvBindings);
}
// Merge and log skipped languages from workers
const skippedLanguages = new Map<string, number>();
for (const result of chunkResults) {
for (const [lang, count] of Object.entries(result.skippedLanguages)) {
skippedLanguages.set(lang, (skippedLanguages.get(lang) || 0) + count);
}
}
if (skippedLanguages.size > 0) {
const summary = Array.from(skippedLanguages.entries())
.map(([lang, count]) => `${lang}: ${count}`)
.join(', ');
console.warn(` Skipped unsupported languages: ${summary}`);
}
// Final progress
onFileProgress?.(total, total, 'done');
return { imports: allImports, calls: allCalls, heritage: allHeritage, routes: allRoutes, constructorBindings: allConstructorBindings };
return { imports: allImports, calls: allCalls, assignments: allAssignments, heritage: allHeritage, routes: allRoutes, constructorBindings: allConstructorBindings, typeEnvBindings: allTypeEnvBindings };
};
// ============================================================================
@ -110,6 +130,7 @@ const processParsingSequential = async (
) => {
const parser = await loadParser();
const total = files.length;
const skippedLanguages = new Map<string, number>();
for (let i = 0; i < files.length; i++) {
const file = files[i];
@ -122,13 +143,19 @@ const processParsingSequential = async (
if (!language) continue;
// Skip unsupported languages (e.g. Swift when tree-sitter-swift not installed)
if (!isLanguageAvailable(language)) {
skippedLanguages.set(language, (skippedLanguages.get(language) || 0) + 1);
continue;
}
// Skip files larger than the max tree-sitter buffer (32 MB)
if (file.content.length > TREE_SITTER_MAX_BUFFER) continue;
try {
await loadLanguage(language, file.path);
} catch {
continue; // parser unavailable — already warned in pipeline
continue; // parser unavailable — safety net
}
let tree;
@ -164,44 +191,14 @@ const processParsingSequential = async (
captureMap[c.name] = c.node;
});
if (captureMap['import']) {
return;
}
if (captureMap['call']) {
return;
}
const nodeLabel = getLabelFromCaptures(captureMap, language);
if (!nodeLabel) return;
const nameNode = captureMap['name'];
// Synthesize name for constructors without explicit @name capture (e.g. Swift init)
if (!nameNode && !captureMap['definition.constructor']) return;
if (!nameNode && nodeLabel !== 'Constructor') return;
const nodeName = nameNode ? nameNode.text : 'init';
let nodeLabel = 'CodeElement';
if (captureMap['definition.function']) nodeLabel = 'Function';
else if (captureMap['definition.class']) nodeLabel = 'Class';
else if (captureMap['definition.interface']) nodeLabel = 'Interface';
else if (captureMap['definition.method']) nodeLabel = 'Method';
else if (captureMap['definition.struct']) nodeLabel = 'Struct';
else if (captureMap['definition.enum']) nodeLabel = 'Enum';
else if (captureMap['definition.namespace']) nodeLabel = 'Namespace';
else if (captureMap['definition.module']) nodeLabel = 'Module';
else if (captureMap['definition.trait']) nodeLabel = 'Trait';
else if (captureMap['definition.impl']) nodeLabel = 'Impl';
else if (captureMap['definition.type']) nodeLabel = 'TypeAlias';
else if (captureMap['definition.const']) nodeLabel = 'Const';
else if (captureMap['definition.static']) nodeLabel = 'Static';
else if (captureMap['definition.typedef']) nodeLabel = 'Typedef';
else if (captureMap['definition.macro']) nodeLabel = 'Macro';
else if (captureMap['definition.union']) nodeLabel = 'Union';
else if (captureMap['definition.property']) nodeLabel = 'Property';
else if (captureMap['definition.record']) nodeLabel = 'Record';
else if (captureMap['definition.delegate']) nodeLabel = 'Delegate';
else if (captureMap['definition.annotation']) nodeLabel = 'Annotation';
else if (captureMap['definition.constructor']) nodeLabel = 'Constructor';
else if (captureMap['definition.template']) nodeLabel = 'Template';
const definitionNodeForRange = getDefinitionNodeFromCaptures(captureMap);
const startLine = definitionNodeForRange ? definitionNodeForRange.startPosition.row : (nameNode ? nameNode.startPosition.row : 0);
const nodeId = generateId(nodeLabel, `${file.path}:${nodeName}`);
@ -217,10 +214,12 @@ const processParsingSequential = async (
: undefined;
// Language-specific return type fallback (e.g. Ruby YARD @return [Type])
if (methodSig && !methodSig.returnType && definitionNode) {
// Also upgrades uninformative AST types like PHP `array` with PHPDoc `@return User[]`
if (methodSig && (!methodSig.returnType || methodSig.returnType === 'array' || methodSig.returnType === 'iterable') && definitionNode) {
const tc = typeConfigs[language as keyof typeof typeConfigs];
if (tc?.extractReturnType) {
methodSig.returnType = tc.extractReturnType(definitionNode);
const docReturn = tc.extractReturnType(definitionNode);
if (docReturn) methodSig.returnType = docReturn;
}
}
@ -240,6 +239,8 @@ const processParsingSequential = async (
} : {}),
...(methodSig ? {
parameterCount: methodSig.parameterCount,
...(methodSig.requiredParameterCount !== undefined ? { requiredParameterCount: methodSig.requiredParameterCount } : {}),
...(methodSig.parameterTypes ? { parameterTypes: methodSig.parameterTypes } : {}),
returnType: methodSig.returnType,
} : {}),
},
@ -252,9 +253,17 @@ const processParsingSequential = async (
const needsOwner = nodeLabel === 'Method' || nodeLabel === 'Constructor' || nodeLabel === 'Property' || nodeLabel === 'Function';
const enclosingClassId = needsOwner ? findEnclosingClassId(nameNode || definitionNodeForRange, file.path) : null;
// Extract declared type for Property nodes (field/property type annotations)
const declaredType = (nodeLabel === 'Property' && definitionNode)
? extractPropertyDeclaredType(definitionNode)
: undefined;
symbolTable.add(file.path, nodeName, nodeId, nodeLabel, {
parameterCount: methodSig?.parameterCount,
requiredParameterCount: methodSig?.requiredParameterCount,
parameterTypes: methodSig?.parameterTypes,
returnType: methodSig?.returnType,
declaredType,
ownerId: enclosingClassId ?? undefined,
});
@ -273,19 +282,27 @@ const processParsingSequential = async (
graph.addRelationship(relationship);
// ── HAS_METHOD: link method/constructor/property to enclosing class ──
// ── HAS_METHOD / HAS_PROPERTY: link member to enclosing class ──
if (enclosingClassId) {
const memberEdgeType = nodeLabel === 'Property' ? 'HAS_PROPERTY' : 'HAS_METHOD';
graph.addRelationship({
id: generateId('HAS_METHOD', `${enclosingClassId}->${nodeId}`),
id: generateId(memberEdgeType, `${enclosingClassId}->${nodeId}`),
sourceId: enclosingClassId,
targetId: nodeId,
type: 'HAS_METHOD',
type: memberEdgeType,
confidence: 1.0,
reason: '',
});
}
});
}
if (skippedLanguages.size > 0) {
const summary = Array.from(skippedLanguages.entries())
.map(([lang, count]) => `${lang}: ${count}`)
.join(', ');
console.warn(` Skipped unsupported languages: ${summary}`);
}
};
// ============================================================================

View file

@ -1,12 +1,14 @@
import { createKnowledgeGraph } from '../graph/graph.js';
import { processStructure } from './structure-processor.js';
import { processMarkdown } from './markdown-processor.js';
import { processParsing } from './parsing-processor.js';
import {
processImports,
processImportsFromExtracted,
buildImportResolutionContext
} from './import-processor.js';
import { processCalls, processCallsFromExtracted, processRoutesFromExtracted } from './call-processor.js';
import { EMPTY_INDEX } from './resolvers/index.js';
import { processCalls, processCallsFromExtracted, processAssignmentsFromExtracted, processRoutesFromExtracted, seedCrossFileReceiverTypes, buildImportedReturnTypes, buildImportedRawReturnTypes, type ExportedTypeMap, buildExportedTypeMapFromGraph } from './call-processor.js';
import { processHeritage, processHeritageFromExtracted } from './heritage-processor.js';
import { computeMRO } from './mro-processor.js';
import { processCommunities } from './community-processor.js';
@ -17,6 +19,7 @@ import { PipelineProgress, PipelineResult } from '../../types/pipeline.js';
import { walkRepositoryPaths, readFileContents } from './filesystem-walker.js';
import { getLanguageFromFilename } from './utils.js';
import { isLanguageAvailable } from '../tree-sitter/parser-loader.js';
import { SupportedLanguages } from '../../config/supported-languages.js';
import { createWorkerPool, WorkerPool } from './workers/worker-pool.js';
import fs from 'node:fs';
import path from 'node:path';
@ -24,6 +27,62 @@ import { fileURLToPath, pathToFileURL } from 'node:url';
const isDev = process.env.NODE_ENV === 'development';
/** A group of files with no mutual dependencies, safe to process in parallel. */
type IndependentFileGroup = readonly string[];
/** Kahn's algorithm: returns files grouped by topological level.
* Files in the same level have no mutual dependencies — safe to process in parallel.
* Files in cycles are returned as a final group (no cross-cycle propagation). */
export function topologicalLevelSort(
importMap: ReadonlyMap<string, ReadonlySet<string>>,
): { levels: readonly IndependentFileGroup[]; cycleCount: number } {
// Build in-degree map and reverse dependency map
const inDegree = new Map<string, number>();
const reverseDeps = new Map<string, string[]>();
for (const [file, deps] of importMap) {
if (!inDegree.has(file)) inDegree.set(file, 0);
for (const dep of deps) {
if (!inDegree.has(dep)) inDegree.set(dep, 0);
// file imports dep, so dep must be processed before file
// In Kahn's terms: dep → file (dep is a prerequisite of file)
inDegree.set(file, (inDegree.get(file) ?? 0) + 1);
let rev = reverseDeps.get(dep);
if (!rev) { rev = []; reverseDeps.set(dep, rev); }
rev.push(file);
}
}
// BFS from zero-in-degree nodes, grouping by level
const levels: string[][] = [];
let currentLevel = [...inDegree.entries()]
.filter(([, d]) => d === 0)
.map(([f]) => f);
while (currentLevel.length > 0) {
levels.push(currentLevel);
const nextLevel: string[] = [];
for (const file of currentLevel) {
for (const dependent of reverseDeps.get(file) ?? []) {
const newDeg = (inDegree.get(dependent) ?? 1) - 1;
inDegree.set(dependent, newDeg);
if (newDeg === 0) nextLevel.push(dependent);
}
}
currentLevel = nextLevel;
}
// Files still with positive in-degree are in cycles — add as final group
const cycleFiles = [...inDegree.entries()]
.filter(([, d]) => d > 0)
.map(([f]) => f);
if (cycleFiles.length > 0) {
levels.push(cycleFiles);
}
return { levels, cycleCount: cycleFiles.length };
}
/** Max bytes of source content to load per parse chunk. Each chunk's source +
* parsed ASTs + extracted records + worker serialization overhead all live in
* memory simultaneously, so this must be conservative. 20MB source ≈ 200-400MB
@ -33,14 +92,276 @@ const CHUNK_BYTE_BUDGET = 20 * 1024 * 1024; // 20MB
/** Max AST trees to keep in LRU cache */
const AST_CACHE_CAP = 50;
/** Minimum percentage of files that must benefit from cross-file seeding to justify the re-resolution pass. */
const CROSS_FILE_SKIP_THRESHOLD = 0.03;
/** Hard cap on files re-processed during cross-file propagation. */
const MAX_CROSS_FILE_REPROCESS = 2000;
/** Node labels that represent top-level importable symbols.
* Excludes Method, Property, Constructor (accessed via receiver, not directly imported),
* and structural labels (File, Folder, Package, Module, Project, etc.). */
const IMPORTABLE_SYMBOL_LABELS = new Set([
'Function', 'Class', 'Interface', 'Struct', 'Enum', 'Trait',
'TypeAlias', 'Const', 'Static', 'Record', 'Union', 'Typedef', 'Macro',
]);
/** Max synthetic bindings per importing file — prevents memory bloat for
* C/C++ files that include many large headers. */
const MAX_SYNTHETIC_BINDINGS_PER_FILE = 1000;
/** Languages with whole-module import semantics (no per-symbol named imports).
* For these languages, namedImportMap entries are synthesized from graph-exported
* symbols after parsing, enabling Phase 14 cross-file binding propagation. */
const WILDCARD_IMPORT_LANGUAGES = new Set([
SupportedLanguages.Go,
SupportedLanguages.Ruby,
SupportedLanguages.C,
SupportedLanguages.CPlusPlus,
SupportedLanguages.Swift,
]);
/** Synthesize namedImportMap entries for languages with whole-module imports.
* These languages (Go, Ruby, C/C++, Swift) import all exported symbols from a file,
* not specific named symbols. After parsing, we know which symbols each file exports
* (via graph isExported), so we can expand ImportMap edges into per-symbol bindings
* that Phase 14 can use for cross-file type propagation. */
function synthesizeWildcardImportBindings(
graph: ReturnType<typeof createKnowledgeGraph>,
ctx: ReturnType<typeof createResolutionContext>,
): number {
// Pre-compute exported symbols per file from graph (single pass)
const exportedSymbolsByFile = new Map<string, { name: string; filePath: string }[]>();
graph.forEachNode(node => {
if (!node.properties?.isExported) return;
if (!IMPORTABLE_SYMBOL_LABELS.has(node.label)) return;
const fp = node.properties.filePath;
const name = node.properties.name;
if (!fp || !name) return;
let symbols = exportedSymbolsByFile.get(fp);
if (!symbols) { symbols = []; exportedSymbolsByFile.set(fp, symbols); }
symbols.push({ name, filePath: fp });
});
if (exportedSymbolsByFile.size === 0) return 0;
// Build a merged import map: ctx.importMap has file-based imports (Ruby, C/C++),
// but Go/C# package imports use graph IMPORTS edges + PackageMap instead.
// Collect graph-level IMPORTS edges for wildcard languages missing from ctx.importMap.
const FILE_PREFIX = 'File:';
const graphImports = new Map<string, Set<string>>();
graph.forEachRelationship(rel => {
if (rel.type !== 'IMPORTS') return;
if (!rel.sourceId.startsWith(FILE_PREFIX) || !rel.targetId.startsWith(FILE_PREFIX)) return;
const srcFile = rel.sourceId.slice(FILE_PREFIX.length);
const tgtFile = rel.targetId.slice(FILE_PREFIX.length);
const lang = getLanguageFromFilename(srcFile);
if (!lang || !WILDCARD_IMPORT_LANGUAGES.has(lang)) return;
// Only add if not already in ctx.importMap (avoid duplicates)
if (ctx.importMap.get(srcFile)?.has(tgtFile)) return;
let set = graphImports.get(srcFile);
if (!set) { set = new Set(); graphImports.set(srcFile, set); }
set.add(tgtFile);
});
let totalSynthesized = 0;
// Helper: synthesize bindings for a file given its imported files
const synthesizeForFile = (filePath: string, importedFiles: Iterable<string>) => {
let fileBindings = ctx.namedImportMap.get(filePath);
let fileCount = fileBindings?.size ?? 0;
for (const importedFile of importedFiles) {
const exportedSymbols = exportedSymbolsByFile.get(importedFile);
if (!exportedSymbols) continue;
for (const sym of exportedSymbols) {
if (fileCount >= MAX_SYNTHETIC_BINDINGS_PER_FILE) return;
if (fileBindings?.has(sym.name)) continue;
if (!fileBindings) {
fileBindings = new Map();
ctx.namedImportMap.set(filePath, fileBindings);
}
fileBindings.set(sym.name, {
sourcePath: importedFile,
exportedName: sym.name,
});
fileCount++;
totalSynthesized++;
}
}
};
// Process files from ctx.importMap (Ruby, C/C++, Swift file-based imports)
for (const [filePath, importedFiles] of ctx.importMap) {
const lang = getLanguageFromFilename(filePath);
if (!lang || !WILDCARD_IMPORT_LANGUAGES.has(lang)) continue;
synthesizeForFile(filePath, importedFiles);
}
// Process files from graph IMPORTS edges (Go package imports)
for (const [filePath, importedFiles] of graphImports) {
synthesizeForFile(filePath, importedFiles);
}
return totalSynthesized;
}
/** Phase 14: Cross-file binding propagation.
* Seeds downstream files with resolved type bindings from upstream exports.
* Files are processed in topological import order so upstream bindings are
* available when downstream files are re-resolved. */
async function runCrossFileBindingPropagation(
graph: ReturnType<typeof createKnowledgeGraph>,
ctx: ReturnType<typeof createResolutionContext>,
exportedTypeMap: ExportedTypeMap,
allPaths: string[],
totalFiles: number,
repoPath: string,
pipelineStart: number,
onProgress: (progress: PipelineProgress) => void,
): Promise<void> {
// For the worker path, buildTypeEnv runs inside workers without SymbolTable,
// so exported bindings must be collected from graph + SymbolTable in main thread.
if (exportedTypeMap.size === 0 && graph.nodeCount > 0) {
const graphExports = buildExportedTypeMapFromGraph(graph, ctx.symbols);
for (const [fp, exports] of graphExports) exportedTypeMap.set(fp, exports);
}
if (exportedTypeMap.size === 0 || ctx.namedImportMap.size === 0) return;
const allPathSet = new Set(allPaths);
const { levels, cycleCount } = topologicalLevelSort(ctx.importMap);
// Cycle diagnostic: only log when actual cycles detected (cycleCount from Kahn's BFS)
if (isDev && cycleCount > 0) {
console.log(`🔄 ${cycleCount} files in import cycles (skipped for cross-file propagation)`);
}
// Quick count of files with cross-file binding gaps (early exit once threshold exceeded)
let filesWithGaps = 0;
const gapThreshold = Math.max(1, Math.ceil(totalFiles * CROSS_FILE_SKIP_THRESHOLD));
outer: for (const level of levels) {
for (const filePath of level) {
const imports = ctx.namedImportMap.get(filePath);
if (!imports) continue;
for (const [, binding] of imports) {
const upstream = exportedTypeMap.get(binding.sourcePath);
if (upstream?.has(binding.exportedName)) { filesWithGaps++; break; }
const def = ctx.symbols.lookupExactFull(binding.sourcePath, binding.exportedName);
if (def?.returnType) { filesWithGaps++; break; }
}
if (filesWithGaps >= gapThreshold) break outer;
}
}
const gapRatio = totalFiles > 0 ? filesWithGaps / totalFiles : 0;
if (gapRatio < CROSS_FILE_SKIP_THRESHOLD && filesWithGaps < gapThreshold) {
if (isDev) {
console.log(`⏭️ Cross-file re-resolution skipped (${filesWithGaps}/${totalFiles} files, ${(gapRatio * 100).toFixed(1)}% < ${CROSS_FILE_SKIP_THRESHOLD * 100}% threshold)`);
}
return;
}
onProgress({
phase: 'parsing',
percent: 82,
message: `Cross-file type propagation (${filesWithGaps}+ files)...`,
stats: { filesProcessed: totalFiles, totalFiles, nodesCreated: graph.nodeCount },
});
let crossFileResolved = 0;
const crossFileStart = Date.now();
let astCache = createASTCache(AST_CACHE_CAP);
for (const level of levels) {
const levelCandidates: { filePath: string; seeded: Map<string, string>; importedReturns: ReadonlyMap<string, string>; importedRawReturns: ReadonlyMap<string, string> }[] = [];
for (const filePath of level) {
if (crossFileResolved + levelCandidates.length >= MAX_CROSS_FILE_REPROCESS) break;
const imports = ctx.namedImportMap.get(filePath);
if (!imports) continue;
const seeded = new Map<string, string>();
for (const [localName, binding] of imports) {
const upstream = exportedTypeMap.get(binding.sourcePath);
if (upstream) {
const type = upstream.get(binding.exportedName);
if (type) seeded.set(localName, type);
}
}
const importedReturns = buildImportedReturnTypes(filePath, ctx.namedImportMap, ctx.symbols);
const importedRawReturns = buildImportedRawReturnTypes(filePath, ctx.namedImportMap, ctx.symbols);
if (seeded.size === 0 && importedReturns.size === 0) continue;
if (!allPathSet.has(filePath)) continue;
const lang = getLanguageFromFilename(filePath);
if (!lang || !isLanguageAvailable(lang)) continue;
levelCandidates.push({ filePath, seeded, importedReturns, importedRawReturns });
}
if (levelCandidates.length === 0) continue;
const levelPaths = levelCandidates.map(c => c.filePath);
const contentMap = await readFileContents(repoPath, levelPaths);
for (const { filePath, seeded, importedReturns, importedRawReturns } of levelCandidates) {
const content = contentMap.get(filePath);
if (!content) continue;
const reFile = [{ path: filePath, content }];
const bindings = new Map<string, ReadonlyMap<string, string>>();
if (seeded.size > 0) bindings.set(filePath, seeded);
const importedReturnTypesMap = new Map<string, ReadonlyMap<string, string>>();
if (importedReturns.size > 0) {
importedReturnTypesMap.set(filePath, importedReturns);
}
const importedRawReturnTypesMap = new Map<string, ReadonlyMap<string, string>>();
if (importedRawReturns.size > 0) {
importedRawReturnTypesMap.set(filePath, importedRawReturns);
}
await processCalls(graph, reFile, astCache, ctx, undefined, exportedTypeMap, bindings.size > 0 ? bindings : undefined, importedReturnTypesMap.size > 0 ? importedReturnTypesMap : undefined, importedRawReturnTypesMap.size > 0 ? importedRawReturnTypesMap : undefined);
crossFileResolved++;
}
if (crossFileResolved >= MAX_CROSS_FILE_REPROCESS) {
if (isDev) console.log(`⚠️ Cross-file re-resolution capped at ${MAX_CROSS_FILE_REPROCESS} files`);
break;
}
}
astCache.clear();
if (isDev) {
const elapsed = Date.now() - crossFileStart;
const totalElapsed = Date.now() - pipelineStart;
const reResolutionPct = totalElapsed > 0 ? ((elapsed / totalElapsed) * 100).toFixed(1) : '0';
console.log(
`🔗 Cross-file re-resolution: ${crossFileResolved} candidates re-processed` +
` in ${elapsed}ms (${reResolutionPct}% of total ingestion time so far)`,
);
}
}
export interface PipelineOptions {
/** Skip MRO, community detection, and process extraction for faster test runs. */
skipGraphPhases?: boolean;
}
export const runPipelineFromRepo = async (
repoPath: string,
onProgress: (progress: PipelineProgress) => void
onProgress: (progress: PipelineProgress) => void,
options?: PipelineOptions,
): Promise<PipelineResult> => {
const graph = createKnowledgeGraph();
const ctx = createResolutionContext();
const symbolTable = ctx.symbols;
let astCache = createASTCache(AST_CACHE_CAP);
const pipelineStart = Date.now();
const cleanup = () => {
astCache.clear();
@ -93,6 +414,21 @@ export const runPipelineFromRepo = async (
stats: { filesProcessed: totalFiles, totalFiles, nodesCreated: graph.nodeCount },
});
// ── Phase 2.5: Markdown processing (headings + cross-links) ────────
const mdScanned = scannedFiles.filter(f => f.path.endsWith('.md') || f.path.endsWith('.mdx'));
if (mdScanned.length > 0) {
const mdContents = await readFileContents(repoPath, mdScanned.map(f => f.path));
const mdFiles = mdScanned
.filter(f => mdContents.has(f.path))
.map(f => ({ path: f.path, content: mdContents.get(f.path)! }));
const allPathSet = new Set(allPaths);
const mdResult = processMarkdown(graph, mdFiles, allPathSet);
if (isDev) {
console.log(` Markdown: ${mdResult.sections} sections, ${mdResult.links} cross-links from ${mdFiles.length} files`);
}
}
// ── Phase 3+4: Chunked read + parse ────────────────────────────────
// Group parseable files into byte-budget chunks so only ~20MB of source
// is in memory at a time. Each chunk is: read → parse → extract → free.
@ -154,22 +490,29 @@ export const runPipelineFromRepo = async (
stats: { filesProcessed: 0, totalFiles: totalParseable, nodesCreated: graph.nodeCount },
});
// Don't spawn workers for tiny repos — overhead exceeds benefit
const MIN_FILES_FOR_WORKERS = 15;
const MIN_BYTES_FOR_WORKERS = 512 * 1024;
const totalBytes = parseableScanned.reduce((s, f) => s + f.size, 0);
// Create worker pool once, reuse across chunks
let workerPool: WorkerPool | undefined;
try {
let workerUrl = new URL('./workers/parse-worker.js', import.meta.url);
// When running under vitest, import.meta.url points to src/ where no .js exists.
// Fall back to the compiled dist/ worker so the pool can spawn real worker threads.
const thisDir = fileURLToPath(new URL('.', import.meta.url));
if (!fs.existsSync(fileURLToPath(workerUrl))) {
const distWorker = path.resolve(thisDir, '..', '..', '..', 'dist', 'core', 'ingestion', 'workers', 'parse-worker.js');
if (fs.existsSync(distWorker)) {
workerUrl = pathToFileURL(distWorker) as URL;
if (totalParseable >= MIN_FILES_FOR_WORKERS || totalBytes >= MIN_BYTES_FOR_WORKERS) {
try {
let workerUrl = new URL('./workers/parse-worker.js', import.meta.url);
// When running under vitest, import.meta.url points to src/ where no .js exists.
// Fall back to the compiled dist/ worker so the pool can spawn real worker threads.
const thisDir = fileURLToPath(new URL('.', import.meta.url));
if (!fs.existsSync(fileURLToPath(workerUrl))) {
const distWorker = path.resolve(thisDir, '..', '..', '..', 'dist', 'core', 'ingestion', 'workers', 'parse-worker.js');
if (fs.existsSync(distWorker)) {
workerUrl = pathToFileURL(distWorker) as URL;
}
}
workerPool = createWorkerPool(workerUrl);
} catch (err) {
if (isDev) console.warn('Worker pool creation failed, using sequential fallback:', (err as Error).message);
}
workerPool = createWorkerPool(workerUrl);
} catch (err) {
if (isDev) console.warn('Worker pool creation failed, using sequential fallback:', (err as Error).message);
}
let filesParsedSoFar = 0;
@ -188,6 +531,10 @@ export const runPipelineFromRepo = async (
// are already registered). This trades ~5% cross-chunk resolution accuracy for
// 200-400MB less memory — critical for Linux-kernel-scale repos.
const sequentialChunkPaths: string[][] = [];
// Phase 14: Collect exported type bindings for cross-file propagation
const exportedTypeMap: ExportedTypeMap = new Map();
// Accumulate file-scope TypeEnv bindings from workers (closes worker/sequential quality gap)
const workerTypeEnvBindings: { filePath: string; bindings: [string, string][] }[] = [];
try {
for (let chunkIdx = 0; chunkIdx < numChunks; chunkIdx++) {
@ -229,6 +576,19 @@ export const runPipelineFromRepo = async (
stats: { filesProcessed: filesParsedSoFar, totalFiles: totalParseable, nodesCreated: graph.nodeCount },
});
}, repoPath, importCtx);
// Phase 14 E1: Seed cross-file receiver types from ExportedTypeMap
// before call resolution — eliminates re-parse for single-hop imported receivers.
// NOTE: In the worker path, exportedTypeMap is empty during chunk processing
// (populated later in runCrossFileBindingPropagation). This block is latent —
// it activates only if incremental export collection is added per-chunk.
if (exportedTypeMap.size > 0 && ctx.namedImportMap.size > 0) {
const { enrichedCount } = seedCrossFileReceiverTypes(
chunkWorkerData.calls, ctx.namedImportMap, exportedTypeMap,
);
if (isDev && enrichedCount > 0) {
console.log(`🔗 E1: Seeded ${enrichedCount} cross-file receiver types (chunk ${chunkIdx + 1})`);
}
}
// Calls + Heritage + Routes — resolve in parallel (no shared mutable state between them)
// This is safe because each writes disjoint relationship types into idempotent id-keyed Maps,
// and the single-threaded event loop prevents races between synchronous addRelationship calls.
@ -277,6 +637,14 @@ export const runPipelineFromRepo = async (
},
),
]);
// Process field write assignments (synchronous, runs after calls resolve)
if (chunkWorkerData.assignments?.length) {
processAssignmentsFromExtracted(graph, chunkWorkerData.assignments, ctx, chunkWorkerData.constructorBindings);
}
// Collect TypeEnv file-scope bindings for exported type enrichment
if (chunkWorkerData.typeEnvBindings?.length) {
workerTypeEnvBindings.push(...chunkWorkerData.typeEnvBindings);
}
} else {
await processImports(graph, chunkFiles, astCache, ctx, undefined, repoPath, allPaths);
sequentialChunkPaths.push(chunkPaths);
@ -299,7 +667,7 @@ export const runPipelineFromRepo = async (
.filter(p => chunkContents.has(p))
.map(p => ({ path: p, content: chunkContents.get(p)! }));
astCache = createASTCache(chunkFiles.length);
const rubyHeritage = await processCalls(graph, chunkFiles, astCache, ctx);
const rubyHeritage = await processCalls(graph, chunkFiles, astCache, ctx, undefined, exportedTypeMap);
await processHeritage(graph, chunkFiles, astCache, ctx);
if (rubyHeritage.length > 0) {
await processHeritageFromExtracted(graph, rubyHeritage, ctx);
@ -315,137 +683,187 @@ export const runPipelineFromRepo = async (
console.log(`🔍 Resolution cache: ${rcStats.cacheHits} hits, ${rcStats.cacheMisses} misses (${hitRate}% hit rate)`);
}
// ── Worker path quality enrichment: merge TypeEnv file-scope bindings into ExportedTypeMap ──
// Workers return file-scope bindings from their TypeEnv fixpoint (includes inferred types
// like `const config = getConfig()` → Config). Filter by graph isExported to match
// the sequential path's collectExportedBindings behavior.
if (workerTypeEnvBindings.length > 0) {
let enriched = 0;
for (const { filePath, bindings } of workerTypeEnvBindings) {
for (const [name, type] of bindings) {
// Verify the symbol is exported via graph node
const nodeId = `Function:${filePath}:${name}`;
const varNodeId = `Variable:${filePath}:${name}`;
const constNodeId = `Const:${filePath}:${name}`;
const node = graph.getNode(nodeId) ?? graph.getNode(varNodeId) ?? graph.getNode(constNodeId);
if (!node?.properties?.isExported) continue;
let fileExports = exportedTypeMap.get(filePath);
if (!fileExports) { fileExports = new Map(); exportedTypeMap.set(filePath, fileExports); }
// Don't overwrite existing entries (Tier 0 from SymbolTable is authoritative)
if (!fileExports.has(name)) {
fileExports.set(name, type);
enriched++;
}
}
}
if (isDev && enriched > 0) {
console.log(`🔗 Worker TypeEnv enrichment: ${enriched} fixpoint-inferred exports added to ExportedTypeMap`);
}
}
// ── Phase 14 pre-pass: Synthesize namedImportMap for whole-module-import languages ──
// Go, Ruby, C/C++, Swift import all exported symbols from a file.
// Expand ImportMap edges into per-symbol namedImportMap entries so Phase 14 can
// propagate types cross-file for these languages.
const synthesized = synthesizeWildcardImportBindings(graph, ctx);
if (isDev && synthesized > 0) {
console.log(`🔗 Synthesized ${synthesized} wildcard import bindings (Go/Ruby/C++/Swift)`);
}
// ── Phase 14: Cross-file binding propagation ──────────────────────
await runCrossFileBindingPropagation(
graph, ctx, exportedTypeMap, allPaths, totalFiles, repoPath, pipelineStart, onProgress,
);
// Free import resolution context — suffix index + resolve cache no longer needed
// (allPathObjects and importCtx hold ~94MB+ for large repos)
allPathObjects.length = 0;
importCtx.resolveCache.clear();
(importCtx as any).suffixIndex = null;
(importCtx as any).normalizedFileList = null;
importCtx.index = EMPTY_INDEX; // Release suffix index memory (~30MB for large repos)
importCtx.normalizedFileList = [];
// ── Phase 4.5: Method Resolution Order ──────────────────────────────
onProgress({
phase: 'parsing',
percent: 81,
message: 'Computing method resolution order...',
stats: { filesProcessed: totalFiles, totalFiles, nodesCreated: graph.nodeCount },
});
let communityResult: Awaited<ReturnType<typeof processCommunities>> | undefined;
let processResult: Awaited<ReturnType<typeof processProcesses>> | undefined;
const mroResult = computeMRO(graph);
if (isDev && mroResult.entries.length > 0) {
console.log(`🔀 MRO: ${mroResult.entries.length} classes analyzed, ${mroResult.ambiguityCount} ambiguities found, ${mroResult.overrideEdges} OVERRIDES edges`);
}
// ── Phase 5: Communities ───────────────────────────────────────────
onProgress({
phase: 'communities',
percent: 82,
message: 'Detecting code communities...',
stats: { filesProcessed: totalFiles, totalFiles, nodesCreated: graph.nodeCount },
});
const communityResult = await processCommunities(graph, (message, progress) => {
const communityProgress = 82 + (progress * 0.10);
if (!options?.skipGraphPhases) {
// ── Phase 4.5: Method Resolution Order ──────────────────────────────
onProgress({
phase: 'communities',
percent: Math.round(communityProgress),
message,
phase: 'parsing',
percent: 81,
message: 'Computing method resolution order...',
stats: { filesProcessed: totalFiles, totalFiles, nodesCreated: graph.nodeCount },
});
});
if (isDev) {
console.log(`🏘️ Community detection: ${communityResult.stats.totalCommunities} communities found (modularity: ${communityResult.stats.modularity.toFixed(3)})`);
}
const mroResult = computeMRO(graph);
if (isDev && mroResult.entries.length > 0) {
console.log(`🔀 MRO: ${mroResult.entries.length} classes analyzed, ${mroResult.ambiguityCount} ambiguities found, ${mroResult.overrideEdges} OVERRIDES edges`);
}
communityResult.communities.forEach(comm => {
graph.addNode({
id: comm.id,
label: 'Community' as const,
properties: {
name: comm.label,
filePath: '',
heuristicLabel: comm.heuristicLabel,
cohesion: comm.cohesion,
symbolCount: comm.symbolCount,
}
// ── Phase 5: Communities ───────────────────────────────────────────
onProgress({
phase: 'communities',
percent: 82,
message: 'Detecting code communities...',
stats: { filesProcessed: totalFiles, totalFiles, nodesCreated: graph.nodeCount },
});
});
communityResult.memberships.forEach(membership => {
graph.addRelationship({
id: `${membership.nodeId}_member_of_${membership.communityId}`,
type: 'MEMBER_OF',
sourceId: membership.nodeId,
targetId: membership.communityId,
confidence: 1.0,
reason: 'leiden-algorithm',
});
});
// ── Phase 6: Processes ─────────────────────────────────────────────
onProgress({
phase: 'processes',
percent: 94,
message: 'Detecting execution flows...',
stats: { filesProcessed: totalFiles, totalFiles, nodesCreated: graph.nodeCount },
});
let symbolCount = 0;
graph.forEachNode(n => { if (n.label !== 'File') symbolCount++; });
const dynamicMaxProcesses = Math.max(20, Math.min(300, Math.round(symbolCount / 10)));
const processResult = await processProcesses(
graph,
communityResult.memberships,
(message, progress) => {
const processProgress = 94 + (progress * 0.05);
communityResult = await processCommunities(graph, (message, progress) => {
const communityProgress = 82 + (progress * 0.10);
onProgress({
phase: 'processes',
percent: Math.round(processProgress),
phase: 'communities',
percent: Math.round(communityProgress),
message,
stats: { filesProcessed: totalFiles, totalFiles, nodesCreated: graph.nodeCount },
});
},
{ maxProcesses: dynamicMaxProcesses, minSteps: 3 }
);
});
if (isDev) {
console.log(`🔄 Process detection: ${processResult.stats.totalProcesses} processes found (${processResult.stats.crossCommunityCount} cross-community)`);
if (isDev) {
console.log(`🏘️ Community detection: ${communityResult.stats.totalCommunities} communities found (modularity: ${communityResult.stats.modularity.toFixed(3)})`);
}
communityResult.communities.forEach(comm => {
graph.addNode({
id: comm.id,
label: 'Community' as const,
properties: {
name: comm.label,
filePath: '',
heuristicLabel: comm.heuristicLabel,
cohesion: comm.cohesion,
symbolCount: comm.symbolCount,
}
});
});
communityResult.memberships.forEach(membership => {
graph.addRelationship({
id: `${membership.nodeId}_member_of_${membership.communityId}`,
type: 'MEMBER_OF',
sourceId: membership.nodeId,
targetId: membership.communityId,
confidence: 1.0,
reason: 'leiden-algorithm',
});
});
// ── Phase 6: Processes ─────────────────────────────────────────────
onProgress({
phase: 'processes',
percent: 94,
message: 'Detecting execution flows...',
stats: { filesProcessed: totalFiles, totalFiles, nodesCreated: graph.nodeCount },
});
let symbolCount = 0;
graph.forEachNode(n => { if (n.label !== 'File') symbolCount++; });
const dynamicMaxProcesses = Math.max(20, Math.min(300, Math.round(symbolCount / 10)));
processResult = await processProcesses(
graph,
communityResult.memberships,
(message, progress) => {
const processProgress = 94 + (progress * 0.05);
onProgress({
phase: 'processes',
percent: Math.round(processProgress),
message,
stats: { filesProcessed: totalFiles, totalFiles, nodesCreated: graph.nodeCount },
});
},
{ maxProcesses: dynamicMaxProcesses, minSteps: 3 }
);
if (isDev) {
console.log(`🔄 Process detection: ${processResult.stats.totalProcesses} processes found (${processResult.stats.crossCommunityCount} cross-community)`);
}
processResult.processes.forEach(proc => {
graph.addNode({
id: proc.id,
label: 'Process' as const,
properties: {
name: proc.label,
filePath: '',
heuristicLabel: proc.heuristicLabel,
processType: proc.processType,
stepCount: proc.stepCount,
communities: proc.communities,
entryPointId: proc.entryPointId,
terminalId: proc.terminalId,
}
});
});
processResult.steps.forEach(step => {
graph.addRelationship({
id: `${step.nodeId}_step_${step.step}_${step.processId}`,
type: 'STEP_IN_PROCESS',
sourceId: step.nodeId,
targetId: step.processId,
confidence: 1.0,
reason: 'trace-detection',
step: step.step,
});
});
}
processResult.processes.forEach(proc => {
graph.addNode({
id: proc.id,
label: 'Process' as const,
properties: {
name: proc.label,
filePath: '',
heuristicLabel: proc.heuristicLabel,
processType: proc.processType,
stepCount: proc.stepCount,
communities: proc.communities,
entryPointId: proc.entryPointId,
terminalId: proc.terminalId,
}
});
});
processResult.steps.forEach(step => {
graph.addRelationship({
id: `${step.nodeId}_step_${step.step}_${step.processId}`,
type: 'STEP_IN_PROCESS',
sourceId: step.nodeId,
targetId: step.processId,
confidence: 1.0,
reason: 'trace-detection',
step: step.step,
});
});
onProgress({
phase: 'complete',
percent: 100,
message: `Graph complete! ${communityResult.stats.totalCommunities} communities, ${processResult.stats.totalProcesses} processes detected.`,
message: communityResult && processResult
? `Graph complete! ${communityResult.stats.totalCommunities} communities, ${processResult.stats.totalProcesses} processes detected.`
: 'Graph complete! (graph phases skipped)',
stats: {
filesProcessed: totalFiles,
totalFiles,

View file

@ -81,10 +81,10 @@ export const createResolutionContext = (): ResolutionContext => {
// --- Core resolution (single implementation of tier logic) ---
const resolveUncached = (name: string, fromFile: string): TieredCandidates | null => {
// Tier 1: Same file — authoritative match
const localDef = symbols.lookupExactFull(fromFile, name);
if (localDef) {
return { candidates: [localDef], tier: 'same-file' };
// Tier 1: Same file — authoritative match (returns all overloads)
const localDefs = symbols.lookupExactAll(fromFile, name);
if (localDefs.length > 0) {
return { candidates: localDefs, tier: 'same-file' };
}
// Get all global definitions for subsequent tiers

View file

@ -3,7 +3,7 @@
* Extracted from import-processor.ts for maintainability.
*/
export { EXTENSIONS, tryResolveWithExtensions, buildSuffixIndex, suffixResolve } from './utils.js';
export { EXTENSIONS, tryResolveWithExtensions, buildSuffixIndex, suffixResolve, EMPTY_INDEX } from './utils.js';
export type { SuffixIndex } from './utils.js';
export { KOTLIN_EXTENSIONS, appendKotlinWildcard, resolveJvmWildcard, resolveJvmMemberImport } from './jvm.js';
@ -21,5 +21,7 @@ export { resolveRustImport, tryRustModulePath } from './rust.js';
export { resolveRubyImport } from './ruby.js';
export { resolvePythonImport } from './python.js';
export { resolveImportPath, RESOLVE_CACHE_CAP } from './standard.js';
export type { TsconfigPaths } from './standard.js';

View file

@ -4,6 +4,7 @@
*/
import type { SuffixIndex } from './utils.js';
import type { SyntaxNode } from '../utils.js';
/** Kotlin file extensions for JVM resolver reuse */
export const KOTLIN_EXTENSIONS: readonly string[] = ['.kt', '.kts'];
@ -12,7 +13,7 @@ export const KOTLIN_EXTENSIONS: readonly string[] = ['.kt', '.kts'];
* Append .* to a Kotlin import path if the AST has a wildcard_import sibling node.
* Pure function — returns a new string without mutating the input.
*/
export const appendKotlinWildcard = (importPath: string, importNode: any): string => {
export const appendKotlinWildcard = (importPath: string, importNode: SyntaxNode): string => {
for (let i = 0; i < importNode.childCount; i++) {
if (importNode.child(i)?.type === 'wildcard_import') {
return importPath.endsWith('.*') ? importPath : `${importPath}.*`;
@ -39,26 +40,39 @@ export function resolveJvmWildcard(
const candidates = extensions.flatMap(ext => index.getFilesInDir(packagePath, ext));
// Filter to only direct children (no subdirectories)
const packageSuffix = '/' + packagePath + '/';
const packagePrefix = packagePath + '/';
return candidates.filter(f => {
const normalized = f.replace(/\\/g, '/');
const idx = normalized.indexOf(packageSuffix);
if (idx < 0) return false;
const afterPkg = normalized.substring(idx + packageSuffix.length);
// Match both nested (src/models/User.kt) and root-level (models/User.kt) packages
let afterPkg: string;
const idx = normalized.lastIndexOf(packageSuffix);
if (idx >= 0) {
afterPkg = normalized.substring(idx + packageSuffix.length);
} else if (normalized.startsWith(packagePrefix)) {
afterPkg = normalized.substring(packagePrefix.length);
} else {
return false;
}
return !afterPkg.includes('/');
});
}
// Fallback: linear scan
const packageSuffix = '/' + packagePath + '/';
const packagePrefix = packagePath + '/';
const matches: string[] = [];
for (let i = 0; i < normalizedFileList.length; i++) {
const normalized = normalizedFileList[i];
if (normalized.includes(packageSuffix) &&
extensions.some(ext => normalized.endsWith(ext))) {
const afterPackage = normalized.substring(normalized.indexOf(packageSuffix) + packageSuffix.length);
if (!afterPackage.includes('/')) {
matches.push(allFileList[i]);
}
if (!extensions.some(ext => normalized.endsWith(ext))) continue;
// Match both nested (src/models/User.kt) and root-level (models/User.kt) packages
let afterPackage: string | null = null;
if (normalized.includes(packageSuffix)) {
afterPackage = normalized.substring(normalized.lastIndexOf(packageSuffix) + packageSuffix.length);
} else if (normalized.startsWith(packagePrefix)) {
afterPackage = normalized.substring(packagePrefix.length);
}
if (afterPackage !== null && !afterPackage.includes('/')) {
matches.push(allFileList[i]);
}
}
return matches;

View file

@ -10,11 +10,34 @@ import { suffixResolve } from './utils.js';
export interface ComposerConfig {
/** Map of namespace prefix -> directory (e.g., "App\\" -> "app/") */
psr4: Map<string, string>;
/** PSR-4 entries sorted by namespace length descending (longest match wins).
* Cached once at config load time to avoid re-sorting on every import. */
psr4Sorted?: readonly [string, string][];
}
/** Get or compute the sorted PSR-4 entries (cached after first call). */
function getSortedPsr4(config: ComposerConfig): readonly [string, string][] {
if (!config.psr4Sorted) {
const sorted = [...config.psr4.entries()].sort((a, b) => b[0].length - a[0].length);
config.psr4Sorted = sorted;
}
return config.psr4Sorted;
}
/**
* Resolve a PHP use-statement import path using PSR-4 mappings.
* e.g. "App\Http\Controllers\UserController" -> "app/Http/Controllers/UserController.php"
*
* For function/constant imports (use function App\Models\getUser), the last
* segment is the symbol name, not a class name, so it may not map directly to
* a file. When PSR-4 class-style resolution fails, we fall back to scanning
* .php files in the namespace directory.
*
* NOTE: The function-import fallback returns the first matching .php file in the
* namespace directory. When multiple files exist in the same namespace directory,
* resolution is non-deterministic (depends on Set/index iteration order). This is
* a known limitation — PHP function imports cannot be resolved to a specific file
* without parsing all candidate files.
*/
export function resolvePhpImport(
importPath: string,
@ -27,20 +50,44 @@ export function resolvePhpImport(
// Normalize: replace backslashes with forward slashes
const normalized = importPath.replace(/\\/g, '/');
// Try PSR-4 resolution if composer.json was found
// Reject path traversal attempts (defense-in-depth — walker whitelist also prevents this)
if (normalized.includes('..')) return null;
if (composerConfig) {
// Sort namespaces by length descending (longest match wins)
const sorted = [...composerConfig.psr4.entries()].sort((a, b) => b[0].length - a[0].length);
const sorted = getSortedPsr4(composerConfig);
for (const [nsPrefix, dirPrefix] of sorted) {
const nsPrefixSlash = nsPrefix.replace(/\\/g, '/');
if (normalized.startsWith(nsPrefixSlash + '/') || normalized === nsPrefixSlash) {
const remainder = normalized.slice(nsPrefixSlash.length).replace(/^\//, '');
// 1. Try class-style PSR-4: full path → file (e.g. App\Models\User → app/Models/User.php)
const filePath = dirPrefix + (remainder ? '/' + remainder : '') + '.php';
if (allFiles.has(filePath)) return filePath;
if (index) {
const result = index.getInsensitive(filePath);
if (result) return result;
}
// 2. Function/constant fallback: strip last segment (symbol name), scan namespace directory.
// e.g. App\Models\getUser → directory app/Models/, find first .php file in that dir.
const lastSlash = remainder.lastIndexOf('/');
const nsDir = lastSlash >= 0
? dirPrefix + '/' + remainder.slice(0, lastSlash)
: dirPrefix;
// Prefer SuffixIndex directory lookup (O(log n + matches)) over linear scan
if (index) {
const candidates = index.getFilesInDir(nsDir, '.php');
if (candidates.length > 0) return candidates[0];
}
// Fallback: linear scan (only when SuffixIndex unavailable)
const nsDirPrefix = nsDir.endsWith('/') ? nsDir : nsDir + '/';
for (const f of allFiles) {
if (f.startsWith(nsDirPrefix) && f.endsWith('.php') && !f.slice(nsDirPrefix.length).includes('/')) {
return f;
}
}
}
}
}

View file

@ -0,0 +1,59 @@
/**
* Python import resolution — PEP 328 relative imports and proximity-based bare imports.
* Import system spec: PEP 302 (original), PEP 451 (current).
*/
import { tryResolveWithExtensions } from './utils.js';
/**
* Resolve a Python import to a file path.
*
* 1. Relative (PEP 328): `.module`, `..module` — 1 dot = current package, each extra dot goes up one level.
* 2. Proximity bare import: static heuristic — checks the importer's own directory first.
* Approximates the common case where co-located files find each other without an installed package.
* Single-segment only — multi-segment (e.g. `os.path`) falls through to suffixResolve.
* Checks package (__init__.py) before module (.py), matching CPython's finder order (PEP 451 §4).
* Coexistence of both is physically impossible (same name = file vs directory), so the order
* only matters for spec compliance.
* Note: namespace packages (PEP 420, directory without __init__.py) are not handled.
*
* Returns null to let the caller fall through to suffixResolve.
*/
export function resolvePythonImport(
currentFile: string,
importPath: string,
allFiles: Set<string>,
): string | null {
// Relative import — PEP 328 (https://peps.python.org/pep-0328/)
if (importPath.startsWith('.')) {
const dotMatch = importPath.match(/^(\.+)(.*)/);
if (!dotMatch) return null;
const dotCount = dotMatch[1].length;
const modulePart = dotMatch[2];
const dirParts = currentFile.split('/').slice(0, -1);
// PEP 328: more dots than directory levels → beyond top-level package → invalid
if (dotCount - 1 > dirParts.length) return null;
for (let i = 1; i < dotCount; i++) dirParts.pop();
if (modulePart) {
dirParts.push(...modulePart.replace(/\./g, '/').split('/'));
}
return tryResolveWithExtensions(dirParts.join('/'), allFiles);
}
// Proximity bare import — single-segment only; package before module (PEP 451 §4)
const pathLike = importPath.replace(/\./g, '/');
if (pathLike.includes('/')) return null;
// Normalize for Windows backslashes
const importerDir = currentFile.replace(/\\/g, '/').split('/').slice(0, -1).join('/');
if (!importerDir) return null;
if (allFiles.has(`${importerDir}/${pathLike}/__init__.py`)) return `${importerDir}/${pathLike}/__init__.py`;
if (allFiles.has(`${importerDir}/${pathLike}.py`)) return `${importerDir}/${pathLike}.py`;
return null;
}

View file

@ -113,32 +113,6 @@ export const resolveImportPath = (
// Fall through to generic resolution if Rust-specific didn't match
}
// ---- Python relative imports (PEP 328): .module, ..module, ... ----
if (language === SupportedLanguages.Python && importPath.startsWith('.')) {
const dotMatch = importPath.match(/^(\.+)(.*)/);
if (dotMatch) {
const dotCount = dotMatch[1].length;
const modulePart = dotMatch[2]; // e.g., "models" from ".models"
const dirParts = currentFile.split('/').slice(0, -1); // remove filename
// Navigate up: 1 dot = same package, 2 dots = parent package, etc.
// First dot means "current package", each additional dot goes up one level
for (let i = 1; i < dotCount; i++) {
dirParts.pop();
}
if (modulePart) {
// from .models import User → resolve "models" relative to current package
const modulePath = modulePart.replace(/\./g, '/');
dirParts.push(...modulePath.split('/'));
}
const basePath = dirParts.join('/');
const resolved = tryResolveWithExtensions(basePath, allFiles);
return cache(resolved);
}
}
// ---- Generic relative import resolution (./ and ../) ----
const currentDir = currentFile.split('/').slice(0, -1);
const parts = importPath.split('/');

View file

@ -3,6 +3,8 @@
* Extracted from import-processor.ts to reduce file size.
*/
import type { SyntaxNode } from '../utils.js';
/** All file extensions to try during resolution */
export const EXTENSIONS = [
'',
@ -63,6 +65,15 @@ export interface SuffixIndex {
getFilesInDir(dirSuffix: string, extension: string): string[];
}
const FROZEN_EMPTY_ARRAY: string[] = Object.freeze([]) as string[];
/** Sentinel index that returns no results. Used to release memory after import resolution. */
export const EMPTY_INDEX: SuffixIndex = Object.freeze({
get: () => undefined,
getInsensitive: () => undefined,
getFilesInDir: () => FROZEN_EMPTY_ARRAY,
});
export function buildSuffixIndex(normalizedFileList: string[], allFileList: string[]): SuffixIndex {
// Map: normalized suffix -> original file path
const exactMap = new Map<string, string>();
@ -156,3 +167,12 @@ export function suffixResolve(
}
return null;
}
/** Find the first direct named child of a tree-sitter node matching the given type. */
export function findChild(node: SyntaxNode, type: string): SyntaxNode | null {
for (let i = 0; i < node.namedChildCount; i++) {
const child = node.namedChild(i);
if (child?.type === type) return child;
}
return null;
}

View file

@ -1,11 +1,22 @@
import type { NodeLabel } from '../graph/types.js';
export interface SymbolDefinition {
nodeId: string;
filePath: string;
type: string; // 'Function', 'Class', etc.
type: NodeLabel;
parameterCount?: number;
/** Number of required (non-optional, non-default) parameters.
* Enables range-based arity filtering: argCount >= requiredParameterCount && argCount <= parameterCount. */
requiredParameterCount?: number;
/** Per-parameter type names for overload disambiguation (e.g. ['int', 'String']).
* Populated when parameter types are resolvable from AST (any typed language).
* Used for disambiguation in overloading languages (Java, Kotlin, C#, C++). */
parameterTypes?: string[];
/** Raw return type text extracted from AST (e.g. 'User', 'Promise<User>') */
returnType?: string;
/** Links Method/Constructor to owning Class/Struct/Trait nodeId */
/** Declared type for non-callable symbols — fields/properties (e.g. 'Address', 'List<User>') */
declaredType?: string;
/** Links Method/Constructor/Property to owning Class/Struct/Trait nodeId */
ownerId?: string;
}
@ -17,10 +28,10 @@ export interface SymbolTable {
filePath: string,
name: string,
nodeId: string,
type: string,
metadata?: { parameterCount?: number; returnType?: string; ownerId?: string }
type: NodeLabel,
metadata?: { parameterCount?: number; requiredParameterCount?: number; parameterTypes?: string[]; returnType?: string; declaredType?: string; ownerId?: string }
) => void;
/**
* High Confidence: Look for a symbol specifically inside a file
* Returns the Node ID if found
@ -30,15 +41,37 @@ export interface SymbolTable {
/**
* High Confidence: Look for a symbol in a specific file, returning full definition.
* Includes type information needed for heritage resolution (Class vs Interface).
* Returns first matching definition — use lookupExactAll for overloaded methods.
*/
lookupExactFull: (filePath: string, name: string) => SymbolDefinition | undefined;
/**
* High Confidence: Look for ALL symbols with this name in a specific file.
* Returns all definitions, including overloaded methods with the same name.
* Used by resolution-context to pass all same-file overloads to candidate filtering.
*/
lookupExactAll: (filePath: string, name: string) => SymbolDefinition[];
/**
* Low Confidence: Look for a symbol anywhere in the project
* Used when imports are missing or for framework magic
*/
lookupFuzzy: (name: string) => SymbolDefinition[];
/**
* Low Confidence: Look for callable symbols (Function/Method/Constructor) by name.
* Faster than `lookupFuzzy` + filter — backed by a lazy callable-only index.
* Used by ReturnTypeLookup to resolve callee → return type.
*/
lookupFuzzyCallable: (name: string) => SymbolDefinition[];
/**
* Look up a field/property by its owning class nodeId and field name.
* O(1) via dedicated eagerly-populated index keyed by `ownerNodeId\0fieldName`.
* Returns undefined when no matching property exists or the owner is ambiguous.
*/
lookupFieldByOwner: (ownerNodeId: string, fieldName: string) => SymbolDefinition | undefined;
/**
* Debugging: See how many symbols are tracked
*/
@ -51,27 +84,42 @@ export interface SymbolTable {
}
export const createSymbolTable = (): SymbolTable => {
// 1. File-Specific Index — stores full SymbolDefinition for O(1) lookupExactFull
// Structure: FilePath -> (SymbolName -> SymbolDefinition)
const fileIndex = new Map<string, Map<string, SymbolDefinition>>();
// 1. File-Specific Index — stores full SymbolDefinition(s) for O(1) lookup.
// Structure: FilePath -> (SymbolName -> SymbolDefinition[])
// Array allows overloaded methods (same name, different signatures) to coexist.
const fileIndex = new Map<string, Map<string, SymbolDefinition[]>>();
// 2. Global Reverse Index (The "Backup")
// Structure: SymbolName -> [List of Definitions]
const globalIndex = new Map<string, SymbolDefinition[]>();
// 3. Lazy Callable Index — populated on first lookupFuzzyCallable call.
// Structure: SymbolName -> [Callable Definitions]
// Only Function, Method, Constructor symbols are indexed.
let callableIndex: Map<string, SymbolDefinition[]> | null = null;
// 4. Eagerly-populated Field/Property Index — keyed by "ownerNodeId\0fieldName".
// Only Property symbols with ownerId and declaredType are indexed.
const fieldByOwner = new Map<string, SymbolDefinition>();
const CALLABLE_TYPES = new Set(['Function', 'Method', 'Constructor']);
const add = (
filePath: string,
name: string,
nodeId: string,
type: string,
metadata?: { parameterCount?: number; returnType?: string; ownerId?: string }
type: NodeLabel,
metadata?: { parameterCount?: number; requiredParameterCount?: number; parameterTypes?: string[]; returnType?: string; declaredType?: string; ownerId?: string }
) => {
const def: SymbolDefinition = {
nodeId,
filePath,
type,
...(metadata?.parameterCount !== undefined ? { parameterCount: metadata.parameterCount } : {}),
...(metadata?.requiredParameterCount !== undefined ? { requiredParameterCount: metadata.requiredParameterCount } : {}),
...(metadata?.parameterTypes !== undefined ? { parameterTypes: metadata.parameterTypes } : {}),
...(metadata?.returnType !== undefined ? { returnType: metadata.returnType } : {}),
...(metadata?.declaredType !== undefined ? { declaredType: metadata.declaredType } : {}),
...(metadata?.ownerId !== undefined ? { ownerId: metadata.ownerId } : {}),
};
@ -79,27 +127,69 @@ export const createSymbolTable = (): SymbolTable => {
if (!fileIndex.has(filePath)) {
fileIndex.set(filePath, new Map());
}
fileIndex.get(filePath)!.set(name, def);
const fileMap = fileIndex.get(filePath)!;
if (!fileMap.has(name)) {
fileMap.set(name, [def]);
} else {
fileMap.get(name)!.push(def);
}
// B. Add to Global Index (same object reference)
// B. Properties go to fieldByOwner index only — skip globalIndex to prevent
// namespace pollution for common names like 'id', 'name', 'type'.
// Index ALL properties (even without declaredType) so write-access tracking
// can resolve field ownership for dynamically-typed languages (Ruby, JS).
if (type === 'Property' && metadata?.ownerId) {
fieldByOwner.set(`${metadata.ownerId}\0${name}`, def);
// Still add to fileIndex above (for lookupExact), but skip globalIndex
return;
}
// C. Add to Global Index (same object reference)
if (!globalIndex.has(name)) {
globalIndex.set(name, []);
}
globalIndex.get(name)!.push(def);
// D. Invalidate the lazy callable index only when adding callable types
if (CALLABLE_TYPES.has(type)) {
callableIndex = null;
}
};
const lookupExact = (filePath: string, name: string): string | undefined => {
return fileIndex.get(filePath)?.get(name)?.nodeId;
const defs = fileIndex.get(filePath)?.get(name);
return defs?.[0]?.nodeId;
};
const lookupExactFull = (filePath: string, name: string): SymbolDefinition | undefined => {
return fileIndex.get(filePath)?.get(name);
const defs = fileIndex.get(filePath)?.get(name);
return defs?.[0];
};
const lookupExactAll = (filePath: string, name: string): SymbolDefinition[] => {
return fileIndex.get(filePath)?.get(name) ?? [];
};
const lookupFuzzy = (name: string): SymbolDefinition[] => {
return globalIndex.get(name) || [];
};
const lookupFuzzyCallable = (name: string): SymbolDefinition[] => {
if (!callableIndex) {
// Build the callable index lazily on first use
callableIndex = new Map();
for (const [symName, defs] of globalIndex) {
const callables = defs.filter(d => CALLABLE_TYPES.has(d.type));
if (callables.length > 0) callableIndex.set(symName, callables);
}
}
return callableIndex.get(name) ?? [];
};
const lookupFieldByOwner = (ownerNodeId: string, fieldName: string): SymbolDefinition | undefined => {
return fieldByOwner.get(`${ownerNodeId}\0${fieldName}`);
};
const getStats = () => ({
fileCount: fileIndex.size,
globalSymbolCount: globalIndex.size
@ -108,7 +198,9 @@ export const createSymbolTable = (): SymbolTable => {
const clear = () => {
fileIndex.clear();
globalIndex.clear();
callableIndex = null;
fieldByOwner.clear();
};
return { add, lookupExact, lookupExactFull, lookupFuzzy, getStats, clear };
return { add, lookupExact, lookupExactFull, lookupExactAll, lookupFuzzy, lookupFuzzyCallable, lookupFieldByOwner, getStats, clear };
};

View file

@ -19,6 +19,10 @@ export const TYPESCRIPT_QUERIES = `
(function_declaration
name: (identifier) @name) @definition.function
; TypeScript overload signatures (function_signature is a separate node type from function_declaration)
(function_signature
name: (identifier) @name) @definition.function
(method_definition
name: (property_identifier) @name) @definition.method
@ -62,6 +66,19 @@ export const TYPESCRIPT_QUERIES = `
(new_expression
constructor: (identifier) @call.name) @call
; Class properties — public_field_definition covers most TS class fields
(public_field_definition
name: (property_identifier) @name) @definition.property
; Private class fields: #address: Address
(public_field_definition
name: (private_property_identifier) @name) @definition.property
; Constructor parameter properties: constructor(public address: Address)
(required_parameter
(accessibility_modifier)
pattern: (identifier) @name) @definition.property
; Heritage queries - class extends
(class_declaration
name: (type_identifier) @heritage.class
@ -75,6 +92,20 @@ export const TYPESCRIPT_QUERIES = `
(class_heritage
(implements_clause
(type_identifier) @heritage.implements))) @heritage.impl
; Write access: obj.field = value
(assignment_expression
left: (member_expression
object: (_) @assignment.receiver
property: (property_identifier) @assignment.property)
right: (_)) @assignment
; Write access: obj.field += value (compound assignment)
(augmented_assignment_expression
left: (member_expression
object: (_) @assignment.receiver
property: (property_identifier) @assignment.property)
right: (_)) @assignment
`;
// JavaScript queries - works with tree-sitter-javascript
@ -128,12 +159,30 @@ export const JAVASCRIPT_QUERIES = `
(new_expression
constructor: (identifier) @call.name) @call
; Class fields — field_definition captures JS class fields (class User { address = ... })
(field_definition
property: (property_identifier) @name) @definition.property
; Heritage queries - class extends (JavaScript uses different AST than TypeScript)
; In tree-sitter-javascript, class_heritage directly contains the parent identifier
(class_declaration
name: (identifier) @heritage.class
(class_heritage
(identifier) @heritage.extends)) @heritage
; Write access: obj.field = value
(assignment_expression
left: (member_expression
object: (_) @assignment.receiver
property: (property_identifier) @assignment.property)
right: (_)) @assignment
; Write access: obj.field += value (compound assignment)
(augmented_assignment_expression
left: (member_expression
object: (_) @assignment.receiver
property: (property_identifier) @assignment.property)
right: (_)) @assignment
`;
// Python queries - works with tree-sitter-python
@ -160,11 +209,33 @@ export const PYTHON_QUERIES = `
function: (attribute
attribute: (identifier) @call.name)) @call
; Class attribute type annotations — PEP 526: address: Address or address: Address = Address()
; Both bare annotations (address: Address) and annotated assignments (name: str = "test")
; are parsed as (assignment left: ... type: ...) in tree-sitter-python.
(expression_statement
(assignment
left: (identifier) @name
type: (type)) @definition.property)
; Heritage queries - Python class inheritance
(class_definition
name: (identifier) @heritage.class
superclasses: (argument_list
(identifier) @heritage.extends)) @heritage
; Write access: obj.field = value
(assignment
left: (attribute
object: (_) @assignment.receiver
attribute: (identifier) @assignment.property)
right: (_)) @assignment
; Write access: obj.field += value (compound assignment)
(augmented_assignment
left: (attribute
object: (_) @assignment.receiver
attribute: (identifier) @assignment.property)
right: (_)) @assignment
`;
// Java queries - works with tree-sitter-java
@ -179,6 +250,11 @@ export const JAVA_QUERIES = `
(method_declaration name: (identifier) @name) @definition.method
(constructor_declaration name: (identifier) @name) @definition.constructor
; Fields — typed field declarations inside class bodies
(field_declaration
declarator: (variable_declarator
name: (identifier) @name)) @definition.property
; Imports - capture any import declaration child as source
(import_declaration (_) @import.source) @import
@ -196,6 +272,13 @@ export const JAVA_QUERIES = `
; Heritage - implements interfaces
(class_declaration name: (identifier) @heritage.class
(super_interfaces (type_list (type_identifier) @heritage.implements))) @heritage.impl
; Write access: obj.field = value
(assignment_expression
left: (field_access
object: (_) @assignment.receiver
field: (identifier) @assignment.property)
right: (_)) @assignment
`;
// C queries - works with tree-sitter-c
@ -243,6 +326,11 @@ export const GO_QUERIES = `
(import_declaration (import_spec path: (interpreted_string_literal) @import.source)) @import
(import_declaration (import_spec_list (import_spec path: (interpreted_string_literal) @import.source))) @import
; Struct fields — named field declarations inside struct types
(field_declaration_list
(field_declaration
name: (field_identifier) @name) @definition.property)
; Struct embedding (anonymous fields = inheritance)
(type_declaration
(type_spec
@ -258,6 +346,24 @@ export const GO_QUERIES = `
; Struct literal construction: User{Name: "Alice"}
(composite_literal type: (type_identifier) @call.name) @call
; Write access: obj.field = value
(assignment_statement
left: (expression_list
(selector_expression
operand: (_) @assignment.receiver
field: (field_identifier) @assignment.property))
right: (_)) @assignment
; Write access: obj.field++ / obj.field--
(inc_statement
(selector_expression
operand: (_) @assignment.receiver
field: (field_identifier) @assignment.property)) @assignment
(dec_statement
(selector_expression
operand: (_) @assignment.receiver
field: (field_identifier) @assignment.property)) @assignment
`;
// C++ queries - works with tree-sitter-cpp
@ -299,8 +405,30 @@ export const CPP_QUERIES = `
(declaration declarator: (function_declarator declarator: (identifier) @name)) @definition.function
(declaration declarator: (pointer_declarator declarator: (function_declarator declarator: (identifier) @name))) @definition.function
; Inline class method declarations (inside class body, no body: void Foo();)
(field_declaration declarator: (function_declarator declarator: (identifier) @name)) @definition.method
; Class/struct data member fields (Address address; int count;)
; Uses field_identifier to exclude method declarations (which use function_declarator)
(field_declaration
declarator: (field_identifier) @name) @definition.property
; Pointer member fields (Address* address;)
(field_declaration
declarator: (pointer_declarator
declarator: (field_identifier) @name)) @definition.property
; Reference member fields (Address& address;)
(field_declaration
declarator: (reference_declarator
(field_identifier) @name)) @definition.property
; Inline class method declarations (inside class body, no body: void save();)
; tree-sitter-cpp uses field_identifier (not identifier) for names inside class bodies
(field_declaration declarator: (function_declarator declarator: [(field_identifier) (identifier)] @name)) @definition.method
; Inline class method declarations returning a pointer (User* lookup();)
(field_declaration declarator: (pointer_declarator declarator: (function_declarator declarator: [(field_identifier) (identifier)] @name))) @definition.method
; Inline class method declarations returning a reference (User& lookup();)
(field_declaration declarator: (reference_declarator (function_declarator declarator: [(field_identifier) (identifier)] @name))) @definition.method
; Inline class method definitions (inside class body, with body: void Foo() { ... })
(field_declaration_list
@ -308,6 +436,20 @@ export const CPP_QUERIES = `
declarator: (function_declarator
declarator: [(field_identifier) (identifier) (operator_name) (destructor_name)] @name)) @definition.method)
; Inline class methods returning a pointer type (User* lookup(int id) { ... })
(field_declaration_list
(function_definition
declarator: (pointer_declarator
declarator: (function_declarator
declarator: [(field_identifier) (identifier) (operator_name)] @name))) @definition.method)
; Inline class methods returning a reference type (User& lookup(int id) { ... })
(field_declaration_list
(function_definition
declarator: (reference_declarator
(function_declarator
declarator: [(field_identifier) (identifier) (operator_name)] @name))) @definition.method)
; Templates
(template_declaration (class_specifier name: (type_identifier) @name)) @definition.template
(template_declaration (function_definition declarator: (function_declarator declarator: (identifier) @name))) @definition.template
@ -329,6 +471,14 @@ export const CPP_QUERIES = `
(base_class_clause (type_identifier) @heritage.extends)) @heritage
(class_specifier name: (type_identifier) @heritage.class
(base_class_clause (access_specifier) (type_identifier) @heritage.extends)) @heritage
; Write access: obj.field = value
(assignment_expression
left: (field_expression
argument: (_) @assignment.receiver
field: (field_identifier) @assignment.property)
right: (_)) @assignment
`;
// C# queries - works with tree-sitter-c-sharp
@ -383,6 +533,13 @@ export const CSHARP_QUERIES = `
(base_list (identifier) @heritage.extends)) @heritage
(class_declaration name: (identifier) @heritage.class
(base_list (generic_name (identifier) @heritage.extends))) @heritage
; Write access: obj.field = value
(assignment_expression
left: (member_access_expression
expression: (_) @assignment.receiver
name: (identifier) @assignment.property)
right: (_)) @assignment
`;
// Rust queries - works with tree-sitter-rust
@ -414,11 +571,30 @@ export const RUST_QUERIES = `
; Struct literal construction: User { name: value }
(struct_expression name: (type_identifier) @call.name) @call
; Struct fields — named field declarations inside struct bodies
(field_declaration_list
(field_declaration
name: (field_identifier) @name) @definition.property)
; Heritage (trait implementation) — all combinations of concrete/generic trait × concrete/generic type
(impl_item trait: (type_identifier) @heritage.trait type: (type_identifier) @heritage.class) @heritage
(impl_item trait: (generic_type type: (type_identifier) @heritage.trait) type: (type_identifier) @heritage.class) @heritage
(impl_item trait: (type_identifier) @heritage.trait type: (generic_type type: (type_identifier) @heritage.class)) @heritage
(impl_item trait: (generic_type type: (type_identifier) @heritage.trait) type: (generic_type type: (type_identifier) @heritage.class)) @heritage
; Write access: obj.field = value
(assignment_expression
left: (field_expression
value: (_) @assignment.receiver
field: (field_identifier) @assignment.property)
right: (_)) @assignment
; Write access: obj.field += value (compound assignment)
(compound_assignment_expr
left: (field_expression
value: (_) @assignment.receiver
field: (field_identifier) @assignment.property)
right: (_)) @assignment
`;
// PHP queries - works with tree-sitter-php (php_only grammar)
@ -457,6 +633,13 @@ export const PHP_QUERIES = `
(variable_name
(name) @name))) @definition.property
; Constructor property promotion (PHP 8.0+: public Address $address in __construct)
(method_declaration
parameters: (formal_parameters
(property_promotion_parameter
name: (variable_name
(name) @name)))) @definition.property
; ── Imports: use statements ──────────────────────────────────────────────────
; Simple: use App\\Models\\User;
(namespace_use_declaration
@ -501,6 +684,20 @@ export const PHP_QUERIES = `
body: (declaration_list
(use_declaration
[(name) (qualified_name)] @heritage.trait))) @heritage
; Write access: $obj->field = value
(assignment_expression
left: (member_access_expression
object: (_) @assignment.receiver
name: (name) @assignment.property)
right: (_)) @assignment
; Write access: ClassName::$field = value (static property)
(assignment_expression
left: (scoped_property_access_expression
scope: (_) @assignment.receiver
name: (variable_name (name) @assignment.property))
right: (_)) @assignment
`;
// Ruby queries - works with tree-sitter-ruby
@ -546,6 +743,20 @@ export const RUBY_QUERIES = `
name: (constant) @heritage.class
superclass: (superclass
(constant) @heritage.extends)) @heritage
; Write access: obj.field = value (Ruby setter — syntactically a method call to field=)
(assignment
left: (call
receiver: (_) @assignment.receiver
method: (identifier) @assignment.property)
right: (_)) @assignment
; Write access: obj.field += value (compound assignment — operator_assignment node, not assignment)
(operator_assignment
left: (call
receiver: (_) @assignment.receiver
method: (identifier) @assignment.property)
right: (_)) @assignment
`;
// Kotlin queries - works with tree-sitter-kotlin (fwcd/tree-sitter-kotlin)
@ -582,6 +793,12 @@ export const KOTLIN_QUERIES = `
(variable_declaration
(simple_identifier) @name)) @definition.property
; Primary constructor val/var parameters (data class, value class, regular class)
; binding_pattern_kind contains "val" or "var" — without it, the param is not a property
(class_parameter
(binding_pattern_kind)
(simple_identifier) @name) @definition.property
; ── Enum entries ─────────────────────────────────────────────────────────
(enum_entry
(simple_identifier) @name) @definition.enum
@ -626,6 +843,15 @@ export const KOTLIN_QUERIES = `
(delegation_specifier
(constructor_invocation
(user_type (type_identifier) @heritage.extends)))) @heritage
; Write access: obj.field = value
(assignment
(directly_assignable_expression
(_) @assignment.receiver
(navigation_suffix
(simple_identifier) @assignment.property))
(_)) @assignment
`;
// Swift queries - works with tree-sitter-swift
@ -684,6 +910,15 @@ export const SWIFT_QUERIES = `
; Extensions wrap the name in user_type unlike class/struct/enum declarations
(class_declaration "extension" name: (user_type (type_identifier) @heritage.class)
(inheritance_specifier inherits_from: (user_type (type_identifier) @heritage.extends))) @heritage
; Write access: obj.field = value
(assignment
(directly_assignable_expression
(_) @assignment.receiver
(navigation_suffix
(simple_identifier) @assignment.property))
(_)) @assignment
`;
export const LANGUAGE_QUERIES: Record<SupportedLanguages, string> = {

View file

@ -1,9 +1,9 @@
import type { SyntaxNode } from './utils.js';
import { FUNCTION_NODE_TYPES, extractFunctionName, CLASS_CONTAINER_TYPES } from './utils.js';
import { FUNCTION_NODE_TYPES, extractFunctionName, CLASS_CONTAINER_TYPES, CALL_EXPRESSION_TYPES, isBuiltInOrNoise } from './utils.js';
import { SupportedLanguages } from '../../config/supported-languages.js';
import { typeConfigs, TYPED_PARAMETER_TYPES } from './type-extractors/index.js';
import type { ClassNameLookup } from './type-extractors/types.js';
import { extractSimpleTypeName } from './type-extractors/shared.js';
import type { ClassNameLookup, ReturnTypeLookup, ForLoopExtractorContext, PendingAssignment } from './type-extractors/types.js';
import { extractSimpleTypeName, extractVarName, stripNullable, extractReturnTypeName } from './type-extractors/shared.js';
import type { SymbolTable } from './symbol-table.js';
/**
@ -12,7 +12,9 @@ import type { SymbolTable } from './symbol-table.js';
* file-level variables use the '' (empty string) scope.
*
* Design constraints:
* - Explicit-only: only type annotations, never inferred types
* - Explicit-only: Tier 0 uses type annotations; Tier 1 infers from constructors
* - Tier 2: single-pass assignment chain propagation in source order — resolves
* `const b = a` when `a` already has a type from Tier 0/1
* - Scope-aware: function-local variables don't collide across functions
* - Conservative: complex/generic types extract the base name only
* - Per-file: built once, used for receiver resolution, then discarded
@ -44,13 +46,68 @@ export interface TypeEnvironment {
readonly constructorBindings: readonly ConstructorBinding[];
/** Raw per-scope type bindings — for testing and debugging. */
readonly env: TypeEnv;
/** Maps `scope\0varName` → constructor type for virtual dispatch override.
* Populated when a variable has BOTH a declared base type AND a more specific
* constructor type (e.g., `Animal a = new Dog()` → key maps to 'Dog'). */
readonly constructorTypeMap: ReadonlyMap<string, string>;
}
/**
* Position-indexed pattern binding: active only within a specific AST range.
* Used for smart-cast narrowing in mutually exclusive branches (e.g., Kotlin when arms).
*/
interface PatternOverride {
rangeStart: number;
rangeEnd: number;
typeName: string;
}
/** scope → varName → overrides (checked in order, first range match wins) */
type PatternOverrides = Map<string, Map<string, PatternOverride[]>>;
/** AST node types that represent mutually exclusive branch containers for pattern bindings.
* Includes both multi-arm pattern-match branches AND if-statement bodies for null-check narrowing. */
const NARROWING_BRANCH_TYPES = new Set([
'when_entry', // Kotlin when
'switch_block_label', // Java switch (enhanced)
'if_statement', // TS/JS, Java, C/C++
'if_expression', // Kotlin (if is an expression)
'statement_block', // TS/JS: { ... } body of if
'control_structure_body', // Kotlin: body of if
]);
/** Walk up the AST from a pattern node to find the enclosing branch container. */
const findNarrowingBranchScope = (node: SyntaxNode): SyntaxNode | undefined => {
let current = node.parent;
while (current) {
if (NARROWING_BRANCH_TYPES.has(current.type)) return current;
if (FUNCTION_NODE_TYPES.has(current.type)) return undefined;
current = current.parent;
}
return undefined;
};
/** Bare nullable keywords that fastStripNullable must reject. */
const FAST_NULLABLE_KEYWORDS = new Set(['null', 'undefined', 'void', 'None', 'nil']);
/**
* Fast-path nullable check: 90%+ of type names are simple identifiers (e.g. "User")
* that don't need the full stripNullable parse. Only call stripNullable when the
* string contains nullable markers ('|' for union types, '?' for nullable suffix).
*/
const fastStripNullable = (typeName: string): string | undefined => {
if (FAST_NULLABLE_KEYWORDS.has(typeName)) return undefined;
return (typeName.indexOf('|') === -1 && typeName.indexOf('?') === -1)
? typeName
: stripNullable(typeName);
};
/** Implementation of the lookup logic — shared between TypeEnvironment and the legacy export. */
const lookupInEnv = (
env: TypeEnv,
varName: string,
callNode: SyntaxNode,
patternOverrides?: PatternOverrides,
): string | undefined => {
// Self/this receiver: resolve to enclosing class name via AST walk
if (varName === 'self' || varName === 'this' || varName === '$this') {
@ -66,18 +123,33 @@ const lookupInEnv = (
// Determine the enclosing function scope for the call
const scopeKey = findEnclosingScopeKey(callNode);
// Check position-indexed pattern overrides first (e.g., Kotlin when/is smart casts).
// These take priority over flat scopeEnv because they represent per-branch narrowing.
if (scopeKey && patternOverrides) {
const varOverrides = patternOverrides.get(scopeKey)?.get(varName);
if (varOverrides) {
const pos = callNode.startIndex;
for (const override of varOverrides) {
if (pos >= override.rangeStart && pos <= override.rangeEnd) {
return fastStripNullable(override.typeName);
}
}
}
}
// Try function-local scope first
if (scopeKey) {
const scopeEnv = env.get(scopeKey);
if (scopeEnv) {
const result = scopeEnv.get(varName);
if (result) return result;
if (result) return fastStripNullable(result);
}
}
// Fall back to file-level scope
const fileEnv = env.get(FILE_SCOPE);
return fileEnv?.get(varName);
const raw = fileEnv?.get(varName);
return raw ? fastStripNullable(raw) : undefined;
};
@ -98,6 +170,23 @@ const findEnclosingClassName = (node: SyntaxNode): string | undefined => {
return undefined;
};
/** Keywords that refer to the current instance across languages. */
const THIS_RECEIVERS = new Set(['this', 'self', '$this', 'Me']);
/**
* If a pending assignment's receiver is this/self/$this/Me, substitute the
* enclosing class name. Returns the item unchanged for non-receiver kinds
* or when the receiver is not a this-keyword. Properties are readonly in the
* discriminated union, so a new object is returned when substitution occurs.
*/
const substituteThisReceiver = (item: PendingAssignment, node: SyntaxNode): PendingAssignment => {
if (item.kind !== 'fieldAccess' && item.kind !== 'methodCallResult') return item;
if (!THIS_RECEIVERS.has(item.receiver)) return item;
const className = findEnclosingClassName(node);
if (!className) return item;
return { ...item, receiver: className };
};
/**
* Walk up the AST to find the enclosing class, then extract its parent class name
* from the heritage/superclass AST node. Used to resolve `super`/`base`/`parent`.
@ -262,7 +351,9 @@ const createClassNameLookup = (
if (localNames.has(name)) return true;
const cached = memo.get(name);
if (cached !== undefined) return cached;
const result = symbolTable.lookupFuzzy(name).some(def => def.type === 'Class');
const result = symbolTable.lookupFuzzy(name).some(def =>
def.type === 'Class' || def.type === 'Enum' || def.type === 'Struct',
);
memo.set(name, result);
return result;
},
@ -278,32 +369,519 @@ const createClassNameLookup = (
* the project are available for constructor inference in languages like Kotlin
* where constructors are syntactically identical to function calls.
*/
/**
* Node types whose subtrees can NEVER contain type-relevant descendants
* (declarations, parameters, for-loops, class definitions, pattern bindings).
* Conservative leaf-only set — verified safe across all 12 supported language grammars.
* IMPORTANT: Do NOT add expression containers (arguments, binary_expression, etc.) —
* they can contain arrow functions with typed parameters.
*/
const SKIP_SUBTREE_TYPES = new Set([
// Plain string literals (NOT template_string — it contains interpolated expressions
// that can hold arrow functions with typed parameters, e.g. `${(x: T) => x}`)
'string', 'string_literal',
'string_content', 'string_fragment', 'heredoc_body',
// Comments
'comment', 'line_comment', 'block_comment',
// Numeric/boolean/null literals
'number', 'integer_literal', 'float_literal',
'true', 'false', 'null',
// Regex
'regex', 'regex_pattern',
]);
const CLASS_LIKE_TYPES = new Set(['Class', 'Struct', 'Interface']);
/** Memoize class definition lookups during fixpoint iteration.
* SymbolTable is immutable during type resolution, so results never change.
* Eliminates redundant array allocations + filter scans across iterations. */
const createClassDefCache = (symbolTable?: SymbolTable) => {
const cache = new Map<string, Array<{ nodeId: string; type: string }>>();
return (typeName: string) => {
let result = cache.get(typeName);
if (result === undefined) {
result = symbolTable
? symbolTable.lookupFuzzy(typeName).filter(d => CLASS_LIKE_TYPES.has(d.type))
: [];
cache.set(typeName, result);
}
return result;
};
};
/** AST node types representing constructor expressions across languages.
* Note: C# also has `implicit_object_creation_expression` (`new()` with type
* inference) which is NOT captured — the type is inferred, not explicit.
* Kotlin constructors use `call_expression` (no `new` keyword) — not detected. */
const CONSTRUCTOR_EXPR_TYPES = new Set([
'new_expression', // TS/JS/C++: new Dog()
'object_creation_expression', // Java/C#: new Dog()
]);
/** Extract the constructor class name from a declaration node's initializer.
* Searches for new_expression / object_creation_expression in the node's subtree.
* Returns the class name or undefined if no constructor is found.
* Depth-limited to 5 to avoid expensive traversals. */
const extractConstructorTypeName = (node: SyntaxNode, depth = 0): string | undefined => {
if (depth > 5) return undefined;
if (CONSTRUCTOR_EXPR_TYPES.has(node.type)) {
// Java/C#: object_creation_expression has 'type' field
const typeField = node.childForFieldName('type');
if (typeField) return extractSimpleTypeName(typeField);
// TS/JS: new_expression has 'constructor' field (but tree-sitter often just has identifier child)
const ctorField = node.childForFieldName('constructor');
if (ctorField) return extractSimpleTypeName(ctorField);
// Fallback: first named child is often the class identifier
if (node.firstNamedChild) return extractSimpleTypeName(node.firstNamedChild);
}
for (let i = 0; i < node.namedChildCount; i++) {
const child = node.namedChild(i);
if (!child) continue;
// Don't descend into nested functions/classes or call expressions (prevents
// finding constructor args inside method calls, e.g. processAll(new Dog()))
if (FUNCTION_NODE_TYPES.has(child.type) || CLASS_CONTAINER_TYPES.has(child.type)
|| CALL_EXPRESSION_TYPES.has(child.type)) continue;
const result = extractConstructorTypeName(child, depth + 1);
if (result) return result;
}
return undefined;
};
/** Max depth for MRO parent chain walking. Real-world inheritance rarely exceeds 3-4 levels. */
const MAX_MRO_DEPTH = 5;
/** Check if `child` is a subclass of `parent` using the parentMap.
* BFS up from child, depth-limited (5), cycle-safe. */
export const isSubclassOf = (
child: string, parent: string,
parentMap: ReadonlyMap<string, readonly string[]> | undefined,
): boolean => {
if (!parentMap || child === parent) return false;
const visited = new Set<string>([child]);
let current = [child];
for (let depth = 0; depth < MAX_MRO_DEPTH && current.length > 0; depth++) {
const next: string[] = [];
for (const cls of current) {
const parents = parentMap.get(cls);
if (!parents) continue;
for (const p of parents) {
if (p === parent) return true;
if (!visited.has(p)) { visited.add(p); next.push(p); }
}
}
current = next;
}
return false;
};
/** Walk up the parent class chain to find a field or method on an ancestor.
* BFS-like traversal with depth limit and cycle detection. First match wins.
* Used by resolveFieldType and resolveMethodReturnType when direct lookup fails. */
const walkParentChain = <T>(
typeName: string,
parentMap: ReadonlyMap<string, readonly string[]> | undefined,
getClassDefs: (name: string) => Array<{ nodeId: string; type: string }>,
lookupOnClass: (nodeId: string) => T | undefined,
): T | undefined => {
if (!parentMap) return undefined;
const visited = new Set<string>([typeName]);
let current = [typeName];
for (let depth = 0; depth < MAX_MRO_DEPTH && current.length > 0; depth++) {
const next: string[] = [];
for (const cls of current) {
const parents = parentMap.get(cls);
if (!parents) continue;
for (const parent of parents) {
if (visited.has(parent)) continue;
visited.add(parent);
const parentDefs = getClassDefs(parent);
if (parentDefs.length === 1) {
const result = lookupOnClass(parentDefs[0].nodeId);
if (result !== undefined) return result;
}
next.push(parent);
}
}
current = next;
}
return undefined;
};
/** Resolve a field's declared type given a receiver variable and field name.
* Uses SymbolTable to find the class nodeId for the receiver's type, then
* looks up the field via the eagerly-populated fieldByOwner index.
* Falls back to MRO parent chain walking if direct lookup fails (Phase 11A). */
const resolveFieldType = (
receiver: string, field: string,
scopeEnv: ReadonlyMap<string, string>, symbolTable?: SymbolTable,
getClassDefs?: (typeName: string) => Array<{ nodeId: string; type: string }>,
parentMap?: ReadonlyMap<string, readonly string[]>,
): string | undefined => {
if (!symbolTable) return undefined;
const receiverType = scopeEnv.get(receiver);
if (!receiverType) return undefined;
const lookup = getClassDefs
?? ((name: string) => symbolTable.lookupFuzzy(name).filter(d => CLASS_LIKE_TYPES.has(d.type)));
const classDefs = lookup(receiverType);
if (classDefs.length !== 1) return undefined;
// Direct lookup first
const fieldDef = symbolTable.lookupFieldByOwner(classDefs[0].nodeId, field);
if (fieldDef?.declaredType) return extractReturnTypeName(fieldDef.declaredType);
// MRO parent chain walking on miss
const inherited = walkParentChain(receiverType, parentMap, lookup, (nodeId) => {
const f = symbolTable.lookupFieldByOwner(nodeId, field);
return f?.declaredType ? extractReturnTypeName(f.declaredType) : undefined;
});
return inherited;
};
/** Resolve a method's return type given a receiver variable and method name.
* Uses SymbolTable to find class nodeIds for the receiver's type, then
* looks up the method via lookupFuzzyCallable filtered by ownerId.
* Falls back to MRO parent chain walking if direct lookup fails (Phase 11A). */
const resolveMethodReturnType = (
receiver: string, method: string,
scopeEnv: ReadonlyMap<string, string>, symbolTable?: SymbolTable,
getClassDefs?: (typeName: string) => Array<{ nodeId: string; type: string }>,
parentMap?: ReadonlyMap<string, readonly string[]>,
): string | undefined => {
if (!symbolTable) return undefined;
const receiverType = scopeEnv.get(receiver);
if (!receiverType) return undefined;
const lookup = getClassDefs
?? ((name: string) => symbolTable.lookupFuzzy(name).filter(d => CLASS_LIKE_TYPES.has(d.type)));
const classDefs = lookup(receiverType);
if (classDefs.length === 0) return undefined;
// Direct lookup first
const classNodeIds = new Set(classDefs.map(d => d.nodeId));
const methods = symbolTable.lookupFuzzyCallable(method)
.filter(d => d.ownerId && classNodeIds.has(d.ownerId));
if (methods.length === 1 && methods[0].returnType) {
return extractReturnTypeName(methods[0].returnType);
}
// MRO parent chain walking on miss
if (methods.length === 0) {
const inherited = walkParentChain(receiverType, parentMap, lookup, (nodeId) => {
const parentMethods = symbolTable.lookupFuzzyCallable(method)
.filter(d => d.ownerId === nodeId);
if (parentMethods.length !== 1 || !parentMethods[0].returnType) return undefined;
return extractReturnTypeName(parentMethods[0].returnType);
});
return inherited;
}
return undefined;
};
/**
* Unified fixpoint propagation: iterate over ALL pending items (copy, callResult,
* fieldAccess, methodCallResult) until no new bindings are produced.
* Handles arbitrary-depth mixed chains:
* const user = getUser(); // callResult → User
* const addr = user.address; // fieldAccess → Address (depends on user)
* const city = addr.getCity(); // methodCallResult → City (depends on addr)
* const alias = city; // copy → City (depends on city)
* Data flow: SymbolTable (immutable) + scopeEnv → resolve → scopeEnv.
* Termination: finite entries, each bound at most once (first-writer-wins), max 10 iterations.
*/
const MAX_FIXPOINT_ITERATIONS = 10;
const resolveFixpointBindings = (
pendingItems: Array<{ scope: string } & PendingAssignment>,
env: TypeEnv,
returnTypeLookup: ReturnTypeLookup,
symbolTable?: SymbolTable,
parentMap?: ReadonlyMap<string, readonly string[]>,
): void => {
if (pendingItems.length === 0) return;
const getClassDefs = createClassDefCache(symbolTable);
const resolved = new Set<number>();
for (let iter = 0; iter < MAX_FIXPOINT_ITERATIONS; iter++) {
let changed = false;
for (let i = 0; i < pendingItems.length; i++) {
if (resolved.has(i)) continue;
const item = pendingItems[i];
const scopeEnv = env.get(item.scope);
if (!scopeEnv || scopeEnv.has(item.lhs)) { resolved.add(i); continue; }
let typeName: string | undefined;
switch (item.kind) {
case 'callResult':
typeName = returnTypeLookup.lookupReturnType(item.callee);
break;
case 'copy':
typeName = scopeEnv.get(item.rhs) ?? env.get(FILE_SCOPE)?.get(item.rhs);
break;
case 'fieldAccess':
typeName = resolveFieldType(item.receiver, item.field, scopeEnv, symbolTable, getClassDefs, parentMap);
break;
case 'methodCallResult':
typeName = resolveMethodReturnType(item.receiver, item.method, scopeEnv, symbolTable, getClassDefs, parentMap);
break;
default: {
// Exhaustive check: TypeScript will error here if a new PendingAssignment
// kind is added without handling it in the switch.
const _exhaustive: never = item;
break;
}
}
if (typeName) {
scopeEnv.set(item.lhs, typeName);
resolved.add(i);
changed = true;
}
}
if (!changed) break;
if (iter === MAX_FIXPOINT_ITERATIONS - 1 && process.env.GITNEXUS_DEBUG) {
const unresolved = pendingItems.length - resolved.size;
if (unresolved > 0) {
console.warn(`[type-env] fixpoint hit iteration cap (${MAX_FIXPOINT_ITERATIONS}), ${unresolved} items unresolved`);
}
}
}
};
/**
* Options for buildTypeEnv.
* Uses an options object to allow future extensions without positional parameter sprawl.
*/
export interface BuildTypeEnvOptions {
symbolTable?: SymbolTable;
parentMap?: ReadonlyMap<string, readonly string[]>;
/** Pre-resolved bindings from upstream files (Phase 14).
* Seeded into FILE_SCOPE after walk() for names with no local binding.
* Local declarations always take precedence (first-writer-wins). */
importedBindings?: ReadonlyMap<string, string>;
/** Cross-file return type fallback for imported callables (Phase 14 E3).
* Consulted ONLY when SymbolTable has no unambiguous match.
* Local definitions always take precedence (local-first principle). */
importedReturnTypes?: ReadonlyMap<string, string>;
/** Cross-file RAW return types for imported callables (Phase 14 E3).
* Stores raw declared return type strings (e.g., 'User[]', 'List<User>').
* Used by lookupRawReturnType for for-loop element extraction. */
importedRawReturnTypes?: ReadonlyMap<string, string>;
}
/** Seed cross-file type bindings into the file scope.
* MUST be called AFTER walk() completes so that local declarations
* (Tier 0/1) always take precedence over imported bindings (first-writer-wins). */
function seedImportedBindings(
env: TypeEnv,
importedBindings: ReadonlyMap<string, string>,
): void {
let fileEnv = env.get(FILE_SCOPE);
if (!fileEnv) { fileEnv = new Map(); env.set(FILE_SCOPE, fileEnv); }
for (const [name, type] of importedBindings) {
if (!fileEnv.has(name)) {
fileEnv.set(name, type);
}
}
}
export const buildTypeEnv = (
tree: { rootNode: SyntaxNode },
language: SupportedLanguages,
symbolTable?: SymbolTable,
options?: BuildTypeEnvOptions,
): TypeEnvironment => {
const symbolTable = options?.symbolTable;
const parentMap = options?.parentMap;
const env: TypeEnv = new Map();
const patternOverrides: PatternOverrides = new Map();
// Phase P: maps `scope\0varName` → constructor type when a declaration has BOTH
// a base type annotation AND a more specific constructor initializer.
// e.g., `Animal a = new Dog()` → constructorTypeMap.set('func@42\0a', 'Dog')
const constructorTypeMap = new Map<string, string>();
const localClassNames = new Set<string>();
const classNames = createClassNameLookup(localClassNames, symbolTable);
const config = typeConfigs[language];
const bindings: ConstructorBinding[] = [];
// Build ReturnTypeLookup: SymbolTable is authoritative when it has an unambiguous match.
// Cross-file importedReturnTypes are consulted ONLY when SymbolTable has 0 matches.
// Ambiguous (2+) → undefined, no cross-file fallback (conservative, local-first principle).
const returnTypeLookup: ReturnTypeLookup = {
lookupReturnType(callee: string): string | undefined {
// SymbolTable is authoritative when it has an unambiguous match
if (symbolTable) {
if (isBuiltInOrNoise(callee)) return undefined;
const callables = symbolTable.lookupFuzzyCallable(callee);
if (callables.length === 1) {
const rawReturn = callables[0].returnType;
if (rawReturn) return extractReturnTypeName(rawReturn);
}
// Ambiguous (2+) → return undefined (conservative, no cross-file fallback)
if (callables.length > 1) return undefined;
}
// No match (0 results or no symbolTable) → fall back to cross-file
return options?.importedReturnTypes?.get(callee);
},
lookupRawReturnType(callee: string): string | undefined {
if (symbolTable) {
if (isBuiltInOrNoise(callee)) return undefined;
const callables = symbolTable.lookupFuzzyCallable(callee);
if (callables.length === 1) return callables[0].returnType;
// Ambiguous (2+) → return undefined (conservative, no cross-file fallback)
if (callables.length > 1) return undefined;
}
// Cross-file fallback uses importedRawReturnTypes (raw declared types, e.g., 'User[]')
// NOT importedReturnTypes (which contains processed/simple types via extractReturnTypeName)
return options?.importedRawReturnTypes?.get(callee);
}
};
// Pre-compute combined set of node types that need extractTypeBinding.
// Single Set.has() replaces 3 separate checks per node in walk().
const interestingNodeTypes = new Set<string>();
TYPED_PARAMETER_TYPES.forEach(t => interestingNodeTypes.add(t));
config.declarationNodeTypes.forEach(t => interestingNodeTypes.add(t));
config.forLoopNodeTypes?.forEach(t => interestingNodeTypes.add(t));
// Tier 2: unified fixpoint propagation — collects copy, callResult, fieldAccess, and
// methodCallResult items during walk(), then iterates until no new bindings are produced.
// Handles arbitrary-depth mixed chains: callResult → fieldAccess → methodCallResult → copy.
const pendingItems: Array<{ scope: string } & PendingAssignment> = [];
// For-loop nodes whose iterable was unresolved at walk-time. Replayed after the fixpoint
// resolves the iterable's type, bridging the walk-time/fixpoint gap (Phase 10 / ex-9B).
const pendingForLoops: Array<{ node: SyntaxNode; scope: string }> = [];
// Maps `scope\0varName` → the type annotation AST node from the original declaration.
// Allows pattern extractors to navigate back to the declaration's generic type arguments
// (e.g., to extract T from Result<T, E> for `if let Ok(x) = res`).
// NOTE: This is a SUPERSET of scopeEnv — entries exist even when extractSimpleTypeName
// returns undefined for container types (User[], []User, List[User]). This is intentional:
// for-loop Strategy 1 needs the raw AST type node for exactly those container types.
const declarationTypeNodes = new Map<string, SyntaxNode>();
/**
* Try to extract a (variableName → typeName) binding from a single AST node.
*
* Resolution tiers (first match wins):
* - Tier 0: explicit type annotations via extractDeclaration
* - Tier 0: explicit type annotations via extractDeclaration / extractForLoopBinding
* - Tier 1: constructor-call inference via extractInitializer (fallback)
*
* Side effect: populates declarationTypeNodes for variables that have an explicit
* type annotation field on the declaration node. This allows pattern extractors to
* retrieve generic type arguments from the original declaration (e.g., extracting T
* from Result<T, E> for `if let Ok(x) = res`).
*/
const extractTypeBinding = (node: SyntaxNode, scopeEnv: Map<string, string>): void => {
const extractTypeBinding = (node: SyntaxNode, scopeEnv: Map<string, string>, scope: string): void => {
// This guard eliminates 90%+ of calls before any language dispatch.
if (TYPED_PARAMETER_TYPES.has(node.type)) {
// Capture the raw type annotation BEFORE extractParameter.
// Most languages use 'name' field; Rust uses 'pattern'; TS uses 'pattern' for some param types.
// Kotlin `parameter` nodes use positional children instead of named fields,
// so we fall back to scanning children by type when childForFieldName returns null.
let typeNode = node.childForFieldName('type');
if (typeNode) {
const nameNode = node.childForFieldName('name')
?? node.childForFieldName('pattern')
// Python typed_parameter: name is a positional child (identifier), not a named field
?? (node.firstNamedChild?.type === 'identifier' ? node.firstNamedChild : null);
if (nameNode) {
const varName = extractVarName(nameNode);
if (varName && !declarationTypeNodes.has(`${scope}\0${varName}`)) {
declarationTypeNodes.set(`${scope}\0${varName}`, typeNode);
}
}
} else {
// Fallback: positional children (Kotlin `parameter` → simple_identifier + user_type)
let fallbackName: SyntaxNode | null = null;
let fallbackType: SyntaxNode | null = null;
for (let i = 0; i < node.namedChildCount; i++) {
const child = node.namedChild(i);
if (!child) continue;
if (!fallbackName && (child.type === 'simple_identifier' || child.type === 'identifier')) {
fallbackName = child;
}
if (!fallbackType && (child.type === 'user_type' || child.type === 'type_identifier'
|| child.type === 'generic_type' || child.type === 'parameterized_type'
|| child.type === 'nullable_type')) {
fallbackType = child;
}
}
if (fallbackName && fallbackType) {
const varName = extractVarName(fallbackName);
if (varName && !declarationTypeNodes.has(`${scope}\0${varName}`)) {
declarationTypeNodes.set(`${scope}\0${varName}`, fallbackType);
}
}
}
config.extractParameter(node, scopeEnv);
return;
}
// For-each loop variable bindings (Java/C#/Kotlin): explicit element types in the AST.
// Checked before declarationNodeTypes — loop variables are not declarations.
if (config.forLoopNodeTypes?.has(node.type)) {
if (config.extractForLoopBinding) {
const sizeBefore = scopeEnv.size;
const forLoopCtx: ForLoopExtractorContext = { scopeEnv, declarationTypeNodes, scope, returnTypeLookup };
config.extractForLoopBinding(node, forLoopCtx);
// If no new binding was produced, the iterable's type may not yet be resolved.
// Store for post-fixpoint replay (Phase 10 / ex-9B loop-fixpoint bridge).
if (scopeEnv.size === sizeBefore) {
pendingForLoops.push({ node, scope });
}
}
return;
}
if (config.declarationNodeTypes.has(node.type)) {
// Capture the raw type annotation AST node BEFORE extractDeclaration.
// This decouples type node capture from scopeEnv success — container types
// (User[], []User, List[User]) that fail extractSimpleTypeName still get
// their AST type node recorded for Strategy 1 for-loop resolution.
// Try direct extraction first (works for Go var_spec, Python assignment, Rust let_declaration).
// Try direct type field first, then unwrap wrapper nodes (C# field_declaration,
// local_declaration_statement wrap their type inside a variable_declaration child).
let typeNode = node.childForFieldName('type');
if (!typeNode) {
// C# field_declaration / local_declaration_statement wrap type inside variable_declaration.
// Use manual loop instead of namedChildren.find() to avoid array allocation on hot path.
let wrapped = node.childForFieldName('declaration');
if (!wrapped) {
for (let i = 0; i < node.namedChildCount; i++) {
const c = node.namedChild(i);
if (c?.type === 'variable_declaration') { wrapped = c; break; }
}
}
if (wrapped) {
typeNode = wrapped.childForFieldName('type');
// Kotlin: variable_declaration stores the type as user_type / nullable_type
// child rather than a named 'type' field.
if (!typeNode) {
for (let i = 0; i < wrapped.namedChildCount; i++) {
const c = wrapped.namedChild(i);
if (c && (c.type === 'user_type' || c.type === 'nullable_type')) {
typeNode = c;
break;
}
}
}
}
}
if (typeNode) {
const nameNode = node.childForFieldName('name')
?? node.childForFieldName('left')
?? node.childForFieldName('pattern');
if (nameNode) {
const varName = extractVarName(nameNode);
if (varName && !declarationTypeNodes.has(`${scope}\0${varName}`)) {
declarationTypeNodes.set(`${scope}\0${varName}`, typeNode);
}
}
}
// Run the language-specific declaration extractor (may or may not add to scopeEnv).
const sizeBefore = typeNode ? scopeEnv.size : -1;
config.extractDeclaration(node, scopeEnv);
// Fallback: for multi-declarator languages (TS, C#, Java) where the type field
// is on variable_declarator children, capture newly-added keys.
// Map preserves insertion order, so new keys are always at the end —
// skip the first sizeBefore entries to find only newly-added variables.
if (sizeBefore >= 0 && scopeEnv.size > sizeBefore) {
let skip = sizeBefore;
for (const varName of scopeEnv.keys()) {
if (skip > 0) { skip--; continue; }
if (!declarationTypeNodes.has(`${scope}\0${varName}`)) {
declarationTypeNodes.set(`${scope}\0${varName}`, typeNode);
}
}
}
// Tier 1: constructor-call inference as fallback.
// Always called when available — each language's extractInitializer
// internally skips declarators that already have explicit annotations,
@ -311,10 +889,38 @@ export const buildTypeEnv = (
if (config.extractInitializer) {
config.extractInitializer(node, scopeEnv, classNames);
}
// Phase P: detect constructor-visible virtual dispatch.
// When a declaration has BOTH a type annotation AND a constructor initializer,
// record the constructor type for receiver override at call resolution time.
// e.g., `Animal a = new Dog()` → constructorTypeMap.set('scope\0a', 'Dog')
if (sizeBefore >= 0 && scopeEnv.size > sizeBefore) {
let ctorSkip = sizeBefore;
for (const varName of scopeEnv.keys()) {
if (ctorSkip > 0) { ctorSkip--; continue; }
const declaredType = scopeEnv.get(varName);
if (!declaredType) continue;
const ctorType = extractConstructorTypeName(node)
?? config.detectConstructorType?.(node, classNames);
if (!ctorType || ctorType === declaredType) continue;
// Unwrap wrapper types (e.g., C++ shared_ptr<Animal> → Animal) for an
// accurate isSubclassOf comparison. Language-specific via config hook.
const declTypeNode = declarationTypeNodes.get(`${scope}\0${varName}`);
const effectiveDeclaredType = (declTypeNode && config.unwrapDeclaredType)
? (config.unwrapDeclaredType(declaredType, declTypeNode) ?? declaredType)
: declaredType;
if (ctorType !== effectiveDeclaredType) {
constructorTypeMap.set(`${scope}\0${varName}`, ctorType);
}
}
}
}
};
const walk = (node: SyntaxNode, currentScope: string): void => {
// Fast skip: subtrees that can never contain type-relevant nodes (leaf-like literals).
if (SKIP_SUBTREE_TYPES.has(node.type)) return;
// Collect class/struct names as we encounter them (used by extractInitializer
// to distinguish constructor calls from function calls, e.g. C++ `User()` vs `getUser()`)
// Currently only C++ uses this locally; other languages rely on the SymbolTable path.
@ -332,18 +938,92 @@ export const buildTypeEnv = (
if (funcName) scope = `${funcName}@${node.startIndex}`;
}
// Get or create the sub-map for this scope
if (!env.has(scope)) env.set(scope, new Map());
const scopeEnv = env.get(scope)!;
// Only create scope map and call extractTypeBinding for interesting node types.
// Single Set.has() replaces 3 separate checks inside extractTypeBinding.
if (interestingNodeTypes.has(node.type)) {
if (!env.has(scope)) env.set(scope, new Map());
const scopeEnv = env.get(scope)!;
extractTypeBinding(node, scopeEnv, scope);
}
extractTypeBinding(node, scopeEnv);
// Pattern binding extraction: handles constructs that introduce NEW typed variables
// via pattern matching (e.g. `if let Some(x) = opt`, `x instanceof T t`)
// or narrow existing variables within a branch (null-check narrowing).
// Runs after Tier 0/1 so scopeEnv already contains the source variable's type.
// Conservative: extractor returns undefined when source type is unknown.
if (config.extractPatternBinding && (!config.patternBindingNodeTypes || config.patternBindingNodeTypes.has(node.type))) {
// Ensure scopeEnv exists for pattern binding reads/writes
if (!env.has(scope)) env.set(scope, new Map());
const scopeEnv = env.get(scope)!;
const patternBinding = config.extractPatternBinding(node, scopeEnv, declarationTypeNodes, scope);
if (patternBinding) {
if (patternBinding.narrowingRange) {
// Explicit narrowing range (null-check narrowing): always store in patternOverrides
// using the extractor-provided range (typically the if-body block).
if (!patternOverrides.has(scope)) patternOverrides.set(scope, new Map());
const varMap = patternOverrides.get(scope)!;
if (!varMap.has(patternBinding.varName)) varMap.set(patternBinding.varName, []);
varMap.get(patternBinding.varName)!.push({
rangeStart: patternBinding.narrowingRange.startIndex,
rangeEnd: patternBinding.narrowingRange.endIndex,
typeName: patternBinding.typeName,
});
} else if (config.allowPatternBindingOverwrite) {
// Position-indexed: store per-branch binding for smart-cast narrowing.
// Each when arm / switch case gets its own type for the variable,
// preventing cross-arm contamination (e.g., Kotlin when/is).
const branchNode = findNarrowingBranchScope(node);
if (branchNode) {
if (!patternOverrides.has(scope)) patternOverrides.set(scope, new Map());
const varMap = patternOverrides.get(scope)!;
if (!varMap.has(patternBinding.varName)) varMap.set(patternBinding.varName, []);
varMap.get(patternBinding.varName)!.push({
rangeStart: branchNode.startIndex,
rangeEnd: branchNode.endIndex,
typeName: patternBinding.typeName,
});
}
// Also store in flat scopeEnv as fallback (last arm wins — same as before
// for code that doesn't use position-indexed lookup).
scopeEnv.set(patternBinding.varName, patternBinding.typeName);
} else if (!scopeEnv.has(patternBinding.varName)) {
// First-writer-wins for languages without smart-cast overwrite (Java instanceof, etc.)
scopeEnv.set(patternBinding.varName, patternBinding.typeName);
}
}
}
// Tier 2: collect plain-identifier RHS assignments for post-walk propagation.
// Delegates to per-language extractPendingAssignment — AST shapes differ widely
// (JS uses variable_declarator/name/value, Rust uses let_declaration/pattern/value,
// Python uses assignment/left/right, Go uses short_var_declaration/expression_list).
// May return a single item or an array (for destructuring: N fieldAccess items).
if (config.extractPendingAssignment && config.declarationNodeTypes.has(node.type)) {
// scopeEnv is guaranteed to exist here because declarationNodeTypes is a subset
// of interestingNodeTypes, so extractTypeBinding already created the scope map above.
const scopeEnv = env.get(scope);
if (scopeEnv) {
const pending = config.extractPendingAssignment(node, scopeEnv);
if (pending) {
const items = Array.isArray(pending) ? pending : [pending];
for (const item of items) {
// Substitute this/self/$this/Me receivers with enclosing class name
const resolved = substituteThisReceiver(item, node);
pendingItems.push({ scope, ...resolved });
}
}
}
}
// Scan for constructor bindings that couldn't be resolved locally.
// Only collect if TypeEnv didn't already resolve this binding.
if (config.scanConstructorBinding) {
const result = config.scanConstructorBinding(node);
if (result && !scopeEnv.has(result.varName)) {
bindings.push({ scope, ...result });
if (result) {
const scopeEnv = env.get(scope);
if (!scopeEnv?.has(result.varName)) {
bindings.push({ scope, ...result });
}
}
}
@ -355,10 +1035,44 @@ export const buildTypeEnv = (
};
walk(tree.rootNode, FILE_SCOPE);
// Phase 14: Seed cross-file bindings from upstream files AFTER walk
// (local declarations from walk() take precedence — first-writer-wins)
if (options?.importedBindings && options.importedBindings.size > 0) {
seedImportedBindings(env, options.importedBindings);
}
resolveFixpointBindings(pendingItems, env, returnTypeLookup, symbolTable, parentMap);
// Post-fixpoint for-loop replay (Phase 10 / ex-9B loop-fixpoint bridge):
// For-loop nodes whose iterables were unresolved at walk-time may now be
// resolvable because the fixpoint bound the iterable's type.
// Example: `const users = getUsers(); for (const u of users) { u.save(); }`
// - walk-time: users untyped → u unresolved
// - fixpoint: users → User[]
// - replay: users now typed → u → User
if (pendingForLoops.length > 0 && config.extractForLoopBinding) {
for (const { node, scope } of pendingForLoops) {
if (!env.has(scope)) env.set(scope, new Map());
const scopeEnv = env.get(scope)!;
config.extractForLoopBinding(node, { scopeEnv, declarationTypeNodes, scope, returnTypeLookup });
}
// Re-run the main fixpoint to resolve items that depended on loop variables.
// Only needed if replay actually produced new bindings.
const unresolvedBefore = pendingItems.filter((item) => {
const scopeEnv = env.get(item.scope);
return scopeEnv && !scopeEnv.has(item.lhs);
});
if (unresolvedBefore.length > 0) {
resolveFixpointBindings(unresolvedBefore, env, returnTypeLookup, symbolTable);
}
}
return {
lookup: (varName, callNode) => lookupInEnv(env, varName, callNode),
lookup: (varName, callNode) => lookupInEnv(env, varName, callNode, patternOverrides),
constructorBindings: bindings,
env,
constructorTypeMap,
};
};

View file

@ -1,12 +1,34 @@
import type { SyntaxNode } from '../utils.js';
import type { LanguageTypeConfig, ParameterExtractor, TypeBindingExtractor, InitializerExtractor, ClassNameLookup, ConstructorBindingScanner } from './types.js';
import { extractSimpleTypeName, extractVarName } from './shared.js';
import type { LanguageTypeConfig, ParameterExtractor, TypeBindingExtractor, InitializerExtractor, ClassNameLookup, ConstructorBindingScanner, PendingAssignmentExtractor, ForLoopExtractor, LiteralTypeInferrer, ConstructorTypeDetector, DeclaredTypeUnwrapper } from './types.js';
import { extractSimpleTypeName, extractVarName, resolveIterableElementType, methodToTypeArgPosition, type TypeArgPosition } from './shared.js';
const DECLARATION_NODE_TYPES: ReadonlySet<string> = new Set([
'declaration',
'for_range_loop',
]);
/** Smart pointer factory function names that create a typed object. */
const SMART_PTR_FACTORIES = new Set([
'make_shared', 'make_unique', 'make_shared_for_overwrite',
]);
/** Smart pointer wrapper type names. When the declared type is a smart pointer,
* the inner template type is extracted for virtual dispatch comparison. */
const SMART_PTR_WRAPPERS = new Set(['shared_ptr', 'unique_ptr', 'weak_ptr']);
/** Extract the first type name from a template_argument_list child.
* Unwraps type_descriptor wrappers common in tree-sitter-cpp ASTs.
* Returns undefined if no template arguments or no type found. */
export const extractFirstTemplateTypeArg = (parentNode: SyntaxNode): string | undefined => {
const templateArgs = parentNode.children.find((c: any) => c.type === 'template_argument_list');
if (!templateArgs?.firstNamedChild) return undefined;
let argNode: any = templateArgs.firstNamedChild;
if (argNode.type === 'type_descriptor') {
const inner = argNode.childForFieldName('type');
if (inner) argNode = inner;
}
return extractSimpleTypeName(argNode) ?? undefined;
};
/** C++: Type x = ...; Type* x; Type& x; */
const extractDeclaration: TypeBindingExtractor = (node: SyntaxNode, env: Map<string, string>): void => {
const typeNode = node.childForFieldName('type');
@ -89,6 +111,27 @@ const extractInitializer: InitializerExtractor = (node: SyntaxNode, env: Map<str
} else if (func.type === 'identifier') {
const text = func.text;
if (text && classNames.has(text)) env.set(varName, text);
} else {
// auto x = std::make_shared<Dog>() — smart pointer factory via template_function.
// AST: call_expression > function: qualified_identifier > template_function
// or: call_expression > function: template_function (unqualified)
const templateFunc = func.type === 'template_function'
? func
: (func.type === 'qualified_identifier' || func.type === 'scoped_identifier')
? func.namedChildren.find((c: any) => c.type === 'template_function') ?? null
: null;
if (templateFunc) {
const nameNode = templateFunc.firstNamedChild;
if (nameNode) {
const funcName = (nameNode.type === 'qualified_identifier' || nameNode.type === 'scoped_identifier')
? nameNode.lastNamedChild?.text ?? ''
: nameNode.text;
if (SMART_PTR_FACTORIES.has(funcName)) {
const typeName = extractFirstTemplateTypeArg(templateFunc);
if (typeName) env.set(varName, typeName);
}
}
}
}
return;
}
@ -160,10 +203,294 @@ const scanConstructorBinding: ConstructorBindingScanner = (node) => {
return { varName, calleeName: func.text };
};
/** C++: auto alias = user → declaration with auto type + init_declarator where value is identifier */
const extractPendingAssignment: PendingAssignmentExtractor = (node, scopeEnv) => {
if (node.type !== 'declaration') return undefined;
const typeNode = node.childForFieldName('type');
if (!typeNode) return undefined;
// Only handle auto — typed declarations already resolved by extractDeclaration
const typeText = typeNode.text;
if (typeText !== 'auto' && typeText !== 'decltype(auto)'
&& typeNode.type !== 'placeholder_type_specifier') return undefined;
const declarator = node.childForFieldName('declarator');
if (!declarator || declarator.type !== 'init_declarator') return undefined;
const value = declarator.childForFieldName('value');
if (!value) return undefined;
const nameNode = declarator.childForFieldName('declarator');
if (!nameNode) return undefined;
const finalName = nameNode.type === 'pointer_declarator' || nameNode.type === 'reference_declarator'
? nameNode.firstNamedChild : nameNode;
if (!finalName) return undefined;
const lhs = extractVarName(finalName);
if (!lhs || scopeEnv.has(lhs)) return undefined;
if (value.type === 'identifier') return { kind: 'copy', lhs, rhs: value.text };
// field_expression RHS → fieldAccess (a.field)
if (value.type === 'field_expression') {
const obj = value.firstNamedChild;
const field = value.lastNamedChild;
if (obj?.type === 'identifier' && field?.type === 'field_identifier') {
return { kind: 'fieldAccess', lhs, receiver: obj.text, field: field.text };
}
}
// call_expression RHS
if (value.type === 'call_expression') {
const funcNode = value.childForFieldName('function');
if (funcNode?.type === 'identifier') {
return { kind: 'callResult', lhs, callee: funcNode.text };
}
// method call with receiver: call_expression → function: field_expression
if (funcNode?.type === 'field_expression') {
const obj = funcNode.firstNamedChild;
const field = funcNode.lastNamedChild;
if (obj?.type === 'identifier' && field?.type === 'field_identifier') {
return { kind: 'methodCallResult', lhs, receiver: obj.text, method: field.text };
}
}
}
return undefined;
};
// --- For-loop Tier 1c ---
const FOR_LOOP_NODE_TYPES: ReadonlySet<string> = new Set(['for_range_loop']);
/** Extract template type arguments from a C++ template_type node.
* C++ template_type uses template_argument_list (not type_arguments), and each
* argument is a type_descriptor with a 'type' field containing the type_specifier. */
const extractCppTemplateTypeArgs = (templateTypeNode: SyntaxNode): string[] => {
const argsNode = templateTypeNode.childForFieldName('arguments');
if (!argsNode || argsNode.type !== 'template_argument_list') return [];
const result: string[] = [];
for (let i = 0; i < argsNode.namedChildCount; i++) {
let argNode = argsNode.namedChild(i);
if (!argNode) continue;
// type_descriptor wraps the actual type specifier in a 'type' field
if (argNode.type === 'type_descriptor') {
const inner = argNode.childForFieldName('type');
if (inner) argNode = inner;
}
const name = extractSimpleTypeName(argNode);
if (name) result.push(name);
}
return result;
};
/** Extract element type from a C++ type annotation AST node.
* Handles: template_type (vector<User>, map<string, User>),
* pointer/reference types (User*, User&). */
const extractCppElementTypeFromTypeNode = (typeNode: SyntaxNode, pos: TypeArgPosition = 'last', depth = 0): string | undefined => {
if (depth > 50) return undefined;
// template_type: vector<User>, map<string, User> — extract type arg based on position
if (typeNode.type === 'template_type') {
const args = extractCppTemplateTypeArgs(typeNode);
if (args.length >= 1) return pos === 'first' ? args[0] : args[args.length - 1];
}
// reference/pointer types: unwrap and recurse (vector<User>& → vector<User>)
if (typeNode.type === 'reference_type' || typeNode.type === 'pointer_type'
|| typeNode.type === 'type_descriptor') {
const inner = typeNode.lastNamedChild;
if (inner) return extractCppElementTypeFromTypeNode(inner, pos, depth + 1);
}
// qualified/scoped types: std::vector<User> → unwrap to template_type child
if (typeNode.type === 'qualified_identifier' || typeNode.type === 'scoped_type_identifier') {
const inner = typeNode.lastNamedChild;
if (inner) return extractCppElementTypeFromTypeNode(inner, pos, depth + 1);
}
return undefined;
};
/** Walk up from a for-range-loop to the enclosing function_definition and search parameters
* for one named `iterableName`. Returns the element type from its annotation. */
const findCppParamElementType = (iterableName: string, startNode: SyntaxNode, pos: TypeArgPosition = 'last'): string | undefined => {
let current: SyntaxNode | null = startNode.parent;
while (current) {
if (current.type === 'function_definition') {
const declarator = current.childForFieldName('declarator');
// function_definition > declarator (function_declarator) > parameters (parameter_list)
const paramsNode = declarator?.childForFieldName('parameters');
if (paramsNode) {
for (let i = 0; i < paramsNode.namedChildCount; i++) {
const param = paramsNode.namedChild(i);
if (!param || param.type !== 'parameter_declaration') continue;
const paramDeclarator = param.childForFieldName('declarator');
if (!paramDeclarator) continue;
// Unwrap reference/pointer declarators: vector<User>& users → &users
let identNode = paramDeclarator;
if (identNode.type === 'reference_declarator' || identNode.type === 'pointer_declarator') {
identNode = identNode.firstNamedChild ?? identNode;
}
if (identNode.text !== iterableName) continue;
const typeNode = param.childForFieldName('type');
if (typeNode) return extractCppElementTypeFromTypeNode(typeNode, pos);
}
}
break;
}
current = current.parent;
}
return undefined;
};
/** C++: for (auto& user : users) — extract loop variable binding.
* Handles explicit types (for (User& user : users)) and auto (for (auto& user : users)).
* For auto, resolves element type from the iterable's container type. */
const extractForLoopBinding: ForLoopExtractor = (node, { scopeEnv, declarationTypeNodes, scope } ): void => {
if (node.type !== 'for_range_loop') return;
const typeNode = node.childForFieldName('type');
const declaratorNode = node.childForFieldName('declarator');
const rightNode = node.childForFieldName('right');
if (!typeNode || !declaratorNode || !rightNode) return;
// Unwrap reference/pointer declarator to get the loop variable name
let nameNode = declaratorNode;
if (nameNode.type === 'reference_declarator' || nameNode.type === 'pointer_declarator') {
nameNode = nameNode.firstNamedChild ?? nameNode;
}
// Handle structured bindings: auto& [key, value] or auto [key, value]
// Bind the last identifier (value heuristic for [key, value] patterns)
let loopVarName: string | undefined;
if (nameNode.type === 'structured_binding_declarator') {
const lastChild = nameNode.lastNamedChild;
if (lastChild?.type === 'identifier') {
loopVarName = lastChild.text;
}
} else if (declaratorNode.type === 'structured_binding_declarator') {
const lastChild = declaratorNode.lastNamedChild;
if (lastChild?.type === 'identifier') {
loopVarName = lastChild.text;
}
}
const varName = loopVarName ?? extractVarName(nameNode);
if (!varName) return;
// Check if the type is auto/placeholder — if not, use the explicit type directly
const isAuto = typeNode.type === 'placeholder_type_specifier'
|| typeNode.text === 'auto'
|| typeNode.text === 'const auto'
|| typeNode.text === 'decltype(auto)';
if (!isAuto) {
// Explicit type: for (User& user : users) — extract directly
const typeName = extractSimpleTypeName(typeNode);
if (typeName) scopeEnv.set(varName, typeName);
return;
}
// auto/const auto/auto& — resolve from the iterable's container type
// Extract iterable name + optional method
let iterableName: string | undefined;
let methodName: string | undefined;
if (rightNode.type === 'identifier') {
iterableName = rightNode.text;
} else if (rightNode.type === 'field_expression') {
const prop = rightNode.lastNamedChild;
if (prop) iterableName = prop.text;
} else if (rightNode.type === 'call_expression') {
// users.begin() is NOT used in range-for, but container.items() etc. might be
const fieldExpr = rightNode.childForFieldName('function');
if (fieldExpr?.type === 'field_expression') {
const obj = fieldExpr.firstNamedChild;
if (obj?.type === 'identifier') iterableName = obj.text;
const field = fieldExpr.lastNamedChild;
if (field?.type === 'field_identifier') methodName = field.text;
}
} else if (rightNode.type === 'pointer_expression') {
// Dereference: for (auto& user : *ptr) → pointer_expression > identifier
// Only handles simple *identifier; *this->field and **ptr are not resolved.
const operand = rightNode.lastNamedChild;
if (operand?.type === 'identifier') iterableName = operand.text;
}
if (!iterableName) return;
const containerTypeName = scopeEnv.get(iterableName);
const typeArgPos = methodToTypeArgPosition(methodName, containerTypeName);
const elementType = resolveIterableElementType(
iterableName, node, scopeEnv, declarationTypeNodes, scope,
extractCppElementTypeFromTypeNode, findCppParamElementType,
typeArgPos,
);
if (elementType) scopeEnv.set(varName, elementType);
};
/** Infer the type of a literal AST node for C++ overload disambiguation. */
const inferLiteralType: LiteralTypeInferrer = (node) => {
switch (node.type) {
case 'number_literal': {
const t = node.text;
// Float suffixes
if (t.endsWith('f') || t.endsWith('F')) return 'float';
if (t.includes('.') || t.includes('e') || t.includes('E')) return 'double';
// Long suffix
if (t.endsWith('L') || t.endsWith('l') || t.endsWith('LL') || t.endsWith('ll')) return 'long';
return 'int';
}
case 'string_literal':
case 'raw_string_literal':
case 'concatenated_string':
return 'string';
case 'char_literal':
return 'char';
case 'true':
case 'false':
return 'bool';
case 'null':
case 'nullptr':
return 'null';
default:
return undefined;
}
};
/** C++: detect constructor type from smart pointer factory calls (make_shared<Dog>()).
* Extracts the template type argument as the constructor type for virtual dispatch. */
const detectCppConstructorType: ConstructorTypeDetector = (node, classNames) => {
// Navigate to the initializer value in the declaration
const declarator = node.childForFieldName('declarator');
const initDecl = declarator?.type === 'init_declarator' ? declarator : undefined;
if (!initDecl) return undefined;
const value = initDecl.childForFieldName('value');
if (!value || value.type !== 'call_expression') return undefined;
// Check for template_function pattern: make_shared<Dog>()
const func = value.childForFieldName('function');
if (!func || func.type !== 'template_function') return undefined;
// Extract function name (possibly qualified: std::make_shared)
const nameNode = func.firstNamedChild;
if (!nameNode) return undefined;
let funcName: string;
if (nameNode.type === 'qualified_identifier' || nameNode.type === 'scoped_identifier') {
funcName = nameNode.lastNamedChild?.text ?? '';
} else {
funcName = nameNode.text;
}
if (!SMART_PTR_FACTORIES.has(funcName)) return undefined;
// Extract template type argument
return extractFirstTemplateTypeArg(func);
};
/** Unwrap a C++ smart pointer declared type to its inner template type.
* E.g., shared_ptr<Animal> → Animal. Returns the original name if not a smart pointer. */
const unwrapCppDeclaredType: DeclaredTypeUnwrapper = (declaredType, typeNode) => {
if (!SMART_PTR_WRAPPERS.has(declaredType)) return declaredType;
if (typeNode.type !== 'template_type') return declaredType;
return extractFirstTemplateTypeArg(typeNode) ?? declaredType;
};
export const typeConfig: LanguageTypeConfig = {
declarationNodeTypes: DECLARATION_NODE_TYPES,
forLoopNodeTypes: FOR_LOOP_NODE_TYPES,
extractDeclaration,
extractParameter,
extractInitializer,
scanConstructorBinding,
extractForLoopBinding,
extractPendingAssignment,
inferLiteralType,
detectConstructorType: detectCppConstructorType,
unwrapDeclaredType: unwrapCppDeclaredType,
};

View file

@ -1,31 +1,19 @@
import type { SyntaxNode } from '../utils.js';
import type { ConstructorBindingScanner, LanguageTypeConfig, ParameterExtractor, TypeBindingExtractor } from './types.js';
import { extractSimpleTypeName, extractVarName, findChildByType, unwrapAwait } from './shared.js';
import type { ConstructorBindingScanner, ForLoopExtractor, LanguageTypeConfig, ParameterExtractor, TypeBindingExtractor, PendingAssignmentExtractor, PatternBindingExtractor, LiteralTypeInferrer } from './types.js';
import { extractSimpleTypeName, extractVarName, unwrapAwait, resolveIterableElementType, methodToTypeArgPosition, extractElementTypeFromString, type TypeArgPosition } from './shared.js';
import { findChild } from '../resolvers/utils.js';
/** Known container property accessors that operate on the container itself (e.g., dict.Keys, dict.Values) */
const KNOWN_CONTAINER_PROPS: ReadonlySet<string> = new Set(['Keys', 'Values']);
const DECLARATION_NODE_TYPES: ReadonlySet<string> = new Set([
'local_declaration_statement',
'variable_declaration',
'field_declaration',
'is_pattern_expression',
]);
/** C#: Type x = ...; var x = new Type(); obj is Type x */
/** C#: Type x = ...; var x = new Type(); */
const extractDeclaration: TypeBindingExtractor = (node: SyntaxNode, env: Map<string, string>): void => {
// C# pattern matching: `obj is User user` → is_pattern_expression > declaration_pattern
if (node.type === 'is_pattern_expression') {
const pattern = node.childForFieldName('pattern');
if (pattern?.type === 'declaration_pattern') {
const typeNode = pattern.childForFieldName('type');
const nameNode = pattern.childForFieldName('name');
if (typeNode && nameNode) {
const typeName = extractSimpleTypeName(typeNode);
const varName = extractVarName(nameNode);
if (typeName && varName) env.set(varName, typeName);
}
}
return;
}
// C# tree-sitter: local_declaration_statement > variable_declaration > ...
// Recursively descend through wrapper nodes
for (let i = 0; i < node.namedChildCount; i++) {
@ -63,8 +51,8 @@ const extractDeclaration: TypeBindingExtractor = (node: SyntaxNode, env: Map<str
// tree-sitter-c-sharp may put object_creation_expression as direct child
// or inside equals_value_clause depending on grammar version
if (declarators.length === 1) {
const initializer = findChildByType(declarators[0], 'object_creation_expression')
?? findChildByType(declarators[0], 'equals_value_clause')?.firstNamedChild;
const initializer = findChild(declarators[0], 'object_creation_expression')
?? findChild(declarators[0], 'equals_value_clause')?.firstNamedChild;
if (initializer?.type === 'object_creation_expression') {
const ctorType = initializer.childForFieldName('type');
if (ctorType) typeName = extractSimpleTypeName(ctorType);
@ -143,9 +131,374 @@ const scanConstructorBinding: ConstructorBindingScanner = (node) => {
return { varName: nameNode.text, calleeName };
};
const FOR_LOOP_NODE_TYPES: ReadonlySet<string> = new Set([
'foreach_statement',
]);
/** Extract element type from a C# type annotation AST node.
* Handles generic_name (List<User>), array_type (User[]), nullable_type (?).
* `pos` selects which type arg: 'first' for keys, 'last' for values (default). */
const extractCSharpElementTypeFromTypeNode = (typeNode: SyntaxNode, pos: TypeArgPosition = 'last', depth = 0): string | undefined => {
if (depth > 50) return undefined;
// generic_name: List<User>, IEnumerable<User>, Dictionary<string, User>
// C# uses generic_name (not generic_type)
if (typeNode.type === 'generic_name') {
const argList = findChild(typeNode, 'type_argument_list');
if (argList && argList.namedChildCount >= 1) {
if (pos === 'first') {
const firstArg = argList.namedChild(0);
if (firstArg) return extractSimpleTypeName(firstArg);
} else {
const lastArg = argList.namedChild(argList.namedChildCount - 1);
if (lastArg) return extractSimpleTypeName(lastArg);
}
}
}
// array_type: User[]
if (typeNode.type === 'array_type') {
const elemNode = typeNode.firstNamedChild;
if (elemNode) return extractSimpleTypeName(elemNode);
}
// nullable_type: unwrap and recurse (List<User>? → List<User> → User)
if (typeNode.type === 'nullable_type') {
const inner = typeNode.firstNamedChild;
if (inner) return extractCSharpElementTypeFromTypeNode(inner, pos, depth + 1);
}
return undefined;
};
/** Walk up from a foreach to the enclosing method and search parameters. */
const findCSharpParamElementType = (iterableName: string, startNode: SyntaxNode, pos: TypeArgPosition = 'last'): string | undefined => {
let current: SyntaxNode | null = startNode.parent;
while (current) {
if (current.type === 'method_declaration' || current.type === 'local_function_statement') {
const paramsNode = current.childForFieldName('parameters');
if (paramsNode) {
for (let i = 0; i < paramsNode.namedChildCount; i++) {
const param = paramsNode.namedChild(i);
if (!param || param.type !== 'parameter') continue;
const nameNode = param.childForFieldName('name');
if (nameNode?.text !== iterableName) continue;
const typeNode = param.childForFieldName('type');
if (typeNode) return extractCSharpElementTypeFromTypeNode(typeNode, pos);
}
}
break;
}
current = current.parent;
}
return undefined;
};
/** C#: foreach (User user in users) — extract loop variable binding.
* Tier 1c: for `foreach (var user in users)`, resolves element type from iterable. */
const extractForLoopBinding: ForLoopExtractor = (node, { scopeEnv, declarationTypeNodes, scope, returnTypeLookup }): void => {
const typeNode = node.childForFieldName('type');
const nameNode = node.childForFieldName('left');
if (!typeNode || !nameNode) return;
const varName = extractVarName(nameNode);
if (!varName) return;
// Explicit type (existing behavior): foreach (User user in users)
if (!(typeNode.type === 'implicit_type' && typeNode.text === 'var')) {
const typeName = extractSimpleTypeName(typeNode);
if (typeName) scopeEnv.set(varName, typeName);
return;
}
// Tier 1c: implicit type (var) — resolve from iterable's container type
const rightNode = node.childForFieldName('right');
let iterableName: string | undefined;
let methodName: string | undefined;
let callExprElementType: string | undefined;
if (rightNode?.type === 'identifier') {
iterableName = rightNode.text;
} else if (rightNode?.type === 'member_access_expression') {
// C# property access: data.Keys, data.Values → member_access_expression
// Also handles bare member access: this.users, repo.users → use property as iterableName
const obj = rightNode.childForFieldName('expression');
const prop = rightNode.childForFieldName('name');
const propText = prop?.type === 'identifier' ? prop.text : undefined;
if (propText && KNOWN_CONTAINER_PROPS.has(propText)) {
if (obj?.type === 'identifier') {
iterableName = obj.text;
} else if (obj?.type === 'member_access_expression') {
// Nested member access: this.data.Values → obj is "this.data", extract "data"
const innerProp = obj.childForFieldName('name');
if (innerProp) iterableName = innerProp.text;
}
methodName = propText;
} else if (propText) {
// Bare member access: this.users → use property name for scopeEnv lookup
iterableName = propText;
}
} else if (rightNode?.type === 'invocation_expression') {
// C# method call: data.Select(...) → invocation_expression > member_access_expression
// Direct function call: GetUsers() → invocation_expression > identifier
const fn = rightNode.firstNamedChild;
if (fn?.type === 'member_access_expression') {
const obj = fn.childForFieldName('expression');
const prop = fn.childForFieldName('name');
if (obj?.type === 'identifier') iterableName = obj.text;
if (prop?.type === 'identifier') methodName = prop.text;
} else if (fn?.type === 'identifier') {
// Direct function call: foreach (var u in GetUsers())
const rawReturn = returnTypeLookup.lookupRawReturnType(fn.text);
if (rawReturn) callExprElementType = extractElementTypeFromString(rawReturn);
}
}
if (!iterableName && !callExprElementType) return;
let elementType: string | undefined;
if (callExprElementType) {
elementType = callExprElementType;
} else {
const containerTypeName = scopeEnv.get(iterableName!);
const typeArgPos = methodToTypeArgPosition(methodName, containerTypeName);
elementType = resolveIterableElementType(
iterableName!, node, scopeEnv, declarationTypeNodes, scope,
extractCSharpElementTypeFromTypeNode, findCSharpParamElementType,
typeArgPos,
);
}
if (elementType) scopeEnv.set(varName, elementType);
};
/**
* C# pattern binding extractor for `obj is Type variable` (type pattern).
*
* AST structure:
* is_pattern_expression
* expression: (the variable being tested)
* pattern: declaration_pattern
* type: (the declared type)
* name: single_variable_designation > identifier (the new variable name)
*
* Conservative: returns undefined when the pattern field is absent, is not a
* declaration_pattern, or when the type/name cannot be extracted.
* No scopeEnv lookup is needed — the pattern explicitly declares the new variable's type.
*/
/**
* Find the if-body (consequence) block for a C# null-check.
* Walks up from the expression to find the enclosing if_statement,
* then returns its first block child (the truthy branch body).
*/
const findCSharpIfConsequenceBlock = (expr: SyntaxNode): SyntaxNode | undefined => {
let current = expr.parent;
while (current) {
if (current.type === 'if_statement') {
// C# if_statement consequence is the 'consequence' field or first block child
const consequence = current.childForFieldName('consequence');
if (consequence) return consequence;
for (let i = 0; i < current.childCount; i++) {
const child = current.child(i);
if (child?.type === 'block') return child;
}
return undefined;
}
if (current.type === 'block' || current.type === 'method_declaration'
|| current.type === 'constructor_declaration' || current.type === 'local_function_statement'
|| current.type === 'lambda_expression') return undefined;
current = current.parent;
}
return undefined;
};
/** Check if a C# declaration type node represents a nullable type.
* Checks for nullable_type AST node or '?' in the type text (e.g., User?). */
const isCSharpNullableDecl = (declTypeNode: SyntaxNode): boolean => {
if (declTypeNode.type === 'nullable_type') return true;
return declTypeNode.text.includes('?');
};
const extractPatternBinding: PatternBindingExtractor = (node, scopeEnv, declarationTypeNodes, scope) => {
// is_pattern_expression: `obj is User user` — has a declaration_pattern child
// Also handles `x is not null` for null-check narrowing
if (node.type === 'is_pattern_expression') {
const pattern = node.childForFieldName('pattern');
if (!pattern) return undefined;
// Standard type pattern: `obj is User user`
if (pattern.type === 'declaration_pattern' || pattern.type === 'recursive_pattern') {
const typeNode = pattern.childForFieldName('type');
const nameNode = pattern.childForFieldName('name');
if (!typeNode || !nameNode) return undefined;
const typeName = extractSimpleTypeName(typeNode);
const varName = extractVarName(nameNode);
if (!typeName || !varName) return undefined;
return { varName, typeName };
}
// Null-check: `x is not null` — negated_pattern > constant_pattern > null_literal
if (pattern.type === 'negated_pattern') {
const inner = pattern.firstNamedChild;
if (inner?.type === 'constant_pattern') {
const literal = inner.firstNamedChild ?? inner.firstChild;
if (literal?.type === 'null_literal' || literal?.text === 'null') {
const expr = node.childForFieldName('expression');
if (!expr || expr.type !== 'identifier') return undefined;
const varName = expr.text;
const resolvedType = scopeEnv.get(varName);
if (!resolvedType) return undefined;
// Verify the original declaration was nullable
const declTypeNode = declarationTypeNodes.get(`${scope}\0${varName}`);
if (!declTypeNode || !isCSharpNullableDecl(declTypeNode)) return undefined;
const ifBody = findCSharpIfConsequenceBlock(node);
if (!ifBody) return undefined;
return {
varName,
typeName: resolvedType,
narrowingRange: { startIndex: ifBody.startIndex, endIndex: ifBody.endIndex },
};
}
}
}
return undefined;
}
// declaration_pattern / recursive_pattern: standalone in switch statements and switch expressions
// `case User u:` or `User u =>` or `User { Name: "Alice" } u =>`
// Both use the same 'type' and 'name' fields.
if (node.type === 'declaration_pattern' || node.type === 'recursive_pattern') {
const typeNode = node.childForFieldName('type');
const nameNode = node.childForFieldName('name');
if (!typeNode || !nameNode) return undefined;
const typeName = extractSimpleTypeName(typeNode);
const varName = extractVarName(nameNode);
if (!typeName || !varName) return undefined;
return { varName, typeName };
}
// Null-check: `x != null` — binary_expression with != operator
if (node.type === 'binary_expression') {
const op = node.children.find(c => !c.isNamed && c.text === '!=');
if (!op) return undefined;
const left = node.namedChild(0);
const right = node.namedChild(1);
if (!left || !right) return undefined;
let varNode: SyntaxNode | undefined;
if (left.type === 'identifier' && (right.type === 'null_literal' || right.text === 'null')) {
varNode = left;
} else if (right.type === 'identifier' && (left.type === 'null_literal' || left.text === 'null')) {
varNode = right;
}
if (!varNode) return undefined;
const varName = varNode.text;
const resolvedType = scopeEnv.get(varName);
if (!resolvedType) return undefined;
// Verify the original declaration was nullable
const declTypeNode = declarationTypeNodes.get(`${scope}\0${varName}`);
if (!declTypeNode || !isCSharpNullableDecl(declTypeNode)) return undefined;
const ifBody = findCSharpIfConsequenceBlock(node);
if (!ifBody) return undefined;
return {
varName,
typeName: resolvedType,
narrowingRange: { startIndex: ifBody.startIndex, endIndex: ifBody.endIndex },
};
}
return undefined;
};
/** C#: var alias = u → variable_declarator with name + equals_value_clause.
* Only local_declaration_statement and variable_declaration contain variable_declarator children;
* is_pattern_expression and field_declaration never do — skip them early. */
const extractPendingAssignment: PendingAssignmentExtractor = (node, scopeEnv) => {
if (node.type === 'is_pattern_expression' || node.type === 'field_declaration') return undefined;
for (let i = 0; i < node.namedChildCount; i++) {
const child = node.namedChild(i);
if (!child || child.type !== 'variable_declarator') continue;
const nameNode = child.childForFieldName('name');
if (!nameNode) continue;
const lhs = nameNode.text;
if (scopeEnv.has(lhs)) continue;
// C# wraps value in equals_value_clause; fall back to last named child
let evc: SyntaxNode | null = null;
for (let j = 0; j < child.childCount; j++) {
if (child.child(j)?.type === 'equals_value_clause') { evc = child.child(j); break; }
}
const valueNode = evc?.firstNamedChild ?? child.namedChild(child.namedChildCount - 1);
if (valueNode && valueNode !== nameNode && (valueNode.type === 'identifier' || valueNode.type === 'simple_identifier')) {
return { kind: 'copy', lhs, rhs: valueNode.text };
}
// member_access_expression RHS → fieldAccess (a.Field)
if (valueNode?.type === 'member_access_expression') {
const expr = valueNode.childForFieldName('expression');
const name = valueNode.childForFieldName('name');
if (expr?.type === 'identifier' && name?.type === 'identifier') {
return { kind: 'fieldAccess', lhs, receiver: expr.text, field: name.text };
}
}
// invocation_expression RHS
if (valueNode?.type === 'invocation_expression') {
const funcNode = valueNode.firstNamedChild;
if (funcNode?.type === 'identifier_name' || funcNode?.type === 'identifier') {
return { kind: 'callResult', lhs, callee: funcNode.text };
}
// method call with receiver → methodCallResult: a.GetC()
if (funcNode?.type === 'member_access_expression') {
const expr = funcNode.childForFieldName('expression');
const name = funcNode.childForFieldName('name');
if (expr?.type === 'identifier' && name?.type === 'identifier') {
return { kind: 'methodCallResult', lhs, receiver: expr.text, method: name.text };
}
}
}
// await_expression → unwrap and check inner
if (valueNode?.type === 'await_expression') {
const inner = valueNode.firstNamedChild;
if (inner?.type === 'invocation_expression') {
const funcNode = inner.firstNamedChild;
if (funcNode?.type === 'identifier_name' || funcNode?.type === 'identifier') {
return { kind: 'callResult', lhs, callee: funcNode.text };
}
if (funcNode?.type === 'member_access_expression') {
const expr = funcNode.childForFieldName('expression');
const name = funcNode.childForFieldName('name');
if (expr?.type === 'identifier' && name?.type === 'identifier') {
return { kind: 'methodCallResult', lhs, receiver: expr.text, method: name.text };
}
}
}
}
}
return undefined;
};
/** Infer the type of a literal AST node for C# overload disambiguation. */
const inferLiteralType: LiteralTypeInferrer = (node) => {
switch (node.type) {
case 'integer_literal':
if (node.text.endsWith('L') || node.text.endsWith('l')) return 'long';
return 'int';
case 'real_literal':
if (node.text.endsWith('f') || node.text.endsWith('F')) return 'float';
if (node.text.endsWith('m') || node.text.endsWith('M')) return 'decimal';
return 'double';
case 'string_literal':
case 'verbatim_string_literal':
case 'raw_string_literal':
case 'interpolated_string_expression':
return 'string';
case 'character_literal':
return 'char';
case 'boolean_literal':
return 'bool';
case 'null_literal':
return 'null';
default:
return undefined;
}
};
export const typeConfig: LanguageTypeConfig = {
declarationNodeTypes: DECLARATION_NODE_TYPES,
forLoopNodeTypes: FOR_LOOP_NODE_TYPES,
patternBindingNodeTypes: new Set(['is_pattern_expression', 'declaration_pattern', 'recursive_pattern', 'binary_expression']),
extractDeclaration,
extractParameter,
scanConstructorBinding,
extractForLoopBinding,
extractPendingAssignment,
extractPatternBinding,
inferLiteralType,
};

View file

@ -1,6 +1,6 @@
import type { SyntaxNode } from '../utils.js';
import type { ConstructorBindingScanner, LanguageTypeConfig, ParameterExtractor, TypeBindingExtractor } from './types.js';
import { extractSimpleTypeName, extractVarName } from './shared.js';
import type { ConstructorBindingScanner, ForLoopExtractor, LanguageTypeConfig, ParameterExtractor, TypeBindingExtractor, PendingAssignmentExtractor } from './types.js';
import { extractSimpleTypeName, extractVarName, extractElementTypeFromString, extractGenericTypeArgs, resolveIterableElementType, methodToTypeArgPosition, type TypeArgPosition } from './shared.js';
const DECLARATION_NODE_TYPES: ReadonlySet<string> = new Set([
'var_declaration',
@ -181,9 +181,303 @@ const scanConstructorBinding: ConstructorBindingScanner = (node) => {
return { varName: leftIds[0].text, calleeName };
};
const FOR_LOOP_NODE_TYPES: ReadonlySet<string> = new Set([
'for_statement',
]);
/** Go function/method node types that carry a parameter list. */
const GO_FUNCTION_NODE_TYPES = new Set([
'function_declaration', 'method_declaration', 'func_literal',
]);
/**
* Extract element type from a Go type annotation AST node.
* Handles:
* slice_type "[]User" → element field → type_identifier "User"
* array_type "[10]User" → element field → type_identifier "User"
* Falls back to text-based extraction via extractElementTypeFromString.
*/
const extractGoElementTypeFromTypeNode = (typeNode: SyntaxNode, pos: TypeArgPosition = 'last'): string | undefined => {
// slice_type: []User — element field is the element type
if (typeNode.type === 'slice_type' || typeNode.type === 'array_type') {
const elemNode = typeNode.childForFieldName('element');
if (elemNode) return extractSimpleTypeName(elemNode);
}
// map_type: map[string]User — value field is the element type (for range, second var gets value)
if (typeNode.type === 'map_type') {
const valueNode = typeNode.childForFieldName('value');
if (valueNode) return extractSimpleTypeName(valueNode);
}
// channel_type: chan User — the type argument is the element type
if (typeNode.type === 'channel_type') {
const valueNode = typeNode.childForFieldName('value') ?? typeNode.lastNamedChild;
if (valueNode) return extractSimpleTypeName(valueNode);
}
// generic_type: Go 1.18+ generics (e.g., MySlice[User], Cache[string, User])
// Use position-aware arg selection: 'first' for keys, 'last' for values.
if (typeNode.type === 'generic_type') {
const args = extractGenericTypeArgs(typeNode);
if (args.length >= 1) return pos === 'first' ? args[0] : args[args.length - 1];
}
// Fallback: text-based extraction ([]User → User, User[] → User)
return extractElementTypeFromString(typeNode.text, pos);
};
/** Check if a Go type node represents a channel type. Used to determine
* whether single-var range yields the element (channels) vs index (slices/maps). */
const isChannelType = (
iterableName: string,
scopeEnv: ReadonlyMap<string, string>,
declarationTypeNodes?: ReadonlyMap<string, SyntaxNode>,
scope?: string,
): boolean => {
if (declarationTypeNodes && scope) {
const typeNode = declarationTypeNodes.get(`${scope}\0${iterableName}`);
if (typeNode) return typeNode.type === 'channel_type';
}
const t = scopeEnv.get(iterableName);
return !!t && t.startsWith('chan ');
};
/**
* Walk up the AST from a for-statement to find the enclosing function declaration,
* then search its parameters for one named `iterableName`.
* Returns the element type extracted from its type annotation, or undefined.
*
* Go parameter_declaration has:
* name field: identifier (the parameter name)
* type field: the type node (slice_type for []User)
*/
const findGoParamElementType = (iterableName: string, startNode: SyntaxNode, pos: TypeArgPosition = 'last'): string | undefined => {
let current: SyntaxNode | null = startNode.parent;
while (current) {
if (GO_FUNCTION_NODE_TYPES.has(current.type)) {
const paramsNode = current.childForFieldName('parameters');
if (paramsNode) {
for (let i = 0; i < paramsNode.namedChildCount; i++) {
const paramDecl = paramsNode.namedChild(i);
if (!paramDecl || paramDecl.type !== 'parameter_declaration') continue;
// parameter_declaration: name type — name field is the identifier
const nameNode = paramDecl.childForFieldName('name');
if (nameNode?.text === iterableName) {
const typeNode = paramDecl.childForFieldName('type');
if (typeNode) return extractGoElementTypeFromTypeNode(typeNode, pos);
}
}
}
break;
}
current = current.parent;
}
return undefined;
};
/**
* Go: for _, user := range users where users has a known slice type.
*
* Go uses a single `for_statement` node for all for-loop forms. We detect
* range-based loops by looking for a `range_clause` child node. C-style for
* loops (with `for_clause`) and infinite loops (no clause) are ignored.
*
* Tier 1c: resolves the element type via three strategies in priority order:
* 1. declarationTypeNodes — raw type annotation AST node
* 2. scopeEnv string — extractElementTypeFromString on the stored type
* 3. AST walk — walks up to the enclosing function's parameters to read []User directly
* For `_, user := range users`, the loop variable is the second identifier in
* the `left` expression_list (index is discarded, value is the element).
*/
const extractForLoopBinding: ForLoopExtractor = (node, { scopeEnv, declarationTypeNodes, scope, returnTypeLookup }): void => {
if (node.type !== 'for_statement') return;
// Find the range_clause child — this distinguishes range loops from other for forms.
let rangeClause: SyntaxNode | null = null;
for (let i = 0; i < node.namedChildCount; i++) {
const child = node.namedChild(i);
if (child?.type === 'range_clause') {
rangeClause = child;
break;
}
}
if (!rangeClause) return;
// The iterable is the `right` field of the range_clause.
const rightNode = rangeClause.childForFieldName('right');
let iterableName: string | undefined;
let callExprElementType: string | undefined;
if (rightNode?.type === 'identifier') {
iterableName = rightNode.text;
} else if (rightNode?.type === 'selector_expression') {
const field = rightNode.childForFieldName('field');
if (field) iterableName = field.text;
} else if (rightNode?.type === 'call_expression') {
// Range over a call result: `for _, v := range getItems()` or `for _, v := range repo.All()`
const funcNode = rightNode.childForFieldName('function');
let callee: string | undefined;
if (funcNode?.type === 'identifier') {
callee = funcNode.text;
} else if (funcNode?.type === 'selector_expression') {
const field = funcNode.childForFieldName('field');
if (field) callee = field.text;
}
if (callee) {
const rawReturn = returnTypeLookup.lookupRawReturnType(callee);
if (rawReturn) callExprElementType = extractElementTypeFromString(rawReturn);
}
}
if (!iterableName && !callExprElementType) return;
let elementType: string | undefined;
if (callExprElementType) {
elementType = callExprElementType;
} else {
const containerTypeName = scopeEnv.get(iterableName!);
const typeArgPos = methodToTypeArgPosition(undefined, containerTypeName);
elementType = resolveIterableElementType(
iterableName!, node, scopeEnv, declarationTypeNodes, scope,
extractGoElementTypeFromTypeNode, findGoParamElementType,
typeArgPos,
);
}
if (!elementType) return;
// The loop variable(s) are in the `left` field.
// Go range semantics:
// Slice/Array/String: single-var → INDEX (int); two-var → (index, element)
// Map: single-var → KEY; two-var → (key, value)
// Channel: single-var → ELEMENT (channels have no index)
const leftNode = rangeClause.childForFieldName('left');
if (!leftNode) return;
let loopVarNode: SyntaxNode | null = null;
if (leftNode.type === 'expression_list') {
if (leftNode.namedChildCount >= 2) {
// Two-var form: `_, user` or `i, user` — second variable gets element/value type
loopVarNode = leftNode.namedChild(1);
} else {
// Single-var in expression_list — yields INDEX for slices/maps, ELEMENT for channels.
// For call-expression iterables (iterableName undefined), conservative: treat as non-channel.
// Channels are rarely returned from function calls, and even if they were, skipping here
// just means we miss a binding rather than create an incorrect one.
if (iterableName && isChannelType(iterableName, scopeEnv, declarationTypeNodes, scope)) {
loopVarNode = leftNode.namedChild(0);
} else {
return; // index-only range on slice/map — skip
}
}
} else {
// Plain identifier (single-var form without expression_list)
// For call-expression iterables (iterableName undefined), conservative: treat as non-channel.
// Channels are rarely returned from function calls, and even if they were, skipping here
// just means we miss a binding rather than create an incorrect one.
if (iterableName && isChannelType(iterableName, scopeEnv, declarationTypeNodes, scope)) {
loopVarNode = leftNode;
} else {
return; // index-only range on slice/map — skip
}
}
if (!loopVarNode) return;
// Skip the blank identifier `_`
if (loopVarNode.text === '_') return;
const loopVarName = extractVarName(loopVarNode);
if (loopVarName) scopeEnv.set(loopVarName, elementType);
};
/** Go: alias := u (short_var_declaration) or var b = u (var_spec) */
const extractPendingAssignment: PendingAssignmentExtractor = (node, scopeEnv) => {
if (node.type === 'short_var_declaration') {
const left = node.childForFieldName('left');
const right = node.childForFieldName('right');
if (!left || !right) return undefined;
const lhsNode = left.type === 'expression_list' ? left.firstNamedChild : left;
const rhsNode = right.type === 'expression_list' ? right.firstNamedChild : right;
if (!lhsNode || !rhsNode) return undefined;
if (lhsNode.type !== 'identifier') return undefined;
const lhs = lhsNode.text;
if (scopeEnv.has(lhs)) return undefined;
if (rhsNode.type === 'identifier') return { kind: 'copy', lhs, rhs: rhsNode.text };
// selector_expression RHS → fieldAccess (a.field)
if (rhsNode.type === 'selector_expression') {
const operand = rhsNode.childForFieldName('operand');
const field = rhsNode.childForFieldName('field');
if (operand?.type === 'identifier' && field) {
return { kind: 'fieldAccess', lhs, receiver: operand.text, field: field.text };
}
}
// call_expression RHS
if (rhsNode.type === 'call_expression') {
const funcNode = rhsNode.childForFieldName('function');
if (funcNode?.type === 'identifier') {
return { kind: 'callResult', lhs, callee: funcNode.text };
}
// method call with receiver: call_expression → function: selector_expression
if (funcNode?.type === 'selector_expression') {
const operand = funcNode.childForFieldName('operand');
const field = funcNode.childForFieldName('field');
if (operand?.type === 'identifier' && field) {
return { kind: 'methodCallResult', lhs, receiver: operand.text, method: field.text };
}
}
}
return undefined;
}
if (node.type === 'var_spec' || node.type === 'var_declaration') {
// var_declaration contains var_spec children; var_spec has name + expression_list value
const specs: SyntaxNode[] = [];
if (node.type === 'var_declaration') {
for (let i = 0; i < node.namedChildCount; i++) {
const c = node.namedChild(i);
if (c?.type === 'var_spec') specs.push(c);
}
} else {
specs.push(node);
}
for (const spec of specs) {
const nameNode = spec.childForFieldName('name');
if (!nameNode || nameNode.type !== 'identifier') continue;
const lhs = nameNode.text;
if (scopeEnv.has(lhs)) continue;
// Check if the last named child is a bare identifier (no type annotation between name and value)
let exprList: SyntaxNode | null = null;
for (let i = 0; i < spec.childCount; i++) {
if (spec.child(i)?.type === 'expression_list') { exprList = spec.child(i); break; }
}
const rhsNode = exprList?.firstNamedChild;
if (rhsNode?.type === 'identifier') return { kind: 'copy', lhs, rhs: rhsNode.text };
// selector_expression RHS → fieldAccess
if (rhsNode?.type === 'selector_expression') {
const operand = rhsNode.childForFieldName('operand');
const field = rhsNode.childForFieldName('field');
if (operand?.type === 'identifier' && field) {
return { kind: 'fieldAccess', lhs, receiver: operand.text, field: field.text };
}
}
// call_expression RHS
if (rhsNode?.type === 'call_expression') {
const funcNode = rhsNode.childForFieldName('function');
if (funcNode?.type === 'identifier') {
return { kind: 'callResult', lhs, callee: funcNode.text };
}
if (funcNode?.type === 'selector_expression') {
const operand = funcNode.childForFieldName('operand');
const field = funcNode.childForFieldName('field');
if (operand?.type === 'identifier' && field) {
return { kind: 'methodCallResult', lhs, receiver: operand.text, method: field.text };
}
}
}
}
}
return undefined;
};
export const typeConfig: LanguageTypeConfig = {
declarationNodeTypes: DECLARATION_NODE_TYPES,
forLoopNodeTypes: FOR_LOOP_NODE_TYPES,
extractDeclaration,
extractParameter,
scanConstructorBinding,
extractForLoopBinding,
extractPendingAssignment,
};

View file

@ -33,12 +33,19 @@ export const typeConfigs = {
[SupportedLanguages.Ruby]: rubyConfig,
} satisfies Record<SupportedLanguages, LanguageTypeConfig>;
export type { LanguageTypeConfig, TypeBindingExtractor, ParameterExtractor, ConstructorBindingScanner } from './types.js';
export type {
LanguageTypeConfig,
TypeBindingExtractor,
ParameterExtractor,
ConstructorBindingScanner,
ForLoopExtractor,
PendingAssignmentExtractor,
PatternBindingExtractor,
} from './types.js';
export {
TYPED_PARAMETER_TYPES,
extractSimpleTypeName,
extractGenericTypeArgs,
extractVarName,
findChildByType,
extractRubyConstructorAssignment
} from './shared.js';

View file

@ -1,6 +1,7 @@
import type { SyntaxNode } from '../utils.js';
import type { LanguageTypeConfig, ParameterExtractor, TypeBindingExtractor, InitializerExtractor, ClassNameLookup, ConstructorBindingScanner } from './types.js';
import { extractSimpleTypeName, extractVarName, findChildByType } from './shared.js';
import type { LanguageTypeConfig, ParameterExtractor, TypeBindingExtractor, InitializerExtractor, ClassNameLookup, ConstructorBindingScanner, ForLoopExtractor, PendingAssignmentExtractor, PatternBindingExtractor, LiteralTypeInferrer, ConstructorTypeDetector } from './types.js';
import { extractSimpleTypeName, extractVarName, extractGenericTypeArgs, resolveIterableElementType, methodToTypeArgPosition, extractElementTypeFromString, type TypeArgPosition } from './shared.js';
import { findChild } from '../resolvers/utils.js';
// ── Java ──────────────────────────────────────────────────────────────────
@ -73,7 +74,7 @@ const scanJavaConstructorBinding: ConstructorBindingScanner = (node) => {
const typeNode = node.childForFieldName('type');
if (!typeNode) return undefined;
if (typeNode.text !== 'var') return undefined;
const declarator = node.namedChildren.find((c: SyntaxNode) => c.type === 'variable_declarator');
const declarator = findChild(node, 'variable_declarator');
if (!declarator) return undefined;
const nameNode = declarator.childForFieldName('name');
const value = declarator.childForFieldName('value');
@ -85,12 +86,233 @@ const scanJavaConstructorBinding: ConstructorBindingScanner = (node) => {
return { varName: nameNode.text, calleeName: methodName.text };
};
const JAVA_FOR_LOOP_NODE_TYPES: ReadonlySet<string> = new Set([
'enhanced_for_statement',
]);
/** Extract element type from a Java type annotation AST node.
* Handles generic_type (List<User>), array_type (User[]). */
const extractJavaElementTypeFromTypeNode = (typeNode: SyntaxNode, pos: TypeArgPosition = 'last'): string | undefined => {
if (typeNode.type === 'generic_type') {
const args = extractGenericTypeArgs(typeNode);
if (args.length >= 1) return pos === 'first' ? args[0] : args[args.length - 1];
}
if (typeNode.type === 'array_type') {
const elemNode = typeNode.firstNamedChild;
if (elemNode) return extractSimpleTypeName(elemNode);
}
return undefined;
};
/** Walk up from a for-each to the enclosing method_declaration and search parameters. */
const findJavaParamElementType = (iterableName: string, startNode: SyntaxNode, pos: TypeArgPosition = 'last'): string | undefined => {
let current: SyntaxNode | null = startNode.parent;
while (current) {
if (current.type === 'method_declaration' || current.type === 'constructor_declaration') {
const paramsNode = current.childForFieldName('parameters');
if (paramsNode) {
for (let i = 0; i < paramsNode.namedChildCount; i++) {
const param = paramsNode.namedChild(i);
if (!param || param.type !== 'formal_parameter') continue;
const nameNode = param.childForFieldName('name');
if (nameNode?.text !== iterableName) continue;
const typeNode = param.childForFieldName('type');
if (typeNode) return extractJavaElementTypeFromTypeNode(typeNode, pos);
}
}
break;
}
current = current.parent;
}
return undefined;
};
/** Java: for (User user : users) — extract loop variable binding.
* Tier 1c: for `for (var user : users)`, resolves element type from iterable. */
const extractJavaForLoopBinding: ForLoopExtractor = (node, { scopeEnv, declarationTypeNodes, scope, returnTypeLookup }): void => {
const typeNode = node.childForFieldName('type');
const nameNode = node.childForFieldName('name');
if (!typeNode || !nameNode) return;
const varName = extractVarName(nameNode);
if (!varName) return;
// Explicit type (existing behavior): for (User user : users)
const typeName = extractSimpleTypeName(typeNode);
if (typeName && typeName !== 'var') {
scopeEnv.set(varName, typeName);
return;
}
// Tier 1c: var — resolve from iterable's container type
const iterableNode = node.childForFieldName('value');
if (!iterableNode) return;
let iterableName: string | undefined;
let methodName: string | undefined;
let callExprElementType: string | undefined;
if (iterableNode.type === 'identifier') {
iterableName = iterableNode.text;
} else if (iterableNode.type === 'field_access') {
const field = iterableNode.childForFieldName('field');
if (field) iterableName = field.text;
} else if (iterableNode.type === 'method_invocation') {
// data.keySet() → method_invocation > object: identifier + name: identifier
// Also handles this.data.values() → object is field_access, extract inner field name
const obj = iterableNode.childForFieldName('object');
const name = iterableNode.childForFieldName('name');
if (obj?.type === 'identifier') {
iterableName = obj.text;
} else if (obj?.type === 'field_access') {
const innerField = obj.childForFieldName('field');
if (innerField) iterableName = innerField.text;
} else if (!obj && name) {
// Direct function call: for (var u : getUsers()) — no receiver object
const rawReturn = returnTypeLookup.lookupRawReturnType(name.text);
if (rawReturn) callExprElementType = extractElementTypeFromString(rawReturn);
}
if (name) methodName = name.text;
}
if (!iterableName && !callExprElementType) return;
let elementType: string | undefined;
if (callExprElementType) {
elementType = callExprElementType;
} else {
const containerTypeName = scopeEnv.get(iterableName!);
const typeArgPos = methodToTypeArgPosition(methodName, containerTypeName);
elementType = resolveIterableElementType(
iterableName!, node, scopeEnv, declarationTypeNodes, scope,
extractJavaElementTypeFromTypeNode, findJavaParamElementType,
typeArgPos,
);
}
if (elementType) scopeEnv.set(varName, elementType);
};
/** Java: var alias = u → local_variable_declaration > variable_declarator with name/value */
const extractJavaPendingAssignment: PendingAssignmentExtractor = (node, scopeEnv) => {
for (let i = 0; i < node.namedChildCount; i++) {
const child = node.namedChild(i);
if (!child || child.type !== 'variable_declarator') continue;
const nameNode = child.childForFieldName('name');
const valueNode = child.childForFieldName('value');
if (!nameNode || !valueNode) continue;
const lhs = nameNode.text;
if (scopeEnv.has(lhs)) continue;
if (valueNode.type === 'identifier' || valueNode.type === 'simple_identifier') return { kind: 'copy', lhs, rhs: valueNode.text };
// field_access RHS → fieldAccess (a.field)
if (valueNode.type === 'field_access') {
const obj = valueNode.childForFieldName('object');
const field = valueNode.childForFieldName('field');
if (obj?.type === 'identifier' && field) {
return { kind: 'fieldAccess', lhs, receiver: obj.text, field: field.text };
}
}
// method_invocation RHS
if (valueNode.type === 'method_invocation') {
const objField = valueNode.childForFieldName('object');
if (!objField) {
// No receiver → callResult
const nameField = valueNode.childForFieldName('name');
if (nameField?.type === 'identifier') {
return { kind: 'callResult', lhs, callee: nameField.text };
}
} else if (objField.type === 'identifier') {
// With receiver → methodCallResult
const nameField = valueNode.childForFieldName('name');
if (nameField?.type === 'identifier') {
return { kind: 'methodCallResult', lhs, receiver: objField.text, method: nameField.text };
}
}
}
}
return undefined;
};
/**
* Java 16+ `instanceof` pattern variable: `x instanceof User user`
*
* AST structure:
* instanceof_expression
* left: expression (the variable being tested)
* instanceof keyword
* right: type (the type to test against)
* name: identifier (the pattern variable — optional, Java 16+)
*
* Conservative: returns undefined when the `name` field is absent (plain instanceof
* without pattern variable, e.g. `x instanceof User`) or when the type cannot be
* extracted. The source variable's existing type is NOT used — the pattern explicitly
* declares the new type, so no scopeEnv lookup is needed.
*/
const extractJavaPatternBinding: PatternBindingExtractor = (node) => {
if (node.type === 'type_pattern') {
// Java 17+ switch pattern: case User u -> ...
// type_pattern has positional children (NO named fields):
// namedChild(0) = type (type_identifier, e.g., User)
// namedChild(1) = identifier (e.g., u)
const typeNode = node.namedChild(0);
const nameNode = node.namedChild(1);
if (!typeNode || !nameNode) return undefined;
const typeName = extractSimpleTypeName(typeNode);
const varName = extractVarName(nameNode);
if (!typeName || !varName) return undefined;
return { varName, typeName };
}
if (node.type !== 'instanceof_expression') return undefined;
const nameNode = node.childForFieldName('name');
if (!nameNode) return undefined;
const typeNode = node.childForFieldName('right');
if (!typeNode) return undefined;
const typeName = extractSimpleTypeName(typeNode);
const varName = extractVarName(nameNode);
if (!typeName || !varName) return undefined;
return { varName, typeName };
};
/** Infer the type of a literal AST node for Java/Kotlin overload disambiguation. */
const inferJvmLiteralType: LiteralTypeInferrer = (node) => {
switch (node.type) {
case 'decimal_integer_literal':
case 'integer_literal':
case 'hex_integer_literal':
case 'octal_integer_literal':
case 'binary_integer_literal':
// Check for long suffix
if (node.text.endsWith('L') || node.text.endsWith('l')) return 'long';
return 'int';
case 'decimal_floating_point_literal':
case 'real_literal':
if (node.text.endsWith('f') || node.text.endsWith('F')) return 'float';
return 'double';
case 'string_literal':
case 'line_string_literal':
case 'multi_line_string_literal':
return 'String';
case 'character_literal':
return 'char';
case 'true':
case 'false':
case 'boolean_literal':
return 'boolean';
case 'null_literal':
return 'null';
default:
return undefined;
}
};
export const javaTypeConfig: LanguageTypeConfig = {
declarationNodeTypes: JAVA_DECLARATION_NODE_TYPES,
forLoopNodeTypes: JAVA_FOR_LOOP_NODE_TYPES,
patternBindingNodeTypes: new Set(['instanceof_expression', 'type_pattern']),
extractDeclaration: extractJavaDeclaration,
extractParameter: extractJavaParameter,
extractInitializer: extractJavaInitializer,
scanConstructorBinding: scanJavaConstructorBinding,
extractForLoopBinding: extractJavaForLoopBinding,
extractPendingAssignment: extractJavaPendingAssignment,
extractPatternBinding: extractJavaPatternBinding,
inferLiteralType: inferJvmLiteralType,
};
// ── Kotlin ────────────────────────────────────────────────────────────────
@ -104,10 +326,11 @@ const KOTLIN_DECLARATION_NODE_TYPES: ReadonlySet<string> = new Set([
const extractKotlinDeclaration: TypeBindingExtractor = (node: SyntaxNode, env: Map<string, string>): void => {
if (node.type === 'property_declaration') {
// Kotlin property_declaration: name/type are inside a variable_declaration child
const varDecl = findChildByType(node, 'variable_declaration');
const varDecl = findChild(node, 'variable_declaration');
if (varDecl) {
const nameNode = findChildByType(varDecl, 'simple_identifier');
const typeNode = findChildByType(varDecl, 'user_type');
const nameNode = findChild(varDecl, 'simple_identifier');
const typeNode = findChild(varDecl, 'user_type')
?? findChild(varDecl, 'nullable_type');
if (!nameNode || !typeNode) return;
const varName = extractVarName(nameNode);
const typeName = extractSimpleTypeName(typeNode);
@ -116,17 +339,17 @@ const extractKotlinDeclaration: TypeBindingExtractor = (node: SyntaxNode, env: M
}
// Fallback: try direct fields
const nameNode = node.childForFieldName('name')
?? findChildByType(node, 'simple_identifier');
?? findChild(node, 'simple_identifier');
const typeNode = node.childForFieldName('type')
?? findChildByType(node, 'user_type');
?? findChild(node, 'user_type');
if (!nameNode || !typeNode) return;
const varName = extractVarName(nameNode);
const typeName = extractSimpleTypeName(typeNode);
if (varName && typeName) env.set(varName, typeName);
} else if (node.type === 'variable_declaration') {
// variable_declaration directly inside functions
const nameNode = findChildByType(node, 'simple_identifier');
const typeNode = findChildByType(node, 'user_type');
const nameNode = findChild(node, 'simple_identifier');
const typeNode = findChild(node, 'user_type');
if (nameNode && typeNode) {
const varName = extractVarName(nameNode);
const typeName = extractSimpleTypeName(typeNode);
@ -135,7 +358,10 @@ const extractKotlinDeclaration: TypeBindingExtractor = (node: SyntaxNode, env: M
}
};
/** Kotlin: formal_parameter → type name */
/** Kotlin: parameter / formal_parameter → type name.
* Kotlin's tree-sitter grammar uses positional children (simple_identifier, user_type)
* rather than named fields (name, type) on `parameter` nodes, so we fall back to
* findChild when childForFieldName returns null. */
const extractKotlinParameter: ParameterExtractor = (node: SyntaxNode, env: Map<string, string>): void => {
let nameNode: SyntaxNode | null = null;
let typeNode: SyntaxNode | null = null;
@ -148,50 +374,67 @@ const extractKotlinParameter: ParameterExtractor = (node: SyntaxNode, env: Map<s
typeNode = node.childForFieldName('type');
}
// Fallback: Kotlin `parameter` nodes use positional children, not named fields
if (!nameNode) nameNode = findChild(node, 'simple_identifier');
if (!typeNode) typeNode = findChild(node, 'user_type')
?? findChild(node, 'nullable_type');
if (!nameNode || !typeNode) return;
const varName = extractVarName(nameNode);
const typeName = extractSimpleTypeName(typeNode);
if (varName && typeName) env.set(varName, typeName);
};
/** Find the constructor callee name in a Kotlin property_declaration's initializer.
* Returns the class name if the callee is a verified class constructor, undefined otherwise. */
const findKotlinConstructorCallee = (node: SyntaxNode, classNames: ClassNameLookup): string | undefined => {
if (node.type !== 'property_declaration') return undefined;
const value = node.childForFieldName('value')
?? findChild(node, 'call_expression');
if (!value || value.type !== 'call_expression') return undefined;
const callee = value.firstNamedChild;
if (!callee || callee.type !== 'simple_identifier') return undefined;
const calleeName = callee.text;
if (!calleeName || !classNames.has(calleeName)) return undefined;
return calleeName;
};
/** Kotlin: val user = User() — infer type from call_expression when callee is a known class.
* Kotlin constructors are syntactically identical to function calls, so we verify
* against classNames (which may include cross-file SymbolTable lookups). */
const extractKotlinInitializer: InitializerExtractor = (node: SyntaxNode, env: Map<string, string>, classNames: ClassNameLookup): void => {
if (node.type !== 'property_declaration') return;
// Skip if there's an explicit type annotation — Tier 0 already handled it
const varDecl = findChildByType(node, 'variable_declaration');
if (varDecl && findChildByType(varDecl, 'user_type')) return;
const varDecl = findChild(node, 'variable_declaration');
if (varDecl && findChild(varDecl, 'user_type')) return;
// Get the initializer value — the call_expression after '='
const value = node.childForFieldName('value')
?? findChildByType(node, 'call_expression');
if (!value || value.type !== 'call_expression') return;
// The callee is the first child of call_expression (simple_identifier for direct calls)
const callee = value.firstNamedChild;
if (!callee || callee.type !== 'simple_identifier') return;
const calleeName = callee.text;
if (!calleeName || !classNames.has(calleeName)) return;
const calleeName = findKotlinConstructorCallee(node, classNames);
if (!calleeName) return;
// Extract the variable name from the variable_declaration inside property_declaration
const nameNode = varDecl
? findChildByType(varDecl, 'simple_identifier')
: findChildByType(node, 'simple_identifier');
? findChild(varDecl, 'simple_identifier')
: findChild(node, 'simple_identifier');
if (!nameNode) return;
const varName = extractVarName(nameNode);
if (varName) env.set(varName, calleeName);
};
/** Kotlin: detect constructor type from call_expression in typed declarations.
* Unlike extractKotlinInitializer (which SKIPS typed declarations), this detects
* the constructor type EVEN when a type annotation exists, enabling virtual dispatch
* for patterns like `val a: Animal = Dog()`. */
const detectKotlinConstructorType: ConstructorTypeDetector = (node, classNames) => {
return findKotlinConstructorCallee(node, classNames);
};
/** Kotlin: val x = User(...) — constructor binding for property_declaration with call_expression */
const scanKotlinConstructorBinding: ConstructorBindingScanner = (node) => {
if (node.type !== 'property_declaration') return undefined;
const varDecl = node.namedChildren.find(c => c.type === 'variable_declaration');
const varDecl = findChild(node, 'variable_declaration');
if (!varDecl) return undefined;
if (varDecl.namedChildren.some(c => c.type === 'user_type')) return undefined;
const callExpr = node.namedChildren.find(c => c.type === 'call_expression');
if (findChild(varDecl, 'user_type')) return undefined;
const callExpr = findChild(node, 'call_expression');
if (!callExpr) return undefined;
const callee = callExpr.firstNamedChild;
if (!callee) return undefined;
@ -210,15 +453,338 @@ const scanKotlinConstructorBinding: ConstructorBindingScanner = (node) => {
}
}
if (!calleeName) return undefined;
const nameNode = varDecl.namedChildren.find(c => c.type === 'simple_identifier');
const nameNode = findChild(varDecl, 'simple_identifier');
if (!nameNode) return undefined;
return { varName: nameNode.text, calleeName };
};
const KOTLIN_FOR_LOOP_NODE_TYPES: ReadonlySet<string> = new Set([
'for_statement',
]);
/** Extract element type from a Kotlin type annotation AST node (user_type wrapping generic).
* Kotlin: user_type → [type_identifier, type_arguments → [type_projection → user_type]]
* Handles the type_projection wrapper that Kotlin uses for generic type arguments. */
const extractKotlinElementTypeFromTypeNode = (typeNode: SyntaxNode, pos: TypeArgPosition = 'last'): string | undefined => {
if (typeNode.type === 'user_type') {
const argsNode = findChild(typeNode, 'type_arguments');
if (argsNode && argsNode.namedChildCount >= 1) {
const targetArg = pos === 'first'
? argsNode.namedChild(0)
: argsNode.namedChild(argsNode.namedChildCount - 1);
if (!targetArg) return undefined;
// Kotlin wraps type args in type_projection — unwrap to get the inner type
const inner = targetArg.type === 'type_projection'
? targetArg.firstNamedChild
: targetArg;
if (inner) return extractSimpleTypeName(inner);
}
}
return undefined;
};
/** Walk up from a for-loop to the enclosing function_declaration and search parameters.
* Kotlin parameters use positional children (simple_identifier, user_type), not named fields. */
const findKotlinParamElementType = (iterableName: string, startNode: SyntaxNode, pos: TypeArgPosition = 'last'): string | undefined => {
let current: SyntaxNode | null = startNode.parent;
while (current) {
if (current.type === 'function_declaration') {
const paramsNode = findChild(current, 'function_value_parameters');
if (paramsNode) {
for (let i = 0; i < paramsNode.namedChildCount; i++) {
const param = paramsNode.namedChild(i);
if (!param || param.type !== 'parameter') continue;
const nameNode = findChild(param, 'simple_identifier');
if (nameNode?.text !== iterableName) continue;
const typeNode = findChild(param, 'user_type');
if (typeNode) return extractKotlinElementTypeFromTypeNode(typeNode, pos);
}
}
break;
}
current = current.parent;
}
return undefined;
};
/** Kotlin: for (user: User in users) — extract loop variable binding.
* Tier 1c: for `for (user in users)` without annotation, resolves from iterable. */
const extractKotlinForLoopBinding: ForLoopExtractor = (node, ctx): void => {
const { scopeEnv, declarationTypeNodes, scope, returnTypeLookup } = ctx;
const varDecl = findChild(node, 'variable_declaration');
if (!varDecl) return;
const nameNode = findChild(varDecl, 'simple_identifier');
if (!nameNode) return;
const varName = extractVarName(nameNode);
if (!varName) return;
// Explicit type annotation (existing behavior): for (user: User in users)
const typeNode = findChild(varDecl, 'user_type');
if (typeNode) {
const typeName = extractSimpleTypeName(typeNode);
if (typeName) scopeEnv.set(varName, typeName);
return;
}
// Tier 1c: no annotation — resolve from iterable's container type
// Kotlin for-loop children: [variable_declaration, iterable_expr, control_structure_body]
// The iterable is the second named child of the for_statement (after variable_declaration)
let iterableName: string | undefined;
let methodName: string | undefined;
let fallbackIterableName: string | undefined;
let callExprElementType: string | undefined;
let foundVarDecl = false;
for (let i = 0; i < node.namedChildCount; i++) {
const child = node.namedChild(i);
if (child === varDecl) { foundVarDecl = true; continue; }
if (!foundVarDecl || !child) continue;
if (child.type === 'simple_identifier') {
iterableName = child.text;
break;
}
if (child.type === 'navigation_expression') {
// data.keys → navigation_expression > simple_identifier(data) + navigation_suffix > simple_identifier(keys)
const obj = child.firstNamedChild;
const suffix = findChild(child, 'navigation_suffix');
const prop = suffix ? findChild(suffix, 'simple_identifier') : null;
const hasCallSuffix = suffix ? findChild(suffix, 'call_suffix') !== null : false;
// Always try object as iterable + property as method first (handles data.values, data.keys).
// For bare property access without call_suffix, also save property as fallback
// (handles this.users, repo.items where the property IS the iterable).
if (obj?.type === 'simple_identifier') iterableName = obj.text;
if (prop) methodName = prop.text;
if (!hasCallSuffix && prop) {
fallbackIterableName = prop.text;
}
break;
}
if (child.type === 'call_expression') {
// data.values() → call_expression > navigation_expression > simple_identifier + navigation_suffix
const callee = child.firstNamedChild;
if (callee?.type === 'navigation_expression') {
const obj = callee.firstNamedChild;
if (obj?.type === 'simple_identifier') iterableName = obj.text;
const suffix = findChild(callee, 'navigation_suffix');
if (suffix) {
const prop = findChild(suffix, 'simple_identifier');
if (prop) methodName = prop.text;
}
} else if (callee?.type === 'simple_identifier') {
// Direct function call: for (u in getUsers())
const rawReturn = returnTypeLookup.lookupRawReturnType(callee.text);
if (rawReturn) callExprElementType = extractElementTypeFromString(rawReturn);
}
break;
}
}
if (!iterableName && !callExprElementType) return;
let elementType: string | undefined;
if (callExprElementType) {
elementType = callExprElementType;
} else {
let containerTypeName = scopeEnv.get(iterableName!);
// Fallback: if object has no type in scope, try the property as the iterable name.
// Handles patterns like this.users where the property itself is the iterable variable.
if (!containerTypeName && fallbackIterableName) {
iterableName = fallbackIterableName;
methodName = undefined;
containerTypeName = scopeEnv.get(iterableName);
}
const typeArgPos = methodToTypeArgPosition(methodName, containerTypeName);
elementType = resolveIterableElementType(
iterableName!, node, scopeEnv, declarationTypeNodes, scope,
extractKotlinElementTypeFromTypeNode, findKotlinParamElementType,
typeArgPos,
);
}
if (elementType) scopeEnv.set(varName, elementType);
};
/** Kotlin: val alias = u → property_declaration or variable_declaration.
* property_declaration has: binding_pattern_kind("val"), variable_declaration("alias"),
* "=", and the RHS value (simple_identifier "u").
* variable_declaration appears directly inside functions and has simple_identifier children. */
const extractKotlinPendingAssignment: PendingAssignmentExtractor = (node, scopeEnv) => {
if (node.type === 'property_declaration') {
// Find the variable name from variable_declaration child
const varDecl = findChild(node, 'variable_declaration');
if (!varDecl) return undefined;
const nameNode = varDecl.firstNamedChild;
if (!nameNode || nameNode.type !== 'simple_identifier') return undefined;
const lhs = nameNode.text;
if (scopeEnv.has(lhs)) return undefined;
// Find the RHS after the "=" token
let foundEq = false;
for (let i = 0; i < node.childCount; i++) {
const child = node.child(i);
if (!child) continue;
if (child.type === '=') { foundEq = true; continue; }
if (foundEq && child.type === 'simple_identifier') {
return { kind: 'copy', lhs, rhs: child.text };
}
// navigation_expression RHS → fieldAccess (a.field)
if (foundEq && child.type === 'navigation_expression') {
const recv = child.firstNamedChild;
const suffix = child.lastNamedChild;
const fieldNode = suffix?.type === 'navigation_suffix' ? suffix.lastNamedChild : suffix;
if (recv?.type === 'simple_identifier' && fieldNode?.type === 'simple_identifier') {
return { kind: 'fieldAccess', lhs, receiver: recv.text, field: fieldNode.text };
}
}
// call_expression RHS
if (foundEq && child.type === 'call_expression') {
const calleeNode = child.firstNamedChild;
if (calleeNode?.type === 'simple_identifier') {
return { kind: 'callResult', lhs, callee: calleeNode.text };
}
// navigation_expression callee → methodCallResult (a.method())
if (calleeNode?.type === 'navigation_expression') {
const recv = calleeNode.firstNamedChild;
const suffix = calleeNode.lastNamedChild;
const methodNode = suffix?.type === 'navigation_suffix' ? suffix.lastNamedChild : suffix;
if (recv?.type === 'simple_identifier' && methodNode?.type === 'simple_identifier') {
return { kind: 'methodCallResult', lhs, receiver: recv.text, method: methodNode.text };
}
}
}
}
return undefined;
}
if (node.type === 'variable_declaration') {
// variable_declaration directly inside functions: simple_identifier children
const nameNode = findChild(node, 'simple_identifier');
if (!nameNode) return undefined;
const lhs = nameNode.text;
if (scopeEnv.has(lhs)) return undefined;
// Look for RHS after "=" in the parent (property_declaration)
const parent = node.parent;
if (!parent) return undefined;
let foundEq = false;
for (let i = 0; i < parent.childCount; i++) {
const child = parent.child(i);
if (!child) continue;
if (child.type === '=') { foundEq = true; continue; }
if (foundEq && child.type === 'simple_identifier') {
return { kind: 'copy', lhs, rhs: child.text };
}
if (foundEq && child.type === 'navigation_expression') {
const recv = child.firstNamedChild;
const suffix = child.lastNamedChild;
const fieldNode = suffix?.type === 'navigation_suffix' ? suffix.lastNamedChild : suffix;
if (recv?.type === 'simple_identifier' && fieldNode?.type === 'simple_identifier') {
return { kind: 'fieldAccess', lhs, receiver: recv.text, field: fieldNode.text };
}
}
if (foundEq && child.type === 'call_expression') {
const calleeNode = child.firstNamedChild;
if (calleeNode?.type === 'simple_identifier') {
return { kind: 'callResult', lhs, callee: calleeNode.text };
}
if (calleeNode?.type === 'navigation_expression') {
const recv = calleeNode.firstNamedChild;
const suffix = calleeNode.lastNamedChild;
const methodNode = suffix?.type === 'navigation_suffix' ? suffix.lastNamedChild : suffix;
if (recv?.type === 'simple_identifier' && methodNode?.type === 'simple_identifier') {
return { kind: 'methodCallResult', lhs, receiver: recv.text, method: methodNode.text };
}
}
}
}
return undefined;
}
return undefined;
};
/** Walk up from a node to find an ancestor of a given type. */
const findAncestorByType = (node: SyntaxNode, type: string): SyntaxNode | undefined => {
let current = node.parent;
while (current) {
if (current.type === type) return current;
current = current.parent;
}
return undefined;
};
const extractKotlinPatternBinding: PatternBindingExtractor = (node, scopeEnv, declarationTypeNodes, scope) => {
// Kotlin when/is smart casts (existing behavior)
if (node.type === 'type_test') {
const typeNode = node.lastNamedChild;
if (!typeNode) return undefined;
const typeName = extractSimpleTypeName(typeNode);
if (!typeName) return undefined;
const whenExpr = findAncestorByType(node, 'when_expression');
if (!whenExpr) return undefined;
const whenSubject = whenExpr.namedChild(0);
const subject = whenSubject?.firstNamedChild ?? whenSubject;
if (!subject) return undefined;
const varName = extractVarName(subject);
if (!varName) return undefined;
return { varName, typeName };
}
// Null-check narrowing: if (x != null) { ... }
// Kotlin AST: equality_expression > simple_identifier, "!=" [anon], "null" [anon]
// Note: `null` is an anonymous node in tree-sitter-kotlin, not `null_literal`.
if (node.type === 'equality_expression') {
const op = node.children.find(c => !c.isNamed && c.text === '!=');
if (!op) return undefined;
// `null` is anonymous in Kotlin grammar — use positional child scan
let varNode: SyntaxNode | undefined;
let hasNull = false;
for (let i = 0; i < node.childCount; i++) {
const c = node.child(i);
if (!c) continue;
if (c.type === 'simple_identifier') varNode = c;
if (!c.isNamed && c.text === 'null') hasNull = true;
}
if (!varNode || !hasNull) return undefined;
const varName = varNode.text;
const resolvedType = scopeEnv.get(varName);
if (!resolvedType) return undefined;
// Check if the original declaration type was nullable (ends with ?)
const declTypeNode = declarationTypeNodes.get(`${scope}\0${varName}`);
if (!declTypeNode) return undefined;
const declText = declTypeNode.text;
if (!declText.includes('?') && !declText.includes('null')) return undefined;
// Find the if-body: walk up to if_expression, then find control_structure_body
const ifExpr = findAncestorByType(node, 'if_expression');
if (!ifExpr) return undefined;
// The consequence is the first control_structure_body child
for (let i = 0; i < ifExpr.childCount; i++) {
const child = ifExpr.child(i);
if (child?.type === 'control_structure_body') {
return {
varName,
typeName: resolvedType,
narrowingRange: { startIndex: child.startIndex, endIndex: child.endIndex },
};
}
}
return undefined;
}
return undefined;
};
export const kotlinTypeConfig: LanguageTypeConfig = {
allowPatternBindingOverwrite: true,
declarationNodeTypes: KOTLIN_DECLARATION_NODE_TYPES,
forLoopNodeTypes: KOTLIN_FOR_LOOP_NODE_TYPES,
patternBindingNodeTypes: new Set(['type_test', 'equality_expression']),
extractDeclaration: extractKotlinDeclaration,
extractParameter: extractKotlinParameter,
extractInitializer: extractKotlinInitializer,
scanConstructorBinding: scanKotlinConstructorBinding,
extractForLoopBinding: extractKotlinForLoopBinding,
extractPendingAssignment: extractKotlinPendingAssignment,
extractPatternBinding: extractKotlinPatternBinding,
inferLiteralType: inferJvmLiteralType,
detectConstructorType: detectKotlinConstructorType,
};

View file

@ -1,6 +1,6 @@
import type { SyntaxNode } from '../utils.js';
import type { LanguageTypeConfig, ParameterExtractor, TypeBindingExtractor, InitializerExtractor, ClassNameLookup, ConstructorBindingScanner, ReturnTypeExtractor } from './types.js';
import { extractSimpleTypeName, extractVarName, extractCalleeName } from './shared.js';
import type { LanguageTypeConfig, ParameterExtractor, TypeBindingExtractor, InitializerExtractor, ClassNameLookup, ConstructorBindingScanner, ReturnTypeExtractor, PendingAssignmentExtractor, ForLoopExtractor } from './types.js';
import { extractSimpleTypeName, extractVarName, extractCalleeName, resolveIterableElementType, extractElementTypeFromString } from './shared.js';
const DECLARATION_NODE_TYPES: ReadonlySet<string> = new Set([
'assignment_expression', // For constructor inference: $x = new User()
@ -61,6 +61,15 @@ const normalizePhpType = (raw: string): string | undefined => {
type = segments[segments.length - 1];
// Skip uninformative types
if (type === 'mixed' || type === 'void' || type === 'self' || type === 'static' || type === 'object') return undefined;
// Extract element type from generic: Collection<User> → User
// PHPDoc generics encode the element type in angle brackets. Since PHP's Strategy B
// uses the scopeEnv value directly as the element type, we must store the inner type,
// not the container name. This mirrors how User[] → User is handled by the [] strip above.
const genericMatch = type.match(/^(\w+)\s*</);
if (genericMatch) {
const elementType = extractElementTypeFromString(type);
return elementType ?? undefined;
}
if (/^\w+$/.test(type)) return type;
return undefined;
};
@ -73,6 +82,67 @@ const SKIP_NODE_TYPES: ReadonlySet<string> = new Set(['attribute_list', 'attribu
const PHPDOC_PARAM_RE = /@param\s+(\S+)\s+\$(\w+)/g;
/** Alternate PHPDoc order: `@param $name Type` (name first) */
const PHPDOC_PARAM_ALT_RE = /@param\s+\$(\w+)\s+(\S+)/g;
/** Regex to extract PHPDoc @var annotations: `@var Type` */
const PHPDOC_VAR_RE = /@var\s+(\S+)/;
/**
* Extract the element type for a class property from its PHPDoc @var annotation or
* PHP 7.4+ native type. Walks backward from the property_declaration node to find
* an immediately preceding comment containing @var.
*
* Returns the normalized element type (e.g. User[] → User, Collection<User> → User).
* Returns undefined when no usable type annotation is found.
*/
const extractClassPropertyElementType = (propDecl: SyntaxNode): string | undefined => {
// Strategy 1: PHPDoc @var annotation on a preceding comment sibling
let sibling = propDecl.previousSibling;
while (sibling) {
if (sibling.type === 'comment') {
const match = PHPDOC_VAR_RE.exec(sibling.text);
if (match) return normalizePhpType(match[1]);
} else if (sibling.isNamed && !SKIP_NODE_TYPES.has(sibling.type)) {
break;
}
sibling = sibling.previousSibling;
}
// Strategy 2: PHP 7.4+ native type field — skip generic 'array' since element type is unknown
const typeNode = propDecl.childForFieldName('type');
if (!typeNode) return undefined;
const typeName = extractSimpleTypeName(typeNode);
if (!typeName || typeName === 'array') return undefined;
return typeName;
};
/**
* Scan a class body for a property_declaration matching the given property name,
* and extract its element type. The class body is the `declaration_list` child of
* a `class_declaration` node.
*
* Used as Strategy C in extractForLoopBinding for `$this->property` iterables
* where Strategy A (resolveIterableElementType) and Strategy B (scopeEnv lookup)
* both fail to find the type.
*/
const findClassPropertyElementType = (propName: string, classNode: SyntaxNode): string | undefined => {
const declList = classNode.childForFieldName('body')
?? (classNode.namedChild(classNode.namedChildCount - 1)?.type === 'declaration_list'
? classNode.namedChild(classNode.namedChildCount - 1)
: null); // fallback: last named child, only if it's a declaration_list
if (!declList) return undefined;
for (let i = 0; i < declList.namedChildCount; i++) {
const child = declList.namedChild(i);
if (child?.type !== 'property_declaration') continue;
// Check if any property_element has a variable_name matching '$propName'
for (let j = 0; j < child.namedChildCount; j++) {
const elem = child.namedChild(j);
if (elem?.type !== 'property_element') continue;
const varNameNode = elem.firstNamedChild; // variable_name node
if (varNameNode?.text === '$' + propName) {
return extractClassPropertyElementType(child);
}
}
}
return undefined;
};
/**
* Collect PHPDoc @param type bindings from comment nodes preceding a method/function.
@ -190,8 +260,12 @@ const extractParameter: ParameterExtractor = (node: SyntaxNode, env: Map<string,
if (!nameNode || !typeNode) return;
const varName = extractVarName(nameNode);
if (!varName) return;
// Don't overwrite PHPDoc-derived types (e.g. @param User[] $users → User)
// with the less-specific AST type annotation (e.g. array).
if (env.has(varName)) return;
const typeName = extractSimpleTypeName(typeNode);
if (varName && typeName) env.set(varName, typeName);
if (typeName) env.set(varName, typeName);
};
/** PHP: $x = SomeFactory() or $x = $this->getUser() — bind variable to call return type */
@ -229,27 +303,251 @@ const scanConstructorBinding: ConstructorBindingScanner = (node) => {
/** Regex to extract PHPDoc @return annotations: `@return User` */
const PHPDOC_RETURN_RE = /@return\s+(\S+)/;
/**
* Normalize a PHPDoc return type for storage in the SymbolTable.
* Unlike normalizePhpType (which strips User[] → User for scopeEnv), this preserves
* array notation so lookupRawReturnType can extract element types for for-loop resolution.
* \App\Models\User[] → User[]
* ?User → User
* Collection<User> → Collection<User> (preserved for extractElementTypeFromString)
*/
const normalizePhpReturnType = (raw: string): string | undefined => {
// Strip nullable prefix: ?User[] → User[]
let type = raw.startsWith('?') ? raw.slice(1) : raw;
// Strip union with null/false/void: User[]|null → User[]
const parts = type.split('|').filter(p => p !== 'null' && p !== 'false' && p !== 'void' && p !== 'mixed');
if (parts.length !== 1) return undefined;
type = parts[0];
// Strip namespace: \App\Models\User[] → User[]
const segments = type.split('\\');
type = segments[segments.length - 1];
// Skip uninformative types
if (type === 'mixed' || type === 'void' || type === 'self' || type === 'static' || type === 'object' || type === 'array') return undefined;
if (/^\w+(\[\])?$/.test(type) || /^\w+\s*</.test(type)) return type;
return undefined;
};
/**
* Extract return type from PHPDoc `@return Type` annotation preceding a method.
* Walks backwards through preceding siblings looking for comment nodes.
* Preserves array notation (e.g., User[]) for for-loop element type extraction.
*/
const extractReturnType: ReturnTypeExtractor = (node) => {
let sibling = node.previousSibling;
while (sibling) {
if (sibling.type === 'comment') {
const match = PHPDOC_RETURN_RE.exec(sibling.text);
if (match) return normalizePhpType(match[1]);
if (match) return normalizePhpReturnType(match[1]);
} else if (sibling.isNamed && !SKIP_NODE_TYPES.has(sibling.type)) break;
sibling = sibling.previousSibling;
}
return undefined;
};
/** PHP: $alias = $user → assignment_expression with variable_name left/right.
* PHP TypeEnv stores variables WITH $ prefix ($user → User), so we keep $ in lhs/rhs. */
const extractPendingAssignment: PendingAssignmentExtractor = (node, scopeEnv) => {
if (node.type !== 'assignment_expression') return undefined;
const left = node.childForFieldName('left');
const right = node.childForFieldName('right');
if (!left || !right) return undefined;
if (left.type !== 'variable_name') return undefined;
const lhs = left.text;
if (!lhs || scopeEnv.has(lhs)) return undefined;
if (right.type === 'variable_name') {
const rhs = right.text;
if (rhs) return { kind: 'copy', lhs, rhs };
}
// member_access_expression RHS → fieldAccess ($a->field)
if (right.type === 'member_access_expression') {
const obj = right.childForFieldName('object');
const name = right.childForFieldName('name');
if (obj?.type === 'variable_name' && name) {
return { kind: 'fieldAccess', lhs, receiver: obj.text, field: name.text };
}
}
// function_call_expression RHS → callResult (bare function calls only)
if (right.type === 'function_call_expression') {
const funcNode = right.childForFieldName('function');
if (funcNode?.type === 'name') {
return { kind: 'callResult', lhs, callee: funcNode.text };
}
}
// member_call_expression RHS → methodCallResult ($a->method())
if (right.type === 'member_call_expression') {
const obj = right.childForFieldName('object');
const name = right.childForFieldName('name');
if (obj?.type === 'variable_name' && name) {
return { kind: 'methodCallResult', lhs, receiver: obj.text, method: name.text };
}
}
return undefined;
};
const FOR_LOOP_NODE_TYPES: ReadonlySet<string> = new Set([
'foreach_statement',
]);
/** Extract element type from a PHP type annotation AST node.
* PHP has limited AST-level container types — `array` is a primitive_type with no generic args.
* Named types (e.g., `Collection`) are returned as-is (container descriptor lookup handles them). */
const extractPhpElementTypeFromTypeNode = (_typeNode: SyntaxNode): string | undefined => {
// PHP AST type nodes don't carry generic parameters (array<User> is PHPDoc-only).
// primitive_type 'array' and named_type 'Collection' don't encode element types.
return undefined;
};
/** Walk up from a foreach to the enclosing function and search parameter type annotations.
* PHP parameter type hints are limited (array, ClassName) — this extracts element type when possible. */
const findPhpParamElementType = (iterableName: string, startNode: SyntaxNode): string | undefined => {
let current: SyntaxNode | null = startNode.parent;
while (current) {
if (current.type === 'method_declaration' || current.type === 'function_definition') {
const paramsNode = current.childForFieldName('parameters');
if (paramsNode) {
for (let i = 0; i < paramsNode.namedChildCount; i++) {
const param = paramsNode.namedChild(i);
if (!param || param.type !== 'simple_parameter') continue;
const nameNode = param.childForFieldName('name');
if (nameNode?.text !== iterableName) continue;
const typeNode = param.childForFieldName('type');
if (typeNode) return extractPhpElementTypeFromTypeNode(typeNode);
}
}
break;
}
current = current.parent;
}
return undefined;
};
/**
* PHP: foreach ($users as $user) — extract loop variable binding.
*
* AST structure (from tree-sitter-php grammar):
* foreach_statement — no named fields for iterable/value (only 'body')
* children[0]: expression (iterable, e.g. $users)
* children[1]: expression (simple value) OR pair ($key => $value)
* pair children: expression (key), expression (value)
*
* PHP's PHPDoc @param normalizes `User[]` → `User` in the env, so the iterable's
* stored type IS the element type. We first try resolveIterableElementType (for
* constructor-binding cases that retain container types), then fall back to direct
* scopeEnv lookup (for PHPDoc-normalized types).
*/
const extractForLoopBinding: ForLoopExtractor = (node, { scopeEnv, declarationTypeNodes, scope, returnTypeLookup }): void => {
if (node.type !== 'foreach_statement') return;
// Collect non-body named children: first is the iterable, second is value or pair
const children: SyntaxNode[] = [];
for (let i = 0; i < node.namedChildCount; i++) {
const child = node.namedChild(i);
if (child && child !== node.childForFieldName('body')) {
children.push(child);
}
}
if (children.length < 2) return;
const iterableNode = children[0];
const valueOrPair = children[1];
// Determine the loop variable node
let loopVarNode: SyntaxNode;
if (valueOrPair.type === 'pair') {
// $key => $value — the value is the last named child of the pair
const lastChild = valueOrPair.namedChild(valueOrPair.namedChildCount - 1);
if (!lastChild) return;
// Handle by_ref: foreach ($arr as $k => &$v)
loopVarNode = lastChild.type === 'by_ref' ? (lastChild.firstNamedChild ?? lastChild) : lastChild;
} else {
// Simple: foreach ($users as $user) or foreach ($users as &$user)
loopVarNode = valueOrPair.type === 'by_ref' ? (valueOrPair.firstNamedChild ?? valueOrPair) : valueOrPair;
}
const varName = extractVarName(loopVarNode);
if (!varName) return;
// Get iterable variable name (PHP vars include $ prefix)
let iterableName: string | undefined;
let callExprElementType: string | undefined;
if (iterableNode.type === 'variable_name') {
iterableName = iterableNode.text;
} else if (iterableNode?.type === 'member_access_expression') {
const name = iterableNode.childForFieldName('name');
// PHP properties are stored in scopeEnv with $ prefix ($users), but
// member_access_expression.name returns without $ (users). Add $ to match.
if (name) iterableName = '$' + name.text;
} else if (iterableNode?.type === 'function_call_expression') {
// foreach (getUsers() as $user) — resolve via return type lookup
const calleeName = extractCalleeName(iterableNode);
if (calleeName) {
const rawReturn = returnTypeLookup.lookupRawReturnType(calleeName);
if (rawReturn) callExprElementType = extractElementTypeFromString(rawReturn);
}
} else if (iterableNode?.type === 'member_call_expression') {
// foreach ($this->getUsers() as $user) — resolve via return type lookup
const methodName = iterableNode.childForFieldName('name');
if (methodName) {
const rawReturn = returnTypeLookup.lookupRawReturnType(methodName.text);
if (rawReturn) callExprElementType = extractElementTypeFromString(rawReturn);
}
}
if (!iterableName && !callExprElementType) return;
// If we resolved the element type from a call expression, bind and return early
if (callExprElementType) {
scopeEnv.set(varName, callExprElementType);
return;
}
// Strategy A: try resolveIterableElementType (handles constructor-binding container types)
const elementType = resolveIterableElementType(
iterableName, node, scopeEnv, declarationTypeNodes, scope,
extractPhpElementTypeFromTypeNode, findPhpParamElementType,
undefined,
);
if (elementType) {
scopeEnv.set(varName, elementType);
return;
}
// Strategy B: direct scopeEnv lookup — PHP normalizePhpType strips User[] → User,
// so the iterable's stored type is already the element type from PHPDoc annotations.
const iterableType = scopeEnv.get(iterableName);
if (iterableType) {
scopeEnv.set(varName, iterableType);
return;
}
// Strategy C: $this->property — scan the enclosing class body for the property
// declaration and extract its element type from @var PHPDoc or native type.
// This handles the common PHP pattern where the property type is declared on the
// class body (/** @var User[] */ private $users) but the foreach is in a method
// whose scopeEnv does not contain the property type.
if (iterableNode?.type === 'member_access_expression') {
const obj = iterableNode.childForFieldName('object');
if (obj?.text === '$this') {
const nameNode = iterableNode.childForFieldName('name');
const propName = nameNode?.text;
if (propName) {
const classNode = findEnclosingClass(iterableNode);
if (classNode) {
const elementType = findClassPropertyElementType(propName, classNode);
if (elementType) scopeEnv.set(varName, elementType);
}
}
}
}
};
export const typeConfig: LanguageTypeConfig = {
declarationNodeTypes: DECLARATION_NODE_TYPES,
forLoopNodeTypes: FOR_LOOP_NODE_TYPES,
extractDeclaration,
extractParameter,
extractInitializer,
scanConstructorBinding,
extractReturnType,
extractForLoopBinding,
extractPendingAssignment,
};

View file

@ -1,21 +1,53 @@
import type { SyntaxNode } from '../utils.js';
import type { LanguageTypeConfig, ParameterExtractor, TypeBindingExtractor, InitializerExtractor, ClassNameLookup, ConstructorBindingScanner } from './types.js';
import { extractSimpleTypeName, extractVarName } from './shared.js';
import type { LanguageTypeConfig, ParameterExtractor, TypeBindingExtractor, InitializerExtractor, ClassNameLookup, ConstructorBindingScanner, PendingAssignmentExtractor, PatternBindingExtractor, ForLoopExtractor } from './types.js';
import { extractSimpleTypeName, extractVarName, extractElementTypeFromString, extractGenericTypeArgs, resolveIterableElementType, methodToTypeArgPosition, type TypeArgPosition } from './shared.js';
const DECLARATION_NODE_TYPES: ReadonlySet<string> = new Set([
'assignment',
'named_expression',
'expression_statement',
]);
/** Python: x: Foo = ... (PEP 484 annotations) */
/** Python: x: Foo = ... (PEP 484 annotated assignment) or x: Foo (standalone annotation).
*
* tree-sitter-python grammar produces two distinct shapes:
*
* 1. Annotated assignment with value: `name: str = ""`
* Node type: `assignment`
* Fields: left=identifier, type=identifier/type, right=value
*
* 2. Standalone annotation (no value): `name: str`
* Node type: `expression_statement`
* Child: `type` node with fields name=identifier, type=identifier/type
*
* Both appear at file scope and inside class bodies (PEP 526 class variable annotations).
*/
const extractDeclaration: TypeBindingExtractor = (node: SyntaxNode, env: Map<string, string>): void => {
// Python annotated assignment: left : type = value
// tree-sitter represents this differently based on grammar version
if (node.type === 'expression_statement') {
// Standalone annotation: expression_statement > type { name: identifier, type: identifier }
const typeChild = node.firstNamedChild;
if (!typeChild || typeChild.type !== 'type') return;
const nameNode = typeChild.childForFieldName('name');
const typeNode = typeChild.childForFieldName('type');
if (!nameNode || !typeNode) return;
const varName = extractVarName(nameNode);
const inner = typeNode.type === 'type' ? (typeNode.firstNamedChild ?? typeNode) : typeNode;
const typeName = extractSimpleTypeName(inner) ?? inner.text;
if (varName && typeName) env.set(varName, typeName);
return;
}
// Annotated assignment: left : type = value
const left = node.childForFieldName('left');
const typeNode = node.childForFieldName('type');
if (!left || !typeNode) return;
const varName = extractVarName(left);
const typeName = extractSimpleTypeName(typeNode);
// extractSimpleTypeName handles identifiers and qualified names.
// Python 3.10+ union syntax `User | None` is parsed as binary_operator,
// which extractSimpleTypeName doesn't handle. Fall back to raw text so
// stripNullable can process it at lookup time (e.g., "User | None" → "User").
const inner = typeNode.type === 'type' ? (typeNode.firstNamedChild ?? typeNode) : typeNode;
const typeName = extractSimpleTypeName(inner) ?? inner.text;
if (varName && typeName) env.set(varName, typeName);
};
@ -30,6 +62,10 @@ const extractParameter: ParameterExtractor = (node: SyntaxNode, env: Map<string,
} else {
nameNode = node.childForFieldName('name') ?? node.childForFieldName('pattern');
typeNode = node.childForFieldName('type');
// Python typed_parameter: name is a positional child (identifier), not a named field
if (!nameNode && node.type === 'typed_parameter') {
nameNode = node.firstNamedChild?.type === 'identifier' ? node.firstNamedChild : null;
}
}
if (!nameNode || !typeNode) return;
@ -102,10 +138,323 @@ const scanConstructorBinding: ConstructorBindingScanner = (node) => {
return { varName: left.text, calleeName };
};
const FOR_LOOP_NODE_TYPES: ReadonlySet<string> = new Set([
'for_statement',
]);
/** Python function/method node types that carry a parameters list. */
const PY_FUNCTION_NODE_TYPES = new Set([
'function_definition', 'decorated_definition',
]);
/**
* Extract element type from a Python type annotation AST node.
* Handles:
* subscript "List[User]" → extractElementTypeFromString("List[User]") → "User"
* generic_type → extractGenericTypeArgs → first arg
* Falls back to text-based extraction.
*/
const extractPyElementTypeFromAnnotation = (typeNode: SyntaxNode, pos: TypeArgPosition = 'last'): string | undefined => {
// Unwrap 'type' wrapper node to get to the actual type (e.g., type > generic_type)
const inner = typeNode.type === 'type' ? (typeNode.firstNamedChild ?? typeNode) : typeNode;
// Python subscript: List[User], Sequence[User] — use raw text
if (inner.type === 'subscript') {
return extractElementTypeFromString(inner.text, pos);
}
// generic_type: dict[str, User] — tree-sitter-python uses type_parameter child
if (inner.type === 'generic_type') {
// Try standard extractGenericTypeArgs first (handles type_arguments)
const args = extractGenericTypeArgs(inner);
if (args.length >= 1) return pos === 'first' ? args[0] : args[args.length - 1];
// Fallback: look for type_parameter child (tree-sitter-python specific)
for (let i = 0; i < inner.namedChildCount; i++) {
const child = inner.namedChild(i);
if (child?.type === 'type_parameter') {
if (pos === 'first') {
const firstArg = child.firstNamedChild;
if (firstArg) return extractSimpleTypeName(firstArg);
} else {
const lastArg = child.lastNamedChild;
if (lastArg) return extractSimpleTypeName(lastArg);
}
}
}
}
// Fallback: raw text extraction (handles User[], [User], etc.)
return extractElementTypeFromString(inner.text, pos);
};
/**
* Walk up the AST from a for-statement to find the enclosing function definition,
* then search its parameters for one named `iterableName`.
* Returns the element type extracted from its type annotation, or undefined.
*
* Handles both `parameter` and `typed_parameter` node types in tree-sitter-python.
* `typed_parameter` may not expose the name as a `name` field — falls back to
* checking the first identifier-type named child.
*/
const findPyParamElementType = (iterableName: string, startNode: SyntaxNode, pos: TypeArgPosition = 'last'): string | undefined => {
let current: SyntaxNode | null = startNode.parent;
while (current) {
if (current.type === 'function_definition') {
const paramsNode = current.childForFieldName('parameters');
if (paramsNode) {
for (let i = 0; i < paramsNode.namedChildCount; i++) {
const param = paramsNode.namedChild(i);
if (!param) continue;
// Try named `name` field first (parameter node), then first identifier child
// (typed_parameter node may store name as first positional child)
const nameNode = param.childForFieldName('name')
?? (param.firstNamedChild?.type === 'identifier' ? param.firstNamedChild : null);
if (nameNode?.text !== iterableName) continue;
// Try `type` field, then last named child (typed_parameter stores type last)
const typeAnnotation = param.childForFieldName('type')
?? (param.namedChildCount >= 2 ? param.namedChild(param.namedChildCount - 1) : null);
if (typeAnnotation && typeAnnotation !== nameNode) {
return extractPyElementTypeFromAnnotation(typeAnnotation, pos);
}
}
}
break;
}
current = current.parent;
}
return undefined;
};
/**
* Extracts iterableName and methodName from a call expression like `data.items()`.
* Returns undefined if the call doesn't match the expected pattern.
*/
const extractMethodCall = (callNode: SyntaxNode): { iterableName: string; methodName?: string } | undefined => {
const fn = callNode.childForFieldName('function');
if (fn?.type !== 'attribute') return undefined;
const obj = fn.firstNamedChild;
if (obj?.type !== 'identifier') return undefined;
const method = fn.lastNamedChild;
const methodName = (method?.type === 'identifier' && method !== obj) ? method.text : undefined;
return { iterableName: obj.text, methodName };
};
/**
* Collects all identifier nodes from a pattern, descending into nested tuple_patterns.
* For `i, (k, v)` returns [i, k, v]. For `key, value` returns [key, value].
*/
const collectPatternIdentifiers = (pattern: SyntaxNode): SyntaxNode[] => {
const vars: SyntaxNode[] = [];
for (let i = 0; i < pattern.namedChildCount; i++) {
const child = pattern.namedChild(i);
if (child?.type === 'identifier') {
vars.push(child);
} else if (child?.type === 'tuple_pattern') {
vars.push(...collectPatternIdentifiers(child));
}
}
return vars;
};
/**
* Python: for user in users: where users has a known container type annotation.
*
* AST node: `for_statement` with `left` (loop variable) and `right` (iterable).
*
* Tier 1c: resolves the element type via three strategies in priority order:
* 1. declarationTypeNodes — raw type annotation AST node (covers stored container types)
* 2. scopeEnv string — extractElementTypeFromString on the stored type
* 3. AST walk — walks up to the enclosing function's parameters to read List[User] directly
*
* Also handles `enumerate(iterable)` — unwraps the outer call and skips the integer
* index variable so the value variable still resolves to the element type.
*/
const extractForLoopBinding: ForLoopExtractor = (node, { scopeEnv, declarationTypeNodes, scope, returnTypeLookup }): void => {
if (node.type !== 'for_statement') return;
const rightNode = node.childForFieldName('right');
let iterableName: string | undefined;
let methodName: string | undefined;
let callExprElementType: string | undefined;
let isEnumerate = false;
// Extract iterable info from the `right` field — may be identifier, attribute, or call.
if (rightNode?.type === 'identifier') {
iterableName = rightNode.text;
} else if (rightNode?.type === 'attribute') {
const prop = rightNode.lastNamedChild;
if (prop) iterableName = prop.text;
} else if (rightNode?.type === 'call') {
const fn = rightNode.childForFieldName('function');
if (fn?.type === 'identifier' && fn.text === 'enumerate') {
// enumerate(iterable) or enumerate(d.items()) — unwrap to inner iterable.
isEnumerate = true;
const innerArg = rightNode.childForFieldName('arguments')?.firstNamedChild;
if (innerArg?.type === 'identifier') {
iterableName = innerArg.text;
} else if (innerArg?.type === 'call') {
const extracted = extractMethodCall(innerArg);
if (extracted) ({ iterableName, methodName } = extracted);
}
} else if (fn?.type === 'attribute') {
// data.items() → call > function: attribute > identifier('data') + identifier('items')
const extracted = extractMethodCall(rightNode);
if (extracted) ({ iterableName, methodName } = extracted);
} else if (fn?.type === 'identifier') {
// Direct function call: for user in get_users() (Phase 7.3 — return-type path)
const rawReturn = returnTypeLookup.lookupRawReturnType(fn.text);
if (rawReturn) callExprElementType = extractElementTypeFromString(rawReturn);
}
}
if (!iterableName && !callExprElementType) return;
let elementType: string | undefined;
if (callExprElementType) {
elementType = callExprElementType;
} else {
const containerTypeName = scopeEnv.get(iterableName!);
const typeArgPos = methodToTypeArgPosition(methodName, containerTypeName);
elementType = resolveIterableElementType(
iterableName!, node, scopeEnv, declarationTypeNodes, scope,
extractPyElementTypeFromAnnotation, findPyParamElementType,
typeArgPos,
);
}
if (!elementType) return;
// The loop variable is the `left` field — identifier or pattern_list.
const leftNode = node.childForFieldName('left');
if (!leftNode) return;
if (leftNode.type === 'pattern_list' || leftNode.type === 'tuple_pattern') {
// Tuple unpacking: `key, value` or `i, (k, v)` or `(k, v)` — bind the last identifier to element type.
// With enumerate, skip binding if there's only one var (just the index, no value to bind).
const vars = collectPatternIdentifiers(leftNode);
if (vars.length > 0 && (!isEnumerate || vars.length > 1)) {
scopeEnv.set(vars[vars.length - 1].text, elementType);
}
return;
}
const loopVarName = extractVarName(leftNode);
if (loopVarName) scopeEnv.set(loopVarName, elementType);
};
/** Python: alias = u → assignment with left/right fields.
* Also handles walrus operator: alias := u → named_expression with name/value fields. */
const extractPendingAssignment: PendingAssignmentExtractor = (node, scopeEnv) => {
let left: SyntaxNode | null;
let right: SyntaxNode | null;
if (node.type === 'assignment') {
left = node.childForFieldName('left');
right = node.childForFieldName('right');
} else if (node.type === 'named_expression') {
left = node.childForFieldName('name');
right = node.childForFieldName('value');
} else {
return undefined;
}
if (!left || !right) return undefined;
const lhs = left.type === 'identifier' ? left.text : undefined;
if (!lhs || scopeEnv.has(lhs)) return undefined;
if (right.type === 'identifier') return { kind: 'copy', lhs, rhs: right.text };
// attribute RHS → fieldAccess (a.field)
if (right.type === 'attribute') {
const obj = right.firstNamedChild;
const field = right.lastNamedChild;
if (obj?.type === 'identifier' && field?.type === 'identifier' && obj !== field) {
return { kind: 'fieldAccess', lhs, receiver: obj.text, field: field.text };
}
}
// call RHS
if (right.type === 'call') {
const funcNode = right.childForFieldName('function');
if (funcNode?.type === 'identifier') {
return { kind: 'callResult', lhs, callee: funcNode.text };
}
// method call with receiver: call → function: attribute
if (funcNode?.type === 'attribute') {
const obj = funcNode.firstNamedChild;
const method = funcNode.lastNamedChild;
if (obj?.type === 'identifier' && method?.type === 'identifier' && obj !== method) {
return { kind: 'methodCallResult', lhs, receiver: obj.text, method: method.text };
}
}
}
return undefined;
};
/**
* Python match/case `as` pattern binding: `case User() as u:`
*
* AST structure (tree-sitter-python):
* as_pattern
* alias: as_pattern_target ← the bound variable name (e.g. "u")
* children[0]: case_pattern ← wraps class_pattern (or is class_pattern directly)
* class_pattern
* dotted_name ← the class name (e.g. "User")
*
* The `alias` field is an `as_pattern_target` node whose `.text` is the identifier.
* The class name lives in the first non-alias named child: either a `case_pattern`
* wrapping a `class_pattern`, or a direct `class_pattern`.
*
* Conservative: returns undefined when:
* - The node is not an `as_pattern`
* - The pattern side is not a class_pattern (e.g. guard or literal match)
* - The variable was already bound in scopeEnv
*/
const extractPatternBinding: PatternBindingExtractor = (node, scopeEnv) => {
if (node.type !== 'as_pattern') return undefined;
// as_pattern: `case User() as u:` — binds matched value to a name.
// Try named field first (future grammar versions may expose it), fall back to positional.
if (node.namedChildCount < 2) return undefined;
const patternChild = node.namedChild(0);
const varNameNode = node.childForFieldName('alias')
?? node.namedChild(node.namedChildCount - 1);
if (!patternChild || !varNameNode) return undefined;
if (varNameNode.type !== 'identifier') return undefined;
const varName = varNameNode.text;
if (!varName || scopeEnv.has(varName)) return undefined;
// Find the class_pattern — may be direct or wrapped in case_pattern.
let classPattern: SyntaxNode | null = null;
if (patternChild.type === 'class_pattern') {
classPattern = patternChild;
} else if (patternChild.type === 'case_pattern') {
// Unwrap one level: case_pattern wraps class_pattern
for (let j = 0; j < patternChild.namedChildCount; j++) {
const inner = patternChild.namedChild(j);
if (inner?.type === 'class_pattern') {
classPattern = inner;
break;
}
}
}
if (!classPattern) return undefined;
// class_pattern children: dotted_name (the class name) + optional keyword_pattern args.
const classNameNode = classPattern.firstNamedChild;
if (!classNameNode || (classNameNode.type !== 'dotted_name' && classNameNode.type !== 'identifier')) return undefined;
const typeName = classNameNode.text;
if (!typeName) return undefined;
return { varName, typeName };
};
const PATTERN_BINDING_NODE_TYPES: ReadonlySet<string> = new Set(['as_pattern']);
export const typeConfig: LanguageTypeConfig = {
declarationNodeTypes: DECLARATION_NODE_TYPES,
forLoopNodeTypes: FOR_LOOP_NODE_TYPES,
patternBindingNodeTypes: PATTERN_BINDING_NODE_TYPES,
extractDeclaration,
extractParameter,
extractInitializer,
scanConstructorBinding,
extractForLoopBinding,
extractPendingAssignment,
extractPatternBinding,
};

View file

@ -1,6 +1,6 @@
import type { LanguageTypeConfig, ParameterExtractor, TypeBindingExtractor, InitializerExtractor, ClassNameLookup, ConstructorBindingScanner, ReturnTypeExtractor } from './types.js';
import { extractRubyConstructorAssignment, extractSimpleTypeName } from './shared.js';
import { SyntaxNode } from '../utils.js';
import type { LanguageTypeConfig, ParameterExtractor, TypeBindingExtractor, InitializerExtractor, ClassNameLookup, ConstructorBindingScanner, ReturnTypeExtractor, PendingAssignmentExtractor, ForLoopExtractor } from './types.js';
import { extractRubyConstructorAssignment, extractSimpleTypeName, extractElementTypeFromString, extractVarName, resolveIterableElementType } from './shared.js';
import type { SyntaxNode } from '../utils.js';
/**
* Ruby type extractor — YARD annotation parsing.
@ -261,11 +261,160 @@ const scanConstructorBinding: ConstructorBindingScanner = (node) => {
return { varName: left.text, calleeName };
};
/** Ruby method node types that carry a parameter list. */
const RUBY_METHOD_NODE_TYPES = new Set(['method', 'singleton_method']);
const FOR_LOOP_NODE_TYPES: ReadonlySet<string> = new Set(['for']);
/**
* Collect raw YARD @param type strings from comment nodes preceding a method.
* Unlike collectYardParams which returns simplified type names, this returns the
* raw bracket content (e.g., "Array<User>" not "Array") for element type extraction.
*/
const collectYardRawParams = (methodNode: SyntaxNode): Map<string, string> => {
const params = new Map<string, string>();
const commentTexts: string[] = [];
const collectComments = (startNode: SyntaxNode): void => {
let sibling = startNode.previousSibling;
while (sibling) {
if (sibling.type === 'comment') {
commentTexts.unshift(sibling.text);
} else if (sibling.isNamed) {
break;
}
sibling = sibling.previousSibling;
}
};
collectComments(methodNode);
if (commentTexts.length === 0 && methodNode.parent?.type === 'body_statement') {
collectComments(methodNode.parent);
}
const commentBlock = commentTexts.join('\n');
let match: RegExpExecArray | null;
YARD_PARAM_RE.lastIndex = 0;
while ((match = YARD_PARAM_RE.exec(commentBlock)) !== null) {
params.set(match[1], match[2]);
}
YARD_PARAM_ALT_RE.lastIndex = 0;
while ((match = YARD_PARAM_ALT_RE.exec(commentBlock)) !== null) {
if (!params.has(match[2])) params.set(match[2], match[1]);
}
return params;
};
/**
* Walk up the AST from a for-statement to find the enclosing method,
* then search its YARD @param annotations for one named `iterableName`.
* Returns the element type extracted from the raw YARD type string.
*
* Example: `@param users [Array<User>]` → extracts "User" from "Array<User>".
*/
const findRubyParamElementType = (iterableName: string, startNode: SyntaxNode): string | undefined => {
let current: SyntaxNode | null = startNode.parent;
while (current) {
if (RUBY_METHOD_NODE_TYPES.has(current.type)) {
const rawParams = collectYardRawParams(current);
const rawType = rawParams.get(iterableName);
if (rawType) return extractElementTypeFromString(rawType);
break;
}
current = current.parent;
}
return undefined;
};
/**
* Ruby: for user in users ... end
*
* tree-sitter-ruby `for` node structure:
* pattern field: the loop variable (identifier)
* value field: `in` node whose child is the iterable expression
*
* Tier 1c: resolves the element type via:
* 1. scopeEnv string — extractElementTypeFromString on the stored type
* 2. AST walk — walks up to the enclosing method's YARD @param to read Array<User> directly
*
* Ruby has no static types on loop variables, so this mainly works when the
* iterable has a YARD-annotated container type (e.g., `@param users [Array<User>]`).
*/
const extractForLoopBinding: ForLoopExtractor = (node, { scopeEnv, declarationTypeNodes, scope }): void => {
if (node.type !== 'for') return;
// The loop variable is the `pattern` field (identifier).
const patternNode = node.childForFieldName('pattern');
if (!patternNode) return;
const loopVarName = extractVarName(patternNode);
if (!loopVarName) return;
// The iterable is inside the `value` field which is an `in` node wrapping the expression.
const inNode = node.childForFieldName('value');
if (!inNode) return;
const iterableNode = inNode.firstNamedChild;
let iterableName: string | undefined;
if (iterableNode?.type === 'identifier') {
iterableName = iterableNode.text;
} else if (iterableNode?.type === 'call') {
const method = iterableNode.childForFieldName('method');
if (method) iterableName = method.text;
}
if (!iterableName) return;
// Ruby has no extractFromTypeNode (no AST type annotations), pass a no-op.
const noopExtractFromTypeNode = (): string | undefined => undefined;
const elementType = resolveIterableElementType(
iterableName, node, scopeEnv, declarationTypeNodes, scope,
noopExtractFromTypeNode, findRubyParamElementType,
undefined,
);
if (!elementType) return;
scopeEnv.set(loopVarName, elementType);
};
/**
* Ruby: alias_user = user → assignment with left/right identifier fields.
* Only handles plain identifier RHS (not calls, not literals).
* Skips if LHS already has a resolved type in scopeEnv.
*/
const extractPendingAssignment: PendingAssignmentExtractor = (node, scopeEnv) => {
if (node.type !== 'assignment') return undefined;
const lhsNode = node.childForFieldName('left');
if (!lhsNode || lhsNode.type !== 'identifier') return undefined;
const varName = lhsNode.text;
if (scopeEnv.has(varName)) return undefined;
const rhsNode = node.childForFieldName('right');
if (!rhsNode) return undefined;
if (rhsNode.type === 'identifier') return { kind: 'copy', lhs: varName, rhs: rhsNode.text };
// call/method_call RHS — Ruby uses method calls for both field access and method calls
if (rhsNode.type === 'call' || rhsNode.type === 'method_call') {
const methodNode = rhsNode.childForFieldName('method');
const receiverNode = rhsNode.childForFieldName('receiver');
if (!receiverNode && methodNode?.type === 'identifier') {
// No receiver → callResult (bare function call)
return { kind: 'callResult', lhs: varName, callee: methodNode.text };
}
if (receiverNode?.type === 'identifier' && methodNode?.type === 'identifier') {
// With receiver → methodCallResult (a.method)
return { kind: 'methodCallResult', lhs: varName, receiver: receiverNode.text, method: methodNode.text };
}
}
return undefined;
};
export const typeConfig: LanguageTypeConfig = {
declarationNodeTypes: DECLARATION_NODE_TYPES,
forLoopNodeTypes: FOR_LOOP_NODE_TYPES,
extractDeclaration,
extractParameter,
extractInitializer,
scanConstructorBinding,
extractReturnType,
extractForLoopBinding,
extractPendingAssignment,
};

View file

@ -1,6 +1,6 @@
import type { SyntaxNode } from '../utils.js';
import type { LanguageTypeConfig, ParameterExtractor, TypeBindingExtractor, InitializerExtractor, ClassNameLookup, ConstructorBindingScanner } from './types.js';
import { extractSimpleTypeName, extractVarName, hasTypeAnnotation, unwrapAwait } from './shared.js';
import type { LanguageTypeConfig, ParameterExtractor, TypeBindingExtractor, InitializerExtractor, ClassNameLookup, ConstructorBindingScanner, PendingAssignmentExtractor, PendingAssignment, PatternBindingExtractor, ForLoopExtractor } from './types.js';
import { extractSimpleTypeName, extractVarName, hasTypeAnnotation, unwrapAwait, extractGenericTypeArgs, resolveIterableElementType, methodToTypeArgPosition, extractElementTypeFromString, type TypeArgPosition } from './shared.js';
const DECLARATION_NODE_TYPES: ReadonlySet<string> = new Set([
'let_declaration',
@ -35,7 +35,8 @@ const extractStructPatternType = (structPattern: SyntaxNode): string | undefined
* Recursively scan a pattern tree for captured_pattern nodes (x @ StructType { .. })
* and extract variable → type bindings from them.
*/
const extractCapturedPatternBindings = (pattern: SyntaxNode, env: Map<string, string>): void => {
const extractCapturedPatternBindings = (pattern: SyntaxNode, env: Map<string, string>, depth = 0): void => {
if (depth > 50) return;
if (pattern.type === 'captured_pattern') {
// captured_pattern: identifier @ inner_pattern
// The first named child is the identifier, followed by the inner pattern.
@ -57,7 +58,7 @@ const extractCapturedPatternBindings = (pattern: SyntaxNode, env: Map<string, st
if (pattern.type === 'tuple_struct_pattern') {
for (let i = 0; i < pattern.namedChildCount; i++) {
const child = pattern.namedChild(i);
if (child) extractCapturedPatternBindings(child, env);
if (child) extractCapturedPatternBindings(child, env, depth + 1);
}
}
};
@ -94,7 +95,7 @@ const extractDeclaration: TypeBindingExtractor = (node: SyntaxNode, env: Map<str
};
/** Rust: let x = User::new(), let x = User::default(), or let x = User { ... } */
const extractInitializer: InitializerExtractor = (node: SyntaxNode, env: Map<string, string>, _classNames: ClassNameLookup): void => {
const extractInitializer: InitializerExtractor = (node: SyntaxNode, env: Map<string, string>, classNames: ClassNameLookup): void => {
// Skip if there's an explicit type annotation — Tier 0 already handled it
if (node.childForFieldName('type') !== null) return;
const pattern = node.childForFieldName('pattern');
@ -115,6 +116,13 @@ const extractInitializer: InitializerExtractor = (node: SyntaxNode, env: Map<str
return;
}
// Unit struct instantiation: let svc = UserService; (bare identifier, no braces or call)
if (value.type === 'identifier' && classNames.has(value.text)) {
const varName = extractVarName(pattern);
if (varName) env.set(varName, value.text);
return;
}
if (value.type !== 'call_expression') return;
const func = value.childForFieldName('function');
if (!func || func.type !== 'scoped_identifier') return;
@ -181,10 +189,299 @@ const scanConstructorBinding: ConstructorBindingScanner = (node) => {
return { varName: patternNode.text, calleeName };
};
/** Rust: let alias = u; → let_declaration with pattern + value fields.
* Also handles struct destructuring: `let Point { x, y } = p` → N fieldAccess items. */
const extractPendingAssignment: PendingAssignmentExtractor = (node, scopeEnv) => {
if (node.type !== 'let_declaration') return undefined;
const pattern = node.childForFieldName('pattern');
const value = node.childForFieldName('value');
if (!pattern || !value) return undefined;
// Struct pattern destructuring: `let Point { x, y } = receiver`
// struct_pattern has a type child (struct name) and field_pattern children
if (pattern.type === 'struct_pattern' && value.type === 'identifier') {
const receiver = value.text;
const items: PendingAssignment[] = [];
for (let j = 0; j < pattern.namedChildCount; j++) {
const field = pattern.namedChild(j);
if (!field) continue;
if (field.type === 'field_pattern') {
// `Point { x: local_x }` → field_pattern with name + pattern children
const nameNode = field.childForFieldName('name');
const patNode = field.childForFieldName('pattern');
if (nameNode && patNode) {
const fieldName = nameNode.text;
const varName = extractVarName(patNode);
if (varName && !scopeEnv.has(varName)) {
items.push({ kind: 'fieldAccess', lhs: varName, receiver, field: fieldName });
}
} else if (nameNode) {
// Shorthand: `Point { x }` → field_pattern with only name (varName = fieldName)
const varName = nameNode.text;
if (!scopeEnv.has(varName)) {
items.push({ kind: 'fieldAccess', lhs: varName, receiver, field: varName });
}
}
}
}
if (items.length > 0) return items;
return undefined;
}
const lhs = extractVarName(pattern);
if (!lhs || scopeEnv.has(lhs)) return undefined;
// Unwrap Rust .await: `let user = get_user().await` → call_expression
const unwrapped = unwrapAwait(value) ?? value;
if (unwrapped.type === 'identifier') return { kind: 'copy', lhs, rhs: unwrapped.text };
// field_expression RHS → fieldAccess (a.field)
if (unwrapped.type === 'field_expression') {
const obj = unwrapped.firstNamedChild;
const field = unwrapped.lastNamedChild;
if (obj?.type === 'identifier' && field?.type === 'field_identifier') {
return { kind: 'fieldAccess', lhs, receiver: obj.text, field: field.text };
}
}
// call_expression RHS → callResult (simple calls only)
if (unwrapped.type === 'call_expression') {
const funcNode = unwrapped.childForFieldName('function');
if (funcNode?.type === 'identifier') {
return { kind: 'callResult', lhs, callee: funcNode.text };
}
}
// method_call_expression RHS → methodCallResult (receiver.method())
if (unwrapped.type === 'method_call_expression') {
const obj = unwrapped.firstNamedChild;
if (obj?.type === 'identifier') {
const methodNode = unwrapped.childForFieldName('name') ?? unwrapped.namedChild(1);
if (methodNode?.type === 'field_identifier') {
return { kind: 'methodCallResult', lhs, receiver: obj.text, method: methodNode.text };
}
}
}
return undefined;
};
/**
* Rust pattern binding extractor for `if let` / `while let` constructs that unwrap
* enum variants and introduce new typed variables.
*
* Supported patterns:
* - `if let Some(x) = opt` → x: T (opt: Option<T>, T already in scopeEnv via NULLABLE_WRAPPER_TYPES)
* - `if let Ok(x) = res` → x: T (res: Result<T, E>, T extracted from declarationTypeNodes)
*
* These complement the captured_pattern support in extractDeclaration (which handles
* `if let x @ Struct { .. } = expr` but NOT tuple struct unwrapping like Some(x) / Ok(x)).
*
* Conservative: returns undefined when:
* - The source variable's type is unknown (not in scopeEnv)
* - The wrapper is not a known single-unwrap variant (Some / Ok)
* - The value side is not a simple identifier
*/
const extractPatternBinding: PatternBindingExtractor = (
node,
scopeEnv,
declarationTypeNodes,
scope,
) => {
let patternNode: SyntaxNode | null = null;
let valueNode: SyntaxNode | null = null;
if (node.type === 'let_condition') {
patternNode = node.childForFieldName('pattern');
valueNode = node.childForFieldName('value');
} else if (node.type === 'match_arm') {
// match_arm → pattern field is match_pattern wrapping the actual pattern
const matchPatternNode = node.childForFieldName('pattern');
// Unwrap match_pattern to get the tuple_struct_pattern inside
patternNode = matchPatternNode?.type === 'match_pattern'
? matchPatternNode.firstNamedChild
: matchPatternNode;
// source variable is in the parent match_expression's 'value' field
const matchExpr = node.parent?.parent; // match_arm → match_block → match_expression
if (matchExpr?.type === 'match_expression') {
valueNode = matchExpr.childForFieldName('value');
}
}
if (!patternNode || !valueNode) return undefined;
// Only handle tuple_struct_pattern: Some(x) or Ok(x)
if (patternNode.type !== 'tuple_struct_pattern') return undefined;
// Extract the wrapper type name: Some | Ok
const wrapperTypeNode = patternNode.childForFieldName('type');
if (!wrapperTypeNode) return undefined;
const wrapperName = extractSimpleTypeName(wrapperTypeNode);
if (wrapperName !== 'Some' && wrapperName !== 'Ok' && wrapperName !== 'Err') return undefined;
// Extract the inner variable name from the single child of the tuple_struct_pattern.
// `Some(x)` → the first named child after the type field is the identifier.
// tree-sitter-rust: tuple_struct_pattern has 'type' field + unnamed children for args.
let innerVar: string | undefined;
for (let i = 0; i < patternNode.namedChildCount; i++) {
const child = patternNode.namedChild(i);
if (!child) continue;
// Skip the type node itself
if (child === wrapperTypeNode) continue;
if (child.type === 'identifier') {
innerVar = child.text;
break;
}
}
if (!innerVar) return undefined;
// The value must be a simple identifier so we can look it up in scopeEnv
const sourceVarName = valueNode.type === 'identifier' ? valueNode.text : undefined;
if (!sourceVarName) return undefined;
// For `Some(x)`: Option<T> is already unwrapped to T in scopeEnv (via NULLABLE_WRAPPER_TYPES).
// For `Ok(x)`: Result<T, E> stores "Result" in scopeEnv — must use declarationTypeNodes.
if (wrapperName === 'Some') {
const innerType = scopeEnv.get(sourceVarName);
if (!innerType) return undefined;
return { varName: innerVar, typeName: innerType };
}
// wrapperName === 'Ok' or 'Err': look up the Result<T, E> type AST node.
// Ok(x) → extract T (typeArgs[0]), Err(e) → extract E (typeArgs[1]).
const typeNodeKey = `${scope}\0${sourceVarName}`;
const typeAstNode = declarationTypeNodes.get(typeNodeKey);
if (!typeAstNode) return undefined;
const typeArgs = extractGenericTypeArgs(typeAstNode);
const argIndex = wrapperName === 'Err' ? 1 : 0;
if (typeArgs.length < argIndex + 1) return undefined;
return { varName: innerVar, typeName: typeArgs[argIndex] };
};
// --- For-loop Tier 1c ---
const FOR_LOOP_NODE_TYPES: ReadonlySet<string> = new Set(['for_expression']);
/** Extract element type from a Rust type annotation AST node.
* Handles: generic_type (Vec<User>), reference_type (&[User]), array_type ([User; N]),
* slice_type ([User]). For call-graph purposes, strips references (&User → User). */
const extractRustElementTypeFromTypeNode = (typeNode: SyntaxNode, pos: TypeArgPosition = 'last', depth = 0): string | undefined => {
if (depth > 50) return undefined;
// generic_type: Vec<User>, HashMap<K, V> — extract type arg based on position
if (typeNode.type === 'generic_type') {
const args = extractGenericTypeArgs(typeNode);
if (args.length >= 1) return pos === 'first' ? args[0] : args[args.length - 1];
}
// reference_type: &[User] or &Vec<User> — unwrap the reference and recurse
if (typeNode.type === 'reference_type') {
const inner = typeNode.lastNamedChild;
if (inner) return extractRustElementTypeFromTypeNode(inner, pos, depth + 1);
}
// array_type: [User; N] — element is the first child
if (typeNode.type === 'array_type') {
const elemNode = typeNode.firstNamedChild;
if (elemNode) return extractSimpleTypeName(elemNode);
}
// slice_type: [User] — element is the first child
if (typeNode.type === 'slice_type') {
const elemNode = typeNode.firstNamedChild;
if (elemNode) return extractSimpleTypeName(elemNode);
}
return undefined;
};
/** Walk up from a for-loop to the enclosing function_item and search parameters
* for one named `iterableName`. Returns the element type from its annotation. */
const findRustParamElementType = (iterableName: string, startNode: SyntaxNode, pos: TypeArgPosition = 'last'): string | undefined => {
let current: SyntaxNode | null = startNode.parent;
while (current) {
if (current.type === 'function_item') {
const paramsNode = current.childForFieldName('parameters');
if (paramsNode) {
for (let i = 0; i < paramsNode.namedChildCount; i++) {
const param = paramsNode.namedChild(i);
if (!param || param.type !== 'parameter') continue;
const nameNode = param.childForFieldName('pattern');
if (!nameNode) continue;
// Unwrap reference patterns: &users, &mut users
let identNode = nameNode;
if (identNode.type === 'reference_pattern') {
identNode = identNode.lastNamedChild ?? identNode;
}
if (identNode.type === 'mut_pattern') {
identNode = identNode.firstNamedChild ?? identNode;
}
if (identNode.text !== iterableName) continue;
const typeNode = param.childForFieldName('type');
if (typeNode) return extractRustElementTypeFromTypeNode(typeNode, pos);
}
}
break;
}
current = current.parent;
}
return undefined;
};
/** Rust: for user in &users where users has a known container type.
* Unwraps reference_expression (&users, &mut users) to get the iterable name. */
const extractForLoopBinding: ForLoopExtractor = (node, { scopeEnv, declarationTypeNodes, scope, returnTypeLookup }): void => {
if (node.type !== 'for_expression') return;
const patternNode = node.childForFieldName('pattern');
const valueNode = node.childForFieldName('value');
if (!patternNode || !valueNode) return;
// Extract iterable name + method — may be &users, users, or users.iter()/keys()/values()
let iterableName: string | undefined;
let methodName: string | undefined;
let callExprElementType: string | undefined;
if (valueNode.type === 'reference_expression') {
const inner = valueNode.lastNamedChild;
if (inner?.type === 'identifier') iterableName = inner.text;
} else if (valueNode.type === 'identifier') {
iterableName = valueNode.text;
} else if (valueNode.type === 'field_expression') {
const prop = valueNode.lastNamedChild;
if (prop) iterableName = prop.text;
} else if (valueNode.type === 'call_expression') {
const funcExpr = valueNode.childForFieldName('function');
if (funcExpr?.type === 'field_expression') {
// users.iter() → field_expression > identifier + field_identifier
const obj = funcExpr.firstNamedChild;
if (obj?.type === 'identifier') iterableName = obj.text;
// Extract method name: iter, keys, values, into_iter, etc.
const field = funcExpr.lastNamedChild;
if (field?.type === 'field_identifier') methodName = field.text;
} else if (funcExpr?.type === 'identifier') {
// Direct function call: for user in get_users()
const rawReturn = returnTypeLookup.lookupRawReturnType(funcExpr.text);
if (rawReturn) callExprElementType = extractElementTypeFromString(rawReturn);
}
}
if (!iterableName && !callExprElementType) return;
let elementType: string | undefined;
if (callExprElementType) {
elementType = callExprElementType;
} else {
const containerTypeName = scopeEnv.get(iterableName!);
const typeArgPos = methodToTypeArgPosition(methodName, containerTypeName);
elementType = resolveIterableElementType(
iterableName!, node, scopeEnv, declarationTypeNodes, scope,
extractRustElementTypeFromTypeNode, findRustParamElementType,
typeArgPos,
);
}
if (!elementType) return;
const loopVarName = extractVarName(patternNode);
if (loopVarName) scopeEnv.set(loopVarName, elementType);
};
export const typeConfig: LanguageTypeConfig = {
declarationNodeTypes: DECLARATION_NODE_TYPES,
forLoopNodeTypes: FOR_LOOP_NODE_TYPES,
patternBindingNodeTypes: new Set(['let_condition', 'match_arm']),
extractDeclaration,
extractInitializer,
extractParameter,
scanConstructorBinding,
extractForLoopBinding,
extractPendingAssignment,
extractPatternBinding,
};

View file

@ -1,12 +1,189 @@
import type { SyntaxNode } from '../utils.js';
/** Which type argument to extract from a multi-arg generic container.
* - 'first': key type (e.g., K from Map<K,V>) — used for .keys(), .keySet()
* - 'last': value type (e.g., V from Map<K,V>) — used for .values(), .items(), .iter() */
export type TypeArgPosition = 'first' | 'last';
// ---------------------------------------------------------------------------
// Container type descriptors — maps container base names to type parameter
// semantics per access method. Replaces the simple KEY_METHODS heuristic.
//
// For user-defined generics (MyCache<K,V> extends Map<K,V>), heritage-aware
// fallback can walk the EXTENDS chain to find a matching descriptor.
// ---------------------------------------------------------------------------
/** Describes which type parameter position each access method yields. */
interface ContainerDescriptor {
/** Number of type parameters (1 = single-element, 2 = key-value) */
arity: number;
/** Methods that yield the first type parameter (key type for maps) */
keyMethods: ReadonlySet<string>;
/** Methods that yield the last type parameter (value type) */
valueMethods: ReadonlySet<string>;
}
/** Empty set for containers that have no key-yielding methods */
const NO_KEYS: ReadonlySet<string> = new Set();
/** Standard key-yielding methods across languages */
const STD_KEY_METHODS: ReadonlySet<string> = new Set(['keys']);
const JAVA_KEY_METHODS: ReadonlySet<string> = new Set(['keySet']);
const CSHARP_KEY_METHODS: ReadonlySet<string> = new Set(['Keys']);
/** Standard value-yielding methods across languages */
const STD_VALUE_METHODS: ReadonlySet<string> = new Set(['values', 'get', 'pop', 'remove']);
const CSHARP_VALUE_METHODS: ReadonlySet<string> = new Set(['Values', 'TryGetValue']);
const SINGLE_ELEMENT_METHODS: ReadonlySet<string> = new Set([
'iter', 'into_iter', 'iterator', 'get', 'first', 'last', 'pop',
'peek', 'poll', 'find', 'filter', 'map',
]);
const CONTAINER_DESCRIPTORS: ReadonlyMap<string, ContainerDescriptor> = new Map([
// --- Map / Dict types (arity 2: key + value) ---
['Map', { arity: 2, keyMethods: STD_KEY_METHODS, valueMethods: STD_VALUE_METHODS }],
['WeakMap', { arity: 2, keyMethods: STD_KEY_METHODS, valueMethods: STD_VALUE_METHODS }],
['HashMap', { arity: 2, keyMethods: STD_KEY_METHODS, valueMethods: STD_VALUE_METHODS }],
['BTreeMap', { arity: 2, keyMethods: STD_KEY_METHODS, valueMethods: STD_VALUE_METHODS }],
['LinkedHashMap', { arity: 2, keyMethods: JAVA_KEY_METHODS, valueMethods: STD_VALUE_METHODS }],
['TreeMap', { arity: 2, keyMethods: JAVA_KEY_METHODS, valueMethods: STD_VALUE_METHODS }],
['dict', { arity: 2, keyMethods: STD_KEY_METHODS, valueMethods: STD_VALUE_METHODS }],
['Dict', { arity: 2, keyMethods: STD_KEY_METHODS, valueMethods: STD_VALUE_METHODS }],
['Dictionary', { arity: 2, keyMethods: CSHARP_KEY_METHODS, valueMethods: CSHARP_VALUE_METHODS }],
['SortedDictionary', { arity: 2, keyMethods: CSHARP_KEY_METHODS, valueMethods: CSHARP_VALUE_METHODS }],
['Record', { arity: 2, keyMethods: STD_KEY_METHODS, valueMethods: STD_VALUE_METHODS }],
['OrderedDict', { arity: 2, keyMethods: STD_KEY_METHODS, valueMethods: STD_VALUE_METHODS }],
['ConcurrentHashMap', { arity: 2, keyMethods: JAVA_KEY_METHODS, valueMethods: STD_VALUE_METHODS }],
['ConcurrentDictionary', { arity: 2, keyMethods: CSHARP_KEY_METHODS, valueMethods: CSHARP_VALUE_METHODS }],
// --- Single-element containers (arity 1) ---
['Array', { arity: 1, keyMethods: NO_KEYS, valueMethods: SINGLE_ELEMENT_METHODS }],
['List', { arity: 1, keyMethods: NO_KEYS, valueMethods: SINGLE_ELEMENT_METHODS }],
['ArrayList', { arity: 1, keyMethods: NO_KEYS, valueMethods: SINGLE_ELEMENT_METHODS }],
['LinkedList',{ arity: 1, keyMethods: NO_KEYS, valueMethods: SINGLE_ELEMENT_METHODS }],
['Vec', { arity: 1, keyMethods: NO_KEYS, valueMethods: SINGLE_ELEMENT_METHODS }],
['VecDeque', { arity: 1, keyMethods: NO_KEYS, valueMethods: SINGLE_ELEMENT_METHODS }],
['Set', { arity: 1, keyMethods: NO_KEYS, valueMethods: SINGLE_ELEMENT_METHODS }],
['HashSet', { arity: 1, keyMethods: NO_KEYS, valueMethods: SINGLE_ELEMENT_METHODS }],
['BTreeSet', { arity: 1, keyMethods: NO_KEYS, valueMethods: SINGLE_ELEMENT_METHODS }],
['TreeSet', { arity: 1, keyMethods: NO_KEYS, valueMethods: SINGLE_ELEMENT_METHODS }],
['Queue', { arity: 1, keyMethods: NO_KEYS, valueMethods: SINGLE_ELEMENT_METHODS }],
['Deque', { arity: 1, keyMethods: NO_KEYS, valueMethods: SINGLE_ELEMENT_METHODS }],
['Stack', { arity: 1, keyMethods: NO_KEYS, valueMethods: SINGLE_ELEMENT_METHODS }],
['Sequence', { arity: 1, keyMethods: NO_KEYS, valueMethods: SINGLE_ELEMENT_METHODS }],
['Iterable', { arity: 1, keyMethods: NO_KEYS, valueMethods: SINGLE_ELEMENT_METHODS }],
['Iterator', { arity: 1, keyMethods: NO_KEYS, valueMethods: SINGLE_ELEMENT_METHODS }],
['IEnumerable', { arity: 1, keyMethods: NO_KEYS, valueMethods: SINGLE_ELEMENT_METHODS }],
['IList', { arity: 1, keyMethods: NO_KEYS, valueMethods: SINGLE_ELEMENT_METHODS }],
['ICollection', { arity: 1, keyMethods: NO_KEYS, valueMethods: SINGLE_ELEMENT_METHODS }],
['Collection', { arity: 1, keyMethods: NO_KEYS, valueMethods: SINGLE_ELEMENT_METHODS }],
['ObservableCollection', { arity: 1, keyMethods: NO_KEYS, valueMethods: SINGLE_ELEMENT_METHODS }],
['IEnumerator', { arity: 1, keyMethods: NO_KEYS, valueMethods: SINGLE_ELEMENT_METHODS }],
['SortedSet', { arity: 1, keyMethods: NO_KEYS, valueMethods: SINGLE_ELEMENT_METHODS }],
['Stream', { arity: 1, keyMethods: NO_KEYS, valueMethods: SINGLE_ELEMENT_METHODS }],
['MutableList', { arity: 1, keyMethods: NO_KEYS, valueMethods: SINGLE_ELEMENT_METHODS }],
['MutableSet', { arity: 1, keyMethods: NO_KEYS, valueMethods: SINGLE_ELEMENT_METHODS }],
['LinkedHashSet', { arity: 1, keyMethods: NO_KEYS, valueMethods: SINGLE_ELEMENT_METHODS }],
['ArrayDeque', { arity: 1, keyMethods: NO_KEYS, valueMethods: SINGLE_ELEMENT_METHODS }],
['PriorityQueue', { arity: 1, keyMethods: NO_KEYS, valueMethods: SINGLE_ELEMENT_METHODS }],
['MutableMap', { arity: 2, keyMethods: STD_KEY_METHODS, valueMethods: STD_VALUE_METHODS }],
['list', { arity: 1, keyMethods: NO_KEYS, valueMethods: SINGLE_ELEMENT_METHODS }],
['set', { arity: 1, keyMethods: NO_KEYS, valueMethods: SINGLE_ELEMENT_METHODS }],
['tuple', { arity: 1, keyMethods: NO_KEYS, valueMethods: SINGLE_ELEMENT_METHODS }],
['frozenset', { arity: 1, keyMethods: NO_KEYS, valueMethods: SINGLE_ELEMENT_METHODS }],
]);
/** Determine which type arg to extract based on container type name and access method.
*
* Resolution order:
* 1. If container is known and method is in keyMethods → 'first'
* 2. If container is known with arity 1 → 'last' (same as 'first' for single-arg)
* 3. If container is unknown → fall back to method name heuristic
* 4. Default: 'last' (value type)
*/
export function methodToTypeArgPosition(methodName: string | undefined, containerTypeName?: string): TypeArgPosition {
if (containerTypeName) {
const desc = CONTAINER_DESCRIPTORS.get(containerTypeName);
if (desc) {
// Single-element container: always 'last' (= only arg)
if (desc.arity === 1) return 'last';
// Multi-element: check if method yields key type
if (methodName && desc.keyMethods.has(methodName)) return 'first';
// Default for multi-element: value type
return 'last';
}
}
// Fallback for unknown containers: simple method name heuristic
if (methodName && (methodName === 'keys' || methodName === 'keySet' || methodName === 'Keys')) {
return 'first';
}
return 'last';
}
/** Look up the container descriptor for a type name. Exported for heritage-chain lookups. */
export function getContainerDescriptor(typeName: string): ContainerDescriptor | undefined {
return CONTAINER_DESCRIPTORS.get(typeName);
}
/**
* Shared 3-strategy fallback for resolving the element type of a container variable.
* Used by all for-loop extractors to resolve the loop variable's type from the iterable.
*
* Strategy 1: declarationTypeNodes — raw AST type annotation node (handles container types
* where extractSimpleTypeName returned undefined, e.g., User[], List[User])
* Strategy 2: scopeEnv string — extractElementTypeFromString on the stored type string
* Strategy 3: AST walk — language-specific upward walk to enclosing function parameters
*
* @param extractFromTypeNode Language-specific function to extract element type from AST node
* @param findParamElementType Optional language-specific AST walk to find parameter type
* @param typeArgPos Which generic type arg to extract: 'first' for keys, 'last' for values (default)
*/
export function resolveIterableElementType(
iterableName: string,
node: SyntaxNode,
scopeEnv: ReadonlyMap<string, string>,
declarationTypeNodes: ReadonlyMap<string, SyntaxNode>,
scope: string,
extractFromTypeNode: (typeNode: SyntaxNode, pos?: TypeArgPosition) => string | undefined,
findParamElementType?: (name: string, startNode: SyntaxNode, pos?: TypeArgPosition) => string | undefined,
typeArgPos: TypeArgPosition = 'last',
): string | undefined {
// Strategy 1: declarationTypeNodes AST node (check current scope, then file scope)
const typeNode = declarationTypeNodes.get(`${scope}\0${iterableName}`)
?? (scope !== '' ? declarationTypeNodes.get(`\0${iterableName}`) : undefined);
if (typeNode) {
const t = extractFromTypeNode(typeNode, typeArgPos);
if (t) return t;
}
// Strategy 2: scopeEnv string → extractElementTypeFromString
const iterableType = scopeEnv.get(iterableName);
if (iterableType) {
const el = extractElementTypeFromString(iterableType, typeArgPos);
if (el) return el;
}
// Strategy 3: AST walk to function parameters
if (findParamElementType) return findParamElementType(iterableName, node, typeArgPos);
return undefined;
}
/** Known single-arg nullable wrapper types that unwrap to their inner type
* for receiver resolution. Optional<User> → "User", Option<User> → "User".
* Only nullable wrappers — NOT containers (List, Vec) or async wrappers (Promise, Future).
* See WRAPPER_GENERICS below for the full set used in return-type inference. */
const NULLABLE_WRAPPER_TYPES = new Set([
'Optional', // Java
'Option', // Rust, Scala
'Maybe', // Haskell-style, Kotlin Arrow
]);
/**
* Extract the simple type name from a type AST node.
* Handles generic types (e.g., List<User> → List), qualified names
* (e.g., models.User → User), and nullable types (e.g., User? → User).
* Returns undefined for complex types (unions, intersections, function types).
*/
export const extractSimpleTypeName = (typeNode: SyntaxNode): string | undefined => {
export const extractSimpleTypeName = (typeNode: SyntaxNode, depth = 0): string | undefined => {
if (depth > 50 || typeNode.text.length > 2048) return undefined;
// Direct type identifier (includes Ruby 'constant' for class names)
if (typeNode.type === 'type_identifier' || typeNode.type === 'identifier'
|| typeNode.type === 'simple_identifier' || typeNode.type === 'constant') {
@ -30,18 +207,33 @@ export const extractSimpleTypeName = (typeNode: SyntaxNode): string | undefined
}
}
// C++ template_type (e.g., vector<User>, map<string, User>): extract base name
if (typeNode.type === 'template_type') {
const base = typeNode.childForFieldName('name') ?? typeNode.firstNamedChild;
if (base) return extractSimpleTypeName(base, depth + 1);
}
// Generic types: extract the base type (e.g., List<User> → List)
if (typeNode.type === 'generic_type' || typeNode.type === 'parameterized_type') {
// For nullable wrappers (Optional<User>, Option<User>), unwrap to inner type.
if (typeNode.type === 'generic_type' || typeNode.type === 'parameterized_type'
|| typeNode.type === 'generic_name') {
const base = typeNode.childForFieldName('name')
?? typeNode.childForFieldName('type')
?? typeNode.firstNamedChild;
if (base) return extractSimpleTypeName(base);
if (!base) return undefined;
const baseName = extractSimpleTypeName(base, depth + 1);
// Unwrap known nullable wrappers: Optional<User> → User, Option<User> → User
if (baseName && NULLABLE_WRAPPER_TYPES.has(baseName)) {
const args = extractGenericTypeArgs(typeNode);
if (args.length >= 1) return args[0];
}
return baseName;
}
// Nullable types (Kotlin User?, C# User?)
if (typeNode.type === 'nullable_type') {
const inner = typeNode.firstNamedChild;
if (inner) return extractSimpleTypeName(inner);
if (inner) return extractSimpleTypeName(inner, depth + 1);
}
// Nullable union types (TS/JS: User | null, User | undefined, User | null | undefined)
@ -58,7 +250,7 @@ export const extractSimpleTypeName = (typeNode: SyntaxNode): string | undefined
}
// Only unwrap if exactly one meaningful type remains
if (nonNullTypes.length === 1) {
return extractSimpleTypeName(nonNullTypes[0]);
return extractSimpleTypeName(nonNullTypes[0], depth + 1);
}
}
@ -66,24 +258,35 @@ export const extractSimpleTypeName = (typeNode: SyntaxNode): string | undefined
if (typeNode.type === 'type_annotation' || typeNode.type === 'type'
|| typeNode.type === 'user_type') {
const inner = typeNode.firstNamedChild;
if (inner) return extractSimpleTypeName(inner);
if (inner) return extractSimpleTypeName(inner, depth + 1);
}
// Pointer/reference types (C++, Rust): User*, &User, &mut User
if (typeNode.type === 'pointer_type' || typeNode.type === 'reference_type') {
const inner = typeNode.firstNamedChild;
if (inner) return extractSimpleTypeName(inner);
// Skip mutable_specifier for Rust &mut references — firstNamedChild would be
// `mutable_specifier` not the actual type. Walk named children to find the type.
for (let i = 0; i < typeNode.namedChildCount; i++) {
const child = typeNode.namedChild(i);
if (child && child.type !== 'mutable_specifier') {
return extractSimpleTypeName(child, depth + 1);
}
}
}
// PHP primitive_type (string, int, float, bool)
if (typeNode.type === 'primitive_type') {
// Primitive/predefined types: string, int, float, bool, number, unknown, any
// PHP: primitive_type; TS/JS: predefined_type
// Java: integral_type (int/long/short/byte), floating_point_type (float/double),
// boolean_type (boolean), void_type (void)
if (typeNode.type === 'primitive_type' || typeNode.type === 'predefined_type'
|| typeNode.type === 'integral_type' || typeNode.type === 'floating_point_type'
|| typeNode.type === 'boolean_type' || typeNode.type === 'void_type') {
return typeNode.text;
}
// PHP named_type / optional_type
if (typeNode.type === 'named_type' || typeNode.type === 'optional_type') {
const inner = typeNode.childForFieldName('name') ?? typeNode.firstNamedChild;
if (inner) return extractSimpleTypeName(inner);
if (inner) return extractSimpleTypeName(inner, depth + 1);
}
// Name node (PHP)
@ -101,7 +304,7 @@ export const extractSimpleTypeName = (typeNode: SyntaxNode): string | undefined
export const extractVarName = (node: SyntaxNode): string | undefined => {
if (node.type === 'identifier' || node.type === 'simple_identifier'
|| node.type === 'variable_name' || node.type === 'name'
|| node.type === 'constant') {
|| node.type === 'constant' || node.type === 'property_identifier') {
return node.text;
}
// variable_declarator (Java/C#): has a 'name' field
@ -123,15 +326,19 @@ export const TYPED_PARAMETER_TYPES = new Set([
'optional_parameter', // TS: (x?: Foo)
'formal_parameter', // Java/Kotlin
'parameter', // C#/Rust/Go/Python/Swift
'typed_parameter', // Python: def f(x: Foo) — distinct from 'parameter' in tree-sitter-python
'parameter_declaration', // C/C++ void f(Type name)
'simple_parameter', // PHP function(Foo $x)
'property_promotion_parameter', // PHP 8.0+ constructor promotion: __construct(private Foo $x)
'closure_parameter', // Rust: |user: User| — typed closure parameters
]);
/**
* Extract type arguments from a generic type node.
* e.g., List<User, String> → ['User', 'String'], Vec<User> → ['User']
*
* Used by extractSimpleTypeName to unwrap nullable wrappers (Optional<User> → User).
*
* Handles language-specific AST structures:
* - TS/Java/Rust/Go: generic_type > type_arguments > type nodes
* - C#: generic_type > type_argument_list > type nodes
@ -144,18 +351,20 @@ export const TYPED_PARAMETER_TYPES = new Set([
* returns [] for non-generic types).
* @returns Array of resolved type argument names. Unresolvable arguments are omitted.
*/
export const extractGenericTypeArgs = (typeNode: SyntaxNode): string[] => {
export const extractGenericTypeArgs = (typeNode: SyntaxNode, depth = 0): string[] => {
if (depth > 50) return [];
// Unwrap wrapper nodes that may sit above the generic_type
if (typeNode.type === 'type_annotation' || typeNode.type === 'type'
|| typeNode.type === 'user_type' || typeNode.type === 'nullable_type'
|| typeNode.type === 'optional_type') {
const inner = typeNode.firstNamedChild;
if (inner) return extractGenericTypeArgs(inner);
if (inner) return extractGenericTypeArgs(inner, depth + 1);
return [];
}
// Only process generic/parameterized type nodes
if (typeNode.type !== 'generic_type' && typeNode.type !== 'parameterized_type') {
// Only process generic/parameterized type nodes (includes C#'s generic_name)
if (typeNode.type !== 'generic_type' && typeNode.type !== 'parameterized_type'
&& typeNode.type !== 'generic_name') {
return [];
}
@ -233,6 +442,42 @@ export const hasTypeAnnotation = (node: SyntaxNode): boolean => {
return false;
};
/** Bare nullable keywords that should not produce a receiver binding. */
const NULLABLE_KEYWORDS = new Set(['null', 'undefined', 'void', 'None', 'nil']);
/**
* Strip nullable wrappers from a type name string.
* Used by both lookupInEnv (TypeEnv annotations) and extractReturnTypeName
* (return-type text) to normalize types before receiver lookup.
*
* "User | null" → "User"
* "User | undefined" → "User"
* "User | null | undefined" → "User"
* "User?" → "User"
* "User | Repo" → undefined (genuine union — refuse)
* "null" → undefined
*/
export const stripNullable = (typeName: string): string | undefined => {
let text = typeName.trim();
if (!text) return undefined;
if (NULLABLE_KEYWORDS.has(text)) return undefined;
// Strip nullable suffix: User? → User
if (text.endsWith('?')) text = text.slice(0, -1).trim();
// Strip union with null/undefined/None/nil/void
if (text.includes('|')) {
const parts = text.split('|').map(p => p.trim()).filter(p =>
p !== '' && !NULLABLE_KEYWORDS.has(p)
);
if (parts.length === 1) return parts[0];
return undefined; // genuine union or all-nullable — refuse
}
return text || undefined;
};
/**
* Unwrap an await_expression to get the inner value.
* Returns the node itself if not an await_expression, or null if input is null.
@ -252,11 +497,351 @@ export const extractCalleeName = (callNode: SyntaxNode): string | undefined => {
return extractSimpleTypeName(func);
};
/** Find the first named child with the given node type */
export const findChildByType = (node: SyntaxNode, type: string): SyntaxNode | null => {
for (let i = 0; i < node.namedChildCount; i++) {
const child = node.namedChild(i);
if (child?.type === type) return child;
// Internal helper: extract the first comma-separated argument from a string,
// respecting nested angle-bracket and square-bracket depth.
function extractFirstArg(args: string): string {
let depth = 0;
for (let i = 0; i < args.length; i++) {
const ch = args[i];
if (ch === '<' || ch === '[') depth++;
else if (ch === '>' || ch === ']') depth--;
else if (ch === ',' && depth === 0) return args.slice(0, i).trim();
}
return null;
return args.trim();
}
/**
* Extract element type from a container type string.
* Uses bracket-balanced parsing (no regex) for generic argument extraction.
* Returns undefined for ambiguous or unparseable strings.
*
* Handles:
* - Array<User> → User (generic angle brackets)
* - User[] → User (array suffix)
* - []User → User (Go slice prefix)
* - List[User] → User (Python subscript)
* - [User] → User (Swift array sugar)
* - vector<User> → User (C++ container)
* - Vec<User> → User (Rust container)
*
* For multi-argument generics (Map<K, V>), returns the first or last type arg
* based on `pos` ('first' for keys, 'last' for values — default 'last').
* Returns undefined when the extracted type is not a simple word.
*/
export function extractElementTypeFromString(typeStr: string, pos: TypeArgPosition = 'last'): string | undefined {
if (!typeStr || typeStr.length === 0 || typeStr.length > 2048) return undefined;
// 1. Array suffix: User[] → User
if (typeStr.endsWith('[]')) {
const base = typeStr.slice(0, -2).trim();
return base && /^\w+$/.test(base) ? base : undefined;
}
// 2. Go slice prefix: []User → User
if (typeStr.startsWith('[]')) {
const element = typeStr.slice(2).trim();
return element && /^\w+$/.test(element) ? element : undefined;
}
// 3. Swift array sugar: [User] → User
// Must start with '[', end with ']', and contain no angle brackets
// (to avoid confusing with List[User] handled below).
if (typeStr.startsWith('[') && typeStr.endsWith(']') && !typeStr.includes('<')) {
const element = typeStr.slice(1, -1).trim();
return element && /^\w+$/.test(element) ? element : undefined;
}
// 4. Generic bracket-balanced extraction: Array<User> / List[User] / Vec<User>
// Find the first opening bracket (< or [) and pick the one that appears first.
const openAngle = typeStr.indexOf('<');
const openSquare = typeStr.indexOf('[');
let openIdx = -1;
let openChar = '';
let closeChar = '';
if (openAngle >= 0 && (openSquare < 0 || openAngle < openSquare)) {
openIdx = openAngle;
openChar = '<';
closeChar = '>';
} else if (openSquare >= 0) {
openIdx = openSquare;
openChar = '[';
closeChar = ']';
}
if (openIdx < 0) return undefined;
// Walk bracket-balanced from the character after the opening bracket to find
// the matching close bracket, tracking depth for nested brackets.
// All bracket types (<, >, [, ]) contribute to depth uniformly, but only the
// selected closeChar can match at depth 0 (prevents cross-bracket miscounting).
let depth = 0;
const start = openIdx + 1;
let lastCommaIdx = -1; // Track last top-level comma for 'last' position
for (let i = start; i < typeStr.length; i++) {
const ch = typeStr[i];
if (ch === '<' || ch === '[') {
depth++;
} else if (ch === '>' || ch === ']') {
if (depth === 0) {
// At depth 0 — only match if it is our selected close bracket.
if (ch !== closeChar) return undefined; // mismatched bracket = malformed
if (pos === 'last' && lastCommaIdx >= 0) {
// Return last arg (text after last comma)
const lastArg = typeStr.slice(lastCommaIdx + 1, i).trim();
return lastArg && /^\w+$/.test(lastArg) ? lastArg : undefined;
}
const inner = typeStr.slice(start, i).trim();
const firstArg = extractFirstArg(inner);
return firstArg && /^\w+$/.test(firstArg) ? firstArg : undefined;
}
depth--;
} else if (ch === ',' && depth === 0) {
if (pos === 'first') {
// Return first arg (text before first comma)
const arg = typeStr.slice(start, i).trim();
return arg && /^\w+$/.test(arg) ? arg : undefined;
}
lastCommaIdx = i;
}
}
return undefined;
}
// ── Return type text helpers ─────────────────────────────────────────────
// extractReturnTypeName works on raw return-type text already stored in
// SymbolDefinition (e.g. "User", "Promise<User>", "User | null", "*User").
// Extracts the base user-defined type name.
/** Primitive / built-in types that should NOT produce a receiver binding. */
const PRIMITIVE_TYPES = new Set([
'string', 'number', 'boolean', 'void', 'int', 'float', 'double', 'long',
'short', 'byte', 'char', 'bool', 'str', 'i8', 'i16', 'i32', 'i64',
'u8', 'u16', 'u32', 'u64', 'f32', 'f64', 'usize', 'isize',
'undefined', 'null', 'None', 'nil',
]);
/**
* Extract a simple type name from raw return-type text.
* Handles common patterns:
* "User" → "User"
* "Promise<User>" → "User" (unwrap wrapper generics)
* "Option<User>" → "User"
* "Result<User, Error>" → "User" (first type arg)
* "User | null" → "User" (strip nullable union)
* "User?" → "User" (strip nullable suffix)
* "*User" → "User" (Go pointer)
* "&User" → "User" (Rust reference)
* Returns undefined for complex types or primitives.
*/
const WRAPPER_GENERICS = new Set([
'Promise', 'Observable', 'Future', 'CompletableFuture', 'Task', 'ValueTask', // async wrappers
'Option', 'Some', 'Optional', 'Maybe', // nullable wrappers
'Result', 'Either', // result wrappers
// Rust smart pointers (Deref to inner type)
'Rc', 'Arc', 'Weak', // pointer types
'MutexGuard', 'RwLockReadGuard', 'RwLockWriteGuard', // guard types
'Ref', 'RefMut', // RefCell guards
'Cow', // copy-on-write
// Containers (List, Array, Vec, Set, etc.) are intentionally excluded —
// methods are called on the container, not the element type.
// Non-wrapper generics return the base type (e.g., List) via the else branch.
]);
/**
* Extracts the first type argument from a comma-separated generic argument string,
* respecting nested angle brackets. For example:
* "Result<User, Error>" → "Result<User, Error>" (no top-level comma)
* "User, Error" → "User"
* "Map<K, V>, string" → "Map<K, V>"
*/
function extractFirstGenericArg(args: string): string {
let depth = 0;
for (let i = 0; i < args.length; i++) {
if (args[i] === '<') depth++;
else if (args[i] === '>') depth--;
else if (args[i] === ',' && depth === 0) return args.slice(0, i).trim();
}
return args.trim();
}
/**
* Extract the first non-lifetime type argument from a generic argument string.
* Skips Rust lifetime parameters (e.g., `'a`, `'_`) to find the actual type.
* "'_, User" → "User"
* "'a, User" → "User"
* "User, Error" → "User" (no lifetime — delegates to extractFirstGenericArg)
*/
function extractFirstTypeArg(args: string): string {
let remaining = args;
while (remaining) {
const first = extractFirstGenericArg(remaining);
if (!first.startsWith("'")) return first;
// Skip past this lifetime arg + the comma separator
const commaIdx = remaining.indexOf(',', first.length);
if (commaIdx < 0) return first; // only lifetimes — fall through
remaining = remaining.slice(commaIdx + 1).trim();
}
return args.trim();
}
const MAX_RETURN_TYPE_INPUT_LENGTH = 2048;
const MAX_RETURN_TYPE_LENGTH = 512;
export const extractReturnTypeName = (raw: string, depth = 0): string | undefined => {
if (depth > 10) return undefined;
if (raw.length > MAX_RETURN_TYPE_INPUT_LENGTH) return undefined;
let text = raw.trim();
if (!text) return undefined;
// Strip pointer/reference prefixes: *User, &User, &mut User
text = text.replace(/^[&*]+\s*(mut\s+)?/, '');
// Strip nullable suffix: User?
text = text.replace(/\?$/, '');
// Handle union types: "User | null" → "User"
if (text.includes('|')) {
const parts = text.split('|').map(p => p.trim()).filter(p =>
p !== 'null' && p !== 'undefined' && p !== 'void' && p !== 'None' && p !== 'nil'
);
if (parts.length === 1) text = parts[0];
else return undefined; // genuine union — too complex
}
// Handle generics: Promise<User> → unwrap if wrapper, else take base
const genericMatch = text.match(/^(\w+)\s*<(.+)>$/);
if (genericMatch) {
const [, base, args] = genericMatch;
if (WRAPPER_GENERICS.has(base)) {
// Take the first non-lifetime type argument, using bracket-balanced splitting
// so that nested generics like Result<User, Error> are not split at the inner
// comma. Lifetime parameters (Rust 'a, '_) are skipped.
const firstArg = extractFirstTypeArg(args);
return extractReturnTypeName(firstArg, depth + 1);
}
// Non-wrapper generic: return the base type (e.g., Map<K,V> → Map)
return PRIMITIVE_TYPES.has(base.toLowerCase()) ? undefined : base;
}
// Bare wrapper type without generic argument (e.g. Task, Promise, Option)
// should not produce a binding — these are meaningless without a type parameter
if (WRAPPER_GENERICS.has(text)) return undefined;
// Handle qualified names: models.User → User, Models::User → User, \App\Models\User → User
if (text.includes('::') || text.includes('.') || text.includes('\\')) {
text = text.split(/::|[.\\]/).pop()!;
}
// Final check: skip primitives
if (PRIMITIVE_TYPES.has(text) || PRIMITIVE_TYPES.has(text.toLowerCase())) return undefined;
// Must start with uppercase (class/type convention) or be a valid identifier
if (!/^[A-Z_]\w*$/.test(text)) return undefined;
// If the final extracted type name is too long, reject it
if (text.length > MAX_RETURN_TYPE_LENGTH) return undefined;
return text;
};
// ── Property declared-type extraction ────────────────────────────────────
// Shared between parse-worker (worker path) and parsing-processor (sequential path).
/**
* Extract the declared type of a property/field from its AST definition node.
* Handles cross-language patterns:
* - TypeScript: `name: Type` → type_annotation child
* - Java: `Type name` → type child on field_declaration
* - C#: `Type Name { get; set; }` → type child on property_declaration
* - Go: `Name Type` → type child on field_declaration
* - Kotlin: `var name: Type` → variable_declaration child with type field
*
* Returns the normalized type name, or undefined if no type can be extracted.
*/
export const extractPropertyDeclaredType = (definitionNode: SyntaxNode | null): string | undefined => {
if (!definitionNode) return undefined;
// Strategy 1: Look for a `type` or `type_annotation` named field
const typeNode = definitionNode.childForFieldName?.('type');
if (typeNode) {
const typeName = extractSimpleTypeName(typeNode);
if (typeName) return typeName;
// Fallback: use the raw text (for complex types like User[] or List<User>)
const text = typeNode.text?.trim();
if (text && text.length < 100) return text;
}
// Strategy 2: Walk children looking for type_annotation (TypeScript pattern)
for (let i = 0; i < definitionNode.childCount; i++) {
const child = definitionNode.child(i);
if (!child) continue;
if (child.type === 'type_annotation') {
// Type annotation has the actual type as a child
for (let j = 0; j < child.childCount; j++) {
const typeChild = child.child(j);
if (typeChild && typeChild.type !== ':') {
const typeName = extractSimpleTypeName(typeChild);
if (typeName) return typeName;
const text = typeChild.text?.trim();
if (text && text.length < 100) return text;
}
}
}
}
// Strategy 3: For Java field_declaration, the type is a sibling of variable_declarator
// AST: (field_declaration type: (type_identifier) declarator: (variable_declarator ...))
const parentDecl = definitionNode.parent;
if (parentDecl) {
const parentType = parentDecl.childForFieldName?.('type');
if (parentType) {
const typeName = extractSimpleTypeName(parentType);
if (typeName) return typeName;
}
}
// Strategy 4: Kotlin property_declaration — type is nested inside variable_declaration child
// AST: (property_declaration (variable_declaration (simple_identifier) ":" (user_type (type_identifier))))
// Kotlin's variable_declaration has NO named 'type' field — children are all positional.
for (let i = 0; i < definitionNode.childCount; i++) {
const child = definitionNode.child(i);
if (child?.type === 'variable_declaration') {
// Try named field first (works for other languages sharing this strategy)
const varType = child.childForFieldName?.('type');
if (varType) {
const typeName = extractSimpleTypeName(varType);
if (typeName) return typeName;
const text = varType.text?.trim();
if (text && text.length < 100) return text;
}
// Fallback: walk unnamed children for user_type / type_identifier (Kotlin)
for (let j = 0; j < child.namedChildCount; j++) {
const varChild = child.namedChild(j);
if (varChild && (varChild.type === 'user_type' || varChild.type === 'type_identifier'
|| varChild.type === 'nullable_type' || varChild.type === 'generic_type')) {
const typeName = extractSimpleTypeName(varChild);
if (typeName) return typeName;
}
}
}
}
// Strategy 5: PHP @var PHPDoc — look for preceding comment with @var Type
// Handles pre-PHP-7.4 code: /** @var Address */ public $address;
const prevSibling = definitionNode.previousNamedSibling ?? definitionNode.parent?.previousNamedSibling;
if (prevSibling?.type === 'comment') {
const commentText = prevSibling.text;
const varMatch = commentText?.match(/@var\s+([A-Z][\w\\]*)/);
if (varMatch) {
// Strip namespace prefix: \App\Models\User → User
const raw = varMatch[1];
const base = raw.includes('\\') ? raw.split('\\').pop()! : raw;
if (base && /^[A-Z]\w*$/.test(base)) return base;
}
}
return undefined;
};

View file

@ -1,6 +1,7 @@
import type { SyntaxNode } from '../utils.js';
import type { LanguageTypeConfig, ParameterExtractor, TypeBindingExtractor, InitializerExtractor, ClassNameLookup, ConstructorBindingScanner } from './types.js';
import { extractSimpleTypeName, extractVarName, findChildByType, hasTypeAnnotation } from './shared.js';
import { extractSimpleTypeName, extractVarName, hasTypeAnnotation } from './shared.js';
import { findChild } from '../resolvers/utils.js';
const DECLARATION_NODE_TYPES: ReadonlySet<string> = new Set([
'property_declaration',
@ -10,9 +11,9 @@ const DECLARATION_NODE_TYPES: ReadonlySet<string> = new Set([
const extractDeclaration: TypeBindingExtractor = (node: SyntaxNode, env: Map<string, string>): void => {
// Swift property_declaration has pattern and type_annotation
const pattern = node.childForFieldName('pattern')
?? findChildByType(node, 'pattern');
?? findChild(node, 'pattern');
const typeAnnotation = node.childForFieldName('type')
?? findChildByType(node, 'type_annotation');
?? findChild(node, 'type_annotation');
if (!pattern || !typeAnnotation) return;
const varName = extractVarName(pattern) ?? pattern.text;
const typeName = extractSimpleTypeName(typeAnnotation);
@ -45,14 +46,14 @@ const extractParameter: ParameterExtractor = (node: SyntaxNode, env: Map<string,
const extractInitializer: InitializerExtractor = (node: SyntaxNode, env: Map<string, string>, classNames: ClassNameLookup): void => {
if (node.type !== 'property_declaration') return;
// Skip if has type annotation — extractDeclaration handled it
if (node.childForFieldName('type') || findChildByType(node, 'type_annotation')) return;
if (node.childForFieldName('type') || findChild(node, 'type_annotation')) return;
// Find pattern (variable name)
const pattern = node.childForFieldName('pattern') ?? findChildByType(node, 'pattern');
const pattern = node.childForFieldName('pattern') ?? findChild(node, 'pattern');
if (!pattern) return;
const varName = extractVarName(pattern) ?? pattern.text;
if (!varName || env.has(varName)) return;
// Find call_expression in the value
const callExpr = findChildByType(node, 'call_expression');
const callExpr = findChild(node, 'call_expression');
if (!callExpr) return;
const callee = callExpr.firstNamedChild;
if (!callee) return;

View file

@ -24,10 +24,119 @@ export type ConstructorBindingScanner = (node: SyntaxNode) => { varName: string;
* rather than in AST fields. Returns undefined if no return type can be determined. */
export type ReturnTypeExtractor = (node: SyntaxNode) => string | undefined;
/** Infer the type name of a literal AST node for overload disambiguation.
* Returns the canonical type name (e.g. 'int', 'String', 'boolean') or undefined
* for non-literal nodes. Only used when resolveCallTarget has multiple candidates
* with parameterTypes — ~1-3% of call sites. */
export type LiteralTypeInferrer = (node: SyntaxNode) => string | undefined;
/** Detect constructor-style call expressions that don't use `new` keyword.
* Returns the constructor class name if the node's initializer is a constructor call,
* or undefined otherwise. Used for virtual dispatch in languages like Kotlin
* where constructors are syntactically identical to function calls, and C++
* where smart pointer factory functions (make_shared/make_unique) wrap constructors. */
export type ConstructorTypeDetector = (node: SyntaxNode, classNames: ClassNameLookup) => string | undefined;
/** Unwrap a declared type name to its inner type for virtual dispatch comparison.
* E.g., C++ shared_ptr<Animal> → Animal. Returns undefined if no unwrapping applies. */
export type DeclaredTypeUnwrapper = (declaredType: string, typeNode: SyntaxNode) => string | undefined;
/** Narrow lookup interface for resolving a callee name → return type name.
* Backed by SymbolTable.lookupFuzzyCallable; passed via ForLoopExtractorContext.
* Conservative: returns undefined when the callee is ambiguous (0 or 2+ matches). */
export interface ReturnTypeLookup {
/** Processed type name after stripping wrappers (e.g., 'User' from 'Promise<User>').
* Use for call-result variable bindings (`const b = foo()`). */
lookupReturnType(callee: string): string | undefined;
/** Raw return type as declared in the symbol (e.g., '[]User', 'List<User>').
* Use for iterable-element extraction (`for v := range foo()`). */
lookupRawReturnType(callee: string): string | undefined;
}
/** Context object passed to ForLoopExtractor.
* Groups the four parameters that were previously positional. */
export interface ForLoopExtractorContext {
/** Mutable type-env for the current scope — extractor writes bindings here */
scopeEnv: Map<string, string>;
/** Maps `scope\0varName` to the declaration's type annotation AST node */
declarationTypeNodes: ReadonlyMap<string, SyntaxNode>;
/** Current scope key, e.g. `"process@42"` */
scope: string;
/** Resolves a callee name to its declared return type (undefined = unknown/ambiguous) */
returnTypeLookup: ReturnTypeLookup;
}
/** Extracts loop variable type binding from a for-each statement. */
export type ForLoopExtractor = (node: SyntaxNode, ctx: ForLoopExtractorContext) => void;
/** Discriminated union for pending Tier-2 propagation items.
* - `copy` — `const b = a` (identifier alias, propagate a's type to b)
* - `callResult` — `const b = foo()` (bind b to foo's declared return type)
* - `fieldAccess` — `const b = a.field` (bind b to field's declaredType on a's type)
* - `methodCallResult` — `const b = a.method()` (bind b to method's returnType on a's type) */
export type PendingAssignment =
| { kind: 'copy'; lhs: string; rhs: string }
| { kind: 'callResult'; lhs: string; callee: string }
| { kind: 'fieldAccess'; lhs: string; receiver: string; field: string }
| { kind: 'methodCallResult'; lhs: string; receiver: string; method: string };
/** Extracts a pending assignment for Tier 2 propagation.
* Returns a PendingAssignment when the RHS is a bare identifier (`copy`), a
* call expression (`callResult`), a field access (`fieldAccess`), or a
* method call with receiver (`methodCallResult`) and the LHS has no resolved type yet.
* May return an array of PendingAssignment items for destructuring patterns
* (e.g., `const { a, b } = obj` emits N fieldAccess items).
* Returns undefined if the node is not a matching assignment. */
export type PendingAssignmentExtractor = (
node: SyntaxNode,
scopeEnv: ReadonlyMap<string, string>,
) => PendingAssignment | PendingAssignment[] | undefined;
/** Result of a pattern binding extraction. */
export interface PatternBindingResult {
varName: string;
typeName: string;
/** Optional: AST node whose position range should be used for the patternOverride.
* When present, the override uses this node's range instead of the auto-detected
* branch scope. Used by null-check narrowing to target the if-body specifically. */
narrowingRange?: { startIndex: number; endIndex: number };
}
/** Extracts a typed variable binding from a pattern-matching construct.
* Returns { varName, typeName } for patterns that introduce NEW variables
* or narrow existing variables (null-check narrowing).
* Examples: `if let Some(user) = opt` (Rust), `x instanceof User user` (Java),
* `if (x != null)` (null-check narrowing in TS/Kotlin/C#).
* Conservative: returns undefined when the source variable's type is unknown.
*
* @param scopeEnv Read-only view of already-resolved type bindings in the current scope.
* @param declarationTypeNodes Maps `scope\0varName` to the original declaration's type
* annotation AST node. Allows extracting generic type arguments (e.g., T from Result<T,E>)
* that are stripped during normal TypeEnv extraction.
* @param scope Current scope key (e.g. `"process@42"`) for declarationTypeNodes lookups. */
export type PatternBindingExtractor = (
node: SyntaxNode,
scopeEnv: ReadonlyMap<string, string>,
declarationTypeNodes: ReadonlyMap<string, SyntaxNode>,
scope: string,
) => PatternBindingResult | undefined;
/** Per-language type extraction configuration */
export interface LanguageTypeConfig {
/** Allow pattern binding to overwrite existing scopeEnv entries.
* WARNING: Enables function-scope type pollution. Only for languages with
* smart-cast semantics (e.g., Kotlin `when/is`) where the subject variable
* already exists in scopeEnv from its declaration. */
readonly allowPatternBindingOverwrite?: boolean;
/** Node types that represent typed declarations for this language */
declarationNodeTypes: ReadonlySet<string>;
/** AST node types for for-each/for-in statements with explicit element types. */
forLoopNodeTypes?: ReadonlySet<string>;
/** Optional allowlist of AST node types on which extractPatternBinding should run.
* When present, extractPatternBinding is only invoked for nodes whose type is in this set,
* short-circuiting the call for all other node types. When absent, every node is passed to
* extractPatternBinding (legacy behaviour). */
patternBindingNodeTypes?: ReadonlySet<string>;
/** Extract a (varName → typeName) binding from a declaration node */
extractDeclaration: TypeBindingExtractor;
/** Extract a (varName → typeName) binding from a parameter node */
@ -44,4 +153,20 @@ export interface LanguageTypeConfig {
/** Extract return type from comment-based annotations (e.g. YARD @return [Type]).
* Called as fallback when extractMethodSignature finds no AST-based return type. */
extractReturnType?: ReturnTypeExtractor;
/** Extract loop variable → type binding from a for-each AST node. */
extractForLoopBinding?: ForLoopExtractor;
/** Extract pending assignment for Tier 2 propagation.
* Called on declaration/assignment nodes; returns a PendingAssignment when the RHS
* is a bare identifier (copy) or call expression (callResult) and the LHS has no
* resolved type yet. Language-specific because AST shapes differ widely. */
extractPendingAssignment?: PendingAssignmentExtractor;
/** Extract a typed variable binding from a pattern-matching construct.
* Called on every AST node; returns { varName, typeName } when the node introduces a new
* typed variable via pattern matching (e.g. `if let Some(x) = opt`, `x instanceof T t`).
* The extractor receives the current scope's resolved bindings (read-only) to look up the
* source variable's type. Returns undefined for non-matching nodes or unknown source types. */
extractPatternBinding?: PatternBindingExtractor;
inferLiteralType?: LiteralTypeInferrer;
detectConstructorType?: ConstructorTypeDetector;
unwrapDeclaredType?: DeclaredTypeUnwrapper;
}

View file

@ -1,12 +1,13 @@
import type { SyntaxNode } from '../utils.js';
import type { LanguageTypeConfig, ParameterExtractor, TypeBindingExtractor, InitializerExtractor, ClassNameLookup, ConstructorBindingScanner, ReturnTypeExtractor } from './types.js';
import { extractSimpleTypeName, extractVarName, hasTypeAnnotation, unwrapAwait, extractCalleeName } from './shared.js';
import type { LanguageTypeConfig, ParameterExtractor, TypeBindingExtractor, InitializerExtractor, ClassNameLookup, ConstructorBindingScanner, ReturnTypeExtractor, PendingAssignmentExtractor, PendingAssignment, ForLoopExtractor, PatternBindingExtractor, LiteralTypeInferrer } from './types.js';
import { extractSimpleTypeName, extractVarName, hasTypeAnnotation, unwrapAwait, extractCalleeName, extractElementTypeFromString, extractGenericTypeArgs, resolveIterableElementType, methodToTypeArgPosition, type TypeArgPosition } from './shared.js';
const DECLARATION_NODE_TYPES: ReadonlySet<string> = new Set([
'lexical_declaration',
'variable_declaration',
'function_declaration', // JSDoc @param on function declarations
'method_definition', // JSDoc @param on class methods
'public_field_definition', // class field: private users: User[]
]);
const normalizeJsDocType = (raw: string): string | undefined => {
@ -80,6 +81,18 @@ const extractDeclaration: TypeBindingExtractor = (node: SyntaxNode, env: Map<str
return;
}
// Class field: `private users: User[]` — public_field_definition has name + type fields directly.
if (node.type === 'public_field_definition') {
const nameNode = node.childForFieldName('name');
const typeAnnotation = node.childForFieldName('type');
if (!nameNode || !typeAnnotation) return;
const varName = nameNode.text;
if (!varName) return;
const typeName = extractSimpleTypeName(typeAnnotation);
if (typeName) env.set(varName, typeName);
return;
}
for (let i = 0; i < node.namedChildCount; i++) {
const declarator = node.namedChild(i);
if (declarator?.type !== 'variable_declarator') continue;
@ -191,11 +204,432 @@ const extractReturnType: ReturnTypeExtractor = (node) => {
return undefined;
};
const FOR_LOOP_NODE_TYPES: ReadonlySet<string> = new Set([
'for_in_statement',
]);
/** TS function/method node types that carry a parameters list. */
const TS_FUNCTION_NODE_TYPES = new Set([
'function_declaration', 'function_expression', 'arrow_function',
'method_definition', 'generator_function', 'generator_function_declaration',
]);
/**
* Extract element type from a TypeScript type annotation AST node.
* Handles:
* type_annotation ": User[]" → array_type → type_identifier "User"
* type_annotation ": Array<User>" → generic_type → extractGenericTypeArgs → "User"
* Falls back to text-based extraction via extractElementTypeFromString.
*/
const extractTsElementTypeFromAnnotation = (typeAnnotation: SyntaxNode, pos: TypeArgPosition = 'last', depth = 0): string | undefined => {
if (depth > 50) return undefined;
// Unwrap type_annotation (the node text includes ': ' prefix)
const inner = typeAnnotation.type === 'type_annotation'
? (typeAnnotation.firstNamedChild ?? typeAnnotation)
: typeAnnotation;
// readonly User[] — readonly_type wraps array_type: unwrap and recurse
if (inner.type === 'readonly_type') {
const wrapped = inner.firstNamedChild;
if (wrapped) return extractTsElementTypeFromAnnotation(wrapped, pos, depth + 1);
}
// User[] — array_type: first named child is the element type
if (inner.type === 'array_type') {
const elem = inner.firstNamedChild;
if (elem) return extractSimpleTypeName(elem);
}
// Array<User>, Map<string, User> — generic_type
// pos determines which type arg: 'first' for keys, 'last' for values
if (inner.type === 'generic_type') {
const args = extractGenericTypeArgs(inner);
if (args.length >= 1) return pos === 'first' ? args[0] : args[args.length - 1];
}
// Fallback: strip ': ' prefix from type_annotation text and use string extraction
const rawText = inner.text;
return extractElementTypeFromString(rawText, pos);
};
/**
* Search a statement_block (function body) for a variable_declarator named `iterableName`
* that has a type annotation, preceding the given `beforeNode`.
* Returns the element type from the type annotation, or undefined.
*/
const findTsLocalDeclElementType = (
iterableName: string,
blockNode: SyntaxNode,
beforeNode: SyntaxNode,
pos: TypeArgPosition = 'last',
): string | undefined => {
for (let i = 0; i < blockNode.namedChildCount; i++) {
const stmt = blockNode.namedChild(i);
if (!stmt) continue;
// Stop when we reach the for-loop itself
if (stmt === beforeNode || stmt.startIndex >= beforeNode.startIndex) break;
// Look for lexical_declaration or variable_declaration
if (stmt.type !== 'lexical_declaration' && stmt.type !== 'variable_declaration') continue;
for (let j = 0; j < stmt.namedChildCount; j++) {
const decl = stmt.namedChild(j);
if (decl?.type !== 'variable_declarator') continue;
const nameNode = decl.childForFieldName('name');
if (nameNode?.text !== iterableName) continue;
const typeAnnotation = decl.childForFieldName('type');
if (typeAnnotation) return extractTsElementTypeFromAnnotation(typeAnnotation, pos);
}
}
return undefined;
};
/**
* Walk up the AST from a for-loop node to find the enclosing function scope,
* then search (1) its parameter list and (2) local declarations in the body
* for a variable named `iterableName` with a container type annotation.
* Returns the element type extracted from the annotation, or undefined.
*/
const findTsIterableElementType = (iterableName: string, startNode: SyntaxNode, pos: TypeArgPosition = 'last'): string | undefined => {
let current: SyntaxNode | null = startNode.parent;
// Capture the immediate statement_block parent to search local declarations
const blockNode = current?.type === 'statement_block' ? current : null;
while (current) {
if (TS_FUNCTION_NODE_TYPES.has(current.type)) {
// Search function parameters
const paramsNode = current.childForFieldName('parameters')
?? current.childForFieldName('formal_parameters');
if (paramsNode) {
for (let i = 0; i < paramsNode.namedChildCount; i++) {
const param = paramsNode.namedChild(i);
if (!param) continue;
const patternNode = param.childForFieldName('pattern') ?? param.childForFieldName('name');
if (patternNode?.text === iterableName) {
const typeAnnotation = param.childForFieldName('type');
if (typeAnnotation) return extractTsElementTypeFromAnnotation(typeAnnotation, pos);
}
}
}
// Search local declarations in the function body (statement_block)
if (blockNode) {
const result = findTsLocalDeclElementType(iterableName, blockNode, startNode, pos);
if (result) return result;
}
break; // stop at the nearest function boundary
}
current = current.parent;
}
return undefined;
};
/**
* TypeScript/JavaScript: for (const user of users) where users has a known array type.
*
* Both `for...of` and `for...in` use the same `for_in_statement` AST node in tree-sitter.
* We differentiate by checking for the `of` keyword among the unnamed children.
*
* Tier 1c: resolves the element type via three strategies in priority order:
* 1. declarationTypeNodes — raw type annotation AST node (covers Array<User> from declarations)
* 2. scopeEnv string — extractElementTypeFromString on the stored type (covers locally annotated vars)
* 3. AST walk — walks up to the enclosing function's parameters to read User[] annotations directly
* Only handles `for...of`; `for...in` produces string keys, not element types.
*/
const extractForLoopBinding: ForLoopExtractor = (node, { scopeEnv, declarationTypeNodes, scope, returnTypeLookup }): void => {
if (node.type !== 'for_in_statement') return;
// Confirm this is `for...of`, not `for...in`, by scanning unnamed children for the keyword text.
let isForOf = false;
for (let i = 0; i < node.childCount; i++) {
const child = node.child(i);
if (child && !child.isNamed && child.text === 'of') {
isForOf = true;
break;
}
}
if (!isForOf) return;
// The iterable is the `right` field — may be identifier, member_expression, or call_expression.
const rightNode = node.childForFieldName('right');
let iterableName: string | undefined;
let methodName: string | undefined;
let callExprElementType: string | undefined;
if (rightNode?.type === 'identifier') {
iterableName = rightNode.text;
} else if (rightNode?.type === 'member_expression') {
const prop = rightNode.childForFieldName('property');
if (prop) iterableName = prop.text;
} else if (rightNode?.type === 'call_expression') {
// entries.values() → call_expression > function: member_expression > object + property
// this.repos.values() → nested member_expression: extract property from inner member
// getUsers() → call_expression > function: identifier (Phase 7.3 — return-type path)
const fn = rightNode.childForFieldName('function');
if (fn?.type === 'member_expression') {
const obj = fn.childForFieldName('object');
const prop = fn.childForFieldName('property');
if (obj?.type === 'identifier') {
iterableName = obj.text;
} else if (obj?.type === 'member_expression') {
// this.repos.values() → obj = this.repos → extract 'repos'
const innerProp = obj.childForFieldName('property');
if (innerProp) iterableName = innerProp.text;
}
if (prop?.type === 'property_identifier') methodName = prop.text;
} else if (fn?.type === 'identifier') {
// Direct function call: for (const user of getUsers())
const rawReturn = returnTypeLookup.lookupRawReturnType(fn.text);
if (rawReturn) callExprElementType = extractElementTypeFromString(rawReturn);
}
}
if (!iterableName && !callExprElementType) return;
let elementType: string | undefined;
if (callExprElementType) {
elementType = callExprElementType;
} else {
// Look up the container's base type name for descriptor-aware resolution
const containerTypeName = scopeEnv.get(iterableName!);
const typeArgPos = methodToTypeArgPosition(methodName, containerTypeName);
elementType = resolveIterableElementType(
iterableName!, node, scopeEnv, declarationTypeNodes, scope,
extractTsElementTypeFromAnnotation, findTsIterableElementType,
typeArgPos,
);
}
if (!elementType) return;
// The loop variable is the `left` field.
const leftNode = node.childForFieldName('left');
if (!leftNode) return;
// Handle destructured for-of: for (const [k, v] of entries)
// AST: left = array_pattern directly (no variable_declarator wrapper)
// Bind the LAST identifier to the element type (value in [key, value] patterns)
if (leftNode.type === 'array_pattern') {
const lastChild = leftNode.lastNamedChild;
if (lastChild?.type === 'identifier') {
scopeEnv.set(lastChild.text, elementType);
}
return;
}
if (leftNode.type === 'object_pattern') {
// Object destructuring (e.g., `for (const { id } of users)`) destructures
// into fields of the element type. Without field-level resolution, we cannot
// bind individual properties to their correct types. Skip to avoid false bindings.
return;
}
let loopVarNode: SyntaxNode | null = leftNode;
// `const user` parses as: left → variable_declarator containing an identifier named `user`
if (loopVarNode.type === 'variable_declarator') {
loopVarNode = loopVarNode.childForFieldName('name') ?? loopVarNode.firstNamedChild;
}
if (!loopVarNode) return;
const loopVarName = extractVarName(loopVarNode);
if (loopVarName) scopeEnv.set(loopVarName, elementType);
};
/** TS/JS: const alias = u → variable_declarator with name/value fields.
* Also handles destructuring: `const { a, b } = obj` → N fieldAccess items. */
const extractPendingAssignment: PendingAssignmentExtractor = (node, scopeEnv) => {
for (let i = 0; i < node.namedChildCount; i++) {
const child = node.namedChild(i);
if (!child || child.type !== 'variable_declarator') continue;
const nameNode = child.childForFieldName('name');
const valueNode = child.childForFieldName('value');
if (!nameNode || !valueNode) continue;
// Object destructuring: `const { address, name } = user`
// Emits N fieldAccess items — one per destructured binding.
if (nameNode.type === 'object_pattern' && valueNode.type === 'identifier') {
const receiver = valueNode.text;
const items: PendingAssignment[] = [];
for (let j = 0; j < nameNode.namedChildCount; j++) {
const prop = nameNode.namedChild(j);
if (!prop) continue;
if (prop.type === 'shorthand_property_identifier_pattern') {
// `const { name } = user` → shorthand: varName = fieldName
const varName = prop.text;
if (!scopeEnv.has(varName)) {
items.push({ kind: 'fieldAccess', lhs: varName, receiver, field: varName });
}
} else if (prop.type === 'pair_pattern') {
// `const { address: addr } = user` → pair_pattern: key=field, value=varName
const keyNode = prop.childForFieldName('key');
const valNode = prop.childForFieldName('value');
if (keyNode && valNode) {
const fieldName = keyNode.text;
const varName = valNode.text;
if (!scopeEnv.has(varName)) {
items.push({ kind: 'fieldAccess', lhs: varName, receiver, field: fieldName });
}
}
}
}
if (items.length > 0) return items;
continue;
}
const lhs = nameNode.text;
if (scopeEnv.has(lhs)) continue;
if (valueNode.type === 'identifier') return { kind: 'copy', lhs, rhs: valueNode.text };
// member_expression RHS → fieldAccess (a.field, this.field)
if (valueNode.type === 'member_expression') {
const obj = valueNode.childForFieldName('object');
const prop = valueNode.childForFieldName('property');
if (obj && prop?.type === 'property_identifier' &&
(obj.type === 'identifier' || obj.type === 'this')) {
return { kind: 'fieldAccess', lhs, receiver: obj.text, field: prop.text };
}
continue;
}
// Unwrap await: `const user = await fetchUser()` or `await a.getC()`
const callNode = unwrapAwait(valueNode);
if (!callNode || callNode.type !== 'call_expression') continue;
const funcNode = callNode.childForFieldName('function');
if (!funcNode) continue;
// Simple call → callResult: getUser()
if (funcNode.type === 'identifier') {
return { kind: 'callResult', lhs, callee: funcNode.text };
}
// Method call with receiver → methodCallResult: a.getC()
if (funcNode.type === 'member_expression') {
const obj = funcNode.childForFieldName('object');
const prop = funcNode.childForFieldName('property');
if (obj && prop?.type === 'property_identifier' &&
(obj.type === 'identifier' || obj.type === 'this')) {
return { kind: 'methodCallResult', lhs, receiver: obj.text, method: prop.text };
}
}
}
return undefined;
};
/** Null-check keywords that indicate a null-comparison in binary expressions. */
const NULL_CHECK_KEYWORDS = new Set(['null', 'undefined']);
/**
* Find the if-body (consequence) block for a null-check binary_expression.
* Walks up from the binary_expression through parenthesized_expression to if_statement,
* then returns the consequence block (statement_block).
*
* AST structure: if_statement > parenthesized_expression > binary_expression
* if_statement > statement_block (consequence)
*/
const findIfConsequenceBlock = (binaryExpr: SyntaxNode): SyntaxNode | undefined => {
// Walk up to find the if_statement (typically: binary_expression > parenthesized_expression > if_statement)
let current = binaryExpr.parent;
while (current) {
if (current.type === 'if_statement') {
// The consequence is the first statement_block child of if_statement
for (let i = 0; i < current.childCount; i++) {
const child = current.child(i);
if (child?.type === 'statement_block') return child;
}
return undefined;
}
// Stop climbing at function/block boundaries — don't cross scope
if (current.type === 'function_declaration' || current.type === 'function_expression'
|| current.type === 'arrow_function' || current.type === 'method_definition') return undefined;
current = current.parent;
}
return undefined;
};
/** TS instanceof narrowing: `x instanceof User` → bind x to User.
* Also handles null-check narrowing: `x !== null`, `x != undefined` etc.
* instanceof: first-writer-wins (no prior type binding).
* null-check: position-indexed narrowing via narrowingRange. */
const extractPatternBinding: PatternBindingExtractor = (node, scopeEnv, declarationTypeNodes, scope) => {
if (node.type !== 'binary_expression') return undefined;
// Check for instanceof first (existing behavior)
const instanceofOp = node.children.find(c => !c.isNamed && c.text === 'instanceof');
if (instanceofOp) {
const left = node.namedChild(0);
const right = node.namedChild(1);
if (left?.type !== 'identifier' || right?.type !== 'identifier') return undefined;
return { varName: left.text, typeName: right.text };
}
// Null-check narrowing: x !== null, x != null, x !== undefined, x != undefined
const op = node.children.find(c => !c.isNamed && (c.text === '!==' || c.text === '!='));
if (!op) return undefined;
const left = node.namedChild(0);
const right = node.namedChild(1);
if (!left || !right) return undefined;
// Determine which side is the variable and which is null/undefined
let varNode: SyntaxNode | undefined;
let isNullCheck = false;
if (left.type === 'identifier' && NULL_CHECK_KEYWORDS.has(right.text)) {
varNode = left;
isNullCheck = true;
} else if (right.type === 'identifier' && NULL_CHECK_KEYWORDS.has(left.text)) {
varNode = right;
isNullCheck = true;
}
if (!isNullCheck || !varNode) return undefined;
const varName = varNode.text;
// Look up the variable's resolved type (already stripped of nullable by extractSimpleTypeName)
const resolvedType = scopeEnv.get(varName);
if (!resolvedType) return undefined;
// Check if the original declaration type was nullable by looking at the raw AST type node.
// extractSimpleTypeName already strips nullable markers, so we need the original to know
// if narrowing is meaningful (i.e., the variable was declared as nullable).
const declTypeNode = declarationTypeNodes.get(`${scope}\0${varName}`);
if (!declTypeNode) return undefined;
const declText = declTypeNode.text;
// Only narrow if the original declaration was nullable
if (!declText.includes('null') && !declText.includes('undefined')) return undefined;
// Find the if-body block to scope the narrowing
const ifBody = findIfConsequenceBlock(node);
if (!ifBody) return undefined;
return {
varName,
typeName: resolvedType,
narrowingRange: { startIndex: ifBody.startIndex, endIndex: ifBody.endIndex },
};
};
/** Infer the type of a literal AST node for TypeScript overload disambiguation. */
const inferTsLiteralType: LiteralTypeInferrer = (node) => {
switch (node.type) {
case 'number':
return 'number';
case 'string':
case 'template_string':
return 'string';
case 'true':
case 'false':
return 'boolean';
case 'null':
return 'null';
case 'undefined':
return 'undefined';
case 'regex':
return 'RegExp';
default:
return undefined;
}
};
export const typeConfig: LanguageTypeConfig = {
declarationNodeTypes: DECLARATION_NODE_TYPES,
forLoopNodeTypes: FOR_LOOP_NODE_TYPES,
patternBindingNodeTypes: new Set(['binary_expression']),
extractDeclaration,
extractParameter,
extractInitializer,
scanConstructorBinding,
extractReturnType,
extractForLoopBinding,
extractPendingAssignment,
extractPatternBinding,
inferLiteralType: inferTsLiteralType,
};

View file

@ -1,98 +1,4 @@
import type Parser from 'tree-sitter';
import { SupportedLanguages } from '../../config/supported-languages.js';
import { generateId } from '../../lib/utils.js';
/** Tree-sitter AST node. Re-exported for use across ingestion modules. */
export type SyntaxNode = Parser.SyntaxNode;
/**
* Ordered list of definition capture keys for tree-sitter query matches.
* Used to extract the definition node from a capture map.
*/
export const DEFINITION_CAPTURE_KEYS = [
'definition.function',
'definition.class',
'definition.interface',
'definition.method',
'definition.struct',
'definition.enum',
'definition.namespace',
'definition.module',
'definition.trait',
'definition.impl',
'definition.type',
'definition.const',
'definition.static',
'definition.typedef',
'definition.macro',
'definition.union',
'definition.property',
'definition.record',
'definition.delegate',
'definition.annotation',
'definition.constructor',
'definition.template',
] as const;
/** Extract the definition node from a tree-sitter query capture map. */
export const getDefinitionNodeFromCaptures = (captureMap: Record<string, any>): any | null => {
for (const key of DEFINITION_CAPTURE_KEYS) {
if (captureMap[key]) return captureMap[key];
}
return null;
};
/**
* Node types that represent function/method definitions across languages.
* Used to find the enclosing function for a call site.
*/
export const FUNCTION_NODE_TYPES = new Set([
// TypeScript/JavaScript
'function_declaration',
'arrow_function',
'function_expression',
'method_definition',
'generator_function_declaration',
// Python
'function_definition',
// Common async variants
'async_function_declaration',
'async_arrow_function',
// Java
'method_declaration',
'constructor_declaration',
// C/C++
// 'function_definition' already included above
// Go
// 'method_declaration' already included from Java
// C#
'local_function_statement',
// Rust
'function_item',
'impl_item', // Methods inside impl blocks
// PHP
'anonymous_function',
// Kotlin
'lambda_literal',
// Swift
'init_declaration',
'deinit_declaration',
// Ruby
'method', // def foo
'singleton_method', // def self.foo
]);
/**
* Node types for standard function declarations that need C/C++ declarator handling.
* Used by extractFunctionName to determine how to extract the function name.
*/
export const FUNCTION_DECLARATION_TYPES = new Set([
'function_declaration',
'function_definition',
'async_function_declaration',
'generator_function_declaration',
'function_item',
]);
/**
* Built-in function/method names that should not be tracked as call targets.
@ -259,191 +165,6 @@ export const BUILT_IN_NAMES = new Set([
/** Check if a name is a built-in function or common noise that should be filtered out */
export const isBuiltInOrNoise = (name: string): boolean => BUILT_IN_NAMES.has(name);
/** AST node types that represent a class-like container (for HAS_METHOD edge extraction) */
export const CLASS_CONTAINER_TYPES = new Set([
'class_declaration', 'abstract_class_declaration',
'interface_declaration', 'struct_declaration', 'record_declaration',
'class_specifier', 'struct_specifier',
'impl_item', 'trait_item',
'class_definition',
'trait_declaration',
'protocol_declaration',
// Ruby
'class',
'module',
// Kotlin
'object_declaration',
'companion_object',
]);
export const CONTAINER_TYPE_TO_LABEL: Record<string, string> = {
class_declaration: 'Class',
abstract_class_declaration: 'Class',
interface_declaration: 'Interface',
struct_declaration: 'Struct',
struct_specifier: 'Struct',
class_specifier: 'Class',
class_definition: 'Class',
impl_item: 'Impl',
trait_item: 'Trait',
trait_declaration: 'Trait',
record_declaration: 'Record',
protocol_declaration: 'Interface',
class: 'Class',
module: 'Module',
object_declaration: 'Class',
companion_object: 'Class',
};
/** Walk up AST to find enclosing class/struct/interface/impl, return its generateId or null.
* For Go method_declaration nodes, extracts receiver type (e.g. `func (u *User) Save()` → User struct). */
export const findEnclosingClassId = (node: any, filePath: string): string | null => {
let current = node.parent;
while (current) {
// Go: method_declaration has a receiver parameter with the struct type
if (current.type === 'method_declaration') {
const receiver = current.childForFieldName?.('receiver');
if (receiver) {
// receiver is a parameter_list: (u *User) or (u User)
const paramDecl = receiver.namedChildren?.find?.((c: any) => c.type === 'parameter_declaration');
if (paramDecl) {
const typeNode = paramDecl.childForFieldName?.('type');
if (typeNode) {
// Unwrap pointer_type (*User → User)
const inner = typeNode.type === 'pointer_type' ? typeNode.firstNamedChild : typeNode;
if (inner && (inner.type === 'type_identifier' || inner.type === 'identifier')) {
return generateId('Struct', `${filePath}:${inner.text}`);
}
}
}
}
}
if (CLASS_CONTAINER_TYPES.has(current.type)) {
// Rust impl_item: for `impl Trait for Struct {}`, pick the type after `for`
if (current.type === 'impl_item') {
const children = current.children ?? [];
const forIdx = children.findIndex((c: any) => c.text === 'for');
if (forIdx !== -1) {
const nameNode = children.slice(forIdx + 1).find((c: any) =>
c.type === 'type_identifier' || c.type === 'identifier'
);
if (nameNode) {
return generateId('Impl', `${filePath}:${nameNode.text}`);
}
}
// Fall through: plain `impl Struct {}` — use first type_identifier below
}
const nameNode = current.childForFieldName?.('name')
?? current.children?.find((c: any) =>
c.type === 'type_identifier' || c.type === 'identifier' || c.type === 'name' || c.type === 'constant'
);
if (nameNode) {
const label = CONTAINER_TYPE_TO_LABEL[current.type] || 'Class';
return generateId(label, `${filePath}:${nameNode.text}`);
}
}
current = current.parent;
}
return null;
};
/**
* Extract function name and label from a function_definition or similar AST node.
* Handles C/C++ qualified_identifier (ClassName::MethodName) and other language patterns.
*/
export const extractFunctionName = (node: any): { funcName: string | null; label: string } => {
let funcName: string | null = null;
let label = 'Function';
// Swift init/deinit
if (node.type === 'init_declaration' || node.type === 'deinit_declaration') {
return {
funcName: node.type === 'init_declaration' ? 'init' : 'deinit',
label: 'Constructor',
};
}
if (FUNCTION_DECLARATION_TYPES.has(node.type)) {
// C/C++: function_definition -> [pointer_declarator ->] function_declarator -> qualified_identifier/identifier
// Unwrap pointer_declarator / reference_declarator wrappers to reach function_declarator
let declarator = node.childForFieldName?.('declarator') ||
node.children?.find((c: any) => c.type === 'function_declarator');
while (declarator && (declarator.type === 'pointer_declarator' || declarator.type === 'reference_declarator')) {
declarator = declarator.childForFieldName?.('declarator') ||
declarator.children?.find((c: any) =>
c.type === 'function_declarator' || c.type === 'pointer_declarator' || c.type === 'reference_declarator');
}
if (declarator) {
const innerDeclarator = declarator.childForFieldName?.('declarator') ||
declarator.children?.find((c: any) =>
c.type === 'qualified_identifier' || c.type === 'identifier' || c.type === 'parenthesized_declarator');
if (innerDeclarator?.type === 'qualified_identifier') {
const nameNode = innerDeclarator.childForFieldName?.('name') ||
innerDeclarator.children?.find((c: any) => c.type === 'identifier');
if (nameNode?.text) {
funcName = nameNode.text;
label = 'Method';
}
} else if (innerDeclarator?.type === 'identifier') {
funcName = innerDeclarator.text;
} else if (innerDeclarator?.type === 'parenthesized_declarator') {
const nestedId = innerDeclarator.children?.find((c: any) =>
c.type === 'qualified_identifier' || c.type === 'identifier');
if (nestedId?.type === 'qualified_identifier') {
const nameNode = nestedId.childForFieldName?.('name') ||
nestedId.children?.find((c: any) => c.type === 'identifier');
if (nameNode?.text) {
funcName = nameNode.text;
label = 'Method';
}
} else if (nestedId?.type === 'identifier') {
funcName = nestedId.text;
}
}
}
// Fallback for other languages (Kotlin uses simple_identifier, Swift uses simple_identifier)
if (!funcName) {
const nameNode = node.childForFieldName?.('name') ||
node.children?.find((c: any) => c.type === 'identifier' || c.type === 'property_identifier' || c.type === 'simple_identifier');
funcName = nameNode?.text;
}
} else if (node.type === 'impl_item') {
const funcItem = node.children?.find((c: any) => c.type === 'function_item');
if (funcItem) {
const nameNode = funcItem.childForFieldName?.('name') ||
funcItem.children?.find((c: any) => c.type === 'identifier');
funcName = nameNode?.text;
label = 'Method';
}
} else if (node.type === 'method_definition') {
const nameNode = node.childForFieldName?.('name') ||
node.children?.find((c: any) => c.type === 'property_identifier');
funcName = nameNode?.text;
label = 'Method';
} else if (node.type === 'method_declaration' || node.type === 'constructor_declaration') {
const nameNode = node.childForFieldName?.('name') ||
node.children?.find((c: any) => c.type === 'identifier');
funcName = nameNode?.text;
label = 'Method';
} else if (node.type === 'arrow_function' || node.type === 'function_expression') {
const parent = node.parent;
if (parent?.type === 'variable_declarator') {
const nameNode = parent.childForFieldName?.('name') ||
parent.children?.find((c: any) => c.type === 'identifier');
funcName = nameNode?.text;
}
} else if (node.type === 'method' || node.type === 'singleton_method') {
const nameNode = node.childForFieldName?.('name') ||
node.children?.find((c: any) => c.type === 'identifier');
funcName = nameNode?.text;
label = 'Method';
}
return { funcName, label };
};
/**
* Yield control to the event loop so spinners/progress can render.
* Call periodically in hot loops to prevent UI freezes.
@ -453,23 +174,6 @@ export const yieldToEventLoop = (): Promise<void> => new Promise(resolve => setI
/** Ruby extensionless filenames recognised as Ruby source */
const RUBY_EXTENSIONLESS_FILES = new Set(['Rakefile', 'Gemfile', 'Guardfile', 'Vagrantfile', 'Brewfile']);
/**
* Find a child of `childType` within a sibling node of `siblingType`.
* Used for Kotlin AST traversal where visibility_modifier lives inside a modifiers sibling.
*/
export const findSiblingChild = (parent: any, siblingType: string, childType: string): any | null => {
for (let i = 0; i < parent.childCount; i++) {
const sibling = parent.child(i);
if (sibling?.type === siblingType) {
for (let j = 0; j < sibling.childCount; j++) {
const child = sibling.child(j);
if (child?.type === childType) return child;
}
}
}
return null;
};
/**
* Map file extension to SupportedLanguage enum
*/
@ -519,389 +223,6 @@ export const getLanguageFromFilename = (filename: string): SupportedLanguages |
return null;
};
export interface MethodSignature {
parameterCount: number | undefined;
returnType: string | undefined;
}
const CALL_ARGUMENT_LIST_TYPES = new Set([
'arguments',
'argument_list',
'value_arguments',
]);
/**
* Extract parameter count and return type text from an AST method/function node.
* Works across languages by looking for common AST patterns.
*/
export const extractMethodSignature = (node: SyntaxNode | null | undefined): MethodSignature => {
let parameterCount: number | undefined = 0;
let returnType: string | undefined;
let isVariadic = false;
if (!node) return { parameterCount, returnType };
const paramListTypes = new Set([
'formal_parameters', 'parameters', 'parameter_list',
'function_parameters', 'method_parameters', 'function_value_parameters',
]);
// Node types that indicate variadic/rest parameters
const VARIADIC_PARAM_TYPES = new Set([
'variadic_parameter_declaration', // Go: ...string
'variadic_parameter', // Rust: extern "C" fn(...)
'spread_parameter', // Java: Object... args
'list_splat_pattern', // Python: *args
'dictionary_splat_pattern', // Python: **kwargs
]);
const findParameterList = (current: SyntaxNode): SyntaxNode | null => {
for (const child of current.children) {
if (paramListTypes.has(child.type)) return child;
}
for (const child of current.children) {
const nested = findParameterList(child);
if (nested) return nested;
}
return null;
};
const parameterList = (
paramListTypes.has(node.type) ? node // node itself IS the parameter list (e.g. C# primary constructors)
: node.childForFieldName?.('parameters')
?? findParameterList(node)
);
if (parameterList && paramListTypes.has(parameterList.type)) {
for (const param of parameterList.namedChildren) {
if (param.type === 'comment') continue;
if (param.text === 'self' || param.text === '&self' || param.text === '&mut self' ||
param.type === 'self_parameter') {
continue;
}
// Check for variadic parameter types
if (VARIADIC_PARAM_TYPES.has(param.type)) {
isVariadic = true;
continue;
}
// TypeScript/JavaScript: rest parameter — required_parameter containing rest_pattern
if (param.type === 'required_parameter' || param.type === 'optional_parameter') {
for (const child of param.children) {
if (child.type === 'rest_pattern') {
isVariadic = true;
break;
}
}
if (isVariadic) continue;
}
// Kotlin: vararg modifier on a regular parameter
if (param.type === 'parameter' || param.type === 'formal_parameter') {
const prev = param.previousSibling;
if (prev?.type === 'parameter_modifiers' && prev.text.includes('vararg')) {
isVariadic = true;
}
}
parameterCount++;
}
// C/C++: bare `...` token in parameter list (not a named child — check all children)
if (!isVariadic) {
for (const child of parameterList.children) {
if (!child.isNamed && child.text === '...') {
isVariadic = true;
break;
}
}
}
}
// Return type extraction — language-specific field names
// Go: 'result' field is either a type_identifier or parameter_list (multi-return)
const goResult = node.childForFieldName?.('result');
if (goResult) {
if (goResult.type === 'parameter_list') {
// Multi-return: extract first parameter's type only (e.g. (*User, error) → *User)
const firstParam = goResult.firstNamedChild;
if (firstParam?.type === 'parameter_declaration') {
const typeNode = firstParam.childForFieldName('type');
if (typeNode) returnType = typeNode.text;
} else if (firstParam) {
// Unnamed return types: (string, error) — first child is a bare type node
returnType = firstParam.text;
}
} else {
returnType = goResult.text;
}
}
// Rust: 'return_type' field — the value IS the type node (e.g. primitive_type, type_identifier).
// Skip if the node is a type_annotation (TS/Python), which is handled by the generic loop below.
if (!returnType) {
const rustReturn = node.childForFieldName?.('return_type');
if (rustReturn && rustReturn.type !== 'type_annotation') {
returnType = rustReturn.text;
}
}
// C/C++: 'type' field on function_definition
if (!returnType) {
const cppType = node.childForFieldName?.('type');
if (cppType && cppType.text !== 'void') {
returnType = cppType.text;
}
}
// C#: 'returns' field on method_declaration
if (!returnType) {
const csReturn = node.childForFieldName?.('returns');
if (csReturn && csReturn.text !== 'void') {
returnType = csReturn.text;
}
}
// TS/Rust/Python/C#/Kotlin: type_annotation or return_type child
if (!returnType) {
for (const child of node.children) {
if (child.type === 'type_annotation' || child.type === 'return_type') {
const typeNode = child.children.find((c) => c.isNamed);
if (typeNode) returnType = typeNode.text;
}
}
}
if (isVariadic) parameterCount = undefined;
return { parameterCount, returnType };
};
/**
* Count direct arguments for a call expression across common tree-sitter grammars.
* Returns undefined when the argument container cannot be located cheaply.
*/
export const countCallArguments = (callNode: SyntaxNode | null | undefined): number | undefined => {
if (!callNode) return undefined;
// Direct field or direct child (most languages)
let argsNode: SyntaxNode | null | undefined = callNode.childForFieldName('arguments')
?? callNode.children.find((child) => CALL_ARGUMENT_LIST_TYPES.has(child.type));
// Kotlin/Swift: call_expression → call_suffix → value_arguments
// Search one level deeper for languages that wrap arguments in a suffix node
if (!argsNode) {
for (const child of callNode.children) {
if (!child.isNamed) continue;
const nested = child.children.find((gc) => CALL_ARGUMENT_LIST_TYPES.has(gc.type));
if (nested) { argsNode = nested; break; }
}
}
if (!argsNode) return undefined;
let count = 0;
for (const child of argsNode.children) {
if (!child.isNamed) continue;
if (child.type === 'comment') continue;
count++;
}
return count;
};
// ── Call-form discrimination (Phase 1, Step D) ─────────────────────────
/**
* AST node types that indicate a member-access wrapper around the callee name.
* When nameNode.parent.type is one of these, the call is a member call.
*/
const MEMBER_ACCESS_NODE_TYPES = new Set([
'member_expression', // TS/JS: obj.method()
'attribute', // Python: obj.method()
'member_access_expression', // C#: obj.Method()
'field_expression', // Rust/C++: obj.method() / ptr->method()
'selector_expression', // Go: obj.Method()
'navigation_suffix', // Kotlin/Swift: obj.method() — nameNode sits inside navigation_suffix
'member_binding_expression', // C#: user?.Method() — null-conditional access
]);
/**
* Call node types that are inherently constructor invocations.
* Only includes patterns that the tree-sitter queries already capture as @call.
*/
const CONSTRUCTOR_CALL_NODE_TYPES = new Set([
'constructor_invocation', // Kotlin: Foo()
'new_expression', // TS/JS/C++: new Foo()
'object_creation_expression', // Java/C#/PHP: new Foo()
'implicit_object_creation_expression', // C# 9: User u = new(...)
'composite_literal', // Go: User{...}
'struct_expression', // Rust: User { ... }
]);
/**
* AST node types for scoped/qualified calls (e.g., Foo::new() in Rust, Foo::bar() in C++).
*/
const SCOPED_CALL_NODE_TYPES = new Set([
'scoped_identifier', // Rust: Foo::new()
'qualified_identifier', // C++: ns::func()
]);
type CallForm = 'free' | 'member' | 'constructor';
/**
* Infer whether a captured call site is a free call, member call, or constructor.
* Returns undefined if the form cannot be determined.
*
* Works by inspecting the AST structure between callNode (@call) and nameNode (@call.name).
* No tree-sitter query changes needed — the distinction is in the node types.
*/
export const inferCallForm = (
callNode: SyntaxNode,
nameNode: SyntaxNode,
): CallForm | undefined => {
// 1. Constructor: callNode itself is a constructor invocation (Kotlin)
if (CONSTRUCTOR_CALL_NODE_TYPES.has(callNode.type)) {
return 'constructor';
}
// 2. Member call: nameNode's parent is a member-access wrapper
const nameParent = nameNode.parent;
if (nameParent && MEMBER_ACCESS_NODE_TYPES.has(nameParent.type)) {
return 'member';
}
// 3. PHP: the callNode itself distinguishes member vs free calls
if (callNode.type === 'member_call_expression' || callNode.type === 'nullsafe_member_call_expression') {
return 'member';
}
if (callNode.type === 'scoped_call_expression') {
return 'member'; // static call Foo::bar()
}
// 4. Java method_invocation: member if it has an 'object' field
if (callNode.type === 'method_invocation' && callNode.childForFieldName('object')) {
return 'member';
}
// 4b. Ruby call with receiver: obj.method
if (callNode.type === 'call' && callNode.childForFieldName('receiver')) {
return 'member';
}
// 5. Scoped calls (Rust Foo::new(), C++ ns::func()): treat as free
// The receiver is a type, not an instance — handled differently in Phase 3
if (nameParent && SCOPED_CALL_NODE_TYPES.has(nameParent.type)) {
return 'free';
}
// 6. Default: if nameNode is a direct child of callNode, it's a free call
if (nameNode.parent === callNode || nameParent?.parent === callNode) {
return 'free';
}
return undefined;
};
/**
* Extract the receiver identifier for member calls.
* Only captures simple identifiers — returns undefined for complex expressions
* like getUser().save() or arr[0].method().
*/
const SIMPLE_RECEIVER_TYPES = new Set([
'identifier',
'simple_identifier',
'variable_name', // PHP $variable (tree-sitter-php)
'name', // PHP name node
'this', // TS/JS/Java/C# this.method()
'self', // Rust/Python self.method()
'super', // TS/JS/Java/Kotlin/Ruby super.method()
'super_expression', // Kotlin wraps super in super_expression
'base', // C# base.Method()
'parent', // PHP parent::method()
'constant', // Ruby CONSTANT.method() (uppercase identifiers)
]);
export const extractReceiverName = (
nameNode: SyntaxNode,
): string | undefined => {
const parent = nameNode.parent;
if (!parent) return undefined;
// PHP: member_call_expression / nullsafe_member_call_expression — receiver is on the callNode
// Java: method_invocation — receiver is the 'object' field on callNode
// For these, parent of nameNode is the call itself, so check the call's object field
const callNode = parent.parent ?? parent;
let receiver: SyntaxNode | null = null;
// Try standard field names used across grammars
receiver = parent.childForFieldName('object') // TS/JS member_expression, Python attribute, PHP, Java
?? parent.childForFieldName('value') // Rust field_expression
?? parent.childForFieldName('operand') // Go selector_expression
?? parent.childForFieldName('expression') // C# member_access_expression
?? parent.childForFieldName('argument'); // C++ field_expression
// Java method_invocation: 'object' field is on the callNode, not on nameNode's parent
if (!receiver && callNode.type === 'method_invocation') {
receiver = callNode.childForFieldName('object');
}
// PHP: member_call_expression has 'object' on the call node
if (!receiver && (callNode.type === 'member_call_expression' || callNode.type === 'nullsafe_member_call_expression')) {
receiver = callNode.childForFieldName('object');
}
// Ruby: call node has 'receiver' field
if (!receiver && parent.type === 'call') {
receiver = parent.childForFieldName('receiver');
}
// PHP scoped_call_expression (parent::method(), self::method()):
// nameNode's direct parent IS the scoped_call_expression (name is a direct child)
if (!receiver && (parent.type === 'scoped_call_expression' || callNode.type === 'scoped_call_expression')) {
const scopedCall = parent.type === 'scoped_call_expression' ? parent : callNode;
receiver = scopedCall.childForFieldName('scope');
// relative_scope wraps 'parent'/'self'/'static' — unwrap to get the keyword
if (receiver?.type === 'relative_scope') {
receiver = receiver.firstChild;
}
}
// C# null-conditional: user?.Save() → conditional_access_expression wraps member_binding_expression
if (!receiver && parent.type === 'member_binding_expression') {
const condAccess = parent.parent;
if (condAccess?.type === 'conditional_access_expression') {
receiver = condAccess.firstNamedChild;
}
}
// Kotlin/Swift: navigation_expression target is the first child
if (!receiver && parent.type === 'navigation_suffix') {
const navExpr = parent.parent;
if (navExpr?.type === 'navigation_expression') {
// First named child is the target (receiver)
for (const child of navExpr.children) {
if (child.isNamed && child !== parent) {
receiver = child;
break;
}
}
}
}
if (!receiver) return undefined;
// Only capture simple identifiers — refuse complex expressions
if (SIMPLE_RECEIVER_TYPES.has(receiver.type)) {
return receiver.text;
}
// Python super().method(): receiver is a call node `super()` — extract the function name
if (receiver.type === 'call') {
const func = receiver.childForFieldName('function');
if (func?.text === 'super') return 'super';
}
return undefined;
};
export const isVerboseIngestionEnabled = (): boolean => {
const raw = process.env.GITNEXUS_VERBOSE;
if (!raw) return false;
@ -909,6 +230,6 @@ export const isVerboseIngestionEnabled = (): boolean => {
return value === '1' || value === 'true' || value === 'yes';
};
// Re-exports for backward compatibility
export * from './ast-helpers.js';
export * from './call-analysis.js';

View file

@ -9,7 +9,6 @@ import CPP from 'tree-sitter-cpp';
import CSharp from 'tree-sitter-c-sharp';
import Go from 'tree-sitter-go';
import Rust from 'tree-sitter-rust';
import Kotlin from 'tree-sitter-kotlin';
import PHP from 'tree-sitter-php';
import Ruby from 'tree-sitter-ruby';
import { createRequire } from 'node:module';
@ -21,17 +20,25 @@ import { getTreeSitterBufferSize, TREE_SITTER_MAX_BUFFER } from '../constants.js
const _require = createRequire(import.meta.url);
let Swift: any = null;
try { Swift = _require('tree-sitter-swift'); } catch {}
import {
// tree-sitter-kotlin is an optionalDependency — may not be installed
let Kotlin: any = null;
try { Kotlin = _require('tree-sitter-kotlin'); } catch {}
import {
getLanguageFromFilename,
FUNCTION_NODE_TYPES,
extractFunctionName,
isBuiltInOrNoise,
getDefinitionNodeFromCaptures,
findEnclosingClassId,
getLabelFromCaptures,
extractMethodSignature,
countCallArguments,
inferCallForm,
extractReceiverName
extractReceiverName,
extractReceiverNode,
extractMixedChain,
type MixedChainStep,
} from '../utils.js';
import { buildTypeEnv } from '../type-env.js';
import type { ConstructorBinding } from '../type-env.js';
@ -39,9 +46,11 @@ import { isNodeExported } from '../export-detection.js';
import { detectFrameworkFromAST } from '../framework-detection.js';
import { typeConfigs } from '../type-extractors/index.js';
import { generateId } from '../../../lib/utils.js';
import { extractNamedBindings } from '../named-binding-extraction.js';
import { appendKotlinWildcard } from '../resolvers/index.js';
import { namedBindingExtractors, preprocessImportPath } from '../import-resolution.js';
import type { NamedBinding } from '../import-resolution.js';
import { callRouters } from '../call-routing.js';
import { extractPropertyDeclaredType } from '../type-extractors/shared.js';
import type { NodeLabel } from '../../graph/types.js';
// ============================================================================
// Types for serializable results
@ -61,6 +70,7 @@ interface ParsedNode {
astFrameworkReason?: string;
description?: string;
parameterCount?: number;
requiredParameterCount?: number;
returnType?: string;
};
}
@ -69,7 +79,7 @@ interface ParsedRelationship {
id: string;
sourceId: string;
targetId: string;
type: 'DEFINES' | 'HAS_METHOD';
type: 'DEFINES' | 'HAS_METHOD' | 'HAS_PROPERTY';
confidence: number;
reason: string;
}
@ -78,9 +88,12 @@ interface ParsedSymbol {
filePath: string;
name: string;
nodeId: string;
type: string;
type: NodeLabel;
parameterCount?: number;
requiredParameterCount?: number;
parameterTypes?: string[];
returnType?: string;
declaredType?: string;
ownerId?: string;
}
@ -89,7 +102,7 @@ export interface ExtractedImport {
rawImportPath: string;
language: SupportedLanguages;
/** Named bindings from the import (e.g., import {User as U} → [{local:'U', exported:'User'}]) */
namedBindings?: { local: string; exported: string }[];
namedBindings?: NamedBinding[];
}
export interface ExtractedCall {
@ -104,6 +117,27 @@ export interface ExtractedCall {
receiverName?: string;
/** Resolved type name of the receiver (e.g., 'User' for user.save() when user: User) */
receiverTypeName?: string;
/**
* Unified mixed chain when the receiver is a chain of field accesses and/or method calls.
* Steps are ordered base-first (innermost to outermost). Examples:
* `svc.getUser().save()` → chain=[{kind:'call',name:'getUser'}], receiverName='svc'
* `user.address.save()` → chain=[{kind:'field',name:'address'}], receiverName='user'
* `svc.getUser().address.save()` → chain=[{kind:'call',name:'getUser'},{kind:'field',name:'address'}]
* Length is capped at MAX_CHAIN_DEPTH (3).
*/
receiverMixedChain?: MixedChainStep[];
}
export interface ExtractedAssignment {
filePath: string;
/** generateId of enclosing function, or generateId('File', filePath) for top-level */
sourceId: string;
/** Receiver text (e.g., 'user' from user.address = value) */
receiverText: string;
/** Property name being written (e.g., 'address') */
propertyName: string;
/** Resolved type name of the receiver if available from TypeEnv */
receiverTypeName?: string;
}
export interface ExtractedHeritage {
@ -131,15 +165,26 @@ export interface FileConstructorBindings {
bindings: ConstructorBinding[];
}
/** File-scope type bindings from TypeEnv fixpoint — used for cross-file ExportedTypeMap. */
export interface FileTypeEnvBindings {
filePath: string;
/** [varName, typeName] pairs from file scope (scope = '') */
bindings: [string, string][];
}
export interface ParseWorkerResult {
nodes: ParsedNode[];
relationships: ParsedRelationship[];
symbols: ParsedSymbol[];
imports: ExtractedImport[];
calls: ExtractedCall[];
assignments: ExtractedAssignment[];
heritage: ExtractedHeritage[];
routes: ExtractedRoute[];
constructorBindings: FileConstructorBindings[];
/** File-scope type bindings from TypeEnv fixpoint for exported symbol collection. */
typeEnvBindings: FileTypeEnvBindings[];
skippedLanguages: Record<string, number>;
fileCount: number;
}
@ -165,12 +210,25 @@ const languageMap: Record<string, any> = {
[SupportedLanguages.CSharp]: CSharp,
[SupportedLanguages.Go]: Go,
[SupportedLanguages.Rust]: Rust,
[SupportedLanguages.Kotlin]: Kotlin,
...(Kotlin ? { [SupportedLanguages.Kotlin]: Kotlin } : {}),
[SupportedLanguages.PHP]: PHP.php_only,
[SupportedLanguages.Ruby]: Ruby,
...(Swift ? { [SupportedLanguages.Swift]: Swift } : {}),
};
/**
* Check if a language grammar is available in this worker.
* Duplicated from parser-loader.ts because workers can't import from the main thread.
* Extra filePath parameter needed to distinguish .tsx from .ts (different grammars
* under the same SupportedLanguages.TypeScript key).
*/
const isLanguageAvailable = (language: SupportedLanguages, filePath: string): boolean => {
const key = language === SupportedLanguages.TypeScript && filePath.endsWith('.tsx')
? `${language}:tsx`
: language;
return key in languageMap && languageMap[key] != null;
};
const setLanguage = (language: SupportedLanguages, filePath: string): void => {
const key = language === SupportedLanguages.TypeScript && filePath.endsWith('.tsx')
? `${language}:tsx`
@ -201,39 +259,7 @@ const findEnclosingFunctionId = (node: any, filePath: string): string | null =>
return null;
};
// ============================================================================
// Label detection from capture map
// ============================================================================
const getLabelFromCaptures = (captureMap: Record<string, any>): string | null => {
// Skip imports (handled separately) and calls
if (captureMap['import'] || captureMap['call']) return null;
if (!captureMap['name']) return null;
if (captureMap['definition.function']) return 'Function';
if (captureMap['definition.class']) return 'Class';
if (captureMap['definition.interface']) return 'Interface';
if (captureMap['definition.method']) return 'Method';
if (captureMap['definition.struct']) return 'Struct';
if (captureMap['definition.enum']) return 'Enum';
if (captureMap['definition.namespace']) return 'Namespace';
if (captureMap['definition.module']) return 'Module';
if (captureMap['definition.trait']) return 'Trait';
if (captureMap['definition.impl']) return 'Impl';
if (captureMap['definition.type']) return 'TypeAlias';
if (captureMap['definition.const']) return 'Const';
if (captureMap['definition.static']) return 'Static';
if (captureMap['definition.typedef']) return 'Typedef';
if (captureMap['definition.macro']) return 'Macro';
if (captureMap['definition.union']) return 'Union';
if (captureMap['definition.property']) return 'Property';
if (captureMap['definition.record']) return 'Record';
if (captureMap['definition.delegate']) return 'Delegate';
if (captureMap['definition.annotation']) return 'Annotation';
if (captureMap['definition.constructor']) return 'Constructor';
if (captureMap['definition.template']) return 'Template';
return 'CodeElement';
};
// Label detection moved to shared getLabelFromCaptures in utils.ts
// DEFINITION_CAPTURE_KEYS and getDefinitionNodeFromCaptures imported from ../utils.js
@ -249,9 +275,12 @@ const processBatch = (files: ParseWorkerInput[], onProgress?: (filesProcessed: n
symbols: [],
imports: [],
calls: [],
assignments: [],
heritage: [],
routes: [],
constructorBindings: [],
typeEnvBindings: [],
skippedLanguages: {},
fileCount: 0,
};
@ -302,21 +331,29 @@ const processBatch = (files: ParseWorkerInput[], onProgress?: (filesProcessed: n
// Process regular files for this language
if (regularFiles.length > 0) {
try {
setLanguage(language, regularFiles[0].path);
processFileGroup(regularFiles, language, queryString, result, onFileProcessed);
} catch {
// parser unavailable — skip this language group
if (isLanguageAvailable(language, regularFiles[0].path)) {
try {
setLanguage(language, regularFiles[0].path);
processFileGroup(regularFiles, language, queryString, result, onFileProcessed);
} catch {
// parser unavailable — skip this language group
}
} else {
result.skippedLanguages[language] = (result.skippedLanguages[language] || 0) + regularFiles.length;
}
}
// Process tsx files separately (different grammar)
if (tsxFiles.length > 0) {
try {
setLanguage(language, tsxFiles[0].path);
processFileGroup(tsxFiles, language, queryString, result, onFileProcessed);
} catch {
// parser unavailable — skip this language group
if (isLanguageAvailable(language, tsxFiles[0].path)) {
try {
setLanguage(language, tsxFiles[0].path);
processFileGroup(tsxFiles, language, queryString, result, onFileProcessed);
} catch {
// parser unavailable — skip this language group
}
} else {
result.skippedLanguages[language] = (result.skippedLanguages[language] || 0) + tsxFiles.length;
}
}
}
@ -835,15 +872,6 @@ const processFileGroup = (
result.fileCount++;
onFileProcessed?.();
// Build per-file type environment + constructor bindings in a single AST walk.
// Constructor bindings are verified against the SymbolTable in processCallsFromExtracted.
const typeEnv = buildTypeEnv(tree, language);
const callRouter = callRouters[language];
if (typeEnv.constructorBindings.length > 0) {
result.constructorBindings.push({ filePath: file.path, bindings: [...typeEnv.constructorBindings] });
}
let matches;
try {
matches = query.matches(tree.rootNode);
@ -852,6 +880,49 @@ const processFileGroup = (
continue;
}
// Pre-pass: extract heritage from query matches to build parentMap for buildTypeEnv.
// Heritage edges (EXTENDS/IMPLEMENTS) are created by heritage-processor which runs
// in PARALLEL with call-processor, so the graph edges don't exist when buildTypeEnv
// runs. This pre-pass makes parent class information available for type resolution.
const fileParentMap = new Map<string, string[]>();
for (const match of matches) {
const captureMap: Record<string, any> = {};
for (const c of match.captures) {
captureMap[c.name] = c.node;
}
if (captureMap['heritage.class'] && captureMap['heritage.extends']) {
const className: string = captureMap['heritage.class'].text;
const parentName: string = captureMap['heritage.extends'].text;
// Skip Go named fields (only anonymous fields are struct embedding)
const extendsNode = captureMap['heritage.extends'];
const fieldDecl = extendsNode.parent;
if (fieldDecl?.type === 'field_declaration' && fieldDecl.childForFieldName('name')) continue;
let parents = fileParentMap.get(className);
if (!parents) { parents = []; fileParentMap.set(className, parents); }
if (!parents.includes(parentName)) parents.push(parentName);
}
}
// Build per-file type environment + constructor bindings in a single AST walk.
// Constructor bindings are verified against the SymbolTable in processCallsFromExtracted.
const parentMap: ReadonlyMap<string, readonly string[]> = fileParentMap;
const typeEnv = buildTypeEnv(tree, language, { parentMap });
const callRouter = callRouters[language];
if (typeEnv.constructorBindings.length > 0) {
result.constructorBindings.push({ filePath: file.path, bindings: [...typeEnv.constructorBindings] });
}
// Extract file-scope bindings for ExportedTypeMap (closes worker/sequential quality gap).
// Sequential path uses collectExportedBindings(typeEnv) directly; worker path serializes
// these bindings so the main thread can merge them into ExportedTypeMap.
const fileScope = typeEnv.env.get('');
if (fileScope && fileScope.size > 0) {
const bindings: [string, string][] = [];
for (const [name, type] of fileScope) bindings.push([name, type]);
result.typeEnvBindings.push({ filePath: file.path, bindings });
}
for (const match of matches) {
const captureMap: Record<string, any> = {};
for (const c of match.captures) {
@ -860,10 +931,10 @@ const processFileGroup = (
// Extract import paths before skipping
if (captureMap['import'] && captureMap['import.source']) {
const rawImportPath = language === SupportedLanguages.Kotlin
? appendKotlinWildcard(captureMap['import.source'].text.replace(/['"<>]/g, ''), captureMap['import'])
: captureMap['import.source'].text.replace(/['"<>]/g, '');
const namedBindings = extractNamedBindings(captureMap['import'], language);
const rawImportPath = preprocessImportPath(captureMap['import.source'].text, captureMap['import'], language);
if (!rawImportPath) continue;
const extractor = namedBindingExtractors[language];
const namedBindings = extractor ? extractor(captureMap['import']) : undefined;
result.imports.push({
filePath: file.path,
rawImportPath,
@ -873,6 +944,28 @@ const processFileGroup = (
continue;
}
// Extract assignment sites (field write access)
if (captureMap['assignment'] && captureMap['assignment.receiver'] && captureMap['assignment.property']) {
const receiverText = captureMap['assignment.receiver'].text;
const propertyName = captureMap['assignment.property'].text;
if (receiverText && propertyName) {
const srcId = findEnclosingFunctionId(captureMap['assignment'], file.path)
|| generateId('File', file.path);
let receiverTypeName: string | undefined;
if (typeEnv) {
receiverTypeName = typeEnv.lookup(receiverText, captureMap['assignment']) ?? undefined;
}
result.assignments.push({
filePath: file.path,
sourceId: srcId,
receiverText,
propertyName,
...(receiverTypeName ? { receiverTypeName } : {}),
});
}
if (!captureMap['call']) continue;
}
// Extract call sites
if (captureMap['call']) {
const callNameNode = captureMap['call.name'];
@ -928,6 +1021,7 @@ const processFileGroup = (
nodeId,
type: 'Property',
...(propEnclosingClassId ? { ownerId: propEnclosingClassId } : {}),
...(item.declaredType ? { declaredType: item.declaredType } : {}),
});
const fileId = generateId('File', file.path);
const relId = generateId('DEFINES', `${fileId}->${nodeId}`);
@ -941,10 +1035,10 @@ const processFileGroup = (
});
if (propEnclosingClassId) {
result.relationships.push({
id: generateId('HAS_METHOD', `${propEnclosingClassId}->${nodeId}`),
id: generateId('HAS_PROPERTY', `${propEnclosingClassId}->${nodeId}`),
sourceId: propEnclosingClassId,
targetId: nodeId,
type: 'HAS_METHOD',
type: 'HAS_PROPERTY',
confidence: 1.0,
reason: '',
});
@ -961,8 +1055,29 @@ const processFileGroup = (
const sourceId = findEnclosingFunctionId(callNode, file.path)
|| generateId('File', file.path);
const callForm = inferCallForm(callNode, callNameNode);
const receiverName = callForm === 'member' ? extractReceiverName(callNameNode) : undefined;
const receiverTypeName = receiverName ? typeEnv.lookup(receiverName, callNode) : undefined;
let receiverName = callForm === 'member' ? extractReceiverName(callNameNode) : undefined;
let receiverTypeName = receiverName ? typeEnv.lookup(receiverName, callNode) : undefined;
let receiverMixedChain: MixedChainStep[] | undefined;
// When the receiver is a complex expression (call chain, field chain, or mixed),
// extractReceiverName returns undefined. Walk the receiver node to build a unified
// mixed chain for deferred resolution in processCallsFromExtracted.
if (callForm === 'member' && receiverName === undefined && !receiverTypeName) {
const receiverNode = extractReceiverNode(callNameNode);
if (receiverNode) {
const extracted = extractMixedChain(receiverNode);
if (extracted && extracted.chain.length > 0) {
receiverMixedChain = extracted.chain;
receiverName = extracted.baseReceiverName;
// Try the type environment immediately for the base receiver
// (covers explicitly-typed locals and annotated parameters).
if (receiverName) {
receiverTypeName = typeEnv.lookup(receiverName, callNode);
}
}
}
}
result.calls.push({
filePath: file.path,
calledName,
@ -971,6 +1086,7 @@ const processFileGroup = (
...(callForm !== undefined ? { callForm } : {}),
...(receiverName !== undefined ? { receiverName } : {}),
...(receiverTypeName !== undefined ? { receiverTypeName } : {}),
...(receiverMixedChain !== undefined ? { receiverMixedChain } : {}),
});
}
}
@ -1017,7 +1133,7 @@ const processFileGroup = (
}
}
const nodeLabel = getLabelFromCaptures(captureMap);
const nodeLabel = getLabelFromCaptures(captureMap, language);
if (!nodeLabel) continue;
const nameNode = captureMap['name'];
@ -1042,19 +1158,30 @@ const processFileGroup = (
: null;
let parameterCount: number | undefined;
let requiredParameterCount: number | undefined;
let parameterTypes: string[] | undefined;
let returnType: string | undefined;
let declaredType: string | undefined;
if (nodeLabel === 'Function' || nodeLabel === 'Method' || nodeLabel === 'Constructor') {
const sig = extractMethodSignature(definitionNode);
parameterCount = sig.parameterCount;
requiredParameterCount = sig.requiredParameterCount;
parameterTypes = sig.parameterTypes;
returnType = sig.returnType;
// Language-specific return type fallback (e.g. Ruby YARD @return [Type])
if (!returnType && definitionNode) {
// Also upgrades uninformative AST types like PHP `array` with PHPDoc `@return User[]`
if ((!returnType || returnType === 'array' || returnType === 'iterable') && definitionNode) {
const tc = typeConfigs[language as keyof typeof typeConfigs];
if (tc?.extractReturnType) {
returnType = tc.extractReturnType(definitionNode);
const docReturn = tc.extractReturnType(definitionNode);
if (docReturn) returnType = docReturn;
}
}
} else if (nodeLabel === 'Property' && definitionNode) {
// Extract the declared type for property/field nodes.
// Walk the definition node for type annotation children.
declaredType = extractPropertyDeclaredType(definitionNode);
}
result.nodes.push({
@ -1073,6 +1200,8 @@ const processFileGroup = (
} : {}),
...(description !== undefined ? { description } : {}),
...(parameterCount !== undefined ? { parameterCount } : {}),
...(requiredParameterCount !== undefined ? { requiredParameterCount } : {}),
...(parameterTypes !== undefined ? { parameterTypes } : {}),
...(returnType !== undefined ? { returnType } : {}),
},
});
@ -1088,7 +1217,10 @@ const processFileGroup = (
nodeId,
type: nodeLabel,
...(parameterCount !== undefined ? { parameterCount } : {}),
...(requiredParameterCount !== undefined ? { requiredParameterCount } : {}),
...(parameterTypes !== undefined ? { parameterTypes } : {}),
...(returnType !== undefined ? { returnType } : {}),
...(declaredType !== undefined ? { declaredType } : {}),
...(enclosingClassId ? { ownerId: enclosingClassId } : {}),
});
@ -1103,13 +1235,14 @@ const processFileGroup = (
reason: '',
});
// ── HAS_METHOD: link method/constructor/property to enclosing class ──
// ── HAS_METHOD / HAS_PROPERTY: link member to enclosing class ──
if (enclosingClassId) {
const memberEdgeType = nodeLabel === 'Property' ? 'HAS_PROPERTY' : 'HAS_METHOD';
result.relationships.push({
id: generateId('HAS_METHOD', `${enclosingClassId}->${nodeId}`),
id: generateId(memberEdgeType, `${enclosingClassId}->${nodeId}`),
sourceId: enclosingClassId,
targetId: nodeId,
type: 'HAS_METHOD',
type: memberEdgeType,
confidence: 1.0,
reason: '',
});
@ -1131,7 +1264,7 @@ const processFileGroup = (
/** Accumulated result across sub-batches */
let accumulated: ParseWorkerResult = {
nodes: [], relationships: [], symbols: [],
imports: [], calls: [], heritage: [], routes: [], constructorBindings: [], fileCount: 0,
imports: [], calls: [], assignments: [], heritage: [], routes: [], constructorBindings: [], typeEnvBindings: [], skippedLanguages: {}, fileCount: 0,
};
let cumulativeProcessed = 0;
@ -1141,9 +1274,14 @@ const mergeResult = (target: ParseWorkerResult, src: ParseWorkerResult) => {
target.symbols.push(...src.symbols);
target.imports.push(...src.imports);
target.calls.push(...src.calls);
target.assignments.push(...src.assignments);
target.heritage.push(...src.heritage);
target.routes.push(...src.routes);
target.constructorBindings.push(...src.constructorBindings);
target.typeEnvBindings.push(...src.typeEnvBindings);
for (const [lang, count] of Object.entries(src.skippedLanguages)) {
target.skippedLanguages[lang] = (target.skippedLanguages[lang] || 0) + count;
}
target.fileCount += src.fileCount;
};
@ -1165,7 +1303,7 @@ parentPort!.on('message', (msg: any) => {
if (msg && msg.type === 'flush') {
parentPort!.postMessage({ type: 'result', data: accumulated });
// Reset for potential reuse
accumulated = { nodes: [], relationships: [], symbols: [], imports: [], calls: [], heritage: [], routes: [], constructorBindings: [], fileCount: 0 };
accumulated = { nodes: [], relationships: [], symbols: [], imports: [], calls: [], assignments: [], heritage: [], routes: [], constructorBindings: [], typeEnvBindings: [], skippedLanguages: {}, fileCount: 0 };
cumulativeProcessed = 0;
return;
}

View file

@ -238,6 +238,9 @@ export const streamAllCSVsToDisk = async (
const communityWriter = new BufferedCSVWriter(path.join(csvDir, 'community.csv'), 'id,label,heuristicLabel,keywords,description,enrichedBy,cohesion,symbolCount');
const processWriter = new BufferedCSVWriter(path.join(csvDir, 'process.csv'), 'id,label,heuristicLabel,processType,stepCount,communities,entryPointId,terminalId');
// Section nodes have an extra 'level' column
const sectionWriter = new BufferedCSVWriter(path.join(csvDir, 'section.csv'), 'id,name,filePath,startLine,endLine,level,content,description');
// Multi-language node types share the same CSV shape (no isExported column)
const multiLangHeader = 'id,name,filePath,startLine,endLine,content,description';
const MULTI_LANG_TYPES = ['Struct', 'Enum', 'Macro', 'Typedef', 'Union', 'Namespace', 'Trait', 'Impl',
@ -324,6 +327,20 @@ export const streamAllCSVsToDisk = async (
].join(','));
break;
}
case 'Section': {
const content = await extractContent(node, contentCache);
await sectionWriter.addRow([
escapeCSVField(node.id),
escapeCSVField(node.properties.name || ''),
escapeCSVField(node.properties.filePath || ''),
escapeCSVNumber(node.properties.startLine, -1),
escapeCSVNumber(node.properties.endLine, -1),
escapeCSVNumber((node.properties as any).level, 1),
escapeCSVField(content),
escapeCSVField((node.properties as any).description || ''),
].join(','));
break;
}
default: {
// Code element nodes (Function, Class, Interface, CodeElement)
const writer = codeWriterMap[node.label];
@ -361,7 +378,7 @@ export const streamAllCSVsToDisk = async (
}
// Finish all node writers
const allWriters = [fileWriter, folderWriter, functionWriter, classWriter, interfaceWriter, methodWriter, codeElemWriter, communityWriter, processWriter, ...multiLangWriters.values()];
const allWriters = [fileWriter, folderWriter, functionWriter, classWriter, interfaceWriter, methodWriter, codeElemWriter, communityWriter, processWriter, sectionWriter, ...multiLangWriters.values()];
await Promise.all(allWriters.map(w => w.finish()));
// --- Stream relationship CSV ---
@ -387,6 +404,7 @@ export const streamAllCSVsToDisk = async (
['Interface', interfaceWriter], ['Method', methodWriter],
['CodeElement', codeElemWriter],
['Community', communityWriter], ['Process', processWriter],
['Section' as NodeTableName, sectionWriter],
...Array.from(multiLangWriters.entries()).map(([name, w]) => [name as NodeTableName, w] as [NodeTableName, BufferedCSVWriter]),
];
for (const [name, writer] of tableMap) {

View file

@ -18,6 +18,9 @@ let conn: lbug.Connection | null = null;
let currentDbPath: string | null = null;
let ftsLoaded = false;
/** Expose the current Database for pool adapter reuse in tests. */
export const getDatabase = (): lbug.Database | null => db;
// Global session lock for operations that touch module-level lbug globals.
// This guarantees no DB switch can happen while an operation is running.
let sessionLock: Promise<void> = Promise.resolve();
@ -333,6 +336,9 @@ const getCopyQuery = (table: NodeTableName, filePath: string): string => {
if (table === 'Process') {
return `COPY ${t}(id, label, heuristicLabel, processType, stepCount, communities, entryPointId, terminalId) FROM "${filePath}" ${COPY_CSV_OPTS}`;
}
if (table === 'Section') {
return `COPY ${t}(id, name, filePath, startLine, endLine, level, content, description) FROM "${filePath}" ${COPY_CSV_OPTS}`;
}
if (table === 'Method') {
return `COPY ${t}(id, name, filePath, startLine, endLine, isExported, content, description, parameterCount, returnType) FROM "${filePath}" ${COPY_CSV_OPTS}`;
}
@ -377,6 +383,9 @@ export const insertNodeToLbug = async (
query = `CREATE (n:File {id: ${escapeValue(properties.id)}, name: ${escapeValue(properties.name)}, filePath: ${escapeValue(properties.filePath)}, content: ${escapeValue(properties.content || '')}})`;
} else if (label === 'Folder') {
query = `CREATE (n:Folder {id: ${escapeValue(properties.id)}, name: ${escapeValue(properties.name)}, filePath: ${escapeValue(properties.filePath)}})`;
} else if (label === 'Section') {
const descPart = properties.description ? `, description: ${escapeValue(properties.description)}` : '';
query = `CREATE (n:Section {id: ${escapeValue(properties.id)}, name: ${escapeValue(properties.name)}, filePath: ${escapeValue(properties.filePath)}, startLine: ${properties.startLine || 0}, endLine: ${properties.endLine || 0}, level: ${properties.level || 1}, content: ${escapeValue(properties.content || '')}${descPart}})`;
} else if (TABLES_WITH_EXPORTED.has(label)) {
const descPart = properties.description ? `, description: ${escapeValue(properties.description)}` : '';
query = `CREATE (n:${t} {id: ${escapeValue(properties.id)}, name: ${escapeValue(properties.name)}, filePath: ${escapeValue(properties.filePath)}, startLine: ${properties.startLine || 0}, endLine: ${properties.endLine || 0}, isExported: ${!!properties.isExported}, content: ${escapeValue(properties.content || '')}${descPart}})`;
@ -448,6 +457,9 @@ export const batchInsertNodesToLbug = async (
query = `MERGE (n:File {id: ${escapeValue(properties.id)}}) SET n.name = ${escapeValue(properties.name)}, n.filePath = ${escapeValue(properties.filePath)}, n.content = ${escapeValue(properties.content || '')}`;
} else if (label === 'Folder') {
query = `MERGE (n:Folder {id: ${escapeValue(properties.id)}}) SET n.name = ${escapeValue(properties.name)}, n.filePath = ${escapeValue(properties.filePath)}`;
} else if (label === 'Section') {
const descPart = properties.description ? `, n.description = ${escapeValue(properties.description)}` : '';
query = `MERGE (n:Section {id: ${escapeValue(properties.id)}}) SET n.name = ${escapeValue(properties.name)}, n.filePath = ${escapeValue(properties.filePath)}, n.startLine = ${properties.startLine || 0}, n.endLine = ${properties.endLine || 0}, n.level = ${properties.level || 1}, n.content = ${escapeValue(properties.content || '')}${descPart}`;
} else if (TABLES_WITH_EXPORTED.has(label)) {
const descPart = properties.description ? `, n.description = ${escapeValue(properties.description)}` : '';
query = `MERGE (n:${t} {id: ${escapeValue(properties.id)}}) SET n.name = ${escapeValue(properties.name)}, n.filePath = ${escapeValue(properties.filePath)}, n.startLine = ${properties.startLine || 0}, n.endLine = ${properties.endLine || 0}, n.isExported = ${!!properties.isExported}, n.content = ${escapeValue(properties.content || '')}${descPart}`;

View file

@ -13,7 +13,7 @@
// NODE TABLE NAMES
// ============================================================================
export const NODE_TABLES = [
'File', 'Folder', 'Function', 'Class', 'Interface', 'Method', 'CodeElement', 'Community', 'Process',
'File', 'Folder', 'Function', 'Class', 'Interface', 'Method', 'CodeElement', 'Community', 'Process', 'Section',
// Multi-language support
'Struct', 'Enum', 'Macro', 'Typedef', 'Union', 'Namespace', 'Trait', 'Impl',
'TypeAlias', 'Const', 'Static', 'Property', 'Record', 'Delegate', 'Annotation', 'Constructor', 'Template', 'Module'
@ -26,7 +26,7 @@ export type NodeTableName = typeof NODE_TABLES[number];
export const REL_TABLE_NAME = 'CodeRelation';
// Valid relation types
export const REL_TYPES = ['CONTAINS', 'DEFINES', 'IMPORTS', 'CALLS', 'EXTENDS', 'IMPLEMENTS', 'HAS_METHOD', 'OVERRIDES', 'MEMBER_OF', 'STEP_IN_PROCESS'] as const;
export const REL_TYPES = ['CONTAINS', 'DEFINES', 'IMPORTS', 'CALLS', 'EXTENDS', 'IMPLEMENTS', 'HAS_METHOD', 'HAS_PROPERTY', 'ACCESSES', 'OVERRIDES', 'MEMBER_OF', 'STEP_IN_PROCESS'] as const;
export type RelType = typeof REL_TYPES[number];
// ============================================================================
@ -192,6 +192,19 @@ export const ANNOTATION_SCHEMA = CODE_ELEMENT_BASE('Annotation');
export const CONSTRUCTOR_SCHEMA = CODE_ELEMENT_BASE('Constructor');
export const TEMPLATE_SCHEMA = CODE_ELEMENT_BASE('Template');
export const MODULE_SCHEMA = CODE_ELEMENT_BASE('Module');
// Markdown heading sections
export const SECTION_SCHEMA = `
CREATE NODE TABLE Section (
id STRING,
name STRING,
filePath STRING,
startLine INT64,
endLine INT64,
level INT64,
content STRING,
description STRING,
PRIMARY KEY (id)
)`;
// ============================================================================
// RELATION TABLE SCHEMA
@ -225,6 +238,7 @@ CREATE REL TABLE ${REL_TABLE_NAME} (
FROM File TO \`Constructor\`,
FROM File TO \`Template\`,
FROM File TO \`Module\`,
FROM File TO Section,
FROM Folder TO Folder,
FROM Folder TO File,
FROM Function TO Function,
@ -289,6 +303,8 @@ CREATE REL TABLE ${REL_TABLE_NAME} (
FROM \`Template\` TO Interface,
FROM \`Template\` TO \`Constructor\`,
FROM \`Module\` TO \`Module\`,
FROM Section TO Section,
FROM Section TO File,
FROM CodeElement TO Community,
FROM Interface TO Community,
FROM Interface TO Function,
@ -447,6 +463,8 @@ export const NODE_SCHEMA_QUERIES = [
CONSTRUCTOR_SCHEMA,
TEMPLATE_SCHEMA,
MODULE_SCHEMA,
// Markdown support
SECTION_SCHEMA,
];
export const REL_SCHEMA_QUERIES = [

View file

@ -8,7 +8,6 @@ import CPP from 'tree-sitter-cpp';
import CSharp from 'tree-sitter-c-sharp';
import Go from 'tree-sitter-go';
import Rust from 'tree-sitter-rust';
import Kotlin from 'tree-sitter-kotlin';
import PHP from 'tree-sitter-php';
import Ruby from 'tree-sitter-ruby';
import { createRequire } from 'node:module';
@ -19,6 +18,10 @@ const _require = createRequire(import.meta.url);
let Swift: any = null;
try { Swift = _require('tree-sitter-swift'); } catch {}
// tree-sitter-kotlin is an optionalDependency — may not be installed
let Kotlin: any = null;
try { Kotlin = _require('tree-sitter-kotlin'); } catch {}
let parser: Parser | null = null;
const languageMap: Record<string, any> = {
@ -32,7 +35,7 @@ const languageMap: Record<string, any> = {
[SupportedLanguages.CSharp]: CSharp,
[SupportedLanguages.Go]: Go,
[SupportedLanguages.Rust]: Rust,
[SupportedLanguages.Kotlin]: Kotlin,
...(Kotlin ? { [SupportedLanguages.Kotlin]: Kotlin } : {}),
[SupportedLanguages.PHP]: PHP.php_only,
[SupportedLanguages.Ruby]: Ruby,
...(Swift ? { [SupportedLanguages.Swift]: Swift } : {}),

View file

@ -27,6 +27,8 @@ interface PoolEntry {
waiters: Array<(conn: lbug.Connection) => void>;
lastUsed: number;
dbPath: string;
/** Set to true when the pool entry is closed — checkin will close orphaned connections */
closed: boolean;
}
const pool = new Map<string, PoolEntry>();
@ -40,6 +42,8 @@ interface SharedDB {
db: lbug.Database;
refCount: number;
ftsLoaded: boolean;
/** When true, closeOne skips db.close() — the Database is owned externally. */
external?: boolean;
}
const dbCache = new Map<string, SharedDB>();
@ -49,14 +53,14 @@ const MAX_POOL_SIZE = 5;
const IDLE_TIMEOUT_MS = 5 * 60 * 1000; // 5 minutes
/** Max connections per repo (caps concurrent queries per repo) */
const MAX_CONNS_PER_REPO = 8;
/** Connections created eagerly on init */
const INITIAL_CONNS_PER_REPO = 2;
let idleTimer: ReturnType<typeof setInterval> | null = null;
/** Saved real stdout.write — used to silence LadybugDB native output without race conditions */
const realStdoutWrite = process.stdout.write.bind(process.stdout);
export const realStdoutWrite = process.stdout.write.bind(process.stdout);
let stdoutSilenceCount = 0;
/** True while pre-warming connections — prevents watchdog from prematurely restoring stdout */
let preWarmActive = false;
/**
* Start the idle cleanup timer (runs every 60s)
@ -96,21 +100,47 @@ function evictLRU(): void {
}
/**
* Remove a repo from the pool and release its shared Database ref.
*
* LadybugDB's native .closeSync() triggers N-API destructor hooks that
* segfault on Linux/macOS. Pool databases are opened read-only, so
* there is no WAL to flush — just deleting the pool entry and letting
* the GC (or process exit) reclaim native resources is safe.
* Remove a repo from the pool, close its connections, and release its
* shared Database ref. Only closes the Database when no other repoIds
* reference it (refCount === 0).
*/
function closeOne(repoId: string): void {
const entry = pool.get(repoId);
if (entry) {
const shared = dbCache.get(entry.dbPath);
if (shared && shared.refCount > 0) {
shared.refCount--;
if (!entry) return;
entry.closed = true;
// Close available connections — fire-and-forget with .catch() to prevent
// unhandled rejections. Native close() returns Promise<void> but can crash
// the N-API destructor on macOS/Windows; deferring to process exit lets
// dangerouslyIgnoreUnhandledErrors absorb the crash.
for (const conn of entry.available) {
conn.close().catch(() => {});
}
entry.available.length = 0;
// Checked-out connections can't be closed here — they're in-flight.
// The checkin() function detects entry.closed and closes them on return.
// Only close the Database when no other repoIds reference it.
// External databases (injected via initLbugWithDb) are never closed here —
// the core adapter owns them and handles their lifecycle.
const shared = dbCache.get(entry.dbPath);
if (shared) {
shared.refCount--;
if (shared.refCount === 0) {
if (shared.external) {
// External databases are owned by the core adapter — don't close
// or remove from cache. Keep the entry so future initLbug() calls
// for the same dbPath reuse it instead of hitting a file lock.
shared.refCount = 0;
} else {
shared.db.close().catch(() => {});
dbCache.delete(entry.dbPath);
}
}
}
pool.delete(repoId);
}
@ -118,6 +148,8 @@ function closeOne(repoId: string): void {
* Create a new Connection from a repo's Database.
* Silences stdout to prevent native module output from corrupting MCP stdio.
*/
let activeQueryCount = 0;
function silenceStdout(): void {
if (stdoutSilenceCount++ === 0) {
process.stdout.write = (() => true) as any;
@ -131,6 +163,17 @@ function restoreStdout(): void {
}
}
// Safety watchdog: restore stdout if it gets stuck silenced (e.g. native crash
// inside createConnection before restoreStdout runs).
// Exempts active queries and pre-warm — these legitimately hold silence for
// longer than 1 second (queries can take up to QUERY_TIMEOUT_MS = 30s).
setInterval(() => {
if (stdoutSilenceCount > 0 && !preWarmActive && activeQueryCount === 0) {
stdoutSilenceCount = 0;
process.stdout.write = realStdoutWrite;
}
}, 1000).unref();
function createConnection(db: lbug.Database): lbug.Connection {
silenceStdout();
try {
@ -148,9 +191,15 @@ const WAITER_TIMEOUT_MS = 15_000;
const LOCK_RETRY_ATTEMPTS = 3;
const LOCK_RETRY_DELAY_MS = 2000;
/** Deduplicates concurrent initLbug calls for the same repoId */
const initPromises = new Map<string, Promise<void>>();
/**
* Initialize (or reuse) a Database + connection pool for a specific repo.
* Retries on lock errors (e.g., when `gitnexus analyze` is running).
*
* Concurrent calls for the same repoId are deduplicated — the second caller
* awaits the first's in-progress init rather than starting a redundant one.
*/
export const initLbug = async (repoId: string, dbPath: string): Promise<void> => {
const existing = pool.get(repoId);
@ -159,6 +208,27 @@ export const initLbug = async (repoId: string, dbPath: string): Promise<void> =>
return;
}
// Deduplicate concurrent init calls for the same repoId —
// prevents double-init race when multiple parallel tool calls
// trigger initialization for the same repo simultaneously.
const pending = initPromises.get(repoId);
if (pending) return pending;
const promise = doInitLbug(repoId, dbPath);
initPromises.set(repoId, promise);
try {
await promise;
} finally {
initPromises.delete(repoId);
}
};
/**
* Internal init — creates DB, pre-warms connections, loads FTS, then registers pool.
* Pool entry is registered LAST so concurrent executeQuery calls see either
* "not initialized" (and throw) or a fully ready pool — never a half-built one.
*/
async function doInitLbug(repoId: string, dbPath: string): Promise<void> {
// Check if database exists
try {
await fs.stat(dbPath);
@ -210,16 +280,22 @@ export const initLbug = async (repoId: string, dbPath: string): Promise<void> =>
shared.refCount++;
const db = shared.db;
// Pre-create a small pool of connections
// Pre-create the full pool upfront so createConnection() (which silences
// stdout) is never called lazily during active query execution.
// Mark preWarmActive so the watchdog timer doesn't interfere.
preWarmActive = true;
const available: lbug.Connection[] = [];
for (let i = 0; i < INITIAL_CONNS_PER_REPO; i++) {
available.push(createConnection(db));
try {
for (let i = 0; i < MAX_CONNS_PER_REPO; i++) {
available.push(createConnection(db));
}
} finally {
preWarmActive = false;
}
pool.set(repoId, { db, available, checkedOut: 0, waiters: [], lastUsed: Date.now(), dbPath });
ensureIdleTimer();
// Load FTS extension once per shared Database
// Load FTS extension once per shared Database.
// Done BEFORE pool registration so no concurrent checkout can grab
// the connection while the async FTS load is in progress.
if (!shared.ftsLoaded) {
try {
await available[0].query('LOAD EXTENSION fts');
@ -228,7 +304,75 @@ export const initLbug = async (repoId: string, dbPath: string): Promise<void> =>
// Extension may not be installed — FTS queries will fail gracefully
}
}
};
// Register pool entry only after all connections are pre-warmed and FTS is
// loaded. Concurrent executeQuery calls see either "not initialized"
// (and throw cleanly) or a fully ready pool — never a half-built one.
pool.set(repoId, { db, available, checkedOut: 0, waiters: [], lastUsed: Date.now(), dbPath, closed: false });
ensureIdleTimer();
}
/**
* Initialize a pool entry from a pre-existing Database object.
*
* Used in tests to avoid the writable→close→read-only cycle that crashes
* on macOS due to N-API destructor segfaults. The pool adapter reuses
* the core adapter's writable Database instead of opening a new read-only one.
*
* The Database is registered in the shared dbCache so closeOne() decrements
* the refCount correctly. If the Database is already cached (e.g. another
* repoId already injected it), the existing entry is reused.
*/
export async function initLbugWithDb(
repoId: string,
existingDb: lbug.Database,
dbPath: string,
): Promise<void> {
const existing = pool.get(repoId);
if (existing) {
existing.lastUsed = Date.now();
return;
}
// Register in dbCache with external: true so other initLbug() calls
// for the same dbPath reuse this Database instead of trying to open
// a new one (which would fail with a file lock error).
// closeOne() respects the external flag and skips db.close().
let shared = dbCache.get(dbPath);
if (!shared) {
shared = { db: existingDb, refCount: 0, ftsLoaded: false, external: true };
dbCache.set(dbPath, shared);
}
shared.refCount++;
const available: lbug.Connection[] = [];
preWarmActive = true;
try {
for (let i = 0; i < MAX_CONNS_PER_REPO; i++) {
available.push(createConnection(existingDb));
}
} finally {
preWarmActive = false;
}
// Load FTS extension if not already loaded on this Database
try {
await available[0].query('LOAD EXTENSION fts');
} catch {
// Extension may already be loaded or not installed
}
pool.set(repoId, {
db: existingDb,
available,
checkedOut: 0,
waiters: [],
lastUsed: Date.now(),
dbPath,
closed: false
});
ensureIdleTimer();
}
/**
* Checkout a connection from the pool.
@ -242,11 +386,16 @@ function checkout(entry: PoolEntry): Promise<lbug.Connection> {
return Promise.resolve(entry.available.pop()!);
}
// Grow the pool if under the cap
// Pool was pre-warmed to MAX_CONNS_PER_REPO during init. If we're here
// with fewer total connections, something leaked — surface the bug rather
// than silently creating a connection (which would silence stdout mid-query).
const totalConns = entry.available.length + entry.checkedOut;
if (totalConns < MAX_CONNS_PER_REPO) {
entry.checkedOut++;
return Promise.resolve(createConnection(entry.db));
throw new Error(
`Connection pool integrity error: expected ${MAX_CONNS_PER_REPO} ` +
`connections but found ${totalConns} (${entry.available.length} available, ` +
`${entry.checkedOut} checked out)`
);
}
// At capacity — queue the caller with a timeout.
@ -266,10 +415,17 @@ function checkout(entry: PoolEntry): Promise<lbug.Connection> {
/**
* Return a connection to the pool after use.
* If the pool entry was closed while the connection was checked out (e.g.
* LRU eviction), close the orphaned connection instead of returning it.
* If there are queued waiters, hand the connection directly to the next one
* instead of putting it back in the available array (avoids race conditions).
*/
function checkin(entry: PoolEntry, conn: lbug.Connection): void {
if (entry.closed) {
// Pool entry was deleted during checkout — close the orphaned connection
conn.close().catch(() => {});
return;
}
if (entry.waiters.length > 0) {
// Hand directly to the next waiter — no intermediate available state
const waiter = entry.waiters.shift()!;
@ -299,15 +455,23 @@ export const executeQuery = async (repoId: string, cypher: string): Promise<any[
throw new Error(`LadybugDB not initialized for repo "${repoId}". Call initLbug first.`);
}
if (isWriteQuery(cypher)) {
throw new Error('Write operations are not allowed. The pool adapter is read-only.');
}
entry.lastUsed = Date.now();
const conn = await checkout(entry);
silenceStdout();
activeQueryCount++;
try {
const queryResult = await withTimeout(conn.query(cypher), QUERY_TIMEOUT_MS, 'Query');
const result = Array.isArray(queryResult) ? queryResult[0] : queryResult;
const rows = await result.getAll();
return rows;
} finally {
activeQueryCount--;
restoreStdout();
checkin(entry, conn);
}
};
@ -329,6 +493,8 @@ export const executeParameterized = async (
entry.lastUsed = Date.now();
const conn = await checkout(entry);
silenceStdout();
activeQueryCount++;
try {
const stmt = await withTimeout(conn.prepare(cypher), QUERY_TIMEOUT_MS, 'Prepare');
if (!stmt.isSuccess()) {
@ -340,6 +506,8 @@ export const executeParameterized = async (
const rows = await result.getAll();
return rows;
} finally {
activeQueryCount--;
restoreStdout();
checkin(entry, conn);
}
};
@ -370,3 +538,11 @@ export const closeLbug = async (repoId?: string): Promise<void> => {
* Check if a specific repo's pool is active
*/
export const isLbugReady = (repoId: string): boolean => pool.has(repoId);
/** Regex to detect write operations in user-supplied Cypher queries */
export const CYPHER_WRITE_RE = /\b(CREATE|DELETE|SET|MERGE|REMOVE|DROP|ALTER|COPY|DETACH)\b/i;
/** Check if a Cypher query contains write operations */
export function isWriteQuery(query: string): boolean {
return CYPHER_WRITE_RE.test(query);
}

View file

@ -47,7 +47,7 @@ export const VALID_NODE_LABELS = new Set([
]);
/** Valid relation types for impact analysis filtering */
export const VALID_RELATION_TYPES = new Set(['CALLS', 'IMPORTS', 'EXTENDS', 'IMPLEMENTS']);
export const VALID_RELATION_TYPES = new Set(['CALLS', 'IMPORTS', 'EXTENDS', 'IMPLEMENTS', 'HAS_METHOD', 'HAS_PROPERTY', 'OVERRIDES', 'ACCESSES']);
/** Regex to detect write operations in user-supplied Cypher queries */
export const CYPHER_WRITE_RE = /\b(CREATE|DELETE|SET|MERGE|REMOVE|DROP|ALTER|COPY|DETACH)\b/i;
@ -88,6 +88,8 @@ export class LocalBackend {
private repos: Map<string, RepoHandle> = new Map();
private contextCache: Map<string, CodebaseContext> = new Map();
private initializedRepos: Set<string> = new Set();
private reinitPromises: Map<string, Promise<void>> = new Map();
private lastStalenessCheck: Map<string, number> = new Map();
// ─── Initialization ──────────────────────────────────────────────
@ -246,12 +248,51 @@ export class LocalBackend {
// ─── Lazy LadybugDB Init ────────────────────────────────────────────
private async ensureInitialized(repoId: string): Promise<void> {
// Always check the actual pool — the idle timer may have evicted the connection
if (this.initializedRepos.has(repoId) && isLbugReady(repoId)) return;
// If a reinit is already in progress for this repo, wait for it
const pending = this.reinitPromises.get(repoId);
if (pending) return pending;
const handle = this.repos.get(repoId);
if (!handle) throw new Error(`Unknown repo: ${repoId}`);
// Check if the index was rebuilt since we opened the connection (#297).
// Throttle staleness checks to at most once per 5 seconds per repo to
// avoid an fs.readFile round-trip on every tool invocation.
if (this.initializedRepos.has(repoId) && isLbugReady(repoId)) {
const now = Date.now();
const lastCheck = this.lastStalenessCheck.get(repoId) ?? 0;
if (now - lastCheck < 5000) return; // Checked recently — skip
this.lastStalenessCheck.set(repoId, now);
try {
const metaPath = path.join(handle.storagePath, 'meta.json');
const metaRaw = await fs.readFile(metaPath, 'utf-8');
const meta = JSON.parse(metaRaw);
if (meta.indexedAt && meta.indexedAt !== handle.indexedAt) {
// Index was rebuilt — close stale connection and re-init.
// Wrap in reinitPromises to prevent TOCTOU race where concurrent
// callers both detect staleness and double-close the pool.
const reinit = (async () => {
try {
await closeLbug(repoId);
this.initializedRepos.delete(repoId);
handle.indexedAt = meta.indexedAt;
await initLbug(repoId, handle.lbugPath);
this.initializedRepos.add(repoId);
} finally {
this.reinitPromises.delete(repoId);
}
})();
this.reinitPromises.set(repoId, reinit);
return reinit;
} else {
return; // Pool is current
}
} catch {
return; // Can't read meta — assume pool is fine
}
}
try {
await initLbug(repoId, handle.lbugPath);
this.initializedRepos.add(repoId);
@ -898,7 +939,7 @@ export class LocalBackend {
// Categorized incoming refs
const incomingRows = await executeParameterized(repo.id, `
MATCH (caller)-[r:CodeRelation]->(n {id: $symId})
WHERE r.type IN ['CALLS', 'IMPORTS', 'EXTENDS', 'IMPLEMENTS']
WHERE r.type IN ['CALLS', 'IMPORTS', 'EXTENDS', 'IMPLEMENTS', 'HAS_METHOD', 'HAS_PROPERTY', 'OVERRIDES', 'ACCESSES']
RETURN r.type AS relType, caller.id AS uid, caller.name AS name, caller.filePath AS filePath, labels(caller)[0] AS kind
LIMIT 30
`, { symId });
@ -906,7 +947,7 @@ export class LocalBackend {
// Categorized outgoing refs
const outgoingRows = await executeParameterized(repo.id, `
MATCH (n {id: $symId})-[r:CodeRelation]->(target)
WHERE r.type IN ['CALLS', 'IMPORTS', 'EXTENDS', 'IMPLEMENTS']
WHERE r.type IN ['CALLS', 'IMPORTS', 'EXTENDS', 'IMPLEMENTS', 'HAS_METHOD', 'HAS_PROPERTY', 'OVERRIDES', 'ACCESSES']
RETURN r.type AS relType, target.id AS uid, target.name AS name, target.filePath AS filePath, labels(target)[0] AS kind
LIMIT 30
`, { symId });
@ -1329,6 +1370,29 @@ export class LocalBackend {
relationTypes?: string[];
includeTests?: boolean;
minConfidence?: number;
}): Promise<any> {
try {
return await this._impactImpl(repo, params);
} catch (err: any) {
// Return structured error instead of crashing (#321)
return {
error: (err instanceof Error ? err.message : String(err)) || 'Impact analysis failed',
target: { name: params.target },
direction: params.direction,
impactedCount: 0,
risk: 'UNKNOWN',
suggestion: 'The graph query failed — try gitnexus context <symbol> as a fallback',
};
}
}
private async _impactImpl(repo: RepoHandle, params: {
target: string;
direction: 'upstream' | 'downstream';
maxDepth?: number;
relationTypes?: string[];
includeTests?: boolean;
minConfidence?: number;
}): Promise<any> {
await this.ensureInitialized(repo.id);
@ -1358,6 +1422,7 @@ export class LocalBackend {
const impacted: any[] = [];
const visited = new Set<string>([symId]);
let frontier = [symId];
let traversalComplete = true;
for (let depth = 1; depth <= maxDepth && frontier.length > 0; depth++) {
const nextFrontier: string[] = [];
@ -1391,7 +1456,13 @@ export class LocalBackend {
});
}
}
} catch (e) { logQueryError('impact:depth-traversal', e); }
} catch (e) {
logQueryError('impact:depth-traversal', e);
// Break out of depth loop on query failure but return partial results
// collected so far, rather than silently swallowing the error (#321)
traversalComplete = false;
break;
}
frontier = nextFrontier;
}
@ -1408,31 +1479,51 @@ export class LocalBackend {
let affectedModules: any[] = [];
if (impacted.length > 0) {
const allIds = impacted.map(i => `'${i.id.replace(/'/g, "''")}'`).join(', ');
const d1Ids = (grouped[1] || []).map((i: any) => `'${i.id.replace(/'/g, "''")}'`).join(', ');
// Cap IN-clause to 100 IDs to prevent oversized queries that crash
// the native DB engine on arm64 macOS (#292)
const cappedImpacted = impacted.slice(0, 100);
const allIds = cappedImpacted.map(i => `'${String(i.id ?? '').replace(/'/g, "''")}'`).join(', ');
const d1Items = (grouped[1] || []).slice(0, 100);
const d1Ids = d1Items.map((i: any) => `'${String(i.id ?? '').replace(/'/g, "''")}'`).join(', ');
// Affected processes: which execution flows are broken and at which step
const [processRows, moduleRows, directModuleRows] = await Promise.all([
executeQuery(repo.id, `
MATCH (s)-[r:CodeRelation {type: 'STEP_IN_PROCESS'}]->(p:Process)
WHERE s.id IN [${allIds}]
RETURN p.heuristicLabel AS name, COUNT(DISTINCT s.id) AS hits, MIN(r.step) AS minStep, p.stepCount AS stepCount
ORDER BY hits DESC
LIMIT 20
`).catch(() => []),
executeQuery(repo.id, `
MATCH (s)-[:CodeRelation {type: 'MEMBER_OF'}]->(c:Community)
WHERE s.id IN [${allIds}]
RETURN c.heuristicLabel AS name, COUNT(DISTINCT s.id) AS hits
ORDER BY hits DESC
LIMIT 20
`).catch(() => []),
d1Ids ? executeQuery(repo.id, `
MATCH (s)-[:CodeRelation {type: 'MEMBER_OF'}]->(c:Community)
WHERE s.id IN [${d1Ids}]
RETURN DISTINCT c.heuristicLabel AS name
`).catch(() => []) : Promise.resolve([]),
]);
// Enrichment queries: sequential on arm64 macOS to avoid SIGSEGV from
// concurrent native DB access (#285, #290, #292); parallel elsewhere
// to preserve performance on unaffected platforms.
const isArm64Mac = process.platform === 'darwin' && process.arch === 'arm64';
const processQuery = executeQuery(repo.id, `
MATCH (s)-[r:CodeRelation {type: 'STEP_IN_PROCESS'}]->(p:Process)
WHERE s.id IN [${allIds}]
RETURN p.heuristicLabel AS name, COUNT(DISTINCT s.id) AS hits, MIN(r.step) AS minStep, p.stepCount AS stepCount
ORDER BY hits DESC
LIMIT 20
`).catch(() => []);
const moduleQuery = () => executeQuery(repo.id, `
MATCH (s)-[:CodeRelation {type: 'MEMBER_OF'}]->(c:Community)
WHERE s.id IN [${allIds}]
RETURN c.heuristicLabel AS name, COUNT(DISTINCT s.id) AS hits
ORDER BY hits DESC
LIMIT 20
`).catch(() => []);
const directModuleQuery = () => d1Ids
? executeQuery(repo.id, `
MATCH (s)-[:CodeRelation {type: 'MEMBER_OF'}]->(c:Community)
WHERE s.id IN [${d1Ids}]
RETURN DISTINCT c.heuristicLabel AS name
`).catch(() => [])
: Promise.resolve([]);
let processRows: any[], moduleRows: any[], directModuleRows: any[];
if (isArm64Mac) {
// Sequential: avoid concurrent native DB access
processRows = await processQuery;
moduleRows = await moduleQuery();
directModuleRows = await directModuleQuery();
} else {
// Parallel: safe on non-arm64 platforms
processRows = await processQuery;
[moduleRows, directModuleRows] = await Promise.all([moduleQuery(), directModuleQuery()]);
}
affectedProcesses = processRows.map((r: any) => ({
name: r.name || r[0],
@ -1474,6 +1565,7 @@ export class LocalBackend {
direction,
impactedCount: impacted.length,
risk,
...(!traversalComplete && { partial: true }),
summary: {
direct: directCount,
processes_affected: processCount,

View file

@ -321,6 +321,13 @@ nodes:
additional_node_types: "Multi-language: Struct, Enum, Macro, Typedef, Union, Namespace, Trait, Impl, TypeAlias, Const, Static, Property, Record, Delegate, Annotation, Constructor, Template, Module (use backticks in queries: \`Struct\`, \`Enum\`, etc.)"
node_properties:
common: "name (STRING), filePath (STRING), startLine (INT32), endLine (INT32)"
Method: "parameterCount (INT32), returnType (STRING), isVariadic (BOOL)"
Function: "parameterCount (INT32), returnType (STRING), isVariadic (BOOL)"
Property: "declaredType (STRING) — the field's type annotation (e.g., 'Address', 'City'). Used for field-access chain resolution."
Constructor: "parameterCount (INT32)"
relationships:
- CONTAINS: File/Folder contains child
- DEFINES: File defines a symbol
@ -328,6 +335,10 @@ relationships:
- IMPORTS: Module imports
- EXTENDS: Class inheritance
- IMPLEMENTS: Interface implementation
- HAS_METHOD: Class/Struct/Interface owns a Method
- HAS_PROPERTY: Class/Struct/Interface owns a Property (field)
- ACCESSES: Function/Method reads or writes a Property (reason: 'read' or 'write')
- OVERRIDES: Method overrides another Method (MRO)
- MEMBER_OF: Symbol belongs to community
- STEP_IN_PROCESS: Symbol is step N in process

View file

@ -24,6 +24,7 @@ import {
GetPromptRequestSchema,
} from '@modelcontextprotocol/sdk/types.js';
import { GITNEXUS_TOOLS } from './tools.js';
import { realStdoutWrite } from './core/lbug-adapter.js';
import type { LocalBackend } from './local/local-backend.js';
import { getResourceDefinitions, getResourceTemplates, readResource } from './resources.js';
@ -276,24 +277,45 @@ Follow these steps:
export async function startMCPServer(backend: LocalBackend): Promise<void> {
const server = createMCPServer(backend);
// Connect to stdio transport
const transport = new CompatibleStdioServerTransport();
// Use the shared stdout reference captured at module-load time by the
// lbug-adapter. Avoids divergence if anything patches stdout between
// module load and server start.
const _safeStdout = new Proxy(process.stdout, {
get(target, prop, receiver) {
if (prop === 'write') return realStdoutWrite;
const val = Reflect.get(target, prop, receiver);
return typeof val === 'function' ? val.bind(target) : val;
}
});
const transport = new CompatibleStdioServerTransport(process.stdin, _safeStdout);
await server.connect(transport);
// Graceful shutdown helper
let shuttingDown = false;
const shutdown = async () => {
const shutdown = async (exitCode = 0) => {
if (shuttingDown) return;
shuttingDown = true;
try { await backend.disconnect(); } catch {}
try { await server.close(); } catch {}
process.exit(0);
process.exit(exitCode);
};
// Handle graceful shutdown
process.on('SIGINT', shutdown);
process.on('SIGTERM', shutdown);
// Log crashes to stderr so they aren't silently lost.
// uncaughtException is fatal — shut down.
// unhandledRejection is logged but kept non-fatal (availability-first):
// killing the server for one missed catch would be worse than logging it.
process.on('uncaughtException', (err) => {
process.stderr.write(`GitNexus MCP uncaughtException: ${err?.stack || err}\n`);
shutdown(1);
});
process.on('unhandledRejection', (reason: any) => {
process.stderr.write(`GitNexus MCP unhandledRejection: ${reason?.stack || reason}\n`);
});
// Handle stdio errors — stdin close means the parent process is gone
process.stdin.on('end', shutdown);
process.stdin.on('error', () => shutdown());

View file

@ -78,7 +78,7 @@ SCHEMA:
- Nodes: File, Folder, Function, Class, Interface, Method, CodeElement, Community, Process
- Multi-language nodes (use backticks): \`Struct\`, \`Enum\`, \`Trait\`, \`Impl\`, etc.
- All edges via single CodeRelation table with 'type' property
- Edge types: CONTAINS, DEFINES, CALLS, IMPORTS, EXTENDS, IMPLEMENTS, HAS_METHOD, OVERRIDES, MEMBER_OF, STEP_IN_PROCESS
- Edge types: CONTAINS, DEFINES, CALLS, IMPORTS, EXTENDS, IMPLEMENTS, HAS_METHOD, HAS_PROPERTY, ACCESSES, OVERRIDES, MEMBER_OF, STEP_IN_PROCESS
- Edge properties: type (STRING), confidence (DOUBLE), reason (STRING), step (INT32)
EXAMPLES:
@ -94,6 +94,12 @@ EXAMPLES:
• Find all methods of a class:
MATCH (c:Class {name: "UserService"})-[r:CodeRelation {type: 'HAS_METHOD'}]->(m:Method) RETURN m.name, m.parameterCount, m.returnType
• Find all properties of a class:
MATCH (c:Class {name: "User"})-[r:CodeRelation {type: 'HAS_PROPERTY'}]->(p:Property) RETURN p.name, p.declaredType
• Find all writers of a field:
MATCH (f:Function)-[r:CodeRelation {type: 'ACCESSES', reason: 'write'}]->(p:Property) WHERE p.name = "address" RETURN f.name, f.filePath
• Find method overrides (MRO resolution):
MATCH (winner:Method)-[r:CodeRelation {type: 'OVERRIDES'}]->(loser:Method) RETURN winner.name, winner.filePath, loser.filePath, r.reason
@ -119,12 +125,14 @@ TIPS:
{
name: 'context',
description: `360-degree view of a single code symbol.
Shows categorized incoming/outgoing references (calls, imports, extends, implements), process participation, and file location.
Shows categorized incoming/outgoing references (calls, imports, extends, implements, methods, properties, overrides), process participation, and file location.
WHEN TO USE: After query() to understand a specific symbol in depth. When you need to know all callers, callees, and what execution flows a symbol participates in.
AFTER THIS: Use impact() if planning changes, or READ gitnexus://repo/{name}/process/{processName} for full execution trace.
Handles disambiguation: if multiple symbols share the same name, returns candidates for you to pick from. Use uid param for zero-ambiguity lookup from prior results.`,
Handles disambiguation: if multiple symbols share the same name, returns candidates for you to pick from. Use uid param for zero-ambiguity lookup from prior results.
NOTE: ACCESSES edges (field read/write tracking) are included in context results with reason 'read' or 'write'. CALLS edges resolve through field access chains and method-call chains (e.g., user.address.getCity().save() produces CALLS edges at each step).`,
inputSchema: {
type: 'object',
properties: {
@ -200,7 +208,9 @@ Depth groups:
- d=2: LIKELY AFFECTED (indirect)
- d=3: MAY NEED TESTING (transitive)
EdgeType: CALLS, IMPORTS, EXTENDS, IMPLEMENTS, HAS_METHOD, OVERRIDES
TIP: Default traversal uses CALLS/IMPORTS/EXTENDS/IMPLEMENTS. For class members, include HAS_METHOD and HAS_PROPERTY in relationTypes. For field access analysis, include ACCESSES in relationTypes.
EdgeType: CALLS, IMPORTS, EXTENDS, IMPLEMENTS, HAS_METHOD, HAS_PROPERTY, OVERRIDES, ACCESSES
Confidence: 1.0 = certain, <0.8 = fuzzy match`,
inputSchema: {
type: 'object',
@ -208,7 +218,7 @@ Confidence: 1.0 = certain, <0.8 = fuzzy match`,
target: { type: 'string', description: 'Name of function, class, or file to analyze' },
direction: { type: 'string', description: 'upstream (what depends on this) or downstream (what this depends on)' },
maxDepth: { type: 'number', description: 'Max relationship depth (default: 3)', default: 3 },
relationTypes: { type: 'array', items: { type: 'string' }, description: 'Filter: CALLS, IMPORTS, EXTENDS, IMPLEMENTS, HAS_METHOD, OVERRIDES (default: usage-based)' },
relationTypes: { type: 'array', items: { type: 'string' }, description: 'Filter: CALLS, IMPORTS, EXTENDS, IMPLEMENTS, HAS_METHOD, HAS_PROPERTY, OVERRIDES, ACCESSES (default: usage-based, ACCESSES excluded by default)' },
includeTests: { type: 'boolean', description: 'Include test files (default: false)' },
minConfidence: { type: 'number', description: 'Minimum confidence 0-1 (default: 0.7)' },
repo: { type: 'string', description: 'Repository name or path. Omit if only one repo is indexed.' },

View file

@ -0,0 +1,7 @@
#include "../models/user_factory.h"
void process() {
User user = get_user();
user.save();
user.get_name();
}

View file

@ -0,0 +1,7 @@
#include "user.h"
void User::save() {}
std::string User::get_name() {
return "";
}

View file

@ -0,0 +1,9 @@
#pragma once
#include <string>
class User {
public:
void save();
std::string get_name();
};

View file

@ -0,0 +1,5 @@
#include "user_factory.h"
User get_user() {
return User();
}

View file

@ -0,0 +1,5 @@
#pragma once
#include "user.h"
User get_user();

View file

@ -0,0 +1,14 @@
using static CrossFile.Models.UserFactory;
namespace CrossFile.App
{
public class Program
{
public void Run()
{
var u = GetUser();
u.Save();
u.GetName();
}
}
}

View file

@ -0,0 +1,5 @@
<Project Sdk="Microsoft.NET.Sdk">
<PropertyGroup>
<RootNamespace>CrossFile</RootNamespace>
</PropertyGroup>
</Project>

Some files were not shown because too many files have changed in this diff Show more